{"id":"W2783287373","doi":"","title":"Quantifying duplication to improve data quality.","year":2017,"lang":"en","type":"article","venue":"Conference of the Centre for Advanced Studies on Collaborative Research","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"McMaster University","funders":"","keywords":"Computer science; Data deduplication; Quality (philosophy); Gene duplication; Database","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2538263,0.00098558,0.002140525,0.011184,0.002828678,0.009270019,0.004218744,0.002609968,0.004349886],"category_scores_gemma":[0.6026192,0.0009548086,0.002162582,0.01643279,0.003834301,0.009919583,0.008396327,0.003205022,0.001006998],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00851664,"about_ca_system_score_gemma":0.02110742,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01037026,"about_ca_topic_score_gemma":0.007563538,"domain_scores_codex":[0.6462875,0.2418808,0.0246252,0.01244271,0.07136218,0.003401692],"domain_scores_gemma":[0.2637197,0.4659767,0.06843974,0.09496896,0.1022323,0.004662643],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001724893,0.0004044698,0.2479933,0.004286822,0.002263074,0.0002032092,0.005376608,0.02375631,0.004444405,0.1056582,0.06298158,0.5409071],"study_design_scores_gemma":[0.0006374932,0.001351059,0.1759651,0.006092748,0.002078517,0.0008289388,0.006693655,0.1931715,0.03945812,0.4094435,0.1636844,0.0005949228],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1327717,0.01775447,0.7592188,0.04224018,0.002875257,0.003813507,0.01100824,0.001982002,0.0283359],"genre_scores_gemma":[0.5834061,0.002726448,0.3993998,0.003392876,0.0008624422,0.001724019,0.005288251,0.000423468,0.002776514],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.7461737,"threshold_uncertainty_score":0.9201651,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.8003330727568276,"score_gpt":0.6486310263808803,"score_spread":0.1517020463759473,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}