{"id":"W2775078427","doi":"10.1177/2053951717745678","title":"Challenges in administrative data linkage for research","year":2017,"lang":"en","type":"article","venue":"Big Data & Society","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":356,"is_retracted":false,"has_abstract":true,"ca_institutions":"Institute for Clinical Evaluative Sciences","funders":"Economic and Social Research Council; Wellcome Trust","keywords":"Linkage (software); Record linkage; Computer science; Data quality; Data science; Data collection; Confidentiality; Data mining; Sample (material); Population; Computer security; Engineering; Statistics; Sociology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[{"model":"gemma","categories":["metaresearch"],"domain":"methods","study_design":"not_applicable","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low","status":"direct model label, unvalidated"},{"model":"gpt","categories":["metaresearch"],"domain":"methods","study_design":"design_other","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low","status":"direct model label, unvalidated"}],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.6349509,0.00124112,0.00464532,0.01068699,0.01210446,0.03324808,0.01277345,0.01078548,0.004573033],"category_scores_gemma":[0.7454242,0.002750949,0.003863978,0.03018361,0.02860501,0.03997223,0.03054885,0.01804472,0.003126342],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01496643,"about_ca_system_score_gemma":0.05795113,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009693236,"about_ca_topic_score_gemma":0.004298186,"domain_scores_codex":[0.2297899,0.6375625,0.04088628,0.01878127,0.06976727,0.003212799],"domain_scores_gemma":[0.1619425,0.6753441,0.02664045,0.07039847,0.06152676,0.004147772],"domain_codex":"methods","domain_gemma":"methods","domain_candidate":"methods","domain_consensus":"methods","study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001622451,0.0000812295,0.005823257,0.004581266,0.0004971627,0.0003403199,0.01672407,0.005488006,0.0001545298,0.6513706,0.04124322,0.2735341],"study_design_scores_gemma":[0.00007716113,0.00009095342,0.00160105,0.007657688,0.0001525064,0.0005326266,0.008444296,0.004753542,0.0003444111,0.6970929,0.2790782,0.0001746981],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.005349892,0.04397675,0.5115274,0.4081247,0.00572695,0.001869902,0.001223555,0.0005696813,0.0216312],"genre_scores_gemma":[0.1181546,0.03760586,0.7707337,0.04823246,0.009510733,0.009100664,0.001759789,0.0008666036,0.004035654],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.3650491,"threshold_uncertainty_score":0.4501706,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.9805021151538041,"score_gpt":0.681924870153455,"score_spread":0.2985772450003491,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}