{"id":"W3040360098","doi":"10.1017/ssh.2020.15","title":"Selection Bias Encountered in the Systematic Linking of Historical Census Records","year":2020,"lang":"en","type":"article","venue":"Social Science History","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"St. Francis Xavier University; University of Guelph","funders":"","keywords":"Census; Selection bias; Matching (statistics); Sample (material); Population; Selection (genetic algorithm); Sampling bias; Econometrics; Statistics; Information bias; Genealogy; Geography; Sample size determination; Computer science; Sociology; Economics; History; Demography; Mathematics; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2040239,0.0004279214,0.0008480379,0.00404832,0.0034985,0.003535286,0.001889752,0.001564939,0.002443756],"category_scores_gemma":[0.489668,0.0007767181,0.0006188514,0.009310365,0.005518643,0.003390396,0.004478945,0.001664796,0.0007264636],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002835715,"about_ca_system_score_gemma":0.004123814,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008970401,"about_ca_topic_score_gemma":0.01199363,"domain_scores_codex":[0.695195,0.2427624,0.01519542,0.009784671,0.03515021,0.001912415],"domain_scores_gemma":[0.4444416,0.4196855,0.04963039,0.05688784,0.02826021,0.001094543],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0006872382,0.0002072367,0.3084886,0.003271961,0.001190792,0.001456809,0.03010772,0.00658407,0.005692856,0.1962747,0.02185122,0.4241868],"study_design_scores_gemma":[0.0003290225,0.001074292,0.269399,0.00845739,0.001438696,0.004434175,0.01905285,0.0262954,0.02454395,0.4004521,0.2440837,0.0004393768],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1991336,0.004401912,0.7488305,0.01211843,0.001467305,0.003147168,0.00169619,0.0004749052,0.02873009],"genre_scores_gemma":[0.7028323,0.001804917,0.2832412,0.003925637,0.0006783655,0.002477988,0.0007184132,0.0001933532,0.004127819],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7959762,"threshold_uncertainty_score":0.9815803,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4476471592264005,"score_gpt":0.3994208451867887,"score_spread":0.04822631403961175,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}