{"id":"W2892160487","doi":"10.23889/ijpds.v3i4.979","title":"Bias, accuracy and sample size in the systematic linking of historical records","year":2018,"lang":"en","type":"article","venue":"International Journal for Population Data Science","topic":"Census and Population Estimation","field":"Mathematics","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"St. Francis Xavier University; University of Guelph","funders":"","keywords":"Representativeness heuristic; Computer science; Sampling bias; Sample size determination; Data quality; Invariant (physics); Data mining; Sample (material); Selection bias; Statistics; Econometrics; Artificial intelligence; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3143033,0.0007644945,0.001438042,0.005818837,0.001990827,0.004317124,0.002826527,0.002943609,0.002263466],"category_scores_gemma":[0.6504667,0.0008816112,0.001848156,0.00594563,0.006030149,0.005754869,0.006268919,0.001383387,0.0006406225],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002414856,"about_ca_system_score_gemma":0.001898735,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00605122,"about_ca_topic_score_gemma":0.00549539,"domain_scores_codex":[0.6222692,0.2823136,0.02395722,0.02490874,0.04408147,0.002469733],"domain_scores_gemma":[0.200676,0.6935636,0.03985982,0.04857294,0.0164397,0.0008878754],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001002788,0.0001560799,0.7873457,0.001207424,0.002433799,0.000454042,0.005255878,0.01324926,0.0007116986,0.01983381,0.003855697,0.1644938],"study_design_scores_gemma":[0.0006024694,0.001899506,0.7182237,0.00300331,0.003870567,0.002117373,0.004939514,0.1093551,0.01789897,0.1045141,0.03315287,0.0004225759],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6407964,0.0127934,0.3150222,0.0060354,0.001263281,0.002856026,0.003031414,0.0007538039,0.0174481],"genre_scores_gemma":[0.9207682,0.0007205372,0.07350899,0.0007668042,0.0003425675,0.001329676,0.0009845824,0.0001052595,0.00147342],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6856967,"threshold_uncertainty_score":0.8455861,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2533297546395775,"score_gpt":0.4503384809297971,"score_spread":0.1970087262902196,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}