{"id":"W4312036666","doi":"10.2196/preprints.44331","title":"Optimizing Patient Record Linkage in a Master Patient Index Using Machine Learning: Algorithm Development and Validation (Preprint)","year":2022,"lang":"en","type":"preprint","venue":"","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Hamilton Health Sciences; McMaster University; Population Health Research Institute; University of Toronto","funders":"","keywords":"Computer science; Matching (statistics); Record linkage; Population; Software; Machine learning; Linkage (software); Interface (matter); Artificial intelligence; Set (abstract data type); Data mining; Programming language; Medicine; Operating system","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01384642,0.001170829,0.001259082,0.00197562,0.0009367539,0.001985485,0.002315752,0.001516037,0.005220936],"category_scores_gemma":[0.03529662,0.0007252601,0.001347088,0.00182165,0.0007202601,0.00125128,0.002006233,0.001558431,0.001667469],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002137292,"about_ca_system_score_gemma":0.003830938,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01052491,"about_ca_topic_score_gemma":0.008093401,"domain_scores_codex":[0.9952304,0.00240039,0.00036674,0.000891488,0.0008897693,0.0002212319],"domain_scores_gemma":[0.9775978,0.01625292,0.0009949305,0.002148339,0.002630215,0.0003758766],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006811324,0.000682647,0.01873505,0.0005786487,0.0004550984,0.0002121402,0.0003187727,0.6143572,0.003134404,0.004405964,0.02653143,0.3299076],"study_design_scores_gemma":[0.0001286357,0.00009836277,0.002252489,0.00003739512,0.0000288898,0.00005715997,0.00003782132,0.9879676,0.004547423,0.002322252,0.002497134,0.0000249066],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.145638,0.0007507162,0.7988625,0.001438277,0.000291612,0.0009153801,0.003892594,0.04466179,0.003549123],"genre_scores_gemma":[0.1866791,0.0001631923,0.8024894,0.0004110524,0.00006152412,0.0008698449,0.006390419,0.001248308,0.001687124],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01384642,"threshold_uncertainty_score":0.07322776,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1697825093542959,"score_gpt":0.3599210885867782,"score_spread":0.1901385792324823,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}