{"id":"W4381572165","doi":"10.2196/44331","title":"Optimizing Patient Record Linkage in a Master Patient Index Using Machine Learning: Algorithm Development and Validation","year":2023,"lang":"en","type":"article","venue":"JMIR Formative Research","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Hamilton Health Sciences; McMaster University; Population Health Research Institute; University of Toronto","funders":"Strong; Bill and Melinda Gates Foundation; United States Agency for International Development","keywords":"Computer science; Matching (statistics); Machine learning; Population; Software; Linkage (software); Record linkage; Artificial intelligence; Interface (matter); Bayesian optimization; Set (abstract data type); Data mining; Programming language; Medicine; Operating system","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008347609,0.0001475449,0.0002379529,0.00162059,0.0004488126,0.0005157602,0.0003567064,0.00007311092,0.0001192339],"category_scores_gemma":[0.0006560818,0.0001165629,0.00003509769,0.002303352,0.0001033475,0.001043491,0.001571808,0.000556029,0.0003639642],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002295533,"about_ca_system_score_gemma":0.00008286518,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001456881,"about_ca_topic_score_gemma":0.00007120748,"domain_scores_codex":[0.9945183,0.00122778,0.0008630813,0.000455043,0.002362593,0.0005731945],"domain_scores_gemma":[0.9982293,0.0008171474,0.0002023614,0.0003051384,0.0003117337,0.0001343688],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00004789691,0.000112314,0.003873042,0.00003721494,0.00001514682,0.00002980406,0.06858306,0.0008101581,0.00003786233,0.0001043689,0.0009646929,0.9253845],"study_design_scores_gemma":[0.001163888,0.0006540258,0.01652528,0.0002598896,0.00000265905,0.000005087765,0.06933504,0.6874129,0.002067152,0.003437431,0.2186852,0.000451452],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.980693,0.00003715136,0.0168404,0.0003573755,0.0001399033,0.0007716509,0.00002789848,0.00004750842,0.001085107],"genre_scores_gemma":[0.9882848,0.00005317932,0.01066695,0.00007102556,0.00002028593,0.0001744388,0.0001151633,0.00001763878,0.0005965814],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.924933,"threshold_uncertainty_score":0.4973487,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3302761360581266,"score_gpt":0.4740721920281339,"score_spread":0.1437960559700073,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}