{"id":"W4411490737","doi":"10.1038/s41598-025-06447-2","title":"Evaluating language model embeddings for Parkinson’s disease cohort harmonization using a novel manually curated variable mapping schema","year":2025,"lang":"en","type":"article","venue":"Scientific Reports","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Institute on Aging; National Institute of Biomedical Imaging and Bioengineering; Canadian Institutes of Health Research; Genentech; National Institutes of Health; IXICO; H. Lundbeck A/S; Mitsubishi Tanabe Pharma Corporation; Servier; Université de Genève; Shionogi; Japan Science and Technology Agency; Astellas Pharma; Fondazione Cariplo; Eisai; Daiichi-Sankyo; European Commission; GHR Foundation; Pfizer; Biogen; BioClinica; F. Hoffmann-La Roche; Wellcome Trust; University of Southern California; U.S. Department of Defense; Eli Lilly and Company; Bristol-Myers Squibb; Meso Scale Diagnostics; Alzheimer's Disease Neuroimaging Initiative; Ministero della Salute; Novartis Pharmaceuticals Corporation; Alzheimer's Association; European Federation of Pharmaceutical Industries and Associations; Brigham and Women's Hospital","keywords":"Harmonization; Computer science; Cohort; Artificial intelligence; Schema (genetic algorithms); Language model; Natural language processing; Data mining; Machine learning; Medicine; Pathology","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001511644,0.0001408979,0.0001549602,0.0001029076,0.0003085333,0.0001739995,0.000133272,0.0001269863,0.000008524408],"category_scores_gemma":[0.001487109,0.0001345785,0.00007466921,0.0003523337,0.0001451102,0.000009425579,0.0001373164,0.00006073576,5.559414e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003933319,"about_ca_system_score_gemma":0.0005554287,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0000125729,"about_ca_topic_score_gemma":0.00000214327,"domain_scores_codex":[0.9982703,0.00002297407,0.0003850106,0.0007809139,0.0002418637,0.0002989401],"domain_scores_gemma":[0.9988956,0.00001699892,0.0002095986,0.0004928809,0.0002912719,0.00009365543],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0000332411,0.00005961982,0.002878339,0.00009226672,0.00005075086,0.000007470069,0.0001166743,0.007246,0.9838879,0.00006133214,0.003292148,0.00227422],"study_design_scores_gemma":[0.0005356311,0.00003452667,0.0006026264,0.0002413881,0.0001287061,0.00002535097,0.0002856131,0.8556122,0.1005224,0.002071171,0.03959348,0.000346964],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5630272,0.0003525812,0.4351201,0.00006234707,0.0009009682,0.0003187155,0.00001285992,0.000038031,0.0001672368],"genre_scores_gemma":[0.7842468,0.00000386176,0.2084642,0.0001440323,0.00008404431,0.00008585443,0.0005605501,0.00001823964,0.006392396],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8833656,"threshold_uncertainty_score":0.548795,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04914273914335597,"score_gpt":0.3592620222019989,"score_spread":0.3101192830586429,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}