{"id":"W4411490737","doi":"10.1038/s41598-025-06447-2","title":"Evaluating language model embeddings for Parkinson’s disease cohort harmonization using a novel manually curated variable mapping schema","year":2025,"lang":"en","type":"article","venue":"Scientific Reports","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Institute on Aging; National Institute of Biomedical Imaging and Bioengineering; Canadian Institutes of Health Research; Genentech; National Institutes of Health; IXICO; H. Lundbeck A/S; Mitsubishi Tanabe Pharma Corporation; Servier; Université de Genève; Shionogi; Japan Science and Technology Agency; Astellas Pharma; Fondazione Cariplo; Eisai; Daiichi-Sankyo; European Commission; GHR Foundation; Pfizer; Biogen; BioClinica; F. Hoffmann-La Roche; Wellcome Trust; University of Southern California; U.S. Department of Defense; Eli Lilly and Company; Bristol-Myers Squibb; Meso Scale Diagnostics; Alzheimer's Disease Neuroimaging Initiative; Ministero della Salute; Novartis Pharmaceuticals Corporation; Alzheimer's Association; European Federation of Pharmaceutical Industries and Associations; Brigham and Women's Hospital","keywords":"Harmonization; Computer science; Cohort; Artificial intelligence; Schema (genetic algorithms); Language model; Natural language processing; Data mining; Machine learning; Medicine; Pathology","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004237426,0.0007714359,0.000349203,0.002362588,0.0003667771,0.001224865,0.0006862643,0.0007612613,0.001758944],"category_scores_gemma":[0.01642286,0.0001955854,0.001030702,0.001344301,0.0003294475,0.001866082,0.001483962,0.0007862479,0.0008224255],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006558057,"about_ca_system_score_gemma":0.001252553,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0037926,"about_ca_topic_score_gemma":0.004895324,"domain_scores_codex":[0.9977719,0.0008627549,0.0003363174,0.0006387368,0.0003077762,0.00008232579],"domain_scores_gemma":[0.9944769,0.003235627,0.0004674428,0.000920009,0.0007884356,0.0001116422],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001329986,0.0008743593,0.1088987,0.001645876,0.0009305229,0.001094461,0.002477293,0.1501821,0.03257177,0.01281016,0.03077505,0.6564096],"study_design_scores_gemma":[0.0001735404,0.0005323401,0.01983048,0.0002339004,0.0002524052,0.0008806541,0.001851492,0.8973254,0.04108328,0.01175592,0.02599094,0.00008971797],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6031591,0.001188904,0.3560709,0.000855284,0.0002942784,0.0007104707,0.0191472,0.01476385,0.003809978],"genre_scores_gemma":[0.616989,0.0003166881,0.3395544,0.0002058478,0.00003972157,0.0004175434,0.04045587,0.0005674351,0.001453572],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.004237426,"threshold_uncertainty_score":0.02240992,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04914273914335597,"score_gpt":0.3592620222019989,"score_spread":0.3101192830586429,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}