{"id":"W4400316892","doi":"10.1038/s41746-024-01180-y","title":"The long but necessary road to responsible use of large language models in healthcare research","year":2024,"lang":"en","type":"editorial","venue":"npj Digital Medicine","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":26,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"University of Toronto","keywords":"Dissemination; Data extraction; Health care; Computer science; Data science; Risk analysis (engineering); Psychology; Cognitive psychology; Medicine; MEDLINE; Political science","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.003317979,0.0003109232,0.0007781468,0.00098128,0.0001632795,0.0001217028,0.0003684972,0.0007173887,0.0000282234],"category_scores_gemma":[0.01929717,0.0002064552,0.0001009677,0.001565094,0.0003126234,0.0003222548,0.0002215565,0.002538528,0.0001470548],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005727623,"about_ca_system_score_gemma":0.002940541,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01848952,"about_ca_topic_score_gemma":0.005500429,"domain_scores_codex":[0.9946672,0.0002583138,0.00125114,0.000672941,0.002237229,0.000913172],"domain_scores_gemma":[0.9919829,0.004794929,0.0001494194,0.0009914124,0.001606495,0.0004748352],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001109394,0.000122023,0.0003092278,0.001223034,0.00003528964,0.0002468947,0.005361625,0.000007086673,0.00004426149,0.000563506,0.9398518,0.05112582],"study_design_scores_gemma":[0.0003374389,0.003561736,0.0005085844,0.01606037,0.00009066681,0.00002473584,0.02137312,0.0006523482,0.0006215998,0.007026488,0.9493012,0.0004416693],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"editorial","genre_gemma":"empirical","genre_scores_codex":[0.1365335,0.02367183,0.00007532414,0.09227898,0.739196,0.003423773,0.0008043403,0.0001555537,0.003860737],"genre_scores_gemma":[0.5384659,0.002134184,0.00003984503,0.0005406778,0.4343598,0.0002024427,0.001016548,0.0001334008,0.0231071],"genre_candidate":"editorial","genre_consensus":null,"teacher_disagreement_score":0.4019325,"threshold_uncertainty_score":0.9997627,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2244390772779246,"score_gpt":0.5133344245479209,"score_spread":0.2888953472699963,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}