{"id":"W4400316892","doi":"10.1038/s41746-024-01180-y","title":"The long but necessary road to responsible use of large language models in healthcare research","year":2024,"lang":"en","type":"editorial","venue":"npj Digital Medicine","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":26,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"University of Toronto","keywords":"Dissemination; Data extraction; Health care; Computer science; Data science; Risk analysis (engineering); Psychology; Cognitive psychology; Medicine; MEDLINE; Political science","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02486363,0.002305221,0.003320393,0.003389052,0.002880122,0.00917445,0.003989668,0.01657563,0.008808427],"category_scores_gemma":[0.101399,0.001288048,0.0026039,0.00165682,0.004956106,0.008943714,0.002295883,0.032608,0.007741976],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002917588,"about_ca_system_score_gemma":0.005331983,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001687588,"about_ca_topic_score_gemma":0.004931806,"domain_scores_codex":[0.9870732,0.00396273,0.00217602,0.0008773932,0.005610635,0.0002999657],"domain_scores_gemma":[0.812614,0.1443042,0.004156962,0.00482823,0.02875664,0.005339946],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00002360066,0.00001053109,0.00002530079,0.0003602692,0.00003505311,0.00008756883,0.00004845978,0.0000468589,0.00004117181,0.001273761,0.9849753,0.01307216],"study_design_scores_gemma":[0.0000513936,0.00001819664,0.0001309491,0.001223803,0.0000682804,0.0002307716,0.00008866944,0.0003288367,0.000113294,0.007762479,0.9899503,0.00003305016],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"editorial","genre_gemma":"editorial","genre_scores_codex":[0.00004537467,0.009920545,0.001068963,0.1385629,0.8491217,0.00002352149,0.0000621432,0.0001416889,0.00105323],"genre_scores_gemma":[0.0005120345,0.008244256,0.0009730839,0.04448377,0.9420456,0.00004019558,0.00002855106,0.00008252845,0.003589916],"genre_candidate":"editorial","genre_consensus":"editorial","teacher_disagreement_score":0.9751363,"threshold_uncertainty_score":0.131493,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2244390772779246,"score_gpt":0.5133344245479209,"score_spread":0.2888953472699963,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}