{"id":"W4388625750","doi":"10.1038/s41591-023-02656-2","title":"Reporting standards for the use of large language model-linked chatbots for health advice","year":2023,"lang":"en","type":"letter","venue":"Nature Medicine","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":53,"is_retracted":false,"has_abstract":false,"ca_institutions":"Impact; McMaster University; Hamilton General Hospital","funders":"Cancer Research UK","keywords":"Advice (programming); Medicine; Computer science; Psychology; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1778294,0.0008156057,0.001245374,0.004377089,0.005738451,0.01046017,0.008212505,0.0434442,0.01420223],"category_scores_gemma":[0.3819886,0.001606868,0.00209816,0.002660607,0.004792323,0.007173322,0.007448426,0.02977491,0.01796084],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.007382796,"about_ca_system_score_gemma":0.01748348,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01190654,"about_ca_topic_score_gemma":0.01244985,"domain_scores_codex":[0.8387421,0.07282066,0.02432919,0.005057068,0.04964252,0.009408546],"domain_scores_gemma":[0.3208892,0.4219499,0.0340138,0.05052146,0.1592906,0.01333503],"domain_codex":null,"domain_gemma":"reporting","domain_candidate":"reporting","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001777875,0.0002841497,0.002018514,0.0003256947,0.00003513811,0.00053027,0.001474714,0.0004157491,0.002620561,0.03069541,0.9225801,0.03884201],"study_design_scores_gemma":[0.0002561948,0.0003450766,0.006457932,0.002142833,0.00008050254,0.0006946007,0.002065342,0.009647315,0.005669747,0.02499275,0.9474237,0.0002240011],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"commentary","genre_gemma":"commentary","genre_scores_codex":[0.006646619,0.001333897,0.09673028,0.7533784,0.01996359,0.003802659,0.003854083,0.007153197,0.1071372],"genre_scores_gemma":[0.07987382,0.001186255,0.1335123,0.6443573,0.008637493,0.0120451,0.005931688,0.001747334,0.1127087],"genre_candidate":"commentary","genre_consensus":"commentary","teacher_disagreement_score":0.8221706,"threshold_uncertainty_score":0.9404629,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2725759343659866,"score_gpt":0.532773826188157,"score_spread":0.2601978918221705,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}