{"id":"W4388625750","doi":"10.1038/s41591-023-02656-2","title":"Reporting standards for the use of large language model-linked chatbots for health advice","year":2023,"lang":"en","type":"letter","venue":"Nature Medicine","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":53,"is_retracted":false,"has_abstract":false,"ca_institutions":"Impact; McMaster University; Hamilton General Hospital","funders":"Cancer Research UK","keywords":"Advice (programming); Medicine; Computer science; Psychology; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.00442793,0.0003003918,0.00110029,0.0002631809,0.0002461499,0.00001086051,0.000181582,0.001268199,0.0000358641],"category_scores_gemma":[0.02027131,0.0001835155,0.0002528262,0.0003736048,0.0001111894,0.00005211216,0.00002932324,0.002561219,0.00000163358],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003273918,"about_ca_system_score_gemma":0.001825885,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008890954,"about_ca_topic_score_gemma":0.0003871009,"domain_scores_codex":[0.9957232,0.00005974014,0.002088848,0.0004894108,0.0009850451,0.0006537967],"domain_scores_gemma":[0.992334,0.00279621,0.002318011,0.0007673165,0.001662501,0.0001219289],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0002326479,0.00002341482,0.00004259733,0.003993489,0.0001030876,0.0000206056,0.004202277,0.00003290688,0.00004839534,0.00005683429,0.9770446,0.01419913],"study_design_scores_gemma":[0.0002839548,0.0006957669,0.0000477272,0.002512313,0.0003283391,0.00002342069,0.001374988,0.01984035,0.0002524595,0.0002926687,0.9741951,0.0001529423],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"commentary","genre_gemma":"commentary","genre_scores_codex":[0.0009211455,0.007025654,0.01766093,0.9663292,0.003289947,0.003619873,0.0009666492,0.0001540925,0.0000325619],"genre_scores_gemma":[0.007728032,0.001013495,0.003370473,0.9519502,0.02201643,0.0005153291,0.00420231,0.0001755502,0.009028195],"genre_candidate":"commentary","genre_consensus":"commentary","teacher_disagreement_score":0.01980744,"threshold_uncertainty_score":0.9997399,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2725759343659866,"score_gpt":0.532773826188157,"score_spread":0.2601978918221705,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}