{"id":"W4400974494","doi":"10.1101/2024.07.24.24310930","title":"The TRIPOD-LLM Statement: A Targeted Guideline For Reporting Large Language Models Use","year":2024,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":22,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Tripod (photography); Guideline; Statement (logic); Computer science; Medicine; Political science; Engineering; Mechanical engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004412655,0.0002596596,0.0004433217,0.0001331287,0.0002550411,0.0001565433,0.0001776551,0.0002567084,0.00005525528],"category_scores_gemma":[0.00532174,0.0001769842,0.0002874678,0.0001773301,0.00004136329,0.00006136263,0.0002683027,0.0007784029,0.00004130965],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000205898,"about_ca_system_score_gemma":0.0008454691,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001227975,"about_ca_topic_score_gemma":0.0004487211,"domain_scores_codex":[0.9954714,0.00007819525,0.002979235,0.0005782483,0.0003606723,0.0005322814],"domain_scores_gemma":[0.9967501,0.0003940405,0.001190998,0.0008532929,0.0006421164,0.000169469],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001588086,0.001311114,0.01755537,0.009822582,0.001621864,0.0006811667,0.06623712,0.004252918,0.005643157,0.006360746,0.6232886,0.2616372],"study_design_scores_gemma":[0.0002439661,0.0004043613,0.0003531196,0.002030988,0.0008562879,0.00005035907,0.01299298,0.4211707,0.02214731,0.06278401,0.4761826,0.0007833402],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.942185,0.009120633,0.01690685,0.02293996,0.004810836,0.003090442,0.0002244193,0.0003061963,0.0004156637],"genre_scores_gemma":[0.9827607,0.0006394592,0.005076786,0.001329683,0.001768332,0.0007239526,0.0009246751,0.00009861758,0.006677835],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4169178,"threshold_uncertainty_score":0.7217208,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2488992623115137,"score_gpt":0.4834255985225807,"score_spread":0.234526336211067,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}