{"id":"W4405597099","doi":"10.1038/s41746-024-01366-4","title":"Probabilistic medical predictions of large language models","year":2024,"lang":"en","type":"article","venue":"npj Digital Medicine","topic":"Topic Modeling","field":"Computer Science","cited_by":34,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"National Institutes of Health","keywords":"Probabilistic logic; Computer science; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01295962,0.001355435,0.0008217496,0.002686749,0.0004855562,0.003130503,0.001507621,0.001452433,0.004004296],"category_scores_gemma":[0.07083464,0.0005729426,0.00147935,0.001412677,0.0007465183,0.003433575,0.001918511,0.002830012,0.002126048],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001183092,"about_ca_system_score_gemma":0.001546612,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003131337,"about_ca_topic_score_gemma":0.004493808,"domain_scores_codex":[0.993462,0.004331572,0.000372723,0.001018436,0.0005889118,0.0002263435],"domain_scores_gemma":[0.9189088,0.07457513,0.001952281,0.001950964,0.002099224,0.0005135749],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001673979,0.0003734933,0.04985631,0.001749667,0.0006398141,0.0008921573,0.001750987,0.4915099,0.004245404,0.0413312,0.04164989,0.3643273],"study_design_scores_gemma":[0.00008279405,0.00007221633,0.002097374,0.0001512088,0.0000803939,0.0001492086,0.0001132723,0.9155532,0.001559403,0.07574017,0.004357489,0.00004323892],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1193644,0.003058161,0.8511013,0.006734556,0.0004412564,0.0005153844,0.007988123,0.006157232,0.004639745],"genre_scores_gemma":[0.8023624,0.001226411,0.1808697,0.001273275,0.0005530397,0.0006218455,0.0101427,0.000617623,0.002333022],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01295962,"threshold_uncertainty_score":0.06853783,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01921885941239498,"score_gpt":0.2821147625464776,"score_spread":0.2628959031340826,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}