{"id":"W7116867186","doi":"10.2196/71178","title":"Evaluating Multiple Input Strategies of Large Language Models for Gallbladder Polyps on Ultrasound: Comparative Study","year":2025,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Cholangiocarcinoma and Gallbladder Cancer Studies","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Gallbladder; Medical imaging; MEDLINE; Unified Medical Language System; Medical diagnosis","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02065511,0.001608941,0.001057647,0.00263952,0.0005869418,0.002218836,0.001229261,0.001181477,0.002267182],"category_scores_gemma":[0.1352612,0.0005133922,0.002130429,0.001025825,0.001007756,0.002886052,0.002337789,0.001013872,0.0009491317],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001628983,"about_ca_system_score_gemma":0.001127358,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002483094,"about_ca_topic_score_gemma":0.003092753,"domain_scores_codex":[0.9814832,0.01239067,0.001802627,0.001940931,0.001981797,0.0004008323],"domain_scores_gemma":[0.8157846,0.1546849,0.007370322,0.005704639,0.01437714,0.002078425],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0296027,0.007277949,0.532867,0.004124116,0.00318893,0.001230305,0.02286649,0.01585878,0.01321949,0.0007915112,0.003499326,0.3654733],"study_design_scores_gemma":[0.002533719,0.03900132,0.6036438,0.00112069,0.005674517,0.003130977,0.01722066,0.2776191,0.03830164,0.003985618,0.006969857,0.0007980858],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.990613,0.0005265007,0.006712994,0.00007260276,0.0000374333,0.0005658827,0.0003052079,0.0002348425,0.0009315109],"genre_scores_gemma":[0.9826194,0.0002382439,0.01538045,0.00007602106,0.0000501635,0.0005157159,0.0006324184,0.00009363332,0.0003938796],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02065511,"threshold_uncertainty_score":0.1092359,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06158663072491496,"score_gpt":0.4195059681936868,"score_spread":0.3579193374687719,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}