{"id":"W4410382902","doi":"10.1136/gutjnl-2025-335091","title":"Large language models for detecting colorectal polyps in endoscopic images","year":2025,"lang":"en","type":"article","venue":"Gut","topic":"Colorectal Cancer Screening and Detection","field":"Medicine","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"Norges Forskningsråd; European Commission","keywords":"Colorectal Polyp; Medicine; Computer science; Colonoscopy; Rectal Polyp; Radiology; Artificial intelligence; Natural language processing; Colorectal cancer; Rectum; Internal medicine; Cancer","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002634864,0.002489775,0.001791606,0.003369217,0.0007731318,0.002171064,0.002008927,0.003027962,0.005978783],"category_scores_gemma":[0.00827177,0.0007781605,0.003413356,0.00155395,0.0005152909,0.002192724,0.001508001,0.003320161,0.006005653],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001471398,"about_ca_system_score_gemma":0.001321422,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02140841,"about_ca_topic_score_gemma":0.02177851,"domain_scores_codex":[0.9984275,0.0006077959,0.0001027994,0.0004918349,0.0001726565,0.0001973255],"domain_scores_gemma":[0.9949328,0.003984966,0.0002140521,0.0003224883,0.0003839337,0.0001617],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.005419223,0.001320261,0.02119165,0.001378415,0.001768207,0.002427456,0.0004223313,0.1993376,0.013588,0.003722992,0.09489463,0.6545293],"study_design_scores_gemma":[0.00009420813,0.0001447054,0.002826948,0.0000841557,0.0001855124,0.0003152968,0.000078692,0.9874521,0.001416846,0.00440589,0.002952795,0.00004287993],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3739822,0.03900776,0.4625601,0.01644766,0.003127548,0.001031218,0.04225617,0.04744441,0.01414288],"genre_scores_gemma":[0.8334108,0.003866029,0.1050943,0.001964357,0.001411341,0.0006836838,0.03875922,0.001099076,0.01371122],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02140841,"threshold_uncertainty_score":0.04256761,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01333324211269625,"score_gpt":0.3012438695644705,"score_spread":0.2879106274517743,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}