{"id":"W4415291590","doi":"10.2196/75556","title":"Automated Esophageal Cancer Staging From Free-Text Radiology Reports: Large Language Model Evaluation Study","year":2025,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Esophageal Cancer Research and Treatment","field":"Medicine","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Esophageal cancer; Verifiable secret sharing; Cancer staging; Neoplasm staging; Lung cancer staging; Gold standard (test)","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.00828112,0.001729049,0.001150722,0.002894487,0.0004455887,0.001426544,0.001408541,0.001095903,0.001189906],"category_scores_gemma":[0.02077085,0.0003432395,0.001872335,0.001594694,0.0004982425,0.001462127,0.001325874,0.001301636,0.0007033176],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001398754,"about_ca_system_score_gemma":0.00123738,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00961638,"about_ca_topic_score_gemma":0.006937932,"domain_scores_codex":[0.9954616,0.002349931,0.0005189068,0.0009147679,0.0005361435,0.0002186277],"domain_scores_gemma":[0.9837387,0.01154352,0.001085688,0.001512952,0.001473741,0.0006454102],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.006938249,0.004483141,0.4653252,0.001822631,0.003020423,0.001891453,0.001176264,0.0907243,0.005446136,0.0007529947,0.02402327,0.3943959],"study_design_scores_gemma":[0.0004110517,0.001847087,0.08617757,0.0001360993,0.0007912604,0.001020633,0.0006148756,0.9018474,0.003173812,0.0009471592,0.002916883,0.0001161895],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.975651,0.00262286,0.01327946,0.0005807418,0.0001453252,0.0003407154,0.005575457,0.0009700409,0.0008344137],"genre_scores_gemma":[0.96733,0.0005319243,0.01185289,0.0001581263,0.0001062356,0.0002061735,0.0193245,0.00005183341,0.0004384007],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9917189,"threshold_uncertainty_score":0.04379529,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02236905037101426,"score_gpt":0.4090225205560591,"score_spread":0.3866534701850448,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}