{"id":"W4409062435","doi":"10.1148/radiol.241554","title":"Accuracy of Large Language Model–based Automatic Calculation of Ovarian-Adnexal Reporting and Data System MRI Scores from Pelvic MRI Reports","year":2025,"lang":"en","type":"article","venue":"Radiology","topic":"Ovarian cancer diagnosis and treatment","field":"Medicine","cited_by":17,"is_retracted":false,"has_abstract":true,"ca_institutions":"Sinai Health System; Princess Margaret Cancer Centre; University Health Network; University of Toronto; Toronto General Hospital; Women's College Hospital; Mount Sinai Hospital","funders":"","keywords":"Medicine; Adnexal Diseases; Feature (linguistics); Radiology; Adnexal mass; Natural language processing; Artificial intelligence; Medical physics; Computer science; Linguistics; Laparoscopy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006346474,0.0001245937,0.0006697655,0.0001197641,0.00003469351,0.000005701635,0.0000764744,0.0001153277,0.00003116341],"category_scores_gemma":[0.0007559757,0.0001021498,0.00005826269,0.0001279285,0.0000521275,0.0000586899,0.00008609991,0.00008054775,6.192615e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008677228,"about_ca_system_score_gemma":0.0003153613,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0007876236,"about_ca_topic_score_gemma":0.00005414645,"domain_scores_codex":[0.998121,0.0000852349,0.001091081,0.0004198583,0.0001246009,0.0001582812],"domain_scores_gemma":[0.9973258,0.0003258576,0.001204472,0.001029668,0.00006507007,0.00004907339],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002364248,0.0005461851,0.9671429,0.002351583,0.001212984,0.001881194,0.001329553,0.002266652,0.007571424,0.00267952,0.003355767,0.009425814],"study_design_scores_gemma":[0.002620364,0.0001286066,0.3539271,0.001308077,0.0008785733,0.0002813007,0.0002706082,0.636779,0.00350349,0.0001113412,0.00007893043,0.0001125793],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9812284,0.003042094,0.01382833,0.0006120574,0.0001898586,0.0005047062,0.0001356218,0.00005236789,0.0004065624],"genre_scores_gemma":[0.9910812,0.00002973324,0.00806588,0.00009133969,0.0000522787,0.00002258726,0.0006129411,0.00001090569,0.0000330924],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6345124,"threshold_uncertainty_score":0.4165548,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02529881637993705,"score_gpt":0.3452238032290642,"score_spread":0.3199249868491272,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}