{"id":"W4409062435","doi":"10.1148/radiol.241554","title":"Accuracy of Large Language Model–based Automatic Calculation of Ovarian-Adnexal Reporting and Data System MRI Scores from Pelvic MRI Reports","year":2025,"lang":"en","type":"article","venue":"Radiology","topic":"Ovarian cancer diagnosis and treatment","field":"Medicine","cited_by":17,"is_retracted":false,"has_abstract":true,"ca_institutions":"Sinai Health System; Princess Margaret Cancer Centre; University Health Network; University of Toronto; Toronto General Hospital; Women's College Hospital; Mount Sinai Hospital","funders":"","keywords":"Medicine; Adnexal Diseases; Feature (linguistics); Radiology; Adnexal mass; Natural language processing; Artificial intelligence; Medical physics; Computer science; Linguistics; Laparoscopy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01667804,0.001023722,0.000523576,0.001683434,0.0003634763,0.002775848,0.001091353,0.001290401,0.00579756],"category_scores_gemma":[0.05016102,0.0003473011,0.001489928,0.0004425303,0.0005206542,0.002023319,0.00180978,0.0009669606,0.005582156],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007287515,"about_ca_system_score_gemma":0.0011895,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006095788,"about_ca_topic_score_gemma":0.008231258,"domain_scores_codex":[0.9924584,0.004016164,0.0006260147,0.001625785,0.0009896997,0.000284019],"domain_scores_gemma":[0.971816,0.02152457,0.001557712,0.002477664,0.001984769,0.0006392624],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.004505802,0.0006820039,0.4139535,0.0006231297,0.001812031,0.0003622653,0.0007913579,0.02395226,0.01209691,0.0007111172,0.03593288,0.5045766],"study_design_scores_gemma":[0.0004177415,0.001289078,0.2922525,0.0005191243,0.001549736,0.001756591,0.001080903,0.6459063,0.0263372,0.006226298,0.02234038,0.0003241843],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.810689,0.002814263,0.1324762,0.003767207,0.001717072,0.0006057887,0.01180279,0.01965779,0.01646992],"genre_scores_gemma":[0.9678514,0.0002139812,0.02187137,0.0007929143,0.000144444,0.00009292248,0.004432342,0.0004655052,0.004135055],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01667804,"threshold_uncertainty_score":0.08820295,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02529881637993705,"score_gpt":0.3452238032290642,"score_spread":0.3199249868491272,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}