{"id":"W4416830923","doi":"10.2196/80182","title":"Performance of ChatGPT-4o, Claude 3 Opus, and DeepSeek-R1 in BI-RADS Category 4 Classification and Malignancy Prediction From Mammography Reports: Retrospective Diagnostic Study","year":2025,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Radiomics and Machine Learning in Medical Imaging","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Mammography; Retrospective cohort study; Malignancy; Screening mammography; MEDLINE; BI-RADS","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008943807,0.0008910164,0.0008192628,0.002993145,0.0003077853,0.00131272,0.001377269,0.0009807064,0.000720424],"category_scores_gemma":[0.04088657,0.000397843,0.001086377,0.001273877,0.0007769435,0.001284547,0.001735826,0.0005887027,0.0004667573],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001004216,"about_ca_system_score_gemma":0.0007857865,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004581034,"about_ca_topic_score_gemma":0.004279964,"domain_scores_codex":[0.9944351,0.002282868,0.000767398,0.001401049,0.0007504188,0.0003631824],"domain_scores_gemma":[0.97105,0.01481542,0.005546445,0.002433932,0.004416634,0.001737548],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.001994226,0.0001247249,0.9818179,0.0001064453,0.0002376651,0.0001182659,0.0002684096,0.001634376,0.0003546429,0.0000442805,0.0006015418,0.01269748],"study_design_scores_gemma":[0.0001586236,0.001600479,0.9301547,0.0001050863,0.0005407882,0.001113129,0.0009561532,0.05991148,0.00345406,0.0003818113,0.001509362,0.0001143421],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9974595,0.000401371,0.0006605711,0.00005163156,0.00001502633,0.00005626993,0.001045842,0.00006603701,0.0002438781],"genre_scores_gemma":[0.9963356,0.00006976674,0.001085988,0.00002581293,0.00001515242,0.00006543905,0.00225786,0.00001635788,0.0001280934],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008943807,"threshold_uncertainty_score":0.04729998,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.00649036711079671,"score_gpt":0.2758797265391429,"score_spread":0.2693893594283462,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}