{"id":"W4414959082","doi":"10.3390/diagnostics15192536","title":"Performance of ChatGPT-4o in Determining Radiology–Pathology Concordance and Management Recommendations Following Image-Guided Breast Biopsies","year":2025,"lang":"en","type":"article","venue":"Diagnostics","topic":"Radiomics and Machine Learning in Medical Imaging","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; Sunnybrook Health Science Centre","funders":"","keywords":"Concordance; Gold standard (test); Biopsy; Conservative management; Breast imaging; Multidisciplinary approach; Retrospective cohort study","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03929957,0.0008605259,0.001021911,0.004219305,0.001055086,0.00275457,0.001627427,0.0009677044,0.001920409],"category_scores_gemma":[0.1091373,0.0004998537,0.00199384,0.0017574,0.001445765,0.001549955,0.004463446,0.0007851173,0.0005606185],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002943074,"about_ca_system_score_gemma":0.003885929,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005494915,"about_ca_topic_score_gemma":0.008295948,"domain_scores_codex":[0.9709418,0.01808378,0.003232086,0.002397437,0.004504026,0.0008409361],"domain_scores_gemma":[0.8605178,0.09427059,0.02279606,0.004568262,0.01536713,0.002480228],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002369172,0.0002780078,0.9479997,0.0005570783,0.0004276175,0.0002876948,0.003069991,0.00315138,0.001247379,0.000340227,0.001348355,0.03892332],"study_design_scores_gemma":[0.0004356409,0.002703298,0.8863558,0.0006708239,0.0007733007,0.002221013,0.006362736,0.08774286,0.005784353,0.001742729,0.004913109,0.0002943593],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.987866,0.0003284697,0.007289567,0.0003401763,0.00007102137,0.0006058674,0.0007235754,0.0002266018,0.002548611],"genre_scores_gemma":[0.9845508,0.00008806117,0.01361462,0.000106723,0.00003877825,0.0004947389,0.0006960342,0.00003972288,0.0003706679],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03929957,"threshold_uncertainty_score":0.2078384,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.009463016648554454,"score_gpt":0.3088086472987807,"score_spread":0.2993456306502262,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}