{"id":"W4406002105","doi":"10.1007/s00330-024-11281-7","title":"Reply to Letter to the Editor: “Comparative analysis of GPT-4 based ChatGPT’s diagnostic performance with radiologists using real-world radiology reports of brain tumors”","year":2025,"lang":"en","type":"letter","venue":"European Radiology","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Medicine; Neuroradiology; Interventional radiology; Radiology; Medical physics; Ultrasound; General surgery; Neurology; Psychiatry","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.005055641,0.0008005404,0.00164505,0.001373979,0.002976333,0.003894721,0.001767515,0.04195342,0.004891555],"category_scores_gemma":[0.04762984,0.0009209101,0.001145839,0.001236261,0.00185118,0.002467597,0.00117412,0.02542934,0.007518989],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004430847,"about_ca_system_score_gemma":0.005433322,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0070256,"about_ca_topic_score_gemma":0.01052923,"domain_scores_codex":[0.9954005,0.001224815,0.0009015285,0.0006748863,0.001180223,0.0006179968],"domain_scores_gemma":[0.978335,0.01117719,0.001911871,0.0004840552,0.005748606,0.002343225],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00009710839,0.000029666,0.001333187,0.00005157147,0.00002289848,0.001180124,0.0001420565,0.00003829295,0.0001479319,0.0003986778,0.9932248,0.003333597],"study_design_scores_gemma":[0.0003093962,0.0002126528,0.008943758,0.0007886548,0.0001637244,0.004199614,0.001618257,0.00105842,0.0008604714,0.005045293,0.9765574,0.0002421377],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"commentary","genre_gemma":"commentary","genre_scores_codex":[0.0007380732,0.0005213062,0.00008618093,0.9746722,0.02297401,0.00001591064,0.0001638676,0.00003694723,0.0007914421],"genre_scores_gemma":[0.003613619,0.0003801522,0.000196617,0.9642888,0.02892235,0.00004254217,0.00006920187,0.00003139029,0.002455355],"genre_candidate":"commentary","genre_consensus":"commentary","teacher_disagreement_score":0.9949443,"threshold_uncertainty_score":0.03214818,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08716996453427181,"score_gpt":0.37412418159377,"score_spread":0.2869542170594982,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}