{"id":"W4386865122","doi":"10.1148/radiol.232082","title":"Evaluating Diagnostic Performance of ChatGPT in Radiology: Delving into Methods","year":2023,"lang":"en","type":"letter","venue":"Radiology","topic":"Radiomics and Machine Learning in Medical Imaging","field":"Medicine","cited_by":18,"is_retracted":false,"has_abstract":false,"ca_institutions":"Western University; London Health Sciences Centre","funders":"","keywords":"Medicine; Radiology; Medical imaging; Medical physics; Nuclear medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","metaepi_narrow","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.003927556,0.0004096947,0.00175103,0.0008451857,0.00007371425,0.000007974779,0.000390825,0.001188359,0.0001010843],"category_scores_gemma":[0.01507582,0.0003640294,0.0002178523,0.0004906965,0.0004907432,0.00004395968,0.0001680106,0.004525636,0.00004507186],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002534677,"about_ca_system_score_gemma":0.0003159788,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002540472,"about_ca_topic_score_gemma":0.000003085669,"domain_scores_codex":[0.9957735,0.001362282,0.001031776,0.0007367005,0.0003070707,0.0007886885],"domain_scores_gemma":[0.9913796,0.007357215,0.0004520677,0.0006092406,0.00009938596,0.0001024797],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001921273,0.00007943918,0.07949217,0.004876986,0.0006945122,0.003466501,0.002678528,0.003721671,0.008706069,0.00007162695,0.5205762,0.3754442],"study_design_scores_gemma":[0.006323577,0.004036815,0.05265954,0.005292465,0.001144983,0.00691309,0.00014809,0.6026488,0.0004717733,0.00181074,0.316938,0.00161207],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"commentary","genre_scores_codex":[0.5719667,0.006858321,0.003011243,0.412443,0.003650303,0.001075412,0.000007042826,0.0002419849,0.0007459811],"genre_scores_gemma":[0.08077362,0.01029391,0.2523728,0.6330885,0.01687476,0.0005950414,0.001516387,0.000703026,0.003782003],"genre_candidate":"commentary","genre_consensus":null,"teacher_disagreement_score":0.5989271,"threshold_uncertainty_score":0.9998811,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03925053290297194,"score_gpt":0.4176554604831245,"score_spread":0.3784049275801525,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}