{"id":"W4387206585","doi":"10.1016/j.ijrobp.2023.06.1727","title":"Evaluating the Performance of ChatGPT at Breast Tumor Board","year":2023,"lang":"en","type":"article","venue":"International Journal of Radiation Oncology*Biology*Physics","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":3,"is_retracted":false,"has_abstract":false,"ca_institutions":"McGill University Health Centre; University of Calgary","funders":"","keywords":"CLARITY; Medicine; Wilcoxon signed-rank test; Medical physics; Test (biology); Human breast; Institutional review board; Medical education; Breast cancer; Internal medicine; Surgery; Cancer","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001566321,0.0004074166,0.0003863701,0.0006602502,0.0003671414,0.0007244581,0.0007578579,0.0007170358,0.004709507],"category_scores_gemma":[0.008816937,0.0001267792,0.0003132444,0.0004710174,0.0002163847,0.0006393544,0.0008056156,0.000603102,0.001698572],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006576828,"about_ca_system_score_gemma":0.0008911077,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01002925,"about_ca_topic_score_gemma":0.009823876,"domain_scores_codex":[0.9990942,0.000352094,0.00005936317,0.0001822929,0.0001756943,0.0001363583],"domain_scores_gemma":[0.9947497,0.002659491,0.0003230885,0.0002546561,0.000976184,0.001036813],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.01257199,0.003910747,0.3512742,0.0005222411,0.0004095046,0.0008194597,0.001184403,0.06093865,0.01146821,0.0007465415,0.03304778,0.5231063],"study_design_scores_gemma":[0.0009183433,0.01181237,0.5534758,0.0001241804,0.0005968859,0.001146225,0.003127833,0.3835877,0.02094179,0.001242236,0.02289565,0.0001309075],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9906156,0.000336646,0.003227544,0.0003163271,0.0001488546,0.0001183678,0.0008731223,0.0008690414,0.003494392],"genre_scores_gemma":[0.9906495,0.0001120763,0.004818099,0.00008662114,0.00005135604,0.00004805219,0.001824054,0.00005078833,0.002359486],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01002925,"threshold_uncertainty_score":0.01994175,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1275671715146966,"score_gpt":0.4650451728030451,"score_spread":0.3374780012883485,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}