{"id":"W4391046210","doi":"10.5858/arpa.2023-0296-oa","title":"Assessment of Pathology Domain-Specific Knowledge of ChatGPT and Comparison to Human Performance","year":2024,"lang":"en","type":"article","venue":"Archives of Pathology & Laboratory Medicine","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":43,"is_retracted":false,"has_abstract":true,"ca_institutions":"London Health Sciences Centre; Western University","funders":"","keywords":"Human Pathology; Pathology; Domain (mathematical analysis); Medicine; Disease; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01101384,0.0006468475,0.0006172674,0.002444507,0.0005373158,0.001906326,0.000916415,0.001299626,0.002790831],"category_scores_gemma":[0.06232273,0.0001902244,0.0008954902,0.0007786323,0.0008075224,0.002119111,0.002587843,0.0007014357,0.0009692666],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001086059,"about_ca_system_score_gemma":0.0008052065,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003588232,"about_ca_topic_score_gemma":0.00540266,"domain_scores_codex":[0.9936727,0.002954457,0.0006000481,0.001174401,0.001280951,0.0003173178],"domain_scores_gemma":[0.9435641,0.0359314,0.007286201,0.003504508,0.007013022,0.002700781],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.003434937,0.001311456,0.7218308,0.001661995,0.0007555705,0.0007593814,0.02803166,0.009156269,0.01274802,0.0006109129,0.005628978,0.2140699],"study_design_scores_gemma":[0.0001415024,0.003005188,0.9539278,0.0002974323,0.0003054239,0.001074744,0.006017714,0.01951043,0.007407639,0.001294528,0.006798951,0.000218809],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9901343,0.0003277922,0.003840406,0.0001439005,0.00003748546,0.0003161779,0.0009579952,0.0003812437,0.003860614],"genre_scores_gemma":[0.9934518,0.0001324259,0.00370143,0.0000853044,0.0000292724,0.0002774214,0.001194204,0.00004336792,0.001084817],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01101384,"threshold_uncertainty_score":0.05824745,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0844600662526074,"score_gpt":0.4409778957139658,"score_spread":0.3565178294613585,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}