{"id":"W4388237833","doi":"10.2196/47532","title":"The Accuracy and Potential Racial and Ethnic Biases of GPT-4 in the Diagnosis and Triage of Health Conditions: Evaluation Study","year":2023,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":64,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Institute on Minority Health and Health Disparities; National Institute on Aging; National Institutes of Health","keywords":"Triage; Ethnic group; Medicine; Medical diagnosis; Health care; Race (biology); Family medicine; Emergency medicine; Pathology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.004012982,0.00006063897,0.0001649965,0.0001332984,0.0001582499,0.00001260186,0.00005237843,0.00006330181,0.00004636514],"category_scores_gemma":[0.008908394,0.00003791816,0.00001680965,0.0004367655,0.0002060434,0.00007080876,0.0000187559,0.0001536538,0.000001323819],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003188818,"about_ca_system_score_gemma":0.00120818,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00114219,"about_ca_topic_score_gemma":0.0004480973,"domain_scores_codex":[0.9981432,0.000561109,0.0005271499,0.0001447186,0.0005066847,0.0001171541],"domain_scores_gemma":[0.9973649,0.002038155,0.0001961626,0.0001486197,0.0001662659,0.00008588551],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.00007165211,0.0008068368,0.06968419,0.0001416205,0.00001198591,4.87505e-7,0.04273166,9.348338e-7,0.00002293172,0.0001541284,0.00229286,0.8840807],"study_design_scores_gemma":[0.0002527311,0.0004878899,0.9083008,0.0002671342,0.00004338965,0.000008052358,0.08770157,0.0007661027,0.00006112033,0.001876678,0.0001990609,0.0000354271],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9617308,0.002896812,0.000003060833,0.03319751,0.0003180662,0.001831179,0.000002723013,0.000006996837,0.00001283245],"genre_scores_gemma":[0.9938652,0.004646293,0.000005541714,0.000496938,0.0001729916,0.0007495667,0.0000498929,0.000004390338,0.000009212233],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8840453,"threshold_uncertainty_score":0.99944,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3388489175866713,"score_gpt":0.5900203259822512,"score_spread":0.2511714083955799,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}