{"id":"W4400549596","doi":"10.2196/51282","title":"Assessing GPT-4’s Performance in Delivering Medical Advice: Comparative Analysis With Human Experts","year":2024,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":34,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Medical diagnosis; CLARITY; Computer science; Proxy (statistics); Grading (engineering); Health care; Vocabulary; Medical education; Medicine; Psychology; Machine learning; Pathology; Engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0006025865,0.0001441493,0.0003112142,0.0005350019,0.0001213292,0.00008048889,0.0001124015,0.0002138039,0.002069115],"category_scores_gemma":[0.0002019569,0.000111573,0.00006290402,0.001464455,0.0001376217,0.0003813867,0.00001975953,0.0005098202,0.00005387095],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003460123,"about_ca_system_score_gemma":0.003678582,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001075388,"about_ca_topic_score_gemma":0.001046501,"domain_scores_codex":[0.997779,0.00007865856,0.0005174084,0.0003712747,0.0009792191,0.0002744496],"domain_scores_gemma":[0.9989364,0.0001547168,0.00005810669,0.0002056798,0.0001682502,0.0004768297],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0001181786,0.001765959,0.4380389,0.001056044,0.000296915,0.000101762,0.04484671,0.00005294387,0.0001471416,0.0009529629,0.003847368,0.5087751],"study_design_scores_gemma":[0.0003033246,0.0007155453,0.7241756,0.009429233,0.000559739,0.00028977,0.05382488,0.1895544,0.001695423,0.0002114758,0.0185624,0.0006781644],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9871023,0.0006514859,0.000289771,0.00689829,0.0004974501,0.0003196657,2.429162e-7,0.0001150547,0.004125737],"genre_scores_gemma":[0.9971732,0.0001050589,0.0002765984,0.001126311,0.0007202211,0.0002487765,0.0001051409,0.00001263499,0.0002320874],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.5080969,"threshold_uncertainty_score":0.9988431,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1181307132958916,"score_gpt":0.5247994825716034,"score_spread":0.4066687692757117,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}