{"id":"W4388025354","doi":"10.2196/51523","title":"Evaluating Large Language Models for the National Premedical Exam in India: Comparative Analysis of GPT-3.5, GPT-4, and Bard","year":2023,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":57,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Mainstream; Test (biology); Artificial intelligence; Psychology; Mathematics education; Computer science; Political science; Biology; Ecology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002344505,0.00008393075,0.0002707846,0.0004240887,0.00009350573,0.000009716106,0.00008852699,0.0001444268,0.0002376105],"category_scores_gemma":[0.002892245,0.00006174815,0.0000686367,0.00130727,0.00011074,0.00007965857,0.00002589938,0.0002303159,0.000006730897],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001073625,"about_ca_system_score_gemma":0.001728922,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002282717,"about_ca_topic_score_gemma":0.0001831733,"domain_scores_codex":[0.9981589,0.0001065547,0.0004932911,0.0002211689,0.0008144457,0.0002056312],"domain_scores_gemma":[0.9979131,0.001324884,0.0001197656,0.0001234476,0.0003484545,0.0001703042],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008137234,0.003860273,0.1320061,0.001185229,0.001257196,0.000004276205,0.3458239,0.001792597,0.0009075212,0.02151157,0.04089039,0.4499472],"study_design_scores_gemma":[0.0002431474,0.0001682422,0.2629004,0.0002008868,0.0002761305,0.000003468562,0.03078631,0.7019994,0.0001962868,0.002430182,0.0007048988,0.00009062752],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9889512,0.0003662237,0.0007082673,0.008198158,0.0002829272,0.0009847607,0.00001403826,0.00002516348,0.0004692658],"genre_scores_gemma":[0.9968135,0.00008565018,0.0002614833,0.001309981,0.0003719677,0.0006837675,0.0003080265,0.000006369081,0.0001591986],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7002068,"threshold_uncertainty_score":0.3462497,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2911824033239588,"score_gpt":0.5812372745878501,"score_spread":0.2900548712638913,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}