{"id":"W7104259919","doi":"10.7910/dvn/kvamlj","title":"Performance of GPT-4o and o1-Pro on United Kingdom Medical Licensing Assessment-style items: a comparative study","year":2025,"lang":"","type":"dataset","venue":"Harvard Dataverse","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"St. Thomas Hospital","funders":"","keywords":"Test (biology); Set (abstract data type); Kingdom; Empirical research; Licensure","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004858127,0.002238407,0.001467923,0.002300977,0.000659968,0.001732519,0.002355793,0.002252295,0.00782331],"category_scores_gemma":[0.02191847,0.0004231893,0.001416169,0.001827184,0.000586747,0.00280607,0.002908001,0.001515189,0.008022928],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001642919,"about_ca_system_score_gemma":0.001831047,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02772561,"about_ca_topic_score_gemma":0.03275156,"domain_scores_codex":[0.9967942,0.001429222,0.0002870051,0.0008135746,0.0004226604,0.0002534104],"domain_scores_gemma":[0.9886063,0.007130809,0.0003810716,0.001565609,0.001543046,0.0007732056],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.008888071,0.002862167,0.05657013,0.002700508,0.001245337,0.0006350797,0.001126829,0.07295571,0.00551714,0.0009653479,0.2327,0.6138338],"study_design_scores_gemma":[0.002997897,0.00477313,0.1200063,0.000715179,0.0009157971,0.001230038,0.002684625,0.7628707,0.01198115,0.004072416,0.08734462,0.00040814],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"dataset","genre_scores_codex":[0.871556,0.005697155,0.01164726,0.001744215,0.0007984209,0.0007372238,0.06135639,0.02594657,0.02051673],"genre_scores_gemma":[0.6945395,0.001045684,0.0322539,0.0008651398,0.0002319582,0.000765995,0.2570282,0.001361995,0.01190744],"genre_candidate":"dataset","genre_consensus":null,"teacher_disagreement_score":0.02772561,"threshold_uncertainty_score":0.05512846,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05761704830978338,"score_gpt":0.3551258816426285,"score_spread":0.2975088333328451,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}