{"id":"W4416222729","doi":"10.2196/73469","title":"Evaluating the Performance of DeepSeek-R1 and DeepSeek-V3 Versus OpenAI Models in the Chinese National Medical Licensing Examination: Cross-Sectional Comparative Study","year":2025,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Academic integrity and plagiarism","field":"Social Sciences","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Licensure","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01227396,0.0001002628,0.0001503635,0.0001061028,0.0007449698,0.00008129619,0.0005004768,0.0003484007,0.0002070442],"category_scores_gemma":[0.00582799,0.00006341447,0.0000296553,0.0007155316,0.0006519642,0.0003618663,0.00007297118,0.00115383,0.000004347924],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001729887,"about_ca_system_score_gemma":0.003509265,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008568965,"about_ca_topic_score_gemma":0.001460205,"domain_scores_codex":[0.9954417,0.001243416,0.0004323809,0.000225496,0.002482484,0.0001745117],"domain_scores_gemma":[0.996555,0.002629407,0.0001404365,0.0001041279,0.00048942,0.00008164329],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"observational","study_design_scores_codex":[0.0003993207,0.0019325,0.3908846,0.00007477812,0.00009102599,0.000001270321,0.4538614,0.0005921033,0.00000806858,0.1008263,0.001989449,0.04933922],"study_design_scores_gemma":[0.0009590117,0.0001255943,0.8495862,0.0001151187,0.00001202425,0.000004113785,0.04186536,0.1035266,0.000002074082,0.003341588,0.0003720416,0.00009026028],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9768704,0.0001601705,0.00009877437,0.006869511,0.0008161585,0.0006781913,0.000001277169,0.0000143132,0.01449125],"genre_scores_gemma":[0.9980798,0.00004592491,0.00007758229,0.0009779405,0.0004598529,0.0001523614,0.00001194948,0.000003274943,0.0001913605],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4587016,"threshold_uncertainty_score":0.6977069,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09535028815819191,"score_gpt":0.5095570624048554,"score_spread":0.4142067742466635,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}