{"id":"W4408251534","doi":"10.2196/67244","title":"Large Language Models in Biochemistry Education: Comparative Evaluation of Performance","year":2025,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":19,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Preprint; Computer science; Course (navigation); Computational biology; Data science; Biology; World Wide Web; Engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009204653,0.00008875855,0.0001918373,0.0002194884,0.0000426695,0.000006086871,0.00009088572,0.0001673113,0.0005569847],"category_scores_gemma":[0.0005305733,0.00008517244,0.00003484074,0.0005756102,0.00006449493,0.0001175053,0.00001590007,0.0002316312,0.0000171285],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003647933,"about_ca_system_score_gemma":0.01389079,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002045266,"about_ca_topic_score_gemma":0.00007072091,"domain_scores_codex":[0.9985215,0.00008154896,0.0004679097,0.0002044011,0.000571169,0.0001534598],"domain_scores_gemma":[0.9987986,0.00006431134,0.0001013166,0.0002532662,0.0006590466,0.0001234777],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"qualitative","study_design_scores_codex":[0.000200957,0.006768581,0.05062249,0.001422045,0.00003196224,3.315784e-7,0.03718924,0.00006148493,0.001072094,0.006095186,0.01518086,0.8813547],"study_design_scores_gemma":[0.001681839,0.0004531642,0.1855546,0.01231027,0.0004028305,0.00004636724,0.2963253,0.2894701,0.1839236,0.01834946,0.01072386,0.0007586777],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9702067,0.001546021,0.0001322899,0.004225142,0.0007069958,0.0007080736,0.000001208069,0.00001614705,0.02245742],"genre_scores_gemma":[0.9967468,0.00007198184,0.0001451108,0.001178611,0.0002322663,0.0004850204,0.0001577143,0.000004822361,0.0009776964],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8805961,"threshold_uncertainty_score":0.9916995,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1161257107869459,"score_gpt":0.5239413793665126,"score_spread":0.4078156685795667,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}