{"id":"W4417474482","doi":"10.2196/77357","title":"Performance of DeepSeek-R1, ChatGPT (GPT-o3-mini), and Gemini 2.0 Flash on German Medical Multiple-Choice Questions: Comparative Evaluation","year":2025,"lang":"en","type":"article","venue":"JMIR Formative Research","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"German; Flash (photography); German government","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.008075246,0.001488241,0.001393903,0.001367319,0.0003661973,0.001171887,0.001410441,0.001394984,0.004214645],"category_scores_gemma":[0.03236056,0.0003185469,0.001070474,0.0006083458,0.0005573255,0.001880545,0.002354721,0.00129717,0.001596731],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001324972,"about_ca_system_score_gemma":0.001163657,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003641083,"about_ca_topic_score_gemma":0.004228864,"domain_scores_codex":[0.994793,0.002655113,0.0004009252,0.0009622549,0.0008825904,0.0003061849],"domain_scores_gemma":[0.9687241,0.02516487,0.001206729,0.0008360576,0.001673589,0.00239471],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.05385781,0.009500078,0.1258011,0.00763219,0.001689399,0.0004051097,0.007664108,0.03260693,0.0132835,0.001557152,0.024236,0.7217667],"study_design_scores_gemma":[0.004597173,0.04280687,0.5259245,0.001073762,0.001876653,0.001077617,0.004422836,0.3666495,0.02137488,0.004797146,0.02477066,0.0006283381],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9805214,0.001231267,0.006180347,0.0003038213,0.0001517069,0.0009431249,0.002358497,0.002631312,0.005678517],"genre_scores_gemma":[0.9732761,0.0005888438,0.01430856,0.0002916957,0.000101081,0.00119143,0.00571644,0.0001691104,0.004356806],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9919248,"threshold_uncertainty_score":0.04270655,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3260883900408559,"score_gpt":0.6023974656951113,"score_spread":0.2763090756542554,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}