{"id":"W4413033493","doi":"10.2196/78320","title":"Impact of Prompt Engineering on the Performance of ChatGPT Variants Across Different Question Types in Medical Student Examinations: Cross-Sectional Study","year":2025,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Preprint; Medical education; Psychology; Engineering; Medicine; Computer science; World Wide Web","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01411748,0.0004943874,0.0005980959,0.0008978529,0.0004002038,0.001376578,0.0006571396,0.0008978081,0.001621159],"category_scores_gemma":[0.05641361,0.0004382379,0.001080879,0.000557376,0.0006790361,0.001323047,0.001546791,0.001411813,0.0008361513],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005962722,"about_ca_system_score_gemma":0.0005916097,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001575723,"about_ca_topic_score_gemma":0.001811222,"domain_scores_codex":[0.9907994,0.003729219,0.001159481,0.001702265,0.002062945,0.0005466469],"domain_scores_gemma":[0.9437807,0.0262044,0.01564529,0.003916773,0.006747617,0.003705142],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0007979737,0.001021216,0.9852524,0.00007179144,0.0001679165,0.00003797504,0.00181638,0.0002081042,0.0006984456,0.0000223336,0.0002688409,0.009636583],"study_design_scores_gemma":[0.00001346177,0.003101148,0.9946116,0.00003105062,0.00007754808,0.00008749308,0.0007108251,0.0004360562,0.0005862983,0.00002237546,0.0003072107,0.00001499129],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9993848,0.00007663694,0.0001720299,0.00002306681,0.000006524137,0.00003905362,0.0001061332,0.000009456457,0.0001822769],"genre_scores_gemma":[0.9990337,0.000059692,0.0002916597,0.00004242982,0.000009553206,0.00007110233,0.0002302657,0.00001193129,0.000249694],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9858825,"threshold_uncertainty_score":0.07466125,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05309610483663259,"score_gpt":0.5138825237555716,"score_spread":0.4607864189189391,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}