{"id":"W4412837419","doi":"10.1371/journal.pone.0325803","title":"Evaluation of large language models as a diagnostic tool for medical learners and clinicians using advanced prompting techniques","year":2025,"lang":"en","type":"article","venue":"PLoS ONE","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"Hamilton Health Sciences","funders":"","keywords":"Computer science; Correctness; Sensitivity (control systems); Medical physics; Artificial intelligence; Medicine; Engineering; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02721786,0.00131916,0.0008837984,0.002005546,0.0004340339,0.002663396,0.001323456,0.001522937,0.002971315],"category_scores_gemma":[0.1529136,0.0003506958,0.0009822826,0.0006689038,0.0006050501,0.00236838,0.00394188,0.001689271,0.001425451],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001341122,"about_ca_system_score_gemma":0.001910443,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001130467,"about_ca_topic_score_gemma":0.001322825,"domain_scores_codex":[0.9785345,0.01472408,0.001886187,0.001895159,0.002580712,0.0003793918],"domain_scores_gemma":[0.7787634,0.1903554,0.007478812,0.007157455,0.01286655,0.003378346],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.01614654,0.004262455,0.235727,0.003328994,0.0008382205,0.001702425,0.009489007,0.0453253,0.02740513,0.001878608,0.01067914,0.6432171],"study_design_scores_gemma":[0.003592116,0.01864269,0.1328681,0.001775459,0.001525766,0.00528976,0.008808346,0.6588767,0.1276532,0.01191924,0.02830028,0.0007484335],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9260964,0.0008806203,0.05858772,0.00128813,0.0001878329,0.001220265,0.001501132,0.007191712,0.003046189],"genre_scores_gemma":[0.9120296,0.000257411,0.08403391,0.0003062095,0.00006745025,0.0005761081,0.001750389,0.0002011662,0.0007778732],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02721786,"threshold_uncertainty_score":0.1439435,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2809413489577854,"score_gpt":0.5119117702232089,"score_spread":0.2309704212654235,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}