{"id":"W4413820752","doi":"10.2196/73419","title":"Development and Validation of a Large Language Model–Based System for Medical History-Taking Training: Prospective Multicase Study on Evaluation Stability, Human-AI Consistency, and Transparency","year":2025,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Shantou University Medical College; Shantou University","keywords":"Generalizability theory; Transparency (behavior); Consistency (knowledge bases); Baseline (sea); Artificial intelligence; Computer science; Stability (learning theory); Medical history; Machine learning; Medical physics; Physical therapy; Medicine; Psychology; Surgery","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002554135,0.0001318059,0.0002960314,0.0001776555,0.000148811,0.000009593287,0.00005688712,0.0001814007,0.0001094399],"category_scores_gemma":[0.002908011,0.0001194901,0.0000338624,0.0001384209,0.0001125839,0.00006083642,0.00001090675,0.0002069425,6.121402e-7],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007805921,"about_ca_system_score_gemma":0.008071935,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00010848,"about_ca_topic_score_gemma":0.0002115175,"domain_scores_codex":[0.9977469,0.0001556555,0.0007255801,0.0003750063,0.0008231079,0.0001737717],"domain_scores_gemma":[0.9985849,0.0002790434,0.0001869838,0.0001920264,0.0005245771,0.0002324549],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"qualitative","study_design_scores_codex":[0.000959466,0.01207208,0.1014711,0.005661753,0.0001565065,0.000005077476,0.4306412,0.0000104447,0.001280007,0.01066131,0.00045448,0.4366265],"study_design_scores_gemma":[0.006522319,0.003036891,0.1202715,0.008767507,0.001063453,0.00002546185,0.537288,0.2885463,0.03192843,0.001319956,0.0005257147,0.0007045052],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9898743,0.0002627448,0.003500047,0.002123405,0.0003557013,0.003455855,0.00000359655,0.00003812255,0.0003862037],"genre_scores_gemma":[0.9968714,0.000001856415,0.0006676622,0.0005393904,0.00008347916,0.001700065,0.00009868453,0.00001067236,0.00002683526],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.435922,"threshold_uncertainty_score":0.9975514,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2213345464877732,"score_gpt":0.5135202631757582,"score_spread":0.292185716687985,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}