{"id":"W4413820752","doi":"10.2196/73419","title":"Development and Validation of a Large Language Model–Based System for Medical History-Taking Training: Prospective Multicase Study on Evaluation Stability, Human-AI Consistency, and Transparency","year":2025,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Shantou University Medical College; Shantou University","keywords":"Generalizability theory; Transparency (behavior); Consistency (knowledge bases); Baseline (sea); Artificial intelligence; Computer science; Stability (learning theory); Medical history; Machine learning; Medical physics; Physical therapy; Medicine; Psychology; Surgery","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02137962,0.001018771,0.00053094,0.0007516473,0.0004220341,0.001483885,0.002139597,0.001140183,0.002634148],"category_scores_gemma":[0.05063853,0.0004993855,0.0006938746,0.000331747,0.0007310368,0.002921214,0.002298023,0.001442317,0.001131183],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001582315,"about_ca_system_score_gemma":0.002415871,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002918,"about_ca_topic_score_gemma":0.002660682,"domain_scores_codex":[0.9885283,0.006785475,0.001115109,0.001955408,0.001331076,0.0002847516],"domain_scores_gemma":[0.9627931,0.02309535,0.001715358,0.004442085,0.006680807,0.001273283],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"nonrandomized_trial","study_design_scores_codex":[0.008345379,0.01020353,0.105336,0.002432767,0.0008794124,0.001270714,0.009557086,0.05221235,0.1397122,0.002690907,0.02292771,0.6444319],"study_design_scores_gemma":[0.002369899,0.01068589,0.08449937,0.0004122858,0.0005787004,0.00108997,0.002138181,0.7759235,0.0980402,0.002258538,0.02157343,0.0004299528],"study_design_candidate":"nonrandomized_trial","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7941033,0.0002840462,0.1812451,0.0005724701,0.0001578361,0.003239935,0.001120215,0.01744816,0.001828956],"genre_scores_gemma":[0.7505011,0.0000841482,0.2420207,0.0004490935,0.00003583102,0.002250484,0.002851833,0.0005260163,0.00128088],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02137962,"threshold_uncertainty_score":0.1130676,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2213345464877732,"score_gpt":0.5135202631757582,"score_spread":0.292185716687985,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}