{"id":"W7098438613","doi":"","title":"The In-Training Examination in Internal Medicine: An Analysis of Resident Performance over Time","year":2016,"lang":"en","type":"article","venue":"","topic":"Diverse Scientific and Economic Studies","field":"Economics, Econometrics and Finance","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Test (biology); Descriptive statistics; Medical school; Educational measurement; Training (meteorology); Residency training","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.003145152,0.0003536858,0.0003447452,0.002806233,0.0002273336,0.0004834625,0.0005753518,0.000395261,0.0009257949],"category_scores_gemma":[0.00719698,0.000177572,0.0005153848,0.002415678,0.0002493228,0.0007942417,0.0009035178,0.0004450207,0.0002918975],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004436553,"about_ca_system_score_gemma":0.0004092552,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004861646,"about_ca_topic_score_gemma":0.005534818,"domain_scores_codex":[0.9987535,0.0003185445,0.0001725485,0.0001766886,0.0004438404,0.0001348147],"domain_scores_gemma":[0.9918594,0.00134587,0.004579586,0.0003702144,0.00110272,0.0007421817],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0001707558,0.00005423474,0.9947245,0.00002636194,0.0001126775,0.00002565738,0.0001259933,0.00009145704,0.0001578888,0.00001398626,0.0002845144,0.004212],"study_design_scores_gemma":[0.00000271306,0.0001257187,0.9992867,0.000003574708,0.00001388167,0.00006547165,0.00006515585,0.000187831,0.0000685804,0.000003979353,0.0001739846,0.000002420012],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9973989,0.0003193444,0.000342026,0.00004518146,0.000008325987,0.00002895805,0.001283712,0.00001815092,0.0005554622],"genre_scores_gemma":[0.9974067,0.0001417053,0.000413188,0.00001350426,0.00001652399,0.0000342322,0.001620484,0.000006007947,0.0003476274],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9968548,"threshold_uncertainty_score":0.01663339,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0401394281318657,"score_gpt":0.2296631996724147,"score_spread":0.1895237715405491,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}