{"id":"W3016806524","doi":"10.3352/jeehp.2020.17.12","title":"Performance of the Ebel standard-setting method for the spring 2019 Royal College of Physicians and Surgeons of Canada internal medicine certification examination consisting of multiple-choice questions","year":2020,"lang":"en","type":"article","venue":"Journal of Educational Evaluation for Health Professions","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":17,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Manitoba; University of Calgary; Royal College of Physicians and Surgeons of Canada","funders":"","keywords":"Specialty; Certification; Nuclear medicine; Medicine; Statistics; Psychology; Mathematics; Family medicine; Management","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.03895415,0.000507578,0.0005716003,0.00202663,0.0007756582,0.001028404,0.0008865441,0.0006508168,0.002573312],"category_scores_gemma":[0.09449459,0.0002823542,0.0007209514,0.001008017,0.0005763513,0.0008046381,0.001782257,0.0005698973,0.0009696],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002641963,"about_ca_system_score_gemma":0.004682942,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0101124,"about_ca_topic_score_gemma":0.02268411,"domain_scores_codex":[0.9633633,0.01430011,0.003771426,0.002539471,0.01470666,0.001318998],"domain_scores_gemma":[0.8994297,0.0313535,0.01591746,0.005840227,0.0444058,0.003053316],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.002078588,0.0006494136,0.8239812,0.0004968696,0.0002402762,0.0001731921,0.002423283,0.001034829,0.003855963,0.0002880264,0.003803636,0.1609748],"study_design_scores_gemma":[0.0001024484,0.001324113,0.9816599,0.0002080746,0.00004489988,0.0003306256,0.0009703499,0.005538328,0.006022345,0.0001407823,0.003598031,0.00006016375],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.97979,0.0007384244,0.008965011,0.0004659246,0.0001669128,0.0007902809,0.0005539142,0.0002261749,0.008303272],"genre_scores_gemma":[0.9819559,0.0001700494,0.01454535,0.0001178369,0.00004296443,0.0003363333,0.0005558051,0.0000284546,0.002247272],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9610459,"threshold_uncertainty_score":0.2060117,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3982188306808459,"score_gpt":0.5391948166882257,"score_spread":0.1409759860073798,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}