{"id":"W2133213181","doi":"10.3138/jvme.31.1.61","title":"Ensuring that the competent are truly competent: an overview of common methods and procedures used to set standards on high-stakes examinations","year":2004,"lang":"en","type":"review","venue":"Journal of Veterinary Medical Education","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":17,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Certification; Licensure; Set (abstract data type); Computer science; Test (biology); Selection (genetic algorithm); Factoring; Norm (philosophy); Task (project management); Medical education; Medicine; Accounting; Political science; Artificial intelligence; Engineering; Business; Law","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01535807,0.0003086618,0.001477057,0.0006583631,0.0002149988,0.0001919656,0.001020977,0.0001819694,0.0005470801],"category_scores_gemma":[0.004549564,0.000167582,0.000284359,0.0006321871,0.0001603926,0.0003651735,0.0001270852,0.000615846,0.000005356435],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004042557,"about_ca_system_score_gemma":0.006365427,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002031442,"about_ca_topic_score_gemma":0.0000515728,"domain_scores_codex":[0.9905888,0.002558693,0.00193682,0.0003322624,0.004388168,0.0001952988],"domain_scores_gemma":[0.9931154,0.002321449,0.002574753,0.0005512926,0.001011771,0.0004253543],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00003981688,0.0004717857,0.00006991167,0.002579445,0.00006709231,0.000007185113,0.001706262,0.00004246333,0.000002415631,0.0006833876,0.000632472,0.9936978],"study_design_scores_gemma":[0.0005817218,0.00289483,0.01697209,0.03926492,0.0003533626,0.0008346696,0.003555861,0.00007986779,0.000009607257,0.00129688,0.9338579,0.0002982552],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.039819,0.9523625,0.0006831541,0.003682995,0.002248227,0.0009568513,0.0001074563,0.00001144016,0.0001283491],"genre_scores_gemma":[0.05627582,0.939153,0.003529988,0.0005267097,0.0003463189,0.00005640851,0.00003398734,0.0000289646,0.00004887201],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.9933995,"threshold_uncertainty_score":0.9992676,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.710843790843106,"score_gpt":0.6648786064823116,"score_spread":0.04596518436079444,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}