{"id":"W63457028","doi":"10.55016/ojs/ajer.v52i1.55109","title":"Setting Cut-Scores for Complex Performance Assessments: A Critical Examination of the Analytic Judgment Method","year":2006,"lang":"en","type":"article","venue":"Alberta Journal of Educational Research","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Psychology; Evaluation methods; Applied psychology; Statistics; Social psychology; Mathematics education; Mathematics; Reliability engineering; Engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.6180485,0.001928818,0.003002433,0.01568453,0.01020213,0.01097315,0.008854651,0.004516674,0.001074465],"category_scores_gemma":[0.8021548,0.002044857,0.001999211,0.006029646,0.01677872,0.01371909,0.007916922,0.01314074,0.0005051555],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01279852,"about_ca_system_score_gemma":0.02055282,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003789613,"about_ca_topic_score_gemma":0.00579046,"domain_scores_codex":[0.3971647,0.4468765,0.05505538,0.008878498,0.0903029,0.001721995],"domain_scores_gemma":[0.08421712,0.7866567,0.01507624,0.01848609,0.09408608,0.001477691],"domain_codex":"methods","domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001376658,0.0005841764,0.02200002,0.00847117,0.0006173187,0.0009782643,0.1269467,0.002170588,0.006650496,0.1752789,0.03140124,0.6235246],"study_design_scores_gemma":[0.001160802,0.002654304,0.02424301,0.04498732,0.001377519,0.004981509,0.09626371,0.0614957,0.03869602,0.4370129,0.2842577,0.002869529],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05751114,0.02338626,0.8388659,0.05352328,0.007314764,0.007097965,0.0001723894,0.0007306408,0.0113977],"genre_scores_gemma":[0.1909402,0.00293859,0.7949368,0.004978582,0.0008565099,0.004264272,0.00005897133,0.0002617552,0.0007644371],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6180485,"threshold_uncertainty_score":0.4710143,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4342625951092038,"score_gpt":0.6329415287968805,"score_spread":0.1986789336876766,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}