{"id":"W4393901296","doi":"10.3138/cjpe-2024-0001","title":"Capturing Evaluation Capacity: Findings from a Mapping of Evaluation Capacity Instruments","year":2024,"lang":"en","type":"article","venue":"Canadian Journal of Program Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University; University of Ottawa","funders":"","keywords":"Rubric; Face validity; Construct validity; Content validity; Reliability (semiconductor); Concurrent validity; Construct (python library); Adaptation (eye); Psychology; Computer science; Applied psychology; Internal consistency; Process management; Psychometrics; Engineering; Clinical psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.05105589,0.0002583632,0.0004629966,0.001971047,0.0002745665,0.0008477197,0.0005175048,0.0001953329,0.005647749],"category_scores_gemma":[0.006609947,0.000220497,0.0002834518,0.001728229,0.000162405,0.001862922,0.00002710455,0.0004562257,0.0001138899],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002145485,"about_ca_system_score_gemma":0.009221726,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003547618,"about_ca_topic_score_gemma":0.01496088,"domain_scores_codex":[0.9871926,0.001720974,0.002026194,0.0004953768,0.008145583,0.000419322],"domain_scores_gemma":[0.9905832,0.0004608103,0.0009927726,0.0004718346,0.007019793,0.0004715816],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001561112,0.00004905748,0.02371707,0.00002755869,0.0001680294,0.000002913782,0.008715455,0.004844323,0.001240753,0.0003259237,0.0009955449,0.9598978],"study_design_scores_gemma":[0.00161779,0.0002958224,0.1474464,0.0006057118,0.0006482531,0.00003901979,0.002587206,0.7869716,0.001761372,0.0502354,0.007503207,0.0002881875],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9891152,0.001109043,0.001392496,0.001187593,0.00289714,0.002224214,0.00005421063,0.00002054597,0.001999503],"genre_scores_gemma":[0.9949582,0.00001387121,0.004317523,0.00007930424,0.0003207516,0.0001752883,0.00007687324,0.00002210363,0.00003606866],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9596096,"threshold_uncertainty_score":0.9963951,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5637636910939985,"score_gpt":0.4808963562487809,"score_spread":0.08286733484521763,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}