{"id":"W4417184697","doi":"10.1111/capa.70043","title":"Escaping the Tunnel: How Formative Evaluation Complements Performance Audit to Improve Decision Navigation","year":2025,"lang":"en","type":"article","venue":"Canadian Public Administration","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University","funders":"","keywords":"Formative assessment; Audit; Performance audit; Evaluation methods; Reliability (semiconductor)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["scholarly_communication","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.008643717,0.0001976921,0.0001914886,0.0007456443,0.001199267,0.001609866,0.0007954799,0.00009938499,0.0009146927],"category_scores_gemma":[0.002586259,0.0001453731,0.00007344354,0.001874626,0.00006733002,0.00194011,0.0000716949,0.0001984094,0.000325613],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00106151,"about_ca_system_score_gemma":0.005478143,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001977583,"about_ca_topic_score_gemma":0.0746985,"domain_scores_codex":[0.995833,0.0003530354,0.0007467805,0.0004956475,0.002117102,0.0004544615],"domain_scores_gemma":[0.9960762,0.0005150246,0.0003384324,0.0007359841,0.001971213,0.0003631727],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00003687746,0.00003099678,0.007466472,0.00001145673,0.0000304359,6.986987e-7,0.0008574316,0.0002761644,0.000171266,0.02395643,0.04998973,0.917172],"study_design_scores_gemma":[0.001126991,0.0003787606,0.141793,0.0001155742,0.00004529711,0.000006554229,0.005562194,0.3546844,0.001042047,0.01059891,0.4842612,0.0003850221],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7846454,0.00003836612,0.04923826,0.1478502,0.002078749,0.002293947,0.00008404378,0.00004575333,0.01372529],"genre_scores_gemma":[0.9947481,0.000004709566,0.0006399104,0.00267127,0.0001221029,0.000309251,0.000230487,0.000008572209,0.001265628],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.916787,"threshold_uncertainty_score":0.9999986,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1439210505811699,"score_gpt":0.4574227160440045,"score_spread":0.3135016654628345,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}