{"id":"W4283391831","doi":"10.1111/capa.12454","title":"Evaluating the evaluators: What have we learned from “neutral assessments” of the Canadian federal evaluation function?","year":2022,"lang":"en","type":"article","venue":"Canadian Public Administration","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Agency (philosophy); Function (biology); Government (linguistics); Quality (philosophy); Business; Political science; Public relations; Psychology; Sociology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["sts","scholarly_communication","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.01836979,0.0002316383,0.0002425304,0.0005134794,0.004249923,0.002324729,0.001368972,0.000118214,0.01790501],"category_scores_gemma":[0.001913924,0.0001616656,0.0001893279,0.001378218,0.0001722165,0.001684768,0.0001128907,0.0005701609,0.00007516785],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002718705,"about_ca_system_score_gemma":0.04300247,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.1415032,"about_ca_topic_score_gemma":0.9816308,"domain_scores_codex":[0.9894519,0.003249618,0.001042447,0.0006557315,0.005021388,0.0005789271],"domain_scores_gemma":[0.9952975,0.0006432733,0.0007409267,0.001208088,0.001543986,0.0005662377],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001802421,0.0001905157,0.09289349,0.000009732954,0.0002912071,0.000005127541,0.003699597,0.0132188,0.0003634328,0.05662284,0.04302187,0.7895032],"study_design_scores_gemma":[0.001660027,0.001156175,0.1407456,0.00003137624,0.0002032251,0.00002146186,0.03488674,0.4584413,0.0001801535,0.0587741,0.3033519,0.0005479564],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.735829,0.0003441364,0.0001343466,0.2490015,0.003927889,0.001965484,0.0002560489,0.00002251883,0.008519084],"genre_scores_gemma":[0.99336,0.000008455521,0.0001085808,0.003328398,0.0002254448,0.0004373686,0.0003468279,0.00001992391,0.002165046],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8401276,"threshold_uncertainty_score":0.9987109,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4379135587071296,"score_gpt":0.5156808570606097,"score_spread":0.07776729835348012,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}