{"id":"W4283391831","doi":"10.1111/capa.12454","title":"Evaluating the evaluators: What have we learned from “neutral assessments” of the Canadian federal evaluation function?","year":2022,"lang":"en","type":"article","venue":"Canadian Public Administration","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Agency (philosophy); Function (biology); Government (linguistics); Quality (philosophy); Business; Political science; Public relations; Psychology; Sociology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2859326,0.0009483293,0.001901791,0.008847986,0.005133261,0.0149553,0.005344371,0.001770303,0.001428597],"category_scores_gemma":[0.3959899,0.0005872801,0.001342265,0.01428504,0.01107578,0.0111088,0.003632133,0.004448496,0.0002161214],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.1030965,"about_ca_system_score_gemma":0.1851923,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.8593156,"about_ca_topic_score_gemma":0.894629,"domain_scores_codex":[0.8235202,0.09412472,0.01029701,0.005631123,0.06242881,0.003998201],"domain_scores_gemma":[0.4605049,0.1701018,0.02388036,0.01558006,0.323037,0.00689589],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"qualitative","study_design_scores_codex":[0.0005819401,0.0001422932,0.03654034,0.01987887,0.0007397314,0.0001551575,0.02085598,0.001867479,0.0005083969,0.0662045,0.03562528,0.8169],"study_design_scores_gemma":[0.0002211926,0.0008431479,0.2255471,0.1190717,0.002482689,0.0004723098,0.0538097,0.005243242,0.003594732,0.03847631,0.5495225,0.0007152671],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"review","genre_gemma":"empirical","genre_scores_codex":[0.08204778,0.6537595,0.03318263,0.1487545,0.005188043,0.001129878,0.001531291,0.0002688488,0.07413758],"genre_scores_gemma":[0.7557341,0.186716,0.03517767,0.01707415,0.0009070706,0.0006457539,0.0009493602,0.000144736,0.002651299],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8969035,"threshold_uncertainty_score":0.8805722,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4379135587071296,"score_gpt":0.5156808570606097,"score_spread":0.07776729835348012,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}