{"id":"W4366384291","doi":"10.3138/cjpe.0027.008","title":"Meta-evaluation: Evaluating the Evaluation of the Paris Declaration","year":2013,"lang":"en","type":"article","venue":"Canadian Journal of Program Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Declaration; Evaluation methods; Meta-analysis; Strengths and weaknesses; Quality (philosophy); Psychology; Computer science; Management science; Political science; Medicine; Engineering; Social psychology; Law; Reliability engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.5605815,0.006593116,0.02241491,0.01828751,0.002970519,0.01288019,0.008840191,0.009127746,0.007304746],"category_scores_gemma":[0.7187894,0.004439696,0.04626917,0.01435215,0.00635216,0.009593047,0.007657841,0.008368249,0.0007502839],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.02273134,"about_ca_system_score_gemma":0.02041284,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00724643,"about_ca_topic_score_gemma":0.008389569,"domain_scores_codex":[0.2452521,0.6534663,0.05570697,0.01111473,0.03302314,0.001436695],"domain_scores_gemma":[0.2137147,0.6713465,0.04080667,0.03996707,0.03204829,0.002116848],"domain_codex":"methods","domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"meta_analysis","study_design_gemma":"qualitative","study_design_scores_codex":[0.0102181,0.0001676672,0.004263362,0.3393535,0.5570449,0.0003118925,0.001361109,0.005262394,0.0006271834,0.01182461,0.008868886,0.06069645],"study_design_scores_gemma":[0.008853332,0.003237807,0.006272707,0.1696706,0.7379692,0.0002209776,0.0007716609,0.007444504,0.002761218,0.02343993,0.03884091,0.0005172326],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"review","genre_gemma":"empirical","genre_scores_codex":[0.02517818,0.6422771,0.1899867,0.02956571,0.01696799,0.06784783,0.0104389,0.002357204,0.0153804],"genre_scores_gemma":[0.3841808,0.08981881,0.3835286,0.008918387,0.002461167,0.1241912,0.003840672,0.0008679447,0.002192429],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.4394185,"threshold_uncertainty_score":0.5418813,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.7957816232407139,"score_gpt":0.5870236538267359,"score_spread":0.208757969413978,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}