{"id":"W4366384291","doi":"10.3138/cjpe.0027.008","title":"Meta-evaluation: Evaluating the Evaluation of the Paris Declaration","year":2013,"lang":"en","type":"article","venue":"Canadian Journal of Program Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Declaration; Evaluation methods; Meta-analysis; Strengths and weaknesses; Quality (philosophy); Psychology; Computer science; Management science; Political science; Medicine; Engineering; Social psychology; Law; Reliability engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","insufficient_payload"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.1405069,0.0002561096,0.0004972088,0.000548324,0.0007808061,0.0007824051,0.001313444,0.0001512244,0.02037896],"category_scores_gemma":[0.01613018,0.0001296875,0.0005753705,0.001973409,0.0002843149,0.001492972,0.00004636222,0.0004207952,0.000144158],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008810804,"about_ca_system_score_gemma":0.01544859,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001493331,"about_ca_topic_score_gemma":0.01114599,"domain_scores_codex":[0.9732264,0.008540806,0.002457898,0.0003717549,0.0149928,0.0004103212],"domain_scores_gemma":[0.9641123,0.001037574,0.003087276,0.00107267,0.03038132,0.0003088746],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001006451,0.00005282935,0.008378584,0.000004432981,0.0004205836,1.058902e-7,0.001989384,0.06408812,0.000251973,0.0003605446,0.006864217,0.9175792],"study_design_scores_gemma":[0.001352175,0.0004303271,0.1838541,0.00003693908,0.00564792,0.00002489755,0.002016436,0.7608584,0.0004697795,0.04103515,0.004106369,0.0001675967],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9609864,0.004065051,0.0007365009,0.01633531,0.002346641,0.009341846,0.000009715868,0.00001081589,0.00616768],"genre_scores_gemma":[0.9962239,0.00001247626,0.001643872,0.0003657647,0.000275237,0.001254714,0.00001518066,0.00001707606,0.0001917869],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9174116,"threshold_uncertainty_score":0.9921574,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.7957816232407139,"score_gpt":0.5870236538267359,"score_spread":0.208757969413978,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}