{"id":"W4384704311","doi":"10.3138/cjpe.76031","title":"<i>Ethics for Evaluation. Beyond “Doing No Harm” to “Tackling Bad” and “Doing Good”</i> , par Rob van den Berg, Penny Hawkins et Nicoletta Stame (dir.)","year":2023,"lang":"en","type":"article","venue":"Canadian Journal of Program Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"","funders":"","keywords":"Harm; Psychology; Social psychology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[{"model":"gpt","categories":[],"domain":null,"study_design":"not_applicable","genre":"review","about_ca_system":false,"about_ca_topic":false,"confidence":"high","status":"direct model label, unvalidated"},{"model":"grok","categories":[],"domain":null,"study_design":"not_applicable","genre":"commentary","about_ca_system":false,"about_ca_topic":false,"confidence":"low","status":"direct model label, unvalidated"},{"model":"opus","categories":[],"domain":null,"study_design":"not_applicable","genre":"commentary","about_ca_system":false,"about_ca_topic":false,"confidence":"low","status":"direct model label, unvalidated"}],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","metaepi_narrow","scholarly_communication","insufficient_payload"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.07420288,0.0002763988,0.0004533144,0.001401287,0.0009937779,0.001380869,0.000640267,0.0002009982,0.001087922],"category_scores_gemma":[0.01128497,0.0002491718,0.0001985072,0.001610137,0.000100147,0.001260608,0.00006908104,0.0005820644,0.0002221435],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006603084,"about_ca_system_score_gemma":0.009781825,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003505757,"about_ca_topic_score_gemma":0.02826473,"domain_scores_codex":[0.9913405,0.001403294,0.001554753,0.0005524119,0.004407763,0.0007413439],"domain_scores_gemma":[0.9877919,0.001844417,0.000951898,0.0004646566,0.00796303,0.0009841194],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00006127976,0.00004787355,0.01345338,0.00003945074,0.0001173392,0.000007984241,0.00725974,0.06465901,0.0005310124,0.001020247,0.02334158,0.8894611],"study_design_scores_gemma":[0.003510135,0.001194241,0.03514216,0.0003487155,0.000488298,0.00004551982,0.003387706,0.4508127,0.0002398239,0.01192327,0.4923358,0.0005716034],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8671578,0.003075873,0.03819168,0.05694571,0.006350007,0.01400333,0.000213124,0.000149931,0.0139126],"genre_scores_gemma":[0.9578635,0.00006492747,0.03723099,0.002956326,0.0005842373,0.000490466,0.0001135741,0.00005454396,0.0006415057],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8888895,"threshold_uncertainty_score":0.9999961,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2641401555920312,"score_gpt":0.5229258898039616,"score_spread":0.2587857342119304,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}