{"id":"W3208939115","doi":"10.1177/1098214020936769","title":"The Use of Evaluability Assessments in Improving Future Evaluations: A Scoping Review of 10 Years of Literature (2008–2018)","year":2021,"lang":"en","type":"review","venue":"American Journal of Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":14,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo; University of Guelph","funders":"","keywords":"Ambiguity; Psychology; Engineering ethics; Equity (law); Relevance (law); Management science; Political science; Engineering; Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1707472,0.001848716,0.004163675,0.0356332,0.002375867,0.007143104,0.002581965,0.00285594,0.003184413],"category_scores_gemma":[0.3521426,0.001652154,0.005458835,0.02698022,0.002985786,0.009955379,0.004614105,0.003138896,0.0005955464],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.009703105,"about_ca_system_score_gemma":0.04526919,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01069714,"about_ca_topic_score_gemma":0.02489615,"domain_scores_codex":[0.8831443,0.06618196,0.02825104,0.003273722,0.01793993,0.001209023],"domain_scores_gemma":[0.5665203,0.3434977,0.03032964,0.007311644,0.05108504,0.001255592],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"systematic_review","study_design_scores_codex":[0.0001393811,0.00005632607,0.001601269,0.4223826,0.001519809,0.0001433965,0.003350172,0.0005047263,0.000303745,0.005759147,0.008148799,0.5560906],"study_design_scores_gemma":[0.00003085478,0.00007082749,0.001955319,0.918065,0.003246182,0.0001461546,0.001468,0.0002054281,0.0003137795,0.002319762,0.07213477,0.00004383843],"study_design_candidate":"systematic_review","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.001118342,0.9895198,0.002940457,0.002761074,0.0004146266,0.0009197474,0.0002521065,0.00002339922,0.002050423],"genre_scores_gemma":[0.01601227,0.9725578,0.007591161,0.001032633,0.0001946185,0.002025092,0.0003054569,0.00002745366,0.0002533914],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.8292528,"threshold_uncertainty_score":0.9030084,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3614297121157451,"score_gpt":0.6078438813411042,"score_spread":0.2464141692253591,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}