{"id":"W4224866913","doi":"10.3138/cjpe.72386","title":"Evaluation Utility Metrics (EUMs) in Reflective Practice","year":2022,"lang":"en","type":"article","venue":"Canadian Journal of Program Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Institute of General Medical Sciences","keywords":"Quality (philosophy); Negotiation; Computer science; Evaluation methods; Management science; Process management; Risk analysis (engineering); Psychology; Medicine; Business; Reliability engineering; Sociology; Engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.3446237,0.002667714,0.002891614,0.01868971,0.002802822,0.01647604,0.003208354,0.003938661,0.003865816],"category_scores_gemma":[0.6086959,0.001299337,0.002912188,0.01691145,0.01100497,0.01944608,0.01250234,0.005198914,0.0007705696],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0125774,"about_ca_system_score_gemma":0.01315636,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002547969,"about_ca_topic_score_gemma":0.001818308,"domain_scores_codex":[0.4137818,0.5057175,0.03088149,0.006626687,0.04069299,0.002299514],"domain_scores_gemma":[0.2804278,0.5846642,0.04233947,0.03271844,0.05629284,0.003557279],"domain_codex":"methods","domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000629212,0.000373281,0.014265,0.005798081,0.0007579206,0.0001228316,0.01096536,0.01642063,0.0006454996,0.4131781,0.01015591,0.5266881],"study_design_scores_gemma":[0.0004276788,0.001416168,0.01583153,0.01327781,0.0007302111,0.0005674278,0.008677859,0.08969446,0.004579085,0.7822038,0.08202924,0.0005647986],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0234277,0.009639557,0.9164072,0.007936486,0.0007021074,0.003730798,0.0003967361,0.001199593,0.03655981],"genre_scores_gemma":[0.3024221,0.001684418,0.6880304,0.0007257251,0.0001878336,0.005451583,0.0002582719,0.000230821,0.001008905],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.3446237,"threshold_uncertainty_score":0.8081957,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5800774051525978,"score_gpt":0.6115541890215195,"score_spread":0.03147678386892172,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}