{"id":"W4366448430","doi":"10.3138/cjpe.23.013","title":"M.J. Bamberger, J. Rugh, and L. Mabry. (2006). <i>RealWorld Evaluation: Working Under Budget, Time, Data, and Political Constraints.</i> Thousand Oaks, CA: Sage.","year":2008,"lang":"en","type":"article","venue":"Canadian Journal of Program Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"","funders":"","keywords":"SAGE; Politics; Psychology; Gerontology; Political science; Medicine; Physics; Law","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[{"model":"gpt","categories":["insufficient_payload"],"domain":null,"study_design":"not_applicable","genre":"other","about_ca_system":false,"about_ca_topic":false,"confidence":"high","status":"direct model label, unvalidated"},{"model":"grok","categories":[],"domain":null,"study_design":"not_applicable","genre":"other","about_ca_system":false,"about_ca_topic":false,"confidence":"low","status":"direct model label, unvalidated"},{"model":"opus","categories":[],"domain":null,"study_design":"not_applicable","genre":"commentary","about_ca_system":false,"about_ca_topic":false,"confidence":"low","status":"direct model label, unvalidated"}],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0216745,0.0002377438,0.0004186246,0.0006490992,0.0006001267,0.0006167089,0.000491778,0.0001382626,0.002856594],"category_scores_gemma":[0.002618834,0.0001971477,0.00007323,0.0006974121,0.0007175662,0.0009914035,0.00006577424,0.0003612408,0.00008281536],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004279709,"about_ca_system_score_gemma":0.007168343,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004793642,"about_ca_topic_score_gemma":0.006397446,"domain_scores_codex":[0.9933903,0.001136095,0.001206914,0.0004994839,0.003186394,0.0005808029],"domain_scores_gemma":[0.994889,0.0007280426,0.0006336538,0.0005774454,0.002069013,0.001102861],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003387907,0.0000680417,0.05025537,0.000008051599,0.00009349023,0.00003512925,0.0005966114,0.0005974706,0.00002882252,0.001576213,0.03242318,0.9142838],"study_design_scores_gemma":[0.006263648,0.0006836988,0.1889898,0.00030443,0.0008470583,0.002801331,0.003556173,0.4636058,0.00004030524,0.02189398,0.3101385,0.0008752664],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.904419,0.01873856,0.006046243,0.02850816,0.003883778,0.005995408,0.0003782584,0.0001261972,0.03190436],"genre_scores_gemma":[0.9943982,0.000112852,0.003618263,0.0008281511,0.000397051,0.000037103,0.0000793087,0.00002949631,0.0004996059],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9134085,"threshold_uncertainty_score":0.9984601,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.214130667283627,"score_gpt":0.4598854060888972,"score_spread":0.2457547388052701,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}