{"id":"W3154034046","doi":"10.1177/1035719x211008263","title":"Thinking with complexity in evaluation: A case study review","year":2021,"lang":"en","type":"article","venue":"Evaluation Journal of Australasia","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"Impact","funders":"","keywords":"Context (archaeology); Management science; Social complexity; Computer science; Complexity management; Exploratory research; Knowledge management; Sociology; Social science; Engineering; Management","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.07097194,0.0001652614,0.0005202802,0.000373412,0.0001685279,0.0002786637,0.0004140929,0.00004416548,0.01353334],"category_scores_gemma":[0.004135991,0.0001121453,0.0001203568,0.001815738,0.00005778907,0.001003498,0.00005394954,0.0003892168,0.00005872296],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003159409,"about_ca_system_score_gemma":0.003074248,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001405958,"about_ca_topic_score_gemma":0.0009294379,"domain_scores_codex":[0.9838009,0.003852737,0.001941116,0.0003379382,0.009859649,0.0002076674],"domain_scores_gemma":[0.9890044,0.0005260991,0.001237178,0.0005239715,0.008572862,0.0001354525],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0001965321,0.002470347,0.5633863,0.0001319776,0.0003018512,0.006193072,0.01717302,0.03923791,0.00008147398,0.0005475288,0.01420631,0.3560737],"study_design_scores_gemma":[0.01647057,0.002377327,0.7433393,0.003046333,0.001591901,0.03999839,0.08924558,0.07460197,0.0001933254,0.02456145,0.003887982,0.0006858894],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.986816,0.002362681,0.000444257,0.006238922,0.0003958947,0.001074097,0.000001668906,0.000005672443,0.002660789],"genre_scores_gemma":[0.99647,0.0001287265,0.002582079,0.0005141794,0.0001140661,0.00002947422,0.000004526922,0.000008114734,0.0001487813],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3553878,"threshold_uncertainty_score":0.9873684,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5494958181463374,"score_gpt":0.5867373010000473,"score_spread":0.03724148285370987,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}