{"id":"W2489879551","doi":"","title":"Evidence-based Principles to Guide Collaborative Approaches to Evaluation: Technical Report","year":2015,"lang":"en","type":"article","venue":"","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":10,"is_retracted":false,"has_abstract":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Management science; Engineering ethics; Data science; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","insufficient_payload"],"consensus_categories":["metaresearch","insufficient_payload"],"category_scores_codex":[0.03630686,0.0001681643,0.0002892157,0.0003792801,0.0001294387,0.000396475,0.0007429154,0.00007740955,0.002257902],"category_scores_gemma":[0.03370766,0.0001118821,0.00006468932,0.002666031,0.00005295207,0.0006313503,0.000225258,0.0000901614,0.002209999],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004745322,"about_ca_system_score_gemma":0.004836346,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002521966,"about_ca_topic_score_gemma":0.0004730248,"domain_scores_codex":[0.991173,0.0005613143,0.001173604,0.0007544043,0.00608347,0.0002542571],"domain_scores_gemma":[0.9931688,0.0009876023,0.0002584475,0.001187548,0.003833084,0.0005644786],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0003519961,0.0003051799,0.05992829,0.000005750746,0.00002633878,0.00002944035,0.002931185,0.2284187,0.000611104,0.02110773,0.5683597,0.1179245],"study_design_scores_gemma":[0.0008235063,0.0008360489,0.05074651,0.0001025243,0.00003669291,0.00002922563,0.005770158,0.09213318,0.004916011,0.003820344,0.8402991,0.0004866998],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.09726898,0.000256171,0.1859298,0.1617864,0.0009874827,0.004535439,0.00001057628,0.0002378369,0.5489873],"genre_scores_gemma":[0.8296793,0.000002353052,0.1446094,0.004611113,0.0001794249,0.0007142816,0.00001077819,0.00001178269,0.02018152],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7324104,"threshold_uncertainty_score":0.9986542,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.9049027782050624,"score_gpt":0.5776800492054441,"score_spread":0.3272227289996182,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}