{"id":"W4388306673","doi":"10.1002/ev.20564","title":"Artificial intelligence and the future of evaluation education: Possibilities and prototypes","year":2023,"lang":"en","type":"article","venue":"New Directions for Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":15,"is_retracted":false,"has_abstract":true,"ca_institutions":"Fraser Health","funders":"","keywords":"Chatbot; Evaluation methods; Engineering ethics; Literacy; Computer science; Program evaluation; Management science; Psychology; Pedagogy; Artificial intelligence; Political science; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04652652,0.000528688,0.000700078,0.00203292,0.002924613,0.01477628,0.002962243,0.004164731,0.006922479],"category_scores_gemma":[0.03292329,0.0004145784,0.0006387548,0.0011818,0.01958588,0.01695871,0.004686157,0.004338104,0.0006021291],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005388487,"about_ca_system_score_gemma":0.005827419,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003153087,"about_ca_topic_score_gemma":0.001776822,"domain_scores_codex":[0.9748905,0.02119292,0.0006471915,0.0005376567,0.001770107,0.0009616725],"domain_scores_gemma":[0.9442628,0.04624326,0.000890327,0.002895908,0.003544377,0.002163364],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0002207172,0.0005669452,0.001591782,0.0005291066,0.00001503308,0.0002057377,0.007126331,0.002406661,0.0005081817,0.8734421,0.004766324,0.1086211],"study_design_scores_gemma":[0.000217884,0.0008176088,0.001272784,0.002882975,0.00002787644,0.0004103971,0.01422166,0.01802012,0.002040655,0.7973178,0.1626463,0.0001237952],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.1317136,0.05004778,0.1831253,0.2710562,0.001886824,0.0007982326,0.0001336076,0.001150916,0.3600874],"genre_scores_gemma":[0.8873028,0.008848689,0.09124804,0.003200431,0.0002922638,0.0005289186,0.00006772172,0.00008733855,0.008423749],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.04652652,"threshold_uncertainty_score":0.2460587,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2331504610129478,"score_gpt":0.5285796286386468,"score_spread":0.295429167625699,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}