{"id":"W4402524684","doi":"10.1145/3696009","title":"Artificial Intelligence and the Future of Evaluation: From Augmented to Automated Evaluation","year":2024,"lang":"en","type":"article","venue":"Digital Government Research and Practice","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université Laval","funders":"","keywords":"Artificial intelligence; Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.09083906,0.001411781,0.002101087,0.007042226,0.002994796,0.02549442,0.002880356,0.004233567,0.006132762],"category_scores_gemma":[0.1525337,0.0007713623,0.0009439429,0.004545894,0.03882634,0.02946495,0.008902283,0.00508164,0.0006864102],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01169472,"about_ca_system_score_gemma":0.008756027,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006648351,"about_ca_topic_score_gemma":0.004092195,"domain_scores_codex":[0.8771792,0.1037308,0.00315275,0.003816387,0.01081359,0.001307346],"domain_scores_gemma":[0.8049867,0.1501371,0.008097927,0.01887801,0.01553204,0.002368349],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001749134,0.00007759261,0.003484893,0.0006994613,0.0001026924,0.00006639557,0.003724028,0.004402241,0.0002495935,0.7328968,0.006053529,0.2480679],"study_design_scores_gemma":[0.00006872145,0.0001054235,0.001900872,0.001455725,0.00004235728,0.0000825401,0.001801746,0.01049022,0.0004692344,0.9324636,0.05103766,0.00008184709],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04522227,0.114496,0.3899063,0.197796,0.002392072,0.000667988,0.0005222969,0.00158752,0.2474095],"genre_scores_gemma":[0.8008724,0.02085255,0.1650681,0.00451527,0.001594997,0.0005914017,0.0001909224,0.0003763181,0.005938022],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.09083906,"threshold_uncertainty_score":0.4804086,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1213552028661894,"score_gpt":0.4269703605473044,"score_spread":0.305615157681115,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}