{"id":"W4200633865","doi":"10.1145/3512943","title":"Human-AI Collaboration for UX Evaluation: Effects of Explanation and Synchronization","year":2022,"lang":"en","type":"article","venue":"Proceedings of the ACM on Human-Computer Interaction","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":68,"is_retracted":false,"has_abstract":true,"ca_institutions":"Microsoft (Canada); University of Waterloo","funders":"","keywords":"Usability; Computer science; Human–computer interaction; Context (archaeology); Wizard of oz; Asynchronous communication; Synchronization (alternating current); Test (biology); User experience design","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03927365,0.001482924,0.001132308,0.001708591,0.001463879,0.003190272,0.001153489,0.001219508,0.003179026],"category_scores_gemma":[0.3282222,0.0006450013,0.0009119106,0.001223395,0.001505794,0.003018906,0.002825736,0.001555481,0.000309812],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001652015,"about_ca_system_score_gemma":0.001643279,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001586263,"about_ca_topic_score_gemma":0.001725945,"domain_scores_codex":[0.9459229,0.04120912,0.004654759,0.002561668,0.004454816,0.001196807],"domain_scores_gemma":[0.3099674,0.6462415,0.01958339,0.009012153,0.01199671,0.003198782],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0271309,0.01270278,0.16506,0.00919831,0.001360528,0.000744618,0.09597142,0.01123439,0.06763965,0.002280322,0.004552825,0.6021243],"study_design_scores_gemma":[0.005261269,0.07751691,0.7212076,0.002642326,0.002550309,0.0006366667,0.04675638,0.05334432,0.06709927,0.005868292,0.01608168,0.001035154],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9820238,0.0004432353,0.01223787,0.0002852888,0.00009681458,0.001409375,0.0001048116,0.0003601212,0.00303866],"genre_scores_gemma":[0.9795385,0.0001410575,0.01697964,0.0001369392,0.00003577831,0.002264783,0.0001514005,0.0001214232,0.0006305046],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03927365,"threshold_uncertainty_score":0.2077014,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04460961568483576,"score_gpt":0.3544957742154282,"score_spread":0.3098861585305924,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}