{"id":"W4415230017","doi":"10.1609/aies.v8i2.36631","title":"Towards Interactive Evaluations for Interaction Harms in Human-AI Systems","year":2025,"lang":"en","type":"article","venue":"Proceedings of the AAAI/ACM Conference on AI Ethics and Society","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Institute on Governance","funders":"","keywords":"Construct (python library); Corporate governance; Cognition; Natural (archaeology); Work (physics); Social relation","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1265776,0.002095151,0.001425586,0.005294552,0.002202189,0.01163497,0.003299852,0.003575741,0.008976852],"category_scores_gemma":[0.3725432,0.0007734645,0.001305791,0.002138018,0.01176276,0.01656069,0.009151886,0.004655152,0.00103277],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005607813,"about_ca_system_score_gemma":0.004921684,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001720108,"about_ca_topic_score_gemma":0.001627268,"domain_scores_codex":[0.7782889,0.187676,0.005637048,0.005007482,0.0214863,0.001904171],"domain_scores_gemma":[0.5748905,0.3516509,0.01924238,0.02269968,0.02810538,0.003411244],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0009784772,0.0009565202,0.01936725,0.002583201,0.0005560747,0.0002742176,0.01751538,0.05679107,0.003925407,0.5975152,0.01035689,0.2891803],"study_design_scores_gemma":[0.0003101412,0.001243424,0.008869506,0.002045406,0.0003046665,0.0002650123,0.007430824,0.1743449,0.007107979,0.7605559,0.03727893,0.0002432917],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05835068,0.001311861,0.8743413,0.008517993,0.0002728755,0.001790911,0.0002286551,0.001181633,0.05400404],"genre_scores_gemma":[0.7276004,0.0003583335,0.2661363,0.001138304,0.0001081257,0.001866482,0.0002018309,0.0002737164,0.00231644],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.1265776,"threshold_uncertainty_score":0.6694142,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08023321290939993,"score_gpt":0.4229158206196218,"score_spread":0.3426826077102219,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}