{"id":"W4382318396","doi":"10.1609/aaai.v37i12.26698","title":"Everyone’s Voice Matters: Quantifying Annotation Disagreement Using Demographic Information","year":2023,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Topic Modeling","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Annotation; Computer science; Demographics; Variance (accounting); Perspective (graphical); Task (project management); Artificial intelligence; Natural language processing; Politeness; Process (computing); Linguistics; Sociology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02029714,0.0009490692,0.000671521,0.002779912,0.001855536,0.001843166,0.001115898,0.001650257,0.001119524],"category_scores_gemma":[0.07717023,0.0003693874,0.0005632956,0.00217286,0.0009947806,0.003500276,0.002481545,0.001485851,0.0008994301],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001273771,"about_ca_system_score_gemma":0.0008947938,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004896206,"about_ca_topic_score_gemma":0.006171933,"domain_scores_codex":[0.9876358,0.007419891,0.001012972,0.001838797,0.001737309,0.000355175],"domain_scores_gemma":[0.9296768,0.04993531,0.006602119,0.004657003,0.008156794,0.0009719364],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00226065,0.0007818303,0.7464964,0.000766376,0.0004267037,0.0003680426,0.01786052,0.03340018,0.01348631,0.009139551,0.02054522,0.1544682],"study_design_scores_gemma":[0.0002243826,0.0004546982,0.4135848,0.0003694716,0.0002838041,0.0006237437,0.01183644,0.4794059,0.02450285,0.03606649,0.03228461,0.0003628492],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8937303,0.0006601561,0.08831621,0.001180431,0.000179009,0.0003497455,0.005396165,0.001055146,0.009132924],"genre_scores_gemma":[0.9566858,0.0001080108,0.03134607,0.0002611883,0.0001102504,0.0005373606,0.009488451,0.0001736125,0.001289183],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02029714,"threshold_uncertainty_score":0.1073428,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1677698905391123,"score_gpt":0.3288002364177132,"score_spread":0.1610303458786008,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}