{"id":"W6928872153","doi":"10.48448/xqtn-hn45","title":"Everyone's Voice Matters: Quantifying Annotation Disagreement Using Demographic Information","year":2023,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Annotation; Task (project management); Variance (accounting); Demographics; Process (computing); Ground truth","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01378798,0.0009883459,0.0005608741,0.002551412,0.001681075,0.001772991,0.001277187,0.001482727,0.002254617],"category_scores_gemma":[0.05017217,0.0003211027,0.0005120545,0.002123876,0.0008294355,0.002565795,0.002708441,0.001299325,0.002082292],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001352441,"about_ca_system_score_gemma":0.001098873,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007588932,"about_ca_topic_score_gemma":0.01131204,"domain_scores_codex":[0.990497,0.004979033,0.0007725235,0.001696205,0.001756299,0.0002989989],"domain_scores_gemma":[0.9659462,0.02163035,0.00319215,0.003536235,0.00498525,0.0007097971],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002714629,0.001068275,0.5533826,0.001258985,0.000426425,0.0005192609,0.01574969,0.03569778,0.01608059,0.01567103,0.1000023,0.2574284],"study_design_scores_gemma":[0.0003394908,0.000392587,0.3400054,0.0005536449,0.0002322222,0.0007231283,0.009765801,0.4498851,0.03405205,0.04705575,0.116572,0.0004228796],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8195227,0.0009216992,0.1137989,0.002303666,0.0004006203,0.0006714959,0.0301556,0.004488194,0.02773717],"genre_scores_gemma":[0.8582982,0.000181537,0.072634,0.0005915985,0.0001543481,0.001240633,0.06032812,0.0007222702,0.005849167],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.986212,"threshold_uncertainty_score":0.07291865,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06876708129350656,"score_gpt":0.3409416048049575,"score_spread":0.2721745235114509,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}