{"id":"W4306694236","doi":"10.1609/hcomp.v10i1.21986","title":"Eliciting and Learning with Soft Labels from Every Annotator","year":2022,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Human Computation and Crowdsourcing","topic":"Mobile Crowdsensing and Crowdsourcing","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Engineering and Physical Sciences Research Council; Canadian Institute for Advanced Research","keywords":"Computer science; Categorical variable; Crowdsourcing; Artificial intelligence; Machine learning; Robustness (evolution); Generalization; Set (abstract data type); Soft skills; Natural language processing; World Wide Web","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04146983,0.002886866,0.002105475,0.002948971,0.004610458,0.004103038,0.003499218,0.003780185,0.008652726],"category_scores_gemma":[0.1106495,0.001626424,0.001822814,0.003641654,0.004149523,0.007918087,0.01155762,0.006739689,0.008014644],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003028555,"about_ca_system_score_gemma":0.008564414,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004838172,"about_ca_topic_score_gemma":0.01888795,"domain_scores_codex":[0.948999,0.03356538,0.002291614,0.008119404,0.005826782,0.001197835],"domain_scores_gemma":[0.8619913,0.06826486,0.007511863,0.04161943,0.01748443,0.003128052],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002411801,0.0009170792,0.03055581,0.003415169,0.0004273346,0.001057483,0.01717464,0.04485076,0.07421295,0.1057422,0.1450794,0.5741553],"study_design_scores_gemma":[0.0005005738,0.0005651163,0.008646972,0.001040115,0.0002817837,0.0009642007,0.007603456,0.2619488,0.04911944,0.4120473,0.2568296,0.0004526556],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.03114075,0.0005100684,0.9391757,0.004017902,0.0005023507,0.0009666719,0.00338992,0.003347327,0.01694942],"genre_scores_gemma":[0.2201826,0.0003900514,0.7511899,0.00258629,0.0004230202,0.004173973,0.007790813,0.001433832,0.01182956],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.04146983,"threshold_uncertainty_score":0.219316,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02357610624059404,"score_gpt":0.240562109562208,"score_spread":0.216986003321614,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}