{"id":"W4393636116","doi":"10.5281/zenodo.5130736","title":"SemEval-2021 Task 12: Learning with Disagreements","year":2021,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Intelligent Tutoring Systems and Adaptive Learning","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Austrian Science Fund; European Commission","keywords":"SemEval; Task (project management); Computer science; Natural language processing; Artificial intelligence; Engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02706224,0.01440477,0.005713674,0.006363254,0.005714148,0.01002195,0.01894375,0.01230865,0.05053979],"category_scores_gemma":[0.08531901,0.002023875,0.006896215,0.005240432,0.003859873,0.01212171,0.01611639,0.01512777,0.06527868],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006530321,"about_ca_system_score_gemma":0.005679352,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01444815,"about_ca_topic_score_gemma":0.02959508,"domain_scores_codex":[0.9647046,0.0139431,0.002433507,0.008638669,0.008147818,0.002132191],"domain_scores_gemma":[0.9496138,0.02041247,0.002269467,0.01612926,0.008663169,0.002911825],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001267685,0.00103635,0.004335904,0.001986936,0.0005354017,0.0002719867,0.0002484913,0.005376874,0.0009701291,0.001856481,0.9153314,0.06678236],"study_design_scores_gemma":[0.004887138,0.003238057,0.01888436,0.003423794,0.0007908774,0.003233987,0.00242642,0.1877352,0.02041401,0.04915484,0.7049701,0.0008412154],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.0581634,0.01801979,0.06711543,0.007894783,0.01047115,0.007220456,0.6627287,0.1181577,0.05022863],"genre_scores_gemma":[0.04286936,0.0006832686,0.05512024,0.002985179,0.0006410074,0.003692157,0.8799421,0.004332948,0.009733728],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.05053979,"threshold_uncertainty_score":0.1690724,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02906174980591631,"score_gpt":0.2382847953472288,"score_spread":0.2092230455413125,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}