{"id":"W4393541042","doi":"10.5281/zenodo.5130737","title":"SemEval-2021 Task 12: Learning with Disagreements","year":2021,"lang":"en","type":"dataset","venue":"IT University Of Copenhagen (IT University of Copenhagen)","topic":"Intelligent Tutoring Systems and Adaptive Learning","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Austrian Science Fund; European Commission","keywords":"SemEval; Task (project management); Computer science; Engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0007352845,0.0007042866,0.001406448,0.0004225046,0.0008895852,0.0001594662,0.003935079,0.0004826984,0.04970485],"category_scores_gemma":[0.00007284564,0.0008780283,0.0004924026,0.0008535599,0.0003457941,0.001171408,0.001989126,0.001050952,0.008617975],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003603989,"about_ca_system_score_gemma":0.001054457,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004225166,"about_ca_topic_score_gemma":0.002007865,"domain_scores_codex":[0.9952739,0.0007489292,0.0005071372,0.001363195,0.001360859,0.0007459898],"domain_scores_gemma":[0.9954396,0.0002838621,0.001490454,0.001564978,0.0008684403,0.0003526084],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000166007,0.0002252259,0.00001567229,0.0003902629,0.0007487125,0.001722537,0.00174577,0.001221569,0.0001915062,0.000151762,0.9924888,0.0009321561],"study_design_scores_gemma":[0.001338134,0.00053892,0.000199779,0.001456281,0.0004017851,0.00002770226,0.006300539,0.0002330814,0.0002729437,0.000001006765,0.9883644,0.0008654512],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"other","genre_scores_codex":[0.003330642,0.003842599,0.3615919,0.001847298,0.002743841,0.003691841,0.4511433,0.0004183682,0.1713902],"genre_scores_gemma":[0.007993947,0.001018855,0.006197337,0.0001280797,0.0001426305,2.649374e-7,0.1353384,0.00005852482,0.8491219],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.6777317,"threshold_uncertainty_score":0.9993671,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0137701187301626,"score_gpt":0.2005554228492351,"score_spread":0.1867853041190725,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}