{"id":"W3203983689","doi":"10.18653/v1/2021.inlg-1.26","title":"Shared Task in Evaluating Accuracy: Leveraging Pre-Annotations in the Validation Process","year":2021,"lang":"en","type":"article","venue":"","topic":"Anomaly Detection Techniques and Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université Laval","funders":"","keywords":"Computer science; Task (project management); Interface (matter); Process (computing); Protocol (science); Web application; Annotation; Information retrieval; User interface; Artificial intelligence; World Wide Web; Natural language processing; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005157653,0.0000619699,0.00006252064,0.00008598794,0.0001213258,0.0002207282,0.0004144673,0.00002792392,0.00002852828],"category_scores_gemma":[0.0001770066,0.00005100372,0.00002326436,0.00126044,0.00000851011,0.0005347757,0.00007031875,0.0001268791,0.000009337061],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003778052,"about_ca_system_score_gemma":0.0001153626,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005947572,"about_ca_topic_score_gemma":0.00006728365,"domain_scores_codex":[0.9990812,0.0001084945,0.0002318162,0.0002630807,0.0001950882,0.0001203247],"domain_scores_gemma":[0.9992893,0.000178939,0.0000662035,0.0003360704,0.0001146958,0.00001477031],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001198966,0.001153996,0.03100351,0.0001339132,0.00002523805,0.00004363094,0.06669315,0.03885399,0.04170628,0.3331304,0.001986215,0.4852576],"study_design_scores_gemma":[0.0004403824,0.00004326035,0.1024195,0.0000671035,0.000006064005,0.00003819697,0.002012118,0.787165,0.04063744,0.06607667,0.0007948239,0.0002993914],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2220422,0.00001985845,0.7704316,0.004096076,0.00002214531,0.000315806,0.00000107338,0.000126646,0.002944526],"genre_scores_gemma":[0.9670456,0.000003331033,0.03195982,0.0004914391,0.00001356442,0.0003507925,0.000009523171,0.000003392945,0.0001225182],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7483111,"threshold_uncertainty_score":0.2128487,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05627887284597749,"score_gpt":0.3708462692766979,"score_spread":0.3145673964307204,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}