{"id":"W3203983689","doi":"10.18653/v1/2021.inlg-1.26","title":"Shared Task in Evaluating Accuracy: Leveraging Pre-Annotations in the Validation Process","year":2021,"lang":"en","type":"article","venue":"","topic":"Anomaly Detection Techniques and Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université Laval","funders":"","keywords":"Computer science; Task (project management); Interface (matter); Process (computing); Protocol (science); Web application; Annotation; Information retrieval; User interface; Artificial intelligence; World Wide Web; Natural language processing; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1375851,0.006336552,0.005233755,0.005370777,0.005115753,0.01342843,0.007763496,0.009264222,0.02094504],"category_scores_gemma":[0.3007066,0.002026255,0.003936925,0.003482721,0.004198335,0.008087642,0.02195185,0.009208239,0.02735777],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003637232,"about_ca_system_score_gemma":0.0102218,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004825667,"about_ca_topic_score_gemma":0.005860172,"domain_scores_codex":[0.782735,0.1250629,0.01646036,0.01798028,0.05077783,0.006983706],"domain_scores_gemma":[0.4656233,0.1881544,0.01036755,0.1431834,0.1757529,0.01691843],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.005013959,0.002564531,0.009630659,0.002612984,0.001103928,0.0009543876,0.002252462,0.01618075,0.0247633,0.007133025,0.6774949,0.250295],"study_design_scores_gemma":[0.003380248,0.005999383,0.03264365,0.002131367,0.0009644645,0.002441353,0.003283867,0.3197234,0.1371138,0.07497615,0.4155228,0.001819562],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.07739575,0.002576497,0.7242469,0.009002676,0.02004715,0.008084344,0.03715644,0.08775993,0.03373034],"genre_scores_gemma":[0.2645214,0.0005898307,0.523729,0.00439282,0.004358219,0.01205279,0.1172429,0.04196198,0.03115093],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.1375851,"threshold_uncertainty_score":0.7276285,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05627887284597749,"score_gpt":0.3708462692766979,"score_spread":0.3145673964307204,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}