{"id":"W3157225713","doi":"10.18653/v1/2021.naacl-main.324","title":"Dynabench: Rethinking Benchmarking in NLP","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University","funders":"Office of Naval Research; Defense Advanced Research Projects Agency","keywords":"Benchmarking; Benchmark (surveying); Computer science; Field (mathematics); Artificial intelligence; Data science; Open source; Machine learning; Programming language; Management","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01968252,0.003269315,0.002126357,0.005789211,0.002611762,0.007945241,0.007088389,0.003533228,0.02256385],"category_scores_gemma":[0.08352941,0.001940551,0.002425446,0.00734752,0.002137455,0.01272092,0.01078463,0.004739875,0.01670418],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002262741,"about_ca_system_score_gemma":0.003611536,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01326843,"about_ca_topic_score_gemma":0.02030404,"domain_scores_codex":[0.9725679,0.01458549,0.002345652,0.004305392,0.005199558,0.0009960062],"domain_scores_gemma":[0.9655352,0.02057136,0.0005098186,0.009370897,0.003276282,0.0007366451],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001551104,0.0005917863,0.006729686,0.003220336,0.001011975,0.0004839703,0.001830685,0.04410926,0.00433333,0.04438995,0.5479149,0.343833],"study_design_scores_gemma":[0.001066677,0.0003257991,0.003982255,0.0007796256,0.0002699324,0.0002930337,0.001359204,0.521293,0.0107301,0.1228615,0.3368216,0.0002171988],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"software","genre_gemma":"methods","genre_scores_codex":[0.0371551,0.0100269,0.4223014,0.005558209,0.004585165,0.0011053,0.03678672,0.4473319,0.03514929],"genre_scores_gemma":[0.2352269,0.002928016,0.4635672,0.002255612,0.0007092394,0.002166494,0.1841341,0.09591017,0.01310223],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9803175,"threshold_uncertainty_score":0.1040923,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04231095882403402,"score_gpt":0.2671027972964434,"score_spread":0.2247918384724094,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}