{"id":"W2911227954","doi":"10.1162/tacl_a_00041","title":"Data Statements for Natural Language Processing: Toward Mitigating System Bias and Enabling Better Science","year":2018,"lang":"en","type":"article","venue":"Transactions of the Association for Computational Linguistics","topic":"Topic Modeling","field":"Computer Science","cited_by":808,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Macquarie University; York University; University of Washington; University of California, San Diego; National Science Foundation","keywords":"Embarrassment; Computer science; Natural (archaeology); Data science; Field (mathematics); Natural language; Lead (geology); Style (visual arts); Engineering ethics; Natural language processing; Psychology; Social psychology","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2543903,0.001285266,0.001519134,0.003926265,0.004139254,0.01144222,0.005334035,0.004844823,0.005221077],"category_scores_gemma":[0.4120362,0.002009786,0.001494599,0.003731263,0.01334146,0.03602486,0.0161058,0.01231708,0.002840397],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003059554,"about_ca_system_score_gemma":0.01102469,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001372118,"about_ca_topic_score_gemma":0.0009106182,"domain_scores_codex":[0.7011495,0.2327128,0.01973961,0.01344035,0.0308386,0.002119228],"domain_scores_gemma":[0.3909179,0.3800997,0.03255863,0.1290555,0.06091979,0.006448459],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0009271836,0.0003164919,0.009593546,0.002263339,0.0001574796,0.0004584835,0.0173786,0.004005092,0.008293481,0.7733111,0.02601784,0.1572773],"study_design_scores_gemma":[0.0003561995,0.0004765967,0.001526683,0.001588329,0.0001688441,0.0004716024,0.003556426,0.03460449,0.02161915,0.6860071,0.2493735,0.0002509995],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.006834033,0.0001822515,0.9646278,0.01934265,0.0004004842,0.0009162328,0.0003886867,0.002939006,0.004368837],"genre_scores_gemma":[0.1102897,0.0002772553,0.871415,0.00985027,0.0004542265,0.003254656,0.001099766,0.001282863,0.002076316],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.7456096,"threshold_uncertainty_score":0.9194695,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06724238139102369,"score_gpt":0.3486267136322955,"score_spread":0.2813843322412718,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}