{"id":"W3166208174","doi":"10.5715/jnlp.28.350","title":"The Effectiveness of Data Augmentation by Removing Unimportant sentence","year":2021,"lang":"en","type":"article","venue":"Journal of Natural Language Processing","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Tokyo Metropolitan University; Institute for Catastrophic Loss Reduction","keywords":"Pointer (user interface); Computer science; Generator (circuit theory); Sentence; Natural language processing; Programming language; Artificial intelligence; Physics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001557391,0.00006730867,0.0001376373,0.00003351466,0.0001121365,0.0001625354,0.0008955447,0.00002416335,8.082385e-7],"category_scores_gemma":[0.0003183104,0.00004312297,0.00003466017,0.0002595547,0.00002407271,0.001222563,0.000259256,0.000228646,1.534254e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005318563,"about_ca_system_score_gemma":0.0002969134,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001138826,"about_ca_topic_score_gemma":0.000004924314,"domain_scores_codex":[0.9988203,0.0001859004,0.0003466935,0.0001538,0.0003706686,0.000122662],"domain_scores_gemma":[0.9985362,0.0003268273,0.0004617805,0.0003318399,0.0003108521,0.00003245011],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00004810722,0.00003293247,0.0004974919,0.0002166573,0.00003548094,0.0002380857,0.001162541,0.0001064442,0.406725,0.000366596,0.00005360396,0.5905171],"study_design_scores_gemma":[0.001539769,0.00008929973,0.003729446,0.002576143,0.00007747539,0.002038607,0.006483342,0.6586515,0.3227194,0.001267633,0.0004554942,0.0003719504],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4151407,0.06788053,0.5158572,0.0004976267,0.0004770703,0.00005338022,0.000002436814,0.00001567386,0.00007539454],"genre_scores_gemma":[0.9532664,0.00004738568,0.04653893,0.00004817953,0.00005703338,1.98525e-7,0.000003326014,0.000003963195,0.00003456314],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.658545,"threshold_uncertainty_score":0.1758504,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01711823828179371,"score_gpt":0.3082740527003838,"score_spread":0.2911558144185901,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}