{"id":"W4389519444","doi":"10.18653/v1/2023.emnlp-main.19","title":"Understanding Compositional Data Augmentation in Typologically Diverse Morphological Inflection","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Inflection; Complement (music); Phonotactics; Set (abstract data type); Artificial intelligence; Natural language processing; Linguistics; Phonology","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003509832,0.000839252,0.0005606998,0.001174293,0.0006098081,0.00179872,0.001434221,0.001198088,0.001773857],"category_scores_gemma":[0.01463455,0.0004486863,0.0008390749,0.001126348,0.001260738,0.002865305,0.002404646,0.001640741,0.001170581],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003681786,"about_ca_system_score_gemma":0.0007703342,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009046459,"about_ca_topic_score_gemma":0.002768495,"domain_scores_codex":[0.9983394,0.0008238487,0.0001092032,0.0004059103,0.0002444356,0.00007712626],"domain_scores_gemma":[0.9909799,0.005944103,0.0004396219,0.001822711,0.0006696322,0.000144143],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001384077,0.0005698319,0.03840235,0.0009981076,0.0002469937,0.0008678125,0.00232296,0.1244537,0.08648256,0.01344738,0.008293433,0.7225309],"study_design_scores_gemma":[0.00009761815,0.00043564,0.0112488,0.0001477754,0.00008266403,0.000870374,0.001352903,0.8698169,0.06138025,0.03926067,0.01523874,0.00006764746],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5591352,0.001492783,0.4258286,0.001003731,0.000153274,0.000189194,0.001986911,0.005520737,0.00468951],"genre_scores_gemma":[0.725943,0.0003205898,0.2660303,0.0002406439,0.00005542595,0.0002331918,0.005597387,0.0003430878,0.001236375],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.003509832,"threshold_uncertainty_score":0.01856202,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3123113950205255,"score_gpt":0.378841462431484,"score_spread":0.06653006741095846,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}