{"id":"W3198457396","doi":"10.18653/v1/2021.emnlp-main.259","title":"Learning from Multiple Noisy Augmented Data Sets for Better Cross-Lingual Spoken Language Understanding","year":2021,"lang":"en","type":"article","venue":"Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Benchmark (surveying); Focus (optics); Noise (video); Code (set theory); Artificial intelligence; Training set; Machine learning; Spoken language; Resource (disambiguation); Noise reduction; Natural language processing; Speech recognition; Image (mathematics); Set (abstract data type)","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042676,0.002701546,0.002076249,0.001730398,0.001056099,0.003122623,0.002706836,0.002644282,0.004371716],"category_scores_gemma":[0.0153662,0.001053637,0.002489953,0.001855541,0.001406558,0.00583303,0.006328544,0.005297908,0.005076002],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007332758,"about_ca_system_score_gemma":0.001883215,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005940411,"about_ca_topic_score_gemma":0.01034889,"domain_scores_codex":[0.9955645,0.001855276,0.0002619963,0.001500295,0.0005838185,0.0002341198],"domain_scores_gemma":[0.991971,0.003786085,0.00030542,0.002572512,0.001196744,0.0001681702],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005967362,0.0007786389,0.005311682,0.0007123644,0.0005924959,0.0005149361,0.001732161,0.1053595,0.04347101,0.005847394,0.01966492,0.8154182],"study_design_scores_gemma":[0.00006635761,0.0002950373,0.002328047,0.0001310441,0.0001269569,0.0002544672,0.0009743865,0.9354633,0.02544989,0.01962884,0.0151581,0.0001235968],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0616038,0.001603002,0.9151431,0.0007920721,0.0004010904,0.0001948661,0.002226178,0.0145724,0.003463434],"genre_scores_gemma":[0.3517685,0.0006649341,0.6231773,0.0006669614,0.0001914008,0.0005908122,0.01607111,0.001329834,0.005539086],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005940411,"threshold_uncertainty_score":0.02256954,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1489856121254087,"score_gpt":0.4560963868956432,"score_spread":0.3071107747702345,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}