{"id":"W4386566490","doi":"10.18653/v1/2023.findings-eacl.74","title":"Improving Prediction Backward-Compatiblility in NLP Model Upgrade with Gated Fusion","year":2023,"lang":"en","type":"article","venue":"","topic":"Domain Adaptation and Few-Shot Learning","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Upgrade; Computer science; Regression; Artificial intelligence; Regression analysis; Machine learning; Ensemble forecasting; Baseline (sea); Data mining; Statistics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00375202,0.002317157,0.002099829,0.001179892,0.0009989928,0.001500729,0.002755282,0.001955956,0.002224668],"category_scores_gemma":[0.01190396,0.0010152,0.001844164,0.001075255,0.0009360513,0.005126885,0.004148183,0.004011328,0.001179015],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008231618,"about_ca_system_score_gemma":0.00173031,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009339998,"about_ca_topic_score_gemma":0.01127107,"domain_scores_codex":[0.9979026,0.0005996512,0.0001176965,0.0006809671,0.0004560731,0.0002429944],"domain_scores_gemma":[0.9956458,0.002162581,0.0002029976,0.001223926,0.0006181258,0.0001464994],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008083174,0.0005986012,0.0065543,0.0002427617,0.0004089232,0.0006128533,0.0005595087,0.3993176,0.02722147,0.004978384,0.0115191,0.5471781],"study_design_scores_gemma":[0.0000198988,0.00008334356,0.0005347875,0.00001426697,0.00006478669,0.00007119044,0.00002750261,0.9881487,0.006742799,0.00314286,0.00112397,0.00002585717],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1393581,0.001290597,0.8373474,0.0006795808,0.0003526498,0.0001599725,0.0005644378,0.01719088,0.003056328],"genre_scores_gemma":[0.8042721,0.0003576439,0.1878457,0.00070446,0.000173002,0.0001139373,0.002238479,0.0008382568,0.003456553],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.009339998,"threshold_uncertainty_score":0.0198428,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02558255835228114,"score_gpt":0.2397477656793226,"score_spread":0.2141652073270414,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}