{"id":"W4375957737","doi":"10.48550/arxiv.2305.04106","title":"On the Usage of Continual Learning for Out-of-Distribution Generalization in Pre-trained Language Models of Code","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Domain Adaptation and Few-Shot Learning","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Robustness (evolution); Encoder; Software; Artificial intelligence; Machine learning; Code (set theory); Language model; Downstream (manufacturing); Source code; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006341424,0.0001231887,0.0002459503,0.0001630174,0.00004765394,0.00001704316,0.0005862692,0.0001219989,0.000004999069],"category_scores_gemma":[0.00023804,0.0001217585,0.0001218069,0.0003842119,0.0000691215,0.0001258392,0.0002835065,0.0002265389,0.000001494044],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005840991,"about_ca_system_score_gemma":0.00008303367,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001171357,"about_ca_topic_score_gemma":0.00008656937,"domain_scores_codex":[0.9989163,0.000219929,0.0002666931,0.0003503624,0.000101834,0.0001448557],"domain_scores_gemma":[0.99858,0.0003962743,0.0004879063,0.000347139,0.0001609775,0.00002773947],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0000408537,0.00003072709,0.0003017244,0.00005846509,0.00001961484,0.000003325996,0.002872379,0.7481979,0.0005030613,0.2477594,0.00003360372,0.0001789644],"study_design_scores_gemma":[0.0004689515,0.00007007438,0.001288447,0.0001322282,0.00001587556,8.217663e-8,0.0004646707,0.9811141,0.000829595,0.01547571,0.0000329168,0.0001073433],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3480211,0.00000680862,0.6513523,0.00003157962,0.000122098,0.0002358604,0.00003349118,0.00003762298,0.0001590913],"genre_scores_gemma":[0.9980735,0.00002192473,0.0008126741,0.00001252575,0.00001284867,0.000001862243,0.0001022812,0.000009794206,0.0009526577],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6505397,"threshold_uncertainty_score":0.4965167,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1116601482219944,"score_gpt":0.2361984450755641,"score_spread":0.1245382968535697,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}