{"id":"W4385572574","doi":"10.18653/v1/2023.findings-acl.823","title":"Better Language Models of Code through Self-Improvement","year":2023,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Automatic summarization; Code (set theory); Benchmark (surveying); Language model; Artificial intelligence; Machine learning; Code generation; Natural language processing; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00013065,0.00006162678,0.00008850709,0.00003722953,0.00002224357,0.00002331379,0.0004524015,0.00002622754,0.00001903888],"category_scores_gemma":[0.0000031218,0.00005139762,0.00003467202,0.0002004672,0.000006985496,0.0003009299,0.0002836976,0.00004359779,0.00006121759],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001402471,"about_ca_system_score_gemma":0.00001785818,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001034242,"about_ca_topic_score_gemma":0.00001036248,"domain_scores_codex":[0.9992498,0.00001070969,0.0001566905,0.0002096446,0.0002015859,0.0001715966],"domain_scores_gemma":[0.9994199,0.00002477912,0.0000326686,0.000475286,0.00002415393,0.00002319179],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000002941502,0.000157593,0.0003324017,0.0001503565,0.0001012869,0.00005356361,0.03678638,0.03258295,0.03224438,0.7454816,0.01232315,0.1397834],"study_design_scores_gemma":[0.0001425773,0.0000229387,0.00003598408,0.000005768646,0.000002166705,6.951243e-7,0.0001104618,0.9601518,0.01832928,0.02059651,0.0005240429,0.00007775975],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1399495,0.00001481033,0.8475964,0.001262428,0.0001228496,0.00008221978,0.000001417832,0.0004285559,0.01054187],"genre_scores_gemma":[0.7807423,0.000006700026,0.2171723,0.001054575,0.00003611335,0.000008553757,0.000001112747,0.00000549512,0.0009728649],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9275689,"threshold_uncertainty_score":0.2095934,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03261112974691838,"score_gpt":0.2701301621696314,"score_spread":0.2375190324227131,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}