{"id":"W6910380341","doi":"10.48448/51d3-bk86","title":"Better Language Models of Code through Self-Improvement","year":2022,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Automatic summarization; Code (set theory); Language model; Code generation; Source code; Data modeling; Code review","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.001023199,0.0004519385,0.0005383198,0.0007939253,0.0001967783,0.00007060391,0.001971503,0.0001516693,0.01247799],"category_scores_gemma":[0.00004225596,0.000416271,0.000110072,0.001579218,0.001037676,0.0003722579,0.0009973308,0.0004582689,0.0007401723],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006603146,"about_ca_system_score_gemma":0.0007454819,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001252831,"about_ca_topic_score_gemma":0.0006656881,"domain_scores_codex":[0.9953435,0.00006878161,0.0005067437,0.001063451,0.002241017,0.0007765167],"domain_scores_gemma":[0.9976279,0.00004187049,0.0006680578,0.001419963,0.0001088317,0.0001334381],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00002294597,0.0011984,0.0001148875,0.000329488,0.0002668194,0.00007108703,0.006154811,0.002206784,0.04874425,0.01750192,0.9161199,0.00726866],"study_design_scores_gemma":[0.001846735,0.000622844,0.00001328205,0.0001956826,0.0002663022,0.00002140063,0.003196637,0.05827633,0.008283105,0.0064335,0.9190612,0.001782927],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.002087887,0.0007414405,0.002571645,0.000283573,0.0006914302,0.001235486,0.002203136,0.001150979,0.9890344],"genre_scores_gemma":[0.1022567,0.0001865072,0.2362825,0.005817906,0.001489765,0.0003952376,0.001100226,0.004883975,0.6475872],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.3414472,"threshold_uncertainty_score":0.9998289,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02440972853770474,"score_gpt":0.2977335995499219,"score_spread":0.2733238710122172,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}