{"id":"W6950269185","doi":"10.5281/zenodo.6977583","title":"NatGen: Generative pre-training by \"Naturalizing\" source code","year":2022,"lang":"en","type":"other","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Radioactive contamination and transfer","field":"Environmental Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Discovery Air (Canada)","funders":"","keywords":"Source code; Code (set theory); Generative grammar; Natural language; Generative model; Code generation; KPI-driven code analysis; Software","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001596483,0.001568546,0.0006311216,0.000922886,0.0004338268,0.0009738843,0.003376386,0.001493137,0.005334158],"category_scores_gemma":[0.006738058,0.0009385393,0.001502019,0.0006226004,0.001540293,0.001967897,0.001889193,0.00358741,0.003228845],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001362379,"about_ca_system_score_gemma":0.001459188,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00537781,"about_ca_topic_score_gemma":0.01496201,"domain_scores_codex":[0.9989285,0.0003348514,0.00003933722,0.0004165996,0.000209414,0.00007130038],"domain_scores_gemma":[0.996388,0.002134326,0.0001323924,0.0008793607,0.0003696413,0.00009626021],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0003691544,0.0004296358,0.005765255,0.000755787,0.0002435419,0.0003549118,0.0005784272,0.4069185,0.02599122,0.02002771,0.04916174,0.4894041],"study_design_scores_gemma":[0.00003072634,0.00006917799,0.0003474346,0.00003025525,0.00001894167,0.00009053083,0.00003402157,0.9694858,0.01166506,0.008737632,0.009468812,0.00002166117],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"software","genre_scores_codex":[0.03329166,0.0005584122,0.9106107,0.0005272288,0.0002174556,0.000283446,0.001427941,0.04717302,0.005910111],"genre_scores_gemma":[0.2969196,0.0003598289,0.6723456,0.001363983,0.00007056893,0.0009457645,0.0109987,0.00558818,0.01140766],"genre_candidate":"software","genre_consensus":null,"teacher_disagreement_score":0.00537781,"threshold_uncertainty_score":0.01784456,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02309372474196294,"score_gpt":0.233985160368437,"score_spread":0.2108914356264741,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}