{"id":"W4221130019","doi":"10.12688/openreseurope.14507.1","title":"An open-source natural language processing toolkit to support software development: addressing automatic bug detection, code summarisation and code search","year":2022,"lang":"en","type":"article","venue":"Open Research Europe","topic":"Software Engineering Research","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Concordia University","funders":"Horizon 2020 Framework Programme; European Commission","keywords":"Computer science; Software engineering; Codebase; Code review; Source code; Parsing; Code generation; Documentation; Software; Code (set theory); KPI-driven code analysis; TRACE (psycholinguistics); Software development; Programming language; World Wide Web; Static program analysis; Operating system; Key (lock)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002759198,0.002356467,0.0009933846,0.003818257,0.0008704657,0.003437189,0.003566266,0.002239875,0.03023666],"category_scores_gemma":[0.0159535,0.001536997,0.002559121,0.002115149,0.001162126,0.006334193,0.005175177,0.003174801,0.03036959],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001171964,"about_ca_system_score_gemma":0.004139531,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005736014,"about_ca_topic_score_gemma":0.01119958,"domain_scores_codex":[0.9965875,0.0006803453,0.000513921,0.0006983089,0.001339035,0.0001809155],"domain_scores_gemma":[0.991307,0.004349689,0.0006435727,0.00142633,0.001841201,0.0004321727],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0007339215,0.0003379129,0.002563045,0.00715304,0.0003293437,0.001670139,0.002428867,0.009595003,0.03468822,0.03155638,0.4467173,0.4622268],"study_design_scores_gemma":[0.000390472,0.000209761,0.004012279,0.00116057,0.0001482964,0.00199493,0.0006218944,0.1411197,0.03278854,0.06365287,0.7534513,0.0004493966],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"software","genre_scores_codex":[0.002551882,0.0005891891,0.5188619,0.0006074315,0.0002918424,0.0005520386,0.02193074,0.4471705,0.007444424],"genre_scores_gemma":[0.02611975,0.0008505061,0.8032973,0.0006023149,0.0001171033,0.001009588,0.09088328,0.06355022,0.01356994],"genre_candidate":"software","genre_consensus":null,"teacher_disagreement_score":0.03023666,"threshold_uncertainty_score":0.1011517,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09172202267189417,"score_gpt":0.3892732159139229,"score_spread":0.2975511932420288,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}