{"id":"W2959891429","doi":"10.1109/icsme.2019.00048","title":"Syntax and Stack Overflow: A Methodology for Extracting a Corpus of Syntax Errors and Fixes","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Syntax error; Computer science; Python (programming language); Syntax; Abstract syntax tree; Parsing; Programming language; Natural language processing; Artificial intelligence; Abstract syntax; Source code","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009192422,0.001922387,0.00109051,0.02067979,0.002310793,0.003245613,0.002288743,0.002171085,0.007872728],"category_scores_gemma":[0.05485969,0.001322126,0.001916758,0.01595701,0.002324949,0.005088875,0.006178858,0.003156539,0.009604675],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001629259,"about_ca_system_score_gemma":0.00789493,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008034229,"about_ca_topic_score_gemma":0.01508501,"domain_scores_codex":[0.9846417,0.002826707,0.003879476,0.00355051,0.004523292,0.000578225],"domain_scores_gemma":[0.9460947,0.02156284,0.007728657,0.01196558,0.01170847,0.0009397239],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007151351,0.000693257,0.05603904,0.009544227,0.0005012681,0.002325414,0.01673751,0.005803806,0.04514198,0.02685911,0.3234942,0.512145],"study_design_scores_gemma":[0.0002580607,0.000338489,0.1141728,0.001233781,0.0002723518,0.002002025,0.00526405,0.03247215,0.05457691,0.03261895,0.7562016,0.0005888485],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"methods","genre_scores_codex":[0.09818535,0.0012197,0.3851356,0.001567653,0.0006648523,0.005575743,0.4346781,0.05557961,0.01739338],"genre_scores_gemma":[0.05186458,0.0004675421,0.5641948,0.0004380079,0.0001468884,0.007767276,0.3620385,0.006586631,0.006495789],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.02067979,"threshold_uncertainty_score":0.04861474,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09427978240340419,"score_gpt":0.3505228612023415,"score_spread":0.2562430787989373,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}