{"id":"W2891582949","doi":"10.18653/v1/d18-1294","title":"Syntax Encoding with Application in Authorship Attribution","year":2018,"lang":"en","type":"article","venue":"","topic":"Authorship Attribution and Profiling","field":"Computer Science","cited_by":46,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa; National Research Council Canada","funders":"National Key Research and Development Program of China; Beijing Advanced Innovation Center for Big Data and Brain Computing; State Key Laboratory of Software Development Environment; National Natural Science Foundation of China","keywords":"Computer science; Syntax; Natural language processing; Sentence; Encoding (memory); Artificial intelligence; Abstract syntax; Embedding; Parsing; ENCODE; Abstract syntax tree; Word (group theory); Representation (politics); Parse tree; Tree (set theory); Lossless compression; Linguistics; Data compression; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001883016,0.001094659,0.000814672,0.002771507,0.0007201408,0.002116267,0.001609157,0.001745608,0.005557085],"category_scores_gemma":[0.01650161,0.0004500731,0.001024017,0.003185831,0.001290813,0.009693616,0.003089441,0.002641221,0.002131433],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00102766,"about_ca_system_score_gemma":0.001447355,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001214652,"about_ca_topic_score_gemma":0.001333929,"domain_scores_codex":[0.9980837,0.0007646951,0.0002196287,0.0004899232,0.0003157928,0.0001263051],"domain_scores_gemma":[0.9938486,0.002817108,0.0006110903,0.00182159,0.0007388659,0.0001627509],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000368459,0.000305621,0.004252607,0.0004915275,0.00009409347,0.0003921291,0.0008380603,0.0867096,0.01152768,0.1221491,0.01576093,0.7571102],"study_design_scores_gemma":[0.00006175503,0.0001114187,0.0009462601,0.0001193144,0.00007206241,0.000342382,0.0002141633,0.5458046,0.01514004,0.422738,0.01435462,0.00009546827],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02031086,0.0004691565,0.9684189,0.0009009428,0.0002101089,0.0001032393,0.001572603,0.005841017,0.002173244],"genre_scores_gemma":[0.5213835,0.0006335392,0.4684877,0.0004707898,0.0002944492,0.0003048842,0.003735293,0.0007247025,0.003965071],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005557085,"threshold_uncertainty_score":0.01859033,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02970367137682244,"score_gpt":0.2849840804760073,"score_spread":0.2552804090991848,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}