{"id":"W3198774825","doi":"10.1109/ccece53047.2021.9569061","title":"Code Authorship Attribution using content-based and non-content-based features","year":2021,"lang":"en","type":"article","venue":"","topic":"Authorship Attribution and Profiling","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Lethbridge","funders":"","keywords":"Computer science; Identification (biology); Field (mathematics); Identity (music); Source code; Natural language processing; Artificial intelligence; Natural language; Code (set theory); Writing style; Focus (optics); Computational linguistics; Machine learning; Linguistics; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002264356,0.0007775579,0.0005941732,0.01017669,0.0006274616,0.002273905,0.0005819025,0.0008658969,0.001331618],"category_scores_gemma":[0.02042248,0.0001585414,0.0006239786,0.004919907,0.0004745253,0.00246688,0.001020978,0.0007725772,0.001131742],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008272384,"about_ca_system_score_gemma":0.0007084419,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001582503,"about_ca_topic_score_gemma":0.002077336,"domain_scores_codex":[0.9981518,0.0004752148,0.0002106632,0.0003912456,0.0005787772,0.0001922861],"domain_scores_gemma":[0.9805172,0.01020281,0.003244084,0.001909865,0.003345477,0.0007805179],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001035789,0.0008961143,0.437315,0.0004249842,0.0002590229,0.0005789289,0.0006878151,0.02810821,0.008956268,0.003043864,0.01029374,0.5084002],"study_design_scores_gemma":[0.00003727896,0.0002002475,0.1465025,0.0001081566,0.0000917041,0.0007761987,0.000478765,0.8246467,0.01327444,0.007841344,0.005959395,0.00008322755],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.9038424,0.000787454,0.08149878,0.0006130245,0.0002763938,0.0002767456,0.00373949,0.00231123,0.006654566],"genre_scores_gemma":[0.9776477,0.0001229207,0.01789187,0.00002027686,0.00007875316,0.00005058891,0.002489277,0.00004935763,0.00164935],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.01017669,"threshold_uncertainty_score":0.01197517,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1634173873262401,"score_gpt":0.3116797845410002,"score_spread":0.1482623972147601,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}