{"id":"W140340549","doi":"10.63317/2id75yrzanby","title":"Using the Complexity of the Distribution of Lexical Elements as a Feature in Authorship Attribution","year":2008,"lang":"en","type":"article","venue":"","topic":"Authorship Attribution and Profiling","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"Punctuation; Authorship attribution; Computer science; Artificial intelligence; Natural language processing; Feature (linguistics); Support vector machine; Noun; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003859967,0.0008225345,0.0008928261,0.005572944,0.0006123842,0.002213079,0.000797908,0.001250066,0.001196789],"category_scores_gemma":[0.03419756,0.0004218061,0.0007751461,0.003594026,0.001066591,0.00518641,0.00134692,0.001181695,0.0007398419],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006531607,"about_ca_system_score_gemma":0.0004236531,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005627148,"about_ca_topic_score_gemma":0.0006002139,"domain_scores_codex":[0.9970521,0.0009468131,0.0003719454,0.0005557194,0.0009085485,0.0001649826],"domain_scores_gemma":[0.9538644,0.03501554,0.004908878,0.00381051,0.001883395,0.000517314],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001162271,0.0006210837,0.2231395,0.0005398668,0.000393991,0.0007752111,0.001181705,0.07765642,0.06562121,0.01102933,0.002358966,0.6155204],"study_design_scores_gemma":[0.00003980326,0.0004823197,0.1086611,0.00006656859,0.0001338281,0.001627987,0.0003200019,0.8027042,0.05200565,0.03160568,0.002101222,0.0002515849],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.53397,0.0003722316,0.4618914,0.000258886,0.00008609093,0.00008823915,0.0005997791,0.001090372,0.001642988],"genre_scores_gemma":[0.9539549,0.0001300282,0.04479859,0.00001597687,0.00006542081,0.00004926448,0.0003902194,0.00006203394,0.0005335411],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.005572944,"threshold_uncertainty_score":0.0204137,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2062589589989079,"score_gpt":0.3485729465110212,"score_spread":0.1423139875121133,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}