{"id":"W2418050709","doi":"10.1109/tcyb.2017.2766189","title":"Learning Stylometric Representations for Authorship Analysis","year":2017,"lang":"en","type":"preprint","venue":"IEEE Transactions on Cybernetics","topic":"Authorship Attribution and Profiling","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"Zayed University; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Computer science; Natural language processing; Word2vec; Artificial intelligence; Latent Dirichlet allocation; Representation (politics); Latent semantic analysis; Feature (linguistics); Sentence; Set (abstract data type); Psycholinguistics; Feature engineering; Attributive; Linguistics; Topic model; Deep learning; Politics; Psychology","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001395991,0.0008773621,0.0007718502,0.005859883,0.0005472648,0.001568541,0.0008920566,0.0008952091,0.0018016],"category_scores_gemma":[0.0115198,0.0002955635,0.000662572,0.004720065,0.00064282,0.002844865,0.001188574,0.001174803,0.00125071],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001120923,"about_ca_system_score_gemma":0.000993741,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00133447,"about_ca_topic_score_gemma":0.002235306,"domain_scores_codex":[0.9988181,0.0004216175,0.0001249916,0.000294931,0.0002345735,0.0001057944],"domain_scores_gemma":[0.9955433,0.002207495,0.00069999,0.0007881402,0.000624625,0.000136567],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002745896,0.0003967172,0.01749718,0.0001413245,0.0001587518,0.0001815447,0.0004292885,0.1705281,0.004737266,0.03100101,0.0106972,0.7639571],"study_design_scores_gemma":[0.0000129572,0.00002721204,0.001681177,0.0000201026,0.00001636139,0.00004990823,0.00006785746,0.9512551,0.001422602,0.04297394,0.002455791,0.00001703029],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.09828579,0.0007905478,0.8927077,0.0005709276,0.0001472938,0.0001353457,0.001761698,0.002647629,0.00295308],"genre_scores_gemma":[0.8662016,0.0005423117,0.1253521,0.00009825144,0.0002853643,0.0002896248,0.004070089,0.0001506507,0.003009981],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005859883,"threshold_uncertainty_score":0.008132935,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07844246020336623,"score_gpt":0.3524052087635211,"score_spread":0.2739627485601549,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}