{"id":"W2250104175","doi":"10.33011/lilt.v8i.1305","title":"Learning to Classify Documents According to Formal and Informal Style","year":2012,"lang":"en","type":"article","venue":"Linguistic Issues in Language Technology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; University of Ottawa","keywords":"Computer science; Artificial intelligence; Naive Bayes classifier; Classifier (UML); Natural language processing; Style (visual arts); Sentence; Support vector machine; Machine learning; Computational linguistics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002703059,0.00103267,0.0009098883,0.006559351,0.0008024189,0.002956486,0.0009218964,0.001118588,0.002424589],"category_scores_gemma":[0.01297167,0.00029888,0.001046455,0.003243872,0.0005609904,0.004260302,0.0007647794,0.001341599,0.002363825],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007568547,"about_ca_system_score_gemma":0.0008636098,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001506052,"about_ca_topic_score_gemma":0.002222168,"domain_scores_codex":[0.9981309,0.0004652395,0.0002718549,0.000489357,0.0004947186,0.000147954],"domain_scores_gemma":[0.990035,0.005702756,0.001115651,0.0007837943,0.002009786,0.0003530729],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003767578,0.0006039366,0.08189279,0.0005945133,0.0002090599,0.0002207096,0.001160684,0.006937195,0.01224151,0.005089467,0.01361054,0.8770629],"study_design_scores_gemma":[0.0003293717,0.0013249,0.1342307,0.0007679242,0.0005874664,0.001727268,0.005485964,0.6720929,0.03896581,0.08714084,0.05700776,0.0003390627],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4515367,0.002444508,0.5185531,0.002133291,0.0005416019,0.001361643,0.006031853,0.003500974,0.01389622],"genre_scores_gemma":[0.5958941,0.001140482,0.3881144,0.0003779994,0.0006093243,0.0006381426,0.008517098,0.0001559254,0.004552459],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006559351,"threshold_uncertainty_score":0.01429528,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.00676897105367181,"score_gpt":0.307116565836935,"score_spread":0.3003475947832632,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}