{"id":"W1997305228","doi":"10.1109/nlpke.2010.5587767","title":"Automatic classification of documents by formality","year":2010,"lang":"en","type":"article","venue":"","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"Formality; Computer science; Naive Bayes classifier; Artificial intelligence; Feature selection; Classifier (UML); Support vector machine; Machine learning; Decision tree; Task (project management); Style (visual arts); Data mining; Natural language processing; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001812396,0.0007710324,0.0008019537,0.009146565,0.0007853853,0.002232893,0.0006996752,0.0005198816,0.001936466],"category_scores_gemma":[0.012998,0.0002437104,0.0006863637,0.004923573,0.0004716464,0.002609891,0.0007840706,0.0009791016,0.001435208],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007588942,"about_ca_system_score_gemma":0.0008199032,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00172188,"about_ca_topic_score_gemma":0.001926642,"domain_scores_codex":[0.9977398,0.0004214641,0.0002947435,0.0003813098,0.0009721749,0.000190486],"domain_scores_gemma":[0.9874471,0.005786377,0.001770901,0.001154043,0.003453638,0.000387961],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005422566,0.0002951619,0.06395314,0.000605014,0.0001124339,0.0002532456,0.0006561862,0.005414502,0.03024001,0.005104483,0.01131719,0.8815064],"study_design_scores_gemma":[0.0002541945,0.0007862133,0.2143134,0.0006087339,0.0003826074,0.00268039,0.00220722,0.5464178,0.1013696,0.05300361,0.07757604,0.00040012],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6312392,0.00434199,0.3294421,0.001078757,0.0005233919,0.0007694658,0.009952767,0.009130212,0.01352211],"genre_scores_gemma":[0.7617556,0.001061202,0.2217862,0.00009763128,0.0003179365,0.000231658,0.01096443,0.0002525846,0.003532722],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.009146565,"threshold_uncertainty_score":0.009585023,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01347406047579485,"score_gpt":0.2704840205753174,"score_spread":0.2570099600995225,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}