{"id":"W57814663","doi":"10.1007/3-540-36618-0_24","title":"Combining Naive Bayes and n-Gram Language Models for Text Classification","year":2003,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Topic Modeling","field":"Computer Science","cited_by":170,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Naive Bayes classifier; Computer science; Artificial intelligence; Machine learning; Bayes error rate; Bayesian programming; Bayes' theorem; n-gram; Smoothing; Conditional independence; Inference; Bayes classifier; Bayes factor; Language model; Bayesian probability; Support vector machine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00794703,0.002260653,0.003344682,0.004969921,0.001872768,0.003267921,0.002651901,0.002402897,0.003631886],"category_scores_gemma":[0.01832936,0.001091563,0.002465962,0.004302016,0.0007997833,0.007657116,0.001642032,0.003156049,0.006397765],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001109216,"about_ca_system_score_gemma":0.002330551,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01000369,"about_ca_topic_score_gemma":0.01769334,"domain_scores_codex":[0.9932085,0.003107093,0.0005814174,0.001140546,0.001658545,0.0003039197],"domain_scores_gemma":[0.9870306,0.009381073,0.0003173358,0.0008626188,0.002160272,0.0002481067],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0006561895,0.0005029219,0.00264452,0.0004633141,0.0004070303,0.0001490618,0.0001915941,0.02859251,0.006200324,0.005337635,0.02014458,0.9347103],"study_design_scores_gemma":[0.0000660967,0.0001116202,0.0007435347,0.00008520859,0.0002586707,0.0001948767,0.00009106134,0.9457187,0.004142385,0.04458082,0.003936024,0.00007100831],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01138954,0.00264891,0.9776731,0.0006480637,0.0005076487,0.0002083241,0.0006260628,0.004469686,0.001828618],"genre_scores_gemma":[0.1966299,0.002726698,0.7854684,0.0008517989,0.001601358,0.0005247526,0.004087979,0.0006410909,0.007468045],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01000369,"threshold_uncertainty_score":0.04202843,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03324207233675179,"score_gpt":0.2656846622277709,"score_spread":0.2324425898910191,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}