{"id":"W2151752770","doi":"10.1023/b:inrt.0000011209.19643.e2","title":"Augmenting Naive Bayes Classifiers with Statistical Language Models","year":2004,"lang":"en","type":"article","venue":"Information Retrieval","topic":"Topic Modeling","field":"Computer Science","cited_by":247,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Naive Bayes classifier; Artificial intelligence; Computer science; Machine learning; Bayes error rate; Bayes classifier; Bayes' theorem; Classifier (UML); Conditional independence; Bayesian programming; Natural language processing; Pattern recognition (psychology); Bayes factor; Bayesian probability; Support vector machine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007708231,0.001956006,0.002820781,0.004448228,0.001149722,0.00313319,0.002370568,0.00246437,0.004933987],"category_scores_gemma":[0.03147248,0.001034707,0.002005483,0.003635254,0.000610854,0.007949124,0.001710059,0.003128775,0.007152606],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008729438,"about_ca_system_score_gemma":0.001638349,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007244185,"about_ca_topic_score_gemma":0.01027938,"domain_scores_codex":[0.9940152,0.00296964,0.0004259123,0.00075319,0.001595023,0.0002410549],"domain_scores_gemma":[0.9773309,0.01723344,0.0004739081,0.001404635,0.003345979,0.0002110581],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005874375,0.0006005458,0.003049169,0.0005393695,0.0004737809,0.0001465533,0.0002047348,0.06320536,0.005727659,0.006089303,0.02235456,0.8970215],"study_design_scores_gemma":[0.00008858077,0.0001329543,0.000735759,0.00008684809,0.0003267741,0.0001773122,0.00007593527,0.9551525,0.003909549,0.03268756,0.006566804,0.00005937217],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02238299,0.004859613,0.9585693,0.001283084,0.0007891072,0.0002622874,0.0008154029,0.006360177,0.00467802],"genre_scores_gemma":[0.3074251,0.003535208,0.6714925,0.001190279,0.001743997,0.0005941053,0.004714453,0.0008093884,0.008495064],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007708231,"threshold_uncertainty_score":0.04076552,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01324613556970664,"score_gpt":0.2371975164125819,"score_spread":0.2239513808428753,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}