{"id":"W2130987418","doi":"10.1109/wi.2005.79","title":"Integrating Compound Terms in Bayesian Text Classification","year":2005,"lang":"en","type":"article","venue":"","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Smoothing; Word (group theory); Artificial intelligence; Natural language processing; Bayesian probability; Term (time); Representation (politics); Naive Bayes classifier; Compound; Component (thermodynamics); Machine learning; Mathematics; Support vector machine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007293961,0.00103976,0.002070152,0.005302378,0.0009289716,0.001958612,0.001438915,0.001900306,0.001598934],"category_scores_gemma":[0.02178446,0.0006998262,0.001357602,0.005728565,0.001241614,0.008228926,0.002164036,0.002085293,0.001357182],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008687009,"about_ca_system_score_gemma":0.001419713,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00287938,"about_ca_topic_score_gemma":0.006489524,"domain_scores_codex":[0.9956294,0.001580949,0.0003501084,0.0007215118,0.001547434,0.0001704927],"domain_scores_gemma":[0.9851537,0.009782447,0.001101837,0.001599334,0.002130877,0.0002319324],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004250541,0.0002125795,0.006576659,0.000363372,0.0003313088,0.0001732344,0.0004531997,0.08282346,0.01626114,0.02764584,0.002368418,0.8623657],"study_design_scores_gemma":[0.00003894274,0.0002316045,0.002444368,0.00006074159,0.0002251351,0.0002099777,0.00006866219,0.8870021,0.008967646,0.09545397,0.005184281,0.0001125594],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02250407,0.0009862477,0.9749288,0.0001617689,0.00004601167,0.00007684239,0.0000666362,0.0005954515,0.0006341831],"genre_scores_gemma":[0.2336213,0.001056565,0.7612415,0.0001538929,0.0002588639,0.0002891385,0.0005786921,0.0001998291,0.002600223],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007293961,"threshold_uncertainty_score":0.03857458,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02413243508252776,"score_gpt":0.2743366962494589,"score_spread":0.2502042611669311,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}