{"id":"W3197492799","doi":"10.1007/978-3-030-86331-9_23","title":"Sparse Document Analysis Using Beta-Liouville Naive Bayes with Vocabulary Knowledge","year":2021,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; BETA (programming language); Bayes' theorem; Artificial intelligence; Naive Bayes classifier; Vocabulary; Natural language processing; Machine learning; Bayesian probability; Programming language; Philosophy; Linguistics; Support vector machine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0008480478,0.000709134,0.00100649,0.001668166,0.0003610621,0.0009011079,0.003144327,0.0003033775,0.000064967],"category_scores_gemma":[0.00003100998,0.0006068122,0.0003040444,0.002323983,0.000501885,0.0007363832,0.00224623,0.0008096354,0.0000200105],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005582542,"about_ca_system_score_gemma":0.001177598,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00006938185,"about_ca_topic_score_gemma":0.0003846679,"domain_scores_codex":[0.9946939,0.00007839777,0.0006572009,0.002509795,0.001233035,0.0008277026],"domain_scores_gemma":[0.9962031,0.0003235288,0.0003599849,0.00238292,0.0004732341,0.0002572572],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000006559907,0.00005684613,0.0005416673,0.00006614502,0.0003680199,0.0005619089,0.002528358,0.718118,0.00007788615,0.02340284,0.000007386928,0.2542644],"study_design_scores_gemma":[0.0002472012,0.0001018711,0.00008854218,0.0004458659,0.0001949304,0.0001015599,0.000001009604,0.9842569,0.001004299,0.01209968,0.000568026,0.000890099],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0008781705,0.00199475,0.9931676,0.0002804578,0.001021931,0.0003100785,0.000003537881,0.000137269,0.00220618],"genre_scores_gemma":[0.1936532,0.00004482763,0.8046982,0.0005731151,0.0004449676,0.000007554325,0.000008949985,0.00004143226,0.0005277743],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.2661389,"threshold_uncertainty_score":0.9996383,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02460182828816797,"score_gpt":0.260545632630907,"score_spread":0.235943804342739,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}