{"id":"W4304806684","doi":"10.1007/s11634-022-00522-6","title":"On smoothing and scaling language model for sentiment based information retrieval","year":2022,"lang":"en","type":"article","venue":"Advances in Data Analysis and Classification","topic":"Sentiment Analysis and Opinion Mining","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Sentiment analysis; Latent Dirichlet allocation; Relevance (law); Social media; Smoothing; Probabilistic logic; Field (mathematics); Information retrieval; Language model; Artificial intelligence; Dirichlet distribution; Topic model; Data mining; World Wide Web; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002782253,0.0007808447,0.001431227,0.001347361,0.0008017309,0.001268848,0.00128988,0.0009715706,0.00236985],"category_scores_gemma":[0.00852502,0.0004100677,0.001358571,0.00180163,0.0006150722,0.003250121,0.0009398902,0.001733877,0.00139732],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000764398,"about_ca_system_score_gemma":0.0009994721,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006687724,"about_ca_topic_score_gemma":0.00496448,"domain_scores_codex":[0.9987638,0.0004388864,0.0001227122,0.0002308648,0.0003317134,0.0001120368],"domain_scores_gemma":[0.9968335,0.001815393,0.00015372,0.0003574342,0.0007620864,0.00007781758],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005661633,0.0003990508,0.002193838,0.0003664278,0.0002872972,0.0002429901,0.0004741639,0.2714321,0.02650361,0.09194465,0.01491754,0.5906722],"study_design_scores_gemma":[0.000008687254,0.00003512726,0.0002065445,0.000005642713,0.0000218821,0.00002200613,0.00001466932,0.9823437,0.001053925,0.01527693,0.000997135,0.00001383472],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01557853,0.0007476521,0.9815026,0.0003293231,0.0001509957,0.00005953692,0.0001356218,0.0007500749,0.0007457561],"genre_scores_gemma":[0.4817678,0.001965324,0.5005162,0.0006039741,0.0006954779,0.0003693293,0.00157143,0.0004736639,0.01203673],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006687724,"threshold_uncertainty_score":0.01471412,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03478667359186854,"score_gpt":0.3244483637770839,"score_spread":0.2896616901852153,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}