{"id":"W4405113218","doi":"10.1016/j.procs.2024.11.126","title":"NLP and Topic Modeling with LDA, LSA, and NMF for Monitoring Psychosocial Well-being in Monthly Surveys","year":2024,"lang":"en","type":"article","venue":"Procedia Computer Science","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université du Québec à Rimouski","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Information retrieval; Topic model; Data science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01389727,0.001354534,0.001312564,0.005271356,0.001200728,0.001894032,0.00100086,0.001331405,0.000962235],"category_scores_gemma":[0.03251844,0.0005128332,0.002671869,0.003405314,0.0007182594,0.002043651,0.001532454,0.002246884,0.0005227652],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001001529,"about_ca_system_score_gemma":0.001567214,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008226607,"about_ca_topic_score_gemma":0.008845959,"domain_scores_codex":[0.9883173,0.008990856,0.0006579487,0.00123212,0.000577839,0.0002238219],"domain_scores_gemma":[0.960045,0.03621264,0.001342257,0.001108511,0.001081015,0.0002106479],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009807694,0.001173543,0.07366017,0.001150958,0.001828847,0.0005001451,0.003901717,0.2295715,0.006686063,0.0115165,0.009085811,0.659944],"study_design_scores_gemma":[0.00004543816,0.000102637,0.009152922,0.00005971225,0.00008475675,0.0000740245,0.0006171656,0.9733262,0.0009438326,0.01345266,0.00208438,0.00005627766],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1256557,0.001445118,0.8668575,0.001031421,0.0001649741,0.0006214959,0.002102057,0.001353406,0.0007682366],"genre_scores_gemma":[0.4498861,0.0005767142,0.541146,0.0001952335,0.0002863148,0.002111533,0.005021481,0.0001302033,0.0006463763],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01389727,"threshold_uncertainty_score":0.07349664,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02895404700687553,"score_gpt":0.3523990624763297,"score_spread":0.3234450154694542,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}