{"id":"W4414464224","doi":"10.1109/acdsa65407.2025.11166276","title":"Topic Modeling Enhancement using Summaries Generated by LLM Models","year":2025,"lang":"en","type":"article","venue":"","topic":"Technology and Data Analysis","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Sheridan College","funders":"Research and Development","keywords":"Topic model; Security token; Language model; Document processing; Segmentation; Semantics (computer science); Named entity","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002472064,0.00141792,0.001139349,0.002623646,0.0004224736,0.001766182,0.0008198399,0.0009055302,0.001433297],"category_scores_gemma":[0.009693895,0.0003599596,0.001149728,0.001530213,0.0002319985,0.002902446,0.0009232013,0.00119652,0.001235553],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006184481,"about_ca_system_score_gemma":0.0007688803,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003151793,"about_ca_topic_score_gemma":0.004208104,"domain_scores_codex":[0.9989359,0.0004264933,0.00009501234,0.0002652825,0.0002134109,0.00006389616],"domain_scores_gemma":[0.995847,0.00250188,0.0003668129,0.0004356837,0.0007379415,0.0001106367],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001368467,0.000345043,0.0122866,0.000711956,0.0003886636,0.000393167,0.001526401,0.1904521,0.03729087,0.00760079,0.008728335,0.7389076],"study_design_scores_gemma":[0.0000355037,0.0002030092,0.0017739,0.00003041197,0.0001239142,0.0001136803,0.0001994368,0.9807768,0.007795957,0.00419465,0.004718219,0.00003433367],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.06065261,0.00146444,0.9310334,0.0003596957,0.000110566,0.0001833769,0.0007375749,0.004175047,0.001283287],"genre_scores_gemma":[0.5359461,0.001190897,0.4548826,0.0001504045,0.0003123158,0.0003411152,0.003931403,0.0004171244,0.002827991],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.003151793,"threshold_uncertainty_score":0.01307362,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02606792469033519,"score_gpt":0.2623315734973197,"score_spread":0.2362636488069845,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}