{"id":"W7117484981","doi":"10.1016/j.eswa.2025.130958","title":"Topic modeling and alignment with large language models for multi-labeled text corpora","year":2025,"lang":"en","type":"article","venue":"Expert Systems with Applications","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Fundamental Research Funds for the Central Universities; China Postdoctoral Science Foundation; National Natural Science Foundation of China","keywords":"Interpretability; Topic model; Language model; Probabilistic logic; Coherence (philosophical gambling strategy); Semantics (computer science); Latent Dirichlet allocation","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008410086,0.001981526,0.002772342,0.004875924,0.002297175,0.004046515,0.003352054,0.002845098,0.003839734],"category_scores_gemma":[0.02868743,0.001855976,0.003372906,0.006107728,0.0009185768,0.006576587,0.002911424,0.005273844,0.005154175],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001386047,"about_ca_system_score_gemma":0.00288589,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00697278,"about_ca_topic_score_gemma":0.01262991,"domain_scores_codex":[0.9910722,0.00472631,0.0006710177,0.002321383,0.0008438457,0.0003652488],"domain_scores_gemma":[0.9786129,0.01610253,0.0008322397,0.002404936,0.001684289,0.0003630923],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001152894,0.0006599505,0.004152684,0.0009898784,0.0009967373,0.0005628041,0.001311676,0.1466841,0.0131948,0.0291733,0.03202434,0.7690967],"study_design_scores_gemma":[0.00006713939,0.00007785491,0.001073225,0.00004987601,0.000167827,0.0001684613,0.0001933864,0.9469582,0.004151816,0.04055899,0.006470679,0.00006239522],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.006957465,0.0009961647,0.9861732,0.0003918943,0.000182242,0.0001338609,0.0009011235,0.003718505,0.0005455544],"genre_scores_gemma":[0.1728056,0.001466819,0.8031104,0.0003202897,0.0009626368,0.001142686,0.014348,0.001822188,0.004021417],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008410086,"threshold_uncertainty_score":0.04447728,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03005171954682459,"score_gpt":0.2874050887933683,"score_spread":0.2573533692465437,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}