{"id":"W7117484981","doi":"10.1016/j.eswa.2025.130958","title":"Topic modeling and alignment with large language models for multi-labeled text corpora","year":2025,"lang":"en","type":"article","venue":"Expert Systems with Applications","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Fundamental Research Funds for the Central Universities; China Postdoctoral Science Foundation; National Natural Science Foundation of China","keywords":"Interpretability; Topic model; Language model; Probabilistic logic; Coherence (philosophical gambling strategy); Semantics (computer science); Latent Dirichlet allocation","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001478597,0.0001471847,0.0001943307,0.00007974396,0.0002322096,0.0001279645,0.0003493374,0.00005174055,4.910145e-7],"category_scores_gemma":[0.000002869983,0.0001113456,0.00001952843,0.0001988439,0.00001709186,0.0002014472,0.00008846861,0.00005839454,0.00000188557],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005631739,"about_ca_system_score_gemma":0.00007704314,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001896131,"about_ca_topic_score_gemma":0.00004675015,"domain_scores_codex":[0.9988777,0.00002086023,0.0002280468,0.0004830991,0.0001490905,0.0002411923],"domain_scores_gemma":[0.9990209,0.00003987466,0.00007385534,0.0006893884,0.0001038627,0.00007206036],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001772964,0.0001907112,0.0002017263,0.0001763005,0.00009437962,0.000002430827,0.005171144,0.0692554,0.001025027,0.9171393,0.0002199892,0.006505874],"study_design_scores_gemma":[0.0009484985,0.000024631,0.000003151382,0.0000854788,0.000008030635,0.000006771848,0.0008698771,0.9953213,0.0001272576,0.0004057598,0.002055295,0.0001439035],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.003197392,0.001868362,0.9922212,0.000413723,0.00004786066,0.001695432,0.000007842596,0.0001673694,0.0003807943],"genre_scores_gemma":[0.7345209,0.00001237534,0.2607259,0.0002055347,0.00004510205,0.00350884,0.000007202481,0.00001144626,0.0009627081],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9260659,"threshold_uncertainty_score":0.4540542,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03005171954682459,"score_gpt":0.2874050887933683,"score_spread":0.2573533692465437,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}