{"id":"W4285138469","doi":"10.1162/tacl_a_00487","title":"Heterogeneous Supervised Topic Models","year":2022,"lang":"en","type":"article","venue":"Transactions of the Association for Computational Linguistics","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal; Mila - Quebec Artificial Intelligence Institute","funders":"Office of Naval Research; Simons Foundation; Alfred P. Sloan Foundation; National Science Foundation","keywords":"Computer science; Inference; Artificial intelligence; Outcome (game theory); Machine learning; Topic model; Bayes' theorem; Bayesian inference; Latent variable; Language model; Bayesian probability; Natural language processing; Probabilistic logic; Tone (literature); Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005525609,0.001037423,0.001583417,0.001808461,0.0006626355,0.001714584,0.002481788,0.001881503,0.003574236],"category_scores_gemma":[0.0151163,0.0006279444,0.001554629,0.001605756,0.001200673,0.00305963,0.001224058,0.002488877,0.001135734],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001232767,"about_ca_system_score_gemma":0.000775667,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004565104,"about_ca_topic_score_gemma":0.006379957,"domain_scores_codex":[0.9971926,0.001512287,0.0001257464,0.0007497307,0.0002580676,0.0001616425],"domain_scores_gemma":[0.9859291,0.01129106,0.0008556317,0.000882067,0.0008596858,0.0001824915],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004469627,0.0002376429,0.007444682,0.0003188367,0.0003539177,0.0002290834,0.0005129277,0.8152267,0.001742747,0.05414791,0.008340361,0.1109982],"study_design_scores_gemma":[0.00001033277,0.000009427908,0.0002610422,0.00001041064,0.0000106459,0.00001260869,0.00001041339,0.9825963,0.0001522207,0.01663852,0.0002828065,0.000005456902],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.09319498,0.001181091,0.8991059,0.001207904,0.0001569898,0.0001284487,0.001361429,0.001043334,0.002619868],"genre_scores_gemma":[0.9010714,0.0005177825,0.08842243,0.0003064465,0.0005397376,0.0002781544,0.003093706,0.0002259118,0.005544361],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005525609,"threshold_uncertainty_score":0.02922255,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04311022111236518,"score_gpt":0.3237508931693172,"score_spread":0.280640672056952,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}