{"id":"W3113091126","doi":"10.1016/j.bdr.2020.100173","title":"Agreeing to Disagree: Choosing Among Eight Topic-Modeling Methods","year":2020,"lang":"en","type":"article","venue":"Big Data Research","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":31,"is_retracted":false,"has_abstract":false,"ca_institutions":"Simon Fraser University; University of British Columbia","funders":"Hong Kong Polytechnic University","keywords":"Latent Dirichlet allocation; Topic model; Computer science; Data science; Artificial intelligence; Coherence (philosophical gambling strategy); Field (mathematics); Context (archaeology); Machine learning; Management science; Natural language processing","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1976716,0.002535491,0.003023694,0.006498765,0.004347725,0.007980894,0.005831443,0.005154657,0.006813993],"category_scores_gemma":[0.3329974,0.001635352,0.004478248,0.004767443,0.003544451,0.01111379,0.007425272,0.007263245,0.002170931],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001750238,"about_ca_system_score_gemma":0.003889166,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00136597,"about_ca_topic_score_gemma":0.004318592,"domain_scores_codex":[0.8347772,0.1381306,0.007954801,0.00594509,0.01091581,0.002276591],"domain_scores_gemma":[0.3774642,0.5812154,0.008933019,0.01187027,0.01562411,0.004893096],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.01206268,0.002374437,0.08158297,0.004764759,0.004213657,0.0002374645,0.02465527,0.008747776,0.00364413,0.04218261,0.02420812,0.7913261],"study_design_scores_gemma":[0.006882404,0.002645764,0.04922847,0.004978783,0.006237485,0.0008355011,0.0321158,0.4033126,0.01330726,0.4376624,0.04152385,0.00126977],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2081795,0.004004317,0.7457227,0.01238216,0.001131329,0.006634318,0.00131667,0.001910983,0.01871799],"genre_scores_gemma":[0.440204,0.00103177,0.5450583,0.001797486,0.0004251892,0.007083504,0.001521211,0.0005649134,0.002313689],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.1976716,"threshold_uncertainty_score":0.9894137,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.8645336607084751,"score_gpt":0.6295130273678069,"score_spread":0.2350206333406683,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}