{"id":"W4404856684","doi":"10.1177/14614456241298915","title":"Topic modelling is a means to an end: On topic modelling in corpus linguistics and discourse analysis","year":2024,"lang":"en","type":"article","venue":"Discourse Studies","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Corpus linguistics; Linguistics; Discourse analysis; Applied linguistics; Computer science; Quantitative linguistics; Sociology; Philosophy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.06727152,0.001911579,0.00276685,0.008461514,0.005506949,0.02084282,0.005185736,0.01008113,0.005414136],"category_scores_gemma":[0.1346229,0.001639193,0.002039607,0.01341292,0.02838922,0.03857501,0.01474015,0.01770976,0.002629492],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006969984,"about_ca_system_score_gemma":0.004197543,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006457916,"about_ca_topic_score_gemma":0.005460497,"domain_scores_codex":[0.9280967,0.06096201,0.001865607,0.003496217,0.004862715,0.000716803],"domain_scores_gemma":[0.7402308,0.2413635,0.003503849,0.007804046,0.005705676,0.001392241],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00006783244,0.00003867853,0.0008307078,0.0006825557,0.00009301068,0.0002020051,0.02015504,0.00417694,0.0002272086,0.9048171,0.01792139,0.05078749],"study_design_scores_gemma":[0.00001892589,0.00002364176,0.0003575637,0.001569384,0.00003210306,0.0001753368,0.00267329,0.01330131,0.0002953522,0.8850594,0.09638865,0.0001050214],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.004832109,0.08342887,0.7096956,0.1838378,0.00402203,0.0001859683,0.0002276266,0.0003450912,0.01342487],"genre_scores_gemma":[0.2375589,0.1017887,0.5644535,0.05417518,0.02622922,0.002291182,0.0008134965,0.002484902,0.01020486],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9327285,"threshold_uncertainty_score":0.3557701,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1362974815568709,"score_gpt":0.4772390681758816,"score_spread":0.3409415866190106,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}