{"id":"W4233989131","doi":"10.1177/1536867x1701700406","title":"Text Mining with n-gram Variables","year":2017,"lang":"en","type":"article","venue":"The Stata Journal Promoting communications on statistics and Stata","topic":"Data Mining Algorithms and Applications","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"n-gram; Gram; Computer science; Statistics; Natural language processing; Mathematics; Geology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003568365,0.001182875,0.001356672,0.004330618,0.001688208,0.002765797,0.001012865,0.001227006,0.008292582],"category_scores_gemma":[0.02499275,0.0005469532,0.00184548,0.005068407,0.0006500113,0.002709481,0.002228484,0.001992034,0.0104773],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004019045,"about_ca_system_score_gemma":0.002270676,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001847102,"about_ca_topic_score_gemma":0.00409901,"domain_scores_codex":[0.9960586,0.001591998,0.0006161751,0.0008442847,0.0006726565,0.0002163244],"domain_scores_gemma":[0.9898823,0.007069179,0.0005249876,0.001362136,0.0009025946,0.0002587904],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001161329,0.0003747818,0.01587087,0.0009161471,0.0005441431,0.0006783761,0.0005346895,0.01091426,0.01148738,0.01803132,0.05751887,0.8819678],"study_design_scores_gemma":[0.0002560289,0.0007280675,0.01339276,0.0005234457,0.0005192304,0.00151579,0.0009200666,0.6752869,0.03186191,0.1834192,0.09138861,0.0001881214],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02712128,0.001650515,0.9333863,0.001400907,0.0006516511,0.0005064405,0.01350164,0.01746655,0.004314691],"genre_scores_gemma":[0.2547529,0.0009083627,0.7086029,0.0005266936,0.0007381697,0.001033085,0.02340378,0.0007987572,0.009235328],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.008292582,"threshold_uncertainty_score":0.02774149,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0460079612838921,"score_gpt":0.321158498507218,"score_spread":0.2751505372233259,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}