{"id":"W2791007156","doi":"10.1111/cogs.12583","title":"A Large‐Scale Analysis of Variance in Written Language","year":2018,"lang":"en","type":"article","venue":"Cognitive Science","topic":"Topic Modeling","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Manitoba","funders":"","keywords":"Computer science; Natural language processing; Semantics (computer science); Artificial intelligence; Natural language; Distributional semantics; Universal Networking Language; Language identification; Language model; Linguistics; Word (group theory); Comprehension approach; Semantic similarity; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00401518,0.0003674886,0.0004504704,0.002115895,0.0005433089,0.001491787,0.0004231948,0.0004119505,0.002161428],"category_scores_gemma":[0.03086003,0.0001714886,0.0008275526,0.002596787,0.001031425,0.001324236,0.001059045,0.001165423,0.0005124838],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004213567,"about_ca_system_score_gemma":0.0003540826,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00173968,"about_ca_topic_score_gemma":0.001818947,"domain_scores_codex":[0.9966136,0.001704935,0.0001211248,0.0008116187,0.0006752604,0.00007337921],"domain_scores_gemma":[0.9791123,0.01603249,0.001040664,0.002822065,0.0008042447,0.0001883535],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0008599262,0.000650473,0.2954228,0.0006295508,0.002700143,0.0008585499,0.009242698,0.0134933,0.03277564,0.04891321,0.02185058,0.5726031],"study_design_scores_gemma":[0.00005799027,0.0005130658,0.8293188,0.0001197334,0.0003514815,0.0009171937,0.001910078,0.0721902,0.005425809,0.06584558,0.02318948,0.0001605638],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7132459,0.002002977,0.2681004,0.001425299,0.0003422122,0.0002223979,0.00259162,0.001123006,0.01094618],"genre_scores_gemma":[0.9689189,0.0002989972,0.02759529,0.0001392333,0.0001592807,0.0001640317,0.001339763,0.0001384187,0.001246088],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.00401518,"threshold_uncertainty_score":0.02123451,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01775587779283431,"score_gpt":0.3001291638628089,"score_spread":0.2823732860699746,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}