{"id":"W2965484644","doi":"10.48550/arxiv.1907.11843","title":"Analyzing Linguistic Complexity and Scientific Impact","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Citation; Proxy (statistics); Scientific literature; Scientific writing; Context (archaeology); Scientific communication; Linguistics; Linguistic context; Linguistic sequence complexity; Linguistic analysis; Psychology; Computer science; Library science; History; Biology; Philosophy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.005903811,0.0005136986,0.0007452385,0.01831946,0.0009724893,0.00457136,0.0004277666,0.0006411432,0.002756549],"category_scores_gemma":[0.07213692,0.0002170399,0.00084567,0.0162916,0.001053853,0.002444852,0.002675001,0.0009232229,0.0004442726],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001015586,"about_ca_system_score_gemma":0.0008941453,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002548012,"about_ca_topic_score_gemma":0.001715514,"domain_scores_codex":[0.9954026,0.001716199,0.0005486996,0.0004218706,0.00161972,0.0002908885],"domain_scores_gemma":[0.854618,0.1170614,0.01800898,0.003065074,0.005256388,0.001990153],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0004106016,0.0002208549,0.9342604,0.0004046832,0.0007214252,0.0002603683,0.003926147,0.006154673,0.002048994,0.003125471,0.0009538164,0.04751269],"study_design_scores_gemma":[0.0000251567,0.0001121627,0.9656019,0.00008376878,0.0002394217,0.0001934653,0.002487765,0.01980291,0.0009029754,0.008056365,0.002431593,0.00006258185],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.991846,0.0006993027,0.002547305,0.0002949965,0.00001876949,0.00004163752,0.0008456401,0.0000375265,0.003668829],"genre_scores_gemma":[0.9971282,0.0001806709,0.001339588,0.00001610946,0.00007289949,0.00004495432,0.0009325396,0.00001349574,0.0002715572],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9940962,"threshold_uncertainty_score":0.0312227,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1372479509848239,"score_gpt":0.2273120980192345,"score_spread":0.0900641470344106,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}