{"id":"W3028968558","doi":"","title":"Collecting Tweets to Investigate Regional Variation in Canadian English","year":2020,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Linguistic Variation and Morphology","field":"Social Sciences","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Timeline; Computer science; Pipeline (software); Variation (astronomy); Word (group theory); Focus (optics); Crawling; Space (punctuation); Information retrieval; Natural language processing; Similarity (geometry); Relevance (law); Vector space model; World Wide Web; Artificial intelligence; Geography; Linguistics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.007711145,0.0002032788,0.0002933204,0.0003973911,0.0006194575,0.0003677614,0.0008897317,0.0003842499,0.0002474273],"category_scores_gemma":[0.03180284,0.0002644924,0.00008699538,0.0008755954,0.0001606563,0.00007307401,0.0004173241,0.0006222734,0.00005699601],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007997058,"about_ca_system_score_gemma":0.004258245,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.7408905,"about_ca_topic_score_gemma":0.9081969,"domain_scores_codex":[0.9920125,0.005890621,0.0004739645,0.0006857751,0.0004163564,0.0005208067],"domain_scores_gemma":[0.9947428,0.001178215,0.0003078337,0.0006429536,0.00234354,0.0007846892],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000008151475,0.0001002861,0.00853805,0.00004012425,0.00003367044,0.00001607851,0.4017055,0.000181797,0.0002488361,0.5806071,0.006916173,0.001604289],"study_design_scores_gemma":[0.001093161,9.580012e-7,0.06494197,0.001580625,0.00006219548,0.000002744594,0.002849398,0.007897209,0.0008674448,0.0549394,0.864566,0.001198938],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.1294132,0.0001814203,0.01813857,0.2726401,0.002549929,0.001817038,0.0001225018,0.0004948671,0.5746424],"genre_scores_gemma":[0.9730238,0.0000764446,0.01989298,0.001977417,0.0002127743,0.00008598279,0.0002141811,0.00002993352,0.004486445],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8576498,"threshold_uncertainty_score":0.9999807,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03841661960124737,"score_gpt":0.2771601306337848,"score_spread":0.2387435110325374,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}