{"id":"W2559617320","doi":"","title":"Models, forests and trees of York English: Was/were variation as a case study for statistical practice","year":2012,"lang":"en","type":"article","venue":"The Mind Research Repository","topic":"Linguistic Variation and Morphology","field":"Social Sciences","cited_by":72,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta; University of Toronto","funders":"","keywords":"Multicollinearity; Variation (astronomy); Plural; Inference; Random forest; Computer science; Verb; Econometrics; Statistics; Variable (mathematics); Artificial intelligence; Natural language processing; Linguistics; Machine learning; Mathematics; Linear regression","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01283101,0.000455837,0.0008295007,0.002259905,0.001700348,0.004018558,0.0015528,0.0009115392,0.004561658],"category_scores_gemma":[0.03541212,0.0004861111,0.0009381585,0.003619203,0.005728391,0.004287416,0.001729113,0.001465943,0.0003355793],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003534181,"about_ca_system_score_gemma":0.001516087,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02266771,"about_ca_topic_score_gemma":0.03839583,"domain_scores_codex":[0.9916339,0.006649618,0.0002966412,0.0007751536,0.0004450577,0.0001996851],"domain_scores_gemma":[0.9585326,0.03769232,0.001078858,0.001601142,0.0007746037,0.0003203606],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","study_design_scores_codex":[0.00008022285,0.00002804183,0.009747171,0.0001903962,0.00005666507,0.0003531013,0.009034278,0.02364814,0.0002812197,0.9195387,0.003518804,0.03352324],"study_design_scores_gemma":[0.00001671698,0.00002458222,0.004446548,0.0001245732,0.00002377043,0.0002502441,0.002317636,0.1160053,0.0002432588,0.8650813,0.01141768,0.00004839346],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1770325,0.00305153,0.7977618,0.00592668,0.0001007599,0.0001137624,0.0009309544,0.0006276126,0.01445438],"genre_scores_gemma":[0.8305165,0.0006038111,0.1648829,0.0001621386,0.00004487479,0.0002214738,0.0005070416,0.0002781774,0.002783158],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02266771,"threshold_uncertainty_score":0.06785762,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1341652799671293,"score_gpt":0.4493904632306894,"score_spread":0.3152251832635601,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}