{"id":"W4320927953","doi":"10.15514//ispras-2022-34(5)-10","title":"Data Mining Methods to Compare Englishes","year":2022,"lang":"en","type":"article","venue":"Proceedings of the Institute for System Programming of RAS","topic":"Linguistics, Language Diversity, and Identity","field":"Arts and Humanities","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"World Englishes; Noun; Covert; Linguistics; Pidgin; Geography","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01100894,0.001317189,0.001374066,0.01386804,0.0008487757,0.004337933,0.0018234,0.0008717543,0.004552797],"category_scores_gemma":[0.04215006,0.0004310777,0.002003237,0.01335686,0.0008194607,0.003146969,0.001917325,0.001558775,0.001871291],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0011427,"about_ca_system_score_gemma":0.001305441,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002256068,"about_ca_topic_score_gemma":0.002176558,"domain_scores_codex":[0.9889991,0.003409493,0.002865708,0.00215936,0.002308875,0.0002573697],"domain_scores_gemma":[0.9666724,0.02332845,0.002549494,0.002755284,0.00436438,0.0003300145],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008527089,0.001007078,0.1231918,0.003835672,0.002246479,0.0006086113,0.004413818,0.01995887,0.005433425,0.01980671,0.02029885,0.7983459],"study_design_scores_gemma":[0.0005269152,0.002174135,0.1895022,0.002945234,0.0019263,0.002166917,0.03430873,0.2770513,0.02915468,0.1379853,0.3216193,0.0006389425],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1490968,0.003526507,0.7641331,0.002941273,0.0008498474,0.005118559,0.05247197,0.004628486,0.01723353],"genre_scores_gemma":[0.2541554,0.001240911,0.6988974,0.0004817707,0.0002310902,0.007015277,0.03515176,0.0003627336,0.002463602],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01386804,"threshold_uncertainty_score":0.05822158,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09825657603574751,"score_gpt":0.3175586179290332,"score_spread":0.2193020418932857,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}