{"id":"W4321445671","doi":"10.1515/9783111017433-005","title":"Dialect Corpora from YouTube","year":2023,"lang":"en","type":"book-chapter","venue":"","topic":"Linguistic Variation and Morphology","field":"Social Sciences","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Scripting language; Upload; Variation (astronomy); Task (project management); Natural language processing; Channel (broadcasting); Support vector machine; World Wide Web; Artificial intelligence; Identification (biology); Limiting; Speech recognition; Engineering; Telecommunications","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004596857,0.0006530574,0.0003973911,0.006177004,0.001263011,0.001084342,0.000892872,0.0004172794,0.04931044],"category_scores_gemma":[0.002059872,0.00020127,0.0002902073,0.006677377,0.0003249745,0.001028008,0.001308607,0.000557466,0.0248013],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001367781,"about_ca_system_score_gemma":0.001374925,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02848337,"about_ca_topic_score_gemma":0.06415328,"domain_scores_codex":[0.999474,0.0001153621,0.00007765455,0.0001215728,0.0001608869,0.00005060843],"domain_scores_gemma":[0.998795,0.0002842698,0.0000477872,0.000211666,0.0005817205,0.00007952174],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0002465235,0.00005176543,0.003178318,0.001761169,0.00003096756,0.001091198,0.003131494,0.0004650145,0.007069415,0.008937426,0.8197746,0.1542621],"study_design_scores_gemma":[0.0000219641,0.00001817966,0.01243282,0.0001199963,0.00001216288,0.0006787776,0.001280529,0.0008225705,0.001802673,0.001497996,0.9812791,0.00003314791],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.028384,0.002271825,0.01058913,0.0006422334,0.0006401216,0.001246487,0.830075,0.003172479,0.1229787],"genre_scores_gemma":[0.02926884,0.001248378,0.01958114,0.0002543393,0.0001040518,0.002580835,0.9023517,0.0009903227,0.04362028],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.04931044,"threshold_uncertainty_score":0.1649598,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08170471255090607,"score_gpt":0.3078860411163288,"score_spread":0.2261813285654227,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}