{"id":"W4321445671","doi":"10.1515/9783111017433-005","title":"Dialect Corpora from YouTube","year":2023,"lang":"en","type":"book-chapter","venue":"","topic":"Linguistic Variation and Morphology","field":"Social Sciences","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Scripting language; Upload; Variation (astronomy); Task (project management); Natural language processing; Channel (broadcasting); Support vector machine; World Wide Web; Artificial intelligence; Identification (biology); Limiting; Speech recognition; Engineering; Telecommunications","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.000248177,0.0001205053,0.0002098784,0.00006913929,0.0001788198,0.00004154818,0.0002000528,0.0004002793,0.02151275],"category_scores_gemma":[0.0005125987,0.0001151447,0.00009345748,0.00002471404,0.0001556017,0.00001605229,0.00004941208,0.0001682711,0.007236081],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006378632,"about_ca_system_score_gemma":0.0003091603,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009991918,"about_ca_topic_score_gemma":0.009841282,"domain_scores_codex":[0.9990841,0.00003078732,0.0001827141,0.0002558006,0.0002767027,0.0001698569],"domain_scores_gemma":[0.9992099,0.0002871506,0.0001194125,0.0001718467,0.0001083111,0.0001033407],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000002019545,0.000002006546,0.00001895202,7.532901e-7,0.00002700473,0.00003084788,0.001336085,8.259387e-8,0.000001003054,0.9583297,0.03920826,0.001043247],"study_design_scores_gemma":[0.00006701473,0.000009066655,0.0000389855,0.00000823076,0.00002666466,1.019383e-7,0.00007927294,0.000001426982,6.480776e-7,0.2876939,0.7119542,0.0001203963],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.000006491402,0.00001786338,0.0001785869,0.00107544,0.00389488,0.0001361035,0.00009114145,0.0003141486,0.9942853],"genre_scores_gemma":[0.0023297,0.0001681036,0.000231333,0.0007910474,0.002621638,0.000003448186,0.0001576281,0.00003091607,0.9936662],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.672746,"threshold_uncertainty_score":0.9966006,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08170471255090607,"score_gpt":0.3078860411163288,"score_spread":0.2261813285654227,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}