{"id":"W4417423602","doi":"10.48550/arxiv.2507.10827","title":"Supporting SENĆOTEN Language Documentation Efforts with Automatic Speech Recognition","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Pipeline (software); Documentation; Vocabulary; Word error rate; Language model; Constructed language; Set (abstract data type); Speech technology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001833414,0.001574366,0.000978599,0.00229418,0.001366617,0.002473606,0.001478858,0.001208008,0.009442844],"category_scores_gemma":[0.007167422,0.0005985858,0.0007456651,0.00157439,0.0007044701,0.002793394,0.003008329,0.002812019,0.01944747],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001102058,"about_ca_system_score_gemma":0.003876395,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.03035883,"about_ca_topic_score_gemma":0.04968528,"domain_scores_codex":[0.9977331,0.0005794424,0.0001728323,0.0006576382,0.0006756819,0.0001812243],"domain_scores_gemma":[0.9951997,0.0008962371,0.0002212174,0.001371663,0.002074652,0.0002364896],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003254934,0.0002180536,0.003893067,0.0005334605,0.00007472867,0.0008487421,0.001233036,0.008051001,0.04037294,0.002679813,0.0996538,0.8421159],"study_design_scores_gemma":[0.0002358852,0.0004149951,0.01330473,0.0003946332,0.0001281805,0.0014858,0.003282313,0.5078349,0.1648723,0.01215825,0.2955672,0.0003207286],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2046669,0.002493114,0.4297511,0.003071286,0.001336798,0.0007022186,0.02043948,0.282402,0.05513715],"genre_scores_gemma":[0.4916914,0.00128924,0.3837042,0.0008503791,0.0002447054,0.0004072323,0.08051552,0.005796309,0.03550105],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.03035883,"threshold_uncertainty_score":0.06036425,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03055018748211533,"score_gpt":0.3052722171440733,"score_spread":0.274722029661958,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}