{"id":"W7132590463","doi":"","title":"Supporting SENĆOŦEN language documentation efforts with Automatic Speech Recognition","year":2025,"lang":"en","type":"article","venue":"NPARC","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Pipeline (software); Documentation; Vocabulary; Word error rate; Speech technology; Language model; Set (abstract data type); Variation (astronomy)","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0002933429,0.00009947863,0.0001189971,0.0001646078,0.00009644271,0.000189924,0.0001998385,0.00003931245,0.0009168769],"category_scores_gemma":[0.00006806835,0.00008510322,0.00003788237,0.0004001269,0.00001939694,0.0004996464,0.00004836417,0.0000727282,0.0002219734],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004514548,"about_ca_system_score_gemma":0.00005750517,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001921653,"about_ca_topic_score_gemma":0.00001561462,"domain_scores_codex":[0.9990608,0.00005642335,0.0002086665,0.0002485607,0.0002149172,0.0002106134],"domain_scores_gemma":[0.9994456,0.0001057533,0.00009592412,0.0002414791,0.00006332224,0.00004790689],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.000003350819,0.00002827502,0.0001941395,0.00003014298,0.00001768237,0.00004659794,0.0003326364,3.277475e-7,0.006499237,0.0005574615,0.001203519,0.9910866],"study_design_scores_gemma":[0.001832068,0.0001909071,0.004561748,0.000649671,0.0001053485,0.0002450977,0.001448778,0.08871977,0.8217565,0.07695023,0.002783568,0.0007563543],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.665457,0.00001193863,0.1637028,0.001390963,0.0002380062,0.0003737432,0.000003976586,0.0005922283,0.1682294],"genre_scores_gemma":[0.6006421,0.000003251238,0.3968523,0.0008049246,0.00002863993,0.00004090103,0.0000199993,0.000007446463,0.001600429],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9903303,"threshold_uncertainty_score":0.9999964,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01088251064359417,"score_gpt":0.2816661806138079,"score_spread":0.2707836699702137,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}