{"id":"W4382464561","doi":"10.1609/aaai.v37i7.25998","title":"Deep Visual Forced Alignment: Learning to Align Transcription with Talking Face Video","year":2023,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Speech and Audio Processing","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Ministry of Science and ICT, South Korea; National Research Foundation of Korea; National Research Foundation","keywords":"Computer science; Speech recognition; Sentence; Two-alternative forced choice; Artificial intelligence; Transcription (linguistics); Computer vision; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000580367,0.0002703028,0.0002676008,0.0002561783,0.0004172009,0.0005119323,0.001455384,0.00008011992,0.00002313106],"category_scores_gemma":[0.0002349578,0.0002020266,0.0000944826,0.001875105,0.0001219441,0.0006863755,0.0002622433,0.0002899206,0.0001883972],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005922165,"about_ca_system_score_gemma":0.00007469046,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002004853,"about_ca_topic_score_gemma":0.00001151042,"domain_scores_codex":[0.997576,0.00002063443,0.0004497912,0.0006440245,0.0007452912,0.0005642411],"domain_scores_gemma":[0.9989013,0.00006999983,0.0002851819,0.0001938267,0.0004010739,0.0001486133],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0001253165,0.00006281155,0.0002817163,0.00007092925,0.00002569407,0.000002583183,0.008546274,0.004596882,0.6999058,0.06455418,0.00006792675,0.2217599],"study_design_scores_gemma":[0.00004005134,0.0003975999,0.0001300162,0.0003606133,0.00001089375,0.000005771861,0.002262158,0.1036889,0.8791953,0.01353,0.0001167463,0.000261945],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5555201,0.0000174502,0.4341399,0.004771556,0.0003205043,0.0006144948,0.000001018655,0.000500188,0.004114864],"genre_scores_gemma":[0.9942412,0.00001679441,0.00491717,0.0002996635,0.0000612074,0.00004633326,7.324174e-7,0.00002194614,0.0003949059],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4387212,"threshold_uncertainty_score":0.8238406,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0453055098017504,"score_gpt":0.2904564255360821,"score_spread":0.2451509157343317,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}