{"id":"W4382464561","doi":"10.1609/aaai.v37i7.25998","title":"Deep Visual Forced Alignment: Learning to Align Transcription with Talking Face Video","year":2023,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Speech and Audio Processing","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Ministry of Science and ICT, South Korea; National Research Foundation of Korea; National Research Foundation","keywords":"Computer science; Speech recognition; Sentence; Two-alternative forced choice; Artificial intelligence; Transcription (linguistics); Computer vision; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006978136,0.001421306,0.0007924897,0.0006258771,0.000484367,0.0008461722,0.001295003,0.001223322,0.004950848],"category_scores_gemma":[0.003253862,0.0004207854,0.000758314,0.0006403255,0.0006003123,0.001440627,0.001424812,0.001870471,0.003048087],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004774684,"about_ca_system_score_gemma":0.001126587,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00464017,"about_ca_topic_score_gemma":0.006605739,"domain_scores_codex":[0.9990293,0.0001407384,0.00004251367,0.000455274,0.0001973691,0.0001348293],"domain_scores_gemma":[0.9993549,0.0001744412,0.00008567963,0.0001479803,0.0001731796,0.00006382812],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003566732,0.000138804,0.001304794,0.0001331216,0.00007502625,0.0002210191,0.0001748156,0.05266871,0.05778498,0.003173748,0.01260182,0.8713663],"study_design_scores_gemma":[0.00004790847,0.0002637644,0.00129754,0.00004333947,0.00004543671,0.0002819201,0.0001384458,0.9374203,0.0450989,0.007342768,0.007980029,0.00003956399],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02393005,0.0006188167,0.965784,0.0002009406,0.0003223715,0.0001029425,0.0003565018,0.005559552,0.003124748],"genre_scores_gemma":[0.3858789,0.0005413871,0.5934506,0.0008169619,0.0003049072,0.0002852104,0.003707352,0.0009728671,0.01404185],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004950848,"threshold_uncertainty_score":0.01656222,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0453055098017504,"score_gpt":0.2904564255360821,"score_spread":0.2451509157343317,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}