{"id":"W3161411634","doi":"10.1109/icassp39728.2021.9413680","title":"A Comparison of Discrete Latent Variable Models for Speech Representation Learning","year":2021,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Latent variable; Speech recognition; Artificial intelligence; Representation (politics); Encoding (memory); Word error rate; Variable (mathematics); Latent variable model; Word (group theory); Pattern recognition (psychology); Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00574706,0.0009752236,0.0008989049,0.001449067,0.000302377,0.001958317,0.001781351,0.001327126,0.002315915],"category_scores_gemma":[0.01394562,0.000470852,0.001234359,0.001453699,0.0004977963,0.003419612,0.00148356,0.00245857,0.001173335],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001554764,"about_ca_system_score_gemma":0.00113255,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00791219,"about_ca_topic_score_gemma":0.007751417,"domain_scores_codex":[0.9973096,0.001398341,0.0001741172,0.0004118225,0.000580154,0.0001258927],"domain_scores_gemma":[0.9928995,0.005243198,0.0001878533,0.0007562057,0.0007453751,0.0001678762],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001527584,0.0003837736,0.005348492,0.000542848,0.0006185207,0.00006123141,0.0002514834,0.3676383,0.003169303,0.02760837,0.006079641,0.5867704],"study_design_scores_gemma":[0.00003088029,0.0001120459,0.0007433904,0.00003384916,0.0000250286,0.0000227896,0.00003292087,0.9905255,0.000669366,0.006789523,0.0009949006,0.0000198559],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.05842924,0.007954122,0.92462,0.001628399,0.0003281975,0.0001137839,0.001139792,0.002718196,0.003068234],"genre_scores_gemma":[0.6496754,0.006628378,0.3321992,0.0004654083,0.0002413042,0.0003845799,0.005468884,0.0005026176,0.004434344],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.00791219,"threshold_uncertainty_score":0.03039372,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09678490460931576,"score_gpt":0.3466243956047883,"score_spread":0.2498394909954725,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}