{"id":"W4392903535","doi":"10.1109/icassp48485.2024.10446349","title":"SoundLoCD: An Efficient Conditional Discrete Contrastive Latent Diffusion Model for Text-to-Sound Generation","year":2024,"lang":"en","type":"article","venue":"","topic":"Music and Audio Processing","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Google (Canada)","funders":"","keywords":"Computer science; Fidelity; Connection (principal bundle); Artificial intelligence; Diffusion; Component (thermodynamics); Speech recognition; Theoretical computer science; Algorithm; Mathematics; Physics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001882811,0.0001246403,0.0001032826,0.00008471158,0.0002656992,0.0006398028,0.0002146415,0.00004168821,0.00002573057],"category_scores_gemma":[0.00001409442,0.00009532343,0.0000552401,0.0001501538,0.00002362862,0.0003841762,0.00008552894,0.00005990565,0.0000335984],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006294964,"about_ca_system_score_gemma":0.0000999549,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000004701898,"about_ca_topic_score_gemma":0.00001318304,"domain_scores_codex":[0.998871,0.00001596882,0.0001760027,0.0004682148,0.0002465676,0.0002222096],"domain_scores_gemma":[0.9995468,0.00005543933,0.00002684016,0.0001547093,0.00009460269,0.0001216617],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0000112127,0.00008160916,0.00001154884,0.00003408662,0.00001991463,0.000003564626,0.002258635,0.4099541,0.01885504,0.5537176,0.00391643,0.01113628],"study_design_scores_gemma":[0.0001764371,0.00006495859,0.00008631534,0.00002204537,0.000008257129,0.00000485322,0.00001327087,0.9814659,0.001209924,0.0165055,0.0002882933,0.000154218],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06729376,0.00005928976,0.9299402,0.001555796,0.0003235555,0.0002803806,0.0000267036,0.0001795278,0.000340739],"genre_scores_gemma":[0.9403527,8.433629e-7,0.05670806,0.001467458,0.0002266872,0.00006110624,0.00005812962,0.00001008669,0.001114955],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8732322,"threshold_uncertainty_score":0.6169633,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04054293453064117,"score_gpt":0.2964657277876325,"score_spread":0.2559227932569913,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}