{"id":"W4390872674","doi":"10.1109/iccv51070.2023.00658","title":"Generating Realistic Images from In-the-wild Sounds","year":2023,"lang":"en","type":"article","venue":"","topic":"Music and Audio Processing","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Computer science; Closed captioning; Task (project management); Sound (geography); Speech recognition; Artificial intelligence; Modalities; Audio signal processing; Audio analyzer; Sound quality; Pattern recognition (psychology); Audio signal; Image (mathematics); Acoustics; Speech coding","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004611669,0.001169774,0.0005335578,0.0006465227,0.0001915953,0.0008456808,0.0009281947,0.001035301,0.003785357],"category_scores_gemma":[0.002763045,0.0002852984,0.0009300984,0.0003565299,0.0004365965,0.001165294,0.0009503921,0.0008984789,0.001555081],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003089048,"about_ca_system_score_gemma":0.0003037436,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009456686,"about_ca_topic_score_gemma":0.001969255,"domain_scores_codex":[0.9996907,0.00005793903,0.00001535918,0.0001095459,0.00009493101,0.0000314764],"domain_scores_gemma":[0.9993182,0.0003586706,0.00003237057,0.0001155559,0.0001314209,0.00004376949],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001134436,0.0005653723,0.002689488,0.001947416,0.000264403,0.00140811,0.0003692213,0.2141653,0.1948954,0.007503701,0.04279504,0.5322622],"study_design_scores_gemma":[0.0001647533,0.0005579387,0.002744714,0.0001212226,0.0001160902,0.001120623,0.0002895752,0.874562,0.08348789,0.01298481,0.02376876,0.0000816736],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1280595,0.001769588,0.8432344,0.0008875173,0.0009288568,0.000755483,0.004377067,0.008979659,0.01100792],"genre_scores_gemma":[0.4754426,0.001117595,0.5043979,0.00079706,0.0003011332,0.0005261548,0.008543214,0.001074745,0.007799663],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003785357,"threshold_uncertainty_score":0.0126633,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03138468770716422,"score_gpt":0.2697742004491064,"score_spread":0.2383895127419421,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}