{"id":"W4390872674","doi":"10.1109/iccv51070.2023.00658","title":"Generating Realistic Images from In-the-wild Sounds","year":2023,"lang":"en","type":"article","venue":"","topic":"Music and Audio Processing","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Computer science; Closed captioning; Task (project management); Sound (geography); Speech recognition; Artificial intelligence; Modalities; Audio signal processing; Audio analyzer; Sound quality; Pattern recognition (psychology); Audio signal; Image (mathematics); Acoustics; Speech coding","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002919033,0.00005903812,0.00006475921,0.00004803384,0.0001270557,0.0003336631,0.0005282527,0.0000196684,0.0000226414],"category_scores_gemma":[0.0000494458,0.00003996684,0.00001846181,0.0004958056,0.00001810356,0.0002166819,0.000125644,0.00007288757,0.0001052695],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000007766925,"about_ca_system_score_gemma":0.00002639793,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002843997,"about_ca_topic_score_gemma":0.00005253181,"domain_scores_codex":[0.9993147,0.00003647497,0.000117647,0.0002009279,0.0001593306,0.0001709438],"domain_scores_gemma":[0.9995279,0.0001560405,0.00002805521,0.0002544454,0.00001393189,0.00001962196],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000002186995,0.00006477643,0.005378839,0.00004510598,0.00001830035,0.000336363,0.01510983,0.005136603,0.01457098,0.07226844,0.3566567,0.5304119],"study_design_scores_gemma":[0.0003239335,0.00002311087,0.01249636,0.00006070806,0.000005221008,0.0000109707,0.0005270498,0.9145448,0.002069906,0.06044359,0.009137552,0.0003567701],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06773156,0.00008561864,0.8935433,0.008188029,0.0002626319,0.00005572041,0.000001850035,0.0003347381,0.02979656],"genre_scores_gemma":[0.9545559,0.000006834538,0.03868809,0.005170592,0.0002158003,0.000008087529,0.000006571495,0.000004691569,0.001343403],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9094082,"threshold_uncertainty_score":0.3217521,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03138468770716422,"score_gpt":0.2697742004491064,"score_spread":0.2383895127419421,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}