{"id":"W4404775376","doi":"10.1049/cit2.12375","title":"Feature pyramid attention network for audio‐visual scene classification","year":2024,"lang":"en","type":"article","venue":"CAAI Transactions on Intelligence Technology","topic":"Music and Audio Processing","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"Basic and Applied Basic Research Foundation of Guangdong Province","keywords":"Pyramid (geometry); Audio visual; Computer science; Artificial intelligence; Feature (linguistics); Computer vision; Pattern recognition (psychology); Multimedia; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004273093,0.0009656515,0.0006371539,0.001106991,0.0003021187,0.0005468688,0.00122959,0.0008539744,0.003373276],"category_scores_gemma":[0.0009264507,0.0002213277,0.000727559,0.0009077097,0.0002524947,0.0009229897,0.0007467975,0.0008766936,0.0006492866],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009962631,"about_ca_system_score_gemma":0.0007190637,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02055611,"about_ca_topic_score_gemma":0.01461526,"domain_scores_codex":[0.9996989,0.00003812161,0.00001211949,0.0001067019,0.00006960755,0.00007457111],"domain_scores_gemma":[0.9997047,0.00009727317,0.00002252292,0.00003024077,0.0001199331,0.00002536874],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0006603103,0.0004292527,0.001937545,0.0001628157,0.000147413,0.0002048634,0.00006194986,0.1565706,0.04638472,0.002207345,0.01245246,0.7787808],"study_design_scores_gemma":[0.00001013532,0.00006870901,0.000799449,0.000004655084,0.00002760148,0.00002830225,0.00001171557,0.9928988,0.004437609,0.001104715,0.0006028711,0.000005408032],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2239748,0.002836052,0.7509708,0.0007411634,0.0004326692,0.0002129927,0.001310255,0.009949741,0.00957153],"genre_scores_gemma":[0.9004648,0.0005089035,0.09010822,0.0003465605,0.0001447969,0.00008985071,0.001905893,0.0001189806,0.006311932],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02055611,"threshold_uncertainty_score":0.04087293,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02904321832120359,"score_gpt":0.3099807059072142,"score_spread":0.2809374875860106,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}