{"id":"W4407304464","doi":"10.1109/access.2025.3540388","title":"BVQA: Connecting Language and Vision Through Multimodal Attention for Open-Ended Question Answering","year":2025,"lang":"en","type":"article","venue":"IEEE Access","topic":"Speech and dialogue systems","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"Athabasca University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Question answering; Natural language processing; Closed-ended question; Artificial intelligence; Human–computer interaction; Linguistics; Philosophy","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003342889,0.0000896682,0.0001364948,0.00006376723,0.0001998014,0.0008681627,0.0006627899,0.000057743,9.956568e-7],"category_scores_gemma":[0.00008829637,0.00008424942,0.00002935412,0.0002067392,0.00001245069,0.00183568,0.0002949971,0.00005464764,0.000002969049],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003134475,"about_ca_system_score_gemma":0.00002576874,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001020758,"about_ca_topic_score_gemma":0.0001867064,"domain_scores_codex":[0.9991899,0.00004944683,0.0001704984,0.0003365098,0.00008520827,0.0001684047],"domain_scores_gemma":[0.9994634,0.0001243896,0.00007772145,0.0002424554,0.00006232022,0.00002972267],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003376962,0.0003566568,0.02153976,0.001050554,0.0001530218,0.00005072664,0.007074869,0.0008264632,0.3061037,0.1027406,0.006968517,0.5527975],"study_design_scores_gemma":[0.01314326,0.000619204,0.1088153,0.002146785,0.000082709,0.00006314612,0.001895706,0.4436241,0.3702374,0.05261855,0.00520178,0.00155199],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3462335,0.0001471798,0.650416,0.0002672429,0.0011896,0.0004174687,0.000002947408,0.0001009533,0.001225088],"genre_scores_gemma":[0.9839489,0.000004757392,0.01562394,0.0001608945,0.00009707313,0.00004202561,0.000006862103,0.000005220267,0.0001102981],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6377154,"threshold_uncertainty_score":0.8371713,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02959379945198485,"score_gpt":0.386820533552708,"score_spread":0.3572267341007231,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}