{"id":"W4385572364","doi":"10.18653/v1/2023.findings-acl.590","title":"Zero-shot Visual Question Answering with Language Model Feedback","year":2023,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"National Natural Science Foundation of China","keywords":"Closed captioning; Computer science; Leverage (statistics); Language model; Task (project management); Artificial intelligence; Context (archaeology); Question answering; Natural language processing; Code (set theory); Scheme (mathematics); Machine learning; Speech recognition; Image (mathematics); Programming language; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002454515,0.002367394,0.001535209,0.0015603,0.0008477391,0.001932596,0.004339433,0.003160649,0.01051034],"category_scores_gemma":[0.01148318,0.0005569017,0.001248418,0.0009323694,0.00120828,0.004659845,0.00313288,0.002706946,0.004046513],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001515173,"about_ca_system_score_gemma":0.001461802,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006622746,"about_ca_topic_score_gemma":0.007930041,"domain_scores_codex":[0.9970162,0.001217804,0.0001093914,0.001004886,0.0004746478,0.0001771052],"domain_scores_gemma":[0.9951081,0.002790024,0.0002171705,0.0008185476,0.0008435056,0.0002225292],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007759331,0.0007822505,0.001403366,0.001555686,0.0002059858,0.0004619479,0.001404451,0.06253485,0.04802319,0.01388441,0.06482504,0.8041428],"study_design_scores_gemma":[0.0001284279,0.0004105097,0.0006708538,0.00009994592,0.0001053389,0.0003366738,0.0003962264,0.9099296,0.02714532,0.03574258,0.02492999,0.0001045728],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01687664,0.001976442,0.9418402,0.0007788063,0.0003869329,0.0004698923,0.001386004,0.03059982,0.005685203],"genre_scores_gemma":[0.4049708,0.0006877096,0.5702156,0.00213807,0.0004942555,0.0006633343,0.009445169,0.001473857,0.009911025],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01051034,"threshold_uncertainty_score":0.03516066,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01670687766372517,"score_gpt":0.3167307179897402,"score_spread":0.300023840326015,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}