{"id":"W3197581510","doi":"10.1609/aaai.v36i10.21306","title":"Retrieve, Caption, Generate: Visual Grounding for Enhancing Commonsense in Text Generation Models","year":2022,"lang":"en","type":"preprint","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Commonsense reasoning; Closed captioning; Computer science; Fluency; Transformer; Commonsense knowledge; Generative grammar; Artificial intelligence; Natural language processing; Image (mathematics); Linguistics; Engineering; Knowledge extraction","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001396303,0.001109896,0.0004023778,0.0009306184,0.0003336195,0.001447661,0.001511046,0.001114597,0.009763787],"category_scores_gemma":[0.007630596,0.0003537649,0.0007883688,0.0004450798,0.0008111649,0.003174642,0.001750914,0.001387733,0.002008065],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006509093,"about_ca_system_score_gemma":0.0004935744,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001896586,"about_ca_topic_score_gemma":0.003145586,"domain_scores_codex":[0.9994905,0.000205837,0.0000229409,0.0001383601,0.0001058961,0.00003631179],"domain_scores_gemma":[0.9975936,0.001493208,0.000108707,0.0005459148,0.0001797743,0.00007880077],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006743526,0.0005442111,0.002388068,0.0007461716,0.0001142087,0.0006642825,0.001676304,0.209278,0.05257025,0.05177972,0.02189815,0.6576661],"study_design_scores_gemma":[0.00007672538,0.000166397,0.0003832192,0.00004200849,0.00004611166,0.0002059538,0.0001560229,0.9070722,0.04190309,0.03586565,0.0140478,0.00003491182],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0639674,0.0003749466,0.8950602,0.0007010518,0.0001497205,0.000375411,0.001277043,0.02627511,0.01181908],"genre_scores_gemma":[0.546347,0.0001916502,0.4442235,0.0003323595,0.00005413192,0.0002508608,0.002597098,0.00171656,0.004286891],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.009763787,"threshold_uncertainty_score":0.03266311,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1422581201807229,"score_gpt":0.355538293976938,"score_spread":0.2132801737962151,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}