{"id":"W2905288264","doi":"10.1609/aaai.v33i01.33018885","title":"Connecting Language to Images: A Progressive Attention-Guided Network for Simultaneous Image Captioning and Language Grounding","year":2019,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"Science and Technology Planning Project of Guangdong Province; National Key Research and Development Program of China; China Scholarship Council; National Natural Science Foundation of China","keywords":"Closed captioning; Bounding overwatch; Computer science; Benchmark (surveying); Artificial intelligence; Image (mathematics); Process (computing); Machine learning; Natural language processing; Pattern recognition (psychology)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007669594,0.00168264,0.0008537954,0.000975541,0.0006060841,0.0009037572,0.003121105,0.002292826,0.003161468],"category_scores_gemma":[0.003156462,0.0006532161,0.001001616,0.0007245616,0.001202465,0.002747865,0.001837265,0.00208037,0.0009895161],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00137621,"about_ca_system_score_gemma":0.001092637,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009106182,"about_ca_topic_score_gemma":0.01071468,"domain_scores_codex":[0.9995703,0.0000939907,0.00001473017,0.0001991279,0.00005855751,0.00006325251],"domain_scores_gemma":[0.999183,0.0003745973,0.00007771824,0.0001387463,0.0001546684,0.00007128549],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005802085,0.000421444,0.002055726,0.0003383116,0.0001512251,0.0005885569,0.0004485749,0.4423081,0.03156727,0.01548373,0.0169531,0.4891038],"study_design_scores_gemma":[0.00001840815,0.00005525755,0.0001555254,0.00001340666,0.00002282766,0.00005506953,0.00002201394,0.9843993,0.005132113,0.008855812,0.00125765,0.00001258755],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0812729,0.001424815,0.898527,0.001219901,0.0002921821,0.0002786376,0.0007252059,0.008442795,0.007816549],"genre_scores_gemma":[0.7429969,0.000543451,0.2435543,0.0009871739,0.0001944398,0.000285187,0.001875688,0.0003928459,0.009169958],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.009106182,"threshold_uncertainty_score":0.01810634,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0302889741580122,"score_gpt":0.3297025353385947,"score_spread":0.2994135611805825,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}