{"id":"W1514535095","doi":"10.48550/arxiv.1502.03044","title":"Show, Attend and Tell: Neural Image Caption Generation with Visual Attention","year":2015,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":7525,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canadian Institute for Advanced Research; University of Toronto; Université de Montréal","funders":"","keywords":"Computer science; Benchmark (surveying); Artificial intelligence; Visualization; Gaze; Object (grammar); Backpropagation; Salient; Sequence (biology); Object detection; Machine translation; Image (mathematics); Artificial neural network; Machine learning; Computer vision; Pattern recognition (psychology)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008708236,0.001429545,0.0006569051,0.0008400625,0.0004916052,0.001031298,0.002432609,0.002228557,0.00756685],"category_scores_gemma":[0.004161246,0.0005923039,0.0007824763,0.0007845811,0.0007864303,0.002500932,0.001588558,0.00190988,0.002448102],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001014995,"about_ca_system_score_gemma":0.0006239883,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007836649,"about_ca_topic_score_gemma":0.01137355,"domain_scores_codex":[0.9996555,0.00009229153,0.00001229826,0.0001238032,0.00007257556,0.00004359007],"domain_scores_gemma":[0.9990698,0.0004825535,0.00004823041,0.0002088273,0.0001304791,0.00006016395],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006173333,0.0002694919,0.0009739755,0.0003818756,0.0001857627,0.0003440555,0.0002956024,0.1508136,0.0299698,0.01622626,0.0841597,0.7157626],"study_design_scores_gemma":[0.00005160049,0.00007078199,0.0002298033,0.00001545525,0.00002286343,0.00008817526,0.00002418372,0.9688085,0.01167903,0.01406701,0.00492134,0.00002121397],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.05071336,0.003153895,0.8817399,0.001608391,0.001054017,0.0003673916,0.002217181,0.04610902,0.01303676],"genre_scores_gemma":[0.521648,0.001022969,0.447942,0.001464084,0.0004796683,0.000443454,0.005082408,0.002350086,0.01956726],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007836649,"threshold_uncertainty_score":0.02531362,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06664802413473758,"score_gpt":0.2082971016600826,"score_spread":0.141649077525345,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}