{"id":"W2560645892","doi":"10.1109/iccv.2017.140","title":"Areas of Attention for Image Captioning","year":2017,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":218,"is_retracted":false,"has_abstract":true,"ca_institutions":"École de Technologie Supérieure","funders":"Agence Nationale de la Recherche","keywords":"Closed captioning; Computer science; Pairwise comparison; Transformer; Artificial intelligence; Convolutional neural network; Image (mathematics); Language model; Pattern recognition (psychology); Computer vision","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001107255,0.00169072,0.0007889174,0.001714679,0.0005119204,0.001560753,0.00216928,0.001540958,0.006123016],"category_scores_gemma":[0.003952969,0.0005401542,0.001455524,0.001189853,0.0009690293,0.00333475,0.002123064,0.002283188,0.002708413],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001325295,"about_ca_system_score_gemma":0.0009001623,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005393265,"about_ca_topic_score_gemma":0.007413249,"domain_scores_codex":[0.9990984,0.000233317,0.0000394038,0.0003570968,0.0001506507,0.0001210611],"domain_scores_gemma":[0.9986603,0.0004740253,0.0001277891,0.0003908764,0.0002695398,0.00007752395],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007371056,0.0002492642,0.002842713,0.0006313706,0.0001988207,0.0002915944,0.0004523749,0.1143268,0.04423166,0.01877627,0.03934645,0.7779157],"study_design_scores_gemma":[0.00003200288,0.0001294607,0.001320615,0.00004236658,0.00007064534,0.0002097205,0.00006707769,0.9267137,0.03274359,0.02047676,0.01816045,0.0000335176],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0496842,0.003529439,0.9048429,0.0008990293,0.0004569732,0.0003860443,0.002185701,0.02612534,0.01189039],"genre_scores_gemma":[0.6140251,0.001416337,0.362097,0.0007602937,0.0003740201,0.0004415552,0.006080111,0.001586934,0.01321867],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006123016,"threshold_uncertainty_score":0.02048349,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03048174510081268,"score_gpt":0.3389131894072359,"score_spread":0.3084314443064232,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}