{"id":"W4405812114","doi":"10.1109/tpami.2024.3522295","title":"A Review of Deep Learning for Video Captioning","year":2024,"lang":"en","type":"review","venue":"IEEE Transactions on Pattern Analysis and Machine Intelligence","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"Closed captioning; Computer science; Artificial intelligence; Deep learning; Computer vision; Natural language processing; Multimedia; Image (mathematics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001152362,0.001191512,0.0009116404,0.00269186,0.0002979482,0.00114883,0.001455393,0.001163617,0.006027457],"category_scores_gemma":[0.003330829,0.0006093941,0.0006580837,0.003717092,0.0004460585,0.002466867,0.0008024632,0.001572577,0.00344023],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009151598,"about_ca_system_score_gemma":0.001579838,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003218893,"about_ca_topic_score_gemma":0.003487656,"domain_scores_codex":[0.9996276,0.0000722259,0.00004644401,0.00008564589,0.0001386241,0.00002942015],"domain_scores_gemma":[0.9986099,0.0007998521,0.00007229715,0.00004714211,0.0004251908,0.00004560481],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00003906215,0.00006063146,0.0002579553,0.0094545,0.00007368459,0.00006569336,0.0000512283,0.002519192,0.0009157933,0.007961229,0.04185257,0.9367486],"study_design_scores_gemma":[0.00001679696,0.0001612328,0.00129737,0.006073137,0.0001897521,0.0006411682,0.00008667963,0.00651495,0.002219611,0.008182297,0.9745544,0.00006267364],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.0005173209,0.9843301,0.008838804,0.0007168289,0.0004868483,0.00002874793,0.0001499227,0.00008689068,0.004844585],"genre_scores_gemma":[0.004413906,0.9844288,0.007044401,0.0005028509,0.0005786122,0.00003968714,0.000335364,0.0000358095,0.002620677],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.006027457,"threshold_uncertainty_score":0.02016383,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03868040932106809,"score_gpt":0.3533350145359122,"score_spread":0.3146546052148441,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}