{"id":"W2973802306","doi":"10.1109/iccv.2019.00900","title":"Watch, Listen and Tell: Multi-Modal Weakly Supervised Dense Event Captioning","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute; Canadian Institute for Advanced Research; University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Closed captioning; Computer science; Event (particle physics); Modal; Focus (optics); Feature (linguistics); Representation (politics); Modalities; Ranging; Speech recognition; Audio visual; Natural language processing; Artificial intelligence; Multimedia; Image (mathematics); Linguistics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001650352,0.002100864,0.00124477,0.00123364,0.000726093,0.001585505,0.002581633,0.002273081,0.004871496],"category_scores_gemma":[0.005498036,0.0005677054,0.001126058,0.00134991,0.00111474,0.002695073,0.00242798,0.002845018,0.002431063],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008778521,"about_ca_system_score_gemma":0.0006948579,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003977765,"about_ca_topic_score_gemma":0.006096236,"domain_scores_codex":[0.9987552,0.0004912089,0.00004334876,0.0004275376,0.0001601857,0.0001225083],"domain_scores_gemma":[0.9977017,0.00116592,0.000150933,0.0004994801,0.0003166271,0.000165404],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001742542,0.0004643541,0.001986685,0.0006654278,0.0002309586,0.0006353173,0.0006368202,0.1853518,0.02660934,0.007424237,0.05896023,0.7152923],"study_design_scores_gemma":[0.00004014392,0.0001144615,0.0009046799,0.00003428917,0.00003319106,0.0001712275,0.0001361418,0.9680541,0.01050031,0.01285848,0.007114844,0.00003804391],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05355845,0.00169994,0.917724,0.001177705,0.0004126322,0.000405176,0.003761815,0.01426583,0.006994524],"genre_scores_gemma":[0.5473368,0.0009309738,0.4144725,0.00108786,0.0008061775,0.0004984767,0.01892065,0.00132329,0.01462322],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004871496,"threshold_uncertainty_score":0.01629674,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02367218328816387,"score_gpt":0.2905838013795247,"score_spread":0.2669116180913608,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}