{"id":"W4416034471","doi":"10.18653/v1/2025.findings-emnlp.1340","title":"QEVA: A Reference-Free Evaluation Metric for Narrative Video Summarization with Multimodal Question Answering","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Institute for Information and Communications Technology Promotion; Ministry of Science and ICT, South Korea; Chung-Ang University","keywords":"Automatic summarization; Question answering; Metric (unit); Narrative; Key (lock)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007323856,0.002049114,0.001218971,0.007019382,0.0008793766,0.002302701,0.001737811,0.00169586,0.004100266],"category_scores_gemma":[0.05180514,0.0002588345,0.0008034107,0.00364839,0.0005922579,0.004245045,0.002700846,0.001081813,0.00211162],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001266377,"about_ca_system_score_gemma":0.001173601,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004423648,"about_ca_topic_score_gemma":0.006069425,"domain_scores_codex":[0.9904937,0.004387974,0.001166685,0.001452872,0.002215345,0.0002834532],"domain_scores_gemma":[0.9748893,0.0142126,0.002015238,0.002392233,0.005846661,0.0006439763],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.002229668,0.0005393497,0.02023561,0.004801234,0.0009528255,0.000352348,0.001574932,0.03350658,0.03510474,0.006478678,0.06925961,0.8249645],"study_design_scores_gemma":[0.0004326648,0.006107138,0.07101952,0.001029026,0.0008238219,0.00169615,0.003027447,0.7043587,0.08299372,0.02688367,0.1010224,0.0006057361],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2194272,0.01682816,0.6675365,0.001168956,0.001112016,0.002161104,0.04114192,0.0327623,0.01786187],"genre_scores_gemma":[0.5709931,0.001531977,0.3500522,0.000335524,0.0003022072,0.001764692,0.06969469,0.001255342,0.004070182],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007323856,"threshold_uncertainty_score":0.03873277,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02417721802176372,"score_gpt":0.3427726956916791,"score_spread":0.3185954776699154,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}