{"id":"W4417517804","doi":"10.48550/arxiv.2505.08455","title":"VCRBench: Exploring Long-form Causal Reasoning Capabilities of Large Video Language Models","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Mitacs","keywords":"Causal reasoning; Causal model; Visual reasoning; Key (lock); Benchmark (surveying); Language model; Simple (philosophy); Modular design; Model-based reasoning","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003770841,0.002054026,0.0008464076,0.001366559,0.0005876452,0.002558055,0.003482931,0.001935499,0.007113093],"category_scores_gemma":[0.02464264,0.0005815263,0.001773518,0.0007300691,0.001026412,0.004913582,0.002285501,0.003291781,0.001862488],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002420453,"about_ca_system_score_gemma":0.002377706,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02554615,"about_ca_topic_score_gemma":0.02957438,"domain_scores_codex":[0.9973134,0.001130707,0.0001509291,0.0007577697,0.0004891857,0.0001580077],"domain_scores_gemma":[0.9876335,0.009863315,0.0004255224,0.001013416,0.0007749196,0.000289418],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009647295,0.0005808013,0.005295284,0.002149111,0.0003690981,0.0006962839,0.0009121677,0.5571337,0.01333689,0.03268789,0.02992704,0.355947],"study_design_scores_gemma":[0.00006000471,0.00009578434,0.0003139067,0.00005528225,0.000022927,0.00006095459,0.0001181017,0.9723218,0.003578311,0.0201882,0.003159754,0.00002494791],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09501944,0.003303661,0.8370188,0.001972229,0.0003853013,0.0006767461,0.008384991,0.04438023,0.008858585],"genre_scores_gemma":[0.5736766,0.0009234049,0.4021665,0.001092846,0.0001061668,0.0005354129,0.01642146,0.001763668,0.003314038],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02554615,"threshold_uncertainty_score":0.0507949,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05864879819581904,"score_gpt":0.3154730716831285,"score_spread":0.2568242734873095,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}