{"id":"W7093322906","doi":"10.1145/3746276.3760469","title":"TemporalCook: Benchmarking Temporal and Procedural Reasoning in Multimodal Large Language Models","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University","funders":"","keywords":"Benchmark (surveying); Benchmarking; Question answering; Inference; Task (project management); Language model; Code (set theory); Visual reasoning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005976774,0.003083167,0.001033274,0.002087226,0.001102836,0.002495802,0.004457594,0.003418705,0.0106694],"category_scores_gemma":[0.02468925,0.0006825979,0.002484643,0.001722352,0.001224608,0.003865291,0.003091249,0.00351564,0.005078654],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002906004,"about_ca_system_score_gemma":0.002919087,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.03606138,"about_ca_topic_score_gemma":0.05235074,"domain_scores_codex":[0.9958169,0.001906217,0.0003141242,0.001055254,0.0006373901,0.0002702278],"domain_scores_gemma":[0.9901564,0.006602093,0.0002451834,0.001602055,0.001000075,0.0003941882],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002036959,0.002275333,0.0112453,0.003993182,0.001115389,0.0005747991,0.000604094,0.372463,0.005719889,0.01043509,0.2384963,0.3510407],"study_design_scores_gemma":[0.0005938027,0.0006600737,0.002535374,0.0002071793,0.0001538472,0.0002532891,0.0004057355,0.9465889,0.0065036,0.01129142,0.03071489,0.00009200622],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.3717642,0.01846787,0.2629467,0.005730457,0.002718264,0.003288354,0.1236968,0.1554005,0.05598696],"genre_scores_gemma":[0.472519,0.002158601,0.2427487,0.002099473,0.0003141778,0.001983751,0.2618367,0.004990079,0.01134939],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03606138,"threshold_uncertainty_score":0.07170296,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.009377410152993777,"score_gpt":0.2913720655981711,"score_spread":0.2819946554451774,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}