{"id":"W4415306945","doi":"10.1109/iccv51701.2025.00863","title":"Perspective-Aware Reasoning in Vision-Language Models via Mental Imagery Simulation","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Perspective (graphical); Visual reasoning; Focus (optics); Construct (python library); Bridge (graph theory); Mental image; Spatial intelligence; Orientation (vector space); Object (grammar)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007570635,0.0007778528,0.0006714439,0.0004573593,0.0004724436,0.001984647,0.002123093,0.00115237,0.003402263],"category_scores_gemma":[0.003918524,0.0004699099,0.001847284,0.0003275331,0.001425815,0.002337083,0.002327735,0.001947924,0.0005614356],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001137604,"about_ca_system_score_gemma":0.001331719,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006556527,"about_ca_topic_score_gemma":0.007777845,"domain_scores_codex":[0.9994336,0.0002029418,0.00002820413,0.0001212687,0.0001507455,0.0000632201],"domain_scores_gemma":[0.9988436,0.0006305986,0.0001154543,0.0002140305,0.0001048271,0.00009152608],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001369449,0.00007822629,0.0006889152,0.000132188,0.0000664468,0.0002027887,0.000427697,0.832086,0.005030672,0.1303106,0.001350541,0.02948902],"study_design_scores_gemma":[0.00001376764,0.00001521668,0.00003654167,0.000007079719,0.000007841409,0.00001974909,0.00002102825,0.9581515,0.0008650033,0.04002745,0.0008275168,0.000007318315],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01697101,0.000119083,0.9783321,0.000283565,0.0000304678,0.0000516042,0.0001391615,0.0008332476,0.003239662],"genre_scores_gemma":[0.5903994,0.0002252979,0.4063222,0.0001853941,0.00003886242,0.0002287379,0.0004151911,0.0002104442,0.001974417],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006556527,"threshold_uncertainty_score":0.01303673,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01329232318466052,"score_gpt":0.3496713324522095,"score_spread":0.336379009267549,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}