{"id":"W4402772286","doi":"10.1109/cvpr52733.2024.01355","title":"EgoThink: Evaluating First-Person Perspective Thinking Capability of Vision-Language Models","year":2024,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Perspective (graphical); Computer science; Human–computer interaction; Artificial intelligence; Natural language processing","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004390691,0.00181249,0.0006694018,0.001445878,0.0004680025,0.001507453,0.001561127,0.00170621,0.00368727],"category_scores_gemma":[0.01633687,0.0002363662,0.000802326,0.0006625929,0.0006046811,0.002536607,0.002290382,0.001615887,0.001111206],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001046628,"about_ca_system_score_gemma":0.0008887474,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004265076,"about_ca_topic_score_gemma":0.005836131,"domain_scores_codex":[0.9972851,0.001354714,0.0001854714,0.0006053014,0.0003923831,0.000176923],"domain_scores_gemma":[0.9927965,0.005033793,0.0004497156,0.0007242698,0.0005673437,0.0004284132],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003728685,0.002681952,0.04813377,0.00313733,0.001320248,0.0004887738,0.002688833,0.1802208,0.01671243,0.006430061,0.03594643,0.6985106],"study_design_scores_gemma":[0.0002893932,0.002342934,0.01964127,0.0002082216,0.0002676289,0.0002576776,0.001248505,0.9405463,0.01601875,0.009423153,0.009647614,0.0001086028],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8472778,0.004528296,0.1134879,0.0008199697,0.0004746068,0.0009487172,0.0049605,0.009348582,0.01815368],"genre_scores_gemma":[0.9313183,0.0004763317,0.05730398,0.0002833433,0.00006540399,0.0003994432,0.007576407,0.0002147639,0.002361974],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.004390691,"threshold_uncertainty_score":0.02322042,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02926812550214529,"score_gpt":0.357759155973917,"score_spread":0.3284910304717717,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}