{"id":"W4416037383","doi":"10.18653/v1/2025.emnlp-demos.68","title":"From Behavioral Performance to Internal Competence: Interpreting Vision-Language Models with VLM-Lens","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Natural (archaeology); Empirical research; Natural language; Action (physics)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003077312,0.0007556032,0.0004786758,0.0009874541,0.0002902262,0.00342736,0.001087868,0.0008899147,0.005839454],"category_scores_gemma":[0.03279773,0.0005367185,0.0006878377,0.0006979041,0.000884001,0.003749494,0.00226152,0.001656297,0.001579557],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001001796,"about_ca_system_score_gemma":0.0006996081,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009003407,"about_ca_topic_score_gemma":0.008384373,"domain_scores_codex":[0.9991401,0.0004405806,0.00003232781,0.0002005298,0.00009652555,0.00008990789],"domain_scores_gemma":[0.9929928,0.004408706,0.0006966342,0.000850739,0.0007358328,0.0003153888],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.002316091,0.0007964054,0.1578087,0.0009023143,0.0006967845,0.0008436683,0.007289207,0.09065472,0.03393197,0.05029895,0.03138271,0.6230784],"study_design_scores_gemma":[0.0001403402,0.0002951374,0.06026416,0.0002315639,0.0001886019,0.0002890813,0.00282415,0.7620947,0.01035821,0.1587742,0.004370769,0.0001690233],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5252227,0.001753381,0.4283381,0.004995226,0.0002923237,0.0002812042,0.005683789,0.006142809,0.02729047],"genre_scores_gemma":[0.9749108,0.0001536799,0.02286795,0.0001456405,0.0000171221,0.00006665708,0.0007023636,0.0002706311,0.0008652898],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.009003407,"threshold_uncertainty_score":0.01953495,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01081601759030118,"score_gpt":0.3170005470036598,"score_spread":0.3061845294133586,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}