{"id":"W4413147826","doi":"10.1109/cvpr52734.2025.01361","title":"Assessing and Learning Alignment of Unimodal Vision and Language Models","year":2025,"lang":"en","type":"article","venue":"","topic":"Categorization, perception, and language","field":"Psychology","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Computer science; Artificial intelligence; Computer vision; Natural language processing","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005301033,0.001481372,0.001134202,0.001667132,0.0006295728,0.003191327,0.002019502,0.001969653,0.003825105],"category_scores_gemma":[0.02687216,0.0008142801,0.001170858,0.00106816,0.001258184,0.00542716,0.00357053,0.002988457,0.002394759],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001549213,"about_ca_system_score_gemma":0.001946529,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005098125,"about_ca_topic_score_gemma":0.003699039,"domain_scores_codex":[0.9961629,0.001423175,0.0001821252,0.001271056,0.0006036984,0.000356925],"domain_scores_gemma":[0.993206,0.003415754,0.0006018066,0.001150871,0.001212974,0.0004125956],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0008764179,0.0004603737,0.01361698,0.0003283738,0.0005667523,0.0002542275,0.0006963962,0.2248672,0.03468168,0.01126035,0.006635017,0.7057563],"study_design_scores_gemma":[0.00002010506,0.0002159203,0.00262934,0.00002812592,0.00004853554,0.000096281,0.0002191787,0.9638281,0.01360702,0.0181645,0.001098583,0.00004423769],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1782144,0.0007514064,0.8080885,0.0007217574,0.0001557898,0.0001761884,0.0004602571,0.00580431,0.005627342],"genre_scores_gemma":[0.8731771,0.0002368056,0.1210707,0.0004795784,0.00006673436,0.0001649593,0.001432388,0.0005397638,0.002831884],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005301033,"threshold_uncertainty_score":0.02803487,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01402788751029604,"score_gpt":0.3499927015957561,"score_spread":0.33596481408546,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}