{"id":"W4411630225","doi":"10.18653/v1/2024.eacl-long.41","title":"Examining Gender and Racial Bias in Large Vision–Language Models Using a Novel Dataset of Parallel Images","year":2024,"lang":"en","type":"article","venue":"","topic":"Categorization, perception, and language","field":"Psychology","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Artificial intelligence; Gender bias; Racial bias; Natural language processing; Computer vision; Race (biology); Psychology; Sociology; Gender studies","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002668991,0.001195482,0.0005161372,0.001077109,0.0006473495,0.001175176,0.001578647,0.001367682,0.002616952],"category_scores_gemma":[0.01040941,0.0002718662,0.0009843456,0.0008256488,0.0006526153,0.001628166,0.001055439,0.001809355,0.001404375],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001250073,"about_ca_system_score_gemma":0.0005904608,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01876561,"about_ca_topic_score_gemma":0.02675248,"domain_scores_codex":[0.998292,0.0007304958,0.0000783368,0.0005322545,0.0002420085,0.0001249033],"domain_scores_gemma":[0.99525,0.003143794,0.0002227495,0.0006910228,0.0004943957,0.0001981222],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.006767438,0.005049098,0.1461016,0.001989957,0.001382511,0.001886186,0.004031397,0.1305466,0.05919771,0.01056994,0.1683088,0.4641688],"study_design_scores_gemma":[0.0004621642,0.0009507842,0.08028584,0.0001562441,0.0001833629,0.001179319,0.002469678,0.8540888,0.01968189,0.01181558,0.0285343,0.0001919686],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9233543,0.002245375,0.03889551,0.001926087,0.000376505,0.0004017011,0.02195066,0.003550337,0.007299607],"genre_scores_gemma":[0.9068456,0.0002341694,0.04342003,0.000685063,0.0001319611,0.0002406682,0.04541814,0.0002309535,0.002793325],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01876561,"threshold_uncertainty_score":0.03731281,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1388596897583776,"score_gpt":0.40015976839756,"score_spread":0.2613000786391824,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}