{"id":"W4415165924","doi":"10.1007/s00146-025-02670-7","title":"Not all AI-generated faces are created equal: impacts of model gender, race, and emotional expression on classification accuracy","year":2025,"lang":"en","type":"article","venue":"AI & Society","topic":"Ethics and Social Impacts of AI","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Facial expression; Emotional expression; Generative grammar; Emotional intelligence; Face (sociological concept); Expression (computer science); Variance (accounting); Generative model; Emotion recognition","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000927365,0.0001513769,0.0002381552,0.00003844895,0.0007627646,0.0001811378,0.000194047,0.0004141257,0.00002729774],"category_scores_gemma":[0.000924902,0.0001384192,0.0001395296,0.0002983326,0.0003457142,0.0005131254,0.00006176873,0.0003827672,0.00000296976],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001454683,"about_ca_system_score_gemma":0.0005390667,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0007362012,"about_ca_topic_score_gemma":0.0002025204,"domain_scores_codex":[0.9983978,0.0002146057,0.0002769765,0.0002905089,0.0005147554,0.0003053761],"domain_scores_gemma":[0.9984445,0.0003995091,0.0002311956,0.0001724988,0.0006046396,0.0001476617],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","study_design_scores_codex":[0.0001661071,0.0008004825,0.01228982,0.0002613862,0.0004236439,0.00000155448,0.1807855,0.002130051,0.3597581,0.2014853,0.2390153,0.002882836],"study_design_scores_gemma":[0.007345341,0.0005664563,0.2950883,0.002083195,0.0004968274,0.000001011038,0.1410495,0.1973582,0.09661652,0.2336194,0.02317168,0.002603543],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9449133,0.0003851596,0.002805299,0.0455512,0.000198274,0.0004093061,0.00009589184,0.0001238237,0.005517763],"genre_scores_gemma":[0.9887892,0.002092902,0.0004793689,0.007830944,0.00007726829,0.000008594486,0.00004237376,0.00001151429,0.000667853],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2827985,"threshold_uncertainty_score":0.5866646,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.14485044819305,"score_gpt":0.4251979108399317,"score_spread":0.2803474626468817,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}