{"id":"W4402667124","doi":"10.18653/v1/2024.acl-long.23","title":"Subtle Biases Need Subtler Measures: Dual Metrics for Evaluating Representative and Affinity Bias in Large Language Models","year":2024,"lang":"en","type":"article","venue":"","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Brock University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Dual (grammatical number); Natural language processing; Language model; Dual language; Artificial intelligence; Machine learning; Psychology; Linguistics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02661588,0.001619735,0.0009186412,0.003948607,0.001133316,0.004251632,0.0009903107,0.001794471,0.001487733],"category_scores_gemma":[0.1639572,0.0004585813,0.0009677163,0.002821072,0.003103638,0.005615359,0.005143665,0.002518205,0.0005392616],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001126804,"about_ca_system_score_gemma":0.001129355,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001169954,"about_ca_topic_score_gemma":0.002193544,"domain_scores_codex":[0.9746935,0.01480244,0.002126762,0.002507807,0.005281246,0.0005882061],"domain_scores_gemma":[0.8170338,0.144451,0.0115148,0.01685111,0.007964921,0.002184475],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.002257465,0.0006162702,0.34152,0.001879179,0.001837429,0.0006676696,0.01522016,0.140066,0.02402864,0.05199156,0.005570101,0.4143456],"study_design_scores_gemma":[0.0002390802,0.001198785,0.07737549,0.0003810546,0.0004501187,0.0008543068,0.004269357,0.7109102,0.02466359,0.17023,0.009109729,0.0003183733],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.4906021,0.001164305,0.4962372,0.0007169751,0.0001013826,0.0005521934,0.001094037,0.001650548,0.007881207],"genre_scores_gemma":[0.9238921,0.0001270033,0.07406301,0.0001482213,0.00005900426,0.0003330655,0.0007117091,0.0002406472,0.0004251399],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02661588,"threshold_uncertainty_score":0.1407599,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4050501291309129,"score_gpt":0.5104728602784459,"score_spread":0.105422731147533,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}