{"id":"W4403988122","doi":"10.15837/ijccc.2024.6.6853","title":"Evaluating and Mitigating Gender Bias in Generative Large Language Models","year":2024,"lang":"en","type":"article","venue":"International Journal of Computers Communications & Control","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Generative grammar; Natural language processing; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009040958,0.0008770616,0.0005904399,0.0008957189,0.0005861011,0.001731626,0.001182907,0.001166002,0.002598112],"category_scores_gemma":[0.04131759,0.000385561,0.0007103149,0.0004820592,0.0008402494,0.001891459,0.002359738,0.00121491,0.001333221],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008855437,"about_ca_system_score_gemma":0.001251249,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002716484,"about_ca_topic_score_gemma":0.005610457,"domain_scores_codex":[0.995069,0.003150886,0.0002110673,0.000529506,0.0008530599,0.0001864913],"domain_scores_gemma":[0.97455,0.02133101,0.0006720457,0.001759318,0.001444594,0.0002428723],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00137744,0.0003189391,0.02321161,0.0007689582,0.0003062636,0.0005625158,0.00262668,0.2654544,0.02732893,0.01746401,0.009574687,0.6510055],"study_design_scores_gemma":[0.00007249916,0.0003676257,0.002392071,0.0001092725,0.00008962831,0.0003000336,0.0006478311,0.9381152,0.0273891,0.02407168,0.006390215,0.00005484912],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2712321,0.001834954,0.7121328,0.00133508,0.0004211261,0.0004190477,0.0008893458,0.005761856,0.005973721],"genre_scores_gemma":[0.7970182,0.0004308335,0.1956001,0.0006818778,0.00009420248,0.0002604608,0.001815822,0.0008249176,0.003273647],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.009040958,"threshold_uncertainty_score":0.04781371,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08070041347036223,"score_gpt":0.3999135031773028,"score_spread":0.3192130897069406,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}