{"id":"W4402671356","doi":"10.18653/v1/2024.arabicnlp-1.18","title":"John vs. Ahmed: Debate-Induced Bias in Multilingual LLMs","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Alliance de recherche numérique du Canada; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Political science; Computer science","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004153104,0.000149762,0.0001457127,0.0003179912,0.00003709654,0.0004072251,0.0008660547,0.0001092909,0.00003027354],"category_scores_gemma":[0.0001524964,0.0001130418,0.0000506913,0.0008372338,0.00002174959,0.000602779,0.0003096431,0.0003303475,0.00009943919],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00007336726,"about_ca_system_score_gemma":0.0001242025,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003520501,"about_ca_topic_score_gemma":0.000164737,"domain_scores_codex":[0.9987111,0.00004345672,0.0002286287,0.0004675773,0.000247881,0.0003013459],"domain_scores_gemma":[0.999351,0.0001387397,0.00002443732,0.0003781429,0.00004360945,0.00006412739],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.000007386656,0.00007751557,0.0003214629,0.0001173118,0.00001519283,0.001180899,0.002837878,0.000004751299,0.02970905,0.1147664,0.002354968,0.8486072],"study_design_scores_gemma":[0.0002940848,0.000122193,0.0002903531,0.0003952699,0.000005580185,0.0001259745,0.00005469134,0.1886074,0.7532016,0.05015036,0.006115263,0.000637267],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4563853,0.01125875,0.4940598,0.00929281,0.002529807,0.0007340708,0.000002952339,0.01375443,0.01198208],"genre_scores_gemma":[0.7535127,0.000008441509,0.245441,0.0004112047,0.00004842724,0.00001046161,9.313391e-7,0.00001103399,0.0005557638],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8479699,"threshold_uncertainty_score":0.4609712,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03865088048033549,"score_gpt":0.3257001436754829,"score_spread":0.2870492631951474,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}