{"id":"W4404781442","doi":"10.18653/v1/2024.nlp4science-1.3","title":"What an Elegant Bridge: Multilingual LLMs are Biased Similarly in Different Languages","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Engineering and Physical Sciences Research Council","keywords":"Bridge (graph theory); Computer science; Linguistics; Natural language processing; Medicine; Philosophy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006338582,0.0004767849,0.000688489,0.0008964378,0.0007676705,0.003595336,0.0008093544,0.001019172,0.004959762],"category_scores_gemma":[0.04588643,0.0003670436,0.0005253128,0.0005775866,0.003053955,0.008720725,0.003122937,0.00188771,0.001200236],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006481872,"about_ca_system_score_gemma":0.0005091279,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002393961,"about_ca_topic_score_gemma":0.002431789,"domain_scores_codex":[0.9966072,0.001384322,0.0001560551,0.001086564,0.0005031655,0.0002626358],"domain_scores_gemma":[0.9857735,0.006759947,0.001747062,0.003959659,0.00113613,0.0006237559],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001705305,0.0002617264,0.31404,0.0006031058,0.001677364,0.0006290831,0.03582575,0.007869855,0.04805471,0.2402619,0.01519503,0.3338763],"study_design_scores_gemma":[0.0001125677,0.0003539186,0.1198748,0.00031018,0.000426559,0.001183068,0.01621232,0.02823608,0.02018614,0.7795327,0.033204,0.0003676884],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8392336,0.001414355,0.1066249,0.01536638,0.0005002479,0.00006089509,0.0009544154,0.0008845079,0.03496073],"genre_scores_gemma":[0.9883317,0.0001546858,0.008092729,0.001656689,0.0001043476,0.00002519602,0.0001804813,0.0002577431,0.001196371],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.006338582,"threshold_uncertainty_score":0.03352201,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02503549506947888,"score_gpt":0.3316981056926324,"score_spread":0.3066626106231535,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}