{"id":"W4285199616","doi":"10.18653/v1/2022.bigscience-1.3","title":"You reap what you sow: On the Challenges of Bias Evaluation Under Multilingual Settings","year":2022,"lang":"en","type":"preprint","venue":"","topic":"Interpreting and Communication in Healthcare","field":"Health Professions","cited_by":62,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University","funders":"","keywords":"oskar; STELLA (programming language); Computer science; Sociology; Psychology; Artificial intelligence; Art; Art history","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1784952,0.001175957,0.002874763,0.004366993,0.00526657,0.01842986,0.003626446,0.006018331,0.007291991],"category_scores_gemma":[0.4852252,0.00119535,0.001287382,0.004232721,0.0151956,0.03385015,0.01866438,0.008161905,0.003019132],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003734574,"about_ca_system_score_gemma":0.007112726,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01008196,"about_ca_topic_score_gemma":0.01518473,"domain_scores_codex":[0.8184619,0.1513578,0.005709595,0.007102468,0.01580878,0.001559533],"domain_scores_gemma":[0.4821908,0.4466659,0.008003067,0.02586929,0.03241542,0.004855506],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.001528317,0.0002068139,0.01902892,0.002050782,0.0007210071,0.0006919771,0.02847293,0.005886072,0.002170829,0.1486852,0.07994844,0.7106087],"study_design_scores_gemma":[0.0003147327,0.000138475,0.00340106,0.001729937,0.0002520363,0.0009532684,0.01783755,0.0244375,0.004335639,0.8694053,0.0769391,0.0002554147],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05830757,0.04552599,0.581551,0.2754245,0.003043164,0.0006283732,0.002156788,0.003510912,0.02985164],"genre_scores_gemma":[0.5008498,0.008606997,0.4545656,0.02032831,0.00292693,0.0008580263,0.001821139,0.00315484,0.006888399],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8215048,"threshold_uncertainty_score":0.943984,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3826491857415912,"score_gpt":0.5224920837813289,"score_spread":0.1398428980397377,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}