{"id":"W7082273943","doi":"10.48448/rqsr-d504","title":"CulturalBench: A Robust, Diverse and Challenging Benchmark for Measuring LMs' Cultural Knowledge Through Human-AI Red-Teaming","year":2025,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Geochemistry and Geologic Mapping","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Overfitting; Set (abstract data type); Benchmark (surveying); Construct (python library); Frontier; Cultural diversity","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006995377,0.001943152,0.0006707064,0.004363768,0.001456732,0.003267838,0.001505053,0.002475578,0.005133471],"category_scores_gemma":[0.04532188,0.000315566,0.0007552829,0.002665001,0.001090968,0.005351818,0.004870774,0.001771024,0.005142377],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00149099,"about_ca_system_score_gemma":0.001604218,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008635038,"about_ca_topic_score_gemma":0.01296936,"domain_scores_codex":[0.9929413,0.003530647,0.0006386369,0.001252141,0.001293242,0.0003441179],"domain_scores_gemma":[0.9726906,0.01620474,0.001653715,0.003643259,0.004372532,0.001435163],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001277133,0.001828363,0.2415454,0.00376061,0.0005777509,0.0009529726,0.01430439,0.03909534,0.0182248,0.007320666,0.1274045,0.5437081],"study_design_scores_gemma":[0.0003922991,0.001833136,0.2402658,0.00146875,0.0002874784,0.0009851189,0.02070832,0.3962168,0.05308284,0.02866935,0.2555158,0.0005740939],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.7976639,0.002462267,0.09350414,0.002245798,0.0006908911,0.00156394,0.02772863,0.0174279,0.05671257],"genre_scores_gemma":[0.8577719,0.0004217702,0.07891034,0.0008390106,0.0001381881,0.001861894,0.04841071,0.0009841912,0.01066203],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.008635038,"threshold_uncertainty_score":0.03699553,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06148170898981459,"score_gpt":0.309999045427913,"score_spread":0.2485173364380984,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}