{"id":"W6910435347","doi":"10.48448/p9w5-af56","title":"Ask LLMs Directly, “What shapes your bias?”: Measuring Social Bias in Large Language Models","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Viewpoints; Perception; Social perception; Social change; Social comparison theory; Social influence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.005162755,0.000940914,0.001001471,0.004155247,0.000328901,0.001576097,0.002134601,0.0005920681,0.001284949],"category_scores_gemma":[0.0005016291,0.0008674381,0.0002583401,0.004680743,0.001115469,0.001585345,0.001062067,0.001250498,0.009513393],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001068228,"about_ca_system_score_gemma":0.001141048,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001693945,"about_ca_topic_score_gemma":0.01525099,"domain_scores_codex":[0.9919953,0.0002622626,0.0007888834,0.002087993,0.002926646,0.001938895],"domain_scores_gemma":[0.9981838,0.0001008148,0.0004861508,0.0007339796,0.0001631622,0.0003320698],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002437714,0.00452121,0.001250235,0.003352811,0.001139823,0.00412195,0.1291906,0.005385684,0.05881757,0.05607196,0.5945814,0.141323],"study_design_scores_gemma":[0.006772702,0.0003540993,0.0005130039,0.01628485,0.0008467123,0.0002054344,0.06631648,0.5489759,0.005428254,0.02662953,0.3156043,0.01206876],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.02077957,0.02591235,0.001081752,0.0007790044,0.004161507,0.002311112,0.002027686,0.005362322,0.9375847],"genre_scores_gemma":[0.6158308,0.0004217103,0.003119314,0.0005107609,0.002820015,0.0001318896,0.0001916213,0.003804374,0.3731695],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.5950513,"threshold_uncertainty_score":0.999628,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1319699398434422,"score_gpt":0.3462782630000419,"score_spread":0.2143083231565997,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}