{"id":"W7138308848","doi":"10.18653/v1/2025.sciprodllm-1.2","title":"Human-Centered Disability Bias Detection in Large Language Models","year":2025,"lang":"","type":"article","venue":"","topic":"Text Readability and Simplification","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Language model; Noise (video); Feature (linguistics); Data collection; Natural language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02547047,0.0006349149,0.0005380025,0.001357864,0.0008997073,0.003234559,0.001077282,0.00110605,0.002479423],"category_scores_gemma":[0.1752128,0.0004376716,0.0006653938,0.0007633196,0.002135849,0.003826166,0.004867896,0.001226441,0.0007122203],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008984738,"about_ca_system_score_gemma":0.00178537,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002287998,"about_ca_topic_score_gemma":0.002980995,"domain_scores_codex":[0.9775665,0.01540157,0.001196356,0.002584504,0.00280045,0.0004505588],"domain_scores_gemma":[0.8417817,0.1290603,0.01034142,0.01236585,0.00544556,0.001005207],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.003300667,0.0004977014,0.2955336,0.001975718,0.0006823368,0.00180473,0.03915546,0.04178284,0.06028468,0.08167592,0.008036236,0.4652701],"study_design_scores_gemma":[0.0002189458,0.001287054,0.1373791,0.0008872511,0.0003757017,0.003088913,0.01202797,0.4776995,0.06533598,0.2709327,0.03038464,0.0003822373],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5503397,0.0005273829,0.4386405,0.001254971,0.00009700411,0.0005202174,0.001099393,0.001256428,0.006264451],"genre_scores_gemma":[0.9384413,0.00007608945,0.05969016,0.000230979,0.0000266626,0.00028611,0.0005791842,0.0001363472,0.000533225],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02547047,"threshold_uncertainty_score":0.1347023,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05454424794295367,"score_gpt":0.3174023805049431,"score_spread":0.2628581325619895,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}