{"id":"W4415230910","doi":"10.21203/rs.3.rs-7623811/v1","title":"Responsible AI in NLP: GUS-Net Span-Level Bias Detection Dataset and Benchmark for Generalizations, Unfairness, and Stereotypes","year":2025,"lang":"en","type":"preprint","venue":"Research Square","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Benchmark (surveying); Pipeline (software); Language model; Limiting; Textual entailment; Identification (biology); Security token; Framing (construction)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.003515007,0.0002627569,0.0003371504,0.001307219,0.0004359698,0.0009437887,0.0009935704,0.0003176015,0.00001620901],"category_scores_gemma":[0.001707138,0.0002729897,0.00004385432,0.001206371,0.000185172,0.0005753869,0.002995135,0.000783699,0.000008188515],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00021978,"about_ca_system_score_gemma":0.0008889626,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002906126,"about_ca_topic_score_gemma":0.009761413,"domain_scores_codex":[0.9962367,0.0008607895,0.0004713564,0.001163213,0.0005981487,0.000669744],"domain_scores_gemma":[0.9967217,0.001216335,0.00009605898,0.001183929,0.000623249,0.0001586797],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001626479,0.001124274,0.02692261,0.01458969,0.0003282142,0.0002634883,0.007993824,0.02600302,0.006204152,0.336396,0.2109767,0.3675716],"study_design_scores_gemma":[0.0007892519,0.0007425465,0.01537479,0.002346006,0.0000294226,0.00002462055,0.0009976493,0.5661052,0.03652676,0.2401459,0.1356473,0.001270492],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05380762,0.004399191,0.9214541,0.007747034,0.0007778135,0.00544562,0.005664467,0.0002049148,0.0004992503],"genre_scores_gemma":[0.9191444,0.007605615,0.05733796,0.00066254,0.0005732463,0.002796135,0.005850253,0.0001023942,0.005927446],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8653368,"threshold_uncertainty_score":0.9999722,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1935936064666533,"score_gpt":0.439637199761039,"score_spread":0.2460435932943857,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}