{"id":"W4225817241","doi":"10.1007/s10489-022-03944-z","title":"Reward modeling for mitigating toxicity in transformer-based language models","year":2022,"lang":"en","type":"article","venue":"Applied Intelligence","topic":"Hate Speech and Cyberbullying Detection","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Language model; Transformer; Artificial intelligence; Unintended consequences; Detoxification (alternative medicine); Machine learning; Natural language processing; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005114217,0.001008122,0.001471562,0.0008586331,0.0007639938,0.001692771,0.002224866,0.001681834,0.004073829],"category_scores_gemma":[0.03115465,0.0007214847,0.0009010251,0.0007252586,0.001070543,0.0042391,0.002212484,0.003452685,0.0008536653],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001578412,"about_ca_system_score_gemma":0.002533109,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008108677,"about_ca_topic_score_gemma":0.01020817,"domain_scores_codex":[0.9975917,0.001206249,0.0001474674,0.0003562371,0.0004290197,0.0002692346],"domain_scores_gemma":[0.9848924,0.01192749,0.0005462914,0.0007922179,0.001527349,0.00031426],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003022191,0.0001052362,0.001542947,0.0001154156,0.00007583899,0.0001330932,0.0001775276,0.8689132,0.001803647,0.07376969,0.002076894,0.05098421],"study_design_scores_gemma":[0.000006673387,0.00001964473,0.00003832023,0.000004955486,0.00001349873,0.00001251746,0.000006801135,0.9813059,0.0003696943,0.01801147,0.000206237,0.000004439392],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02246505,0.0002246379,0.9742332,0.0006062101,0.00007160091,0.000043735,0.0001475906,0.0007785481,0.001429394],"genre_scores_gemma":[0.9174244,0.0002840537,0.07699329,0.0003573203,0.00007344876,0.0001156322,0.0003351248,0.0002508073,0.004165788],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008108677,"threshold_uncertainty_score":0.02704686,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0284011211217057,"score_gpt":0.2561019994612627,"score_spread":0.227700878339557,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}