{"id":"W4285891774","doi":"10.1007/s10489-022-03944-z","title":"Reward Modeling for Mitigating Toxicity in Transformer-based Language Models","year":2022,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Language model; Transformer; Artificial intelligence; Machine learning; Natural language processing; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002902413,0.0008329425,0.0006937701,0.0005294151,0.0003732193,0.0007630261,0.00121064,0.0008079373,0.002073762],"category_scores_gemma":[0.01179011,0.0003786529,0.0006762123,0.0003491857,0.0007439414,0.00202052,0.00119274,0.001949133,0.0006879164],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001032476,"about_ca_system_score_gemma":0.001161791,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003141577,"about_ca_topic_score_gemma":0.00505238,"domain_scores_codex":[0.9988108,0.0006313092,0.00006515681,0.000232838,0.0001747544,0.00008512456],"domain_scores_gemma":[0.9944125,0.004335142,0.0003373206,0.000284047,0.0004812772,0.0001496436],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004631336,0.0002065072,0.003428918,0.0002038808,0.00007409819,0.0002129024,0.0002994216,0.7885953,0.0107044,0.02969086,0.003115108,0.1630055],"study_design_scores_gemma":[0.00001370075,0.00003447648,0.00007583073,0.000003215725,0.000008218088,0.00001578113,0.000006571354,0.9922069,0.001233699,0.006093194,0.0003034369,0.000004941536],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04424654,0.0001907786,0.9523178,0.0004405099,0.00004249793,0.00009572236,0.000117975,0.001411171,0.001137191],"genre_scores_gemma":[0.8419373,0.0001974949,0.1530647,0.0004126466,0.00005198815,0.0002576146,0.0003731128,0.0002386459,0.003466486],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003141577,"threshold_uncertainty_score":0.01534963,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08285491862021747,"score_gpt":0.1938981566713769,"score_spread":0.1110432380511594,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}