{"id":"W4385764438","doi":"10.24963/ijcai.2023/694","title":"Evaluating GPT-3 Generated Explanations for Hateful Content Moderation","year":2023,"lang":"en","type":"article","venue":"","topic":"Hate Speech and Cyberbullying Detection","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Moderation; Fluency; Content (measure theory); Social psychology; Psychology; Quality (philosophy); Cognitive psychology; Computer science; Epistemology; Mathematics education","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004450801,0.0000704987,0.00006801778,0.0001243056,0.0002920762,0.0002071309,0.0001675005,0.00003834875,0.00001587696],"category_scores_gemma":[0.0001072896,0.00006352139,0.00003989103,0.0004284339,0.000005862446,0.0003384012,0.00003769912,0.00003765802,0.0001648442],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003103763,"about_ca_system_score_gemma":0.00004811791,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003291228,"about_ca_topic_score_gemma":0.00003076022,"domain_scores_codex":[0.9991807,0.00003624642,0.000162427,0.000241255,0.0001865355,0.0001927954],"domain_scores_gemma":[0.9994411,0.00006777058,0.00004013184,0.0001948764,0.000209478,0.00004668209],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002330708,0.00006542598,0.00005948972,0.00002417192,0.00006123714,0.000006029184,0.00121323,0.0660968,0.3532695,0.1734331,0.01866744,0.3870802],"study_design_scores_gemma":[0.0002962757,0.00008894611,0.0001897957,0.000003983935,0.000003122322,0.000002703468,0.00004923983,0.9462652,0.04974899,0.002102314,0.001161804,0.00008763759],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0998128,0.000008014304,0.8966269,0.001455832,0.0004345776,0.000314709,0.000002481846,0.0006937858,0.0006508653],"genre_scores_gemma":[0.7999741,0.00001091683,0.1891316,0.0004851691,0.000161194,0.0003637263,0.00009431528,0.00001461196,0.009764348],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8801684,"threshold_uncertainty_score":0.2590327,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2562098803625319,"score_gpt":0.3650106648018543,"score_spread":0.1088007844393225,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}