{"id":"W4385764438","doi":"10.24963/ijcai.2023/694","title":"Evaluating GPT-3 Generated Explanations for Hateful Content Moderation","year":2023,"lang":"en","type":"article","venue":"","topic":"Hate Speech and Cyberbullying Detection","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Moderation; Fluency; Content (measure theory); Social psychology; Psychology; Quality (philosophy); Cognitive psychology; Computer science; Epistemology; Mathematics education","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01817249,0.001296556,0.0004132708,0.00148567,0.0005717894,0.0020384,0.001406372,0.001667686,0.006666231],"category_scores_gemma":[0.1306393,0.0003991041,0.0008990986,0.0008477023,0.0008936189,0.002304303,0.002368722,0.0017562,0.001616083],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001607919,"about_ca_system_score_gemma":0.001594047,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001441375,"about_ca_topic_score_gemma":0.002578521,"domain_scores_codex":[0.9877061,0.008961231,0.0005976242,0.0007232933,0.001750095,0.000261542],"domain_scores_gemma":[0.7751828,0.1968996,0.007242233,0.01065721,0.008892142,0.001125997],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.004197411,0.003091425,0.1381902,0.007634401,0.0004762124,0.002645302,0.07790876,0.06960807,0.04643127,0.02204536,0.04362992,0.5841417],"study_design_scores_gemma":[0.001969346,0.005215235,0.06850256,0.001927442,0.0007818418,0.001856944,0.01967491,0.6725488,0.06062106,0.03453072,0.1317579,0.0006133108],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7972839,0.0004851399,0.1578303,0.001387956,0.000217392,0.003041697,0.004136382,0.02114923,0.01446803],"genre_scores_gemma":[0.7868285,0.0002257523,0.1984517,0.0005094633,0.00006473632,0.002152838,0.006253981,0.001467237,0.004045784],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01817249,"threshold_uncertainty_score":0.09610647,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2562098803625319,"score_gpt":0.3650106648018543,"score_spread":0.1088007844393225,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}