{"id":"W4206166506","doi":"10.1109/smc52423.2021.9659134","title":"Semi-Supervised Self-Learning for Arabic Hate Speech Detection","year":2021,"lang":"en","type":"article","venue":"2021 IEEE International Conference on Systems, Man, and Cybernetics (SMC)","topic":"Hate Speech and Cyberbullying Detection","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Regina","funders":"","keywords":"Computer science; Arabic; Classifier (UML); Voice activity detection; Artificial intelligence; Speech recognition; Supervised learning; Natural language processing; Social media; Training set; Machine learning; Speech processing; Linguistics; World Wide Web; Artificial neural network","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0004245438,0.0002827991,0.0003004254,0.0001908587,0.0002575686,0.001003083,0.0004898388,0.0001848718,0.00007465286],"category_scores_gemma":[0.00009307693,0.0002934863,0.0001156428,0.0002492132,0.00003471616,0.0003250108,0.0001090191,0.0003277047,0.0001436461],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001262909,"about_ca_system_score_gemma":0.0001324685,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001150931,"about_ca_topic_score_gemma":0.0000993127,"domain_scores_codex":[0.9976382,0.0001801398,0.0004635697,0.0007675195,0.0005876073,0.0003629193],"domain_scores_gemma":[0.9982495,0.0001397088,0.0002125442,0.0003800972,0.0008615363,0.0001566001],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003519038,0.0009001385,0.001803583,0.0008248411,0.001317587,0.0006266921,0.003841932,0.005252214,0.2259036,0.3351566,0.002853401,0.4211676],"study_design_scores_gemma":[0.001699577,0.0005424239,0.0006564664,0.0004329349,0.00005942623,0.0004481954,0.0008246992,0.829215,0.08610142,0.002941628,0.07624233,0.0008358281],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4420026,0.0006293779,0.4959843,0.00189901,0.01157396,0.001185138,0.00003988836,0.000655031,0.04603074],"genre_scores_gemma":[0.9852132,0.0005332767,0.002674125,0.0001598793,0.0005428848,0.00009569577,0.00003056107,0.00002703052,0.01072336],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8239629,"threshold_uncertainty_score":0.9999517,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03250911837304178,"score_gpt":0.2649787246159093,"score_spread":0.2324696062428675,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}