{"id":"W4312084660","doi":"10.1016/j.dib.2022.108832","title":"A curated dataset for hate speech detection on social media text","year":2022,"lang":"en","type":"article","venue":"Data in Brief","topic":"Hate Speech and Cyberbullying Detection","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"ca_institutions":"Lakehead University","funders":"Natural Sciences and Engineering Research Council of Canada; Lakehead University","keywords":"Computer science; Slang; Social media; Natural language processing; Preprocessor; Vocabulary; Classifier (UML); Sentence; Artificial intelligence; Perplexity; Speech recognition; Information retrieval; World Wide Web; Linguistics; Language model","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001143685,0.002342935,0.001071041,0.004306301,0.001586875,0.001281407,0.00177631,0.002345179,0.009800845],"category_scores_gemma":[0.005081556,0.0003138485,0.001307687,0.00262254,0.0006429199,0.001603915,0.002066871,0.001831596,0.01772105],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001206698,"about_ca_system_score_gemma":0.001522548,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009383424,"about_ca_topic_score_gemma":0.02321478,"domain_scores_codex":[0.9974649,0.0004721286,0.0003835795,0.000608898,0.0008113369,0.0002591771],"domain_scores_gemma":[0.9966872,0.0009026955,0.000367842,0.0006510928,0.001071964,0.0003192277],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0004992682,0.0008768477,0.01040367,0.00301673,0.0001441491,0.0008693585,0.0005583881,0.00187226,0.009349769,0.001117522,0.8879263,0.08336574],"study_design_scores_gemma":[0.0003000764,0.0005429921,0.09247568,0.0008463951,0.0001803953,0.002184482,0.001634394,0.01789722,0.01547802,0.001748399,0.8664458,0.0002660802],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.03491829,0.00148494,0.005032131,0.0006012947,0.0007653724,0.0009176615,0.9434591,0.005086898,0.007734368],"genre_scores_gemma":[0.0107118,0.0002088696,0.005275788,0.0001476666,0.00009123941,0.0007556533,0.9802201,0.0001260745,0.002462726],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.009800845,"threshold_uncertainty_score":0.03278708,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05926111213507972,"score_gpt":0.2942761866840025,"score_spread":0.2350150745489228,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}