{"id":"W2987134874","doi":"10.48550/arxiv.1911.01217","title":"Detect Toxic Content to Improve Online Conversations","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Sentiment Analysis and Opinion Mining","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"Python (programming language); Computer science; Naive Bayes classifier; Artificial intelligence; Support vector machine; Social media; Machine learning; Resampling; Natural language processing; Information retrieval; Deep learning; Content (measure theory); World Wide Web; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0001807525,0.0002638751,0.000360019,0.0003913935,0.0001047601,0.0001588005,0.001536423,0.0001701368,0.00007863601],"category_scores_gemma":[0.00003469155,0.0002942614,0.0003148276,0.0005746515,0.00003249836,0.0002408045,0.001924855,0.0003320249,0.0004503132],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002022247,"about_ca_system_score_gemma":0.0001596624,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001663153,"about_ca_topic_score_gemma":0.00006467483,"domain_scores_codex":[0.9981682,0.00008028686,0.0002350831,0.001088408,0.0001186206,0.0003093465],"domain_scores_gemma":[0.9979777,0.00009064405,0.0002327028,0.001302701,0.0001985777,0.0001976928],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001025894,0.0006181056,0.02038711,0.0001973884,0.001652449,0.0003232286,0.001972278,0.7971586,0.005091592,0.1559773,0.004924785,0.01159454],"study_design_scores_gemma":[0.0007155752,0.0001321226,0.003784401,0.00009677914,0.0001390377,0.000001172524,0.0003078972,0.9888219,0.001205382,0.002065763,0.002077051,0.0006528813],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3331456,0.00003839822,0.6641896,0.0003357512,0.001177465,0.0003390386,0.00002852218,0.000134611,0.0006109409],"genre_scores_gemma":[0.9900936,0.00005437355,0.004228837,0.000406866,0.00008848018,8.912185e-7,0.00003865101,0.00001274137,0.00507551],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6599608,"threshold_uncertainty_score":0.9999509,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1210382773475492,"score_gpt":0.2110812787755628,"score_spread":0.0900430014280136,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}