{"id":"W3158822421","doi":"10.18653/v1/2021.acl-short.24","title":"What’s in the Box? An Analysis of Undesirable Content in the Common Crawl Corpus","year":2021,"lang":"en","type":"article","venue":"","topic":"Hermeneutics and Narrative Identity","field":"Arts and Humanities","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Language model; Natural language processing; Content (measure theory); Artificial intelligence; Content analysis; Corpus linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004537796,0.0005299203,0.0009350392,0.008783558,0.003718743,0.00433246,0.001250566,0.001833921,0.003226336],"category_scores_gemma":[0.03169149,0.0007301676,0.0004971566,0.01053898,0.003008329,0.004259555,0.003364916,0.002033282,0.002659316],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002263196,"about_ca_system_score_gemma":0.003695847,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.03007147,"about_ca_topic_score_gemma":0.06741312,"domain_scores_codex":[0.9942978,0.001855147,0.0005659647,0.0008333619,0.002071009,0.0003766771],"domain_scores_gemma":[0.9613094,0.02605687,0.001839853,0.003953058,0.006313496,0.0005272928],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"observational","study_design_scores_codex":[0.00102489,0.0006969156,0.06623,0.005001058,0.0002175835,0.003539228,0.04573926,0.001744431,0.02199019,0.05249311,0.5232801,0.2780431],"study_design_scores_gemma":[0.0001587454,0.0001234475,0.1092317,0.001354748,0.0002600389,0.004790949,0.02771855,0.02403088,0.02693004,0.01859815,0.786588,0.0002146718],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7353861,0.01087186,0.07620419,0.01460522,0.001546797,0.0007220202,0.1031586,0.00713932,0.05036596],"genre_scores_gemma":[0.6844614,0.003608774,0.07792702,0.002765491,0.0005132324,0.001191122,0.2019286,0.006467606,0.02113682],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03007147,"threshold_uncertainty_score":0.05979288,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1369919535072696,"score_gpt":0.2910123372420831,"score_spread":0.1540203837348135,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}