{"id":"W6948191620","doi":"10.48448/ztc5-5r72","title":"What's in the Box? An Analysis of Undesirable Content in the Common Crawl Corpus","year":2021,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute; Université de Montréal","funders":"","keywords":"Language model; Corpus linguistics; Computational linguistics; Content (measure theory); Language identification; Training set; Artificial neural network","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.007553264,0.0004320751,0.0009973136,0.003059336,0.000170184,0.0007869885,0.003930275,0.0002249273,0.0008302067],"category_scores_gemma":[0.0002816461,0.0002586184,0.0001939518,0.01333447,0.002314249,0.0004985413,0.0002431494,0.0006996171,0.00005875644],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003891816,"about_ca_system_score_gemma":0.0007363108,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.02410007,"about_ca_topic_score_gemma":0.309978,"domain_scores_codex":[0.9940888,0.001171137,0.0008139383,0.0009651243,0.002206094,0.00075491],"domain_scores_gemma":[0.99615,0.0004931339,0.0007247115,0.002324765,0.0002083594,0.00009907249],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"qualitative","study_design_scores_codex":[0.0005877034,0.03077516,0.3546836,0.000994246,0.004687088,0.002655701,0.08776893,0.06772983,0.05884776,0.1989753,0.1429491,0.04934551],"study_design_scores_gemma":[0.005602275,0.001530119,0.245012,0.003879349,0.004663646,0.0001362276,0.3438476,0.2579861,0.001462482,0.005862323,0.1258692,0.004148523],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.3318876,0.02356387,0.00119581,0.008145703,0.003450848,0.008713554,0.001142696,0.0005311939,0.6213687],"genre_scores_gemma":[0.9780858,0.000380149,0.0007817752,0.001817401,0.0001071969,0.00006623035,0.0004538771,0.0002777119,0.01802983],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6461982,"threshold_uncertainty_score":0.9999866,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1021472914619384,"score_gpt":0.3475380771891824,"score_spread":0.245390785727244,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}