{"id":"W4387561465","doi":"10.48550/arxiv.2310.06786","title":"OpenWebMath: An Open Dataset of High-Quality Mathematical Web Text","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Mathematics, Computing, and Information Processing","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Fujitsu; Natural Sciences and Engineering Research Council of Canada; Government of Canada; Canadian Institute for Advanced Research; Brigham Young University","keywords":"Computer science; Language model; Quality (philosophy); Preprocessor; Information retrieval; Code (set theory); Natural language processing; Artificial intelligence; World Wide Web; Programming language","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["open_science"],"consensus_categories":[],"category_scores_codex":[0.0008908983,0.001710715,0.0006979623,0.006517996,0.001041218,0.001734503,0.002103139,0.002074124,0.01487336],"category_scores_gemma":[0.007891827,0.0004722159,0.0009675688,0.005457261,0.0009347805,0.003193618,0.002012233,0.001829966,0.02322401],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001283904,"about_ca_system_score_gemma":0.001725193,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007967979,"about_ca_topic_score_gemma":0.01781748,"domain_scores_codex":[0.9986688,0.0002174605,0.0002276748,0.0003066959,0.0004765154,0.0001028928],"domain_scores_gemma":[0.9961592,0.001076151,0.0004473322,0.001096258,0.0009174583,0.0003035149],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0004988776,0.0005504967,0.007645066,0.001894728,0.00009835997,0.0007632212,0.0003914832,0.00314541,0.006584108,0.005529671,0.9103261,0.06257252],"study_design_scores_gemma":[0.000702308,0.0002007982,0.03179333,0.0003803734,0.00007527519,0.00141985,0.0008215536,0.03433139,0.02180095,0.01640831,0.891852,0.0002138617],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.03785021,0.001052679,0.01232722,0.0008059056,0.0003380675,0.0003473962,0.9020308,0.03518967,0.01005801],"genre_scores_gemma":[0.01665489,0.0001712221,0.01207846,0.0001452922,0.00004833237,0.0003064563,0.9672199,0.0008679859,0.002507524],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.9978969,"threshold_uncertainty_score":0.04975635,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2422326316852966,"score_gpt":0.2815237572692198,"score_spread":0.03929112558392323,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}