{"id":"W4404573785","doi":"10.48550/arxiv.2411.12372","title":"RedPajama: an Open Dataset for Training Large Language Models","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Oak Ridge National Laboratory; Office of Naval Research; Canadian Institute for Advanced Research; U.S. Army Combat Capabilities Development Command; National Institutes of Health; National Science Foundation; VMware; Office of Science; Accenture; Canada Excellence Research Chairs, Government of Canada; U.S. Department of Energy","keywords":"Training (meteorology); Computer science; Natural language processing; Artificial intelligence; Geography","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003570548,0.001907575,0.0008241982,0.002241476,0.001570004,0.002179027,0.004934833,0.002627277,0.01291813],"category_scores_gemma":[0.02229944,0.0008861666,0.002163756,0.002337628,0.001255513,0.004519192,0.003686334,0.004599643,0.01930777],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001352326,"about_ca_system_score_gemma":0.00313902,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02195009,"about_ca_topic_score_gemma":0.04430265,"domain_scores_codex":[0.9964243,0.001295054,0.0003461096,0.0008892807,0.0008054103,0.0002398158],"domain_scores_gemma":[0.9879463,0.005257592,0.0005379806,0.004072009,0.001660902,0.0005251275],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0008991411,0.0006749732,0.009797354,0.001613829,0.0003326013,0.0004976931,0.0008223794,0.01169925,0.005779133,0.006435221,0.8686353,0.0928131],"study_design_scores_gemma":[0.0008287736,0.0004805646,0.01755452,0.0004689,0.0001829033,0.0008230607,0.0008518131,0.09931479,0.01489313,0.01599788,0.8482597,0.0003439477],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.05065367,0.002084165,0.05960014,0.003037544,0.001337849,0.0009763382,0.7735432,0.09344415,0.01532298],"genre_scores_gemma":[0.03353515,0.0002674272,0.04875783,0.00065895,0.00009144544,0.001317897,0.908457,0.002642558,0.004271782],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.02195009,"threshold_uncertainty_score":0.04364467,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.145504564827458,"score_gpt":0.2749186405515542,"score_spread":0.1294140757240962,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}