{"id":"W4403536522","doi":"10.1145/3691620.3695061","title":"What Makes a High-Quality Training Dataset for Large Language Models: A Practitioners' Perspective","year":2024,"lang":"en","type":"article","venue":"","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":15,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"Natural Science Foundation of Chongqing","keywords":"Perspective (graphical); Computer science; Training (meteorology); Quality (philosophy); Language model; Artificial intelligence; Natural language processing; Data science; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.2356959,0.0007753474,0.001653654,0.005103671,0.002989537,0.01210771,0.004156417,0.004489942,0.003083591],"category_scores_gemma":[0.5573184,0.001114253,0.001117883,0.006314619,0.005699308,0.02112105,0.006255557,0.005191939,0.00130673],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004443755,"about_ca_system_score_gemma":0.01551607,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008041882,"about_ca_topic_score_gemma":0.009401197,"domain_scores_codex":[0.7924526,0.1454156,0.01693751,0.008386473,0.03312947,0.003678406],"domain_scores_gemma":[0.330692,0.4773246,0.03297769,0.05170763,0.09641886,0.01087919],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"qualitative","study_design_scores_codex":[0.001433287,0.001560945,0.2228888,0.009849685,0.0005509255,0.001294218,0.04952516,0.01714913,0.01675557,0.07353497,0.07496366,0.5304937],"study_design_scores_gemma":[0.0007607878,0.002383128,0.1619608,0.02242651,0.0005590668,0.002637256,0.1699408,0.1249713,0.03686956,0.2304234,0.2461163,0.0009510905],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3336805,0.008628826,0.3396237,0.2888948,0.001092099,0.002448185,0.005026028,0.002570357,0.01803548],"genre_scores_gemma":[0.6000823,0.002156091,0.3829164,0.007662991,0.0004810132,0.001173153,0.004235824,0.0005062192,0.0007859267],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.2356959,"threshold_uncertainty_score":0.9425231,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3320277085142438,"score_gpt":0.5121126675199051,"score_spread":0.1800849590056613,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}