{"id":"W4389519618","doi":"10.18653/v1/2023.emnlp-main.125","title":"Self-Influence Guided Data Reweighting for Language Model Pre-training","year":2023,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canadian Institute for Advanced Research; Polytechnique Montréal; Mila - Quebec Artificial Intelligence Institute","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Computer science; Novelty; Artificial intelligence; Language model; Context (archaeology); Task (project management); Machine learning; Relevance (law); Sample (material); Training set; Stability (learning theory); Point (geometry); Data modeling; Natural language processing","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006127829,0.001949281,0.001876634,0.001352855,0.0009531875,0.002074525,0.002972324,0.002190339,0.002901145],"category_scores_gemma":[0.03058355,0.0009701553,0.001497852,0.0009914034,0.001596028,0.003821406,0.003779896,0.005338696,0.002148709],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008089122,"about_ca_system_score_gemma":0.001660164,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002149827,"about_ca_topic_score_gemma":0.006009087,"domain_scores_codex":[0.9968584,0.001284303,0.0002329469,0.0007846638,0.0006217459,0.0002179599],"domain_scores_gemma":[0.9886494,0.006687737,0.0005424467,0.002151506,0.001566538,0.0004023206],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001071188,0.000711013,0.008740951,0.0006281666,0.0005040221,0.000287197,0.001066731,0.2946725,0.03723402,0.01378646,0.01171833,0.6295794],"study_design_scores_gemma":[0.0000581315,0.0001861929,0.0007459244,0.00004204842,0.00005190687,0.0000796713,0.00007641927,0.9735874,0.01320132,0.009242835,0.002700081,0.00002809796],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03589287,0.0007931038,0.9571903,0.0003500328,0.0001794285,0.0001602072,0.0001892424,0.004072092,0.001172811],"genre_scores_gemma":[0.478911,0.0004214509,0.5099714,0.0006677943,0.0003767147,0.0006742972,0.002082002,0.002090601,0.004804787],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006127829,"threshold_uncertainty_score":0.03240746,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1264040835937797,"score_gpt":0.3563720179076763,"score_spread":0.2299679343138966,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}