{"id":"W6929147930","doi":"10.48448/53rb-h963","title":"Self-Influence Guided Data Reweighting for Language Model Pre-training","year":2023,"lang":"en","type":"other","venue":"Open MIND","topic":"Cellular transport and secretion","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Context (archaeology); Language model; Relevance (law); Novelty; Sample (material); Point (geometry); Stability (learning theory); Data modeling","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005000275,0.001865968,0.001469035,0.001150802,0.0008369592,0.002049969,0.002790121,0.001974667,0.003913573],"category_scores_gemma":[0.02337576,0.000870708,0.001377901,0.0008855011,0.001422696,0.003476132,0.00363856,0.004939605,0.002901629],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008130016,"about_ca_system_score_gemma":0.00148574,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002313325,"about_ca_topic_score_gemma":0.005977918,"domain_scores_codex":[0.9975315,0.0009824244,0.0001759902,0.0006598104,0.000473178,0.0001770011],"domain_scores_gemma":[0.9918521,0.004617016,0.0003998372,0.001619782,0.001199855,0.0003112946],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0008989172,0.0005605819,0.006301599,0.0005270428,0.0004622395,0.0003164401,0.000728175,0.2935569,0.03342222,0.01552426,0.01317255,0.6345291],"study_design_scores_gemma":[0.00003832382,0.0001089243,0.0005191836,0.0000300911,0.00003358662,0.00006971573,0.00005006746,0.9741934,0.01291149,0.009204865,0.00281644,0.00002395827],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.03193698,0.0008166753,0.9588642,0.0004261906,0.0001957467,0.0001180393,0.0002528748,0.005660498,0.001728788],"genre_scores_gemma":[0.4510813,0.0004654755,0.5337826,0.0007647813,0.0003187051,0.0004887799,0.002422143,0.002644485,0.008031701],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005000275,"threshold_uncertainty_score":0.02644432,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06899219198716,"score_gpt":0.3465851193036998,"score_spread":0.2775929273165398,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}