{"id":"W4365441057","doi":"10.48550/arxiv.2304.04858","title":"Simulated Annealing in Early Layers Leads to Better Generalization","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Domain Adaptation and Few-Shot Learning","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Compute Canada","keywords":"Initialization; Computer science; Margin (machine learning); Artificial intelligence; Forgetting; Benchmark (surveying); Gradient descent; Transfer of learning; Machine learning; Generalization; Artificial neural network; Mathematics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001961178,0.00174522,0.001729857,0.0007806713,0.0007563395,0.001181926,0.001831484,0.001875136,0.003816013],"category_scores_gemma":[0.009073368,0.0008093497,0.001344521,0.0004628001,0.001069519,0.003388115,0.001455395,0.003088762,0.00123148],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001103034,"about_ca_system_score_gemma":0.001310128,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006103116,"about_ca_topic_score_gemma":0.01003104,"domain_scores_codex":[0.9992427,0.0001891558,0.00005153008,0.0002910374,0.0001202643,0.0001052685],"domain_scores_gemma":[0.9970714,0.001446002,0.0001742795,0.0008450658,0.0003498352,0.0001135078],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002357767,0.0002096388,0.002989829,0.0001877055,0.0001685896,0.0001364567,0.0002072035,0.8118232,0.01289576,0.007977076,0.004532788,0.1586359],"study_design_scores_gemma":[0.00000983446,0.00003990804,0.0002348754,0.00001073068,0.00001218772,0.00002569734,0.00001132378,0.9907857,0.003207674,0.005080847,0.0005727832,0.000008369864],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09777579,0.001157018,0.8899384,0.0007800575,0.0001907338,0.0001129078,0.0001300506,0.005017761,0.004897334],"genre_scores_gemma":[0.7795689,0.0004536864,0.2134202,0.0006270451,0.00009156535,0.0001814823,0.0004790528,0.0006887747,0.004489414],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006103116,"threshold_uncertainty_score":0.01276582,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1059978539784337,"score_gpt":0.2160249857094276,"score_spread":0.1100271317309939,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}