{"id":"W4391125455","doi":"10.48550/arxiv.2401.10359","title":"Keeping Deep Learning Models in Check: A History-Based Approach to Mitigate Overfitting","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Software Engineering Research","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Huawei Technologies (Canada); University of Alberta","funders":"","keywords":"Overfitting; Computer science; Machine learning; Artificial intelligence; Classifier (UML); Early stopping; Deep learning; Artificial neural network","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005307241,0.002446112,0.002076081,0.002103285,0.00106427,0.001909822,0.003806468,0.002298218,0.00204635],"category_scores_gemma":[0.02238231,0.001233603,0.001466592,0.001062711,0.00129387,0.00387464,0.002807728,0.004760059,0.0009410949],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001483568,"about_ca_system_score_gemma":0.002426811,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01025168,"about_ca_topic_score_gemma":0.01403163,"domain_scores_codex":[0.9973652,0.0004816803,0.0002518298,0.0007103626,0.0008971967,0.0002937525],"domain_scores_gemma":[0.9888766,0.004853991,0.001699415,0.001990091,0.002110063,0.0004697315],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004924881,0.0006253878,0.02516915,0.0002353319,0.0003903384,0.0005561516,0.0005019318,0.4686149,0.0111803,0.003442701,0.008159235,0.4806321],"study_design_scores_gemma":[0.00001216348,0.0001399273,0.001251512,0.00004103175,0.00006103878,0.00008714507,0.00003027471,0.9903172,0.004314002,0.002702127,0.001017584,0.00002599659],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1140908,0.001646292,0.8707483,0.001410566,0.0002104809,0.0001755015,0.0002933446,0.008693683,0.002731019],"genre_scores_gemma":[0.8569569,0.0004869051,0.1350336,0.0010887,0.0001662587,0.0001836914,0.0008881572,0.0006816443,0.004514176],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01025168,"threshold_uncertainty_score":0.02806771,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09718714126425883,"score_gpt":0.1949020764319976,"score_spread":0.09771493516773876,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}