{"id":"W2947548407","doi":"10.48550/arxiv.1905.12787","title":"The Theory Behind Overfitting, Cross Validation, Regularization, Bagging, and Boosting: Tutorial","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Domain Adaptation and Few-Shot Learning","field":"Computer Science","cited_by":73,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Overfitting; Boosting (machine learning); Early stopping; Artificial intelligence; AdaBoost; Gradient boosting; Machine learning; Estimator; Mathematics; Support vector machine; Cross-validation; Generalization error; Computer science; Ensemble learning; Regularization (linguistics); Bias of an estimator; Random forest; Algorithm; Statistics; Minimum-variance unbiased estimator; Artificial neural network","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0104893,0.003736787,0.003598033,0.004325037,0.0006934112,0.004059912,0.002755447,0.003659062,0.005269329],"category_scores_gemma":[0.01665004,0.00177885,0.002956004,0.008058252,0.00278402,0.005432402,0.002235069,0.007405804,0.006035467],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001818196,"about_ca_system_score_gemma":0.001431469,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00216569,"about_ca_topic_score_gemma":0.001261825,"domain_scores_codex":[0.9925261,0.003459662,0.0005589119,0.0009747764,0.002274714,0.0002058769],"domain_scores_gemma":[0.9919227,0.006304325,0.0003540567,0.0004712312,0.0008354908,0.0001121409],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00007640841,0.0001578196,0.001382397,0.002695357,0.0004015517,0.0002277799,0.0004188799,0.05458426,0.001402085,0.3087547,0.07594507,0.5539536],"study_design_scores_gemma":[0.0000308957,0.000236726,0.001614674,0.001221566,0.000145696,0.0009933362,0.00009016324,0.1161834,0.001467977,0.6014364,0.2764031,0.0001760678],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0008382157,0.135391,0.8504137,0.001768544,0.0023307,0.0001083767,0.0002251091,0.001006702,0.007917679],"genre_scores_gemma":[0.03445844,0.2460238,0.6847748,0.005147169,0.01416428,0.001255404,0.001436727,0.001260307,0.0114791],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.0104893,"threshold_uncertainty_score":0.05547339,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04054486057077095,"score_gpt":0.2000192836718457,"score_spread":0.1594744231010747,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}