{"id":"W4316469706","doi":"10.3390/info14010054","title":"A Comparison of Undersampling, Oversampling, and SMOTE Methods for Dealing with Imbalanced Classification in Educational Data Mining","year":2023,"lang":"en","type":"article","venue":"Information","topic":"Imbalanced Data Classification Techniques","field":"Computer Science","cited_by":340,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Undersampling; Oversampling; Resampling; Random forest; Computer science; Machine learning; Class (philosophy); Data mining; Sampling (signal processing); Artificial intelligence; Bandwidth (computing)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01416285,0.001047933,0.001459435,0.002574177,0.0009016976,0.0009602769,0.001322771,0.001087868,0.0004298837],"category_scores_gemma":[0.0271285,0.0003856423,0.001408612,0.001761205,0.0007959108,0.002060934,0.001391945,0.001573451,0.0001719491],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00071674,"about_ca_system_score_gemma":0.001730406,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005630831,"about_ca_topic_score_gemma":0.007916315,"domain_scores_codex":[0.9936543,0.003645141,0.0004066482,0.0005727663,0.00146936,0.0002517663],"domain_scores_gemma":[0.9855759,0.00946067,0.000826259,0.001432977,0.002404564,0.0002995941],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001650491,0.0005713325,0.02512867,0.0005369937,0.0008484118,0.0001170066,0.0007477621,0.2507116,0.006171947,0.01029661,0.004046755,0.6991723],"study_design_scores_gemma":[0.00006209952,0.0003193576,0.004182845,0.00005856554,0.00008130027,0.00008264864,0.0001492332,0.9869021,0.002897478,0.003499737,0.001736496,0.00002801475],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1171351,0.004400431,0.8744875,0.0006418643,0.0003194069,0.0004015696,0.0001989723,0.001042152,0.001373016],"genre_scores_gemma":[0.4325219,0.001740768,0.5631222,0.0003323987,0.0002867591,0.0003941434,0.0007787157,0.0001116781,0.0007114488],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01416285,"threshold_uncertainty_score":0.07490122,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1915929103132564,"score_gpt":0.4664089199150849,"score_spread":0.2748160096018285,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}