{"id":"W1651734266","doi":"10.48550/arxiv.1109.3532","title":"A Characterization of the Combined Effects of Overlap and Imbalance on the SVM Classifier","year":2011,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Imbalanced Data Classification Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"","keywords":"Overfitting; Support vector machine; Covert; Artificial intelligence; Computer science; Machine learning; Regularization (linguistics); Classifier (UML); Parametric statistics; Pattern recognition (psychology); Artificial neural network; Mathematics; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01154016,0.0009254541,0.001690707,0.00218644,0.001166032,0.002276052,0.001158087,0.001513505,0.001645637],"category_scores_gemma":[0.07747643,0.00045088,0.0006309644,0.001816305,0.002973804,0.005292207,0.003243616,0.002650293,0.0004565996],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001121663,"about_ca_system_score_gemma":0.0009987835,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005559463,"about_ca_topic_score_gemma":0.0005965508,"domain_scores_codex":[0.9911732,0.002550014,0.0004442265,0.001289007,0.004027584,0.0005159382],"domain_scores_gemma":[0.8992162,0.07895953,0.007637693,0.008764111,0.004206939,0.001215405],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001594981,0.0006850329,0.1045307,0.0005300757,0.0004276548,0.001093576,0.002012565,0.3500195,0.0625139,0.08482724,0.004033752,0.387731],"study_design_scores_gemma":[0.00001993987,0.0004140815,0.0232547,0.00007999282,0.000102637,0.001261047,0.0002726434,0.8918911,0.02210155,0.05753443,0.002982598,0.00008529657],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2450264,0.0008943769,0.7481425,0.001079411,0.0001011478,0.0001096303,0.000206214,0.0003487421,0.004091624],"genre_scores_gemma":[0.9367589,0.000386355,0.06082738,0.0002250236,0.000283794,0.0001119524,0.0002527086,0.0001589437,0.0009949266],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01154016,"threshold_uncertainty_score":0.06103092,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04711872809783901,"score_gpt":0.1694499975787661,"score_spread":0.1223312694809271,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}