{"id":"W4411969915","doi":"10.1101/2025.06.26.661884","title":"Novel binning-based methods for model fitting and data splitting improved machine learning imbalanced data","year":2025,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Imbalanced Data Classification Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Machine learning; Artificial intelligence; Data mining; Algorithm","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication","open_science"],"consensus_categories":["open_science"],"category_scores_codex":[0.006040203,0.0007870487,0.0009137492,0.0005038694,0.0005185053,0.001061141,0.009206825,0.0005822407,0.00000135353],"category_scores_gemma":[0.00437558,0.0008858205,0.00008719321,0.0008040873,0.0001373842,0.001215986,0.01444646,0.001401594,0.000001328729],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002081408,"about_ca_system_score_gemma":0.001478616,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00007467616,"about_ca_topic_score_gemma":0.00000246686,"domain_scores_codex":[0.9936645,0.000289935,0.001088303,0.003785045,0.0003376326,0.0008345626],"domain_scores_gemma":[0.9864429,0.0008913253,0.001251236,0.01053389,0.0006445989,0.0002360382],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003448543,0.000158896,0.0008991056,0.001600051,0.0001778065,0.00000273929,0.0000153767,0.001386224,0.98962,0.004880301,0.0003307691,0.0008942171],"study_design_scores_gemma":[0.0006693575,0.00002677512,0.0005665888,0.0004807717,0.00009480515,1.872988e-8,0.000001032903,0.9205692,0.07345794,0.00001677472,0.003307925,0.0008087425],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0004242393,0.000949869,0.9897088,0.0007220069,0.0005111907,0.001423204,0.004335215,0.001920527,0.000004948417],"genre_scores_gemma":[0.05810362,0.0001038166,0.9406679,0.0004534949,0.0001557411,0.0003274221,0.00007593944,0.0001001796,0.00001184745],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.919183,"threshold_uncertainty_score":0.9999759,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06597440103141576,"score_gpt":0.3401000506503915,"score_spread":0.2741256496189757,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}