{"id":"W4415926978","doi":"10.1101/2025.11.04.686583","title":"Adaptive resampling for improved machine learning in imbalanced single-cell datasets","year":2025,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Single-cell and spatial transcriptomics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Resampling; Representation (politics); Training set; Labeled data; Feature learning; External Data Representation; Support vector machine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005701756,0.001359597,0.001212643,0.0008171545,0.0005497521,0.0009283517,0.001878198,0.001187491,0.001129192],"category_scores_gemma":[0.01536451,0.0005318497,0.001008285,0.0008973176,0.0008954224,0.001790064,0.001832904,0.002488126,0.0007572572],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007509478,"about_ca_system_score_gemma":0.0008467766,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00214715,"about_ca_topic_score_gemma":0.003338644,"domain_scores_codex":[0.9986067,0.0007001229,0.0000733064,0.0003187019,0.0002153999,0.00008575767],"domain_scores_gemma":[0.9956332,0.002344556,0.0002795698,0.001280436,0.0003395816,0.0001227676],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006054256,0.0003789989,0.007184426,0.0002497513,0.0002919808,0.000273893,0.000348216,0.6660877,0.03621059,0.01855644,0.007276462,0.2625361],"study_design_scores_gemma":[0.00001333492,0.00003805403,0.0004735965,0.000005456858,0.000007889571,0.0000235628,0.00001439951,0.9853473,0.00432034,0.009079677,0.0006661545,0.0000102637],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03260241,0.0003075585,0.963729,0.0003628462,0.00007211541,0.00007077002,0.0003518479,0.002149879,0.0003534896],"genre_scores_gemma":[0.4190793,0.0002867782,0.5751741,0.0003454515,0.0001410686,0.0004553717,0.002906151,0.0004529028,0.001158789],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005701756,"threshold_uncertainty_score":0.03015411,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01801844834119599,"score_gpt":0.2278284629522935,"score_spread":0.2098100146110975,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}