{"id":"W4303672047","doi":"10.1101/2022.10.06.511156","title":"The differential impacts of dataset imbalance in single-cell data integration","year":2022,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Single-cell and spatial transcriptomics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"Lunenfeld-Tanenbaum Research Institute; Vector Institute; University of Toronto; University Health Network","funders":"","keywords":"Cluster analysis; Data integration; Benchmarking; Computer science; Pipeline (software); Data mining; Sample (material); Annotation; Data type; Sample size determination; Artificial intelligence; Statistics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0005375621,0.0003720788,0.0003510475,0.00008825703,0.0001366763,0.0001191376,0.00161405,0.0003254679,0.00002302296],"category_scores_gemma":[0.0002028,0.000331731,0.00008663047,0.0001932334,0.0001424346,0.0000130387,0.001423646,0.0005856198,0.000002502592],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008061242,"about_ca_system_score_gemma":0.0003828876,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002050993,"about_ca_topic_score_gemma":0.0001261985,"domain_scores_codex":[0.9976826,0.0002062543,0.0005635392,0.0008704678,0.0002890101,0.000388172],"domain_scores_gemma":[0.9967862,0.0000412179,0.0004000054,0.002554491,0.0001173951,0.0001007179],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0001402528,0.0002970023,0.006941113,0.0001205547,0.00005145788,0.000006145287,0.000004661615,0.00003547469,0.9908749,0.00002280639,0.001500502,0.000005130266],"study_design_scores_gemma":[0.0008405093,0.0001891716,0.03129037,0.0001139943,0.00007255286,1.779096e-8,0.000008261855,0.0006395705,0.9456432,0.000001217905,0.02065364,0.0005475308],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9870229,0.00193658,0.001299221,0.00007813382,0.001130494,0.0004474381,0.008053707,0.00002361448,0.000007852361],"genre_scores_gemma":[0.9976803,0.001006263,0.0005778761,0.00006927369,0.0002791362,0.00004898413,0.000264473,0.00006941274,0.000004293197],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.04523173,"threshold_uncertainty_score":0.9999135,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02242270811863101,"score_gpt":0.2370273541591535,"score_spread":0.2146046460405225,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}