{"id":"W7106482260","doi":"10.1609/aaaiss.v7i1.36931","title":"From Bias to Breakdown: Benchmarking Failure Mode Analysisof Single-cell RNA Sequencing Foundation Models in AcuteMyeloid Leukemia","year":2025,"lang":"","type":"article","venue":"Proceedings of the AAAI Symposium Series","topic":"Single-cell and spatial transcriptomics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute","funders":"Natural Sciences and Engineering Research Council of Canada; Canada First Research Excellence Fund; York University","keywords":"Benchmarking; Myeloid leukemia; Disease; Foundation (evidence); Set (abstract data type); Benchmark (surveying); Myeloid","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0004529835,0.0006944906,0.0008355847,0.0003809807,0.0003523085,0.0003936882,0.001259668,0.0006125111,0.00001713752],"category_scores_gemma":[0.0001091704,0.0006467486,0.0004671636,0.001274285,0.0002623361,0.0001728517,0.0006884945,0.0004469319,0.000003234841],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007180175,"about_ca_system_score_gemma":0.0006377005,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00213402,"about_ca_topic_score_gemma":0.0008848532,"domain_scores_codex":[0.9963147,0.00005632557,0.001264706,0.001189982,0.0004643483,0.0007099631],"domain_scores_gemma":[0.9980085,0.00004888784,0.0006349134,0.000514134,0.0006451979,0.0001483559],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.000646168,0.0001492231,0.00444045,0.0004503203,0.0002884933,8.161372e-7,0.003373204,0.01336975,0.9754072,0.0003518421,0.0004447363,0.001077833],"study_design_scores_gemma":[0.0009310177,0.0003118949,0.0001488255,0.0009156821,0.0004504553,0.000004531491,0.001910957,0.01481516,0.9761724,0.002012855,0.001682591,0.0006436709],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9843255,0.0005350815,0.001319103,0.002781835,0.0007129993,0.0007460938,0.0000868459,0.00002843853,0.009464104],"genre_scores_gemma":[0.9928812,0.0005388481,0.003954612,0.0004319579,0.0004402876,0.00005302971,0.00009132038,0.00006785272,0.001540926],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008555667,"threshold_uncertainty_score":0.9995984,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01511818684366384,"score_gpt":0.22297528231365,"score_spread":0.2078570954699861,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}