{"id":"W7106482260","doi":"10.1609/aaaiss.v7i1.36931","title":"From Bias to Breakdown: Benchmarking Failure Mode Analysisof Single-cell RNA Sequencing Foundation Models in AcuteMyeloid Leukemia","year":2025,"lang":"","type":"article","venue":"Proceedings of the AAAI Symposium Series","topic":"Single-cell and spatial transcriptomics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute","funders":"Natural Sciences and Engineering Research Council of Canada; Canada First Research Excellence Fund; York University","keywords":"Benchmarking; Myeloid leukemia; Disease; Foundation (evidence); Set (abstract data type); Benchmark (surveying); Myeloid","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004929333,0.001511155,0.0009217976,0.0007519037,0.0004142401,0.0007916224,0.001180376,0.001259372,0.00135942],"category_scores_gemma":[0.009020041,0.0004226664,0.001216236,0.0004325832,0.0005406337,0.0008006995,0.001022514,0.001301283,0.0007797175],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009740936,"about_ca_system_score_gemma":0.001333181,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009151375,"about_ca_topic_score_gemma":0.009663556,"domain_scores_codex":[0.9989707,0.0004234363,0.00005769469,0.0002665365,0.0001611281,0.0001205228],"domain_scores_gemma":[0.9958895,0.002803022,0.0001867823,0.000469938,0.0004490216,0.0002018894],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007946662,0.0002023899,0.03487606,0.0003196026,0.000389304,0.0001648664,0.0001882057,0.8449693,0.007549832,0.001063201,0.007786679,0.1016959],"study_design_scores_gemma":[0.00002361872,0.0001588338,0.003872109,0.00001639906,0.00002096931,0.00003975351,0.00002909999,0.9902235,0.00331673,0.00162157,0.0006602653,0.00001728127],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8412024,0.002109739,0.1366725,0.0007867641,0.0002822727,0.0001809488,0.003495301,0.01281465,0.002455432],"genre_scores_gemma":[0.9528974,0.0003075523,0.03646249,0.0003094007,0.0000596359,0.0001599395,0.007857771,0.0005229791,0.001422835],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.009151375,"threshold_uncertainty_score":0.0260691,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01511818684366384,"score_gpt":0.22297528231365,"score_spread":0.2078570954699861,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}