{"id":"W4399734222","doi":"10.1186/s13059-024-03304-9","title":"Beyond benchmarking and towards predictive models of dataset-specific single-cell RNA-seq pipeline performance","year":2024,"lang":"en","type":"article","venue":"Genome biology","topic":"Single-cell and spatial transcriptomics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute; Lunenfeld-Tanenbaum Research Institute; Ontario Institute for Cancer Research; University of Toronto","funders":"Canadian Institutes of Health Research; University of Toronto; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Pipeline (software); Benchmarking; Computer science; Machine learning; Cluster analysis; Normalization (sociology); Pipeline transport; Artificial intelligence; Data mining; Predictive modelling; Range (aeronautics); Set (abstract data type); Engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02181828,0.002264329,0.001733742,0.002636908,0.0006867416,0.003902432,0.002746112,0.002073645,0.001621],"category_scores_gemma":[0.0548235,0.0005411488,0.002105508,0.003353995,0.001335406,0.005495075,0.001611859,0.004087938,0.001327127],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002400204,"about_ca_system_score_gemma":0.001918328,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006359787,"about_ca_topic_score_gemma":0.00620736,"domain_scores_codex":[0.9936232,0.003087435,0.0003186217,0.001581948,0.001043229,0.0003456897],"domain_scores_gemma":[0.9575633,0.02720528,0.003384517,0.00698285,0.004052205,0.000811745],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004966382,0.0003178565,0.06787473,0.00101759,0.0008328964,0.0001110847,0.000327272,0.837975,0.00542628,0.007387035,0.0108535,0.06738024],"study_design_scores_gemma":[0.00001771277,0.0002031492,0.009305834,0.0001566879,0.00007499836,0.00005644828,0.00009345365,0.9645534,0.005652882,0.01713299,0.002694005,0.00005839486],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3705544,0.005413982,0.5855307,0.003779579,0.0003297494,0.0006774274,0.01506689,0.01068753,0.007959697],"genre_scores_gemma":[0.8240947,0.001321874,0.145663,0.001364199,0.0001316304,0.0007349629,0.02428633,0.0009791051,0.001424225],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02181828,"threshold_uncertainty_score":0.1153875,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02194880880511308,"score_gpt":0.2261055599729756,"score_spread":0.2041567511678626,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}