{"id":"W4390656815","doi":"10.1101/2024.01.02.572650","title":"Beyond benchmarking: towards predictive models of dataset-specific single-cell RNA-seq pipeline performance","year":2024,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Single-cell and spatial transcriptomics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute; Lunenfeld-Tanenbaum Research Institute; Ontario Institute for Cancer Research; University of Toronto","funders":"Canadian Institutes of Health Research; University of Toronto; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Pipeline (software); Benchmarking; Computer science; Normalization (sociology); Cluster analysis; Pipeline transport; Machine learning; Data mining; Artificial intelligence; Range (aeronautics); Set (abstract data type); Predictive modelling; Engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01772521,0.001989073,0.001656814,0.002376007,0.0006395665,0.003261816,0.002285498,0.001758031,0.001346685],"category_scores_gemma":[0.05212945,0.0005759156,0.00160556,0.002869522,0.001164854,0.004499588,0.001351532,0.003195293,0.001049299],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00199279,"about_ca_system_score_gemma":0.001603329,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006504562,"about_ca_topic_score_gemma":0.005806071,"domain_scores_codex":[0.9953484,0.002193464,0.0001991459,0.00117978,0.0007763403,0.0003028699],"domain_scores_gemma":[0.9645282,0.02412583,0.002490456,0.005070145,0.003167659,0.0006176759],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003699299,0.0002169059,0.03978744,0.0004086769,0.0004486556,0.00008140624,0.0001444931,0.9032196,0.003850668,0.005042313,0.007145521,0.03928433],"study_design_scores_gemma":[0.000008392125,0.00006000931,0.003910198,0.00003190866,0.00002411861,0.00001962069,0.00002810229,0.9841684,0.002227031,0.008855914,0.0006453138,0.0000209959],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3446293,0.002827043,0.622908,0.002319828,0.0002268966,0.0004130949,0.01084303,0.0109282,0.004904608],"genre_scores_gemma":[0.874963,0.000617632,0.1065354,0.0006162147,0.00008917282,0.0004758858,0.01475783,0.0009254577,0.00101938],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9822748,"threshold_uncertainty_score":0.09374094,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01940723924096349,"score_gpt":0.2092493754798289,"score_spread":0.1898421362388654,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}