{"id":"W4392190154","doi":"10.1101/2024.02.21.581497","title":"A highly scalable approach to topic modelling in single-cell data by approximate pseudobulk projection","year":2024,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Single-cell and spatial transcriptomics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Scalability; Preprocessor; Probabilistic logic; Projection (relational algebra); Representation (politics); Data mining; Theoretical computer science; Artificial intelligence; Algorithm; Database","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0005979181,0.0006356773,0.0005190847,0.0002628121,0.00009436333,0.0003456384,0.001091069,0.0008316926,0.000003111451],"category_scores_gemma":[0.00004557647,0.0006966233,0.0001205361,0.0004885794,0.00006509422,0.0000171896,0.001348149,0.00079524,0.00002487176],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001703887,"about_ca_system_score_gemma":0.0004050016,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002856538,"about_ca_topic_score_gemma":0.000009184592,"domain_scores_codex":[0.996153,0.0001080823,0.0006391229,0.002107564,0.0003141222,0.0006781043],"domain_scores_gemma":[0.9973966,0.00001068596,0.0001702262,0.002021917,0.0001689629,0.0002316337],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00007372201,0.0006176411,0.0004204371,0.0009692934,0.000072221,0.000008307319,0.00001647465,0.00365328,0.9923801,0.00004541225,0.001737209,0.000005843596],"study_design_scores_gemma":[0.0007493123,0.0001755855,0.00008821756,0.0003903979,0.0001315429,7.374459e-8,0.000007397722,0.07803145,0.9043769,0.000006426955,0.01473473,0.001308017],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9300181,0.003463031,0.06269213,0.0001243937,0.001253832,0.001359724,0.0006782995,0.000202055,0.0002084771],"genre_scores_gemma":[0.9741353,0.0002761431,0.02427009,0.0001786481,0.00057726,0.0002380343,0.00003304252,0.0002017499,0.00008977838],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.08800331,"threshold_uncertainty_score":0.9995485,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02987459943069743,"score_gpt":0.2195691386022869,"score_spread":0.1896945391715894,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}