{"id":"W4384925646","doi":"10.1038/s41467-023-39985-2","title":"Identification of transcriptional programs using dense vector representations defined by mutual information with GeneVector","year":2023,"lang":"en","type":"article","venue":"Nature Communications","topic":"Single-cell and spatial transcriptomics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"Congressionally Directed Medical Research Programs; National Cancer Institute; National Human Genome Research Institute; Canadian Cancer Society Research Institute; Ovarian Cancer Research Fund; National Institutes of Health; Cancer Research UK; Cycle for Survival; Canadian Institutes of Health Research; Mark Foundation For Cancer Research; Memorial Sloan-Kettering Cancer Center; Marie-Josée and Henry R. Kravis Center for Molecular Oncology; U.S. Department of Defense; Breast Cancer Research Foundation; Stand Up To Cancer","keywords":"Dimensionality reduction; Computer science; Computational biology; Curse of dimensionality; Mutual information; Embedding; Scalability; Gene; Leverage (statistics); Identification (biology); Principal component analysis; Feature vector; Artificial intelligence; Biology; Genetics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001339788,0.0001062495,0.00009766869,0.00009109081,0.000192156,0.00003782746,0.0003985993,0.0002029553,0.000004130762],"category_scores_gemma":[0.0000800937,0.0001035025,0.00006702221,0.0004474522,0.0001413735,0.00003255773,0.00004854368,0.0001980731,0.000007744861],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001949697,"about_ca_system_score_gemma":0.00009523435,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000042585,"about_ca_topic_score_gemma":0.0002709159,"domain_scores_codex":[0.9991502,0.0000734292,0.000320117,0.0001444507,0.0001791314,0.000132728],"domain_scores_gemma":[0.9987115,0.00003043397,0.0001523958,0.0007598757,0.0003019922,0.00004379023],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.000060496,0.0001410917,0.006039175,0.00001816317,0.0000644735,9.965011e-8,0.0002153524,0.0001957139,0.9896984,0.001062956,0.001780459,0.0007236163],"study_design_scores_gemma":[0.00430917,0.000780783,0.09489348,0.0001280059,0.0004553697,0.00006331078,0.002082708,0.03369103,0.6961523,0.0003183019,0.1658097,0.001315828],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9894342,0.0008479044,0.00801596,0.0005946875,0.0001477483,0.0004096568,0.0003450744,0.00006345263,0.0001412943],"genre_scores_gemma":[0.9898048,0.0002024193,0.004114249,0.00007070682,0.00003309956,0.00006007201,0.00561034,0.00001567745,0.00008862563],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2935461,"threshold_uncertainty_score":0.4220708,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02417645493662699,"score_gpt":0.2824276157368955,"score_spread":0.2582511608002684,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}