{"id":"W4380550560","doi":"10.1101/2023.06.13.544850","title":"The impacts of active and self-supervised learning on efficient annotation of single-cell expression data","year":2023,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Single-cell and spatial transcriptomics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute; Sinai Health System; Lunenfeld-Tanenbaum Research Institute; Institute of Cancer Research; Ontario Institute for Cancer Research; University of Toronto","funders":"Canadian Institutes of Health Research; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs; Princess Margaret Cancer Foundation","keywords":"Annotation; Computer science; Benchmarking; Classifier (UML); Machine learning; Artificial intelligence; Heuristic; Supervised learning; Active learning (machine learning); Similarity (geometry); Data mining","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0004461313,0.0002945255,0.0003000023,0.00008816984,0.0001430317,0.00005356621,0.0005157245,0.0003511132,0.000001123864],"category_scores_gemma":[0.0003072797,0.0002497126,0.0000760026,0.0001528327,0.0001174165,0.000006546024,0.0005721535,0.0003273126,0.000001749478],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003270939,"about_ca_system_score_gemma":0.0002148474,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003858007,"about_ca_topic_score_gemma":0.00000361713,"domain_scores_codex":[0.9982247,0.0001692079,0.0003973577,0.0006807358,0.0002629679,0.0002649711],"domain_scores_gemma":[0.9979684,0.00009717511,0.0004379218,0.00110805,0.0002899261,0.00009846604],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0001838823,0.0002607368,0.001563777,0.0002999694,0.00007541237,0.000001830057,0.00003673638,0.0007175808,0.9967886,0.000007283027,0.00005047751,0.00001368444],"study_design_scores_gemma":[0.0005402748,0.0002869197,0.009975852,0.0002755296,0.00006285877,5.855293e-9,0.00001956357,0.00266326,0.9854329,3.723741e-7,0.0004926945,0.0002498146],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9975821,0.00065718,0.0006552765,0.00003580108,0.0003780302,0.0003821716,0.0002517441,0.00005021924,0.000007518237],"genre_scores_gemma":[0.9978297,0.0007969093,0.001124829,0.00001568087,0.0001301902,0.00001689856,0.000008520514,0.00007353348,0.000003745672],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01135577,"threshold_uncertainty_score":0.9999955,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02303407408858868,"score_gpt":0.2301645880025953,"score_spread":0.2071305139140066,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}