{"id":"W4393818958","doi":"10.5281/zenodo.10403474","title":"The impacts of active and self-supervised learning on efficient annotation of single-cell expression data - source data","year":2023,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Single-cell and spatial transcriptomics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Lunenfeld-Tanenbaum Research Institute; University of Toronto","funders":"","keywords":"Annotation; Data source; Computer science; Expression (computer science); Artificial intelligence; Computational biology; Machine learning; Data mining; Biology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006115296,0.0001686482,0.0001704301,0.00009987875,0.0007529815,0.0001650299,0.001795502,0.0001474241,0.00004924549],"category_scores_gemma":[0.0009830752,0.0001431343,0.00002886274,0.0001928955,0.0001541497,0.00001360577,0.002834625,0.0002678119,0.0001171509],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000216567,"about_ca_system_score_gemma":0.00001188121,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005976938,"about_ca_topic_score_gemma":0.0000027351,"domain_scores_codex":[0.9982451,0.0003729579,0.0002830955,0.0005588317,0.0003223052,0.0002176853],"domain_scores_gemma":[0.9979013,0.00006237002,0.0002742201,0.001417695,0.0002631299,0.00008124675],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0002771487,0.0002942951,7.677237e-7,0.000216156,0.00005762594,0.000001386855,0.000163608,0.0002010316,0.2498688,0.000001221137,0.7382744,0.01064353],"study_design_scores_gemma":[0.000517681,0.0006339825,0.00003546069,0.00009503538,0.00004392083,0.000005620371,0.000246962,0.001410862,0.02332298,0.000001009087,0.9735397,0.0001468534],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.06015689,0.0002849829,0.0005637349,0.00007057281,0.000150252,0.0005481718,0.9376336,0.0000997216,0.0004920998],"genre_scores_gemma":[0.03330213,0.001089829,0.00006731277,0.00001504495,0.0001007994,3.010271e-8,0.9649289,0.0004213378,0.00007460328],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.2352653,"threshold_uncertainty_score":0.5836847,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04563327408192022,"score_gpt":0.258690667010738,"score_spread":0.2130573929288178,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}