{"id":"W4322200617","doi":"10.1101/2023.02.26.530100","title":"A generalizable Cas9/sgRNA prediction model using machine transfer learning with small high-quality datasets","year":2023,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"CRISPR and Genetic Engineering","field":"Biochemistry, Genetics and Molecular Biology","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Western University","funders":"Canadian Institutes of Health Research; Mitacs","keywords":"Cas9; CRISPR; Nuclease; Computer science; Guide RNA; Genome editing; Cleavage (geology); Subgenomic mRNA; Artificial intelligence; Machine learning; Computational biology; Biology; DNA; Genetics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002719039,0.001326315,0.0007959995,0.0007842052,0.00037801,0.0007697256,0.001706081,0.001499658,0.001498502],"category_scores_gemma":[0.005159648,0.0004264668,0.0009412528,0.0006768603,0.0005114987,0.001181325,0.0009864785,0.002190767,0.0007280175],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001121837,"about_ca_system_score_gemma":0.0009107335,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008536558,"about_ca_topic_score_gemma":0.006851009,"domain_scores_codex":[0.9993327,0.0002105387,0.00003458578,0.0002821376,0.0000855081,0.00005452035],"domain_scores_gemma":[0.997943,0.001169458,0.0001227437,0.0002750921,0.0004029059,0.00008680984],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002829601,0.0003327043,0.005988027,0.00006919788,0.0001646076,0.0001291009,0.00002677113,0.900438,0.004670256,0.0006422693,0.003107349,0.08414873],"study_design_scores_gemma":[0.000006583402,0.00002361493,0.0002191144,0.000001503078,0.000004558811,0.000005112204,0.000002283961,0.9983743,0.0008474995,0.0004261188,0.00008591508,0.00000322758],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4271035,0.0008945039,0.5550827,0.001411401,0.0001581641,0.0002959115,0.002729039,0.00953688,0.002788013],"genre_scores_gemma":[0.8776208,0.0001807243,0.1128795,0.0003363551,0.00005074543,0.0002938695,0.005391923,0.0001613302,0.003084878],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008536558,"threshold_uncertainty_score":0.01697373,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02636413462386981,"score_gpt":0.2671971836966561,"score_spread":0.2408330490727863,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}