{"id":"W7114904848","doi":"10.5281/zenodo.17900952","title":"crisprHAL — Better data for better predictions: data curation improves deep learning for sgRNA/Cas9 prediction","year":2025,"lang":"","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"RNA regulation and disease","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Western University","funders":"","keywords":"Deep learning; Data curation; Training set; Data modeling; Missing data","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.001225764,0.000339285,0.0002581071,0.0002762985,0.003608226,0.001322844,0.002338585,0.0003079709,0.0009083824],"category_scores_gemma":[0.003115585,0.0003892021,0.0001254398,0.000398544,0.000244928,0.0002877095,0.003530098,0.0003056442,0.0001832825],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001149495,"about_ca_system_score_gemma":0.00004261228,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000007303481,"about_ca_topic_score_gemma":0.000001708679,"domain_scores_codex":[0.996379,0.0004218679,0.0007067384,0.001562531,0.0003437152,0.0005861073],"domain_scores_gemma":[0.9959157,0.00006052017,0.0003072684,0.002335716,0.001136535,0.0002443217],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001134528,0.0004203044,0.00007781891,0.0005799475,0.0004836179,0.000001612228,0.0001616411,0.0005369381,0.04469199,0.0006535467,0.583932,0.3673261],"study_design_scores_gemma":[0.002018091,0.000472414,0.0009393323,0.0000629296,0.0002613965,0.00001425881,0.0002565023,0.1515662,0.000770447,0.0002004631,0.843157,0.0002809824],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.008903679,0.001689667,0.9569842,0.006743571,0.001075105,0.003740253,0.01344289,0.0004555999,0.006965092],"genre_scores_gemma":[0.6593741,0.0009166105,0.003512578,0.001324296,0.002339788,0.00000321868,0.3264901,0.001569762,0.004469489],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9534715,"threshold_uncertainty_score":0.999856,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03732011026066585,"score_gpt":0.2887285183086397,"score_spread":0.2514084080479738,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}