{"id":"W4404261746","doi":"10.1101/2024.11.11.622097","title":"Codebook: sequence specificity and genomic binding of poorly-characterized human transcription factors","year":2024,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Genomics and Chromatin Dynamics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill Genome Centre; University of British Columbia; BC Children's Hospital; University of Toronto","funders":"Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung","keywords":"Codebook; Computational biology; Sequence (biology); Transcription (linguistics); Genetics; Biology; Computer science; Artificial intelligence; Linguistics; Philosophy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009182098,0.0004496308,0.0008799411,0.001256291,0.000331741,0.00053824,0.0005917366,0.0003636062,0.00362807],"category_scores_gemma":[0.002855858,0.0002699775,0.0002292526,0.001251465,0.0003471834,0.0004049246,0.0005437707,0.0004550428,0.00222632],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005394995,"about_ca_system_score_gemma":0.0006529434,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00141507,"about_ca_topic_score_gemma":0.002378297,"domain_scores_codex":[0.9992141,0.0001331864,0.00007366701,0.000229354,0.0003021061,0.00004766685],"domain_scores_gemma":[0.9985411,0.0006501531,0.0001914377,0.0003115506,0.000212619,0.00009317887],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002520231,0.00009950021,0.03569675,0.001590139,0.00009592836,0.0003794702,0.0003659209,0.004426907,0.8517056,0.004486452,0.01655106,0.08208219],"study_design_scores_gemma":[0.0002442531,0.0005778337,0.164736,0.0002148169,0.0002275636,0.004552512,0.000259333,0.03737479,0.6523827,0.00947513,0.1298127,0.0001423496],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"dataset","genre_scores_codex":[0.8159262,0.009062837,0.0977454,0.000526757,0.0001768991,0.0001488296,0.06509534,0.003750988,0.007566754],"genre_scores_gemma":[0.7573639,0.002892205,0.09407769,0.0003743247,0.0001179994,0.0002652295,0.1398436,0.0007969463,0.00426817],"genre_candidate":"dataset","genre_consensus":null,"teacher_disagreement_score":0.00362807,"threshold_uncertainty_score":0.01213712,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01825551914376264,"score_gpt":0.2256876600856431,"score_spread":0.2074321409418804,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}