{"id":"W3196311544","doi":"10.1109/tcbb.2021.3109557","title":"GapPredict – A Language Model for Resolving Gaps in Draft Genome Assemblies","year":2021,"lang":"en","type":"article","venue":"IEEE/ACM Transactions on Computational Biology and Bioinformatics","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canada's Michael Smith Genome Sciences Centre","funders":"National Human Genome Research Institute; National Institutes of Health","keywords":"Genome; Sequence assembly; Computer science; Similarity (geometry); Character (mathematics); Scaffold; Sequence (biology); k-mer; State (computer science); Computational biology; DNA sequencing; Reference genome; Artificial intelligence; Theoretical computer science; Algorithm; Biology; DNA; Programming language; Genetics; Gene; Image (mathematics); Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001270496,0.0001345277,0.000153095,0.00007990274,0.0001540711,0.00002039475,0.00009565387,0.0001392921,0.000003491541],"category_scores_gemma":[0.00003543869,0.0001289616,0.00006948781,0.00007959047,0.00007993423,0.000002961472,0.00001453137,0.0000848382,0.000001849906],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001133054,"about_ca_system_score_gemma":0.0001007316,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000002684419,"about_ca_topic_score_gemma":0.00005733045,"domain_scores_codex":[0.9992768,0.00002481106,0.0002700937,0.0001921102,0.00005151032,0.0001847324],"domain_scores_gemma":[0.9995461,0.0001027665,0.00005571131,0.0001587638,0.00009369106,0.00004303153],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003325642,0.0003365786,0.001248247,0.0002495834,0.0004408839,0.000005232503,0.004293384,0.730944,0.2306591,0.000779699,0.0002708288,0.03043991],"study_design_scores_gemma":[0.003245254,0.0007769627,0.005912201,0.00005558611,0.00008863496,0.00009866509,0.002142395,0.9516381,0.02353743,0.008800272,0.002933691,0.0007708269],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4728712,0.0008364779,0.525228,0.0002580666,0.0001186566,0.0001633707,0.0003761282,0.000006270099,0.000141925],"genre_scores_gemma":[0.9137369,0.000407172,0.0848779,0.0004780885,0.00004777847,0.00003855478,0.0002606071,0.000009182198,0.0001438354],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.4408657,"threshold_uncertainty_score":0.5258902,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01601469205008969,"score_gpt":0.2758623415886273,"score_spread":0.2598476495385376,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}