{"id":"W3196311544","doi":"10.1109/tcbb.2021.3109557","title":"GapPredict – A Language Model for Resolving Gaps in Draft Genome Assemblies","year":2021,"lang":"en","type":"article","venue":"IEEE/ACM Transactions on Computational Biology and Bioinformatics","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canada's Michael Smith Genome Sciences Centre","funders":"National Human Genome Research Institute; National Institutes of Health","keywords":"Genome; Sequence assembly; Computer science; Similarity (geometry); Character (mathematics); Scaffold; Sequence (biology); k-mer; State (computer science); Computational biology; DNA sequencing; Reference genome; Artificial intelligence; Theoretical computer science; Algorithm; Biology; DNA; Programming language; Genetics; Gene; Image (mathematics); Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002467541,0.001209745,0.0006479865,0.0006933712,0.0005303687,0.001229313,0.002829007,0.001429892,0.003839792],"category_scores_gemma":[0.01097603,0.0007976256,0.00162611,0.0005562532,0.0007232144,0.00273439,0.001544703,0.002368411,0.00241935],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009260331,"about_ca_system_score_gemma":0.003013239,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004293052,"about_ca_topic_score_gemma":0.00728766,"domain_scores_codex":[0.9986694,0.0003844388,0.0001417466,0.0003798476,0.0003326872,0.00009185092],"domain_scores_gemma":[0.9946301,0.003519926,0.0004070413,0.0005776557,0.0007256013,0.0001395525],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00151513,0.0003722947,0.0139815,0.001263819,0.0002640747,0.0006469132,0.000923149,0.6184906,0.02256223,0.03285428,0.03473141,0.2723947],"study_design_scores_gemma":[0.00003159876,0.00007146694,0.0001993086,0.00002303333,0.0000186304,0.00006358468,0.00002726301,0.9797371,0.005483257,0.00934073,0.004983816,0.00002015237],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02686927,0.0003467113,0.9162517,0.0004539645,0.0001647165,0.000188219,0.004879331,0.04926394,0.00158221],"genre_scores_gemma":[0.2631319,0.0003708959,0.7164964,0.0006532393,0.0001056709,0.0007854435,0.01208606,0.0036078,0.002762542],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004293052,"threshold_uncertainty_score":0.01304978,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01601469205008969,"score_gpt":0.2758623415886273,"score_spread":0.2598476495385376,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}