{"id":"W4323662853","doi":"10.1101/2023.03.05.531190","title":"Split-Transformer Impute (STI): A Transformer Framework for Genotype Imputation","year":2023,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Gene expression and cancer classification","field":"Biochemistry, Genetics and Molecular Biology","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Imputation (statistics); Preprocessor; Computer science; Missing data; 1000 Genomes Project; Transformer; Data mining; Data pre-processing; Benchmarking; Artificial intelligence; Machine learning; Genotype; Biology; Single-nucleotide polymorphism; Engineering; Gene; Genetics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003731974,0.001114579,0.001381355,0.001289817,0.0004834629,0.00124869,0.004146856,0.001600275,0.005813259],"category_scores_gemma":[0.008813179,0.000847416,0.002224199,0.001624654,0.0009244978,0.002068975,0.002298993,0.003500897,0.002427933],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001039635,"about_ca_system_score_gemma":0.002189824,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008663998,"about_ca_topic_score_gemma":0.01491765,"domain_scores_codex":[0.998755,0.0005242004,0.00007009882,0.0003222907,0.0002087884,0.0001195243],"domain_scores_gemma":[0.9976497,0.001096165,0.0001523231,0.0005202257,0.0004351084,0.000146433],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006469507,0.0003017479,0.01156261,0.000282172,0.0005790751,0.000419643,0.000309896,0.5583736,0.003164501,0.04639193,0.02130637,0.3566615],"study_design_scores_gemma":[0.00001886377,0.00003040888,0.0002615036,0.00001353595,0.00002339533,0.00005210208,0.000009917595,0.979165,0.0006173445,0.01846613,0.001329523,0.00001237195],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.005370835,0.0002426007,0.9893139,0.0002545304,0.00003659452,0.00004266903,0.0007296533,0.003256567,0.0007527461],"genre_scores_gemma":[0.3730827,0.0006889699,0.6081145,0.001379316,0.0001730656,0.0003660009,0.008242508,0.001241321,0.006711669],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008663998,"threshold_uncertainty_score":0.01973683,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02070529690257752,"score_gpt":0.2681291536866069,"score_spread":0.2474238567840294,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}