{"id":"W4393236229","doi":"10.1101/2024.03.24.586462","title":"BetaAlign: a deep learning approach for multiple sequence alignment","year":2024,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Azrieli Foundation; Tel Aviv University","keywords":"Computer science; Artificial intelligence; Subspace topology; Multiple sequence alignment; Indel; Deep learning; Transformer; Machine learning; Sequence (biology); Phylogenomics; Smith–Waterman algorithm; Genomics; Sequence alignment; Phylogenetic tree; Biology; Genome","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001758436,0.00126105,0.0009911436,0.001284966,0.0007571407,0.001514737,0.003278561,0.001600081,0.00576227],"category_scores_gemma":[0.003684039,0.0008005595,0.0008988613,0.00174276,0.0008842305,0.002274447,0.002658935,0.004151592,0.002881678],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001080064,"about_ca_system_score_gemma":0.001742497,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002542076,"about_ca_topic_score_gemma":0.003883165,"domain_scores_codex":[0.9989678,0.0002684461,0.00005253286,0.0002761133,0.0003363389,0.00009890199],"domain_scores_gemma":[0.9988812,0.0004123812,0.00009517383,0.0002470374,0.0002693578,0.00009491033],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0004628364,0.0002805848,0.001699459,0.0003248991,0.000215164,0.0001717906,0.0001681411,0.2014708,0.02419107,0.04412026,0.03016917,0.6967258],"study_design_scores_gemma":[0.00002376498,0.00003932082,0.0001234072,0.00001309475,0.000009361692,0.00004050475,0.00001708282,0.9659755,0.005781609,0.0235715,0.004395632,0.000009178081],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.00632079,0.0002822927,0.9815412,0.0002242095,0.00008559177,0.00005161024,0.0003192253,0.00992933,0.00124584],"genre_scores_gemma":[0.1110645,0.0003578821,0.8794858,0.0004355146,0.0001011896,0.0002439133,0.002243206,0.0014056,0.004662304],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.00576227,"threshold_uncertainty_score":0.01927674,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01980617216349289,"score_gpt":0.2317610329314344,"score_spread":0.2119548607679415,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}