{"id":"W2100033550","doi":"10.1093/bioinformatics/16.12.1105","title":"Saturated BLAST: an automated multiple intermediate sequence search used to detect distant homology","year":2000,"lang":"en","type":"article","venue":"Bioinformatics","topic":"Protein Structure and Dynamics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":65,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University","funders":"National Institute of General Medical Sciences; National Institutes of Health","keywords":"Perl; Computer science; False positive paradox; Sequence (biology); Sequence database; Cluster analysis; Parsing; Data mining; Multiple sequence alignment; Sequence alignment; Artificial intelligence; Programming language; Biology; Peptide sequence; Gene; Genetics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000171221,0.0002125098,0.0001846919,0.0000671594,0.0001182679,0.00005975626,0.0003541002,0.0002256539,0.00005152804],"category_scores_gemma":[0.00008441553,0.0001841559,0.0000569299,0.000191349,0.0001120802,0.00001779933,0.0001002649,0.0001598138,0.0001163854],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003373844,"about_ca_system_score_gemma":0.0001062796,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0000365224,"about_ca_topic_score_gemma":0.0002034064,"domain_scores_codex":[0.9987733,0.00006681287,0.0003542796,0.0002266629,0.0001586079,0.0004203311],"domain_scores_gemma":[0.9991076,0.00001625274,0.00005898823,0.0005122616,0.0000850799,0.0002198194],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008557592,0.00007646687,0.001452386,0.0001443411,0.0001370274,0.00004466749,0.002826795,0.005077872,0.8476109,0.00004650148,0.001464091,0.1402632],"study_design_scores_gemma":[0.002579388,0.003022426,0.0100122,0.00007285538,0.00004587447,0.0002917286,0.0005509533,0.5909132,0.3675125,0.0001663821,0.02344755,0.001384974],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9931053,0.00003059206,0.005675173,0.00003864964,0.000124752,0.0003709161,0.0001418801,0.0001563769,0.0003563847],"genre_scores_gemma":[0.9826291,0.00003231741,0.01607863,0.000403397,0.00007598588,0.00001948716,0.0005696116,0.00002330241,0.0001681495],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.5858353,"threshold_uncertainty_score":0.7509661,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01420292838673811,"score_gpt":0.279582879899132,"score_spread":0.2653799515123939,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}