{"id":"W2883727091","doi":"10.1021/acs.biochem.8b00473","title":"Revealing Unexplored Sequence-Function Space Using Sequence Similarity Networks","year":2018,"lang":"en","type":"article","venue":"Biochemistry","topic":"Bioinformatics and Genomic Networks","field":"Biochemistry, Genetics and Molecular Biology","cited_by":105,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canada's Michael Smith Genome Sciences Centre; University of British Columbia","funders":"National Institute of General Medical Sciences; Natural Sciences and Engineering Research Council of Canada; Canadian Institutes of Health Research; Michael Smith Health Research BC","keywords":"Computational biology; Sequence (biology); Function (biology); Similarity (geometry); Protein sequencing; Sequence space; Biology; Protein function; Repertoire; Sequence alignment; Protein structure database; Alignment-free sequence analysis; Peptide sequence; Computer science; Genetics; Sequence database; Artificial intelligence; Gene; Mathematics; Physics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001751888,0.0007965986,0.001056784,0.007268769,0.0006860302,0.001938358,0.000726165,0.0008010005,0.0009755252],"category_scores_gemma":[0.006129299,0.0003418312,0.0008477527,0.004423167,0.001036891,0.002621487,0.001502569,0.000786777,0.0002891799],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001096235,"about_ca_system_score_gemma":0.0007275417,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002013516,"about_ca_topic_score_gemma":0.002237771,"domain_scores_codex":[0.9987112,0.0004812704,0.0001006553,0.000361068,0.0002821585,0.00006367274],"domain_scores_gemma":[0.9961655,0.002570067,0.0006069026,0.000331399,0.000200939,0.0001251626],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001064009,0.0005876018,0.08002427,0.001664633,0.001031559,0.001338582,0.0008619421,0.4400072,0.0451619,0.08743145,0.002177093,0.3386497],"study_design_scores_gemma":[0.00001683358,0.0001006585,0.007126477,0.0000771755,0.0001162297,0.0002788943,0.0002277861,0.8577785,0.004122829,0.1255338,0.004584918,0.00003588871],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.484857,0.005185276,0.4994599,0.000927095,0.00004001799,0.0001524023,0.003523021,0.00121389,0.004641516],"genre_scores_gemma":[0.8732722,0.002295326,0.1197088,0.00013124,0.00005683508,0.000162913,0.003872504,0.00007145289,0.0004287006],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007268769,"threshold_uncertainty_score":0.009264946,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03823930741795281,"score_gpt":0.2771345886916788,"score_spread":0.238895281273726,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}