{"id":"W2985622460","doi":"10.1074/mcp.tir119.001752","title":"Assessing Protein Sequence Database Suitability Using De Novo Sequencing","year":2019,"lang":"en","type":"article","venue":"Molecular & Cellular Proteomics","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":43,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Office of Extramural Research, National Institutes of Health; National Institute of General Medical Sciences","keywords":"Computational biology; Sequence (biology); Sequence database; Protein sequencing; Computer science; Biology; Database; Peptide sequence; Genetics; Gene","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01654787,0.0009480683,0.001103719,0.003243396,0.001147625,0.003116253,0.001298702,0.001347372,0.001928098],"category_scores_gemma":[0.02876303,0.0007609566,0.00101629,0.002837572,0.0005759957,0.002353251,0.001682194,0.001135237,0.001658769],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001239711,"about_ca_system_score_gemma":0.001161721,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001703532,"about_ca_topic_score_gemma":0.003159787,"domain_scores_codex":[0.9910467,0.002762466,0.001192485,0.002061208,0.002626693,0.0003104036],"domain_scores_gemma":[0.9820719,0.009219791,0.002461312,0.00209182,0.003725763,0.0004293516],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002769049,0.0004675851,0.04328782,0.001347102,0.0006637066,0.001331911,0.001049757,0.02186259,0.7136901,0.003210016,0.005675343,0.2046451],"study_design_scores_gemma":[0.0002159926,0.001140285,0.05367879,0.000279567,0.0005402641,0.002622971,0.0007714447,0.2468806,0.6461108,0.006818827,0.04056224,0.0003783619],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6714637,0.004539418,0.3070191,0.000824508,0.0002087041,0.001055922,0.006538142,0.004237081,0.004113337],"genre_scores_gemma":[0.5299059,0.002094701,0.447384,0.0005883639,0.0000482922,0.0007011878,0.01643353,0.0008417496,0.002002348],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01654787,"threshold_uncertainty_score":0.08751458,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03392472601878817,"score_gpt":0.2786965261732743,"score_spread":0.2447718001544862,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}