{"id":"W2809728348","doi":"10.4236/jbise.2018.116012","title":"Improving Protein Sequence Classification Performance Using Adjacent and Overlapped Segments on Existing Protein Descriptors","year":2018,"lang":"en","type":"article","venue":"Journal of Biomedical Science and Engineering","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Japan Society for the Promotion of Science; Institute of Genetics; University of Tokyo; Institute of Medical Science, University of Tokyo; Research Organization of Information and Systems","keywords":"Subsequence; Sequence (biology); Computer science; Protein sequencing; Pattern recognition (psychology); Segmentation; Feature selection; Feature (linguistics); Feature vector; Artificial intelligence; Peptide sequence; Mathematics; Biology; Genetics; Gene","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009909912,0.0009487519,0.001226566,0.002732289,0.0004991578,0.0007030616,0.0007018537,0.0006782561,0.0008800275],"category_scores_gemma":[0.001918315,0.0001611009,0.0005648076,0.002822614,0.0003258329,0.001340615,0.0006892318,0.0006193028,0.000796632],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004210494,"about_ca_system_score_gemma":0.0007813728,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003209295,"about_ca_topic_score_gemma":0.003262927,"domain_scores_codex":[0.9993018,0.00009308814,0.00006793102,0.0002068734,0.0001961421,0.0001341456],"domain_scores_gemma":[0.9988217,0.0004820871,0.0001353776,0.0001451246,0.0003283071,0.0000873942],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001450526,0.0007517097,0.01846423,0.0001842319,0.0001288264,0.0002213933,0.0001601673,0.045472,0.1354616,0.0007000226,0.002998683,0.7940066],"study_design_scores_gemma":[0.00006101151,0.0005360482,0.01547474,0.00001587581,0.00007880777,0.0001891113,0.0001808284,0.9357996,0.04417241,0.001182634,0.002273106,0.0000358246],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.8022822,0.001563562,0.1892351,0.0001665476,0.0001004534,0.0001388851,0.0004823846,0.003735862,0.002295027],"genre_scores_gemma":[0.8749424,0.0003040757,0.1201893,0.0001023035,0.00007428116,0.00009215392,0.002713018,0.0001232565,0.00145926],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.003209295,"threshold_uncertainty_score":0.006381214,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02626929420757971,"score_gpt":0.273701352229618,"score_spread":0.2474320580220383,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}