{"id":"W1982088273","doi":"10.1016/j.ymeth.2014.10.027","title":"Text as data: Using text-based features for proteins representation and for computational prediction of their characteristics","year":2014,"lang":"en","type":"review","venue":"Methods","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":24,"is_retracted":false,"has_abstract":true,"ca_institutions":"Mount Sinai Hospital; University of Toronto; Queen's University","funders":"U.S. National Library of Medicine; Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Representation (politics); Function (biology); Computational biology; Protein sequencing; Value (mathematics); Protein structure database; Process (computing); Genome; Sequence (biology); Protein function prediction; Protein function; Data mining; Information retrieval; Bioinformatics; Biology; Machine learning; Gene; Sequence database; Genetics; Peptide sequence","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001433323,0.001328608,0.001708324,0.005777986,0.0002243364,0.002515398,0.002301351,0.001556026,0.003536585],"category_scores_gemma":[0.004330429,0.000363276,0.001201704,0.006737675,0.0008860197,0.004477598,0.0009655079,0.001692949,0.003941684],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007632763,"about_ca_system_score_gemma":0.0009582809,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001362011,"about_ca_topic_score_gemma":0.001166287,"domain_scores_codex":[0.9991535,0.0001686765,0.00009384126,0.0001769847,0.0003739664,0.00003304222],"domain_scores_gemma":[0.9976604,0.001669784,0.0001766156,0.0001260357,0.0003076834,0.00005953967],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00005109619,0.00006347572,0.0004625483,0.007385297,0.000137323,0.0001271032,0.0001106275,0.001027168,0.002968147,0.008191369,0.02753678,0.951939],"study_design_scores_gemma":[0.00005145267,0.0001109439,0.004059471,0.005984694,0.0003290936,0.00156341,0.0002920296,0.007016511,0.006840672,0.03786667,0.9357179,0.0001672022],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.003427256,0.8652245,0.1033983,0.004192722,0.001626717,0.0003038285,0.006035774,0.002018247,0.01377263],"genre_scores_gemma":[0.02729205,0.8315419,0.1167519,0.002977481,0.001670153,0.0005612238,0.01111492,0.0003822237,0.007708097],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.005777986,"threshold_uncertainty_score":0.01183105,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2246045402564346,"score_gpt":0.4994441404292237,"score_spread":0.2748396001727891,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}