{"id":"W1997344234","doi":"10.1016/j.aca.2006.01.072","title":"Creating hierarchical models of protein families based on Expressed Sequence Tags: The “Sprockets” analysis pipeline","year":2006,"lang":"en","type":"article","venue":"Analytica Chimica Acta","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Calgary","funders":"Genome Prairie; Genome Canada","keywords":"Computational biology; Sequence analysis; Pipeline (software); Sequence (biology); Genome; Alignment-free sequence analysis; Sequence alignment; Multiple sequence alignment; Genetics; Gene; Data mining; Biology; Computer science; Peptide sequence","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002806418,0.003475606,0.001980397,0.002586557,0.001642662,0.003678846,0.003129231,0.00181912,0.007151315],"category_scores_gemma":[0.005408339,0.001844936,0.006628332,0.002179596,0.0006489126,0.002834429,0.002108072,0.003886789,0.005775368],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001421703,"about_ca_system_score_gemma":0.002882362,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004768476,"about_ca_topic_score_gemma":0.008852508,"domain_scores_codex":[0.9991876,0.0001452807,0.00007813835,0.000335265,0.0001784075,0.00007525837],"domain_scores_gemma":[0.9980189,0.001038824,0.0001516305,0.0004376524,0.0002061077,0.0001470176],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.004165047,0.001022939,0.02671971,0.002966369,0.003141547,0.002182087,0.002800449,0.1567135,0.1300895,0.06246921,0.1306743,0.4770553],"study_design_scores_gemma":[0.0002446999,0.0001414893,0.002677332,0.0001106885,0.0004770284,0.0004663986,0.0003115173,0.8232498,0.03446898,0.09464469,0.04301963,0.0001878152],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01885653,0.000197481,0.8845814,0.0003165283,0.00008107479,0.0002177908,0.01137205,0.08331284,0.001064216],"genre_scores_gemma":[0.08079847,0.0005022932,0.867389,0.000257026,0.00004270428,0.0007237366,0.03803215,0.01068758,0.001566979],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007151315,"threshold_uncertainty_score":0.02392352,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01488628344734453,"score_gpt":0.2410835843302372,"score_spread":0.2261973008828927,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}