{"id":"W6976728766","doi":"10.6084/m9.figshare.14623466","title":"Additional file 1 of Profile hidden Markov model sequence analysis can help remove putative pseudogenes from DNA barcoding and metabarcoding datasets","year":2021,"lang":"en","type":"article","venue":"Figshare","topic":"Environmental DNA in Biodiversity Studies","field":"Environmental Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Guelph","funders":"","keywords":"DNA barcoding; Pseudogene; Hidden Markov model; Sequence (biology); Sequence analysis; Markov chain","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.002083727,0.001731217,0.001532686,0.003369585,0.001456601,0.001996436,0.00288674,0.001711712,0.7992164],"category_scores_gemma":[0.01347065,0.0009288206,0.001247023,0.004980407,0.0004591383,0.00267915,0.001833501,0.001747706,0.2889514],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001018929,"about_ca_system_score_gemma":0.001731331,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005360082,"about_ca_topic_score_gemma":0.01093703,"domain_scores_codex":[0.9991971,0.000133241,0.00009740996,0.0002665534,0.0001674572,0.0001382245],"domain_scores_gemma":[0.9906554,0.006728115,0.0004320435,0.0008160382,0.001031512,0.0003369434],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0002369713,0.00007553321,0.001550938,0.001670692,0.00004601704,0.00007149895,0.00008084557,0.0006504366,0.0007206677,0.001084131,0.986078,0.007734273],"study_design_scores_gemma":[0.001552474,0.0001528349,0.01229128,0.001169812,0.0001621392,0.0003745327,0.0003132441,0.003732769,0.004029105,0.01181763,0.9642246,0.0001795784],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"other","genre_scores_codex":[0.000163545,0.00001394689,0.0009799993,0.00005425713,0.00002873295,0.00003165442,0.9964827,0.001409953,0.0008351606],"genre_scores_gemma":[0.002855052,0.00006164936,0.006197021,0.0002421015,0.00004285887,0.0004408031,0.9842264,0.002956848,0.002977163],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.7992164,"threshold_uncertainty_score":0.2863934,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04128832294681259,"score_gpt":0.2374476124995104,"score_spread":0.1961592895526978,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}