{"id":"W3045121768","doi":"10.1101/2020.07.21.214270","title":"Assessment of current taxonomic assignment strategies for metabarcoding eukaryotes","year":2020,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Environmental DNA in Biodiversity Studies","field":"Environmental Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Guelph; McGill University","funders":"Compute Canada","keywords":"Computer science; Biodiversity; Completeness (order theory); Data mining; Machine learning; Selection (genetic algorithm); Data science; Ecology; Biology; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02995252,0.001146707,0.00080138,0.00306457,0.001238727,0.003467377,0.00223852,0.001474132,0.0007208655],"category_scores_gemma":[0.05636964,0.0006204462,0.000768167,0.001945214,0.0008109996,0.002517977,0.001437595,0.0009736946,0.0007861371],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001504087,"about_ca_system_score_gemma":0.001432963,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004344648,"about_ca_topic_score_gemma":0.006206978,"domain_scores_codex":[0.9898149,0.004774132,0.0009803942,0.001564989,0.002498327,0.0003672674],"domain_scores_gemma":[0.971807,0.01411344,0.003012283,0.002363264,0.007917124,0.0007868611],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.003394408,0.0007657307,0.1432149,0.002533313,0.001236033,0.0001940957,0.00220119,0.0724243,0.1807302,0.004143714,0.004546264,0.584616],"study_design_scores_gemma":[0.0001900611,0.002793114,0.08745439,0.0008664886,0.0007953716,0.0006841428,0.001883648,0.6487643,0.2362685,0.006421158,0.01354733,0.0003314311],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7794146,0.006922993,0.2030394,0.001472198,0.0002094538,0.0004397799,0.001249786,0.003859861,0.003392021],"genre_scores_gemma":[0.5925693,0.001178156,0.4026769,0.0002120664,0.00003433254,0.0001890844,0.002223301,0.0003143348,0.0006025604],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9700475,"threshold_uncertainty_score":0.1584059,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04168535933777667,"score_gpt":0.2560996619982327,"score_spread":0.214414302660456,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}