{"id":"W2745161505","doi":"10.12705/664.9","title":"Building the “Plant Glossary”—A controlled botanical vocabulary using terms extracted from the Floras of North America and China","year":2017,"lang":"en","type":"article","venue":"Taxon","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":14,"is_retracted":false,"has_abstract":true,"ca_institutions":"Agriculture and Agri-Food Canada","funders":"National Science Foundation","keywords":"Glossary; Categorization; Vocabulary; Taxon; Synonym (taxonomy); Computer science; Taxonomic rank; Natural language processing; Term (time); Set (abstract data type); Linguistics; Artificial intelligence; Interpretation (philosophy); Ecology; Biology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001396884,0.0001165967,0.000225736,0.000006272221,0.0003269385,0.00004876285,0.0003896585,0.00009715007,0.000005568268],"category_scores_gemma":[0.0004763381,0.00005843604,0.00007334174,0.0000208283,0.0005735916,0.000003133534,0.0002034698,0.000133566,6.193448e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000003760941,"about_ca_system_score_gemma":0.00002546172,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003942667,"about_ca_topic_score_gemma":0.0001158077,"domain_scores_codex":[0.9992573,0.00008155964,0.0001820714,0.0002043686,0.0001143026,0.0001604042],"domain_scores_gemma":[0.9991107,0.00009011377,0.0002239931,0.0005143652,0.00001831004,0.00004253119],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","study_design_scores_codex":[0.001695524,0.0001625701,0.1283159,0.00002019683,0.0004960321,0.00004061396,0.00022659,0.00002490162,0.5748731,0.00004975323,0.002201223,0.2918935],"study_design_scores_gemma":[0.002318269,0.000255564,0.9523542,0.00005495815,0.0001325465,0.00002822973,0.00006841167,0.003099642,0.008179855,0.00011662,0.03318471,0.0002069847],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9944878,0.001449393,0.002900902,0.0007191183,0.0001338528,0.000151992,0.00005666288,0.000008272187,0.00009206776],"genre_scores_gemma":[0.9966002,0.0002343903,0.002689979,0.0001671944,0.0002467228,0.000006942969,0.00002568497,0.000007667433,0.00002120385],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8240383,"threshold_uncertainty_score":0.251458,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02429857359007057,"score_gpt":0.2736396031255826,"score_spread":0.249341029535512,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}