{"id":"W3097585533","doi":"10.2196/22333","title":"Automatic Structuring of Ontology Terms Based on Lexical Granularity and Machine Learning: Algorithm Development and Validation","year":2020,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Natural language processing; Task (project management); Artificial intelligence; Convolutional neural network; Ontology; Relation (database); Pairwise comparison; Word embedding; Granularity; Set (abstract data type); Structuring; Machine learning; Embedding; Data mining; Programming language","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00274685,0.0007828544,0.0007614572,0.002751984,0.0005863637,0.001511344,0.001454462,0.001326344,0.002988363],"category_scores_gemma":[0.008581412,0.0003460644,0.0007875684,0.002079652,0.0005933676,0.0021242,0.00150537,0.001296639,0.001332545],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001380605,"about_ca_system_score_gemma":0.00198984,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006299876,"about_ca_topic_score_gemma":0.007800872,"domain_scores_codex":[0.9987932,0.0003603806,0.0001429668,0.0003631244,0.0002482017,0.00009191533],"domain_scores_gemma":[0.9954709,0.002774694,0.0003499428,0.000580641,0.0007243699,0.00009947598],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0002107873,0.0002401621,0.005477319,0.0003052915,0.00009470639,0.00009259822,0.0002036301,0.05852233,0.01464881,0.004212863,0.005071727,0.9109198],"study_design_scores_gemma":[0.00003876217,0.00004400984,0.001560169,0.00004379873,0.00002553459,0.00009922406,0.0001126359,0.9792417,0.009425315,0.00734066,0.002054751,0.00001354542],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04886407,0.0005210128,0.9426818,0.0002877097,0.00004519923,0.0003299579,0.000514366,0.005636576,0.001119215],"genre_scores_gemma":[0.1729809,0.0001916928,0.8240165,0.00009962091,0.00002687984,0.0002800043,0.001469505,0.0002081355,0.0007268074],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006299876,"threshold_uncertainty_score":0.0145269,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02046656088449061,"score_gpt":0.2732670438334047,"score_spread":0.2528004829489141,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}