{"id":"W3097585533","doi":"10.2196/22333","title":"Automatic Structuring of Ontology Terms Based on Lexical Granularity and Machine Learning: Algorithm Development and Validation","year":2020,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Natural language processing; Task (project management); Artificial intelligence; Convolutional neural network; Ontology; Relation (database); Pairwise comparison; Word embedding; Granularity; Set (abstract data type); Structuring; Machine learning; Embedding; Data mining; Programming language","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002026399,0.000118685,0.0002051178,0.00003247493,0.00005742382,0.00001237246,0.0001042924,0.0002536324,0.00003143555],"category_scores_gemma":[0.000356807,0.000091175,0.00002423711,0.00004984135,0.0002470764,0.000004033537,0.0001101862,0.0002251247,0.000001387163],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000005517413,"about_ca_system_score_gemma":0.000069078,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000002136175,"about_ca_topic_score_gemma":0.000001693137,"domain_scores_codex":[0.999036,0.00004694062,0.0003764526,0.0001136746,0.0002788401,0.0001480489],"domain_scores_gemma":[0.9995167,0.00004736567,0.0001125336,0.00008382582,0.00002027887,0.0002193035],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00004285221,0.00004684283,0.003988255,0.0003996575,0.00003340295,0.000005434242,0.001492231,0.000007340643,0.0004416391,0.00005254868,0.0001080294,0.9933817],"study_design_scores_gemma":[0.00407844,0.002497465,0.01965284,0.0002426552,0.00004305524,0.00008265793,0.0009316892,0.8763985,0.05441763,0.000189231,0.04089023,0.0005755924],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9671242,0.00007882252,0.03169511,0.0007345872,0.0000378671,0.0001223303,0.000008620535,0.00003547624,0.0001629695],"genre_scores_gemma":[0.9505194,0.0000272983,0.04832137,0.0009670499,0.00004219628,0.000009231182,0.0001011847,0.000006498093,0.000005798184],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9928062,"threshold_uncertainty_score":0.3718009,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02046656088449061,"score_gpt":0.2732670438334047,"score_spread":0.2528004829489141,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}