{"id":"W4402592798","doi":"10.2196/60665","title":"An Automatic and End-to-End System for Rare Disease Knowledge Graph Construction Based on Ontology-Enhanced Large Language Models: Development Study","year":2024,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Information extraction; Named-entity recognition; Relationship extraction; Domain knowledge; Ontology; Data science; Information retrieval; Data mining; Artificial intelligence","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002157309,0.001332136,0.0007792109,0.002249087,0.0006141734,0.001559501,0.002377063,0.0009899603,0.007680943],"category_scores_gemma":[0.006829933,0.0008589427,0.001514694,0.001301937,0.0003633397,0.003355404,0.002418023,0.001452974,0.005025571],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009714553,"about_ca_system_score_gemma":0.002741124,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007870907,"about_ca_topic_score_gemma":0.01050031,"domain_scores_codex":[0.9985636,0.0003213451,0.000147129,0.0005599857,0.0003275793,0.00008036095],"domain_scores_gemma":[0.9960639,0.001998824,0.0002098592,0.0006630024,0.0008718548,0.0001925714],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005077717,0.001046268,0.006504408,0.001163432,0.0003155993,0.001008186,0.0006327168,0.01838106,0.03284603,0.005336033,0.06715003,0.8651084],"study_design_scores_gemma":[0.0003432544,0.000555589,0.006313198,0.0001645008,0.0002864338,0.001688541,0.0004572319,0.8432567,0.06407448,0.006710287,0.07594008,0.0002097828],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.03328389,0.0005561952,0.7441306,0.0004864043,0.000136918,0.001322432,0.008513264,0.2092411,0.002329192],"genre_scores_gemma":[0.0622391,0.0003598007,0.9049541,0.0002706991,0.00002732246,0.0006741753,0.02633876,0.002394749,0.002741359],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007870907,"threshold_uncertainty_score":0.02569532,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01387842618304462,"score_gpt":0.3118774236526712,"score_spread":0.2979989974696265,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}