{"id":"W4402592798","doi":"10.2196/60665","title":"An Automatic and End-to-End System for Rare Disease Knowledge Graph Construction Based on Ontology-Enhanced Large Language Models: Development Study","year":2024,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Information extraction; Named-entity recognition; Relationship extraction; Domain knowledge; Ontology; Data science; Information retrieval; Data mining; Artificial intelligence","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005125393,0.0001949724,0.0002350342,0.0001345197,0.00012639,0.00005791957,0.0001947857,0.0002246307,0.00001973585],"category_scores_gemma":[0.0001438914,0.0001478713,0.00005497871,0.0001304412,0.000127779,0.00001229699,0.00007484686,0.0001524909,0.00001047112],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003649685,"about_ca_system_score_gemma":0.0003517391,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000001533546,"about_ca_topic_score_gemma":0.00003070939,"domain_scores_codex":[0.9986369,0.00006944685,0.0004507872,0.0002335339,0.0003113071,0.0002980507],"domain_scores_gemma":[0.9990878,0.00008192534,0.000057556,0.0002711504,0.00005189432,0.000449646],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003984666,0.001170027,0.0003916297,0.004933621,0.0002910871,0.00005808083,0.02919309,0.00009682628,0.0005972845,0.0009683138,0.003080072,0.9588215],"study_design_scores_gemma":[0.004804634,0.00323165,0.001739542,0.001781739,0.000164655,0.00004758437,0.06421295,0.9027825,0.003259984,0.0001021809,0.01692308,0.0009495007],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.85931,0.0003961421,0.1382815,0.0001145351,0.0003676432,0.0008495568,0.00007918319,0.0002242214,0.000377177],"genre_scores_gemma":[0.9884318,0.000005656172,0.01043818,0.0003158813,0.0001020113,0.0004185196,0.0002346521,0.00001572724,0.00003754599],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.957872,"threshold_uncertainty_score":0.6030015,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01387842618304462,"score_gpt":0.3118774236526712,"score_spread":0.2979989974696265,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}