{"id":"W2089697781","doi":"10.3414/me10-01-0020","title":"Effectiveness of Lexico-syntactic Pattern Matching for Ontology Enrichment with Clinical Documents","year":2010,"lang":"en","type":"article","venue":"Methods of Information in Medicine","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":27,"is_retracted":false,"has_abstract":true,"ca_institutions":"Lockheed Martin (Canada)","funders":"National Human Genome Research Institute; National Cancer Institute; University of Pittsburgh","keywords":"Computer science; Matching (statistics); Sentence; Natural language processing; Set (abstract data type); Ontology; Information retrieval; Domain (mathematical analysis); Artificial intelligence; Mathematics; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01495641,0.001572562,0.001025687,0.005287445,0.0008548256,0.001747985,0.001618745,0.001636137,0.001852021],"category_scores_gemma":[0.0747392,0.0003808184,0.001121878,0.003385533,0.0005992274,0.003825296,0.002026985,0.0007762601,0.001268357],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007968153,"about_ca_system_score_gemma":0.002484699,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003900124,"about_ca_topic_score_gemma":0.003938792,"domain_scores_codex":[0.9858757,0.007370833,0.00166895,0.001604458,0.00310874,0.0003712631],"domain_scores_gemma":[0.9174861,0.07028343,0.003260953,0.00352568,0.004688786,0.0007550507],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.006097096,0.001974736,0.06441837,0.003018398,0.0009608365,0.0009240569,0.001686986,0.0242789,0.05013103,0.001309108,0.006862216,0.8383384],"study_design_scores_gemma":[0.001277363,0.004950485,0.06712058,0.0006606538,0.002068081,0.00502964,0.002787455,0.7300282,0.1569222,0.00655422,0.02223941,0.0003617416],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.7734967,0.002690496,0.1984828,0.001666722,0.0001486474,0.001423433,0.002220253,0.01108725,0.008783688],"genre_scores_gemma":[0.5494259,0.0007783714,0.4429776,0.0003578219,0.00008429708,0.0003504158,0.003567559,0.0005175566,0.001940465],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.01495641,"threshold_uncertainty_score":0.07909805,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02669870627091075,"score_gpt":0.4329955275027238,"score_spread":0.406296821231813,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}