{"id":"W2612992329","doi":"10.2196/medinform.7235","title":"Effective Information Extraction Framework for Heterogeneous Clinical Reports Using Online Machine Learning and Controlled Vocabularies","year":2017,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":27,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Information extraction; Controlled vocabulary; Machine learning; Artificial intelligence; Unified Medical Language System; Information retrieval; Data science; Data mining","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009704759,0.001706053,0.001502977,0.008793053,0.001339494,0.003620579,0.003647614,0.00118135,0.003290794],"category_scores_gemma":[0.02819194,0.0007574856,0.002455724,0.00368424,0.001021706,0.007264676,0.004126266,0.001686409,0.001904627],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002353793,"about_ca_system_score_gemma":0.005967319,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01166883,"about_ca_topic_score_gemma":0.01007952,"domain_scores_codex":[0.99164,0.002215412,0.002321357,0.001775389,0.001813577,0.0002341801],"domain_scores_gemma":[0.9797217,0.01168263,0.002329225,0.00221319,0.003695965,0.0003573219],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005221181,0.0004735444,0.008122111,0.002701074,0.000298124,0.001593413,0.002173548,0.03181736,0.04158759,0.02384218,0.03072692,0.8561419],"study_design_scores_gemma":[0.0004147372,0.0004261778,0.005170045,0.0008526716,0.0004475351,0.001571542,0.001432702,0.7800725,0.07643685,0.04957344,0.08327136,0.0003305099],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01523401,0.000867942,0.9529904,0.001081495,0.00009046924,0.001295973,0.004780098,0.02187652,0.001783179],"genre_scores_gemma":[0.08273669,0.0003483817,0.9006593,0.0002621911,0.00008272719,0.0009270186,0.01358757,0.0005291693,0.0008668657],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01166883,"threshold_uncertainty_score":0.05132425,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02581817604947437,"score_gpt":0.3984720093010376,"score_spread":0.3726538332515632,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}