{"id":"W4206522381","doi":"10.2196/32903","title":"Strategies to Address the Lack of Labeled Data for Supervised Machine Learning Training With Electronic Health Records: Case Study for the Extraction of Symptoms From Clinical Notes","year":2021,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Cancer Institute","keywords":"Artificial intelligence; Machine learning; Computer science; Task (project management); Bottleneck; Natural language processing; Information extraction; Named-entity recognition; Supervised learning; Set (abstract data type); Process (computing); Health records; Precision and recall; Health care; Artificial neural network","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01841357,0.0007732172,0.0006502471,0.001236691,0.001325588,0.001310088,0.002290991,0.002276348,0.0007612403],"category_scores_gemma":[0.05612585,0.0004318857,0.000775018,0.001503737,0.001110611,0.002332459,0.001309145,0.001915765,0.0003477653],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001266117,"about_ca_system_score_gemma":0.001643621,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002952376,"about_ca_topic_score_gemma":0.005288139,"domain_scores_codex":[0.9879553,0.008786635,0.0007547308,0.0009637177,0.001252345,0.0002872034],"domain_scores_gemma":[0.9144605,0.06802305,0.003759321,0.006859418,0.006065591,0.000832214],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001388536,0.003969863,0.1893442,0.001469675,0.0002961518,0.007440321,0.005856628,0.09102128,0.02448559,0.01417437,0.01361585,0.6469376],"study_design_scores_gemma":[0.00044453,0.001973201,0.04318046,0.0004886749,0.000274343,0.007044865,0.003656623,0.7962335,0.08068474,0.03648593,0.02929814,0.0002348595],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.4803913,0.0006826452,0.5058419,0.007717725,0.00009930601,0.0007578259,0.000445099,0.001098035,0.002966103],"genre_scores_gemma":[0.5828749,0.0001874797,0.4143324,0.0005732914,0.00005908715,0.0005564449,0.000603066,0.0001046146,0.0007086351],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01841357,"threshold_uncertainty_score":0.09738147,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.177950069762978,"score_gpt":0.4690022669547648,"score_spread":0.2910521971917868,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}