{"id":"W4366121787","doi":"10.2196/44876","title":"Identifying Patient Populations in Texts Describing Drug Approvals Through Deep Learning–Based Information Extraction: Development of a Natural Language Processing Algorithm","year":2023,"lang":"en","type":"article","venue":"JMIR Formative Research","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"AstraZeneca","keywords":"Computer science; Artificial intelligence; Machine learning; Deep learning; Information extraction; Population; Subject-matter expert; Automation; Natural language processing; Data science; Data mining; Expert system; Medicine; Engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008546045,0.0001028849,0.0001278575,0.0003088415,0.0002835533,0.00005599332,0.0001410458,0.0001027349,0.000006315608],"category_scores_gemma":[0.0002539125,0.00009016231,0.00004121898,0.0007575053,0.0001161007,0.00007042615,0.0001538228,0.0003657583,0.00002091069],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00007121942,"about_ca_system_score_gemma":0.0001486784,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002868504,"about_ca_topic_score_gemma":0.00004547052,"domain_scores_codex":[0.9984283,0.0001759399,0.0004088635,0.0001539477,0.0004823967,0.0003504965],"domain_scores_gemma":[0.9994377,0.00004813768,0.0001442967,0.0001117185,0.0002180096,0.00004006239],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00003558623,0.00006118479,0.0004105597,0.0001773168,0.00001104871,0.000002848969,0.0477099,0.0001955933,0.007092237,0.000006435821,0.0002358796,0.9440614],"study_design_scores_gemma":[0.003605919,0.0007800905,0.08750134,0.001691224,0.00001307356,0.00002853784,0.2866956,0.236295,0.3513057,0.0003465283,0.0306369,0.001100027],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.966296,0.0007253612,0.03210927,0.00008817796,0.00009677529,0.0003428057,0.000005216083,0.0000513381,0.0002850561],"genre_scores_gemma":[0.9800975,0.0000130424,0.01927674,0.00001452632,0.00002363534,0.0001420151,0.0003633257,0.000008253437,0.00006098809],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9429614,"threshold_uncertainty_score":0.3676712,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1027807238280939,"score_gpt":0.4281712360084775,"score_spread":0.3253905121803835,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}