{"id":"W4366121787","doi":"10.2196/44876","title":"Identifying Patient Populations in Texts Describing Drug Approvals Through Deep Learning–Based Information Extraction: Development of a Natural Language Processing Algorithm","year":2023,"lang":"en","type":"article","venue":"JMIR Formative Research","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"AstraZeneca","keywords":"Computer science; Artificial intelligence; Machine learning; Deep learning; Information extraction; Population; Subject-matter expert; Automation; Natural language processing; Data science; Data mining; Expert system; Medicine; Engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002564326,0.001573162,0.0008606694,0.005290232,0.0006577875,0.001808284,0.001602129,0.001598759,0.003549701],"category_scores_gemma":[0.007953006,0.0006361139,0.001691582,0.002199756,0.0006648464,0.00268847,0.001597241,0.002475552,0.003267167],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001340772,"about_ca_system_score_gemma":0.00265528,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004822013,"about_ca_topic_score_gemma":0.00687172,"domain_scores_codex":[0.9979298,0.0004267567,0.0005026195,0.0006851034,0.0003447775,0.0001109775],"domain_scores_gemma":[0.9936426,0.00445478,0.0005259156,0.0004229493,0.0008269632,0.0001268498],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0003032803,0.0004169837,0.007600544,0.0008238959,0.0001432819,0.0005404846,0.0006448419,0.02936807,0.02236334,0.004118316,0.01784928,0.9158277],"study_design_scores_gemma":[0.0001062402,0.0001719646,0.00417179,0.0002152606,0.0001173916,0.0007632036,0.0004882519,0.9186573,0.03597323,0.01835522,0.02090957,0.00007051627],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02858363,0.0004933216,0.942929,0.0009516975,0.0001009957,0.0009135389,0.007415437,0.01737966,0.001232625],"genre_scores_gemma":[0.07570115,0.000201672,0.9096349,0.0002619145,0.00005736787,0.0005771567,0.01216173,0.0002289091,0.001175141],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005290232,"threshold_uncertainty_score":0.01356161,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1027807238280939,"score_gpt":0.4281712360084775,"score_spread":0.3253905121803835,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}