{"id":"W4414161315","doi":"10.2196/68707","title":"Performance of Natural Language Processing for Information Extraction From Electronic Health Records Within Cancer: Systematic Review","year":2025,"lang":"en","type":"review","venue":"JMIR Medical Informatics","topic":"Topic Modeling","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Aalborg Universitetshospital; Aalborg Universitet","keywords":"Information extraction; Health records; Electronic health record; Natural language; Unstructured data; Biomedical text mining; Text mining; Text processing; Information processing","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.001780583,0.0003642933,0.002376622,0.0002794341,0.0001125583,0.0001007626,0.001366209,0.0002988233,0.000007404866],"category_scores_gemma":[0.0005080288,0.0002651077,0.0002872095,0.0006845818,0.00003001461,0.001777772,0.0001884515,0.0008406432,0.000009522998],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005279686,"about_ca_system_score_gemma":0.004894843,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00004579712,"about_ca_topic_score_gemma":0.00001835051,"domain_scores_codex":[0.9947397,0.0001313246,0.003545061,0.0001699149,0.0009817636,0.0004322248],"domain_scores_gemma":[0.9951515,0.0003418858,0.003471031,0.0006788653,0.0002160326,0.0001406704],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"systematic_review","study_design_gemma":"systematic_review","study_design_scores_codex":[7.226485e-7,0.000005662508,6.106742e-8,0.5171881,0.00002479917,6.913536e-8,0.0008681106,0.000001750967,1.814499e-9,0.0001151502,0.000152293,0.4816433],"study_design_scores_gemma":[0.0001381443,0.00005017417,7.662955e-8,0.709406,0.0002151842,0.00001561364,0.0001041532,0.2673717,4.377104e-7,0.000009842491,0.0224823,0.0002063138],"study_design_candidate":"systematic_review","study_design_consensus":"systematic_review","genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.000004055301,0.9150145,0.08036819,0.0001111614,0.0005724635,0.003684434,0.0000301272,0.0001198944,0.00009514711],"genre_scores_gemma":[0.00003571349,0.9859758,0.01110394,0.001120606,0.00008387335,0.001356815,0.0002470652,0.00001077967,0.00006538428],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.481437,"threshold_uncertainty_score":0.9999801,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01986415516832182,"score_gpt":0.3726917215292588,"score_spread":0.352827566360937,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}