{"id":"W4415312233","doi":"10.2196/63908","title":"Clinical Information Extraction From Notes of Veterans With Lymphoid Malignancies: Natural Language Processing Study","year":2025,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Center for Advancing Translational Sciences; National Cancer Institute","keywords":"Veterans Affairs; Information extraction; Pipeline (software); Information system; Natural language; Data extraction; Work (physics)","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005773712,0.0001233318,0.0002530783,0.0001333304,0.00006757725,0.0001227726,0.0006048253,0.0001206051,0.000008880036],"category_scores_gemma":[0.0002406652,0.000089696,0.00004637079,0.0003514629,0.00006826771,0.001987038,0.0001638628,0.0002981663,0.00001009675],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004074394,"about_ca_system_score_gemma":0.000367493,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00008127723,"about_ca_topic_score_gemma":0.00002975486,"domain_scores_codex":[0.9978669,0.00004973126,0.001053002,0.00009873362,0.0007560446,0.0001755888],"domain_scores_gemma":[0.9987692,0.0002782682,0.0003342371,0.0003837818,0.00014962,0.00008486985],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003539684,0.0001964336,0.01196867,0.0002054166,0.00004121394,0.00000615078,0.046161,0.00009596394,0.00001514769,0.0002710205,0.00008238987,0.9409212],"study_design_scores_gemma":[0.002044598,0.0001931379,0.01629303,0.000582847,0.00002368977,0.00001387524,0.01341211,0.9663811,0.0002333325,0.00006524954,0.0005734519,0.0001835288],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5219647,0.00004739408,0.4768741,0.000118117,0.0002151283,0.0002220986,0.000001388772,0.00007961256,0.0004775287],"genre_scores_gemma":[0.9514892,0.000005750414,0.04763035,0.0007807157,0.00004494802,0.00002314235,0.00001029855,0.000002537919,0.00001301648],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9662852,"threshold_uncertainty_score":0.3657697,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01644192540689026,"score_gpt":0.3441439778989147,"score_spread":0.3277020524920244,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}