{"id":"W4415740074","doi":"10.2196/78432","title":"Medical Feature Extraction From Clinical Examination Notes: Development and Evaluation of a Two-Phase Large Language Model Framework","year":2025,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Generalization; Feature (linguistics); Feature extraction; Calibration; Unified Medical Language System; Language model","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003956676,0.001411975,0.0009881096,0.001103608,0.0004486559,0.001103203,0.002541636,0.001598487,0.002378311],"category_scores_gemma":[0.006786277,0.0005049949,0.001398172,0.0005701812,0.0004169885,0.001651559,0.001517464,0.002619153,0.002070895],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001355431,"about_ca_system_score_gemma":0.002534522,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01770714,"about_ca_topic_score_gemma":0.02204766,"domain_scores_codex":[0.9986236,0.0004601398,0.0001055739,0.0003702726,0.0003282354,0.0001121868],"domain_scores_gemma":[0.9970106,0.001748438,0.000113368,0.0002661307,0.0006825784,0.0001789155],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0008105797,0.001000567,0.004986924,0.0003203228,0.0003157748,0.0004987735,0.0002572132,0.1839919,0.01866149,0.002199948,0.01748596,0.7694705],"study_design_scores_gemma":[0.00005253297,0.0001178085,0.0006041575,0.00001429302,0.00003550608,0.0001057494,0.00004115888,0.993056,0.003822076,0.0008386571,0.001289426,0.00002262954],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.07977022,0.001764582,0.8880952,0.001632475,0.0002277132,0.0007826823,0.001846112,0.0239767,0.00190426],"genre_scores_gemma":[0.3746858,0.0006556438,0.6080077,0.001140015,0.0001611455,0.0007583352,0.008709643,0.000687569,0.00519417],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01770714,"threshold_uncertainty_score":0.03520817,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04682460930576575,"score_gpt":0.4736675890042354,"score_spread":0.4268429796984697,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}