{"id":"W4416035400","doi":"10.18653/v1/2025.emnlp-main.1637","title":"Not What the Doctor Ordered: Surveying LLM-based De-identification and Quantifying Clinical Information Loss","year":2025,"lang":"en","type":"article","venue":"","topic":"Imbalanced Data Classification Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Arthritis Society; Canadian Institute for Advanced Research; Alberta Health Services","keywords":"Information system; MEDLINE; Data loss; Information loss; Risk assessment","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.002881727,0.00009649379,0.0001154387,0.0001404397,0.0002452173,0.001575401,0.0007876128,0.00008527408,0.000005959912],"category_scores_gemma":[0.0005726728,0.00007276807,0.00003649657,0.000480818,0.000105221,0.003479644,0.0001877565,0.0001621362,0.00003184016],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004956048,"about_ca_system_score_gemma":0.0001603721,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005329855,"about_ca_topic_score_gemma":0.00002688259,"domain_scores_codex":[0.9985688,0.0002584952,0.0005637745,0.0002483716,0.0001903703,0.0001701605],"domain_scores_gemma":[0.9980619,0.0006804932,0.0001975171,0.0008281335,0.0001944385,0.00003749738],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.00002020764,0.00007913342,0.02504621,0.00007489968,0.00001946141,6.363381e-7,0.0003232477,0.00003171748,0.001141909,0.3491075,0.005294987,0.6188601],"study_design_scores_gemma":[0.000682972,0.00004560879,0.4600315,0.0001434851,0.00001585347,0.000005033719,0.0002968774,0.4474853,0.04300608,0.003571849,0.04438354,0.0003319018],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01340282,0.00003334713,0.9759025,0.009399263,0.0003639557,0.0002650015,0.000003637326,0.0003539147,0.0002755859],"genre_scores_gemma":[0.9507307,0.0001947495,0.04506357,0.003730068,0.00001640175,0.00004848401,0.00003596988,0.000003375174,0.000176639],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9373279,"threshold_uncertainty_score":0.9994611,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06983332277345129,"score_gpt":0.3749101502068776,"score_spread":0.3050768274334263,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}