{"id":"W7082254231","doi":"10.48448/cdtf-hm31","title":"Robust Utility-Preserving Text Anonymization Based on Large Language Models","year":2025,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Geochemistry and Geologic Mapping","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"Robustness (evolution); Context (archaeology); Key (lock); Data modeling; Information privacy; Language model; Code (set theory); Face (sociological concept)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005567134,0.001679814,0.001762297,0.001422773,0.001306121,0.003625275,0.00245909,0.001560197,0.002834426],"category_scores_gemma":[0.01640499,0.0006066901,0.001980698,0.001873535,0.001666664,0.006771204,0.004465138,0.003218715,0.002195913],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002460297,"about_ca_system_score_gemma":0.002976109,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004172227,"about_ca_topic_score_gemma":0.006162009,"domain_scores_codex":[0.9946226,0.002592817,0.0002837482,0.001062814,0.001012245,0.0004258987],"domain_scores_gemma":[0.992108,0.003244285,0.0006598601,0.003087447,0.0006710541,0.0002292937],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007425067,0.0003608125,0.002966142,0.0003258861,0.000248504,0.0004283927,0.0005297119,0.6186554,0.007663859,0.08884231,0.02471169,0.2545248],"study_design_scores_gemma":[0.00002463711,0.0000327199,0.0001572361,0.00001671282,0.00002672141,0.0001030641,0.00005666052,0.9434848,0.002860903,0.04926696,0.003948583,0.00002102827],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01586591,0.0006639161,0.9720334,0.001268065,0.000127804,0.0002352788,0.001237219,0.005815282,0.002753227],"genre_scores_gemma":[0.5420189,0.001009285,0.4354268,0.0009983105,0.0003775486,0.0005804517,0.006681522,0.001330291,0.0115768],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005567134,"threshold_uncertainty_score":0.02944219,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02542903830547049,"score_gpt":0.2561528963576901,"score_spread":0.2307238580522196,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}