{"id":"W4401164363","doi":"10.1109/csp62567.2024.00010","title":"Linkage Deanonymization Risks, Data-Matching and Privacy: A Case Study","year":2024,"lang":"en","type":"article","venue":"","topic":"Names, Identity, and Discrimination Research","field":"Social Sciences","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"New York Institute of Technology","funders":"","keywords":"Computer science; Linkage (software); Record linkage; Matching (statistics); Information privacy; Computer security; Internet privacy; Statistics; Environmental health; Mathematics; Genetics; Medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01490358,0.0005482804,0.0005859221,0.002518732,0.002554475,0.002639443,0.001617723,0.003109103,0.001389841],"category_scores_gemma":[0.03805162,0.0002902925,0.001395108,0.00659799,0.002411042,0.004447801,0.002566234,0.002299414,0.0002522341],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002904915,"about_ca_system_score_gemma":0.001711782,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009603282,"about_ca_topic_score_gemma":0.009178482,"domain_scores_codex":[0.9834924,0.01118606,0.0006803586,0.0009326533,0.002916838,0.0007916481],"domain_scores_gemma":[0.9156448,0.06989389,0.004009344,0.00626584,0.003092851,0.00109329],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.001512937,0.004201296,0.402288,0.001600034,0.0009771129,0.02064897,0.01137135,0.2432151,0.002762489,0.1397077,0.02918404,0.1425309],"study_design_scores_gemma":[0.0006893593,0.002735067,0.1374798,0.0007504263,0.000727497,0.02337835,0.0372933,0.5739303,0.01795341,0.1111027,0.09359743,0.0003623979],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9450559,0.001665042,0.03596848,0.004999259,0.00005857852,0.0005205632,0.001831682,0.000159086,0.009741442],"genre_scores_gemma":[0.9687595,0.001175134,0.02693068,0.0003820232,0.00005940993,0.0001848292,0.0009334157,0.00003022674,0.00154471],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01490358,"threshold_uncertainty_score":0.07881868,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2251580555565904,"score_gpt":0.4977252111921275,"score_spread":0.2725671556355372,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}