{"id":"W3091951858","doi":"10.2196/23139","title":"Evaluating Identity Disclosure Risk in Fully Synthetic Health Data: Model Development and Validation","year":2020,"lang":"en","type":"article","venue":"Journal of Medical Internet Research","topic":"Data Analysis and Archiving","field":"Social Sciences","cited_by":85,"is_retracted":false,"has_abstract":true,"ca_institutions":"Children's Hospital of Eastern Ontario; University of Ottawa","funders":"Health Canada","keywords":"Computer science; Identity (music); Data science; Psychology","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.04502236,0.0009642086,0.0009901252,0.001395681,0.0005432498,0.001572521,0.002201157,0.001829394,0.001284736],"category_scores_gemma":[0.09866424,0.0004967417,0.001889127,0.001175628,0.00178122,0.001862309,0.002146776,0.002374188,0.0001645461],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003067106,"about_ca_system_score_gemma":0.002607682,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009060289,"about_ca_topic_score_gemma":0.00464087,"domain_scores_codex":[0.9866771,0.009865456,0.0005945,0.001204903,0.001411126,0.0002469857],"domain_scores_gemma":[0.8469939,0.135251,0.005208233,0.007119932,0.004920339,0.0005066597],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002770881,0.0001839345,0.01134861,0.0001723226,0.000185151,0.00008019077,0.0002527521,0.9563482,0.0004685882,0.01118974,0.0005590143,0.01893428],"study_design_scores_gemma":[0.00002388048,0.00009275577,0.001187767,0.00002876406,0.00001946144,0.00002568234,0.00003925025,0.9916892,0.0004001545,0.006263037,0.0002191687,0.00001088347],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2929238,0.0007330455,0.6999664,0.0014707,0.00005484403,0.001359295,0.001637572,0.0003157179,0.00153859],"genre_scores_gemma":[0.7736008,0.0002647982,0.2226219,0.0002081333,0.00002731338,0.001264683,0.001559262,0.00002451786,0.0004285771],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9549776,"threshold_uncertainty_score":0.2381038,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3742093669327674,"score_gpt":0.5630940315766862,"score_spread":0.1888846646439189,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}