{"id":"W4415063342","doi":"10.1093/jamia/ocaf169","title":"Should we synthesize more than we need: impact of synthetic data generation for high-dimensional cross-sectional medical data","year":2025,"lang":"en","type":"article","venue":"Journal of the American Medical Informatics Association","topic":"Privacy-Preserving Technologies in Data","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Ottawa Public Health; Children's Hospital of Eastern Ontario; University of Ottawa","funders":"Canadian Institutes of Health Research; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs; Deutsche Forschungsgemeinschaft","keywords":"Synthetic data; Medical research; Data collection; Big data; Data modeling","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02866964,0.0009111319,0.0007230096,0.0007179716,0.00050174,0.002504717,0.001118316,0.001346021,0.001188966],"category_scores_gemma":[0.1426925,0.0005030409,0.001373841,0.000684337,0.001672542,0.002553891,0.002592833,0.002198153,0.0003605335],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009639381,"about_ca_system_score_gemma":0.001218981,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001806542,"about_ca_topic_score_gemma":0.001678678,"domain_scores_codex":[0.982899,0.01338901,0.0006273516,0.001535675,0.001306095,0.0002429925],"domain_scores_gemma":[0.8412708,0.1336721,0.004032294,0.01719224,0.002911385,0.0009212131],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.004616036,0.001047608,0.1447817,0.001197809,0.001321568,0.0006377791,0.001712173,0.612325,0.01025346,0.01403488,0.005132601,0.2029395],"study_design_scores_gemma":[0.0003771566,0.002614752,0.02782152,0.0005168973,0.0005061949,0.0008666089,0.0008796896,0.9034779,0.01815565,0.03780418,0.00683774,0.0001416797],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7614848,0.001493196,0.2273358,0.003140442,0.0002780028,0.0007852425,0.002158141,0.0008611681,0.002463189],"genre_scores_gemma":[0.9163381,0.0002965335,0.07900954,0.0006431574,0.00006097226,0.0002950185,0.002792507,0.0001315187,0.0004326648],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02866964,"threshold_uncertainty_score":0.1516213,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06764900475797032,"score_gpt":0.3819244041408982,"score_spread":0.3142753993829279,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}