{"id":"W4411952500","doi":"10.1136/bmjebm-2024-113617","title":"Understanding synthetic data: artificial datasets for real-world evidence","year":2025,"lang":"en","type":"article","venue":"BMJ evidence-based medicine","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[{"model":"gemma","categories":["metaresearch"],"domain":"methods","study_design":"theoretical_or_conceptual","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"low","status":"direct model label, unvalidated"},{"model":"gpt","categories":[],"domain":null,"study_design":"theoretical_or_conceptual","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"low","status":"direct model label, unvalidated"}],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01163291,0.0008299893,0.0007280022,0.003205444,0.0005665608,0.003244022,0.002003353,0.002356234,0.005670403],"category_scores_gemma":[0.09952826,0.0004781263,0.001233295,0.002793979,0.001325988,0.002661357,0.001779259,0.002234583,0.0007278434],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001442333,"about_ca_system_score_gemma":0.001691352,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002299161,"about_ca_topic_score_gemma":0.003511064,"domain_scores_codex":[0.9913707,0.006107592,0.0007986077,0.0006971309,0.0008994123,0.0001265134],"domain_scores_gemma":[0.8915088,0.09505644,0.002847051,0.006984324,0.002796152,0.0008072968],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00213758,0.001513846,0.04827198,0.006950872,0.001327541,0.001700505,0.001568572,0.4821263,0.002670852,0.1170871,0.08907125,0.2455738],"study_design_scores_gemma":[0.000369888,0.0003847969,0.005324843,0.001295086,0.0002394083,0.0007808478,0.001032765,0.717244,0.002411223,0.226462,0.04433995,0.0001151108],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"review","genre_scores_codex":[0.27947,0.007022222,0.5906603,0.01791871,0.001662288,0.001929041,0.08528548,0.002503688,0.01354822],"genre_scores_gemma":[0.6049941,0.001876862,0.3448342,0.001495439,0.0003484787,0.001245414,0.04381115,0.0001922278,0.00120217],"genre_candidate":"review","genre_consensus":null,"teacher_disagreement_score":0.9883671,"threshold_uncertainty_score":0.06152141,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5243495068028722,"score_gpt":0.4894084368243749,"score_spread":0.03494106997849727,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}