{"id":"W4396883491","doi":"10.1145/3665252.3665266","title":"Technical Perspective: Synthetic Data Needs a Reproducibility Benchmark","year":2024,"lang":"en","type":"article","venue":"ACM SIGMOD Record","topic":"Scientific Computing and Data Management","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Benchmark (surveying); Perspective (graphical); Reproducibility; Data mining; Data science; Database; Artificial intelligence; Statistics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1526892,0.001791425,0.002307889,0.00411904,0.002621539,0.01140706,0.006528432,0.005589562,0.01136376],"category_scores_gemma":[0.4922293,0.001046074,0.001924805,0.007323953,0.005634068,0.01563632,0.00788222,0.006647281,0.005660807],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003258057,"about_ca_system_score_gemma":0.007081829,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004236171,"about_ca_topic_score_gemma":0.002371681,"domain_scores_codex":[0.8744061,0.07847718,0.007081236,0.01149587,0.02668077,0.001858828],"domain_scores_gemma":[0.4617153,0.291689,0.0141527,0.1603419,0.06737428,0.00472692],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.002133701,0.001218052,0.03750109,0.003037169,0.001013114,0.001146915,0.002084033,0.1300208,0.01447008,0.4000445,0.1436171,0.2637134],"study_design_scores_gemma":[0.0002865542,0.001023656,0.007141934,0.001576428,0.0002240009,0.001942136,0.001742945,0.2022993,0.01442847,0.6251168,0.1439724,0.0002452428],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"commentary","genre_scores_codex":[0.02786983,0.002922892,0.8752707,0.03900256,0.004691218,0.001143714,0.009299514,0.004425143,0.03537443],"genre_scores_gemma":[0.427343,0.002159189,0.5144117,0.0157364,0.004788112,0.002964767,0.02105487,0.00411565,0.007426377],"genre_candidate":"commentary","genre_consensus":null,"teacher_disagreement_score":0.8473108,"threshold_uncertainty_score":0.8075072,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2209871303754449,"score_gpt":0.4407284741139029,"score_spread":0.219741343738458,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}