{"id":"W1982723861","doi":"10.1145/2746539.2746580","title":"Preserving Statistical Validity in Adaptive Data Analysis","year":2015,"lang":"en","type":"article","venue":"","topic":"Machine Learning and Data Classification","field":"Computer Science","cited_by":265,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"National Science Foundation","keywords":"Spurious relationship; Computer science; Statistical inference; Statistical hypothesis testing; Inference; Data mining; Machine learning; Statistical analysis; Data collection; Data science; Artificial intelligence; Statistics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.3153098,0.00270875,0.004944179,0.005841988,0.003687107,0.009346747,0.008212823,0.006346032,0.002656762],"category_scores_gemma":[0.707212,0.003133912,0.004145336,0.006106342,0.03301642,0.01192173,0.01676875,0.01729901,0.0009878281],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003651798,"about_ca_system_score_gemma":0.01366499,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002393279,"about_ca_topic_score_gemma":0.001468515,"domain_scores_codex":[0.5915508,0.3375226,0.01747462,0.02450914,0.02694451,0.001998379],"domain_scores_gemma":[0.1888938,0.7083583,0.01327981,0.07394217,0.01395229,0.00157372],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001042547,0.0002649288,0.01635722,0.002331981,0.002153899,0.001162664,0.005549012,0.05474374,0.002928523,0.6843786,0.005684813,0.2234022],"study_design_scores_gemma":[0.0002533577,0.00023089,0.001200268,0.0004024568,0.0001354536,0.0002811208,0.0002109517,0.07587625,0.001585073,0.9141186,0.005629152,0.00007638751],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.002862438,0.001050102,0.990262,0.003709746,0.0002366005,0.0002814292,0.00009652582,0.0002624104,0.001238747],"genre_scores_gemma":[0.1736735,0.001352459,0.8138074,0.004738664,0.001348085,0.003405704,0.0003643459,0.0004512386,0.0008585543],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.3153098,"threshold_uncertainty_score":0.844345,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3190507828462666,"score_gpt":0.3803631002506314,"score_spread":0.06131231740436482,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}