{"id":"W3190043208","doi":"10.1037/met0000678","title":"Why multiple hypothesis test corrections provide poor control of false positives in the real world.","year":2024,"lang":"en","type":"article","venue":"Psychological Methods","topic":"Meta-analysis and systematic reviews","field":"Decision Sciences","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Prioris.ai (Canada)","funders":"","keywords":"False positive paradox; Statistical hypothesis testing; False positives and false negatives; Statistics; Test (biology); Econometrics; Psychology; Multiple comparisons problem; Statistical analysis; Control (management); Mathematics; Artificial intelligence; Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3188252,0.002607198,0.003902484,0.007804493,0.001806757,0.005136908,0.009031539,0.004747279,0.01396051],"category_scores_gemma":[0.8352908,0.001998748,0.005044354,0.01079342,0.008544682,0.008909599,0.00443808,0.01110964,0.003152611],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0023002,"about_ca_system_score_gemma":0.005274585,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00329259,"about_ca_topic_score_gemma":0.003449258,"domain_scores_codex":[0.3588706,0.5588945,0.03073232,0.02441565,0.02531583,0.001771053],"domain_scores_gemma":[0.1093463,0.8302451,0.01722204,0.02791462,0.01415834,0.001113693],"domain_codex":"methods","domain_gemma":"methods","domain_candidate":"methods","domain_consensus":"methods","study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.004030934,0.0003276144,0.04070875,0.0291506,0.03159065,0.002000489,0.005886373,0.00714062,0.00284739,0.07724044,0.3675478,0.4315283],"study_design_scores_gemma":[0.004708773,0.001915854,0.04881249,0.02138567,0.01746872,0.002662465,0.00306334,0.08793867,0.013415,0.5831066,0.2144966,0.001025896],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02444841,0.04072838,0.7862716,0.07832686,0.03510316,0.004280677,0.007652974,0.006826248,0.01636167],"genre_scores_gemma":[0.3649772,0.005240191,0.5814423,0.02520673,0.005343996,0.009131646,0.001867212,0.002378675,0.004412096],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6811748,"threshold_uncertainty_score":0.8400098,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.7374936027261868,"score_gpt":0.6119179092144343,"score_spread":0.1255756935117525,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}