{"id":"W1987762341","doi":"10.1177/0959354307086923","title":"Why <i>P</i> Values Are Not a Useful Measure of Evidence in Statistical Significance Testing","year":2008,"lang":"en","type":"article","venue":"Theory & Psychology","topic":"Forecasting Techniques and Applications","field":"Decision Sciences","cited_by":217,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Lethbridge","funders":"","keywords":"p-value; Replication (statistics); Null hypothesis; Statistical hypothesis testing; Value (mathematics); Psychology; Measure (data warehouse); Social psychology; Statistics; Epistemology; Cognitive psychology; Econometrics; Computer science; Mathematics; Philosophy; Data mining","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.4277495,0.002163224,0.00752461,0.01366426,0.003453837,0.01757966,0.01212152,0.01886989,0.004871442],"category_scores_gemma":[0.8163325,0.002583438,0.003700171,0.01552608,0.0389462,0.02115386,0.007900916,0.03016366,0.005052073],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006078237,"about_ca_system_score_gemma":0.006853686,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001967555,"about_ca_topic_score_gemma":0.00136833,"domain_scores_codex":[0.4353653,0.366462,0.0754908,0.02356566,0.09608144,0.003034861],"domain_scores_gemma":[0.0660544,0.8404399,0.02720918,0.02271778,0.04164651,0.001932277],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001321728,0.0002875279,0.01096134,0.02895945,0.003729056,0.002078554,0.01018311,0.002193375,0.002370874,0.2411213,0.3975803,0.2992134],"study_design_scores_gemma":[0.0002671972,0.0004799991,0.006430431,0.02168602,0.0007525134,0.002275906,0.003434561,0.004778167,0.00288047,0.7775972,0.1789191,0.00049836],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"commentary","genre_gemma":"methods","genre_scores_codex":[0.007181181,0.1244582,0.2144921,0.578364,0.05104438,0.0007676994,0.001401325,0.001769253,0.02052187],"genre_scores_gemma":[0.2292698,0.04003691,0.3997692,0.2781398,0.04102007,0.005225912,0.000706045,0.001900697,0.003931654],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.5722505,"threshold_uncertainty_score":0.7056868,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.6032423430011634,"score_gpt":0.4851480948478012,"score_spread":0.1180942481533622,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}