{"id":"W4414092373","doi":"10.1561/3600000002","title":"Hypothesis Testing with E-values","year":2025,"lang":"en","type":"article","venue":"Foundations and Trends® in Statistics","topic":"Neural Networks and Applications","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Statistical hypothesis testing; Variety (cybernetics); Null hypothesis; Alternative hypothesis; Core (optical fiber); Test statistic; Value (mathematics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00005021845,0.00006376416,0.00007225947,0.0001370377,0.0001777649,0.0001548121,0.0001291775,0.00001442223,0.000009245156],"category_scores_gemma":[0.00003972152,0.0000536898,0.000005517441,0.0009006726,0.00005064872,0.0001059589,0.00004877173,0.00005653632,0.000003443365],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001252393,"about_ca_system_score_gemma":0.00002341568,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00006192497,"about_ca_topic_score_gemma":0.0002149307,"domain_scores_codex":[0.9995181,0.00001458207,0.0001193953,0.0001804112,0.00005791808,0.0001095976],"domain_scores_gemma":[0.9992398,0.0004761899,0.00003381375,0.0001800252,0.00004576096,0.00002439425],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[7.529963e-7,0.00002646238,0.002516982,0.00000367819,0.000004821396,0.000002625301,0.00003913854,0.0002113988,0.000005895405,0.4219775,0.001428645,0.5737821],"study_design_scores_gemma":[0.0005683626,0.00009111917,0.3375472,0.00007878022,0.0000299697,0.00001320022,0.00004652942,0.3132676,0.00002827001,0.3350087,0.01303785,0.0002824597],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.003712368,0.00003538364,0.9856143,0.001031459,0.00004283449,0.00005256477,0.00002721906,0.00005350486,0.009430353],"genre_scores_gemma":[0.4662128,0.00001145945,0.53263,0.0001002323,0.00001130429,0.00002401372,0.000007954932,0.000002827949,0.0009994241],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.5734997,"threshold_uncertainty_score":0.2189406,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03986766164455913,"score_gpt":0.3050822617816032,"score_spread":0.2652146001370441,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}