{"id":"W4287996197","doi":"10.5281/zenodo.3566771","title":"Null hypothesis significance testing interpreted and calibrated by estimating probabilities of sign errors: A Bayes-frequentist continuum","year":2019,"lang":"en","type":"article","venue":"","topic":"Statistical Methods in Clinical Trials","field":"Mathematics","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"Frequentist inference; Null hypothesis; Sign (mathematics); Statistical hypothesis testing; Bayes factor; Statistics; Bayes' theorem; Alternative hypothesis; Econometrics; Null (SQL); Bayesian probability; Mathematics; Computer science; Bayesian inference; Data mining; Mathematical analysis","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.002459825,0.0002799718,0.001010652,0.00005600748,0.00004679732,0.00007061586,0.0002728281,0.000170578,0.0006985202],"category_scores_gemma":[0.1794253,0.0002299575,0.00008793876,0.0002803013,0.0003909277,0.0001383541,0.0001280529,0.0002424696,0.00001590394],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003800123,"about_ca_system_score_gemma":0.0000635472,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001277213,"about_ca_topic_score_gemma":0.00001064436,"domain_scores_codex":[0.9965301,0.0008225667,0.001438221,0.0005411314,0.0003086247,0.0003593882],"domain_scores_gemma":[0.8865956,0.1119594,0.0006116981,0.0004700906,0.0002354816,0.0001276772],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001146306,0.001776543,0.06336708,0.01327454,0.00114354,0.00003587832,0.003009364,0.0000473516,0.6049939,0.1441583,0.02995163,0.1370956],"study_design_scores_gemma":[0.001157987,0.0004852379,0.0003999542,0.0009246908,0.0001147313,0.000006628934,0.0003431296,0.02862229,0.03474486,0.9327396,0.00002379237,0.0004371471],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.5409758,0.00005984608,0.442449,0.0002128833,0.0004818274,0.002331588,0.0002831495,0.0004264451,0.01277945],"genre_scores_gemma":[0.2967449,7.400224e-7,0.7023257,0.00005860561,0.00003220895,0.00004059686,8.873756e-7,0.00004014103,0.0007561789],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.7885813,"threshold_uncertainty_score":0.9377394,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3165564284266273,"score_gpt":0.4439534014564465,"score_spread":0.1273969730298192,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}