{"id":"W4403764486","doi":"10.1080/02664763.2024.2418473","title":"Evaluating the median <i>p</i> -value method for assessing the statistical significance of tests when using multiple imputation","year":2024,"lang":"en","type":"article","venue":"Journal of Applied Statistics","topic":"Statistical Methods and Bayesian Inference","field":"Mathematics","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Institute for Clinical Evaluative Sciences; Sunnybrook Hospital; University of Toronto","funders":"Canadian Institutes of Health Research","keywords":"Statistics; Mathematics; Wilcoxon signed-rank test; Test statistic; Statistical significance; Statistic; Statistical hypothesis testing; Logistic regression; Student's t-test; p-value; F-test; Type I and type II errors; Pearson's chi-squared test; Linear regression; Pooling; Nominal level; Imputation (statistics); Missing data; Mann–Whitney U test; Computer science; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1520058,0.002442303,0.004858535,0.008549628,0.002238934,0.00607126,0.006027289,0.005165371,0.0124379],"category_scores_gemma":[0.4841022,0.00137533,0.006319019,0.008681114,0.005165964,0.004550032,0.004495603,0.00910665,0.002913545],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002098812,"about_ca_system_score_gemma":0.006632147,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002193995,"about_ca_topic_score_gemma":0.001517449,"domain_scores_codex":[0.8497927,0.1092894,0.008306933,0.01466308,0.016545,0.001402792],"domain_scores_gemma":[0.4547684,0.4818129,0.01816737,0.0280366,0.01563223,0.001582416],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002749269,0.0006722549,0.04732412,0.007508279,0.01002799,0.001762043,0.001949548,0.04199168,0.004676865,0.1422543,0.05899676,0.6800869],"study_design_scores_gemma":[0.001220496,0.003137971,0.03166891,0.004885089,0.005013384,0.004854445,0.001388303,0.3757039,0.02587084,0.4057844,0.1394807,0.0009915801],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.004659166,0.001880666,0.9860761,0.0007283763,0.0004116376,0.0008273226,0.001202545,0.001607197,0.002607078],"genre_scores_gemma":[0.08338178,0.0008359567,0.907077,0.000952997,0.00038019,0.003621657,0.001423417,0.001143702,0.001183255],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8479942,"threshold_uncertainty_score":0.8038929,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1524259777304091,"score_gpt":0.5050042235663473,"score_spread":0.3525782458359382,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}