{"id":"W4404691850","doi":"10.4171/owr/2024/24","title":"Game-theoretic Statistical Inference: Optional Sampling, Universal Inference, and Multiple Testing Based on E-values","year":2024,"lang":"en","type":"article","venue":"Oberwolfach Reports","topic":"Statistical Methods in Clinical Trials","field":"Mathematics","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Inference; Statistical inference; Computer science; Martingale (probability theory); Notation; Statistical hypothesis testing; Artificial intelligence; Machine learning; Interpretation (philosophy); Meaning (existential); Game theory; Theoretical computer science; Data science; Mathematical economics; Mathematics; Epistemology; Statistics; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.003333564,0.0003441693,0.00060641,0.0001804862,0.0001333296,0.0002398117,0.000130441,0.0002435933,0.000762187],"category_scores_gemma":[0.2266953,0.0002914008,0.00009719719,0.0003277038,0.0005472601,0.00009657833,0.0001271427,0.0006341352,0.00003318299],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001044263,"about_ca_system_score_gemma":0.0003426243,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002587093,"about_ca_topic_score_gemma":0.000002487866,"domain_scores_codex":[0.9962389,0.0005210633,0.001184729,0.0008620829,0.0007535138,0.0004396671],"domain_scores_gemma":[0.8681273,0.1306479,0.0002473832,0.0004920083,0.0001958809,0.0002895393],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.000126384,0.0003957146,0.0343693,0.0009926712,0.0001144845,0.002097361,0.00017851,0.0006779871,0.0002266333,0.9108768,0.001014883,0.04892927],"study_design_scores_gemma":[0.0003240018,0.0002468862,0.005854554,0.000672092,0.0001352566,0.00007107822,0.00002466118,0.09959272,0.00009713323,0.8920559,0.0005941678,0.0003315143],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07506264,0.00005668789,0.9105698,0.000167928,0.001131653,0.0006201828,0.0002462911,0.0005580748,0.01158668],"genre_scores_gemma":[0.5228798,0.000005570877,0.4766752,0.00005496714,0.0001501715,0.00001579517,0.00001596847,0.00003674396,0.0001657651],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.4478171,"threshold_uncertainty_score":0.9999538,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4245189506016742,"score_gpt":0.526775350920045,"score_spread":0.1022564003183708,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}