{"id":"W3198238708","doi":"10.48550/arxiv.2102.05196","title":"Once is Never Enough: Foundations for Sound Statistical Inference in Tor\\n Network Experimentation","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Sports Analytics and Performance","field":"Economics, Econometrics and Finance","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Inference; Sound (geography); Computer science; Statistical inference; Artificial intelligence; Mathematics; Statistics; Acoustics; Physics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.04006777,0.001222559,0.001978017,0.002724314,0.001704936,0.003876649,0.004468953,0.002394144,0.004867551],"category_scores_gemma":[0.234675,0.00137667,0.0021191,0.001993491,0.007929156,0.008387811,0.005234947,0.006683658,0.0005677203],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002701486,"about_ca_system_score_gemma":0.005032748,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007614916,"about_ca_topic_score_gemma":0.005686263,"domain_scores_codex":[0.9816278,0.01239565,0.0008274837,0.001903688,0.002650282,0.0005952626],"domain_scores_gemma":[0.7699615,0.2011807,0.006491443,0.01560361,0.005639635,0.001123105],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00006486901,0.000070179,0.003534728,0.0001540568,0.0001076514,0.0001778989,0.0004232137,0.2757589,0.0005353919,0.6977699,0.001156405,0.02024669],"study_design_scores_gemma":[0.00002096993,0.00003666405,0.0002824246,0.0000494331,0.00001348565,0.0000333029,0.00007508461,0.6154784,0.0004569761,0.3823843,0.001148254,0.00002072246],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.003926734,0.00008556232,0.9929646,0.0007228715,0.00003464486,0.00008491873,0.0001355997,0.0001939081,0.001851101],"genre_scores_gemma":[0.3060546,0.000462982,0.6892005,0.0006713673,0.0002372932,0.001162258,0.0005023459,0.0002600803,0.001448635],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9599322,"threshold_uncertainty_score":0.2119011,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1885894387928196,"score_gpt":0.2435710650988728,"score_spread":0.05498162630605324,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}