{"id":"W4417465482","doi":"10.48550/arxiv.2512.13997","title":"Maximum Mean Discrepancy with Unequal Sample Sizes via Generalized U-Statistics","year":2025,"lang":"","type":"preprint","venue":"ArXiv.org","topic":"Statistical Methods and Inference","field":"Mathematics","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Estimator; Sample size determination; Generalization; Degenerate energy levels; Kernel (algebra); Sample (material); Statistical hypothesis testing; Variance (accounting)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","metaepi_narrow","insufficient_payload"],"consensus_categories":["metaepi_narrow"],"category_scores_codex":[0.001311979,0.001780341,0.002903121,0.0002758449,0.0006115186,0.0002988857,0.001443088,0.0008954447,0.005441587],"category_scores_gemma":[0.008500338,0.001493449,0.0004238152,0.0006569791,0.0008671653,0.0001355308,0.001732389,0.002084788,0.000284626],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003068509,"about_ca_system_score_gemma":0.001023915,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002816428,"about_ca_topic_score_gemma":0.0008642119,"domain_scores_codex":[0.9909371,0.001351748,0.002621916,0.002302918,0.001174888,0.001611465],"domain_scores_gemma":[0.9827217,0.01124833,0.001413153,0.002885239,0.001046062,0.0006855062],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001045426,0.001559817,0.0671045,0.006069707,0.002111246,0.0002559375,0.003242632,0.0002179564,0.0004561111,0.8257715,0.001430329,0.09073487],"study_design_scores_gemma":[0.002713454,0.0008317723,0.0232677,0.002215696,0.00229295,0.00001613765,0.0003887735,0.01851036,0.001411815,0.9424717,0.003277188,0.002602424],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.07395633,0.0002901353,0.9103524,0.000333279,0.001636458,0.001636965,0.00769846,0.0002437631,0.00385221],"genre_scores_gemma":[0.08787411,0.0006405857,0.9066754,0.000449319,0.000504763,0.0003336597,0.0004828151,0.0001824236,0.002856934],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.1167003,"threshold_uncertainty_score":0.9998515,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1049423830570551,"score_gpt":0.3638554360211775,"score_spread":0.2589130529641225,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}