{"id":"W4417465482","doi":"10.48550/arxiv.2512.13997","title":"Maximum Mean Discrepancy with Unequal Sample Sizes via Generalized U-Statistics","year":2025,"lang":"","type":"preprint","venue":"ArXiv.org","topic":"Statistical Methods and Inference","field":"Mathematics","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Estimator; Sample size determination; Generalization; Degenerate energy levels; Kernel (algebra); Sample (material); Statistical hypothesis testing; Variance (accounting)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0540245,0.001548241,0.00358942,0.004465797,0.001572673,0.003582659,0.004172237,0.003013783,0.002540113],"category_scores_gemma":[0.2472029,0.001513013,0.002567985,0.003368224,0.01041061,0.006887993,0.008517281,0.00568958,0.0006780903],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001929689,"about_ca_system_score_gemma":0.002314224,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001042297,"about_ca_topic_score_gemma":0.0006978086,"domain_scores_codex":[0.9430697,0.04399667,0.001559883,0.004814064,0.005478902,0.001080708],"domain_scores_gemma":[0.7478274,0.2198834,0.008272796,0.01759952,0.004959282,0.001457679],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0003127774,0.0001071426,0.006363591,0.0003388239,0.0003031507,0.0003885727,0.0006143836,0.07741418,0.001981138,0.8296519,0.002517987,0.08000637],"study_design_scores_gemma":[0.00007794044,0.0001457151,0.0009388549,0.0001057415,0.00004262422,0.0001946446,0.00007487421,0.253221,0.00161146,0.7418393,0.001682543,0.00006522696],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.006505062,0.0002198811,0.9919518,0.0002535872,0.00002640787,0.00005337875,0.00005225882,0.0001908763,0.0007467506],"genre_scores_gemma":[0.3221421,0.0005677121,0.6722008,0.0009984637,0.0003442086,0.00142197,0.0003802255,0.0004940369,0.001450538],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.0540245,"threshold_uncertainty_score":0.2857123,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1049423830570551,"score_gpt":0.3638554360211775,"score_spread":0.2589130529641225,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}