{"id":"W4396860058","doi":"10.1080/01621459.2024.2353948","title":"Generalized Data Thinning Using Sufficient Statistics","year":2024,"lang":"en","type":"article","venue":"Journal of the American Statistical Association","topic":"Statistical Methods and Inference","field":"Mathematics","cited_by":16,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"National Institute of Biomedical Imaging and Bioengineering; National Institute of General Medical Sciences; National Institute on Drug Abuse; Natural Sciences and Engineering Research Council of Canada; Office of Naval Research; W. M. Keck Foundation; National Institutes of Health; National Science Foundation","keywords":"Generalization; Random variable; Mathematics; Thinning; Inference; Sum of normally distributed random variables; Set (abstract data type); Variables; Sample (material); Exponential function; Statistics; Variable (mathematics); Function (biology); Exponential family; Applied mathematics; Computer science; Marginal distribution; Artificial intelligence; Mathematical analysis","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03227434,0.001470133,0.002273961,0.003948736,0.001350065,0.001953651,0.00263576,0.00206088,0.003300019],"category_scores_gemma":[0.07529718,0.001326416,0.003691199,0.002859985,0.004290234,0.005560032,0.005069901,0.004860408,0.0009958252],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001160081,"about_ca_system_score_gemma":0.003613219,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001020679,"about_ca_topic_score_gemma":0.001044197,"domain_scores_codex":[0.9884199,0.006543054,0.000932169,0.001582408,0.002024703,0.0004978226],"domain_scores_gemma":[0.9501343,0.03130663,0.003709303,0.01039704,0.00362136,0.0008314215],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.000254972,0.0001118323,0.004567456,0.0003784895,0.0002295707,0.0006574851,0.0006108158,0.08483502,0.006931154,0.7986179,0.003999785,0.09880559],"study_design_scores_gemma":[0.00007700156,0.0001255883,0.001094219,0.0001584867,0.00004631509,0.0004079591,0.00009462795,0.3552561,0.005337811,0.6317098,0.005626754,0.00006532084],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.004524931,0.0001241072,0.9942577,0.0001477661,0.00002564789,0.00006075351,0.00008955425,0.0001902026,0.0005792867],"genre_scores_gemma":[0.1219817,0.0005914775,0.8728606,0.0005610575,0.0002018528,0.001036857,0.0007683072,0.0004210265,0.001577152],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.03227434,"threshold_uncertainty_score":0.1706851,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1489105408152893,"score_gpt":0.44730948182901,"score_spread":0.2983989410137208,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}