{"id":"W4415886174","doi":"10.48550/arxiv.2510.15839","title":"Learning Correlated Reward Models: Statistical Barriers and Opportunities","year":2025,"lang":"","type":"preprint","venue":"ArXiv.org","topic":"Recommender Systems and Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Office of Naval Research; Multidisciplinary University Research Initiative; National Science Foundation","keywords":"Pairwise comparison; Preference; Statistical model; Estimator; Preference learning; Reinforcement learning; Range (aeronautics); Key (lock); Independence (probability theory)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.001610833,0.0008833813,0.001243045,0.0004480547,0.0008001421,0.0007234348,0.001491151,0.0008888772,0.0001435318],"category_scores_gemma":[0.0002729117,0.0009225658,0.0002127735,0.0002845997,0.0004294846,0.0007192653,0.003913536,0.002584871,0.00002542587],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002063921,"about_ca_system_score_gemma":0.001096754,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009639951,"about_ca_topic_score_gemma":0.00001645768,"domain_scores_codex":[0.994198,0.001038574,0.001418951,0.001881621,0.0005546698,0.0009081916],"domain_scores_gemma":[0.9962566,0.0005952148,0.0006368049,0.001360686,0.0003964495,0.0007542104],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001618083,0.0002574646,0.2147499,0.003423396,0.001808074,0.001170646,0.03086715,0.005989324,0.00004770783,0.3634281,0.02344171,0.3546546],"study_design_scores_gemma":[0.0008898239,0.0007063429,0.006890913,0.003262439,0.0002789275,0.0001401288,0.002963212,0.8841624,0.0001635436,0.02901976,0.06925553,0.002267036],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05884505,0.002031652,0.9112725,0.002103232,0.002957001,0.0009235044,0.00009298147,0.0009089731,0.02086516],"genre_scores_gemma":[0.9724108,0.007608214,0.009679196,0.0006420458,0.0001355832,0.0001613266,0.00006362866,0.00004952433,0.009249681],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9135658,"threshold_uncertainty_score":0.9997162,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1148276253556892,"score_gpt":0.2929756032361907,"score_spread":0.1781479778805015,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}