{"id":"W4415886174","doi":"10.48550/arxiv.2510.15839","title":"Learning Correlated Reward Models: Statistical Barriers and Opportunities","year":2025,"lang":"","type":"preprint","venue":"ArXiv.org","topic":"Recommender Systems and Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Office of Naval Research; Multidisciplinary University Research Initiative; National Science Foundation","keywords":"Pairwise comparison; Preference; Statistical model; Estimator; Preference learning; Reinforcement learning; Range (aeronautics); Key (lock); Independence (probability theory)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02565627,0.001522382,0.003883449,0.001287275,0.001484465,0.004424776,0.004557961,0.003483181,0.002532417],"category_scores_gemma":[0.1212131,0.002125139,0.001870862,0.002223246,0.004229065,0.01093118,0.006199071,0.01080127,0.0006935433],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002747208,"about_ca_system_score_gemma":0.003071536,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004843093,"about_ca_topic_score_gemma":0.005160279,"domain_scores_codex":[0.9855296,0.01026579,0.0004806498,0.001925954,0.00129928,0.0004988505],"domain_scores_gemma":[0.83213,0.1465793,0.004992539,0.01197355,0.002838406,0.001486329],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0004177725,0.000315898,0.006284548,0.0003011575,0.0002509622,0.0002512079,0.0005778986,0.3867746,0.0004353223,0.5471197,0.004241479,0.05302949],"study_design_scores_gemma":[0.00003167074,0.00004397651,0.0002717972,0.00003766963,0.00001450493,0.00003475861,0.00003879944,0.7238992,0.0001546903,0.2746286,0.0008268263,0.00001757016],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02755001,0.0010367,0.965402,0.003560169,0.00004659767,0.00007727333,0.0002286738,0.0002748915,0.001823685],"genre_scores_gemma":[0.6506428,0.002249719,0.3386991,0.001922247,0.0005238279,0.0007533561,0.00105429,0.0003360032,0.003818613],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02565627,"threshold_uncertainty_score":0.1356849,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1148276253556892,"score_gpt":0.2929756032361907,"score_spread":0.1781479778805015,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}