{"id":"W7095254530","doi":"","title":"Regularized Reinforcement Learning with Performance Guarantees","year":2014,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Reinforcement learning; Control (management); Reinforcement; Active learning (machine learning)","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004610521,0.0001692144,0.0001699636,0.00009679748,0.0002347605,0.000201618,0.0007367309,0.00004434159,0.00006976554],"category_scores_gemma":[0.00005338778,0.0001236599,0.0000361463,0.0002856868,0.0000519499,0.0005874309,0.0002097858,0.0002251536,0.0002093523],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003355735,"about_ca_system_score_gemma":0.00003475648,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000009105795,"about_ca_topic_score_gemma":8.458794e-7,"domain_scores_codex":[0.9985336,0.00005965008,0.0002412279,0.000305267,0.000487658,0.0003725855],"domain_scores_gemma":[0.9990243,0.00007271138,0.0001302942,0.0006078124,0.00009030881,0.00007454312],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001135791,0.000004428269,0.002106092,0.0000160752,0.00001415457,0.000001093347,0.0002001305,0.9484962,0.000215915,0.04372993,0.0001017423,0.00510295],"study_design_scores_gemma":[0.0006176649,0.0005963442,0.001242701,0.00003869194,0.000004883491,0.00001290758,0.00001554414,0.9807215,0.00182635,0.00003303532,0.01467417,0.0002161656],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01064019,0.000003579963,0.9163108,0.0002588635,0.00009315069,0.0001218463,5.086608e-9,0.0004198,0.07215171],"genre_scores_gemma":[0.8890082,0.00001047281,0.09009469,0.0003229474,0.0000369775,0.000009941005,0.000002096833,0.00001200804,0.02050272],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.878368,"threshold_uncertainty_score":0.5042705,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.006797794977967789,"score_gpt":0.1919661569729464,"score_spread":0.1851683619949786,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}