{"id":"W7095254530","doi":"","title":"Regularized Reinforcement Learning with Performance Guarantees","year":2014,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Reinforcement learning; Control (management); Reinforcement; Active learning (machine learning)","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001967413,0.001330892,0.001324946,0.0004436562,0.0004162337,0.001152618,0.001193946,0.001005386,0.007278478],"category_scores_gemma":[0.009936024,0.0004972789,0.0004688789,0.0005018961,0.001148409,0.00155039,0.001849029,0.002806544,0.001295419],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001537806,"about_ca_system_score_gemma":0.001694705,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002593376,"about_ca_topic_score_gemma":0.002155213,"domain_scores_codex":[0.9984245,0.0005798744,0.00006846499,0.000369432,0.0003257715,0.000232039],"domain_scores_gemma":[0.9952031,0.003159539,0.0003783506,0.000573548,0.0003994731,0.0002860012],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005717041,0.0002204081,0.0008604311,0.0001708405,0.00007598331,0.0001119966,0.000075296,0.8403259,0.001460727,0.07030084,0.009773809,0.07605205],"study_design_scores_gemma":[0.00005285024,0.00007050297,0.00009401621,0.00001215808,0.000007180668,0.00001456922,0.000005927744,0.9629948,0.0002760559,0.03586757,0.000598061,0.000006337007],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.03891865,0.001129195,0.9377961,0.00155652,0.0002277865,0.0001413656,0.0002528851,0.001899944,0.01807771],"genre_scores_gemma":[0.925845,0.0003976095,0.06257918,0.0002260284,0.0001699537,0.0002384794,0.000280155,0.0002244776,0.01003905],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007278478,"threshold_uncertainty_score":0.02434891,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.006797794977967789,"score_gpt":0.1919661569729464,"score_spread":0.1851683619949786,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}