{"id":"W7124143215","doi":"10.65109/vuye5463","title":"Optimal policy switching algorithms for reinforcement learning","year":2010,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Task (project management); Q-learning; Function (biology); Markov decision process; Function approximation; Control (management); Optimal control","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.001988269,0.0007767571,0.0006341972,0.0006649547,0.001535683,0.001911075,0.002549323,0.0004678855,0.000544537],"category_scores_gemma":[0.001534696,0.0007899889,0.0004769575,0.0009877414,0.0001613724,0.001534729,0.001509383,0.002287864,0.0004139692],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002128388,"about_ca_system_score_gemma":0.001051287,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002960952,"about_ca_topic_score_gemma":0.000008868184,"domain_scores_codex":[0.9940652,0.0001012394,0.001403186,0.001267921,0.001148382,0.002014019],"domain_scores_gemma":[0.9960197,0.0005525657,0.0007669469,0.001484413,0.0005448512,0.0006315556],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002262927,0.00003389284,0.0001235614,0.00007517273,0.00009356697,0.000005476261,0.001887985,0.8313979,0.003124156,0.1019223,0.0003588872,0.06095447],"study_design_scores_gemma":[0.001413528,0.001017601,0.00008314721,0.00006040045,0.00004467571,0.00003915851,0.0001698113,0.9263088,0.002617862,0.0001655494,0.06718995,0.0008895406],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.001737393,0.00002756838,0.9678888,0.002869974,0.003937313,0.001303565,8.053345e-7,0.0004803246,0.02175423],"genre_scores_gemma":[0.5691891,0.00004668015,0.3857586,0.0007348288,0.001937439,0.00007155426,0.00001398142,0.0000816107,0.04216613],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.5821302,"threshold_uncertainty_score":0.9997642,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02392354187161438,"score_gpt":0.3013734190586969,"score_spread":0.2774498771870825,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}