{"id":"W1491129446","doi":"10.1007/978-3-540-39857-8_29","title":"Using MDP Characteristics to Guide Exploration in Reinforcement Learning","year":2003,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Markov decision process; Artificial intelligence; Bellman equation; Machine learning; State space; Computation; Focus (optics); Variance (accounting); Markov process; Mathematical optimization; Algorithm","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00168813,0.0008123267,0.000898946,0.0009391892,0.0003536656,0.0007876667,0.0008219551,0.0008403286,0.002182249],"category_scores_gemma":[0.01236352,0.0007095213,0.0004191791,0.0005901018,0.0007775149,0.001973673,0.001186726,0.001689583,0.0002684247],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008333622,"about_ca_system_score_gemma":0.0009269786,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00193916,"about_ca_topic_score_gemma":0.002461202,"domain_scores_codex":[0.9995469,0.0001532493,0.00003746092,0.00005997587,0.0001651996,0.00003720777],"domain_scores_gemma":[0.9938207,0.004707455,0.0004501302,0.0002459969,0.0005851838,0.0001906116],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001073285,0.00003898388,0.0009800561,0.00007888596,0.00001843329,0.00005210157,0.0000611246,0.9256065,0.001664543,0.02298675,0.0006683517,0.04773706],"study_design_scores_gemma":[0.00000718927,0.00001809155,0.00006128137,0.000005857531,0.000002509981,0.0000122723,0.000003868627,0.9933612,0.0003729933,0.005947025,0.0002045917,0.000003006254],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02186935,0.0001281916,0.9755418,0.00013391,0.00002756448,0.00004482998,0.00004537737,0.0003111847,0.001897759],"genre_scores_gemma":[0.687607,0.0002306647,0.3093768,0.00009960523,0.00003475608,0.0002881893,0.000140144,0.0002587412,0.001964065],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.002182249,"threshold_uncertainty_score":0.008927763,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04465666491859917,"score_gpt":0.2856795728748343,"score_spread":0.2410229079562351,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}