{"id":"W2156760524","doi":"10.1109/ijcnn.1999.833417","title":"Adaptive exploration in reinforcement learning","year":2003,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Guelph; University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; Implementation; Artificial intelligence; Connectionism; Machine learning; Reinforcement; Learning classifier system; Artificial neural network; Engineering; Software engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001302574,0.0006420457,0.0007314996,0.0003063797,0.0003449939,0.0009865909,0.0009043261,0.001055474,0.002111112],"category_scores_gemma":[0.004919053,0.000281034,0.0003741045,0.0003784706,0.001729246,0.001271684,0.001008979,0.001461612,0.0003264735],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009738379,"about_ca_system_score_gemma":0.0007336767,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001750433,"about_ca_topic_score_gemma":0.001154547,"domain_scores_codex":[0.9993497,0.0003157757,0.00003337494,0.00009577528,0.0001485934,0.00005692568],"domain_scores_gemma":[0.9984972,0.001079271,0.0001200674,0.00008465056,0.000148915,0.00006987312],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00008075932,0.00005936024,0.0006732265,0.000167985,0.00007409656,0.0001216952,0.0001482305,0.6581506,0.001372147,0.2795694,0.001851522,0.05773098],"study_design_scores_gemma":[0.00004673173,0.0000497966,0.0001048013,0.00001844744,0.00001192128,0.00003004619,0.00001258312,0.8238552,0.0004731224,0.1727851,0.002601038,0.0000112061],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01195385,0.001752006,0.9760165,0.000580253,0.0001103849,0.00004401172,0.00002572855,0.0002160049,0.009301226],"genre_scores_gemma":[0.8415294,0.00187273,0.1490258,0.0002874199,0.0001902393,0.0003394948,0.00005738336,0.0000623718,0.006635262],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002111112,"threshold_uncertainty_score":0.007065713,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03664087277169577,"score_gpt":0.2510125592442868,"score_spread":0.214371686472591,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}