{"id":"W2114500662","doi":"10.1142/s0129183101002851","title":"DEEP-SARSA: A REINFORCEMENT LEARNING ALGORITHM FOR AUTONOMOUS NAVIGATION","year":2001,"lang":"en","type":"article","venue":"International Journal of Modern Physics C","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Lethbridge","funders":"","keywords":"Reinforcement learning; Computer science; Convergence (economics); Algorithm; Artificial intelligence; Q-learning; Learning classifier system; Robot; Graph; Theoretical computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008289851,0.0005813434,0.0006794973,0.0003413894,0.0002773577,0.0004490825,0.001015674,0.0009608429,0.0024059],"category_scores_gemma":[0.001850576,0.0003097966,0.0003526215,0.000273239,0.0007573672,0.0008278961,0.0008182122,0.001357471,0.0005220014],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000561672,"about_ca_system_score_gemma":0.001162666,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002330818,"about_ca_topic_score_gemma":0.002228562,"domain_scores_codex":[0.9997813,0.00007242089,0.00001244038,0.00004597359,0.00006369357,0.00002423262],"domain_scores_gemma":[0.9993783,0.0003350431,0.00005831467,0.000051694,0.0001321404,0.00004457806],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001320189,0.00007554646,0.0004919454,0.00009470157,0.00004457365,0.0000498249,0.00005191534,0.8124688,0.004831839,0.03161855,0.002290477,0.1478498],"study_design_scores_gemma":[0.00001452795,0.00003176314,0.0000293828,0.000003769246,0.000002872785,0.00001114841,0.000002380121,0.9925072,0.0006942292,0.005738156,0.0009609383,0.000003674253],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.004075816,0.0001136481,0.9945765,0.00007766375,0.00003861788,0.00002178768,0.00001524861,0.0003830751,0.0006976541],"genre_scores_gemma":[0.3996918,0.0002417412,0.595279,0.0001847989,0.00004416664,0.0002179025,0.0001093856,0.0001250581,0.00410613],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0024059,"threshold_uncertainty_score":0.008048594,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01994058600414072,"score_gpt":0.2823315718161346,"score_spread":0.2623909858119939,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}