{"id":"W2060652090","doi":"10.1142/s0129183104006662","title":"REINFORCEMENT LEARNING WITH GOAL-DIRECTED ELIGIBILITY TRACES","year":2004,"lang":"en","type":"article","venue":"International Journal of Modern Physics C","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Lethbridge","funders":"","keywords":"Reinforcement learning; TRACE (psycholinguistics); Computer science; Reinforcement; Artificial intelligence; Mechanism (biology); Machine learning; Psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003022638,0.0001499368,0.0001877978,0.00009731649,0.0000723499,0.0002278953,0.001169971,0.00003586693,0.000006654746],"category_scores_gemma":[0.00007877681,0.0001210615,0.0001092527,0.0001535023,0.00004908679,0.001010974,0.0001400783,0.000402596,0.00001409244],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002443833,"about_ca_system_score_gemma":0.0002213878,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001369983,"about_ca_topic_score_gemma":0.000001556081,"domain_scores_codex":[0.9979337,0.00003924984,0.0004626381,0.0001809993,0.001193065,0.0001903871],"domain_scores_gemma":[0.9980654,0.00007259464,0.0006627067,0.0002115674,0.0009013059,0.00008639698],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00005088787,0.00005814593,0.0007466875,0.000003840474,0.0001499147,0.00004197241,0.0009918751,0.9785616,0.001004294,0.007122343,0.00002544988,0.01124299],"study_design_scores_gemma":[0.003008738,0.000766158,0.001913393,0.0002568204,0.00002977828,0.0002106924,0.00004738837,0.9513844,0.01176405,0.02927603,0.0009915489,0.0003510243],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0383935,0.00003531522,0.9587401,0.000794095,0.0003909389,0.00006376076,2.609451e-7,0.00007552104,0.001506473],"genre_scores_gemma":[0.9636542,0.00002860498,0.03573874,0.0001347843,0.0002481162,0.000001538442,0.000002855315,0.00001146012,0.0001797684],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9252607,"threshold_uncertainty_score":0.4936746,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01414252212207897,"score_gpt":0.2760133025861901,"score_spread":0.2618707804641111,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}