{"id":"W1561485809","doi":"10.1007/978-3-540-70829-2_11","title":"The Concept of Opposition and Its Use in Q-Learning and Q(λ) Techniques","year":2008,"lang":"en","type":"book-chapter","venue":"Studies in computational intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; Opposition (politics); Q-learning; Artificial intelligence; TRACE (psycholinguistics); Mathematical optimization; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003440209,0.0008115662,0.001133662,0.001100267,0.0009788582,0.002084061,0.001967645,0.001859707,0.004922268],"category_scores_gemma":[0.0118029,0.0005353073,0.0010401,0.00251849,0.008822698,0.005063361,0.003069816,0.005611728,0.001049071],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009336161,"about_ca_system_score_gemma":0.0006150719,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008000039,"about_ca_topic_score_gemma":0.0005011817,"domain_scores_codex":[0.9975997,0.001412101,0.0001043613,0.0002355395,0.0005668092,0.00008151026],"domain_scores_gemma":[0.9930083,0.006010381,0.0002439104,0.0003045457,0.0003161997,0.0001167726],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00001721151,0.00001065786,0.00004775411,0.00005816419,0.000006887235,0.00001909732,0.0000623026,0.004165839,0.0002242922,0.9726042,0.001068337,0.02171517],"study_design_scores_gemma":[0.00001363617,0.0000270886,0.00005242976,0.00003184866,0.00000777199,0.00005885533,0.00001976846,0.03212463,0.0002396839,0.9583342,0.009077661,0.00001245828],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.002282807,0.002780491,0.9695585,0.001198425,0.0003870131,0.00002674131,0.00002321146,0.00007325439,0.02366948],"genre_scores_gemma":[0.3338208,0.006734401,0.6359086,0.001557137,0.001457388,0.0005189157,0.00007810693,0.0002398457,0.0196848],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004922268,"threshold_uncertainty_score":0.01819384,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09245566855671966,"score_gpt":0.3442205430760504,"score_spread":0.2517648745193307,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}