{"id":"W2011033111","doi":"10.1109/adprl.2007.368201","title":"Opposition-Based Q(&amp;#x003BB;) with Non-Markovian Update","year":2007,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Opposition (politics); Computer science; Markov process; Algorithm; Theoretical computer science; Statistical physics; Mathematics; Physics; Law; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001599568,0.0005190344,0.00071123,0.0006648856,0.0004405794,0.0007067941,0.002146213,0.0009444759,0.004873447],"category_scores_gemma":[0.00673325,0.0003051132,0.0005014607,0.0006766701,0.0008555293,0.001332284,0.001512413,0.001223426,0.0007321712],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000636614,"about_ca_system_score_gemma":0.001787281,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004231357,"about_ca_topic_score_gemma":0.004601266,"domain_scores_codex":[0.9989377,0.0003203262,0.00006560294,0.0001598317,0.0004179738,0.00009863339],"domain_scores_gemma":[0.9967803,0.002077762,0.0002170706,0.0002927538,0.0004895858,0.0001424711],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004549049,0.0004129194,0.001549782,0.0002428984,0.0000765867,0.0001512053,0.0002288051,0.1845423,0.008106159,0.0594806,0.004533084,0.7402207],"study_design_scores_gemma":[0.00007534815,0.0001833455,0.0003021395,0.0000121725,0.00001680701,0.00009428253,0.00002039094,0.9761562,0.003357383,0.01665371,0.003111585,0.00001655958],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.008023033,0.00006622076,0.9893561,0.00008116649,0.00004794302,0.00008512123,0.00001864235,0.0004815573,0.001840318],"genre_scores_gemma":[0.4063289,0.0001267551,0.5864031,0.0003270361,0.00006107147,0.0003357237,0.0001127362,0.0001975267,0.006107116],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004873447,"threshold_uncertainty_score":0.0163033,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.009002528772758794,"score_gpt":0.2387725281728582,"score_spread":0.2297699994000994,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}