{"id":"W2943366292","doi":"10.1038/s42256-019-0053-0","title":"Moving beyond reward prediction errors","year":2019,"lang":"en","type":"article","venue":"Nature Machine Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"The Scarborough Hospital; Canadian Institute for Advanced Research; University of Toronto; Vector Institute","funders":"","keywords":"Business; Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004187543,0.00137598,0.001742017,0.0008373221,0.0006879446,0.002695453,0.002900707,0.003061139,0.01254148],"category_scores_gemma":[0.03099857,0.0008339726,0.0006537645,0.0006868813,0.002684981,0.01103135,0.002754678,0.007480556,0.001828581],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001418739,"about_ca_system_score_gemma":0.00153984,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003999275,"about_ca_topic_score_gemma":0.00308674,"domain_scores_codex":[0.997806,0.0008681104,0.00008532252,0.0005168609,0.0005547218,0.0001689421],"domain_scores_gemma":[0.9813778,0.01350239,0.0008405723,0.001917933,0.001780075,0.0005811892],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0003129074,0.0002726269,0.002643301,0.0003227545,0.0001998206,0.000138491,0.0001894376,0.2697171,0.001544262,0.4979624,0.0124949,0.2142019],"study_design_scores_gemma":[0.00003160007,0.00006192661,0.0003383668,0.00006866611,0.00002770066,0.00002564378,0.00002460385,0.6198157,0.0006956669,0.3755282,0.003360135,0.00002185934],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02969722,0.003816278,0.9216798,0.01391781,0.001136726,0.00006096882,0.0001965834,0.0007753526,0.02871935],"genre_scores_gemma":[0.8509098,0.001883969,0.1196423,0.002333247,0.0007710717,0.0000915194,0.0001923081,0.0004062596,0.02376951],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01254148,"threshold_uncertainty_score":0.04195541,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.006224225736806918,"score_gpt":0.2522100027226873,"score_spread":0.2459857769858804,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}