{"id":"W2403054493","doi":"10.1371/journal.pone.0180234","title":"A neural model of hierarchical reinforcement learning","year":2017,"lang":"en","type":"article","venue":"PLoS ONE","topic":"Neural Networks and Applications","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"Office of Naval Research; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs; Air Force Office of Scientific Research; Ontario Innovation Trust","keywords":"Reinforcement learning; Artificial intelligence; Computer science; Leverage (statistics); Transfer of learning; Artificial neural network; Reinforcement; Cognitive science; Machine learning; Psychology","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00005476188,0.00004379289,0.00008874972,0.00001541707,0.0002235841,0.00007016474,0.000642644,0.00001704559,0.000004166084],"category_scores_gemma":[0.00002214623,0.00004039762,0.00002607213,0.00002846379,0.00003989821,0.0001729022,0.0002850804,0.000120665,0.000008427129],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000004635006,"about_ca_system_score_gemma":0.00001133293,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000006997803,"about_ca_topic_score_gemma":9.130388e-7,"domain_scores_codex":[0.999466,0.000008165794,0.0001060869,0.0001290329,0.000174272,0.000116438],"domain_scores_gemma":[0.9993092,0.00001920296,0.00009252477,0.000499888,0.00003243679,0.00004672005],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001821805,0.0009328297,0.002946118,0.00007541282,0.00009923797,0.000004212005,0.0005854868,0.380977,0.1459869,0.4361959,0.0003280918,0.03185068],"study_design_scores_gemma":[0.00008665133,0.00003691198,0.0004012635,0.00001861965,0.000005124074,2.137023e-7,5.772613e-7,0.9906884,0.00602269,0.002677529,0.00001657259,0.00004540889],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6280918,0.00001706289,0.3600191,0.004431246,0.0000192733,0.0001815696,5.681287e-7,0.00008333158,0.007156025],"genre_scores_gemma":[0.9794058,0.00001296606,0.0195771,0.00008149434,0.00003428294,0.000014841,6.549067e-7,0.00000301215,0.0008698592],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6097115,"threshold_uncertainty_score":0.1719651,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08724812942187102,"score_gpt":0.2616324953613466,"score_spread":0.1743843659394756,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}