{"id":"W3034833014","doi":"","title":"OPtions as REsponses: Grounding behavioural hierarchies in multi-agent reinforcement learning","year":2020,"lang":"en","type":"article","venue":"International Conference on Machine Learning","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Ground; Reinforcement; Artificial intelligence; Cognitive psychology; Human–computer interaction; Cognitive science; Psychology; Engineering; Social psychology; Electrical engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00235586,0.0005889152,0.0007977224,0.0005255658,0.0005544263,0.001251907,0.001783505,0.001295395,0.004822354],"category_scores_gemma":[0.0120798,0.0006077595,0.0005185204,0.000378973,0.002154014,0.002576538,0.002335364,0.002115828,0.0002884285],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008961641,"about_ca_system_score_gemma":0.0008729218,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004308442,"about_ca_topic_score_gemma":0.005229015,"domain_scores_codex":[0.9989789,0.0005755264,0.00004475999,0.0001683209,0.0001133575,0.0001192106],"domain_scores_gemma":[0.9942457,0.003876978,0.0004724062,0.0005877593,0.0004025082,0.0004146638],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002514696,0.0001486217,0.002437349,0.0001123917,0.00006847967,0.0001156879,0.0005798488,0.8209447,0.002804999,0.1165396,0.0009337448,0.05506302],"study_design_scores_gemma":[0.00002487368,0.0000370703,0.0001784757,0.00001056043,0.00000810479,0.000006519702,0.00004140136,0.9271498,0.0002646463,0.07206038,0.0002095244,0.00000854415],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.142398,0.0001437292,0.8498492,0.0007890675,0.00006091486,0.00008418212,0.00009800489,0.0005121082,0.006064921],"genre_scores_gemma":[0.9439145,0.00004596342,0.05449433,0.0001012665,0.00001094629,0.00007879031,0.00005388874,0.00004504695,0.001255259],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004822354,"threshold_uncertainty_score":0.01613235,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1289016132082067,"score_gpt":0.3483189519549212,"score_spread":0.2194173387467145,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}