{"id":"W4406121228","doi":"10.1007/s11633-023-1482-0","title":"Latent Landmark Graph for Efficient Exploration-exploitation Balance in Hierarchical Reinforcement Learning","year":2025,"lang":"en","type":"article","venue":"Machine Intelligence Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Landmark; Reinforcement learning; Computer science; Reinforcement; Graph; Artificial intelligence; Balance (ability); Machine learning; Cognitive psychology; Psychology; Theoretical computer science; Social psychology; Neuroscience","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001074754,0.0007209634,0.00142878,0.0006664355,0.0005417988,0.000678292,0.001742159,0.001274177,0.005388852],"category_scores_gemma":[0.005311747,0.0005579423,0.0003785616,0.0005853815,0.001158993,0.001627345,0.001810656,0.001665494,0.0005635515],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001018843,"about_ca_system_score_gemma":0.001609877,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004032827,"about_ca_topic_score_gemma":0.006029177,"domain_scores_codex":[0.9995793,0.0001595155,0.00001665348,0.00008443171,0.00008359094,0.00007650557],"domain_scores_gemma":[0.9977876,0.001514095,0.0001719367,0.0001663002,0.0001946281,0.0001654494],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000178085,0.0001059239,0.0006303976,0.00008303803,0.00003271088,0.00005160175,0.00008427727,0.905373,0.002309662,0.04269305,0.002316398,0.04614186],"study_design_scores_gemma":[0.00001274333,0.00001770273,0.00003426302,0.000002975209,0.000002977327,0.000003082854,0.000003130473,0.9914146,0.0001232001,0.008299986,0.00008262662,0.000002653377],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02820509,0.0001104443,0.9693558,0.0001852677,0.00002330517,0.0000452845,0.00007186896,0.0004531817,0.001549784],"genre_scores_gemma":[0.8826362,0.000088127,0.1135389,0.0001408263,0.00002747793,0.0002169522,0.000166868,0.0001458689,0.003038767],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005388852,"threshold_uncertainty_score":0.01802754,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0638528454781579,"score_gpt":0.3792239764301926,"score_spread":0.3153711309520347,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}