{"id":"W7125578405","doi":"10.1109/mecatronics-rem67547.2025.11349467","title":"Learning Multistage Robotic Manipulation Using Chained Options and Composable Subtask Rewards","year":2025,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Interpretability; Task (project management); Benchmark (surveying); Stability (learning theory); Decomposition; Sequence (biology); Process (computing)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001111428,0.0007816501,0.0006786469,0.0002842978,0.0003176229,0.0005483426,0.001074476,0.0008850785,0.001688278],"category_scores_gemma":[0.003256041,0.0005152915,0.000513519,0.00020088,0.00117824,0.001220198,0.001244759,0.001428058,0.0001922933],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008434979,"about_ca_system_score_gemma":0.0009670187,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002966829,"about_ca_topic_score_gemma":0.004360643,"domain_scores_codex":[0.9996037,0.0001372645,0.00002154644,0.00009775322,0.00007455178,0.00006518354],"domain_scores_gemma":[0.9987212,0.000798144,0.0001384844,0.000124475,0.0000960598,0.000121625],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00007554027,0.00003888005,0.0006686957,0.000023853,0.00001713134,0.00004893465,0.00004816882,0.9730584,0.002153526,0.005163681,0.0001490638,0.01855419],"study_design_scores_gemma":[0.000005685412,0.000017299,0.00004891072,0.000002229412,0.000001854528,0.00000337216,0.000002546345,0.9961531,0.000361095,0.003350353,0.00005152535,0.000002113293],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1195995,0.000141805,0.8777695,0.0001926372,0.00002000476,0.00006515635,0.00003889495,0.0004529544,0.00171942],"genre_scores_gemma":[0.933774,0.00005214471,0.06454057,0.00005499738,0.000007457473,0.0001048194,0.00004493381,0.00003940792,0.001381696],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002966829,"threshold_uncertainty_score":0.006120026,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03100935994228316,"score_gpt":0.2912058435706634,"score_spread":0.2601964836283802,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}