{"id":"W7125578405","doi":"10.1109/mecatronics-rem67547.2025.11349467","title":"Learning Multistage Robotic Manipulation Using Chained Options and Composable Subtask Rewards","year":2025,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Interpretability; Task (project management); Benchmark (surveying); Stability (learning theory); Decomposition; Sequence (biology); Process (computing)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.0009160973,0.0004770242,0.0005362108,0.0005666583,0.001428471,0.001281489,0.0006371148,0.0002494162,0.0001316745],"category_scores_gemma":[0.000231174,0.0005213184,0.0001254517,0.001304603,0.0002238206,0.001025931,0.001045523,0.0007262407,0.00004699558],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003499173,"about_ca_system_score_gemma":0.0002657673,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003767629,"about_ca_topic_score_gemma":0.00001043224,"domain_scores_codex":[0.9964542,0.0003942214,0.0009129909,0.0009275898,0.0005198887,0.0007911223],"domain_scores_gemma":[0.9980704,0.0003129919,0.0003715665,0.0007632817,0.0002926339,0.0001891779],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001340004,0.00003974853,0.008224838,0.0001829914,0.00008209849,0.00001154684,0.0007020809,0.9343398,0.001402369,0.05282768,0.0000442473,0.002129242],"study_design_scores_gemma":[0.000882482,0.0001761168,0.01084759,0.0003820533,0.000117479,0.00001806883,0.0002642614,0.9857923,0.0001692286,0.0001982468,0.0006821438,0.0004699925],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0147088,0.0004326027,0.9758695,0.0008046663,0.001235027,0.0006240131,6.347414e-7,0.0002970344,0.00602768],"genre_scores_gemma":[0.822042,0.0001676535,0.1411139,0.0001766826,0.00006277565,0.000005511591,0.00001441809,0.00002656199,0.03639047],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8347557,"threshold_uncertainty_score":0.9998716,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03100935994228316,"score_gpt":0.2912058435706634,"score_spread":0.2601964836283802,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}