{"id":"W4408810302","doi":"10.1021/acs.iecr.4c04900","title":"Recurrent Reinforcement Learning Strategy with a Parameterized Agent for Online Scheduling of a State Task Network Under Uncertainty","year":2025,"lang":"en","type":"article","venue":"Industrial & Engineering Chemistry Research","topic":"Scheduling and Optimization Algorithms","field":"Engineering","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"Consejo Nacional de Ciencia y Tecnología","keywords":"Reinforcement learning; Parameterized complexity; Computer science; Scheduling (production processes); Task (project management); Reinforcement; Artificial intelligence; Error-driven learning; State (computer science); Machine learning; Mathematical optimization; Algorithm; Psychology; Mathematics; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008073047,0.000617541,0.0005781243,0.0002385556,0.0002973721,0.0005560059,0.000898357,0.0006813,0.00177349],"category_scores_gemma":[0.002077235,0.0002822495,0.0003801176,0.00017059,0.0006491888,0.0007210768,0.0006941305,0.0008426833,0.0002462559],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006700302,"about_ca_system_score_gemma":0.0008749628,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004606987,"about_ca_topic_score_gemma":0.003783426,"domain_scores_codex":[0.9997208,0.0000831417,0.00001606654,0.00006862773,0.00006778829,0.00004353408],"domain_scores_gemma":[0.9993216,0.0003372789,0.0001269755,0.0000663391,0.00009945656,0.00004833891],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00005703155,0.00003176603,0.000294906,0.00002083464,0.00002288519,0.00006688247,0.00004527924,0.9740629,0.002508511,0.006935495,0.0002618072,0.01569166],"study_design_scores_gemma":[0.000004708633,0.00001399642,0.00002050702,8.972735e-7,0.000002591894,0.000004671821,0.00000158387,0.99897,0.0002397357,0.0006328613,0.0001067754,0.000001558683],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05178801,0.00009666511,0.9441185,0.0001400597,0.00002876503,0.00005251092,0.00002785347,0.0005100025,0.003237708],"genre_scores_gemma":[0.9393923,0.00004338396,0.05848142,0.00004515078,0.00001431754,0.00009319724,0.00003513058,0.00003176197,0.00186347],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004606987,"threshold_uncertainty_score":0.00916034,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07084368239796389,"score_gpt":0.3286161707403337,"score_spread":0.2577724883423698,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}