{"id":"W4413211058","doi":"10.2139/ssrn.5387658","title":"Experimental Study on the Effect of Multi-Step Deep Reinforcement Learning in Pomdps","year":2025,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Space Satellite Systems and Control","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Reinforcement; Computer science; Artificial intelligence; Mathematical optimization; Psychology; Mathematics; Social psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001530301,0.0008966479,0.0006338632,0.0002377862,0.0003548872,0.0006044132,0.001042612,0.001027768,0.006603903],"category_scores_gemma":[0.01318895,0.0003278236,0.0003811703,0.000224837,0.0007136865,0.001187463,0.001011362,0.002161939,0.0003267683],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005803904,"about_ca_system_score_gemma":0.0007934576,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003193544,"about_ca_topic_score_gemma":0.002036066,"domain_scores_codex":[0.9993716,0.0002374785,0.0000488385,0.0001289954,0.00009522818,0.000117982],"domain_scores_gemma":[0.9874001,0.010089,0.0004911565,0.001064809,0.0004953071,0.0004595058],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.005157566,0.004256112,0.00361061,0.0009731765,0.000206682,0.00021756,0.000230391,0.8936442,0.0243386,0.01005148,0.001737787,0.05557584],"study_design_scores_gemma":[0.000258059,0.001192115,0.0009460666,0.00002614973,0.00002975161,0.0000201954,0.00004303347,0.9876517,0.006232261,0.003252349,0.0003348712,0.00001338417],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9356501,0.0004791535,0.05545266,0.0003908613,0.0001893312,0.0001299359,0.0003754284,0.0005867123,0.006745972],"genre_scores_gemma":[0.9913833,0.00005496334,0.007742973,0.00003146206,0.000006977123,0.00004652913,0.0001112248,0.00002270912,0.0006000623],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.006603903,"threshold_uncertainty_score":0.02209222,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.007431018518091173,"score_gpt":0.2500148783517177,"score_spread":0.2425838598336265,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}