{"id":"W4378531216","doi":"10.1007/978-3-031-33377-4_26","title":"A Dynamic and Task-Independent Reward Shaping Approach for Discrete Partially Observable Markov Decision Processes","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Context-Aware Activity Recognition Systems","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"York University","funders":"","keywords":"Partially observable Markov decision process; Computer science; Markov decision process; Bellman equation; Graph; Dimension (graph theory); Task (project management); State space; Reinforcement learning; Observable; Function (biology); Action (physics); Markov chain; Markov process; Artificial intelligence; Domain (mathematical analysis); Markov model; Machine learning; Mathematical optimization; Theoretical computer science; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002387455,0.001243223,0.002484867,0.0007058763,0.0007848305,0.001738379,0.00360418,0.001768993,0.005647069],"category_scores_gemma":[0.005894694,0.001166728,0.001586669,0.001321032,0.001555022,0.002024536,0.003142809,0.003085565,0.0009624807],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002339923,"about_ca_system_score_gemma":0.00327871,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008795382,"about_ca_topic_score_gemma":0.008935818,"domain_scores_codex":[0.9985564,0.0004433136,0.00006270618,0.0003115913,0.000339216,0.0002866178],"domain_scores_gemma":[0.9964663,0.002580142,0.0001893375,0.0002089062,0.0003546616,0.0002006263],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00009469167,0.00009136098,0.0001992356,0.00009693583,0.00006470227,0.00008274362,0.0001155552,0.8660319,0.001101402,0.09382779,0.001761124,0.03653257],"study_design_scores_gemma":[0.000005671546,0.00001417496,0.0000352747,0.000005527024,0.000008311056,0.000008637508,0.000005076905,0.9681383,0.00008881576,0.03141315,0.0002697971,0.000007133774],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.003556558,0.0002561353,0.993238,0.0001737186,0.00004461164,0.00003699644,0.00005465025,0.0001670517,0.002472205],"genre_scores_gemma":[0.6564354,0.001381865,0.3201043,0.0003525743,0.0003210214,0.0005028362,0.0003624746,0.000334703,0.02020487],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.008795382,"threshold_uncertainty_score":0.01889133,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04587839046642977,"score_gpt":0.2765145901837316,"score_spread":0.2306361997173019,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}