{"id":"W4385729101","doi":"10.1016/j.ins.2023.119481","title":"Reward shaping using convolutional neural network","year":2023,"lang":"en","type":"article","venue":"Information Sciences","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Markov decision process; Reinforcement learning; Convolutional neural network; Stochastic matrix; Representation (politics); Convolution (computer science); Artificial intelligence; Matrix (chemical analysis); Function (biology); Markov chain; Bellman equation; Algorithm; Artificial neural network; Markov process; Machine learning; Mathematical optimization; Mathematics; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005180545,0.0007198931,0.0005743743,0.000369019,0.0002615489,0.000588022,0.001354063,0.0007620985,0.002300876],"category_scores_gemma":[0.001679529,0.0003256885,0.0004322351,0.0003216125,0.000605925,0.0009066634,0.000808834,0.000998216,0.0003895339],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001195304,"about_ca_system_score_gemma":0.001060227,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00707682,"about_ca_topic_score_gemma":0.007491525,"domain_scores_codex":[0.9997467,0.0000405974,0.00001260142,0.00007432341,0.00007241135,0.00005322288],"domain_scores_gemma":[0.9995976,0.0001681606,0.00005977031,0.00005246081,0.0000868033,0.00003525884],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00008921856,0.00006377732,0.0007856735,0.00007094155,0.00003991745,0.00007204539,0.00003241115,0.8697034,0.005241031,0.01147297,0.001591814,0.1108368],"study_design_scores_gemma":[0.000003111887,0.00001598111,0.00006416258,0.000003718958,0.000003897248,0.000007776733,0.000001440838,0.99544,0.0009534002,0.003115848,0.0003874029,0.000003217197],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03855881,0.0007560504,0.9512902,0.0003515753,0.0001200235,0.00006574291,0.000125221,0.002519453,0.006212856],"genre_scores_gemma":[0.8979196,0.0003361123,0.09607769,0.0002137071,0.00002759338,0.00009760748,0.0001783881,0.0001049314,0.005044412],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00707682,"threshold_uncertainty_score":0.01407129,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1262881818370389,"score_gpt":0.3253652114570306,"score_spread":0.1990770296199917,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}