{"id":"W2167748058","doi":"","title":"Average Reward Optimization Objective In Partially Observable Domains","year":2013,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Observability; Dimension (graph theory); Observable; Representation (politics); Function (biology); Mathematics; Process (computing); Mathematical optimization; Bellman equation; Computer science; Computation; Key (lock); Applied mathematics; Algorithm","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001910174,0.0009782667,0.001611628,0.000572169,0.0003940454,0.001365221,0.001046404,0.001360968,0.001794227],"category_scores_gemma":[0.006668587,0.0004666547,0.000664202,0.0006071216,0.001592588,0.002087686,0.001244914,0.001432595,0.0001809927],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001851446,"about_ca_system_score_gemma":0.001183775,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004799244,"about_ca_topic_score_gemma":0.002273174,"domain_scores_codex":[0.9991165,0.0003858726,0.00003864837,0.0001868971,0.0001612725,0.0001109007],"domain_scores_gemma":[0.9969836,0.002194083,0.0003423869,0.0001321496,0.0002080009,0.0001397955],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002979573,0.00001513636,0.0002130818,0.00004263055,0.00001609759,0.00003756805,0.00002734361,0.9584724,0.0003203093,0.03585707,0.000258285,0.004710278],"study_design_scores_gemma":[0.000004721233,0.00001249416,0.00005077281,0.000004919866,0.00000295972,0.000004421046,0.000003745257,0.9804149,0.0001171932,0.01928081,0.00009956962,0.000003440248],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0459943,0.000452687,0.9495853,0.0005926426,0.0000264789,0.00003953091,0.0001424197,0.0001958051,0.002970929],"genre_scores_gemma":[0.90644,0.0004100385,0.089793,0.00009576825,0.00004277743,0.0001211473,0.0001524106,0.0000767193,0.002868109],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004799244,"threshold_uncertainty_score":0.01343322,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01299934096471681,"score_gpt":0.2207024881727568,"score_spread":0.20770314720804,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}