{"id":"W2142097483","doi":"","title":"On Average Reward Policy Evaluation in Infinite-State Partially Observable Systems","year":2012,"lang":"en","type":"article","venue":"","topic":"Advanced Control Systems Optimization","field":"Engineering","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Observable; State space; State (computer science); Ergodic theory; Dynamical systems theory; Computer science; Q-learning; Key (lock); Dynamical system (definition); Reward system; Mathematics; Reinforcement learning; Mathematical optimization; Control theory (sociology); Artificial intelligence; Control (management); Algorithm","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006156805,0.0001458622,0.0001841932,0.0001935742,0.000025773,0.00003598907,0.00007453479,0.00006992587,0.00005333832],"category_scores_gemma":[0.00018997,0.0001438977,0.00002528834,0.0003244177,0.000005381789,0.0004991422,0.00001099074,0.00009975112,0.0003234667],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004345267,"about_ca_system_score_gemma":0.00003094027,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001775699,"about_ca_topic_score_gemma":0.00006444474,"domain_scores_codex":[0.9987729,0.0001009202,0.00034867,0.0001184311,0.0002757415,0.0003833701],"domain_scores_gemma":[0.9994748,0.00009026065,0.00004560744,0.0002410344,0.00006158382,0.00008671395],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00000933407,0.00001412455,0.001325989,0.00003240631,0.00001022452,5.08655e-7,0.0002596495,0.9927486,0.0004861366,0.003804724,0.0001147505,0.001193608],"study_design_scores_gemma":[0.00073356,0.00002023275,0.002155508,0.0000555506,0.000005697564,0.000001492819,0.00002747182,0.9947067,0.0002383407,0.0003392257,0.001534581,0.0001816242],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.41637,0.001332764,0.4563742,0.0001423799,0.002484592,0.002734512,0.00001655579,0.001052802,0.1194922],"genre_scores_gemma":[0.9984163,0.00002559863,0.0003794109,0.00005188338,0.0002059209,0.0001769644,0.0000154036,0.00003864361,0.0006898888],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.5820463,"threshold_uncertainty_score":0.5867977,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02210244359152734,"score_gpt":0.2562949962152085,"score_spread":0.2341925526236811,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}