{"id":"W1552684655","doi":"","title":"Using linear programming for Bayesian exploration in Markov decision processes","year":2007,"lang":"en","type":"article","venue":"International Joint Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Markov decision process; Computer science; Bellman equation; Reinforcement learning; Mathematical optimization; Markov process; Linear programming; Machine learning; Representation (politics); Artificial intelligence; Markov chain; Key (lock); Bayesian probability; Partially observable Markov decision process; Markov model; Algorithm; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005429449,0.001746528,0.002431212,0.001169428,0.0007312061,0.002001735,0.001928613,0.002100803,0.00510543],"category_scores_gemma":[0.01922418,0.001230014,0.001271504,0.001840999,0.002730521,0.002883603,0.002424606,0.00387709,0.0007216192],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002958881,"about_ca_system_score_gemma":0.002516607,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008297607,"about_ca_topic_score_gemma":0.008343421,"domain_scores_codex":[0.997115,0.001929027,0.00007797917,0.0002839121,0.0003771482,0.0002169226],"domain_scores_gemma":[0.9859586,0.01283591,0.0005173181,0.0001920376,0.0003100446,0.0001860604],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00005443228,0.00006057691,0.0003629464,0.0001034416,0.00004078874,0.00004671535,0.0001003188,0.8628524,0.000159665,0.1183311,0.0007718895,0.01711575],"study_design_scores_gemma":[0.00001130209,0.00001104683,0.00002480928,0.00001110201,0.000003652657,0.000004173749,0.000005993588,0.9287184,0.0000465999,0.07089499,0.0002624415,0.00000545611],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.003499975,0.0004855718,0.9937435,0.0004606768,0.00001677441,0.00003040711,0.00003503135,0.0001307525,0.001597274],"genre_scores_gemma":[0.457115,0.002044464,0.5301758,0.0006198586,0.0002652164,0.001176781,0.0003608175,0.00032834,0.007913793],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.008297607,"threshold_uncertainty_score":0.028714,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2143689443380605,"score_gpt":0.3942220206217933,"score_spread":0.1798530762837328,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}