{"id":"W2184461682","doi":"10.82308/33420","title":"A Bayesian Framework for Online Parameter Learning in POMDPs","year":2011,"lang":"en","type":"article","venue":"eScholarship@McGill (McGill)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Partially observable Markov decision process; Computer science; Reinforcement learning; Artificial intelligence; Markov decision process; Machine learning; Ambiguity; Robotics; Bayesian probability; Robot; Markov process; Markov chain; Markov model","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.001182654,0.0004871244,0.0005075182,0.0004429139,0.0005951807,0.0001492674,0.001811959,0.000428728,0.00008291841],"category_scores_gemma":[0.00242455,0.0005127162,0.0002560506,0.0009344034,0.00006418687,0.001381098,0.0006331814,0.001687037,0.0001503386],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003621884,"about_ca_system_score_gemma":0.0000441583,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00008528476,"about_ca_topic_score_gemma":0.000049231,"domain_scores_codex":[0.9961155,0.0003495803,0.0008439161,0.001031947,0.0005675876,0.001091441],"domain_scores_gemma":[0.9970821,0.0009131181,0.0003888752,0.001110961,0.0001876314,0.0003172797],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00007701379,0.0004073394,0.00169602,0.0001023541,0.000108735,0.00009280372,0.0001060655,0.02149036,0.000731995,0.8431109,0.000002438942,0.1320739],"study_design_scores_gemma":[0.003902986,0.002376261,0.009561338,0.0008500776,0.0001185123,0.0001061098,0.0002174026,0.4021683,0.01617329,0.4745122,0.08674499,0.003268596],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4759212,0.0001407006,0.483634,0.0002580978,0.002589985,0.002734829,0.00009124446,0.003137615,0.03149242],"genre_scores_gemma":[0.5812483,0.00002005775,0.4176655,0.0005122176,0.00002448443,0.00006335555,0.00001106196,0.00006012269,0.0003949364],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.3806779,"threshold_uncertainty_score":0.9997324,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04305426771065417,"score_gpt":0.2616776303496849,"score_spread":0.2186233626390308,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}