{"id":"W2407045295","doi":"10.13140/2.1.3356.8002","title":"Efficient Abstraction Selection in Reinforcement Learning --- Extended Abstract","year":2013,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Abstraction; Reinforcement learning; Computer science; Markov decision process; Selection (genetic algorithm); Set (abstract data type); Artificial intelligence; State (computer science); Markov process; Machine learning; Theoretical computer science; Programming language; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001828485,0.0005897459,0.001260691,0.0003016453,0.00033764,0.0008094029,0.001097674,0.0006643428,0.002440399],"category_scores_gemma":[0.005039601,0.0003470342,0.0005554903,0.0004137789,0.0009209566,0.001396968,0.00175753,0.001370898,0.0003223074],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008972128,"about_ca_system_score_gemma":0.001009483,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002763004,"about_ca_topic_score_gemma":0.001682127,"domain_scores_codex":[0.9988525,0.0005289656,0.00005683348,0.0001830718,0.0002380057,0.0001405838],"domain_scores_gemma":[0.9980026,0.001232655,0.0001690475,0.0002526747,0.0002256297,0.0001173499],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0002295704,0.00007457939,0.0009966837,0.00009822497,0.00005561998,0.0001378455,0.00008021477,0.8830867,0.002491823,0.05022691,0.001259473,0.06126243],"study_design_scores_gemma":[0.00001582124,0.00002617449,0.000052418,0.000005303687,0.000004927497,0.000009564388,0.000003641543,0.981196,0.0003863154,0.018005,0.0002915272,0.000003271919],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02738021,0.0002025311,0.9701531,0.0002139249,0.00004096353,0.00005393795,0.00004812536,0.0003661036,0.001541138],"genre_scores_gemma":[0.8698692,0.0001427913,0.1276962,0.00009384981,0.0000353763,0.0001407934,0.0001043578,0.00006821666,0.001849181],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002763004,"threshold_uncertainty_score":0.009670079,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01256500116340309,"score_gpt":0.2434278053132065,"score_spread":0.2308628041498034,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}