{"id":"W3037571487","doi":"","title":"Value Preserving State-Action Abstractions","year":2020,"lang":"en","type":"article","venue":"International Conference on Artificial Intelligence and Statistics","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Value (mathematics); State (computer science); Action (physics); Programming language; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001789238,0.0008924238,0.0007551191,0.0006508761,0.0005953131,0.001941548,0.001661951,0.001019167,0.006722853],"category_scores_gemma":[0.004625426,0.000615711,0.001566135,0.0006931228,0.001510194,0.003187372,0.004262074,0.003369358,0.001043257],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008753447,"about_ca_system_score_gemma":0.001326728,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001380471,"about_ca_topic_score_gemma":0.002099546,"domain_scores_codex":[0.9988556,0.0002774318,0.00009480288,0.0002281487,0.0003935902,0.0001505278],"domain_scores_gemma":[0.998218,0.0007403555,0.000119405,0.00060357,0.0001817242,0.0001370339],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002193154,0.0001224571,0.0005583838,0.0001783899,0.00008169668,0.0002424957,0.0003466263,0.1609781,0.005806975,0.7132933,0.002668574,0.1155037],"study_design_scores_gemma":[0.00003139029,0.00006846241,0.0001206649,0.0000307181,0.00004248289,0.00006115897,0.00004575683,0.4112064,0.004880922,0.5767396,0.006755474,0.00001687884],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.006751448,0.00008771121,0.9878893,0.0001666083,0.00005748206,0.00004661067,0.0001434711,0.0007904244,0.004066946],"genre_scores_gemma":[0.5599312,0.0003685474,0.4244834,0.0003016259,0.00007328342,0.0002336618,0.0005562101,0.000378942,0.01367315],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006722853,"threshold_uncertainty_score":0.02249014,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2378914496979285,"score_gpt":0.3741205699300081,"score_spread":0.1362291202320796,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}