{"id":"W4307490062","doi":"10.1613/jair.1.13854","title":"Low-Rank Representation of Reinforcement Learning Policies","year":2022,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal; Canadian Institute for Advanced Research; McGill University","funders":"Canadian Institute for Advanced Research","keywords":"Reproducing kernel Hilbert space; Reinforcement learning; Representation (politics); Computer science; Embedding; Rank (graph theory); Stability (learning theory); Kernel (algebra); Convergence (economics); Space (punctuation); Hilbert space; Hilbert curve; Mathematical optimization; Artificial intelligence; Theoretical computer science; Machine learning; Mathematics; Algorithm; Discrete mathematics; Economics; Pure mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001807293,0.0009563294,0.001279509,0.000571706,0.0002648785,0.001394118,0.001173272,0.001272023,0.003540235],"category_scores_gemma":[0.007397827,0.0003510018,0.0006306767,0.0005248319,0.001156557,0.001823582,0.001214562,0.002034976,0.0007472243],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009547762,"about_ca_system_score_gemma":0.001325505,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001944227,"about_ca_topic_score_gemma":0.001338473,"domain_scores_codex":[0.9988711,0.0004752176,0.00007158433,0.0001940403,0.0002735786,0.0001143466],"domain_scores_gemma":[0.9975901,0.001204729,0.0003739851,0.0003547402,0.0003543945,0.0001221147],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00007883414,0.00005639635,0.0002369383,0.00009625554,0.00002245873,0.00005208532,0.00004939378,0.8987291,0.002512343,0.06353839,0.0007350456,0.03389273],"study_design_scores_gemma":[0.000004439871,0.00002221731,0.00002580938,0.000003263381,0.000001579065,0.000005409899,0.000002166166,0.9864956,0.0003082547,0.01294949,0.0001778775,0.000003874305],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.004597228,0.00007607558,0.9944077,0.000111039,0.00001432582,0.00002324732,0.00004496578,0.0001895141,0.0005359949],"genre_scores_gemma":[0.6688945,0.0004480215,0.3240256,0.0002166177,0.0001038809,0.0004265147,0.0004526389,0.0001605632,0.005271601],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003540235,"threshold_uncertainty_score":0.01184332,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1582119241585682,"score_gpt":0.4211595401058433,"score_spread":0.2629476159472751,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}