{"id":"W2990933479","doi":"10.48550/arxiv.1911.08610","title":"Efficient decorrelation of features using Gramian in Reinforcement Learning","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Decorrelation; Regularization (linguistics); Computer science; Sample complexity; Artificial intelligence; Gramian matrix; Temporal difference learning; Computational complexity theory; Mathematical optimization; Machine learning; Mathematics; Algorithm","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0004163428,0.000284206,0.0003727415,0.000617131,0.00008756913,0.00007534272,0.001164933,0.0003087879,0.00001423815],"category_scores_gemma":[0.00008224216,0.0003436511,0.0001437718,0.0006955148,0.00006427379,0.0001551414,0.001691139,0.0009771165,0.00003664946],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005002284,"about_ca_system_score_gemma":0.0002329966,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000311863,"about_ca_topic_score_gemma":0.00001117178,"domain_scores_codex":[0.9981567,0.0001715919,0.000377946,0.0007273296,0.0002012693,0.0003651434],"domain_scores_gemma":[0.9980916,0.0001237227,0.0006595027,0.000906502,0.0001425671,0.00007604295],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001561946,0.00002133109,0.01485648,0.000102126,0.00003045904,0.00002938273,0.0004811533,0.9659405,0.00005334891,0.01837102,0.000006525281,0.00009202261],"study_design_scores_gemma":[0.0004450756,0.00007784725,0.004676201,0.0002897453,0.00002995297,0.000001989583,0.00008435985,0.9937159,0.00007214719,0.0002621972,0.00003243551,0.0003120993],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2813605,0.0000231868,0.7160642,0.000008535268,0.000487986,0.0003026719,4.169783e-7,0.00006728783,0.001685224],"genre_scores_gemma":[0.9951616,0.00002922175,0.003266289,0.00001572725,0.00001975458,3.209502e-7,0.00001693654,0.00001698005,0.001473162],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7138011,"threshold_uncertainty_score":0.9999015,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05318134821492707,"score_gpt":0.2026254375092948,"score_spread":0.1494440892943678,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}