{"id":"W2925234320","doi":"10.48550/arxiv.1903.09762","title":"TTR-Based Reward for Reinforcement Learning with Implicit Model Priors","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Reinforcement learning; Computer science; Key (lock); Function (biology); Inefficiency; Artificial intelligence; Process (computing); Exploit; Temporal difference learning; State (computer science); Machine learning; Algorithm","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0004097113,0.0005488257,0.0005384191,0.0003903672,0.0002854078,0.0002526101,0.002386465,0.0003615127,0.00001273337],"category_scores_gemma":[0.00006046814,0.000579886,0.0003040933,0.0004469784,0.00009351291,0.0004146022,0.00152686,0.0009664318,0.00007111429],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005134555,"about_ca_system_score_gemma":0.0008165539,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00004542785,"about_ca_topic_score_gemma":0.000004745638,"domain_scores_codex":[0.9971322,0.00008336637,0.0003567202,0.001435896,0.0002618461,0.0007299878],"domain_scores_gemma":[0.9966968,0.0002020032,0.0006709403,0.001855101,0.0003679423,0.0002072331],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001024607,0.00002029024,0.00142032,0.0002275378,0.0001087471,0.0000221078,0.0001935652,0.9587315,0.00002068629,0.03892776,0.000156381,0.00006868828],"study_design_scores_gemma":[0.001325935,0.0004369988,0.00007243012,0.0002179994,0.0001080847,0.000001481506,0.00004373309,0.9952367,0.0001261873,0.000890116,0.0008096861,0.0007306738],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01540967,0.000008610238,0.9791676,0.0001078984,0.0002947553,0.00127795,0.000003204333,0.0004621846,0.00326812],"genre_scores_gemma":[0.9575335,0.00002498407,0.02920164,0.0002113094,0.00004132036,0.000007768915,0.00006024131,0.00005721702,0.01286199],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.949966,"threshold_uncertainty_score":0.9996653,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06577840482045562,"score_gpt":0.198769425024594,"score_spread":0.1329910202041384,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}