{"id":"W2930281309","doi":"10.48550/arxiv.1904.01191","title":"Planning with Expectation Models","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Parametrization (atmospheric modeling); Convergence (economics); Mathematical optimization; Bellman equation; Function (biology); State (computer science); Artificial intelligence; Mathematics; Algorithm","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002022912,0.00124115,0.001281973,0.0005611754,0.0004572665,0.001798557,0.001911422,0.001532104,0.005165026],"category_scores_gemma":[0.01059087,0.0006947867,0.001196366,0.0008606566,0.00156231,0.00343956,0.002052257,0.002802934,0.0008192444],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001479037,"about_ca_system_score_gemma":0.001672221,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004583531,"about_ca_topic_score_gemma":0.004724278,"domain_scores_codex":[0.9979973,0.0009421583,0.0001073177,0.0003961289,0.0003914246,0.0001656058],"domain_scores_gemma":[0.9955348,0.003436772,0.0002949553,0.0003428561,0.0002635298,0.0001271064],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001046968,0.00005154536,0.0004843133,0.0001157307,0.00004115293,0.0001002016,0.0001156192,0.7659137,0.0005476314,0.1978513,0.001581767,0.03309238],"study_design_scores_gemma":[0.00001684838,0.00002986223,0.00004717113,0.00001430823,0.000008621992,0.00002458336,0.00001010148,0.9031315,0.0002960206,0.0954311,0.0009804006,0.000009461581],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.00344733,0.0001449167,0.9938188,0.0002448364,0.00001880178,0.00002733439,0.00006999699,0.0002338011,0.001994122],"genre_scores_gemma":[0.6184328,0.0007229415,0.3719481,0.0004259461,0.00009969672,0.0004867178,0.0005383944,0.0002256095,0.007119875],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005165026,"threshold_uncertainty_score":0.01727867,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1031923110966533,"score_gpt":0.1905398815660638,"score_spread":0.08734757046941054,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}