{"id":"W3152801112","doi":"10.48550/arxiv.2104.08543","title":"Planning with Expectation Models for Control","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Guelph","funders":"","keywords":"Reinforcement learning; Variance (accounting); Computer science; Bellman equation; Function (biology); Feature (linguistics); Artificial intelligence; Sample (material); Action (physics); Stochastic modelling; Control (management); Mathematical optimization; Mathematics; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002545938,0.001796709,0.001370517,0.0006484319,0.0005846499,0.001953945,0.001537925,0.001738522,0.006516162],"category_scores_gemma":[0.01147923,0.0007104272,0.001211671,0.0009633413,0.002099628,0.003432924,0.002126971,0.004406784,0.0009533731],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002504299,"about_ca_system_score_gemma":0.001956566,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006147763,"about_ca_topic_score_gemma":0.004455293,"domain_scores_codex":[0.9984331,0.000712399,0.00006778316,0.00032779,0.0003350082,0.0001239216],"domain_scores_gemma":[0.9935145,0.005319454,0.0003073462,0.0003747282,0.0003483998,0.0001354682],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001137374,0.00008351412,0.000526292,0.0001683376,0.00005282775,0.00007340564,0.0001372929,0.6711638,0.0005522108,0.2797463,0.002572951,0.04480934],"study_design_scores_gemma":[0.00002119012,0.00003892975,0.0000546635,0.00002504301,0.000008874553,0.00001766114,0.00001235941,0.8221648,0.0002718719,0.1757622,0.001609248,0.00001311056],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.002945888,0.0007101811,0.9912125,0.0005923224,0.00004844725,0.00003525699,0.00006013693,0.0002455804,0.004149618],"genre_scores_gemma":[0.5918798,0.001993806,0.3964116,0.0006981741,0.0002338502,0.0005932532,0.0003515573,0.0002476248,0.007590318],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006516162,"threshold_uncertainty_score":0.02179873,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0911888882141892,"score_gpt":0.1957893315222335,"score_spread":0.1046004433080443,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}