{"id":"W2996283994","doi":"10.48550/arxiv.2002.05822","title":"Frequency-based Search-control in Dyna","year":2020,"lang":"en","type":"article","venue":"PolyPublie (École Polytechnique de Montréal)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Reinforcement learning; Bellman equation; Benchmark (surveying); Artificial intelligence; Function (biology); Machine learning; Mathematical optimization; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.000751534,0.0003554201,0.000409288,0.0004299697,0.0001499033,0.0003625581,0.002026036,0.0002361219,0.00003933859],"category_scores_gemma":[0.0003511246,0.0003756965,0.0001666355,0.001355221,0.00008830548,0.0006599383,0.0003069523,0.0008157521,0.0001091472],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004563926,"about_ca_system_score_gemma":0.0005411549,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002180704,"about_ca_topic_score_gemma":0.0003009884,"domain_scores_codex":[0.9967917,0.0002685852,0.0006000432,0.0006813643,0.0006658632,0.000992398],"domain_scores_gemma":[0.997935,0.0002054314,0.000172876,0.001075443,0.0001069945,0.0005042704],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003014842,0.00007028075,0.0456764,0.00004193823,0.00002080971,0.0002078934,0.0004202217,0.8967614,0.005862565,0.04429157,0.000666781,0.005949973],"study_design_scores_gemma":[0.001057714,0.000213234,0.009691426,0.00003659585,0.00000702175,0.00001073072,0.00002032213,0.9842556,0.003029694,0.0006935981,0.0006008876,0.0003832497],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.00413249,0.0001984246,0.9613023,0.03156829,0.00009049806,0.0007060705,0.000008207416,0.001109124,0.0008845506],"genre_scores_gemma":[0.7310141,0.00001683265,0.2541306,0.01448648,0.00008660238,0.0001379877,0.000008606564,0.00003970234,0.00007909734],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.7268816,"threshold_uncertainty_score":0.9998695,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01453165029516121,"score_gpt":0.2300616995247914,"score_spread":0.2155300492296302,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}