{"id":"W4385484633","doi":"10.1109/ijcnn54540.2023.10191867","title":"Reducing the Cost of Cycle-Time Tuning for Real-World Policy Optimization","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"Alberta Machine Intelligence Institute","keywords":"Benchmark (surveying); Baseline (sea); Task (project management); Computer science; Robotics; Artificial intelligence; Reinforcement learning; Machine learning; Robot; Engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003127314,0.001449119,0.001521821,0.0006713869,0.0006428295,0.001285635,0.002065237,0.001779411,0.005463794],"category_scores_gemma":[0.02253261,0.0009089101,0.0006233812,0.0005691994,0.001205947,0.002550752,0.001598962,0.003979232,0.001359392],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001241593,"about_ca_system_score_gemma":0.002880991,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006122227,"about_ca_topic_score_gemma":0.00702154,"domain_scores_codex":[0.9980655,0.0006881286,0.0001400567,0.0004119333,0.0004784614,0.0002159084],"domain_scores_gemma":[0.9909645,0.00602927,0.0005193788,0.001433751,0.0007307021,0.0003223604],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006686822,0.000648534,0.003517731,0.0004335501,0.0001492889,0.0001437296,0.0003021658,0.6807085,0.009494872,0.01121045,0.00586037,0.2868621],"study_design_scores_gemma":[0.00009543704,0.0001734768,0.0007457612,0.00004877283,0.00003291413,0.00006250956,0.00006010866,0.9833583,0.003211153,0.009358403,0.002819778,0.00003326358],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09756731,0.003459725,0.8784577,0.001095998,0.0004073072,0.0003767166,0.0001258851,0.007903416,0.0106059],"genre_scores_gemma":[0.8860277,0.0003862395,0.1100219,0.0006402454,0.00005130507,0.0003190814,0.0001665242,0.0005583069,0.001828788],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006122227,"threshold_uncertainty_score":0.01827824,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02834986602689119,"score_gpt":0.3042799630545102,"score_spread":0.2759300970276189,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}