{"id":"W4385484633","doi":"10.1109/ijcnn54540.2023.10191867","title":"Reducing the Cost of Cycle-Time Tuning for Real-World Policy Optimization","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"Alberta Machine Intelligence Institute","keywords":"Benchmark (surveying); Baseline (sea); Task (project management); Computer science; Robotics; Artificial intelligence; Reinforcement learning; Machine learning; Robot; Engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004955489,0.00007473396,0.0001023536,0.0002312999,0.0001616593,0.0000933044,0.0006009797,0.00002367409,0.00001956442],"category_scores_gemma":[0.0002233199,0.00005577572,0.00004702137,0.001271176,0.00003280766,0.0002249345,0.0002231827,0.00005880848,0.00003815331],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004619506,"about_ca_system_score_gemma":0.00008496534,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002020188,"about_ca_topic_score_gemma":0.00000284129,"domain_scores_codex":[0.9991435,0.00003548851,0.0002324764,0.0001661701,0.0001919068,0.0002304081],"domain_scores_gemma":[0.9989524,0.0003599617,0.0001321407,0.0004337408,0.00008841417,0.00003336266],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000001437141,0.000002219651,0.00003689388,0.000008040071,0.000007635924,1.862773e-7,0.0003444977,0.9599922,0.0002947682,0.03521689,0.001641591,0.002453679],"study_design_scores_gemma":[0.0001235296,0.00002774682,0.0002173349,0.00002108549,0.000003580315,5.447799e-7,0.00002566171,0.9980304,0.0005059605,0.0001720679,0.0008067481,0.00006537166],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0001245582,0.000001219803,0.9716912,0.002460925,0.0001093603,0.0002990605,5.927057e-7,0.0002627657,0.02505031],"genre_scores_gemma":[0.3504191,0.00007166573,0.5675665,0.0007178631,0.000397642,0.00008981256,0.00004769876,0.00005493238,0.08063485],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.4041247,"threshold_uncertainty_score":0.2274468,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02834986602689119,"score_gpt":0.3042799630545102,"score_spread":0.2759300970276189,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}