{"id":"W4360764167","doi":"10.1109/icmla55696.2022.00101","title":"Hyperparameter Tuning in Offline Reinforcement Learning","year":2022,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Lakehead University","funders":"","keywords":"Hyperparameter; Reinforcement learning; Benchmark (surveying); Metric (unit); Computer science; Machine learning; Performance metric; Artificial intelligence; Scheme (mathematics); Hyperparameter optimization; Friedman test; Statistical hypothesis testing; Statistics; Mathematics; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004388205,0.00217583,0.001763409,0.0008322067,0.0005792055,0.001777219,0.002738543,0.001927933,0.00307411],"category_scores_gemma":[0.02348758,0.0006792144,0.0005768558,0.0004547932,0.001928132,0.002440932,0.002231834,0.003983528,0.001397302],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00127757,"about_ca_system_score_gemma":0.001847335,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002156321,"about_ca_topic_score_gemma":0.002477139,"domain_scores_codex":[0.9968165,0.001487585,0.0001782501,0.0006818242,0.0005320002,0.0003038658],"domain_scores_gemma":[0.9932305,0.003641822,0.0006293257,0.001414999,0.0007880733,0.0002952838],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003793006,0.0002625533,0.002577147,0.0001595257,0.00009190156,0.0001097386,0.0001133922,0.8621983,0.004386362,0.006520212,0.003821179,0.1193805],"study_design_scores_gemma":[0.00005186483,0.00009537164,0.0001518254,0.00002008439,0.000009381119,0.00002984218,0.00001512961,0.9909729,0.001710642,0.006241618,0.0006871898,0.00001434264],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.04680138,0.0009445363,0.9432147,0.0004470703,0.0001388382,0.0001908951,0.0001480421,0.004732714,0.003381805],"genre_scores_gemma":[0.8502154,0.0001683484,0.1457973,0.0005164733,0.00007808326,0.0004181734,0.0003420426,0.0005945303,0.001869602],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004388205,"threshold_uncertainty_score":0.02320731,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01903993677232508,"score_gpt":0.2398741299863006,"score_spread":0.2208341932139755,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}