{"id":"W4360764167","doi":"10.1109/icmla55696.2022.00101","title":"Hyperparameter Tuning in Offline Reinforcement Learning","year":2022,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Lakehead University","funders":"","keywords":"Hyperparameter; Reinforcement learning; Benchmark (surveying); Metric (unit); Computer science; Machine learning; Performance metric; Artificial intelligence; Scheme (mathematics); Hyperparameter optimization; Friedman test; Statistical hypothesis testing; Statistics; Mathematics; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006829737,0.0001215548,0.0001447705,0.0002579813,0.000256682,0.0001002135,0.0008554537,0.00002145486,0.0006237041],"category_scores_gemma":[0.00008283419,0.000126289,0.00004887956,0.0006297987,0.00001654331,0.0003201366,0.001265335,0.0005477693,0.00007367597],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000179263,"about_ca_system_score_gemma":0.00005306304,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005641,"about_ca_topic_score_gemma":0.000001795052,"domain_scores_codex":[0.9983346,0.000135438,0.000340182,0.0003025546,0.0005177269,0.000369459],"domain_scores_gemma":[0.9993019,0.0001135272,0.0001016114,0.0004034098,0.00002388162,0.00005568777],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000003500429,0.00001121041,0.003366865,0.000003073665,0.000005152418,0.00001824458,0.0005336803,0.9797524,0.0002051597,0.01404468,0.0002057523,0.001850307],"study_design_scores_gemma":[0.0003582065,0.0002056752,0.0004911462,0.000004562109,0.000001397646,0.00001336863,0.000151514,0.9683756,0.0001054489,0.0000707122,0.03005835,0.0001640614],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.00811636,0.00001863731,0.9550312,0.0004592377,0.0002564286,0.0001448119,3.83944e-8,0.0002141456,0.0357591],"genre_scores_gemma":[0.9559676,0.000004544459,0.02780227,0.0008169026,0.0000176397,0.00003868953,0.00000564493,0.00001100521,0.01533573],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9478512,"threshold_uncertainty_score":0.6829122,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01903993677232508,"score_gpt":0.2398741299863006,"score_spread":0.2208341932139755,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}