{"id":"W4360764815","doi":"10.1109/icmla55696.2022.00044","title":"Benchmarking Offline Reinforcement Learning","year":2022,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Lakehead University","funders":"","keywords":"Benchmarking; Reinforcement learning; Computer science; Hyperparameter; Machine learning; Artificial intelligence; Offline learning; Performance improvement; Online learning; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003463295,0.002217615,0.001735167,0.001091337,0.0006312813,0.001729597,0.003255778,0.002070485,0.008192163],"category_scores_gemma":[0.01476365,0.0005832271,0.0008809311,0.001091125,0.001227262,0.002030501,0.001806707,0.002603596,0.003625237],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001532085,"about_ca_system_score_gemma":0.002210116,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009267435,"about_ca_topic_score_gemma":0.008013245,"domain_scores_codex":[0.9961888,0.001280818,0.0002624677,0.001023814,0.0008240179,0.0004200916],"domain_scores_gemma":[0.9925673,0.003942952,0.0002868081,0.001836713,0.00106935,0.0002968086],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009297007,0.0007919881,0.004031412,0.0007508561,0.0002498519,0.000107626,0.00006416783,0.806254,0.002510625,0.00493269,0.0201417,0.1592354],"study_design_scores_gemma":[0.0001507788,0.0002854014,0.0009752497,0.00004444776,0.00002157829,0.00004825398,0.00002690838,0.9839103,0.005913787,0.003412521,0.005184313,0.00002643185],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.291904,0.006273308,0.5814791,0.001192045,0.001495334,0.0007001641,0.01105931,0.06748007,0.03841674],"genre_scores_gemma":[0.7977999,0.0007174527,0.1761447,0.0004826491,0.0001221849,0.0006231599,0.0162881,0.001999125,0.005822662],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.009267435,"threshold_uncertainty_score":0.0274055,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0137320246154256,"score_gpt":0.2294180182797683,"score_spread":0.2156859936643427,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}