{"id":"W4402352014","doi":"10.1109/ijcnn60899.2024.10650768","title":"Conservative In-Distribution Q-Learning for Offline Reinforcement Learning","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"National Natural Science Foundation of China","keywords":"Reinforcement learning; Computer science; Distribution (mathematics); Reinforcement; Q-learning; Artificial intelligence; Machine learning; Engineering; Mathematics; Structural engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006135416,0.0001656929,0.0001708054,0.0001475862,0.000152933,0.0003460609,0.0003654964,0.000075151,0.00006512504],"category_scores_gemma":[0.000429623,0.0001552802,0.00008167761,0.0006124739,0.00003357639,0.0005897834,0.0001925776,0.0004330925,0.0001079978],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002025362,"about_ca_system_score_gemma":0.0001226222,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003423555,"about_ca_topic_score_gemma":0.000004996997,"domain_scores_codex":[0.9984598,0.00007234428,0.0004011924,0.000393355,0.0002753033,0.000398005],"domain_scores_gemma":[0.9990561,0.0004867564,0.00007810445,0.0001996019,0.0001168941,0.00006256639],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000006626894,0.000004254857,0.0005681082,0.00005663293,0.00001781352,0.000007807331,0.0003902169,0.8617268,0.0001351241,0.132324,0.001100839,0.003661783],"study_design_scores_gemma":[0.0002855673,0.0002440551,0.0004418372,0.00009740963,0.000005000788,0.000004075858,0.00006890182,0.8775666,0.0004911683,0.0002843286,0.1203407,0.000170332],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0009283249,0.00008069917,0.990263,0.001009807,0.000362821,0.0003611705,3.859862e-7,0.0005605489,0.006433263],"genre_scores_gemma":[0.9616418,0.00003200568,0.01594323,0.0001865347,0.00007038907,0.0000659922,0.0001103347,0.00001716841,0.02193248],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9743198,"threshold_uncertainty_score":0.6332143,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02239324758151313,"score_gpt":0.2819598571834541,"score_spread":0.259566609601941,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}