{"id":"W4402352014","doi":"10.1109/ijcnn60899.2024.10650768","title":"Conservative In-Distribution Q-Learning for Offline Reinforcement Learning","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"National Natural Science Foundation of China","keywords":"Reinforcement learning; Computer science; Distribution (mathematics); Reinforcement; Q-learning; Artificial intelligence; Machine learning; Engineering; Mathematics; Structural engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005031079,0.00131242,0.001528505,0.0005943187,0.0005322823,0.001328678,0.002500588,0.001349881,0.003507093],"category_scores_gemma":[0.01921455,0.0005740287,0.0005364949,0.0005491687,0.001918705,0.001656469,0.002216871,0.003496558,0.0007038252],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001625303,"about_ca_system_score_gemma":0.002677144,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002495605,"about_ca_topic_score_gemma":0.002451842,"domain_scores_codex":[0.9976327,0.001089285,0.0001168727,0.0004062323,0.0005359486,0.0002188689],"domain_scores_gemma":[0.9912272,0.006145577,0.0005633441,0.0009965401,0.0007301907,0.0003372302],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0003205514,0.0002369737,0.002045246,0.0002274358,0.00006825786,0.0001006007,0.0001224316,0.8476379,0.002728555,0.04000086,0.003278548,0.1032327],"study_design_scores_gemma":[0.00002094737,0.00004896756,0.00008787103,0.0000110991,0.000004322545,0.00001344974,0.000005716456,0.9850875,0.0007507808,0.01339362,0.0005706924,0.000005095865],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.007855391,0.0001724441,0.9896536,0.0002009109,0.00003735802,0.00005855015,0.00003649401,0.0006944338,0.001290858],"genre_scores_gemma":[0.7097112,0.0002329353,0.2849195,0.0006454912,0.00009566724,0.0004833478,0.0003205749,0.0003346637,0.003256631],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005031079,"threshold_uncertainty_score":0.02660722,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02239324758151313,"score_gpt":0.2819598571834541,"score_spread":0.259566609601941,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}