{"id":"W7009802497","doi":"","title":"Evaluation of sample efficiency in offline reinforcement learning","year":2023,"lang":"en","type":"other","venue":"Espace École de technologie supérieure (École de technologie supérieure)","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Reinforcement learning; Robustness (evolution); Offline learning; Process (computing); Online machine learning; Online and offline; Set (abstract data type); Function (biology)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008764444,0.001557632,0.001373279,0.001031793,0.0006754234,0.001700853,0.002184743,0.001565783,0.002822844],"category_scores_gemma":[0.04566317,0.0004549758,0.0006382671,0.0008428692,0.001588839,0.002922356,0.001964384,0.002069937,0.001207083],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001470271,"about_ca_system_score_gemma":0.001890791,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004115187,"about_ca_topic_score_gemma":0.003425599,"domain_scores_codex":[0.994846,0.001984707,0.000431194,0.001123024,0.001254438,0.0003605781],"domain_scores_gemma":[0.9614871,0.02528021,0.0015022,0.0073465,0.00363168,0.0007522693],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003426408,0.001230352,0.01527902,0.0004715884,0.0002848185,0.000143718,0.0002169537,0.7250338,0.008314336,0.008762719,0.00740796,0.2294284],"study_design_scores_gemma":[0.000136808,0.0005680527,0.001960498,0.00003187421,0.00002768598,0.00006329991,0.00006550338,0.9811792,0.009733253,0.005072873,0.001140583,0.00002038845],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4803246,0.001952053,0.4940634,0.00113606,0.0003596137,0.0005013416,0.001823,0.008104998,0.01173496],"genre_scores_gemma":[0.827917,0.0002461989,0.1645023,0.0002388441,0.00007367754,0.0004869242,0.002866421,0.0007875661,0.002881001],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008764444,"threshold_uncertainty_score":0.04635137,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03321411594561836,"score_gpt":0.3128450508131989,"score_spread":0.2796309348675806,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}