{"id":"W4391157052","doi":"","title":"Using Confounded Data in Latent Model-Based Reinforcement Learning","year":2023,"lang":"en","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Ubisoft (Canada); Montfort Hospital; Minnow Environmental (Canada)","funders":"Canada First Research Excellence Fund; Canada Excellence Research Chairs, Government of Canada","keywords":"Reinforcement learning; Computer science; Reinforcement; Latent inhibition; Artificial intelligence; Cognitive psychology; Machine learning; Psychology; Statistics; Social psychology; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01017443,0.001315579,0.003418941,0.0007657161,0.0007446848,0.001867461,0.003367984,0.003365435,0.003818823],"category_scores_gemma":[0.05324675,0.001529226,0.0009079324,0.001037048,0.002656735,0.00412324,0.004157359,0.005068934,0.0006235921],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001801305,"about_ca_system_score_gemma":0.001864174,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005903688,"about_ca_topic_score_gemma":0.005435613,"domain_scores_codex":[0.9959601,0.002583086,0.0001707463,0.000663391,0.0003855073,0.0002372884],"domain_scores_gemma":[0.9488193,0.04450164,0.001840172,0.002058072,0.001849339,0.0009315074],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00060567,0.0001784784,0.001519925,0.0001863585,0.0001141692,0.00009382505,0.0001230817,0.9348584,0.0008072277,0.02082529,0.0009853635,0.0397022],"study_design_scores_gemma":[0.00001811909,0.00002109036,0.00005199742,0.000006021963,0.000005672053,0.000004360206,0.000002397035,0.9927618,0.0001062994,0.006957294,0.00005985631,0.00000496033],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02378368,0.0003054987,0.9742951,0.0004466413,0.00006045844,0.00005273424,0.00008159465,0.0004329203,0.0005413733],"genre_scores_gemma":[0.877867,0.0002051851,0.1183973,0.0003655064,0.0001049494,0.0002604799,0.0002867302,0.0001466325,0.002366176],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01017443,"threshold_uncertainty_score":0.05380815,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06579403889571468,"score_gpt":0.2846853523510118,"score_spread":0.2188913134552972,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}