{"id":"W7006393979","doi":"","title":"Towards building model-based reinforcement learning agents that effectively adapt and generalize","year":2025,"lang":"en","type":"dissertation","venue":"eScholarship@McGill (McGill)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Control (management); Key (lock); Action (physics); Feature (linguistics); Stability (learning theory)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006701713,0.0005640797,0.0005876031,0.0002887876,0.0003671292,0.0008551364,0.001028394,0.000928142,0.002235015],"category_scores_gemma":[0.002490688,0.000495969,0.0005905987,0.0001767478,0.0006629365,0.001042774,0.001660725,0.001552131,0.0008266813],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005029311,"about_ca_system_score_gemma":0.0009011041,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003244981,"about_ca_topic_score_gemma":0.004253737,"domain_scores_codex":[0.9997733,0.00006191001,0.00001442319,0.00006104507,0.00005840408,0.00003091386],"domain_scores_gemma":[0.9993961,0.0002723846,0.00005768718,0.0001092363,0.0001144697,0.00005018621],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00008292049,0.0001805397,0.001111408,0.00007112249,0.00005561304,0.00008013984,0.0001230125,0.8614561,0.01232969,0.01407736,0.00242943,0.1080026],"study_design_scores_gemma":[0.0000114799,0.00002194724,0.00004371967,0.000003362552,0.000006413625,0.000009802468,0.000007190857,0.994498,0.001469892,0.003248597,0.0006767994,0.000002837741],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02895408,0.00009616094,0.9660364,0.0002855408,0.00005160483,0.00006524751,0.00003266337,0.001512707,0.002965552],"genre_scores_gemma":[0.5208539,0.0001683472,0.4721358,0.0002546838,0.00002782038,0.0002430652,0.0001602902,0.0002531031,0.005902933],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003244981,"threshold_uncertainty_score":0.007476926,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02959898004858515,"score_gpt":0.268028614514628,"score_spread":0.2384296344660428,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}