{"id":"W7006393979","doi":"","title":"Towards building model-based reinforcement learning agents that effectively adapt and generalize","year":2025,"lang":"en","type":"dissertation","venue":"eScholarship@McGill (McGill)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Control (management); Key (lock); Action (physics); Feature (linguistics); Stability (learning theory)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.001450861,0.001223199,0.001079347,0.000941956,0.001908906,0.0006146842,0.002017444,0.0008762866,0.00004176876],"category_scores_gemma":[0.000861944,0.001346601,0.0004297188,0.0008656983,0.00005860224,0.001375937,0.0006411736,0.002386807,0.00004327658],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001184903,"about_ca_system_score_gemma":0.0002756884,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002542647,"about_ca_topic_score_gemma":0.0000437871,"domain_scores_codex":[0.9936041,0.0005729567,0.001039509,0.001880406,0.001717427,0.001185662],"domain_scores_gemma":[0.9964672,0.0003522851,0.00102171,0.001244253,0.0004770433,0.0004374938],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0000766417,0.00003451355,0.00003098045,0.0005573319,0.0002145788,0.00003755813,0.00002330091,0.8642599,0.003306813,0.08203117,0.00001130109,0.04941589],"study_design_scores_gemma":[0.001627905,0.0003472762,0.0004093728,0.001331226,0.0002407169,0.000007947579,0.00004528716,0.9281761,0.04849283,0.003361081,0.01433392,0.001626308],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4534213,0.001235023,0.2381286,0.0002152024,0.007991128,0.006173604,0.000156468,0.003830113,0.2888486],"genre_scores_gemma":[0.951793,0.0002546015,0.02297812,0.0005345563,0.00003406938,0.0001534144,0.0003901761,0.0001571542,0.02370489],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4983718,"threshold_uncertainty_score":0.9999147,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02959898004858515,"score_gpt":0.268028614514628,"score_spread":0.2384296344660428,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}