{"id":"W6991803794","doi":"","title":"Inductive biases and generalisation for deep reinforcement learning","year":2021,"lang":"en","type":"dissertation","venue":"Oxford University Research Archive (ORA) (University of Oxford)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kellogg's (Canada)","funders":"","keywords":"Reinforcement learning; Focus (optics); Scope (computer science); Transfer of learning; Artificial neural network; Deep learning; Moment (physics); Inductive bias","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts"],"consensus_categories":[],"category_scores_codex":[0.0008329519,0.0003855276,0.0005977717,0.001523936,0.001684018,0.0001663721,0.001826089,0.000334303,0.00008403006],"category_scores_gemma":[0.0003791498,0.0005359598,0.0002973947,0.001088449,0.0004219259,0.001198034,0.001231635,0.001213134,0.000003638275],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004153235,"about_ca_system_score_gemma":0.0007546219,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0006397894,"about_ca_topic_score_gemma":0.001038005,"domain_scores_codex":[0.9962018,0.0005820633,0.000271396,0.001003887,0.001121468,0.0008193701],"domain_scores_gemma":[0.9962498,0.0009527518,0.0005308914,0.000683043,0.001255796,0.0003276963],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003573203,0.0004312619,0.005260586,0.00304798,0.002391623,0.0007085487,0.09185699,0.5746467,0.006004466,0.2264147,0.0057098,0.07995414],"study_design_scores_gemma":[0.003552161,0.002150542,0.005815085,0.0008440276,0.000241235,0.00001242509,0.1094619,0.6918239,0.0005813339,0.001805508,0.1824386,0.001273253],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05586721,0.0001181639,0.9143881,0.0003063437,0.0002675701,0.001193986,0.00002841858,0.0001244617,0.02770579],"genre_scores_gemma":[0.5721554,0.01433364,0.2344369,0.00006681485,0.0002935318,0.000005797342,0.00936716,0.000168576,0.1691722],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.6799511,"threshold_uncertainty_score":0.9997092,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04855249049508199,"score_gpt":0.2839006775695266,"score_spread":0.2353481870744447,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}