{"id":"W2946669606","doi":"10.65109/ugvb1408","title":"Building Knowledge for AI Agents with Reinforcement Learning","year":2019,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Generalization; Artificial intelligence; Function (biology); Reinforcement; Machine learning; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001604625,0.0007428235,0.0008214395,0.0005455415,0.0006801105,0.001828856,0.001882154,0.001602061,0.002668699],"category_scores_gemma":[0.007512589,0.0004913175,0.0007595062,0.0004307449,0.002690752,0.003505005,0.002677595,0.002967376,0.0005735926],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001451432,"about_ca_system_score_gemma":0.001388062,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004386687,"about_ca_topic_score_gemma":0.005012959,"domain_scores_codex":[0.9992238,0.0002570462,0.00006976737,0.0001334151,0.0002500988,0.00006574075],"domain_scores_gemma":[0.9979562,0.001090148,0.0001853786,0.0003472245,0.0002884764,0.0001326171],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00006219074,0.00008074973,0.0007210457,0.0001895334,0.00008290537,0.0001320802,0.0002764454,0.6747615,0.00181935,0.2498268,0.002081262,0.06996625],"study_design_scores_gemma":[0.0000209869,0.00002925633,0.0000672899,0.00003905666,0.00001483493,0.00002550872,0.00003040207,0.7752737,0.001000109,0.220061,0.003424834,0.00001304944],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.008540421,0.0003356022,0.9847483,0.001297646,0.00005963044,0.00007517634,0.00005539689,0.0004786497,0.004409153],"genre_scores_gemma":[0.4045023,0.0008551427,0.590044,0.0004218955,0.0001092249,0.000384531,0.0001898899,0.0001377842,0.003355175],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004386687,"threshold_uncertainty_score":0.01053095,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01839217257182383,"score_gpt":0.2856183917926542,"score_spread":0.2672262192208303,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}