{"id":"W2946669606","doi":"10.65109/ugvb1408","title":"Building Knowledge for AI Agents with Reinforcement Learning","year":2019,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Generalization; Artificial intelligence; Function (biology); Reinforcement; Machine learning; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003173524,0.0001696875,0.0001703133,0.0001182655,0.0001502443,0.0002256976,0.0007002175,0.00004963467,0.0001427727],"category_scores_gemma":[0.00003780441,0.0001350436,0.00006096069,0.0002640668,0.00001753012,0.0005495141,0.0002817974,0.0001855643,0.0003647088],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008207352,"about_ca_system_score_gemma":0.00007764645,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00000629082,"about_ca_topic_score_gemma":8.38015e-7,"domain_scores_codex":[0.9986644,0.00002653072,0.0002335471,0.00036938,0.0002797669,0.000426419],"domain_scores_gemma":[0.9990621,0.0001231193,0.0001135007,0.0004685642,0.0001447871,0.0000879718],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000008817993,0.000008587284,0.001599065,0.00004038466,0.00002776631,7.747044e-7,0.0002931968,0.8964993,0.0002104177,0.09786142,0.001167222,0.002283057],"study_design_scores_gemma":[0.0007443922,0.00052061,0.000205639,0.00004720006,0.000006104549,0.000003488445,0.00002560817,0.9178354,0.001167485,0.00008442525,0.07914468,0.000214923],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.002459424,0.00001094973,0.9530084,0.0002095713,0.0003124662,0.0005334624,4.078256e-8,0.0002719782,0.04319374],"genre_scores_gemma":[0.8069296,0.000002671742,0.146452,0.0004237932,0.00004281887,0.00003008305,0.000003025851,0.00001981007,0.04609619],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8065563,"threshold_uncertainty_score":0.550692,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01839217257182383,"score_gpt":0.2856183917926542,"score_spread":0.2672262192208303,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}