{"id":"W2133552775","doi":"","title":"Learning from Limited Demonstrations","year":2013,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":72,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Variety (cybernetics); Task (project management); Key (lock); Path (computing); Artificial intelligence; Machine learning; Mathematical optimization; Temporal difference learning; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002062479,0.0007929308,0.001347381,0.0005296243,0.0003809226,0.0009137292,0.001989214,0.001344207,0.002846328],"category_scores_gemma":[0.0151894,0.0008444638,0.0004761583,0.0004302526,0.00123449,0.002434259,0.002187565,0.001866285,0.0006318325],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007423583,"about_ca_system_score_gemma":0.001444054,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002491084,"about_ca_topic_score_gemma":0.002838331,"domain_scores_codex":[0.9988212,0.0004620682,0.00006321898,0.0002903489,0.0002743413,0.00008883359],"domain_scores_gemma":[0.9929302,0.004766246,0.0005714397,0.0008310479,0.0006179285,0.0002831974],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001764435,0.00007799525,0.001228516,0.0001357615,0.00004777823,0.0001108524,0.000115978,0.8797263,0.002370472,0.01530744,0.001412993,0.09928946],"study_design_scores_gemma":[0.00001400782,0.0000294227,0.00007562981,0.000008254283,0.000003087372,0.00001567558,0.000005449664,0.9926934,0.0005388372,0.00632985,0.0002811763,0.000005074377],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02522689,0.0002013663,0.9721025,0.0002009107,0.00002147098,0.00004115995,0.00006336283,0.0006312011,0.001511161],"genre_scores_gemma":[0.7503097,0.000171668,0.246101,0.0001986033,0.00003952067,0.0002323111,0.000287163,0.0001194312,0.002540654],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002846328,"threshold_uncertainty_score":0.01090759,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01983794697850236,"score_gpt":0.2250760590633588,"score_spread":0.2052381120848564,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}