{"id":"W3151722529","doi":"10.15607/rss.2021.xvii.012","title":"Learning Generalizable Robotic Reward Functions from “In-The-Wild” Human Videos","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Domain Adaptation and Few-Shot Learning","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Office of Naval Research; Canadian Institute for Advanced Research; National Science Foundation","keywords":"Computer science; Artificial intelligence; Generalization; Reinforcement learning; Task (project management); Robot; Function (biology); Discriminator; Machine learning; Robotics; Human–computer interaction","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.0007099799,0.0003135915,0.0003901346,0.0002202117,0.0004620251,0.001447354,0.001366792,0.0002338241,0.000628278],"category_scores_gemma":[0.0001264799,0.000278213,0.0002242059,0.0004991384,0.0000418688,0.0003935033,0.001153331,0.001567055,0.0002042442],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001059384,"about_ca_system_score_gemma":0.0001902419,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002512634,"about_ca_topic_score_gemma":0.0004649592,"domain_scores_codex":[0.9968237,0.0007076138,0.0005085582,0.0009873239,0.0005470217,0.0004258336],"domain_scores_gemma":[0.9981491,0.0002351881,0.0002236522,0.001173277,0.0001150325,0.0001037441],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000001612122,0.00008919952,0.002519516,0.00002578823,0.00005828293,0.00009220412,0.005710019,0.9726325,0.000321728,0.01237176,0.002219993,0.003957342],"study_design_scores_gemma":[0.0008851964,0.0001199702,0.02219032,0.0003726368,0.00007672302,0.00004275391,0.004310167,0.9216506,0.000086022,0.00605882,0.0428582,0.001348519],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02006108,0.0003462879,0.9530897,0.001907037,0.001166445,0.000249234,8.541519e-7,0.0003709771,0.02280845],"genre_scores_gemma":[0.8644133,0.00005224088,0.1128163,0.001728288,0.0003570279,0.0001037244,0.0002599948,0.00003886073,0.02023029],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8443522,"threshold_uncertainty_score":0.999967,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04143465606680853,"score_gpt":0.2716297009467223,"score_spread":0.2301950448799138,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}