{"id":"W4301808900","doi":"10.48550/arxiv.2011.04118","title":"Joint Estimation of Expertise and Reward Preferences From Human Demonstrations","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Inference; Leverage (statistics); Robot; Computer science; Set (abstract data type); Artificial intelligence; Function (biology); Human–robot interaction; Machine learning; Human–computer interaction; Space (punctuation)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00007958702,0.0001752847,0.0002509505,0.0001165896,0.0001158499,0.0001037554,0.0007033638,0.0001480741,0.00001691328],"category_scores_gemma":[0.00005678831,0.0002034048,0.000073497,0.0001961223,0.0001198439,0.0003166735,0.001090816,0.0002981948,0.00001096061],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005866869,"about_ca_system_score_gemma":0.0001052453,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003166578,"about_ca_topic_score_gemma":0.00001994719,"domain_scores_codex":[0.9988484,0.00008788906,0.0002526702,0.0005849474,0.00009698384,0.0001291331],"domain_scores_gemma":[0.9988328,0.00006013352,0.0003513427,0.0005770326,0.00007711314,0.0001015972],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000003160535,0.00001380489,0.001721536,0.00004003431,0.00004540665,0.00001368786,0.0008007613,0.9538656,0.000262176,0.04273732,0.00004522482,0.0004512886],"study_design_scores_gemma":[0.0001705905,0.00006270996,0.005697495,0.000112908,0.00004459648,3.419523e-7,0.00006535621,0.9645393,0.0005316202,0.02857611,0.000009884115,0.0001890412],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2308167,0.00002098981,0.7677718,0.00009449859,0.0001230488,0.0001470431,0.000005515902,0.0001009266,0.0009195272],"genre_scores_gemma":[0.9801653,0.00005675248,0.01958294,0.00002287253,0.0000216792,6.618292e-7,0.00003204688,0.000006612516,0.0001111092],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7493486,"threshold_uncertainty_score":0.8294606,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1498991308694945,"score_gpt":0.2126569037306524,"score_spread":0.06275777286115786,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}