{"id":"W4301808900","doi":"10.48550/arxiv.2011.04118","title":"Joint Estimation of Expertise and Reward Preferences From Human Demonstrations","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Inference; Leverage (statistics); Robot; Computer science; Set (abstract data type); Artificial intelligence; Function (biology); Human–robot interaction; Machine learning; Human–computer interaction; Space (punctuation)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00252607,0.0005527742,0.0008677945,0.0005637662,0.0002036774,0.0006890367,0.0008026077,0.0009797777,0.001660509],"category_scores_gemma":[0.01885344,0.0004434464,0.000418039,0.0002850797,0.0009346184,0.001409071,0.0009821015,0.001314445,0.0002921461],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00063357,"about_ca_system_score_gemma":0.0006151168,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002746005,"about_ca_topic_score_gemma":0.003574238,"domain_scores_codex":[0.9988182,0.0005830781,0.00004997451,0.0002798541,0.0001772884,0.00009146616],"domain_scores_gemma":[0.9908414,0.006672074,0.000942527,0.0006446411,0.0005034222,0.0003958677],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007617306,0.0002965711,0.02694103,0.0002573035,0.000182609,0.0003517375,0.000460156,0.8246109,0.007666525,0.007170006,0.001224977,0.1300765],"study_design_scores_gemma":[0.00002710158,0.0001073876,0.006161517,0.00001677849,0.00001300698,0.00007998001,0.00004061551,0.9825794,0.001799463,0.008932077,0.0002237789,0.00001892919],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.3559405,0.0003384646,0.6399477,0.000480197,0.00001690195,0.00008008245,0.0001786822,0.0005120364,0.002505467],"genre_scores_gemma":[0.968289,0.00005769754,0.0308804,0.0000410974,0.000008646945,0.00002870353,0.00008605874,0.00001868214,0.0005897127],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.002746005,"threshold_uncertainty_score":0.01335931,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1498991308694945,"score_gpt":0.2126569037306524,"score_spread":0.06275777286115786,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}