{"id":"W3103898737","doi":"10.1109/tro.2022.3192969","title":"Joint Estimation of Expertise and Reward Preferences From Human Demonstrations","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Robotics","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Inference; Robot; Leverage (statistics); Computer science; Human–robot interaction; Artificial intelligence; Set (abstract data type); Function (biology); Machine learning; Human–computer interaction; Human behavior; Space (punctuation)","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002509573,0.0005807838,0.0009082872,0.0005882572,0.0002060288,0.0007406923,0.0007843191,0.0010565,0.001716635],"category_scores_gemma":[0.01943565,0.0004785152,0.0004642044,0.0002968586,0.0009233728,0.001539403,0.00109973,0.001310082,0.0002809948],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005670856,"about_ca_system_score_gemma":0.0006180439,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002719268,"about_ca_topic_score_gemma":0.003564295,"domain_scores_codex":[0.9987751,0.0005659384,0.00005547924,0.0002957035,0.0002064032,0.0001012898],"domain_scores_gemma":[0.9908704,0.006681649,0.0009455423,0.0006278549,0.0004835694,0.0003909369],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007754891,0.0002648285,0.02527427,0.0002343263,0.0001793167,0.0003745162,0.0004464583,0.8380653,0.007611177,0.006155399,0.0009110942,0.1197078],"study_design_scores_gemma":[0.0000258312,0.0001284684,0.006927511,0.00001658578,0.0000139822,0.000100313,0.00004542206,0.9832758,0.001846542,0.007402983,0.0001937718,0.00002276712],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3685457,0.0003538213,0.6276942,0.0003529062,0.0000146121,0.00008089862,0.0001634421,0.0004347311,0.002359735],"genre_scores_gemma":[0.9704756,0.00006282362,0.02876136,0.00003675828,0.00000841868,0.0000298578,0.00007953762,0.00001726459,0.0005284068],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002719268,"threshold_uncertainty_score":0.01327211,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03686872366786004,"score_gpt":0.2569113578331829,"score_spread":0.2200426341653229,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}