{"id":"W3127605640","doi":"10.1007/s00521-021-06375-y","title":"Improving reinforcement learning with human assistance: an argument for human subject studies with HIPPO Gym","year":2021,"lang":"en","type":"preprint","venue":"Neural Computing and Applications","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"École de Technologie Supérieure; University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; University of Alberta; Alberta Machine Intelligence Institute; Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada; Amazon Web Services; Canadian Institute for Advanced Research","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Human–computer interaction; Subject (documents); Parsing; World Wide Web","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009300175,0.0008065212,0.001267666,0.0004318466,0.0007496446,0.001109371,0.001543637,0.002488817,0.008239507],"category_scores_gemma":[0.03540395,0.0002839419,0.0004933809,0.0004016989,0.003711985,0.002934225,0.002458317,0.002554289,0.0009995274],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007305677,"about_ca_system_score_gemma":0.00156035,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001930287,"about_ca_topic_score_gemma":0.001313771,"domain_scores_codex":[0.9959318,0.002632129,0.00008405963,0.0005523277,0.0006495832,0.0001501181],"domain_scores_gemma":[0.9820037,0.0125924,0.0007696001,0.003021948,0.0009634746,0.000648867],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.004547535,0.001699804,0.01748817,0.001275036,0.0008170772,0.0003034344,0.001710365,0.1064662,0.01559434,0.2678783,0.02140073,0.560819],"study_design_scores_gemma":[0.00090467,0.00252731,0.0157525,0.0002270928,0.0002352842,0.0004076697,0.0005363976,0.2940933,0.01394735,0.6356454,0.03562536,0.00009773493],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1528026,0.002008246,0.7617952,0.0247726,0.0005549102,0.0004011601,0.0002962472,0.001762575,0.05560652],"genre_scores_gemma":[0.9191466,0.0005881989,0.07006444,0.002466913,0.0002784939,0.0003272815,0.0001043219,0.0001925476,0.006831175],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.009300175,"threshold_uncertainty_score":0.04918462,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04079369190192602,"score_gpt":0.3265417845465773,"score_spread":0.2857480926446513,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}