{"id":"W4287333197","doi":"10.48550/arxiv.2102.02639","title":"Improving Reinforcement Learning with Human Assistance: An Argument for\\n Human Subject Studies with HIPPO Gym","year":2021,"lang":"","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Northern Alberta Institute of Technology","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Human–computer interaction; Subject (documents); World Wide Web","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00801839,0.000933144,0.0006573645,0.0004944145,0.000904389,0.002034174,0.002155273,0.002221223,0.01237028],"category_scores_gemma":[0.02689632,0.000387075,0.0006110336,0.0004549159,0.006529384,0.005918117,0.003761406,0.004346148,0.001944668],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001984974,"about_ca_system_score_gemma":0.002552174,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003524204,"about_ca_topic_score_gemma":0.002493668,"domain_scores_codex":[0.9952709,0.002815712,0.00008540902,0.0007558872,0.0008685067,0.0002035647],"domain_scores_gemma":[0.9791542,0.01538424,0.0007969479,0.002688732,0.00110361,0.0008722129],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0007511994,0.0006798893,0.003557609,0.0006704561,0.0001726594,0.0001659067,0.00170658,0.07736798,0.003278029,0.631272,0.02236606,0.2580116],"study_design_scores_gemma":[0.0005203006,0.0006909383,0.002274976,0.0002725129,0.00007744449,0.0001573581,0.0003805352,0.2510933,0.006895756,0.6277183,0.1098426,0.00007598658],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02282516,0.001783893,0.8879724,0.03232243,0.0003857403,0.000143748,0.00009349269,0.002965704,0.05150749],"genre_scores_gemma":[0.755356,0.001264356,0.2168035,0.0051205,0.0004637764,0.0004703232,0.0001227882,0.0007470956,0.01965171],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01237028,"threshold_uncertainty_score":0.04240578,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09487229354682383,"score_gpt":0.2386730474669452,"score_spread":0.1438007539201214,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}