{"id":"W4287333197","doi":"10.48550/arxiv.2102.02639","title":"Improving Reinforcement Learning with Human Assistance: An Argument for\\n Human Subject Studies with HIPPO Gym","year":2021,"lang":"","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Northern Alberta Institute of Technology","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Human–computer interaction; Subject (documents); World Wide Web","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts","scholarly_communication","research_integrity"],"consensus_categories":["metaepi_narrow"],"category_scores_codex":[0.001666826,0.001746915,0.001777353,0.0007230197,0.003759756,0.001583023,0.003306813,0.0005298103,0.00009841102],"category_scores_gemma":[0.0001211567,0.001779191,0.0004986386,0.001553725,0.0009161659,0.002240939,0.003387555,0.002416262,0.00001437346],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002435691,"about_ca_system_score_gemma":0.001068347,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004335818,"about_ca_topic_score_gemma":0.0006813187,"domain_scores_codex":[0.990953,0.0006787445,0.001196899,0.004330982,0.0008670476,0.001973305],"domain_scores_gemma":[0.9908639,0.0003456899,0.002670128,0.003410275,0.002054678,0.0006553647],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003206404,0.0001995557,0.01046695,0.001127016,0.001701905,0.0007115409,0.004617704,0.9495903,0.0005912809,0.03024579,0.00001387925,0.0004135019],"study_design_scores_gemma":[0.005336358,0.008024834,0.001690273,0.002410361,0.001400928,0.00003107393,0.01288493,0.9630706,0.001207217,0.0004247963,0.0003757115,0.003142874],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3222375,0.0001205751,0.6741806,0.00003304251,0.0003882413,0.001552633,0.000001982183,0.0003290873,0.001156321],"genre_scores_gemma":[0.9756898,0.000260792,0.008826072,0.00009999012,0.0002251694,0.00003127893,0.0001794796,0.0001628784,0.01452452],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6653545,"threshold_uncertainty_score":0.9998852,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09487229354682383,"score_gpt":0.2386730474669452,"score_spread":0.1438007539201214,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}