{"id":"W7124289677","doi":"10.65109/ldoy7418","title":"Leveraging Sub-Optimal Data for Human-in-the-Loop Reinforcement Learning","year":2024,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Function (biology); Data collection; Human-in-the-loop; Temporal difference learning; Work (physics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002155679,0.001113976,0.001121351,0.0004910338,0.0004594874,0.0009877237,0.001285904,0.001066308,0.002406666],"category_scores_gemma":[0.01123221,0.0006225571,0.0003618178,0.0003310551,0.001719277,0.001656097,0.002492277,0.002434743,0.0005367284],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007283762,"about_ca_system_score_gemma":0.001717094,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002825218,"about_ca_topic_score_gemma":0.004196095,"domain_scores_codex":[0.9991912,0.0003115463,0.00004118143,0.0001611738,0.0002048776,0.00009002471],"domain_scores_gemma":[0.9961312,0.002457512,0.000380478,0.0004476649,0.000382573,0.0002005841],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001233426,0.0001347425,0.001050495,0.00008175019,0.00002819018,0.00007063671,0.0001179072,0.9321096,0.003359864,0.008269439,0.0009322796,0.05372173],"study_design_scores_gemma":[0.00001215915,0.00004523719,0.00009724985,0.000007673149,0.000002473871,0.00001115093,0.000007300962,0.9941055,0.001014299,0.004329923,0.0003610215,0.000006097833],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02286742,0.0001399947,0.9743401,0.0002174854,0.00002868772,0.00004675602,0.00003018145,0.0005709326,0.001758383],"genre_scores_gemma":[0.8201376,0.00009821528,0.1777368,0.0002074656,0.00002408552,0.0001720592,0.0001048948,0.0001503406,0.001368571],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002825218,"threshold_uncertainty_score":0.01140046,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09075955790988156,"score_gpt":0.3330438801573506,"score_spread":0.242284322247469,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}