{"id":"W7124241732","doi":"10.65109/xgyf2237","title":"Interpretable Preference-based Reinforcement Learning with Tree-Structured Reward Functions","year":2022,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Thales (Canada)","funders":"","keywords":"Interpretability; Reinforcement learning; Robustness (evolution); Function (biology); Heuristic; Preference","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001895125,0.0007822828,0.000878378,0.0003794435,0.0003509192,0.0008903412,0.001093209,0.001137347,0.002101946],"category_scores_gemma":[0.01222292,0.0004557009,0.0004610689,0.0003452955,0.001432289,0.001598155,0.001079127,0.001908678,0.0003438958],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009670716,"about_ca_system_score_gemma":0.001136042,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002061676,"about_ca_topic_score_gemma":0.002646865,"domain_scores_codex":[0.9991666,0.0004217256,0.0000418554,0.0001410403,0.0001424483,0.00008636318],"domain_scores_gemma":[0.9954125,0.003148538,0.0004448277,0.000364702,0.0004266727,0.0002026719],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001648202,0.00009216411,0.001338872,0.00007185111,0.00002640856,0.00009359415,0.0001391339,0.9286723,0.003203019,0.02777684,0.0006258684,0.03779517],"study_design_scores_gemma":[0.00001422376,0.00002635291,0.00005745327,0.000003831933,0.000002480049,0.000006908849,0.000004587471,0.9880087,0.0004185227,0.01136641,0.00008659215,0.000003987005],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05437326,0.00008059584,0.9436297,0.0001999406,0.0000195548,0.00005065714,0.00005040731,0.0004254773,0.001170279],"genre_scores_gemma":[0.8551107,0.00005587826,0.1433874,0.0001141193,0.00001280716,0.0001219931,0.00007585721,0.0000881502,0.001033066],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002101946,"threshold_uncertainty_score":0.01002252,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02839159268205176,"score_gpt":0.2199169094736677,"score_spread":0.1915253167916159,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}