{"id":"W4287181747","doi":"10.48550/arxiv.2105.07099","title":"Feature-Based Interpretable Reinforcement Learning based on State-Transition Models","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University","funders":"","keywords":"Reinforcement learning; Computer science; Locality; Artificial intelligence; Feature (linguistics); Reinforcement; State (computer science); Function (biology); Action (physics); Machine learning; Engineering; Algorithm","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0004506296,0.0004577365,0.000412347,0.0004683197,0.0003032307,0.0004659229,0.001517379,0.0003075257,0.00005977086],"category_scores_gemma":[0.00004872193,0.0005625908,0.0003287317,0.0008333242,0.00007396031,0.0008584771,0.0006582514,0.001209649,0.00007735859],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006475069,"about_ca_system_score_gemma":0.0006100399,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003844966,"about_ca_topic_score_gemma":0.00007951836,"domain_scores_codex":[0.9970616,0.0003452449,0.0002585156,0.001470778,0.0002656275,0.000598268],"domain_scores_gemma":[0.9974758,0.0001671121,0.0002796797,0.001464385,0.0003915764,0.0002214368],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00009731867,0.0001017932,0.00003382713,0.00009421424,0.00003718531,0.0004403556,0.0005473422,0.9833494,0.00009758074,0.01480122,0.0001073259,0.0002923628],"study_design_scores_gemma":[0.0002598172,0.000188455,0.000007503954,0.0004204701,0.00003528842,8.316167e-7,0.0002374337,0.9837357,0.006787883,0.007632725,0.0001502922,0.0005435474],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02620122,0.00002461239,0.9688629,0.0004650279,0.0004581086,0.0004011703,0.000004523272,0.0003877111,0.00319471],"genre_scores_gemma":[0.9941153,0.00003577627,0.003345736,0.0007328605,0.00003039622,0.000004535409,0.00008713021,0.00003125979,0.001616997],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9679141,"threshold_uncertainty_score":0.9996825,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06063147237957493,"score_gpt":0.1949988079869347,"score_spread":0.1343673356073598,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}