{"id":"W4287181747","doi":"10.48550/arxiv.2105.07099","title":"Feature-Based Interpretable Reinforcement Learning based on State-Transition Models","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University","funders":"","keywords":"Reinforcement learning; Computer science; Locality; Artificial intelligence; Feature (linguistics); Reinforcement; State (computer science); Function (biology); Action (physics); Machine learning; Engineering; Algorithm","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009510652,0.0007331424,0.00074928,0.000469534,0.0002790653,0.0007807323,0.001261873,0.000789525,0.002331379],"category_scores_gemma":[0.006448707,0.0003179268,0.0006494605,0.0003570033,0.0008799314,0.001527736,0.000988757,0.001945222,0.0002326241],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001002467,"about_ca_system_score_gemma":0.0007761757,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003336648,"about_ca_topic_score_gemma":0.00451545,"domain_scores_codex":[0.9993758,0.0002637955,0.00003554558,0.0001373576,0.0001370949,0.00005054461],"domain_scores_gemma":[0.9964808,0.002566185,0.0003267374,0.0003015437,0.000232302,0.00009237267],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000112758,0.00006899644,0.001271352,0.00007679899,0.00004934674,0.0001435676,0.0001957915,0.9037742,0.001728644,0.03500357,0.0008934246,0.05668155],"study_design_scores_gemma":[0.00000775085,0.00001373916,0.0000532776,0.000003339645,0.000004035036,0.000007910547,0.000003121575,0.9887465,0.0002494458,0.01077454,0.0001330733,0.000003300255],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01628422,0.00007617527,0.9818334,0.0002000748,0.00001853348,0.0000292725,0.0000705866,0.0007179752,0.000769663],"genre_scores_gemma":[0.8329707,0.00009859259,0.1652684,0.00009896148,0.00002908072,0.0001546469,0.0001636458,0.0000910093,0.001124933],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003336648,"threshold_uncertainty_score":0.007799208,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06063147237957493,"score_gpt":0.1949988079869347,"score_spread":0.1343673356073598,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}