{"id":"W3161449188","doi":"10.1109/smc52423.2021.9658917","title":"Feature-Based Interpretable Reinforcement Learning based on State-Transition Models","year":2021,"lang":"en","type":"article","venue":"2021 IEEE International Conference on Systems, Man, and Cybernetics (SMC)","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University","funders":"","keywords":"Reinforcement learning; Computer science; Feature (linguistics); Transition (genetics); Artificial intelligence; State (computer science); Machine learning; Pattern recognition (psychology); Algorithm","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009866412,0.0007969831,0.0008428218,0.0004878023,0.0002915282,0.0008208206,0.001293549,0.0008231727,0.002336384],"category_scores_gemma":[0.00624338,0.0003293721,0.0006435071,0.0003383448,0.0008616364,0.001480293,0.0009859322,0.001937799,0.0002402442],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009721372,"about_ca_system_score_gemma":0.0008273689,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003514535,"about_ca_topic_score_gemma":0.004420963,"domain_scores_codex":[0.9993631,0.0002497553,0.00003854475,0.0001420037,0.0001516068,0.00005487621],"domain_scores_gemma":[0.9967636,0.002311142,0.000321553,0.0002587849,0.0002530222,0.00009200961],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001091299,0.00006421335,0.001075606,0.00006877947,0.00004383002,0.0001380513,0.0001724,0.9170943,0.001618372,0.02624029,0.0006868999,0.05268807],"study_design_scores_gemma":[0.000007474257,0.00001487008,0.00004420956,0.000003102214,0.000003652318,0.000007924894,0.000002743915,0.9925122,0.0002257544,0.007066413,0.0001084299,0.000003242834],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01372601,0.00006843942,0.9845659,0.0001543171,0.00001822003,0.00003026844,0.00005405522,0.0006739142,0.000708792],"genre_scores_gemma":[0.8473023,0.00008919353,0.1510523,0.00009395422,0.00002800936,0.000164256,0.0001319502,0.00008409457,0.001053896],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003514535,"threshold_uncertainty_score":0.007815957,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04755566164385049,"score_gpt":0.27826063322828,"score_spread":0.2307049715844295,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}