{"id":"W3216764023","doi":"10.48550/arxiv.2110.03184","title":"Explaining Deep Reinforcement Learning Agents In The Atari Domain through a Surrogate Model","year":2021,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Suite; Domain (mathematical analysis); Transformation (genetics); Replicate; Deep learning; Representation (politics); Deep neural networks; Surrogate model; Machine learning; Mathematics; Law","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001370141,0.0007313964,0.0006190478,0.0002622766,0.000308051,0.0008689827,0.001208817,0.001414793,0.002388979],"category_scores_gemma":[0.006972444,0.0003397488,0.0005708158,0.0001811084,0.001341452,0.001428755,0.001251409,0.002657888,0.0003064046],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009478597,"about_ca_system_score_gemma":0.0008999784,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002465256,"about_ca_topic_score_gemma":0.00349623,"domain_scores_codex":[0.9994298,0.0002939967,0.00002172919,0.0001011417,0.00009758478,0.00005579638],"domain_scores_gemma":[0.9975814,0.001624519,0.0002414825,0.0002705278,0.0001699887,0.0001120913],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00006801115,0.00004553325,0.0008759012,0.00004801236,0.00002632863,0.0001021971,0.0001529707,0.9429709,0.001381483,0.04060899,0.0007241477,0.01299543],"study_design_scores_gemma":[0.000007224293,0.00001720691,0.00005274718,0.000004193447,0.000002087686,0.000006894718,0.000006866731,0.985374,0.0002170469,0.01412904,0.0001799082,0.000002882087],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0936293,0.0001500713,0.899895,0.0010705,0.00005193976,0.00008232312,0.0001549169,0.0005613652,0.004404576],"genre_scores_gemma":[0.8956662,0.00008338753,0.1008983,0.0001937264,0.00002207957,0.0001311781,0.0001450538,0.0000728438,0.002787251],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002465256,"threshold_uncertainty_score":0.00799197,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1407612288090627,"score_gpt":0.2314626391637081,"score_spread":0.0907014103546454,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}