{"id":"W3176022961","doi":"10.1145/3449639.3459348","title":"On the impact of tangled program graph marking schemes under the atari reinforcement learning benchmark","year":2021,"lang":"en","type":"article","venue":"Proceedings of the Genetic and Evolutionary Computation Conference","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Reinforcement learning; Benchmark (surveying); Graph; Adaptation (eye); Heuristic; Scheme (mathematics); Theoretical computer science; Modularity (biology); Artificial intelligence; Node (physics); Machine learning; Engineering; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004013534,0.0009272197,0.0005331166,0.0009376683,0.0005932075,0.0009867746,0.001314551,0.001032569,0.002426487],"category_scores_gemma":[0.02343968,0.0002155092,0.0003056029,0.0006385919,0.0009995369,0.001532242,0.001082386,0.001457332,0.0002687853],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001414798,"about_ca_system_score_gemma":0.001211096,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007742818,"about_ca_topic_score_gemma":0.01199671,"domain_scores_codex":[0.998123,0.0008871703,0.0001108401,0.0002516391,0.0002982605,0.0003290011],"domain_scores_gemma":[0.9673177,0.02596503,0.001565652,0.002083499,0.00154849,0.001519684],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001861516,0.001263095,0.01182569,0.0002907728,0.0001182599,0.000129994,0.0001087029,0.9114919,0.00455171,0.006124481,0.003435387,0.05879855],"study_design_scores_gemma":[0.0001809718,0.001397302,0.004123457,0.00004956938,0.00005442823,0.00004307122,0.0001405742,0.9823889,0.004522883,0.006208568,0.0008694576,0.00002077173],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9829288,0.0006289054,0.006829748,0.0005195426,0.00006277929,0.00005518169,0.0003058686,0.0006843286,0.007984968],"genre_scores_gemma":[0.991602,0.00009209328,0.0070393,0.00008141765,0.00001053517,0.00003074034,0.0003466074,0.00007659977,0.0007207352],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.007742818,"threshold_uncertainty_score":0.02122587,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0203845929356071,"score_gpt":0.258709105662269,"score_spread":0.2383245127266619,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}