{"id":"W4285805521","doi":"10.1145/3520304.3528766","title":"Benchmarking genetic programming in a multi-action reinforcement learning locomotion task","year":2022,"lang":"en","type":"article","venue":"Proceedings of the Genetic and Evolutionary Computation Conference Companion","topic":"Evolutionary Algorithms and Applications","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"","keywords":"Benchmarking; Reinforcement learning; Benchmark (surveying); Computer science; Task (project management); Scalability; Artificial intelligence; Action (physics); Genetic programming; Machine learning; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002159315,0.0007557105,0.0006875205,0.0005593309,0.0003896397,0.0007388931,0.001296735,0.001639844,0.001709785],"category_scores_gemma":[0.005647564,0.0002254509,0.0005425101,0.0006024285,0.0007560721,0.0006162983,0.000610817,0.00122827,0.0002822991],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007876851,"about_ca_system_score_gemma":0.001016951,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005844802,"about_ca_topic_score_gemma":0.004655512,"domain_scores_codex":[0.9992414,0.0003373098,0.00003661238,0.0001245318,0.0001561621,0.0001040471],"domain_scores_gemma":[0.9975561,0.001735983,0.0001019539,0.0001706836,0.0002922265,0.0001431775],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002118626,0.000232654,0.001226333,0.0000985384,0.00004755753,0.00007627082,0.00004067396,0.9715376,0.00125116,0.003042361,0.001079691,0.0211553],"study_design_scores_gemma":[0.00004181847,0.0001684317,0.0004317191,0.00001029869,0.000007919932,0.00001125228,0.00002378077,0.9959001,0.001061876,0.001852257,0.0004838735,0.00000671042],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8601046,0.000947754,0.1190359,0.001015262,0.0002463746,0.0002035066,0.0005089905,0.001147999,0.01678954],"genre_scores_gemma":[0.9377414,0.0001652547,0.05907408,0.0001717502,0.00001939025,0.0001510322,0.0005317349,0.00009182173,0.002053586],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.005844802,"threshold_uncertainty_score":0.01162153,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03241674665769873,"score_gpt":0.2598119153093554,"score_spread":0.2273951686516567,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}