{"id":"W4311681064","doi":"10.22215/etd/2022-15178","title":"Learning Transition Dynamics via Rewarded Exploration: A Study using Unity's MLAgents","year":2022,"lang":"en","type":"dissertation","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Popularity; Artificial intelligence; Hyperparameter; Transition (genetics); Action (physics); Dynamics (music); Variable (mathematics); Focus (optics); Artificial neural network; State (computer science); Game engine; Machine learning; Work (physics); Human–computer interaction; Engineering; Psychology; Social psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.000586794,0.0003808365,0.0003664483,0.0004571945,0.0008638314,0.0004232738,0.0009710055,0.000168812,0.0003601297],"category_scores_gemma":[0.00005238764,0.000441126,0.0001306501,0.0009433812,0.00001192707,0.00103797,0.0001546351,0.0009977759,0.00003652348],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005471492,"about_ca_system_score_gemma":0.0002108242,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002437037,"about_ca_topic_score_gemma":0.0002134004,"domain_scores_codex":[0.9967247,0.0005084162,0.0006169527,0.0006698515,0.001131564,0.0003485308],"domain_scores_gemma":[0.9985086,0.00006502742,0.0004719556,0.0006425778,0.0002245521,0.00008726081],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002245384,0.0001169817,0.00007903142,0.00005613716,0.00008018126,0.00003512703,0.02294409,0.9722642,0.00005914562,0.0005210748,0.00002388779,0.003797675],"study_design_scores_gemma":[0.0004242592,0.0006161021,0.0001610464,0.0000339763,0.00007977247,0.000006196283,0.0216312,0.976219,0.0000369676,0.0001491333,0.0001876162,0.0004547232],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0443594,0.00001321826,0.9480271,0.00005269928,0.00154896,0.0008620223,9.048558e-7,0.0004765694,0.004659156],"genre_scores_gemma":[0.9534001,0.00002099939,0.01737851,0.0001031733,0.0001145677,0.000116901,0.002532697,0.0001024739,0.02623057],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9306486,"threshold_uncertainty_score":0.9998041,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03340026244345544,"score_gpt":0.3000187166623439,"score_spread":0.2666184542188884,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}