{"id":"W4311681064","doi":"10.22215/etd/2022-15178","title":"Learning Transition Dynamics via Rewarded Exploration: A Study using Unity's MLAgents","year":2022,"lang":"en","type":"dissertation","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Popularity; Artificial intelligence; Hyperparameter; Transition (genetics); Action (physics); Dynamics (music); Variable (mathematics); Focus (optics); Artificial neural network; State (computer science); Game engine; Machine learning; Work (physics); Human–computer interaction; Engineering; Psychology; Social psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002704496,0.0004670495,0.0005890937,0.0004538495,0.0004587334,0.001448426,0.001532637,0.001112153,0.00241124],"category_scores_gemma":[0.02704983,0.0003759092,0.0005808273,0.0005289012,0.001179873,0.003079236,0.001463542,0.001854247,0.0004081643],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008880605,"about_ca_system_score_gemma":0.0005699253,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007968894,"about_ca_topic_score_gemma":0.004380895,"domain_scores_codex":[0.9986501,0.0007745144,0.0000736673,0.0002401763,0.0001794218,0.00008211959],"domain_scores_gemma":[0.9833824,0.01412003,0.0005202895,0.001137857,0.0005416268,0.0002978488],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002168506,0.005181907,0.04080123,0.001534238,0.0004933859,0.000631091,0.007006842,0.5596346,0.01031298,0.05510611,0.005588894,0.3115402],"study_design_scores_gemma":[0.0001782017,0.001331849,0.005407935,0.00006366186,0.00006107781,0.000125469,0.0006579048,0.972047,0.004393273,0.01157799,0.004113772,0.00004183162],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9251538,0.00116909,0.06476918,0.0005331555,0.00004295181,0.0001804504,0.0001369272,0.0003410073,0.007673535],"genre_scores_gemma":[0.9720246,0.0002567646,0.02514656,0.00009965834,0.00001417895,0.00007560149,0.0001348495,0.00005112165,0.002196592],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.007968894,"threshold_uncertainty_score":0.01584506,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03340026244345544,"score_gpt":0.3000187166623439,"score_spread":0.2666184542188884,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}