{"id":"W3184031856","doi":"10.22215/etd/2021-14448","title":"Several Reinforcement Learning Methods in Mean-Field Games with Binary Action Spaces","year":2021,"lang":"en","type":"dissertation","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University","funders":"","keywords":"Reinforcement learning; Action (physics); Convergence (economics); Computer science; Binary number; Task (project management); Field (mathematics); Artificial intelligence; Population; Binary classification; Machine learning; Space (punctuation); Mathematical optimization; Mathematics; Engineering; Support vector machine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0006174364,0.0004080014,0.0004584088,0.0004686766,0.0001499838,0.0004918473,0.0006564724,0.0003224594,0.0003106052],"category_scores_gemma":[0.0001434362,0.0003588552,0.0001072449,0.000671319,0.00001439187,0.0007771869,0.0001604549,0.00111115,0.00003054893],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001881287,"about_ca_system_score_gemma":0.0003239306,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003608915,"about_ca_topic_score_gemma":0.0003751661,"domain_scores_codex":[0.9973592,0.0003012905,0.0005179646,0.0006771082,0.0006861981,0.0004582439],"domain_scores_gemma":[0.9983836,0.0003232471,0.0004422068,0.000581339,0.0001830785,0.00008653504],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00005466135,0.00001655553,0.0003198291,0.0001623968,0.00006978001,0.00004674833,0.003485954,0.9771668,0.0006041426,0.002438438,0.000264082,0.01537062],"study_design_scores_gemma":[0.0006291447,0.0009307212,0.001342586,0.0006396847,0.00004395629,0.0000185974,0.007134029,0.9658023,0.01611287,0.00004214637,0.006424781,0.0008791706],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.003991433,0.0001088881,0.9421174,0.0001809672,0.0008599422,0.0002834029,3.582903e-8,0.0002328754,0.05222509],"genre_scores_gemma":[0.0988836,0.0004887229,0.5817887,0.0003762347,0.0001333745,0.00009577617,0.0003661623,0.00007695817,0.3177904],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.3603286,"threshold_uncertainty_score":0.9998863,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02751152124695856,"score_gpt":0.3446213576646282,"score_spread":0.3171098364176697,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}