{"id":"W2946045694","doi":"10.24963/ijcai.2019/85","title":"A Regularized Opponent Model with Maximum Entropy Objective","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Principle of maximum entropy; Computer science; Inference; Binary number; Mathematical optimization; Probabilistic logic; Iterated function; Random variable; Artificial intelligence; Algorithm; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001641792,0.000696377,0.001183803,0.0004895339,0.0003413447,0.001072655,0.00186736,0.001477426,0.00380077],"category_scores_gemma":[0.004295556,0.0004689629,0.0005534468,0.0003526964,0.001509787,0.001612685,0.00144647,0.001596826,0.0005156356],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001144065,"about_ca_system_score_gemma":0.001283353,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002751174,"about_ca_topic_score_gemma":0.002575897,"domain_scores_codex":[0.9992328,0.0003374215,0.00002035054,0.0001632881,0.0001528768,0.00009335909],"domain_scores_gemma":[0.9984161,0.001048886,0.0001746652,0.00009872056,0.0001492549,0.000112316],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00008145504,0.0000460375,0.0004966617,0.00006240802,0.00003376808,0.0000818738,0.0000704319,0.8859115,0.001371667,0.09436473,0.001251274,0.01622813],"study_design_scores_gemma":[0.000009354128,0.00001128871,0.00003214625,0.000003818833,0.000002273879,0.000008996352,0.00000264973,0.987371,0.0001164231,0.01224445,0.0001943367,0.000003253472],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02273385,0.0001532225,0.9712905,0.0004897422,0.00002930363,0.00004697229,0.00007054081,0.0001915639,0.004994284],"genre_scores_gemma":[0.7875684,0.0001351993,0.2023089,0.0003310994,0.00004976461,0.0001882494,0.0001377575,0.0001157223,0.009164941],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00380077,"threshold_uncertainty_score":0.01271486,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01899547368216535,"score_gpt":0.2364777973377845,"score_spread":0.2174823236556192,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}