{"id":"W3211741236","doi":"","title":"Learning in two-player zero-sum partially observable Markov games with perfect recall","year":2021,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Observable; Zero (linguistics); Computer science; Markov chain; Recall; Zero-sum game; Markov process; Mathematics; Mathematical economics; Game theory; Statistics; Machine learning; Cognitive psychology; Psychology; Physics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003463481,0.001513479,0.003779226,0.0007665221,0.0008630526,0.00294524,0.003184033,0.002926146,0.00517834],"category_scores_gemma":[0.01553964,0.001219744,0.0009729419,0.000689575,0.003628721,0.004403847,0.003203436,0.002790432,0.0004247927],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002296489,"about_ca_system_score_gemma":0.002432427,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008502237,"about_ca_topic_score_gemma":0.006892487,"domain_scores_codex":[0.9977403,0.0008736433,0.0001190616,0.0004366094,0.0002859054,0.0005444482],"domain_scores_gemma":[0.9822226,0.01482482,0.001145728,0.0004653656,0.0005721671,0.0007693397],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007590022,0.0001956345,0.001134724,0.0002284538,0.0001320932,0.0002879717,0.0002303545,0.8212072,0.000833207,0.1622743,0.0012394,0.01147782],"study_design_scores_gemma":[0.0001132173,0.00007414071,0.0001387239,0.00001575821,0.00002004532,0.00002269343,0.00002497315,0.9271914,0.0002253703,0.07200921,0.0001429119,0.00002150367],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.334663,0.000722973,0.6459879,0.00270108,0.0001529605,0.0002070785,0.0004901668,0.0004451385,0.01462978],"genre_scores_gemma":[0.9809278,0.0001916154,0.01102856,0.0001777033,0.00004558053,0.0001264196,0.0001154647,0.00002821738,0.007358606],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008502237,"threshold_uncertainty_score":0.01831686,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0187638513781867,"score_gpt":0.2540259796019725,"score_spread":0.2352621282237858,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}