{"id":"W2562174057","doi":"10.1609/aaai.v32i1.11481","title":"AIVAT: A New Variance Reduction Technique for Agent Evaluation in Imperfect Information Games","year":2018,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Artificial Intelligence in Games","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Variance reduction; Variance (accounting); Perfect information; Computer science; Estimator; Imperfect; Heuristic; Limit (mathematics); Value (mathematics); Victory; Mathematical optimization; Artificial intelligence; Statistics; Machine learning; Mathematics; Monte Carlo method; Mathematical economics; Economics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01342149,0.001805976,0.002173756,0.00246843,0.0009010574,0.002215981,0.002811089,0.001680414,0.002669489],"category_scores_gemma":[0.05475974,0.0009341821,0.001693435,0.0009835571,0.001949435,0.002729362,0.003412825,0.003636068,0.0006158749],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001625132,"about_ca_system_score_gemma":0.0022884,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002540819,"about_ca_topic_score_gemma":0.002648502,"domain_scores_codex":[0.9858478,0.008332886,0.0006472133,0.001028124,0.003549549,0.0005944668],"domain_scores_gemma":[0.9649253,0.02585939,0.002812914,0.002423545,0.003301822,0.0006771231],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005600177,0.0002656616,0.006538784,0.0003358186,0.0004378306,0.0003205639,0.0005800031,0.6203521,0.005629899,0.1027511,0.004773449,0.2574547],"study_design_scores_gemma":[0.00003003853,0.00009230574,0.00028477,0.00002441347,0.00001872094,0.00004170441,0.00002206276,0.9742405,0.001195628,0.02336684,0.0006639767,0.0000190476],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.004513151,0.00009909483,0.9933206,0.0000847641,0.00003298916,0.0001090885,0.00004519002,0.0007273933,0.00106765],"genre_scores_gemma":[0.374546,0.000141548,0.6220067,0.0002008838,0.0001248823,0.0007407735,0.0002578302,0.0003948518,0.001586661],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01342149,"threshold_uncertainty_score":0.07098049,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09062847487418954,"score_gpt":0.3394626692515252,"score_spread":0.2488341943773356,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}