{"id":"W66652580","doi":"10.5555/2484920.2485079","title":"Baseline: practical control variates for agent evaluation in zero-sum domains","year":2013,"lang":"en","type":"article","venue":"","topic":"Artificial Intelligence in Games","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Variance reduction; Control variates; Baseline (sea); Variance (accounting); Computer science; Reduction (mathematics); Estimator; Monte Carlo method; Domain (mathematical analysis); Overhead (engineering); Algorithm; Mathematical optimization; Artificial intelligence; Statistics; Mathematics; Hybrid Monte Carlo","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00919495,0.001125466,0.001201868,0.001159677,0.0007002542,0.002185406,0.002191812,0.001412835,0.006069597],"category_scores_gemma":[0.03258758,0.0006839034,0.0007494268,0.0006522487,0.001321204,0.002860631,0.002716708,0.002502541,0.001167581],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001357084,"about_ca_system_score_gemma":0.001955654,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002350091,"about_ca_topic_score_gemma":0.002845109,"domain_scores_codex":[0.9948621,0.002537579,0.0002841388,0.0005622396,0.001484344,0.0002694833],"domain_scores_gemma":[0.987422,0.007907179,0.0006432628,0.001816267,0.00188106,0.0003302543],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009952296,0.0005609661,0.003316212,0.0003644506,0.0001638785,0.0001432904,0.0002591017,0.4072205,0.01030518,0.09700462,0.00778242,0.4718841],"study_design_scores_gemma":[0.00006633853,0.0001371668,0.000289166,0.00002957537,0.0000121608,0.000038181,0.00002908651,0.9693972,0.004732427,0.02331181,0.001937689,0.00001921946],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.005424351,0.0001044031,0.991568,0.00007081051,0.00003295483,0.00009474057,0.00004556279,0.001300742,0.001358467],"genre_scores_gemma":[0.2311048,0.00009746277,0.7654234,0.0001698292,0.00004592984,0.0004170153,0.0003038312,0.0005454361,0.001892379],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.00919495,"threshold_uncertainty_score":0.04862815,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0631569794152325,"score_gpt":0.359566842125435,"score_spread":0.2964098627102025,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}