{"id":"W2156523042","doi":"10.1109/aamas.2004.238","title":"Run the GAMUT: A Comprehensive Approach to Evaluating Game-Theoretic Algorithms","year":2004,"lang":"en","type":"article","venue":"","topic":"Artificial Intelligence in Games","field":"Computer Science","cited_by":201,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Gamut; Computer science; Suite; Benchmarking; Generator (circuit theory); Test suite; Nash equilibrium; Theoretical computer science; Variation (astronomy); Architecture; Algorithm; Artificial intelligence; Machine learning; Mathematical optimization; Test case; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01258691,0.002999775,0.001773911,0.004129222,0.0007762316,0.003045338,0.0035366,0.001896465,0.003718252],"category_scores_gemma":[0.05833088,0.0008758031,0.001248212,0.00268312,0.001996451,0.003725859,0.003392496,0.003674167,0.0009239057],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001518047,"about_ca_system_score_gemma":0.001843558,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002710493,"about_ca_topic_score_gemma":0.004876463,"domain_scores_codex":[0.9908516,0.005439449,0.0004961301,0.0007641133,0.002081631,0.0003670963],"domain_scores_gemma":[0.9663256,0.02558013,0.001027604,0.004656256,0.001664807,0.0007455472],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008867511,0.001359514,0.01113212,0.001017062,0.0008020084,0.0002761289,0.0005487849,0.6378581,0.003903428,0.08540756,0.03288302,0.2239255],"study_design_scores_gemma":[0.0001338718,0.0003466276,0.000858437,0.00005993847,0.00004129797,0.00008919664,0.00007878155,0.9364335,0.003362005,0.05377599,0.004776058,0.00004426309],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07477766,0.0007617187,0.897903,0.0006883649,0.000209114,0.001065833,0.001943215,0.01266093,0.009990157],"genre_scores_gemma":[0.3227871,0.0004348533,0.6691953,0.0003814499,0.00006665926,0.001903538,0.002161428,0.001613313,0.001456337],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01258691,"threshold_uncertainty_score":0.06656671,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08309830794725798,"score_gpt":0.3423465536720401,"score_spread":0.2592482457247821,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}