{"id":"W2403345031","doi":"10.1609/aiide.v7i1.12440","title":"Behavior Learning-Based Testing of Starcraft Competition Entries","year":2011,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence and Interactive Digital Entertainment","topic":"Artificial Intelligence in Games","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Macro; Competition (biology); Computer science; Margin (machine learning); Action (physics); Strengths and weaknesses; Artificial intelligence; Test (biology); Machine learning; Psychology; Social psychology; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002127095,0.001259627,0.0005515889,0.0008843591,0.0003000082,0.0007793479,0.002065873,0.0007875815,0.001792165],"category_scores_gemma":[0.01919705,0.0003671833,0.000309236,0.0002775796,0.0006955681,0.001450557,0.0008109435,0.001080983,0.0007905738],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006891257,"about_ca_system_score_gemma":0.0006782066,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003281598,"about_ca_topic_score_gemma":0.004293735,"domain_scores_codex":[0.9968513,0.0009851548,0.0002750573,0.0008319147,0.0008646824,0.0001919228],"domain_scores_gemma":[0.986262,0.007475772,0.001799396,0.001301658,0.002258339,0.0009028271],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003731939,0.00386391,0.205954,0.000878602,0.0004670762,0.001050118,0.002473848,0.2040694,0.1919179,0.002940571,0.004328972,0.3783236],"study_design_scores_gemma":[0.0001238256,0.002670608,0.05119513,0.00005570049,0.00006122323,0.0004641515,0.0004260125,0.847611,0.09275561,0.001790596,0.002715299,0.0001308059],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8516074,0.00009820699,0.1373854,0.0001247466,0.00006617363,0.0005412265,0.0005617976,0.005453588,0.004161401],"genre_scores_gemma":[0.9635885,0.00002423522,0.03392036,0.00008701128,0.000006435084,0.0001560628,0.000554857,0.0001754889,0.001487128],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.003281598,"threshold_uncertainty_score":0.0112493,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08308497060880611,"score_gpt":0.2814262492827505,"score_spread":0.1983412786739444,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}