{"id":"W2403345031","doi":"10.1609/aiide.v7i1.12440","title":"Behavior Learning-Based Testing of Starcraft Competition Entries","year":2011,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence and Interactive Digital Entertainment","topic":"Artificial Intelligence in Games","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Macro; Competition (biology); Computer science; Margin (machine learning); Action (physics); Strengths and weaknesses; Artificial intelligence; Test (biology); Machine learning; Psychology; Social psychology; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002310047,0.0002683027,0.0003076912,0.0001791698,0.0001437037,0.0002523863,0.0009950645,0.00006512602,0.00007684211],"category_scores_gemma":[0.0006005889,0.0002081454,0.0001357448,0.0003533328,0.000542701,0.00108255,0.0004066273,0.0002979649,0.00002399414],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006471384,"about_ca_system_score_gemma":0.00006582392,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00007349192,"about_ca_topic_score_gemma":0.000005195037,"domain_scores_codex":[0.9980153,0.00002358447,0.0007244585,0.0004648749,0.0004626133,0.0003091532],"domain_scores_gemma":[0.99801,0.0002255533,0.0006317536,0.0002107175,0.0008290366,0.0000928907],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0006116813,0.002326287,0.04245438,0.0001401056,0.00009495186,0.000004892909,0.01478774,0.0001558253,0.04431774,0.5942903,0.00002121751,0.3007949],"study_design_scores_gemma":[0.00003557139,0.001406987,0.001847114,0.000511873,0.00002136426,0.000005336119,0.00431334,0.028682,0.9359871,0.02686948,0.00006463768,0.0002551662],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9363343,0.00001482109,0.04948986,0.0003825852,0.0003017432,0.0006220178,0.00001619413,0.00008309317,0.01275545],"genre_scores_gemma":[0.9984742,0.000009275962,0.001271027,0.00007683346,0.00001847751,0.00004693388,0.000001140943,0.00001272348,0.00008937202],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8916694,"threshold_uncertainty_score":0.8487923,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08308497060880611,"score_gpt":0.2814262492827505,"score_spread":0.1983412786739444,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}