{"id":"W4288723601","doi":"10.1007/s10664-022-10158-x","title":"GBGallery : A benchmark and framework for game testing","year":2022,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Benchmark (surveying); Software engineering; Video game development; Test strategy; Software development; Game testing; Software; Database; Game Developer; Game design; Artificial intelligence; Programming language; Game design document","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01039787,0.002597812,0.001387326,0.006183542,0.0009811064,0.004377573,0.007758392,0.002997756,0.009800072],"category_scores_gemma":[0.07107691,0.001290909,0.00153271,0.003518854,0.001906577,0.006404637,0.005151144,0.004304515,0.004293337],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001829896,"about_ca_system_score_gemma":0.003557915,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007388865,"about_ca_topic_score_gemma":0.007911489,"domain_scores_codex":[0.9882515,0.005074975,0.001466504,0.0008703926,0.00346864,0.0008679527],"domain_scores_gemma":[0.9619957,0.02177425,0.001891775,0.008415544,0.004518898,0.001403833],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001240223,0.00140613,0.01501842,0.001818587,0.0002801038,0.0005642026,0.0006934455,0.1101887,0.007305918,0.2029137,0.1797464,0.4788242],"study_design_scores_gemma":[0.000447403,0.0004896988,0.003439829,0.0006947197,0.0001008122,0.000743031,0.0002260854,0.6935531,0.01491586,0.2119015,0.07330607,0.000181909],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01316862,0.0006244063,0.8387747,0.0009731802,0.0002669347,0.0005381795,0.003905101,0.1283823,0.01336664],"genre_scores_gemma":[0.1539991,0.0005101744,0.8125084,0.0006216019,0.0001142628,0.001164738,0.00998501,0.01703544,0.004061202],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01039787,"threshold_uncertainty_score":0.05498981,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04408833245119895,"score_gpt":0.284560538152745,"score_spread":0.240472205701546,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}