{"id":"W4405452958","doi":"10.1145/3705013","title":"A Comprehensive Model of Automated Evaluation of Difficulty in Platformer Games","year":2024,"lang":"en","type":"article","venue":"Games Research and Practice","topic":"Artificial Intelligence in Games","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université du Québec à Chicoutimi","funders":"","keywords":"Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003652965,0.00008742451,0.0001662295,0.0003867606,0.00004158735,0.0001337655,0.0003523771,0.00006869101,0.00001715203],"category_scores_gemma":[0.00290754,0.0000733062,0.00002925369,0.0009277998,0.000263675,0.001315985,0.0002429713,0.0003139685,0.00001955111],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006067572,"about_ca_system_score_gemma":0.000406037,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002764552,"about_ca_topic_score_gemma":0.00003528233,"domain_scores_codex":[0.9973273,0.000533212,0.0003457363,0.0003214737,0.001196987,0.0002753424],"domain_scores_gemma":[0.9945932,0.003648191,0.00007874928,0.0003106188,0.001304539,0.00006467916],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004651122,0.0006793342,0.0002375142,0.0006899044,0.000184925,0.00004727479,0.03562435,0.04749235,0.1232457,0.08938024,0.00462695,0.6973264],"study_design_scores_gemma":[0.00009044842,0.0001545006,0.000514,0.0001221041,0.000009220952,0.00001432821,0.001576656,0.97064,0.01229997,0.01340263,0.001107403,0.00006876289],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9555623,0.007235761,0.02561788,0.002656199,0.0001485675,0.0007205553,0.00001448202,0.0001597574,0.007884505],"genre_scores_gemma":[0.990289,0.0005773026,0.008958076,0.00002327574,0.00001349927,0.00002898823,0.000002004603,0.000006262393,0.0001015371],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9231476,"threshold_uncertainty_score":0.3480808,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3253714657586089,"score_gpt":0.4986562545373149,"score_spread":0.173284788778706,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}