{"id":"W4386875701","doi":"10.48550/arxiv.2309.09476","title":"Mechanic Maker 2.0: Reinforcement Learning for Evaluating Generated Rules","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Artificial Intelligence in Games","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Alberta Machine Intelligence Institute; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"USable; Reinforcement learning; Computer science; Artificial intelligence; Meaning (existential); Machine learning; Reinforcement; Baseline (sea); Psychology; World Wide Web; Social psychology","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004055368,0.001040998,0.0008241333,0.0009286299,0.0003162219,0.001409948,0.002809752,0.001648142,0.009339888],"category_scores_gemma":[0.01489948,0.0006957766,0.0005206413,0.0003901696,0.001063708,0.001523422,0.001721726,0.002014043,0.001941008],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008610812,"about_ca_system_score_gemma":0.001483489,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001874615,"about_ca_topic_score_gemma":0.003131433,"domain_scores_codex":[0.9984832,0.0006191512,0.00009890117,0.00030402,0.0003920229,0.0001025909],"domain_scores_gemma":[0.9940794,0.004503491,0.0003253485,0.0005696968,0.0003610785,0.0001611317],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001083451,0.0007382262,0.006246232,0.0009899666,0.0001952552,0.0002955112,0.0003771501,0.5533003,0.01218374,0.05426043,0.01751126,0.3528185],"study_design_scores_gemma":[0.00006522766,0.00006331784,0.0001795028,0.00001758075,0.00000967642,0.0000340841,0.000007557192,0.9862404,0.00439421,0.00695644,0.002021729,0.00001028621],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02861152,0.0002761181,0.9259416,0.0002284297,0.0001126193,0.0003618004,0.0004420953,0.0373644,0.00666143],"genre_scores_gemma":[0.3391101,0.0001419368,0.6539168,0.0002475746,0.00003448285,0.0005336056,0.0005619468,0.00203567,0.003417804],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.009339888,"threshold_uncertainty_score":0.03124505,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.265515225508008,"score_gpt":0.2751630985045367,"score_spread":0.009647872996528695,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}