{"id":"W4403564633","doi":"10.48550/arxiv.2410.09368","title":"Towards a Domain-Specific Modelling Environment for Reinforcement Learning","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Advanced Software Engineering Methodologies","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Reinforcement; Domain (mathematical analysis); Computer science; Cognitive science; Psychology; Artificial intelligence; Social psychology; Mathematics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0005605062,0.0004031177,0.0003806936,0.0002737535,0.0001544021,0.000134042,0.001365737,0.000257946,0.00000947522],"category_scores_gemma":[0.00001928499,0.0004670733,0.0003019825,0.0002417708,0.00007050503,0.0001757338,0.003378673,0.0008961561,0.00006345201],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005911329,"about_ca_system_score_gemma":0.00009520246,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001026062,"about_ca_topic_score_gemma":3.095126e-7,"domain_scores_codex":[0.9977393,0.00009579563,0.0002530754,0.001307802,0.0001264106,0.0004776619],"domain_scores_gemma":[0.9983387,0.0003197983,0.0001586466,0.001008195,0.0000516044,0.0001230466],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001226409,0.000008291915,0.000001817385,0.0001138102,0.0000525038,0.00007079411,0.0003168634,0.729842,0.00002012669,0.267895,0.00003721477,0.001629247],"study_design_scores_gemma":[0.0001737187,0.00005597844,0.000001910231,0.00009066852,0.00002350482,0.000002212491,0.00005160022,0.6451588,0.0001281171,0.326482,0.02744516,0.0003862961],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.003707249,0.0005669459,0.9929148,0.00007210831,0.0009766554,0.0004947745,0.000004827069,0.0009074845,0.0003551431],"genre_scores_gemma":[0.2693044,0.000920716,0.7279975,0.00001652304,0.0001077507,0.00001113875,0.00001369158,0.0000437049,0.001584532],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.2655971,"threshold_uncertainty_score":0.9997781,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1393643901273245,"score_gpt":0.2123780578668136,"score_spread":0.07301366773948906,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}