{"id":"W4391021135","doi":"10.1109/cdc49753.2023.10383590","title":"Weighted-Norm Bounds on Model Approximation in MDPs with Unbounded Per-Step Cost","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"National Science Foundation","keywords":"Bounding overwatch; Norm (philosophy); Function (biology); Computer science; Algorithm; Discrete mathematics; Combinatorics; Mathematics; Artificial intelligence; Philosophy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003300397,0.0001613126,0.000145449,0.0003130395,0.0001270168,0.0003165308,0.0006034238,0.00006849067,0.00001481836],"category_scores_gemma":[0.00002027298,0.0001280938,0.00002858862,0.0009775144,0.00004101569,0.0004297538,0.000163518,0.0002089187,0.0004445098],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001567417,"about_ca_system_score_gemma":0.0001114827,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00004367932,"about_ca_topic_score_gemma":0.00006432424,"domain_scores_codex":[0.9985265,0.00003505194,0.0002392579,0.0003668383,0.0004575978,0.0003748191],"domain_scores_gemma":[0.9991453,0.00008111032,0.00007854022,0.0005638045,0.00006182838,0.00006945916],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000008803408,0.00001974909,0.0004347178,0.00001144855,0.000005350366,0.000005624247,0.0004234351,0.8351998,0.00002819356,0.1608245,0.0004350541,0.002603365],"study_design_scores_gemma":[0.0005023037,0.0001679343,0.0006111042,0.00003040746,0.000002192239,0.0000029765,0.00004974235,0.9959995,0.0001798713,0.001708773,0.0005581803,0.0001869945],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.015148,0.000001498998,0.9428928,0.00064616,0.0001013502,0.0003273184,3.461691e-7,0.0004963437,0.04038613],"genre_scores_gemma":[0.8284571,0.00001205845,0.1581455,0.0005557929,0.00002941493,0.0000710364,0.00002889293,0.00002631104,0.01267384],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8133091,"threshold_uncertainty_score":0.5713423,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02607248693463956,"score_gpt":0.2568377678037991,"score_spread":0.2307652808691595,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}