{"id":"W4402706509","doi":"10.48550/arxiv.2408.16262","title":"On Convergence of Average-Reward Q-Learning in Weakly Communicating Markov Decision Processes","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Bayesian Modeling and Causal Inference","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; University of Alberta; DeepMind","keywords":"Markov decision process; Convergence (economics); Markov chain; Computer science; Artificial intelligence; Markov process; Mathematics; Psychology; Machine learning; Statistics; Economics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0005485739,0.0002917791,0.0003860477,0.0004094836,0.00009816066,0.0001104675,0.002327038,0.0002595642,0.00001931055],"category_scores_gemma":[0.0002747664,0.0003233708,0.0001215145,0.00110535,0.00009813542,0.0001772071,0.003547505,0.001495763,0.00007734178],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001402008,"about_ca_system_score_gemma":0.0004789791,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002257275,"about_ca_topic_score_gemma":0.00007307986,"domain_scores_codex":[0.9980191,0.0002128551,0.0003583538,0.000955127,0.0001561657,0.0002983696],"domain_scores_gemma":[0.9977456,0.0006309263,0.0002763117,0.001021133,0.0002256936,0.0001002889],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00006487255,0.0001410709,0.002068299,0.0007010151,0.00004266596,0.00020716,0.001012583,0.8836993,0.00009666623,0.1032606,0.00007365497,0.008632102],"study_design_scores_gemma":[0.0001545317,0.00008359061,0.0002723509,0.002022367,0.00001610576,0.000002360741,0.00006143867,0.8406163,0.0002514921,0.1561726,0.00003069408,0.0003161831],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4653144,0.0002593573,0.5310401,0.00007397697,0.0002782159,0.0001426933,0.000003477749,0.0001877524,0.002700034],"genre_scores_gemma":[0.9940463,0.0006462641,0.00482213,0.00004485793,0.00001374108,0.000001330833,0.000004160787,0.00001778097,0.0004034698],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.5287318,"threshold_uncertainty_score":0.9999219,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06133400688008574,"score_gpt":0.2172719103401338,"score_spread":0.1559379034600481,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}