{"id":"W162907590","doi":"10.82308/44454","title":"On characteristics of Markov decision processes and reinforcement learning in large domains","year":2005,"lang":"en","type":"article","venue":"eScholarship@McGill (McGill)","topic":"Neural Networks and Applications","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Markov decision process; Artificial intelligence; Machine learning; Task (project management); Instance-based learning; Process (computing); Temporal difference learning; Stability (learning theory); Markov process; Mathematics; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006155832,0.001218288,0.001704103,0.001344373,0.000858103,0.002872952,0.001617183,0.00216094,0.003749066],"category_scores_gemma":[0.04646743,0.0007489785,0.001568867,0.001576257,0.004442608,0.006317776,0.002322171,0.004825896,0.0004316743],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002890884,"about_ca_system_score_gemma":0.001978629,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003428409,"about_ca_topic_score_gemma":0.001827659,"domain_scores_codex":[0.9965175,0.001677175,0.0001973005,0.0005676433,0.0007896061,0.0002507417],"domain_scores_gemma":[0.9379128,0.05280216,0.003769185,0.002118004,0.002195525,0.001202232],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00007941167,0.00006858783,0.001706798,0.0001942063,0.0000615384,0.0001224323,0.0002602033,0.2903218,0.0008530641,0.6919293,0.000846195,0.01355647],"study_design_scores_gemma":[0.00002568728,0.00005888209,0.0004256583,0.00003820952,0.00001236868,0.00004949044,0.00002968524,0.6176061,0.0003242323,0.3803053,0.001105824,0.00001852459],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03600134,0.001605122,0.9486246,0.002227412,0.0001009442,0.0001260092,0.0001638349,0.0001846804,0.01096609],"genre_scores_gemma":[0.8096029,0.003171663,0.1792217,0.0005840547,0.0004019112,0.0008699842,0.0003272807,0.0001630835,0.005657525],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006155832,"threshold_uncertainty_score":0.03255552,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01057071576771858,"score_gpt":0.2355505879691684,"score_spread":0.2249798722014499,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}