{"id":"W2161500856","doi":"10.1109/ichis.2005.75","title":"Monte Carlo off-policy reinforcement learning: a rough set approach","year":2005,"lang":"en","type":"article","venue":"","topic":"Rough Sets and Fuzzy Logic","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Winnipeg; University of Manitoba","funders":"Natural Sciences and Engineering Research Council of Canada; Manitoba Hydro","keywords":"Reinforcement learning; Monte Carlo method; Computer science; Context (archaeology); Swarm behaviour; Artificial intelligence; Mathematical optimization; Machine learning; Mathematics; Statistics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00467322,0.0008911658,0.002360677,0.001210735,0.0006736478,0.00190664,0.002440222,0.001432727,0.00139558],"category_scores_gemma":[0.01126979,0.0006597737,0.001574595,0.0008724748,0.002444427,0.001782644,0.001228691,0.00201071,0.000185189],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002392894,"about_ca_system_score_gemma":0.001664322,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005759097,"about_ca_topic_score_gemma":0.003866615,"domain_scores_codex":[0.9968613,0.001754981,0.0001606821,0.000313765,0.0007498044,0.0001593642],"domain_scores_gemma":[0.9935063,0.004753194,0.000522044,0.0004501176,0.0005705843,0.0001976313],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002889556,0.00002647303,0.0003295362,0.00005095082,0.00006403073,0.00004410327,0.0000592522,0.946983,0.0003133894,0.03889312,0.0002172628,0.01298984],"study_design_scores_gemma":[0.00001028199,0.00002641758,0.00006603567,0.00001138689,0.00001069743,0.00001175454,0.000008187117,0.976025,0.0001906591,0.02315638,0.0004721864,0.00001106853],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.003869089,0.0001968601,0.994724,0.0001713681,0.00002993015,0.00004002771,0.0000136112,0.00005570882,0.0008993367],"genre_scores_gemma":[0.5707373,0.0006883269,0.425559,0.0002034736,0.0001386795,0.0004358206,0.00009149582,0.00006909043,0.00207682],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005759097,"threshold_uncertainty_score":0.02471459,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02617794070649943,"score_gpt":0.2556230226707021,"score_spread":0.2294450819642027,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}