{"id":"W4281383770","doi":"10.2200/s01170ed1v01y202202aim052","title":"Applying Reinforcement Learning on Real-World Data with Practical Examples in Python","year":2022,"lang":"en","type":"article","venue":"Synthesis lectures on artificial intelligence and machine learning","topic":"Robot Manipulation and Learning","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Ambrose University; Canadian Institute for Advanced Research","funders":"University of Manchester; University of Oxford","keywords":"Python (programming language); Computer science; Reinforcement learning; Real world data; Artificial intelligence; Data science; Machine learning; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001768228,0.0009635027,0.0006693607,0.0005264211,0.000493398,0.001509235,0.00253225,0.0008539509,0.0229624],"category_scores_gemma":[0.01097978,0.0007427145,0.0008827451,0.0006346107,0.001027266,0.001950529,0.002208903,0.002347774,0.004842911],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007280837,"about_ca_system_score_gemma":0.001495884,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004052728,"about_ca_topic_score_gemma":0.006011568,"domain_scores_codex":[0.9990739,0.0002475102,0.0001019226,0.0001846044,0.0002998429,0.00009218111],"domain_scores_gemma":[0.9965242,0.002175619,0.0001272992,0.0006847575,0.0003774046,0.0001107386],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0005991947,0.0004061519,0.004127033,0.001104163,0.0001757966,0.0006737768,0.0005577406,0.4629483,0.008310839,0.05258062,0.05367298,0.4148434],"study_design_scores_gemma":[0.00007277019,0.00002647864,0.0003302161,0.00004815085,0.00001154529,0.00008649416,0.00003416931,0.955878,0.005660418,0.02543725,0.01239884,0.00001577951],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01699739,0.0001031784,0.9065273,0.0004325446,0.0001354851,0.0001615998,0.001045821,0.06229629,0.01230047],"genre_scores_gemma":[0.2240426,0.0001608462,0.761003,0.0002540167,0.00003690516,0.0003780379,0.002009502,0.00555166,0.006563384],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.0229624,"threshold_uncertainty_score":0.07681686,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1123371378000468,"score_gpt":0.3250342094362956,"score_spread":0.2126970716362487,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}