{"id":"W2953351167","doi":"10.48550/arxiv.1704.02544","title":"A Linearly Relaxed Approximate Linear Program for Markov Decision Processes","year":2017,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Linear programming; Constraint (computer-aided design); Markov decision process; Mathematical optimization; Markov chain; Markov process; Computer science; Mathematics; Linear approximation; Applied mathematics; Nonlinear system; Statistics; Physics; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002047576,0.001355018,0.001400982,0.0005793828,0.0003635953,0.001783555,0.001439574,0.001496808,0.004581667],"category_scores_gemma":[0.009473356,0.0006440603,0.001034341,0.0009986616,0.001462634,0.001990258,0.00226776,0.003708659,0.0006688314],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001461582,"about_ca_system_score_gemma":0.001873793,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002737161,"about_ca_topic_score_gemma":0.002290647,"domain_scores_codex":[0.9975213,0.001139628,0.00008482856,0.0004928998,0.0005329656,0.0002283767],"domain_scores_gemma":[0.9955733,0.003421495,0.0003308957,0.0001987429,0.0003037996,0.0001717072],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001171806,0.00009934721,0.0003585743,0.000186101,0.00003882301,0.0001011003,0.0001060242,0.8717837,0.00128037,0.09571069,0.002501323,0.02771686],"study_design_scores_gemma":[0.000009626922,0.0000303624,0.00003413936,0.00001330251,0.0000047599,0.00001753251,0.000009876878,0.9742582,0.0002267805,0.02470525,0.0006849413,0.000005183044],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.009666622,0.0003769688,0.9847748,0.0004239686,0.00003223372,0.00008286419,0.000209268,0.0002067014,0.004226614],"genre_scores_gemma":[0.4679749,0.0009414851,0.5205617,0.0004781155,0.0001870031,0.0009232553,0.0009779789,0.0002496548,0.007705903],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004581667,"threshold_uncertainty_score":0.01532716,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08128687683702827,"score_gpt":0.2421828787003502,"score_spread":0.1608960018633219,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}