{"id":"W2604173138","doi":"10.1609/aaai.v31i1.11056","title":"Hindsight Optimization for Hybrid State and Action MDPs","year":2017,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Bayesian Modeling and Causal Inference","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"National Science Foundation","keywords":"Computer science; Mathematical optimization; Markov decision process; Upper and lower bounds; Piecewise linear function; Formalism (music); Action selection; Scaling; Markov process; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003080574,0.001676451,0.001378481,0.0007730906,0.0004979038,0.001378649,0.001441241,0.001312527,0.005442026],"category_scores_gemma":[0.008410472,0.0009417955,0.001117171,0.0006469036,0.001478397,0.001704158,0.002027628,0.002603088,0.0005456863],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001700611,"about_ca_system_score_gemma":0.002112024,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004987238,"about_ca_topic_score_gemma":0.005670914,"domain_scores_codex":[0.9984702,0.0007100258,0.00007392354,0.0003586946,0.0002396613,0.000147351],"domain_scores_gemma":[0.9937413,0.005210035,0.0003967817,0.000200814,0.0002371573,0.0002138828],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00005069778,0.00002752792,0.0002570892,0.00004677992,0.00001916102,0.00002908362,0.00002240735,0.9829486,0.0001975364,0.008565721,0.0003986619,0.007436785],"study_design_scores_gemma":[0.000008698145,0.000015539,0.00002528584,0.000005245511,0.000003015871,0.00000469626,0.000005007039,0.9923394,0.000123371,0.007279778,0.0001874539,0.000002447477],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01907866,0.0003192436,0.9762834,0.0002833895,0.00003587725,0.00008717028,0.0001754555,0.0005699839,0.00316675],"genre_scores_gemma":[0.6383772,0.0002793147,0.3556122,0.0003644503,0.00004943536,0.0003678104,0.0005863046,0.0002687209,0.004094592],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005442026,"threshold_uncertainty_score":0.0182054,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1316459554872544,"score_gpt":0.330156495202359,"score_spread":0.1985105397151047,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}