{"id":"W2962951833","doi":"","title":"Cumulative Prospect Theory Meets Reinforcement Learning: Prediction and Control","year":2015,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Traffic control and management","field":"Engineering","cited_by":48,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Simultaneous perturbation stochastic approximation; Cumulative prospect theory; Computer science; Mathematical optimization; Bellman equation; Empirical distribution function; Convergence (economics); Random variable; Cumulative distribution function; Stochastic approximation; Stochastic process; Artificial intelligence; Mathematics; Probability density function; Expected utility hypothesis; Statistics; Key (lock)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000184366,0.000108258,0.0001105973,0.00006991671,0.00005134904,0.00001648704,0.00006691194,0.00003901509,0.00001943568],"category_scores_gemma":[0.00001848904,0.0001133718,0.00003008809,0.0001120231,0.00003557995,0.0001682168,0.00003082147,0.00009941425,0.0000235944],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00009765857,"about_ca_system_score_gemma":0.000008763728,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000009159713,"about_ca_topic_score_gemma":0.000007737977,"domain_scores_codex":[0.9995052,0.00003551698,0.00008059917,0.0001750865,0.00004622942,0.0001573818],"domain_scores_gemma":[0.999685,0.0000320312,0.00002451158,0.000112678,0.00003541175,0.0001103082],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00007790256,0.000009037152,0.001160453,0.00001117548,0.00009366582,0.00001995126,0.0002618683,0.9399723,0.00004144252,0.05745964,0.0002015909,0.0006909039],"study_design_scores_gemma":[0.002717033,0.00014823,0.004368045,0.00001195916,0.00009868618,0.000001427801,0.0004190188,0.9796891,0.00002403589,0.002505726,0.009861898,0.0001548188],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6131125,0.0002590678,0.336199,0.00006834268,0.000384522,0.0006994729,0.00000484934,0.001094691,0.0481775],"genre_scores_gemma":[0.997099,0.00004914109,0.00001229005,0.0000171783,0.00003537055,0.000001259343,0.000003604056,0.00001094692,0.002771176],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3839865,"threshold_uncertainty_score":0.4623169,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02721012820076477,"score_gpt":0.1492170774226893,"score_spread":0.1220069492219245,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}