{"id":"W4379878604","doi":"10.1016/j.eswa.2023.120702","title":"<mml:math xmlns:mml=\"http://www.w3.org/1998/Math/MathML\" altimg=\"si144.svg\" display=\"inline\" id=\"d1e549\"><mml:mi>Δ</mml:mi></mml:math>V-learning: An adaptive reinforcement learning algorithm for the optimal stopping problem","year":2023,"lang":"en","type":"article","venue":"Expert Systems with Applications","topic":"Auction Theory and Applications","field":"Decision Sciences","cited_by":5,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto; University of New Brunswick","funders":"","keywords":"Reinforcement learning; Optimal stopping; Markov decision process; Computer science; Algorithm; Q-learning; Benchmark (surveying); Bellman equation; Artificial intelligence; Machine learning; Mathematical optimization; Markov process; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.001167407,0.001758389,0.001135561,0.001525205,0.0006452648,0.004859324,0.003896423,0.002395264,0.656679],"category_scores_gemma":[0.008046649,0.001120226,0.0009200076,0.00280277,0.0005613583,0.00379321,0.001933294,0.002339617,0.530808],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001816448,"about_ca_system_score_gemma":0.001349758,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008022635,"about_ca_topic_score_gemma":0.008665277,"domain_scores_codex":[0.999271,0.0001221003,0.00007824643,0.0001381751,0.0003218431,0.00006864862],"domain_scores_gemma":[0.9973426,0.0009922053,0.0001579625,0.0005160056,0.0008439232,0.0001471997],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00008082335,0.0000448609,0.0001305563,0.0002729985,0.00001079723,0.00004498952,0.00005683065,0.00125258,0.0009091821,0.02200382,0.9030938,0.07209878],"study_design_scores_gemma":[0.0001019764,0.00002139312,0.0004154278,0.00009923352,0.000007480163,0.000102658,0.00003171612,0.01347467,0.003804966,0.02915768,0.9527401,0.0000426884],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.000985121,0.0003563876,0.396372,0.003758064,0.0009881827,0.0005353941,0.1118338,0.1798722,0.3052988],"genre_scores_gemma":[0.02620269,0.001200612,0.2518151,0.002427649,0.0006137298,0.001575168,0.1459178,0.1458802,0.4243671],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.656679,"threshold_uncertainty_score":0.4897057,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04489854708964441,"score_gpt":0.3107925151993854,"score_spread":0.265893968109741,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}