{"id":"W4379878604","doi":"10.1016/j.eswa.2023.120702","title":"<mml:math xmlns:mml=\"http://www.w3.org/1998/Math/MathML\" altimg=\"si144.svg\" display=\"inline\" id=\"d1e549\"><mml:mi>Δ</mml:mi></mml:math>V-learning: An adaptive reinforcement learning algorithm for the optimal stopping problem","year":2023,"lang":"en","type":"article","venue":"Expert Systems with Applications","topic":"Auction Theory and Applications","field":"Decision Sciences","cited_by":5,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto; University of New Brunswick","funders":"","keywords":"Reinforcement learning; Optimal stopping; Markov decision process; Computer science; Algorithm; Q-learning; Benchmark (surveying); Bellman equation; Artificial intelligence; Machine learning; Mathematical optimization; Markov process; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["sts","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00208561,0.0002843624,0.0002286167,0.0001653516,0.002545732,0.0005408239,0.0009360235,0.0002191441,0.00003807262],"category_scores_gemma":[0.0001998062,0.0002381555,0.0002073802,0.0009656832,0.0003046725,0.0004980331,0.0002636612,0.000431056,0.001355061],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003070772,"about_ca_system_score_gemma":0.0002227286,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001647106,"about_ca_topic_score_gemma":0.00002067773,"domain_scores_codex":[0.9966437,0.0001627542,0.0008443689,0.000818187,0.0009973369,0.0005336631],"domain_scores_gemma":[0.996236,0.001372612,0.0008448596,0.001056364,0.0002535613,0.000236633],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0000535851,0.00005089249,0.000002827782,0.00001917857,0.0001069555,0.000002975497,0.002379775,0.1071229,0.0002545764,0.8817067,0.000965569,0.007334059],"study_design_scores_gemma":[0.0002798756,0.0002852841,0.000009833678,0.00007113654,0.00005321157,0.00006427392,0.01880515,0.8419696,0.0003259152,0.0004287021,0.1374719,0.0002350757],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07930385,0.0002655126,0.9158983,0.0007821508,0.0002617156,0.0008437773,0.00006723349,0.0004771971,0.002100281],"genre_scores_gemma":[0.9670792,0.00008801142,0.01184543,0.0001878802,0.0008893759,0.01661959,0.000327484,0.0001106376,0.00285238],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9040529,"threshold_uncertainty_score":0.9994225,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04489854708964441,"score_gpt":0.3107925151993854,"score_spread":0.265893968109741,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}