{"id":"W4372259970","doi":"10.1109/icassp49357.2023.10095236","title":"MEET: A Monte Carlo Exploration-Exploitation Trade-Off for Buffer Sampling","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Infineon Technologies (Canada)","funders":"","keywords":"Reinforcement learning; Computer science; Sampling (signal processing); Convergence (economics); Task (project management); Importance sampling; Bellman equation; Selection (genetic algorithm); Monte Carlo method; Function (biology); Machine learning; State (computer science); Mathematical optimization; Artificial intelligence; Algorithm; Statistics; Mathematics; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003242439,0.0001167411,0.0001171624,0.0001585275,0.0001806506,0.0002223381,0.000418398,0.00005012231,0.000008173664],"category_scores_gemma":[0.0001506082,0.0001104012,0.00007201093,0.0005145232,0.00001396986,0.001036925,0.0000861523,0.00005862886,0.00009506036],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004542277,"about_ca_system_score_gemma":0.00004096409,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001001294,"about_ca_topic_score_gemma":0.000009373332,"domain_scores_codex":[0.9988582,0.00002455045,0.0002554543,0.0002895713,0.0002842439,0.0002880265],"domain_scores_gemma":[0.9991697,0.0002815291,0.00007528818,0.0003442958,0.00006815486,0.00006102481],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000002980879,0.000005848539,0.00007898964,0.00001485982,0.00001400953,8.203923e-7,0.003007196,0.9512169,0.0002807917,0.03278191,0.005076527,0.007519132],"study_design_scores_gemma":[0.000264682,0.00008862815,0.000386321,0.00001232084,0.000005101599,6.679622e-7,0.0004550213,0.987051,0.0004265796,0.001436144,0.009724855,0.0001487169],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.003164441,0.00001495925,0.9908797,0.003411396,0.0004709552,0.0003657608,9.624798e-7,0.0007371546,0.0009546307],"genre_scores_gemma":[0.758444,0.00004899978,0.2342055,0.000759815,0.0001990533,0.0003011124,0.00002391283,0.00003937021,0.005978178],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.7566742,"threshold_uncertainty_score":0.4502032,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08781016052838872,"score_gpt":0.3141202326013707,"score_spread":0.226310072072982,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}