{"id":"W4372259970","doi":"10.1109/icassp49357.2023.10095236","title":"MEET: A Monte Carlo Exploration-Exploitation Trade-Off for Buffer Sampling","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Infineon Technologies (Canada)","funders":"","keywords":"Reinforcement learning; Computer science; Sampling (signal processing); Convergence (economics); Task (project management); Importance sampling; Bellman equation; Selection (genetic algorithm); Monte Carlo method; Function (biology); Machine learning; State (computer science); Mathematical optimization; Artificial intelligence; Algorithm; Statistics; Mathematics; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003897645,0.001209816,0.001545276,0.000883231,0.0006398046,0.001316841,0.002673821,0.001962607,0.003340715],"category_scores_gemma":[0.01399843,0.0007183977,0.0006694404,0.000504452,0.00118097,0.002491295,0.002262122,0.002053697,0.0005977015],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001069819,"about_ca_system_score_gemma":0.001718862,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001942662,"about_ca_topic_score_gemma":0.00241481,"domain_scores_codex":[0.99847,0.0006171169,0.00007979794,0.000249416,0.0004127372,0.0001709035],"domain_scores_gemma":[0.9944707,0.003790139,0.0003666701,0.0004634685,0.0004577466,0.0004513148],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000852165,0.0003208288,0.00245866,0.0001665974,0.0001205179,0.0001115712,0.0001812368,0.8188533,0.004111227,0.02466108,0.002199512,0.1459633],"study_design_scores_gemma":[0.00002823458,0.00009004578,0.00008599229,0.000008940764,0.000007623063,0.00002085633,0.000007575622,0.9945627,0.0008296265,0.004069037,0.0002822473,0.000007039935],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02427267,0.0004308205,0.9720879,0.0002599408,0.00005916896,0.0001490868,0.0000429026,0.0008980123,0.001799576],"genre_scores_gemma":[0.7653271,0.0001765119,0.230999,0.0002849844,0.00008664206,0.0003890163,0.0001279818,0.000234587,0.002374087],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003897645,"threshold_uncertainty_score":0.02061301,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08781016052838872,"score_gpt":0.3141202326013707,"score_spread":0.226310072072982,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}