{"id":"W4405930443","doi":"10.1007/978-3-031-74640-6_18","title":"Learning When to Observe: A Frugal Reinforcement Learning Framework for a High-Cost World","year":2024,"lang":"en","type":"book-chapter","venue":"Communications in computer and information science","topic":"Innovation and Socioeconomic Development","field":"Business, Management and Accounting","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Waterloo; University of Ottawa; Vector Institute; National Research Council Canada","funders":"","keywords":"Reinforcement learning; Computer science; Reinforcement; Artificial intelligence; Psychology; Social psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002320288,0.0009952342,0.001260582,0.0005611783,0.0007993958,0.002375032,0.003404259,0.00303315,0.008040179],"category_scores_gemma":[0.007722578,0.0005914171,0.0007430876,0.000693429,0.003085041,0.003520172,0.002351696,0.003313443,0.00078259],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00221977,"about_ca_system_score_gemma":0.001585552,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0110623,"about_ca_topic_score_gemma":0.01024867,"domain_scores_codex":[0.9988494,0.0005574168,0.00003821689,0.0002106448,0.0002057831,0.0001385231],"domain_scores_gemma":[0.9974375,0.001771482,0.0001839466,0.0001480982,0.0002655493,0.0001934743],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00009303268,0.00008778872,0.0004200307,0.00008473799,0.00006120165,0.0001323633,0.0002668126,0.6719823,0.0007128084,0.2869152,0.0025565,0.0366874],"study_design_scores_gemma":[0.0000158148,0.00002298038,0.00005028954,0.000007729211,0.000009060092,0.00001148291,0.0000146274,0.9033354,0.00008831731,0.09587119,0.0005634138,0.000009671074],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01433974,0.0003531652,0.9679087,0.001432352,0.00009446442,0.00005882717,0.00009124138,0.0002534282,0.01546806],"genre_scores_gemma":[0.7351974,0.0005125974,0.24093,0.0005139006,0.0001814879,0.0002899277,0.0000953738,0.0001491723,0.02213014],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0110623,"threshold_uncertainty_score":0.02689713,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04815994935224382,"score_gpt":0.2880743671744916,"score_spread":0.2399144178222478,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}