{"id":"W4301895924","doi":"10.21428/594757db.8e09102d","title":"Balancing Information with Observation Costs in Deep Reinforcement Learning","year":2022,"lang":"en","type":"article","venue":"","topic":"Fuel Cells and Related Materials","field":"Engineering","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada; Vector Institute; University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; State (computer science); Artificial intelligence; Reduction (mathematics); Action (physics); Measure (data warehouse); Machine learning; Risk analysis (engineering); Data mining; Algorithm; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004432265,0.00130927,0.001411687,0.0003942677,0.0004162746,0.001321527,0.001997277,0.001660311,0.001640535],"category_scores_gemma":[0.01933907,0.0006926483,0.0003557306,0.000371361,0.001968197,0.003066605,0.001798928,0.003140302,0.0002074681],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001698721,"about_ca_system_score_gemma":0.001643526,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004412808,"about_ca_topic_score_gemma":0.004159166,"domain_scores_codex":[0.9985043,0.0006181636,0.00008680301,0.0002557167,0.000363091,0.0001720361],"domain_scores_gemma":[0.9899489,0.007714715,0.0008157012,0.0005613733,0.0006252749,0.0003340943],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002068918,0.0001240974,0.001225609,0.00007483083,0.00004240104,0.00005544502,0.00005740462,0.9484943,0.001429264,0.02001357,0.0004791295,0.02779698],"study_design_scores_gemma":[0.00001810278,0.00003647256,0.00008613683,0.00000571295,0.00000507196,0.000006087862,0.000003188818,0.9924005,0.000412049,0.006912483,0.0001089496,0.000005179805],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09584485,0.0006288662,0.8984317,0.001318253,0.0001005474,0.00008080029,0.00006907538,0.0005996799,0.002926211],"genre_scores_gemma":[0.9549964,0.0001281416,0.04312227,0.0001949997,0.0000439973,0.00008951881,0.00004308867,0.0000495843,0.001332112],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004432265,"threshold_uncertainty_score":0.0234403,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0041193932377063,"score_gpt":0.1618116194556102,"score_spread":0.1576922262179039,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}