{"id":"W4283787033","doi":"10.1088/2632-2153/ac7ddc","title":"Curiosity in exploring chemical spaces: intrinsic rewards for molecular reinforcement learning","year":2022,"lang":"en","type":"article","venue":"Machine Learning Science and Technology","topic":"Computational Drug Discovery Methods","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canadian Institute for Advanced Research; Vector Institute; University of Toronto","funders":"Natural Resources Canada; Mitacs; Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada; Austrian Science Fund; Compute Canada","keywords":"Curiosity; Reinforcement learning; Pace; Chemical space; Computer science; Space (punctuation); Field (mathematics); Reinforcement; Artificial intelligence; Human–computer interaction; Cognitive science; Drug discovery; Engineering; Psychology; Neuroscience; Biology; Bioinformatics; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002952512,0.0007786616,0.0007815589,0.0004422461,0.0005128075,0.001193508,0.001177106,0.001136702,0.002386953],"category_scores_gemma":[0.01578029,0.0003034356,0.0004398077,0.0002662938,0.001863601,0.001408616,0.001617903,0.001846214,0.0001926896],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001229089,"about_ca_system_score_gemma":0.001112327,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001254594,"about_ca_topic_score_gemma":0.001474362,"domain_scores_codex":[0.9988976,0.0006806566,0.00003840339,0.0001137194,0.0001452377,0.0001243531],"domain_scores_gemma":[0.9884284,0.008587601,0.001169847,0.0004923719,0.0005883275,0.0007334836],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003663909,0.0003190432,0.005350784,0.0002004339,0.0001299087,0.0001301289,0.0002243978,0.8775505,0.002469425,0.06820058,0.001726446,0.04333194],"study_design_scores_gemma":[0.00004347895,0.00009788979,0.0003162348,0.00001737131,0.00001148663,0.00001743589,0.00001692396,0.9777972,0.0003599238,0.02098692,0.0003243664,0.0000108424],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4592665,0.0009165605,0.5219834,0.002784641,0.0001250217,0.0002060695,0.00009386331,0.0005189255,0.01410502],"genre_scores_gemma":[0.9706435,0.00009546449,0.0280354,0.0001338716,0.00002007375,0.00008091703,0.00001998567,0.00002587206,0.0009449457],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002952512,"threshold_uncertainty_score":0.01561457,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01967815809897593,"score_gpt":0.2802007172333917,"score_spread":0.2605225591344157,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}