{"id":"W6892661701","doi":"10.5281/zenodo.10626005","title":"Learning Conditional Policies for Crystal Design Using Offline Reinforcement Learning","year":2024,"lang":"en","type":"other","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Machine Learning in Materials Science","field":"Materials Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal; Polytechnique Montréal; Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Reinforcement learning; Crystal (programming language); Online and offline; Key (lock); Offline learning; Trajectory","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts","scholarly_communication","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.00215016,0.0003611622,0.0003664169,0.0005736799,0.001968883,0.001933974,0.001148418,0.0001879028,0.07636062],"category_scores_gemma":[0.001592706,0.0003702178,0.0001010603,0.0003905757,0.0003604667,0.0001912447,0.001270212,0.000528912,0.01281901],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002523528,"about_ca_system_score_gemma":0.00002042532,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0000839619,"about_ca_topic_score_gemma":3.304655e-7,"domain_scores_codex":[0.9965855,0.0007274802,0.000459896,0.0008211054,0.0007208285,0.0006851462],"domain_scores_gemma":[0.9986154,0.00007250028,0.0004283033,0.000369912,0.0003341042,0.0001797226],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00008074461,0.00004271313,5.1551e-7,0.0006005423,0.00005114408,0.00001403475,0.0008849414,0.1158414,0.2622617,0.003160167,0.6162295,0.0008325519],"study_design_scores_gemma":[0.0003356806,0.000422634,0.000003121698,0.0002771055,0.00004678636,0.00009971156,0.0001818214,0.03197355,0.001765451,0.0002470866,0.9642407,0.0004062928],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"other","genre_scores_codex":[0.0009330607,0.000333544,0.5911992,0.0003539392,0.001090357,0.002056265,0.0006133603,0.005359823,0.3980605],"genre_scores_gemma":[0.08397238,0.0001427926,0.034381,0.0002797911,0.003187218,0.000001617927,0.007749875,0.0243229,0.8459624],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.5568181,"threshold_uncertainty_score":0.9998749,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05128356517961073,"score_gpt":0.2943236453669126,"score_spread":0.2430400801873019,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}