{"id":"W6929314077","doi":"10.48448/71gc-k031","title":"Episodic Policy Gradient Training","year":2022,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"COVID-19 Impact on Reproduction","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Hyperparameter; Episodic memory; Reinforcement learning; Markov decision process; Gradient boosting; Boosting (machine learning); Hyperparameter optimization; Scheduling (production processes)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00101099,0.0002880337,0.0004414906,0.002455687,0.0002698027,0.00005092456,0.0004282565,0.0001116434,0.008429741],"category_scores_gemma":[0.001440061,0.0002616994,0.00009991274,0.002390117,0.0008616421,0.00008710878,0.0002370253,0.0005209647,0.0001770503],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001397945,"about_ca_system_score_gemma":0.004005128,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001375857,"about_ca_topic_score_gemma":0.0001115074,"domain_scores_codex":[0.9968467,0.00003996736,0.0002924377,0.000950488,0.001215688,0.0006546912],"domain_scores_gemma":[0.9982079,0.00004235558,0.0002443629,0.001060373,0.00006759685,0.0003773581],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00008437737,0.0007503974,0.002824571,0.0005307405,0.0001837394,0.0002667955,0.004023419,0.0001477233,0.006465279,0.02157992,0.5751306,0.3880124],"study_design_scores_gemma":[0.0006151626,0.0003416622,0.0008121891,0.00009999416,0.00007220218,0.0002675,0.0005824423,0.0004752546,0.00006011897,0.0005553023,0.9958153,0.0003028947],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.001347087,0.0007985434,0.000713517,0.02869233,0.002231307,0.001430632,0.00006277814,0.001135344,0.9635885],"genre_scores_gemma":[0.07791517,0.0002057543,0.004584766,0.008295063,0.004533418,0.0000695213,0.0001630824,0.0006045877,0.9036286],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.4206846,"threshold_uncertainty_score":0.9999835,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05818920261221472,"score_gpt":0.3718802003393841,"score_spread":0.3136909977271694,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}