{"id":"W6966711030","doi":"10.48448/gjda-fy41","title":"PiCor: Multi-Task Deep Reinforcement Learning with Policy Correction","year":2023,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Task (project management); Constraint (computer-aided design); Control (management); Sample (material); Compounding; Interference (communication)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0009989144,0.0005692118,0.0004383008,0.003231503,0.0005941412,0.0002750241,0.0009048491,0.000239654,0.0005940041],"category_scores_gemma":[0.0007934506,0.0004811011,0.00006781105,0.004859374,0.001463044,0.0003548032,0.0003240267,0.0007782996,0.01192988],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001523095,"about_ca_system_score_gemma":0.001500156,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.007519623,"about_ca_topic_score_gemma":0.01815284,"domain_scores_codex":[0.9955291,0.00007455862,0.0004082542,0.001156318,0.001701694,0.001130069],"domain_scores_gemma":[0.9978947,0.00007655279,0.0006805498,0.0007686049,0.0002398259,0.0003397703],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001879809,0.000534043,0.005593112,0.0001964968,0.0003869601,0.0001548447,0.00300256,0.5856512,0.007760164,0.003970017,0.2778358,0.1147268],"study_design_scores_gemma":[0.001524003,0.0006424896,0.0005488909,0.0006690652,0.00009064097,0.00006491483,0.001034878,0.6397883,0.0003142393,0.00002104509,0.3540407,0.001260873],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.0001560582,0.0002209181,0.1552358,0.0002877879,0.004747454,0.002454225,0.00003581754,0.01312446,0.8237375],"genre_scores_gemma":[0.05845482,0.00009309953,0.005096486,0.0001614812,0.001052993,0.00007977911,0.0001110126,0.002763939,0.9321864],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.1501393,"threshold_uncertainty_score":0.9997641,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02307826103668365,"score_gpt":0.3051186165160267,"score_spread":0.2820403554793431,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}