{"id":"W6947682406","doi":"10.48448/r66m-5q64","title":"Learning to Shape Rewards using a Game of Two Partners","year":2023,"lang":"en","type":"other","venue":"Open MIND","topic":"Biochemical and Structural Characterization","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada); University of Alberta","funders":"","keywords":"Reinforcement learning; Task (project management); Construct (python library); Function (biology); Convergence (economics); Markov decision process; Domain (mathematical analysis); Temporal difference learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001379361,0.001090891,0.0007041093,0.0003521993,0.0006943604,0.0009826738,0.001527372,0.00144607,0.006359438],"category_scores_gemma":[0.005678402,0.0004180669,0.0005941534,0.0002449077,0.001885783,0.001536486,0.001923561,0.00140775,0.0007328035],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009465646,"about_ca_system_score_gemma":0.00130041,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002436359,"about_ca_topic_score_gemma":0.00265179,"domain_scores_codex":[0.999294,0.0003009837,0.00002669829,0.0001565885,0.0001345136,0.00008721542],"domain_scores_gemma":[0.9979619,0.001212044,0.0002285113,0.0002120375,0.0001334491,0.0002520391],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002634943,0.0002208373,0.001535379,0.00009481579,0.0000688296,0.0003369354,0.0003191361,0.8431926,0.005624648,0.09190118,0.001790698,0.05465137],"study_design_scores_gemma":[0.00003399685,0.00005236891,0.00009666421,0.00000848831,0.000007098778,0.00003850967,0.00001984718,0.9742963,0.0009633084,0.02309096,0.001383183,0.000009259576],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.08068004,0.0001133813,0.9010364,0.0006467656,0.00005074544,0.0002055996,0.00007018522,0.0007676695,0.0164293],"genre_scores_gemma":[0.8010115,0.00008808873,0.1871642,0.0002257212,0.00002030116,0.0002898862,0.00006864375,0.0001154645,0.0110162],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006359438,"threshold_uncertainty_score":0.02127439,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04118907524627249,"score_gpt":0.3566924011245187,"score_spread":0.3155033258782461,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}