{"id":"W6947682406","doi":"10.48448/r66m-5q64","title":"Learning to Shape Rewards using a Game of Two Partners","year":2023,"lang":"en","type":"other","venue":"Open MIND","topic":"Biochemical and Structural Characterization","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada); University of Alberta","funders":"","keywords":"Reinforcement learning; Task (project management); Construct (python library); Function (biology); Convergence (economics); Markov decision process; Domain (mathematical analysis); Temporal difference learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00005844997,0.0001293631,0.0001850556,0.00003841069,0.00001374895,0.00002504772,0.0002228685,0.0001924296,0.001093986],"category_scores_gemma":[0.00004999587,0.0001182424,0.00004672715,0.00009106621,0.00002781966,0.000001434291,0.0002555092,0.00007119038,0.00005367197],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000007373502,"about_ca_system_score_gemma":0.00004122756,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0000796641,"about_ca_topic_score_gemma":0.00005594015,"domain_scores_codex":[0.9993376,0.00002311685,0.0001360487,0.0002975274,0.00007742969,0.0001283237],"domain_scores_gemma":[0.9996341,0.000002615071,0.0001199642,0.0001611842,0.00002486615,0.00005723237],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00003101684,0.000005544139,0.0001915965,0.00001383423,0.00004505253,0.000001776187,0.00005014575,0.00002185941,0.9643202,8.743011e-7,0.003538413,0.03177963],"study_design_scores_gemma":[0.0002934147,0.0001274292,0.0001146696,0.0001755116,0.00003149763,0.000004248124,0.0000343128,0.0001404446,0.2810899,0.000003638417,0.7176947,0.0002902499],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"other","genre_scores_codex":[0.9299569,0.00008549693,0.00006654577,0.00004640761,0.0001993385,0.0004100267,0.000106648,0.00000616215,0.06912244],"genre_scores_gemma":[0.3678598,0.00008715926,0.006016976,0.0000826878,0.0007520887,0.00001464535,0.0008167283,0.000545831,0.6238241],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7141563,"threshold_uncertainty_score":0.9998192,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04118907524627249,"score_gpt":0.3566924011245187,"score_spread":0.3155033258782461,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}