{"id":"W2952370515","doi":"","title":"Social Influence as Intrinsic Motivation for Multi-Agent Deep Reinforcement Learning","year":2018,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Evolutionary Game Theory and Cooperation","field":"Social Sciences","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Reinforcement learning; Counterfactual thinking; Computer science; Social dilemma; Dilemma; Mechanism (biology); Social learning; Artificial intelligence; Reinforcement; Psychology; Knowledge management; Social psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["sts"],"consensus_categories":[],"category_scores_codex":[0.0006606301,0.000184181,0.000179454,0.0001327697,0.00154696,0.00007604829,0.0003958222,0.000324419,0.0001752718],"category_scores_gemma":[0.0005027654,0.0002360648,0.0001407597,0.0002861891,0.0003533677,0.0003680545,0.0002743327,0.000331636,0.0001363318],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006213555,"about_ca_system_score_gemma":0.0003630457,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005546944,"about_ca_topic_score_gemma":0.0003929285,"domain_scores_codex":[0.9985183,0.0003449735,0.0001916824,0.0005100272,0.0001246422,0.0003103644],"domain_scores_gemma":[0.9988353,0.0001168246,0.0002666454,0.0001765666,0.0005099449,0.00009473148],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001651274,0.00005964268,0.003499927,0.0000397577,0.00006210995,0.000005517389,0.007710128,0.5086585,0.00002833231,0.4791407,0.000104928,0.0005253222],"study_design_scores_gemma":[0.00509688,0.0009896661,0.0349163,0.0004441152,0.000733831,0.000001872736,0.02256519,0.596566,0.0003956511,0.2906726,0.04448551,0.003132417],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7962298,0.00001498052,0.1979483,0.0001254424,0.0003000307,0.0007114608,0.000002227664,0.0001516422,0.004516202],"genre_scores_gemma":[0.9904708,0.0001098726,0.0001503122,0.00009905115,0.0004362737,0.000007367416,0.00007530091,0.00001376041,0.008637259],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.197798,"threshold_uncertainty_score":0.9997529,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1057084921767029,"score_gpt":0.2582526959228985,"score_spread":0.1525442037461955,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}