{"id":"W4382197597","doi":"10.21203/rs.3.rs-3080402/v1","title":"Long short-term prediction guides human metacognitive reinforcement learning","year":2023,"lang":"en","type":"preprint","venue":"Research Square","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Samsung; Ministry of Science and ICT, South Korea; University of Oxford; Korea Advanced Institute of Science and Technology; National Research Foundation; National Research Foundation of Korea; Somerville College, University of Oxford","keywords":"Term (time); Reinforcement learning; Metacognition; Reinforcement; Cognitive psychology; Psychology; Computer science; Artificial intelligence; Machine learning; Social psychology; Cognition; Neuroscience","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts","scholarly_communication","open_science","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.005559465,0.0005557559,0.00059481,0.00154569,0.001305521,0.001720183,0.002996517,0.0005387416,0.0001250314],"category_scores_gemma":[0.00142523,0.0005718546,0.0003609966,0.001174505,0.0002723964,0.0006378929,0.009968016,0.004777181,0.0007082712],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008608457,"about_ca_system_score_gemma":0.0005051795,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001994809,"about_ca_topic_score_gemma":0.00002358547,"domain_scores_codex":[0.990626,0.001188577,0.001047316,0.001609487,0.003983843,0.001544783],"domain_scores_gemma":[0.9949754,0.0007186027,0.000291436,0.001906435,0.001726934,0.000381217],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001108626,0.00003568616,0.01651881,0.0008322469,0.00028181,0.0002035694,0.001650071,0.9718572,0.0002301443,0.003428519,0.002029157,0.002921687],"study_design_scores_gemma":[0.000530131,0.001231542,0.06311522,0.003341352,0.00006940919,0.00001253431,0.0005747271,0.9245808,0.001456182,0.001939042,0.002144795,0.001004246],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.007422018,0.0001907241,0.9747743,0.0003955476,0.001077317,0.002251347,0.00001110576,0.001756461,0.01212118],"genre_scores_gemma":[0.9712536,0.0004921958,0.002327071,0.00002288234,0.0006364546,0.0006213648,0.0007565059,0.0001237469,0.02376616],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9724472,"threshold_uncertainty_score":0.9999946,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1644300741039804,"score_gpt":0.4243602209578892,"score_spread":0.2599301468539088,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}