{"id":"W4412459220","doi":"10.1167/jov.25.9.1880","title":"Deep Reinforcement Learning's Struggle with Visuospatial Reasoning: Insights from the Same-Different Task","year":2025,"lang":"en","type":"article","venue":"Journal of Vision","topic":"Decision-Making and Behavioral Economics","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University","funders":"","keywords":"Task (project management); Reinforcement learning; Reinforcement; Cognitive psychology; Psychology; Computer science; Cognitive science; Artificial intelligence; Social psychology; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001529305,0.0004840488,0.0005510894,0.000256004,0.0002788922,0.0008152906,0.0008399872,0.001023919,0.001395409],"category_scores_gemma":[0.008025333,0.0002337121,0.000412124,0.0002329801,0.00138653,0.001282826,0.0009924949,0.00210076,0.0001540135],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006953928,"about_ca_system_score_gemma":0.0005931751,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003574472,"about_ca_topic_score_gemma":0.002363087,"domain_scores_codex":[0.9995369,0.0002139025,0.00002409515,0.00008319215,0.00008695114,0.00005498781],"domain_scores_gemma":[0.9969556,0.002291848,0.0001947398,0.0001935443,0.0001942351,0.0001700341],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004456221,0.0004316041,0.009022255,0.0003604343,0.0001191497,0.0004991552,0.0009276196,0.749175,0.01409228,0.06965645,0.003806841,0.1514635],"study_design_scores_gemma":[0.00001362083,0.00005285603,0.0006646459,0.00001024263,0.000003857826,0.00003488357,0.00003341032,0.9654029,0.001384563,0.0317596,0.000631307,0.000008003771],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4287981,0.0013213,0.5541193,0.003607487,0.00009730419,0.00007490064,0.000156905,0.0005047789,0.01131992],"genre_scores_gemma":[0.9389863,0.0003801487,0.05772014,0.0002095308,0.00004372362,0.00004813531,0.0000892109,0.00004952708,0.002473237],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003574472,"threshold_uncertainty_score":0.008087814,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03132511678826457,"score_gpt":0.3472424157034376,"score_spread":0.315917298915173,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}