{"id":"W4401022147","doi":"10.1167/jov.24.7.16","title":"Visual working memory models of delayed estimation do not generalize to whole-report tasks","year":2024,"lang":"en","type":"article","venue":"Journal of Vision","topic":"Neural and Behavioral Psychology Studies","field":"Neuroscience","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Working memory; Computer science; Estimation; Visual memory; Cognitive psychology; Psychology; Artificial intelligence; Neuroscience; Cognition; Engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01866305,0.001190543,0.001621364,0.0007065595,0.0004940475,0.003322013,0.004285277,0.001164023,0.003778449],"category_scores_gemma":[0.126354,0.0009943959,0.002291872,0.0006681163,0.001679963,0.007670203,0.002553545,0.003869636,0.001148904],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001049677,"about_ca_system_score_gemma":0.0008120218,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00220779,"about_ca_topic_score_gemma":0.001762047,"domain_scores_codex":[0.994498,0.001586313,0.0005824202,0.001881782,0.001138702,0.0003127994],"domain_scores_gemma":[0.9059155,0.05016094,0.008123896,0.03194019,0.002834331,0.001025167],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.008291352,0.002370582,0.188718,0.00240788,0.003576969,0.0005747263,0.00669249,0.1876663,0.1073284,0.06227245,0.007073028,0.4230278],"study_design_scores_gemma":[0.0003861305,0.001290556,0.1370754,0.0002195473,0.0004584749,0.0009271195,0.0004367667,0.6123111,0.03321212,0.2102845,0.003044079,0.0003543363],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4791159,0.0003387464,0.5134149,0.0004199954,0.0001164942,0.0004326989,0.0009844393,0.0009539248,0.004222818],"genre_scores_gemma":[0.9482638,0.0001430736,0.04804687,0.0002907742,0.00004905968,0.0004519178,0.001394773,0.0003642741,0.0009954426],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01866305,"threshold_uncertainty_score":0.09870082,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1454124225504867,"score_gpt":0.4353740759323322,"score_spread":0.2899616533818455,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}