{"id":"W2607284162","doi":"10.1016/j.jecp.2017.03.010","title":"Testing the validity of a continuous false belief task in 3- to 7-year-old children","year":2017,"lang":"en","type":"article","venue":"Journal of Experimental Child Psychology","topic":"Child and Animal Learning Development","field":"Psychology","cited_by":18,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Ottawa; Kwantlen Polytechnic University; Brock University","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"False belief; Psychology; Sandbox (software development); Task (project management); Theory of mind; Egocentrism; Peabody Picture Vocabulary Test; Cognitive psychology; Developmental psychology; Vocabulary; Cognition","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001001386,0.0002210532,0.0005346097,0.0001956192,0.0002664901,0.00005111652,0.0009973397,0.000130928,0.0003354717],"category_scores_gemma":[0.0003305306,0.0001619037,0.0001748852,0.0001332601,0.0002106928,0.0000955123,0.0001780608,0.0006390884,0.00007656587],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004697415,"about_ca_system_score_gemma":0.0000293755,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001298104,"about_ca_topic_score_gemma":0.000009642024,"domain_scores_codex":[0.997959,0.000237902,0.0008226942,0.0003360347,0.0002599375,0.0003844393],"domain_scores_gemma":[0.998017,0.0001385081,0.0009622815,0.0006551102,0.00008713576,0.0001399838],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0007161021,0.001500066,0.9446338,0.000003464682,0.0001962405,0.0001143522,0.005711875,0.000005459197,0.03253489,0.003047672,0.007049193,0.00448689],"study_design_scores_gemma":[0.002430967,0.001362761,0.9898466,0.0001247813,0.00001564129,0.0007101947,0.0005047244,3.119034e-7,0.001631346,0.0001097831,0.003092661,0.0001701711],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9533962,0.0006401491,0.00001463013,0.00245682,0.001223896,0.0002757309,0.000008592742,0.00001023096,0.04197377],"genre_scores_gemma":[0.9970444,0.00000972653,0.000772398,0.001474068,0.0005194339,0.000009080713,0.000001490676,0.00002757016,0.0001418232],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.04521286,"threshold_uncertainty_score":0.6602243,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04818459548791645,"score_gpt":0.3637943955334106,"score_spread":0.3156098000454942,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}