{"id":"W2607284162","doi":"10.1016/j.jecp.2017.03.010","title":"Testing the validity of a continuous false belief task in 3- to 7-year-old children","year":2017,"lang":"en","type":"article","venue":"Journal of Experimental Child Psychology","topic":"Child and Animal Learning Development","field":"Psychology","cited_by":18,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Ottawa; Kwantlen Polytechnic University; Brock University","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"False belief; Psychology; Sandbox (software development); Task (project management); Theory of mind; Egocentrism; Peabody Picture Vocabulary Test; Cognitive psychology; Developmental psychology; Vocabulary; Cognition","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00365507,0.0008727013,0.0007603575,0.0009713456,0.000472159,0.001620971,0.001358538,0.001660107,0.001663507],"category_scores_gemma":[0.02110693,0.0005231948,0.0007199288,0.0002907144,0.002201204,0.001272038,0.001259605,0.001934548,0.0007527306],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006487018,"about_ca_system_score_gemma":0.00102267,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01062767,"about_ca_topic_score_gemma":0.007639563,"domain_scores_codex":[0.9980877,0.0002781761,0.000271404,0.000294142,0.000766419,0.0003020987],"domain_scores_gemma":[0.9773212,0.01300492,0.003991118,0.001686246,0.002706533,0.001289957],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.003596441,0.0035952,0.8839475,0.0002324573,0.0002317121,0.004839954,0.01446507,0.0007423605,0.06321188,0.001745809,0.001180803,0.02221072],"study_design_scores_gemma":[0.0002224224,0.002933534,0.9617377,0.00009924139,0.0002008306,0.003149558,0.005098663,0.001850444,0.0226513,0.0008598785,0.001132055,0.00006442505],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9989204,0.00005448383,0.00009759553,0.00003116902,0.00001555552,0.0000116454,0.00004627061,0.000008504193,0.0008144137],"genre_scores_gemma":[0.9986192,0.00009583122,0.0003776421,0.00004548921,0.000007535783,0.00003506699,0.0002118468,0.000007728465,0.000599635],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01062767,"threshold_uncertainty_score":0.02113163,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04818459548791645,"score_gpt":0.3637943955334106,"score_spread":0.3156098000454942,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}