{"id":"W4405290985","doi":"10.1098/rsos.241875","title":"When children can explain why they believe a claim, they suggest a better empirical test for that claim","year":2024,"lang":"en","type":"article","venue":"Royal Society Open Science","topic":"Child and Animal Learning Development","field":"Psychology","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"Universitetet i Oslo","keywords":"Test (biology); Empirical research; Psychology; Object (grammar); Developmental psychology; Social psychology; Computer science; Mathematics; Artificial intelligence; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["scholarly_communication","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.002023373,0.0003156478,0.0003291758,0.00005515392,0.001001343,0.001217758,0.002174204,0.0001681434,0.0009693819],"category_scores_gemma":[0.0001263367,0.0002427822,0.0002793467,0.0003134339,0.0004343742,0.0002600738,0.0008545262,0.0005254553,0.0003366268],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002390143,"about_ca_system_score_gemma":0.0004820701,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001444968,"about_ca_topic_score_gemma":0.0004310334,"domain_scores_codex":[0.9967467,0.00007399012,0.0003052495,0.001354092,0.000592977,0.0009269794],"domain_scores_gemma":[0.9985646,0.0004792015,0.00008264287,0.0005176195,0.00008464982,0.0002712638],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"observational","study_design_scores_codex":[0.00004412309,0.0003086508,0.2606454,0.00003022317,0.0001552702,0.00001564103,0.0642053,0.0000167025,0.000329849,0.003399142,0.6553113,0.01553837],"study_design_scores_gemma":[0.001112059,0.0004769848,0.7197303,0.0002197074,0.00005987854,0.00004917718,0.002484448,0.00112337,0.0001648909,0.003421506,0.2702847,0.0008729948],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8609816,0.0006913212,0.00140504,0.08948841,0.001921625,0.003284369,0.0004566047,0.0005112558,0.04125981],"genre_scores_gemma":[0.973343,0.00001006587,0.005187386,0.01229717,0.0004104109,0.0003031569,0.00004477922,0.00005088103,0.008353114],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4590849,"threshold_uncertainty_score":0.9999439,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03606854729262957,"score_gpt":0.3387746782709916,"score_spread":0.302706130978362,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}