{"id":"W4386596863","doi":"10.1111/cogs.13334","title":"Predicting Age of Acquisition for Children's Early Vocabulary in Five Languages Using Language Model Surprisal","year":2023,"lang":"en","type":"article","venue":"Cognitive Science","topic":"Language Development and Disorders","field":"Psychology","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Predictability; Concreteness; Age of Acquisition; Vocabulary; Computer science; Word (group theory); Word lists by frequency; Noun; Natural language processing; Linguistics; Artificial intelligence; Psychology; Cognitive psychology; Cognition; Sentence; Mathematics; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006442614,0.0001019817,0.0001386372,0.0003973595,0.0001083611,0.00002555836,0.0001930643,0.00005432948,0.00003619054],"category_scores_gemma":[0.0003052073,0.0001005218,0.00004141004,0.001001089,0.0002990141,0.0002200671,0.00008631463,0.00007849398,0.00001811121],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002609172,"about_ca_system_score_gemma":0.00009356684,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003922839,"about_ca_topic_score_gemma":0.00004713754,"domain_scores_codex":[0.9988338,0.00003620987,0.0001804088,0.0003498417,0.0002245163,0.0003752256],"domain_scores_gemma":[0.9995021,0.0001568684,0.00008072022,0.0001109576,0.0001022695,0.00004704294],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0003277027,0.0002403265,0.6494967,0.00006238923,0.00007192349,0.0002404618,0.243981,0.001240527,0.07961202,0.0008723955,0.0001102631,0.02374423],"study_design_scores_gemma":[0.001894723,0.00009793875,0.9117443,0.0001595861,0.00002873313,0.000005999807,0.0572284,0.01915677,0.008882719,0.0005123166,2.089412e-7,0.0002883355],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9917125,0.0001424021,0.005288458,0.00001238926,0.000121082,0.0004078771,0.0001027169,0.00006670664,0.002145834],"genre_scores_gemma":[0.9988877,0.000002078563,0.0006069686,0.00008103743,0.00004258794,0.00003324776,0.0001079193,0.00001331736,0.0002251],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2622476,"threshold_uncertainty_score":0.409916,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02620242596379231,"score_gpt":0.3570726994257877,"score_spread":0.3308702734619954,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}