{"id":"W4386596863","doi":"10.1111/cogs.13334","title":"Predicting Age of Acquisition for Children's Early Vocabulary in Five Languages Using Language Model Surprisal","year":2023,"lang":"en","type":"article","venue":"Cognitive Science","topic":"Language Development and Disorders","field":"Psychology","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Predictability; Concreteness; Age of Acquisition; Vocabulary; Computer science; Word (group theory); Word lists by frequency; Noun; Natural language processing; Linguistics; Artificial intelligence; Psychology; Cognitive psychology; Cognition; Sentence; Mathematics; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001147085,0.0005693389,0.0002301106,0.0008815679,0.0001702006,0.0009076102,0.0002066405,0.0003835749,0.00116925],"category_scores_gemma":[0.005574649,0.0002427607,0.0004750172,0.0003904628,0.0002582161,0.0006531777,0.0007961422,0.0005765678,0.0003102092],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004357589,"about_ca_system_score_gemma":0.0004242288,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009086647,"about_ca_topic_score_gemma":0.01325918,"domain_scores_codex":[0.9997435,0.00005396977,0.00002512998,0.00009898823,0.00003675122,0.00004163355],"domain_scores_gemma":[0.9955445,0.002765564,0.0009247129,0.000170027,0.0003055301,0.0002896938],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0001406082,0.00003420758,0.9853077,0.00001989009,0.00005850254,0.0001136633,0.0005285403,0.002942941,0.003769977,0.0001406677,0.00009127756,0.006852013],"study_design_scores_gemma":[0.00000614947,0.0001756241,0.9642206,0.00002088911,0.000048178,0.0002331853,0.0006251614,0.03116847,0.002672229,0.0005755707,0.0002335365,0.00002028573],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9988624,0.00004175831,0.0007610224,0.00001025436,0.000001029694,0.000001858588,0.0001537934,0.00001879816,0.0001490584],"genre_scores_gemma":[0.9986305,0.00003594421,0.0008765524,0.000003431039,7.109166e-7,0.000006846909,0.000347147,0.000008182995,0.00009066111],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.009086647,"threshold_uncertainty_score":0.01806748,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02620242596379231,"score_gpt":0.3570726994257877,"score_spread":0.3308702734619954,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}