{"id":"W4385953436","doi":"10.31234/osf.io/vpx57","title":"Predicting age of acquisition for children’s early vocabulary in five languages using language model surprisal","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Language Development and Disorders","field":"Psychology","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Predictability; Concreteness; Age of Acquisition; Vocabulary; Computer science; Word (group theory); Word lists by frequency; Linguistics; Natural language processing; Noun; Predicate (mathematical logic); Artificial intelligence; Psychology; Cognitive psychology; Cognition; Mathematics; Sentence; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001314013,0.0006055791,0.0002514468,0.0009414412,0.0001698914,0.0009966575,0.0002224528,0.0004424909,0.001522557],"category_scores_gemma":[0.005781395,0.0002584821,0.0005279057,0.0004123791,0.0002895424,0.0006738292,0.00087515,0.0006154936,0.0004190876],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004693361,"about_ca_system_score_gemma":0.0004180715,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009206812,"about_ca_topic_score_gemma":0.01227244,"domain_scores_codex":[0.9997192,0.00005686772,0.00002466223,0.0001137468,0.00003989789,0.00004551201],"domain_scores_gemma":[0.9953798,0.002980037,0.0008395125,0.0001941102,0.0003000405,0.0003064211],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0002181846,0.00004004815,0.9811926,0.00002836607,0.00007474377,0.00014452,0.0007070452,0.00372308,0.004924146,0.0001767968,0.0001287692,0.008641672],"study_design_scores_gemma":[0.000006449841,0.0001748504,0.9689348,0.00002336527,0.00004695817,0.0002500621,0.0006574912,0.02645584,0.002543676,0.0006081509,0.0002773452,0.00002111508],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9986179,0.00004933239,0.0008783261,0.00001331051,0.00000135828,0.000002195085,0.0002235024,0.00002430754,0.0001897239],"genre_scores_gemma":[0.9983183,0.00004117748,0.0009733734,0.00000391275,8.718212e-7,0.000007970418,0.0005104028,0.00001129517,0.0001326544],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.009206812,"threshold_uncertainty_score":0.01830643,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03382932446985801,"score_gpt":0.3443756611534295,"score_spread":0.3105463366835716,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}