{"id":"W4404782836","doi":"10.18653/v1/2024.emnlp-main.1231","title":"Is Child-Directed Speech Effective Training Data for Language Models?","year":2024,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Training (meteorology); Speech recognition; Training set; Language model; Natural language processing; Artificial intelligence","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004327286,0.00101832,0.0007033026,0.0004866825,0.0004639668,0.00128309,0.001610322,0.001403976,0.004263906],"category_scores_gemma":[0.02178869,0.0005993853,0.0007762009,0.0005655139,0.001016216,0.002840814,0.001578014,0.002151074,0.00428414],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006766736,"about_ca_system_score_gemma":0.001559336,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006468946,"about_ca_topic_score_gemma":0.01315417,"domain_scores_codex":[0.9967614,0.002001396,0.0001417917,0.0006470911,0.0003247599,0.0001236235],"domain_scores_gemma":[0.9885689,0.006833035,0.0003423374,0.002589988,0.001329909,0.000335801],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002529105,0.001068303,0.1438282,0.001703383,0.0007371112,0.00133459,0.004803902,0.1092671,0.07637411,0.01150343,0.05683697,0.5900139],"study_design_scores_gemma":[0.0004842065,0.002255556,0.06276964,0.001143042,0.0005597004,0.00229996,0.0058352,0.61729,0.1495998,0.02463344,0.132819,0.0003104649],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7037922,0.003268838,0.2417677,0.004701351,0.0006511618,0.000261545,0.02201273,0.008444715,0.01509985],"genre_scores_gemma":[0.8632518,0.0008579337,0.09757648,0.0008516734,0.00006812854,0.0004639503,0.03220992,0.000669718,0.004050452],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.006468946,"threshold_uncertainty_score":0.02288514,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07857330905148112,"score_gpt":0.3169085696781402,"score_spread":0.2383352606266591,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}