{"id":"W4394168003","doi":"10.6084/m9.figshare.21431332","title":"Creating a Large-Scale Audio-Aligned Parsed Corpus of Bilingual Russian Child and Child-Directed Speech (BiRCh): Challenges, Solutions, and Implications for Research","year":2022,"lang":"en","type":"dataset","venue":"Figshare","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Parsing; Scale (ratio); Linguistics; Computer science; Psychology; Natural language processing; Geography; Cartography","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003052681,0.001095777,0.0009203255,0.003204333,0.001556179,0.001375379,0.001998483,0.001879453,0.009849786],"category_scores_gemma":[0.005824489,0.0005841405,0.000696224,0.002821377,0.0009503714,0.001077415,0.003104213,0.001833705,0.01074542],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001480672,"about_ca_system_score_gemma":0.002240789,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.03096513,"about_ca_topic_score_gemma":0.06513595,"domain_scores_codex":[0.9974026,0.000948966,0.0002589667,0.0006629825,0.0004673463,0.0002590654],"domain_scores_gemma":[0.9956611,0.001746198,0.00022494,0.001015726,0.001029087,0.0003230381],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001666459,0.001012571,0.02558542,0.004244727,0.0003198881,0.002568693,0.003638857,0.005657091,0.03126434,0.005932738,0.7752448,0.1428644],"study_design_scores_gemma":[0.0009267001,0.0003594618,0.1670526,0.0005753721,0.000208968,0.00205659,0.006260606,0.01407715,0.01579043,0.003288528,0.7890862,0.0003174373],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.0975884,0.001038641,0.009035536,0.0009542544,0.0004145176,0.0007687804,0.8772225,0.004532306,0.008445012],"genre_scores_gemma":[0.02314163,0.000125274,0.01051285,0.0001072033,0.00002935167,0.001147462,0.9627706,0.0002007554,0.001964828],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.03096513,"threshold_uncertainty_score":0.06156975,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.127277236659284,"score_gpt":0.3427014884403706,"score_spread":0.2154242517810866,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}