{"id":"W4394168003","doi":"10.6084/m9.figshare.21431332","title":"Creating a Large-Scale Audio-Aligned Parsed Corpus of Bilingual Russian Child and Child-Directed Speech (BiRCh): Challenges, Solutions, and Implications for Research","year":2022,"lang":"en","type":"dataset","venue":"Figshare","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Parsing; Scale (ratio); Linguistics; Computer science; Psychology; Natural language processing; Geography; Cartography","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0008089015,0.0002944108,0.0004884735,0.0005563374,0.001347051,0.0001910639,0.0009447327,0.000282984,0.02037146],"category_scores_gemma":[0.002769995,0.0003062531,0.0001252696,0.0006284604,0.00004414862,0.0001625135,0.001120542,0.0004725403,0.00003321314],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00007517972,"about_ca_system_score_gemma":0.0002821938,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005947965,"about_ca_topic_score_gemma":0.0004512596,"domain_scores_codex":[0.9971191,0.0003694536,0.0004719505,0.0009578991,0.0004546822,0.0006269423],"domain_scores_gemma":[0.9969622,0.001185579,0.0003256569,0.001000182,0.0003086558,0.0002177379],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00001222886,0.0001790158,0.000001393468,0.0005985691,0.00005509881,0.000005901669,0.0002062777,2.0047e-7,0.000005495259,0.0002810248,0.9758413,0.02281353],"study_design_scores_gemma":[0.0005777514,0.0001555306,0.0007517897,0.001457113,0.00002921049,0.0001587939,0.0001754072,0.0005088141,0.0001239074,0.0004668092,0.9952248,0.0003700801],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.000006646562,0.00501526,0.00001723148,0.001594606,0.00006033065,0.001112256,0.9907531,0.0001561305,0.001284471],"genre_scores_gemma":[0.00008457193,0.001136097,0.003389573,0.0001008805,0.0001472784,0.0009898772,0.9940764,0.00002953679,0.00004583271],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.02244345,"threshold_uncertainty_score":0.999953,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.127277236659284,"score_gpt":0.3427014884403706,"score_spread":0.2154242517810866,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}