{"id":"W3032201847","doi":"","title":"SpiCE: A New Open-Access Corpus of Conversational Bilingual Speech in Cantonese and English.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Transcription (linguistics); Spice; Sentence; Annotation; Natural language processing; Speech corpus; Storyboard; Phonetic transcription; Speech recognition; Linguistics; Artificial intelligence; Speech synthesis; Multimedia; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["open_science"],"consensus_categories":[],"category_scores_codex":[0.001410597,0.001101486,0.0007409334,0.003325559,0.001792832,0.001402179,0.001320241,0.000876927,0.01745474],"category_scores_gemma":[0.006742451,0.0003459453,0.000297857,0.002779855,0.0008624537,0.001621078,0.002911014,0.001002546,0.004922333],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001167908,"about_ca_system_score_gemma":0.003678437,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.06782387,"about_ca_topic_score_gemma":0.1025796,"domain_scores_codex":[0.9985657,0.0004423434,0.0001823456,0.0003260995,0.0003210031,0.0001624493],"domain_scores_gemma":[0.9952591,0.001540232,0.0002556547,0.0006179137,0.001774386,0.0005526578],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.003967754,0.001401783,0.04658138,0.007127294,0.0004079627,0.00424291,0.01999146,0.00279955,0.1252282,0.008758069,0.4309106,0.3485832],"study_design_scores_gemma":[0.0009013352,0.0005803974,0.3678876,0.001103859,0.0004040513,0.003614065,0.01548167,0.009502767,0.02773397,0.003194288,0.5692571,0.0003388344],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.4031602,0.003070674,0.01520615,0.001127867,0.0005689238,0.001671395,0.5331783,0.005060051,0.03695647],"genre_scores_gemma":[0.3381445,0.0006790278,0.01744675,0.000255834,0.0001597613,0.00265277,0.627647,0.0008236541,0.01219058],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.9986798,"threshold_uncertainty_score":0.1348582,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04815606913529093,"score_gpt":0.3652228133420278,"score_spread":0.3170667442067369,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}