{"id":"W4394509641","doi":"10.6084/m9.figshare.19969795","title":"Developing and implementing an English-Spanish literary parallel audio-textual corpus for data-driven ESL learning","year":2022,"lang":"en","type":"dataset","venue":"Figshare","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Linguistics; Corpus linguistics; Natural language processing; Artificial intelligence; Philosophy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003179924,0.001589938,0.0008703484,0.003852429,0.001632416,0.001915668,0.002620935,0.001823302,0.01058934],"category_scores_gemma":[0.009047044,0.000412236,0.0009539657,0.002799676,0.001029836,0.002051919,0.003422501,0.002052772,0.01398047],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001298518,"about_ca_system_score_gemma":0.00182808,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0121148,"about_ca_topic_score_gemma":0.02382055,"domain_scores_codex":[0.9976712,0.0007981935,0.0002305504,0.0007707379,0.0003863301,0.0001430049],"domain_scores_gemma":[0.9952888,0.001890154,0.0001677255,0.001187259,0.001107565,0.0003583919],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001115124,0.001766935,0.01160468,0.003193929,0.000205665,0.001280448,0.00163012,0.008941315,0.01198452,0.005214113,0.7024736,0.2505895],"study_design_scores_gemma":[0.0008724071,0.0004502547,0.02902047,0.0007131939,0.0001549805,0.001075353,0.004811344,0.06420273,0.02141533,0.00700397,0.870077,0.0002030692],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.1212958,0.002227886,0.05336853,0.002242686,0.001714479,0.002533856,0.7666211,0.02895427,0.02104145],"genre_scores_gemma":[0.03614564,0.0002100233,0.04893086,0.0002463363,0.0000713468,0.002241817,0.9067604,0.0006050986,0.004788546],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.0121148,"threshold_uncertainty_score":0.03542483,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06614477133137872,"score_gpt":0.3325839432235653,"score_spread":0.2664391718921866,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}