{"id":"W3031229455","doi":"","title":"SEDAR: a Large Scale French-English Financial Domain Parallel Corpus","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Machine translation; Computer science; Domain (mathematical analysis); Preprocessor; Natural language processing; Sentence; Translation (biology); Artificial intelligence; Scale (ratio); Speech recognition; Chemistry","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00177393,0.001241071,0.0007534519,0.00438736,0.00212779,0.001677125,0.0014399,0.001433059,0.0222195],"category_scores_gemma":[0.007047587,0.0004436807,0.0006326935,0.003111232,0.001014873,0.001985593,0.001856933,0.001323009,0.009744931],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001455152,"about_ca_system_score_gemma":0.003420402,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.04281013,"about_ca_topic_score_gemma":0.04490782,"domain_scores_codex":[0.9985623,0.000552464,0.0001315275,0.0003325605,0.0002925288,0.0001287136],"domain_scores_gemma":[0.9950516,0.002121869,0.0001736339,0.0005859228,0.001739104,0.000327863],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.002438765,0.001536986,0.01375937,0.004503166,0.0005985738,0.00497691,0.003115494,0.008787898,0.05996235,0.02116169,0.6064276,0.2727313],"study_design_scores_gemma":[0.001291356,0.0005878602,0.07005242,0.0004518992,0.0005377599,0.004406522,0.00410619,0.03391825,0.03335688,0.008594587,0.8423951,0.0003011929],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.3599276,0.004966531,0.05351289,0.004001608,0.001308697,0.001363764,0.5016189,0.01960641,0.05369358],"genre_scores_gemma":[0.2726698,0.001143492,0.06100544,0.0008267665,0.0003155982,0.001221858,0.6451708,0.002067097,0.01557907],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.04281013,"threshold_uncertainty_score":0.08512193,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01198234994470673,"score_gpt":0.2627608228998279,"score_spread":0.2507784729551212,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}