{"id":"W3120459072","doi":"10.18653/v1/2020.wmt-1.110","title":"Improving Parallel Data Identification using Iteratively Refined Sentence Alignments and Bilingual Mappings of Pre-trained Language Models","year":2020,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Machine translation; Natural language processing; Sentence; Artificial intelligence; Context (archaeology); Metric (unit); Task (project management); Language model; Similarity (geometry)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003957412,0.002025566,0.001664317,0.002231308,0.0009280328,0.001829923,0.001887691,0.001303042,0.004992487],"category_scores_gemma":[0.01426373,0.001004772,0.001552332,0.002095259,0.0006131063,0.004155117,0.002525016,0.002563463,0.006521924],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000901178,"about_ca_system_score_gemma":0.003153069,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007678352,"about_ca_topic_score_gemma":0.01550318,"domain_scores_codex":[0.9965135,0.001268153,0.0002670605,0.001051622,0.0007241514,0.0001755414],"domain_scores_gemma":[0.9934826,0.002428565,0.0003431591,0.001464298,0.002096621,0.0001847011],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007095042,0.0006992445,0.004460384,0.000402783,0.0003463902,0.0004453789,0.0006157572,0.08017517,0.05749413,0.006357787,0.02279199,0.8255014],"study_design_scores_gemma":[0.0001396782,0.0003018763,0.00170418,0.00002680633,0.00009809474,0.000233362,0.0003032535,0.9424132,0.03692773,0.007852281,0.009938086,0.00006154076],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07487601,0.0006091747,0.8968086,0.0003668735,0.0002719831,0.0001822214,0.000952549,0.02353342,0.002399169],"genre_scores_gemma":[0.2475033,0.0002495131,0.7329323,0.0003208501,0.0001358827,0.0003940421,0.009978201,0.002198704,0.006287254],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007678352,"threshold_uncertainty_score":0.02092904,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06009170999017288,"score_gpt":0.314486357809468,"score_spread":0.2543946478192952,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}