{"id":"W4411413826","doi":"10.33767/osf.io/t3b62_v2","title":"Measuring Lexical Distance between Parallel Corpora: The Case of AI-Generated News Translation","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Institut de Valorisation des Données; Canada First Research Excellence Fund","keywords":"Computer science; Jaccard index; Source text; Natural language processing; Compiler; Artificial intelligence; Translation (biology); Strengths and weaknesses; Heuristic; Machine translation; Value (mathematics); Information retrieval; Linguistics; Programming language; Cluster analysis","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01196093,0.0007111155,0.001183192,0.006338453,0.002540735,0.004588097,0.001863373,0.001885156,0.001718806],"category_scores_gemma":[0.09395844,0.0006930425,0.0006430031,0.01320505,0.003215417,0.006694297,0.003259272,0.001536075,0.0009832291],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001949141,"about_ca_system_score_gemma":0.001335192,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004097017,"about_ca_topic_score_gemma":0.004149984,"domain_scores_codex":[0.9784557,0.01157452,0.001900507,0.002771246,0.004831877,0.0004661153],"domain_scores_gemma":[0.9096325,0.06254276,0.005698944,0.01068913,0.01081661,0.0006200741],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00200304,0.0007661678,0.05844035,0.002450925,0.0005675033,0.006541517,0.02592385,0.07039864,0.02587376,0.1370178,0.007189194,0.6628273],"study_design_scores_gemma":[0.0002983014,0.000846491,0.05428776,0.0005312015,0.0003648514,0.005876095,0.01643503,0.5737076,0.05673815,0.233643,0.05690644,0.0003650971],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5908662,0.003315658,0.3800107,0.001693149,0.0002913532,0.0005672488,0.001254191,0.000949074,0.0210525],"genre_scores_gemma":[0.6796799,0.0005429828,0.3148983,0.000107532,0.0001131495,0.0003645765,0.001808785,0.0003153936,0.00216954],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01196093,"threshold_uncertainty_score":0.06325626,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07683607276715325,"score_gpt":0.3152620369339938,"score_spread":0.2384259641668406,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}