{"id":"W4391428684","doi":"10.1371/journal.pone.0297183","title":"Performance of machine translators in translating French medical research abstracts to English: A comparative study of DeepL, Google Translate, and CUBBITT","year":2024,"lang":"en","type":"article","venue":"PLoS ONE","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":35,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"ROUGE; Recall; Fluency; Natural language processing; Computer science; Artificial intelligence; Psychology; Cognitive psychology; Mathematics education","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01546582,0.001857923,0.00185963,0.005489272,0.001178758,0.004190245,0.00112606,0.001830826,0.001562556],"category_scores_gemma":[0.06765994,0.0004787126,0.001198953,0.004369637,0.00148504,0.003553573,0.002372227,0.001078082,0.002550773],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002120285,"about_ca_system_score_gemma":0.002194796,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00971659,"about_ca_topic_score_gemma":0.00995285,"domain_scores_codex":[0.9846931,0.008106633,0.001972931,0.00209965,0.002614876,0.0005128829],"domain_scores_gemma":[0.9230633,0.0525621,0.004674137,0.004534526,0.0131961,0.00196982],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.01701691,0.001560558,0.09852311,0.01031739,0.003066831,0.001493518,0.009313722,0.02760573,0.02446734,0.001225171,0.03239654,0.7730131],"study_design_scores_gemma":[0.00311273,0.02746742,0.3245069,0.001944931,0.003897362,0.006551358,0.01673538,0.4612689,0.08308961,0.004374719,0.06557163,0.001478932],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9469877,0.01181571,0.01460501,0.001361344,0.0004311469,0.0004230639,0.004673105,0.009752897,0.009950036],"genre_scores_gemma":[0.9468743,0.002365357,0.03438647,0.0005554981,0.0002438103,0.0002883825,0.01118956,0.0009801699,0.003116456],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9845342,"threshold_uncertainty_score":0,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3388232812038178,"score_gpt":0.4704020086829002,"score_spread":0.1315787274790824,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}