一键重装系统工具 | U盘启动盘制作工具 | 误删文件恢复软件 | 硬盘数据抢救专家 | 电脑蓝屏修复助手 | C盘空间清理神器 | 电脑驱动离线安装工具 | 微信聊天记录恢复工具 | 照片误格式化恢复 | 电脑密码破解清除工具 | 系统崩溃紧急救援盘 | 电脑加速优化大师 | 电脑开不了机怎么重装系统 | 回收站清空了怎么恢复 | 硬盘分区丢失数据恢复 | 电脑卡顿重装系统有用吗 | U盘插入提示格式化数据恢复 | 电脑中毒文件被隐藏恢复 | 忘记电脑开机密码怎么办 | 新硬盘分区对齐工具 | 旧电脑装Win10流畅工具 | SD卡照片删除恢复免费版 | 移动硬盘打不开提示损坏修复 | 电脑无故重启系统修复工具 | 电脑小白一键重装神器 | 程序员电脑环境配置助手 | 设计师电脑字体/素材恢复工具 | 网吧网管系统维护工具箱 | 财务人员电脑发票备份恢复 | 学生党免费电脑系统安装包 | 电脑维修师傅必备工具盘 | 游戏玩家电脑性能优化助手 | 办公白领误删文档恢复软件 | 自媒体视频素材恢复工具 | 网课录制视频损坏修复工具 | 最好的U盘PE系统排名 | 数据恢复软件哪个最强 | 免费电脑助手与收费版区别 | 国产装机工具哪款无广告 | 离线版驱动助手推荐 | 轻量级电脑优化工具对比 | 支持NVMe驱动的PE工具 | 带网络功能的应急启动盘 | 2026最新版万能装机工具 | 支持Win11 24H2的PE工具 | 最新免激活系统重装工具 | 2026数据恢复软件破解版合集 | 纯净无捆绑装机助手V3.0 | 支持苹果M芯片的电脑助手 | 秋季更新版系统维护工具箱 | 电脑系统崩了怎么用U盘把重要资料拷贝出来 | 重装系统前哪些文件夹必须备份 | 固态硬盘误格式化还能恢复数据吗 | 如何制作一个既带PE又能存数据的双分区U盘 | 电脑总是弹窗广告用什么助手彻底拦截 后台管理
📢 欢迎访问系统之家!所有资源均经过安全检测。

Should you still care about translation quality?

发布时间:2026-08-29 | 浏览:2
📥 下载地址(文章开头)
装机神器,一键装机,安装任何系统。纯净版,原版,软件版,精简版,英文版。
Morana Peric, Director of Product - Trust & Transparency, DeepL Ask about the quality of any AI translation service, and someone will probably quote you a number. It’s usually a number out of 100. It will often be calculated using one of the standard formulas for ranking translation quality. It will seem like a simple, objective, definitive view of which provider’s translations are “best”. And it will almost certainly show the translations from the provider you’re talking to, at the top. After decades of ‘data is king’ mantras, we now also know that data can be interpreted in the way that suits the storyteller the best. I’ll be honest. If you ask DeepL about translation quality, you’ll get the same response. We’ll share with you the in-depth quality benchmarking analysis that we ran in March across 48,000 blind tests with language experts, that showed DeepL winning 94% of test groups across 16 language pairs and all five major competitors, including all of the most powerful LLMs. No apologies for this. However, there’s something that nobody in AI translation likes to admit. It’s an inconvenient truth that gets brushed under the carpet whenever test results are released. We’re measuring gaps in quality that are getting smaller all the time. So small, in fact, that some might argue they no longer matter. At this point, all AI translations of written text are generally good. There are few errors that a non-expert would spot at first glance. The differences are marginal, nuanced and quite subjective. That’s partly why there are so many different rankings out there, and why it’s easy for leading models to find a ranking that favors them. So, if you’re choosing an AI translation provider, should you still care about quality as a differentiator? I believe that you should, but quality as measured in test scores isn’t really what you should be focusing on. There’s a growing difference between how AI translation performs in standardized quality tests and how AI translation performs in practice, for your organization and for your content. In a test, quality gaps look marginal. Out in the real world, the differences are far greater, and they have very real business impacts. What AI translation quality tests measure To explain what I mean, let’s take a closer look at what standardized quality tests actually measure, how they measure it and, most importantly, what they don’t include. When human linguists evaluate a translation, they typically use the MQM (Multidimensional Quality Metrics) framework, which includes eight different dimensions of translation quality. By attaching metrics to these different dimensions, and choosing how to weight them, you can create a formula that represents quality mathematically, through a single-number score. MQM’s dimensions include accuracy (how precisely a translation into a target language matches the content of an original text in the source language), but also elements such as translation fluency (how naturally the translation flows in the target language), domain-specific terminology (whether it gets specialized terms right), how well it reflects style, whether it follows local conventions, and more. What the different translation quality formulas mean Using expert linguists to evaluate translations is complicated and costly. In practice, many of the translation quality rankings that are attached to AI providers don’t use human evaluation at all. Instead, they use one of the following, automated scoring systems. Over the years, these have become increasingly sophisticated. In 2001, IBM introduced BLEU (Bilingual Evaluation Understudy). It analyzes how closely a machine translation matches a human translation of the same source text. The problem? It assumes this reference translation is the best possible translation and any deviation from it is an error. In 2006, this was adapted for Translation Edit Rate (TER) evaluation, which counts the number of edits a human translator would need to make to turn the machine translation into the human one. It’s similar to BLEU, but in this system, lower numbers are better. Unbabel launched COMET (Crosslingual Optimized Metric for Evaluation of Translation) in 2020, using a neural network model to compare a machine translation to both a human translation and the source text. It references the original text to try to work out the meaning, which BLEU and TER don’t. Automated evaluation entered the LLM era in 2023 with the launch of GEMBA (GPT Estimation Metric Based Assessment), which prompts LLMs to apply the different MQM dimensions to a translation and decide how it performs. Automated translation assessment has advanced to become more and more useful, but still can’t quite match the judgment of human experts. When LLMs take on the heavy lifting of quality evaluation, they run into problems of bias, consistently favoring translations produced by their own model, even if these lack a degree of accuracy. This is partly due to the subjective nature of translation quality. In effect, an AI model is marking its own translation homework and deciding its own solution was best.
📥 下载地址(文章中间)
装机神器,一键装机,安装任何系统。纯净版,原版,软件版,精简版,英文版。
WMT25, the tenth edition of an annual conference on automated evaluation, came to a very clear conclusion about this: “Systems that topped automated metrics did not consistently win under human evaluation, pointing to persistent metric bias and reinforcing that human evaluation should remain the final arbiter of translation quality.” In other words, if you really want to know about translation quality, ask a human expert (as DeepL does in our benchmark testing). As the differences in AI translation quality become narrower, human judgment about how to balance all the different elements becomes more valuable. Quality differences are no longer obvious. At least, not in the standard quality tests. What translation quality tests miss: context, determinism and consistency As I mentioned, though, the quality of translation measured in tests is not the same as the quality of translation that you experience in practice, especially when you’re managing translation and localization across a complex organization. In fact, it’s the aspects of translation quality that the tests don’t measure which have the greatest business impact. Most of the frameworks and scoring systems I’ve mentioned in this blog are “analytic”. They oftentimes work by comparing “segments” of original and translated text, which are usually just individual sentences. They will also rate the translation quality of each sentence in isolation, as if it had nothing to do with sentences elsewhere in a text. Our actual experience of consuming translated content is nothing like this. It’s “holistic”. It includes not just the sentence that we’re reading at the time, but the entire piece of text that it’s part of. It also includes every other piece of communication that we encounter from an organization. In this holistic reality, the quality of an individually translated sentence arguably matters less than the quality of consistency and coherence across translated output as a whole. Every organization has its own distinctive vocabulary: terminology and phrases for products and features, areas of the business, aspects of service, strategic concepts and more. Translating these terms consistently is a huge aspect of translation quality. If a translation in the Help Center doesn’t match the terms featured on products or call-to-action buttons, then the translation quality is poor. It doesn’t matter if the individual translations are technically correct or not. When an AI translation platform determines terminology, style and tone at scale (as DeepL does with glossaries, style rules and translation memories) it becomes massively more valuable. In contrast, the fundamentally probabilistic nature of large general-purpose LLMs (Claude, Gemini, ChatGPT) trades away determinism, compliance, and consistency. They lose out on what really matters when it comes to translation quality in practice. What translation quality tests miss: how quality is achieved Standard quality tests for translation also don’t concern themselves with how quality performance is achieved, and how scalable this really is. DeepL’s own blind tests with human experts show that the latest general-purpose LLMs at high-reasoning settings can produce competitive translation quality on specific language pairs. However, they do so by using larger models, broader task surfaces and wider context windows. This means that they use a lot of tokens (with unpredictable costs) and they take a lot longer (at least 4x and up to 29x slower), which has a big impact if you’re deploying AI translation in products, through an API. Most importantly for quality, they are inherently less predictable in the translations they choose. They essentially reinvent the translation wheel for each task, and so the terms and phrases they use in one context are far less likely to be repeated in another. If you’re running translation at enterprise scale, that’s the aspect of quality you need to pay closest attention to. Quality test scores tell you almost nothing useful about it. Why we’re developing a holistic view of quality that puts customers in control For these reasons, automated test scores are a very limited guide to translation quality. However, this doesn’t mean they aren’t a hugely useful tool when it comes to taking control of quality for your translations. They’re actually extremely valuable, if you define their role differently. At DeepL, we’ve built a new Translation Quality Evaluation (TQE) model , which uses a purpose-built AI model to assess the quality of DeepL translations. Crucially, though, it doesn’t just produce a “score” for the translation and move on. Its role is to highlight potential errors and areas of a translation that can be improved by looping in your human experts. It recognizes that this human judgment is, ultimately, what counts. When our TQE model analyzes a text, it does so while referring to all of the customization features that you work with on DeepL. It doesn’t analyze translation quality in isolation. It checks how your organization’s terminology, style and tone are applied, in context. It gives the holistic view of quality that really counts in practice. If you care about translation quality, and I believe you still should, this is the view of it that you should really care about. It’s not about tests. It’s about finding the perspective on translation quality that’s right for you, and works for your content, in the wild. Interested in using TQE to assess translation quality? You’ll find all the details, and options for joining our TQE testing environment, on DeepL AI Labs .
📥 下载地址(文章结尾)
装机神器,一键装机,安装任何系统。纯净版,原版,软件版,精简版,英文版。