Measured against M2M100 418M on real content: the OpenNMT ⦅0⦆ convention was dropped on every single occurrence of a protected term. SentencePiece fragments punctuation runs, and the model then has nothing it recognises as a unit to copy across. The brand paid for it. "Viena Latina" came back as "Wien Latin" in one file and "Vienna Latina" in another — the same name rendered two different wrong ways across two pages, which is worse than being consistently wrong. "Grätzl" survived untouched throughout, because it is genuinely unknown to the model, whereas "Viena" reads as a city name and gets translated. Masks are now word-shaped (Zq0Xv), which a model treats like an unknown proper noun and carries through rather than translating. Matching is case- and space-insensitive, since models re-case and pad these. The tiered fallback stays as the safety net for when it is still dropped. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01NizVpJ2dwzCbjCrTLCjeHn
173 lines
6.1 KiB
Python
173 lines
6.1 KiB
Python
"""Markdown-safe translation.
|
|
|
|
DeepL preserved markup server-side with tag_handling=html. A self-hosted NMT
|
|
model has no equivalent, so structure is protected here instead: non-prose
|
|
blocks pass through untouched, and inline constructs are masked with opaque
|
|
placeholders whose survival is verified after the round trip.
|
|
|
|
Placeholders are word-shaped (``Zq0Xv``) rather than punctuation. Measured on
|
|
M2M100 418M: the OpenNMT ``⦅0⦆`` convention was dropped on every single
|
|
occurrence, because SentencePiece fragments punctuation runs and the model then
|
|
fails to copy them. A token that looks like an unknown proper noun gets carried
|
|
through, since the model has nothing to translate it to.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
|
|
from .provider import SITE_TO_MODEL, Provider
|
|
|
|
_MASK = "Zq{}Xv"
|
|
_MASK_RE = re.compile(r"Zq\s*(\d+)\s*Xv", re.IGNORECASE)
|
|
|
|
# Community vocabulary that must reach readers unchanged. Not inherited from
|
|
# the WordPress plugin, which had no glossary at all — edit freely.
|
|
PROTECTED_TERMS = [
|
|
"Viena Latina",
|
|
"Grätzl",
|
|
"empanadas de viento",
|
|
"Naschmarkt",
|
|
]
|
|
|
|
_FENCE = re.compile(r"^\s*(?:```|~~~)")
|
|
_HEADING = re.compile(r"^(#{1,6}\s+)(.*)$")
|
|
_LIST = re.compile(r"^(\s*(?:[-*+]|\d+[.)])\s+)(.*)$")
|
|
_QUOTE = re.compile(r"^(\s*>\s?)(.*)$")
|
|
_HTML_BLOCK = re.compile(r"^\s*<")
|
|
_PREFIXED = (_HEADING, _LIST, _QUOTE)
|
|
|
|
# Inline spans that must never reach the model. Order matters: inline code is
|
|
# taken first so a URL inside backticks is masked once, not twice.
|
|
_INLINE = (
|
|
re.compile(r"`[^`]*`"), # inline code
|
|
re.compile(r"\]\([^)]*\)"), # link/image target — the label stays translatable
|
|
re.compile(r"<[^>\s][^>]*>"), # raw HTML tags, autolinks
|
|
re.compile(r"https?://\S+"), # bare URLs
|
|
)
|
|
|
|
class PlaceholderError(RuntimeError):
|
|
"""A masked span did not survive translation intact."""
|
|
|
|
|
|
class _Masker:
|
|
def __init__(self, protect_terms: bool = True) -> None:
|
|
self.spans: list[str] = []
|
|
self.protect_terms = protect_terms
|
|
|
|
def _take(self, match: re.Match) -> str:
|
|
self.spans.append(match.group(0))
|
|
return _MASK.format(len(self.spans) - 1)
|
|
|
|
def mask(self, text: str) -> str:
|
|
for pattern in _INLINE:
|
|
text = pattern.sub(self._take, text)
|
|
if self.protect_terms:
|
|
for term in PROTECTED_TERMS:
|
|
text = re.sub(re.escape(term), self._take, text, flags=re.IGNORECASE)
|
|
return text
|
|
|
|
def restore(self, text: str) -> str:
|
|
# Models pad, re-case and reorder placeholders; normalise before matching.
|
|
text = _MASK_RE.sub(lambda m: _MASK.format(m.group(1)), text)
|
|
for index, span in enumerate(self.spans):
|
|
token = _MASK.format(index)
|
|
seen = text.count(token)
|
|
if seen != 1:
|
|
raise PlaceholderError(
|
|
f"masked span {span!r} came back {seen} times, expected once"
|
|
)
|
|
text = text.replace(token, span)
|
|
return text
|
|
|
|
|
|
def _sentences(text: str, lang: str) -> list[str]:
|
|
from sentencex import segment
|
|
|
|
return [s.strip() for s in segment(SITE_TO_MODEL[lang], text) if s.strip()]
|
|
|
|
|
|
def _attempt(text: str, src: str, tgt: str, provider: Provider, protect_terms: bool) -> str:
|
|
masker = _Masker(protect_terms=protect_terms)
|
|
pieces = _sentences(masker.mask(text), src)
|
|
if not pieces:
|
|
return text
|
|
return masker.restore(" ".join(provider.translate(pieces, src, tgt)))
|
|
|
|
|
|
def translate_text(text: str, src: str, tgt: str, provider: Provider) -> str:
|
|
"""Translate one prose string, protecting inline markup and fixed terms.
|
|
|
|
Small models drop placeholders now and then. Rather than failing the whole
|
|
publish over one proper noun, degrade in steps — but never emit a stray
|
|
placeholder, and never silently corrupt markup.
|
|
"""
|
|
if not text.strip():
|
|
return text
|
|
try:
|
|
return _attempt(text, src, tgt, provider, protect_terms=True)
|
|
except PlaceholderError as exc:
|
|
lost_term = exc # `exc` is cleared when the except block ends
|
|
|
|
# Terminology is a nice-to-have; markup is not. Retry guarding only markup
|
|
# and accept that a protected term may come back translated.
|
|
try:
|
|
result = _attempt(text, src, tgt, provider, protect_terms=False)
|
|
print(f" note: protected terms not preserved in one segment ({lost_term})")
|
|
return result
|
|
except PlaceholderError as lost_markup:
|
|
# Markup itself didn't survive. Shipping the source text is the only
|
|
# outcome that is neither corrupt nor silently wrong.
|
|
print(f" warning: one segment left untranslated ({lost_markup})")
|
|
return text
|
|
|
|
|
|
def _is_prose(line: str) -> bool:
|
|
return bool(
|
|
line.strip()
|
|
and not _FENCE.match(line)
|
|
and not _HTML_BLOCK.match(line)
|
|
and not any(p.match(line) for p in _PREFIXED)
|
|
)
|
|
|
|
|
|
def translate_markdown(body: str, src: str, tgt: str, provider: Provider) -> str:
|
|
"""Translate a markdown body, leaving every non-prose construct intact."""
|
|
lines = body.split("\n")
|
|
out: list[str] = []
|
|
i = 0
|
|
|
|
while i < len(lines):
|
|
line = lines[i]
|
|
|
|
if _FENCE.match(line):
|
|
out.append(line)
|
|
i += 1
|
|
while i < len(lines) and not _FENCE.match(lines[i]):
|
|
out.append(lines[i])
|
|
i += 1
|
|
if i < len(lines):
|
|
out.append(lines[i])
|
|
i += 1
|
|
continue
|
|
|
|
if not line.strip() or _HTML_BLOCK.match(line):
|
|
out.append(line)
|
|
i += 1
|
|
continue
|
|
|
|
prefixed = next((m for m in (p.match(line) for p in _PREFIXED) if m), None)
|
|
if prefixed:
|
|
out.append(prefixed.group(1) + translate_text(prefixed.group(2), src, tgt, provider))
|
|
i += 1
|
|
continue
|
|
|
|
# A soft-wrapped paragraph: rejoin it so sentences are translated whole.
|
|
para: list[str] = []
|
|
while i < len(lines) and _is_prose(lines[i]):
|
|
para.append(lines[i].strip())
|
|
i += 1
|
|
out.append(translate_text(" ".join(para), src, tgt, provider))
|
|
|
|
return "\n".join(out)
|