vienalatina/scripts/translation/markdown.py
Claude cc32cf49b5
Degrade instead of failing when a mask is lost in translation
First real pipeline run died with:
  PlaceholderError: masked span 'Viena Latina' came back 0 times

M2M100 drops placeholder tokens often enough that failing the pipeline on
mismatch would block the whole site deploy over a single proper noun. The
verification itself was right — it stopped a literal ⦅0⦆ reaching a
reader — but the policy was too blunt.

Masks are now tiered by how much they actually matter. Markup must survive;
terminology is a preference. So: try markup + terms, and on a lost term retry
guarding only markup, accepting the term may come back translated. Only if
markup itself is lost does the segment stay in the source language.

A stray placeholder or mangled URL still never reaches a reader, but one
awkward proper noun no longer blocks a publish.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01NizVpJ2dwzCbjCrTLCjeHn
2026-09-16 09:46:05 +00:00

173 lines
5.9 KiB
Python

"""Markdown-safe translation.
DeepL preserved markup server-side with tag_handling=html. A self-hosted NMT
model has no equivalent, so structure is protected here instead: non-prose
blocks pass through untouched, and inline constructs are masked with opaque
placeholders whose survival is verified after the round trip.
Placeholders use OpenNMT's protected-sequence convention (U+FF5F/U+FF60).
SentencePiece keeps these atomic; ``{{x}}``, ``<x>`` and ``%s`` get fragmented
by BPE and dropped by the model.
"""
from __future__ import annotations
import re
from .provider import SITE_TO_MODEL, Provider
OPEN, CLOSE = "⦅", "⦆"
# Community vocabulary that must reach readers unchanged. Not inherited from
# the WordPress plugin, which had no glossary at all — edit freely.
PROTECTED_TERMS = [
"Viena Latina",
"Grätzl",
"empanadas de viento",
"Naschmarkt",
]
_FENCE = re.compile(r"^\s*(?:```|~~~)")
_HEADING = re.compile(r"^(#{1,6}\s+)(.*)$")
_LIST = re.compile(r"^(\s*(?:[-*+]|\d+[.)])\s+)(.*)$")
_QUOTE = re.compile(r"^(\s*>\s?)(.*)$")
_HTML_BLOCK = re.compile(r"^\s*<")
_PREFIXED = (_HEADING, _LIST, _QUOTE)
# Inline spans that must never reach the model. Order matters: inline code is
# taken first so a URL inside backticks is masked once, not twice.
_INLINE = (
re.compile(r"`[^`]*`"), # inline code
re.compile(r"\]\([^)]*\)"), # link/image target — the label stays translatable
re.compile(r"<[^>\s][^>]*>"), # raw HTML tags, autolinks
re.compile(r"https?://\S+"), # bare URLs
)
_PLACEHOLDER = re.compile(re.escape(OPEN) + r"\s*(\d+)\s*" + re.escape(CLOSE))
class PlaceholderError(RuntimeError):
"""A masked span did not survive translation intact."""
class _Masker:
def __init__(self, protect_terms: bool = True) -> None:
self.spans: list[str] = []
self.protect_terms = protect_terms
def _take(self, match: re.Match) -> str:
self.spans.append(match.group(0))
return f"{OPEN}{len(self.spans) - 1}{CLOSE}"
def mask(self, text: str) -> str:
for pattern in _INLINE:
text = pattern.sub(self._take, text)
if self.protect_terms:
for term in PROTECTED_TERMS:
text = re.sub(re.escape(term), self._take, text, flags=re.IGNORECASE)
return text
def restore(self, text: str) -> str:
# Models pad and reorder placeholders; normalise spacing before matching.
text = _PLACEHOLDER.sub(lambda m: f"{OPEN}{m.group(1)}{CLOSE}", text)
for index, span in enumerate(self.spans):
token = f"{OPEN}{index}{CLOSE}"
seen = text.count(token)
if seen != 1:
raise PlaceholderError(
f"masked span {span!r} came back {seen} times, expected once"
)
text = text.replace(token, span)
return text
def _sentences(text: str, lang: str) -> list[str]:
from sentencex import segment
return [s.strip() for s in segment(SITE_TO_MODEL[lang], text) if s.strip()]
def _attempt(text: str, src: str, tgt: str, provider: Provider, protect_terms: bool) -> str:
masker = _Masker(protect_terms=protect_terms)
pieces = _sentences(masker.mask(text), src)
if not pieces:
return text
return masker.restore(" ".join(provider.translate(pieces, src, tgt)))
def translate_text(text: str, src: str, tgt: str, provider: Provider) -> str:
"""Translate one prose string, protecting inline markup and fixed terms.
Small models drop placeholders now and then. Rather than failing the whole
publish over one proper noun, degrade in steps — but never emit a stray
placeholder, and never silently corrupt markup.
"""
if not text.strip():
return text
try:
return _attempt(text, src, tgt, provider, protect_terms=True)
except PlaceholderError as exc:
lost_term = exc # `exc` is cleared when the except block ends
# Terminology is a nice-to-have; markup is not. Retry guarding only markup
# and accept that a protected term may come back translated.
try:
result = _attempt(text, src, tgt, provider, protect_terms=False)
print(f" note: protected terms not preserved in one segment ({lost_term})")
return result
except PlaceholderError as lost_markup:
# Markup itself didn't survive. Shipping the source text is the only
# outcome that is neither corrupt nor silently wrong.
print(f" warning: one segment left untranslated ({lost_markup})")
return text
def _is_prose(line: str) -> bool:
return bool(
line.strip()
and not _FENCE.match(line)
and not _HTML_BLOCK.match(line)
and not any(p.match(line) for p in _PREFIXED)
)
def translate_markdown(body: str, src: str, tgt: str, provider: Provider) -> str:
"""Translate a markdown body, leaving every non-prose construct intact."""
lines = body.split("\n")
out: list[str] = []
i = 0
while i < len(lines):
line = lines[i]
if _FENCE.match(line):
out.append(line)
i += 1
while i < len(lines) and not _FENCE.match(lines[i]):
out.append(lines[i])
i += 1
if i < len(lines):
out.append(lines[i])
i += 1
continue
if not line.strip() or _HTML_BLOCK.match(line):
out.append(line)
i += 1
continue
prefixed = next((m for m in (p.match(line) for p in _PREFIXED) if m), None)
if prefixed:
out.append(prefixed.group(1) + translate_text(prefixed.group(2), src, tgt, provider))
i += 1
continue
# A soft-wrapped paragraph: rejoin it so sentences are translated whole.
para: list[str] = []
while i < len(lines) and _is_prose(lines[i]):
para.append(lines[i].strip())
i += 1
out.append(translate_text(" ".join(para), src, tgt, provider))
return "\n".join(out)