From cc32cf49b594a3e000ff2b852143eb22b6c27346 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 09:46:05 +0000 Subject: [PATCH] Degrade instead of failing when a mask is lost in translation MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit First real pipeline run died with: PlaceholderError: masked span 'Viena Latina' came back 0 times M2M100 drops placeholder tokens often enough that failing the pipeline on mismatch would block the whole site deploy over a single proper noun. The verification itself was right — it stopped a literal ⦅0⦆ reaching a reader — but the policy was too blunt. Masks are now tiered by how much they actually matter. Markup must survive; terminology is a preference. So: try markup + terms, and on a lost term retry guarding only markup, accepting the term may come back translated. Only if markup itself is lost does the segment stay in the source language. A stray placeholder or mangled URL still never reaches a reader, but one awkward proper noun no longer blocks a publish. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01NizVpJ2dwzCbjCrTLCjeHn --- README.md | 13 ++++++++-- scripts/translation/markdown.py | 42 ++++++++++++++++++++++++++------- 2 files changed, 45 insertions(+), 10 deletions(-) diff --git a/README.md b/README.md index cbedf25..a917f92 100644 --- a/README.md +++ b/README.md @@ -65,8 +65,17 @@ Generated siblings additionally carry `translated_from: es`. Markup never reaches the model: code blocks and raw HTML pass through untouched, and link targets, inline code and protected community terms (`Grätzl`, `Naschmarkt`, …) are masked and verified to survive the round trip. -A mask that doesn't come back fails the pipeline rather than shipping corrupted -text — no half-translated sets ever ship. + +Small models drop those masks occasionally, so a failed round trip degrades in +steps rather than failing the publish: + +1. mask markup **and** protected terms — the normal path; +2. if a term is lost, retry guarding only markup, and log that the term may now + be translated; +3. if markup itself is lost, leave that segment in the source language and warn. + +A stray `⦅0⦆` or a mangled URL therefore never reaches a reader, and one +awkward proper noun never blocks a deploy. Backfill anything missing siblings (after the WP migration, or for pages that predate the pipeline): diff --git a/scripts/translation/markdown.py b/scripts/translation/markdown.py index ddda525..936143c 100644 --- a/scripts/translation/markdown.py +++ b/scripts/translation/markdown.py @@ -51,8 +51,9 @@ class PlaceholderError(RuntimeError): class _Masker: - def __init__(self) -> None: + def __init__(self, protect_terms: bool = True) -> None: self.spans: list[str] = [] + self.protect_terms = protect_terms def _take(self, match: re.Match) -> str: self.spans.append(match.group(0)) @@ -61,8 +62,9 @@ class _Masker: def mask(self, text: str) -> str: for pattern in _INLINE: text = pattern.sub(self._take, text) - for term in PROTECTED_TERMS: - text = re.sub(re.escape(term), self._take, text, flags=re.IGNORECASE) + if self.protect_terms: + for term in PROTECTED_TERMS: + text = re.sub(re.escape(term), self._take, text, flags=re.IGNORECASE) return text def restore(self, text: str) -> str: @@ -85,17 +87,41 @@ def _sentences(text: str, lang: str) -> list[str]: return [s.strip() for s in segment(SITE_TO_MODEL[lang], text) if s.strip()] -def translate_text(text: str, src: str, tgt: str, provider: Provider) -> str: - """Translate one prose string, protecting inline markup and fixed terms.""" - if not text.strip(): - return text - masker = _Masker() +def _attempt(text: str, src: str, tgt: str, provider: Provider, protect_terms: bool) -> str: + masker = _Masker(protect_terms=protect_terms) pieces = _sentences(masker.mask(text), src) if not pieces: return text return masker.restore(" ".join(provider.translate(pieces, src, tgt))) +def translate_text(text: str, src: str, tgt: str, provider: Provider) -> str: + """Translate one prose string, protecting inline markup and fixed terms. + + Small models drop placeholders now and then. Rather than failing the whole + publish over one proper noun, degrade in steps — but never emit a stray + placeholder, and never silently corrupt markup. + """ + if not text.strip(): + return text + try: + return _attempt(text, src, tgt, provider, protect_terms=True) + except PlaceholderError as exc: + lost_term = exc # `exc` is cleared when the except block ends + + # Terminology is a nice-to-have; markup is not. Retry guarding only markup + # and accept that a protected term may come back translated. + try: + result = _attempt(text, src, tgt, provider, protect_terms=False) + print(f" note: protected terms not preserved in one segment ({lost_term})") + return result + except PlaceholderError as lost_markup: + # Markup itself didn't survive. Shipping the source text is the only + # outcome that is neither corrupt nor silently wrong. + print(f" warning: one segment left untranslated ({lost_markup})") + return text + + def _is_prose(line: str) -> bool: return bool( line.strip()