Merge branch 'claude/relaxed-faraday-h4zd09' of https://github.com/pablovolenski/vienalatina
All checks were successful
ci/woodpecker/push/woodpecker Pipeline was successful
All checks were successful
ci/woodpecker/push/woodpecker Pipeline was successful
This commit is contained in:
commit
3f9e46253a
13
README.md
13
README.md
@ -65,8 +65,17 @@ Generated siblings additionally carry `translated_from: es`.
|
|||||||
Markup never reaches the model: code blocks and raw HTML pass through
|
Markup never reaches the model: code blocks and raw HTML pass through
|
||||||
untouched, and link targets, inline code and protected community terms
|
untouched, and link targets, inline code and protected community terms
|
||||||
(`Grätzl`, `Naschmarkt`, …) are masked and verified to survive the round trip.
|
(`Grätzl`, `Naschmarkt`, …) are masked and verified to survive the round trip.
|
||||||
A mask that doesn't come back fails the pipeline rather than shipping corrupted
|
|
||||||
text — no half-translated sets ever ship.
|
Small models drop those masks occasionally, so a failed round trip degrades in
|
||||||
|
steps rather than failing the publish:
|
||||||
|
|
||||||
|
1. mask markup **and** protected terms — the normal path;
|
||||||
|
2. if a term is lost, retry guarding only markup, and log that the term may now
|
||||||
|
be translated;
|
||||||
|
3. if markup itself is lost, leave that segment in the source language and warn.
|
||||||
|
|
||||||
|
A stray `⦅0⦆` or a mangled URL therefore never reaches a reader, and one
|
||||||
|
awkward proper noun never blocks a deploy.
|
||||||
|
|
||||||
Backfill anything missing siblings (after the WP migration, or for pages that
|
Backfill anything missing siblings (after the WP migration, or for pages that
|
||||||
predate the pipeline):
|
predate the pipeline):
|
||||||
|
|||||||
@ -51,8 +51,9 @@ class PlaceholderError(RuntimeError):
|
|||||||
|
|
||||||
|
|
||||||
class _Masker:
|
class _Masker:
|
||||||
def __init__(self) -> None:
|
def __init__(self, protect_terms: bool = True) -> None:
|
||||||
self.spans: list[str] = []
|
self.spans: list[str] = []
|
||||||
|
self.protect_terms = protect_terms
|
||||||
|
|
||||||
def _take(self, match: re.Match) -> str:
|
def _take(self, match: re.Match) -> str:
|
||||||
self.spans.append(match.group(0))
|
self.spans.append(match.group(0))
|
||||||
@ -61,6 +62,7 @@ class _Masker:
|
|||||||
def mask(self, text: str) -> str:
|
def mask(self, text: str) -> str:
|
||||||
for pattern in _INLINE:
|
for pattern in _INLINE:
|
||||||
text = pattern.sub(self._take, text)
|
text = pattern.sub(self._take, text)
|
||||||
|
if self.protect_terms:
|
||||||
for term in PROTECTED_TERMS:
|
for term in PROTECTED_TERMS:
|
||||||
text = re.sub(re.escape(term), self._take, text, flags=re.IGNORECASE)
|
text = re.sub(re.escape(term), self._take, text, flags=re.IGNORECASE)
|
||||||
return text
|
return text
|
||||||
@ -85,17 +87,41 @@ def _sentences(text: str, lang: str) -> list[str]:
|
|||||||
return [s.strip() for s in segment(SITE_TO_MODEL[lang], text) if s.strip()]
|
return [s.strip() for s in segment(SITE_TO_MODEL[lang], text) if s.strip()]
|
||||||
|
|
||||||
|
|
||||||
def translate_text(text: str, src: str, tgt: str, provider: Provider) -> str:
|
def _attempt(text: str, src: str, tgt: str, provider: Provider, protect_terms: bool) -> str:
|
||||||
"""Translate one prose string, protecting inline markup and fixed terms."""
|
masker = _Masker(protect_terms=protect_terms)
|
||||||
if not text.strip():
|
|
||||||
return text
|
|
||||||
masker = _Masker()
|
|
||||||
pieces = _sentences(masker.mask(text), src)
|
pieces = _sentences(masker.mask(text), src)
|
||||||
if not pieces:
|
if not pieces:
|
||||||
return text
|
return text
|
||||||
return masker.restore(" ".join(provider.translate(pieces, src, tgt)))
|
return masker.restore(" ".join(provider.translate(pieces, src, tgt)))
|
||||||
|
|
||||||
|
|
||||||
|
def translate_text(text: str, src: str, tgt: str, provider: Provider) -> str:
|
||||||
|
"""Translate one prose string, protecting inline markup and fixed terms.
|
||||||
|
|
||||||
|
Small models drop placeholders now and then. Rather than failing the whole
|
||||||
|
publish over one proper noun, degrade in steps — but never emit a stray
|
||||||
|
placeholder, and never silently corrupt markup.
|
||||||
|
"""
|
||||||
|
if not text.strip():
|
||||||
|
return text
|
||||||
|
try:
|
||||||
|
return _attempt(text, src, tgt, provider, protect_terms=True)
|
||||||
|
except PlaceholderError as exc:
|
||||||
|
lost_term = exc # `exc` is cleared when the except block ends
|
||||||
|
|
||||||
|
# Terminology is a nice-to-have; markup is not. Retry guarding only markup
|
||||||
|
# and accept that a protected term may come back translated.
|
||||||
|
try:
|
||||||
|
result = _attempt(text, src, tgt, provider, protect_terms=False)
|
||||||
|
print(f" note: protected terms not preserved in one segment ({lost_term})")
|
||||||
|
return result
|
||||||
|
except PlaceholderError as lost_markup:
|
||||||
|
# Markup itself didn't survive. Shipping the source text is the only
|
||||||
|
# outcome that is neither corrupt nor silently wrong.
|
||||||
|
print(f" warning: one segment left untranslated ({lost_markup})")
|
||||||
|
return text
|
||||||
|
|
||||||
|
|
||||||
def _is_prose(line: str) -> bool:
|
def _is_prose(line: str) -> bool:
|
||||||
return bool(
|
return bool(
|
||||||
line.strip()
|
line.strip()
|
||||||
|
|||||||
Loading…
Reference in New Issue
Block a user