Compare commits

..

No commits in common. "ec1e790713e851f52f80878dadf9f75e7d604739" and "3f9e46253aecd73d4d4c7297b13bea24656f1e50" have entirely different histories.

7 changed files with 11 additions and 65 deletions

View File

@ -1,8 +0,0 @@
---
title: über
lang: de
manual_translation: false
translated_from: es
---
**Viena Latina** ist die lateinamerikanische Gemeinschaft in Wien: Tourismusführer, kulturelle Agenda, Gastronomie, Gemeinschaftsleben und lateinamerikanische Geschäfte in der Stadt.

View File

@ -1,8 +0,0 @@
---
title: Sobre
lang: pt-br
manual_translation: false
translated_from: es
---
**Viena Latina** é a comunidade latino-americana em Viena: guias de turismo, agenda cultural, gastronomia, vida comunitária e lojas latinas na cidade.

View File

@ -1,8 +0,0 @@
---
title: Kontakt
lang: de
manual_translation: false
translated_from: es
---
Schreiben Sie uns an **hola@vienalatina.com**.

View File

@ -1,8 +0,0 @@
---
title: Contacto
lang: pt-br
manual_translation: false
translated_from: es
---
Escreva-nos a **hola@vienalatina.com**.

View File

@ -1,11 +0,0 @@
---
title: Testartikel
date: 2026-09-16
lang: de
manual_translation: false
categories:
- Comunidad
translated_from: es
---
Dies ist ein Test des automatischen Übersetzungsflusses im Grätzl. Viena Latina veröffentlicht in Spanisch, Deutsch und Portugiesisch.

View File

@ -1,11 +0,0 @@
---
title: Artigo de prova
date: 2026-09-16
lang: pt-br
manual_translation: false
categories:
- Comunidad
translated_from: es
---
Este é um teste do fluxo de tradução automática no Grätzl. Viena Latina publica em espanhol, alemão e português.

View File

@ -5,11 +5,9 @@ model has no equivalent, so structure is protected here instead: non-prose
blocks pass through untouched, and inline constructs are masked with opaque blocks pass through untouched, and inline constructs are masked with opaque
placeholders whose survival is verified after the round trip. placeholders whose survival is verified after the round trip.
Placeholders are word-shaped (``Zq0Xv``) rather than punctuation. Measured on Placeholders use OpenNMT's protected-sequence convention (U+FF5F/U+FF60).
M2M100 418M: the OpenNMT ``⦅0⦆`` convention was dropped on every single SentencePiece keeps these atomic; ``{{x}}``, ``<x>`` and ``%s`` get fragmented
occurrence, because SentencePiece fragments punctuation runs and the model then by BPE and dropped by the model.
fails to copy them. A token that looks like an unknown proper noun gets carried
through, since the model has nothing to translate it to.
""" """
from __future__ import annotations from __future__ import annotations
@ -18,8 +16,7 @@ import re
from .provider import SITE_TO_MODEL, Provider from .provider import SITE_TO_MODEL, Provider
_MASK = "Zq{}Xv" OPEN, CLOSE = "⦅", "⦆"
_MASK_RE = re.compile(r"Zq\s*(\d+)\s*Xv", re.IGNORECASE)
# Community vocabulary that must reach readers unchanged. Not inherited from # Community vocabulary that must reach readers unchanged. Not inherited from
# the WordPress plugin, which had no glossary at all — edit freely. # the WordPress plugin, which had no glossary at all — edit freely.
@ -46,6 +43,9 @@ _INLINE = (
re.compile(r"https?://\S+"), # bare URLs re.compile(r"https?://\S+"), # bare URLs
) )
_PLACEHOLDER = re.compile(re.escape(OPEN) + r"\s*(\d+)\s*" + re.escape(CLOSE))
class PlaceholderError(RuntimeError): class PlaceholderError(RuntimeError):
"""A masked span did not survive translation intact.""" """A masked span did not survive translation intact."""
@ -57,7 +57,7 @@ class _Masker:
def _take(self, match: re.Match) -> str: def _take(self, match: re.Match) -> str:
self.spans.append(match.group(0)) self.spans.append(match.group(0))
return _MASK.format(len(self.spans) - 1) return f"{OPEN}{len(self.spans) - 1}{CLOSE}"
def mask(self, text: str) -> str: def mask(self, text: str) -> str:
for pattern in _INLINE: for pattern in _INLINE:
@ -68,10 +68,10 @@ class _Masker:
return text return text
def restore(self, text: str) -> str: def restore(self, text: str) -> str:
# Models pad, re-case and reorder placeholders; normalise before matching. # Models pad and reorder placeholders; normalise spacing before matching.
text = _MASK_RE.sub(lambda m: _MASK.format(m.group(1)), text) text = _PLACEHOLDER.sub(lambda m: f"{OPEN}{m.group(1)}{CLOSE}", text)
for index, span in enumerate(self.spans): for index, span in enumerate(self.spans):
token = _MASK.format(index) token = f"{OPEN}{index}{CLOSE}"
seen = text.count(token) seen = text.count(token)
if seen != 1: if seen != 1:
raise PlaceholderError( raise PlaceholderError(