diff --git a/.woodpecker.yml b/.woodpecker.yml index b4d8104..889037c 100644 --- a/.woodpecker.yml +++ b/.woodpecker.yml @@ -1,12 +1,16 @@ # Woodpecker CI: translate → build → deploy # # Triggered by the Gitea push webhook. The translate step replaces the old -# synchronous WordPress save_post hook: publishing in Decap is one git -# commit, translation happens here, asynchronously, and every DeepL output -# lands as a reviewable bot commit. +# synchronous WordPress save_post hook: publishing in Decap is one git commit, +# translation happens here, asynchronously, and every generated sibling lands +# as a reviewable bot commit. +# +# Translation is self-hosted — the model ships inside vienalatina/translate, +# so there is no API key and no third-party request. Build that image on the +# server before the first run: +# docker build -t vienalatina/translate:1 docker/translate # # Secrets to configure in Woodpecker (repo settings → secrets): -# deepl_api_key — DeepL free-tier key (500k chars/mo) # gitea_push_token — Gitea token for the translations bot user # (translations@vienalatina.com) with repo write access # @@ -19,17 +23,12 @@ when: steps: translate: - image: python:3.12-slim + image: vienalatina/translate:1 + pull: false # built locally on the server, never fetched from a registry environment: - DEEPL_API_KEY: - from_secret: deepl_api_key GITEA_PUSH_TOKEN: from_secret: gitea_push_token commands: - # translate.py shells out to git (rev-list/diff/add/commit/push) — the - # slim image doesn't ship it. - - apt-get update -qq && apt-get install -qq -y --no-install-recommends git - - pip install --quiet requests pyyaml - python scripts/translate.py build: diff --git a/README.md b/README.md index 915e8b4..cbedf25 100644 --- a/README.md +++ b/README.md @@ -5,30 +5,39 @@ Vienna — Hugo + Decap CMS + Gitea + Woodpecker CI on a single Hetzner CX22 (Nuremberg, DE). Replaces the previous WordPress + Polylang + synchronous DeepL stack. +Translation is self-hosted too: M2M100 418M (MIT) runs on CPU via CTranslate2 +inside the pipeline image. No API key, no quota, and no visitor or content data +leaving the server. + ``` Pablo ──► Decap CMS (/admin) ──commit──► Gitea ──webhook──► Woodpecker CI │ - translate (DeepL, async) ─┤ - build (hugo) ─┤ - deploy (rsync) ─┘ + translate (M2M100, async) ──┤ + build (hugo) ─┤ + deploy (rsync) ─┘ ▼ Caddy 2 serves /var/www/vienalatina.com ``` ## Languages -Spanish is the authoring language; German (`/de/`) and Brazilian Portuguese -(`/pt-br/`) siblings are generated by CI as reviewable git commits. Siblings -pair by filename basename: +Spanish, German (`/de/`) and Brazilian Portuguese (`/pt-br/`). **Any of the +three can be the authored original**; the other two are generated by CI as +reviewable git commits. Siblings pair by filename basename: ``` -content/post/mi-articulo.es.md ← authored in Decap +content/post/mi-articulo.es.md ← authored (no `translated_from`) content/post/mi-articulo.de.md ← written by scripts/translate.py content/post/mi-articulo.pt-br.md ← written by scripts/translate.py ``` -Set `manual_translation: true` in a sibling's frontmatter to freeze it — -CI will never overwrite it again. +A generated file carries `translated_from`, which is what stops CI translating +its own output back into a loop. Machine output is never treated as a source. + +`manual_translation: true` means "hands off", on both sides: + +- on an **authored original** — don't generate siblings for this post at all +- on a **generated sibling** — never overwrite it again ### Frontmatter contract @@ -37,23 +46,59 @@ CI will never overwrite it again. title: "Mi artículo" date: 2026-07-31 lang: es -manual_translation: false # set true on *.de.md/*.pt-br.md to freeze -categories: [Gastronomía] +manual_translation: false +categories: [Gastronomía] # taxonomy terms stay Spanish in every language --- ``` +Generated siblings additionally carry `translated_from: es`. + ## Translation flow -1. Publish `*.es.md` in Decap → one git commit, no waiting on DeepL. +1. Publish in Decap → one git commit. 2. Gitea webhook fires Woodpecker. -3. `scripts/translate.py` diffs the push, translates changed `*.es.md` via - DeepL (`tag_handling=html`, protected community terms), writes the - siblings, and pushes them back as a bot commit - (`translations@vienalatina.com`). +3. `scripts/translate.py` diffs the push, finds changed authored files in any + language, generates the missing siblings, and pushes them back as a bot + commit (`translations@vienalatina.com`) marked `[skip-translate]`. 4. `hugo --minify` builds, `rsync --delete` deploys, Caddy serves. -Any DeepL error fails the pipeline visibly (red X, one-click retry) — -no half-translated sets ever ship. +Markup never reaches the model: code blocks and raw HTML pass through +untouched, and link targets, inline code and protected community terms +(`Grätzl`, `Naschmarkt`, …) are masked and verified to survive the round trip. +A mask that doesn't come back fails the pipeline rather than shipping corrupted +text — no half-translated sets ever ship. + +Backfill anything missing siblings (after the WP migration, or for pages that +predate the pipeline): + +```sh +python scripts/translate.py --backfill +``` + +## Translation engine + +Built once on the server, and again only when changing models: + +```sh +docker build -t vienalatina/translate:1 docker/translate +``` + +The image bakes in a pre-converted CTranslate2 build of M2M100 418M plus its +tokenizer, so a publish makes no network calls. Tunable via `MT_MODEL_DIR`, +`MT_TOKENIZER`, `MT_COMPUTE_TYPE` and `MT_THREADS`. + +Swapping engines means writing one class against +`scripts/translation/provider.py` — nothing in the pipeline changes. Two +upgrades worth knowing about: + +- **M2M100 1.2B** — same MIT licence and same code path, materially better + output, but ~2–2.5GB peak RAM. Needs an 8GB box (Hetzner CX32), not the CX22. +- **opus-mt-tc-big** — better still for these specific language pairs and the + only permissive option that handles Brazilian Portuguese distinctly (`>>pob<<`), + at the cost of CC-BY-4.0 attribution and one model per directed pair. + +Do **not** build on NLLB-200 (CC-BY-NC, non-commercial) or LibreTranslate +(AGPL-3.0) if this stack is ever to be sold or offered as a service. ## Server setup @@ -70,9 +115,10 @@ hugo server # http://localhost:1313 | Secret | Value | |---|---| -| `deepl_api_key` | DeepL free-tier key (500k chars/mo) | | `gitea_push_token` | Gitea token for the translations bot user, repo write access | +Translation needs no secret — the model is local. + The repo must be marked **trusted** in Woodpecker so the deploy step can mount `/var/www/vienalatina.com`. diff --git a/docker/translate/Dockerfile b/docker/translate/Dockerfile new file mode 100644 index 0000000..43ef7ee --- /dev/null +++ b/docker/translate/Dockerfile @@ -0,0 +1,42 @@ +# Pipeline image for the translate step. +# +# Model and tokenizer are baked in, so a publish makes zero network calls and +# needs no API key. torch is deliberately absent: it is only required to +# *convert* a model, and the default PyPI wheel drags in ~2.5GB of CUDA +# libraries this CPU-only box will never use. +# +# Build on the server (once, and again only when changing models): +# docker build -t vienalatina/translate:1 docker/translate +# +# The default model is a pre-converted CTranslate2 build, which avoids running +# ct2-transformers-converter on a 4GB box — it gets OOM-killed there. + +FROM python:3.12-slim + +ARG MT_MODEL_REPO=michaelfeil/ct2fast-m2m100_418M +ARG MT_TOKENIZER_REPO=facebook/m2m100_418M + +# translate.py shells out to git to diff the push and commit the siblings back. +RUN apt-get update -qq \ + && apt-get install -qq -y --no-install-recommends git \ + && rm -rf /var/lib/apt/lists/* + +RUN pip install --no-cache-dir \ + ctranslate2 \ + transformers \ + sentencepiece \ + sentencex \ + pyyaml \ + huggingface_hub + +RUN python -c "from huggingface_hub import snapshot_download; \ +snapshot_download('${MT_MODEL_REPO}', local_dir='/opt/mt/model')" \ + && python -c "import transformers; \ +transformers.AutoTokenizer.from_pretrained('${MT_TOKENIZER_REPO}') \ + .save_pretrained('/opt/mt/tokenizer')" + +# The published artifact is float16; CTranslate2 quantises to int8 on load. +ENV MT_MODEL_DIR=/opt/mt/model \ + MT_TOKENIZER=/opt/mt/tokenizer \ + MT_COMPUTE_TYPE=int8 \ + MT_THREADS=2 diff --git a/docs/server-setup.md b/docs/server-setup.md index 361e76e..e47358a 100644 --- a/docs/server-setup.md +++ b/docs/server-setup.md @@ -196,15 +196,23 @@ In Woodpecker (**https://ci.vienalatina.com**): (this auto-creates the push webhook in Gitea). 2. Repo → Settings → *Project settings* → check **Trusted** (needed so the deploy step may mount `/var/www/vienalatina.com`). -3. Repo → Settings → *Secrets* → add: - - `deepl_api_key` — your DeepL key (the same one from the WP plugin - settings page). - - `gitea_push_token` — the bot token from step 6.3. +3. Repo → Settings → *Secrets* → add `gitea_push_token`, the bot token from + step 6.3. That is the only secret — translation runs locally and needs no key. -The push in the step above has already triggered a first pipeline — it likely -ran before the secrets existed, so open it and press the retry button. All -three steps (translate → build → deploy) should go green, and -`/var/www/vienalatina.com/` on the server now contains the built site: +Build the translation image before the first run (~5 minutes; it downloads +about 1GB of model): + +```sh +cd ~/vienalatina +docker build -t vienalatina/translate:1 docker/translate +``` + +The push in the step above has already triggered a first pipeline — it ran +before the image and secret existed, so **push a new commit rather than using +Restart**. Restart replays the old commit, and a restart's empty diff range +makes the translate step find nothing to do. All three steps (translate → +build → deploy) should go green, and `/var/www/vienalatina.com/` on the server +now contains the built site: ```sh ls /var/www/vienalatina.com # index.html, de/, pt-br/, robots.txt, llms.txt … @@ -247,8 +255,14 @@ Saturday complete. 🎉 ## Cutover day (Sunday evening) 1. Run the content migration and push (see README, "One-shot content - migration"), spot-check the built site by IP or with - `curl -H "Host: vienalatina.com" http://127.0.0.1/...` on the server. + migration"). Migrated Polylang siblings arrive frozen + (`manual_translation: true`) because they are *human* translations — the + machine engine must never overwrite them. Then + `python scripts/translate.py --backfill` fills in any set that WordPress + had no translation for. Spot-check the built site with + `grep` on `/var/www/vienalatina.com/index.html`; the + `curl -H "Host: vienalatina.com" http://127.0.0.1/` trick only works after + step 3, since Caddy has no matching site block until then. 2. At the registrar: lower the `vienalatina.com` A record TTL to 300, wait for the old TTL to expire, then change the A record to `` (and `www` too, as CNAME to `vienalatina.com` or A to the same IP). @@ -256,8 +270,10 @@ Saturday complete. 🎉 `/etc/caddy/Caddyfile`, then `sudo systemctl reload caddy`. Caddy fetches the certificate as soon as DNS resolves to this server. 4. Verify: the checklist in the migration plan (hreflang tags, robots.txt, - llms.txt, Lighthouse, red-pipeline DeepL failure test, - `manual_translation: true` freeze test). + llms.txt, Lighthouse, `manual_translation: true` freeze test, and the + loop-prevention test — after the bot pushes siblings, the pipeline it + triggers must report "nothing to translate" rather than translating the + siblings back). 5. Keep the WP host untouched for 30 days as fallback; watch Google Search Console and add Caddy 301s for any 404s it reports. diff --git a/scripts/translate.py b/scripts/translate.py index e2f771f..1b684b7 100644 --- a/scripts/translate.py +++ b/scripts/translate.py @@ -1,65 +1,57 @@ #!/usr/bin/env python3 -"""Async DeepL translation step for Woodpecker CI. +"""Self-hosted translation step for Woodpecker CI. -Replaces the synchronous `save_post` hook from the WordPress plugin -(plataforma_deepl_translate in plugin/plataforma-social/plataforma-social.php). +Replaces the DeepL API call this script used to make, and restores the +multi-source behaviour of the original WordPress plugin: any of the three site +languages may be the authored original, and the other two are generated from +it. Translation runs on a model shipped inside the pipeline image, so there is +no API key, no quota and no third-party request. -For each *.es.md changed in the pushed commit range, calls DeepL with -tag_handling=html (markup survives translation) plus a glossary of -community-specific terms, and writes the *.de.md and *.pt-br.md siblings -next to the source. Sibling files carrying `manual_translation: true` in -their frontmatter are never overwritten. The siblings are committed back -to the same branch as a bot commit, then the pipeline builds and deploys. +Loop prevention, which was structural back when only Spanish could be a source: + * a generated sibling carries `translated_from`, and a file carrying it is + never itself treated as a source; + * the bot's own commits carry [skip-translate] and are skipped outright. -Failure behaviour: any DeepL error exits non-zero, the pipeline goes red, +`manual_translation: true` means "hands off", on both sides: + * on an authored source — do not generate siblings for this post at all; + * on a generated sibling — never overwrite it again. + +Failure behaviour: any translation error exits non-zero, the pipeline goes red, and nothing partial is committed — no half-translated sets. Environment: - DEEPL_API_KEY required (Woodpecker secret) - DEEPL_API_URL optional, defaults to the free-tier endpoint - CI_COMMIT_SHA / CI_PREV_COMMIT_SHA provided by Woodpecker + MT_MODEL_DIR / MT_TOKENIZER / MT_COMPUTE_TYPE / MT_THREADS see translation/ + CI_COMMIT_SHA / CI_PREV_COMMIT_SHA / CI_COMMIT_MESSAGE from Woodpecker + GITEA_PUSH_TOKEN bot push token -Dependencies: requests, pyyaml +Dependencies: ctranslate2, transformers, sentencepiece, sentencex, pyyaml """ from __future__ import annotations +import argparse import os import re import subprocess -import sys from pathlib import Path -import requests import yaml +from translation.ctranslate_provider import CTranslate2Provider +from translation.markdown import translate_markdown, translate_text +from translation.provider import SITE_LANGS + REPO_ROOT = Path(__file__).resolve().parent.parent -DEEPL_URL = os.environ.get("DEEPL_API_URL", "https://api-free.deepl.com/v2/translate") +CONTENT_DIR = REPO_ROOT / "content" -# Language slug -> DeepL target code (ported from the plugin's $lang_map). -TARGETS = { - "de": "DE", - "pt-br": "PT-BR", -} -SOURCE_SLUG = "es" -DEEPL_SOURCE = "ES" - -# Fixed community-specific terms DeepL must not "translate". -# Sent as ignored tags via tag_handling=html: each term is wrapped in -# … before the call and unwrapped after, which pins proper -# nouns without needing a server-side DeepL glossary resource. -PROTECTED_TERMS = [ - "Viena Latina", - "Grätzl", - "empanadas de viento", - "Naschmarkt", -] - -# Frontmatter keys whose string values get translated alongside the body. +# Frontmatter strings translated alongside the body. Note `categories` is +# deliberately absent: the taxonomy terms stay Spanish in every language, or +# Hugo would fork the taxonomy per language. TRANSLATED_KEYS = ("title", "description") BOT_NAME = "vienalatina-translations" BOT_EMAIL = "translations@vienalatina.com" +SKIP_MARKER = "[skip-translate]" def run(*args: str, check: bool = True) -> str: @@ -67,33 +59,15 @@ def run(*args: str, check: bool = True) -> str: return result.stdout.strip() -def changed_source_files() -> list[Path]: - """Spanish sources touched by the pushed commits. - - Filtering to *.es.md is what makes bot commits (which only add .de.md / - .pt-br.md) a no-op round — the re-fire loop the WP hook had to guard - against with meta flags cannot happen here. - """ - head = os.environ.get("CI_COMMIT_SHA", "HEAD") - prev = os.environ.get("CI_PREV_COMMIT_SHA", "") - if prev and not set(prev) <= {"0"}: - diff_range = [prev, head] - else: - diff_range = ["HEAD~1", "HEAD"] if run("git", "rev-list", "--count", "HEAD") != "1" else None - - if diff_range: - out = run("git", "diff", "--name-only", "--diff-filter=AM", *diff_range) - else: # very first commit in the repo: translate everything - out = run("git", "ls-files") - - files = [] - for line in out.splitlines(): - p = Path(line.strip()) - if p.suffix == ".md" and p.name.endswith(f".{SOURCE_SLUG}.md") and p.parts[:1] == ("content",): - full = REPO_ROOT / p - if full.exists(): - files.append(full) - return files +def split_lang(path: Path) -> tuple[str, str] | None: + """('mi-articulo', 'es') for mi-articulo.es.md, else None.""" + if path.suffix != ".md": + return None + stem = path.name[: -len(".md")] + for lang in SITE_LANGS: + if stem.endswith(f".{lang}"): + return stem[: -(len(lang) + 1)], lang + return None def split_frontmatter(text: str) -> tuple[dict, str]: @@ -108,93 +82,109 @@ def join_frontmatter(fm: dict, body: str) -> str: return f"---\n{front}---\n\n{body.lstrip()}" -def protect(text: str) -> str: - for term in PROTECTED_TERMS: - text = re.sub(re.escape(term), lambda m: f"{m.group(0)}", text, flags=re.IGNORECASE) - return text - - -def unprotect(text: str) -> str: - return re.sub(r"", "", text) - - -def deepl_translate(texts: list[str], target: str, html: bool) -> list[str]: - """Direct port of plataforma_deepl_translate(): same endpoint, same - tag_handling=html behaviour, but errors abort the pipeline instead of - silently shipping a half-translated post.""" - texts = [t for t in texts if isinstance(t, str) and t.strip()] - if not texts: - return [] - - key = os.environ.get("DEEPL_API_KEY", "") - if not key: - sys.exit("DEEPL_API_KEY is not set — configure the Woodpecker secret.") - - data: list[tuple[str, str]] = [ - ("target_lang", target), - ("source_lang", DEEPL_SOURCE), - ("tag_handling", "html"), - ("ignore_tags", "keep"), - ] - if not html: - # Titles/descriptions are plain strings; still use tag handling so - # protection works, DeepL just has no other tags to preserve. - pass - for t in texts: - data.append(("text", protect(t))) - - resp = requests.post( - DEEPL_URL, - headers={"Authorization": f"DeepL-Auth-Key {key}"}, - data=data, - timeout=60, - ) - if resp.status_code != 200: - sys.exit(f"DeepL HTTP {resp.status_code}: {resp.text[:300]}") - - return [unprotect(item["text"]) for item in resp.json().get("translations", [])] - - -def sibling_path(source: Path, slug: str) -> Path: - return source.with_name(source.name.replace(f".{SOURCE_SLUG}.md", f".{slug}.md")) - - -def is_frozen(path: Path) -> bool: +def read_frontmatter(path: Path) -> dict: if not path.exists(): - return False + return {} fm, _ = split_frontmatter(path.read_text(encoding="utf-8")) + return fm + + +def is_generated(fm: dict) -> bool: + return bool(fm.get("translated_from")) + + +def is_frozen(fm: dict) -> bool: return bool(fm.get("manual_translation")) -def translate_file(source: Path) -> list[Path]: - raw = source.read_text(encoding="utf-8") - fm, body = split_frontmatter(raw) - written = [] +def changed_markdown() -> list[Path]: + """Content files touched by the pushed commits.""" + head = os.environ.get("CI_COMMIT_SHA", "HEAD") + prev = os.environ.get("CI_PREV_COMMIT_SHA", "") + if prev and not set(prev) <= {"0"}: + out = run("git", "diff", "--name-only", "--diff-filter=AM", prev, head) + elif run("git", "rev-list", "--count", "HEAD") != "1": + out = run("git", "diff", "--name-only", "--diff-filter=AM", "HEAD~1", "HEAD") + else: # first commit in the repo + out = run("git", "ls-files") - for slug, deepl_target in TARGETS.items(): - target_file = sibling_path(source, slug) - if is_frozen(target_file): - print(f" {target_file.relative_to(REPO_ROOT)}: manual_translation=true — skipped") + paths = [] + for line in out.splitlines(): + rel = Path(line.strip()) + if rel.parts[:1] == ("content",) and (REPO_ROOT / rel).exists(): + paths.append(REPO_ROOT / rel) + return paths + + +def authored_sources(paths: list[Path]) -> list[tuple[Path, str, str]]: + """(path, basename, lang) for files that may act as a translation source.""" + sources = [] + for path in paths: + parsed = split_lang(path) + if not parsed: + continue + basename, lang = parsed + fm = read_frontmatter(path) + if is_generated(fm): + continue # machine output is never a source + if is_frozen(fm): + print(f"{path.relative_to(REPO_ROOT)}: manual_translation=true — not translated") + continue + sources.append((path, basename, lang)) + return sources + + +def missing_siblings() -> list[Path]: + """Every authored source missing at least one sibling.""" + incomplete = [] + for path in sorted(CONTENT_DIR.rglob("*.md")): + parsed = split_lang(path) + if not parsed: + continue + basename, lang = parsed + if is_generated(read_frontmatter(path)): + continue + for target in SITE_LANGS: + if target != lang and not path.with_name(f"{basename}.{target}.md").exists(): + incomplete.append(path) + break + return incomplete + + +def translate_file(source: Path, basename: str, src_lang: str, provider) -> list[Path]: + fm, body = split_frontmatter(source.read_text(encoding="utf-8")) + written: list[Path] = [] + + for tgt in SITE_LANGS: + if tgt == src_lang: + continue + target = source.with_name(f"{basename}.{tgt}.md") + target_fm = read_frontmatter(target) + if is_frozen(target_fm): + print(f" {target.relative_to(REPO_ROOT)}: manual_translation=true — skipped") continue - strings = [str(fm[k]) for k in TRANSLATED_KEYS if fm.get(k)] - translated_strings = deepl_translate(strings, deepl_target, html=False) - translated_body = deepl_translate([body], deepl_target, html=True) if body.strip() else [""] - new_fm = dict(fm) - it = iter(translated_strings) - for k in TRANSLATED_KEYS: - if fm.get(k): - new_fm[k] = next(it) - new_fm["lang"] = slug + for key in TRANSLATED_KEYS: + if fm.get(key): + new_fm[key] = translate_text(str(fm[key]), src_lang, tgt, provider) + new_fm["lang"] = tgt + new_fm["translated_from"] = src_lang new_fm["manual_translation"] = False - # Pin the URL to the shared basename so it never drifts when a title - # is retranslated (plan: /de/mi-articulo/ pairs with /mi-articulo/). - new_fm.setdefault("slug", source.name.removesuffix(f".{SOURCE_SLUG}.md")) - target_file.write_text(join_frontmatter(new_fm, translated_body[0] if translated_body else ""), encoding="utf-8") - written.append(target_file) - print(f" {target_file.relative_to(REPO_ROOT)}: written") + # Never inherit the source's slug: that would move a migrated sibling's + # URL onto the source's and break inbound links. Keep the slug the + # target already had; otherwise let Hugo fall back to the filename. + new_fm.pop("slug", None) + if target_fm.get("slug"): + new_fm["slug"] = target_fm["slug"] + + target.write_text( + join_frontmatter(new_fm, translate_markdown(body, src_lang, tgt, provider)), + encoding="utf-8", + ) + written.append(target) + print(f" {target.relative_to(REPO_ROOT)}: written") return written @@ -205,30 +195,49 @@ def commit_and_push(files: list[Path]) -> None: run("git", "config", "user.email", BOT_EMAIL) token = os.environ.get("GITEA_PUSH_TOKEN", "") repo = os.environ.get("CI_REPO", "pablo/vienalatina") - if token: # Woodpecker's clone credentials are read-only; push needs its own token - run("git", "remote", "set-url", "origin", f"https://{BOT_NAME}:{token}@git.vienalatina.com/{repo}.git") + if token: # clone credentials are read-only; pushing needs the bot token + run("git", "remote", "set-url", "origin", + f"https://{BOT_NAME}:{token}@git.vienalatina.com/{repo}.git") run("git", "add", *[str(f) for f in files]) if not run("git", "status", "--porcelain"): print("Translations identical to committed siblings — nothing to push.") return - run("git", "commit", "-m", "translate: update DE and PT-BR siblings [skip-translate]") + run("git", "commit", "-m", f"translate: update generated siblings {SKIP_MARKER}") run("git", "push", "origin", f"HEAD:{branch}") print(f"Pushed sibling commit to {branch}.") def main() -> None: - sources = changed_source_files() - if not sources: - print("No changed *.es.md files — nothing to translate.") + parser = argparse.ArgumentParser(description="Generate translated content siblings.") + parser.add_argument("--backfill", action="store_true", + help="translate every source missing siblings, not just changed files") + parser.add_argument("--no-push", action="store_true", + help="write siblings but do not commit or push (local testing)") + args = parser.parse_args() + + if SKIP_MARKER in os.environ.get("CI_COMMIT_MESSAGE", ""): + print(f"{SKIP_MARKER} commit — nothing to translate.") return - all_written: list[Path] = [] - for source in sources: - print(f"Translating {source.relative_to(REPO_ROOT)}:") - all_written.extend(translate_file(source)) + paths = missing_siblings() if args.backfill else changed_markdown() + sources = authored_sources(paths) + if not sources: + print("No authored content changed — nothing to translate.") + return - if all_written: - commit_and_push(all_written) + provider = CTranslate2Provider() + written: list[Path] = [] + for source, basename, lang in sources: + print(f"Translating {source.relative_to(REPO_ROOT)} (from {lang}):") + written.extend(translate_file(source, basename, lang, provider)) + + if not written: + print("Every sibling is frozen — nothing written.") + return + if args.no_push: + print(f"--no-push: wrote {len(written)} files, leaving them uncommitted.") + return + commit_and_push(written) if __name__ == "__main__": diff --git a/scripts/translation/__init__.py b/scripts/translation/__init__.py new file mode 100644 index 0000000..cc4ea7b --- /dev/null +++ b/scripts/translation/__init__.py @@ -0,0 +1 @@ +"""Self-hosted translation for the Viena Latina pipeline.""" diff --git a/scripts/translation/ctranslate_provider.py b/scripts/translation/ctranslate_provider.py new file mode 100644 index 0000000..6960ed6 --- /dev/null +++ b/scripts/translation/ctranslate_provider.py @@ -0,0 +1,58 @@ +"""M2M100 via CTranslate2, on CPU. + +Replaces the DeepL HTTP call. No API key, no quota, no third-party request — +the model ships inside the pipeline image. +""" + +from __future__ import annotations + +import os + +from .provider import Provider, SITE_TO_MODEL + + +class CTranslate2Provider(Provider): + def __init__( + self, + model_dir: str | None = None, + tokenizer: str | None = None, + compute_type: str | None = None, + threads: int | None = None, + ): + self.model_dir = model_dir or os.environ.get("MT_MODEL_DIR", "/opt/mt/model") + self.tokenizer = tokenizer or os.environ.get("MT_TOKENIZER", "facebook/m2m100_418M") + self.compute_type = compute_type or os.environ.get("MT_COMPUTE_TYPE", "int8") + self.threads = threads or int(os.environ.get("MT_THREADS", "2")) + self._translator = None + self._tokenizer = None + + def _load(self) -> None: + if self._translator is not None: + return + import ctranslate2 + import transformers + + self._tokenizer = transformers.AutoTokenizer.from_pretrained(self.tokenizer) + self._translator = ctranslate2.Translator( + self.model_dir, + device="cpu", + compute_type=self.compute_type, + intra_threads=self.threads, + ) + + def translate(self, texts: list[str], src: str, tgt: str) -> list[str]: + if not texts: + return [] + self._load() + tok = self._tokenizer + tok.src_lang = SITE_TO_MODEL[src] + target_token = tok.lang_code_to_token[SITE_TO_MODEL[tgt]] + + batch = [tok.convert_ids_to_tokens(tok.encode(t)) for t in texts] + results = self._translator.translate_batch( + batch, target_prefix=[[target_token]] * len(batch) + ) + # hypotheses[0][0] is the target-language token we forced; drop it. + return [ + tok.decode(tok.convert_tokens_to_ids(r.hypotheses[0][1:])) for r in results + ] diff --git a/scripts/translation/markdown.py b/scripts/translation/markdown.py new file mode 100644 index 0000000..ddda525 --- /dev/null +++ b/scripts/translation/markdown.py @@ -0,0 +1,146 @@ +"""Markdown-safe translation. + +DeepL preserved markup server-side with tag_handling=html. A self-hosted NMT +model has no equivalent, so structure is protected here instead: non-prose +blocks pass through untouched, and inline constructs are masked with opaque +placeholders whose survival is verified after the round trip. + +Placeholders use OpenNMT's protected-sequence convention (U+FF5F/U+FF60). +SentencePiece keeps these atomic; ``{{x}}``, ```` and ``%s`` get fragmented +by BPE and dropped by the model. +""" + +from __future__ import annotations + +import re + +from .provider import SITE_TO_MODEL, Provider + +OPEN, CLOSE = "⦅", "⦆" + +# Community vocabulary that must reach readers unchanged. Not inherited from +# the WordPress plugin, which had no glossary at all — edit freely. +PROTECTED_TERMS = [ + "Viena Latina", + "Grätzl", + "empanadas de viento", + "Naschmarkt", +] + +_FENCE = re.compile(r"^\s*(?:```|~~~)") +_HEADING = re.compile(r"^(#{1,6}\s+)(.*)$") +_LIST = re.compile(r"^(\s*(?:[-*+]|\d+[.)])\s+)(.*)$") +_QUOTE = re.compile(r"^(\s*>\s?)(.*)$") +_HTML_BLOCK = re.compile(r"^\s*<") +_PREFIXED = (_HEADING, _LIST, _QUOTE) + +# Inline spans that must never reach the model. Order matters: inline code is +# taken first so a URL inside backticks is masked once, not twice. +_INLINE = ( + re.compile(r"`[^`]*`"), # inline code + re.compile(r"\]\([^)]*\)"), # link/image target — the label stays translatable + re.compile(r"<[^>\s][^>]*>"), # raw HTML tags, autolinks + re.compile(r"https?://\S+"), # bare URLs +) + +_PLACEHOLDER = re.compile(re.escape(OPEN) + r"\s*(\d+)\s*" + re.escape(CLOSE)) + + +class PlaceholderError(RuntimeError): + """A masked span did not survive translation intact.""" + + +class _Masker: + def __init__(self) -> None: + self.spans: list[str] = [] + + def _take(self, match: re.Match) -> str: + self.spans.append(match.group(0)) + return f"{OPEN}{len(self.spans) - 1}{CLOSE}" + + def mask(self, text: str) -> str: + for pattern in _INLINE: + text = pattern.sub(self._take, text) + for term in PROTECTED_TERMS: + text = re.sub(re.escape(term), self._take, text, flags=re.IGNORECASE) + return text + + def restore(self, text: str) -> str: + # Models pad and reorder placeholders; normalise spacing before matching. + text = _PLACEHOLDER.sub(lambda m: f"{OPEN}{m.group(1)}{CLOSE}", text) + for index, span in enumerate(self.spans): + token = f"{OPEN}{index}{CLOSE}" + seen = text.count(token) + if seen != 1: + raise PlaceholderError( + f"masked span {span!r} came back {seen} times, expected once" + ) + text = text.replace(token, span) + return text + + +def _sentences(text: str, lang: str) -> list[str]: + from sentencex import segment + + return [s.strip() for s in segment(SITE_TO_MODEL[lang], text) if s.strip()] + + +def translate_text(text: str, src: str, tgt: str, provider: Provider) -> str: + """Translate one prose string, protecting inline markup and fixed terms.""" + if not text.strip(): + return text + masker = _Masker() + pieces = _sentences(masker.mask(text), src) + if not pieces: + return text + return masker.restore(" ".join(provider.translate(pieces, src, tgt))) + + +def _is_prose(line: str) -> bool: + return bool( + line.strip() + and not _FENCE.match(line) + and not _HTML_BLOCK.match(line) + and not any(p.match(line) for p in _PREFIXED) + ) + + +def translate_markdown(body: str, src: str, tgt: str, provider: Provider) -> str: + """Translate a markdown body, leaving every non-prose construct intact.""" + lines = body.split("\n") + out: list[str] = [] + i = 0 + + while i < len(lines): + line = lines[i] + + if _FENCE.match(line): + out.append(line) + i += 1 + while i < len(lines) and not _FENCE.match(lines[i]): + out.append(lines[i]) + i += 1 + if i < len(lines): + out.append(lines[i]) + i += 1 + continue + + if not line.strip() or _HTML_BLOCK.match(line): + out.append(line) + i += 1 + continue + + prefixed = next((m for m in (p.match(line) for p in _PREFIXED) if m), None) + if prefixed: + out.append(prefixed.group(1) + translate_text(prefixed.group(2), src, tgt, provider)) + i += 1 + continue + + # A soft-wrapped paragraph: rejoin it so sentences are translated whole. + para: list[str] = [] + while i < len(lines) and _is_prose(lines[i]): + para.append(lines[i].strip()) + i += 1 + out.append(translate_text(" ".join(para), src, tgt, provider)) + + return "\n".join(out) diff --git a/scripts/translation/provider.py b/scripts/translation/provider.py new file mode 100644 index 0000000..d5d3966 --- /dev/null +++ b/scripts/translation/provider.py @@ -0,0 +1,26 @@ +"""Translation provider interface. + +The pipeline talks to a provider, never to a model directly, so the engine can +be swapped — a larger M2M100, OPUS-MT, or a hosted API — without touching +translate.py. +""" + +from __future__ import annotations + +from abc import ABC, abstractmethod + +# Site language codes (Hugo, and the ..md filename suffix) -> model codes. +# M2M100 has no Brazilian variant, so pt-br translates as generic Portuguese. +SITE_TO_MODEL = {"es": "es", "de": "de", "pt-br": "pt"} + +SITE_LANGS = tuple(SITE_TO_MODEL) + + +class Provider(ABC): + @abstractmethod + def translate(self, texts: list[str], src: str, tgt: str) -> list[str]: + """Translate plain-text strings between two site language codes. + + Input must carry no markup: callers mask it first (see markdown.py). + Returns one string per input, in order. + """ diff --git a/scripts/wp-to-hugo.py b/scripts/wp-to-hugo.py index d56a57f..ce6bc33 100644 --- a/scripts/wp-to-hugo.py +++ b/scripts/wp-to-hugo.py @@ -81,7 +81,8 @@ def to_markdown(html: str) -> str: return conv.handle(html).strip() -def frontmatter(post: dict, lang: str, categories: dict[int, str]) -> str: +def frontmatter(post: dict, lang: str, categories: dict[int, str], + derived_from: str | None = None) -> str: title = to_markdown(post["title"]["rendered"]).replace('"', '\\"') cats = [categories[c] for c in post.get("categories", []) if c in categories] lines = [ @@ -90,9 +91,15 @@ def frontmatter(post: dict, lang: str, categories: dict[int, str]) -> str: f"date: {post['date']}", f"slug: {post['slug']}", # preserve the exact WP slug per language f"lang: {lang}", - "manual_translation: false", - f"categories: [{', '.join(cats)}]", ] + if derived_from: + # A Polylang sibling: derived, so translate.py never treats it as a + # source — and frozen, because these are *human* WordPress translations + # and regenerating them would replace them with weaker machine output. + lines.extend([f"translated_from: {derived_from}", "manual_translation: true"]) + else: + lines.append("manual_translation: false") + lines.append(f"categories: [{', '.join(cats)}]") excerpt = to_markdown(post.get("excerpt", {}).get("rendered", "")) if excerpt: lines.append(f'description: "{excerpt[:300].replace(chr(34), chr(39))}"') @@ -125,10 +132,16 @@ def main() -> None: # sibling's slug as the shared basename so Hugo pairs the set. translations = post.get("translations", {}) es_id = translations.get("es", post["id"]) - basename = by_id.get(es_id, post)["slug"] + origin = by_id.get(es_id, post) + basename = origin["slug"] + + # Anything that isn't the origin of its Polylang group is a translation. + origin_lang = LANG_MAP.get(origin.get("lang", "es")) + derived_from = origin_lang if origin_lang and origin_lang != lang else None body_html = rewrite_images(post["content"]["rendered"], base_url) - md = frontmatter(post, lang, categories) + "\n\n" + to_markdown(body_html) + "\n" + md = (frontmatter(post, lang, categories, derived_from) + + "\n\n" + to_markdown(body_html) + "\n") out = POSTS_DIR / f"{basename}.{lang}.md" out.write_text(md, encoding="utf-8")