diff --git a/.woodpecker.yml b/.woodpecker.yml index 889037c..eca2f89 100644 --- a/.woodpecker.yml +++ b/.woodpecker.yml @@ -5,10 +5,14 @@ # translation happens here, asynchronously, and every generated sibling lands # as a reviewable bot commit. # -# Translation is self-hosted — the model ships inside vienalatina/translate, -# so there is no API key and no third-party request. Build that image on the -# server before the first run: -# docker build -t vienalatina/translate:1 docker/translate +# Translation is self-hosted — OPUS-MT runs on CPU inside the image, so there +# is no API key and no third-party request. Before the first run, on the server: +# bash scripts/fetch-models.sh # -> /srv/mt-models +# docker build -t vienalatina/translate:2 docker/translate +# +# The models are mounted rather than baked in, so swapping them later does not +# mean rebuilding the image. The repo must be marked "trusted" in Woodpecker +# for this mount, which it already is for the deploy step. # # Secrets to configure in Woodpecker (repo settings → secrets): # gitea_push_token — Gitea token for the translations bot user @@ -23,8 +27,10 @@ when: steps: translate: - image: vienalatina/translate:1 + image: vienalatina/translate:2 pull: false # built locally on the server, never fetched from a registry + volumes: + - /srv/mt-models:/opt/mt/models:ro environment: GITEA_PUSH_TOKEN: from_secret: gitea_push_token diff --git a/docker/translate/Dockerfile b/docker/translate/Dockerfile index 43ef7ee..ccbc5f9 100644 --- a/docker/translate/Dockerfile +++ b/docker/translate/Dockerfile @@ -1,21 +1,20 @@ # Pipeline image for the translate step. # -# Model and tokenizer are baked in, so a publish makes zero network calls and -# needs no API key. torch is deliberately absent: it is only required to -# *convert* a model, and the default PyPI wheel drags in ~2.5GB of CUDA -# libraries this CPU-only box will never use. +# Models are NOT baked in — they live on the host at /srv/mt-models (see +# scripts/fetch-models.sh) and are mounted read-only by .woodpecker.yml. That +# keeps this image small and lets a model change be a directory swap instead of +# a 1GB image rebuild, which matters because model choice turned out to need +# iteration. # -# Build on the server (once, and again only when changing models): -# docker build -t vienalatina/translate:1 docker/translate +# torch is deliberately absent: it is only needed to *convert* a model, and the +# default PyPI wheel drags in ~2.5GB of CUDA libraries this CPU-only box will +# never use. Conversion happens in fetch-models.sh, not here. # -# The default model is a pre-converted CTranslate2 build, which avoids running -# ct2-transformers-converter on a 4GB box — it gets OOM-killed there. +# Build on the server: +# docker build -t vienalatina/translate:2 docker/translate FROM python:3.12-slim -ARG MT_MODEL_REPO=michaelfeil/ct2fast-m2m100_418M -ARG MT_TOKENIZER_REPO=facebook/m2m100_418M - # translate.py shells out to git to diff the push and commit the siblings back. RUN apt-get update -qq \ && apt-get install -qq -y --no-install-recommends git \ @@ -26,17 +25,9 @@ RUN pip install --no-cache-dir \ transformers \ sentencepiece \ sentencex \ - pyyaml \ - huggingface_hub + pyyaml -RUN python -c "from huggingface_hub import snapshot_download; \ -snapshot_download('${MT_MODEL_REPO}', local_dir='/opt/mt/model')" \ - && python -c "import transformers; \ -transformers.AutoTokenizer.from_pretrained('${MT_TOKENIZER_REPO}') \ - .save_pretrained('/opt/mt/tokenizer')" - -# The published artifact is float16; CTranslate2 quantises to int8 on load. -ENV MT_MODEL_DIR=/opt/mt/model \ - MT_TOKENIZER=/opt/mt/tokenizer \ +ENV MT_PROVIDER=opus \ + MT_MODEL_DIR=/opt/mt/models \ MT_COMPUTE_TYPE=int8 \ MT_THREADS=2 diff --git a/scripts/fetch-models.sh b/scripts/fetch-models.sh new file mode 100644 index 0000000..9b0d8d4 --- /dev/null +++ b/scripts/fetch-models.sh @@ -0,0 +1,46 @@ +#!/usr/bin/env bash +# Download and convert the OPUS-MT models the pipeline needs. +# +# Run once on the server; the pipeline mounts the result read-only, so changing +# models later means re-running this rather than rebuilding a 1GB image. +# +# bash scripts/fetch-models.sh # -> /srv/mt-models +# bash scripts/fetch-models.sh /tmp/mt # -> somewhere else +# +# Conversion holds the fp32 model in RAM. These are 74M-237M parameter models, +# so each peaks well under the 4GB box's headroom — unlike M2M100 418M, whose +# conversion was OOM-killed here at a 3GB cap. One container per model keeps a +# failure contained to that model. + +set -euo pipefail +DEST="${1:-/srv/mt-models}" + +MODELS=( + "Helsinki-NLP/opus-mt-es-de" + "Helsinki-NLP/opus-mt-tc-big-de-es" + "Helsinki-NLP/opus-mt-tc-big-itc-itc" +) + +sudo mkdir -p "$DEST" +sudo chown "$(id -u):$(id -g)" "$DEST" + +for REPO in "${MODELS[@]}"; do + NAME="${REPO##*/}" + if [ -f "$DEST/$NAME/model.bin" ]; then + echo "== $NAME already converted, skipping" + continue + fi + echo "== converting $NAME" + docker run --rm -u "$(id -u):$(id -g)" -e HOME=/tmp --memory=3g \ + -v "$DEST:/out" python:3.12-slim bash -c " + set -e + pip install --quiet 'transformers[torch]' ctranslate2 sentencepiece \ + --extra-index-url https://download.pytorch.org/whl/cpu + ct2-transformers-converter --model $REPO --output_dir /out/$NAME \ + --quantization int8 --copy_files source.spm target.spm vocab.json tokenizer_config.json + " +done + +echo +du -sh "$DEST"/* +echo "Models ready in $DEST" diff --git a/scripts/translate.py b/scripts/translate.py index 1b684b7..76d7a24 100644 --- a/scripts/translate.py +++ b/scripts/translate.py @@ -37,10 +37,18 @@ from pathlib import Path import yaml -from translation.ctranslate_provider import CTranslate2Provider from translation.markdown import translate_markdown, translate_text from translation.provider import SITE_LANGS + +def build_provider(): + """MT_PROVIDER=m2m100 falls back to the single-model engine.""" + if os.environ.get("MT_PROVIDER", "opus") == "m2m100": + from translation.ctranslate_provider import CTranslate2Provider + return CTranslate2Provider() + from translation.opus_provider import OpusMTProvider + return OpusMTProvider() + REPO_ROOT = Path(__file__).resolve().parent.parent CONTENT_DIR = REPO_ROOT / "content" @@ -225,7 +233,7 @@ def main() -> None: print("No authored content changed — nothing to translate.") return - provider = CTranslate2Provider() + provider = build_provider() written: list[Path] = [] for source, basename, lang in sources: print(f"Translating {source.relative_to(REPO_ROOT)} (from {lang}):") diff --git a/scripts/translation/opus_provider.py b/scripts/translation/opus_provider.py new file mode 100644 index 0000000..0b5643b --- /dev/null +++ b/scripts/translation/opus_provider.py @@ -0,0 +1,83 @@ +"""OPUS-MT (Helsinki-NLP) via CTranslate2, on CPU. + +Replaces M2M100 418M, which was measured producing unusable German on real +articles — "frijoles" became "Beeren" (berries), "rompe la idea" became +"breitet die Idee" (spreads it), and the site's own name came back mangled. + +Not every pair exists as a published model, so routes are explicit rather than +assumed. Verified against Hugging Face: + + es -> de opus-mt-es-de (small, ~74M) + de -> es opus-mt-tc-big-de-es (tc-big, ~237M) + es <-> pt-br opus-mt-tc-big-itc-itc (Italic multilingual) + de <-> pt-br no direct model — pivots through Spanish + +Marian models take the target language as a `>>xxx<<` token at the start of +the *source* text, unlike M2M100's decoder-side target_prefix. +""" + +from __future__ import annotations + +import os + +from .provider import Provider + +# (source, target) -> [(model directory, target-language token or None), ...] +# More than one hop means a pivot translation. +ROUTES: dict[tuple[str, str], list[tuple[str, str | None]]] = { + ("es", "de"): [("opus-mt-es-de", None)], + ("de", "es"): [("opus-mt-tc-big-de-es", None)], + ("es", "pt-br"): [("opus-mt-tc-big-itc-itc", ">>pob<<")], + ("pt-br", "es"): [("opus-mt-tc-big-itc-itc", ">>spa<<")], + ("de", "pt-br"): [("opus-mt-tc-big-de-es", None), + ("opus-mt-tc-big-itc-itc", ">>pob<<")], + ("pt-br", "de"): [("opus-mt-tc-big-itc-itc", ">>spa<<"), + ("opus-mt-es-de", None)], +} + + +class OpusMTProvider(Provider): + def __init__(self, model_root: str | None = None, compute_type: str | None = None, + threads: int | None = None): + self.model_root = model_root or os.environ.get("MT_MODEL_DIR", "/opt/mt/models") + self.compute_type = compute_type or os.environ.get("MT_COMPUTE_TYPE", "int8") + self.threads = threads or int(os.environ.get("MT_THREADS", "2")) + self._name: str | None = None + self._translator = None + self._tokenizer = None + + def _load(self, name: str): + """Keep exactly one model resident — three at once would not fit a 4GB box.""" + if self._name == name: + return self._translator, self._tokenizer + import ctranslate2 + import transformers + + self._translator = None # free the previous model before allocating the next + path = os.path.join(self.model_root, name) + self._tokenizer = transformers.AutoTokenizer.from_pretrained(path) + self._translator = ctranslate2.Translator( + path, device="cpu", compute_type=self.compute_type, intra_threads=self.threads + ) + self._name = name + return self._translator, self._tokenizer + + def _hop(self, texts: list[str], model: str, token: str | None) -> list[str]: + translator, tok = self._load(model) + prepared = [f"{token} {t}" if token else t for t in texts] + batch = [tok.convert_ids_to_tokens(tok.encode(t)) for t in prepared] + results = translator.translate_batch(batch) + return [ + tok.decode(tok.convert_tokens_to_ids(r.hypotheses[0]), skip_special_tokens=True) + for r in results + ] + + def translate(self, texts: list[str], src: str, tgt: str) -> list[str]: + if not texts: + return [] + route = ROUTES.get((src, tgt)) + if route is None: + raise ValueError(f"no translation route from {src} to {tgt}") + for model, token in route: + texts = self._hop(texts, model, token) + return texts