Merge branch 'claude/relaxed-faraday-h4zd09' of https://github.com/pablovolenski/vienalatina

This commit is contained in:
admin 2026-09-18 12:43:26 +00:00
commit 5898f96edb
5 changed files with 163 additions and 29 deletions

View File

@ -5,10 +5,14 @@
# translation happens here, asynchronously, and every generated sibling lands
# as a reviewable bot commit.
#
# Translation is self-hosted — the model ships inside vienalatina/translate,
# so there is no API key and no third-party request. Build that image on the
# server before the first run:
# docker build -t vienalatina/translate:1 docker/translate
# Translation is self-hosted — OPUS-MT runs on CPU inside the image, so there
# is no API key and no third-party request. Before the first run, on the server:
# bash scripts/fetch-models.sh # -> /srv/mt-models
# docker build -t vienalatina/translate:2 docker/translate
#
# The models are mounted rather than baked in, so swapping them later does not
# mean rebuilding the image. The repo must be marked "trusted" in Woodpecker
# for this mount, which it already is for the deploy step.
#
# Secrets to configure in Woodpecker (repo settings → secrets):
# gitea_push_token — Gitea token for the translations bot user
@ -23,8 +27,10 @@ when:
steps:
translate:
image: vienalatina/translate:1
image: vienalatina/translate:2
pull: false # built locally on the server, never fetched from a registry
volumes:
- /srv/mt-models:/opt/mt/models:ro
environment:
GITEA_PUSH_TOKEN:
from_secret: gitea_push_token

View File

@ -1,21 +1,20 @@
# Pipeline image for the translate step.
#
# Model and tokenizer are baked in, so a publish makes zero network calls and
# needs no API key. torch is deliberately absent: it is only required to
# *convert* a model, and the default PyPI wheel drags in ~2.5GB of CUDA
# libraries this CPU-only box will never use.
# Models are NOT baked in — they live on the host at /srv/mt-models (see
# scripts/fetch-models.sh) and are mounted read-only by .woodpecker.yml. That
# keeps this image small and lets a model change be a directory swap instead of
# a 1GB image rebuild, which matters because model choice turned out to need
# iteration.
#
# Build on the server (once, and again only when changing models):
# docker build -t vienalatina/translate:1 docker/translate
# torch is deliberately absent: it is only needed to *convert* a model, and the
# default PyPI wheel drags in ~2.5GB of CUDA libraries this CPU-only box will
# never use. Conversion happens in fetch-models.sh, not here.
#
# The default model is a pre-converted CTranslate2 build, which avoids running
# ct2-transformers-converter on a 4GB box — it gets OOM-killed there.
# Build on the server:
# docker build -t vienalatina/translate:2 docker/translate
FROM python:3.12-slim
ARG MT_MODEL_REPO=michaelfeil/ct2fast-m2m100_418M
ARG MT_TOKENIZER_REPO=facebook/m2m100_418M
# translate.py shells out to git to diff the push and commit the siblings back.
RUN apt-get update -qq \
&& apt-get install -qq -y --no-install-recommends git \
@ -26,17 +25,9 @@ RUN pip install --no-cache-dir \
transformers \
sentencepiece \
sentencex \
pyyaml \
huggingface_hub
pyyaml
RUN python -c "from huggingface_hub import snapshot_download; \
snapshot_download('${MT_MODEL_REPO}', local_dir='/opt/mt/model')" \
&& python -c "import transformers; \
transformers.AutoTokenizer.from_pretrained('${MT_TOKENIZER_REPO}') \
.save_pretrained('/opt/mt/tokenizer')"
# The published artifact is float16; CTranslate2 quantises to int8 on load.
ENV MT_MODEL_DIR=/opt/mt/model \
MT_TOKENIZER=/opt/mt/tokenizer \
ENV MT_PROVIDER=opus \
MT_MODEL_DIR=/opt/mt/models \
MT_COMPUTE_TYPE=int8 \
MT_THREADS=2

46
scripts/fetch-models.sh Normal file
View File

@ -0,0 +1,46 @@
#!/usr/bin/env bash
# Download and convert the OPUS-MT models the pipeline needs.
#
# Run once on the server; the pipeline mounts the result read-only, so changing
# models later means re-running this rather than rebuilding a 1GB image.
#
# bash scripts/fetch-models.sh # -> /srv/mt-models
# bash scripts/fetch-models.sh /tmp/mt # -> somewhere else
#
# Conversion holds the fp32 model in RAM. These are 74M-237M parameter models,
# so each peaks well under the 4GB box's headroom — unlike M2M100 418M, whose
# conversion was OOM-killed here at a 3GB cap. One container per model keeps a
# failure contained to that model.
set -euo pipefail
DEST="${1:-/srv/mt-models}"
MODELS=(
"Helsinki-NLP/opus-mt-es-de"
"Helsinki-NLP/opus-mt-tc-big-de-es"
"Helsinki-NLP/opus-mt-tc-big-itc-itc"
)
sudo mkdir -p "$DEST"
sudo chown "$(id -u):$(id -g)" "$DEST"
for REPO in "${MODELS[@]}"; do
NAME="${REPO##*/}"
if [ -f "$DEST/$NAME/model.bin" ]; then
echo "== $NAME already converted, skipping"
continue
fi
echo "== converting $NAME"
docker run --rm -u "$(id -u):$(id -g)" -e HOME=/tmp --memory=3g \
-v "$DEST:/out" python:3.12-slim bash -c "
set -e
pip install --quiet 'transformers[torch]' ctranslate2 sentencepiece \
--extra-index-url https://download.pytorch.org/whl/cpu
ct2-transformers-converter --model $REPO --output_dir /out/$NAME \
--quantization int8 --copy_files source.spm target.spm vocab.json tokenizer_config.json
"
done
echo
du -sh "$DEST"/*
echo "Models ready in $DEST"

View File

@ -37,10 +37,18 @@ from pathlib import Path
import yaml
from translation.ctranslate_provider import CTranslate2Provider
from translation.markdown import translate_markdown, translate_text
from translation.provider import SITE_LANGS
def build_provider():
"""MT_PROVIDER=m2m100 falls back to the single-model engine."""
if os.environ.get("MT_PROVIDER", "opus") == "m2m100":
from translation.ctranslate_provider import CTranslate2Provider
return CTranslate2Provider()
from translation.opus_provider import OpusMTProvider
return OpusMTProvider()
REPO_ROOT = Path(__file__).resolve().parent.parent
CONTENT_DIR = REPO_ROOT / "content"
@ -225,7 +233,7 @@ def main() -> None:
print("No authored content changed — nothing to translate.")
return
provider = CTranslate2Provider()
provider = build_provider()
written: list[Path] = []
for source, basename, lang in sources:
print(f"Translating {source.relative_to(REPO_ROOT)} (from {lang}):")

View File

@ -0,0 +1,83 @@
"""OPUS-MT (Helsinki-NLP) via CTranslate2, on CPU.
Replaces M2M100 418M, which was measured producing unusable German on real
articles — "frijoles" became "Beeren" (berries), "rompe la idea" became
"breitet die Idee" (spreads it), and the site's own name came back mangled.
Not every pair exists as a published model, so routes are explicit rather than
assumed. Verified against Hugging Face:
es -> de opus-mt-es-de (small, ~74M)
de -> es opus-mt-tc-big-de-es (tc-big, ~237M)
es <-> pt-br opus-mt-tc-big-itc-itc (Italic multilingual)
de <-> pt-br no direct model — pivots through Spanish
Marian models take the target language as a `>>xxx<<` token at the start of
the *source* text, unlike M2M100's decoder-side target_prefix.
"""
from __future__ import annotations
import os
from .provider import Provider
# (source, target) -> [(model directory, target-language token or None), ...]
# More than one hop means a pivot translation.
ROUTES: dict[tuple[str, str], list[tuple[str, str | None]]] = {
("es", "de"): [("opus-mt-es-de", None)],
("de", "es"): [("opus-mt-tc-big-de-es", None)],
("es", "pt-br"): [("opus-mt-tc-big-itc-itc", ">>pob<<")],
("pt-br", "es"): [("opus-mt-tc-big-itc-itc", ">>spa<<")],
("de", "pt-br"): [("opus-mt-tc-big-de-es", None),
("opus-mt-tc-big-itc-itc", ">>pob<<")],
("pt-br", "de"): [("opus-mt-tc-big-itc-itc", ">>spa<<"),
("opus-mt-es-de", None)],
}
class OpusMTProvider(Provider):
def __init__(self, model_root: str | None = None, compute_type: str | None = None,
threads: int | None = None):
self.model_root = model_root or os.environ.get("MT_MODEL_DIR", "/opt/mt/models")
self.compute_type = compute_type or os.environ.get("MT_COMPUTE_TYPE", "int8")
self.threads = threads or int(os.environ.get("MT_THREADS", "2"))
self._name: str | None = None
self._translator = None
self._tokenizer = None
def _load(self, name: str):
"""Keep exactly one model resident — three at once would not fit a 4GB box."""
if self._name == name:
return self._translator, self._tokenizer
import ctranslate2
import transformers
self._translator = None # free the previous model before allocating the next
path = os.path.join(self.model_root, name)
self._tokenizer = transformers.AutoTokenizer.from_pretrained(path)
self._translator = ctranslate2.Translator(
path, device="cpu", compute_type=self.compute_type, intra_threads=self.threads
)
self._name = name
return self._translator, self._tokenizer
def _hop(self, texts: list[str], model: str, token: str | None) -> list[str]:
translator, tok = self._load(model)
prepared = [f"{token} {t}" if token else t for t in texts]
batch = [tok.convert_ids_to_tokens(tok.encode(t)) for t in prepared]
results = translator.translate_batch(batch)
return [
tok.decode(tok.convert_tokens_to_ids(r.hypotheses[0]), skip_special_tokens=True)
for r in results
]
def translate(self, texts: list[str], src: str, tgt: str) -> list[str]:
if not texts:
return []
route = ROUTES.get((src, tgt))
if route is None:
raise ValueError(f"no translation route from {src} to {tgt}")
for model, token in route:
texts = self._hop(texts, model, token)
return texts