Phase B. Images on threads and comments, stored in /data/uploads — inside the volume that already holds board.db, so there is one directory to back up rather than two and no second mount to remember. Not in the site repository, which is the distinction that matters: the content editor uploads by committing, and a private photo committed there would go through the build pipeline and out onto vienalatina.com. The type is decided by the first bytes, not the filename. content.py trusts the extension, which is tolerable where Hugo serves the result; here we serve it back, so an HTML file called gato.png would be a script running on our own origin. Five magic-number checks, no new dependency. The uploaded name is kept only as text to show a person; the name on disk is generated. Staging is separate from saving so a refused picture cannot leave a half-made thread behind. Threads and comments are soft-deleted, so the serving route checks the parent is still live. Without it, taking a post down leaves its photo readable by anyone who noted the URL. And the hole this phase existed to close: backup-board.sh archived board.db and nothing else, so the first upload would have made the nightly backup silently incomplete while still reporting success. It now archives the uploads directory too, verifies the archive, and names the directory when it is empty rather than passing over it in silence. Tested by restoring: destroyed the directory, restored from the archive, compared bytes. IMAGE_EXTENSIONS moves to uploads.py and content.py imports it, so the editor and the board cannot drift apart about what counts as a picture. 46 new tests, 176 in total. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01NizVpJ2dwzCbjCrTLCjeHn
215 lines
7.9 KiB
Python
215 lines
7.9 KiB
Python
"""Pictures on the board.
|
|
|
|
Three decisions worth stating, because each one is a place this could have
|
|
gone wrong quietly.
|
|
|
|
**Not in the site repository.** `content.py` uploads pictures by committing
|
|
them, which is right for a post that is about to be published. A photo attached
|
|
to a thread in here is the opposite: it is private, and committing it would put
|
|
it through the build pipeline and out onto vienalatina.com. These go to a
|
|
directory on disk that only this app serves, behind the same login as the
|
|
thread they belong to.
|
|
|
|
**The file says what it is; the filename is an opinion.** `content.py` trusts
|
|
the extension, which is tolerable there because Hugo serves the result as a
|
|
static file. Here *we* serve it back, so the type is read from the first bytes
|
|
and the name the browser sent is never used for anything but display. A file
|
|
called `gato.png` that is really HTML is refused, rather than stored and later
|
|
handed back with a content type that invites the browser to run it.
|
|
|
|
**The directory lives inside the volume that gets backed up.** `/data/uploads`
|
|
sits next to `board.db` in the same mount, so there is one thing to back up,
|
|
not two — and `scripts/backup-board.sh` covers both.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
import secrets
|
|
from pathlib import Path
|
|
|
|
from flask import Blueprint, abort, current_app, send_from_directory
|
|
|
|
from .db import get_db
|
|
from .security import login_required
|
|
|
|
bp = Blueprint("uploads", __name__)
|
|
|
|
# How many pictures one post or comment may carry. Not a security limit —
|
|
# MAX_CONTENT_LENGTH is — just a bound on what one form submission can become.
|
|
MAX_FILES = 4
|
|
|
|
# extension -> content type. The extension is ours, derived from the bytes,
|
|
# never taken from the upload.
|
|
CONTENT_TYPES = {
|
|
"jpg": "image/jpeg",
|
|
"png": "image/png",
|
|
"gif": "image/gif",
|
|
"webp": "image/webp",
|
|
"avif": "image/avif",
|
|
}
|
|
|
|
# What content.py offers writers, kept here so there is one list rather than
|
|
# two that drift apart. `jpeg` appears only as an accepted spelling; anything
|
|
# stored is named `.jpg`.
|
|
IMAGE_EXTENSIONS = set(CONTENT_TYPES) | {"jpeg"}
|
|
|
|
# Stored names are generated by this module, so the pattern is a check on our
|
|
# own output — which is exactly why it is worth having. It is the last thing
|
|
# between a crafted request and send_from_directory.
|
|
STORED_NAME = re.compile(r"^[a-z0-9]+(?:-[a-z0-9]+)*-[0-9a-f]{12}\.[a-z]{3,4}$")
|
|
|
|
|
|
class RejectedUpload(ValueError):
|
|
"""The message is written for the member who chose the file."""
|
|
|
|
|
|
def _detect(data: bytes) -> str | None:
|
|
"""The extension these bytes actually deserve, or None.
|
|
|
|
Magic numbers rather than a library: five formats, each identified by a
|
|
fixed prefix, is less code than a dependency and has no version to track.
|
|
"""
|
|
if data.startswith(b"\xff\xd8\xff"):
|
|
return "jpg"
|
|
if data.startswith(b"\x89PNG\r\n\x1a\n"):
|
|
return "png"
|
|
if data.startswith((b"GIF87a", b"GIF89a")):
|
|
return "gif"
|
|
# Both of these are container formats: the marker sits at a fixed offset
|
|
# rather than at the very start, so a prefix check would miss them.
|
|
if data[:4] == b"RIFF" and data[8:12] == b"WEBP":
|
|
return "webp"
|
|
if data[4:8] == b"ftyp" and data[8:12] in (b"avif", b"avis"):
|
|
return "avif"
|
|
return None
|
|
|
|
|
|
def slugify(text: str) -> str:
|
|
"""Only ever applied to a display name to build a readable stored name.
|
|
|
|
Falls back to "imagen" rather than the empty string: a name that is all
|
|
punctuation, or written in a script this strips entirely, must still
|
|
produce something that matches STORED_NAME.
|
|
"""
|
|
slug = re.sub(r"[^a-z0-9]+", "-", text.lower()).strip("-")
|
|
return slug[:48] or "imagen"
|
|
|
|
|
|
def directory() -> Path:
|
|
path = Path(current_app.config["UPLOAD_DIR"])
|
|
path.mkdir(parents=True, exist_ok=True)
|
|
return path
|
|
|
|
|
|
def stage(files) -> list[dict]:
|
|
"""Read and check everything before anything is written or inserted.
|
|
|
|
Separate from `save` on purpose. A picture that is refused must not leave a
|
|
half-made thread behind, so nothing touches the database until every file
|
|
in the submission has passed.
|
|
"""
|
|
staged = []
|
|
real = [f for f in files if f and f.filename]
|
|
if len(real) > MAX_FILES:
|
|
raise RejectedUpload(f"Como máximo {MAX_FILES} imágenes por mensaje.")
|
|
|
|
for upload in real:
|
|
data = upload.read()
|
|
if not data:
|
|
continue
|
|
maximum = current_app.config["UPLOAD_MAX_BYTES"]
|
|
if len(data) > maximum:
|
|
raise RejectedUpload(
|
|
f"«{upload.filename}» pesa {len(data) // 1024}KB y el máximo "
|
|
f"es {maximum // 1024}KB."
|
|
)
|
|
extension = _detect(data)
|
|
if extension is None:
|
|
raise RejectedUpload(
|
|
f"«{upload.filename}» no parece una imagen. Se aceptan "
|
|
f"{', '.join(sorted(CONTENT_TYPES))}."
|
|
)
|
|
stem = slugify(upload.filename.rsplit(".", 1)[0])
|
|
staged.append({
|
|
"data": data,
|
|
"original_name": upload.filename[:200],
|
|
"content_type": CONTENT_TYPES[extension],
|
|
# A random suffix rather than a counter: two people uploading
|
|
# "foto.jpg" in the same second must not race for one path.
|
|
"stored_name": f"{stem}-{secrets.token_hex(6)}.{extension}",
|
|
})
|
|
return staged
|
|
|
|
|
|
def save(staged: list[dict], member_id: int,
|
|
thread_id: int | None = None, comment_id: int | None = None) -> None:
|
|
"""Write the files, then record them. In that order.
|
|
|
|
A row pointing at a file that does not exist renders as a broken image on
|
|
every future visit. A file with no row is invisible and gets swept up by
|
|
the next audit — so if one of the two has to happen first, it is the file.
|
|
"""
|
|
folder = directory()
|
|
for item in staged:
|
|
(folder / item["stored_name"]).write_bytes(item["data"])
|
|
get_db().execute(
|
|
"""INSERT INTO attachments
|
|
(thread_id, comment_id, stored_name, original_name,
|
|
content_type, bytes, uploaded_by)
|
|
VALUES (?, ?, ?, ?, ?, ?, ?)""",
|
|
(thread_id, comment_id, item["stored_name"], item["original_name"],
|
|
item["content_type"], len(item["data"]), member_id),
|
|
)
|
|
|
|
|
|
def for_threads(thread_ids: list[int]) -> dict[int, list]:
|
|
return _grouped("thread_id", thread_ids)
|
|
|
|
|
|
def for_comments(comment_ids: list[int]) -> dict[int, list]:
|
|
return _grouped("comment_id", comment_ids)
|
|
|
|
|
|
def _grouped(column: str, ids: list[int]) -> dict[int, list]:
|
|
"""One query for a whole page rather than one per comment."""
|
|
if not ids:
|
|
return {}
|
|
marks = ",".join("?" * len(ids))
|
|
rows = get_db().execute(
|
|
f"""SELECT * FROM attachments WHERE {column} IN ({marks})
|
|
ORDER BY id""",
|
|
ids,
|
|
).fetchall()
|
|
grouped: dict[int, list] = {}
|
|
for row in rows:
|
|
grouped.setdefault(row[column], []).append(row)
|
|
return grouped
|
|
|
|
|
|
@bp.route("/media/<name>")
|
|
@login_required
|
|
def serve(name: str):
|
|
"""Behind the login, like the thread the picture belongs to.
|
|
|
|
The deleted check is the part that is easy to leave out: threads and
|
|
comments are *soft*-deleted, so without it, taking a post down would leave
|
|
its photo readable forever by anyone who noted the URL.
|
|
"""
|
|
if not STORED_NAME.match(name):
|
|
abort(404)
|
|
row = get_db().execute(
|
|
"""SELECT a.content_type
|
|
FROM attachments a
|
|
LEFT JOIN threads t ON t.id = a.thread_id
|
|
LEFT JOIN comments c ON c.id = a.comment_id
|
|
WHERE a.stored_name = ?
|
|
AND COALESCE(t.deleted_at, c.deleted_at) IS NULL""",
|
|
(name,),
|
|
).fetchone()
|
|
if row is None:
|
|
abort(404)
|
|
# mimetype from our own column, never guessed from the name on disk, and
|
|
# paired with the X-Content-Type-Options: nosniff set in app.py.
|
|
return send_from_directory(directory(), name, mimetype=row["content_type"])
|