✨(backend) attach media bytes to DocumentNodes during zip parsing

When parsing a zip export, scan each document's markdown content for
local media links and store the corresponding file bytes in
`node.media`, keyed by the relative reference as written in the
markdown.
The endpoint (in a future commit) will use this to upload media to S3
and substitute the correct URLs before converting to YJS.

Signed-off-by: Mathieu Agopian <mathieu@agopian.info>
This commit is contained in:
Mathieu Agopian
2026-09-24 18:38:26 +02:00
parent dd7c7dab10
commit 9990435262
2 changed files with 66 additions and 5 deletions
+38 -3
View File
@@ -1,10 +1,24 @@
"""Parse ZIP exports into a document tree."""
import re
import zipfile
from collections import defaultdict
from dataclasses import dataclass, field
from pathlib import PurePosixPath
# Captures the URL portion of a markdown image reference: ![alt text](URL "optional title")
_IMAGE_REF_RE = re.compile(
r"""
!\[ # image marker: exclamation mark + opening bracket
[^\]]* # alt text: any characters except closing bracket
\]\( # closing bracket + opening parenthesis
( # start capture group: the URL
[^) "]+ # URL: any characters except closing paren, space, or quote
) # end capture group
""",
re.VERBOSE,
)
@dataclass
class DocumentNode:
@@ -12,12 +26,14 @@ class DocumentNode:
title: str
content: bytes | None = None
# Media files referenced in content: {relative ref as written in markdown -> raw bytes}
media: dict[str, bytes] = field(default_factory=dict)
children: list["DocumentNode"] = field(default_factory=list)
def parse_zip(zf: zipfile.ZipFile) -> list[DocumentNode]:
"""Return root DocumentNodes parsed from a ZIP export."""
return _build_tree(_read_md_files(zf))
return _build_tree(_read_md_files(zf), zf)
def _read_md_files(zf: zipfile.ZipFile) -> dict[str, bytes]:
@@ -30,13 +46,16 @@ def _read_md_files(zf: zipfile.ZipFile) -> dict[str, bytes]:
return result
def _build_tree(md_files: dict[str, bytes]) -> list[DocumentNode]:
def _build_tree(md_files: dict[str, bytes], zf: zipfile.ZipFile) -> list[DocumentNode]:
"""
Build a DocumentNode tree from a flat dict of path -> markdown content.
A folder and a .md file with the same name at the same level merge into a
single node: the .md provides content, the folder provides children.
Folders without a matching .md become container nodes with no content.
Media files referenced in each node's content are read from the zip and
stored in node.media keyed by the relative reference as written in the markdown.
"""
paths = {PurePosixPath(k): v for k, v in md_files.items()}
@@ -54,12 +73,28 @@ def _build_tree(md_files: dict[str, bytes]) -> list[DocumentNode]:
for d in all_dirs:
dirs_by_parent[d.parent].add(d.name)
zip_names = set(zf.namelist())
def build_children(parent: PurePosixPath) -> list[DocumentNode]:
files = by_parent.get(parent, {})
subdirs = dirs_by_parent.get(parent, set())
nodes = []
# Union of .md stems and subdir names: a name present in both means the
# .md file and the folder represent the same document (content + children).
for name in sorted(set(files) | subdirs):
node = DocumentNode(title=name, content=files.get(name))
content = files.get(name)
node = DocumentNode(title=name, content=content)
if content is not None:
for ref in _IMAGE_REF_RE.findall(
content.decode("utf-8", errors="replace")
):
if ref.startswith(("http://", "https://")):
continue
zip_path = str(parent / ref)
if zip_path in zip_names:
# Store the raw bytes under the original relative reference so
# the caller can upload the file and substitute the URL.
node.media[ref] = zf.read(zip_path)
subdir = parent / name
if subdir in all_dirs:
node.children = build_children(subdir)
@@ -49,7 +49,7 @@ def test_parse_zip_nested_children():
with zipfile.ZipFile(FIXTURES / "outline-export.zip") as zf:
roots = parse_zip(zf)
by_title = {c.title: c for c in roots[0].children}
by_title = {child.title: child for child in roots[0].children}
getting_started = by_title["Getting Started"]
assert len(getting_started.children) == 1
assert getting_started.children[0].title == "rich nested doc"
@@ -60,11 +60,37 @@ def test_parse_zip_content_present():
with zipfile.ZipFile(FIXTURES / "outline-export.zip") as zf:
roots = parse_zip(zf)
by_title = {c.title: c for c in roots[0].children}
by_title = {child.title: child for child in roots[0].children}
assert by_title["Our Editor"].content is not None
assert b"editor" in by_title["Our Editor"].content.lower()
def test_parse_zip_media_attached():
"""Image files referenced in a .md are stored in node.media keyed by their relative ref."""
with zipfile.ZipFile(FIXTURES / "outline-export.zip") as zf:
roots = parse_zip(zf)
by_title = {child.title: child for child in roots[0].children}
getting_started = by_title["Getting Started"]
rich = getting_started.children[0]
assert rich.title == "rich nested doc"
assert len(rich.media) == 1
ref, media_bytes = next(iter(rich.media.items()))
assert ref.endswith("harley-benton.jpeg")
assert media_bytes[:3] == b"\xff\xd8\xff" # JPEG magic bytes
def test_parse_zip_external_links_not_collected():
"""Absolute http(s) image URLs are not added to node.media."""
buf = io.BytesIO()
with zipfile.ZipFile(buf, "w") as zf:
zf.writestr("doc.md", b"![](https://example.com/image.png)")
buf.seek(0)
with zipfile.ZipFile(buf) as zf:
nodes = parse_zip(zf)
assert nodes[0].media == {}
def test_parse_folder_alongside_md_merges():
"""A folder and a same-name .md at the same level merge into one node."""
buf = io.BytesIO()