From 9990435262f7d74417f84513fd2eef6eb4c91f92 Mon Sep 17 00:00:00 2001 From: Mathieu Agopian Date: Thu, 24 Sep 2026 15:08:20 +0200 Subject: [PATCH] =?UTF-8?q?=E2=9C=A8(backend)=20attach=20media=20bytes=20t?= =?UTF-8?q?o=20DocumentNodes=20during=20zip=20parsing?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit When parsing a zip export, scan each document's markdown content for local media links and store the corresponding file bytes in `node.media`, keyed by the relative reference as written in the markdown. The endpoint (in a future commit) will use this to upload media to S3 and substitute the correct URLs before converting to YJS. Signed-off-by: Mathieu Agopian --- src/backend/core/services/zip_import.py | 41 +++++++++++++++++-- .../core/tests/test_zip_import_service.py | 30 +++++++++++++- 2 files changed, 66 insertions(+), 5 deletions(-) diff --git a/src/backend/core/services/zip_import.py b/src/backend/core/services/zip_import.py index 831e37fdf..ad49798cc 100644 --- a/src/backend/core/services/zip_import.py +++ b/src/backend/core/services/zip_import.py @@ -1,10 +1,24 @@ """Parse ZIP exports into a document tree.""" +import re import zipfile from collections import defaultdict from dataclasses import dataclass, field from pathlib import PurePosixPath +# Captures the URL portion of a markdown image reference: ![alt text](URL "optional title") +_IMAGE_REF_RE = re.compile( + r""" + !\[ # image marker: exclamation mark + opening bracket + [^\]]* # alt text: any characters except closing bracket + \]\( # closing bracket + opening parenthesis + ( # start capture group: the URL + [^) "]+ # URL: any characters except closing paren, space, or quote + ) # end capture group + """, + re.VERBOSE, +) + @dataclass class DocumentNode: @@ -12,12 +26,14 @@ class DocumentNode: title: str content: bytes | None = None + # Media files referenced in content: {relative ref as written in markdown -> raw bytes} + media: dict[str, bytes] = field(default_factory=dict) children: list["DocumentNode"] = field(default_factory=list) def parse_zip(zf: zipfile.ZipFile) -> list[DocumentNode]: """Return root DocumentNodes parsed from a ZIP export.""" - return _build_tree(_read_md_files(zf)) + return _build_tree(_read_md_files(zf), zf) def _read_md_files(zf: zipfile.ZipFile) -> dict[str, bytes]: @@ -30,13 +46,16 @@ def _read_md_files(zf: zipfile.ZipFile) -> dict[str, bytes]: return result -def _build_tree(md_files: dict[str, bytes]) -> list[DocumentNode]: +def _build_tree(md_files: dict[str, bytes], zf: zipfile.ZipFile) -> list[DocumentNode]: """ Build a DocumentNode tree from a flat dict of path -> markdown content. A folder and a .md file with the same name at the same level merge into a single node: the .md provides content, the folder provides children. Folders without a matching .md become container nodes with no content. + + Media files referenced in each node's content are read from the zip and + stored in node.media keyed by the relative reference as written in the markdown. """ paths = {PurePosixPath(k): v for k, v in md_files.items()} @@ -54,12 +73,28 @@ def _build_tree(md_files: dict[str, bytes]) -> list[DocumentNode]: for d in all_dirs: dirs_by_parent[d.parent].add(d.name) + zip_names = set(zf.namelist()) + def build_children(parent: PurePosixPath) -> list[DocumentNode]: files = by_parent.get(parent, {}) subdirs = dirs_by_parent.get(parent, set()) nodes = [] + # Union of .md stems and subdir names: a name present in both means the + # .md file and the folder represent the same document (content + children). for name in sorted(set(files) | subdirs): - node = DocumentNode(title=name, content=files.get(name)) + content = files.get(name) + node = DocumentNode(title=name, content=content) + if content is not None: + for ref in _IMAGE_REF_RE.findall( + content.decode("utf-8", errors="replace") + ): + if ref.startswith(("http://", "https://")): + continue + zip_path = str(parent / ref) + if zip_path in zip_names: + # Store the raw bytes under the original relative reference so + # the caller can upload the file and substitute the URL. + node.media[ref] = zf.read(zip_path) subdir = parent / name if subdir in all_dirs: node.children = build_children(subdir) diff --git a/src/backend/core/tests/test_zip_import_service.py b/src/backend/core/tests/test_zip_import_service.py index 2a0439d4f..a12ba8abe 100644 --- a/src/backend/core/tests/test_zip_import_service.py +++ b/src/backend/core/tests/test_zip_import_service.py @@ -49,7 +49,7 @@ def test_parse_zip_nested_children(): with zipfile.ZipFile(FIXTURES / "outline-export.zip") as zf: roots = parse_zip(zf) - by_title = {c.title: c for c in roots[0].children} + by_title = {child.title: child for child in roots[0].children} getting_started = by_title["Getting Started"] assert len(getting_started.children) == 1 assert getting_started.children[0].title == "rich nested doc" @@ -60,11 +60,37 @@ def test_parse_zip_content_present(): with zipfile.ZipFile(FIXTURES / "outline-export.zip") as zf: roots = parse_zip(zf) - by_title = {c.title: c for c in roots[0].children} + by_title = {child.title: child for child in roots[0].children} assert by_title["Our Editor"].content is not None assert b"editor" in by_title["Our Editor"].content.lower() +def test_parse_zip_media_attached(): + """Image files referenced in a .md are stored in node.media keyed by their relative ref.""" + with zipfile.ZipFile(FIXTURES / "outline-export.zip") as zf: + roots = parse_zip(zf) + + by_title = {child.title: child for child in roots[0].children} + getting_started = by_title["Getting Started"] + rich = getting_started.children[0] + assert rich.title == "rich nested doc" + assert len(rich.media) == 1 + ref, media_bytes = next(iter(rich.media.items())) + assert ref.endswith("harley-benton.jpeg") + assert media_bytes[:3] == b"\xff\xd8\xff" # JPEG magic bytes + + +def test_parse_zip_external_links_not_collected(): + """Absolute http(s) image URLs are not added to node.media.""" + buf = io.BytesIO() + with zipfile.ZipFile(buf, "w") as zf: + zf.writestr("doc.md", b"![](https://example.com/image.png)") + buf.seek(0) + with zipfile.ZipFile(buf) as zf: + nodes = parse_zip(zf) + assert nodes[0].media == {} + + def test_parse_folder_alongside_md_merges(): """A folder and a same-name .md at the same level merge into one node.""" buf = io.BytesIO()