✨(backend) import csv files in zip imports as tables in markdown

Parse .csv files found in a zip export as DocumentNodes whose content is
a Markdown table.
CSV links in .md files ([text](file.csv)) are rewritten to /docs/{id} of
the corresponding imported document.
URL-encoded link targets are decoded before resolution.

Directories that contain only CSV files (no .md) are now included in the
document tree (previously only directories with .md descendants were
discovered).

Signed-off-by: Mathieu Agopian <mathieu@agopian.info>
This commit is contained in:
Mathieu Agopian
2026-09-24 18:38:27 +02:00
parent 9990435262
commit fdd7961043
3 changed files with 211 additions and 21 deletions
+129 -18
View File
@@ -1,10 +1,14 @@
"""Parse ZIP exports into a document tree."""
import csv
import io
import re
import uuid
import zipfile
from collections import defaultdict
from dataclasses import dataclass, field
from pathlib import PurePosixPath
from urllib.parse import unquote
# Captures the URL portion of a markdown image reference: ![alt text](URL "optional title")
_IMAGE_REF_RE = re.compile(
@@ -19,12 +23,28 @@ _IMAGE_REF_RE = re.compile(
re.VERBOSE,
)
# Captures a markdown link whose target is a .csv file: [text](path/to/file.csv)
_CSV_LINK_RE = re.compile(
r"""
\[ # opening bracket
([^\]]*) # link text: anything except closing bracket
\]\( # closing bracket + opening parenthesis
( # start capture group: the path
.*? # non-greedy: allows literal () in path segments (e.g. "Income (Monthly)")
\.csv # must end with .csv extension
) # end capture group
\) # closing parenthesis
""",
re.VERBOSE | re.IGNORECASE,
)
@dataclass
class DocumentNode:
"""A document node in the import tree."""
title: str
id: uuid.UUID = field(default_factory=uuid.uuid4)
content: bytes | None = None
# Media files referenced in content: {relative ref as written in markdown -> raw bytes}
media: dict[str, bytes] = field(default_factory=dict)
@@ -33,7 +53,7 @@ class DocumentNode:
def parse_zip(zf: zipfile.ZipFile) -> list[DocumentNode]:
"""Return root DocumentNodes parsed from a ZIP export."""
return _build_tree(_read_md_files(zf), zf)
return _build_tree(_read_md_files(zf), _read_csv_files(zf), zf)
def _read_md_files(zf: zipfile.ZipFile) -> dict[str, bytes]:
@@ -46,27 +66,75 @@ def _read_md_files(zf: zipfile.ZipFile) -> dict[str, bytes]:
return result
def _build_tree(md_files: dict[str, bytes], zf: zipfile.ZipFile) -> list[DocumentNode]:
"""
Build a DocumentNode tree from a flat dict of path -> markdown content.
def _read_csv_files(zf: zipfile.ZipFile) -> dict[str, bytes]:
"""Read .csv files from the zip."""
result = {}
for info in zf.infolist():
if not info.filename.lower().endswith(".csv"):
continue
result[info.filename] = zf.read(info.filename)
return result
A folder and a .md file with the same name at the same level merge into a
single node: the .md provides content, the folder provides children.
Folders without a matching .md become container nodes with no content.
def _csv_to_markdown(data: bytes) -> bytes:
"""Convert CSV bytes to a markdown table."""
# "utf-8-sig" is a UTF-8 variant that automatically strips the BOM
# (\xef\xbb\xbf) that Excel and other Windows tools prepend to CSV files.
# errors="replace" substitutes any undecodable byte with the Unicode
# replacement character (U+FFFD) rather than raising an exception.
text = data.decode("utf-8-sig", errors="replace")
rows = [
row
for row in csv.reader(io.StringIO(text))
if any(cell.strip() for cell in row)
]
if not rows:
return b""
def escape(cell: str) -> str:
# Pipe characters break markdown table syntax; newlines collapse to a space.
return cell.replace("|", "\\|").replace("\n", " ")
header, *body = rows
col_count = len(header)
lines = [
"| " + " | ".join(escape(heading) for heading in header) + " |",
"| " + " | ".join("---" for _ in header) + " |",
]
for row in body:
padded = row[:col_count] + [""] * max(0, col_count - len(row))
lines.append("| " + " | ".join(escape(cell) for cell in padded) + " |")
return "\n".join(lines).encode("utf-8")
def _build_tree(
md_files: dict[str, bytes],
csv_files: dict[str, bytes],
zf: zipfile.ZipFile,
) -> list[DocumentNode]:
"""
Build a DocumentNode tree from .md and .csv files found in the zip.
.md files become content nodes; .csv files become markdown-table nodes.
A folder and a same-name .md (or .csv) at the same level merge into one
node: the file provides content, the folder provides children.
Folders with no matching file become container nodes with no content.
Media files referenced in each node's content are read from the zip and
stored in node.media keyed by the relative reference as written in the markdown.
CSV links ([text](file.csv)) are rewritten to /docs/{node.id} of the target node.
"""
paths = {PurePosixPath(k): v for k, v in md_files.items()}
md_paths = {PurePosixPath(k): v for k, v in md_files.items()}
csv_paths = [PurePosixPath(k) for k in csv_files]
all_dirs: set[PurePosixPath] = set()
for path in paths:
for path in [*md_paths, *csv_paths]:
for ancestor in path.parents:
if str(ancestor) != ".":
all_dirs.add(ancestor)
by_parent: dict[PurePosixPath, dict[str, bytes]] = defaultdict(dict)
for path, content in paths.items():
for path, content in md_paths.items():
by_parent[path.parent][path.stem] = content
dirs_by_parent: dict[PurePosixPath, set[str]] = defaultdict(set)
@@ -75,19 +143,61 @@ def _build_tree(md_files: dict[str, bytes], zf: zipfile.ZipFile) -> list[Documen
zip_names = set(zf.namelist())
# Pre-create all CSV nodes so their IDs are stable before any link rewriting.
csv_nodes: dict[str, DocumentNode] = {
zip_path: DocumentNode(
title=PurePosixPath(zip_path).stem,
content=_csv_to_markdown(csv_bytes),
)
for zip_path, csv_bytes in csv_files.items()
}
# Index by parent dir for tree placement, and by full zip path for link rewriting.
csv_by_parent: dict[PurePosixPath, dict[str, DocumentNode]] = defaultdict(dict)
for zip_path, node in csv_nodes.items():
p = PurePosixPath(zip_path)
csv_by_parent[p.parent][p.stem] = node
def build_children(parent: PurePosixPath) -> list[DocumentNode]:
files = by_parent.get(parent, {})
md_here = by_parent.get(parent, {})
subdirs = dirs_by_parent.get(parent, set())
csv_here = csv_by_parent.get(parent, {})
# Union of .md stems, subdir names, and .csv stems: a name present in
# multiple sources merges into one node (.md takes priority over .csv
# for content; the folder adds children).
nodes = []
# Union of .md stems and subdir names: a name present in both means the
# .md file and the folder represent the same document (content + children).
for name in sorted(set(files) | subdirs):
content = files.get(name)
for name in sorted(set(md_here) | subdirs | set(csv_here)):
if name in md_here:
content = md_here[name]
elif name in csv_here:
# Reuse the pre-created CSV node directly (keeps the stable ID).
node = csv_here[name]
subdir = parent / name
if subdir in all_dirs:
node.children = build_children(subdir)
nodes.append(node)
continue
else:
content = None
node = DocumentNode(title=name, content=content)
if content is not None:
for ref in _IMAGE_REF_RE.findall(
content.decode("utf-8", errors="replace")
):
text = content.decode("utf-8", errors="replace")
# Rewrite local CSV links to point at the imported document.
def _replace_csv_link(m, _parent=parent):
link_text, raw_ref = m.group(1), m.group(2)
zip_path = str(_parent / unquote(raw_ref))
target = csv_nodes.get(zip_path)
if target is None:
return m.group(0)
return f"[{link_text}](/docs/{target.id})"
text = _CSV_LINK_RE.sub(_replace_csv_link, text)
node.content = text.encode("utf-8")
# Collect local media files referenced in the (possibly rewritten) content.
for ref in _IMAGE_REF_RE.findall(text):
if ref.startswith(("http://", "https://")):
continue
zip_path = str(parent / ref)
@@ -95,6 +205,7 @@ def _build_tree(md_files: dict[str, bytes], zf: zipfile.ZipFile) -> list[Documen
# Store the raw bytes under the original relative reference so
# the caller can upload the file and substitute the URL.
node.media[ref] = zf.read(zip_path)
subdir = parent / name
if subdir in all_dirs:
node.children = build_children(subdir)
Binary file not shown.
@@ -19,13 +19,12 @@ def test_parse_empty_zip():
assert not parse_zip(zf)
def test_parse_zip_ignores_non_md_files():
"""Only .md files produce nodes; images, CSVs etc. are silently skipped."""
def test_parse_zip_ignores_non_md_non_csv_files():
"""Non-.md and non-.csv files (images, binaries, etc.) produce no nodes."""
buf = io.BytesIO()
with zipfile.ZipFile(buf, "w") as zf:
zf.writestr("doc.md", b"# Hello")
zf.writestr("image.png", b"\x89PNG")
zf.writestr("data.csv", b"a,b,c")
buf.seek(0)
with zipfile.ZipFile(buf) as zf:
nodes = parse_zip(zf)
@@ -91,6 +90,86 @@ def test_parse_zip_external_links_not_collected():
assert nodes[0].media == {}
def test_parse_csv_becomes_markdown_table():
"""A .csv file is turned into a DocumentNode whose content is a markdown table."""
buf = io.BytesIO()
with zipfile.ZipFile(buf, "w") as zf:
zf.writestr("Collection/data.csv", b"name,score\nAlice,10\nBob,20")
buf.seek(0)
with zipfile.ZipFile(buf) as zf:
roots = parse_zip(zf)
assert len(roots) == 1
assert roots[0].title == "Collection"
assert len(roots[0].children) == 1
node = roots[0].children[0]
assert node.title == "data"
content = node.content.decode()
assert "| name | score |" in content
assert "| Alice | 10 |" in content
assert "| Bob | 20 |" in content
def test_parse_csv_link_rewritten_to_doc_url():
"""URL-encoded CSV links are decoded then rewritten to /docs/{id}."""
buf = io.BytesIO()
with zipfile.ZipFile(buf, "w") as zf:
zf.writestr("Collection/page.md", b"[table](my%20data.csv)")
zf.writestr("Collection/my data.csv", b"x\n1")
buf.seek(0)
with zipfile.ZipFile(buf) as zf:
roots = parse_zip(zf)
by_title = {child.title: child for child in roots[0].children}
assert f"/docs/{by_title['my data'].id}" in by_title["page"].content.decode()
def test_parse_notion_csv_content():
"""Notion CSV (BOM-prefixed, empty cells) becomes a clean markdown table."""
with zipfile.ZipFile(FIXTURES / "notion-export.zip") as zf:
roots = parse_zip(zf)
# The export root container wraps everything; drill into its children.
children = {child.title: child for child in roots[0].children}
# People database exported as a CSV alongside a same-name .md
people_csv = children["People d3d17bcb381d82dfb0d8014512d331ec_all"]
content = people_csv.content.decode()
assert "| Name | About | Membership Type | Person |" in content
assert "| John Smith |" in content
def test_parse_notion_csv_link_rewritten_same_level():
"""A URL-encoded CSV link at the same directory level is rewritten to /docs/{id}."""
with zipfile.ZipFile(FIXTURES / "notion-export.zip") as zf:
roots = parse_zip(zf)
children = {child.title: child for child in roots[0].children}
people_csv = children["People d3d17bcb381d82dfb0d8014512d331ec_all"]
people_md = children["People d3d17bcb381d82dfb0d8014512d331ec"]
assert f"/docs/{people_csv.id}" in people_md.content.decode()
def test_parse_notion_csv_link_rewritten_nested_path():
"""A URL-encoded CSV link whose path crosses a subdirectory is rewritten to /docs/{id}."""
with zipfile.ZipFile(FIXTURES / "notion-export.zip") as zf:
roots = parse_zip(zf)
children = {child.title: child for child in roots[0].children}
# Monthly Budget folder container holds the Expenses CSV as a child
monthly_budget_folder = children["Monthly Budget"]
budget_children = {child.title: child for child in monthly_budget_folder.children}
expenses_csv = budget_children[
"Expenses (Monthly) c8a17bcb381d82a2972601154a09516c_all"
]
# Monthly Budget .md (title includes its Notion UUID) should have the link rewritten
monthly_budget_md = children["Monthly Budget 34c17bcb381d8218a33c01896366c349"]
assert f"/docs/{expenses_csv.id}" in monthly_budget_md.content.decode()
def test_parse_folder_alongside_md_merges():
"""A folder and a same-name .md at the same level merge into one node."""
buf = io.BytesIO()