From 4da35babda464ab6d9ed17d99ae90fec438a047c Mon Sep 17 00:00:00 2001
From: Xin Jin <39755499+EugeneJinXin@users.noreply.github.com>
Date: Fri, 11 Jul 2025 12:13:55 -0700
Subject: [PATCH] feat: add copy page button functionality and fix llms-text
output (#5419)
* feat: add copy page button functionality and fix llms-text output
- Add copy page button with CSS and JS implementation
- Implement copy page hooks for MkDocs integration
- Fix HTML filtering and DOM text reinterpreted as HTML issues
- Update llms-text target to generate docs/llm.txt instead of docs/llms-full.txt
- Add necessary styling and package.json dependencies
* fix missing button in preview
* remove the over-processing
* disable API reference
---
docs/_scripts/copy_page_hooks.py | 162 +++++++++++++++++++++++++++++++
docs/mkdocs.yml | 1 +
docs/overrides/copy-page.css | 16 +++
docs/overrides/copy-page.js | 38 ++++++++
docs/overrides/main.html | 135 ++++++++++++++++++++++++++
docs/package.json | 3 +-
6 files changed, 354 insertions(+), 1 deletion(-)
create mode 100644 docs/_scripts/copy_page_hooks.py
create mode 100644 docs/overrides/copy-page.css
create mode 100644 docs/overrides/copy-page.js
diff --git a/docs/_scripts/copy_page_hooks.py b/docs/_scripts/copy_page_hooks.py
new file mode 100644
index 000000000..2dd42b29c
--- /dev/null
+++ b/docs/_scripts/copy_page_hooks.py
@@ -0,0 +1,162 @@
+"""
+Copy page functionality hooks for MkDocs.
+
+This module provides hooks to inject original markdown content into HTML pages
+for the copy page functionality, allowing users to copy clean markdown content
+optimized for LLMs.
+"""
+
+import json
+import re
+from pathlib import Path
+from typing import Optional
+
+from mkdocs.config.defaults import MkDocsConfig
+from mkdocs.structure.pages import Page
+
+
+def _process_includes(content: str, docs_dir: Path) -> str:
+ """Process MkDocs includes like {!../README.md!}."""
+ include_pattern = r'\{!([^!]+)!\}'
+
+ def replace_include(match):
+ include_path = match.group(1)
+ # Resolve relative path
+ if include_path.startswith('../'):
+ # Go up from docs dir
+ include_file = docs_dir.parent / include_path[3:]
+ else:
+ include_file = docs_dir / include_path
+
+ try:
+ with open(include_file, 'r', encoding='utf-8') as f:
+ included_content = f.read()
+ # Remove frontmatter from included content to avoid duplication
+ included_content = re.sub(r'^---\n.*?\n---\n', '', included_content, flags=re.DOTALL)
+ return included_content
+ except:
+ return f"[Content from {include_path}]"
+
+ return re.sub(include_pattern, replace_include, content)
+
+
+def _clean_markdown(content: str) -> str:
+ """Minimal cleanup of markdown content - preserve original as much as possible."""
+ # Remove frontmatter
+ content = re.sub(r'^---\n.*?\n---\n', '', content, flags=re.DOTALL)
+
+ # Remove script tags (security)
+ content = re.sub(r'', '', content, flags=re.DOTALL | re.IGNORECASE)
+
+ # Remove style tags (security)
+ content = re.sub(r'', '', content, flags=re.DOTALL | re.IGNORECASE)
+
+ # Remove HTML comments
+ content = re.sub(r'', '', content, flags=re.DOTALL)
+
+ # Just strip and return - preserve original structure
+ return content.strip()
+
+
+def inject_markdown_content(html: str, page: Page, config: MkDocsConfig) -> str:
+ """
+ Inject the original markdown content into the HTML for copy page functionality.
+
+ Args:
+ html: The HTML content to inject into
+ page: The MkDocs page object
+ config: The MkDocs configuration
+
+ Returns:
+ Modified HTML with markdown content injected as JSON
+ """
+ if not hasattr(page, 'file') or not page.file:
+ return html
+
+ # Get the original markdown file path
+ docs_dir = Path(config.get('docs_dir', 'docs'))
+ src_path = page.file.src_path
+
+ # Handle different file types
+ if src_path.endswith('.ipynb'):
+ # For notebook files, we might want to use the converted markdown
+ # For now, just return the HTML as-is
+ return html
+
+ markdown_file = docs_dir / src_path
+
+ if not markdown_file.exists():
+ return html
+
+ try:
+ # Read the original markdown content
+ with open(markdown_file, 'r', encoding='utf-8') as f:
+ markdown_content = f.read()
+
+ # Special handling for index page - use relative path to the actual README.md
+ if src_path == 'index.md':
+ # Relative path to the repository README.md file (go up two levels from docs/docs)
+ readme_path = docs_dir.parent.parent / 'README.md'
+
+ try:
+ with open(readme_path, 'r', encoding='utf-8') as f:
+ readme_content = f.read()
+ # Remove frontmatter if present
+ processed_markdown = re.sub(r'^---\n.*?\n---\n', '', readme_content, flags=re.DOTALL)
+ processed_markdown = processed_markdown.strip()
+ except Exception as e:
+ # If we can't read the README, fallback to original behavior
+ processed_markdown = _process_includes(markdown_content, docs_dir)
+ processed_markdown = re.sub(r'^---\n.*?\n---\n', '', processed_markdown, flags=re.DOTALL)
+ processed_markdown = processed_markdown.strip()
+ else:
+ # Process any includes in the markdown to get the full content
+ processed_markdown = _process_includes(markdown_content, docs_dir)
+ # Clean up the processed markdown normally for other pages
+ processed_markdown = _clean_markdown(processed_markdown)
+
+ # Create the JSON data
+ markdown_data = {
+ 'markdown': processed_markdown,
+ 'title': page.title or 'Page Content',
+ 'url': page.url or ''
+ }
+
+ # Properly escape the JSON for HTML
+ json_content = json.dumps(markdown_data, ensure_ascii=False)
+ json_content = json_content.replace('', '\\u003c/')
+ json_content = json_content.replace(''
+
+ # Insert before if it exists, otherwise before