mirror of
https://github.com/langchain-ai/langgraph.git
synced 2026-09-06 01:37:49 +02:00
docs: generate llms text (#3243)
```shell python docs/_scripts/generate_llms_text.py llms_text.md ```
This commit is contained in:
@@ -0,0 +1,95 @@
|
||||
"""Experimental script to generate consolidated llms text from the docs."""
|
||||
|
||||
import glob
|
||||
import os
|
||||
import pathlib
|
||||
|
||||
from mkdocs.structure.files import File
|
||||
from mkdocs.structure.pages import Page
|
||||
|
||||
from notebook_hooks import _on_page_markdown_with_config
|
||||
|
||||
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
# Get source directory (parent of HERE / docs)
|
||||
SOURCE_DIR = os.path.abspath(os.path.join(os.path.dirname(HERE), "docs"))
|
||||
|
||||
|
||||
def _make_llms_text(output_file: str) -> str:
|
||||
"""Generate a consolidated text file from markdown/notebook files for LLM training.
|
||||
|
||||
Args:
|
||||
output_file: Path to output the consolidated text file
|
||||
"""
|
||||
# Collect all markdown and notebook files
|
||||
relative_paths = [
|
||||
# Files relative to docs/docs/
|
||||
"tutorials/introduction.ipynb",
|
||||
]
|
||||
all_files = [os.path.join(SOURCE_DIR, path) for path in relative_paths]
|
||||
|
||||
all_files.extend(
|
||||
glob.glob(os.path.join(SOURCE_DIR, "how-tos/*.md"), recursive=True)
|
||||
)
|
||||
all_files.extend(
|
||||
glob.glob(os.path.join(SOURCE_DIR, "how-tos/*.ipynb"), recursive=True)
|
||||
)
|
||||
# Add all concepts
|
||||
all_files.extend(
|
||||
glob.glob(os.path.join(SOURCE_DIR, "concepts/*.md"), recursive=True)
|
||||
)
|
||||
all_files.extend(
|
||||
glob.glob(os.path.join(SOURCE_DIR, "concepts/*.ipynb"), recursive=True)
|
||||
)
|
||||
|
||||
all_files = [
|
||||
path if isinstance(pathlib.Path) else pathlib.Path(path) for path in all_files
|
||||
]
|
||||
|
||||
all_content = []
|
||||
|
||||
# Process each file
|
||||
for file_path in all_files:
|
||||
print(f"Processing {file_path}")
|
||||
rel_path = os.path.relpath(file_path, SOURCE_DIR)
|
||||
|
||||
# Create File and Page objects to match mkdocs structure
|
||||
file_obj = File(
|
||||
path=rel_path, src_dir=SOURCE_DIR, dest_dir="", use_directory_urls=True
|
||||
)
|
||||
page = Page(
|
||||
title="",
|
||||
file=file_obj,
|
||||
config={},
|
||||
)
|
||||
|
||||
# Read raw content
|
||||
with open(file_path, "r", encoding="utf-8") as f:
|
||||
content = f.read()
|
||||
|
||||
# Convert to markdown without logic to resolve API references
|
||||
processed_content = _on_page_markdown_with_config(
|
||||
content, page, add_api_references=False, remove_base64_images=True
|
||||
)
|
||||
if processed_content:
|
||||
# Add file name
|
||||
all_content.append(f"---\n{rel_path}\n---")
|
||||
# Add content
|
||||
all_content.append(processed_content)
|
||||
|
||||
# Write consolidated output
|
||||
with open(output_file, "w", encoding="utf-8") as f:
|
||||
f.write("\n\n".join(all_content))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import argparse
|
||||
|
||||
parser = argparse.ArgumentParser(
|
||||
description=(
|
||||
"Generate consolidated text file from markdown/notebook files for LLMs."
|
||||
)
|
||||
)
|
||||
parser.add_argument("output_file", help="Path to output the consolidated text file")
|
||||
|
||||
args = parser.parse_args()
|
||||
_make_llms_text(args.output_file)
|
||||
@@ -7,8 +7,8 @@ from mkdocs.structure.files import Files, File
|
||||
from mkdocs.structure.pages import Page
|
||||
import posixpath
|
||||
|
||||
from notebook_convert import convert_notebook
|
||||
from generate_api_reference_links import update_markdown_with_imports
|
||||
from notebook_convert import convert_notebook
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
logging.basicConfig()
|
||||
@@ -125,7 +125,14 @@ def _highlight_code_blocks(markdown: str) -> str:
|
||||
return markdown
|
||||
|
||||
|
||||
def on_page_markdown(markdown: str, page: Page, **kwargs: Dict[str, Any]):
|
||||
def _on_page_markdown_with_config(
|
||||
markdown: str,
|
||||
page: Page,
|
||||
*,
|
||||
add_api_references: bool = True,
|
||||
remove_base64_images: bool = False,
|
||||
**kwargs: Any,
|
||||
) -> str:
|
||||
if DISABLED:
|
||||
return markdown
|
||||
if page.file.src_path.endswith(".ipynb"):
|
||||
@@ -133,12 +140,26 @@ def on_page_markdown(markdown: str, page: Page, **kwargs: Dict[str, Any]):
|
||||
markdown = convert_notebook(page.file.abs_src_path)
|
||||
|
||||
# Append API reference links to code blocks
|
||||
markdown = update_markdown_with_imports(markdown)
|
||||
if add_api_references:
|
||||
markdown = update_markdown_with_imports(markdown)
|
||||
# Apply highlight comments to code blocks
|
||||
markdown = _highlight_code_blocks(markdown)
|
||||
|
||||
if remove_base64_images:
|
||||
# Remove base64 encoded images from markdown
|
||||
markdown = re.sub(r"!\[.*?\]\(data:image/[^;]+;base64,[^\)]+\)", "", markdown)
|
||||
|
||||
return markdown
|
||||
|
||||
|
||||
def on_page_markdown(markdown: str, page: Page, **kwargs: Dict[str, Any]):
|
||||
return _on_page_markdown_with_config(
|
||||
markdown,
|
||||
page,
|
||||
add_api_references=True,
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
# redirects
|
||||
|
||||
HTML_TEMPLATE = """
|
||||
|
||||
Reference in New Issue
Block a user