Compare commits

...
6 Commits
9 changed files with 287 additions and 47 deletions
+13
View File
@@ -1,6 +1,19 @@
Changelog Changelog
========= =========
0.3.2
-----
- Fix image paths to deployed images
`#30 <https://github.com/jdillard/sphinx-llms-txt/pull/30>`_
0.3.1
-----
- Fix issue when ``source_suffix`` equals ``source_link_suffix``
`#29 <https://github.com/jdillard/sphinx-llms-txt/pull/29>`_
0.3.0 0.3.0
----- -----
+1
View File
@@ -3,6 +3,7 @@
A Sphinx extension that generates a summary `llms.txt` file and a single combined documentation `llms-full.txt` file. A Sphinx extension that generates a summary `llms.txt` file and a single combined documentation `llms-full.txt` file.
[![PyPI version](https://img.shields.io/pypi/v/sphinx-llms-txt.svg)](https://pypi.python.org/pypi/sphinx-llms-txt) [![PyPI version](https://img.shields.io/pypi/v/sphinx-llms-txt.svg)](https://pypi.python.org/pypi/sphinx-llms-txt)
[![Conda Version](https://img.shields.io/conda/vn/conda-forge/sphinx-llms-txt.svg)](https://anaconda.org/conda-forge/sphinx-llms-txt)
[![Downloads](https://static.pepy.tech/badge/sphinx-llms-txt/month)](https://pepy.tech/project/sphinx-llms-txt) [![Downloads](https://static.pepy.tech/badge/sphinx-llms-txt/month)](https://pepy.tech/project/sphinx-llms-txt)
[![Parallel Safe](https://img.shields.io/badge/parallel%20safe-true-brightgreen)](#) [![Parallel Safe](https://img.shields.io/badge/parallel%20safe-true-brightgreen)](#)
+5 -1
View File
@@ -81,7 +81,11 @@ html_theme = "furo"
# further. For a list of options available for each theme, see the # further. For a list of options available for each theme, see the
# documentation. # documentation.
# #
html_theme_options = {} html_theme_options = {
"source_repository": "https://github.com/jdillard/sphinx-llms-txt/",
"source_branch": "main",
"source_directory": "docs/source/",
}
html_baseurl = "https://sphinx-llms-txt.readthedocs.org/" html_baseurl = "https://sphinx-llms-txt.readthedocs.org/"
+4 -1
View File
@@ -3,7 +3,7 @@ Sphinx llms.txt Generator
A `Sphinx`_ extension that generates a summary ``llms.txt`` file, written in Markdown, and a single combined documentation ``llms-full.txt`` file, written in reStructuredText. A `Sphinx`_ extension that generates a summary ``llms.txt`` file, written in Markdown, and a single combined documentation ``llms-full.txt`` file, written in reStructuredText.
|PyPI version| |Downloads| |Parallel Safe| |GitHub Stars| |PyPI version| |Conda Version| |Downloads| |Parallel Safe| |GitHub Stars|
.. toctree:: .. toctree::
:maxdepth: 2 :maxdepth: 2
@@ -20,6 +20,9 @@ A `Sphinx`_ extension that generates a summary ``llms.txt`` file, written in Mar
.. |PyPI version| image:: https://img.shields.io/pypi/v/sphinx-llms-txt.svg .. |PyPI version| image:: https://img.shields.io/pypi/v/sphinx-llms-txt.svg
:target: https://pypi.python.org/pypi/sphinx-llms-txt :target: https://pypi.python.org/pypi/sphinx-llms-txt
:alt: Latest PyPi Version :alt: Latest PyPi Version
.. |Conda Version| image:: https://img.shields.io/conda/vn/conda-forge/sphinx-llms-txt.svg
:target: https://anaconda.org/conda-forge/sphinx-llms-txt
:alt: Latest Conda Version
.. |Downloads| image:: https://static.pepy.tech/badge/sphinx-llms-txt/month .. |Downloads| image:: https://static.pepy.tech/badge/sphinx-llms-txt/month
:target: https://pepy.tech/project/sphinx-llms-txt :target: https://pepy.tech/project/sphinx-llms-txt
:alt: PyPi Downloads per month :alt: PyPi Downloads per month
+1 -1
View File
@@ -12,7 +12,7 @@ from .manager import LLMSFullManager
from .processor import DocumentProcessor from .processor import DocumentProcessor
from .writer import FileWriter from .writer import FileWriter
__version__ = "0.3.0" __version__ = "0.3.2"
# Export classes needed by tests # Export classes needed by tests
__all__ = [ __all__ = [
+7 -1
View File
@@ -90,7 +90,13 @@ class DocumentCollector:
# Try to find the source file with any of the valid source suffixes # Try to find the source file with any of the valid source suffixes
for src_suffix in source_suffixes: for src_suffix in source_suffixes:
candidate_file = sources_dir / f"{docname}{src_suffix}{source_link_suffix}" # Avoid duplicate extensions when source_suffix == source_link_suffix
if src_suffix == source_link_suffix:
candidate_file = sources_dir / f"{docname}{src_suffix}"
else:
candidate_file = (
sources_dir / f"{docname}{src_suffix}{source_link_suffix}"
)
if candidate_file.exists(): if candidate_file.exists():
return src_suffix return src_suffix
+22 -4
View File
@@ -138,13 +138,22 @@ class LLMSFullManager:
# Build the source file path directly using the known suffix # Build the source file path directly using the known suffix
if src_suffix: if src_suffix:
source_file = sources_dir / f"{docname}{src_suffix}{source_link_suffix}" # Avoid duplicate extensions when source_suffix == source_link_suffix
if src_suffix == source_link_suffix:
source_file = sources_dir / f"{docname}{src_suffix}"
expected_suffix = src_suffix
else:
source_file = (
sources_dir / f"{docname}{src_suffix}{source_link_suffix}"
)
expected_suffix = f"{src_suffix}{source_link_suffix}"
if source_file.exists(): if source_file.exists():
docname_to_file[docname] = source_file docname_to_file[docname] = source_file
else: else:
logger.warning( logger.warning(
f"sphinx-llms-txt: Source file not found for: {docname}." f"sphinx-llms-txt: Source file not found for: {docname}."
f"Expected: {docname}{src_suffix}{source_link_suffix}" f"Expected: {docname}{expected_suffix}"
) )
else: else:
logger.warning( logger.warning(
@@ -205,7 +214,11 @@ class LLMSFullManager:
source_suffixes = self._get_source_suffixes() source_suffixes = self._get_source_suffixes()
all_source_files = [] all_source_files = []
for src_suffix in source_suffixes: for src_suffix in source_suffixes:
glob_pattern = f"**/*{src_suffix}{source_link_suffix}" # Avoid duplicate extensions when source_suffix == source_link_suffix
if src_suffix == source_link_suffix:
glob_pattern = f"**/*{src_suffix}"
else:
glob_pattern = f"**/*{src_suffix}{source_link_suffix}"
all_source_files.extend(sources_dir.glob(glob_pattern)) all_source_files.extend(sources_dir.glob(glob_pattern))
processed_paths = set(file.resolve() for file in docname_to_file.values()) processed_paths = set(file.resolve() for file in docname_to_file.values())
@@ -231,7 +244,12 @@ class LLMSFullManager:
# Try each source suffix to find which one this file uses # Try each source suffix to find which one this file uses
for src_suffix in source_suffixes: for src_suffix in source_suffixes:
combined_suffix = f"{src_suffix}{source_link_suffix}" # Avoid duplicate extensions when suffixes match
if src_suffix == source_link_suffix:
combined_suffix = src_suffix
else:
combined_suffix = f"{src_suffix}{source_link_suffix}"
if rel_path.endswith(combined_suffix): if rel_path.endswith(combined_suffix):
docname = rel_path[: -len(combined_suffix)] # Remove suffix docname = rel_path[: -len(combined_suffix)] # Remove suffix
break break
+78 -36
View File
@@ -94,8 +94,14 @@ class DocumentProcessor:
if not base_url: if not base_url:
return path return path
# Ensure base URL ends with slash
if not base_url.endswith("/"): if not base_url.endswith("/"):
base_url += "/" base_url += "/"
# Remove leading slash from path to avoid double slashes
if path.startswith("/"):
path = path[1:]
return f"{base_url}{path}" return f"{base_url}{path}"
def _is_absolute_or_url(self, path: str) -> bool: def _is_absolute_or_url(self, path: str) -> bool:
@@ -137,12 +143,73 @@ class DocumentProcessor:
prefix = match.group(1) # The entire directive prefix including whitespace prefix = match.group(1) # The entire directive prefix including whitespace
path = match.group(3).strip() # The path argument path = match.group(3).strip() # The path argument
# Only process relative paths, not absolute paths or URLs # Handle URLs and data URIs - leave unchanged
if not self._is_absolute_or_url(path): if path.startswith(("http://", "https://", "data:")):
# Special case for test files return match.group(0)
if is_test:
# Add subdir/ prefix to match test expectations # For ALL paths, check if image exists in _images first
full_path = "subdir/" + path # Extract filename from the path
filename = os.path.basename(path)
# Check if image exists in _images directory
# First determine the build directory from source_path
build_dir = None
if "_sources" in str(source_path):
# Extract build directory (parent of _sources)
path_parts = str(source_path).split("_sources/")
if len(path_parts) > 1:
build_dir = path_parts[0].rstrip("/")
# If we can determine the build directory, check if image exists in _images
if build_dir:
images_path = os.path.join(build_dir, "_images", filename)
if os.path.exists(images_path):
# Image exists in _images, use _images path
full_path = f"/_images/{filename}"
# Add base URL if configured
full_path = self._add_base_url(full_path, base_url)
return f"{prefix}{full_path}"
# Image doesn't exist in _images, handle based on path type
# Handle absolute paths (starting with /) - add base URL if configured
if path.startswith("/"):
# Add base URL to absolute paths if configured
full_path = self._add_base_url(path, base_url)
return f"{prefix}{full_path}"
# Handle relative paths with original logic for backward compatibility
# Special case for test files
if is_test:
# Add subdir/ prefix to match test expectations
full_path = "subdir/" + path
# If base_url is set, prepend it to the path
full_path = self._add_base_url(full_path, base_url)
# Return the updated directive with the full path
return f"{prefix}{full_path}"
# Production case (not in test)
elif "_sources" in str(source_path):
# Extract the part after _sources/
rel_doc_path, rel_doc_dir, rel_doc_path_parts = (
self._extract_relative_document_path(source_path)
)
if rel_doc_path_parts:
# For test subdirectory handling - this is for our test cases
if (
len(rel_doc_path_parts) > 0
and rel_doc_path_parts[0] == "subdir"
):
full_path = os.path.normpath(os.path.join("subdir", path))
# Only add the rel_doc_dir if it's not empty
elif rel_doc_dir:
# Join with the original path to form full path relative
# to srcdir
full_path = os.path.normpath(os.path.join(rel_doc_dir, path))
else:
full_path = path
# If base_url is set, prepend it to the path # If base_url is set, prepend it to the path
full_path = self._add_base_url(full_path, base_url) full_path = self._add_base_url(full_path, base_url)
@@ -150,37 +217,12 @@ class DocumentProcessor:
# Return the updated directive with the full path # Return the updated directive with the full path
return f"{prefix}{full_path}" return f"{prefix}{full_path}"
# Production case (not in test) # Fallback for relative paths - add base URL if configured
elif "_sources" in str(source_path): else:
# Extract the part after _sources/ full_path = self._add_base_url(path, base_url)
rel_doc_path, rel_doc_dir, rel_doc_path_parts = ( return f"{prefix}{full_path}"
self._extract_relative_document_path(source_path)
)
if rel_doc_path_parts: # If we couldn't resolve the path, return unchanged
# For test subdirectory handling - this is for our test cases
if (
len(rel_doc_path_parts) > 0
and rel_doc_path_parts[0] == "subdir"
):
full_path = os.path.normpath(os.path.join("subdir", path))
# Only add the rel_doc_dir if it's not empty
elif rel_doc_dir:
# Join with the original path to form full path relative
# to srcdir
full_path = os.path.normpath(
os.path.join(rel_doc_dir, path)
)
else:
full_path = path
# If base_url is set, prepend it to the path
full_path = self._add_base_url(full_path, base_url)
# Return the updated directive with the full path
return f"{prefix}{full_path}"
# If we couldn't resolve the path or it's already absolute, return unchanged
return match.group(0) return match.group(0)
# Replace directive paths in the content # Replace directive paths in the content
+156 -3
View File
@@ -102,7 +102,7 @@ def test_process_path_directives_with_html_baseurl(tmp_path):
def test_process_path_directives_absolute_urls(tmp_path): def test_process_path_directives_absolute_urls(tmp_path):
"""Test that absolute URLs are not modified.""" """Test that absolute URLs are not modified but absolute paths get base URL."""
# Create a processor # Create a processor
config = { config = {
"llms_txt_directives": [], "llms_txt_directives": [],
@@ -127,10 +127,17 @@ def test_process_path_directives_absolute_urls(tmp_path):
with open(source_file, "w", encoding="utf-8") as f: with open(source_file, "w", encoding="utf-8") as f:
f.write(source_content) f.write(source_content)
# Process the directives (should remain unchanged) # Process the directives
processed_content = processor._process_path_directives(source_content, source_file) processed_content = processor._process_path_directives(source_content, source_file)
assert processed_content == source_content # Expected: URLs and data URIs unchanged, absolute paths get base URL
expected_content = (
".. image:: https://othersite.com/images/test.png\n"
".. image:: https://example.com/docs/absolute/path/image.png\n"
".. image:: data:image/png;base64,iVBORw0KG...\n"
)
assert processed_content == expected_content
def test_process_path_directives_custom_directives(tmp_path): def test_process_path_directives_custom_directives(tmp_path):
@@ -251,3 +258,149 @@ def test_process_content_end_to_end(tmp_path):
) )
assert processed_content == expected_content assert processed_content == expected_content
def test_process_path_directives_images_directory(tmp_path):
"""Test that _images directory paths are handled correctly."""
# Create a processor with base URL
config = {
"llms_txt_directives": [],
"html_baseurl": "https://example.com/docs",
}
processor = DocumentProcessor(config)
# Create source directory structure
src_dir = tmp_path / "src"
src_dir.mkdir()
processor.srcdir = str(src_dir)
# Create _sources directory to mimic Sphinx output
build_dir = tmp_path / "build"
build_dir.mkdir()
sources_dir = build_dir / "_sources"
sources_dir.mkdir()
# Create a source file with various _images directory paths
source_content = (
"Some content.\n"
".. image:: _images/test.png\n" # Relative _images should become /_images
".. image:: /_images/absolute.png\n" # Absolute _images should get base URL
".. figure:: _images/figure.png\n" # Test with figure directive too
" :alt: A test figure\n"
".. image:: images/normal.png\n" # Normal relative path should be unchanged
)
# Create source file in sources directory to simulate Sphinx build output
source_file = sources_dir / "page.txt"
with open(source_file, "w", encoding="utf-8") as f:
f.write(source_content)
# Process the directives
processed_content = processor._process_path_directives(source_content, source_file)
# Expected: _images paths should be converted and get base URL
expected_content = (
"Some content.\n"
".. image:: https://example.com/docs/_images/test.png\n"
".. image:: https://example.com/docs/_images/absolute.png\n"
".. figure:: https://example.com/docs/_images/figure.png\n"
" :alt: A test figure\n"
".. image:: https://example.com/docs/images/normal.png\n"
)
assert processed_content == expected_content
def test_process_path_directives_images_directory_no_baseurl(tmp_path):
"""
Test that _images directory paths work correctly without base URL.
Only converts when image exists.
"""
# Create a processor without base URL
config = {
"llms_txt_directives": [],
"html_baseurl": "",
}
processor = DocumentProcessor(config)
# Create source directory structure
src_dir = tmp_path / "src"
src_dir.mkdir()
processor.srcdir = str(src_dir)
# Create _sources directory to mimic Sphinx output
build_dir = tmp_path / "build"
build_dir.mkdir()
sources_dir = build_dir / "_sources"
sources_dir.mkdir()
# Create _images directory and one test image
images_dir = build_dir / "_images"
images_dir.mkdir()
(images_dir / "test.png").write_text("fake image content")
# Note: absolute.png is not created, so it won't be converted
# Create a source file with _images directory paths
source_content = (
".. image:: _images/test.png\n" # Should become /_images (image exists)
".. image:: /_images/absolute.png\n" # Should stay unchanged (absolute path)
)
# Create source file in sources directory to simulate Sphinx build output
source_file = sources_dir / "page.txt"
with open(source_file, "w", encoding="utf-8") as f:
f.write(source_content)
# Process the directives
processed_content = processor._process_path_directives(source_content, source_file)
# Expected: only test.png gets converted because it exists in _images
expected_content = (
".. image:: /_images/test.png\n" # Converted because image exists
".. image:: /_images/absolute.png\n" # Absolute path unchanged
)
assert processed_content == expected_content
def test_process_path_directives_all_absolute_paths_get_baseurl(tmp_path):
"""Test that all absolute paths (starting with /) get base URL prepended."""
# Create a processor with base URL
config = {
"llms_txt_directives": [],
"html_baseurl": "https://mysite.com/docs/",
}
processor = DocumentProcessor(config)
# Create source directory structure
src_dir = tmp_path / "src"
src_dir.mkdir()
processor.srcdir = str(src_dir)
# Create a source file with various absolute paths
source_content = (
".. image:: /static/images/logo.png\n"
".. figure:: /assets/diagrams/flow.svg\n"
".. image:: /media/photos/team.jpg\n"
" :alt: Team photo\n"
".. image:: relative/path.png\n" # This should still get normal processing
)
# Create source file
source_file = src_dir / "page.txt"
with open(source_file, "w", encoding="utf-8") as f:
f.write(source_content)
# Process the directives
processed_content = processor._process_path_directives(source_content, source_file)
# Expected: All absolute paths get base URL prepended
expected_content = (
".. image:: https://mysite.com/docs/static/images/logo.png\n"
".. figure:: https://mysite.com/docs/assets/diagrams/flow.svg\n"
".. image:: https://mysite.com/docs/media/photos/team.jpg\n"
" :alt: Team photo\n"
".. image:: https://mysite.com/docs/relative/path.png\n"
)
assert processed_content == expected_content