Compare commits

..
Author SHA1 Message Date
Jared DillardandGitHub b63801bcff Improve _sources directory handling (#47) 2025-10-13 23:53:46 -07:00
dependabot[bot]GitHubdependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com>
c45ebb0369 Bump the all-github-actions group with 2 updates (#45)
Bumps the all-github-actions group with 2 updates: [actions/checkout](https://github.com/actions/checkout) and [actions/setup-python](https://github.com/actions/setup-python).


Updates `actions/checkout` from 4 to 5
- [Release notes](https://github.com/actions/checkout/releases)
- [Changelog](https://github.com/actions/checkout/blob/main/CHANGELOG.md)
- [Commits](https://github.com/actions/checkout/compare/v4...v5)

Updates `actions/setup-python` from 5 to 6
- [Release notes](https://github.com/actions/setup-python/releases)
- [Commits](https://github.com/actions/setup-python/compare/v5...v6)

---
updated-dependencies:
- dependency-name: actions/checkout
  dependency-version: '5'
  dependency-type: direct:production
  update-type: version-update:semver-major
  dependency-group: all-github-actions
- dependency-name: actions/setup-python
  dependency-version: '6'
  dependency-type: direct:production
  update-type: version-update:semver-major
  dependency-group: all-github-actions
...

Signed-off-by: dependabot[bot] <support@github.com>
Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com>
2025-10-04 18:59:29 -07:00
Jared DillardandGitHub f5dcd15889 Fix optional sphinx dependency (#44) 2025-09-17 10:51:07 -07:00
Jared DillardandGitHub e64e20133a Remove support for singlehtml (#40) 2025-08-29 15:22:32 -07:00
Jared Dillard 52949a952a Update changelog 2025-08-20 16:01:44 -07:00
Jared DillardandGitHub 3d7edbf7d9 Only allow builders that have a _sources directory (#38) 2025-08-20 15:57:47 -07:00
11 changed files with 272 additions and 85 deletions
+5 -5
View File
@@ -10,9 +10,9 @@ jobs:
pre-commit:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: actions/checkout@v5
- name: Set up Python 3.10
uses: actions/setup-python@v5
uses: actions/setup-python@v6
with:
python-version: "3.10"
- uses: pre-commit/action@v3.0.1
@@ -23,17 +23,17 @@ jobs:
python-version: ['3.9', '3.10', '3.11', '3.12']
steps:
- uses: actions/checkout@v4
- uses: actions/checkout@v5
- name: Set up Python ${{ matrix.python-version }}
uses: actions/setup-python@v5
uses: actions/setup-python@v6
with:
python-version: ${{ matrix.python-version }}
- name: Install dependencies
run: |
python -m pip install --upgrade pip
pip install -e ".[dev]"
pip install -e . --group dev
# - name: Run mypy
# run: |
-7
View File
@@ -18,13 +18,6 @@ repos:
hooks:
- id: flake8
- repo: https://github.com/pre-commit/mirrors-mypy
rev: v1.11.2
hooks:
- id: mypy
files: ^sphinx_llms_txt/
additional_dependencies: [types-docutils]
- repo: https://github.com/sphinx-contrib/sphinx-lint
rev: v1.0.0
hooks:
+24
View File
@@ -1,6 +1,30 @@
Changelog
=========
0.6.0
-----
- Improve _sources directory handling
`#47 <https://github.com/jdillard/sphinx-llms-txt/pull/47>`_
0.5.3
-----
- Make sphinx a required dependency since there are imports from Sphinx
`#44 <https://github.com/jdillard/sphinx-llms-txt/pull/44>`_
0.5.2
-----
- Remove support for singlehtml
`#40 <https://github.com/jdillard/sphinx-llms-txt/pull/40>`_
0.5.1
-----
- Only allow builders that have _sources directory
`#38 <https://github.com/jdillard/sphinx-llms-txt/pull/38>`_
0.5.0
-----
+1 -1
View File
@@ -19,7 +19,7 @@ Local development
.. code-block:: console
pip install -e ".[dev]"
pip install -e . --group dev
#. Install pre-commit Git hook scripts:
+4 -2
View File
@@ -26,13 +26,16 @@ classifiers = [
license = {text = "MIT"}
readme = "README.md"
dynamic = ["version"]
dependencies = [
"sphinx",
]
[project.urls]
download = "https://pypi.org/project/sphinx-llms-txt/"
source = "https://github.com/jdillard/sphinx-llms-txt"
changelog = "https://github.com/jdillard/sphinx-llms-txt/blob/master/CHANGELOG.rst"
[project.optional-dependencies]
[dependency-groups]
dev = [
"pytest>=7.0.0",
"black",
@@ -40,7 +43,6 @@ dev = [
"mypy",
"isort",
"pre-commit",
"sphinx",
]
test = [
"pytest>=7.0.0",
+11 -6
View File
@@ -21,7 +21,7 @@ from .manager import LLMSFullManager
from .processor import DocumentProcessor
from .writer import FileWriter
__version__ = "0.5.0"
__version__ = "0.6.0"
# Export classes needed by tests
__all__ = [
@@ -107,13 +107,12 @@ def build_finished(app: Sphinx, exception):
_manager.update_page_title(docname, title)
# Create the combined file
_manager.combine_sources(str(app.outdir), str(app.srcdir))
_manager.combine_sources(app.outdir, app.srcdir)
def setup(app: Sphinx) -> Dict[str, Any]:
"""Set up the Sphinx extension."""
# Add configuration options
app.add_config_value("llms_txt_file", True, "env")
app.add_config_value("llms_txt_filename", "llms.txt", "env")
app.add_config_value("llms_txt_full_file", True, "env")
@@ -127,15 +126,21 @@ def setup(app: Sphinx) -> Dict[str, Any]:
app.add_config_value("llms_txt_code_files", [], "env")
app.add_config_value("llms_txt_code_base_path", None, "env")
# Connect to Sphinx events
app.connect("doctree-resolved", doctree_resolved)
app.connect("build-finished", build_finished)
def builder_inited(app):
"""Used to limit what builders are allowed to run the extension."""
allowed_builders = ["html", "dirhtml"]
if hasattr(app, "builder") and app.builder.name in allowed_builders:
# Reset manager and root paragraph for each build
global _manager, _root_first_paragraph
_manager = LLMSFullManager()
_root_first_paragraph = ""
app.connect("doctree-resolved", doctree_resolved)
app.connect("build-finished", build_finished)
app.connect("builder-inited", builder_inited)
return {
"version": __version__,
"parallel_read_safe": True,
+15 -22
View File
@@ -3,7 +3,7 @@ Document collector module for sphinx-llms-txt.
"""
import fnmatch
from typing import Any, Dict, List, Optional, Tuple
from typing import Any, Dict, List, Tuple
from sphinx.environment import BuildEnvironment
from sphinx.util import logging
@@ -16,8 +16,8 @@ class DocumentCollector:
def __init__(self):
self.page_titles: Dict[str, str] = {}
self.master_doc: Optional[str] = None
self.env: Optional[BuildEnvironment] = None
self.master_doc: str = None
self.env: BuildEnvironment = None
self.config: Dict[str, Any] = {}
self.app = None
@@ -60,7 +60,7 @@ class DocumentCollector:
else:
return [source_suffix] # String format
def _get_docname_suffix(self, docname: str, sources_dir) -> Optional[str]:
def _get_docname_suffix(self, docname: str, sources_dir) -> str:
"""
Determine the source suffix for a given docname by checking which
file exists.
@@ -102,7 +102,7 @@ class DocumentCollector:
return None
def get_page_order(self, sources_dir=None) -> List[Tuple[str, Optional[str]]]:
def get_page_order(self, sources_dir=None) -> List[Tuple[str, str]]:
"""Get the correct page order from the toctree structure.
Args:
@@ -114,7 +114,7 @@ class DocumentCollector:
if not self.env or not self.master_doc:
return []
page_order: List[Tuple[str, Optional[str]]] = []
page_order = []
visited = set()
def collect_from_toctree(docname: str):
@@ -126,7 +126,7 @@ class DocumentCollector:
# Add the current document with its suffix
if docname not in [doc for doc, _ in page_order]:
suffix: Optional[str] = None
suffix = None
if sources_dir:
suffix = self._get_docname_suffix(docname, sources_dir)
page_order.append((docname, suffix))
@@ -135,21 +135,18 @@ class DocumentCollector:
try:
# Look for toctree_includes which contains the direct children
if (
self.env
and hasattr(self.env, "toctree_includes")
hasattr(self.env, "toctree_includes")
and docname in self.env.toctree_includes
):
for child_docname in self.env.toctree_includes[docname]:
collect_from_toctree(str(child_docname))
collect_from_toctree(child_docname)
# Try to use dependencies to find related documents
elif (
self.env
and hasattr(self.env, "dependencies")
hasattr(self.env, "dependencies")
and docname in self.env.dependencies
):
# Extract the dependent documents from the dependencies dict
for child_docname_obj in self.env.dependencies[docname]:
child_docname = str(child_docname_obj)
for child_docname in self.env.dependencies[docname]:
# Only add documents actually in the document set
if (
hasattr(self.env, "all_docs")
@@ -157,11 +154,7 @@ class DocumentCollector:
):
collect_from_toctree(child_docname)
# Fallback to titles or other available references
elif (
self.env
and hasattr(self.env, "titles")
and hasattr(self.env, "all_docs")
):
elif hasattr(self.env, "titles") and hasattr(self.env, "all_docs"):
# Get all document names
all_docnames = list(self.env.all_docs.keys())
@@ -192,7 +185,7 @@ class DocumentCollector:
]
)
for docname in remaining:
suffix: Optional[str] = None
suffix = None
if sources_dir:
suffix = self._get_docname_suffix(docname, sources_dir)
page_order.append((docname, suffix))
@@ -200,8 +193,8 @@ class DocumentCollector:
return page_order
def filter_excluded_pages(
self, page_order: List[Tuple[str, Optional[str]]]
) -> List[Tuple[str, Optional[str]]]:
self, page_order: List[Tuple[str, str]]
) -> List[Tuple[str, str]]:
"""Filter out excluded pages from the page order."""
exclude_patterns = self.config.get("llms_txt_exclude")
if exclude_patterns:
+40 -26
View File
@@ -5,7 +5,7 @@ Main manager module for sphinx-llms-txt.
import glob
import subprocess
from pathlib import Path
from typing import Any, Dict, List, Optional, Tuple, Union, cast
from typing import Any, Dict, List, Optional, Tuple, Union
from sphinx.application import Sphinx
from sphinx.environment import BuildEnvironment
@@ -150,8 +150,8 @@ class LLMSFullManager:
self.ignored_pages.add(docname)
def _filter_ignored_pages(
self, page_order: Union[List[str], List[Tuple[str, Optional[str]]]]
) -> Union[List[str], List[Tuple[str, Optional[str]]]]:
self, page_order: Union[List[str], List[Tuple[str, str]]]
) -> Union[List[str], List[Tuple[str, str]]]:
"""Filter out ignored pages from page_order."""
filtered_pages = []
for item in page_order:
@@ -164,7 +164,7 @@ class LLMSFullManager:
if docname not in self.ignored_pages:
filtered_pages.append(item)
return cast(Union[List[str], List[Tuple[str, Optional[str]]]], filtered_pages)
return filtered_pages
def set_config(self, config: Dict[str, Any]):
"""Set configuration options."""
@@ -197,7 +197,6 @@ class LLMSFullManager:
possible_sources = [
Path(outdir) / "_sources",
Path(outdir) / "html" / "_sources",
Path(outdir) / "singlehtml" / "_sources",
]
for path in possible_sources:
@@ -205,27 +204,44 @@ class LLMSFullManager:
sources_dir = path
break
if not sources_dir:
logger.warning(
"Could not find _sources directory, skipping llms-full creation"
)
return
# Get the correct page order with source suffixes
# Get the correct page order (with or without source suffixes)
page_order = self.collector.get_page_order(sources_dir)
if not page_order:
logger.warning(
"Could not determine page order, skipping llms-full creation"
)
logger.warning("Could not determine page order, skipping file generation")
return
# Apply exclusion filter if configured
page_order = self.collector.filter_excluded_pages(page_order)
# Determine output file name and location
# If no sources directory, only generate llms.txt and return early
if not sources_dir:
# Generate llms.txt if requested
if self.config.get("llms_txt_file"):
filtered_page_order = self._filter_ignored_pages(page_order)
self.writer.write_verbose_info_to_file(
filtered_page_order,
self.collector.page_titles,
0, # No line count since no llms-full.txt
)
# Only warn if user explicitly wants llms-full.txt
if self.config.get("llms_txt_full_file"):
# Check if html_copy_source is False
if self.app and not self.app.config.html_copy_source:
logger.warning(
"Could not find _sources directory, skipping llms-full.txt."
"Set html_copy_source = True in conf.py to enable."
)
else:
logger.warning(
"Could not find _sources directory, skipping llms-full.txt"
)
return
# Determine output file name and location for llms-full.txt
output_filename = self.config.get("llms_txt_full_filename")
output_path = Path(outdir) / str(output_filename)
output_path = Path(outdir) / output_filename
# Log discovered files and page order
logger.debug(f"sphinx-llms-txt: Page order (after exclusion): {page_order}")
@@ -286,7 +302,7 @@ class LLMSFullManager:
content_parts = []
# Track code files for later processing
code_file_parts: List[str] = []
code_file_parts = []
# Count lines in code files (initially 0)
code_files_line_count = 0
@@ -365,7 +381,7 @@ class LLMSFullManager:
if not (size_limit_exceeded and should_abort_early):
# Get all source files in the _sources directory using configured suffixes
source_suffixes = self._get_source_suffixes()
all_source_files: List[Path] = []
all_source_files = []
for src_suffix in source_suffixes:
# Avoid duplicate extensions when source_suffix == source_link_suffix
if src_suffix == source_link_suffix:
@@ -735,9 +751,9 @@ class LLMSFullManager:
title = Path(title_str[len(base_path) :])
except ValueError:
# File is not relative to srcdir, use filename
title = Path(file_path.name)
title = file_path.name
else:
title = Path(file_path.name)
title = file_path.name
# Format as code block with equals underline
title_str = str(title)
@@ -769,9 +785,7 @@ class LLMSFullManager:
return code_parts, sorted(processed_files)
def _create_code_files_section_header(
self, file_paths: Optional[List[Path]] = None
) -> str:
def _create_code_files_section_header(self, file_paths: List[Path] = None) -> str:
"""Create the section header for source code files.
Args:
@@ -819,7 +833,7 @@ class LLMSFullManager:
return ""
# Convert to relative paths if possible and create tree structure
tree_data: Dict[str, Any] = {}
tree_data = {}
for file_path in sorted(file_paths):
# Get relative path from source directory for display
@@ -872,7 +886,7 @@ class LLMSFullManager:
current[parts[-1]] = None # None indicates it's a file
# Convert tree structure to string representation
lines: List[str] = []
lines = []
self._format_tree_node(tree_data, lines, "", True)
# Indent each line for reStructuredText code block
+1 -1
View File
@@ -130,7 +130,7 @@ class DocumentProcessor:
"""
# Get the configured path directives to process
default_path_directives = ["image", "figure"]
custom_path_directives = self.config.get("llms_txt_directives") or []
custom_path_directives = self.config.get("llms_txt_directives")
path_directives = set(default_path_directives + custom_path_directives)
# Build the regex pattern to match all configured directives
+5 -10
View File
@@ -3,7 +3,7 @@ File writer module for sphinx-llms-txt.
"""
from pathlib import Path
from typing import Any, Dict, List, Optional, Tuple, Union
from typing import Any, Dict, List, Tuple, Union
from sphinx.application import Sphinx
from sphinx.util import logging
@@ -14,12 +14,7 @@ logger = logging.getLogger(__name__)
class FileWriter:
"""Handles writing processed content to output files."""
def __init__(
self,
config: Dict[str, Any],
outdir: Optional[str] = None,
app: Optional[Sphinx] = None,
):
def __init__(self, config: Dict[str, Any], outdir: str = None, app: Sphinx = None):
self.config = config
self.outdir = outdir
self.app = app
@@ -52,7 +47,7 @@ class FileWriter:
def write_verbose_info_to_file(
self,
page_order: Union[List[str], List[Tuple[str, Optional[str]]]],
page_order: Union[List[str], List[Tuple[str, str]]],
page_titles: Dict[str, str],
total_line_count: int = 0,
) -> bool:
@@ -72,13 +67,13 @@ class FileWriter:
)
return False
output_path = Path(self.outdir) / str(self.config.get("llms_txt_filename"))
output_path = Path(self.outdir) / self.config.get("llms_txt_filename")
try:
with open(output_path, "w", encoding="utf-8") as f:
project_name = "llms-txt Summary"
# First priority: use title from config if available
if self.config.get("llms_txt_title"):
project_name = str(self.config.get("llms_txt_title"))
project_name = self.config.get("llms_txt_title")
# Second priority: use project name from Sphinx app if available
elif (
self.app
+161
View File
@@ -41,6 +41,42 @@ def test_setup_returns_valid_dict():
assert "parallel_write_safe" in result
def test_builder_inited_with_disallowed_builder():
"""Test that disallowed builders do not trigger extension setup."""
import sphinx_llms_txt
# Reset global state
sphinx_llms_txt._manager = sphinx_llms_txt.LLMSFullManager()
sphinx_llms_txt._root_first_paragraph = ""
# Mock a Sphinx app with a disallowed builder
class MockBuilder:
name = "text" # Not in allowed list
class MockApp:
def __init__(self):
self.config_values = {}
self.connections = {}
self.builder = MockBuilder()
def add_config_value(self, name, default, rebuild):
self.config_values[name] = (default, rebuild)
def connect(self, event, handler):
self.connections[event] = handler
app = MockApp()
setup(app)
# Trigger builder-inited
builder_inited_handler = app.connections["builder-inited"]
builder_inited_handler(app)
# With disallowed builder, other events should NOT be connected
assert "doctree-resolved" not in app.connections
assert "build-finished" not in app.connections
def test_document_collector_initialization():
"""Test initialization of DocumentCollector."""
collector = DocumentCollector()
@@ -995,3 +1031,128 @@ def test_code_files_ignored_patterns(tmp_path, caplog):
assert (
"Code file pattern 'docs/**/*.rst' ignored." in captured_warnings[0]
), f"Warning message should contain expected text. Got: {captured_warnings[0]}"
def test_llms_txt_generated_without_sources_dir(tmp_path):
"""Test that llms.txt is generated even when _sources directory doesn't exist."""
from sphinx_llms_txt.manager import LLMSFullManager
# Create manager
manager = LLMSFullManager()
# Set config to enable llms.txt
config = {
"llms_txt_file": True,
"llms_txt_filename": "llms.txt",
"llms_txt_full_file": True,
"llms_txt_full_filename": "llms-full.txt",
"llms_txt_exclude": [],
"llms_txt_directives": [],
}
manager.set_config(config)
# Create directories (but no _sources)
outdir = tmp_path / "build"
srcdir = tmp_path / "source"
outdir.mkdir()
srcdir.mkdir()
# Mock env with documents
class MockEnv:
all_docs = {"index": None, "about": None}
titles = {
"index": type("TitleNode", (), {"astext": lambda self: "Home"})(),
"about": type("TitleNode", (), {"astext": lambda self: "About"})(),
}
toctree_includes = {"index": ["about"]}
manager.set_env(MockEnv())
manager.set_master_doc("index")
# Update page titles directly in the collector
manager.update_page_title("index", "Home")
manager.update_page_title("about", "About")
# Call combine_sources - should generate llms.txt even without _sources
manager.combine_sources(str(outdir), str(srcdir))
# Verify llms.txt was created
llms_txt = outdir / "llms.txt"
assert llms_txt.exists(), "llms.txt should be generated even without _sources"
# Verify llms-full.txt was NOT created (since no _sources)
llms_full_txt = outdir / "llms-full.txt"
assert (
not llms_full_txt.exists()
), "llms-full.txt should not be generated without _sources"
# Read llms.txt and verify it has content
with open(llms_txt, "r", encoding="utf-8") as f:
content = f.read()
# Should contain page titles and links
assert "Home" in content
assert "About" in content
assert "index.html" in content
assert "about.html" in content
def test_llms_txt_no_warning_when_full_file_disabled(tmp_path, caplog):
"""
Test that no warning is logged when llms_txt_full_file=False and
_sources doesn't exist.
"""
from unittest.mock import patch
from sphinx_llms_txt.manager import LLMSFullManager
# Create manager
manager = LLMSFullManager()
# Set config with llms_txt_full_file=False
config = {
"llms_txt_file": True,
"llms_txt_filename": "llms.txt",
"llms_txt_full_file": False, # User doesn't want llms-full.txt
"llms_txt_full_filename": "llms-full.txt",
"llms_txt_exclude": [],
"llms_txt_directives": [],
}
manager.set_config(config)
# Create directories (but no _sources)
outdir = tmp_path / "build"
srcdir = tmp_path / "source"
outdir.mkdir()
srcdir.mkdir()
# Mock env with documents
class MockEnv:
all_docs = {"index": None}
titles = {"index": type("TitleNode", (), {"astext": lambda self: "Home"})()}
toctree_includes = {"index": []}
manager.set_env(MockEnv())
manager.set_master_doc("index")
manager.update_page_title("index", "Home")
# Capture warnings
captured_warnings = []
def capture_warning(message, *args, **kwargs):
if "_sources" in str(message):
captured_warnings.append(message)
with patch("sphinx_llms_txt.manager.logger.warning", side_effect=capture_warning):
# Call combine_sources
manager.combine_sources(str(outdir), str(srcdir))
# Verify NO warning was logged since llms_txt_full_file=False
assert (
len(captured_warnings) == 0
), "No warning should be logged when llms_txt_full_file=False"
# Verify llms.txt was still created
llms_txt = outdir / "llms.txt"
assert llms_txt.exists()