Compare commits

...
4 Commits
Author SHA1 Message Date
Jared Dillard 6962ea1f80 Merge branch 'main' into feature/add-mypy 2025-08-18 18:26:11 -07:00
Jared Dillard ad987f77f0 Fix mypy issues 2025-08-10 15:05:15 -07:00
Jared Dillard bb75a8ba77 Add mypy to pre-commit 2025-08-10 15:02:01 -07:00
Jared Dillard 8da5aa9052 Improve docstring 2025-08-10 14:53:56 -07:00
6 changed files with 55 additions and 34 deletions
+7
View File
@@ -18,6 +18,13 @@ repos:
hooks:
- id: flake8
- repo: https://github.com/pre-commit/mirrors-mypy
rev: v1.11.2
hooks:
- id: mypy
files: ^sphinx_llms_txt/
additional_dependencies: [types-docutils]
- repo: https://github.com/sphinx-contrib/sphinx-lint
rev: v1.0.0
hooks:
+1 -1
View File
@@ -107,7 +107,7 @@ def build_finished(app: Sphinx, exception):
_manager.update_page_title(docname, title)
# Create the combined file
_manager.combine_sources(app.outdir, app.srcdir)
_manager.combine_sources(str(app.outdir), str(app.srcdir))
def setup(app: Sphinx) -> Dict[str, Any]:
+22 -15
View File
@@ -3,7 +3,7 @@ Document collector module for sphinx-llms-txt.
"""
import fnmatch
from typing import Any, Dict, List, Tuple
from typing import Any, Dict, List, Optional, Tuple
from sphinx.environment import BuildEnvironment
from sphinx.util import logging
@@ -16,8 +16,8 @@ class DocumentCollector:
def __init__(self):
self.page_titles: Dict[str, str] = {}
self.master_doc: str = None
self.env: BuildEnvironment = None
self.master_doc: Optional[str] = None
self.env: Optional[BuildEnvironment] = None
self.config: Dict[str, Any] = {}
self.app = None
@@ -60,7 +60,7 @@ class DocumentCollector:
else:
return [source_suffix] # String format
def _get_docname_suffix(self, docname: str, sources_dir) -> str:
def _get_docname_suffix(self, docname: str, sources_dir) -> Optional[str]:
"""
Determine the source suffix for a given docname by checking which
file exists.
@@ -102,7 +102,7 @@ class DocumentCollector:
return None
def get_page_order(self, sources_dir=None) -> List[Tuple[str, str]]:
def get_page_order(self, sources_dir=None) -> List[Tuple[str, Optional[str]]]:
"""Get the correct page order from the toctree structure.
Args:
@@ -114,7 +114,7 @@ class DocumentCollector:
if not self.env or not self.master_doc:
return []
page_order = []
page_order: List[Tuple[str, Optional[str]]] = []
visited = set()
def collect_from_toctree(docname: str):
@@ -126,7 +126,7 @@ class DocumentCollector:
# Add the current document with its suffix
if docname not in [doc for doc, _ in page_order]:
suffix = None
suffix: Optional[str] = None
if sources_dir:
suffix = self._get_docname_suffix(docname, sources_dir)
page_order.append((docname, suffix))
@@ -135,18 +135,21 @@ class DocumentCollector:
try:
# Look for toctree_includes which contains the direct children
if (
hasattr(self.env, "toctree_includes")
self.env
and hasattr(self.env, "toctree_includes")
and docname in self.env.toctree_includes
):
for child_docname in self.env.toctree_includes[docname]:
collect_from_toctree(child_docname)
collect_from_toctree(str(child_docname))
# Try to use dependencies to find related documents
elif (
hasattr(self.env, "dependencies")
self.env
and hasattr(self.env, "dependencies")
and docname in self.env.dependencies
):
# Extract the dependent documents from the dependencies dict
for child_docname in self.env.dependencies[docname]:
for child_docname_obj in self.env.dependencies[docname]:
child_docname = str(child_docname_obj)
# Only add documents actually in the document set
if (
hasattr(self.env, "all_docs")
@@ -154,7 +157,11 @@ class DocumentCollector:
):
collect_from_toctree(child_docname)
# Fallback to titles or other available references
elif hasattr(self.env, "titles") and hasattr(self.env, "all_docs"):
elif (
self.env
and hasattr(self.env, "titles")
and hasattr(self.env, "all_docs")
):
# Get all document names
all_docnames = list(self.env.all_docs.keys())
@@ -185,7 +192,7 @@ class DocumentCollector:
]
)
for docname in remaining:
suffix = None
suffix: Optional[str] = None
if sources_dir:
suffix = self._get_docname_suffix(docname, sources_dir)
page_order.append((docname, suffix))
@@ -193,8 +200,8 @@ class DocumentCollector:
return page_order
def filter_excluded_pages(
self, page_order: List[Tuple[str, str]]
) -> List[Tuple[str, str]]:
self, page_order: List[Tuple[str, Optional[str]]]
) -> List[Tuple[str, Optional[str]]]:
"""Filter out excluded pages from the page order."""
exclude_patterns = self.config.get("llms_txt_exclude")
if exclude_patterns:
+14 -12
View File
@@ -5,7 +5,7 @@ Main manager module for sphinx-llms-txt.
import glob
import subprocess
from pathlib import Path
from typing import Any, Dict, List, Optional, Tuple, Union
from typing import Any, Dict, List, Optional, Tuple, Union, cast
from sphinx.application import Sphinx
from sphinx.environment import BuildEnvironment
@@ -150,8 +150,8 @@ class LLMSFullManager:
self.ignored_pages.add(docname)
def _filter_ignored_pages(
self, page_order: Union[List[str], List[Tuple[str, str]]]
) -> Union[List[str], List[Tuple[str, str]]]:
self, page_order: Union[List[str], List[Tuple[str, Optional[str]]]]
) -> Union[List[str], List[Tuple[str, Optional[str]]]]:
"""Filter out ignored pages from page_order."""
filtered_pages = []
for item in page_order:
@@ -164,7 +164,7 @@ class LLMSFullManager:
if docname not in self.ignored_pages:
filtered_pages.append(item)
return filtered_pages
return cast(Union[List[str], List[Tuple[str, Optional[str]]]], filtered_pages)
def set_config(self, config: Dict[str, Any]):
"""Set configuration options."""
@@ -225,7 +225,7 @@ class LLMSFullManager:
# Determine output file name and location
output_filename = self.config.get("llms_txt_full_filename")
output_path = Path(outdir) / output_filename
output_path = Path(outdir) / str(output_filename)
# Log discovered files and page order
logger.debug(f"sphinx-llms-txt: Page order (after exclusion): {page_order}")
@@ -286,7 +286,7 @@ class LLMSFullManager:
content_parts = []
# Track code files for later processing
code_file_parts = []
code_file_parts: List[str] = []
# Count lines in code files (initially 0)
code_files_line_count = 0
@@ -365,7 +365,7 @@ class LLMSFullManager:
if not (size_limit_exceeded and should_abort_early):
# Get all source files in the _sources directory using configured suffixes
source_suffixes = self._get_source_suffixes()
all_source_files = []
all_source_files: List[Path] = []
for src_suffix in source_suffixes:
# Avoid duplicate extensions when source_suffix == source_link_suffix
if src_suffix == source_link_suffix:
@@ -735,9 +735,9 @@ class LLMSFullManager:
title = Path(title_str[len(base_path) :])
except ValueError:
# File is not relative to srcdir, use filename
title = file_path.name
title = Path(file_path.name)
else:
title = file_path.name
title = Path(file_path.name)
# Format as code block with equals underline
title_str = str(title)
@@ -769,7 +769,9 @@ class LLMSFullManager:
return code_parts, sorted(processed_files)
def _create_code_files_section_header(self, file_paths: List[Path] = None) -> str:
def _create_code_files_section_header(
self, file_paths: Optional[List[Path]] = None
) -> str:
"""Create the section header for source code files.
Args:
@@ -817,7 +819,7 @@ class LLMSFullManager:
return ""
# Convert to relative paths if possible and create tree structure
tree_data = {}
tree_data: Dict[str, Any] = {}
for file_path in sorted(file_paths):
# Get relative path from source directory for display
@@ -870,7 +872,7 @@ class LLMSFullManager:
current[parts[-1]] = None # None indicates it's a file
# Convert tree structure to string representation
lines = []
lines: List[str] = []
self._format_tree_node(tree_data, lines, "", True)
# Indent each line for reStructuredText code block
+1 -1
View File
@@ -130,7 +130,7 @@ class DocumentProcessor:
"""
# Get the configured path directives to process
default_path_directives = ["image", "figure"]
custom_path_directives = self.config.get("llms_txt_directives")
custom_path_directives = self.config.get("llms_txt_directives") or []
path_directives = set(default_path_directives + custom_path_directives)
# Build the regex pattern to match all configured directives
+10 -5
View File
@@ -3,7 +3,7 @@ File writer module for sphinx-llms-txt.
"""
from pathlib import Path
from typing import Any, Dict, List, Tuple, Union
from typing import Any, Dict, List, Optional, Tuple, Union
from sphinx.application import Sphinx
from sphinx.util import logging
@@ -14,7 +14,12 @@ logger = logging.getLogger(__name__)
class FileWriter:
"""Handles writing processed content to output files."""
def __init__(self, config: Dict[str, Any], outdir: str = None, app: Sphinx = None):
def __init__(
self,
config: Dict[str, Any],
outdir: Optional[str] = None,
app: Optional[Sphinx] = None,
):
self.config = config
self.outdir = outdir
self.app = app
@@ -47,7 +52,7 @@ class FileWriter:
def write_verbose_info_to_file(
self,
page_order: Union[List[str], List[Tuple[str, str]]],
page_order: Union[List[str], List[Tuple[str, Optional[str]]]],
page_titles: Dict[str, str],
total_line_count: int = 0,
) -> bool:
@@ -67,13 +72,13 @@ class FileWriter:
)
return False
output_path = Path(self.outdir) / self.config.get("llms_txt_filename")
output_path = Path(self.outdir) / str(self.config.get("llms_txt_filename"))
try:
with open(output_path, "w", encoding="utf-8") as f:
project_name = "llms-txt Summary"
# First priority: use title from config if available
if self.config.get("llms_txt_title"):
project_name = self.config.get("llms_txt_title")
project_name = str(self.config.get("llms_txt_title"))
# Second priority: use project name from Sphinx app if available
elif (
self.app