Support source file suffix detection (#21)

This commit is contained in:
Jared Dillard
2025-06-22 19:28:54 -07:00
committed by GitHub
parent 5db5395889
commit b56d93d265
6 changed files with 585 additions and 55 deletions
+1 -1
View File
@@ -12,7 +12,7 @@ from .manager import LLMSFullManager
from .processor import DocumentProcessor
from .writer import FileWriter
__version__ = "0.2.3"
__version__ = "0.2.4"
# Export classes needed by tests
__all__ = [
+92 -12
View File
@@ -3,7 +3,7 @@ Document collector module for sphinx-llms-txt.
"""
import fnmatch
from typing import Any, Dict, List
from typing import Any, Dict, List, Tuple
from sphinx.environment import BuildEnvironment
from sphinx.util import logging
@@ -19,6 +19,7 @@ class DocumentCollector:
self.master_doc: str = None
self.env: BuildEnvironment = None
self.config: Dict[str, Any] = {}
self.app = None
def set_master_doc(self, master_doc: str):
"""Set the master document name."""
@@ -37,8 +38,73 @@ class DocumentCollector:
"""Set configuration options."""
self.config = config
def get_page_order(self) -> List[str]:
"""Get the correct page order from the toctree structure."""
def set_app(self, app):
"""Set the Sphinx application reference."""
self.app = app
def _get_source_suffixes(self):
"""Get all valid source file suffixes from Sphinx configuration.
Returns:
list: List of source file suffixes (e.g., ['.rst', '.md', '.txt'])
"""
if not self.app:
return [".rst"] # Default fallback
source_suffix = self.app.config.source_suffix
if isinstance(source_suffix, dict):
return list(source_suffix.keys())
elif isinstance(source_suffix, list):
return source_suffix
else:
return [source_suffix] # String format
def _get_docname_suffix(self, docname: str, sources_dir) -> str:
"""
Determine the source suffix for a given docname by checking which
file exists.
Args:
docname: The document name to check
sources_dir: Path to the _sources directory
Returns:
The source suffix if found, or None if no matching file exists
"""
if not sources_dir or not sources_dir.exists():
return None
# Get the source link suffix from Sphinx config
source_link_suffix = ""
if self.app and hasattr(self.app.config, "html_sourcelink_suffix"):
source_link_suffix = self.app.config.html_sourcelink_suffix
# Handle empty string case specially
if source_link_suffix == "":
source_link_suffix = "" # Keep it empty
elif not source_link_suffix.startswith("."):
source_link_suffix = "." + source_link_suffix
# Get the source file suffixes from Sphinx config
source_suffixes = self._get_source_suffixes()
# Try to find the source file with any of the valid source suffixes
for src_suffix in source_suffixes:
candidate_file = sources_dir / f"{docname}{src_suffix}{source_link_suffix}"
if candidate_file.exists():
return src_suffix
return None
def get_page_order(self, sources_dir=None) -> List[Tuple[str, str]]:
"""Get the correct page order from the toctree structure.
Args:
sources_dir: Optional path to _sources directory for suffix detection
Returns:
List of tuples (docname, source_suffix) in toctree order
"""
if not self.env or not self.master_doc:
return []
@@ -52,9 +118,12 @@ class DocumentCollector:
visited.add(docname)
# Add the current document
if docname not in page_order:
page_order.append(docname)
# Add the current document with its suffix
if docname not in [doc for doc, _ in page_order]:
suffix = None
if sources_dir:
suffix = self._get_docname_suffix(docname, sources_dir)
page_order.append((docname, suffix))
# Check for toctree entries in this document
try:
@@ -101,22 +170,33 @@ class DocumentCollector:
# Add any remaining documents not in the toctree (sorted)
if hasattr(self.env, "all_docs"):
processed_docnames = {doc for doc, _ in page_order}
remaining = sorted(
[doc for doc in self.env.all_docs.keys() if doc not in page_order]
[
doc
for doc in self.env.all_docs.keys()
if doc not in processed_docnames
]
)
page_order.extend(remaining)
for docname in remaining:
suffix = None
if sources_dir:
suffix = self._get_docname_suffix(docname, sources_dir)
page_order.append((docname, suffix))
return page_order
def filter_excluded_pages(self, page_order: List[str]) -> List[str]:
def filter_excluded_pages(
self, page_order: List[Tuple[str, str]]
) -> List[Tuple[str, str]]:
"""Filter out excluded pages from the page order."""
exclude_patterns = self.config.get("llms_txt_exclude")
if exclude_patterns:
return [
page
for page in page_order
(docname, suffix)
for docname, suffix in page_order
if not any(
self._match_exclude_pattern(page, pattern)
self._match_exclude_pattern(docname, pattern)
for pattern in exclude_patterns
)
]
+83 -37
View File
@@ -56,6 +56,7 @@ class LLMSFullManager:
def set_app(self, app: Sphinx):
"""Set the Sphinx application reference."""
self.app = app
self.collector.set_app(app)
if self.writer:
self.writer.app = app
@@ -69,23 +70,7 @@ class LLMSFullManager:
self.processor = DocumentProcessor(self.config, srcdir)
self.writer = FileWriter(self.config, outdir, self.app)
# Get the correct page order
page_order = self.collector.get_page_order()
if not page_order:
logger.warning(
"Could not determine page order, skipping llms-full creation"
)
return
# Apply exclusion filter if configured
page_order = self.collector.filter_excluded_pages(page_order)
# Determine output file name and location
output_filename = self.config.get("llms_txt_full_filename")
output_path = Path(outdir) / output_filename
# Find sources directory
# Find sources directory first so we can pass it to get_page_order
sources_dir = None
possible_sources = [
Path(outdir) / "_sources",
@@ -104,6 +89,22 @@ class LLMSFullManager:
)
return
# Get the correct page order with source suffixes
page_order = self.collector.get_page_order(sources_dir)
if not page_order:
logger.warning(
"Could not determine page order, skipping llms-full creation"
)
return
# Apply exclusion filter if configured
page_order = self.collector.filter_excluded_pages(page_order)
# Determine output file name and location
output_filename = self.config.get("llms_txt_full_filename")
output_path = Path(outdir) / output_filename
# Log discovered files and page order
logger.debug(f"sphinx-llms-txt: Page order (after exclusion): {page_order}")
@@ -115,8 +116,19 @@ class LLMSFullManager:
# Create a mapping from docnames to source files
docname_to_file = {}
# Process each docname in the page order
for docname in page_order:
# Get the source link suffix from Sphinx config
source_link_suffix = (
self.app.config.html_sourcelink_suffix if self.app else ".txt"
)
# Handle empty string case specially
if source_link_suffix == "":
source_link_suffix = "" # Keep it empty
elif not source_link_suffix.startswith("."):
source_link_suffix = "." + source_link_suffix
# Process each (docname, suffix) in the page order
for docname, src_suffix in page_order:
# Skip excluded pages
if exclude_patterns and any(
self.collector._match_exclude_pattern(docname, pattern)
@@ -124,15 +136,19 @@ class LLMSFullManager:
):
continue
# Construct expected source file path directly from docname
source_file = sources_dir / f"{docname}.rst.txt"
if source_file.exists():
docname_to_file[docname] = source_file
# Build the source file path directly using the known suffix
if src_suffix:
source_file = sources_dir / f"{docname}{src_suffix}{source_link_suffix}"
if source_file.exists():
docname_to_file[docname] = source_file
else:
logger.warning(
f"sphinx-llms-txt: Source file not found for: {docname}."
f"Expected: {docname}{src_suffix}{source_link_suffix}"
)
else:
logger.warning(
f"sphinx-llm-txt: Source file not found for: {docname}. Expected"
f" at {source_file}"
f"sphinx-llms-txt: No source suffix determined for: {docname}"
)
# Generate content
@@ -144,7 +160,7 @@ class LLMSFullManager:
max_lines = self.config.get("llms_txt_full_max_size")
abort_due_to_max_lines = False
for docname in page_order:
for docname, _ in page_order:
if docname in docname_to_file:
file_path = docname_to_file[docname]
content, line_count = self._read_source_file(file_path, docname)
@@ -179,14 +195,19 @@ class LLMSFullManager:
total_line_count += line_count
else:
logger.warning(
f"sphinx-llm-txt: Source file not found for: {docname}. Check that"
f" the file exists at _sources/{docname}.rst.txt"
f"sphinx-llms-txt: Source file not found for: {docname}. Check that"
f" file exists at _sources/{docname}[suffix]{source_link_suffix}"
)
# Add any remaining files (in alphabetical order) that aren't in the page order
if not abort_due_to_max_lines:
# Get all .rst.txt files in the _sources directory
all_source_files = list(sources_dir.glob("**/*.rst.txt"))
# Get all source files in the _sources directory using configured suffixes
source_suffixes = self._get_source_suffixes()
all_source_files = []
for src_suffix in source_suffixes:
glob_pattern = f"**/*{src_suffix}{source_link_suffix}"
all_source_files.extend(sources_dir.glob(glob_pattern))
processed_paths = set(file.resolve() for file in docname_to_file.values())
# Find files that haven't been processed yet
@@ -204,11 +225,18 @@ class LLMSFullManager:
)
for file_path in remaining_source_files:
# Extract docname from path by removing the .rst.txt extension
# Extract docname from path by removing the source and link suffixes
rel_path = str(file_path.relative_to(sources_dir))
if rel_path.endswith(".rst.txt"):
docname = rel_path[:-8] # Remove .rst.txt extension
else:
docname = None
# Try each source suffix to find which one this file uses
for src_suffix in source_suffixes:
combined_suffix = f"{src_suffix}{source_link_suffix}"
if rel_path.endswith(combined_suffix):
docname = rel_path[: -len(combined_suffix)] # Remove suffix
break
if docname is None:
continue
# Skip excluded docnames
@@ -237,7 +265,7 @@ class LLMSFullManager:
max_lines is not None and total_line_count > max_lines
):
logger.warning(
f"sphinx-llm-txt: Max line limit ({max_lines}) exceeded:"
f"sphinx-llms-txt: Max line limit ({max_lines}) exceeded:"
f" {total_line_count} > {max_lines}. "
f"Not creating llms-full.txt file."
)
@@ -304,5 +332,23 @@ class LLMSFullManager:
return content_str, line_count + 1
except Exception as e:
logger.error(f"sphinx-llm-txt: Error reading source file {file_path}: {e}")
logger.error(f"sphinx-llms-txt: Error reading source file {file_path}: {e}")
return "", 0
def _get_source_suffixes(self):
"""Get all valid source file suffixes from Sphinx configuration.
Returns:
list: List of source file suffixes (e.g., ['.rst', '.md', '.txt'])
"""
if not self.app:
return [".rst"] # Default fallback
source_suffix = self.app.config.source_suffix
if isinstance(source_suffix, dict):
return list(source_suffix.keys())
elif isinstance(source_suffix, list):
return source_suffix
else:
return [source_suffix] # String format
+10 -5
View File
@@ -3,7 +3,7 @@ File writer module for sphinx-llms-txt.
"""
from pathlib import Path
from typing import Any, Dict, List
from typing import Any, Dict, List, Tuple, Union
from sphinx.application import Sphinx
from sphinx.util import logging
@@ -42,19 +42,19 @@ class FileWriter:
)
return True
except Exception as e:
logger.error(f"sphinx-llm-txt: Error writing combined sources file: {e}")
logger.error(f"sphinx-llms-txt: Error writing combined sources file: {e}")
return False
def write_verbose_info_to_file(
self,
page_order: List[str],
page_order: Union[List[str], List[Tuple[str, str]]],
page_titles: Dict[str, str],
total_line_count: int = 0,
) -> bool:
"""Write summary information to the llms.txt file.
Args:
page_order: Ordered list of document names
page_order: Ordered list of document names or (docname, suffix) tuples
page_titles: Dictionary mapping docnames to titles
total_line_count: Total number of lines in the combined content
@@ -100,7 +100,12 @@ class FileWriter:
if not base_url.endswith("/"):
base_url += "/"
for docname in page_order:
for item in page_order:
# Handle both old format (str) and new format (tuple)
if isinstance(item, tuple):
docname, _ = item
else:
docname = item
title = page_titles.get(docname, docname)
f.write(f"- [{title}]({base_url}{docname}.html)\n")