diff --git a/CHANGELOG.rst b/CHANGELOG.rst index 9f7a4fd..fa57a0f 100644 --- a/CHANGELOG.rst +++ b/CHANGELOG.rst @@ -1,6 +1,11 @@ Changelog ========= +0.2.2 +----- + +- Refactor LLMSFullManager with clearer class structure + 0.2.1 ----- diff --git a/sphinx_llms_txt/__init__.py b/sphinx_llms_txt/__init__.py index f627a9e..5a826f3 100644 --- a/sphinx_llms_txt/__init__.py +++ b/sphinx_llms_txt/__init__.py @@ -2,654 +2,25 @@ Sphinx extension to create a combined sources file (llms-full.txt) """ -import os -import re -from pathlib import Path -from typing import Any, Dict, List, Optional +from typing import Any, Dict +from docutils import nodes from sphinx.application import Sphinx -from sphinx.environment import BuildEnvironment -from sphinx.util import logging -__version__ = "0.2.1" +from .collector import DocumentCollector +from .manager import LLMSFullManager +from .processor import DocumentProcessor +from .writer import FileWriter -logger = logging.getLogger(__name__) - - -class LLMSFullManager: - """Manages the collection and ordering of documentation sources.""" - - def __init__(self): - self.page_titles: Dict[str, str] = {} - self.config: Dict[str, Any] = {} - self.master_doc: str = None - self.env: BuildEnvironment = None - self.srcdir: Optional[str] = None - self.outdir: Optional[str] = None - self.app: Optional[Sphinx] = None - - def set_master_doc(self, master_doc: str): - """Set the master document name.""" - self.master_doc = master_doc - - def set_env(self, env: BuildEnvironment): - """Set the Sphinx environment.""" - self.env = env - - def update_page_title(self, docname: str, title: str): - """Update the title for a page.""" - if title: - self.page_titles[docname] = title - - def set_config(self, config: Dict[str, Any]): - """Set configuration options.""" - self.config = config - - def set_app(self, app: Sphinx): - """Set the Sphinx application reference.""" - self.app = app - - def get_page_order(self) -> List[str]: - """Get the correct page order from the toctree structure.""" - if not self.env or not self.master_doc: - return [] - - page_order = [] - visited = set() - - def collect_from_toctree(docname: str): - """Recursively collect documents from toctree.""" - if docname in visited: - return - - visited.add(docname) - - # Add the current document - if docname not in page_order: - page_order.append(docname) - - # Check for toctree entries in this document - try: - # Look for toctree_includes which contains the direct children - if ( - hasattr(self.env, "toctree_includes") - and docname in self.env.toctree_includes - ): - for child_docname in self.env.toctree_includes[docname]: - collect_from_toctree(child_docname) - else: - # Fallback: try to resolve and parse the toctree - toctree = self.env.get_and_resolve_toctree(docname, None) - if toctree: - from docutils import nodes - - for node in list(toctree.findall(nodes.reference)): - if "refuri" in node.attributes: - refuri = node.attributes["refuri"] - if refuri and refuri.endswith(".html"): - child_docname = refuri[:-5] # Remove .html - if ( - child_docname != docname - ): # Avoid circular references - collect_from_toctree(child_docname) - except Exception as e: - logger.debug(f"Could not get toctree for {docname}: {e}") - - # Start from the master document - collect_from_toctree(self.master_doc) - - # Add any remaining documents not in the toctree (sorted) - if hasattr(self.env, "all_docs"): - remaining = sorted( - [doc for doc in self.env.all_docs.keys() if doc not in page_order] - ) - page_order.extend(remaining) - - return page_order - - def combine_sources(self, outdir: str, srcdir: str): - """Combine all source files into a single file.""" - # Store the source directory for resolving include directives - self.srcdir = srcdir - self.outdir = outdir - - # Get the correct page order - page_order = self.get_page_order() - - if not page_order: - logger.warning( - "Could not determine page order, skipping llms-full creation" - ) - return - - # Apply exclusion filter if configured - exclude_patterns = self.config.get("llms_txt_exclude") - if exclude_patterns: - page_order = [ - page - for page in page_order - if not any( - self._match_exclude_pattern(page, pattern) - for pattern in exclude_patterns - ) - ] - - # Determine output file name and location - output_filename = self.config.get("llms_txt_full_filename") - output_path = Path(outdir) / output_filename - - # Find sources directory - sources_dir = None - possible_sources = [ - Path(outdir) / "_sources", - Path(outdir) / "html" / "_sources", - Path(outdir) / "singlehtml" / "_sources", - ] - - for path in possible_sources: - if path.exists(): - sources_dir = path - break - - if not sources_dir: - logger.warning( - "Could not find _sources directory, skipping llms-full creation" - ) - return - - # Collect all available source files - txt_files = {} - for f in sources_dir.glob("*.txt"): - logger.debug(f"sphinx-llms-txt: Found source file: {f.stem} at {f}") - txt_files[f.stem] = f - - # Log discovered files and page order - logger.debug(f"sphinx-llms-txt: Found {len(txt_files)} source files") - logger.debug(f"sphinx-llms-txt: Page order (after exclusion): {page_order}") - - # Log exclusion patterns - exclude_patterns = self.config.get("llms_txt_exclude") - if exclude_patterns: - logger.debug(f"sphinx-llms-txt: Exclusion patterns: {exclude_patterns}") - - # Create a mapping from docnames to actual file names - docname_to_file = {} - - # Try exact matches first - for docname in page_order: - # Skip excluded pages - if any( - self._match_exclude_pattern(docname, pattern) - for pattern in exclude_patterns - ): - continue - - if docname in txt_files: - docname_to_file[docname] = txt_files[docname] - else: - # Try with .rst extension - if f"{docname}.rst" in txt_files: - docname_to_file[docname] = txt_files[f"{docname}.rst"] - # Try with .txt extension - elif f"{docname}.txt" in txt_files: - docname_to_file[docname] = txt_files[f"{docname}.txt"] - # Try with underscores instead of hyphens - elif docname.replace("-", "_") in txt_files: - docname_to_file[docname] = txt_files[docname.replace("-", "_")] - # Try with hyphens instead of underscores - elif docname.replace("_", "-") in txt_files: - docname_to_file[docname] = txt_files[docname.replace("_", "-")] - - # Generate content - content_parts = [] - - # Add pages in order - added_files = set() - total_line_count = 0 - max_lines = self.config.get("llms_txt_full_max_size") - abort_due_to_max_lines = False - - for docname in page_order: - if docname in docname_to_file: - file_path = docname_to_file[docname] - content, line_count = self._read_source_file(file_path, docname) - - # Check if adding this file would exceed the maximum line count - if max_lines is not None and total_line_count + line_count > max_lines: - abort_due_to_max_lines = True - break - - # Double-check this file should be included (not in excluded patterns) - exclude_patterns = self.config.get("llms_txt_exclude") - file_stem = file_path.stem - should_include = True - - if exclude_patterns: - # Check stem and docname against exclusion patterns - if any( - self._match_exclude_pattern(file_stem, pattern) - for pattern in exclude_patterns - ) or any( - self._match_exclude_pattern(docname, pattern) - for pattern in exclude_patterns - ): - logger.debug( - f"sphinx-llms-txt: Final exclusion check removed: {docname}" - ) - should_include = False - - if content and should_include: - content_parts.append(content) - added_files.add(file_path.stem) - total_line_count += line_count - else: - logger.warning(f"sphinx-llm-txt: Source file not found for: {docname}") - - # Add any remaining files (in alphabetical order) if not aborted - if not abort_due_to_max_lines: - # Apply the same exclusion filter to remaining files - exclude_patterns = self.config.get("llms_txt_exclude") - - # Create a set of files to exclude based on their basename - excluded_files = set() - for pattern in exclude_patterns: - if "*" not in pattern and "?" not in pattern: - # For exact patterns, add variants - excluded_files.add(pattern) - excluded_files.add(f"{pattern}.rst") - excluded_files.add(f"{pattern}.txt") - excluded_files.add(pattern.replace("-", "_")) - excluded_files.add(pattern.replace("_", "-")) - - # Filter remaining files - remaining_files = sorted( - [ - name - for name in txt_files - if name not in added_files - and name not in excluded_files - and not any( - self._match_exclude_pattern(name, pattern) - for pattern in exclude_patterns - ) - ] - ) - if remaining_files: - logger.info(f"Adding remaining files: {remaining_files}") - for file_stem in remaining_files: - file_path = txt_files[file_stem] - content, line_count = self._read_source_file(file_path, file_stem) - - # Check if adding this file would exceed the maximum line count - if max_lines is not None and total_line_count + line_count > max_lines: - break - - # Double-check that this file should be included - should_include = True - file_stem = file_path.stem - exclude_patterns = self.config.get("llms_txt_exclude") - - if exclude_patterns: - # Check stem against exclusion patterns - if any( - self._match_exclude_pattern(file_stem, pattern) - for pattern in exclude_patterns - ): - logger.debug( - "sphinx-llms-txt: Final exclusion check removed remaining" - f" file: {file_stem}" - ) - should_include = False - - if content and should_include: - content_parts.append(content) - total_line_count += line_count - - # Check if line limit was exceeded before creating the file - max_lines = self.config.get("llms_txt_full_max_size") - if abort_due_to_max_lines or ( - max_lines is not None and total_line_count > max_lines - ): - logger.warning( - f"sphinx-llm-txt: Max line limit ({max_lines}) exceeded:" - f" {total_line_count} > {max_lines}. " - f"Not creating llms-full.txt file." - ) - - # Log summary information if requested - if self.config.get("llms_txt_file"): - self._write_verbose_info_to_file(page_order, total_line_count) - - return - - # Write combined file if limit wasn't exceeded - try: - with open(output_path, "w", encoding="utf-8") as f: - f.write("\n".join(content_parts)) - - logger.info( - f"sphinx-llms-txt: created {output_path} with {len(txt_files)}" - f" sources and {total_line_count} lines" - ) - - # Log summary information if requested - if self.config.get("llms_txt_file"): - self._write_verbose_info_to_file(page_order, total_line_count) - - except Exception as e: - logger.error(f"sphinx-llm-txt: Error writing combined sources file: {e}") - - def _read_source_file(self, file_path: Path, docname: str) -> tuple: - """Read and format a single source file. - - Handles include directives by replacing them with the content of the included - file, and processes directives with paths that need to be resolved. - - Returns: - tuple: (content_str, line_count) where line_count is the number of lines - in the file - """ - # Check if this file should be excluded by looking at the doc name - exclude_patterns = self.config.get("llms_txt_exclude") - if exclude_patterns and any( - self._match_exclude_pattern(docname, pattern) - for pattern in exclude_patterns - ): - return "", 0 - - try: - # Check if the file stem (without extension) should be excluded - file_stem = file_path.stem - if exclude_patterns and any( - self._match_exclude_pattern(file_stem, pattern) - for pattern in exclude_patterns - ): - return "", 0 - - with open(file_path, "r", encoding="utf-8") as f: - content = f.read() - - # Process include directives and directives with paths - content = self._process_content(content, file_path) - - # Count the lines in the content - line_count = content.count("\n") + (0 if content.endswith("\n") else 1) - - section_lines = [content, ""] - content_str = "\n".join(section_lines) - - # Add 2 for the section_lines (content + empty line) - return content_str, line_count + 1 - - except Exception as e: - logger.error(f"sphinx-llm-txt: Error reading source file {file_path}: {e}") - return "", 0 - - def _process_content(self, content: str, source_path: Path) -> str: - """Process directives in content that need path resolution. - - Args: - content: The source content to process - source_path: Path to the source file (to resolve relative paths) - - Returns: - Processed content with directives properly resolved - """ - # First process include directives - content = self._process_includes(content, source_path) - - # Then process path directives (image, figure, etc.) - content = self._process_path_directives(content, source_path) - - return content - - def _process_path_directives(self, content: str, source_path: Path) -> str: - """Process directives with paths that need to be resolved. - - Args: - content: The source content to process - source_path: Path to the source file (to resolve relative paths) - - Returns: - Processed content with directive paths properly resolved - """ - # Get the configured path directives to process - default_path_directives = ["image", "figure"] - custom_path_directives = self.config.get("llms_txt_directives") - path_directives = set(default_path_directives + custom_path_directives) - - # Build the regex pattern to match all configured directives - directives_pattern = "|".join(re.escape(d) for d in path_directives) - directive_pattern = re.compile( - r"^(\s*\.\.\s+(" + directives_pattern + r")::\s+)([^\s].+?)$", re.MULTILINE - ) - - # Get the base URL from Sphinx's html_baseurl if set - base_url = self.config.get("html_baseurl", "") - - # Handle test case specially - is_test = "pytest" in str(source_path) and "subdir" in str(source_path) - - def replace_directive_path(match, base_url=base_url, is_test=is_test): - prefix = match.group(1) # The entire directive prefix including whitespace - path = match.group(3).strip() # The path argument - - # Only process relative paths, not absolute paths or URLs - if not path.startswith(("http://", "https://", "/", "data:")): - # Special case for test files - if is_test: - # Add subdir/ prefix to match test expectations - full_path = "subdir/" + path - - # If base_url is set, prepend it to the path - if base_url: - if not base_url.endswith("/"): - base_url += "/" - full_path = f"{base_url}{full_path}" - - # Return the updated directive with the full path - return f"{prefix}{full_path}" - - # Production case (not in test) - elif "_sources" in str(source_path): - # Extract the part after _sources/ - try: - path_parts = str(source_path).split("_sources/") - if len(path_parts) > 1: - rel_doc_path = path_parts[1] - # Remove .txt extension if present - if rel_doc_path.endswith(".txt"): - rel_doc_path = rel_doc_path[:-4] - # Get the directory containing the current document - rel_doc_dir = os.path.dirname(rel_doc_path) - rel_doc_path_parts = rel_doc_path.split("/") - - # For test subdirectory handling - this is for our test - # cases - if ( - len(rel_doc_path_parts) > 0 - and rel_doc_path_parts[0] == "subdir" - ): - full_path = os.path.normpath( - os.path.join("subdir", path) - ) - # Only add the rel_doc_dir if it's not empty - elif rel_doc_dir: - # Join with the original path to form full path - # relative to srcdir - full_path = os.path.normpath( - os.path.join(rel_doc_dir, path) - ) - else: - full_path = path - - # If base_url is set, prepend it to the path - if base_url: - if not base_url.endswith("/"): - base_url += "/" - full_path = f"{base_url}{full_path}" - - # Return the updated directive with the full path - return f"{prefix}{full_path}" - except Exception as e: - logger.debug( - f"sphinx-llms-txt: Error resolving path {path}: {e}" - ) - - # If we couldn't resolve the path or it's already absolute, return unchanged - return match.group(0) - - # Replace directive paths in the content - processed_content = directive_pattern.sub(replace_directive_path, content) - return processed_content - - def _process_includes(self, content: str, source_path: Path) -> str: - """Process include directives in content. - - Args: - content: The source content to process - source_path: Path to the source file (to resolve relative paths) - - Returns: - Processed content with include directives replaced with included content - """ - # Find all include directives using regex - include_pattern = re.compile(r"^\.\.\s+include::\s+([^\s]+)\s*$", re.MULTILINE) - - # Function to replace each include with content - def replace_include(match): - include_path = match.group(1) - - # Try multiple possible paths for the include file - possible_paths = [] - - # If it's an absolute path, use it directly - if os.path.isabs(include_path): - possible_paths.append(Path(include_path)) - else: - # Relative to the source file (in _sources directory) - possible_paths.append((source_path.parent / include_path).resolve()) - - # If we're in _sources directory, try relative to the original source - # directory - if "_sources" in str(source_path): - # Extract the relative path portion from the source path - rel_path = None - try: - # Get the part after _sources/ - path_parts = str(source_path).split("_sources/") - if len(path_parts) > 1: - rel_path = path_parts[1] - # Remove .txt extension if present - if rel_path.endswith(".txt"): - rel_path = rel_path[:-4] - except Exception: - pass - - # If we have the original source directory from Sphinx - if hasattr(self, "srcdir") and self.srcdir: - # Try in the srcdir root - possible_paths.append( - (Path(self.srcdir) / include_path).resolve() - ) - - # If we have a relative path, try in the corresponding source - # subdirectory - if rel_path: - rel_dir = os.path.dirname(rel_path) - if rel_dir: - possible_paths.append( - ( - Path(self.srcdir) / rel_dir / include_path - ).resolve() - ) - - # Try each possible path - for path_to_try in possible_paths: - try: - if path_to_try.exists(): - with open(path_to_try, "r", encoding="utf-8") as f: - included_content = f.read() - return included_content - except Exception as e: - logger.error( - f"sphinx-llms-txt: Error reading include file {path_to_try}:" - f" {e}" - ) - continue - - # If we get here, we couldn't find the file - paths_tried = ", ".join(str(p) for p in possible_paths) - logger.warning(f"sphinx-llms-txt: Include file not found: {include_path}") - logger.debug(f"sphinx-llms-txt: Tried paths: {paths_tried}") - return f"[Include file not found: {include_path}]" - - # Replace all includes with their content - processed_content = include_pattern.sub(replace_include, content) - return processed_content - - def _match_exclude_pattern(self, docname: str, pattern: str) -> bool: - """Check if a document name matches an exclude pattern. - - Args: - docname: The document name to check - pattern: The pattern to match against - - Returns: - True if the document should be excluded, False otherwise - """ - # Exact match - if docname == pattern: - return True - - # Glob-style pattern matching - import fnmatch - - if fnmatch.fnmatch(docname, pattern): - return True - - return False - - def _write_verbose_info_to_file( - self, page_order: List[str], total_line_count: int = 0 - ): - """Write summary information to the llms.txt file.""" - if not self.outdir: - logger.warning( - "sphinx-llms-txt: Cannot write verbose info to file: outdir not set" - ) - return - - output_path = Path(self.outdir) / self.config.get("llms_txt_filename") - try: - with open(output_path, "w", encoding="utf-8") as f: - project_name = "llms-txt Summary" - # First priority: use title from config if available - if self.config.get("llms_txt_title"): - project_name = self.config.get("llms_txt_title") - # Second priority: use project name from Sphinx app if available - elif ( - self.app - and hasattr(self.app, "config") - and hasattr(self.app.config, "project") - ): - project_name = self.app.config.project - f.write(f"# {project_name}\n\n") - - # Add description if available - description = self.config.get("llms_txt_summary", "") - if description: - f.write(f"> {description}\n\n") - - f.write("## Docs\n\n") - for i, docname in enumerate(page_order, 1): - title = self.page_titles.get(docname, docname) - f.write(f"- [{title}](/{docname}.html)\n") - - logger.info(f"sphinx-llms-txt: created {output_path}") - except Exception as e: - logger.error(f"sphinx-llms-txt: Error writing verbose info to file: {e}") +__version__ = "0.2.2" +# Export classes needed by tests +__all__ = [ + "DocumentCollector", + "DocumentProcessor", + "FileWriter", + "LLMSFullManager", +] # Global manager instance _manager = LLMSFullManager() @@ -658,8 +29,6 @@ _manager = LLMSFullManager() def doctree_resolved(app: Sphinx, doctree, docname: str): """Called when a docname has been resolved to a document.""" # Extract title from the document - from docutils import nodes - title = None # findall() returns a generator, convert to list to check if it has elements title_nodes = list(doctree.findall(nodes.title)) diff --git a/sphinx_llms_txt/collector.py b/sphinx_llms_txt/collector.py new file mode 100644 index 0000000..6ee0091 --- /dev/null +++ b/sphinx_llms_txt/collector.py @@ -0,0 +1,130 @@ +""" +Document collector module for sphinx-llms-txt. +""" + +import fnmatch +from typing import Any, Dict, List + +from sphinx.environment import BuildEnvironment +from sphinx.util import logging + +logger = logging.getLogger(__name__) + + +class DocumentCollector: + """Collects and orders documentation sources based on toctree structure.""" + + def __init__(self): + self.page_titles: Dict[str, str] = {} + self.master_doc: str = None + self.env: BuildEnvironment = None + self.config: Dict[str, Any] = {} + + def set_master_doc(self, master_doc: str): + """Set the master document name.""" + self.master_doc = master_doc + + def set_env(self, env: BuildEnvironment): + """Set the Sphinx environment.""" + self.env = env + + def update_page_title(self, docname: str, title: str): + """Update the title for a page.""" + if title: + self.page_titles[docname] = title + + def set_config(self, config: Dict[str, Any]): + """Set configuration options.""" + self.config = config + + def get_page_order(self) -> List[str]: + """Get the correct page order from the toctree structure.""" + if not self.env or not self.master_doc: + return [] + + page_order = [] + visited = set() + + def collect_from_toctree(docname: str): + """Recursively collect documents from toctree.""" + if docname in visited: + return + + visited.add(docname) + + # Add the current document + if docname not in page_order: + page_order.append(docname) + + # Check for toctree entries in this document + try: + # Look for toctree_includes which contains the direct children + if ( + hasattr(self.env, "toctree_includes") + and docname in self.env.toctree_includes + ): + for child_docname in self.env.toctree_includes[docname]: + collect_from_toctree(child_docname) + else: + # Fallback: try to resolve and parse the toctree + toctree = self.env.get_and_resolve_toctree(docname, None) + if toctree: + from docutils import nodes + + for node in list(toctree.findall(nodes.reference)): + if "refuri" in node.attributes: + refuri = node.attributes["refuri"] + if refuri and refuri.endswith(".html"): + child_docname = refuri[:-5] # Remove .html + if ( + child_docname != docname + ): # Avoid circular references + collect_from_toctree(child_docname) + except Exception as e: + logger.debug(f"Could not get toctree for {docname}: {e}") + + # Start from the master document + collect_from_toctree(self.master_doc) + + # Add any remaining documents not in the toctree (sorted) + if hasattr(self.env, "all_docs"): + remaining = sorted( + [doc for doc in self.env.all_docs.keys() if doc not in page_order] + ) + page_order.extend(remaining) + + return page_order + + def filter_excluded_pages(self, page_order: List[str]) -> List[str]: + """Filter out excluded pages from the page order.""" + exclude_patterns = self.config.get("llms_txt_exclude") + if exclude_patterns: + return [ + page + for page in page_order + if not any( + self._match_exclude_pattern(page, pattern) + for pattern in exclude_patterns + ) + ] + return page_order + + def _match_exclude_pattern(self, docname: str, pattern: str) -> bool: + """Check if a document name matches an exclude pattern. + + Args: + docname: The document name to check + pattern: The pattern to match against + + Returns: + True if the document should be excluded, False otherwise + """ + # Exact match + if docname == pattern: + return True + + # Glob-style pattern matching + if fnmatch.fnmatch(docname, pattern): + return True + + return False diff --git a/sphinx_llms_txt/manager.py b/sphinx_llms_txt/manager.py new file mode 100644 index 0000000..7c86ded --- /dev/null +++ b/sphinx_llms_txt/manager.py @@ -0,0 +1,329 @@ +""" +Main manager module for sphinx-llms-txt. +""" + +from pathlib import Path +from typing import Any, Dict, Optional, Tuple + +from sphinx.application import Sphinx +from sphinx.environment import BuildEnvironment +from sphinx.util import logging + +from .collector import DocumentCollector +from .processor import DocumentProcessor +from .writer import FileWriter + +logger = logging.getLogger(__name__) + + +class LLMSFullManager: + """Manages the collection and ordering of documentation sources.""" + + def __init__(self): + self.config: Dict[str, Any] = {} + self.collector = DocumentCollector() + self.processor = None + self.writer = None + self.master_doc: str = None + self.env: BuildEnvironment = None + self.srcdir: Optional[str] = None + self.outdir: Optional[str] = None + self.app: Optional[Sphinx] = None + + def set_master_doc(self, master_doc: str): + """Set the master document name.""" + self.master_doc = master_doc + self.collector.set_master_doc(master_doc) + + def set_env(self, env: BuildEnvironment): + """Set the Sphinx environment.""" + self.env = env + self.collector.set_env(env) + + def update_page_title(self, docname: str, title: str): + """Update the title for a page.""" + self.collector.update_page_title(docname, title) + + def set_config(self, config: Dict[str, Any]): + """Set configuration options.""" + self.config = config + self.collector.set_config(config) + + # Initialize processor and writer with config + self.processor = DocumentProcessor(config, self.srcdir) + self.writer = FileWriter(config, self.outdir, self.app) + + def set_app(self, app: Sphinx): + """Set the Sphinx application reference.""" + self.app = app + if self.writer: + self.writer.app = app + + def combine_sources(self, outdir: str, srcdir: str): + """Combine all source files into a single file.""" + # Store the source directory for resolving include directives + self.srcdir = srcdir + self.outdir = outdir + + # Update processor and writer with directories + self.processor = DocumentProcessor(self.config, srcdir) + self.writer = FileWriter(self.config, outdir, self.app) + + # Get the correct page order + page_order = self.collector.get_page_order() + + if not page_order: + logger.warning( + "Could not determine page order, skipping llms-full creation" + ) + return + + # Apply exclusion filter if configured + page_order = self.collector.filter_excluded_pages(page_order) + + # Determine output file name and location + output_filename = self.config.get("llms_txt_full_filename") + output_path = Path(outdir) / output_filename + + # Find sources directory + sources_dir = None + possible_sources = [ + Path(outdir) / "_sources", + Path(outdir) / "html" / "_sources", + Path(outdir) / "singlehtml" / "_sources", + ] + + for path in possible_sources: + if path.exists(): + sources_dir = path + break + + if not sources_dir: + logger.warning( + "Could not find _sources directory, skipping llms-full creation" + ) + return + + # Collect all available source files + txt_files = {} + for f in sources_dir.glob("*.txt"): + logger.debug(f"sphinx-llms-txt: Found source file: {f.stem} at {f}") + txt_files[f.stem] = f + + # Log discovered files and page order + logger.debug(f"sphinx-llms-txt: Found {len(txt_files)} source files") + logger.debug(f"sphinx-llms-txt: Page order (after exclusion): {page_order}") + + # Log exclusion patterns + exclude_patterns = self.config.get("llms_txt_exclude") + if exclude_patterns: + logger.debug(f"sphinx-llms-txt: Exclusion patterns: {exclude_patterns}") + + # Create a mapping from docnames to actual file names + docname_to_file = {} + + # Try exact matches first + for docname in page_order: + # Skip excluded pages + if any( + self.collector._match_exclude_pattern(docname, pattern) + for pattern in exclude_patterns + ): + continue + + if docname in txt_files: + docname_to_file[docname] = txt_files[docname] + else: + # Try with .rst extension + if f"{docname}.rst" in txt_files: + docname_to_file[docname] = txt_files[f"{docname}.rst"] + # Try with .txt extension + elif f"{docname}.txt" in txt_files: + docname_to_file[docname] = txt_files[f"{docname}.txt"] + # Try with underscores instead of hyphens + elif docname.replace("-", "_") in txt_files: + docname_to_file[docname] = txt_files[docname.replace("-", "_")] + # Try with hyphens instead of underscores + elif docname.replace("_", "-") in txt_files: + docname_to_file[docname] = txt_files[docname.replace("_", "-")] + + # Generate content + content_parts = [] + + # Add pages in order + added_files = set() + total_line_count = 0 + max_lines = self.config.get("llms_txt_full_max_size") + abort_due_to_max_lines = False + + for docname in page_order: + if docname in docname_to_file: + file_path = docname_to_file[docname] + content, line_count = self._read_source_file(file_path, docname) + + # Check if adding this file would exceed the maximum line count + if max_lines is not None and total_line_count + line_count > max_lines: + abort_due_to_max_lines = True + break + + # Double-check this file should be included (not in excluded patterns) + exclude_patterns = self.config.get("llms_txt_exclude") + file_stem = file_path.stem + should_include = True + + if exclude_patterns: + # Check stem and docname against exclusion patterns + if any( + self.collector._match_exclude_pattern(file_stem, pattern) + for pattern in exclude_patterns + ) or any( + self.collector._match_exclude_pattern(docname, pattern) + for pattern in exclude_patterns + ): + logger.debug( + f"sphinx-llms-txt: Final exclusion check removed: {docname}" + ) + should_include = False + + if content and should_include: + content_parts.append(content) + added_files.add(file_path.stem) + total_line_count += line_count + else: + logger.warning(f"sphinx-llm-txt: Source file not found for: {docname}") + + # Add any remaining files (in alphabetical order) if not aborted + if not abort_due_to_max_lines: + # Apply the same exclusion filter to remaining files + exclude_patterns = self.config.get("llms_txt_exclude") + + # Create a set of files to exclude based on their basename + excluded_files = set() + for pattern in exclude_patterns: + if "*" not in pattern and "?" not in pattern: + # For exact patterns, add variants + excluded_files.add(pattern) + excluded_files.add(f"{pattern}.rst") + excluded_files.add(f"{pattern}.txt") + excluded_files.add(pattern.replace("-", "_")) + excluded_files.add(pattern.replace("_", "-")) + + # Filter remaining files + remaining_files = sorted( + [ + name + for name in txt_files + if name not in added_files + and name not in excluded_files + and not any( + self.collector._match_exclude_pattern(name, pattern) + for pattern in exclude_patterns + ) + ] + ) + if remaining_files: + logger.info(f"Adding remaining files: {remaining_files}") + for file_stem in remaining_files: + file_path = txt_files[file_stem] + content, line_count = self._read_source_file(file_path, file_stem) + + # Check if adding this file would exceed the maximum line count + if max_lines is not None and total_line_count + line_count > max_lines: + break + + # Double-check that this file should be included + should_include = True + file_stem = file_path.stem + exclude_patterns = self.config.get("llms_txt_exclude") + + if exclude_patterns: + # Check stem against exclusion patterns + if any( + self.collector._match_exclude_pattern(file_stem, pattern) + for pattern in exclude_patterns + ): + logger.debug( + "sphinx-llms-txt: Final exclusion check removed remaining" + f" file: {file_stem}" + ) + should_include = False + + if content and should_include: + content_parts.append(content) + total_line_count += line_count + + # Check if line limit was exceeded before creating the file + max_lines = self.config.get("llms_txt_full_max_size") + if abort_due_to_max_lines or ( + max_lines is not None and total_line_count > max_lines + ): + logger.warning( + f"sphinx-llm-txt: Max line limit ({max_lines}) exceeded:" + f" {total_line_count} > {max_lines}. " + f"Not creating llms-full.txt file." + ) + + # Log summary information if requested + if self.config.get("llms_txt_file"): + self.writer.write_verbose_info_to_file( + page_order, self.collector.page_titles, total_line_count + ) + + return + + # Write combined file if limit wasn't exceeded + success = self.writer.write_combined_file( + content_parts, output_path, total_line_count + ) + + # Log summary information if requested + if success and self.config.get("llms_txt_file"): + self.writer.write_verbose_info_to_file( + page_order, self.collector.page_titles, total_line_count + ) + + def _read_source_file(self, file_path: Path, docname: str) -> Tuple[str, int]: + """Read and format a single source file. + + Handles include directives by replacing them with the content of the included + file, and processes directives with paths that need to be resolved. + + Returns: + tuple: (content_str, line_count) where line_count is the number of lines + in the file + """ + # Check if this file should be excluded by looking at the doc name + exclude_patterns = self.config.get("llms_txt_exclude") + if exclude_patterns and any( + self.collector._match_exclude_pattern(docname, pattern) + for pattern in exclude_patterns + ): + return "", 0 + + try: + # Check if the file stem (without extension) should be excluded + file_stem = file_path.stem + if exclude_patterns and any( + self.collector._match_exclude_pattern(file_stem, pattern) + for pattern in exclude_patterns + ): + return "", 0 + + with open(file_path, "r", encoding="utf-8") as f: + content = f.read() + + # Process include directives and directives with paths + content = self.processor.process_content(content, file_path) + + # Count the lines in the content + line_count = content.count("\n") + (0 if content.endswith("\n") else 1) + + section_lines = [content, ""] + content_str = "\n".join(section_lines) + + # Add 2 for the section_lines (content + empty line) + return content_str, line_count + 1 + + except Exception as e: + logger.error(f"sphinx-llm-txt: Error reading source file {file_path}: {e}") + return "", 0 diff --git a/sphinx_llms_txt/processor.py b/sphinx_llms_txt/processor.py new file mode 100644 index 0000000..ef7024b --- /dev/null +++ b/sphinx_llms_txt/processor.py @@ -0,0 +1,273 @@ +""" +Document processor module for sphinx-llms-txt. +""" + +import os +import re +from pathlib import Path +from typing import Any, Dict, List, Optional, Tuple + +from sphinx.util import logging + +logger = logging.getLogger(__name__) + + +def build_directive_pattern(directives): + """Build a regex pattern for directives. + + Args: + directives: List of directive names to match + + Returns: + A compiled regex pattern that matches the specified directives + """ + directives_pattern = "|".join(re.escape(d) for d in directives) + return re.compile( + r"^(\s*\.\.\s+(" + directives_pattern + r")::\s+)([^\s].+?)$", re.MULTILINE + ) + + +class DocumentProcessor: + """Processes document content, handling includes and directives.""" + + def __init__(self, config: Dict[str, Any], srcdir: Optional[str] = None): + self.config = config + self.srcdir = srcdir + + def process_content(self, content: str, source_path: Path) -> str: + """Process directives in content that need path resolution. + + Args: + content: The source content to process + source_path: Path to the source file (to resolve relative paths) + + Returns: + Processed content with directives properly resolved + """ + # First process include directives + content = self._process_includes(content, source_path) + + # Then process path directives (image, figure, etc.) + content = self._process_path_directives(content, source_path) + + return content + + def _extract_relative_document_path( + self, source_path: Path + ) -> Tuple[Optional[str], Optional[str], Optional[List[str]]]: + """Extract the relative document path from a source file in _sources directory. + + Args: + source_path: Path to the source file + + Returns: + Tuple of (rel_doc_path, rel_doc_dir, rel_doc_path_parts) + """ + try: + # Extract the part after _sources/ + path_parts = str(source_path).split("_sources/") + if len(path_parts) > 1: + rel_doc_path = path_parts[1] + # Remove .txt extension if present + if rel_doc_path.endswith(".txt"): + rel_doc_path = rel_doc_path[:-4] + # Get the directory containing the current document + rel_doc_dir = os.path.dirname(rel_doc_path) + rel_doc_path_parts = rel_doc_path.split("/") + + return rel_doc_path, rel_doc_dir, rel_doc_path_parts + except Exception as e: + logger.debug(f"sphinx-llms-txt: Error extracting relative path: {e}") + + return None, None, None + + def _add_base_url(self, path: str, base_url: str) -> str: + """Add base URL to a path if needed. + + Args: + path: The path to add the base URL to + base_url: The base URL to add + + Returns: + Path with base URL added if applicable + """ + if not base_url: + return path + + if not base_url.endswith("/"): + base_url += "/" + return f"{base_url}{path}" + + def _is_absolute_or_url(self, path: str) -> bool: + """Check if a path is absolute or a URL. + + Args: + path: The path to check + + Returns: + True if the path is absolute or a URL, False otherwise + """ + return path.startswith(("http://", "https://", "/", "data:")) + + def _process_path_directives(self, content: str, source_path: Path) -> str: + """Process directives with paths that need to be resolved. + + Args: + content: The source content to process + source_path: Path to the source file (to resolve relative paths) + + Returns: + Processed content with directive paths properly resolved + """ + # Get the configured path directives to process + default_path_directives = ["image", "figure"] + custom_path_directives = self.config.get("llms_txt_directives") + path_directives = set(default_path_directives + custom_path_directives) + + # Build the regex pattern to match all configured directives + directive_pattern = build_directive_pattern(path_directives) + + # Get the base URL from Sphinx's html_baseurl if set + base_url = self.config.get("html_baseurl", "") + + # Handle test case specially + is_test = "pytest" in str(source_path) and "subdir" in str(source_path) + + def replace_directive_path(match, base_url=base_url, is_test=is_test): + prefix = match.group(1) # The entire directive prefix including whitespace + path = match.group(3).strip() # The path argument + + # Only process relative paths, not absolute paths or URLs + if not self._is_absolute_or_url(path): + # Special case for test files + if is_test: + # Add subdir/ prefix to match test expectations + full_path = "subdir/" + path + + # If base_url is set, prepend it to the path + full_path = self._add_base_url(full_path, base_url) + + # Return the updated directive with the full path + return f"{prefix}{full_path}" + + # Production case (not in test) + elif "_sources" in str(source_path): + # Extract the part after _sources/ + rel_doc_path, rel_doc_dir, rel_doc_path_parts = ( + self._extract_relative_document_path(source_path) + ) + + if rel_doc_path_parts: + # For test subdirectory handling - this is for our test cases + if ( + len(rel_doc_path_parts) > 0 + and rel_doc_path_parts[0] == "subdir" + ): + full_path = os.path.normpath(os.path.join("subdir", path)) + # Only add the rel_doc_dir if it's not empty + elif rel_doc_dir: + # Join with the original path to form full path relative + # to srcdir + full_path = os.path.normpath( + os.path.join(rel_doc_dir, path) + ) + else: + full_path = path + + # If base_url is set, prepend it to the path + full_path = self._add_base_url(full_path, base_url) + + # Return the updated directive with the full path + return f"{prefix}{full_path}" + + # If we couldn't resolve the path or it's already absolute, return unchanged + return match.group(0) + + # Replace directive paths in the content + processed_content = directive_pattern.sub(replace_directive_path, content) + return processed_content + + def _resolve_include_paths( + self, include_path: str, source_path: Path + ) -> List[Path]: + """Resolve possible paths for an include directive. + + Args: + include_path: The path from the include directive + source_path: The path to the source file + + Returns: + List of possible paths to try + """ + possible_paths = [] + + # If it's an absolute path, use it directly + if os.path.isabs(include_path): + possible_paths.append(Path(include_path)) + else: + # Relative to the source file (in _sources directory) + possible_paths.append((source_path.parent / include_path).resolve()) + + # If we're in _sources directory, try relative to the original source + # directory + if "_sources" in str(source_path): + # Extract the relative path portion from the source path + rel_path, rel_dir, _ = self._extract_relative_document_path(source_path) + + # If we have the original source directory from Sphinx + if self.srcdir: + # Try in the srcdir root + possible_paths.append((Path(self.srcdir) / include_path).resolve()) + + # If we have a relative path, try in the corresponding source + # subdirectory + if rel_path and rel_dir: + possible_paths.append( + (Path(self.srcdir) / rel_dir / include_path).resolve() + ) + + return possible_paths + + def _process_includes(self, content: str, source_path: Path) -> str: + """Process include directives in content. + + Args: + content: The source content to process + source_path: Path to the source file (to resolve relative paths) + + Returns: + Processed content with include directives replaced with included content + """ + # Find all include directives using regex + include_pattern = build_directive_pattern(["include"]) + + # Function to replace each include with content + def replace_include(match): + include_path = match.group(3) + + # Get all possible paths to try + possible_paths = self._resolve_include_paths(include_path, source_path) + + # Try each possible path + for path_to_try in possible_paths: + try: + if path_to_try.exists(): + with open(path_to_try, "r", encoding="utf-8") as f: + included_content = f.read() + return included_content + except Exception as e: + logger.error( + f"sphinx-llms-txt: Error reading include file {path_to_try}:" + f" {e}" + ) + continue + + # If we get here, we couldn't find the file + paths_tried = ", ".join(str(p) for p in possible_paths) + logger.warning(f"sphinx-llms-txt: Include file not found: {include_path}") + logger.debug(f"sphinx-llms-txt: Tried paths: {paths_tried}") + return f"[Include file not found: {include_path}]" + + # Replace all includes with their content + processed_content = include_pattern.sub(replace_include, content) + return processed_content diff --git a/sphinx_llms_txt/writer.py b/sphinx_llms_txt/writer.py new file mode 100644 index 0000000..70d6d77 --- /dev/null +++ b/sphinx_llms_txt/writer.py @@ -0,0 +1,100 @@ +""" +File writer module for sphinx-llms-txt. +""" + +from pathlib import Path +from typing import Any, Dict, List + +from sphinx.application import Sphinx +from sphinx.util import logging + +logger = logging.getLogger(__name__) + + +class FileWriter: + """Handles writing processed content to output files.""" + + def __init__(self, config: Dict[str, Any], outdir: str = None, app: Sphinx = None): + self.config = config + self.outdir = outdir + self.app = app + + def write_combined_file( + self, content_parts: List[str], output_path: Path, total_line_count: int + ) -> bool: + """Write the combined content to a file. + + Args: + content_parts: List of content strings to combine + output_path: Path to write the output file + total_line_count: Total number of lines in the content + + Returns: + True if successful, False otherwise + """ + try: + with open(output_path, "w", encoding="utf-8") as f: + f.write("\n".join(content_parts)) + + logger.info( + f"sphinx-llms-txt: created {output_path} with {len(content_parts)}" + f" sources and {total_line_count} lines" + ) + return True + except Exception as e: + logger.error(f"sphinx-llm-txt: Error writing combined sources file: {e}") + return False + + def write_verbose_info_to_file( + self, + page_order: List[str], + page_titles: Dict[str, str], + total_line_count: int = 0, + ) -> bool: + """Write summary information to the llms.txt file. + + Args: + page_order: Ordered list of document names + page_titles: Dictionary mapping docnames to titles + total_line_count: Total number of lines in the combined content + + Returns: + True if successful, False otherwise + """ + if not self.outdir: + logger.warning( + "sphinx-llms-txt: Cannot write verbose info to file: outdir not set" + ) + return False + + output_path = Path(self.outdir) / self.config.get("llms_txt_filename") + try: + with open(output_path, "w", encoding="utf-8") as f: + project_name = "llms-txt Summary" + # First priority: use title from config if available + if self.config.get("llms_txt_title"): + project_name = self.config.get("llms_txt_title") + # Second priority: use project name from Sphinx app if available + elif ( + self.app + and hasattr(self.app, "config") + and hasattr(self.app.config, "project") + ): + project_name = self.app.config.project + f.write(f"# {project_name}\n\n") + + # Add description if available + description = self.config.get("llms_txt_summary", "") + if description: + f.write(f"> {description}\n\n") + + f.write("## Docs\n\n") + for docname in page_order: + title = page_titles.get(docname, docname) + f.write(f"- [{title}](/{docname}.html)\n") + + logger.info(f"sphinx-llms-txt: created {output_path}") + return True + except Exception as e: + logger.error(f"sphinx-llms-txt: Error writing verbose info to file: {e}") + return False diff --git a/tests/test_llms_txt.py b/tests/test_llms_txt.py index 9a8fd3c..3a6cd0e 100644 --- a/tests/test_llms_txt.py +++ b/tests/test_llms_txt.py @@ -1,6 +1,12 @@ """Test the sphinx_llms_txt extension.""" -from sphinx_llms_txt import LLMSFullManager, setup +from sphinx_llms_txt import ( + DocumentCollector, + DocumentProcessor, + FileWriter, + LLMSFullManager, + setup, +) def test_version(): @@ -35,23 +41,61 @@ def test_setup_returns_valid_dict(): assert "parallel_write_safe" in result +def test_document_collector_initialization(): + """Test initialization of DocumentCollector.""" + collector = DocumentCollector() + assert collector.page_titles == {} + assert collector.config == {} + assert collector.master_doc is None + assert collector.env is None + + +def test_document_processor_initialization(): + """Test initialization of DocumentProcessor.""" + config = {"llms_txt_directives": []} + processor = DocumentProcessor(config) + assert processor.config == config + assert processor.srcdir is None + + +def test_file_writer_initialization(): + """Test initialization of FileWriter.""" + config = {"llms_txt_filename": "llms.txt"} + writer = FileWriter(config) + assert writer.config == config + assert writer.outdir is None + assert writer.app is None + + def test_llms_full_manager_initialization(): """Test initialization of LLMSFullManager.""" manager = LLMSFullManager() - assert manager.page_titles == {} assert manager.config == {} + assert isinstance(manager.collector, DocumentCollector) + assert manager.processor is None + assert manager.writer is None assert manager.master_doc is None assert manager.env is None -def test_manager_page_title_update(): +def test_collector_page_title_update(): """Test updating page titles.""" + collector = DocumentCollector() + collector.update_page_title("doc1", "Title 1") + collector.update_page_title("doc2", "Title 2") + + assert collector.page_titles["doc1"] == "Title 1" + assert collector.page_titles["doc2"] == "Title 2" + + +def test_manager_page_title_update(): + """Test updating page titles through manager.""" manager = LLMSFullManager() manager.update_page_title("doc1", "Title 1") manager.update_page_title("doc2", "Title 2") - assert manager.page_titles["doc1"] == "Title 1" - assert manager.page_titles["doc2"] == "Title 2" + assert manager.collector.page_titles["doc1"] == "Title 1" + assert manager.collector.page_titles["doc2"] == "Title 2" def test_set_config(): @@ -64,6 +108,9 @@ def test_set_config(): } manager.set_config(config) assert manager.config == config + assert manager.collector.config == config + assert isinstance(manager.processor, DocumentProcessor) + assert isinstance(manager.writer, FileWriter) def test_set_master_doc(): @@ -71,22 +118,24 @@ def test_set_master_doc(): manager = LLMSFullManager() manager.set_master_doc("index") assert manager.master_doc == "index" + assert manager.collector.master_doc == "index" def test_empty_page_order(): """Test get_page_order returns empty list when env or master_doc not set.""" - manager = LLMSFullManager() - assert manager.get_page_order() == [] + collector = DocumentCollector() + assert collector.get_page_order() == [] # Set only master_doc, but not env - manager.set_master_doc("index") - assert manager.get_page_order() == [] + collector.set_master_doc("index") + assert collector.get_page_order() == [] def test_process_includes(tmp_path): """Test that include directives are processed correctly.""" - # Create a manager - manager = LLMSFullManager() + # Create a processor + config = {"llms_txt_directives": []} + processor = DocumentProcessor(config) # Create a test file with an include directive include_content = "This is included content.\nWith multiple lines." @@ -103,7 +152,7 @@ def test_process_includes(tmp_path): f.write(source_content) # Process the include directive - processed_content = manager._process_includes(source_content, source_file) + processed_content = processor._process_includes(source_content, source_file) # Check that the include directive was replaced with the content expected_content = ( @@ -115,8 +164,8 @@ def test_process_includes(tmp_path): def test_process_includes_with_relative_paths(tmp_path): """Test that include directives with relative paths are processed correctly.""" - # Create a manager - manager = LLMSFullManager() + # Create a processor + config = {"llms_txt_directives": []} # Set up a more complex directory structure docs_dir = tmp_path / "docs" @@ -134,8 +183,8 @@ def test_process_includes_with_relative_paths(tmp_path): includes_dir = source_dir / "includes" includes_dir.mkdir() - # Set the srcdir on the manager - manager.srcdir = str(source_dir) + # Create a processor with srcdir + processor = DocumentProcessor(config, str(source_dir)) # Create the included file in the includes directory include_content = "This is included content from another directory." @@ -167,7 +216,7 @@ def test_process_includes_with_relative_paths(tmp_path): f.write(source_content) # Process the include directive from the _sources file - processed_content = manager._process_includes(source_content, sources_file) + processed_content = processor._process_includes(source_content, sources_file) # Check that the include directive was replaced with the content expected_content = ( @@ -179,50 +228,46 @@ def test_process_includes_with_relative_paths(tmp_path): def test_match_exclude_pattern(): """Test the _match_exclude_pattern method.""" - # Create a manager - manager = LLMSFullManager() + # Create a collector + collector = DocumentCollector() # Test exact match - assert manager._match_exclude_pattern("page1", "page1") is True - assert manager._match_exclude_pattern("page1", "page2") is False + assert collector._match_exclude_pattern("page1", "page1") is True + assert collector._match_exclude_pattern("page1", "page2") is False # Test glob-style patterns - assert manager._match_exclude_pattern("page1", "page*") is True - assert manager._match_exclude_pattern("page_with_include", "page_with_*") is True - assert manager._match_exclude_pattern("page1", "*1") is True - assert manager._match_exclude_pattern("subdir/page1", "*/page1") is True - assert manager._match_exclude_pattern("page1", "subdir/*") is False + assert collector._match_exclude_pattern("page1", "page*") is True + assert collector._match_exclude_pattern("page_with_include", "page_with_*") is True + assert collector._match_exclude_pattern("page1", "*1") is True + assert collector._match_exclude_pattern("subdir/page1", "*/page1") is True + assert collector._match_exclude_pattern("page1", "subdir/*") is False def test_write_verbose_info_to_file(tmp_path): """Test writing verbose info to a file.""" - # Create a manager - manager = LLMSFullManager() - - # Set up a build directory + # Create a build directory build_dir = tmp_path / "build" build_dir.mkdir() - # Set the outdir on the manager - manager.outdir = str(build_dir) - - # Set configuration with verbose_file enabled + # Create writer with configuration and outdir config = { "llms_txt_file": True, "llms_txt_full_max_size": 1000, "llms_txt_filename": "llms.txt", } - manager.set_config(config) + writer = FileWriter(config, str(build_dir)) - # Add some page titles - manager.update_page_title("index", "Home Page") - manager.update_page_title("about", "About Us") + # Create page titles + page_titles = { + "index": "Home Page", + "about": "About Us", + } # Create a page order page_order = ["index", "about"] # Call the method to write verbose info to file - manager._write_verbose_info_to_file(page_order, 500) + writer.write_verbose_info_to_file(page_order, page_titles) # Check that the file was created verbose_file = build_dir / "llms.txt" diff --git a/tests/test_path_directives.py b/tests/test_path_directives.py index e8086dd..ce72e38 100644 --- a/tests/test_path_directives.py +++ b/tests/test_path_directives.py @@ -1,25 +1,21 @@ """Test the path directive processing functionality in sphinx_llms_txt.""" -from sphinx_llms_txt import LLMSFullManager +from sphinx_llms_txt import DocumentProcessor def test_process_path_directives(tmp_path): """Test that path directives are processed correctly.""" - # Create a manager - manager = LLMSFullManager() - - # Configure the manager with default directives - manager.set_config( - { - "llms_txt_directives": [], - "html_baseurl": "", - } - ) + # Create a processor + config = { + "llms_txt_directives": [], + "html_baseurl": "", + } + processor = DocumentProcessor(config) # Create source directory structure src_dir = tmp_path / "src" src_dir.mkdir() - manager.srcdir = str(src_dir) + processor.srcdir = str(src_dir) # Create _sources directory to mimic Sphinx output build_dir = tmp_path / "build" @@ -48,7 +44,7 @@ def test_process_path_directives(tmp_path): f.write(source_content) # Process the directives - processed_content = manager._process_path_directives(source_content, source_file) + processed_content = processor._process_path_directives(source_content, source_file) # With our implementation, the paths should have subdirectory paths added expected_content = ( @@ -64,21 +60,17 @@ def test_process_path_directives(tmp_path): def test_process_path_directives_with_html_baseurl(tmp_path): """Test path directives with base_url configured using html_baseurl.""" - # Create a manager - manager = LLMSFullManager() - - # Configure the manager with default directives and base_url using html_baseurl - manager.set_config( - { - "llms_txt_directives": [], - "html_baseurl": "https://sphinx-docs.org/", - } - ) + # Create a processor + config = { + "llms_txt_directives": [], + "html_baseurl": "https://sphinx-docs.org/", + } + processor = DocumentProcessor(config) # Create source directory structure src_dir = tmp_path / "src" src_dir.mkdir() - manager.srcdir = str(src_dir) + processor.srcdir = str(src_dir) # Create _sources directory to mimic Sphinx output build_dir = tmp_path / "build" @@ -101,7 +93,7 @@ def test_process_path_directives_with_html_baseurl(tmp_path): f.write(source_content) # Process the directives - processed_content = manager._process_path_directives(source_content, source_file) + processed_content = processor._process_path_directives(source_content, source_file) # Expected: The paths should include the base URL with 'subdir' prefix expected_content = ".. image:: https://sphinx-docs.org/subdir/images/test.png\n" @@ -111,21 +103,17 @@ def test_process_path_directives_with_html_baseurl(tmp_path): def test_process_path_directives_absolute_urls(tmp_path): """Test that absolute URLs are not modified.""" - # Create a manager - manager = LLMSFullManager() - - # Configure the manager with default directives - manager.set_config( - { - "llms_txt_directives": [], - "html_baseurl": "https://example.com/docs", - } - ) + # Create a processor + config = { + "llms_txt_directives": [], + "html_baseurl": "https://example.com/docs", + } + processor = DocumentProcessor(config) # Create source directory structure src_dir = tmp_path / "src" src_dir.mkdir() - manager.srcdir = str(src_dir) + processor.srcdir = str(src_dir) # Create a source file with absolute URL image directives source_content = ( @@ -140,28 +128,24 @@ def test_process_path_directives_absolute_urls(tmp_path): f.write(source_content) # Process the directives (should remain unchanged) - processed_content = manager._process_path_directives(source_content, source_file) + processed_content = processor._process_path_directives(source_content, source_file) assert processed_content == source_content def test_process_path_directives_custom_directives(tmp_path): """Test that custom directives are processed correctly.""" - # Create a manager - manager = LLMSFullManager() - - # Configure the manager with custom directives - manager.set_config( - { - "llms_txt_directives": ["drawio-figure", "drawio-image"], - "html_baseurl": "", - } - ) + # Create a processor + config = { + "llms_txt_directives": ["drawio-figure", "drawio-image"], + "html_baseurl": "", + } + processor = DocumentProcessor(config) # Create source directory structure src_dir = tmp_path / "src" src_dir.mkdir() - manager.srcdir = str(src_dir) + processor.srcdir = str(src_dir) # Create _sources directory to mimic Sphinx output build_dir = tmp_path / "build" @@ -182,7 +166,7 @@ def test_process_path_directives_custom_directives(tmp_path): f.write(source_content) # Process the directives - processed_content = manager._process_path_directives(source_content, source_file) + processed_content = processor._process_path_directives(source_content, source_file) # Expected: The paths should be resolved to full paths expected_content = ( @@ -196,23 +180,18 @@ def test_process_path_directives_custom_directives(tmp_path): def test_process_content_end_to_end(tmp_path): """ - Test the full _process_content method handling both includes and path directives. + Test the full process_content method handling both includes and path directives. """ - # Create a manager - manager = LLMSFullManager() - - # Configure the manager - manager.set_config( - { - "llms_txt_directives": ["drawio-figure"], - "html_baseurl": "https://sphinx-docs.org/", - } - ) + # Create a processor + config = { + "llms_txt_directives": ["drawio-figure"], + "html_baseurl": "https://sphinx-docs.org/", + } + processor = DocumentProcessor(config, str(tmp_path / "src")) # Create source directory structure src_dir = tmp_path / "src" src_dir.mkdir() - manager.srcdir = str(src_dir) # Create an includes directory includes_dir = src_dir / "includes" @@ -255,7 +234,7 @@ def test_process_content_end_to_end(tmp_path): f.write(source_content) # Process the content - processed_content = manager._process_content(source_content, source_file) + processed_content = processor.process_content(source_content, source_file) # Expected: Both includes and path directives should be processed expected_content = (