diff --git a/sphinx_llms_txt/__init__.py b/sphinx_llms_txt/__init__.py index ca56d77..029c6d3 100644 --- a/sphinx_llms_txt/__init__.py +++ b/sphinx_llms_txt/__init__.py @@ -107,7 +107,7 @@ def build_finished(app: Sphinx, exception): _manager.update_page_title(docname, title) # Create the combined file - _manager.combine_sources(app.outdir, app.srcdir) + _manager.combine_sources(str(app.outdir), str(app.srcdir)) def setup(app: Sphinx) -> Dict[str, Any]: diff --git a/sphinx_llms_txt/collector.py b/sphinx_llms_txt/collector.py index d99b237..c2c6973 100644 --- a/sphinx_llms_txt/collector.py +++ b/sphinx_llms_txt/collector.py @@ -3,7 +3,7 @@ Document collector module for sphinx-llms-txt. """ import fnmatch -from typing import Any, Dict, List, Tuple +from typing import Any, Dict, List, Optional, Tuple from sphinx.environment import BuildEnvironment from sphinx.util import logging @@ -16,8 +16,8 @@ class DocumentCollector: def __init__(self): self.page_titles: Dict[str, str] = {} - self.master_doc: str = None - self.env: BuildEnvironment = None + self.master_doc: Optional[str] = None + self.env: Optional[BuildEnvironment] = None self.config: Dict[str, Any] = {} self.app = None @@ -60,7 +60,7 @@ class DocumentCollector: else: return [source_suffix] # String format - def _get_docname_suffix(self, docname: str, sources_dir) -> str: + def _get_docname_suffix(self, docname: str, sources_dir) -> Optional[str]: """ Determine the source suffix for a given docname by checking which file exists. @@ -102,7 +102,7 @@ class DocumentCollector: return None - def get_page_order(self, sources_dir=None) -> List[Tuple[str, str]]: + def get_page_order(self, sources_dir=None) -> List[Tuple[str, Optional[str]]]: """Get the correct page order from the toctree structure. Args: @@ -114,7 +114,7 @@ class DocumentCollector: if not self.env or not self.master_doc: return [] - page_order = [] + page_order: List[Tuple[str, Optional[str]]] = [] visited = set() def collect_from_toctree(docname: str): @@ -126,7 +126,7 @@ class DocumentCollector: # Add the current document with its suffix if docname not in [doc for doc, _ in page_order]: - suffix = None + suffix: Optional[str] = None if sources_dir: suffix = self._get_docname_suffix(docname, sources_dir) page_order.append((docname, suffix)) @@ -135,18 +135,21 @@ class DocumentCollector: try: # Look for toctree_includes which contains the direct children if ( - hasattr(self.env, "toctree_includes") + self.env + and hasattr(self.env, "toctree_includes") and docname in self.env.toctree_includes ): for child_docname in self.env.toctree_includes[docname]: - collect_from_toctree(child_docname) + collect_from_toctree(str(child_docname)) # Try to use dependencies to find related documents elif ( - hasattr(self.env, "dependencies") + self.env + and hasattr(self.env, "dependencies") and docname in self.env.dependencies ): # Extract the dependent documents from the dependencies dict - for child_docname in self.env.dependencies[docname]: + for child_docname_obj in self.env.dependencies[docname]: + child_docname = str(child_docname_obj) # Only add documents actually in the document set if ( hasattr(self.env, "all_docs") @@ -154,7 +157,11 @@ class DocumentCollector: ): collect_from_toctree(child_docname) # Fallback to titles or other available references - elif hasattr(self.env, "titles") and hasattr(self.env, "all_docs"): + elif ( + self.env + and hasattr(self.env, "titles") + and hasattr(self.env, "all_docs") + ): # Get all document names all_docnames = list(self.env.all_docs.keys()) @@ -185,7 +192,7 @@ class DocumentCollector: ] ) for docname in remaining: - suffix = None + suffix: Optional[str] = None if sources_dir: suffix = self._get_docname_suffix(docname, sources_dir) page_order.append((docname, suffix)) @@ -193,8 +200,8 @@ class DocumentCollector: return page_order def filter_excluded_pages( - self, page_order: List[Tuple[str, str]] - ) -> List[Tuple[str, str]]: + self, page_order: List[Tuple[str, Optional[str]]] + ) -> List[Tuple[str, Optional[str]]]: """Filter out excluded pages from the page order.""" exclude_patterns = self.config.get("llms_txt_exclude") if exclude_patterns: diff --git a/sphinx_llms_txt/manager.py b/sphinx_llms_txt/manager.py index 1f97da5..7338ec0 100644 --- a/sphinx_llms_txt/manager.py +++ b/sphinx_llms_txt/manager.py @@ -5,7 +5,7 @@ Main manager module for sphinx-llms-txt. import glob import subprocess from pathlib import Path -from typing import Any, Dict, List, Optional, Tuple, Union +from typing import Any, Dict, List, Optional, Tuple, Union, cast from sphinx.application import Sphinx from sphinx.environment import BuildEnvironment @@ -150,8 +150,8 @@ class LLMSFullManager: self.ignored_pages.add(docname) def _filter_ignored_pages( - self, page_order: Union[List[str], List[Tuple[str, str]]] - ) -> Union[List[str], List[Tuple[str, str]]]: + self, page_order: Union[List[str], List[Tuple[str, Optional[str]]]] + ) -> Union[List[str], List[Tuple[str, Optional[str]]]]: """Filter out ignored pages from page_order.""" filtered_pages = [] for item in page_order: @@ -164,7 +164,7 @@ class LLMSFullManager: if docname not in self.ignored_pages: filtered_pages.append(item) - return filtered_pages + return cast(Union[List[str], List[Tuple[str, Optional[str]]]], filtered_pages) def set_config(self, config: Dict[str, Any]): """Set configuration options.""" @@ -225,7 +225,7 @@ class LLMSFullManager: # Determine output file name and location output_filename = self.config.get("llms_txt_full_filename") - output_path = Path(outdir) / output_filename + output_path = Path(outdir) / str(output_filename) # Log discovered files and page order logger.debug(f"sphinx-llms-txt: Page order (after exclusion): {page_order}") @@ -286,7 +286,7 @@ class LLMSFullManager: content_parts = [] # Track code files for later processing - code_file_parts = [] + code_file_parts: List[str] = [] # Count lines in code files (initially 0) code_files_line_count = 0 @@ -365,7 +365,7 @@ class LLMSFullManager: if not (size_limit_exceeded and should_abort_early): # Get all source files in the _sources directory using configured suffixes source_suffixes = self._get_source_suffixes() - all_source_files = [] + all_source_files: List[Path] = [] for src_suffix in source_suffixes: # Avoid duplicate extensions when source_suffix == source_link_suffix if src_suffix == source_link_suffix: @@ -735,9 +735,9 @@ class LLMSFullManager: title = Path(title_str[len(base_path) :]) except ValueError: # File is not relative to srcdir, use filename - title = file_path.name + title = Path(file_path.name) else: - title = file_path.name + title = Path(file_path.name) # Format as code block with equals underline title_str = str(title) @@ -769,7 +769,9 @@ class LLMSFullManager: return code_parts, sorted(processed_files) - def _create_code_files_section_header(self, file_paths: List[Path] = None) -> str: + def _create_code_files_section_header( + self, file_paths: Optional[List[Path]] = None + ) -> str: """Create the section header for source code files. Args: @@ -817,7 +819,7 @@ class LLMSFullManager: return "" # Convert to relative paths if possible and create tree structure - tree_data = {} + tree_data: Dict[str, Any] = {} for file_path in sorted(file_paths): # Get relative path from source directory for display @@ -870,7 +872,7 @@ class LLMSFullManager: current[parts[-1]] = None # None indicates it's a file # Convert tree structure to string representation - lines = [] + lines: List[str] = [] self._format_tree_node(tree_data, lines, "", True) # Indent each line for reStructuredText code block diff --git a/sphinx_llms_txt/processor.py b/sphinx_llms_txt/processor.py index f04ddfd..65abf24 100644 --- a/sphinx_llms_txt/processor.py +++ b/sphinx_llms_txt/processor.py @@ -130,7 +130,7 @@ class DocumentProcessor: """ # Get the configured path directives to process default_path_directives = ["image", "figure"] - custom_path_directives = self.config.get("llms_txt_directives") + custom_path_directives = self.config.get("llms_txt_directives") or [] path_directives = set(default_path_directives + custom_path_directives) # Build the regex pattern to match all configured directives diff --git a/sphinx_llms_txt/writer.py b/sphinx_llms_txt/writer.py index e5f873b..ebabefe 100644 --- a/sphinx_llms_txt/writer.py +++ b/sphinx_llms_txt/writer.py @@ -3,7 +3,7 @@ File writer module for sphinx-llms-txt. """ from pathlib import Path -from typing import Any, Dict, List, Tuple, Union +from typing import Any, Dict, List, Optional, Tuple, Union from sphinx.application import Sphinx from sphinx.util import logging @@ -14,7 +14,12 @@ logger = logging.getLogger(__name__) class FileWriter: """Handles writing processed content to output files.""" - def __init__(self, config: Dict[str, Any], outdir: str = None, app: Sphinx = None): + def __init__( + self, + config: Dict[str, Any], + outdir: Optional[str] = None, + app: Optional[Sphinx] = None, + ): self.config = config self.outdir = outdir self.app = app @@ -47,7 +52,7 @@ class FileWriter: def write_verbose_info_to_file( self, - page_order: Union[List[str], List[Tuple[str, str]]], + page_order: Union[List[str], List[Tuple[str, Optional[str]]]], page_titles: Dict[str, str], total_line_count: int = 0, ) -> bool: @@ -67,13 +72,13 @@ class FileWriter: ) return False - output_path = Path(self.outdir) / self.config.get("llms_txt_filename") + output_path = Path(self.outdir) / str(self.config.get("llms_txt_filename")) try: with open(output_path, "w", encoding="utf-8") as f: project_name = "llms-txt Summary" # First priority: use title from config if available if self.config.get("llms_txt_title"): - project_name = self.config.get("llms_txt_title") + project_name = str(self.config.get("llms_txt_title")) # Second priority: use project name from Sphinx app if available elif ( self.app