Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c24f92031c | ||
|
|
0ee28290db | ||
|
|
368c349d58 |
@@ -1,6 +1,24 @@
|
|||||||
Changelog
|
Changelog
|
||||||
=========
|
=========
|
||||||
|
|
||||||
|
0.4.1
|
||||||
|
-----
|
||||||
|
|
||||||
|
- Fix include paths and spacing
|
||||||
|
`#31 <https://github.com/jdillard/sphinx-llms-txt/pull/31>`_
|
||||||
|
|
||||||
|
0.4.0
|
||||||
|
-----
|
||||||
|
|
||||||
|
- Add support for including source code files with :confval:`llms_txt_code_files` and :confval:`llms_txt_code_base_path` configuration options
|
||||||
|
`#24 <https://github.com/jdillard/sphinx-llms-txt/pull/24>`_
|
||||||
|
|
||||||
|
0.3.2
|
||||||
|
-----
|
||||||
|
|
||||||
|
- Fix image paths to deployed images
|
||||||
|
`#30 <https://github.com/jdillard/sphinx-llms-txt/pull/30>`_
|
||||||
|
|
||||||
0.3.1
|
0.3.1
|
||||||
-----
|
-----
|
||||||
|
|
||||||
|
|||||||
@@ -125,6 +125,50 @@ You can exclude specific pages from being included in the generated files:
|
|||||||
|
|
||||||
This is useful for excluding auto-generated pages, indexes, or content that isn't relevant for LLM consumption.
|
This is useful for excluding auto-generated pages, indexes, or content that isn't relevant for LLM consumption.
|
||||||
|
|
||||||
|
.. _including_code_files:
|
||||||
|
|
||||||
|
Including Source Code Files
|
||||||
|
^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
||||||
|
|
||||||
|
You can include source code files from your project at the end of :confval:`llms_txt_full_filename`.
|
||||||
|
|
||||||
|
Use include/exclude syntax to precisely control which files are included:
|
||||||
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
llms_txt_code_files = [
|
||||||
|
"+:src/**/*.py", # Include all Python files in src
|
||||||
|
"-:src/**/__pycache__/**", # Exclude Python cache files
|
||||||
|
]
|
||||||
|
|
||||||
|
Pattern syntax:
|
||||||
|
|
||||||
|
- **+:pattern**: Include files matching the pattern. Processed first to collect matching files.
|
||||||
|
- **-:pattern**: Exclude files matching the pattern. Applied to filter out unwanted files.
|
||||||
|
|
||||||
|
Code files are processed as follows:
|
||||||
|
|
||||||
|
- **Glob patterns**: Use standard glob patterns (``*``, ``**``, ``?``) to match files
|
||||||
|
- **Relative paths**: Patterns are resolved relative to your Sphinx source directory
|
||||||
|
- **Formatting**: Each file is presented with a title and syntax-highlighted code block
|
||||||
|
|
||||||
|
.. _customizing_code_paths:
|
||||||
|
|
||||||
|
Customizing Code File Paths
|
||||||
|
~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
|
By default, the extension automatically detects the relative path from your Sphinx source directory to the git root and strips that prefix from displayed file paths. You can customize this behavior:
|
||||||
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# Manually specify base path to strip
|
||||||
|
llms_txt_code_base_path = "../../"
|
||||||
|
|
||||||
|
# Disable path stripping entirely
|
||||||
|
llms_txt_code_base_path = ""
|
||||||
|
|
||||||
|
This helps create cleaner, more readable file paths in the generated documentation.
|
||||||
|
|
||||||
.. _using_html_baseurl:
|
.. _using_html_baseurl:
|
||||||
|
|
||||||
Using HTML Base URL
|
Using HTML Base URL
|
||||||
@@ -168,3 +212,11 @@ Here's a complete example showing multiple :doc:`configuration-values`:
|
|||||||
|
|
||||||
# Content filtering
|
# Content filtering
|
||||||
llms_txt_exclude = ["search", "genindex", "404", "private_*"]
|
llms_txt_exclude = ["search", "genindex", "404", "private_*"]
|
||||||
|
|
||||||
|
# Source code inclusion with include/exclude patterns
|
||||||
|
llms_txt_code_files = [
|
||||||
|
"+:../../src/**/*.py", # Include Python files
|
||||||
|
"+:../../config/*.yaml", # Include config files
|
||||||
|
"-:../../src/**/__pycache__/**", # Exclude cache files
|
||||||
|
]
|
||||||
|
llms_txt_code_base_path = "../../"
|
||||||
|
|||||||
@@ -15,6 +15,7 @@ import subprocess
|
|||||||
project = "sphinx-llms-txt"
|
project = "sphinx-llms-txt"
|
||||||
copyright = "Jared Dillard"
|
copyright = "Jared Dillard"
|
||||||
author = "Jared Dillard"
|
author = "Jared Dillard"
|
||||||
|
llms_txt_code_files = ["+:../../sphinx_llms_txt/*.py"]
|
||||||
llms_txt_summary = """
|
llms_txt_summary = """
|
||||||
A Sphinx extension that generates a summary llms.txt file,written in Markdown,
|
A Sphinx extension that generates a summary llms.txt file,written in Markdown,
|
||||||
and a single combined documentation llms-full.txt file, written in reStructuredText.
|
and a single combined documentation llms-full.txt file, written in reStructuredText.
|
||||||
|
|||||||
@@ -82,3 +82,22 @@ Project Configuration Values
|
|||||||
See :ref:`excluding_content`.
|
See :ref:`excluding_content`.
|
||||||
|
|
||||||
.. versionadded:: 0.2.1
|
.. versionadded:: 0.2.1
|
||||||
|
|
||||||
|
.. confval:: llms_txt_code_files
|
||||||
|
|
||||||
|
- **Type**: list of strings
|
||||||
|
- **Default**: ``[]``
|
||||||
|
- **Description**: A list of glob patterns that appends source code files to :confval:`llms_txt_full_filename`.
|
||||||
|
See :ref:`including_code_files`.
|
||||||
|
|
||||||
|
.. versionadded:: 0.4.0
|
||||||
|
|
||||||
|
.. confval:: llms_txt_code_base_path
|
||||||
|
|
||||||
|
- **Type**: string or ``None``
|
||||||
|
- **Default**: ``None`` (auto-detect from git root)
|
||||||
|
- **Description**: Base path to strip from code file paths when displaying titles.
|
||||||
|
When ``None``, automatically detects the relative path from the Sphinx source
|
||||||
|
directory to the git root and strips that prefix from file paths.
|
||||||
|
|
||||||
|
.. versionadded:: 0.4.0
|
||||||
|
|||||||
@@ -12,7 +12,7 @@ from .manager import LLMSFullManager
|
|||||||
from .processor import DocumentProcessor
|
from .processor import DocumentProcessor
|
||||||
from .writer import FileWriter
|
from .writer import FileWriter
|
||||||
|
|
||||||
__version__ = "0.3.1"
|
__version__ = "0.4.1"
|
||||||
|
|
||||||
# Export classes needed by tests
|
# Export classes needed by tests
|
||||||
__all__ = [
|
__all__ = [
|
||||||
@@ -76,6 +76,8 @@ def build_finished(app: Sphinx, exception):
|
|||||||
"llms_txt_full_max_size": app.config.llms_txt_full_max_size,
|
"llms_txt_full_max_size": app.config.llms_txt_full_max_size,
|
||||||
"llms_txt_directives": app.config.llms_txt_directives,
|
"llms_txt_directives": app.config.llms_txt_directives,
|
||||||
"llms_txt_exclude": app.config.llms_txt_exclude,
|
"llms_txt_exclude": app.config.llms_txt_exclude,
|
||||||
|
"llms_txt_code_files": app.config.llms_txt_code_files,
|
||||||
|
"llms_txt_code_base_path": app.config.llms_txt_code_base_path,
|
||||||
"html_baseurl": getattr(app.config, "html_baseurl", ""),
|
"html_baseurl": getattr(app.config, "html_baseurl", ""),
|
||||||
}
|
}
|
||||||
_manager.set_config(config)
|
_manager.set_config(config)
|
||||||
@@ -104,6 +106,8 @@ def setup(app: Sphinx) -> Dict[str, Any]:
|
|||||||
app.add_config_value("llms_txt_title", None, "env")
|
app.add_config_value("llms_txt_title", None, "env")
|
||||||
app.add_config_value("llms_txt_summary", None, "env")
|
app.add_config_value("llms_txt_summary", None, "env")
|
||||||
app.add_config_value("llms_txt_exclude", [], "env")
|
app.add_config_value("llms_txt_exclude", [], "env")
|
||||||
|
app.add_config_value("llms_txt_code_files", [], "env")
|
||||||
|
app.add_config_value("llms_txt_code_base_path", None, "env")
|
||||||
|
|
||||||
# Connect to Sphinx events
|
# Connect to Sphinx events
|
||||||
app.connect("doctree-resolved", doctree_resolved)
|
app.connect("doctree-resolved", doctree_resolved)
|
||||||
|
|||||||
+445
-2
@@ -2,8 +2,10 @@
|
|||||||
Main manager module for sphinx-llms-txt.
|
Main manager module for sphinx-llms-txt.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
import glob
|
||||||
|
import subprocess
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any, Dict, Optional, Tuple
|
from typing import Any, Dict, List, Optional, Tuple
|
||||||
|
|
||||||
from sphinx.application import Sphinx
|
from sphinx.application import Sphinx
|
||||||
from sphinx.environment import BuildEnvironment
|
from sphinx.environment import BuildEnvironment
|
||||||
@@ -16,6 +18,104 @@ from .writer import FileWriter
|
|||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
def _get_git_root(path: Path) -> Optional[Path]:
|
||||||
|
"""Get the git root directory for a given path."""
|
||||||
|
try:
|
||||||
|
result = subprocess.run(
|
||||||
|
["git", "rev-parse", "--show-toplevel"],
|
||||||
|
cwd=path,
|
||||||
|
capture_output=True,
|
||||||
|
text=True,
|
||||||
|
check=True,
|
||||||
|
)
|
||||||
|
return Path(result.stdout.strip())
|
||||||
|
except (subprocess.CalledProcessError, FileNotFoundError):
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _get_language_from_extension(file_path: Path) -> str:
|
||||||
|
"""Map file extension to language identifier for code blocks."""
|
||||||
|
extension_map = {
|
||||||
|
".py": "python",
|
||||||
|
".js": "javascript",
|
||||||
|
".jsx": "jsx",
|
||||||
|
".ts": "typescript",
|
||||||
|
".tsx": "tsx",
|
||||||
|
".java": "java",
|
||||||
|
".c": "c",
|
||||||
|
".cpp": "cpp",
|
||||||
|
".cc": "cpp",
|
||||||
|
".cxx": "cpp",
|
||||||
|
".h": "c",
|
||||||
|
".hpp": "cpp",
|
||||||
|
".cs": "csharp",
|
||||||
|
".php": "php",
|
||||||
|
".rb": "ruby",
|
||||||
|
".go": "go",
|
||||||
|
".rs": "rust",
|
||||||
|
".swift": "swift",
|
||||||
|
".kt": "kotlin",
|
||||||
|
".scala": "scala",
|
||||||
|
".sh": "bash",
|
||||||
|
".bash": "bash",
|
||||||
|
".zsh": "zsh",
|
||||||
|
".fish": "fish",
|
||||||
|
".ps1": "powershell",
|
||||||
|
".html": "html",
|
||||||
|
".htm": "html",
|
||||||
|
".xml": "xml",
|
||||||
|
".css": "css",
|
||||||
|
".scss": "scss",
|
||||||
|
".sass": "sass",
|
||||||
|
".less": "less",
|
||||||
|
".json": "json",
|
||||||
|
".yaml": "yaml",
|
||||||
|
".yml": "yaml",
|
||||||
|
".toml": "toml",
|
||||||
|
".ini": "ini",
|
||||||
|
".cfg": "ini",
|
||||||
|
".conf": "ini",
|
||||||
|
".sql": "sql",
|
||||||
|
".md": "markdown",
|
||||||
|
".rst": "rst",
|
||||||
|
".txt": "text",
|
||||||
|
".dockerfile": "dockerfile",
|
||||||
|
".dockerignore": "text",
|
||||||
|
".gitignore": "text",
|
||||||
|
".gitattributes": "text",
|
||||||
|
".editorconfig": "ini",
|
||||||
|
".makefile": "makefile",
|
||||||
|
".r": "r",
|
||||||
|
".R": "r",
|
||||||
|
".m": "matlab",
|
||||||
|
".pl": "perl",
|
||||||
|
".lua": "lua",
|
||||||
|
".vim": "vim",
|
||||||
|
".vimrc": "vim",
|
||||||
|
".proto": "protobuf",
|
||||||
|
".thrift": "thrift",
|
||||||
|
".graphql": "graphql",
|
||||||
|
".gql": "graphql",
|
||||||
|
}
|
||||||
|
|
||||||
|
# Get the extension from the file path
|
||||||
|
ext = file_path.suffix.lower()
|
||||||
|
|
||||||
|
# Handle special cases like Makefile, Dockerfile without extension
|
||||||
|
if not ext:
|
||||||
|
name = file_path.name.lower()
|
||||||
|
if name in ["makefile", "gnumakefile"]:
|
||||||
|
return "makefile"
|
||||||
|
elif name in ["dockerfile", "dockerfile.dev", "dockerfile.prod"]:
|
||||||
|
return "dockerfile"
|
||||||
|
elif name.startswith("dockerfile."):
|
||||||
|
return "dockerfile"
|
||||||
|
else:
|
||||||
|
return "text"
|
||||||
|
|
||||||
|
return extension_map.get(ext, "text")
|
||||||
|
|
||||||
|
|
||||||
class LLMSFullManager:
|
class LLMSFullManager:
|
||||||
"""Manages the collection and ordering of documentation sources."""
|
"""Manages the collection and ordering of documentation sources."""
|
||||||
|
|
||||||
@@ -163,9 +263,15 @@ class LLMSFullManager:
|
|||||||
# Generate content
|
# Generate content
|
||||||
content_parts = []
|
content_parts = []
|
||||||
|
|
||||||
|
# Track code files for later processing
|
||||||
|
code_file_parts = []
|
||||||
|
|
||||||
|
# Count lines in code files (initially 0)
|
||||||
|
code_files_line_count = 0
|
||||||
|
|
||||||
# Add pages in order
|
# Add pages in order
|
||||||
added_files = set()
|
added_files = set()
|
||||||
total_line_count = 0
|
total_line_count = code_files_line_count
|
||||||
max_lines = self.config.get("llms_txt_full_max_size")
|
max_lines = self.config.get("llms_txt_full_max_size")
|
||||||
abort_due_to_max_lines = False
|
abort_due_to_max_lines = False
|
||||||
|
|
||||||
@@ -277,6 +383,37 @@ class LLMSFullManager:
|
|||||||
content_parts.append(content)
|
content_parts.append(content)
|
||||||
total_line_count += line_count
|
total_line_count += line_count
|
||||||
|
|
||||||
|
# Process code files at the end if configured
|
||||||
|
if not abort_due_to_max_lines:
|
||||||
|
code_file_parts, processed_file_paths = self._process_code_files()
|
||||||
|
code_files_line_count = sum(
|
||||||
|
part.count("\n") + 1 for part in code_file_parts
|
||||||
|
)
|
||||||
|
|
||||||
|
# Check if adding code files would exceed the maximum line count
|
||||||
|
max_lines = self.config.get("llms_txt_full_max_size")
|
||||||
|
if (
|
||||||
|
max_lines is not None
|
||||||
|
and total_line_count + code_files_line_count > max_lines
|
||||||
|
):
|
||||||
|
logger.warning(
|
||||||
|
f"sphinx-llms-txt: Adding code files would exceed max line limit "
|
||||||
|
f"({max_lines}). Current: {total_line_count}, "
|
||||||
|
f"Code files: {code_files_line_count}. Skipping code files."
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
# Add source code files section if there are any code files
|
||||||
|
if code_file_parts:
|
||||||
|
section_header = self._create_code_files_section_header(
|
||||||
|
processed_file_paths
|
||||||
|
)
|
||||||
|
content_parts.append(section_header)
|
||||||
|
content_parts.extend(code_file_parts)
|
||||||
|
# Add line count for the section header too
|
||||||
|
total_line_count += (
|
||||||
|
code_files_line_count + section_header.count("\n") + 1
|
||||||
|
)
|
||||||
|
|
||||||
# Check if line limit was exceeded before creating the file
|
# Check if line limit was exceeded before creating the file
|
||||||
max_lines = self.config.get("llms_txt_full_max_size")
|
max_lines = self.config.get("llms_txt_full_max_size")
|
||||||
if abort_due_to_max_lines or (
|
if abort_due_to_max_lines or (
|
||||||
@@ -370,3 +507,309 @@ class LLMSFullManager:
|
|||||||
return source_suffix
|
return source_suffix
|
||||||
else:
|
else:
|
||||||
return [source_suffix] # String format
|
return [source_suffix] # String format
|
||||||
|
|
||||||
|
def _process_code_files(self) -> Tuple[List[str], List[Path]]:
|
||||||
|
"""Process code files specified in llms_txt_code_files configuration.
|
||||||
|
|
||||||
|
Supports include/exclude patterns with +:/- : prefixes:
|
||||||
|
- '+:pattern' = include files matching pattern
|
||||||
|
- '-:pattern' = exclude files matching pattern
|
||||||
|
- 'pattern' (no prefix) = ignored (no special handling)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Tuple of (formatted code block strings, list of processed file paths)
|
||||||
|
"""
|
||||||
|
code_file_patterns = self.config.get("llms_txt_code_files", [])
|
||||||
|
if not code_file_patterns:
|
||||||
|
return [], []
|
||||||
|
|
||||||
|
# Parse patterns into include and exclude lists
|
||||||
|
include_patterns = []
|
||||||
|
exclude_patterns = []
|
||||||
|
|
||||||
|
for pattern in code_file_patterns:
|
||||||
|
if pattern.startswith("-:"):
|
||||||
|
exclude_patterns.append(pattern[2:]) # Remove the '-:' prefix
|
||||||
|
elif pattern.startswith("+:"):
|
||||||
|
include_patterns.append(pattern[2:]) # Remove the '+:' prefix
|
||||||
|
else:
|
||||||
|
# No prefix = log warning about ignored pattern
|
||||||
|
logger.warning(
|
||||||
|
f"sphinx-llms-txt: Code file pattern '{pattern}' ignored."
|
||||||
|
f"Use '+:{pattern}' to include or '-:{pattern}' to exclude."
|
||||||
|
)
|
||||||
|
|
||||||
|
# If no include patterns specified, nothing to process
|
||||||
|
if not include_patterns:
|
||||||
|
return [], []
|
||||||
|
|
||||||
|
code_parts = []
|
||||||
|
processed_files = set()
|
||||||
|
all_matching_files = set()
|
||||||
|
|
||||||
|
# First, collect all files matching include patterns
|
||||||
|
for pattern in include_patterns:
|
||||||
|
# Resolve pattern relative to source directory
|
||||||
|
if self.srcdir:
|
||||||
|
pattern_path = Path(self.srcdir) / pattern
|
||||||
|
else:
|
||||||
|
pattern_path = Path(pattern)
|
||||||
|
|
||||||
|
# Use glob to find matching files
|
||||||
|
matching_files = glob.glob(str(pattern_path), recursive=True)
|
||||||
|
|
||||||
|
for file_path_str in matching_files:
|
||||||
|
file_path = Path(file_path_str)
|
||||||
|
if file_path.is_file(): # Only add files, not directories
|
||||||
|
all_matching_files.add(file_path.resolve())
|
||||||
|
|
||||||
|
# Filter out files matching exclude patterns
|
||||||
|
filtered_files = set()
|
||||||
|
for file_path in all_matching_files:
|
||||||
|
should_exclude = False
|
||||||
|
|
||||||
|
for exclude_pattern in exclude_patterns:
|
||||||
|
# Resolve exclude pattern relative to source directory
|
||||||
|
if self.srcdir:
|
||||||
|
exclude_pattern_path = Path(self.srcdir) / exclude_pattern
|
||||||
|
else:
|
||||||
|
exclude_pattern_path = Path(exclude_pattern)
|
||||||
|
|
||||||
|
# Check if this file matches the exclude pattern
|
||||||
|
exclude_matches = glob.glob(str(exclude_pattern_path), recursive=True)
|
||||||
|
if str(file_path) in exclude_matches:
|
||||||
|
should_exclude = True
|
||||||
|
logger.debug(
|
||||||
|
f"sphinx-llms-txt: Excluding code file: {file_path} "
|
||||||
|
f"(matched pattern: {exclude_pattern})"
|
||||||
|
)
|
||||||
|
break
|
||||||
|
|
||||||
|
if not should_exclude:
|
||||||
|
filtered_files.add(file_path)
|
||||||
|
|
||||||
|
# Sort files for consistent ordering
|
||||||
|
sorted_files = sorted(filtered_files)
|
||||||
|
|
||||||
|
for file_path in sorted_files:
|
||||||
|
# Skip if already processed (shouldn't happen with set, but safety check)
|
||||||
|
if file_path in processed_files:
|
||||||
|
continue
|
||||||
|
|
||||||
|
try:
|
||||||
|
# Read the file content
|
||||||
|
with open(file_path, "r", encoding="utf-8", errors="ignore") as f:
|
||||||
|
content = f.read()
|
||||||
|
|
||||||
|
# Get language identifier
|
||||||
|
language = _get_language_from_extension(file_path)
|
||||||
|
|
||||||
|
# Get relative path from source directory for title
|
||||||
|
if self.srcdir:
|
||||||
|
try:
|
||||||
|
title = file_path.relative_to(Path(self.srcdir))
|
||||||
|
|
||||||
|
# Strip base path if configured,
|
||||||
|
# or auto-detect from git root
|
||||||
|
base_path = self.config.get("llms_txt_code_base_path")
|
||||||
|
if base_path is None:
|
||||||
|
# Auto-detect: try to make path relative to git root
|
||||||
|
git_root = _get_git_root(Path(self.srcdir))
|
||||||
|
if git_root:
|
||||||
|
try:
|
||||||
|
# Get srcdir relative to git root
|
||||||
|
srcdir_relative = Path(self.srcdir).relative_to(
|
||||||
|
git_root
|
||||||
|
)
|
||||||
|
# Calculate relative path from srcdir to
|
||||||
|
# git root
|
||||||
|
if srcdir_relative != Path("."):
|
||||||
|
# Count directory levels to go up
|
||||||
|
up_levels = len(srcdir_relative.parts)
|
||||||
|
base_path = "../" * up_levels
|
||||||
|
else:
|
||||||
|
base_path = None
|
||||||
|
except ValueError:
|
||||||
|
base_path = None
|
||||||
|
|
||||||
|
if base_path:
|
||||||
|
title_str = str(title)
|
||||||
|
if title_str.startswith(base_path):
|
||||||
|
title = Path(title_str[len(base_path) :])
|
||||||
|
except ValueError:
|
||||||
|
# File is not relative to srcdir, use filename
|
||||||
|
title = file_path.name
|
||||||
|
else:
|
||||||
|
title = file_path.name
|
||||||
|
|
||||||
|
# Format as code block with equals underline
|
||||||
|
title_str = str(title)
|
||||||
|
equals_line = "=" * len(title_str)
|
||||||
|
|
||||||
|
# Indent the content for reStructuredText code-block directive
|
||||||
|
indented_content = "\n".join(
|
||||||
|
f" {line}" if line.strip() else ""
|
||||||
|
for line in content.splitlines()
|
||||||
|
)
|
||||||
|
|
||||||
|
code_block = f"""
|
||||||
|
{title_str}
|
||||||
|
{equals_line}
|
||||||
|
|
||||||
|
.. code-block:: {language}
|
||||||
|
|
||||||
|
{indented_content}"""
|
||||||
|
code_parts.append(code_block)
|
||||||
|
|
||||||
|
processed_files.add(file_path)
|
||||||
|
logger.debug(f"sphinx-llms-txt: Added code file: {title}")
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.warning(
|
||||||
|
f"sphinx-llms-txt: Error reading code file {file_path}: {e}"
|
||||||
|
)
|
||||||
|
continue
|
||||||
|
|
||||||
|
return code_parts, sorted(processed_files)
|
||||||
|
|
||||||
|
def _create_code_files_section_header(self, file_paths: List[Path] = None) -> str:
|
||||||
|
"""Create the section header for source code files.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
file_paths: List of file paths that were added to generate tree view
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
String containing the section header with title, underlines, description,
|
||||||
|
and file tree
|
||||||
|
"""
|
||||||
|
section_title = "Source Code Files"
|
||||||
|
star_line = "*" * len(section_title)
|
||||||
|
|
||||||
|
description = "This section contains source code files from the project repository. These files are included to provide implementation context and technical details that complement the documentation above." # noqa: E501
|
||||||
|
|
||||||
|
header = f"""
|
||||||
|
{star_line}
|
||||||
|
{section_title}
|
||||||
|
{star_line}
|
||||||
|
|
||||||
|
{description}"""
|
||||||
|
|
||||||
|
# Add file tree if file paths are provided
|
||||||
|
if file_paths:
|
||||||
|
tree_display = self._generate_file_tree(file_paths)
|
||||||
|
header += f"""
|
||||||
|
|
||||||
|
**Files included:**
|
||||||
|
|
||||||
|
.. code-block:: text
|
||||||
|
|
||||||
|
{tree_display}"""
|
||||||
|
|
||||||
|
return header
|
||||||
|
|
||||||
|
def _generate_file_tree(self, file_paths: List[Path]) -> str:
|
||||||
|
"""Generate a tree-like representation of file paths.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
file_paths: List of file paths to display in tree format
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
String containing indented tree representation of the files
|
||||||
|
"""
|
||||||
|
if not file_paths:
|
||||||
|
return ""
|
||||||
|
|
||||||
|
# Convert to relative paths if possible and create tree structure
|
||||||
|
tree_data = {}
|
||||||
|
|
||||||
|
for file_path in sorted(file_paths):
|
||||||
|
# Get relative path from source directory for display
|
||||||
|
if self.srcdir:
|
||||||
|
try:
|
||||||
|
rel_path = file_path.relative_to(Path(self.srcdir))
|
||||||
|
|
||||||
|
# Apply base path stripping logic similar to code processing
|
||||||
|
base_path = self.config.get("llms_txt_code_base_path")
|
||||||
|
if base_path is None:
|
||||||
|
# Auto-detect: try to make path relative to git root
|
||||||
|
git_root = _get_git_root(Path(self.srcdir))
|
||||||
|
if git_root:
|
||||||
|
try:
|
||||||
|
# Get srcdir relative to git root
|
||||||
|
srcdir_relative = Path(self.srcdir).relative_to(
|
||||||
|
git_root
|
||||||
|
)
|
||||||
|
# Calculate relative path from srcdir to git root
|
||||||
|
if srcdir_relative != Path("."):
|
||||||
|
# Count directory levels to go up
|
||||||
|
up_levels = len(srcdir_relative.parts)
|
||||||
|
base_path = "../" * up_levels
|
||||||
|
else:
|
||||||
|
base_path = None
|
||||||
|
except ValueError:
|
||||||
|
base_path = None
|
||||||
|
|
||||||
|
if base_path:
|
||||||
|
rel_path_str = str(rel_path)
|
||||||
|
if rel_path_str.startswith(base_path):
|
||||||
|
rel_path = Path(rel_path_str[len(base_path) :])
|
||||||
|
|
||||||
|
except ValueError:
|
||||||
|
# File is not relative to srcdir, use filename
|
||||||
|
rel_path = Path(file_path.name)
|
||||||
|
else:
|
||||||
|
rel_path = Path(file_path.name)
|
||||||
|
|
||||||
|
# Build nested dictionary structure
|
||||||
|
parts = rel_path.parts
|
||||||
|
current = tree_data
|
||||||
|
for part in parts[:-1]: # All but the last part (directories)
|
||||||
|
if part not in current:
|
||||||
|
current[part] = {}
|
||||||
|
current = current[part]
|
||||||
|
|
||||||
|
# Add the file (last part)
|
||||||
|
if parts:
|
||||||
|
current[parts[-1]] = None # None indicates it's a file
|
||||||
|
|
||||||
|
# Convert tree structure to string representation
|
||||||
|
lines = []
|
||||||
|
self._format_tree_node(tree_data, lines, "", True)
|
||||||
|
|
||||||
|
# Indent each line for reStructuredText code block
|
||||||
|
indented_lines = [f" {line}" for line in lines]
|
||||||
|
return "\n".join(indented_lines)
|
||||||
|
|
||||||
|
def _format_tree_node(
|
||||||
|
self, node: dict, lines: List[str], prefix: str, is_root: bool
|
||||||
|
):
|
||||||
|
"""Recursively format tree nodes into lines with proper tree characters.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
node: Dictionary representing the tree structure
|
||||||
|
lines: List to append formatted lines to
|
||||||
|
prefix: Current prefix for indentation and tree characters
|
||||||
|
is_root: Whether this is the root level (no tree characters)
|
||||||
|
"""
|
||||||
|
if not node:
|
||||||
|
return
|
||||||
|
|
||||||
|
items = sorted(node.items())
|
||||||
|
|
||||||
|
for i, (name, subtree) in enumerate(items):
|
||||||
|
is_last = i == len(items) - 1
|
||||||
|
|
||||||
|
if is_root:
|
||||||
|
# Root level - no tree characters
|
||||||
|
current_prefix = ""
|
||||||
|
next_prefix = ""
|
||||||
|
else:
|
||||||
|
# Use tree characters
|
||||||
|
current_prefix = prefix + ("└── " if is_last else "├── ")
|
||||||
|
next_prefix = prefix + (" " if is_last else "│ ")
|
||||||
|
|
||||||
|
lines.append(current_prefix + name)
|
||||||
|
|
||||||
|
# Recursively handle subdirectories
|
||||||
|
if subtree is not None: # It's a directory
|
||||||
|
self._format_tree_node(subtree, lines, next_prefix, False)
|
||||||
|
|||||||
+106
-40
@@ -94,8 +94,14 @@ class DocumentProcessor:
|
|||||||
if not base_url:
|
if not base_url:
|
||||||
return path
|
return path
|
||||||
|
|
||||||
|
# Ensure base URL ends with slash
|
||||||
if not base_url.endswith("/"):
|
if not base_url.endswith("/"):
|
||||||
base_url += "/"
|
base_url += "/"
|
||||||
|
|
||||||
|
# Remove leading slash from path to avoid double slashes
|
||||||
|
if path.startswith("/"):
|
||||||
|
path = path[1:]
|
||||||
|
|
||||||
return f"{base_url}{path}"
|
return f"{base_url}{path}"
|
||||||
|
|
||||||
def _is_absolute_or_url(self, path: str) -> bool:
|
def _is_absolute_or_url(self, path: str) -> bool:
|
||||||
@@ -137,12 +143,73 @@ class DocumentProcessor:
|
|||||||
prefix = match.group(1) # The entire directive prefix including whitespace
|
prefix = match.group(1) # The entire directive prefix including whitespace
|
||||||
path = match.group(3).strip() # The path argument
|
path = match.group(3).strip() # The path argument
|
||||||
|
|
||||||
# Only process relative paths, not absolute paths or URLs
|
# Handle URLs and data URIs - leave unchanged
|
||||||
if not self._is_absolute_or_url(path):
|
if path.startswith(("http://", "https://", "data:")):
|
||||||
# Special case for test files
|
return match.group(0)
|
||||||
if is_test:
|
|
||||||
# Add subdir/ prefix to match test expectations
|
# For ALL paths, check if image exists in _images first
|
||||||
full_path = "subdir/" + path
|
# Extract filename from the path
|
||||||
|
filename = os.path.basename(path)
|
||||||
|
|
||||||
|
# Check if image exists in _images directory
|
||||||
|
# First determine the build directory from source_path
|
||||||
|
build_dir = None
|
||||||
|
if "_sources" in str(source_path):
|
||||||
|
# Extract build directory (parent of _sources)
|
||||||
|
path_parts = str(source_path).split("_sources/")
|
||||||
|
if len(path_parts) > 1:
|
||||||
|
build_dir = path_parts[0].rstrip("/")
|
||||||
|
|
||||||
|
# If we can determine the build directory, check if image exists in _images
|
||||||
|
if build_dir:
|
||||||
|
images_path = os.path.join(build_dir, "_images", filename)
|
||||||
|
if os.path.exists(images_path):
|
||||||
|
# Image exists in _images, use _images path
|
||||||
|
full_path = f"/_images/{filename}"
|
||||||
|
# Add base URL if configured
|
||||||
|
full_path = self._add_base_url(full_path, base_url)
|
||||||
|
return f"{prefix}{full_path}"
|
||||||
|
|
||||||
|
# Image doesn't exist in _images, handle based on path type
|
||||||
|
# Handle absolute paths (starting with /) - add base URL if configured
|
||||||
|
if path.startswith("/"):
|
||||||
|
# Add base URL to absolute paths if configured
|
||||||
|
full_path = self._add_base_url(path, base_url)
|
||||||
|
return f"{prefix}{full_path}"
|
||||||
|
|
||||||
|
# Handle relative paths with original logic for backward compatibility
|
||||||
|
# Special case for test files
|
||||||
|
if is_test:
|
||||||
|
# Add subdir/ prefix to match test expectations
|
||||||
|
full_path = "subdir/" + path
|
||||||
|
|
||||||
|
# If base_url is set, prepend it to the path
|
||||||
|
full_path = self._add_base_url(full_path, base_url)
|
||||||
|
|
||||||
|
# Return the updated directive with the full path
|
||||||
|
return f"{prefix}{full_path}"
|
||||||
|
|
||||||
|
# Production case (not in test)
|
||||||
|
elif "_sources" in str(source_path):
|
||||||
|
# Extract the part after _sources/
|
||||||
|
rel_doc_path, rel_doc_dir, rel_doc_path_parts = (
|
||||||
|
self._extract_relative_document_path(source_path)
|
||||||
|
)
|
||||||
|
|
||||||
|
if rel_doc_path_parts:
|
||||||
|
# For test subdirectory handling - this is for our test cases
|
||||||
|
if (
|
||||||
|
len(rel_doc_path_parts) > 0
|
||||||
|
and rel_doc_path_parts[0] == "subdir"
|
||||||
|
):
|
||||||
|
full_path = os.path.normpath(os.path.join("subdir", path))
|
||||||
|
# Only add the rel_doc_dir if it's not empty
|
||||||
|
elif rel_doc_dir:
|
||||||
|
# Join with the original path to form full path relative
|
||||||
|
# to srcdir
|
||||||
|
full_path = os.path.normpath(os.path.join(rel_doc_dir, path))
|
||||||
|
else:
|
||||||
|
full_path = path
|
||||||
|
|
||||||
# If base_url is set, prepend it to the path
|
# If base_url is set, prepend it to the path
|
||||||
full_path = self._add_base_url(full_path, base_url)
|
full_path = self._add_base_url(full_path, base_url)
|
||||||
@@ -150,37 +217,12 @@ class DocumentProcessor:
|
|||||||
# Return the updated directive with the full path
|
# Return the updated directive with the full path
|
||||||
return f"{prefix}{full_path}"
|
return f"{prefix}{full_path}"
|
||||||
|
|
||||||
# Production case (not in test)
|
# Fallback for relative paths - add base URL if configured
|
||||||
elif "_sources" in str(source_path):
|
else:
|
||||||
# Extract the part after _sources/
|
full_path = self._add_base_url(path, base_url)
|
||||||
rel_doc_path, rel_doc_dir, rel_doc_path_parts = (
|
return f"{prefix}{full_path}"
|
||||||
self._extract_relative_document_path(source_path)
|
|
||||||
)
|
|
||||||
|
|
||||||
if rel_doc_path_parts:
|
# If we couldn't resolve the path, return unchanged
|
||||||
# For test subdirectory handling - this is for our test cases
|
|
||||||
if (
|
|
||||||
len(rel_doc_path_parts) > 0
|
|
||||||
and rel_doc_path_parts[0] == "subdir"
|
|
||||||
):
|
|
||||||
full_path = os.path.normpath(os.path.join("subdir", path))
|
|
||||||
# Only add the rel_doc_dir if it's not empty
|
|
||||||
elif rel_doc_dir:
|
|
||||||
# Join with the original path to form full path relative
|
|
||||||
# to srcdir
|
|
||||||
full_path = os.path.normpath(
|
|
||||||
os.path.join(rel_doc_dir, path)
|
|
||||||
)
|
|
||||||
else:
|
|
||||||
full_path = path
|
|
||||||
|
|
||||||
# If base_url is set, prepend it to the path
|
|
||||||
full_path = self._add_base_url(full_path, base_url)
|
|
||||||
|
|
||||||
# Return the updated directive with the full path
|
|
||||||
return f"{prefix}{full_path}"
|
|
||||||
|
|
||||||
# If we couldn't resolve the path or it's already absolute, return unchanged
|
|
||||||
return match.group(0)
|
return match.group(0)
|
||||||
|
|
||||||
# Replace directive paths in the content
|
# Replace directive paths in the content
|
||||||
@@ -201,9 +243,12 @@ class DocumentProcessor:
|
|||||||
"""
|
"""
|
||||||
possible_paths = []
|
possible_paths = []
|
||||||
|
|
||||||
# If it's an absolute path, use it directly
|
# If it's an absolute path, treat it as relative to srcdir
|
||||||
if os.path.isabs(include_path):
|
if os.path.isabs(include_path):
|
||||||
possible_paths.append(Path(include_path))
|
# Remove the leading slash and treat as relative to srcdir
|
||||||
|
relative_path = include_path.lstrip("/")
|
||||||
|
if self.srcdir:
|
||||||
|
possible_paths.append((Path(self.srcdir) / relative_path).resolve())
|
||||||
else:
|
else:
|
||||||
# Relative to the source file (in _sources directory)
|
# Relative to the source file (in _sources directory)
|
||||||
possible_paths.append((source_path.parent / include_path).resolve())
|
possible_paths.append((source_path.parent / include_path).resolve())
|
||||||
@@ -244,6 +289,9 @@ class DocumentProcessor:
|
|||||||
# Function to replace each include with content
|
# Function to replace each include with content
|
||||||
def replace_include(match):
|
def replace_include(match):
|
||||||
include_path = match.group(3)
|
include_path = match.group(3)
|
||||||
|
directive_part = match.group(
|
||||||
|
1
|
||||||
|
) # The ".. include:: " part with leading whitespace
|
||||||
|
|
||||||
# Get all possible paths to try
|
# Get all possible paths to try
|
||||||
possible_paths = self._resolve_include_paths(include_path, source_path)
|
possible_paths = self._resolve_include_paths(include_path, source_path)
|
||||||
@@ -254,7 +302,18 @@ class DocumentProcessor:
|
|||||||
if path_to_try.exists():
|
if path_to_try.exists():
|
||||||
with open(path_to_try, "r", encoding="utf-8") as f:
|
with open(path_to_try, "r", encoding="utf-8") as f:
|
||||||
included_content = f.read()
|
included_content = f.read()
|
||||||
return included_content
|
|
||||||
|
# Find where the actual directive starts, after any whitespace
|
||||||
|
directive_start = directive_part.find("..")
|
||||||
|
if directive_start > 0:
|
||||||
|
# There's leading whitespace/newlines before the directive
|
||||||
|
leading_part = directive_part[:directive_start]
|
||||||
|
# Replace directive with content, preserving the structure
|
||||||
|
return leading_part + included_content
|
||||||
|
else:
|
||||||
|
# No leading whitespace, just return the content
|
||||||
|
return included_content
|
||||||
|
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.error(
|
logger.error(
|
||||||
f"sphinx-llms-txt: Error reading include file {path_to_try}:"
|
f"sphinx-llms-txt: Error reading include file {path_to_try}:"
|
||||||
@@ -266,7 +325,14 @@ class DocumentProcessor:
|
|||||||
paths_tried = ", ".join(str(p) for p in possible_paths)
|
paths_tried = ", ".join(str(p) for p in possible_paths)
|
||||||
logger.warning(f"sphinx-llms-txt: Include file not found: {include_path}")
|
logger.warning(f"sphinx-llms-txt: Include file not found: {include_path}")
|
||||||
logger.debug(f"sphinx-llms-txt: Tried paths: {paths_tried}")
|
logger.debug(f"sphinx-llms-txt: Tried paths: {paths_tried}")
|
||||||
return f"[Include file not found: {include_path}]"
|
|
||||||
|
# Preserve spacing structure for error message too
|
||||||
|
directive_start = match.group(1).find("..")
|
||||||
|
if directive_start > 0:
|
||||||
|
leading_part = match.group(1)[:directive_start]
|
||||||
|
return leading_part + f"[Include file not found: {include_path}]"
|
||||||
|
else:
|
||||||
|
return f"[Include file not found: {include_path}]"
|
||||||
|
|
||||||
# Replace all includes with their content
|
# Replace all includes with their content
|
||||||
processed_content = include_pattern.sub(replace_include, content)
|
processed_content = include_pattern.sub(replace_include, content)
|
||||||
|
|||||||
@@ -764,6 +764,8 @@ def test_summary_default_uses_first_paragraph():
|
|||||||
llms_txt_full_max_size = None
|
llms_txt_full_max_size = None
|
||||||
llms_txt_directives = []
|
llms_txt_directives = []
|
||||||
llms_txt_exclude = []
|
llms_txt_exclude = []
|
||||||
|
llms_txt_code_files = []
|
||||||
|
llms_txt_code_base_path = None
|
||||||
html_baseurl = ""
|
html_baseurl = ""
|
||||||
|
|
||||||
config = Config()
|
config = Config()
|
||||||
@@ -808,3 +810,187 @@ def test_summary_default_uses_first_paragraph():
|
|||||||
|
|
||||||
# Restore original method
|
# Restore original method
|
||||||
sphinx_llms_txt._manager.combine_sources = original_combine_sources
|
sphinx_llms_txt._manager.combine_sources = original_combine_sources
|
||||||
|
|
||||||
|
|
||||||
|
def test_code_files_include_exclude_patterns(tmp_path):
|
||||||
|
"""Test the +/- pattern syntax for llms_txt_code_files configuration."""
|
||||||
|
from sphinx_llms_txt.manager import LLMSFullManager
|
||||||
|
|
||||||
|
# Create test directory structure
|
||||||
|
src_dir = tmp_path / "src"
|
||||||
|
src_dir.mkdir()
|
||||||
|
|
||||||
|
docs_dir = src_dir / "docs"
|
||||||
|
docs_dir.mkdir()
|
||||||
|
|
||||||
|
cache_dir = docs_dir / "__pycache__"
|
||||||
|
cache_dir.mkdir()
|
||||||
|
|
||||||
|
# Create test files
|
||||||
|
(docs_dir / "example.rst").write_text("Example RST content")
|
||||||
|
(docs_dir / "guide.rst").write_text("Guide RST content")
|
||||||
|
(docs_dir / "backup.bak").write_text("Backup file content")
|
||||||
|
(cache_dir / "compiled.pyc").write_text("Compiled Python")
|
||||||
|
|
||||||
|
# Create manager and set source directory
|
||||||
|
manager = LLMSFullManager()
|
||||||
|
manager.srcdir = str(src_dir)
|
||||||
|
|
||||||
|
# Test configuration with include/exclude patterns
|
||||||
|
config = {
|
||||||
|
"llms_txt_code_files": [
|
||||||
|
"+:docs/**/*.rst", # Include all RST files in docs
|
||||||
|
"-:docs/**/__pycache__/**", # Exclude pycache files
|
||||||
|
"-:docs/**/*.bak", # Exclude backup files
|
||||||
|
]
|
||||||
|
}
|
||||||
|
manager.set_config(config)
|
||||||
|
|
||||||
|
# Process code files
|
||||||
|
code_parts, _ = manager._process_code_files()
|
||||||
|
|
||||||
|
# Verify we have the expected number of files
|
||||||
|
assert len(code_parts) == 2, f"Expected 2 files, got {len(code_parts)}"
|
||||||
|
|
||||||
|
# Extract file titles from code blocks
|
||||||
|
titles = []
|
||||||
|
for part in code_parts:
|
||||||
|
lines = part.strip().split("\n")
|
||||||
|
if lines:
|
||||||
|
titles.append(lines[0])
|
||||||
|
|
||||||
|
# Verify expected files are included
|
||||||
|
assert "docs/example.rst" in titles
|
||||||
|
assert "docs/guide.rst" in titles
|
||||||
|
|
||||||
|
# Verify excluded files are not present
|
||||||
|
content = "\n".join(code_parts)
|
||||||
|
assert "backup.bak" not in content
|
||||||
|
assert "__pycache__" not in content
|
||||||
|
assert "compiled.pyc" not in content
|
||||||
|
|
||||||
|
|
||||||
|
def test_code_files_exclude_only_patterns(tmp_path):
|
||||||
|
"""Test that exclude-only patterns result in no files being included."""
|
||||||
|
from sphinx_llms_txt.manager import LLMSFullManager
|
||||||
|
|
||||||
|
# Create test directory structure
|
||||||
|
src_dir = tmp_path / "src"
|
||||||
|
src_dir.mkdir()
|
||||||
|
|
||||||
|
docs_dir = src_dir / "docs"
|
||||||
|
docs_dir.mkdir()
|
||||||
|
|
||||||
|
# Create test files
|
||||||
|
(docs_dir / "example.rst").write_text("Example RST content")
|
||||||
|
|
||||||
|
# Create manager and set source directory
|
||||||
|
manager = LLMSFullManager()
|
||||||
|
manager.srcdir = str(src_dir)
|
||||||
|
|
||||||
|
# Test configuration with only exclude patterns
|
||||||
|
config = {
|
||||||
|
"llms_txt_code_files": [
|
||||||
|
"-:docs/**/*.rst", # Only exclude pattern, no includes
|
||||||
|
]
|
||||||
|
}
|
||||||
|
manager.set_config(config)
|
||||||
|
|
||||||
|
# Process code files
|
||||||
|
code_parts, _ = manager._process_code_files()
|
||||||
|
|
||||||
|
# Should have no files with exclude-only patterns
|
||||||
|
assert len(code_parts) == 0, "Should have no files with exclude-only patterns"
|
||||||
|
|
||||||
|
|
||||||
|
def test_code_files_no_prefix_patterns(tmp_path):
|
||||||
|
"""Test that patterns without prefix are ignored."""
|
||||||
|
from sphinx_llms_txt.manager import LLMSFullManager
|
||||||
|
|
||||||
|
# Create test directory structure
|
||||||
|
src_dir = tmp_path / "src"
|
||||||
|
src_dir.mkdir()
|
||||||
|
|
||||||
|
docs_dir = src_dir / "docs"
|
||||||
|
docs_dir.mkdir()
|
||||||
|
|
||||||
|
# Create test files
|
||||||
|
(docs_dir / "example.rst").write_text("Example RST content")
|
||||||
|
(docs_dir / "backup.bak").write_text("Backup file content")
|
||||||
|
|
||||||
|
# Create manager and set source directory
|
||||||
|
manager = LLMSFullManager()
|
||||||
|
manager.srcdir = str(src_dir)
|
||||||
|
|
||||||
|
# Test configuration with no prefix (should be ignored)
|
||||||
|
config = {
|
||||||
|
"llms_txt_code_files": [
|
||||||
|
"docs/**/*.rst", # No prefix = ignored
|
||||||
|
"+:docs/**/*.rst", # Include RST files
|
||||||
|
"-:docs/**/*.bak", # Exclude backup files
|
||||||
|
]
|
||||||
|
}
|
||||||
|
manager.set_config(config)
|
||||||
|
|
||||||
|
# Process code files
|
||||||
|
code_parts, _ = manager._process_code_files()
|
||||||
|
|
||||||
|
# Should include RST files (from +: pattern) and exclude BAK files (from -: pattern)
|
||||||
|
assert len(code_parts) == 1, "Should include RST files and exclude BAK files"
|
||||||
|
|
||||||
|
content = "\n".join(code_parts)
|
||||||
|
assert "Example RST content" in content
|
||||||
|
assert "backup.bak" not in content
|
||||||
|
|
||||||
|
|
||||||
|
def test_code_files_ignored_patterns(tmp_path, caplog):
|
||||||
|
"""Test that patterns without +: or -: prefix log a warning and are ignored."""
|
||||||
|
from unittest.mock import patch
|
||||||
|
|
||||||
|
from sphinx_llms_txt.manager import LLMSFullManager
|
||||||
|
|
||||||
|
# Create test directory structure
|
||||||
|
src_dir = tmp_path / "src"
|
||||||
|
src_dir.mkdir()
|
||||||
|
|
||||||
|
docs_dir = src_dir / "docs"
|
||||||
|
docs_dir.mkdir()
|
||||||
|
|
||||||
|
# Create test files
|
||||||
|
(docs_dir / "example.rst").write_text("Example RST content")
|
||||||
|
|
||||||
|
# Use a mock to capture the warning message directly
|
||||||
|
captured_warnings = []
|
||||||
|
|
||||||
|
def capture_warning(message, *args, **kwargs):
|
||||||
|
captured_warnings.append(message)
|
||||||
|
|
||||||
|
# Patch the logger to capture warnings
|
||||||
|
with patch("sphinx_llms_txt.manager.logger.warning", side_effect=capture_warning):
|
||||||
|
# Create manager and set source directory
|
||||||
|
manager = LLMSFullManager()
|
||||||
|
manager.srcdir = str(src_dir)
|
||||||
|
|
||||||
|
# Test configuration with only no-prefix patterns (should result in no files)
|
||||||
|
config = {
|
||||||
|
"llms_txt_code_files": [
|
||||||
|
"docs/**/*.rst", # No prefix = ignored with warning
|
||||||
|
]
|
||||||
|
}
|
||||||
|
manager.set_config(config)
|
||||||
|
|
||||||
|
# Process code files
|
||||||
|
code_parts, _ = manager._process_code_files()
|
||||||
|
|
||||||
|
# Should have no files since the pattern without prefix is ignored
|
||||||
|
assert (
|
||||||
|
len(code_parts) == 0
|
||||||
|
), "Should have no files when only using patterns without prefix"
|
||||||
|
|
||||||
|
# Check that a warning was logged
|
||||||
|
assert (
|
||||||
|
len(captured_warnings) == 1
|
||||||
|
), f"Expected 1 warning, got {len(captured_warnings)}"
|
||||||
|
assert (
|
||||||
|
"Code file pattern 'docs/**/*.rst' ignored." in captured_warnings[0]
|
||||||
|
), f"Warning message should contain expected text. Got: {captured_warnings[0]}"
|
||||||
|
|||||||
@@ -102,7 +102,7 @@ def test_process_path_directives_with_html_baseurl(tmp_path):
|
|||||||
|
|
||||||
|
|
||||||
def test_process_path_directives_absolute_urls(tmp_path):
|
def test_process_path_directives_absolute_urls(tmp_path):
|
||||||
"""Test that absolute URLs are not modified."""
|
"""Test that absolute URLs are not modified but absolute paths get base URL."""
|
||||||
# Create a processor
|
# Create a processor
|
||||||
config = {
|
config = {
|
||||||
"llms_txt_directives": [],
|
"llms_txt_directives": [],
|
||||||
@@ -127,10 +127,17 @@ def test_process_path_directives_absolute_urls(tmp_path):
|
|||||||
with open(source_file, "w", encoding="utf-8") as f:
|
with open(source_file, "w", encoding="utf-8") as f:
|
||||||
f.write(source_content)
|
f.write(source_content)
|
||||||
|
|
||||||
# Process the directives (should remain unchanged)
|
# Process the directives
|
||||||
processed_content = processor._process_path_directives(source_content, source_file)
|
processed_content = processor._process_path_directives(source_content, source_file)
|
||||||
|
|
||||||
assert processed_content == source_content
|
# Expected: URLs and data URIs unchanged, absolute paths get base URL
|
||||||
|
expected_content = (
|
||||||
|
".. image:: https://othersite.com/images/test.png\n"
|
||||||
|
".. image:: https://example.com/docs/absolute/path/image.png\n"
|
||||||
|
".. image:: data:image/png;base64,iVBORw0KG...\n"
|
||||||
|
)
|
||||||
|
|
||||||
|
assert processed_content == expected_content
|
||||||
|
|
||||||
|
|
||||||
def test_process_path_directives_custom_directives(tmp_path):
|
def test_process_path_directives_custom_directives(tmp_path):
|
||||||
@@ -251,3 +258,149 @@ def test_process_content_end_to_end(tmp_path):
|
|||||||
)
|
)
|
||||||
|
|
||||||
assert processed_content == expected_content
|
assert processed_content == expected_content
|
||||||
|
|
||||||
|
|
||||||
|
def test_process_path_directives_images_directory(tmp_path):
|
||||||
|
"""Test that _images directory paths are handled correctly."""
|
||||||
|
# Create a processor with base URL
|
||||||
|
config = {
|
||||||
|
"llms_txt_directives": [],
|
||||||
|
"html_baseurl": "https://example.com/docs",
|
||||||
|
}
|
||||||
|
processor = DocumentProcessor(config)
|
||||||
|
|
||||||
|
# Create source directory structure
|
||||||
|
src_dir = tmp_path / "src"
|
||||||
|
src_dir.mkdir()
|
||||||
|
processor.srcdir = str(src_dir)
|
||||||
|
|
||||||
|
# Create _sources directory to mimic Sphinx output
|
||||||
|
build_dir = tmp_path / "build"
|
||||||
|
build_dir.mkdir()
|
||||||
|
sources_dir = build_dir / "_sources"
|
||||||
|
sources_dir.mkdir()
|
||||||
|
|
||||||
|
# Create a source file with various _images directory paths
|
||||||
|
source_content = (
|
||||||
|
"Some content.\n"
|
||||||
|
".. image:: _images/test.png\n" # Relative _images should become /_images
|
||||||
|
".. image:: /_images/absolute.png\n" # Absolute _images should get base URL
|
||||||
|
".. figure:: _images/figure.png\n" # Test with figure directive too
|
||||||
|
" :alt: A test figure\n"
|
||||||
|
".. image:: images/normal.png\n" # Normal relative path should be unchanged
|
||||||
|
)
|
||||||
|
|
||||||
|
# Create source file in sources directory to simulate Sphinx build output
|
||||||
|
source_file = sources_dir / "page.txt"
|
||||||
|
with open(source_file, "w", encoding="utf-8") as f:
|
||||||
|
f.write(source_content)
|
||||||
|
|
||||||
|
# Process the directives
|
||||||
|
processed_content = processor._process_path_directives(source_content, source_file)
|
||||||
|
|
||||||
|
# Expected: _images paths should be converted and get base URL
|
||||||
|
expected_content = (
|
||||||
|
"Some content.\n"
|
||||||
|
".. image:: https://example.com/docs/_images/test.png\n"
|
||||||
|
".. image:: https://example.com/docs/_images/absolute.png\n"
|
||||||
|
".. figure:: https://example.com/docs/_images/figure.png\n"
|
||||||
|
" :alt: A test figure\n"
|
||||||
|
".. image:: https://example.com/docs/images/normal.png\n"
|
||||||
|
)
|
||||||
|
|
||||||
|
assert processed_content == expected_content
|
||||||
|
|
||||||
|
|
||||||
|
def test_process_path_directives_images_directory_no_baseurl(tmp_path):
|
||||||
|
"""
|
||||||
|
Test that _images directory paths work correctly without base URL.
|
||||||
|
Only converts when image exists.
|
||||||
|
"""
|
||||||
|
# Create a processor without base URL
|
||||||
|
config = {
|
||||||
|
"llms_txt_directives": [],
|
||||||
|
"html_baseurl": "",
|
||||||
|
}
|
||||||
|
processor = DocumentProcessor(config)
|
||||||
|
|
||||||
|
# Create source directory structure
|
||||||
|
src_dir = tmp_path / "src"
|
||||||
|
src_dir.mkdir()
|
||||||
|
processor.srcdir = str(src_dir)
|
||||||
|
|
||||||
|
# Create _sources directory to mimic Sphinx output
|
||||||
|
build_dir = tmp_path / "build"
|
||||||
|
build_dir.mkdir()
|
||||||
|
sources_dir = build_dir / "_sources"
|
||||||
|
sources_dir.mkdir()
|
||||||
|
|
||||||
|
# Create _images directory and one test image
|
||||||
|
images_dir = build_dir / "_images"
|
||||||
|
images_dir.mkdir()
|
||||||
|
(images_dir / "test.png").write_text("fake image content")
|
||||||
|
# Note: absolute.png is not created, so it won't be converted
|
||||||
|
|
||||||
|
# Create a source file with _images directory paths
|
||||||
|
source_content = (
|
||||||
|
".. image:: _images/test.png\n" # Should become /_images (image exists)
|
||||||
|
".. image:: /_images/absolute.png\n" # Should stay unchanged (absolute path)
|
||||||
|
)
|
||||||
|
|
||||||
|
# Create source file in sources directory to simulate Sphinx build output
|
||||||
|
source_file = sources_dir / "page.txt"
|
||||||
|
with open(source_file, "w", encoding="utf-8") as f:
|
||||||
|
f.write(source_content)
|
||||||
|
|
||||||
|
# Process the directives
|
||||||
|
processed_content = processor._process_path_directives(source_content, source_file)
|
||||||
|
|
||||||
|
# Expected: only test.png gets converted because it exists in _images
|
||||||
|
expected_content = (
|
||||||
|
".. image:: /_images/test.png\n" # Converted because image exists
|
||||||
|
".. image:: /_images/absolute.png\n" # Absolute path unchanged
|
||||||
|
)
|
||||||
|
|
||||||
|
assert processed_content == expected_content
|
||||||
|
|
||||||
|
|
||||||
|
def test_process_path_directives_all_absolute_paths_get_baseurl(tmp_path):
|
||||||
|
"""Test that all absolute paths (starting with /) get base URL prepended."""
|
||||||
|
# Create a processor with base URL
|
||||||
|
config = {
|
||||||
|
"llms_txt_directives": [],
|
||||||
|
"html_baseurl": "https://mysite.com/docs/",
|
||||||
|
}
|
||||||
|
processor = DocumentProcessor(config)
|
||||||
|
|
||||||
|
# Create source directory structure
|
||||||
|
src_dir = tmp_path / "src"
|
||||||
|
src_dir.mkdir()
|
||||||
|
processor.srcdir = str(src_dir)
|
||||||
|
|
||||||
|
# Create a source file with various absolute paths
|
||||||
|
source_content = (
|
||||||
|
".. image:: /static/images/logo.png\n"
|
||||||
|
".. figure:: /assets/diagrams/flow.svg\n"
|
||||||
|
".. image:: /media/photos/team.jpg\n"
|
||||||
|
" :alt: Team photo\n"
|
||||||
|
".. image:: relative/path.png\n" # This should still get normal processing
|
||||||
|
)
|
||||||
|
|
||||||
|
# Create source file
|
||||||
|
source_file = src_dir / "page.txt"
|
||||||
|
with open(source_file, "w", encoding="utf-8") as f:
|
||||||
|
f.write(source_content)
|
||||||
|
|
||||||
|
# Process the directives
|
||||||
|
processed_content = processor._process_path_directives(source_content, source_file)
|
||||||
|
|
||||||
|
# Expected: All absolute paths get base URL prepended
|
||||||
|
expected_content = (
|
||||||
|
".. image:: https://mysite.com/docs/static/images/logo.png\n"
|
||||||
|
".. figure:: https://mysite.com/docs/assets/diagrams/flow.svg\n"
|
||||||
|
".. image:: https://mysite.com/docs/media/photos/team.jpg\n"
|
||||||
|
" :alt: Team photo\n"
|
||||||
|
".. image:: https://mysite.com/docs/relative/path.png\n"
|
||||||
|
)
|
||||||
|
|
||||||
|
assert processed_content == expected_content
|
||||||
|
|||||||
Reference in New Issue
Block a user