Compare commits

...
5 Commits
Author SHA1 Message Date
Jared Dillard 9bfdc38b60 Apply black formatting 2025-12-16 14:21:37 -08:00
Kayce Basques 10af26acfa Address feedback 2025-12-09 07:18:39 -08:00
Kayce Basques fcfc9c6f70 Update test comments 2025-12-07 10:02:37 -08:00
Kayce Basques b397cce765 Add fix and update test 2025-12-07 09:59:20 -08:00
Kayce Basques af34e0f293 Don't process includes within code blocks
Fixes #57
2025-12-07 09:32:09 -08:00
2 changed files with 106 additions and 1 deletions
+81 -1
View File
@@ -128,8 +128,11 @@ class DocumentProcessor:
Returns:
Processed content with directive paths properly resolved
"""
# Get code block ranges to skip directives inside them
code_block_ranges = self._get_code_block_ranges(content)
# Get the configured path directives to process
default_path_directives = ["image", "figure"]
default_path_directives = ["image", "figure", "literalinclude"]
custom_path_directives = self.config.get("llms_txt_directives")
path_directives = set(default_path_directives + custom_path_directives)
@@ -143,6 +146,11 @@ class DocumentProcessor:
is_test = "pytest" in str(source_path) and "subdir" in str(source_path)
def replace_directive_path(match, base_url=base_url, is_test=is_test):
# Check if this directive is within a code block
if self._is_in_code_block(match.start(), code_block_ranges):
# This directive is inside a code block, don't process it
return match.group(0)
prefix = match.group(1) # The entire directive prefix including whitespace
path = match.group(3).strip() # The path argument
@@ -276,6 +284,71 @@ class DocumentProcessor:
return possible_paths
def _get_code_block_ranges(self, content: str) -> List[Tuple[int, int]]:
"""Find all code block ranges in the content.
Args:
content: The source content to analyze
Returns:
List of (start, end) tuples representing code block character
ranges
"""
code_block_ranges = []
# Match code block as well as `code` and `sourcecode` aliases
code_block_pattern = re.compile(
r"^(\s*)\.\.\s+(code-block|code|sourcecode)::\s*\S*\s*$", re.MULTILINE
)
for match in code_block_pattern.finditer(content):
start_pos = match.start()
indent = match.group(1)
indent_len = len(indent)
# Find the end of the code block by looking for the next line
# that is not indented more than the directive
block_start = match.end()
pos = block_start
# Skip any blank lines immediately after the directive
while pos < len(content) and content[pos] in "\n":
pos += 1
# Find where the code block ends
lines = content[pos:].split("\n")
block_end = pos
for line in lines:
if line.strip(): # Non-empty line
# Check indentation level
line_indent = len(line) - len(line.lstrip())
if line_indent <= indent_len:
# The block ends when we find a line that is indented
# less than the directive itself
break
block_end += len(line) + 1 # +1 for the newline
code_block_ranges.append((start_pos, block_end))
return code_block_ranges
def _is_in_code_block(
self, match_start: int, code_block_ranges: List[Tuple[int, int]]
) -> bool:
"""Check if a match position is within a code block.
Args:
match_start: The starting position of the match
code_block_ranges: List of (start, end) tuples for code blocks
Returns:
True if the match is within a code block, False otherwise
"""
for block_start, block_end in code_block_ranges:
if block_start <= match_start < block_end:
return True
return False
def _process_includes(self, content: str, source_path: Path) -> str:
"""Process include directives in content.
@@ -286,11 +359,18 @@ class DocumentProcessor:
Returns:
Processed content with include directives replaced with included content
"""
code_block_ranges = self._get_code_block_ranges(content)
# Find all include directives using regex
include_pattern = build_directive_pattern(["include"])
# Function to replace each include with content
def replace_include(match):
# Check if this include is within a code block
if self._is_in_code_block(match.start(), code_block_ranges):
# This include is inside a code block, don't process it
return match.group(0)
include_path = match.group(3)
directive_part = match.group(
1
+25
View File
@@ -198,6 +198,31 @@ def test_process_includes(tmp_path):
assert processed_content == expected_content
def test_process_includes_in_code_block(tmp_path):
"""Test that an `include` within a `code-block` is not processed."""
# Create a processor
config = {"llms_txt_directives": []}
processor = DocumentProcessor(config)
# Create a source file that uses include syntax within a `code-block`
source_content = (
"Normal paragraph.\n\n"
".. code-block:: rst\n\n"
" .. include:: foo.txt\n\n"
"Another normal paragraph."
)
source_file = tmp_path / "source.txt"
with open(source_file, "w", encoding="utf-8") as f:
f.write(source_content)
# Run the include directive processor
processed_content = processor._process_includes(source_content, source_file)
# Check that the include directive was not processed
expected_content = source_content
assert processed_content == expected_content
def test_process_includes_with_relative_paths(tmp_path):
"""Test that include directives with relative paths are processed correctly."""
# Create a processor