Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
327ae7c2d7 | ||
|
|
10794dce65 | ||
|
|
f6596dd1ec | ||
|
|
29c3932488 | ||
|
|
1dd8119cfd | ||
|
|
dacabd62b4 | ||
|
|
2f8e64cc6d | ||
|
|
71dc28e7d8 | ||
|
|
cbbf4b572b | ||
|
|
8a1a73512d | ||
|
|
097f2c1084 | ||
|
|
61f9b38d4d | ||
|
|
8254002622 | ||
|
|
5b1b72bafb | ||
|
|
3a4ccca5bb | ||
|
|
0a3da6b1f9 | ||
|
|
f1a3963a0d | ||
|
|
c23c6d4468 | ||
|
|
db32dc60a7 | ||
|
|
543efabebb | ||
|
|
141e0e29f6 | ||
|
|
1b08b2f362 | ||
|
|
5bfbb0168f | ||
|
|
e2a80faf04 | ||
|
|
b63801bcff | ||
|
|
c45ebb0369 | ||
|
|
f5dcd15889 | ||
|
|
e64e20133a | ||
|
|
52949a952a | ||
|
|
3d7edbf7d9 | ||
|
|
7e390546ba | ||
|
|
75380589e1 | ||
|
|
19c224c199 | ||
|
|
ebd0e13594 | ||
|
|
b4cab5ab52 | ||
|
|
c24f92031c | ||
|
|
0ee28290db |
@@ -10,9 +10,9 @@ jobs:
|
|||||||
pre-commit:
|
pre-commit:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v4
|
- uses: actions/checkout@v7
|
||||||
- name: Set up Python 3.10
|
- name: Set up Python 3.10
|
||||||
uses: actions/setup-python@v5
|
uses: actions/setup-python@v6
|
||||||
with:
|
with:
|
||||||
python-version: "3.10"
|
python-version: "3.10"
|
||||||
- uses: pre-commit/action@v3.0.1
|
- uses: pre-commit/action@v3.0.1
|
||||||
@@ -23,17 +23,17 @@ jobs:
|
|||||||
python-version: ['3.9', '3.10', '3.11', '3.12']
|
python-version: ['3.9', '3.10', '3.11', '3.12']
|
||||||
|
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v4
|
- uses: actions/checkout@v7
|
||||||
|
|
||||||
- name: Set up Python ${{ matrix.python-version }}
|
- name: Set up Python ${{ matrix.python-version }}
|
||||||
uses: actions/setup-python@v5
|
uses: actions/setup-python@v6
|
||||||
with:
|
with:
|
||||||
python-version: ${{ matrix.python-version }}
|
python-version: ${{ matrix.python-version }}
|
||||||
|
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
run: |
|
run: |
|
||||||
python -m pip install --upgrade pip
|
python -m pip install --upgrade pip
|
||||||
pip install -e ".[dev]"
|
pip install -e . --group dev
|
||||||
|
|
||||||
# - name: Run mypy
|
# - name: Run mypy
|
||||||
# run: |
|
# run: |
|
||||||
|
|||||||
+14
-11
@@ -1,15 +1,18 @@
|
|||||||
version: 2
|
version: 2
|
||||||
|
|
||||||
build:
|
build:
|
||||||
os: "ubuntu-20.04"
|
os: ubuntu-24.04
|
||||||
tools:
|
tools:
|
||||||
python: "3.10"
|
python: "3.13"
|
||||||
|
commands:
|
||||||
sphinx:
|
- pip install cmake
|
||||||
configuration: docs/source/conf.py
|
- pip install -r docs/requirements.txt
|
||||||
|
- pip install -e .
|
||||||
python:
|
- cmake --workflow --preset documentation-workflow
|
||||||
install:
|
# Generate llms.txt variants for demo purposes
|
||||||
- requirements: docs/requirements.txt
|
- python docs/generate_llms_variants.py build/html
|
||||||
- method: pip
|
# Copy built documentation to Read the Docs output directory
|
||||||
path: .
|
- mkdir -p $READTHEDOCS_OUTPUT/html
|
||||||
|
- cp -r build/html/* $READTHEDOCS_OUTPUT/html/
|
||||||
|
- cp -r build/markdown/* $READTHEDOCS_OUTPUT/html/
|
||||||
|
- cp -r build/rst/* $READTHEDOCS_OUTPUT/html/
|
||||||
|
|||||||
+55
-1
@@ -1,13 +1,67 @@
|
|||||||
Changelog
|
Changelog
|
||||||
=========
|
=========
|
||||||
|
|
||||||
|
0.7.1
|
||||||
|
-----
|
||||||
|
|
||||||
|
- Don't process includes within code blocks
|
||||||
|
|
||||||
|
0.7.0
|
||||||
|
-----
|
||||||
|
|
||||||
|
- Add :confval:`llms_txt_uri_template` configuration option to control the link behavior in :confval:`llms_txt_filename`.
|
||||||
|
`#48 <https://github.com/jdillard/sphinx-llms-txt/pull/48>`_
|
||||||
|
|
||||||
|
0.6.0
|
||||||
|
-----
|
||||||
|
|
||||||
|
- Improve _sources directory handling
|
||||||
|
`#47 <https://github.com/jdillard/sphinx-llms-txt/pull/47>`_
|
||||||
|
|
||||||
|
0.5.3
|
||||||
|
-----
|
||||||
|
|
||||||
|
- Make sphinx a required dependency since there are imports from Sphinx
|
||||||
|
`#44 <https://github.com/jdillard/sphinx-llms-txt/pull/44>`_
|
||||||
|
|
||||||
|
0.5.2
|
||||||
|
-----
|
||||||
|
|
||||||
|
- Remove support for singlehtml
|
||||||
|
`#40 <https://github.com/jdillard/sphinx-llms-txt/pull/40>`_
|
||||||
|
|
||||||
|
0.5.1
|
||||||
|
-----
|
||||||
|
|
||||||
|
- Only allow builders that have _sources directory
|
||||||
|
`#38 <https://github.com/jdillard/sphinx-llms-txt/pull/38>`_
|
||||||
|
|
||||||
|
0.5.0
|
||||||
|
-----
|
||||||
|
|
||||||
|
- Add :ref:`block_level_ignore` and :ref:`page_level_ignore`
|
||||||
|
`#33 <https://github.com/jdillard/sphinx-llms-txt/pull/33>`_
|
||||||
|
- Add :confval:`llms_txt_full_size_policy` configuration option to control behavior when :confval:`llms_txt_full_max_size` is exceeded.
|
||||||
|
`#35 <https://github.com/jdillard/sphinx-llms-txt/pull/35>`_
|
||||||
|
|
||||||
|
0.4.1
|
||||||
|
-----
|
||||||
|
|
||||||
|
- Fix include paths and spacing
|
||||||
|
`#31 <https://github.com/jdillard/sphinx-llms-txt/pull/31>`_
|
||||||
|
|
||||||
|
0.4.0
|
||||||
|
-----
|
||||||
|
|
||||||
|
- Add support for including source code files with :confval:`llms_txt_code_files` and :confval:`llms_txt_code_base_path` configuration options
|
||||||
|
`#24 <https://github.com/jdillard/sphinx-llms-txt/pull/24>`_
|
||||||
|
|
||||||
0.3.2
|
0.3.2
|
||||||
-----
|
-----
|
||||||
|
|
||||||
- Fix image paths to deployed images
|
- Fix image paths to deployed images
|
||||||
`#30 <https://github.com/jdillard/sphinx-llms-txt/pull/30>`_
|
`#30 <https://github.com/jdillard/sphinx-llms-txt/pull/30>`_
|
||||||
|
|
||||||
|
|
||||||
0.3.1
|
0.3.1
|
||||||
-----
|
-----
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,15 @@
|
|||||||
|
cmake_minimum_required(VERSION 3.15)
|
||||||
|
project(SphinxDocs VERSION 1.0.0 LANGUAGES NONE)
|
||||||
|
|
||||||
|
# Fetch Sphinx CMake modules
|
||||||
|
include(FetchContent)
|
||||||
|
FetchContent_Declare(
|
||||||
|
sphinx_cmake_modules
|
||||||
|
GIT_REPOSITORY https://github.com/jdillard/sphinx-cmake-modules.git
|
||||||
|
GIT_TAG main
|
||||||
|
)
|
||||||
|
FetchContent_MakeAvailable(sphinx_cmake_modules)
|
||||||
|
list(APPEND CMAKE_MODULE_PATH "${sphinx_cmake_modules_SOURCE_DIR}/cmake/modules")
|
||||||
|
|
||||||
|
# Add documentation
|
||||||
|
add_subdirectory(docs)
|
||||||
@@ -0,0 +1,53 @@
|
|||||||
|
{
|
||||||
|
"version": 6,
|
||||||
|
"configurePresets": [
|
||||||
|
{
|
||||||
|
"name": "documentation",
|
||||||
|
"displayName": "Documentation Build",
|
||||||
|
"description": "Configure project with documentation environment setup",
|
||||||
|
"binaryDir": "${sourceDir}/build"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"buildPresets": [
|
||||||
|
{
|
||||||
|
"name": "html",
|
||||||
|
"displayName": "Build HTML Documentation",
|
||||||
|
"configurePreset": "documentation",
|
||||||
|
"targets": ["html"]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"name": "markdown",
|
||||||
|
"displayName": "Build Markdown Documentation",
|
||||||
|
"configurePreset": "documentation",
|
||||||
|
"targets": ["markdown"]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"name": "rst",
|
||||||
|
"displayName": "Build reStructuredText Documentation",
|
||||||
|
"configurePreset": "documentation",
|
||||||
|
"targets": ["rst"]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"name": "docs-parallel",
|
||||||
|
"displayName": "Build all output formats in parallel",
|
||||||
|
"configurePreset": "documentation",
|
||||||
|
"targets": ["html", "markdown", "rst"]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"workflowPresets": [
|
||||||
|
{
|
||||||
|
"name": "documentation-workflow",
|
||||||
|
"displayName": "Documentation Build Workflow",
|
||||||
|
"steps": [
|
||||||
|
{
|
||||||
|
"type": "configure",
|
||||||
|
"name": "documentation"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"type": "build",
|
||||||
|
"name": "docs-parallel"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
@@ -11,6 +11,10 @@ A Sphinx extension that generates a summary `llms.txt` file and a single combine
|
|||||||
|
|
||||||
See [sphinx-llms-txt documentation](https://sphinx-llms-txt.readthedocs.io/en/latest/index.html) for installation and configuration instructions.
|
See [sphinx-llms-txt documentation](https://sphinx-llms-txt.readthedocs.io/en/latest/index.html) for installation and configuration instructions.
|
||||||
|
|
||||||
|
## Contributing
|
||||||
|
|
||||||
|
Pull Requests welcome! See [Contributing](https://sphinx-llms-txt.readthedocs.io/en/latest/contributing.html) for instructions on how best to contribute.
|
||||||
|
|
||||||
## License
|
## License
|
||||||
|
|
||||||
MIT License - see LICENSE file for details.
|
MIT License - see LICENSE file for details.
|
||||||
|
|||||||
@@ -0,0 +1,7 @@
|
|||||||
|
include(SphinxUtils)
|
||||||
|
|
||||||
|
setup_sphinx_environment()
|
||||||
|
|
||||||
|
add_sphinx_builder(html)
|
||||||
|
add_sphinx_builder(markdown)
|
||||||
|
add_sphinx_builder(rst)
|
||||||
@@ -0,0 +1,59 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""
|
||||||
|
Generate variant llms.txt files for demo purposes.
|
||||||
|
|
||||||
|
Takes the generated llms.txt (with _sources links) and creates:
|
||||||
|
- llms.txt - default with _sources links (unchanged)
|
||||||
|
- llms.md.txt - .html.md links
|
||||||
|
- llms.rst.txt - .rst links
|
||||||
|
"""
|
||||||
|
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
|
||||||
|
def get_base_url() -> str:
|
||||||
|
"""Import base_url from conf.py."""
|
||||||
|
sys.path.insert(0, str(Path(__file__).parent / "source"))
|
||||||
|
from conf import html_baseurl # noqa: E402
|
||||||
|
|
||||||
|
return html_baseurl
|
||||||
|
|
||||||
|
|
||||||
|
def generate_variants(build_dir: Path) -> None:
|
||||||
|
"""Generate llms.txt variants from the original file."""
|
||||||
|
original = build_dir / "llms.txt"
|
||||||
|
|
||||||
|
if not original.exists():
|
||||||
|
print(f"Error: {original} not found")
|
||||||
|
sys.exit(1)
|
||||||
|
|
||||||
|
content = original.read_text()
|
||||||
|
base_url = get_base_url()
|
||||||
|
|
||||||
|
# Pattern to match links like: https://.../_sources/{docname}.rst.txt
|
||||||
|
link_pattern = re.compile(
|
||||||
|
rf"({re.escape(base_url)})_sources/([a-zA-Z0-9_/\-]+)\.rst\.txt"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Generate .html.md variant
|
||||||
|
md_content = link_pattern.sub(r"\1\2.html.md", content)
|
||||||
|
(build_dir / "llms.md.txt").write_text(md_content)
|
||||||
|
print(f"Generated: {build_dir / 'llms.md.txt'} (.html.md links)")
|
||||||
|
|
||||||
|
# Generate .rst variant
|
||||||
|
rst_content = link_pattern.sub(r"\1\2.rst", content)
|
||||||
|
(build_dir / "llms.rst.txt").write_text(rst_content)
|
||||||
|
print(f"Generated: {build_dir / 'llms.rst.txt'} (.rst links)")
|
||||||
|
|
||||||
|
print(f"Kept: {build_dir / 'llms.txt'} (_sources links)")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
if len(sys.argv) > 1:
|
||||||
|
build_dir = Path(sys.argv[1])
|
||||||
|
else:
|
||||||
|
build_dir = Path("build/html")
|
||||||
|
|
||||||
|
generate_variants(build_dir)
|
||||||
@@ -2,5 +2,9 @@ furo
|
|||||||
esbonio
|
esbonio
|
||||||
sphinx-contributors
|
sphinx-contributors
|
||||||
sphinx
|
sphinx
|
||||||
|
sphinx-design
|
||||||
sphinx-llms-txt
|
sphinx-llms-txt
|
||||||
|
sphinx-inline-tabs
|
||||||
sphinxext-opengraph
|
sphinxext-opengraph
|
||||||
|
sphinx-markdown-builder
|
||||||
|
sphinxcontrib-restbuilder
|
||||||
|
|||||||
@@ -76,15 +76,26 @@ Handling Large Documentation
|
|||||||
^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
||||||
|
|
||||||
For very large documentation sets, generating the full documentation file might exceed reasonable size limits.
|
For very large documentation sets, generating the full documentation file might exceed reasonable size limits.
|
||||||
You can set a maximum line count:
|
You can set a maximum line count and control what happens when that limit is exceeded:
|
||||||
|
|
||||||
.. code-block:: python
|
.. code-block:: python
|
||||||
|
|
||||||
llms_txt_full_max_size = 10000 # Maximum 10,000 lines
|
llms_txt_full_max_size = 10000 # Maximum 10,000 lines
|
||||||
|
llms_txt_full_size_policy = "warn_skip" # Default behavior
|
||||||
|
|
||||||
If the generated file would exceed this limit, the extension will skip its generation and show a warning, allowing the build to complete.
|
The ``llms_txt_full_size_policy`` setting controls both the log level and action taken when the size limit is exceeded.
|
||||||
|
It uses the format ``"<loglevel>_<action>"``:
|
||||||
|
|
||||||
.. tip:: Use :ref:`excluding_content` to remove less relevant pages.
|
**Log levels:**
|
||||||
|
- ``warn``: Log as a warning (default)
|
||||||
|
- ``info``: Log as informational message
|
||||||
|
|
||||||
|
**Actions:**
|
||||||
|
- ``skip``: Don't create the file (default)
|
||||||
|
- ``keep``: Create the file anyway, ignoring the size limit
|
||||||
|
- ``note``: Create a placeholder file explaining why the full file wasn't generated
|
||||||
|
|
||||||
|
.. tip:: Use :ref:`excluding_content` to remove less relevant pages and reduce the file size.
|
||||||
|
|
||||||
.. _custom_directive_handling:
|
.. _custom_directive_handling:
|
||||||
|
|
||||||
@@ -113,6 +124,13 @@ This ensures that paths in your custom directives are properly resolved in the g
|
|||||||
Excluding Content
|
Excluding Content
|
||||||
^^^^^^^^^^^^^^^^^
|
^^^^^^^^^^^^^^^^^
|
||||||
|
|
||||||
|
There are several ways to exclude content from the generated ``llms-full.txt`` file:
|
||||||
|
|
||||||
|
.. _global_exclusion:
|
||||||
|
|
||||||
|
Global Page Exclusion
|
||||||
|
~~~~~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
You can exclude specific pages from being included in the generated files:
|
You can exclude specific pages from being included in the generated files:
|
||||||
|
|
||||||
.. code-block:: python
|
.. code-block:: python
|
||||||
@@ -124,6 +142,111 @@ You can exclude specific pages from being included in the generated files:
|
|||||||
]
|
]
|
||||||
|
|
||||||
This is useful for excluding auto-generated pages, indexes, or content that isn't relevant for LLM consumption.
|
This is useful for excluding auto-generated pages, indexes, or content that isn't relevant for LLM consumption.
|
||||||
|
It can also be used to reduce the size of llms-full.txt.
|
||||||
|
|
||||||
|
.. _page_level_ignore:
|
||||||
|
|
||||||
|
Page-Level Ignore Metadata
|
||||||
|
~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
|
You can exclude individual pages by adding metadata at the top of any reStructuredText file:
|
||||||
|
|
||||||
|
.. code-block:: restructuredtext
|
||||||
|
|
||||||
|
:llms-txt-ignore: true
|
||||||
|
|
||||||
|
Page Title
|
||||||
|
==========
|
||||||
|
|
||||||
|
This entire page will be excluded from llms-full.txt
|
||||||
|
|
||||||
|
When this metadata is present, the entire page is skipped during processing.
|
||||||
|
|
||||||
|
.. _block_level_ignore:
|
||||||
|
|
||||||
|
Block-Level Ignore Directives
|
||||||
|
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
|
You can exclude specific sections within a page using ignore directives:
|
||||||
|
|
||||||
|
.. code-block:: restructuredtext
|
||||||
|
|
||||||
|
Page Title
|
||||||
|
==========
|
||||||
|
|
||||||
|
This content will be included in llms-full.txt.
|
||||||
|
|
||||||
|
.. llms-txt-ignore-start
|
||||||
|
|
||||||
|
This content will be excluded from llms-full.txt.
|
||||||
|
|
||||||
|
Section To Ignore
|
||||||
|
-----------------
|
||||||
|
|
||||||
|
This entire section and any nested content will be ignored.
|
||||||
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# This code block will also be ignored
|
||||||
|
def ignored_function():
|
||||||
|
pass
|
||||||
|
|
||||||
|
.. llms-txt-ignore-end
|
||||||
|
|
||||||
|
This content will be included again.
|
||||||
|
|
||||||
|
Block-level ignores can be useful for:
|
||||||
|
|
||||||
|
- Removing internal notes or TODOs
|
||||||
|
- Hiding implementation details while keeping user-facing documentation
|
||||||
|
|
||||||
|
.. note::
|
||||||
|
- Multiple ignore blocks can be used within the same file
|
||||||
|
- Ignore directives work with any indentation level
|
||||||
|
|
||||||
|
.. _including_code_files:
|
||||||
|
|
||||||
|
Including Source Code Files
|
||||||
|
^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
||||||
|
|
||||||
|
You can include source code files from your project at the end of :confval:`llms_txt_full_filename`.
|
||||||
|
|
||||||
|
Use include/exclude syntax to precisely control which files are included:
|
||||||
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
llms_txt_code_files = [
|
||||||
|
"+:src/**/*.py", # Include all Python files in src
|
||||||
|
"-:src/**/__pycache__/**", # Exclude Python cache files
|
||||||
|
]
|
||||||
|
|
||||||
|
Pattern syntax:
|
||||||
|
|
||||||
|
- **+:pattern**: Include files matching the pattern. Processed first to collect matching files.
|
||||||
|
- **-:pattern**: Exclude files matching the pattern. Applied to filter out unwanted files.
|
||||||
|
|
||||||
|
Code files are processed as follows:
|
||||||
|
|
||||||
|
- **Glob patterns**: Use standard glob patterns (``*``, ``**``, ``?``) to match files
|
||||||
|
- **Relative paths**: Patterns are resolved relative to your Sphinx source directory
|
||||||
|
- **Formatting**: Each file is presented with a title and syntax-highlighted code block
|
||||||
|
|
||||||
|
.. _customizing_code_paths:
|
||||||
|
|
||||||
|
Customizing Code File Paths
|
||||||
|
~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
|
By default, the extension automatically detects the relative path from your Sphinx source directory to the git root and strips that prefix from displayed file paths. You can customize this behavior:
|
||||||
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# Manually specify base path to strip
|
||||||
|
llms_txt_code_base_path = "../../"
|
||||||
|
|
||||||
|
# Disable path stripping entirely
|
||||||
|
llms_txt_code_base_path = ""
|
||||||
|
|
||||||
|
This helps create cleaner, more readable file paths in the generated documentation.
|
||||||
|
|
||||||
.. _using_html_baseurl:
|
.. _using_html_baseurl:
|
||||||
|
|
||||||
@@ -138,6 +261,178 @@ If you want to include absolute URLs for resources in your documentation, you ca
|
|||||||
|
|
||||||
When this option is set, all resolved paths in directives will be prefixed with this URL, creating absolute paths in the generated files.
|
When this option is set, all resolved paths in directives will be prefixed with this URL, creating absolute paths in the generated files.
|
||||||
|
|
||||||
|
.. _customizing_uri_links:
|
||||||
|
|
||||||
|
Customizing URI Links in llms.txt
|
||||||
|
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
||||||
|
|
||||||
|
By default, the ``llms.txt`` file links to source files in the ``_sources`` directory when available, falling back to HTML pages when sources aren't available.
|
||||||
|
You can customize this behavior using URI templates with :confval:`llms_txt_uri_template`:
|
||||||
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# Default: Link to source files, if _sources exists
|
||||||
|
llms_txt_uri_template = "{base_url}_sources/{docname}{suffix}{sourcelink_suffix}"
|
||||||
|
|
||||||
|
# Default: Link to HTML pages instead, if _sources doesn't exist
|
||||||
|
llms_txt_uri_template = "{base_url}{docname}.html"
|
||||||
|
|
||||||
|
# Manual: Link to a custom markdown build
|
||||||
|
llms_txt_uri_template = "{base_url}{docname}.md"
|
||||||
|
|
||||||
|
.. _available_template_variables:
|
||||||
|
|
||||||
|
Available Template Variables
|
||||||
|
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
|
Your URI template can use the following variables:
|
||||||
|
|
||||||
|
- ``{base_url}`` - The base URL from ``html_baseurl`` configuration (includes trailing slash)
|
||||||
|
- ``{docname}`` - The document name (e.g., ``index``, ``guide/intro``)
|
||||||
|
- ``{suffix}`` - The source file suffix (e.g., ``.rst``, ``.md``) - may be empty if no source file exists
|
||||||
|
- ``{sourcelink_suffix}`` - The suffix from ``html_sourcelink_suffix`` configuration (e.g., ``.txt``)
|
||||||
|
|
||||||
|
.. tip::
|
||||||
|
Instead of using the default of linking to ``_sources``, you can generate Markdown and/or reStructuredText files from your documentation and link to those in ``llms.txt``.
|
||||||
|
See :ref:`cmake_workflow` for an example of building both HTML and Markdown and/or reStructuredText in parallel.
|
||||||
|
Note that ``_sources`` is still needed for ``llms-full.txt`` at this time.
|
||||||
|
|
||||||
|
.. _cmake_workflow:
|
||||||
|
|
||||||
|
CMake Workflow
|
||||||
|
^^^^^^^^^^^^^^
|
||||||
|
|
||||||
|
This project uses CMake to orchestrate documentation builds across multiple output formats, serving as a simple demo of the functionality.
|
||||||
|
This approach enables parallel builds and integrates well with CI/CD platforms like Read the Docs.
|
||||||
|
|
||||||
|
Building multiple formats allows you to compare what works best for your docs, as well as allows users to choose which format to feed to their LLM.
|
||||||
|
Use :confval:`llms_txt_uri_template` to configure links to point to your preferred format.
|
||||||
|
|
||||||
|
Key Files
|
||||||
|
~~~~~~~~~
|
||||||
|
|
||||||
|
These configuration files serve as a simple example of a Sphinx site hosted on Read The Docs, some modification may be needed.
|
||||||
|
|
||||||
|
.. code-block:: text
|
||||||
|
|
||||||
|
.
|
||||||
|
├── .readthedocs.yml
|
||||||
|
├── CMakeLists.txt
|
||||||
|
├── CMakePresets.json
|
||||||
|
└── docs/
|
||||||
|
└── CMakeLists.txt
|
||||||
|
|
||||||
|
Each section below contains a summary of the file's purpose, the full contents of the file, and a table describing key lines that may need modification.
|
||||||
|
|
||||||
|
.. dropdown:: .readthedocs.yml
|
||||||
|
:chevron: down-up
|
||||||
|
|
||||||
|
A Read The Docs config file that installs dependencies, then runs the full documentation workflow which builds all output formats in parallel, and copies them into a single deploy location.
|
||||||
|
|
||||||
|
.. literalinclude:: ../../.readthedocs.yml
|
||||||
|
:language: yaml
|
||||||
|
:lines: 1-9,11,14-
|
||||||
|
:linenos:
|
||||||
|
:emphasize-lines: 9, 14-15
|
||||||
|
|
||||||
|
.. list-table::
|
||||||
|
:header-rows: 1
|
||||||
|
:width: 100%
|
||||||
|
:widths: 15 85
|
||||||
|
|
||||||
|
* - Line
|
||||||
|
- Description
|
||||||
|
* - **9**
|
||||||
|
- Update the path if your requirements file is in a different location
|
||||||
|
* - **13-14**
|
||||||
|
- Modify the copy commands for the output formats you deploy
|
||||||
|
|
||||||
|
.. dropdown:: CMakeLists.txt
|
||||||
|
:chevron: down-up
|
||||||
|
|
||||||
|
A CMake config file that sets up the project, fetches the shared `sphinx-cmake-modules <https://github.com/jdillard/sphinx-cmake-modules>`_, and includes the docs subdirectory.
|
||||||
|
|
||||||
|
.. literalinclude:: ../../CMakeLists.txt
|
||||||
|
:language: cmake
|
||||||
|
:linenos:
|
||||||
|
:emphasize-lines: 9, 15
|
||||||
|
|
||||||
|
.. list-table::
|
||||||
|
:header-rows: 1
|
||||||
|
:width: 100%
|
||||||
|
:widths: 15 85
|
||||||
|
|
||||||
|
* - Line
|
||||||
|
- Description
|
||||||
|
* - **9**
|
||||||
|
- Update the ``GIT_TAG`` to use a different version or commit hash
|
||||||
|
* - **15**
|
||||||
|
- Change if your docs subdirectory has a different location
|
||||||
|
|
||||||
|
.. dropdown:: docs/CMakeLists.txt
|
||||||
|
:chevron: down-up
|
||||||
|
|
||||||
|
A CMake config file that includes the `SphinxUtils <https://github.com/jdillard/sphinx-cmake-modules/blob/v0.1.0/SphinxUtils.cmake>`_ module from FetchContent and defines the documentation-specific build targets.
|
||||||
|
|
||||||
|
.. literalinclude:: ../CMakeLists.txt
|
||||||
|
:language: cmake
|
||||||
|
:linenos:
|
||||||
|
:emphasize-lines: 5-7
|
||||||
|
|
||||||
|
.. list-table::
|
||||||
|
:header-rows: 1
|
||||||
|
:width: 100%
|
||||||
|
:widths: 15 85
|
||||||
|
|
||||||
|
* - Line
|
||||||
|
- Description
|
||||||
|
* - **5-7**
|
||||||
|
- Add or remove calls based on which output formats you need
|
||||||
|
|
||||||
|
|
||||||
|
.. dropdown:: CMakePresets.json
|
||||||
|
:chevron: down-up
|
||||||
|
|
||||||
|
Defines presets for configuring and building documentation:
|
||||||
|
|
||||||
|
- **Configure Presets:** Sets up the build directory.
|
||||||
|
- **Build Presets:** Defines build formats individually and all in parallel.
|
||||||
|
- **Workflow Presets:** Runs the configure preset followed by the parallel build preset.
|
||||||
|
|
||||||
|
.. literalinclude:: ../../CMakePresets.json
|
||||||
|
:language: json
|
||||||
|
:linenos:
|
||||||
|
:emphasize-lines: 18-23, 24-29, 34
|
||||||
|
|
||||||
|
.. list-table::
|
||||||
|
:header-rows: 1
|
||||||
|
:width: 100%
|
||||||
|
:widths: 15 85
|
||||||
|
|
||||||
|
* - Line
|
||||||
|
- Description
|
||||||
|
* - **18-23**
|
||||||
|
- Remove this preset to disable Markdown documentation builds
|
||||||
|
* - **24-29**
|
||||||
|
- Remove this preset to disable reStructuredText documentation builds
|
||||||
|
* - **34**
|
||||||
|
- Modify the targets list to build only the output formats you need in parallel
|
||||||
|
|
||||||
|
Usage
|
||||||
|
~~~~~
|
||||||
|
|
||||||
|
To build documentation locally using CMake:
|
||||||
|
|
||||||
|
.. code-block:: console
|
||||||
|
|
||||||
|
# Run the full workflow (configure + build all formats)
|
||||||
|
cmake --workflow --preset documentation-workflow
|
||||||
|
|
||||||
|
# Or configure and build separately
|
||||||
|
cmake --preset documentation
|
||||||
|
cmake --build --preset html # Build HTML only
|
||||||
|
cmake --build --preset docs-parallel # Build all formats
|
||||||
|
|
||||||
.. _integration_examples:
|
.. _integration_examples:
|
||||||
|
|
||||||
Integration Examples
|
Integration Examples
|
||||||
@@ -154,6 +449,7 @@ Here's a complete example showing multiple :doc:`configuration-values`:
|
|||||||
llms_txt_filename = "ai-summary.txt"
|
llms_txt_filename = "ai-summary.txt"
|
||||||
llms_txt_full_filename = "ai-full-docs.txt"
|
llms_txt_full_filename = "ai-full-docs.txt"
|
||||||
llms_txt_full_max_size = 50000
|
llms_txt_full_max_size = 50000
|
||||||
|
llms_txt_full_size_policy = "warn_note"
|
||||||
|
|
||||||
# Content customization
|
# Content customization
|
||||||
llms_txt_title = "Project Documentation for AI Assistants"
|
llms_txt_title = "Project Documentation for AI Assistants"
|
||||||
@@ -161,6 +457,7 @@ Here's a complete example showing multiple :doc:`configuration-values`:
|
|||||||
This is a comprehensive documentation set for our project.
|
This is a comprehensive documentation set for our project.
|
||||||
It includes API references, usage examples, and tutorials.
|
It includes API references, usage examples, and tutorials.
|
||||||
"""
|
"""
|
||||||
|
llms_txt_uri_template = "{base_url}{docname}.md"
|
||||||
|
|
||||||
# Path handling
|
# Path handling
|
||||||
html_baseurl = "https://docs.example.com/"
|
html_baseurl = "https://docs.example.com/"
|
||||||
@@ -168,3 +465,11 @@ Here's a complete example showing multiple :doc:`configuration-values`:
|
|||||||
|
|
||||||
# Content filtering
|
# Content filtering
|
||||||
llms_txt_exclude = ["search", "genindex", "404", "private_*"]
|
llms_txt_exclude = ["search", "genindex", "404", "private_*"]
|
||||||
|
|
||||||
|
# Source code inclusion with include/exclude patterns
|
||||||
|
llms_txt_code_files = [
|
||||||
|
"+:../../src/**/*.py", # Include Python files
|
||||||
|
"+:../../config/*.yaml", # Include config files
|
||||||
|
"-:../../src/**/__pycache__/**", # Exclude cache files
|
||||||
|
]
|
||||||
|
llms_txt_code_base_path = "../../"
|
||||||
|
|||||||
+11
-1
@@ -15,11 +15,17 @@ import subprocess
|
|||||||
project = "sphinx-llms-txt"
|
project = "sphinx-llms-txt"
|
||||||
copyright = "Jared Dillard"
|
copyright = "Jared Dillard"
|
||||||
author = "Jared Dillard"
|
author = "Jared Dillard"
|
||||||
|
|
||||||
|
llms_txt_code_files = ["+:../../sphinx_llms_txt/*.py"]
|
||||||
llms_txt_summary = """
|
llms_txt_summary = """
|
||||||
A Sphinx extension that generates a summary llms.txt file,written in Markdown,
|
A Sphinx extension that generates a summary llms.txt file,written in Markdown,
|
||||||
and a single combined documentation llms-full.txt file, written in reStructuredText.
|
and a single combined documentation llms-full.txt file, written in reStructuredText.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
# This doesn't seem to be supported
|
||||||
|
# rst_file_suffix = ".html.rst"
|
||||||
|
markdown_file_suffix = ".html.md"
|
||||||
|
|
||||||
# check if the current commit is tagged as a release (vX.Y.Z)
|
# check if the current commit is tagged as a release (vX.Y.Z)
|
||||||
try:
|
try:
|
||||||
GIT_TAG_OUTPUT = subprocess.check_output(["git", "tag", "--points-at", "HEAD"])
|
GIT_TAG_OUTPUT = subprocess.check_output(["git", "tag", "--points-at", "HEAD"])
|
||||||
@@ -48,6 +54,10 @@ extensions = [
|
|||||||
"sphinx.ext.intersphinx",
|
"sphinx.ext.intersphinx",
|
||||||
"sphinx_contributors",
|
"sphinx_contributors",
|
||||||
"sphinx_llms_txt",
|
"sphinx_llms_txt",
|
||||||
|
"sphinxcontrib.restbuilder",
|
||||||
|
"sphinx_inline_tabs",
|
||||||
|
"sphinx.ext.extlinks",
|
||||||
|
"sphinx_design",
|
||||||
]
|
]
|
||||||
|
|
||||||
# The language for content autogenerated by Sphinx. Refer to documentation
|
# The language for content autogenerated by Sphinx. Refer to documentation
|
||||||
@@ -87,7 +97,7 @@ html_theme_options = {
|
|||||||
"source_directory": "docs/source/",
|
"source_directory": "docs/source/",
|
||||||
}
|
}
|
||||||
|
|
||||||
html_baseurl = "https://sphinx-llms-txt.readthedocs.org/"
|
html_baseurl = "https://sphinx-llms-txt.readthedocs.org/en/latest/"
|
||||||
|
|
||||||
|
|
||||||
# -- Options for HTMLHelp output ---------------------------------------------
|
# -- Options for HTMLHelp output ---------------------------------------------
|
||||||
|
|||||||
@@ -24,11 +24,22 @@ Project Configuration Values
|
|||||||
- **Type**: integer or ``None``
|
- **Type**: integer or ``None``
|
||||||
- **Default**: ``None`` (no limit)
|
- **Default**: ``None`` (no limit)
|
||||||
- **Description**: Sets a maximum line count for ``llms_txt_full_filename``.
|
- **Description**: Sets a maximum line count for ``llms_txt_full_filename``.
|
||||||
If exceeded, the file is skipped and a warning is shown, but the build still completes.
|
Behavior when exceeded is controlled by :confval:`llms_txt_full_size_policy`.
|
||||||
See :ref:`handling_large_documentation`.
|
See :ref:`handling_large_documentation`.
|
||||||
|
|
||||||
.. versionadded:: 0.2.0
|
.. versionadded:: 0.2.0
|
||||||
|
|
||||||
|
.. confval:: llms_txt_full_size_policy
|
||||||
|
|
||||||
|
- **Type**: string
|
||||||
|
- **Default**: ``'warn_skip'``
|
||||||
|
- **Description**: Controls what happens when :confval:`llms_txt_full_max_size` is exceeded.
|
||||||
|
Format is ``<loglevel>_<action>``. Log levels: ``warn``, ``info``.
|
||||||
|
Actions: ``skip``, ``keep``, ``note``.
|
||||||
|
See :ref:`handling_large_documentation`.
|
||||||
|
|
||||||
|
.. versionadded:: 0.5.0
|
||||||
|
|
||||||
.. confval:: llms_txt_file
|
.. confval:: llms_txt_file
|
||||||
|
|
||||||
- **Type**: boolean
|
- **Type**: boolean
|
||||||
@@ -47,6 +58,15 @@ Project Configuration Values
|
|||||||
|
|
||||||
.. versionadded:: 0.2.0
|
.. versionadded:: 0.2.0
|
||||||
|
|
||||||
|
.. confval:: llms_txt_uri_template
|
||||||
|
|
||||||
|
- **Type**: string or ``None``
|
||||||
|
- **Default**: ``None``
|
||||||
|
- **Description**: Template string for generating URIs in ``llms.txt``.
|
||||||
|
See :ref:`customizing_uri_links`.
|
||||||
|
|
||||||
|
.. versionadded:: 0.7.0
|
||||||
|
|
||||||
.. confval:: llms_txt_directives
|
.. confval:: llms_txt_directives
|
||||||
|
|
||||||
- **Type**: list of strings
|
- **Type**: list of strings
|
||||||
@@ -78,7 +98,26 @@ Project Configuration Values
|
|||||||
|
|
||||||
- **Type**: list of strings
|
- **Type**: list of strings
|
||||||
- **Default**: ``[]``
|
- **Default**: ``[]``
|
||||||
- **Description**: A list of pages to ignore.
|
- **Description**: A list of pages to ignore using glob patterns.
|
||||||
See :ref:`excluding_content`.
|
See :ref:`excluding_content`.
|
||||||
|
|
||||||
.. versionadded:: 0.2.1
|
.. versionadded:: 0.2.1
|
||||||
|
|
||||||
|
.. confval:: llms_txt_code_files
|
||||||
|
|
||||||
|
- **Type**: list of strings
|
||||||
|
- **Default**: ``[]``
|
||||||
|
- **Description**: A list of glob patterns that appends source code files to :confval:`llms_txt_full_filename`.
|
||||||
|
See :ref:`including_code_files`.
|
||||||
|
|
||||||
|
.. versionadded:: 0.4.0
|
||||||
|
|
||||||
|
.. confval:: llms_txt_code_base_path
|
||||||
|
|
||||||
|
- **Type**: string or ``None``
|
||||||
|
- **Default**: ``None`` (auto-detect from git root)
|
||||||
|
- **Description**: Base path to strip from code file paths when displaying titles.
|
||||||
|
When ``None``, automatically detects the relative path from the Sphinx source
|
||||||
|
directory to the git root and strips that prefix from file paths.
|
||||||
|
|
||||||
|
.. versionadded:: 0.4.0
|
||||||
|
|||||||
@@ -19,7 +19,7 @@ Local development
|
|||||||
|
|
||||||
.. code-block:: console
|
.. code-block:: console
|
||||||
|
|
||||||
pip install -e ".[dev]"
|
pip install -e . --group dev
|
||||||
|
|
||||||
#. Install pre-commit Git hook scripts:
|
#. Install pre-commit Git hook scripts:
|
||||||
|
|
||||||
|
|||||||
@@ -1,19 +1,22 @@
|
|||||||
Getting Started
|
Getting Started
|
||||||
===============
|
===============
|
||||||
|
|
||||||
Demo
|
|
||||||
----
|
|
||||||
|
|
||||||
You can see this Sphinx project's `llms.txt`_ and `llms-full.txt`_ files as a simple example.
|
|
||||||
|
|
||||||
Installation
|
Installation
|
||||||
------------
|
------------
|
||||||
|
|
||||||
Directly install via ``pip`` by using:
|
Directly install by using:
|
||||||
|
|
||||||
.. code-block:: bash
|
.. tab:: via pip
|
||||||
|
|
||||||
pip install sphinx-llms-txt
|
.. code-block:: bash
|
||||||
|
|
||||||
|
pip install sphinx-llms-txt
|
||||||
|
|
||||||
|
.. tab:: via conda:
|
||||||
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
conda install -c conda-forge sphinx-llms-txt
|
||||||
|
|
||||||
Usage
|
Usage
|
||||||
-----
|
-----
|
||||||
@@ -26,25 +29,64 @@ Add the extension to your Sphinx configuration (``conf.py``):
|
|||||||
'sphinx_llms_txt',
|
'sphinx_llms_txt',
|
||||||
]
|
]
|
||||||
|
|
||||||
Once added, the extension will automatically generate the LLMs.txt files during the build process.
|
After the HTML finishes building, **sphinx-llms-txt** will output the location of the output files::
|
||||||
|
|
||||||
See :doc:`advanced-configuration` for more information about how to use **sphinx-llms-txt**.
|
sphinx-llms-txt: Created /path/to/_build/html/llms-full.txt with 45 sources and 6879 lines
|
||||||
|
sphinx-llms-txt: created /path/to/_build/html/llms.txt
|
||||||
|
|
||||||
How It Works
|
.. _choosing-output-format:
|
||||||
------------
|
|
||||||
|
|
||||||
During the Sphinx build process:
|
Choosing an Output Format
|
||||||
|
-------------------------
|
||||||
|
|
||||||
1. **Content Collection**: Scans all of your documentation's ``_source`` pages and collects their content
|
By default, **sphinx-llms-txt** requires no additional configuration and links to raw reStructuredText source files created by the HTML builder.
|
||||||
2. **Directive Processing**: Resolves ``include`` directives by automatically incorporating their content
|
For optimal LLM support, see the alternative builders below and the :ref:`CMake workflow <cmake_workflow>` for setup.
|
||||||
3. **Path Resolution**: Transforms relative paths in directives to full paths
|
|
||||||
4. **Output Generation**: Creates two optional files:
|
|
||||||
|
|
||||||
- ``llms.txt``: A concise summary of your documentation, in Markdown
|
.. list-table:: Output Format Comparison
|
||||||
- ``llms-full.txt``: A comprehensive version with all documentation content, in reStructuredText
|
:header-rows: 1
|
||||||
|
:widths: 18 27 27 27
|
||||||
|
|
||||||
5. **Content Filtering**: Allows you to exclude specific pages from the generated files
|
* -
|
||||||
|
- Default
|
||||||
|
- Markdown
|
||||||
|
- reStructuredText
|
||||||
|
* - **Setup**
|
||||||
|
- No config
|
||||||
|
- CMake [#sphinxllm]_
|
||||||
|
- CMake
|
||||||
|
* - **Builder**
|
||||||
|
- Native [#native]_
|
||||||
|
- `sphinx-markdown-builder`_
|
||||||
|
- `sphinxcontrib-restbuilder`_
|
||||||
|
* - **Format**
|
||||||
|
- Raw reStructuredText source
|
||||||
|
- Rendered Markdown [#rendered]_
|
||||||
|
- Rendered reStructuredText [#rendered]_
|
||||||
|
* - **LLM Readability**
|
||||||
|
- Good - preserves structure for simple syntax
|
||||||
|
- Excellent - native LLM format
|
||||||
|
- Good - Can provide more structured content
|
||||||
|
* - **Key Advantage**
|
||||||
|
- Zero setup required
|
||||||
|
- More compact (less input tokens)
|
||||||
|
- Can preserve Sphinx semantics
|
||||||
|
* - **Key Disadvantage**
|
||||||
|
- Raw directives won't be parsed [#autodoc]_
|
||||||
|
- Loses structure from complex directives
|
||||||
|
- Can lose structure from complex directives
|
||||||
|
* - **llms-full.txt support**
|
||||||
|
- Supported with above caveats
|
||||||
|
- Pending `support <https://github.com/liran-funaro/sphinx-markdown-builder/pull/37>`__ [#pending]_
|
||||||
|
- Pending `support <https://github.com/sphinx-contrib/restbuilder/pull/35>`__ [#pending]_
|
||||||
|
|
||||||
|
.. _sphinx-markdown-builder: https://pypi.org/project/sphinx-markdown-builder/
|
||||||
|
.. _sphinxcontrib-restbuilder: https://pypi.org/project/sphinxcontrib-restbuilder/
|
||||||
|
|
||||||
|
.. rubric:: Footnotes
|
||||||
|
|
||||||
|
.. [#sphinxllm] See `sphinx-llm <https://github.com/NVIDIA/sphinx-llm>`_ as an alternative for CMake-free Markdown builds.
|
||||||
|
.. [#native] Uses raw :confval:`_sources/ <sphinx:html_copy_source>` files created by Sphinx's HTML builder with some minor enhancements.
|
||||||
|
.. [#autodoc] Directives like ``autodoc`` will appear as raw directive syntax rather than the extracted docstrings.
|
||||||
|
.. [#pending] PRs that add ``llms-full.txt`` concatenation support have yet to be released.
|
||||||
|
.. [#rendered] Directives are expanded and processed before output, so content like autodoc docstrings will be included.
|
||||||
|
|
||||||
.. _llms.txt: https://sphinx-llms-txt.readthedocs.io/en/latest/llms.txt
|
|
||||||
.. _llms-full.txt: https://sphinx-llms-txt.readthedocs.io/en/latest/llms-full.txt
|
|
||||||
|
|||||||
@@ -5,6 +5,32 @@ A `Sphinx`_ extension that generates a summary ``llms.txt`` file, written in Mar
|
|||||||
|
|
||||||
|PyPI version| |Conda Version| |Downloads| |Parallel Safe| |GitHub Stars|
|
|PyPI version| |Conda Version| |Downloads| |Parallel Safe| |GitHub Stars|
|
||||||
|
|
||||||
|
Demo
|
||||||
|
----
|
||||||
|
|
||||||
|
This Sphinx project's `llms.txt`_ and `llms-full.txt`_ files as an example of the default output format.
|
||||||
|
|
||||||
|
Alternative :ref:`output formats <choosing-output-format>` are also available. For example: `Markdown`_ and `reStructuredText`_.
|
||||||
|
|
||||||
|
Highlights
|
||||||
|
----------
|
||||||
|
|
||||||
|
**Zero Configuration**
|
||||||
|
Add the extension to your ``conf.py`` and you're done.
|
||||||
|
The extension automatically collects your documentation and generates both ``llms.txt`` and ``llms-full.txt`` during your normal Sphinx build.
|
||||||
|
|
||||||
|
**Intelligent Content Processing**
|
||||||
|
Automatically resolves ``include`` directives, transforms relative paths, and handles your documentation structure without manual intervention.
|
||||||
|
|
||||||
|
**Customizable When Needed**
|
||||||
|
Filter content, include source code files, or integrate with alternative output formats like Markdown for even better LLM compatibility.
|
||||||
|
See :doc:`getting-started` for output format options and :doc:`configuration-values` for all settings.
|
||||||
|
|
||||||
|
.. seealso::
|
||||||
|
|
||||||
|
For better default output without configuration, see `sphinx-llm <https://github.com/NVIDIA/sphinx-llm>`_ from NVIDIA.
|
||||||
|
sphinx-llms-txt is best when customized with alternative output formats, content filtering, or source code inclusion.
|
||||||
|
|
||||||
.. toctree::
|
.. toctree::
|
||||||
:maxdepth: 2
|
:maxdepth: 2
|
||||||
|
|
||||||
@@ -15,6 +41,10 @@ A `Sphinx`_ extension that generates a summary ``llms.txt`` file, written in Mar
|
|||||||
changelog
|
changelog
|
||||||
|
|
||||||
|
|
||||||
|
.. _llms.txt: https://sphinx-llms-txt.readthedocs.io/en/latest/llms.txt
|
||||||
|
.. _llms-full.txt: https://sphinx-llms-txt.readthedocs.io/en/latest/llms-full.txt
|
||||||
|
.. _Markdown: https://sphinx-llms-txt.readthedocs.io/en/latest/llms.md.txt
|
||||||
|
.. _reStructuredText: https://sphinx-llms-txt.readthedocs.io/en/latest/llms.rst.txt
|
||||||
.. _Sphinx: http://sphinx-doc.org/
|
.. _Sphinx: http://sphinx-doc.org/
|
||||||
|
|
||||||
.. |PyPI version| image:: https://img.shields.io/pypi/v/sphinx-llms-txt.svg
|
.. |PyPI version| image:: https://img.shields.io/pypi/v/sphinx-llms-txt.svg
|
||||||
|
|||||||
+6
-2
@@ -26,13 +26,16 @@ classifiers = [
|
|||||||
license = {text = "MIT"}
|
license = {text = "MIT"}
|
||||||
readme = "README.md"
|
readme = "README.md"
|
||||||
dynamic = ["version"]
|
dynamic = ["version"]
|
||||||
|
dependencies = [
|
||||||
|
"sphinx",
|
||||||
|
]
|
||||||
|
|
||||||
[project.urls]
|
[project.urls]
|
||||||
download = "https://pypi.org/project/sphinx-llms-txt/"
|
download = "https://pypi.org/project/sphinx-llms-txt/"
|
||||||
source = "https://github.com/jdillard/sphinx-llms-txt"
|
source = "https://github.com/jdillard/sphinx-llms-txt"
|
||||||
changelog = "https://github.com/jdillard/sphinx-llms-txt/blob/master/CHANGELOG.rst"
|
changelog = "https://github.com/jdillard/sphinx-llms-txt/blob/master/CHANGELOG.rst"
|
||||||
|
|
||||||
[project.optional-dependencies]
|
[dependency-groups]
|
||||||
dev = [
|
dev = [
|
||||||
"pytest>=7.0.0",
|
"pytest>=7.0.0",
|
||||||
"black",
|
"black",
|
||||||
@@ -40,12 +43,13 @@ dev = [
|
|||||||
"mypy",
|
"mypy",
|
||||||
"isort",
|
"isort",
|
||||||
"pre-commit",
|
"pre-commit",
|
||||||
"sphinx",
|
|
||||||
]
|
]
|
||||||
test = [
|
test = [
|
||||||
"pytest>=7.0.0",
|
"pytest>=7.0.0",
|
||||||
]
|
]
|
||||||
|
|
||||||
|
[tool.setuptools]
|
||||||
|
packages = ["sphinx_llms_txt"]
|
||||||
|
|
||||||
[tool.setuptools.dynamic]
|
[tool.setuptools.dynamic]
|
||||||
version = {attr = "sphinx_llms_txt.__version__"}
|
version = {attr = "sphinx_llms_txt.__version__"}
|
||||||
|
|||||||
+39
-10
@@ -1,5 +1,14 @@
|
|||||||
"""
|
"""
|
||||||
Sphinx extension to create a combined sources file (llms-full.txt)
|
Sphinx extension that generates llms.txt and llms-full.txt files for LLM consumption.
|
||||||
|
|
||||||
|
This extension collects documentation content from Sphinx projects and generates
|
||||||
|
two output files:
|
||||||
|
- llms.txt: A concise Markdown summary with project overview and page links
|
||||||
|
- llms-full.txt: A comprehensive reStructuredText file containing all documentation
|
||||||
|
content with resolved includes and path references
|
||||||
|
|
||||||
|
The extension processes content during the build phase, handles page-level and
|
||||||
|
block-level ignore directives, and can optionally include source code files.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from typing import Any, Dict
|
from typing import Any, Dict
|
||||||
@@ -12,7 +21,7 @@ from .manager import LLMSFullManager
|
|||||||
from .processor import DocumentProcessor
|
from .processor import DocumentProcessor
|
||||||
from .writer import FileWriter
|
from .writer import FileWriter
|
||||||
|
|
||||||
__version__ = "0.3.2"
|
__version__ = "0.7.1"
|
||||||
|
|
||||||
# Export classes needed by tests
|
# Export classes needed by tests
|
||||||
__all__ = [
|
__all__ = [
|
||||||
@@ -33,6 +42,13 @@ def doctree_resolved(app: Sphinx, doctree, docname: str):
|
|||||||
"""Called when a docname has been resolved to a document."""
|
"""Called when a docname has been resolved to a document."""
|
||||||
global _root_first_paragraph
|
global _root_first_paragraph
|
||||||
|
|
||||||
|
# Check for llms-txt-ignore metadata at the page level
|
||||||
|
if hasattr(app.env, "metadata") and docname in app.env.metadata:
|
||||||
|
metadata = app.env.metadata[docname]
|
||||||
|
if metadata.get("llms-txt-ignore", "").lower() in ("true", "1", "yes"):
|
||||||
|
_manager.mark_page_ignored(docname)
|
||||||
|
return
|
||||||
|
|
||||||
# Extract title from the document
|
# Extract title from the document
|
||||||
title = None
|
title = None
|
||||||
# findall() returns a generator, convert to list to check if it has elements
|
# findall() returns a generator, convert to list to check if it has elements
|
||||||
@@ -69,13 +85,17 @@ def build_finished(app: Sphinx, exception):
|
|||||||
config = {
|
config = {
|
||||||
"llms_txt_file": app.config.llms_txt_file,
|
"llms_txt_file": app.config.llms_txt_file,
|
||||||
"llms_txt_filename": app.config.llms_txt_filename,
|
"llms_txt_filename": app.config.llms_txt_filename,
|
||||||
|
"llms_txt_uri_template": app.config.llms_txt_uri_template,
|
||||||
"llms_txt_title": app.config.llms_txt_title,
|
"llms_txt_title": app.config.llms_txt_title,
|
||||||
"llms_txt_summary": summary,
|
"llms_txt_summary": summary,
|
||||||
"llms_txt_full_file": app.config.llms_txt_full_file,
|
"llms_txt_full_file": app.config.llms_txt_full_file,
|
||||||
"llms_txt_full_filename": app.config.llms_txt_full_filename,
|
"llms_txt_full_filename": app.config.llms_txt_full_filename,
|
||||||
"llms_txt_full_max_size": app.config.llms_txt_full_max_size,
|
"llms_txt_full_max_size": app.config.llms_txt_full_max_size,
|
||||||
|
"llms_txt_full_size_policy": app.config.llms_txt_full_size_policy,
|
||||||
"llms_txt_directives": app.config.llms_txt_directives,
|
"llms_txt_directives": app.config.llms_txt_directives,
|
||||||
"llms_txt_exclude": app.config.llms_txt_exclude,
|
"llms_txt_exclude": app.config.llms_txt_exclude,
|
||||||
|
"llms_txt_code_files": app.config.llms_txt_code_files,
|
||||||
|
"llms_txt_code_base_path": app.config.llms_txt_code_base_path,
|
||||||
"html_baseurl": getattr(app.config, "html_baseurl", ""),
|
"html_baseurl": getattr(app.config, "html_baseurl", ""),
|
||||||
}
|
}
|
||||||
_manager.set_config(config)
|
_manager.set_config(config)
|
||||||
@@ -94,25 +114,34 @@ def build_finished(app: Sphinx, exception):
|
|||||||
def setup(app: Sphinx) -> Dict[str, Any]:
|
def setup(app: Sphinx) -> Dict[str, Any]:
|
||||||
"""Set up the Sphinx extension."""
|
"""Set up the Sphinx extension."""
|
||||||
|
|
||||||
# Add configuration options
|
|
||||||
app.add_config_value("llms_txt_file", True, "env")
|
app.add_config_value("llms_txt_file", True, "env")
|
||||||
app.add_config_value("llms_txt_filename", "llms.txt", "env")
|
app.add_config_value("llms_txt_filename", "llms.txt", "env")
|
||||||
|
app.add_config_value("llms_txt_uri_template", None, "env")
|
||||||
app.add_config_value("llms_txt_full_file", True, "env")
|
app.add_config_value("llms_txt_full_file", True, "env")
|
||||||
app.add_config_value("llms_txt_full_filename", "llms-full.txt", "env")
|
app.add_config_value("llms_txt_full_filename", "llms-full.txt", "env")
|
||||||
app.add_config_value("llms_txt_full_max_size", None, "env")
|
app.add_config_value("llms_txt_full_max_size", None, "env")
|
||||||
|
app.add_config_value("llms_txt_full_size_policy", "warn_skip", "env")
|
||||||
app.add_config_value("llms_txt_directives", [], "env")
|
app.add_config_value("llms_txt_directives", [], "env")
|
||||||
app.add_config_value("llms_txt_title", None, "env")
|
app.add_config_value("llms_txt_title", None, "env")
|
||||||
app.add_config_value("llms_txt_summary", None, "env")
|
app.add_config_value("llms_txt_summary", None, "env")
|
||||||
app.add_config_value("llms_txt_exclude", [], "env")
|
app.add_config_value("llms_txt_exclude", [], "env")
|
||||||
|
app.add_config_value("llms_txt_code_files", [], "env")
|
||||||
|
app.add_config_value("llms_txt_code_base_path", None, "env")
|
||||||
|
|
||||||
# Connect to Sphinx events
|
def builder_inited(app):
|
||||||
app.connect("doctree-resolved", doctree_resolved)
|
"""Used to limit what builders are allowed to run the extension."""
|
||||||
app.connect("build-finished", build_finished)
|
|
||||||
|
|
||||||
# Reset manager and root paragraph for each build
|
allowed_builders = ["html", "dirhtml"]
|
||||||
global _manager, _root_first_paragraph
|
if hasattr(app, "builder") and app.builder.name in allowed_builders:
|
||||||
_manager = LLMSFullManager()
|
# Reset manager and root paragraph for each build
|
||||||
_root_first_paragraph = ""
|
global _manager, _root_first_paragraph
|
||||||
|
_manager = LLMSFullManager()
|
||||||
|
_root_first_paragraph = ""
|
||||||
|
|
||||||
|
app.connect("doctree-resolved", doctree_resolved)
|
||||||
|
app.connect("build-finished", build_finished)
|
||||||
|
|
||||||
|
app.connect("builder-inited", builder_inited)
|
||||||
|
|
||||||
return {
|
return {
|
||||||
"version": __version__,
|
"version": __version__,
|
||||||
|
|||||||
+671
-40
@@ -2,8 +2,10 @@
|
|||||||
Main manager module for sphinx-llms-txt.
|
Main manager module for sphinx-llms-txt.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
import glob
|
||||||
|
import subprocess
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any, Dict, Optional, Tuple
|
from typing import Any, Dict, List, Optional, Tuple, Union
|
||||||
|
|
||||||
from sphinx.application import Sphinx
|
from sphinx.application import Sphinx
|
||||||
from sphinx.environment import BuildEnvironment
|
from sphinx.environment import BuildEnvironment
|
||||||
@@ -16,6 +18,104 @@ from .writer import FileWriter
|
|||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
def _get_git_root(path: Path) -> Optional[Path]:
|
||||||
|
"""Get the git root directory for a given path."""
|
||||||
|
try:
|
||||||
|
result = subprocess.run(
|
||||||
|
["git", "rev-parse", "--show-toplevel"],
|
||||||
|
cwd=path,
|
||||||
|
capture_output=True,
|
||||||
|
text=True,
|
||||||
|
check=True,
|
||||||
|
)
|
||||||
|
return Path(result.stdout.strip())
|
||||||
|
except (subprocess.CalledProcessError, FileNotFoundError):
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _get_language_from_extension(file_path: Path) -> str:
|
||||||
|
"""Map file extension to language identifier for code blocks."""
|
||||||
|
extension_map = {
|
||||||
|
".py": "python",
|
||||||
|
".js": "javascript",
|
||||||
|
".jsx": "jsx",
|
||||||
|
".ts": "typescript",
|
||||||
|
".tsx": "tsx",
|
||||||
|
".java": "java",
|
||||||
|
".c": "c",
|
||||||
|
".cpp": "cpp",
|
||||||
|
".cc": "cpp",
|
||||||
|
".cxx": "cpp",
|
||||||
|
".h": "c",
|
||||||
|
".hpp": "cpp",
|
||||||
|
".cs": "csharp",
|
||||||
|
".php": "php",
|
||||||
|
".rb": "ruby",
|
||||||
|
".go": "go",
|
||||||
|
".rs": "rust",
|
||||||
|
".swift": "swift",
|
||||||
|
".kt": "kotlin",
|
||||||
|
".scala": "scala",
|
||||||
|
".sh": "bash",
|
||||||
|
".bash": "bash",
|
||||||
|
".zsh": "zsh",
|
||||||
|
".fish": "fish",
|
||||||
|
".ps1": "powershell",
|
||||||
|
".html": "html",
|
||||||
|
".htm": "html",
|
||||||
|
".xml": "xml",
|
||||||
|
".css": "css",
|
||||||
|
".scss": "scss",
|
||||||
|
".sass": "sass",
|
||||||
|
".less": "less",
|
||||||
|
".json": "json",
|
||||||
|
".yaml": "yaml",
|
||||||
|
".yml": "yaml",
|
||||||
|
".toml": "toml",
|
||||||
|
".ini": "ini",
|
||||||
|
".cfg": "ini",
|
||||||
|
".conf": "ini",
|
||||||
|
".sql": "sql",
|
||||||
|
".md": "markdown",
|
||||||
|
".rst": "rst",
|
||||||
|
".txt": "text",
|
||||||
|
".dockerfile": "dockerfile",
|
||||||
|
".dockerignore": "text",
|
||||||
|
".gitignore": "text",
|
||||||
|
".gitattributes": "text",
|
||||||
|
".editorconfig": "ini",
|
||||||
|
".makefile": "makefile",
|
||||||
|
".r": "r",
|
||||||
|
".R": "r",
|
||||||
|
".m": "matlab",
|
||||||
|
".pl": "perl",
|
||||||
|
".lua": "lua",
|
||||||
|
".vim": "vim",
|
||||||
|
".vimrc": "vim",
|
||||||
|
".proto": "protobuf",
|
||||||
|
".thrift": "thrift",
|
||||||
|
".graphql": "graphql",
|
||||||
|
".gql": "graphql",
|
||||||
|
}
|
||||||
|
|
||||||
|
# Get the extension from the file path
|
||||||
|
ext = file_path.suffix.lower()
|
||||||
|
|
||||||
|
# Handle special cases like Makefile, Dockerfile without extension
|
||||||
|
if not ext:
|
||||||
|
name = file_path.name.lower()
|
||||||
|
if name in ["makefile", "gnumakefile"]:
|
||||||
|
return "makefile"
|
||||||
|
elif name in ["dockerfile", "dockerfile.dev", "dockerfile.prod"]:
|
||||||
|
return "dockerfile"
|
||||||
|
elif name.startswith("dockerfile."):
|
||||||
|
return "dockerfile"
|
||||||
|
else:
|
||||||
|
return "text"
|
||||||
|
|
||||||
|
return extension_map.get(ext, "text")
|
||||||
|
|
||||||
|
|
||||||
class LLMSFullManager:
|
class LLMSFullManager:
|
||||||
"""Manages the collection and ordering of documentation sources."""
|
"""Manages the collection and ordering of documentation sources."""
|
||||||
|
|
||||||
@@ -29,6 +129,7 @@ class LLMSFullManager:
|
|||||||
self.srcdir: Optional[str] = None
|
self.srcdir: Optional[str] = None
|
||||||
self.outdir: Optional[str] = None
|
self.outdir: Optional[str] = None
|
||||||
self.app: Optional[Sphinx] = None
|
self.app: Optional[Sphinx] = None
|
||||||
|
self.ignored_pages: set = set()
|
||||||
|
|
||||||
def set_master_doc(self, master_doc: str):
|
def set_master_doc(self, master_doc: str):
|
||||||
"""Set the master document name."""
|
"""Set the master document name."""
|
||||||
@@ -44,6 +145,27 @@ class LLMSFullManager:
|
|||||||
"""Update the title for a page."""
|
"""Update the title for a page."""
|
||||||
self.collector.update_page_title(docname, title)
|
self.collector.update_page_title(docname, title)
|
||||||
|
|
||||||
|
def mark_page_ignored(self, docname: str):
|
||||||
|
"""Mark a page as ignored due to llms-txt-ignore metadata."""
|
||||||
|
self.ignored_pages.add(docname)
|
||||||
|
|
||||||
|
def _filter_ignored_pages(
|
||||||
|
self, page_order: Union[List[str], List[Tuple[str, str]]]
|
||||||
|
) -> Union[List[str], List[Tuple[str, str]]]:
|
||||||
|
"""Filter out ignored pages from page_order."""
|
||||||
|
filtered_pages = []
|
||||||
|
for item in page_order:
|
||||||
|
# Handle both old format (str) and new format (tuple)
|
||||||
|
if isinstance(item, tuple):
|
||||||
|
docname, _ = item
|
||||||
|
else:
|
||||||
|
docname = item
|
||||||
|
|
||||||
|
if docname not in self.ignored_pages:
|
||||||
|
filtered_pages.append(item)
|
||||||
|
|
||||||
|
return filtered_pages
|
||||||
|
|
||||||
def set_config(self, config: Dict[str, Any]):
|
def set_config(self, config: Dict[str, Any]):
|
||||||
"""Set configuration options."""
|
"""Set configuration options."""
|
||||||
self.config = config
|
self.config = config
|
||||||
@@ -75,7 +197,6 @@ class LLMSFullManager:
|
|||||||
possible_sources = [
|
possible_sources = [
|
||||||
Path(outdir) / "_sources",
|
Path(outdir) / "_sources",
|
||||||
Path(outdir) / "html" / "_sources",
|
Path(outdir) / "html" / "_sources",
|
||||||
Path(outdir) / "singlehtml" / "_sources",
|
|
||||||
]
|
]
|
||||||
|
|
||||||
for path in possible_sources:
|
for path in possible_sources:
|
||||||
@@ -83,25 +204,43 @@ class LLMSFullManager:
|
|||||||
sources_dir = path
|
sources_dir = path
|
||||||
break
|
break
|
||||||
|
|
||||||
if not sources_dir:
|
# Get the correct page order (with or without source suffixes)
|
||||||
logger.warning(
|
|
||||||
"Could not find _sources directory, skipping llms-full creation"
|
|
||||||
)
|
|
||||||
return
|
|
||||||
|
|
||||||
# Get the correct page order with source suffixes
|
|
||||||
page_order = self.collector.get_page_order(sources_dir)
|
page_order = self.collector.get_page_order(sources_dir)
|
||||||
|
|
||||||
if not page_order:
|
if not page_order:
|
||||||
logger.warning(
|
logger.warning("Could not determine page order, skipping file generation")
|
||||||
"Could not determine page order, skipping llms-full creation"
|
|
||||||
)
|
|
||||||
return
|
return
|
||||||
|
|
||||||
# Apply exclusion filter if configured
|
# Apply exclusion filter if configured
|
||||||
page_order = self.collector.filter_excluded_pages(page_order)
|
page_order = self.collector.filter_excluded_pages(page_order)
|
||||||
|
|
||||||
# Determine output file name and location
|
# If no sources directory, only generate llms.txt and return early
|
||||||
|
if not sources_dir:
|
||||||
|
# Generate llms.txt if requested
|
||||||
|
if self.config.get("llms_txt_file"):
|
||||||
|
filtered_page_order = self._filter_ignored_pages(page_order)
|
||||||
|
self.writer.write_verbose_info_to_file(
|
||||||
|
filtered_page_order,
|
||||||
|
self.collector.page_titles,
|
||||||
|
0, # No line count since no llms-full.txt
|
||||||
|
sources_dir,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Only warn if user explicitly wants llms-full.txt
|
||||||
|
if self.config.get("llms_txt_full_file"):
|
||||||
|
# Check if html_copy_source is False
|
||||||
|
if self.app and not self.app.config.html_copy_source:
|
||||||
|
logger.warning(
|
||||||
|
"Could not find _sources directory, skipping llms-full.txt."
|
||||||
|
"Set html_copy_source = True in conf.py to enable."
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
logger.warning(
|
||||||
|
"Could not find _sources directory, skipping llms-full.txt"
|
||||||
|
)
|
||||||
|
return
|
||||||
|
|
||||||
|
# Determine output file name and location for llms-full.txt
|
||||||
output_filename = self.config.get("llms_txt_full_filename")
|
output_filename = self.config.get("llms_txt_full_filename")
|
||||||
output_path = Path(outdir) / output_filename
|
output_path = Path(outdir) / output_filename
|
||||||
|
|
||||||
@@ -163,20 +302,49 @@ class LLMSFullManager:
|
|||||||
# Generate content
|
# Generate content
|
||||||
content_parts = []
|
content_parts = []
|
||||||
|
|
||||||
|
# Track code files for later processing
|
||||||
|
code_file_parts = []
|
||||||
|
|
||||||
|
# Count lines in code files (initially 0)
|
||||||
|
code_files_line_count = 0
|
||||||
|
|
||||||
# Add pages in order
|
# Add pages in order
|
||||||
added_files = set()
|
added_files = set()
|
||||||
total_line_count = 0
|
total_line_count = code_files_line_count
|
||||||
max_lines = self.config.get("llms_txt_full_max_size")
|
max_lines = self.config.get("llms_txt_full_max_size")
|
||||||
abort_due_to_max_lines = False
|
|
||||||
|
# Parse size_policy configuration early to determine collection strategy
|
||||||
|
size_policy_action = None
|
||||||
|
aborted_due_to_size = False
|
||||||
|
if max_lines is not None:
|
||||||
|
size_policy = self.config.get("llms_txt_full_size_policy", "warn_skip")
|
||||||
|
_, size_policy_action = self._parse_size_policy_config(size_policy)
|
||||||
|
|
||||||
|
# Only collect all files if action is "keep"
|
||||||
|
# For "skip" and "note", we can abort early when size limit is exceeded
|
||||||
|
should_abort_early = size_policy_action in ["skip", "note"]
|
||||||
|
|
||||||
for docname, _ in page_order:
|
for docname, _ in page_order:
|
||||||
|
# Skip pages marked as ignored
|
||||||
|
if docname in self.ignored_pages:
|
||||||
|
logger.debug(f"sphinx-llms-txt: Skipping ignored page: {docname}")
|
||||||
|
continue
|
||||||
|
|
||||||
if docname in docname_to_file:
|
if docname in docname_to_file:
|
||||||
file_path = docname_to_file[docname]
|
file_path = docname_to_file[docname]
|
||||||
content, line_count = self._read_source_file(file_path, docname)
|
content, line_count = self._read_source_file(file_path, docname)
|
||||||
|
|
||||||
# Check if adding this file would exceed the maximum line count
|
# Abort early for skip/note actions
|
||||||
if max_lines is not None and total_line_count + line_count > max_lines:
|
if (
|
||||||
abort_due_to_max_lines = True
|
max_lines is not None
|
||||||
|
and total_line_count + line_count > max_lines
|
||||||
|
and should_abort_early
|
||||||
|
):
|
||||||
|
logger.debug(
|
||||||
|
f"sphinx-llms-txt: Stopping collection due to size limit. "
|
||||||
|
f"File {docname} would exceed limit."
|
||||||
|
)
|
||||||
|
aborted_due_to_size = True
|
||||||
break
|
break
|
||||||
|
|
||||||
# Double-check this file should be included (not in excluded patterns)
|
# Double-check this file should be included (not in excluded patterns)
|
||||||
@@ -209,7 +377,9 @@ class LLMSFullManager:
|
|||||||
)
|
)
|
||||||
|
|
||||||
# Add any remaining files (in alphabetical order) that aren't in the page order
|
# Add any remaining files (in alphabetical order) that aren't in the page order
|
||||||
if not abort_due_to_max_lines:
|
# Only skip this if we aborted early due to size limits for skip/note actions
|
||||||
|
size_limit_exceeded = max_lines is not None and total_line_count > max_lines
|
||||||
|
if not (size_limit_exceeded and should_abort_early):
|
||||||
# Get all source files in the _sources directory using configured suffixes
|
# Get all source files in the _sources directory using configured suffixes
|
||||||
source_suffixes = self._get_source_suffixes()
|
source_suffixes = self._get_source_suffixes()
|
||||||
all_source_files = []
|
all_source_files = []
|
||||||
@@ -257,6 +427,13 @@ class LLMSFullManager:
|
|||||||
if docname is None:
|
if docname is None:
|
||||||
continue
|
continue
|
||||||
|
|
||||||
|
# Skip pages marked as ignored
|
||||||
|
if docname in self.ignored_pages:
|
||||||
|
logger.debug(
|
||||||
|
f"sphinx-llms-txt: Skipping ignored remaining file: {docname}"
|
||||||
|
)
|
||||||
|
continue
|
||||||
|
|
||||||
# Skip excluded docnames
|
# Skip excluded docnames
|
||||||
if exclude_patterns and any(
|
if exclude_patterns and any(
|
||||||
self.collector._match_exclude_pattern(docname, pattern)
|
self.collector._match_exclude_pattern(docname, pattern)
|
||||||
@@ -268,8 +445,13 @@ class LLMSFullManager:
|
|||||||
# Read and process the file
|
# Read and process the file
|
||||||
content, line_count = self._read_source_file(file_path, docname)
|
content, line_count = self._read_source_file(file_path, docname)
|
||||||
|
|
||||||
# Check if adding this file would exceed the maximum line count
|
# Abort early for skip/note actions
|
||||||
if max_lines is not None and total_line_count + line_count > max_lines:
|
if (
|
||||||
|
max_lines is not None
|
||||||
|
and total_line_count + line_count > max_lines
|
||||||
|
and should_abort_early
|
||||||
|
):
|
||||||
|
aborted_due_to_size = True
|
||||||
break
|
break
|
||||||
|
|
||||||
if content:
|
if content:
|
||||||
@@ -277,34 +459,108 @@ class LLMSFullManager:
|
|||||||
content_parts.append(content)
|
content_parts.append(content)
|
||||||
total_line_count += line_count
|
total_line_count += line_count
|
||||||
|
|
||||||
# Check if line limit was exceeded before creating the file
|
# Process code files at the end if configured
|
||||||
max_lines = self.config.get("llms_txt_full_max_size")
|
# Only skip this if we aborted early due to size limits for skip/note actions
|
||||||
if abort_due_to_max_lines or (
|
if not (size_limit_exceeded and should_abort_early):
|
||||||
max_lines is not None and total_line_count > max_lines
|
code_file_parts, processed_file_paths = self._process_code_files()
|
||||||
):
|
code_files_line_count = sum(
|
||||||
logger.warning(
|
part.count("\n") + 1 for part in code_file_parts
|
||||||
f"sphinx-llms-txt: Max line limit ({max_lines}) exceeded:"
|
|
||||||
f" {total_line_count} > {max_lines}. "
|
|
||||||
f"Not creating llms-full.txt file."
|
|
||||||
)
|
)
|
||||||
|
|
||||||
# Log summary information if requested
|
# Check if adding code files would exceed the maximum line count
|
||||||
if self.config.get("llms_txt_file"):
|
# For "keep" action, we include code files regardless of size
|
||||||
self.writer.write_verbose_info_to_file(
|
if (
|
||||||
page_order, self.collector.page_titles, total_line_count
|
max_lines is not None
|
||||||
|
and total_line_count + code_files_line_count > max_lines
|
||||||
|
and should_abort_early
|
||||||
|
):
|
||||||
|
logger.warning(
|
||||||
|
f"sphinx-llms-txt: Adding code files would exceed max line limit "
|
||||||
|
f"({max_lines}). Current: {total_line_count}, "
|
||||||
|
f"Code files: {code_files_line_count}. Skipping code files."
|
||||||
)
|
)
|
||||||
|
aborted_due_to_size = True
|
||||||
|
else:
|
||||||
|
# Add source code files section if there are any code files
|
||||||
|
if code_file_parts:
|
||||||
|
section_header = self._create_code_files_section_header(
|
||||||
|
processed_file_paths
|
||||||
|
)
|
||||||
|
content_parts.append(section_header)
|
||||||
|
content_parts.extend(code_file_parts)
|
||||||
|
# Add line count for the section header too
|
||||||
|
total_line_count += (
|
||||||
|
code_files_line_count + section_header.count("\n") + 1
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
# If we aborted early for skip/note actions, set empty code file parts
|
||||||
|
code_file_parts = []
|
||||||
|
|
||||||
return
|
# Handle size limit exceeded cases
|
||||||
|
if max_lines is not None and (
|
||||||
|
total_line_count > max_lines or aborted_due_to_size
|
||||||
|
):
|
||||||
|
# Parse the size_policy configuration (reuse what we parsed earlier)
|
||||||
|
size_policy = self.config.get("llms_txt_full_size_policy", "warn_skip")
|
||||||
|
log_level, action = self._parse_size_policy_config(size_policy)
|
||||||
|
|
||||||
# Write combined file if limit wasn't exceeded
|
# Log with the specified level
|
||||||
success = self.writer.write_combined_file(
|
filename = self.config.get("llms_txt_full_filename", "llms-full.txt")
|
||||||
content_parts, output_path, total_line_count
|
message = f"sphinx-llms-txt: Max lines ({max_lines}) exceeded for {filename}" # noqa: E501
|
||||||
)
|
|
||||||
|
if log_level == "info":
|
||||||
|
logger.info(message)
|
||||||
|
else:
|
||||||
|
logger.warning(message)
|
||||||
|
|
||||||
|
# Handle different actions
|
||||||
|
if action == "skip":
|
||||||
|
filename = self.config.get("llms_txt_full_filename", "llms-full.txt")
|
||||||
|
logger.info(f"sphinx-llms-txt: Skipping {filename} generation")
|
||||||
|
# Log summary information if requested
|
||||||
|
if self.config.get("llms_txt_file"):
|
||||||
|
filtered_page_order = self._filter_ignored_pages(page_order)
|
||||||
|
self.writer.write_verbose_info_to_file(
|
||||||
|
filtered_page_order,
|
||||||
|
self.collector.page_titles,
|
||||||
|
total_line_count,
|
||||||
|
sources_dir,
|
||||||
|
)
|
||||||
|
return
|
||||||
|
elif action == "note":
|
||||||
|
logger.info(f"sphinx-llms-txt: Creating placeholder {output_path}")
|
||||||
|
self._write_placeholder_file(output_path, max_lines)
|
||||||
|
|
||||||
|
# Log summary information if requested
|
||||||
|
if self.config.get("llms_txt_file"):
|
||||||
|
filtered_page_order = self._filter_ignored_pages(page_order)
|
||||||
|
self.writer.write_verbose_info_to_file(
|
||||||
|
filtered_page_order,
|
||||||
|
self.collector.page_titles,
|
||||||
|
total_line_count,
|
||||||
|
sources_dir,
|
||||||
|
)
|
||||||
|
return
|
||||||
|
elif action == "keep":
|
||||||
|
filename = self.config.get("llms_txt_full_filename", "llms-full.txt")
|
||||||
|
# Fall through to write the file
|
||||||
|
|
||||||
|
# Write combined file only if we have content to write
|
||||||
|
if content_parts:
|
||||||
|
success = self.writer.write_combined_file(
|
||||||
|
content_parts, output_path, total_line_count
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
success = False
|
||||||
|
|
||||||
# Log summary information if requested
|
# Log summary information if requested
|
||||||
if success and self.config.get("llms_txt_file"):
|
if success and self.config.get("llms_txt_file"):
|
||||||
|
filtered_page_order = self._filter_ignored_pages(page_order)
|
||||||
self.writer.write_verbose_info_to_file(
|
self.writer.write_verbose_info_to_file(
|
||||||
page_order, self.collector.page_titles, total_line_count
|
filtered_page_order,
|
||||||
|
self.collector.page_titles,
|
||||||
|
total_line_count,
|
||||||
|
sources_dir,
|
||||||
)
|
)
|
||||||
|
|
||||||
def _read_source_file(self, file_path: Path, docname: str) -> Tuple[str, int]:
|
def _read_source_file(self, file_path: Path, docname: str) -> Tuple[str, int]:
|
||||||
@@ -370,3 +626,378 @@ class LLMSFullManager:
|
|||||||
return source_suffix
|
return source_suffix
|
||||||
else:
|
else:
|
||||||
return [source_suffix] # String format
|
return [source_suffix] # String format
|
||||||
|
|
||||||
|
def _process_code_files(self) -> Tuple[List[str], List[Path]]:
|
||||||
|
"""Process code files specified in llms_txt_code_files configuration.
|
||||||
|
|
||||||
|
Supports include/exclude patterns with +:/- : prefixes:
|
||||||
|
- '+:pattern' = include files matching pattern
|
||||||
|
- '-:pattern' = exclude files matching pattern
|
||||||
|
- 'pattern' (no prefix) = ignored (no special handling)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Tuple of (formatted code block strings, list of processed file paths)
|
||||||
|
"""
|
||||||
|
code_file_patterns = self.config.get("llms_txt_code_files", [])
|
||||||
|
if not code_file_patterns:
|
||||||
|
return [], []
|
||||||
|
|
||||||
|
# Parse patterns into include and exclude lists
|
||||||
|
include_patterns = []
|
||||||
|
exclude_patterns = []
|
||||||
|
|
||||||
|
for pattern in code_file_patterns:
|
||||||
|
if pattern.startswith("-:"):
|
||||||
|
exclude_patterns.append(pattern[2:]) # Remove the '-:' prefix
|
||||||
|
elif pattern.startswith("+:"):
|
||||||
|
include_patterns.append(pattern[2:]) # Remove the '+:' prefix
|
||||||
|
else:
|
||||||
|
# No prefix = log warning about ignored pattern
|
||||||
|
logger.warning(
|
||||||
|
f"sphinx-llms-txt: Code file pattern '{pattern}' ignored."
|
||||||
|
f"Use '+:{pattern}' to include or '-:{pattern}' to exclude."
|
||||||
|
)
|
||||||
|
|
||||||
|
# If no include patterns specified, nothing to process
|
||||||
|
if not include_patterns:
|
||||||
|
return [], []
|
||||||
|
|
||||||
|
code_parts = []
|
||||||
|
processed_files = set()
|
||||||
|
all_matching_files = set()
|
||||||
|
|
||||||
|
# First, collect all files matching include patterns
|
||||||
|
for pattern in include_patterns:
|
||||||
|
# Resolve pattern relative to source directory
|
||||||
|
if self.srcdir:
|
||||||
|
pattern_path = Path(self.srcdir) / pattern
|
||||||
|
else:
|
||||||
|
pattern_path = Path(pattern)
|
||||||
|
|
||||||
|
# Use glob to find matching files
|
||||||
|
matching_files = glob.glob(str(pattern_path), recursive=True)
|
||||||
|
|
||||||
|
for file_path_str in matching_files:
|
||||||
|
file_path = Path(file_path_str)
|
||||||
|
if file_path.is_file(): # Only add files, not directories
|
||||||
|
all_matching_files.add(file_path.resolve())
|
||||||
|
|
||||||
|
# Filter out files matching exclude patterns
|
||||||
|
filtered_files = set()
|
||||||
|
for file_path in all_matching_files:
|
||||||
|
should_exclude = False
|
||||||
|
|
||||||
|
for exclude_pattern in exclude_patterns:
|
||||||
|
# Resolve exclude pattern relative to source directory
|
||||||
|
if self.srcdir:
|
||||||
|
exclude_pattern_path = Path(self.srcdir) / exclude_pattern
|
||||||
|
else:
|
||||||
|
exclude_pattern_path = Path(exclude_pattern)
|
||||||
|
|
||||||
|
# Check if this file matches the exclude pattern
|
||||||
|
exclude_matches = glob.glob(str(exclude_pattern_path), recursive=True)
|
||||||
|
if str(file_path) in exclude_matches:
|
||||||
|
should_exclude = True
|
||||||
|
logger.debug(
|
||||||
|
f"sphinx-llms-txt: Excluding code file: {file_path} "
|
||||||
|
f"(matched pattern: {exclude_pattern})"
|
||||||
|
)
|
||||||
|
break
|
||||||
|
|
||||||
|
if not should_exclude:
|
||||||
|
filtered_files.add(file_path)
|
||||||
|
|
||||||
|
# Sort files for consistent ordering
|
||||||
|
sorted_files = sorted(filtered_files)
|
||||||
|
|
||||||
|
for file_path in sorted_files:
|
||||||
|
# Skip if already processed (shouldn't happen with set, but safety check)
|
||||||
|
if file_path in processed_files:
|
||||||
|
continue
|
||||||
|
|
||||||
|
try:
|
||||||
|
# Read the file content
|
||||||
|
with open(file_path, "r", encoding="utf-8", errors="ignore") as f:
|
||||||
|
content = f.read()
|
||||||
|
|
||||||
|
# Get language identifier
|
||||||
|
language = _get_language_from_extension(file_path)
|
||||||
|
|
||||||
|
# Get relative path from source directory for title
|
||||||
|
if self.srcdir:
|
||||||
|
try:
|
||||||
|
title = file_path.relative_to(Path(self.srcdir))
|
||||||
|
|
||||||
|
# Strip base path if configured,
|
||||||
|
# or auto-detect from git root
|
||||||
|
base_path = self.config.get("llms_txt_code_base_path")
|
||||||
|
if base_path is None:
|
||||||
|
# Auto-detect: try to make path relative to git root
|
||||||
|
git_root = _get_git_root(Path(self.srcdir))
|
||||||
|
if git_root:
|
||||||
|
try:
|
||||||
|
# Get srcdir relative to git root
|
||||||
|
srcdir_relative = Path(self.srcdir).relative_to(
|
||||||
|
git_root
|
||||||
|
)
|
||||||
|
# Calculate relative path from srcdir to
|
||||||
|
# git root
|
||||||
|
if srcdir_relative != Path("."):
|
||||||
|
# Count directory levels to go up
|
||||||
|
up_levels = len(srcdir_relative.parts)
|
||||||
|
base_path = "../" * up_levels
|
||||||
|
else:
|
||||||
|
base_path = None
|
||||||
|
except ValueError:
|
||||||
|
base_path = None
|
||||||
|
|
||||||
|
if base_path:
|
||||||
|
title_str = str(title)
|
||||||
|
if title_str.startswith(base_path):
|
||||||
|
title = Path(title_str[len(base_path) :])
|
||||||
|
except ValueError:
|
||||||
|
# File is not relative to srcdir, use filename
|
||||||
|
title = file_path.name
|
||||||
|
else:
|
||||||
|
title = file_path.name
|
||||||
|
|
||||||
|
# Format as code block with equals underline
|
||||||
|
title_str = str(title)
|
||||||
|
equals_line = "=" * len(title_str)
|
||||||
|
|
||||||
|
# Indent the content for reStructuredText code-block directive
|
||||||
|
indented_content = "\n".join(
|
||||||
|
f" {line}" if line.strip() else ""
|
||||||
|
for line in content.splitlines()
|
||||||
|
)
|
||||||
|
|
||||||
|
code_block = f"""
|
||||||
|
{title_str}
|
||||||
|
{equals_line}
|
||||||
|
|
||||||
|
.. code-block:: {language}
|
||||||
|
|
||||||
|
{indented_content}"""
|
||||||
|
code_parts.append(code_block)
|
||||||
|
|
||||||
|
processed_files.add(file_path)
|
||||||
|
logger.debug(f"sphinx-llms-txt: Added code file: {title}")
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.warning(
|
||||||
|
f"sphinx-llms-txt: Error reading code file {file_path}: {e}"
|
||||||
|
)
|
||||||
|
continue
|
||||||
|
|
||||||
|
return code_parts, sorted(processed_files)
|
||||||
|
|
||||||
|
def _create_code_files_section_header(self, file_paths: List[Path] = None) -> str:
|
||||||
|
"""Create the section header for source code files.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
file_paths: List of file paths that were added to generate tree view
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
String containing the section header with title, underlines, description,
|
||||||
|
and file tree
|
||||||
|
"""
|
||||||
|
section_title = "Source Code Files"
|
||||||
|
star_line = "*" * len(section_title)
|
||||||
|
|
||||||
|
description = "This section contains source code files from the project repository. These files are included to provide implementation context and technical details that complement the documentation above." # noqa: E501
|
||||||
|
|
||||||
|
header = f"""
|
||||||
|
{star_line}
|
||||||
|
{section_title}
|
||||||
|
{star_line}
|
||||||
|
|
||||||
|
{description}"""
|
||||||
|
|
||||||
|
# Add file tree if file paths are provided
|
||||||
|
if file_paths:
|
||||||
|
tree_display = self._generate_file_tree(file_paths)
|
||||||
|
header += f"""
|
||||||
|
|
||||||
|
**Files included:**
|
||||||
|
|
||||||
|
.. code-block:: text
|
||||||
|
|
||||||
|
{tree_display}"""
|
||||||
|
|
||||||
|
return header
|
||||||
|
|
||||||
|
def _generate_file_tree(self, file_paths: List[Path]) -> str:
|
||||||
|
"""Generate a tree-like representation of file paths.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
file_paths: List of file paths to display in tree format
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
String containing indented tree representation of the files
|
||||||
|
"""
|
||||||
|
if not file_paths:
|
||||||
|
return ""
|
||||||
|
|
||||||
|
# Convert to relative paths if possible and create tree structure
|
||||||
|
tree_data = {}
|
||||||
|
|
||||||
|
for file_path in sorted(file_paths):
|
||||||
|
# Get relative path from source directory for display
|
||||||
|
if self.srcdir:
|
||||||
|
try:
|
||||||
|
rel_path = file_path.relative_to(Path(self.srcdir))
|
||||||
|
|
||||||
|
# Apply base path stripping logic similar to code processing
|
||||||
|
base_path = self.config.get("llms_txt_code_base_path")
|
||||||
|
if base_path is None:
|
||||||
|
# Auto-detect: try to make path relative to git root
|
||||||
|
git_root = _get_git_root(Path(self.srcdir))
|
||||||
|
if git_root:
|
||||||
|
try:
|
||||||
|
# Get srcdir relative to git root
|
||||||
|
srcdir_relative = Path(self.srcdir).relative_to(
|
||||||
|
git_root
|
||||||
|
)
|
||||||
|
# Calculate relative path from srcdir to git root
|
||||||
|
if srcdir_relative != Path("."):
|
||||||
|
# Count directory levels to go up
|
||||||
|
up_levels = len(srcdir_relative.parts)
|
||||||
|
base_path = "../" * up_levels
|
||||||
|
else:
|
||||||
|
base_path = None
|
||||||
|
except ValueError:
|
||||||
|
base_path = None
|
||||||
|
|
||||||
|
if base_path:
|
||||||
|
rel_path_str = str(rel_path)
|
||||||
|
if rel_path_str.startswith(base_path):
|
||||||
|
rel_path = Path(rel_path_str[len(base_path) :])
|
||||||
|
|
||||||
|
except ValueError:
|
||||||
|
# File is not relative to srcdir, use filename
|
||||||
|
rel_path = Path(file_path.name)
|
||||||
|
else:
|
||||||
|
rel_path = Path(file_path.name)
|
||||||
|
|
||||||
|
# Build nested dictionary structure
|
||||||
|
parts = rel_path.parts
|
||||||
|
current = tree_data
|
||||||
|
for part in parts[:-1]: # All but the last part (directories)
|
||||||
|
if part not in current:
|
||||||
|
current[part] = {}
|
||||||
|
current = current[part]
|
||||||
|
|
||||||
|
# Add the file (last part)
|
||||||
|
if parts:
|
||||||
|
current[parts[-1]] = None # None indicates it's a file
|
||||||
|
|
||||||
|
# Convert tree structure to string representation
|
||||||
|
lines = []
|
||||||
|
self._format_tree_node(tree_data, lines, "", True)
|
||||||
|
|
||||||
|
# Indent each line for reStructuredText code block
|
||||||
|
indented_lines = [f" {line}" for line in lines]
|
||||||
|
return "\n".join(indented_lines)
|
||||||
|
|
||||||
|
def _format_tree_node(
|
||||||
|
self, node: dict, lines: List[str], prefix: str, is_root: bool
|
||||||
|
):
|
||||||
|
"""Recursively format tree nodes into lines with proper tree characters.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
node: Dictionary representing the tree structure
|
||||||
|
lines: List to append formatted lines to
|
||||||
|
prefix: Current prefix for indentation and tree characters
|
||||||
|
is_root: Whether this is the root level (no tree characters)
|
||||||
|
"""
|
||||||
|
if not node:
|
||||||
|
return
|
||||||
|
|
||||||
|
items = sorted(node.items())
|
||||||
|
|
||||||
|
for i, (name, subtree) in enumerate(items):
|
||||||
|
is_last = i == len(items) - 1
|
||||||
|
|
||||||
|
if is_root:
|
||||||
|
# Root level - no tree characters
|
||||||
|
current_prefix = ""
|
||||||
|
next_prefix = ""
|
||||||
|
else:
|
||||||
|
# Use tree characters
|
||||||
|
current_prefix = prefix + ("└── " if is_last else "├── ")
|
||||||
|
next_prefix = prefix + (" " if is_last else "│ ")
|
||||||
|
|
||||||
|
lines.append(current_prefix + name)
|
||||||
|
|
||||||
|
# Recursively handle subdirectories
|
||||||
|
if subtree is not None: # It's a directory
|
||||||
|
self._format_tree_node(subtree, lines, next_prefix, False)
|
||||||
|
|
||||||
|
def _parse_size_policy_config(self, size_policy: str) -> tuple[str, str]:
|
||||||
|
"""Parse the llms_txt_full_size_policy configuration value.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
size_policy: Configuration string in format "loglevel_action"
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Tuple of (log_level, action) where:
|
||||||
|
- log_level is "warn" or "info"
|
||||||
|
- action is "keep", "skip", or "note"
|
||||||
|
"""
|
||||||
|
if not size_policy or "_" not in size_policy:
|
||||||
|
logger.warning(
|
||||||
|
f"sphinx-llms-txt: Invalid llms_txt_full_size_policy "
|
||||||
|
f"format: '{size_policy}'. "
|
||||||
|
f"Using default 'warn_skip'."
|
||||||
|
)
|
||||||
|
return "warn", "skip"
|
||||||
|
|
||||||
|
parts = size_policy.split("_", 1) # Split on first underscore only
|
||||||
|
log_level, action = parts[0], parts[1]
|
||||||
|
|
||||||
|
# Validate log level
|
||||||
|
if log_level not in ["warn", "info"]:
|
||||||
|
logger.warning(
|
||||||
|
f"sphinx-llms-txt: Invalid log level '{log_level}' in "
|
||||||
|
f"llms_txt_full_size_policy. "
|
||||||
|
f"Valid options: warn, info. Using 'warn'."
|
||||||
|
)
|
||||||
|
log_level = "warn"
|
||||||
|
|
||||||
|
# Validate action
|
||||||
|
if action not in ["keep", "skip", "note"]:
|
||||||
|
logger.warning(
|
||||||
|
f"sphinx-llms-txt: Invalid action '{action}' in "
|
||||||
|
f"llms_txt_full_size_policy. "
|
||||||
|
f"Valid options: keep, skip, note. Using 'skip'."
|
||||||
|
)
|
||||||
|
action = "skip"
|
||||||
|
|
||||||
|
return log_level, action
|
||||||
|
|
||||||
|
def _write_placeholder_file(self, output_path: Path, max_lines: int):
|
||||||
|
"""Write a placeholder llms-full.txt file with a note about size limit.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
output_path: Path where the placeholder file should be written
|
||||||
|
max_lines: The configured maximum line limit
|
||||||
|
"""
|
||||||
|
# Create the placeholder note content
|
||||||
|
placeholder_content = (
|
||||||
|
f".. This file was not generated because it exceeded the configured size limit.\n" # noqa: E501
|
||||||
|
" See the conf.py ``llms_txt_full_max_size`` and ``llms_txt_full_size_policy``\n" # noqa: E501
|
||||||
|
" for configuration options.\n"
|
||||||
|
"\n"
|
||||||
|
f" Configured max size: {max_lines} lines\n"
|
||||||
|
"\n"
|
||||||
|
" For more information, see: https://sphinx-llms-txt.readthedocs.io/en/latest/configuration-values.html#llms-txt-full-max-size\n" # noqa: E501
|
||||||
|
)
|
||||||
|
|
||||||
|
try:
|
||||||
|
with open(output_path, "w", encoding="utf-8") as f:
|
||||||
|
f.write(placeholder_content)
|
||||||
|
logger.debug(f"sphinx-llms-txt: Wrote placeholder file: {output_path}")
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(
|
||||||
|
f"sphinx-llms-txt: Error writing placeholder file {output_path}: {e}"
|
||||||
|
)
|
||||||
|
|||||||
@@ -44,7 +44,10 @@ class DocumentProcessor:
|
|||||||
Returns:
|
Returns:
|
||||||
Processed content with directives properly resolved
|
Processed content with directives properly resolved
|
||||||
"""
|
"""
|
||||||
# First process include directives
|
# First process llms-txt-ignore blocks
|
||||||
|
content = self._process_ignore_blocks(content)
|
||||||
|
|
||||||
|
# Then process include directives
|
||||||
content = self._process_includes(content, source_path)
|
content = self._process_includes(content, source_path)
|
||||||
|
|
||||||
# Then process path directives (image, figure, etc.)
|
# Then process path directives (image, figure, etc.)
|
||||||
@@ -125,8 +128,11 @@ class DocumentProcessor:
|
|||||||
Returns:
|
Returns:
|
||||||
Processed content with directive paths properly resolved
|
Processed content with directive paths properly resolved
|
||||||
"""
|
"""
|
||||||
|
# Get code block ranges to skip directives inside them
|
||||||
|
code_block_ranges = self._get_code_block_ranges(content)
|
||||||
|
|
||||||
# Get the configured path directives to process
|
# Get the configured path directives to process
|
||||||
default_path_directives = ["image", "figure"]
|
default_path_directives = ["image", "figure", "literalinclude"]
|
||||||
custom_path_directives = self.config.get("llms_txt_directives")
|
custom_path_directives = self.config.get("llms_txt_directives")
|
||||||
path_directives = set(default_path_directives + custom_path_directives)
|
path_directives = set(default_path_directives + custom_path_directives)
|
||||||
|
|
||||||
@@ -140,6 +146,11 @@ class DocumentProcessor:
|
|||||||
is_test = "pytest" in str(source_path) and "subdir" in str(source_path)
|
is_test = "pytest" in str(source_path) and "subdir" in str(source_path)
|
||||||
|
|
||||||
def replace_directive_path(match, base_url=base_url, is_test=is_test):
|
def replace_directive_path(match, base_url=base_url, is_test=is_test):
|
||||||
|
# Check if this directive is within a code block
|
||||||
|
if self._is_in_code_block(match.start(), code_block_ranges):
|
||||||
|
# This directive is inside a code block, don't process it
|
||||||
|
return match.group(0)
|
||||||
|
|
||||||
prefix = match.group(1) # The entire directive prefix including whitespace
|
prefix = match.group(1) # The entire directive prefix including whitespace
|
||||||
path = match.group(3).strip() # The path argument
|
path = match.group(3).strip() # The path argument
|
||||||
|
|
||||||
@@ -243,9 +254,12 @@ class DocumentProcessor:
|
|||||||
"""
|
"""
|
||||||
possible_paths = []
|
possible_paths = []
|
||||||
|
|
||||||
# If it's an absolute path, use it directly
|
# If it's an absolute path, treat it as relative to srcdir
|
||||||
if os.path.isabs(include_path):
|
if os.path.isabs(include_path):
|
||||||
possible_paths.append(Path(include_path))
|
# Remove the leading slash and treat as relative to srcdir
|
||||||
|
relative_path = include_path.lstrip("/")
|
||||||
|
if self.srcdir:
|
||||||
|
possible_paths.append((Path(self.srcdir) / relative_path).resolve())
|
||||||
else:
|
else:
|
||||||
# Relative to the source file (in _sources directory)
|
# Relative to the source file (in _sources directory)
|
||||||
possible_paths.append((source_path.parent / include_path).resolve())
|
possible_paths.append((source_path.parent / include_path).resolve())
|
||||||
@@ -270,6 +284,71 @@ class DocumentProcessor:
|
|||||||
|
|
||||||
return possible_paths
|
return possible_paths
|
||||||
|
|
||||||
|
def _get_code_block_ranges(self, content: str) -> List[Tuple[int, int]]:
|
||||||
|
"""Find all code block ranges in the content.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
content: The source content to analyze
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
List of (start, end) tuples representing code block character
|
||||||
|
ranges
|
||||||
|
"""
|
||||||
|
code_block_ranges = []
|
||||||
|
|
||||||
|
# Match code block as well as `code` and `sourcecode` aliases
|
||||||
|
code_block_pattern = re.compile(
|
||||||
|
r"^(\s*)\.\.\s+(code-block|code|sourcecode)::\s*\S*\s*$", re.MULTILINE
|
||||||
|
)
|
||||||
|
|
||||||
|
for match in code_block_pattern.finditer(content):
|
||||||
|
start_pos = match.start()
|
||||||
|
indent = match.group(1)
|
||||||
|
indent_len = len(indent)
|
||||||
|
|
||||||
|
# Find the end of the code block by looking for the next line
|
||||||
|
# that is not indented more than the directive
|
||||||
|
block_start = match.end()
|
||||||
|
pos = block_start
|
||||||
|
|
||||||
|
# Skip any blank lines immediately after the directive
|
||||||
|
while pos < len(content) and content[pos] in "\n":
|
||||||
|
pos += 1
|
||||||
|
|
||||||
|
# Find where the code block ends
|
||||||
|
lines = content[pos:].split("\n")
|
||||||
|
block_end = pos
|
||||||
|
for line in lines:
|
||||||
|
if line.strip(): # Non-empty line
|
||||||
|
# Check indentation level
|
||||||
|
line_indent = len(line) - len(line.lstrip())
|
||||||
|
if line_indent <= indent_len:
|
||||||
|
# The block ends when we find a line that is indented
|
||||||
|
# less than the directive itself
|
||||||
|
break
|
||||||
|
block_end += len(line) + 1 # +1 for the newline
|
||||||
|
|
||||||
|
code_block_ranges.append((start_pos, block_end))
|
||||||
|
|
||||||
|
return code_block_ranges
|
||||||
|
|
||||||
|
def _is_in_code_block(
|
||||||
|
self, match_start: int, code_block_ranges: List[Tuple[int, int]]
|
||||||
|
) -> bool:
|
||||||
|
"""Check if a match position is within a code block.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
match_start: The starting position of the match
|
||||||
|
code_block_ranges: List of (start, end) tuples for code blocks
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True if the match is within a code block, False otherwise
|
||||||
|
"""
|
||||||
|
for block_start, block_end in code_block_ranges:
|
||||||
|
if block_start <= match_start < block_end:
|
||||||
|
return True
|
||||||
|
return False
|
||||||
|
|
||||||
def _process_includes(self, content: str, source_path: Path) -> str:
|
def _process_includes(self, content: str, source_path: Path) -> str:
|
||||||
"""Process include directives in content.
|
"""Process include directives in content.
|
||||||
|
|
||||||
@@ -280,12 +359,22 @@ class DocumentProcessor:
|
|||||||
Returns:
|
Returns:
|
||||||
Processed content with include directives replaced with included content
|
Processed content with include directives replaced with included content
|
||||||
"""
|
"""
|
||||||
|
code_block_ranges = self._get_code_block_ranges(content)
|
||||||
|
|
||||||
# Find all include directives using regex
|
# Find all include directives using regex
|
||||||
include_pattern = build_directive_pattern(["include"])
|
include_pattern = build_directive_pattern(["include"])
|
||||||
|
|
||||||
# Function to replace each include with content
|
# Function to replace each include with content
|
||||||
def replace_include(match):
|
def replace_include(match):
|
||||||
|
# Check if this include is within a code block
|
||||||
|
if self._is_in_code_block(match.start(), code_block_ranges):
|
||||||
|
# This include is inside a code block, don't process it
|
||||||
|
return match.group(0)
|
||||||
|
|
||||||
include_path = match.group(3)
|
include_path = match.group(3)
|
||||||
|
directive_part = match.group(
|
||||||
|
1
|
||||||
|
) # The ".. include:: " part with leading whitespace
|
||||||
|
|
||||||
# Get all possible paths to try
|
# Get all possible paths to try
|
||||||
possible_paths = self._resolve_include_paths(include_path, source_path)
|
possible_paths = self._resolve_include_paths(include_path, source_path)
|
||||||
@@ -296,7 +385,18 @@ class DocumentProcessor:
|
|||||||
if path_to_try.exists():
|
if path_to_try.exists():
|
||||||
with open(path_to_try, "r", encoding="utf-8") as f:
|
with open(path_to_try, "r", encoding="utf-8") as f:
|
||||||
included_content = f.read()
|
included_content = f.read()
|
||||||
return included_content
|
|
||||||
|
# Find where the actual directive starts, after any whitespace
|
||||||
|
directive_start = directive_part.find("..")
|
||||||
|
if directive_start > 0:
|
||||||
|
# There's leading whitespace/newlines before the directive
|
||||||
|
leading_part = directive_part[:directive_start]
|
||||||
|
# Replace directive with content, preserving the structure
|
||||||
|
return leading_part + included_content
|
||||||
|
else:
|
||||||
|
# No leading whitespace, just return the content
|
||||||
|
return included_content
|
||||||
|
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.error(
|
logger.error(
|
||||||
f"sphinx-llms-txt: Error reading include file {path_to_try}:"
|
f"sphinx-llms-txt: Error reading include file {path_to_try}:"
|
||||||
@@ -308,8 +408,48 @@ class DocumentProcessor:
|
|||||||
paths_tried = ", ".join(str(p) for p in possible_paths)
|
paths_tried = ", ".join(str(p) for p in possible_paths)
|
||||||
logger.warning(f"sphinx-llms-txt: Include file not found: {include_path}")
|
logger.warning(f"sphinx-llms-txt: Include file not found: {include_path}")
|
||||||
logger.debug(f"sphinx-llms-txt: Tried paths: {paths_tried}")
|
logger.debug(f"sphinx-llms-txt: Tried paths: {paths_tried}")
|
||||||
return f"[Include file not found: {include_path}]"
|
|
||||||
|
# Preserve spacing structure for error message too
|
||||||
|
directive_start = match.group(1).find("..")
|
||||||
|
if directive_start > 0:
|
||||||
|
leading_part = match.group(1)[:directive_start]
|
||||||
|
return leading_part + f"[Include file not found: {include_path}]"
|
||||||
|
else:
|
||||||
|
return f"[Include file not found: {include_path}]"
|
||||||
|
|
||||||
# Replace all includes with their content
|
# Replace all includes with their content
|
||||||
processed_content = include_pattern.sub(replace_include, content)
|
processed_content = include_pattern.sub(replace_include, content)
|
||||||
return processed_content
|
return processed_content
|
||||||
|
|
||||||
|
def _process_ignore_blocks(self, content: str) -> str:
|
||||||
|
"""Process llms-txt-ignore-start/end blocks by removing their content.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
content: The source content to process
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Processed content with ignore blocks removed
|
||||||
|
"""
|
||||||
|
# Process ignore blocks iteratively to handle nested cases correctly
|
||||||
|
while True:
|
||||||
|
# Pattern to match ignore blocks - handles whitespace and indentation
|
||||||
|
ignore_pattern = re.compile(
|
||||||
|
r"^\s*\.\.\s+llms-txt-ignore-start\s*\n" # Start directive line
|
||||||
|
r"(.*?)" # Content to ignore (non-greedy)
|
||||||
|
r"^\s*\.\.\s+llms-txt-ignore-end\s*$", # End directive line
|
||||||
|
re.MULTILINE | re.DOTALL,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Find and remove one ignore block at a time
|
||||||
|
match = ignore_pattern.search(content)
|
||||||
|
if not match:
|
||||||
|
break
|
||||||
|
|
||||||
|
# Remove the matched block
|
||||||
|
content = content[: match.start()] + content[match.end() :]
|
||||||
|
|
||||||
|
# Clean up any extra blank lines that might be left
|
||||||
|
# Replace multiple consecutive newlines with at most 2 newlines
|
||||||
|
processed_content = re.sub(r"\n\n\n+", "\n\n", content)
|
||||||
|
|
||||||
|
return processed_content
|
||||||
|
|||||||
@@ -19,6 +19,42 @@ class FileWriter:
|
|||||||
self.outdir = outdir
|
self.outdir = outdir
|
||||||
self.app = app
|
self.app = app
|
||||||
|
|
||||||
|
def _resolve_uri_template(self, sources_dir: Path = None) -> str:
|
||||||
|
"""Resolve which URI template to use based on configuration and sources_dir.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
sources_dir: Path to _sources directory (None if not found)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
The template string to use for generating URIs
|
||||||
|
"""
|
||||||
|
# If custom template exists
|
||||||
|
custom_template = self.config.get("llms_txt_uri_template")
|
||||||
|
|
||||||
|
if custom_template:
|
||||||
|
# Validate user's template by checking for valid variable names
|
||||||
|
try:
|
||||||
|
# Try formatting with test valid values to validate syntax
|
||||||
|
test_values = {
|
||||||
|
"base_url": "http://example.com/",
|
||||||
|
"docname": "test",
|
||||||
|
"suffix": ".rst",
|
||||||
|
"sourcelink_suffix": ".txt",
|
||||||
|
}
|
||||||
|
custom_template.format(**test_values)
|
||||||
|
return custom_template
|
||||||
|
except (KeyError, ValueError) as e:
|
||||||
|
logger.warning(
|
||||||
|
f"sphinx-llms-txt: Invalid llms_txt_uri_template: {e}. "
|
||||||
|
f"Falling back to default."
|
||||||
|
)
|
||||||
|
|
||||||
|
# Else, use one of the default templates
|
||||||
|
if sources_dir:
|
||||||
|
return "{base_url}_sources/{docname}{suffix}{sourcelink_suffix}"
|
||||||
|
else:
|
||||||
|
return "{base_url}{docname}.html"
|
||||||
|
|
||||||
def write_combined_file(
|
def write_combined_file(
|
||||||
self, content_parts: List[str], output_path: Path, total_line_count: int
|
self, content_parts: List[str], output_path: Path, total_line_count: int
|
||||||
) -> bool:
|
) -> bool:
|
||||||
@@ -37,7 +73,7 @@ class FileWriter:
|
|||||||
f.write("\n".join(content_parts))
|
f.write("\n".join(content_parts))
|
||||||
|
|
||||||
logger.info(
|
logger.info(
|
||||||
f"sphinx-llms-txt: created {output_path} with {len(content_parts)}"
|
f"sphinx-llms-txt: Created {output_path} with {len(content_parts)}"
|
||||||
f" sources and {total_line_count} lines"
|
f" sources and {total_line_count} lines"
|
||||||
)
|
)
|
||||||
return True
|
return True
|
||||||
@@ -50,6 +86,7 @@ class FileWriter:
|
|||||||
page_order: Union[List[str], List[Tuple[str, str]]],
|
page_order: Union[List[str], List[Tuple[str, str]]],
|
||||||
page_titles: Dict[str, str],
|
page_titles: Dict[str, str],
|
||||||
total_line_count: int = 0,
|
total_line_count: int = 0,
|
||||||
|
sources_dir: Path = None,
|
||||||
) -> bool:
|
) -> bool:
|
||||||
"""Write summary information to the llms.txt file.
|
"""Write summary information to the llms.txt file.
|
||||||
|
|
||||||
@@ -57,6 +94,7 @@ class FileWriter:
|
|||||||
page_order: Ordered list of document names or (docname, suffix) tuples
|
page_order: Ordered list of document names or (docname, suffix) tuples
|
||||||
page_titles: Dictionary mapping docnames to titles
|
page_titles: Dictionary mapping docnames to titles
|
||||||
total_line_count: Total number of lines in the combined content
|
total_line_count: Total number of lines in the combined content
|
||||||
|
sources_dir: Path to _sources directory (None if not found)
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
True if successful, False otherwise
|
True if successful, False otherwise
|
||||||
@@ -102,14 +140,37 @@ class FileWriter:
|
|||||||
if not base_url.endswith("/"):
|
if not base_url.endswith("/"):
|
||||||
base_url += "/"
|
base_url += "/"
|
||||||
|
|
||||||
|
# Get sourcelink suffix from Sphinx config
|
||||||
|
sourcelink_suffix = ""
|
||||||
|
if self.app and hasattr(self.app.config, "html_sourcelink_suffix"):
|
||||||
|
sourcelink_suffix = self.app.config.html_sourcelink_suffix
|
||||||
|
# Handle empty string case specially
|
||||||
|
if sourcelink_suffix == "":
|
||||||
|
sourcelink_suffix = "" # Keep it empty
|
||||||
|
elif not sourcelink_suffix.startswith("."):
|
||||||
|
sourcelink_suffix = "." + sourcelink_suffix
|
||||||
|
|
||||||
|
# Resolve which template to use
|
||||||
|
uri_template = self._resolve_uri_template(sources_dir)
|
||||||
|
|
||||||
for item in page_order:
|
for item in page_order:
|
||||||
# Handle both old format (str) and new format (tuple)
|
# Handle both old format (str) and new format (tuple)
|
||||||
if isinstance(item, tuple):
|
if isinstance(item, tuple):
|
||||||
docname, _ = item
|
docname, suffix = item
|
||||||
else:
|
else:
|
||||||
docname = item
|
docname = item
|
||||||
|
suffix = None
|
||||||
|
|
||||||
title = page_titles.get(docname, docname)
|
title = page_titles.get(docname, docname)
|
||||||
f.write(f"- [{title}]({base_url}{docname}.html)\n")
|
|
||||||
|
uri = uri_template.format(
|
||||||
|
base_url=base_url,
|
||||||
|
docname=docname,
|
||||||
|
suffix=suffix or "",
|
||||||
|
sourcelink_suffix=sourcelink_suffix,
|
||||||
|
)
|
||||||
|
|
||||||
|
f.write(f"- [{title}]({uri})\n")
|
||||||
|
|
||||||
logger.info(f"sphinx-llms-txt: created {output_path}")
|
logger.info(f"sphinx-llms-txt: created {output_path}")
|
||||||
return True
|
return True
|
||||||
|
|||||||
@@ -8,6 +8,8 @@ Welcome to Test Project's documentation!
|
|||||||
page1
|
page1
|
||||||
page2
|
page2
|
||||||
page_with_include
|
page_with_include
|
||||||
|
page_ignored_metadata
|
||||||
|
page_with_ignore_blocks
|
||||||
|
|
||||||
Indices and tables
|
Indices and tables
|
||||||
==================
|
==================
|
||||||
|
|||||||
@@ -0,0 +1,16 @@
|
|||||||
|
:llms-txt-ignore: true
|
||||||
|
|
||||||
|
Page Ignored by Metadata
|
||||||
|
========================
|
||||||
|
|
||||||
|
This page should not appear in llms-full.txt because of the metadata directive.
|
||||||
|
|
||||||
|
Section 1
|
||||||
|
---------
|
||||||
|
|
||||||
|
This content should be completely ignored.
|
||||||
|
|
||||||
|
Section 2
|
||||||
|
---------
|
||||||
|
|
||||||
|
This content should also be ignored.
|
||||||
@@ -0,0 +1,39 @@
|
|||||||
|
Page With Ignore Blocks
|
||||||
|
=======================
|
||||||
|
|
||||||
|
This content should appear in llms-full.txt.
|
||||||
|
|
||||||
|
.. llms-txt-ignore-start
|
||||||
|
|
||||||
|
This content should be ignored and not appear in llms-full.txt.
|
||||||
|
|
||||||
|
Section Ignored
|
||||||
|
---------------
|
||||||
|
|
||||||
|
This section should also be ignored.
|
||||||
|
|
||||||
|
.. llms-txt-ignore-end
|
||||||
|
|
||||||
|
This content after the ignore block should appear in llms-full.txt.
|
||||||
|
|
||||||
|
Another Section
|
||||||
|
---------------
|
||||||
|
|
||||||
|
This content should definitely appear.
|
||||||
|
|
||||||
|
.. llms-txt-ignore-start
|
||||||
|
|
||||||
|
Another ignored block with multiple lines.
|
||||||
|
|
||||||
|
- Item 1 (ignored)
|
||||||
|
- Item 2 (ignored)
|
||||||
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# This code should be ignored
|
||||||
|
def ignored_function():
|
||||||
|
pass
|
||||||
|
|
||||||
|
.. llms-txt-ignore-end
|
||||||
|
|
||||||
|
Final content that should appear.
|
||||||
@@ -0,0 +1,220 @@
|
|||||||
|
"""Tests for llms-txt ignore features."""
|
||||||
|
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from sphinx_llms_txt import DocumentProcessor
|
||||||
|
|
||||||
|
|
||||||
|
def test_process_ignore_blocks():
|
||||||
|
"""Test that ignore blocks are properly removed from content."""
|
||||||
|
processor = DocumentProcessor({}, None)
|
||||||
|
|
||||||
|
content = """This content should remain.
|
||||||
|
|
||||||
|
.. llms-txt-ignore-start
|
||||||
|
|
||||||
|
This content should be removed.
|
||||||
|
|
||||||
|
Section Ignored
|
||||||
|
---------------
|
||||||
|
|
||||||
|
This section should also be removed.
|
||||||
|
|
||||||
|
.. llms-txt-ignore-end
|
||||||
|
|
||||||
|
This content should remain after the ignore block.
|
||||||
|
|
||||||
|
.. llms-txt-ignore-start
|
||||||
|
|
||||||
|
Another ignored block.
|
||||||
|
Multiple lines here.
|
||||||
|
|
||||||
|
.. llms-txt-ignore-end
|
||||||
|
|
||||||
|
Final content that should remain."""
|
||||||
|
|
||||||
|
processed = processor._process_ignore_blocks(content)
|
||||||
|
|
||||||
|
# Check that ignored content is removed
|
||||||
|
assert "This content should be removed." not in processed
|
||||||
|
assert "Section Ignored" not in processed
|
||||||
|
assert "Another ignored block." not in processed
|
||||||
|
assert "Multiple lines here." not in processed
|
||||||
|
|
||||||
|
# Check that non-ignored content remains
|
||||||
|
assert "This content should remain." in processed
|
||||||
|
assert "This content should remain after the ignore block." in processed
|
||||||
|
assert "Final content that should remain." in processed
|
||||||
|
|
||||||
|
|
||||||
|
def test_process_ignore_blocks_with_indentation():
|
||||||
|
"""Test that ignore blocks work with different indentation levels."""
|
||||||
|
processor = DocumentProcessor({}, None)
|
||||||
|
|
||||||
|
content = """Section Title
|
||||||
|
=============
|
||||||
|
|
||||||
|
Normal content.
|
||||||
|
|
||||||
|
.. llms-txt-ignore-start
|
||||||
|
|
||||||
|
Indented ignored content.
|
||||||
|
More indented content.
|
||||||
|
|
||||||
|
.. llms-txt-ignore-end
|
||||||
|
|
||||||
|
Back to normal content."""
|
||||||
|
|
||||||
|
processed = processor._process_ignore_blocks(content)
|
||||||
|
|
||||||
|
# Check that ignored content is removed
|
||||||
|
assert "Indented ignored content." not in processed
|
||||||
|
assert "More indented content." not in processed
|
||||||
|
|
||||||
|
# Check that non-ignored content remains
|
||||||
|
assert "Section Title" in processed
|
||||||
|
assert "Normal content." in processed
|
||||||
|
assert "Back to normal content." in processed
|
||||||
|
|
||||||
|
|
||||||
|
def test_process_ignore_blocks_multiple():
|
||||||
|
"""Test that multiple ignore blocks are handled correctly."""
|
||||||
|
processor = DocumentProcessor({}, None)
|
||||||
|
|
||||||
|
content = """Start content.
|
||||||
|
|
||||||
|
.. llms-txt-ignore-start
|
||||||
|
|
||||||
|
First ignore block.
|
||||||
|
|
||||||
|
.. llms-txt-ignore-end
|
||||||
|
|
||||||
|
Middle content that should remain.
|
||||||
|
|
||||||
|
.. llms-txt-ignore-start
|
||||||
|
|
||||||
|
Second ignore block.
|
||||||
|
|
||||||
|
.. llms-txt-ignore-end
|
||||||
|
|
||||||
|
End content."""
|
||||||
|
|
||||||
|
processed = processor._process_ignore_blocks(content)
|
||||||
|
|
||||||
|
# Check that ignored content is removed
|
||||||
|
assert "First ignore block." not in processed
|
||||||
|
assert "Second ignore block." not in processed
|
||||||
|
|
||||||
|
# Check that non-ignored content remains
|
||||||
|
assert "Start content." in processed
|
||||||
|
assert "Middle content that should remain." in processed
|
||||||
|
assert "End content." in processed
|
||||||
|
|
||||||
|
|
||||||
|
def test_build_with_ignore_features(basic_sphinx_app):
|
||||||
|
"""Test building HTML documentation with ignore features."""
|
||||||
|
app = basic_sphinx_app
|
||||||
|
app.build()
|
||||||
|
|
||||||
|
# Check if the output file was created
|
||||||
|
output_file = Path(app.outdir) / "test-llms-full.txt"
|
||||||
|
assert output_file.exists(), f"Output file {output_file} does not exist"
|
||||||
|
|
||||||
|
# Read the content of the output file
|
||||||
|
content = output_file.read_text()
|
||||||
|
|
||||||
|
# Check that page with metadata ignore is completely excluded
|
||||||
|
assert "Page Ignored by Metadata" not in content
|
||||||
|
assert "This page should not appear in llms-full.txt" not in content
|
||||||
|
|
||||||
|
# Check that page with ignore blocks has the right content
|
||||||
|
assert "Page With Ignore Blocks" in content
|
||||||
|
assert "This content should appear in llms-full.txt." in content
|
||||||
|
assert "This content after the ignore block should appear" in content
|
||||||
|
assert "Another Section" in content
|
||||||
|
assert "Final content that should appear." in content
|
||||||
|
|
||||||
|
# Check that ignored block content is not present
|
||||||
|
assert "This content should be ignored and not appear" not in content
|
||||||
|
assert "Section Ignored" not in content
|
||||||
|
assert "Another ignored block with multiple lines." not in content
|
||||||
|
assert "Item 1 (ignored)" not in content
|
||||||
|
assert "def ignored_function():" not in content
|
||||||
|
|
||||||
|
|
||||||
|
def test_manager_mark_page_ignored():
|
||||||
|
"""Test that manager can mark pages as ignored."""
|
||||||
|
from sphinx_llms_txt import LLMSFullManager
|
||||||
|
|
||||||
|
manager = LLMSFullManager()
|
||||||
|
|
||||||
|
# Initially no pages are ignored
|
||||||
|
assert len(manager.ignored_pages) == 0
|
||||||
|
|
||||||
|
# Mark a page as ignored
|
||||||
|
manager.mark_page_ignored("test_page")
|
||||||
|
|
||||||
|
# Check that page is in ignored set
|
||||||
|
assert "test_page" in manager.ignored_pages
|
||||||
|
assert len(manager.ignored_pages) == 1
|
||||||
|
|
||||||
|
# Mark another page as ignored
|
||||||
|
manager.mark_page_ignored("another_page")
|
||||||
|
|
||||||
|
# Check both pages are ignored
|
||||||
|
assert "test_page" in manager.ignored_pages
|
||||||
|
assert "another_page" in manager.ignored_pages
|
||||||
|
assert len(manager.ignored_pages) == 2
|
||||||
|
|
||||||
|
|
||||||
|
def test_process_ignore_blocks_empty_blocks():
|
||||||
|
"""Test that empty ignore blocks are handled correctly."""
|
||||||
|
processor = DocumentProcessor({}, None)
|
||||||
|
|
||||||
|
content = """Content before.
|
||||||
|
|
||||||
|
.. llms-txt-ignore-start
|
||||||
|
|
||||||
|
.. llms-txt-ignore-end
|
||||||
|
|
||||||
|
Content after."""
|
||||||
|
|
||||||
|
processed = processor._process_ignore_blocks(content)
|
||||||
|
|
||||||
|
# Check that content remains
|
||||||
|
assert "Content before." in processed
|
||||||
|
assert "Content after." in processed
|
||||||
|
|
||||||
|
# Check that we don't have excessive newlines
|
||||||
|
lines = processed.strip().split("\n")
|
||||||
|
non_empty_lines = [line for line in lines if line.strip()]
|
||||||
|
assert len(non_empty_lines) == 2
|
||||||
|
|
||||||
|
|
||||||
|
def test_ignore_metadata_affects_both_files(basic_sphinx_app):
|
||||||
|
"""Test that :llms-txt-ignore: true affects both files."""
|
||||||
|
app = basic_sphinx_app
|
||||||
|
# Enable both llms.txt and llms-full.txt file generation
|
||||||
|
app.config.llms_txt_file = True
|
||||||
|
app.config.llms_txt_filename = "test-llms.txt"
|
||||||
|
app.build()
|
||||||
|
|
||||||
|
# Check if both output files were created
|
||||||
|
llms_full_file = Path(app.outdir) / "test-llms-full.txt"
|
||||||
|
llms_summary_file = Path(app.outdir) / "test-llms.txt"
|
||||||
|
|
||||||
|
assert llms_full_file.exists(), f"Output file {llms_full_file} does not exist"
|
||||||
|
assert llms_summary_file.exists(), f"Output file {llms_summary_file} does not exist"
|
||||||
|
|
||||||
|
# Read the content of both files
|
||||||
|
llms_full_content = llms_full_file.read_text()
|
||||||
|
llms_summary_content = llms_summary_file.read_text()
|
||||||
|
|
||||||
|
# Check that page with metadata ignore is excluded from llms-full.txt
|
||||||
|
assert "Page Ignored by Metadata" not in llms_full_content
|
||||||
|
assert "This page should not appear in llms-full.txt" not in llms_full_content
|
||||||
|
|
||||||
|
# Check that page with metadata ignore is also excluded from llms.txt
|
||||||
|
# This should NOT contain a link to the ignored page
|
||||||
|
assert "Page Ignored by Metadata" not in llms_summary_content
|
||||||
|
assert "page_ignored_metadata.html" not in llms_summary_content
|
||||||
@@ -117,6 +117,152 @@ def test_max_lines_limit(temp_dir, rootdir):
|
|||||||
app.docutils_conf_path.unlink()
|
app.docutils_conf_path.unlink()
|
||||||
|
|
||||||
|
|
||||||
|
def test_on_exceed_skip(temp_dir, rootdir):
|
||||||
|
"""Test that skip action works when size limit is exceeded."""
|
||||||
|
from sphinx.testing.util import SphinxTestApp
|
||||||
|
|
||||||
|
src_dir = rootdir / "basic"
|
||||||
|
|
||||||
|
app = SphinxTestApp(
|
||||||
|
srcdir=src_dir,
|
||||||
|
builddir=temp_dir,
|
||||||
|
buildername="html",
|
||||||
|
freshenv=True,
|
||||||
|
confoverrides={
|
||||||
|
"llms_txt_full_filename": "skip-test.txt",
|
||||||
|
"llms_txt_full_max_size": 20,
|
||||||
|
"llms_txt_full_size_policy": "warn_skip",
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
app.build()
|
||||||
|
|
||||||
|
# Check that the output file was NOT created
|
||||||
|
output_file = Path(app.outdir) / "skip-test.txt"
|
||||||
|
assert (
|
||||||
|
not output_file.exists()
|
||||||
|
), f"Output file {output_file} should not exist with skip action"
|
||||||
|
|
||||||
|
# Cleanup
|
||||||
|
sys.path[:] = app._saved_path
|
||||||
|
_clean_up_global_state()
|
||||||
|
if hasattr(app, "docutils_conf_path") and app.docutils_conf_path.exists():
|
||||||
|
app.docutils_conf_path.unlink()
|
||||||
|
|
||||||
|
|
||||||
|
def test_on_exceed_keep(temp_dir, rootdir):
|
||||||
|
"""Test that keep action works when size limit is exceeded."""
|
||||||
|
from sphinx.testing.util import SphinxTestApp
|
||||||
|
|
||||||
|
src_dir = rootdir / "basic"
|
||||||
|
|
||||||
|
app = SphinxTestApp(
|
||||||
|
srcdir=src_dir,
|
||||||
|
builddir=temp_dir,
|
||||||
|
buildername="html",
|
||||||
|
freshenv=True,
|
||||||
|
confoverrides={
|
||||||
|
"llms_txt_full_filename": "keep-test.txt",
|
||||||
|
"llms_txt_full_max_size": 20,
|
||||||
|
"llms_txt_full_size_policy": "info_keep",
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
app.build()
|
||||||
|
|
||||||
|
# Check that the output file WAS created despite exceeding limit
|
||||||
|
output_file = Path(app.outdir) / "keep-test.txt"
|
||||||
|
assert (
|
||||||
|
output_file.exists()
|
||||||
|
), f"Output file {output_file} should exist with keep action"
|
||||||
|
|
||||||
|
# Verify it has content
|
||||||
|
content = output_file.read_text()
|
||||||
|
assert len(content) > 0, "Output file should have content with keep action"
|
||||||
|
|
||||||
|
# Cleanup
|
||||||
|
sys.path[:] = app._saved_path
|
||||||
|
_clean_up_global_state()
|
||||||
|
if hasattr(app, "docutils_conf_path") and app.docutils_conf_path.exists():
|
||||||
|
app.docutils_conf_path.unlink()
|
||||||
|
|
||||||
|
|
||||||
|
def test_on_exceed_note(temp_dir, rootdir):
|
||||||
|
"""Test that note action works when size limit is exceeded."""
|
||||||
|
from sphinx.testing.util import SphinxTestApp
|
||||||
|
|
||||||
|
src_dir = rootdir / "basic"
|
||||||
|
|
||||||
|
app = SphinxTestApp(
|
||||||
|
srcdir=src_dir,
|
||||||
|
builddir=temp_dir,
|
||||||
|
buildername="html",
|
||||||
|
freshenv=True,
|
||||||
|
confoverrides={
|
||||||
|
"llms_txt_full_filename": "note-test.txt",
|
||||||
|
"llms_txt_full_max_size": 20,
|
||||||
|
"llms_txt_full_size_policy": "warn_note",
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
app.build()
|
||||||
|
|
||||||
|
# Check that the output file WAS created with placeholder content
|
||||||
|
output_file = Path(app.outdir) / "note-test.txt"
|
||||||
|
assert (
|
||||||
|
output_file.exists()
|
||||||
|
), f"Output file {output_file} should exist with note action"
|
||||||
|
|
||||||
|
# Verify it has the placeholder content
|
||||||
|
content = output_file.read_text()
|
||||||
|
assert (
|
||||||
|
"This file was not generated because it exceeded the configured size limit."
|
||||||
|
in content
|
||||||
|
)
|
||||||
|
assert "llms_txt_full_max_size" in content
|
||||||
|
assert "llms_txt_full_size_policy" in content
|
||||||
|
assert "Configured max size: 20 lines" in content
|
||||||
|
|
||||||
|
# Cleanup
|
||||||
|
sys.path[:] = app._saved_path
|
||||||
|
_clean_up_global_state()
|
||||||
|
if hasattr(app, "docutils_conf_path") and app.docutils_conf_path.exists():
|
||||||
|
app.docutils_conf_path.unlink()
|
||||||
|
|
||||||
|
|
||||||
|
def test_on_exceed_invalid_config(temp_dir, rootdir):
|
||||||
|
"""Test behavior with invalid configuration values."""
|
||||||
|
from sphinx.testing.util import SphinxTestApp
|
||||||
|
|
||||||
|
src_dir = rootdir / "basic"
|
||||||
|
|
||||||
|
app = SphinxTestApp(
|
||||||
|
srcdir=src_dir,
|
||||||
|
builddir=temp_dir,
|
||||||
|
buildername="html",
|
||||||
|
freshenv=True,
|
||||||
|
confoverrides={
|
||||||
|
"llms_txt_full_filename": "invalid-test.txt",
|
||||||
|
"llms_txt_full_max_size": 20,
|
||||||
|
"llms_txt_full_size_policy": "invalid_format", # Invalid config
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
app.build()
|
||||||
|
|
||||||
|
# Should fall back to default behavior (warn_skip)
|
||||||
|
output_file = Path(app.outdir) / "invalid-test.txt"
|
||||||
|
assert (
|
||||||
|
not output_file.exists()
|
||||||
|
), f"Output file {output_file} should not exist with invalid config fallback"
|
||||||
|
|
||||||
|
# Cleanup
|
||||||
|
sys.path[:] = app._saved_path
|
||||||
|
_clean_up_global_state()
|
||||||
|
if hasattr(app, "docutils_conf_path") and app.docutils_conf_path.exists():
|
||||||
|
app.docutils_conf_path.unlink()
|
||||||
|
|
||||||
|
|
||||||
def test_title_override(temp_dir, rootdir):
|
def test_title_override(temp_dir, rootdir):
|
||||||
"""Test that the title override works correctly."""
|
"""Test that the title override works correctly."""
|
||||||
from sphinx.testing.util import SphinxTestApp
|
from sphinx.testing.util import SphinxTestApp
|
||||||
|
|||||||
@@ -41,6 +41,42 @@ def test_setup_returns_valid_dict():
|
|||||||
assert "parallel_write_safe" in result
|
assert "parallel_write_safe" in result
|
||||||
|
|
||||||
|
|
||||||
|
def test_builder_inited_with_disallowed_builder():
|
||||||
|
"""Test that disallowed builders do not trigger extension setup."""
|
||||||
|
import sphinx_llms_txt
|
||||||
|
|
||||||
|
# Reset global state
|
||||||
|
sphinx_llms_txt._manager = sphinx_llms_txt.LLMSFullManager()
|
||||||
|
sphinx_llms_txt._root_first_paragraph = ""
|
||||||
|
|
||||||
|
# Mock a Sphinx app with a disallowed builder
|
||||||
|
class MockBuilder:
|
||||||
|
name = "text" # Not in allowed list
|
||||||
|
|
||||||
|
class MockApp:
|
||||||
|
def __init__(self):
|
||||||
|
self.config_values = {}
|
||||||
|
self.connections = {}
|
||||||
|
self.builder = MockBuilder()
|
||||||
|
|
||||||
|
def add_config_value(self, name, default, rebuild):
|
||||||
|
self.config_values[name] = (default, rebuild)
|
||||||
|
|
||||||
|
def connect(self, event, handler):
|
||||||
|
self.connections[event] = handler
|
||||||
|
|
||||||
|
app = MockApp()
|
||||||
|
setup(app)
|
||||||
|
|
||||||
|
# Trigger builder-inited
|
||||||
|
builder_inited_handler = app.connections["builder-inited"]
|
||||||
|
builder_inited_handler(app)
|
||||||
|
|
||||||
|
# With disallowed builder, other events should NOT be connected
|
||||||
|
assert "doctree-resolved" not in app.connections
|
||||||
|
assert "build-finished" not in app.connections
|
||||||
|
|
||||||
|
|
||||||
def test_document_collector_initialization():
|
def test_document_collector_initialization():
|
||||||
"""Test initialization of DocumentCollector."""
|
"""Test initialization of DocumentCollector."""
|
||||||
collector = DocumentCollector()
|
collector = DocumentCollector()
|
||||||
@@ -162,6 +198,31 @@ def test_process_includes(tmp_path):
|
|||||||
assert processed_content == expected_content
|
assert processed_content == expected_content
|
||||||
|
|
||||||
|
|
||||||
|
def test_process_includes_in_code_block(tmp_path):
|
||||||
|
"""Test that an `include` within a `code-block` is not processed."""
|
||||||
|
# Create a processor
|
||||||
|
config = {"llms_txt_directives": []}
|
||||||
|
processor = DocumentProcessor(config)
|
||||||
|
|
||||||
|
# Create a source file that uses include syntax within a `code-block`
|
||||||
|
source_content = (
|
||||||
|
"Normal paragraph.\n\n"
|
||||||
|
".. code-block:: rst\n\n"
|
||||||
|
" .. include:: foo.txt\n\n"
|
||||||
|
"Another normal paragraph."
|
||||||
|
)
|
||||||
|
source_file = tmp_path / "source.txt"
|
||||||
|
with open(source_file, "w", encoding="utf-8") as f:
|
||||||
|
f.write(source_content)
|
||||||
|
|
||||||
|
# Run the include directive processor
|
||||||
|
processed_content = processor._process_includes(source_content, source_file)
|
||||||
|
|
||||||
|
# Check that the include directive was not processed
|
||||||
|
expected_content = source_content
|
||||||
|
assert processed_content == expected_content
|
||||||
|
|
||||||
|
|
||||||
def test_process_includes_with_relative_paths(tmp_path):
|
def test_process_includes_with_relative_paths(tmp_path):
|
||||||
"""Test that include directives with relative paths are processed correctly."""
|
"""Test that include directives with relative paths are processed correctly."""
|
||||||
# Create a processor
|
# Create a processor
|
||||||
@@ -758,12 +819,16 @@ def test_summary_default_uses_first_paragraph():
|
|||||||
llms_txt_summary = None # Not configured
|
llms_txt_summary = None # Not configured
|
||||||
llms_txt_file = True
|
llms_txt_file = True
|
||||||
llms_txt_filename = "llms.txt"
|
llms_txt_filename = "llms.txt"
|
||||||
|
llms_txt_uri_template = None
|
||||||
llms_txt_title = None
|
llms_txt_title = None
|
||||||
llms_txt_full_file = True
|
llms_txt_full_file = True
|
||||||
llms_txt_full_filename = "llms-full.txt"
|
llms_txt_full_filename = "llms-full.txt"
|
||||||
llms_txt_full_max_size = None
|
llms_txt_full_max_size = None
|
||||||
|
llms_txt_full_size_policy = "warn_skip"
|
||||||
llms_txt_directives = []
|
llms_txt_directives = []
|
||||||
llms_txt_exclude = []
|
llms_txt_exclude = []
|
||||||
|
llms_txt_code_files = []
|
||||||
|
llms_txt_code_base_path = None
|
||||||
html_baseurl = ""
|
html_baseurl = ""
|
||||||
|
|
||||||
config = Config()
|
config = Config()
|
||||||
@@ -808,3 +873,312 @@ def test_summary_default_uses_first_paragraph():
|
|||||||
|
|
||||||
# Restore original method
|
# Restore original method
|
||||||
sphinx_llms_txt._manager.combine_sources = original_combine_sources
|
sphinx_llms_txt._manager.combine_sources = original_combine_sources
|
||||||
|
|
||||||
|
|
||||||
|
def test_code_files_include_exclude_patterns(tmp_path):
|
||||||
|
"""Test the +/- pattern syntax for llms_txt_code_files configuration."""
|
||||||
|
from sphinx_llms_txt.manager import LLMSFullManager
|
||||||
|
|
||||||
|
# Create test directory structure
|
||||||
|
src_dir = tmp_path / "src"
|
||||||
|
src_dir.mkdir()
|
||||||
|
|
||||||
|
docs_dir = src_dir / "docs"
|
||||||
|
docs_dir.mkdir()
|
||||||
|
|
||||||
|
cache_dir = docs_dir / "__pycache__"
|
||||||
|
cache_dir.mkdir()
|
||||||
|
|
||||||
|
# Create test files
|
||||||
|
(docs_dir / "example.rst").write_text("Example RST content")
|
||||||
|
(docs_dir / "guide.rst").write_text("Guide RST content")
|
||||||
|
(docs_dir / "backup.bak").write_text("Backup file content")
|
||||||
|
(cache_dir / "compiled.pyc").write_text("Compiled Python")
|
||||||
|
|
||||||
|
# Create manager and set source directory
|
||||||
|
manager = LLMSFullManager()
|
||||||
|
manager.srcdir = str(src_dir)
|
||||||
|
|
||||||
|
# Test configuration with include/exclude patterns
|
||||||
|
config = {
|
||||||
|
"llms_txt_code_files": [
|
||||||
|
"+:docs/**/*.rst", # Include all RST files in docs
|
||||||
|
"-:docs/**/__pycache__/**", # Exclude pycache files
|
||||||
|
"-:docs/**/*.bak", # Exclude backup files
|
||||||
|
]
|
||||||
|
}
|
||||||
|
manager.set_config(config)
|
||||||
|
|
||||||
|
# Process code files
|
||||||
|
code_parts, _ = manager._process_code_files()
|
||||||
|
|
||||||
|
# Verify we have the expected number of files
|
||||||
|
assert len(code_parts) == 2, f"Expected 2 files, got {len(code_parts)}"
|
||||||
|
|
||||||
|
# Extract file titles from code blocks
|
||||||
|
titles = []
|
||||||
|
for part in code_parts:
|
||||||
|
lines = part.strip().split("\n")
|
||||||
|
if lines:
|
||||||
|
titles.append(lines[0])
|
||||||
|
|
||||||
|
# Verify expected files are included
|
||||||
|
assert "docs/example.rst" in titles
|
||||||
|
assert "docs/guide.rst" in titles
|
||||||
|
|
||||||
|
# Verify excluded files are not present
|
||||||
|
content = "\n".join(code_parts)
|
||||||
|
assert "backup.bak" not in content
|
||||||
|
assert "__pycache__" not in content
|
||||||
|
assert "compiled.pyc" not in content
|
||||||
|
|
||||||
|
|
||||||
|
def test_code_files_exclude_only_patterns(tmp_path):
|
||||||
|
"""Test that exclude-only patterns result in no files being included."""
|
||||||
|
from sphinx_llms_txt.manager import LLMSFullManager
|
||||||
|
|
||||||
|
# Create test directory structure
|
||||||
|
src_dir = tmp_path / "src"
|
||||||
|
src_dir.mkdir()
|
||||||
|
|
||||||
|
docs_dir = src_dir / "docs"
|
||||||
|
docs_dir.mkdir()
|
||||||
|
|
||||||
|
# Create test files
|
||||||
|
(docs_dir / "example.rst").write_text("Example RST content")
|
||||||
|
|
||||||
|
# Create manager and set source directory
|
||||||
|
manager = LLMSFullManager()
|
||||||
|
manager.srcdir = str(src_dir)
|
||||||
|
|
||||||
|
# Test configuration with only exclude patterns
|
||||||
|
config = {
|
||||||
|
"llms_txt_code_files": [
|
||||||
|
"-:docs/**/*.rst", # Only exclude pattern, no includes
|
||||||
|
]
|
||||||
|
}
|
||||||
|
manager.set_config(config)
|
||||||
|
|
||||||
|
# Process code files
|
||||||
|
code_parts, _ = manager._process_code_files()
|
||||||
|
|
||||||
|
# Should have no files with exclude-only patterns
|
||||||
|
assert len(code_parts) == 0, "Should have no files with exclude-only patterns"
|
||||||
|
|
||||||
|
|
||||||
|
def test_code_files_no_prefix_patterns(tmp_path):
|
||||||
|
"""Test that patterns without prefix are ignored."""
|
||||||
|
from sphinx_llms_txt.manager import LLMSFullManager
|
||||||
|
|
||||||
|
# Create test directory structure
|
||||||
|
src_dir = tmp_path / "src"
|
||||||
|
src_dir.mkdir()
|
||||||
|
|
||||||
|
docs_dir = src_dir / "docs"
|
||||||
|
docs_dir.mkdir()
|
||||||
|
|
||||||
|
# Create test files
|
||||||
|
(docs_dir / "example.rst").write_text("Example RST content")
|
||||||
|
(docs_dir / "backup.bak").write_text("Backup file content")
|
||||||
|
|
||||||
|
# Create manager and set source directory
|
||||||
|
manager = LLMSFullManager()
|
||||||
|
manager.srcdir = str(src_dir)
|
||||||
|
|
||||||
|
# Test configuration with no prefix (should be ignored)
|
||||||
|
config = {
|
||||||
|
"llms_txt_code_files": [
|
||||||
|
"docs/**/*.rst", # No prefix = ignored
|
||||||
|
"+:docs/**/*.rst", # Include RST files
|
||||||
|
"-:docs/**/*.bak", # Exclude backup files
|
||||||
|
]
|
||||||
|
}
|
||||||
|
manager.set_config(config)
|
||||||
|
|
||||||
|
# Process code files
|
||||||
|
code_parts, _ = manager._process_code_files()
|
||||||
|
|
||||||
|
# Should include RST files (from +: pattern) and exclude BAK files (from -: pattern)
|
||||||
|
assert len(code_parts) == 1, "Should include RST files and exclude BAK files"
|
||||||
|
|
||||||
|
content = "\n".join(code_parts)
|
||||||
|
assert "Example RST content" in content
|
||||||
|
assert "backup.bak" not in content
|
||||||
|
|
||||||
|
|
||||||
|
def test_code_files_ignored_patterns(tmp_path, caplog):
|
||||||
|
"""Test that patterns without +: or -: prefix log a warning and are ignored."""
|
||||||
|
from unittest.mock import patch
|
||||||
|
|
||||||
|
from sphinx_llms_txt.manager import LLMSFullManager
|
||||||
|
|
||||||
|
# Create test directory structure
|
||||||
|
src_dir = tmp_path / "src"
|
||||||
|
src_dir.mkdir()
|
||||||
|
|
||||||
|
docs_dir = src_dir / "docs"
|
||||||
|
docs_dir.mkdir()
|
||||||
|
|
||||||
|
# Create test files
|
||||||
|
(docs_dir / "example.rst").write_text("Example RST content")
|
||||||
|
|
||||||
|
# Use a mock to capture the warning message directly
|
||||||
|
captured_warnings = []
|
||||||
|
|
||||||
|
def capture_warning(message, *args, **kwargs):
|
||||||
|
captured_warnings.append(message)
|
||||||
|
|
||||||
|
# Patch the logger to capture warnings
|
||||||
|
with patch("sphinx_llms_txt.manager.logger.warning", side_effect=capture_warning):
|
||||||
|
# Create manager and set source directory
|
||||||
|
manager = LLMSFullManager()
|
||||||
|
manager.srcdir = str(src_dir)
|
||||||
|
|
||||||
|
# Test configuration with only no-prefix patterns (should result in no files)
|
||||||
|
config = {
|
||||||
|
"llms_txt_code_files": [
|
||||||
|
"docs/**/*.rst", # No prefix = ignored with warning
|
||||||
|
]
|
||||||
|
}
|
||||||
|
manager.set_config(config)
|
||||||
|
|
||||||
|
# Process code files
|
||||||
|
code_parts, _ = manager._process_code_files()
|
||||||
|
|
||||||
|
# Should have no files since the pattern without prefix is ignored
|
||||||
|
assert (
|
||||||
|
len(code_parts) == 0
|
||||||
|
), "Should have no files when only using patterns without prefix"
|
||||||
|
|
||||||
|
# Check that a warning was logged
|
||||||
|
assert (
|
||||||
|
len(captured_warnings) == 1
|
||||||
|
), f"Expected 1 warning, got {len(captured_warnings)}"
|
||||||
|
assert (
|
||||||
|
"Code file pattern 'docs/**/*.rst' ignored." in captured_warnings[0]
|
||||||
|
), f"Warning message should contain expected text. Got: {captured_warnings[0]}"
|
||||||
|
|
||||||
|
|
||||||
|
def test_llms_txt_generated_without_sources_dir(tmp_path):
|
||||||
|
"""Test that llms.txt is generated even when _sources directory doesn't exist."""
|
||||||
|
|
||||||
|
from sphinx_llms_txt.manager import LLMSFullManager
|
||||||
|
|
||||||
|
# Create manager
|
||||||
|
manager = LLMSFullManager()
|
||||||
|
|
||||||
|
# Set config to enable llms.txt
|
||||||
|
config = {
|
||||||
|
"llms_txt_file": True,
|
||||||
|
"llms_txt_filename": "llms.txt",
|
||||||
|
"llms_txt_full_file": True,
|
||||||
|
"llms_txt_full_filename": "llms-full.txt",
|
||||||
|
"llms_txt_exclude": [],
|
||||||
|
"llms_txt_directives": [],
|
||||||
|
}
|
||||||
|
manager.set_config(config)
|
||||||
|
|
||||||
|
# Create directories (but no _sources)
|
||||||
|
outdir = tmp_path / "build"
|
||||||
|
srcdir = tmp_path / "source"
|
||||||
|
outdir.mkdir()
|
||||||
|
srcdir.mkdir()
|
||||||
|
|
||||||
|
# Mock env with documents
|
||||||
|
class MockEnv:
|
||||||
|
all_docs = {"index": None, "about": None}
|
||||||
|
titles = {
|
||||||
|
"index": type("TitleNode", (), {"astext": lambda self: "Home"})(),
|
||||||
|
"about": type("TitleNode", (), {"astext": lambda self: "About"})(),
|
||||||
|
}
|
||||||
|
toctree_includes = {"index": ["about"]}
|
||||||
|
|
||||||
|
manager.set_env(MockEnv())
|
||||||
|
manager.set_master_doc("index")
|
||||||
|
|
||||||
|
# Update page titles directly in the collector
|
||||||
|
manager.update_page_title("index", "Home")
|
||||||
|
manager.update_page_title("about", "About")
|
||||||
|
|
||||||
|
# Call combine_sources - should generate llms.txt even without _sources
|
||||||
|
manager.combine_sources(str(outdir), str(srcdir))
|
||||||
|
|
||||||
|
# Verify llms.txt was created
|
||||||
|
llms_txt = outdir / "llms.txt"
|
||||||
|
assert llms_txt.exists(), "llms.txt should be generated even without _sources"
|
||||||
|
|
||||||
|
# Verify llms-full.txt was NOT created (since no _sources)
|
||||||
|
llms_full_txt = outdir / "llms-full.txt"
|
||||||
|
assert (
|
||||||
|
not llms_full_txt.exists()
|
||||||
|
), "llms-full.txt should not be generated without _sources"
|
||||||
|
|
||||||
|
# Read llms.txt and verify it has content
|
||||||
|
with open(llms_txt, "r", encoding="utf-8") as f:
|
||||||
|
content = f.read()
|
||||||
|
|
||||||
|
# Should contain page titles and links
|
||||||
|
assert "Home" in content
|
||||||
|
assert "About" in content
|
||||||
|
assert "index.html" in content
|
||||||
|
assert "about.html" in content
|
||||||
|
|
||||||
|
|
||||||
|
def test_llms_txt_no_warning_when_full_file_disabled(tmp_path, caplog):
|
||||||
|
"""
|
||||||
|
Test that no warning is logged when llms_txt_full_file=False and
|
||||||
|
_sources doesn't exist.
|
||||||
|
"""
|
||||||
|
from unittest.mock import patch
|
||||||
|
|
||||||
|
from sphinx_llms_txt.manager import LLMSFullManager
|
||||||
|
|
||||||
|
# Create manager
|
||||||
|
manager = LLMSFullManager()
|
||||||
|
|
||||||
|
# Set config with llms_txt_full_file=False
|
||||||
|
config = {
|
||||||
|
"llms_txt_file": True,
|
||||||
|
"llms_txt_filename": "llms.txt",
|
||||||
|
"llms_txt_full_file": False, # User doesn't want llms-full.txt
|
||||||
|
"llms_txt_full_filename": "llms-full.txt",
|
||||||
|
"llms_txt_exclude": [],
|
||||||
|
"llms_txt_directives": [],
|
||||||
|
}
|
||||||
|
manager.set_config(config)
|
||||||
|
|
||||||
|
# Create directories (but no _sources)
|
||||||
|
outdir = tmp_path / "build"
|
||||||
|
srcdir = tmp_path / "source"
|
||||||
|
outdir.mkdir()
|
||||||
|
srcdir.mkdir()
|
||||||
|
|
||||||
|
# Mock env with documents
|
||||||
|
class MockEnv:
|
||||||
|
all_docs = {"index": None}
|
||||||
|
titles = {"index": type("TitleNode", (), {"astext": lambda self: "Home"})()}
|
||||||
|
toctree_includes = {"index": []}
|
||||||
|
|
||||||
|
manager.set_env(MockEnv())
|
||||||
|
manager.set_master_doc("index")
|
||||||
|
manager.update_page_title("index", "Home")
|
||||||
|
|
||||||
|
# Capture warnings
|
||||||
|
captured_warnings = []
|
||||||
|
|
||||||
|
def capture_warning(message, *args, **kwargs):
|
||||||
|
if "_sources" in str(message):
|
||||||
|
captured_warnings.append(message)
|
||||||
|
|
||||||
|
with patch("sphinx_llms_txt.manager.logger.warning", side_effect=capture_warning):
|
||||||
|
# Call combine_sources
|
||||||
|
manager.combine_sources(str(outdir), str(srcdir))
|
||||||
|
|
||||||
|
# Verify NO warning was logged since llms_txt_full_file=False
|
||||||
|
assert (
|
||||||
|
len(captured_warnings) == 0
|
||||||
|
), "No warning should be logged when llms_txt_full_file=False"
|
||||||
|
|
||||||
|
# Verify llms.txt was still created
|
||||||
|
llms_txt = outdir / "llms.txt"
|
||||||
|
assert llms_txt.exists()
|
||||||
|
|||||||
@@ -0,0 +1,175 @@
|
|||||||
|
"""Test URI template functionality for llms.txt links."""
|
||||||
|
|
||||||
|
from sphinx_llms_txt import FileWriter
|
||||||
|
|
||||||
|
|
||||||
|
def test_uri_template_with_sources_dir(tmp_path):
|
||||||
|
"""Test that default template uses _sources links when sources_dir exists."""
|
||||||
|
build_dir = tmp_path / "build"
|
||||||
|
build_dir.mkdir()
|
||||||
|
|
||||||
|
# Create _sources directory to simulate its existence
|
||||||
|
sources_dir = build_dir / "_sources"
|
||||||
|
sources_dir.mkdir()
|
||||||
|
|
||||||
|
# Mock app with html_sourcelink_suffix
|
||||||
|
class MockApp:
|
||||||
|
class Config:
|
||||||
|
html_sourcelink_suffix = ".txt"
|
||||||
|
|
||||||
|
config = Config()
|
||||||
|
|
||||||
|
config = {
|
||||||
|
"llms_txt_file": True,
|
||||||
|
"llms_txt_filename": "llms.txt",
|
||||||
|
"llms_txt_uri_template": (
|
||||||
|
"{base_url}_sources/{docname}{suffix}{sourcelink_suffix}"
|
||||||
|
),
|
||||||
|
"html_baseurl": "https://example.com",
|
||||||
|
}
|
||||||
|
writer = FileWriter(config, str(build_dir), MockApp())
|
||||||
|
|
||||||
|
page_titles = {
|
||||||
|
"index": "Home Page",
|
||||||
|
"about": "About Us",
|
||||||
|
}
|
||||||
|
|
||||||
|
# Page order with suffixes (simulating _sources files exist)
|
||||||
|
page_order = [("index", ".rst"), ("about", ".md")]
|
||||||
|
|
||||||
|
writer.write_verbose_info_to_file(page_order, page_titles, 0, sources_dir)
|
||||||
|
|
||||||
|
# Check that the file was created
|
||||||
|
verbose_file = build_dir / "llms.txt"
|
||||||
|
assert verbose_file.exists()
|
||||||
|
|
||||||
|
# Read the file content
|
||||||
|
with open(verbose_file, "r", encoding="utf-8") as f:
|
||||||
|
content = f.read()
|
||||||
|
|
||||||
|
# Should link to _sources files
|
||||||
|
assert "- [Home Page](https://example.com/_sources/index.rst.txt)" in content
|
||||||
|
assert "- [About Us](https://example.com/_sources/about.md.txt)" in content
|
||||||
|
|
||||||
|
|
||||||
|
def test_uri_template_without_sources_dir(tmp_path):
|
||||||
|
"""
|
||||||
|
Test that HTML template is used when sources_dir doesn't exist and no custom
|
||||||
|
template.
|
||||||
|
"""
|
||||||
|
build_dir = tmp_path / "build"
|
||||||
|
build_dir.mkdir()
|
||||||
|
|
||||||
|
config = {
|
||||||
|
"llms_txt_file": True,
|
||||||
|
"llms_txt_filename": "llms.txt",
|
||||||
|
# No custom template set
|
||||||
|
"html_baseurl": "https://example.com",
|
||||||
|
}
|
||||||
|
writer = FileWriter(config, str(build_dir))
|
||||||
|
|
||||||
|
page_titles = {
|
||||||
|
"index": "Home Page",
|
||||||
|
"about": "About Us",
|
||||||
|
}
|
||||||
|
|
||||||
|
# Page order without suffixes (simulating no _sources)
|
||||||
|
page_order = [("index", None), ("about", None)]
|
||||||
|
|
||||||
|
# Pass None for sources_dir to simulate it doesn't exist
|
||||||
|
writer.write_verbose_info_to_file(page_order, page_titles, 0, None)
|
||||||
|
|
||||||
|
# Check that the file was created
|
||||||
|
verbose_file = build_dir / "llms.txt"
|
||||||
|
assert verbose_file.exists()
|
||||||
|
|
||||||
|
# Read the file content
|
||||||
|
with open(verbose_file, "r", encoding="utf-8") as f:
|
||||||
|
content = f.read()
|
||||||
|
|
||||||
|
# Should fallback to HTML links
|
||||||
|
assert "- [Home Page](https://example.com/index.html)" in content
|
||||||
|
assert "- [About Us](https://example.com/about.html)" in content
|
||||||
|
|
||||||
|
|
||||||
|
def test_uri_template_custom(tmp_path):
|
||||||
|
"""Test that custom URI template works correctly."""
|
||||||
|
build_dir = tmp_path / "build"
|
||||||
|
build_dir.mkdir()
|
||||||
|
|
||||||
|
sources_dir = build_dir / "_sources"
|
||||||
|
sources_dir.mkdir()
|
||||||
|
|
||||||
|
# Mock app with html_sourcelink_suffix
|
||||||
|
class MockApp:
|
||||||
|
class Config:
|
||||||
|
html_sourcelink_suffix = ".txt"
|
||||||
|
|
||||||
|
config = Config()
|
||||||
|
|
||||||
|
# Custom template that uses different path
|
||||||
|
config = {
|
||||||
|
"llms_txt_file": True,
|
||||||
|
"llms_txt_filename": "llms.txt",
|
||||||
|
"llms_txt_uri_template": "{base_url}raw/{docname}{suffix}",
|
||||||
|
"html_baseurl": "https://example.com/",
|
||||||
|
}
|
||||||
|
writer = FileWriter(config, str(build_dir), MockApp())
|
||||||
|
|
||||||
|
page_titles = {
|
||||||
|
"index": "Home Page",
|
||||||
|
}
|
||||||
|
|
||||||
|
page_order = [("index", ".rst")]
|
||||||
|
|
||||||
|
writer.write_verbose_info_to_file(page_order, page_titles, 0, sources_dir)
|
||||||
|
|
||||||
|
verbose_file = build_dir / "llms.txt"
|
||||||
|
with open(verbose_file, "r", encoding="utf-8") as f:
|
||||||
|
content = f.read()
|
||||||
|
|
||||||
|
# Should use custom template
|
||||||
|
assert "- [Home Page](https://example.com/raw/index.rst)" in content
|
||||||
|
|
||||||
|
|
||||||
|
def test_uri_template_invalid_fallback(tmp_path):
|
||||||
|
"""
|
||||||
|
Test that invalid template falls back to default sources template when
|
||||||
|
sources_dir exists.
|
||||||
|
"""
|
||||||
|
build_dir = tmp_path / "build"
|
||||||
|
build_dir.mkdir()
|
||||||
|
|
||||||
|
sources_dir = build_dir / "_sources"
|
||||||
|
sources_dir.mkdir()
|
||||||
|
|
||||||
|
# Mock app with html_sourcelink_suffix
|
||||||
|
class MockApp:
|
||||||
|
class Config:
|
||||||
|
html_sourcelink_suffix = ".txt"
|
||||||
|
|
||||||
|
config = Config()
|
||||||
|
|
||||||
|
# Invalid template with typo in variable name
|
||||||
|
config = {
|
||||||
|
"llms_txt_file": True,
|
||||||
|
"llms_txt_filename": "llms.txt",
|
||||||
|
"llms_txt_uri_template": "{base_urll}/{docname}",
|
||||||
|
"html_baseurl": "https://example.com",
|
||||||
|
}
|
||||||
|
writer = FileWriter(config, str(build_dir), MockApp())
|
||||||
|
|
||||||
|
page_titles = {
|
||||||
|
"index": "Home Page",
|
||||||
|
}
|
||||||
|
|
||||||
|
page_order = [("index", ".rst")]
|
||||||
|
|
||||||
|
writer.write_verbose_info_to_file(page_order, page_titles, 0, sources_dir)
|
||||||
|
|
||||||
|
verbose_file = build_dir / "llms.txt"
|
||||||
|
with open(verbose_file, "r", encoding="utf-8") as f:
|
||||||
|
content = f.read()
|
||||||
|
|
||||||
|
# Should fallback to default sources template
|
||||||
|
assert "- [Home Page](https://example.com/_sources/index.rst.txt)" in content
|
||||||
Reference in New Issue
Block a user