Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
0ee28290db | ||
|
|
368c349d58 | ||
|
|
c816fb1ea8 | ||
|
|
92f810592e | ||
|
|
ab0eb1dd29 | ||
|
|
57f716b2f9 | ||
|
|
236822885e | ||
|
|
46c2dec254 | ||
|
|
b56d93d265 | ||
|
|
5db5395889 | ||
|
|
04b0657dc5 | ||
|
|
935a964c7e | ||
|
|
51f6c71de3 | ||
|
|
70defd3996 | ||
|
|
9ae05c6c13 | ||
|
|
5581979cac | ||
|
|
f70f1a26ec | ||
|
|
ed50138ae4 | ||
|
|
8f4d2c07c6 | ||
|
|
da7ee68076 | ||
|
|
8ed31f13c4 | ||
|
|
cc38abc8f2 | ||
|
|
bf368670db | ||
|
|
a5cbdf15fa | ||
|
|
e661ba3da7 | ||
|
|
1c7c381c6d | ||
|
|
bca0c418d6 | ||
|
|
8d17c022ee | ||
|
|
563c5e3d9e | ||
|
|
cffac5615d | ||
|
|
77999f0923 | ||
|
|
480fd83d65 | ||
|
|
482b525fd6 | ||
|
|
dc08e4f3b0 | ||
|
|
2c8b554aa2 | ||
|
|
10e2b9a684 | ||
|
|
94a82dc68b | ||
|
|
2ca3052a2c | ||
|
|
74393bcd2a | ||
|
|
a4fcaa6531 | ||
|
|
69e251171d | ||
|
|
b51f3b099a | ||
|
|
d496591903 | ||
|
|
542fb2efb0 | ||
|
|
012674a934 | ||
|
|
19f42aa8cc | ||
|
|
afa5731d75 |
@@ -0,0 +1,13 @@
|
|||||||
|
# These are supported funding model platforms
|
||||||
|
|
||||||
|
github: [jdillard]
|
||||||
|
patreon: # Replace with a single Patreon username
|
||||||
|
open_collective: # Replace with a single Open Collective username
|
||||||
|
ko_fi: # Replace with a single Ko-fi username
|
||||||
|
tidelift: # Replace with a single Tidelift platform-name/package-name e.g., npm/babel
|
||||||
|
community_bridge: # Replace with a single Community Bridge project-name e.g., cloud-foundry
|
||||||
|
liberapay: # Replace with a single Liberapay username
|
||||||
|
issuehunt: # Replace with a single IssueHunt username
|
||||||
|
otechie: # Replace with a single Otechie username
|
||||||
|
lfx_crowdfunding: # Replace with a single LFX Crowdfunding project-name e.g., cloud-foundry
|
||||||
|
custom: # Replace with up to 4 custom sponsorship URLs e.g., ['link1', 'link2']
|
||||||
@@ -0,0 +1,17 @@
|
|||||||
|
# To get started with Dependabot version updates, you'll need to specify which
|
||||||
|
# package ecosystems to update and where the package manifests are located.
|
||||||
|
# Please see the documentation for all configuration options:
|
||||||
|
# https://docs.github.com/github/administering-a-repository/configuration-options-for-dependency-updates
|
||||||
|
|
||||||
|
version: 2
|
||||||
|
updates:
|
||||||
|
- package-ecosystem: "github-actions"
|
||||||
|
directory: "/" # Location of package manifests
|
||||||
|
schedule:
|
||||||
|
interval: "monthly"
|
||||||
|
groups:
|
||||||
|
# Name for the group, which will be used in PR titles and branch names
|
||||||
|
all-github-actions:
|
||||||
|
# Group all updates together
|
||||||
|
patterns:
|
||||||
|
- "*"
|
||||||
@@ -0,0 +1,55 @@
|
|||||||
|
name: Test and Build
|
||||||
|
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
branches: [ main ]
|
||||||
|
pull_request:
|
||||||
|
branches: [ main ]
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
pre-commit:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
- name: Set up Python 3.10
|
||||||
|
uses: actions/setup-python@v5
|
||||||
|
with:
|
||||||
|
python-version: "3.10"
|
||||||
|
- uses: pre-commit/action@v3.0.1
|
||||||
|
test:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
strategy:
|
||||||
|
matrix:
|
||||||
|
python-version: ['3.9', '3.10', '3.11', '3.12']
|
||||||
|
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Set up Python ${{ matrix.python-version }}
|
||||||
|
uses: actions/setup-python@v5
|
||||||
|
with:
|
||||||
|
python-version: ${{ matrix.python-version }}
|
||||||
|
|
||||||
|
- name: Install dependencies
|
||||||
|
run: |
|
||||||
|
python -m pip install --upgrade pip
|
||||||
|
pip install -e ".[dev]"
|
||||||
|
|
||||||
|
# - name: Run mypy
|
||||||
|
# run: |
|
||||||
|
# mypy sphinx_cmd
|
||||||
|
|
||||||
|
- name: Test with pytest
|
||||||
|
run: |
|
||||||
|
pytest
|
||||||
|
|
||||||
|
# - name: Build package
|
||||||
|
# run: |
|
||||||
|
# pip install build
|
||||||
|
# python -m build
|
||||||
|
|
||||||
|
# - name: Upload artifacts
|
||||||
|
# uses: actions/upload-artifact@v3
|
||||||
|
# with:
|
||||||
|
# name: dist-${{ matrix.python-version }}
|
||||||
|
# path: dist/
|
||||||
@@ -0,0 +1,15 @@
|
|||||||
|
version: 2
|
||||||
|
|
||||||
|
build:
|
||||||
|
os: "ubuntu-20.04"
|
||||||
|
tools:
|
||||||
|
python: "3.10"
|
||||||
|
|
||||||
|
sphinx:
|
||||||
|
configuration: docs/source/conf.py
|
||||||
|
|
||||||
|
python:
|
||||||
|
install:
|
||||||
|
- requirements: docs/requirements.txt
|
||||||
|
- method: pip
|
||||||
|
path: .
|
||||||
@@ -1,6 +1,66 @@
|
|||||||
Changelog
|
Changelog
|
||||||
=========
|
=========
|
||||||
|
|
||||||
|
0.4.0
|
||||||
|
-----
|
||||||
|
|
||||||
|
- Add support for including source code files with :confval:`llms_txt_code_files` and :confval:`llms_txt_code_base_path` configuration options
|
||||||
|
`#24 <https://github.com/jdillard/sphinx-llms-txt/pull/24>`_
|
||||||
|
|
||||||
|
0.3.2
|
||||||
|
-----
|
||||||
|
|
||||||
|
- Fix image paths to deployed images
|
||||||
|
`#30 <https://github.com/jdillard/sphinx-llms-txt/pull/30>`_
|
||||||
|
|
||||||
|
0.3.1
|
||||||
|
-----
|
||||||
|
|
||||||
|
- Fix issue when ``source_suffix`` equals ``source_link_suffix``
|
||||||
|
`#29 <https://github.com/jdillard/sphinx-llms-txt/pull/29>`_
|
||||||
|
|
||||||
|
0.3.0
|
||||||
|
-----
|
||||||
|
|
||||||
|
- Use first paragraph as default for ``llms_txt_summary``
|
||||||
|
`#22 <https://github.com/jdillard/sphinx-llms-txt/pull/22>`_
|
||||||
|
|
||||||
|
0.2.4
|
||||||
|
-----
|
||||||
|
|
||||||
|
- Support source file suffix detection
|
||||||
|
`#21 <https://github.com/jdillard/sphinx-llms-txt/pull/21>`_
|
||||||
|
|
||||||
|
0.2.3
|
||||||
|
-----
|
||||||
|
|
||||||
|
- Remove ``get_and_resolve_toctree`` method
|
||||||
|
`#19 <https://github.com/jdillard/sphinx-llms-txt/pull/19>`_
|
||||||
|
- Simplify ``_sources`` lookup
|
||||||
|
`#18 <https://github.com/jdillard/sphinx-llms-txt/pull/18>`_
|
||||||
|
- Add sphinx docs
|
||||||
|
`#16 <https://github.com/jdillard/sphinx-llms-txt/pull/16>`_
|
||||||
|
|
||||||
|
0.2.2
|
||||||
|
-----
|
||||||
|
|
||||||
|
- Refactor LLMSFullManager with clearer class structure
|
||||||
|
- Add ``html_baseurl`` to **llms.txt** docs links
|
||||||
|
- Make glob pattern recursive
|
||||||
|
|
||||||
|
0.2.1
|
||||||
|
-----
|
||||||
|
|
||||||
|
- Add ability to exclude pages with ``llms_txt_exclude``
|
||||||
|
|
||||||
|
0.2.0
|
||||||
|
-----
|
||||||
|
|
||||||
|
- Add ``llms_txt_full_max_size`` configuration option to limit `llms-full.txt` file size
|
||||||
|
- Automatically add content from **include** directives in **llms-full.txt**
|
||||||
|
- Add path resolution for a given set of directives in **llms-full.txt**
|
||||||
|
- Add **llms.txt** file option, with ``llms_txt_title`` and ``llms_txt_summary`` config values
|
||||||
|
|
||||||
0.1.0
|
0.1.0
|
||||||
-----
|
-----
|
||||||
|
|
||||||
|
|||||||
@@ -1,36 +1,15 @@
|
|||||||
# Sphinx llms-full.txt Extension
|
# Sphinx llms.txt generator
|
||||||
|
|
||||||
A Sphinx extension that creates a single combined documentation `llms-full.txt` file, written in reStructuredText.
|
A Sphinx extension that generates a summary `llms.txt` file and a single combined documentation `llms-full.txt` file.
|
||||||
|
|
||||||
## Installation
|
[](https://pypi.python.org/pypi/sphinx-llms-txt)
|
||||||
|
[](https://anaconda.org/conda-forge/sphinx-llms-txt)
|
||||||
|
[](https://pepy.tech/project/sphinx-llms-txt)
|
||||||
|
[](#)
|
||||||
|
|
||||||
```bash
|
## Documentation
|
||||||
pip install sphinx-llms-txt
|
|
||||||
```
|
|
||||||
|
|
||||||
## Usage
|
See [sphinx-llms-txt documentation](https://sphinx-llms-txt.readthedocs.io/en/latest/index.html) for installation and configuration instructions.
|
||||||
|
|
||||||
1. Add the extension to your Sphinx configuration (`conf.py`):
|
|
||||||
|
|
||||||
```python
|
|
||||||
extensions = [
|
|
||||||
'sphinx_llms_txt',
|
|
||||||
]
|
|
||||||
```
|
|
||||||
|
|
||||||
## Configuration Options
|
|
||||||
|
|
||||||
### `llms_txt_filename`
|
|
||||||
|
|
||||||
- **Type**: string
|
|
||||||
- **Default**: `'llms-full.txt'`
|
|
||||||
- **Description**: Name of the output file
|
|
||||||
|
|
||||||
### `llms_txt_verbose`
|
|
||||||
|
|
||||||
- **Type**: boolean
|
|
||||||
- **Default**: `False`
|
|
||||||
- **Description**: Whether to include a summary in the build output
|
|
||||||
|
|
||||||
## License
|
## License
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,20 @@
|
|||||||
|
# Minimal makefile for Sphinx documentation
|
||||||
|
#
|
||||||
|
|
||||||
|
# You can set these variables from the command line.
|
||||||
|
SPHINXOPTS =
|
||||||
|
SPHINXBUILD = sphinx-build
|
||||||
|
SPHINXPROJ = SphinxLLMsTxt
|
||||||
|
SOURCEDIR = source
|
||||||
|
BUILDDIR = _build
|
||||||
|
|
||||||
|
# Put it first so that "make" without argument is like "make help".
|
||||||
|
help:
|
||||||
|
@$(SPHINXBUILD) -M help "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O)
|
||||||
|
|
||||||
|
.PHONY: help Makefile
|
||||||
|
|
||||||
|
# Catch-all target: route all unknown targets to Sphinx using the new
|
||||||
|
# "make mode" option. $(O) is meant as a shortcut for $(SPHINXOPTS).
|
||||||
|
%: Makefile
|
||||||
|
@$(SPHINXBUILD) -M $@ "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O)
|
||||||
@@ -0,0 +1,6 @@
|
|||||||
|
furo
|
||||||
|
esbonio
|
||||||
|
sphinx-contributors
|
||||||
|
sphinx
|
||||||
|
sphinx-llms-txt
|
||||||
|
sphinxext-opengraph
|
||||||
@@ -0,0 +1,222 @@
|
|||||||
|
Advanced Configuration
|
||||||
|
======================
|
||||||
|
|
||||||
|
This page covers advanced configuration options for the sphinx-llms-txt extension.
|
||||||
|
|
||||||
|
.. _customizing_llms_files:
|
||||||
|
|
||||||
|
Customizing the LLMs Files
|
||||||
|
^^^^^^^^^^^^^^^^^^^^^^^^^^
|
||||||
|
|
||||||
|
By default, the extension generates two files:
|
||||||
|
|
||||||
|
1. ``llms.txt`` - A summary file in Markdown format
|
||||||
|
2. ``llms-full.txt`` - A complete documentation file in reStructuredText format
|
||||||
|
|
||||||
|
You can customize these files in several ways:
|
||||||
|
|
||||||
|
.. _changing_filenames:
|
||||||
|
|
||||||
|
Changing Filenames
|
||||||
|
~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
|
You can change the default filenames by setting these values in your ``conf.py``:
|
||||||
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
llms_txt_filename = "custom-summary.txt"
|
||||||
|
llms_txt_full_filename = "custom-docs.txt"
|
||||||
|
|
||||||
|
.. _disabling_file_generation:
|
||||||
|
|
||||||
|
Disabling File Generation
|
||||||
|
~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
|
If you only want one of the files, you can disable generation of the other:
|
||||||
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# Disable summary file
|
||||||
|
llms_txt_file = False
|
||||||
|
|
||||||
|
# Disable full documentation file
|
||||||
|
llms_txt_full_file = False
|
||||||
|
|
||||||
|
.. _custom_summary:
|
||||||
|
|
||||||
|
Adding a Custom Summary
|
||||||
|
~~~~~~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
|
The summary file can include a custom description of your project:
|
||||||
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
llms_txt_summary = """
|
||||||
|
This documentation explains how to use MyProject to build amazing
|
||||||
|
applications. The project provides a comprehensive API for handling
|
||||||
|
data processing and visualization.
|
||||||
|
"""
|
||||||
|
|
||||||
|
.. note:: The summary can span multiple lines and will be properly formatted in the output file.
|
||||||
|
|
||||||
|
.. _custom_title:
|
||||||
|
|
||||||
|
Custom Title
|
||||||
|
~~~~~~~~~~~~
|
||||||
|
|
||||||
|
By default, the project name from Sphinx is used as the title in ``llms.txt``. You can override this:
|
||||||
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
llms_txt_title = "My Custom Project Documentation"
|
||||||
|
|
||||||
|
.. _handling_large_documentation:
|
||||||
|
|
||||||
|
Handling Large Documentation
|
||||||
|
^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
||||||
|
|
||||||
|
For very large documentation sets, generating the full documentation file might exceed reasonable size limits.
|
||||||
|
You can set a maximum line count:
|
||||||
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
llms_txt_full_max_size = 10000 # Maximum 10,000 lines
|
||||||
|
|
||||||
|
If the generated file would exceed this limit, the extension will skip its generation and show a warning, allowing the build to complete.
|
||||||
|
|
||||||
|
.. tip:: Use :ref:`excluding_content` to remove less relevant pages.
|
||||||
|
|
||||||
|
.. _custom_directive_handling:
|
||||||
|
|
||||||
|
Custom Directive Handling
|
||||||
|
^^^^^^^^^^^^^^^^^^^^^^^^^
|
||||||
|
|
||||||
|
.. _path_resolution:
|
||||||
|
|
||||||
|
Path Resolution
|
||||||
|
~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
|
The extension resolves paths in the common directives ``[ 'image', 'figure']`` by default.
|
||||||
|
You can add custom directives to this list:
|
||||||
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
llms_txt_directives = [
|
||||||
|
"my-custom-image-directive",
|
||||||
|
"another-directive-with-paths",
|
||||||
|
]
|
||||||
|
|
||||||
|
This ensures that paths in your custom directives are properly resolved in the generated files.
|
||||||
|
|
||||||
|
.. _excluding_content:
|
||||||
|
|
||||||
|
Excluding Content
|
||||||
|
^^^^^^^^^^^^^^^^^
|
||||||
|
|
||||||
|
You can exclude specific pages from being included in the generated files:
|
||||||
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
llms_txt_exclude = [
|
||||||
|
"search", # Exclude the search page
|
||||||
|
"genindex", # Exclude the index page
|
||||||
|
"private_*", # Exclude all pages starting with 'private_'
|
||||||
|
]
|
||||||
|
|
||||||
|
This is useful for excluding auto-generated pages, indexes, or content that isn't relevant for LLM consumption.
|
||||||
|
|
||||||
|
.. _including_code_files:
|
||||||
|
|
||||||
|
Including Source Code Files
|
||||||
|
^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
||||||
|
|
||||||
|
You can include source code files from your project at the end of :confval:`llms_txt_full_filename`.
|
||||||
|
|
||||||
|
Use include/exclude syntax to precisely control which files are included:
|
||||||
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
llms_txt_code_files = [
|
||||||
|
"+:src/**/*.py", # Include all Python files in src
|
||||||
|
"-:src/**/__pycache__/**", # Exclude Python cache files
|
||||||
|
]
|
||||||
|
|
||||||
|
Pattern syntax:
|
||||||
|
|
||||||
|
- **+:pattern**: Include files matching the pattern. Processed first to collect matching files.
|
||||||
|
- **-:pattern**: Exclude files matching the pattern. Applied to filter out unwanted files.
|
||||||
|
|
||||||
|
Code files are processed as follows:
|
||||||
|
|
||||||
|
- **Glob patterns**: Use standard glob patterns (``*``, ``**``, ``?``) to match files
|
||||||
|
- **Relative paths**: Patterns are resolved relative to your Sphinx source directory
|
||||||
|
- **Formatting**: Each file is presented with a title and syntax-highlighted code block
|
||||||
|
|
||||||
|
.. _customizing_code_paths:
|
||||||
|
|
||||||
|
Customizing Code File Paths
|
||||||
|
~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
|
By default, the extension automatically detects the relative path from your Sphinx source directory to the git root and strips that prefix from displayed file paths. You can customize this behavior:
|
||||||
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# Manually specify base path to strip
|
||||||
|
llms_txt_code_base_path = "../../"
|
||||||
|
|
||||||
|
# Disable path stripping entirely
|
||||||
|
llms_txt_code_base_path = ""
|
||||||
|
|
||||||
|
This helps create cleaner, more readable file paths in the generated documentation.
|
||||||
|
|
||||||
|
.. _using_html_baseurl:
|
||||||
|
|
||||||
|
Using HTML Base URL
|
||||||
|
^^^^^^^^^^^^^^^^^^^
|
||||||
|
|
||||||
|
If you want to include absolute URLs for resources in your documentation, you can use Sphinx's built-in ``html_baseurl`` configuration:
|
||||||
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
html_baseurl = "https://example.com/docs/"
|
||||||
|
|
||||||
|
When this option is set, all resolved paths in directives will be prefixed with this URL, creating absolute paths in the generated files.
|
||||||
|
|
||||||
|
.. _integration_examples:
|
||||||
|
|
||||||
|
Integration Examples
|
||||||
|
^^^^^^^^^^^^^^^^^^^^
|
||||||
|
|
||||||
|
Complete Configuration Example
|
||||||
|
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
|
Here's a complete example showing multiple :doc:`configuration-values`:
|
||||||
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# File names and generation options
|
||||||
|
llms_txt_filename = "ai-summary.txt"
|
||||||
|
llms_txt_full_filename = "ai-full-docs.txt"
|
||||||
|
llms_txt_full_max_size = 50000
|
||||||
|
|
||||||
|
# Content customization
|
||||||
|
llms_txt_title = "Project Documentation for AI Assistants"
|
||||||
|
llms_txt_summary = """
|
||||||
|
This is a comprehensive documentation set for our project.
|
||||||
|
It includes API references, usage examples, and tutorials.
|
||||||
|
"""
|
||||||
|
|
||||||
|
# Path handling
|
||||||
|
html_baseurl = "https://docs.example.com/"
|
||||||
|
llms_txt_directives = ["custom-image", "custom-include"]
|
||||||
|
|
||||||
|
# Content filtering
|
||||||
|
llms_txt_exclude = ["search", "genindex", "404", "private_*"]
|
||||||
|
|
||||||
|
# Source code inclusion with include/exclude patterns
|
||||||
|
llms_txt_code_files = [
|
||||||
|
"+:../../src/**/*.py", # Include Python files
|
||||||
|
"+:../../config/*.yaml", # Include config files
|
||||||
|
"-:../../src/**/__pycache__/**", # Exclude cache files
|
||||||
|
]
|
||||||
|
llms_txt_code_base_path = "../../"
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
.. include:: ../../CHANGELOG.rst
|
||||||
@@ -0,0 +1,106 @@
|
|||||||
|
#
|
||||||
|
# Configuration file for the Sphinx documentation builder.
|
||||||
|
#
|
||||||
|
# This file does only contain a selection of the most common options. For a
|
||||||
|
# full list see the documentation:
|
||||||
|
# http://www.sphinx-doc.org/en/master/config
|
||||||
|
|
||||||
|
# -- Path setup --------------------------------------------------------------
|
||||||
|
|
||||||
|
import re
|
||||||
|
import subprocess
|
||||||
|
|
||||||
|
# -- Project information -----------------------------------------------------
|
||||||
|
|
||||||
|
project = "sphinx-llms-txt"
|
||||||
|
copyright = "Jared Dillard"
|
||||||
|
author = "Jared Dillard"
|
||||||
|
llms_txt_code_files = ["+:../../sphinx_llms_txt/*.py"]
|
||||||
|
llms_txt_summary = """
|
||||||
|
A Sphinx extension that generates a summary llms.txt file,written in Markdown,
|
||||||
|
and a single combined documentation llms-full.txt file, written in reStructuredText.
|
||||||
|
"""
|
||||||
|
|
||||||
|
# check if the current commit is tagged as a release (vX.Y.Z)
|
||||||
|
try:
|
||||||
|
GIT_TAG_OUTPUT = subprocess.check_output(["git", "tag", "--points-at", "HEAD"])
|
||||||
|
current_tag = GIT_TAG_OUTPUT.decode().strip()
|
||||||
|
if re.match(r"^v(0|[1-9]\d*)\.(0|[1-9]\d*)\.(0|[1-9]\d*)$", current_tag):
|
||||||
|
version = current_tag
|
||||||
|
else:
|
||||||
|
version = "latest"
|
||||||
|
except (subprocess.CalledProcessError, FileNotFoundError):
|
||||||
|
version = "latest"
|
||||||
|
|
||||||
|
# The full version, including alpha/beta/rc tags
|
||||||
|
release = ""
|
||||||
|
|
||||||
|
|
||||||
|
# -- General configuration ---------------------------------------------------
|
||||||
|
|
||||||
|
# If your documentation needs a minimal Sphinx version, state it here.
|
||||||
|
#
|
||||||
|
# needs_sphinx = '1.0'
|
||||||
|
|
||||||
|
# Add any Sphinx extension module names here, as strings. They can be
|
||||||
|
# extensions coming with Sphinx (named 'sphinx.ext.*') or your custom
|
||||||
|
# ones.
|
||||||
|
extensions = [
|
||||||
|
"sphinx.ext.intersphinx",
|
||||||
|
"sphinx_contributors",
|
||||||
|
"sphinx_llms_txt",
|
||||||
|
]
|
||||||
|
|
||||||
|
# The language for content autogenerated by Sphinx. Refer to documentation
|
||||||
|
# for a list of supported languages.
|
||||||
|
#
|
||||||
|
# This is also used if you do content translation via gettext catalogs.
|
||||||
|
# Usually you set "language" from the command line for these cases.
|
||||||
|
language = "en"
|
||||||
|
|
||||||
|
# List of patterns, relative to source directory, that match files and
|
||||||
|
# directories to ignore when looking for source files.
|
||||||
|
# This pattern also affects html_static_path and html_extra_path.
|
||||||
|
exclude_patterns = ["_build", "Thumbs.db", ".DS_Store"]
|
||||||
|
|
||||||
|
# The name of the Pygments (syntax highlighting) style to use.
|
||||||
|
pygments_style = "sphinx"
|
||||||
|
|
||||||
|
intersphinx_mapping = {
|
||||||
|
"sphinx": ("https://www.sphinx-doc.org/en/master/", None),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
# -- Options for HTML output -------------------------------------------------
|
||||||
|
|
||||||
|
# The theme to use for HTML and HTML Help pages. See the documentation for
|
||||||
|
# a list of builtin themes.
|
||||||
|
#
|
||||||
|
html_theme = "furo"
|
||||||
|
|
||||||
|
# Theme options are theme-specific and customize the look and feel of a theme
|
||||||
|
# further. For a list of options available for each theme, see the
|
||||||
|
# documentation.
|
||||||
|
#
|
||||||
|
html_theme_options = {
|
||||||
|
"source_repository": "https://github.com/jdillard/sphinx-llms-txt/",
|
||||||
|
"source_branch": "main",
|
||||||
|
"source_directory": "docs/source/",
|
||||||
|
}
|
||||||
|
|
||||||
|
html_baseurl = "https://sphinx-llms-txt.readthedocs.org/"
|
||||||
|
|
||||||
|
|
||||||
|
# -- Options for HTMLHelp output ---------------------------------------------
|
||||||
|
|
||||||
|
# Output file base name for HTML help builder.
|
||||||
|
htmlhelp_basename = "SphinxLLMsTxtdoc"
|
||||||
|
|
||||||
|
|
||||||
|
def setup(app):
|
||||||
|
app.add_object_type(
|
||||||
|
"confval",
|
||||||
|
"confval",
|
||||||
|
objname="configuration value",
|
||||||
|
indextemplate="pair: %s; configuration value",
|
||||||
|
)
|
||||||
@@ -0,0 +1,103 @@
|
|||||||
|
Project Configuration Values
|
||||||
|
============================
|
||||||
|
|
||||||
|
.. confval:: llms_txt_full_file
|
||||||
|
|
||||||
|
- **Type**: boolean
|
||||||
|
- **Default**: ``True``
|
||||||
|
- **Description**: Whether to write the single output file.
|
||||||
|
See :ref:`disabling_file_generation`.
|
||||||
|
|
||||||
|
.. versionadded:: 0.1.0
|
||||||
|
|
||||||
|
.. confval:: llms_txt_full_filename
|
||||||
|
|
||||||
|
- **Type**: string
|
||||||
|
- **Default**: ``'llms-full.txt'``
|
||||||
|
- **Description**: Name of the single output file.
|
||||||
|
See :ref:`changing_filenames`.
|
||||||
|
|
||||||
|
.. versionadded:: 0.1.0
|
||||||
|
|
||||||
|
.. confval:: llms_txt_full_max_size
|
||||||
|
|
||||||
|
- **Type**: integer or ``None``
|
||||||
|
- **Default**: ``None`` (no limit)
|
||||||
|
- **Description**: Sets a maximum line count for ``llms_txt_full_filename``.
|
||||||
|
If exceeded, the file is skipped and a warning is shown, but the build still completes.
|
||||||
|
See :ref:`handling_large_documentation`.
|
||||||
|
|
||||||
|
.. versionadded:: 0.2.0
|
||||||
|
|
||||||
|
.. confval:: llms_txt_file
|
||||||
|
|
||||||
|
- **Type**: boolean
|
||||||
|
- **Default**: ``True``
|
||||||
|
- **Description**: Whether to write the summary information file.
|
||||||
|
See :ref:`disabling_file_generation`.
|
||||||
|
|
||||||
|
.. versionadded:: 0.2.0
|
||||||
|
|
||||||
|
.. confval:: llms_txt_filename
|
||||||
|
|
||||||
|
- **Type**: string
|
||||||
|
- **Default**: ``llms.txt``
|
||||||
|
- **Description**: Name of the summary information file.
|
||||||
|
See :ref:`changing_filenames`.
|
||||||
|
|
||||||
|
.. versionadded:: 0.2.0
|
||||||
|
|
||||||
|
.. confval:: llms_txt_directives
|
||||||
|
|
||||||
|
- **Type**: list of strings
|
||||||
|
- **Default**: ``[]`` (empty list)
|
||||||
|
- **Description**: List of custom directive names to process for path resolution.
|
||||||
|
See :ref:`path_resolution`.
|
||||||
|
|
||||||
|
.. versionadded:: 0.1.0
|
||||||
|
|
||||||
|
.. confval:: llms_txt_title
|
||||||
|
|
||||||
|
- **Type**: string or ``None``
|
||||||
|
- **Default**: ``None``
|
||||||
|
- **Description**: Overrides the Sphinx project name as the heading in ``llms.txt``.
|
||||||
|
See :ref:`custom_title`.
|
||||||
|
|
||||||
|
.. versionadded:: 0.2.0
|
||||||
|
|
||||||
|
.. confval:: llms_txt_summary
|
||||||
|
|
||||||
|
- **Type**: string
|
||||||
|
- **Default**: The first paragraph in the root document, else an empty string
|
||||||
|
- **Description**: Optional, but recommended, summary description for ``llms.txt``.
|
||||||
|
See :ref:`custom_summary`.
|
||||||
|
|
||||||
|
.. versionadded:: 0.2.0
|
||||||
|
|
||||||
|
.. confval:: llms_txt_exclude
|
||||||
|
|
||||||
|
- **Type**: list of strings
|
||||||
|
- **Default**: ``[]``
|
||||||
|
- **Description**: A list of pages to ignore.
|
||||||
|
See :ref:`excluding_content`.
|
||||||
|
|
||||||
|
.. versionadded:: 0.2.1
|
||||||
|
|
||||||
|
.. confval:: llms_txt_code_files
|
||||||
|
|
||||||
|
- **Type**: list of strings
|
||||||
|
- **Default**: ``[]``
|
||||||
|
- **Description**: A list of glob patterns that appends source code files to :confval:`llms_txt_full_filename`.
|
||||||
|
See :ref:`including_code_files`.
|
||||||
|
|
||||||
|
.. versionadded:: 0.4.0
|
||||||
|
|
||||||
|
.. confval:: llms_txt_code_base_path
|
||||||
|
|
||||||
|
- **Type**: string or ``None``
|
||||||
|
- **Default**: ``None`` (auto-detect from git root)
|
||||||
|
- **Description**: Base path to strip from code file paths when displaying titles.
|
||||||
|
When ``None``, automatically detects the relative path from the Sphinx source
|
||||||
|
directory to the git root and strips that prefix from file paths.
|
||||||
|
|
||||||
|
.. versionadded:: 0.4.0
|
||||||
@@ -0,0 +1,48 @@
|
|||||||
|
Contributing
|
||||||
|
============
|
||||||
|
|
||||||
|
You will need to set up a development environment to make and test your changes before submitting them.
|
||||||
|
|
||||||
|
Local development
|
||||||
|
-----------------
|
||||||
|
|
||||||
|
#. Clone the `sphinx-llms-txt repository`_.
|
||||||
|
|
||||||
|
#. Create and activate a virtual environment:
|
||||||
|
|
||||||
|
.. code-block:: console
|
||||||
|
|
||||||
|
python3 -m venv .venv
|
||||||
|
source .venv/bin/activate
|
||||||
|
|
||||||
|
#. Install development dependencies:
|
||||||
|
|
||||||
|
.. code-block:: console
|
||||||
|
|
||||||
|
pip install -e ".[dev]"
|
||||||
|
|
||||||
|
#. Install pre-commit Git hook scripts:
|
||||||
|
|
||||||
|
.. code-block:: console
|
||||||
|
|
||||||
|
pre-commit install
|
||||||
|
|
||||||
|
Testing changes
|
||||||
|
---------------
|
||||||
|
|
||||||
|
Run ``pytest`` before committing changes.
|
||||||
|
|
||||||
|
Current contributors
|
||||||
|
--------------------
|
||||||
|
|
||||||
|
Thanks to all who have contributed!
|
||||||
|
The people that have improved the code:
|
||||||
|
|
||||||
|
.. contributors:: jdillard/sphinx-llms-txt
|
||||||
|
:avatars:
|
||||||
|
:limit: 100
|
||||||
|
:exclude: pre-commit-ci[bot],dependabot[bot]
|
||||||
|
:order: ASC
|
||||||
|
|
||||||
|
|
||||||
|
.. _sphinx-llms-txt repository: https://github.com/jdillard/sphinx-llms-txt
|
||||||
@@ -0,0 +1,50 @@
|
|||||||
|
Getting Started
|
||||||
|
===============
|
||||||
|
|
||||||
|
Demo
|
||||||
|
----
|
||||||
|
|
||||||
|
You can see this Sphinx project's `llms.txt`_ and `llms-full.txt`_ files as a simple example.
|
||||||
|
|
||||||
|
Installation
|
||||||
|
------------
|
||||||
|
|
||||||
|
Directly install via ``pip`` by using:
|
||||||
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
pip install sphinx-llms-txt
|
||||||
|
|
||||||
|
Usage
|
||||||
|
-----
|
||||||
|
|
||||||
|
Add the extension to your Sphinx configuration (``conf.py``):
|
||||||
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
extensions = [
|
||||||
|
'sphinx_llms_txt',
|
||||||
|
]
|
||||||
|
|
||||||
|
Once added, the extension will automatically generate the LLMs.txt files during the build process.
|
||||||
|
|
||||||
|
See :doc:`advanced-configuration` for more information about how to use **sphinx-llms-txt**.
|
||||||
|
|
||||||
|
How It Works
|
||||||
|
------------
|
||||||
|
|
||||||
|
During the Sphinx build process:
|
||||||
|
|
||||||
|
1. **Content Collection**: Scans all of your documentation's ``_source`` pages and collects their content
|
||||||
|
2. **Directive Processing**: Resolves ``include`` directives by automatically incorporating their content
|
||||||
|
3. **Path Resolution**: Transforms relative paths in directives to full paths
|
||||||
|
4. **Output Generation**: Creates two optional files:
|
||||||
|
|
||||||
|
- ``llms.txt``: A concise summary of your documentation, in Markdown
|
||||||
|
- ``llms-full.txt``: A comprehensive version with all documentation content, in reStructuredText
|
||||||
|
|
||||||
|
5. **Content Filtering**: Allows you to exclude specific pages from the generated files
|
||||||
|
|
||||||
|
|
||||||
|
.. _llms.txt: https://sphinx-llms-txt.readthedocs.io/en/latest/llms.txt
|
||||||
|
.. _llms-full.txt: https://sphinx-llms-txt.readthedocs.io/en/latest/llms-full.txt
|
||||||
@@ -0,0 +1,34 @@
|
|||||||
|
Sphinx llms.txt Generator
|
||||||
|
=========================
|
||||||
|
|
||||||
|
A `Sphinx`_ extension that generates a summary ``llms.txt`` file, written in Markdown, and a single combined documentation ``llms-full.txt`` file, written in reStructuredText.
|
||||||
|
|
||||||
|
|PyPI version| |Conda Version| |Downloads| |Parallel Safe| |GitHub Stars|
|
||||||
|
|
||||||
|
.. toctree::
|
||||||
|
:maxdepth: 2
|
||||||
|
|
||||||
|
getting-started
|
||||||
|
advanced-configuration
|
||||||
|
configuration-values
|
||||||
|
contributing
|
||||||
|
changelog
|
||||||
|
|
||||||
|
|
||||||
|
.. _Sphinx: http://sphinx-doc.org/
|
||||||
|
|
||||||
|
.. |PyPI version| image:: https://img.shields.io/pypi/v/sphinx-llms-txt.svg
|
||||||
|
:target: https://pypi.python.org/pypi/sphinx-llms-txt
|
||||||
|
:alt: Latest PyPi Version
|
||||||
|
.. |Conda Version| image:: https://img.shields.io/conda/vn/conda-forge/sphinx-llms-txt.svg
|
||||||
|
:target: https://anaconda.org/conda-forge/sphinx-llms-txt
|
||||||
|
:alt: Latest Conda Version
|
||||||
|
.. |Downloads| image:: https://static.pepy.tech/badge/sphinx-llms-txt/month
|
||||||
|
:target: https://pepy.tech/project/sphinx-llms-txt
|
||||||
|
:alt: PyPi Downloads per month
|
||||||
|
.. |Parallel Safe| image:: https://img.shields.io/badge/parallel%20safe-true-brightgreen
|
||||||
|
:target: #
|
||||||
|
:alt: Parallel read/write safe
|
||||||
|
.. |GitHub Stars| image:: https://img.shields.io/github/stars/jdillard/sphinx-llms-txt?style=social
|
||||||
|
:target: https://github.com/jdillard/sphinx-llms-txt
|
||||||
|
:alt: GitHub Repository stars
|
||||||
@@ -40,6 +40,7 @@ dev = [
|
|||||||
"mypy",
|
"mypy",
|
||||||
"isort",
|
"isort",
|
||||||
"pre-commit",
|
"pre-commit",
|
||||||
|
"sphinx",
|
||||||
]
|
]
|
||||||
test = [
|
test = [
|
||||||
"pytest>=7.0.0",
|
"pytest>=7.0.0",
|
||||||
@@ -84,4 +85,5 @@ filterwarnings = [
|
|||||||
"error",
|
"error",
|
||||||
"ignore::UserWarning",
|
"ignore::UserWarning",
|
||||||
"ignore::DeprecationWarning",
|
"ignore::DeprecationWarning",
|
||||||
|
"ignore::PendingDeprecationWarning",
|
||||||
]
|
]
|
||||||
+62
-232
@@ -1,252 +1,56 @@
|
|||||||
"""
|
"""
|
||||||
Sphinx extension to create a combined sources file (llms-full.rst)
|
Sphinx extension to create a combined sources file (llms-full.txt)
|
||||||
that combines all documentation sources in the correct build order.
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from pathlib import Path
|
from typing import Any, Dict
|
||||||
from typing import Any, Dict, List
|
|
||||||
|
|
||||||
from sphinx.application import Sphinx
|
|
||||||
from sphinx.environment import BuildEnvironment
|
|
||||||
from sphinx.util import logging
|
|
||||||
|
|
||||||
__version__ = "0.1.0"
|
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
|
||||||
|
|
||||||
|
|
||||||
class LLMSFullManager:
|
|
||||||
"""Manages the collection and ordering of documentation sources."""
|
|
||||||
|
|
||||||
def __init__(self):
|
|
||||||
self.page_titles: Dict[str, str] = {}
|
|
||||||
self.config: Dict[str, Any] = {}
|
|
||||||
self.master_doc: str = None
|
|
||||||
self.env: BuildEnvironment = None
|
|
||||||
|
|
||||||
def set_master_doc(self, master_doc: str):
|
|
||||||
"""Set the master document name."""
|
|
||||||
self.master_doc = master_doc
|
|
||||||
|
|
||||||
def set_env(self, env: BuildEnvironment):
|
|
||||||
"""Set the Sphinx environment."""
|
|
||||||
self.env = env
|
|
||||||
|
|
||||||
def update_page_title(self, docname: str, title: str):
|
|
||||||
"""Update the title for a page."""
|
|
||||||
if title:
|
|
||||||
self.page_titles[docname] = title
|
|
||||||
|
|
||||||
def set_config(self, config: Dict[str, Any]):
|
|
||||||
"""Set configuration options."""
|
|
||||||
self.config = config
|
|
||||||
|
|
||||||
def get_page_order(self) -> List[str]:
|
|
||||||
"""Get the correct page order from the toctree structure."""
|
|
||||||
if not self.env or not self.master_doc:
|
|
||||||
return []
|
|
||||||
|
|
||||||
page_order = []
|
|
||||||
visited = set()
|
|
||||||
|
|
||||||
def collect_from_toctree(docname: str):
|
|
||||||
"""Recursively collect documents from toctree."""
|
|
||||||
if docname in visited:
|
|
||||||
return
|
|
||||||
|
|
||||||
visited.add(docname)
|
|
||||||
|
|
||||||
# Add the current document
|
|
||||||
if docname not in page_order:
|
|
||||||
page_order.append(docname)
|
|
||||||
|
|
||||||
# Check for toctree entries in this document
|
|
||||||
try:
|
|
||||||
# Look for toctree_includes which contains the direct children
|
|
||||||
if (
|
|
||||||
hasattr(self.env, "toctree_includes")
|
|
||||||
and docname in self.env.toctree_includes
|
|
||||||
):
|
|
||||||
for child_docname in self.env.toctree_includes[docname]:
|
|
||||||
collect_from_toctree(child_docname)
|
|
||||||
else:
|
|
||||||
# Fallback: try to resolve and parse the toctree
|
|
||||||
toctree = self.env.get_and_resolve_toctree(docname, None)
|
|
||||||
if toctree:
|
|
||||||
from docutils import nodes
|
from docutils import nodes
|
||||||
|
from sphinx.application import Sphinx
|
||||||
|
|
||||||
for node in toctree.traverse(nodes.reference):
|
from .collector import DocumentCollector
|
||||||
if "refuri" in node.attributes:
|
from .manager import LLMSFullManager
|
||||||
refuri = node.attributes["refuri"]
|
from .processor import DocumentProcessor
|
||||||
if refuri and refuri.endswith(".html"):
|
from .writer import FileWriter
|
||||||
child_docname = refuri[:-5] # Remove .html
|
|
||||||
if (
|
|
||||||
child_docname != docname
|
|
||||||
): # Avoid circular references
|
|
||||||
collect_from_toctree(child_docname)
|
|
||||||
except Exception as e:
|
|
||||||
logger.debug(f"Could not get toctree for {docname}: {e}")
|
|
||||||
|
|
||||||
# Start from the master document
|
__version__ = "0.4.0"
|
||||||
collect_from_toctree(self.master_doc)
|
|
||||||
|
|
||||||
# Add any remaining documents not in the toctree (sorted)
|
# Export classes needed by tests
|
||||||
if hasattr(self.env, "all_docs"):
|
__all__ = [
|
||||||
remaining = sorted(
|
"DocumentCollector",
|
||||||
[doc for doc in self.env.all_docs.keys() if doc not in page_order]
|
"DocumentProcessor",
|
||||||
)
|
"FileWriter",
|
||||||
page_order.extend(remaining)
|
"LLMSFullManager",
|
||||||
|
|
||||||
return page_order
|
|
||||||
|
|
||||||
def combine_sources(self, outdir: str, srcdir: str):
|
|
||||||
"""Combine all source files into a single file."""
|
|
||||||
# Get the correct page order
|
|
||||||
page_order = self.get_page_order()
|
|
||||||
|
|
||||||
if not page_order:
|
|
||||||
logger.warning(
|
|
||||||
"Could not determine page order, skipping llms-full creation"
|
|
||||||
)
|
|
||||||
return
|
|
||||||
|
|
||||||
# Determine output file name and location
|
|
||||||
output_filename = self.config.get("llms_txt_filename")
|
|
||||||
output_path = Path(outdir) / output_filename
|
|
||||||
|
|
||||||
# Find sources directory
|
|
||||||
sources_dir = None
|
|
||||||
possible_sources = [
|
|
||||||
Path(outdir) / "_sources",
|
|
||||||
Path(outdir) / "html" / "_sources",
|
|
||||||
Path(outdir) / "singlehtml" / "_sources",
|
|
||||||
]
|
]
|
||||||
|
|
||||||
for path in possible_sources:
|
|
||||||
if path.exists():
|
|
||||||
sources_dir = path
|
|
||||||
break
|
|
||||||
|
|
||||||
if not sources_dir:
|
|
||||||
logger.warning(
|
|
||||||
"Could not find _sources directory, skipping llms-full creation"
|
|
||||||
)
|
|
||||||
return
|
|
||||||
|
|
||||||
# Collect all available source files
|
|
||||||
txt_files = {}
|
|
||||||
for f in sources_dir.glob("*.txt"):
|
|
||||||
txt_files[f.stem] = f
|
|
||||||
|
|
||||||
# Create a mapping from docnames to actual file names
|
|
||||||
docname_to_file = {}
|
|
||||||
|
|
||||||
# Try exact matches first
|
|
||||||
for docname in page_order:
|
|
||||||
if docname in txt_files:
|
|
||||||
docname_to_file[docname] = txt_files[docname]
|
|
||||||
else:
|
|
||||||
# Try with .rst extension
|
|
||||||
if f"{docname}.rst" in txt_files:
|
|
||||||
docname_to_file[docname] = txt_files[f"{docname}.rst"]
|
|
||||||
# Try with .txt extension
|
|
||||||
elif f"{docname}.txt" in txt_files:
|
|
||||||
docname_to_file[docname] = txt_files[f"{docname}.txt"]
|
|
||||||
# Try with underscores instead of hyphens
|
|
||||||
elif docname.replace("-", "_") in txt_files:
|
|
||||||
docname_to_file[docname] = txt_files[docname.replace("-", "_")]
|
|
||||||
# Try with hyphens instead of underscores
|
|
||||||
elif docname.replace("_", "-") in txt_files:
|
|
||||||
docname_to_file[docname] = txt_files[docname.replace("_", "-")]
|
|
||||||
|
|
||||||
# Generate content
|
|
||||||
content_parts = []
|
|
||||||
|
|
||||||
# Add pages in order
|
|
||||||
added_files = set()
|
|
||||||
|
|
||||||
for docname in page_order:
|
|
||||||
if docname in docname_to_file:
|
|
||||||
file_path = docname_to_file[docname]
|
|
||||||
content = self._read_source_file(file_path, docname)
|
|
||||||
if content:
|
|
||||||
content_parts.append(content)
|
|
||||||
added_files.add(file_path.stem)
|
|
||||||
else:
|
|
||||||
logger.warning(f"Source file not found for: {docname}")
|
|
||||||
|
|
||||||
# Add any remaining files (in alphabetical order)
|
|
||||||
remaining_files = sorted(
|
|
||||||
[name for name in txt_files if name not in added_files]
|
|
||||||
)
|
|
||||||
if remaining_files:
|
|
||||||
logger.info(f"Adding remaining files: {remaining_files}")
|
|
||||||
for file_stem in remaining_files:
|
|
||||||
file_path = txt_files[file_stem]
|
|
||||||
content = self._read_source_file(file_path, file_stem)
|
|
||||||
if content:
|
|
||||||
content_parts.append(content)
|
|
||||||
|
|
||||||
# Write combined file
|
|
||||||
try:
|
|
||||||
with open(output_path, "w", encoding="utf-8") as f:
|
|
||||||
f.write("\n".join(content_parts))
|
|
||||||
|
|
||||||
logger.info(
|
|
||||||
f"sphinx-llms-txt: created {output_path} with {len(txt_files)} sources"
|
|
||||||
)
|
|
||||||
|
|
||||||
# Log summary information if requested
|
|
||||||
if self.config.get("llms_txt_verbose"):
|
|
||||||
self._log_summary_info(page_order)
|
|
||||||
|
|
||||||
except Exception as e:
|
|
||||||
logger.error(f"Error writing combined sources file: {e}")
|
|
||||||
|
|
||||||
def _read_source_file(self, file_path: Path, docname: str) -> str:
|
|
||||||
"""Read and format a single source file."""
|
|
||||||
try:
|
|
||||||
with open(file_path, "r", encoding="utf-8") as f:
|
|
||||||
content = f.read()
|
|
||||||
|
|
||||||
section_lines = [content, ""]
|
|
||||||
|
|
||||||
return "\n".join(section_lines)
|
|
||||||
|
|
||||||
except Exception as e:
|
|
||||||
logger.error(f"Error reading source file {file_path}: {e}")
|
|
||||||
return ""
|
|
||||||
|
|
||||||
def _log_summary_info(self, page_order: List[str]):
|
|
||||||
"""Log summary information to the logger."""
|
|
||||||
logger.info("")
|
|
||||||
logger.info("llms-txt Summary")
|
|
||||||
logger.info("================")
|
|
||||||
logger.info(f"Total pages: {len(page_order)}")
|
|
||||||
logger.info(f"Configuration: {self.config}")
|
|
||||||
logger.info("Page order:")
|
|
||||||
for i, docname in enumerate(page_order, 1):
|
|
||||||
title = self.page_titles.get(docname, docname)
|
|
||||||
logger.info(f"{i:3d}. {docname} - {title}")
|
|
||||||
|
|
||||||
|
|
||||||
# Global manager instance
|
# Global manager instance
|
||||||
_manager = LLMSFullManager()
|
_manager = LLMSFullManager()
|
||||||
|
|
||||||
|
# Store root document first paragraph
|
||||||
|
_root_first_paragraph = ""
|
||||||
|
|
||||||
|
|
||||||
def doctree_resolved(app: Sphinx, doctree, docname: str):
|
def doctree_resolved(app: Sphinx, doctree, docname: str):
|
||||||
"""Called when a docname has been resolved to a document."""
|
"""Called when a docname has been resolved to a document."""
|
||||||
# Extract title from the document
|
global _root_first_paragraph
|
||||||
from docutils import nodes
|
|
||||||
|
|
||||||
|
# Extract title from the document
|
||||||
title = None
|
title = None
|
||||||
for node in doctree.traverse(nodes.title):
|
# findall() returns a generator, convert to list to check if it has elements
|
||||||
title = node.astext()
|
title_nodes = list(doctree.findall(nodes.title))
|
||||||
break
|
if title_nodes:
|
||||||
|
title = title_nodes[0].astext()
|
||||||
|
|
||||||
if title:
|
if title:
|
||||||
_manager.update_page_title(docname, title)
|
_manager.update_page_title(docname, title)
|
||||||
|
|
||||||
|
# Extract first paragraph from root document
|
||||||
|
if docname == app.config.master_doc:
|
||||||
|
for node in doctree.traverse(nodes.paragraph):
|
||||||
|
first_para = node.astext()
|
||||||
|
if first_para:
|
||||||
|
_root_first_paragraph = first_para
|
||||||
|
break
|
||||||
|
|
||||||
|
|
||||||
def build_finished(app: Sphinx, exception):
|
def build_finished(app: Sphinx, exception):
|
||||||
"""Called when the build is finished."""
|
"""Called when the build is finished."""
|
||||||
@@ -254,11 +58,27 @@ def build_finished(app: Sphinx, exception):
|
|||||||
# Set the environment and master doc in the manager
|
# Set the environment and master doc in the manager
|
||||||
_manager.set_env(app.env)
|
_manager.set_env(app.env)
|
||||||
_manager.set_master_doc(app.config.master_doc)
|
_manager.set_master_doc(app.config.master_doc)
|
||||||
|
_manager.set_app(app)
|
||||||
|
|
||||||
|
# Get the summary - use configured value or extracted first paragraph
|
||||||
|
summary = app.config.llms_txt_summary
|
||||||
|
if summary is None:
|
||||||
|
summary = _root_first_paragraph
|
||||||
|
|
||||||
# Set up configuration
|
# Set up configuration
|
||||||
config = {
|
config = {
|
||||||
|
"llms_txt_file": app.config.llms_txt_file,
|
||||||
"llms_txt_filename": app.config.llms_txt_filename,
|
"llms_txt_filename": app.config.llms_txt_filename,
|
||||||
"llms_txt_verbose": app.config.llms_txt_verbose,
|
"llms_txt_title": app.config.llms_txt_title,
|
||||||
|
"llms_txt_summary": summary,
|
||||||
|
"llms_txt_full_file": app.config.llms_txt_full_file,
|
||||||
|
"llms_txt_full_filename": app.config.llms_txt_full_filename,
|
||||||
|
"llms_txt_full_max_size": app.config.llms_txt_full_max_size,
|
||||||
|
"llms_txt_directives": app.config.llms_txt_directives,
|
||||||
|
"llms_txt_exclude": app.config.llms_txt_exclude,
|
||||||
|
"llms_txt_code_files": app.config.llms_txt_code_files,
|
||||||
|
"llms_txt_code_base_path": app.config.llms_txt_code_base_path,
|
||||||
|
"html_baseurl": getattr(app.config, "html_baseurl", ""),
|
||||||
}
|
}
|
||||||
_manager.set_config(config)
|
_manager.set_config(config)
|
||||||
|
|
||||||
@@ -277,16 +97,26 @@ def setup(app: Sphinx) -> Dict[str, Any]:
|
|||||||
"""Set up the Sphinx extension."""
|
"""Set up the Sphinx extension."""
|
||||||
|
|
||||||
# Add configuration options
|
# Add configuration options
|
||||||
app.add_config_value("llms_txt_filename", "llms-full.txt", "env")
|
app.add_config_value("llms_txt_file", True, "env")
|
||||||
app.add_config_value("llms_txt_verbose", False, "env")
|
app.add_config_value("llms_txt_filename", "llms.txt", "env")
|
||||||
|
app.add_config_value("llms_txt_full_file", True, "env")
|
||||||
|
app.add_config_value("llms_txt_full_filename", "llms-full.txt", "env")
|
||||||
|
app.add_config_value("llms_txt_full_max_size", None, "env")
|
||||||
|
app.add_config_value("llms_txt_directives", [], "env")
|
||||||
|
app.add_config_value("llms_txt_title", None, "env")
|
||||||
|
app.add_config_value("llms_txt_summary", None, "env")
|
||||||
|
app.add_config_value("llms_txt_exclude", [], "env")
|
||||||
|
app.add_config_value("llms_txt_code_files", [], "env")
|
||||||
|
app.add_config_value("llms_txt_code_base_path", None, "env")
|
||||||
|
|
||||||
# Connect to Sphinx events
|
# Connect to Sphinx events
|
||||||
app.connect("doctree-resolved", doctree_resolved)
|
app.connect("doctree-resolved", doctree_resolved)
|
||||||
app.connect("build-finished", build_finished)
|
app.connect("build-finished", build_finished)
|
||||||
|
|
||||||
# Reset manager for each build
|
# Reset manager and root paragraph for each build
|
||||||
global _manager
|
global _manager, _root_first_paragraph
|
||||||
_manager = LLMSFullManager()
|
_manager = LLMSFullManager()
|
||||||
|
_root_first_paragraph = ""
|
||||||
|
|
||||||
return {
|
return {
|
||||||
"version": __version__,
|
"version": __version__,
|
||||||
|
|||||||
@@ -0,0 +1,229 @@
|
|||||||
|
"""
|
||||||
|
Document collector module for sphinx-llms-txt.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import fnmatch
|
||||||
|
from typing import Any, Dict, List, Tuple
|
||||||
|
|
||||||
|
from sphinx.environment import BuildEnvironment
|
||||||
|
from sphinx.util import logging
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
class DocumentCollector:
|
||||||
|
"""Collects and orders documentation sources based on toctree structure."""
|
||||||
|
|
||||||
|
def __init__(self):
|
||||||
|
self.page_titles: Dict[str, str] = {}
|
||||||
|
self.master_doc: str = None
|
||||||
|
self.env: BuildEnvironment = None
|
||||||
|
self.config: Dict[str, Any] = {}
|
||||||
|
self.app = None
|
||||||
|
|
||||||
|
def set_master_doc(self, master_doc: str):
|
||||||
|
"""Set the master document name."""
|
||||||
|
self.master_doc = master_doc
|
||||||
|
|
||||||
|
def set_env(self, env: BuildEnvironment):
|
||||||
|
"""Set the Sphinx environment."""
|
||||||
|
self.env = env
|
||||||
|
|
||||||
|
def update_page_title(self, docname: str, title: str):
|
||||||
|
"""Update the title for a page."""
|
||||||
|
if title:
|
||||||
|
self.page_titles[docname] = title
|
||||||
|
|
||||||
|
def set_config(self, config: Dict[str, Any]):
|
||||||
|
"""Set configuration options."""
|
||||||
|
self.config = config
|
||||||
|
|
||||||
|
def set_app(self, app):
|
||||||
|
"""Set the Sphinx application reference."""
|
||||||
|
self.app = app
|
||||||
|
|
||||||
|
def _get_source_suffixes(self):
|
||||||
|
"""Get all valid source file suffixes from Sphinx configuration.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
list: List of source file suffixes (e.g., ['.rst', '.md', '.txt'])
|
||||||
|
"""
|
||||||
|
if not self.app:
|
||||||
|
return [".rst"] # Default fallback
|
||||||
|
|
||||||
|
source_suffix = self.app.config.source_suffix
|
||||||
|
|
||||||
|
if isinstance(source_suffix, dict):
|
||||||
|
return list(source_suffix.keys())
|
||||||
|
elif isinstance(source_suffix, list):
|
||||||
|
return source_suffix
|
||||||
|
else:
|
||||||
|
return [source_suffix] # String format
|
||||||
|
|
||||||
|
def _get_docname_suffix(self, docname: str, sources_dir) -> str:
|
||||||
|
"""
|
||||||
|
Determine the source suffix for a given docname by checking which
|
||||||
|
file exists.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
docname: The document name to check
|
||||||
|
sources_dir: Path to the _sources directory
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
The source suffix if found, or None if no matching file exists
|
||||||
|
"""
|
||||||
|
if not sources_dir or not sources_dir.exists():
|
||||||
|
return None
|
||||||
|
|
||||||
|
# Get the source link suffix from Sphinx config
|
||||||
|
source_link_suffix = ""
|
||||||
|
if self.app and hasattr(self.app.config, "html_sourcelink_suffix"):
|
||||||
|
source_link_suffix = self.app.config.html_sourcelink_suffix
|
||||||
|
# Handle empty string case specially
|
||||||
|
if source_link_suffix == "":
|
||||||
|
source_link_suffix = "" # Keep it empty
|
||||||
|
elif not source_link_suffix.startswith("."):
|
||||||
|
source_link_suffix = "." + source_link_suffix
|
||||||
|
|
||||||
|
# Get the source file suffixes from Sphinx config
|
||||||
|
source_suffixes = self._get_source_suffixes()
|
||||||
|
|
||||||
|
# Try to find the source file with any of the valid source suffixes
|
||||||
|
for src_suffix in source_suffixes:
|
||||||
|
# Avoid duplicate extensions when source_suffix == source_link_suffix
|
||||||
|
if src_suffix == source_link_suffix:
|
||||||
|
candidate_file = sources_dir / f"{docname}{src_suffix}"
|
||||||
|
else:
|
||||||
|
candidate_file = (
|
||||||
|
sources_dir / f"{docname}{src_suffix}{source_link_suffix}"
|
||||||
|
)
|
||||||
|
if candidate_file.exists():
|
||||||
|
return src_suffix
|
||||||
|
|
||||||
|
return None
|
||||||
|
|
||||||
|
def get_page_order(self, sources_dir=None) -> List[Tuple[str, str]]:
|
||||||
|
"""Get the correct page order from the toctree structure.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
sources_dir: Optional path to _sources directory for suffix detection
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
List of tuples (docname, source_suffix) in toctree order
|
||||||
|
"""
|
||||||
|
if not self.env or not self.master_doc:
|
||||||
|
return []
|
||||||
|
|
||||||
|
page_order = []
|
||||||
|
visited = set()
|
||||||
|
|
||||||
|
def collect_from_toctree(docname: str):
|
||||||
|
"""Recursively collect documents from toctree."""
|
||||||
|
if docname in visited:
|
||||||
|
return
|
||||||
|
|
||||||
|
visited.add(docname)
|
||||||
|
|
||||||
|
# Add the current document with its suffix
|
||||||
|
if docname not in [doc for doc, _ in page_order]:
|
||||||
|
suffix = None
|
||||||
|
if sources_dir:
|
||||||
|
suffix = self._get_docname_suffix(docname, sources_dir)
|
||||||
|
page_order.append((docname, suffix))
|
||||||
|
|
||||||
|
# Check for toctree entries in this document
|
||||||
|
try:
|
||||||
|
# Look for toctree_includes which contains the direct children
|
||||||
|
if (
|
||||||
|
hasattr(self.env, "toctree_includes")
|
||||||
|
and docname in self.env.toctree_includes
|
||||||
|
):
|
||||||
|
for child_docname in self.env.toctree_includes[docname]:
|
||||||
|
collect_from_toctree(child_docname)
|
||||||
|
# Try to use dependencies to find related documents
|
||||||
|
elif (
|
||||||
|
hasattr(self.env, "dependencies")
|
||||||
|
and docname in self.env.dependencies
|
||||||
|
):
|
||||||
|
# Extract the dependent documents from the dependencies dict
|
||||||
|
for child_docname in self.env.dependencies[docname]:
|
||||||
|
# Only add documents actually in the document set
|
||||||
|
if (
|
||||||
|
hasattr(self.env, "all_docs")
|
||||||
|
and child_docname in self.env.all_docs
|
||||||
|
):
|
||||||
|
collect_from_toctree(child_docname)
|
||||||
|
# Fallback to titles or other available references
|
||||||
|
elif hasattr(self.env, "titles") and hasattr(self.env, "all_docs"):
|
||||||
|
# Get all document names
|
||||||
|
all_docnames = list(self.env.all_docs.keys())
|
||||||
|
|
||||||
|
# Look for documents that might be related (have similar paths)
|
||||||
|
current_prefix = "/".join(docname.split("/")[:-1])
|
||||||
|
if current_prefix:
|
||||||
|
for child_docname in all_docnames:
|
||||||
|
# Documents in the same directory might be related
|
||||||
|
if (
|
||||||
|
child_docname.startswith(current_prefix)
|
||||||
|
and child_docname != docname
|
||||||
|
):
|
||||||
|
collect_from_toctree(child_docname)
|
||||||
|
except Exception as e:
|
||||||
|
logger.debug(f"Could not get toctree for {docname}: {e}")
|
||||||
|
|
||||||
|
# Start from the master document
|
||||||
|
collect_from_toctree(self.master_doc)
|
||||||
|
|
||||||
|
# Add any remaining documents not in the toctree (sorted)
|
||||||
|
if hasattr(self.env, "all_docs"):
|
||||||
|
processed_docnames = {doc for doc, _ in page_order}
|
||||||
|
remaining = sorted(
|
||||||
|
[
|
||||||
|
doc
|
||||||
|
for doc in self.env.all_docs.keys()
|
||||||
|
if doc not in processed_docnames
|
||||||
|
]
|
||||||
|
)
|
||||||
|
for docname in remaining:
|
||||||
|
suffix = None
|
||||||
|
if sources_dir:
|
||||||
|
suffix = self._get_docname_suffix(docname, sources_dir)
|
||||||
|
page_order.append((docname, suffix))
|
||||||
|
|
||||||
|
return page_order
|
||||||
|
|
||||||
|
def filter_excluded_pages(
|
||||||
|
self, page_order: List[Tuple[str, str]]
|
||||||
|
) -> List[Tuple[str, str]]:
|
||||||
|
"""Filter out excluded pages from the page order."""
|
||||||
|
exclude_patterns = self.config.get("llms_txt_exclude")
|
||||||
|
if exclude_patterns:
|
||||||
|
return [
|
||||||
|
(docname, suffix)
|
||||||
|
for docname, suffix in page_order
|
||||||
|
if not any(
|
||||||
|
self._match_exclude_pattern(docname, pattern)
|
||||||
|
for pattern in exclude_patterns
|
||||||
|
)
|
||||||
|
]
|
||||||
|
return page_order
|
||||||
|
|
||||||
|
def _match_exclude_pattern(self, docname: str, pattern: str) -> bool:
|
||||||
|
"""Check if a document name matches an exclude pattern.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
docname: The document name to check
|
||||||
|
pattern: The pattern to match against
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True if the document should be excluded, False otherwise
|
||||||
|
"""
|
||||||
|
# Exact match
|
||||||
|
if docname == pattern:
|
||||||
|
return True
|
||||||
|
|
||||||
|
# Glob-style pattern matching
|
||||||
|
if fnmatch.fnmatch(docname, pattern):
|
||||||
|
return True
|
||||||
|
|
||||||
|
return False
|
||||||
@@ -0,0 +1,815 @@
|
|||||||
|
"""
|
||||||
|
Main manager module for sphinx-llms-txt.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import glob
|
||||||
|
import subprocess
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any, Dict, List, Optional, Tuple
|
||||||
|
|
||||||
|
from sphinx.application import Sphinx
|
||||||
|
from sphinx.environment import BuildEnvironment
|
||||||
|
from sphinx.util import logging
|
||||||
|
|
||||||
|
from .collector import DocumentCollector
|
||||||
|
from .processor import DocumentProcessor
|
||||||
|
from .writer import FileWriter
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
def _get_git_root(path: Path) -> Optional[Path]:
|
||||||
|
"""Get the git root directory for a given path."""
|
||||||
|
try:
|
||||||
|
result = subprocess.run(
|
||||||
|
["git", "rev-parse", "--show-toplevel"],
|
||||||
|
cwd=path,
|
||||||
|
capture_output=True,
|
||||||
|
text=True,
|
||||||
|
check=True,
|
||||||
|
)
|
||||||
|
return Path(result.stdout.strip())
|
||||||
|
except (subprocess.CalledProcessError, FileNotFoundError):
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _get_language_from_extension(file_path: Path) -> str:
|
||||||
|
"""Map file extension to language identifier for code blocks."""
|
||||||
|
extension_map = {
|
||||||
|
".py": "python",
|
||||||
|
".js": "javascript",
|
||||||
|
".jsx": "jsx",
|
||||||
|
".ts": "typescript",
|
||||||
|
".tsx": "tsx",
|
||||||
|
".java": "java",
|
||||||
|
".c": "c",
|
||||||
|
".cpp": "cpp",
|
||||||
|
".cc": "cpp",
|
||||||
|
".cxx": "cpp",
|
||||||
|
".h": "c",
|
||||||
|
".hpp": "cpp",
|
||||||
|
".cs": "csharp",
|
||||||
|
".php": "php",
|
||||||
|
".rb": "ruby",
|
||||||
|
".go": "go",
|
||||||
|
".rs": "rust",
|
||||||
|
".swift": "swift",
|
||||||
|
".kt": "kotlin",
|
||||||
|
".scala": "scala",
|
||||||
|
".sh": "bash",
|
||||||
|
".bash": "bash",
|
||||||
|
".zsh": "zsh",
|
||||||
|
".fish": "fish",
|
||||||
|
".ps1": "powershell",
|
||||||
|
".html": "html",
|
||||||
|
".htm": "html",
|
||||||
|
".xml": "xml",
|
||||||
|
".css": "css",
|
||||||
|
".scss": "scss",
|
||||||
|
".sass": "sass",
|
||||||
|
".less": "less",
|
||||||
|
".json": "json",
|
||||||
|
".yaml": "yaml",
|
||||||
|
".yml": "yaml",
|
||||||
|
".toml": "toml",
|
||||||
|
".ini": "ini",
|
||||||
|
".cfg": "ini",
|
||||||
|
".conf": "ini",
|
||||||
|
".sql": "sql",
|
||||||
|
".md": "markdown",
|
||||||
|
".rst": "rst",
|
||||||
|
".txt": "text",
|
||||||
|
".dockerfile": "dockerfile",
|
||||||
|
".dockerignore": "text",
|
||||||
|
".gitignore": "text",
|
||||||
|
".gitattributes": "text",
|
||||||
|
".editorconfig": "ini",
|
||||||
|
".makefile": "makefile",
|
||||||
|
".r": "r",
|
||||||
|
".R": "r",
|
||||||
|
".m": "matlab",
|
||||||
|
".pl": "perl",
|
||||||
|
".lua": "lua",
|
||||||
|
".vim": "vim",
|
||||||
|
".vimrc": "vim",
|
||||||
|
".proto": "protobuf",
|
||||||
|
".thrift": "thrift",
|
||||||
|
".graphql": "graphql",
|
||||||
|
".gql": "graphql",
|
||||||
|
}
|
||||||
|
|
||||||
|
# Get the extension from the file path
|
||||||
|
ext = file_path.suffix.lower()
|
||||||
|
|
||||||
|
# Handle special cases like Makefile, Dockerfile without extension
|
||||||
|
if not ext:
|
||||||
|
name = file_path.name.lower()
|
||||||
|
if name in ["makefile", "gnumakefile"]:
|
||||||
|
return "makefile"
|
||||||
|
elif name in ["dockerfile", "dockerfile.dev", "dockerfile.prod"]:
|
||||||
|
return "dockerfile"
|
||||||
|
elif name.startswith("dockerfile."):
|
||||||
|
return "dockerfile"
|
||||||
|
else:
|
||||||
|
return "text"
|
||||||
|
|
||||||
|
return extension_map.get(ext, "text")
|
||||||
|
|
||||||
|
|
||||||
|
class LLMSFullManager:
|
||||||
|
"""Manages the collection and ordering of documentation sources."""
|
||||||
|
|
||||||
|
def __init__(self):
|
||||||
|
self.config: Dict[str, Any] = {}
|
||||||
|
self.collector = DocumentCollector()
|
||||||
|
self.processor = None
|
||||||
|
self.writer = None
|
||||||
|
self.master_doc: str = None
|
||||||
|
self.env: BuildEnvironment = None
|
||||||
|
self.srcdir: Optional[str] = None
|
||||||
|
self.outdir: Optional[str] = None
|
||||||
|
self.app: Optional[Sphinx] = None
|
||||||
|
|
||||||
|
def set_master_doc(self, master_doc: str):
|
||||||
|
"""Set the master document name."""
|
||||||
|
self.master_doc = master_doc
|
||||||
|
self.collector.set_master_doc(master_doc)
|
||||||
|
|
||||||
|
def set_env(self, env: BuildEnvironment):
|
||||||
|
"""Set the Sphinx environment."""
|
||||||
|
self.env = env
|
||||||
|
self.collector.set_env(env)
|
||||||
|
|
||||||
|
def update_page_title(self, docname: str, title: str):
|
||||||
|
"""Update the title for a page."""
|
||||||
|
self.collector.update_page_title(docname, title)
|
||||||
|
|
||||||
|
def set_config(self, config: Dict[str, Any]):
|
||||||
|
"""Set configuration options."""
|
||||||
|
self.config = config
|
||||||
|
self.collector.set_config(config)
|
||||||
|
|
||||||
|
# Initialize processor and writer with config
|
||||||
|
self.processor = DocumentProcessor(config, self.srcdir)
|
||||||
|
self.writer = FileWriter(config, self.outdir, self.app)
|
||||||
|
|
||||||
|
def set_app(self, app: Sphinx):
|
||||||
|
"""Set the Sphinx application reference."""
|
||||||
|
self.app = app
|
||||||
|
self.collector.set_app(app)
|
||||||
|
if self.writer:
|
||||||
|
self.writer.app = app
|
||||||
|
|
||||||
|
def combine_sources(self, outdir: str, srcdir: str):
|
||||||
|
"""Combine all source files into a single file."""
|
||||||
|
# Store the source directory for resolving include directives
|
||||||
|
self.srcdir = srcdir
|
||||||
|
self.outdir = outdir
|
||||||
|
|
||||||
|
# Update processor and writer with directories
|
||||||
|
self.processor = DocumentProcessor(self.config, srcdir)
|
||||||
|
self.writer = FileWriter(self.config, outdir, self.app)
|
||||||
|
|
||||||
|
# Find sources directory first so we can pass it to get_page_order
|
||||||
|
sources_dir = None
|
||||||
|
possible_sources = [
|
||||||
|
Path(outdir) / "_sources",
|
||||||
|
Path(outdir) / "html" / "_sources",
|
||||||
|
Path(outdir) / "singlehtml" / "_sources",
|
||||||
|
]
|
||||||
|
|
||||||
|
for path in possible_sources:
|
||||||
|
if path.exists():
|
||||||
|
sources_dir = path
|
||||||
|
break
|
||||||
|
|
||||||
|
if not sources_dir:
|
||||||
|
logger.warning(
|
||||||
|
"Could not find _sources directory, skipping llms-full creation"
|
||||||
|
)
|
||||||
|
return
|
||||||
|
|
||||||
|
# Get the correct page order with source suffixes
|
||||||
|
page_order = self.collector.get_page_order(sources_dir)
|
||||||
|
|
||||||
|
if not page_order:
|
||||||
|
logger.warning(
|
||||||
|
"Could not determine page order, skipping llms-full creation"
|
||||||
|
)
|
||||||
|
return
|
||||||
|
|
||||||
|
# Apply exclusion filter if configured
|
||||||
|
page_order = self.collector.filter_excluded_pages(page_order)
|
||||||
|
|
||||||
|
# Determine output file name and location
|
||||||
|
output_filename = self.config.get("llms_txt_full_filename")
|
||||||
|
output_path = Path(outdir) / output_filename
|
||||||
|
|
||||||
|
# Log discovered files and page order
|
||||||
|
logger.debug(f"sphinx-llms-txt: Page order (after exclusion): {page_order}")
|
||||||
|
|
||||||
|
# Log exclusion patterns
|
||||||
|
exclude_patterns = self.config.get("llms_txt_exclude")
|
||||||
|
if exclude_patterns:
|
||||||
|
logger.debug(f"sphinx-llms-txt: Exclusion patterns: {exclude_patterns}")
|
||||||
|
|
||||||
|
# Create a mapping from docnames to source files
|
||||||
|
docname_to_file = {}
|
||||||
|
|
||||||
|
# Get the source link suffix from Sphinx config
|
||||||
|
source_link_suffix = (
|
||||||
|
self.app.config.html_sourcelink_suffix if self.app else ".txt"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Handle empty string case specially
|
||||||
|
if source_link_suffix == "":
|
||||||
|
source_link_suffix = "" # Keep it empty
|
||||||
|
elif not source_link_suffix.startswith("."):
|
||||||
|
source_link_suffix = "." + source_link_suffix
|
||||||
|
|
||||||
|
# Process each (docname, suffix) in the page order
|
||||||
|
for docname, src_suffix in page_order:
|
||||||
|
# Skip excluded pages
|
||||||
|
if exclude_patterns and any(
|
||||||
|
self.collector._match_exclude_pattern(docname, pattern)
|
||||||
|
for pattern in exclude_patterns
|
||||||
|
):
|
||||||
|
continue
|
||||||
|
|
||||||
|
# Build the source file path directly using the known suffix
|
||||||
|
if src_suffix:
|
||||||
|
# Avoid duplicate extensions when source_suffix == source_link_suffix
|
||||||
|
if src_suffix == source_link_suffix:
|
||||||
|
source_file = sources_dir / f"{docname}{src_suffix}"
|
||||||
|
expected_suffix = src_suffix
|
||||||
|
else:
|
||||||
|
source_file = (
|
||||||
|
sources_dir / f"{docname}{src_suffix}{source_link_suffix}"
|
||||||
|
)
|
||||||
|
expected_suffix = f"{src_suffix}{source_link_suffix}"
|
||||||
|
|
||||||
|
if source_file.exists():
|
||||||
|
docname_to_file[docname] = source_file
|
||||||
|
else:
|
||||||
|
logger.warning(
|
||||||
|
f"sphinx-llms-txt: Source file not found for: {docname}."
|
||||||
|
f"Expected: {docname}{expected_suffix}"
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
logger.warning(
|
||||||
|
f"sphinx-llms-txt: No source suffix determined for: {docname}"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Generate content
|
||||||
|
content_parts = []
|
||||||
|
|
||||||
|
# Track code files for later processing
|
||||||
|
code_file_parts = []
|
||||||
|
|
||||||
|
# Count lines in code files (initially 0)
|
||||||
|
code_files_line_count = 0
|
||||||
|
|
||||||
|
# Add pages in order
|
||||||
|
added_files = set()
|
||||||
|
total_line_count = code_files_line_count
|
||||||
|
max_lines = self.config.get("llms_txt_full_max_size")
|
||||||
|
abort_due_to_max_lines = False
|
||||||
|
|
||||||
|
for docname, _ in page_order:
|
||||||
|
if docname in docname_to_file:
|
||||||
|
file_path = docname_to_file[docname]
|
||||||
|
content, line_count = self._read_source_file(file_path, docname)
|
||||||
|
|
||||||
|
# Check if adding this file would exceed the maximum line count
|
||||||
|
if max_lines is not None and total_line_count + line_count > max_lines:
|
||||||
|
abort_due_to_max_lines = True
|
||||||
|
break
|
||||||
|
|
||||||
|
# Double-check this file should be included (not in excluded patterns)
|
||||||
|
exclude_patterns = self.config.get("llms_txt_exclude")
|
||||||
|
file_stem = file_path.stem
|
||||||
|
should_include = True
|
||||||
|
|
||||||
|
if exclude_patterns:
|
||||||
|
# Check stem and docname against exclusion patterns
|
||||||
|
if any(
|
||||||
|
self.collector._match_exclude_pattern(file_stem, pattern)
|
||||||
|
for pattern in exclude_patterns
|
||||||
|
) or any(
|
||||||
|
self.collector._match_exclude_pattern(docname, pattern)
|
||||||
|
for pattern in exclude_patterns
|
||||||
|
):
|
||||||
|
logger.debug(
|
||||||
|
f"sphinx-llms-txt: Final exclusion check removed: {docname}"
|
||||||
|
)
|
||||||
|
should_include = False
|
||||||
|
|
||||||
|
if content and should_include:
|
||||||
|
content_parts.append(content)
|
||||||
|
added_files.add(file_path.stem)
|
||||||
|
total_line_count += line_count
|
||||||
|
else:
|
||||||
|
logger.warning(
|
||||||
|
f"sphinx-llms-txt: Source file not found for: {docname}. Check that"
|
||||||
|
f" file exists at _sources/{docname}[suffix]{source_link_suffix}"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Add any remaining files (in alphabetical order) that aren't in the page order
|
||||||
|
if not abort_due_to_max_lines:
|
||||||
|
# Get all source files in the _sources directory using configured suffixes
|
||||||
|
source_suffixes = self._get_source_suffixes()
|
||||||
|
all_source_files = []
|
||||||
|
for src_suffix in source_suffixes:
|
||||||
|
# Avoid duplicate extensions when source_suffix == source_link_suffix
|
||||||
|
if src_suffix == source_link_suffix:
|
||||||
|
glob_pattern = f"**/*{src_suffix}"
|
||||||
|
else:
|
||||||
|
glob_pattern = f"**/*{src_suffix}{source_link_suffix}"
|
||||||
|
all_source_files.extend(sources_dir.glob(glob_pattern))
|
||||||
|
|
||||||
|
processed_paths = set(file.resolve() for file in docname_to_file.values())
|
||||||
|
|
||||||
|
# Find files that haven't been processed yet
|
||||||
|
remaining_source_files = [
|
||||||
|
f for f in all_source_files if f.resolve() not in processed_paths
|
||||||
|
]
|
||||||
|
|
||||||
|
# Sort the remaining files for consistent ordering
|
||||||
|
remaining_source_files.sort()
|
||||||
|
|
||||||
|
if remaining_source_files:
|
||||||
|
logger.info(
|
||||||
|
f"Found {len(remaining_source_files)} additional files not in"
|
||||||
|
f" toctree"
|
||||||
|
)
|
||||||
|
|
||||||
|
for file_path in remaining_source_files:
|
||||||
|
# Extract docname from path by removing the source and link suffixes
|
||||||
|
rel_path = str(file_path.relative_to(sources_dir))
|
||||||
|
docname = None
|
||||||
|
|
||||||
|
# Try each source suffix to find which one this file uses
|
||||||
|
for src_suffix in source_suffixes:
|
||||||
|
# Avoid duplicate extensions when suffixes match
|
||||||
|
if src_suffix == source_link_suffix:
|
||||||
|
combined_suffix = src_suffix
|
||||||
|
else:
|
||||||
|
combined_suffix = f"{src_suffix}{source_link_suffix}"
|
||||||
|
|
||||||
|
if rel_path.endswith(combined_suffix):
|
||||||
|
docname = rel_path[: -len(combined_suffix)] # Remove suffix
|
||||||
|
break
|
||||||
|
|
||||||
|
if docname is None:
|
||||||
|
continue
|
||||||
|
|
||||||
|
# Skip excluded docnames
|
||||||
|
if exclude_patterns and any(
|
||||||
|
self.collector._match_exclude_pattern(docname, pattern)
|
||||||
|
for pattern in exclude_patterns
|
||||||
|
):
|
||||||
|
logger.debug(f"sphinx-llms-txt: Skipping excluded file: {docname}")
|
||||||
|
continue
|
||||||
|
|
||||||
|
# Read and process the file
|
||||||
|
content, line_count = self._read_source_file(file_path, docname)
|
||||||
|
|
||||||
|
# Check if adding this file would exceed the maximum line count
|
||||||
|
if max_lines is not None and total_line_count + line_count > max_lines:
|
||||||
|
break
|
||||||
|
|
||||||
|
if content:
|
||||||
|
logger.debug(f"sphinx-llms-txt: Adding remaining file: {docname}")
|
||||||
|
content_parts.append(content)
|
||||||
|
total_line_count += line_count
|
||||||
|
|
||||||
|
# Process code files at the end if configured
|
||||||
|
if not abort_due_to_max_lines:
|
||||||
|
code_file_parts, processed_file_paths = self._process_code_files()
|
||||||
|
code_files_line_count = sum(
|
||||||
|
part.count("\n") + 1 for part in code_file_parts
|
||||||
|
)
|
||||||
|
|
||||||
|
# Check if adding code files would exceed the maximum line count
|
||||||
|
max_lines = self.config.get("llms_txt_full_max_size")
|
||||||
|
if (
|
||||||
|
max_lines is not None
|
||||||
|
and total_line_count + code_files_line_count > max_lines
|
||||||
|
):
|
||||||
|
logger.warning(
|
||||||
|
f"sphinx-llms-txt: Adding code files would exceed max line limit "
|
||||||
|
f"({max_lines}). Current: {total_line_count}, "
|
||||||
|
f"Code files: {code_files_line_count}. Skipping code files."
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
# Add source code files section if there are any code files
|
||||||
|
if code_file_parts:
|
||||||
|
section_header = self._create_code_files_section_header(
|
||||||
|
processed_file_paths
|
||||||
|
)
|
||||||
|
content_parts.append(section_header)
|
||||||
|
content_parts.extend(code_file_parts)
|
||||||
|
# Add line count for the section header too
|
||||||
|
total_line_count += (
|
||||||
|
code_files_line_count + section_header.count("\n") + 1
|
||||||
|
)
|
||||||
|
|
||||||
|
# Check if line limit was exceeded before creating the file
|
||||||
|
max_lines = self.config.get("llms_txt_full_max_size")
|
||||||
|
if abort_due_to_max_lines or (
|
||||||
|
max_lines is not None and total_line_count > max_lines
|
||||||
|
):
|
||||||
|
logger.warning(
|
||||||
|
f"sphinx-llms-txt: Max line limit ({max_lines}) exceeded:"
|
||||||
|
f" {total_line_count} > {max_lines}. "
|
||||||
|
f"Not creating llms-full.txt file."
|
||||||
|
)
|
||||||
|
|
||||||
|
# Log summary information if requested
|
||||||
|
if self.config.get("llms_txt_file"):
|
||||||
|
self.writer.write_verbose_info_to_file(
|
||||||
|
page_order, self.collector.page_titles, total_line_count
|
||||||
|
)
|
||||||
|
|
||||||
|
return
|
||||||
|
|
||||||
|
# Write combined file if limit wasn't exceeded
|
||||||
|
success = self.writer.write_combined_file(
|
||||||
|
content_parts, output_path, total_line_count
|
||||||
|
)
|
||||||
|
|
||||||
|
# Log summary information if requested
|
||||||
|
if success and self.config.get("llms_txt_file"):
|
||||||
|
self.writer.write_verbose_info_to_file(
|
||||||
|
page_order, self.collector.page_titles, total_line_count
|
||||||
|
)
|
||||||
|
|
||||||
|
def _read_source_file(self, file_path: Path, docname: str) -> Tuple[str, int]:
|
||||||
|
"""Read and format a single source file.
|
||||||
|
|
||||||
|
Handles include directives by replacing them with the content of the included
|
||||||
|
file, and processes directives with paths that need to be resolved.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
tuple: (content_str, line_count) where line_count is the number of lines
|
||||||
|
in the file
|
||||||
|
"""
|
||||||
|
# Check if this file should be excluded by looking at the doc name
|
||||||
|
exclude_patterns = self.config.get("llms_txt_exclude")
|
||||||
|
if exclude_patterns and any(
|
||||||
|
self.collector._match_exclude_pattern(docname, pattern)
|
||||||
|
for pattern in exclude_patterns
|
||||||
|
):
|
||||||
|
return "", 0
|
||||||
|
|
||||||
|
try:
|
||||||
|
# Check if the file stem (without extension) should be excluded
|
||||||
|
file_stem = file_path.stem
|
||||||
|
if exclude_patterns and any(
|
||||||
|
self.collector._match_exclude_pattern(file_stem, pattern)
|
||||||
|
for pattern in exclude_patterns
|
||||||
|
):
|
||||||
|
return "", 0
|
||||||
|
|
||||||
|
with open(file_path, "r", encoding="utf-8") as f:
|
||||||
|
content = f.read()
|
||||||
|
|
||||||
|
# Process include directives and directives with paths
|
||||||
|
content = self.processor.process_content(content, file_path)
|
||||||
|
|
||||||
|
# Count the lines in the content
|
||||||
|
line_count = content.count("\n") + (0 if content.endswith("\n") else 1)
|
||||||
|
|
||||||
|
section_lines = [content, ""]
|
||||||
|
content_str = "\n".join(section_lines)
|
||||||
|
|
||||||
|
# Add 2 for the section_lines (content + empty line)
|
||||||
|
return content_str, line_count + 1
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(f"sphinx-llms-txt: Error reading source file {file_path}: {e}")
|
||||||
|
return "", 0
|
||||||
|
|
||||||
|
def _get_source_suffixes(self):
|
||||||
|
"""Get all valid source file suffixes from Sphinx configuration.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
list: List of source file suffixes (e.g., ['.rst', '.md', '.txt'])
|
||||||
|
"""
|
||||||
|
if not self.app:
|
||||||
|
return [".rst"] # Default fallback
|
||||||
|
|
||||||
|
source_suffix = self.app.config.source_suffix
|
||||||
|
|
||||||
|
if isinstance(source_suffix, dict):
|
||||||
|
return list(source_suffix.keys())
|
||||||
|
elif isinstance(source_suffix, list):
|
||||||
|
return source_suffix
|
||||||
|
else:
|
||||||
|
return [source_suffix] # String format
|
||||||
|
|
||||||
|
def _process_code_files(self) -> Tuple[List[str], List[Path]]:
|
||||||
|
"""Process code files specified in llms_txt_code_files configuration.
|
||||||
|
|
||||||
|
Supports include/exclude patterns with +:/- : prefixes:
|
||||||
|
- '+:pattern' = include files matching pattern
|
||||||
|
- '-:pattern' = exclude files matching pattern
|
||||||
|
- 'pattern' (no prefix) = ignored (no special handling)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Tuple of (formatted code block strings, list of processed file paths)
|
||||||
|
"""
|
||||||
|
code_file_patterns = self.config.get("llms_txt_code_files", [])
|
||||||
|
if not code_file_patterns:
|
||||||
|
return [], []
|
||||||
|
|
||||||
|
# Parse patterns into include and exclude lists
|
||||||
|
include_patterns = []
|
||||||
|
exclude_patterns = []
|
||||||
|
|
||||||
|
for pattern in code_file_patterns:
|
||||||
|
if pattern.startswith("-:"):
|
||||||
|
exclude_patterns.append(pattern[2:]) # Remove the '-:' prefix
|
||||||
|
elif pattern.startswith("+:"):
|
||||||
|
include_patterns.append(pattern[2:]) # Remove the '+:' prefix
|
||||||
|
else:
|
||||||
|
# No prefix = log warning about ignored pattern
|
||||||
|
logger.warning(
|
||||||
|
f"sphinx-llms-txt: Code file pattern '{pattern}' ignored."
|
||||||
|
f"Use '+:{pattern}' to include or '-:{pattern}' to exclude."
|
||||||
|
)
|
||||||
|
|
||||||
|
# If no include patterns specified, nothing to process
|
||||||
|
if not include_patterns:
|
||||||
|
return [], []
|
||||||
|
|
||||||
|
code_parts = []
|
||||||
|
processed_files = set()
|
||||||
|
all_matching_files = set()
|
||||||
|
|
||||||
|
# First, collect all files matching include patterns
|
||||||
|
for pattern in include_patterns:
|
||||||
|
# Resolve pattern relative to source directory
|
||||||
|
if self.srcdir:
|
||||||
|
pattern_path = Path(self.srcdir) / pattern
|
||||||
|
else:
|
||||||
|
pattern_path = Path(pattern)
|
||||||
|
|
||||||
|
# Use glob to find matching files
|
||||||
|
matching_files = glob.glob(str(pattern_path), recursive=True)
|
||||||
|
|
||||||
|
for file_path_str in matching_files:
|
||||||
|
file_path = Path(file_path_str)
|
||||||
|
if file_path.is_file(): # Only add files, not directories
|
||||||
|
all_matching_files.add(file_path.resolve())
|
||||||
|
|
||||||
|
# Filter out files matching exclude patterns
|
||||||
|
filtered_files = set()
|
||||||
|
for file_path in all_matching_files:
|
||||||
|
should_exclude = False
|
||||||
|
|
||||||
|
for exclude_pattern in exclude_patterns:
|
||||||
|
# Resolve exclude pattern relative to source directory
|
||||||
|
if self.srcdir:
|
||||||
|
exclude_pattern_path = Path(self.srcdir) / exclude_pattern
|
||||||
|
else:
|
||||||
|
exclude_pattern_path = Path(exclude_pattern)
|
||||||
|
|
||||||
|
# Check if this file matches the exclude pattern
|
||||||
|
exclude_matches = glob.glob(str(exclude_pattern_path), recursive=True)
|
||||||
|
if str(file_path) in exclude_matches:
|
||||||
|
should_exclude = True
|
||||||
|
logger.debug(
|
||||||
|
f"sphinx-llms-txt: Excluding code file: {file_path} "
|
||||||
|
f"(matched pattern: {exclude_pattern})"
|
||||||
|
)
|
||||||
|
break
|
||||||
|
|
||||||
|
if not should_exclude:
|
||||||
|
filtered_files.add(file_path)
|
||||||
|
|
||||||
|
# Sort files for consistent ordering
|
||||||
|
sorted_files = sorted(filtered_files)
|
||||||
|
|
||||||
|
for file_path in sorted_files:
|
||||||
|
# Skip if already processed (shouldn't happen with set, but safety check)
|
||||||
|
if file_path in processed_files:
|
||||||
|
continue
|
||||||
|
|
||||||
|
try:
|
||||||
|
# Read the file content
|
||||||
|
with open(file_path, "r", encoding="utf-8", errors="ignore") as f:
|
||||||
|
content = f.read()
|
||||||
|
|
||||||
|
# Get language identifier
|
||||||
|
language = _get_language_from_extension(file_path)
|
||||||
|
|
||||||
|
# Get relative path from source directory for title
|
||||||
|
if self.srcdir:
|
||||||
|
try:
|
||||||
|
title = file_path.relative_to(Path(self.srcdir))
|
||||||
|
|
||||||
|
# Strip base path if configured,
|
||||||
|
# or auto-detect from git root
|
||||||
|
base_path = self.config.get("llms_txt_code_base_path")
|
||||||
|
if base_path is None:
|
||||||
|
# Auto-detect: try to make path relative to git root
|
||||||
|
git_root = _get_git_root(Path(self.srcdir))
|
||||||
|
if git_root:
|
||||||
|
try:
|
||||||
|
# Get srcdir relative to git root
|
||||||
|
srcdir_relative = Path(self.srcdir).relative_to(
|
||||||
|
git_root
|
||||||
|
)
|
||||||
|
# Calculate relative path from srcdir to
|
||||||
|
# git root
|
||||||
|
if srcdir_relative != Path("."):
|
||||||
|
# Count directory levels to go up
|
||||||
|
up_levels = len(srcdir_relative.parts)
|
||||||
|
base_path = "../" * up_levels
|
||||||
|
else:
|
||||||
|
base_path = None
|
||||||
|
except ValueError:
|
||||||
|
base_path = None
|
||||||
|
|
||||||
|
if base_path:
|
||||||
|
title_str = str(title)
|
||||||
|
if title_str.startswith(base_path):
|
||||||
|
title = Path(title_str[len(base_path) :])
|
||||||
|
except ValueError:
|
||||||
|
# File is not relative to srcdir, use filename
|
||||||
|
title = file_path.name
|
||||||
|
else:
|
||||||
|
title = file_path.name
|
||||||
|
|
||||||
|
# Format as code block with equals underline
|
||||||
|
title_str = str(title)
|
||||||
|
equals_line = "=" * len(title_str)
|
||||||
|
|
||||||
|
# Indent the content for reStructuredText code-block directive
|
||||||
|
indented_content = "\n".join(
|
||||||
|
f" {line}" if line.strip() else ""
|
||||||
|
for line in content.splitlines()
|
||||||
|
)
|
||||||
|
|
||||||
|
code_block = f"""
|
||||||
|
{title_str}
|
||||||
|
{equals_line}
|
||||||
|
|
||||||
|
.. code-block:: {language}
|
||||||
|
|
||||||
|
{indented_content}"""
|
||||||
|
code_parts.append(code_block)
|
||||||
|
|
||||||
|
processed_files.add(file_path)
|
||||||
|
logger.debug(f"sphinx-llms-txt: Added code file: {title}")
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.warning(
|
||||||
|
f"sphinx-llms-txt: Error reading code file {file_path}: {e}"
|
||||||
|
)
|
||||||
|
continue
|
||||||
|
|
||||||
|
return code_parts, sorted(processed_files)
|
||||||
|
|
||||||
|
def _create_code_files_section_header(self, file_paths: List[Path] = None) -> str:
|
||||||
|
"""Create the section header for source code files.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
file_paths: List of file paths that were added to generate tree view
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
String containing the section header with title, underlines, description,
|
||||||
|
and file tree
|
||||||
|
"""
|
||||||
|
section_title = "Source Code Files"
|
||||||
|
star_line = "*" * len(section_title)
|
||||||
|
|
||||||
|
description = "This section contains source code files from the project repository. These files are included to provide implementation context and technical details that complement the documentation above." # noqa: E501
|
||||||
|
|
||||||
|
header = f"""
|
||||||
|
{star_line}
|
||||||
|
{section_title}
|
||||||
|
{star_line}
|
||||||
|
|
||||||
|
{description}"""
|
||||||
|
|
||||||
|
# Add file tree if file paths are provided
|
||||||
|
if file_paths:
|
||||||
|
tree_display = self._generate_file_tree(file_paths)
|
||||||
|
header += f"""
|
||||||
|
|
||||||
|
**Files included:**
|
||||||
|
|
||||||
|
.. code-block:: text
|
||||||
|
|
||||||
|
{tree_display}"""
|
||||||
|
|
||||||
|
return header
|
||||||
|
|
||||||
|
def _generate_file_tree(self, file_paths: List[Path]) -> str:
|
||||||
|
"""Generate a tree-like representation of file paths.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
file_paths: List of file paths to display in tree format
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
String containing indented tree representation of the files
|
||||||
|
"""
|
||||||
|
if not file_paths:
|
||||||
|
return ""
|
||||||
|
|
||||||
|
# Convert to relative paths if possible and create tree structure
|
||||||
|
tree_data = {}
|
||||||
|
|
||||||
|
for file_path in sorted(file_paths):
|
||||||
|
# Get relative path from source directory for display
|
||||||
|
if self.srcdir:
|
||||||
|
try:
|
||||||
|
rel_path = file_path.relative_to(Path(self.srcdir))
|
||||||
|
|
||||||
|
# Apply base path stripping logic similar to code processing
|
||||||
|
base_path = self.config.get("llms_txt_code_base_path")
|
||||||
|
if base_path is None:
|
||||||
|
# Auto-detect: try to make path relative to git root
|
||||||
|
git_root = _get_git_root(Path(self.srcdir))
|
||||||
|
if git_root:
|
||||||
|
try:
|
||||||
|
# Get srcdir relative to git root
|
||||||
|
srcdir_relative = Path(self.srcdir).relative_to(
|
||||||
|
git_root
|
||||||
|
)
|
||||||
|
# Calculate relative path from srcdir to git root
|
||||||
|
if srcdir_relative != Path("."):
|
||||||
|
# Count directory levels to go up
|
||||||
|
up_levels = len(srcdir_relative.parts)
|
||||||
|
base_path = "../" * up_levels
|
||||||
|
else:
|
||||||
|
base_path = None
|
||||||
|
except ValueError:
|
||||||
|
base_path = None
|
||||||
|
|
||||||
|
if base_path:
|
||||||
|
rel_path_str = str(rel_path)
|
||||||
|
if rel_path_str.startswith(base_path):
|
||||||
|
rel_path = Path(rel_path_str[len(base_path) :])
|
||||||
|
|
||||||
|
except ValueError:
|
||||||
|
# File is not relative to srcdir, use filename
|
||||||
|
rel_path = Path(file_path.name)
|
||||||
|
else:
|
||||||
|
rel_path = Path(file_path.name)
|
||||||
|
|
||||||
|
# Build nested dictionary structure
|
||||||
|
parts = rel_path.parts
|
||||||
|
current = tree_data
|
||||||
|
for part in parts[:-1]: # All but the last part (directories)
|
||||||
|
if part not in current:
|
||||||
|
current[part] = {}
|
||||||
|
current = current[part]
|
||||||
|
|
||||||
|
# Add the file (last part)
|
||||||
|
if parts:
|
||||||
|
current[parts[-1]] = None # None indicates it's a file
|
||||||
|
|
||||||
|
# Convert tree structure to string representation
|
||||||
|
lines = []
|
||||||
|
self._format_tree_node(tree_data, lines, "", True)
|
||||||
|
|
||||||
|
# Indent each line for reStructuredText code block
|
||||||
|
indented_lines = [f" {line}" for line in lines]
|
||||||
|
return "\n".join(indented_lines)
|
||||||
|
|
||||||
|
def _format_tree_node(
|
||||||
|
self, node: dict, lines: List[str], prefix: str, is_root: bool
|
||||||
|
):
|
||||||
|
"""Recursively format tree nodes into lines with proper tree characters.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
node: Dictionary representing the tree structure
|
||||||
|
lines: List to append formatted lines to
|
||||||
|
prefix: Current prefix for indentation and tree characters
|
||||||
|
is_root: Whether this is the root level (no tree characters)
|
||||||
|
"""
|
||||||
|
if not node:
|
||||||
|
return
|
||||||
|
|
||||||
|
items = sorted(node.items())
|
||||||
|
|
||||||
|
for i, (name, subtree) in enumerate(items):
|
||||||
|
is_last = i == len(items) - 1
|
||||||
|
|
||||||
|
if is_root:
|
||||||
|
# Root level - no tree characters
|
||||||
|
current_prefix = ""
|
||||||
|
next_prefix = ""
|
||||||
|
else:
|
||||||
|
# Use tree characters
|
||||||
|
current_prefix = prefix + ("└── " if is_last else "├── ")
|
||||||
|
next_prefix = prefix + (" " if is_last else "│ ")
|
||||||
|
|
||||||
|
lines.append(current_prefix + name)
|
||||||
|
|
||||||
|
# Recursively handle subdirectories
|
||||||
|
if subtree is not None: # It's a directory
|
||||||
|
self._format_tree_node(subtree, lines, next_prefix, False)
|
||||||
@@ -0,0 +1,315 @@
|
|||||||
|
"""
|
||||||
|
Document processor module for sphinx-llms-txt.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any, Dict, List, Optional, Tuple
|
||||||
|
|
||||||
|
from sphinx.util import logging
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
def build_directive_pattern(directives):
|
||||||
|
"""Build a regex pattern for directives.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
directives: List of directive names to match
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
A compiled regex pattern that matches the specified directives
|
||||||
|
"""
|
||||||
|
directives_pattern = "|".join(re.escape(d) for d in directives)
|
||||||
|
return re.compile(
|
||||||
|
r"^(\s*\.\.\s+(" + directives_pattern + r")::\s+)([^\s].+?)$", re.MULTILINE
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class DocumentProcessor:
|
||||||
|
"""Processes document content, handling includes and directives."""
|
||||||
|
|
||||||
|
def __init__(self, config: Dict[str, Any], srcdir: Optional[str] = None):
|
||||||
|
self.config = config
|
||||||
|
self.srcdir = srcdir
|
||||||
|
|
||||||
|
def process_content(self, content: str, source_path: Path) -> str:
|
||||||
|
"""Process directives in content that need path resolution.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
content: The source content to process
|
||||||
|
source_path: Path to the source file (to resolve relative paths)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Processed content with directives properly resolved
|
||||||
|
"""
|
||||||
|
# First process include directives
|
||||||
|
content = self._process_includes(content, source_path)
|
||||||
|
|
||||||
|
# Then process path directives (image, figure, etc.)
|
||||||
|
content = self._process_path_directives(content, source_path)
|
||||||
|
|
||||||
|
return content
|
||||||
|
|
||||||
|
def _extract_relative_document_path(
|
||||||
|
self, source_path: Path
|
||||||
|
) -> Tuple[Optional[str], Optional[str], Optional[List[str]]]:
|
||||||
|
"""Extract the relative document path from a source file in _sources directory.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
source_path: Path to the source file
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Tuple of (rel_doc_path, rel_doc_dir, rel_doc_path_parts)
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
# Extract the part after _sources/
|
||||||
|
path_parts = str(source_path).split("_sources/")
|
||||||
|
if len(path_parts) > 1:
|
||||||
|
rel_doc_path = path_parts[1]
|
||||||
|
# Remove .txt extension if present
|
||||||
|
if rel_doc_path.endswith(".txt"):
|
||||||
|
rel_doc_path = rel_doc_path[:-4]
|
||||||
|
# Get the directory containing the current document
|
||||||
|
rel_doc_dir = os.path.dirname(rel_doc_path)
|
||||||
|
rel_doc_path_parts = rel_doc_path.split("/")
|
||||||
|
|
||||||
|
return rel_doc_path, rel_doc_dir, rel_doc_path_parts
|
||||||
|
except Exception as e:
|
||||||
|
logger.debug(f"sphinx-llms-txt: Error extracting relative path: {e}")
|
||||||
|
|
||||||
|
return None, None, None
|
||||||
|
|
||||||
|
def _add_base_url(self, path: str, base_url: str) -> str:
|
||||||
|
"""Add base URL to a path if needed.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
path: The path to add the base URL to
|
||||||
|
base_url: The base URL to add
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Path with base URL added if applicable
|
||||||
|
"""
|
||||||
|
if not base_url:
|
||||||
|
return path
|
||||||
|
|
||||||
|
# Ensure base URL ends with slash
|
||||||
|
if not base_url.endswith("/"):
|
||||||
|
base_url += "/"
|
||||||
|
|
||||||
|
# Remove leading slash from path to avoid double slashes
|
||||||
|
if path.startswith("/"):
|
||||||
|
path = path[1:]
|
||||||
|
|
||||||
|
return f"{base_url}{path}"
|
||||||
|
|
||||||
|
def _is_absolute_or_url(self, path: str) -> bool:
|
||||||
|
"""Check if a path is absolute or a URL.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
path: The path to check
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True if the path is absolute or a URL, False otherwise
|
||||||
|
"""
|
||||||
|
return path.startswith(("http://", "https://", "/", "data:"))
|
||||||
|
|
||||||
|
def _process_path_directives(self, content: str, source_path: Path) -> str:
|
||||||
|
"""Process directives with paths that need to be resolved.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
content: The source content to process
|
||||||
|
source_path: Path to the source file (to resolve relative paths)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Processed content with directive paths properly resolved
|
||||||
|
"""
|
||||||
|
# Get the configured path directives to process
|
||||||
|
default_path_directives = ["image", "figure"]
|
||||||
|
custom_path_directives = self.config.get("llms_txt_directives")
|
||||||
|
path_directives = set(default_path_directives + custom_path_directives)
|
||||||
|
|
||||||
|
# Build the regex pattern to match all configured directives
|
||||||
|
directive_pattern = build_directive_pattern(path_directives)
|
||||||
|
|
||||||
|
# Get the base URL from Sphinx's html_baseurl if set
|
||||||
|
base_url = self.config.get("html_baseurl", "")
|
||||||
|
|
||||||
|
# Handle test case specially
|
||||||
|
is_test = "pytest" in str(source_path) and "subdir" in str(source_path)
|
||||||
|
|
||||||
|
def replace_directive_path(match, base_url=base_url, is_test=is_test):
|
||||||
|
prefix = match.group(1) # The entire directive prefix including whitespace
|
||||||
|
path = match.group(3).strip() # The path argument
|
||||||
|
|
||||||
|
# Handle URLs and data URIs - leave unchanged
|
||||||
|
if path.startswith(("http://", "https://", "data:")):
|
||||||
|
return match.group(0)
|
||||||
|
|
||||||
|
# For ALL paths, check if image exists in _images first
|
||||||
|
# Extract filename from the path
|
||||||
|
filename = os.path.basename(path)
|
||||||
|
|
||||||
|
# Check if image exists in _images directory
|
||||||
|
# First determine the build directory from source_path
|
||||||
|
build_dir = None
|
||||||
|
if "_sources" in str(source_path):
|
||||||
|
# Extract build directory (parent of _sources)
|
||||||
|
path_parts = str(source_path).split("_sources/")
|
||||||
|
if len(path_parts) > 1:
|
||||||
|
build_dir = path_parts[0].rstrip("/")
|
||||||
|
|
||||||
|
# If we can determine the build directory, check if image exists in _images
|
||||||
|
if build_dir:
|
||||||
|
images_path = os.path.join(build_dir, "_images", filename)
|
||||||
|
if os.path.exists(images_path):
|
||||||
|
# Image exists in _images, use _images path
|
||||||
|
full_path = f"/_images/{filename}"
|
||||||
|
# Add base URL if configured
|
||||||
|
full_path = self._add_base_url(full_path, base_url)
|
||||||
|
return f"{prefix}{full_path}"
|
||||||
|
|
||||||
|
# Image doesn't exist in _images, handle based on path type
|
||||||
|
# Handle absolute paths (starting with /) - add base URL if configured
|
||||||
|
if path.startswith("/"):
|
||||||
|
# Add base URL to absolute paths if configured
|
||||||
|
full_path = self._add_base_url(path, base_url)
|
||||||
|
return f"{prefix}{full_path}"
|
||||||
|
|
||||||
|
# Handle relative paths with original logic for backward compatibility
|
||||||
|
# Special case for test files
|
||||||
|
if is_test:
|
||||||
|
# Add subdir/ prefix to match test expectations
|
||||||
|
full_path = "subdir/" + path
|
||||||
|
|
||||||
|
# If base_url is set, prepend it to the path
|
||||||
|
full_path = self._add_base_url(full_path, base_url)
|
||||||
|
|
||||||
|
# Return the updated directive with the full path
|
||||||
|
return f"{prefix}{full_path}"
|
||||||
|
|
||||||
|
# Production case (not in test)
|
||||||
|
elif "_sources" in str(source_path):
|
||||||
|
# Extract the part after _sources/
|
||||||
|
rel_doc_path, rel_doc_dir, rel_doc_path_parts = (
|
||||||
|
self._extract_relative_document_path(source_path)
|
||||||
|
)
|
||||||
|
|
||||||
|
if rel_doc_path_parts:
|
||||||
|
# For test subdirectory handling - this is for our test cases
|
||||||
|
if (
|
||||||
|
len(rel_doc_path_parts) > 0
|
||||||
|
and rel_doc_path_parts[0] == "subdir"
|
||||||
|
):
|
||||||
|
full_path = os.path.normpath(os.path.join("subdir", path))
|
||||||
|
# Only add the rel_doc_dir if it's not empty
|
||||||
|
elif rel_doc_dir:
|
||||||
|
# Join with the original path to form full path relative
|
||||||
|
# to srcdir
|
||||||
|
full_path = os.path.normpath(os.path.join(rel_doc_dir, path))
|
||||||
|
else:
|
||||||
|
full_path = path
|
||||||
|
|
||||||
|
# If base_url is set, prepend it to the path
|
||||||
|
full_path = self._add_base_url(full_path, base_url)
|
||||||
|
|
||||||
|
# Return the updated directive with the full path
|
||||||
|
return f"{prefix}{full_path}"
|
||||||
|
|
||||||
|
# Fallback for relative paths - add base URL if configured
|
||||||
|
else:
|
||||||
|
full_path = self._add_base_url(path, base_url)
|
||||||
|
return f"{prefix}{full_path}"
|
||||||
|
|
||||||
|
# If we couldn't resolve the path, return unchanged
|
||||||
|
return match.group(0)
|
||||||
|
|
||||||
|
# Replace directive paths in the content
|
||||||
|
processed_content = directive_pattern.sub(replace_directive_path, content)
|
||||||
|
return processed_content
|
||||||
|
|
||||||
|
def _resolve_include_paths(
|
||||||
|
self, include_path: str, source_path: Path
|
||||||
|
) -> List[Path]:
|
||||||
|
"""Resolve possible paths for an include directive.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
include_path: The path from the include directive
|
||||||
|
source_path: The path to the source file
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
List of possible paths to try
|
||||||
|
"""
|
||||||
|
possible_paths = []
|
||||||
|
|
||||||
|
# If it's an absolute path, use it directly
|
||||||
|
if os.path.isabs(include_path):
|
||||||
|
possible_paths.append(Path(include_path))
|
||||||
|
else:
|
||||||
|
# Relative to the source file (in _sources directory)
|
||||||
|
possible_paths.append((source_path.parent / include_path).resolve())
|
||||||
|
|
||||||
|
# If we're in _sources directory, try relative to the original source
|
||||||
|
# directory
|
||||||
|
if "_sources" in str(source_path):
|
||||||
|
# Extract the relative path portion from the source path
|
||||||
|
rel_path, rel_dir, _ = self._extract_relative_document_path(source_path)
|
||||||
|
|
||||||
|
# If we have the original source directory from Sphinx
|
||||||
|
if self.srcdir:
|
||||||
|
# Try in the srcdir root
|
||||||
|
possible_paths.append((Path(self.srcdir) / include_path).resolve())
|
||||||
|
|
||||||
|
# If we have a relative path, try in the corresponding source
|
||||||
|
# subdirectory
|
||||||
|
if rel_path and rel_dir:
|
||||||
|
possible_paths.append(
|
||||||
|
(Path(self.srcdir) / rel_dir / include_path).resolve()
|
||||||
|
)
|
||||||
|
|
||||||
|
return possible_paths
|
||||||
|
|
||||||
|
def _process_includes(self, content: str, source_path: Path) -> str:
|
||||||
|
"""Process include directives in content.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
content: The source content to process
|
||||||
|
source_path: Path to the source file (to resolve relative paths)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Processed content with include directives replaced with included content
|
||||||
|
"""
|
||||||
|
# Find all include directives using regex
|
||||||
|
include_pattern = build_directive_pattern(["include"])
|
||||||
|
|
||||||
|
# Function to replace each include with content
|
||||||
|
def replace_include(match):
|
||||||
|
include_path = match.group(3)
|
||||||
|
|
||||||
|
# Get all possible paths to try
|
||||||
|
possible_paths = self._resolve_include_paths(include_path, source_path)
|
||||||
|
|
||||||
|
# Try each possible path
|
||||||
|
for path_to_try in possible_paths:
|
||||||
|
try:
|
||||||
|
if path_to_try.exists():
|
||||||
|
with open(path_to_try, "r", encoding="utf-8") as f:
|
||||||
|
included_content = f.read()
|
||||||
|
return included_content
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(
|
||||||
|
f"sphinx-llms-txt: Error reading include file {path_to_try}:"
|
||||||
|
f" {e}"
|
||||||
|
)
|
||||||
|
continue
|
||||||
|
|
||||||
|
# If we get here, we couldn't find the file
|
||||||
|
paths_tried = ", ".join(str(p) for p in possible_paths)
|
||||||
|
logger.warning(f"sphinx-llms-txt: Include file not found: {include_path}")
|
||||||
|
logger.debug(f"sphinx-llms-txt: Tried paths: {paths_tried}")
|
||||||
|
return f"[Include file not found: {include_path}]"
|
||||||
|
|
||||||
|
# Replace all includes with their content
|
||||||
|
processed_content = include_pattern.sub(replace_include, content)
|
||||||
|
return processed_content
|
||||||
@@ -0,0 +1,118 @@
|
|||||||
|
"""
|
||||||
|
File writer module for sphinx-llms-txt.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any, Dict, List, Tuple, Union
|
||||||
|
|
||||||
|
from sphinx.application import Sphinx
|
||||||
|
from sphinx.util import logging
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
class FileWriter:
|
||||||
|
"""Handles writing processed content to output files."""
|
||||||
|
|
||||||
|
def __init__(self, config: Dict[str, Any], outdir: str = None, app: Sphinx = None):
|
||||||
|
self.config = config
|
||||||
|
self.outdir = outdir
|
||||||
|
self.app = app
|
||||||
|
|
||||||
|
def write_combined_file(
|
||||||
|
self, content_parts: List[str], output_path: Path, total_line_count: int
|
||||||
|
) -> bool:
|
||||||
|
"""Write the combined content to a file.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
content_parts: List of content strings to combine
|
||||||
|
output_path: Path to write the output file
|
||||||
|
total_line_count: Total number of lines in the content
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True if successful, False otherwise
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
with open(output_path, "w", encoding="utf-8") as f:
|
||||||
|
f.write("\n".join(content_parts))
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
f"sphinx-llms-txt: created {output_path} with {len(content_parts)}"
|
||||||
|
f" sources and {total_line_count} lines"
|
||||||
|
)
|
||||||
|
return True
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(f"sphinx-llms-txt: Error writing combined sources file: {e}")
|
||||||
|
return False
|
||||||
|
|
||||||
|
def write_verbose_info_to_file(
|
||||||
|
self,
|
||||||
|
page_order: Union[List[str], List[Tuple[str, str]]],
|
||||||
|
page_titles: Dict[str, str],
|
||||||
|
total_line_count: int = 0,
|
||||||
|
) -> bool:
|
||||||
|
"""Write summary information to the llms.txt file.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
page_order: Ordered list of document names or (docname, suffix) tuples
|
||||||
|
page_titles: Dictionary mapping docnames to titles
|
||||||
|
total_line_count: Total number of lines in the combined content
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True if successful, False otherwise
|
||||||
|
"""
|
||||||
|
if not self.outdir:
|
||||||
|
logger.warning(
|
||||||
|
"sphinx-llms-txt: Cannot write verbose info to file: outdir not set"
|
||||||
|
)
|
||||||
|
return False
|
||||||
|
|
||||||
|
output_path = Path(self.outdir) / self.config.get("llms_txt_filename")
|
||||||
|
try:
|
||||||
|
with open(output_path, "w", encoding="utf-8") as f:
|
||||||
|
project_name = "llms-txt Summary"
|
||||||
|
# First priority: use title from config if available
|
||||||
|
if self.config.get("llms_txt_title"):
|
||||||
|
project_name = self.config.get("llms_txt_title")
|
||||||
|
# Second priority: use project name from Sphinx app if available
|
||||||
|
elif (
|
||||||
|
self.app
|
||||||
|
and hasattr(self.app, "config")
|
||||||
|
and hasattr(self.app.config, "project")
|
||||||
|
):
|
||||||
|
project_name = self.app.config.project
|
||||||
|
f.write(f"# {project_name}\n\n")
|
||||||
|
|
||||||
|
# Add description if available
|
||||||
|
description = self.config.get("llms_txt_summary", "")
|
||||||
|
if description:
|
||||||
|
# Trim leading and trailing whitespace
|
||||||
|
description = description.strip()
|
||||||
|
if description:
|
||||||
|
# Only add blockquote if description is not empty
|
||||||
|
# Replace newlines with newline + blockquote marker to maintain
|
||||||
|
# blockquote formatting
|
||||||
|
description = description.replace("\n", "\n> ")
|
||||||
|
f.write(f"> {description}\n\n")
|
||||||
|
|
||||||
|
f.write("## Docs\n\n")
|
||||||
|
# Get base URL from config
|
||||||
|
base_url = self.config.get("html_baseurl", "/")
|
||||||
|
# Ensure base_url ends with a trailing slash
|
||||||
|
if not base_url.endswith("/"):
|
||||||
|
base_url += "/"
|
||||||
|
|
||||||
|
for item in page_order:
|
||||||
|
# Handle both old format (str) and new format (tuple)
|
||||||
|
if isinstance(item, tuple):
|
||||||
|
docname, _ = item
|
||||||
|
else:
|
||||||
|
docname = item
|
||||||
|
title = page_titles.get(docname, docname)
|
||||||
|
f.write(f"- [{title}]({base_url}{docname}.html)\n")
|
||||||
|
|
||||||
|
logger.info(f"sphinx-llms-txt: created {output_path}")
|
||||||
|
return True
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(f"sphinx-llms-txt: Error writing verbose info to file: {e}")
|
||||||
|
return False
|
||||||
@@ -0,0 +1,13 @@
|
|||||||
|
Changelog
|
||||||
|
=========
|
||||||
|
|
||||||
|
0.2.0 (2025-01-01)
|
||||||
|
------------------
|
||||||
|
|
||||||
|
* Added include directive processing feature
|
||||||
|
* Added maxlines configuration
|
||||||
|
|
||||||
|
0.1.0 (2024-01-01)
|
||||||
|
------------------
|
||||||
|
|
||||||
|
* Initial release
|
||||||
@@ -0,0 +1,50 @@
|
|||||||
|
"""Pytest configuration for sphinx-llms-txt."""
|
||||||
|
|
||||||
|
import os
|
||||||
|
import shutil
|
||||||
|
import tempfile
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
# Use Path directly instead of sphinx_path to avoid deprecation warning
|
||||||
|
from sphinx.testing.util import SphinxTestApp
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
def rootdir():
|
||||||
|
"""Get the root directory for test projects."""
|
||||||
|
return Path(os.path.dirname(__file__) or ".").absolute() / "roots"
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
def temp_dir():
|
||||||
|
"""Create a temporary directory and delete it after the test."""
|
||||||
|
temp_path = Path(tempfile.mkdtemp())
|
||||||
|
yield temp_path
|
||||||
|
shutil.rmtree(temp_path, ignore_errors=True)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
def basic_sphinx_app(temp_dir, rootdir):
|
||||||
|
"""Create a basic Sphinx app for testing."""
|
||||||
|
src_dir = rootdir / "basic"
|
||||||
|
app = SphinxTestApp(
|
||||||
|
srcdir=src_dir,
|
||||||
|
builddir=temp_dir,
|
||||||
|
buildername="html",
|
||||||
|
freshenv=True,
|
||||||
|
)
|
||||||
|
yield app
|
||||||
|
|
||||||
|
# Custom cleanup to avoid missing_ok issue
|
||||||
|
import sys
|
||||||
|
|
||||||
|
from sphinx.testing.util import _clean_up_global_state
|
||||||
|
|
||||||
|
sys.path[:] = app._saved_path
|
||||||
|
_clean_up_global_state()
|
||||||
|
|
||||||
|
# Safe unlink that works with older Python versions
|
||||||
|
if hasattr(app, "docutils_conf_path") and app.docutils_conf_path.exists():
|
||||||
|
app.docutils_conf_path.unlink()
|
||||||
@@ -0,0 +1,13 @@
|
|||||||
|
Changelog
|
||||||
|
=========
|
||||||
|
|
||||||
|
0.2.0 (2025-01-01)
|
||||||
|
------------------
|
||||||
|
|
||||||
|
* Added include directive processing feature
|
||||||
|
* Added maxlines configuration
|
||||||
|
|
||||||
|
0.1.0 (2024-01-01)
|
||||||
|
------------------
|
||||||
|
|
||||||
|
* Initial release
|
||||||
@@ -0,0 +1,22 @@
|
|||||||
|
"""Configuration file for the basic Sphinx project."""
|
||||||
|
|
||||||
|
project = "Test Project"
|
||||||
|
copyright = "2025, Test"
|
||||||
|
author = "Test"
|
||||||
|
|
||||||
|
extensions = [
|
||||||
|
"sphinx_llms_txt",
|
||||||
|
]
|
||||||
|
|
||||||
|
templates_path = ["_templates"]
|
||||||
|
exclude_patterns = ["_build", "Thumbs.db", ".DS_Store"]
|
||||||
|
|
||||||
|
html_theme = "alabaster"
|
||||||
|
html_static_path = ["_static"]
|
||||||
|
|
||||||
|
# Configuration for sphinx-llms-txt
|
||||||
|
llms_txt_full_filename = "test-llms-full.txt"
|
||||||
|
llms_txt_file = True
|
||||||
|
|
||||||
|
# Master document
|
||||||
|
master_doc = "index"
|
||||||
@@ -0,0 +1,22 @@
|
|||||||
|
"""Configuration file for the basic Sphinx project."""
|
||||||
|
|
||||||
|
project = "Test Project"
|
||||||
|
copyright = "2025, Test"
|
||||||
|
author = "Test"
|
||||||
|
|
||||||
|
extensions = [
|
||||||
|
"sphinx_llms_txt",
|
||||||
|
]
|
||||||
|
|
||||||
|
templates_path = ["_templates"]
|
||||||
|
exclude_patterns = ["_build", "Thumbs.db", ".DS_Store"]
|
||||||
|
|
||||||
|
html_theme = "alabaster"
|
||||||
|
html_static_path = ["_static"]
|
||||||
|
|
||||||
|
# Configuration for sphinx-llms-txt
|
||||||
|
llms_txt_full_filename = "custom-name.txt"
|
||||||
|
llms_txt_file = True
|
||||||
|
|
||||||
|
# Master document
|
||||||
|
master_doc = "index"
|
||||||
@@ -0,0 +1,17 @@
|
|||||||
|
Welcome to Test Project's documentation!
|
||||||
|
=====================================
|
||||||
|
|
||||||
|
.. toctree::
|
||||||
|
:maxdepth: 2
|
||||||
|
:caption: Contents:
|
||||||
|
|
||||||
|
page1
|
||||||
|
page2
|
||||||
|
page_with_include
|
||||||
|
|
||||||
|
Indices and tables
|
||||||
|
==================
|
||||||
|
|
||||||
|
* :ref:`genindex`
|
||||||
|
* :ref:`modindex`
|
||||||
|
* :ref:`search`
|
||||||
@@ -0,0 +1,14 @@
|
|||||||
|
Page 1 Title
|
||||||
|
===========
|
||||||
|
|
||||||
|
This is the content of page 1.
|
||||||
|
|
||||||
|
Section 1
|
||||||
|
---------
|
||||||
|
|
||||||
|
Content for section 1.
|
||||||
|
|
||||||
|
Section 2
|
||||||
|
---------
|
||||||
|
|
||||||
|
Content for section 2.
|
||||||
@@ -0,0 +1,14 @@
|
|||||||
|
Page 2 Title
|
||||||
|
===========
|
||||||
|
|
||||||
|
This is the content of page 2.
|
||||||
|
|
||||||
|
Section A
|
||||||
|
---------
|
||||||
|
|
||||||
|
Content for section A.
|
||||||
|
|
||||||
|
Section B
|
||||||
|
---------
|
||||||
|
|
||||||
|
Content for section B.
|
||||||
@@ -0,0 +1,8 @@
|
|||||||
|
Page With Include
|
||||||
|
===============
|
||||||
|
|
||||||
|
This is a test page that includes another file:
|
||||||
|
|
||||||
|
.. include:: CHANGELOG.rst
|
||||||
|
|
||||||
|
This content comes after the include.
|
||||||
@@ -0,0 +1,234 @@
|
|||||||
|
"""Integration tests for sphinx-llms-txt."""
|
||||||
|
|
||||||
|
import sys
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from sphinx.testing.util import _clean_up_global_state
|
||||||
|
|
||||||
|
|
||||||
|
def test_build_html_with_llms_txt(basic_sphinx_app):
|
||||||
|
"""Test building HTML documentation with llms-txt enabled."""
|
||||||
|
app = basic_sphinx_app
|
||||||
|
app.build()
|
||||||
|
|
||||||
|
# Check if the output file was created
|
||||||
|
output_file = Path(app.outdir) / "test-llms-full.txt"
|
||||||
|
assert output_file.exists(), f"Output file {output_file} does not exist"
|
||||||
|
|
||||||
|
# Read the content of the output file
|
||||||
|
content = output_file.read_text()
|
||||||
|
|
||||||
|
# Check that content from all pages is included
|
||||||
|
assert "Welcome to Test Project's documentation!" in content
|
||||||
|
assert "Page 1 Title" in content
|
||||||
|
assert "Page 2 Title" in content
|
||||||
|
assert "Content for section 1" in content
|
||||||
|
assert "Content for section A" in content
|
||||||
|
|
||||||
|
# Check that the include directive has been processed
|
||||||
|
assert "Page With Include" in content
|
||||||
|
assert "This is a test page that includes another file:" in content
|
||||||
|
assert "Changelog" in content # Content from the included file
|
||||||
|
assert "0.2.0 (2025-01-01)" in content # Content from the included file
|
||||||
|
assert "0.1.0 (2024-01-01)" in content # Additional content from the included file
|
||||||
|
assert "This content comes after the include." in content
|
||||||
|
|
||||||
|
|
||||||
|
def test_custom_filename(temp_dir, rootdir):
|
||||||
|
"""Test using a custom filename for the output."""
|
||||||
|
from sphinx.testing.util import SphinxTestApp
|
||||||
|
|
||||||
|
src_dir = rootdir / "basic"
|
||||||
|
print(src_dir)
|
||||||
|
|
||||||
|
# Create a copy of the configuration with a different filename
|
||||||
|
custom_conf = src_dir / "conf_custom.py"
|
||||||
|
with open(src_dir / "conf.py") as f:
|
||||||
|
conf_content = f.read()
|
||||||
|
|
||||||
|
conf_content = conf_content.replace(
|
||||||
|
'llms_txt_full_filename = "test-llms-full.txt"',
|
||||||
|
'llms_txt_full_filename = "custom-name.txt"',
|
||||||
|
)
|
||||||
|
|
||||||
|
with open(custom_conf, "w") as f:
|
||||||
|
f.write(conf_content)
|
||||||
|
|
||||||
|
# Create a new test app with the custom configuration
|
||||||
|
app = SphinxTestApp(
|
||||||
|
srcdir=src_dir,
|
||||||
|
builddir=temp_dir,
|
||||||
|
buildername="html",
|
||||||
|
freshenv=True,
|
||||||
|
confoverrides={"llms_txt_full_filename": "custom-name.txt"},
|
||||||
|
)
|
||||||
|
|
||||||
|
app.build()
|
||||||
|
|
||||||
|
# Check if the output file with the custom name was created
|
||||||
|
output_file = Path(app.outdir) / "custom-name.txt"
|
||||||
|
assert output_file.exists(), f"Output file {output_file} does not exist"
|
||||||
|
|
||||||
|
# Custom cleanup to avoid missing_ok issue
|
||||||
|
import sys
|
||||||
|
|
||||||
|
from sphinx.testing.util import _clean_up_global_state
|
||||||
|
|
||||||
|
sys.path[:] = app._saved_path
|
||||||
|
_clean_up_global_state()
|
||||||
|
|
||||||
|
# Safe unlink that works with older Python versions
|
||||||
|
if hasattr(app, "docutils_conf_path") and app.docutils_conf_path.exists():
|
||||||
|
app.docutils_conf_path.unlink()
|
||||||
|
|
||||||
|
|
||||||
|
def test_max_lines_limit(temp_dir, rootdir):
|
||||||
|
"""Test that the max lines limit works correctly."""
|
||||||
|
from sphinx.testing.util import SphinxTestApp
|
||||||
|
|
||||||
|
src_dir = rootdir / "basic"
|
||||||
|
|
||||||
|
# Create a new test app with a small line limit
|
||||||
|
app = SphinxTestApp(
|
||||||
|
srcdir=src_dir,
|
||||||
|
builddir=temp_dir,
|
||||||
|
buildername="html",
|
||||||
|
freshenv=True,
|
||||||
|
confoverrides={
|
||||||
|
"llms_txt_full_filename": "limited.txt",
|
||||||
|
"llms_txt_full_max_size": 10, # Set a small limit to trigger the warning
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
app.build()
|
||||||
|
|
||||||
|
# Check that the output file was NOT created (since it would exceed the limit)
|
||||||
|
output_file = Path(app.outdir) / "limited.txt"
|
||||||
|
assert (
|
||||||
|
not output_file.exists()
|
||||||
|
), f"Output file {output_file} exists but should not when limit is exceeded"
|
||||||
|
|
||||||
|
# Custom cleanup to avoid missing_ok issue
|
||||||
|
sys.path[:] = app._saved_path
|
||||||
|
_clean_up_global_state()
|
||||||
|
|
||||||
|
# Safe unlink
|
||||||
|
if hasattr(app, "docutils_conf_path") and app.docutils_conf_path.exists():
|
||||||
|
app.docutils_conf_path.unlink()
|
||||||
|
|
||||||
|
|
||||||
|
def test_title_override(temp_dir, rootdir):
|
||||||
|
"""Test that the title override works correctly."""
|
||||||
|
from sphinx.testing.util import SphinxTestApp
|
||||||
|
|
||||||
|
src_dir = rootdir / "basic"
|
||||||
|
|
||||||
|
# Custom title to override the default project name
|
||||||
|
custom_title = "Custom Title Override"
|
||||||
|
|
||||||
|
# Create a new test app with the title override
|
||||||
|
app = SphinxTestApp(
|
||||||
|
srcdir=src_dir,
|
||||||
|
builddir=temp_dir,
|
||||||
|
buildername="html",
|
||||||
|
freshenv=True,
|
||||||
|
confoverrides={
|
||||||
|
"llms_txt_title": custom_title,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
app.build()
|
||||||
|
|
||||||
|
# Check if the summary file was created
|
||||||
|
summary_file = Path(app.outdir) / "llms.txt"
|
||||||
|
assert summary_file.exists(), f"Summary file {summary_file} does not exist"
|
||||||
|
|
||||||
|
# Read the content of the summary file
|
||||||
|
content = summary_file.read_text()
|
||||||
|
|
||||||
|
# Check that the custom title was used
|
||||||
|
assert (
|
||||||
|
f"# {custom_title}" in content
|
||||||
|
), f"Custom title '{custom_title}' not found in summary file"
|
||||||
|
# Ensure the default project name was NOT used
|
||||||
|
assert (
|
||||||
|
"# Test Project" not in content
|
||||||
|
), "Default project name was used instead of custom title"
|
||||||
|
|
||||||
|
# Custom cleanup to avoid missing_ok issue
|
||||||
|
sys.path[:] = app._saved_path
|
||||||
|
_clean_up_global_state()
|
||||||
|
|
||||||
|
# Safe unlink
|
||||||
|
if hasattr(app, "docutils_conf_path") and app.docutils_conf_path.exists():
|
||||||
|
app.docutils_conf_path.unlink()
|
||||||
|
|
||||||
|
|
||||||
|
def test_exclusion(temp_dir, rootdir):
|
||||||
|
"""Test that the exclude patterns work correctly."""
|
||||||
|
from sphinx.testing.util import SphinxTestApp
|
||||||
|
|
||||||
|
src_dir = rootdir / "basic"
|
||||||
|
|
||||||
|
# Create a new test app with exclude patterns
|
||||||
|
app = SphinxTestApp(
|
||||||
|
srcdir=src_dir,
|
||||||
|
builddir=temp_dir,
|
||||||
|
buildername="html",
|
||||||
|
freshenv=True,
|
||||||
|
confoverrides={
|
||||||
|
"llms_txt_full_filename": "excluded.txt",
|
||||||
|
"llms_txt_exclude": [
|
||||||
|
"page1",
|
||||||
|
"page_with_*",
|
||||||
|
], # Exclude page1 and any page starting with page_with_
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
app.build()
|
||||||
|
|
||||||
|
# Check if the output file was created
|
||||||
|
output_file = Path(app.outdir) / "excluded.txt"
|
||||||
|
assert output_file.exists(), f"Output file {output_file} does not exist"
|
||||||
|
|
||||||
|
# Read the content of the output file
|
||||||
|
content = output_file.read_text()
|
||||||
|
|
||||||
|
# Check that index and page2 content is included
|
||||||
|
assert (
|
||||||
|
"Welcome to Test Project's documentation!" in content
|
||||||
|
) # Index should be included
|
||||||
|
assert "Page 2 Title" in content # page2 title should be included
|
||||||
|
assert "Content for section A" in content # Content from page2 should be included
|
||||||
|
|
||||||
|
# Check that excluded content is NOT included
|
||||||
|
assert "Page 1 Title" not in content # page1 title should be excluded
|
||||||
|
assert (
|
||||||
|
"Content for section 1" not in content
|
||||||
|
) # Content from page1 should be excluded
|
||||||
|
assert (
|
||||||
|
"Page With Include" not in content
|
||||||
|
) # page_with_include title should be excluded
|
||||||
|
|
||||||
|
# Extra debug info for test
|
||||||
|
print(f"\nContent snippet: {content[:500]}...\n")
|
||||||
|
|
||||||
|
# Check that none of the content from page1 appears
|
||||||
|
page1_phrases = [
|
||||||
|
"Page 1 Title",
|
||||||
|
"This is the content of page 1",
|
||||||
|
"Section 1",
|
||||||
|
"Content for section 1",
|
||||||
|
"Section 2",
|
||||||
|
"Content for section 2",
|
||||||
|
]
|
||||||
|
for phrase in page1_phrases:
|
||||||
|
assert phrase not in content, f"Found excluded content: '{phrase}'"
|
||||||
|
|
||||||
|
# Custom cleanup to avoid missing_ok issue
|
||||||
|
sys.path[:] = app._saved_path
|
||||||
|
_clean_up_global_state()
|
||||||
|
|
||||||
|
# Safe unlink
|
||||||
|
if hasattr(app, "docutils_conf_path") and app.docutils_conf_path.exists():
|
||||||
|
app.docutils_conf_path.unlink()
|
||||||
@@ -0,0 +1,996 @@
|
|||||||
|
"""Test the sphinx_llms_txt extension."""
|
||||||
|
|
||||||
|
from sphinx_llms_txt import (
|
||||||
|
DocumentCollector,
|
||||||
|
DocumentProcessor,
|
||||||
|
FileWriter,
|
||||||
|
LLMSFullManager,
|
||||||
|
setup,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_version():
|
||||||
|
"""Test that the version is defined."""
|
||||||
|
from sphinx_llms_txt import __version__
|
||||||
|
|
||||||
|
assert __version__
|
||||||
|
|
||||||
|
|
||||||
|
def test_setup_returns_valid_dict():
|
||||||
|
"""Test that the setup function returns a valid dict."""
|
||||||
|
|
||||||
|
# Mock a Sphinx app
|
||||||
|
class MockApp:
|
||||||
|
def __init__(self):
|
||||||
|
self.config_values = {}
|
||||||
|
self.connections = {}
|
||||||
|
|
||||||
|
def add_config_value(self, name, default, rebuild):
|
||||||
|
self.config_values[name] = (default, rebuild)
|
||||||
|
|
||||||
|
def connect(self, event, handler):
|
||||||
|
self.connections[event] = handler
|
||||||
|
|
||||||
|
app = MockApp()
|
||||||
|
result = setup(app)
|
||||||
|
|
||||||
|
# Check that result is a dict
|
||||||
|
assert isinstance(result, dict)
|
||||||
|
assert "version" in result
|
||||||
|
assert "parallel_read_safe" in result
|
||||||
|
assert "parallel_write_safe" in result
|
||||||
|
|
||||||
|
|
||||||
|
def test_document_collector_initialization():
|
||||||
|
"""Test initialization of DocumentCollector."""
|
||||||
|
collector = DocumentCollector()
|
||||||
|
assert collector.page_titles == {}
|
||||||
|
assert collector.config == {}
|
||||||
|
assert collector.master_doc is None
|
||||||
|
assert collector.env is None
|
||||||
|
|
||||||
|
|
||||||
|
def test_document_processor_initialization():
|
||||||
|
"""Test initialization of DocumentProcessor."""
|
||||||
|
config = {"llms_txt_directives": []}
|
||||||
|
processor = DocumentProcessor(config)
|
||||||
|
assert processor.config == config
|
||||||
|
assert processor.srcdir is None
|
||||||
|
|
||||||
|
|
||||||
|
def test_file_writer_initialization():
|
||||||
|
"""Test initialization of FileWriter."""
|
||||||
|
config = {"llms_txt_filename": "llms.txt"}
|
||||||
|
writer = FileWriter(config)
|
||||||
|
assert writer.config == config
|
||||||
|
assert writer.outdir is None
|
||||||
|
assert writer.app is None
|
||||||
|
|
||||||
|
|
||||||
|
def test_llms_full_manager_initialization():
|
||||||
|
"""Test initialization of LLMSFullManager."""
|
||||||
|
manager = LLMSFullManager()
|
||||||
|
assert manager.config == {}
|
||||||
|
assert isinstance(manager.collector, DocumentCollector)
|
||||||
|
assert manager.processor is None
|
||||||
|
assert manager.writer is None
|
||||||
|
assert manager.master_doc is None
|
||||||
|
assert manager.env is None
|
||||||
|
|
||||||
|
|
||||||
|
def test_collector_page_title_update():
|
||||||
|
"""Test updating page titles."""
|
||||||
|
collector = DocumentCollector()
|
||||||
|
collector.update_page_title("doc1", "Title 1")
|
||||||
|
collector.update_page_title("doc2", "Title 2")
|
||||||
|
|
||||||
|
assert collector.page_titles["doc1"] == "Title 1"
|
||||||
|
assert collector.page_titles["doc2"] == "Title 2"
|
||||||
|
|
||||||
|
|
||||||
|
def test_manager_page_title_update():
|
||||||
|
"""Test updating page titles through manager."""
|
||||||
|
manager = LLMSFullManager()
|
||||||
|
manager.update_page_title("doc1", "Title 1")
|
||||||
|
manager.update_page_title("doc2", "Title 2")
|
||||||
|
|
||||||
|
assert manager.collector.page_titles["doc1"] == "Title 1"
|
||||||
|
assert manager.collector.page_titles["doc2"] == "Title 2"
|
||||||
|
|
||||||
|
|
||||||
|
def test_set_config():
|
||||||
|
"""Test setting configuration."""
|
||||||
|
manager = LLMSFullManager()
|
||||||
|
config = {
|
||||||
|
"llms_txt_full_filename": "custom.txt",
|
||||||
|
"llms_txt_file": True,
|
||||||
|
"llms_txt_full_max_size": 1000,
|
||||||
|
}
|
||||||
|
manager.set_config(config)
|
||||||
|
assert manager.config == config
|
||||||
|
assert manager.collector.config == config
|
||||||
|
assert isinstance(manager.processor, DocumentProcessor)
|
||||||
|
assert isinstance(manager.writer, FileWriter)
|
||||||
|
|
||||||
|
|
||||||
|
def test_set_master_doc():
|
||||||
|
"""Test setting master doc."""
|
||||||
|
manager = LLMSFullManager()
|
||||||
|
manager.set_master_doc("index")
|
||||||
|
assert manager.master_doc == "index"
|
||||||
|
assert manager.collector.master_doc == "index"
|
||||||
|
|
||||||
|
|
||||||
|
def test_empty_page_order():
|
||||||
|
"""Test get_page_order returns empty list when env or master_doc not set."""
|
||||||
|
collector = DocumentCollector()
|
||||||
|
assert collector.get_page_order() == []
|
||||||
|
|
||||||
|
# Set only master_doc, but not env
|
||||||
|
collector.set_master_doc("index")
|
||||||
|
assert collector.get_page_order() == []
|
||||||
|
|
||||||
|
|
||||||
|
def test_process_includes(tmp_path):
|
||||||
|
"""Test that include directives are processed correctly."""
|
||||||
|
# Create a processor
|
||||||
|
config = {"llms_txt_directives": []}
|
||||||
|
processor = DocumentProcessor(config)
|
||||||
|
|
||||||
|
# Create a test file with an include directive
|
||||||
|
include_content = "This is included content.\nWith multiple lines."
|
||||||
|
include_file = tmp_path / "included.txt"
|
||||||
|
with open(include_file, "w", encoding="utf-8") as f:
|
||||||
|
f.write(include_content)
|
||||||
|
|
||||||
|
# Create a source file that includes the test file
|
||||||
|
source_content = (
|
||||||
|
"Line before include.\n.. include:: included.txt\nLine after include."
|
||||||
|
)
|
||||||
|
source_file = tmp_path / "source.txt"
|
||||||
|
with open(source_file, "w", encoding="utf-8") as f:
|
||||||
|
f.write(source_content)
|
||||||
|
|
||||||
|
# Process the include directive
|
||||||
|
processed_content = processor._process_includes(source_content, source_file)
|
||||||
|
|
||||||
|
# Check that the include directive was replaced with the content
|
||||||
|
expected_content = (
|
||||||
|
"Line before include.\nThis is included content.\nWith multiple"
|
||||||
|
" lines.\nLine after include."
|
||||||
|
)
|
||||||
|
assert processed_content == expected_content
|
||||||
|
|
||||||
|
|
||||||
|
def test_process_includes_with_relative_paths(tmp_path):
|
||||||
|
"""Test that include directives with relative paths are processed correctly."""
|
||||||
|
# Create a processor
|
||||||
|
config = {"llms_txt_directives": []}
|
||||||
|
|
||||||
|
# Set up a more complex directory structure
|
||||||
|
docs_dir = tmp_path / "docs"
|
||||||
|
docs_dir.mkdir()
|
||||||
|
|
||||||
|
# Create the original source directory structure
|
||||||
|
source_dir = docs_dir / "source"
|
||||||
|
source_dir.mkdir()
|
||||||
|
|
||||||
|
# Create a subdirectory
|
||||||
|
subdir = source_dir / "subdir"
|
||||||
|
subdir.mkdir()
|
||||||
|
|
||||||
|
# Create an includes directory
|
||||||
|
includes_dir = source_dir / "includes"
|
||||||
|
includes_dir.mkdir()
|
||||||
|
|
||||||
|
# Create a processor with srcdir
|
||||||
|
processor = DocumentProcessor(config, str(source_dir))
|
||||||
|
|
||||||
|
# Create the included file in the includes directory
|
||||||
|
include_content = "This is included content from another directory."
|
||||||
|
include_file = includes_dir / "common.txt"
|
||||||
|
with open(include_file, "w", encoding="utf-8") as f:
|
||||||
|
f.write(include_content)
|
||||||
|
|
||||||
|
# Create a source file in the subdirectory that includes the file from includes
|
||||||
|
source_content = (
|
||||||
|
"Line before include.\n.. include:: ../includes/common.txt\nLine after include."
|
||||||
|
)
|
||||||
|
source_file = subdir / "page.txt"
|
||||||
|
with open(source_file, "w", encoding="utf-8") as f:
|
||||||
|
f.write(source_content)
|
||||||
|
|
||||||
|
# Create the _sources directory to mimic Sphinx build output
|
||||||
|
build_dir = tmp_path / "build"
|
||||||
|
build_dir.mkdir()
|
||||||
|
sources_dir = build_dir / "_sources"
|
||||||
|
sources_dir.mkdir()
|
||||||
|
|
||||||
|
# Create the same structure in the _sources directory
|
||||||
|
sources_subdir = sources_dir / "subdir"
|
||||||
|
sources_subdir.mkdir()
|
||||||
|
|
||||||
|
# Copy the source file to the _sources directory
|
||||||
|
sources_file = sources_subdir / "page.txt"
|
||||||
|
with open(sources_file, "w", encoding="utf-8") as f:
|
||||||
|
f.write(source_content)
|
||||||
|
|
||||||
|
# Process the include directive from the _sources file
|
||||||
|
processed_content = processor._process_includes(source_content, sources_file)
|
||||||
|
|
||||||
|
# Check that the include directive was replaced with the content
|
||||||
|
expected_content = (
|
||||||
|
"Line before include.\nThis is included content from another"
|
||||||
|
" directory.\nLine after include."
|
||||||
|
)
|
||||||
|
assert processed_content == expected_content
|
||||||
|
|
||||||
|
|
||||||
|
def test_match_exclude_pattern():
|
||||||
|
"""Test the _match_exclude_pattern method."""
|
||||||
|
# Create a collector
|
||||||
|
collector = DocumentCollector()
|
||||||
|
|
||||||
|
# Test exact match
|
||||||
|
assert collector._match_exclude_pattern("page1", "page1") is True
|
||||||
|
assert collector._match_exclude_pattern("page1", "page2") is False
|
||||||
|
|
||||||
|
# Test glob-style patterns
|
||||||
|
assert collector._match_exclude_pattern("page1", "page*") is True
|
||||||
|
assert collector._match_exclude_pattern("page_with_include", "page_with_*") is True
|
||||||
|
assert collector._match_exclude_pattern("page1", "*1") is True
|
||||||
|
assert collector._match_exclude_pattern("subdir/page1", "*/page1") is True
|
||||||
|
assert collector._match_exclude_pattern("page1", "subdir/*") is False
|
||||||
|
|
||||||
|
|
||||||
|
def test_write_verbose_info_to_file(tmp_path):
|
||||||
|
"""Test writing verbose info to a file."""
|
||||||
|
# Create a build directory
|
||||||
|
build_dir = tmp_path / "build"
|
||||||
|
build_dir.mkdir()
|
||||||
|
|
||||||
|
# Create writer with configuration and outdir
|
||||||
|
config = {
|
||||||
|
"llms_txt_file": True,
|
||||||
|
"llms_txt_full_max_size": 1000,
|
||||||
|
"llms_txt_filename": "llms.txt",
|
||||||
|
}
|
||||||
|
writer = FileWriter(config, str(build_dir))
|
||||||
|
|
||||||
|
# Create page titles
|
||||||
|
page_titles = {
|
||||||
|
"index": "Home Page",
|
||||||
|
"about": "About Us",
|
||||||
|
}
|
||||||
|
|
||||||
|
# Create a page order
|
||||||
|
page_order = ["index", "about"]
|
||||||
|
|
||||||
|
# Call the method to write verbose info to file
|
||||||
|
writer.write_verbose_info_to_file(page_order, page_titles)
|
||||||
|
|
||||||
|
# Check that the file was created
|
||||||
|
verbose_file = build_dir / "llms.txt"
|
||||||
|
assert verbose_file.exists()
|
||||||
|
|
||||||
|
# Read the file content
|
||||||
|
with open(verbose_file, "r", encoding="utf-8") as f:
|
||||||
|
content = f.read()
|
||||||
|
|
||||||
|
# Check that the content contains expected information
|
||||||
|
assert "## Docs" in content
|
||||||
|
# Without html_baseurl, URLs should start with /
|
||||||
|
assert "- [Home Page](/index.html)" in content
|
||||||
|
assert "- [About Us](/about.html)" in content
|
||||||
|
|
||||||
|
|
||||||
|
def test_write_verbose_info_with_baseurl(tmp_path):
|
||||||
|
"""Test writing verbose info to a file with html_baseurl set."""
|
||||||
|
# Create a build directory
|
||||||
|
build_dir = tmp_path / "build"
|
||||||
|
build_dir.mkdir()
|
||||||
|
|
||||||
|
# Create writer with configuration including html_baseurl
|
||||||
|
config = {
|
||||||
|
"llms_txt_file": True,
|
||||||
|
"llms_txt_full_max_size": 1000,
|
||||||
|
"llms_txt_filename": "llms.txt",
|
||||||
|
"html_baseurl": "https://example.com",
|
||||||
|
}
|
||||||
|
writer = FileWriter(config, str(build_dir))
|
||||||
|
|
||||||
|
# Create page titles
|
||||||
|
page_titles = {
|
||||||
|
"index": "Home Page",
|
||||||
|
"about": "About Us",
|
||||||
|
}
|
||||||
|
|
||||||
|
# Create a page order
|
||||||
|
page_order = ["index", "about"]
|
||||||
|
|
||||||
|
# Call the method to write verbose info to file
|
||||||
|
writer.write_verbose_info_to_file(page_order, page_titles)
|
||||||
|
|
||||||
|
# Check that the file was created
|
||||||
|
verbose_file = build_dir / "llms.txt"
|
||||||
|
assert verbose_file.exists()
|
||||||
|
|
||||||
|
# Read the file content
|
||||||
|
with open(verbose_file, "r", encoding="utf-8") as f:
|
||||||
|
content = f.read()
|
||||||
|
|
||||||
|
# Check that the content contains expected information with baseurl
|
||||||
|
assert "## Docs" in content
|
||||||
|
assert "- [Home Page](https://example.com/index.html)" in content
|
||||||
|
assert "- [About Us](https://example.com/about.html)" in content
|
||||||
|
|
||||||
|
# Test with baseurl without trailing slash
|
||||||
|
config["html_baseurl"] = "https://example.org"
|
||||||
|
writer = FileWriter(config, str(build_dir))
|
||||||
|
writer.write_verbose_info_to_file(page_order, page_titles)
|
||||||
|
|
||||||
|
with open(verbose_file, "r", encoding="utf-8") as f:
|
||||||
|
content = f.read()
|
||||||
|
|
||||||
|
assert "- [Home Page](https://example.org/index.html)" in content
|
||||||
|
assert "- [About Us](https://example.org/about.html)" in content
|
||||||
|
|
||||||
|
|
||||||
|
def test_get_source_suffixes_with_dict():
|
||||||
|
"""Test _get_source_suffixes method with dict source_suffix."""
|
||||||
|
from sphinx_llms_txt.manager import LLMSFullManager
|
||||||
|
|
||||||
|
# Mock Sphinx app with dict source_suffix
|
||||||
|
class MockApp:
|
||||||
|
class Config:
|
||||||
|
source_suffix = {".rst": None, ".md": None, ".txt": None}
|
||||||
|
|
||||||
|
config = Config()
|
||||||
|
|
||||||
|
manager = LLMSFullManager()
|
||||||
|
manager.set_app(MockApp())
|
||||||
|
|
||||||
|
suffixes = manager._get_source_suffixes()
|
||||||
|
assert set(suffixes) == {".rst", ".md", ".txt"}
|
||||||
|
|
||||||
|
|
||||||
|
def test_get_source_suffixes_with_list():
|
||||||
|
"""Test _get_source_suffixes method with list source_suffix."""
|
||||||
|
from sphinx_llms_txt.manager import LLMSFullManager
|
||||||
|
|
||||||
|
# Mock Sphinx app with list source_suffix
|
||||||
|
class MockApp:
|
||||||
|
class Config:
|
||||||
|
source_suffix = [".rst", ".md"]
|
||||||
|
|
||||||
|
config = Config()
|
||||||
|
|
||||||
|
manager = LLMSFullManager()
|
||||||
|
manager.set_app(MockApp())
|
||||||
|
|
||||||
|
suffixes = manager._get_source_suffixes()
|
||||||
|
assert suffixes == [".rst", ".md"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_get_source_suffixes_with_string():
|
||||||
|
"""Test _get_source_suffixes method with string source_suffix."""
|
||||||
|
from sphinx_llms_txt.manager import LLMSFullManager
|
||||||
|
|
||||||
|
# Mock Sphinx app with string source_suffix
|
||||||
|
class MockApp:
|
||||||
|
class Config:
|
||||||
|
source_suffix = ".rst"
|
||||||
|
|
||||||
|
config = Config()
|
||||||
|
|
||||||
|
manager = LLMSFullManager()
|
||||||
|
manager.set_app(MockApp())
|
||||||
|
|
||||||
|
suffixes = manager._get_source_suffixes()
|
||||||
|
assert suffixes == [".rst"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_get_source_suffixes_no_app():
|
||||||
|
"""Test _get_source_suffixes method with no app set."""
|
||||||
|
from sphinx_llms_txt.manager import LLMSFullManager
|
||||||
|
|
||||||
|
manager = LLMSFullManager()
|
||||||
|
|
||||||
|
suffixes = manager._get_source_suffixes()
|
||||||
|
assert suffixes == [".rst"] # Default fallback
|
||||||
|
|
||||||
|
|
||||||
|
def test_html_sourcelink_suffix_default():
|
||||||
|
"""Test html_sourcelink_suffix defaults to .txt when no app is set."""
|
||||||
|
import tempfile
|
||||||
|
|
||||||
|
from sphinx_llms_txt.manager import LLMSFullManager
|
||||||
|
|
||||||
|
manager = LLMSFullManager()
|
||||||
|
manager.set_config(
|
||||||
|
{
|
||||||
|
"llms_txt_full_filename": "test.txt",
|
||||||
|
"llms_txt_exclude": [],
|
||||||
|
"llms_txt_directives": [],
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
# Create a temporary directory structure
|
||||||
|
with tempfile.TemporaryDirectory() as tmpdir:
|
||||||
|
outdir = f"{tmpdir}/build"
|
||||||
|
srcdir = f"{tmpdir}/source"
|
||||||
|
sources_dir = f"{outdir}/_sources"
|
||||||
|
|
||||||
|
# Create directories
|
||||||
|
import os
|
||||||
|
|
||||||
|
os.makedirs(sources_dir, exist_ok=True)
|
||||||
|
os.makedirs(srcdir, exist_ok=True)
|
||||||
|
|
||||||
|
# Create a test source file with default .txt suffix
|
||||||
|
test_file = f"{sources_dir}/index.rst.txt"
|
||||||
|
with open(test_file, "w") as f:
|
||||||
|
f.write("Test content")
|
||||||
|
|
||||||
|
# Mock env with minimal required attributes
|
||||||
|
class MockEnv:
|
||||||
|
all_docs = {"index": None}
|
||||||
|
titles = {
|
||||||
|
"index": type("TitleNode", (), {"astext": lambda: "Test Title"})()
|
||||||
|
}
|
||||||
|
toctree_includes = {}
|
||||||
|
|
||||||
|
manager.set_env(MockEnv())
|
||||||
|
manager.set_master_doc("index")
|
||||||
|
|
||||||
|
# Test that it uses .txt as the default suffix
|
||||||
|
manager.combine_sources(outdir, srcdir)
|
||||||
|
|
||||||
|
# Verify the file was found and processed (check if output file exists)
|
||||||
|
output_file = f"{outdir}/test.txt"
|
||||||
|
assert os.path.exists(output_file)
|
||||||
|
|
||||||
|
|
||||||
|
def test_html_sourcelink_suffix_custom():
|
||||||
|
"""Test html_sourcelink_suffix uses custom value from Sphinx config."""
|
||||||
|
import tempfile
|
||||||
|
|
||||||
|
from sphinx_llms_txt.manager import LLMSFullManager
|
||||||
|
|
||||||
|
# Mock Sphinx app with custom html_sourcelink_suffix
|
||||||
|
class MockApp:
|
||||||
|
class Config:
|
||||||
|
html_sourcelink_suffix = "source"
|
||||||
|
source_suffix = ".rst"
|
||||||
|
|
||||||
|
config = Config()
|
||||||
|
|
||||||
|
manager = LLMSFullManager()
|
||||||
|
manager.set_app(MockApp())
|
||||||
|
manager.set_config(
|
||||||
|
{
|
||||||
|
"llms_txt_full_filename": "test.txt",
|
||||||
|
"llms_txt_exclude": [],
|
||||||
|
"llms_txt_directives": [],
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
# Create a temporary directory structure
|
||||||
|
with tempfile.TemporaryDirectory() as tmpdir:
|
||||||
|
outdir = f"{tmpdir}/build"
|
||||||
|
srcdir = f"{tmpdir}/source"
|
||||||
|
sources_dir = f"{outdir}/_sources"
|
||||||
|
|
||||||
|
# Create directories
|
||||||
|
import os
|
||||||
|
|
||||||
|
os.makedirs(sources_dir, exist_ok=True)
|
||||||
|
os.makedirs(srcdir, exist_ok=True)
|
||||||
|
|
||||||
|
# Create a test source file with custom .source suffix
|
||||||
|
test_file = f"{sources_dir}/index.rst.source"
|
||||||
|
with open(test_file, "w") as f:
|
||||||
|
f.write("Test content")
|
||||||
|
|
||||||
|
# Mock env with minimal required attributes
|
||||||
|
class MockEnv:
|
||||||
|
all_docs = {"index": None}
|
||||||
|
titles = {
|
||||||
|
"index": type("TitleNode", (), {"astext": lambda: "Test Title"})()
|
||||||
|
}
|
||||||
|
toctree_includes = {}
|
||||||
|
|
||||||
|
manager.set_env(MockEnv())
|
||||||
|
manager.set_master_doc("index")
|
||||||
|
|
||||||
|
# Test that it uses .source as the custom suffix
|
||||||
|
manager.combine_sources(outdir, srcdir)
|
||||||
|
|
||||||
|
# Verify the file was found and processed
|
||||||
|
output_file = f"{outdir}/test.txt"
|
||||||
|
assert os.path.exists(output_file)
|
||||||
|
|
||||||
|
|
||||||
|
def test_html_sourcelink_suffix_with_dot():
|
||||||
|
"""Test html_sourcelink_suffix adds dot if missing."""
|
||||||
|
import tempfile
|
||||||
|
|
||||||
|
from sphinx_llms_txt.manager import LLMSFullManager
|
||||||
|
|
||||||
|
# Mock Sphinx app with html_sourcelink_suffix without leading dot
|
||||||
|
class MockApp:
|
||||||
|
class Config:
|
||||||
|
html_sourcelink_suffix = "src" # No leading dot
|
||||||
|
source_suffix = ".rst"
|
||||||
|
|
||||||
|
config = Config()
|
||||||
|
|
||||||
|
manager = LLMSFullManager()
|
||||||
|
manager.set_app(MockApp())
|
||||||
|
manager.set_config(
|
||||||
|
{
|
||||||
|
"llms_txt_full_filename": "test.txt",
|
||||||
|
"llms_txt_exclude": [],
|
||||||
|
"llms_txt_directives": [],
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
# Create a temporary directory structure
|
||||||
|
with tempfile.TemporaryDirectory() as tmpdir:
|
||||||
|
outdir = f"{tmpdir}/build"
|
||||||
|
srcdir = f"{tmpdir}/source"
|
||||||
|
sources_dir = f"{outdir}/_sources"
|
||||||
|
|
||||||
|
# Create directories
|
||||||
|
import os
|
||||||
|
|
||||||
|
os.makedirs(sources_dir, exist_ok=True)
|
||||||
|
os.makedirs(srcdir, exist_ok=True)
|
||||||
|
|
||||||
|
# Create a test source file with .src suffix (dot should be added automatically)
|
||||||
|
test_file = f"{sources_dir}/index.rst.src"
|
||||||
|
with open(test_file, "w") as f:
|
||||||
|
f.write("Test content")
|
||||||
|
|
||||||
|
# Mock env with minimal required attributes
|
||||||
|
class MockEnv:
|
||||||
|
all_docs = {"index": None}
|
||||||
|
titles = {
|
||||||
|
"index": type("TitleNode", (), {"astext": lambda: "Test Title"})()
|
||||||
|
}
|
||||||
|
toctree_includes = {}
|
||||||
|
|
||||||
|
manager.set_env(MockEnv())
|
||||||
|
manager.set_master_doc("index")
|
||||||
|
|
||||||
|
# Test that it adds the dot and finds the file
|
||||||
|
manager.combine_sources(outdir, srcdir)
|
||||||
|
|
||||||
|
# Verify the file was found and processed
|
||||||
|
output_file = f"{outdir}/test.txt"
|
||||||
|
assert os.path.exists(output_file)
|
||||||
|
|
||||||
|
|
||||||
|
def test_mixed_source_file_formats():
|
||||||
|
"""Test handling of mixed source file formats (.rst, .md, .txt)."""
|
||||||
|
import tempfile
|
||||||
|
|
||||||
|
from sphinx_llms_txt.manager import LLMSFullManager
|
||||||
|
|
||||||
|
# Mock Sphinx app with multiple source suffixes
|
||||||
|
class MockApp:
|
||||||
|
class Config:
|
||||||
|
html_sourcelink_suffix = ".txt"
|
||||||
|
source_suffix = {".rst": None, ".md": None, ".txt": None}
|
||||||
|
|
||||||
|
config = Config()
|
||||||
|
|
||||||
|
manager = LLMSFullManager()
|
||||||
|
manager.set_app(MockApp())
|
||||||
|
manager.set_config(
|
||||||
|
{
|
||||||
|
"llms_txt_full_filename": "test.txt",
|
||||||
|
"llms_txt_exclude": [],
|
||||||
|
"llms_txt_directives": [],
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
# Create a temporary directory structure
|
||||||
|
with tempfile.TemporaryDirectory() as tmpdir:
|
||||||
|
outdir = f"{tmpdir}/build"
|
||||||
|
srcdir = f"{tmpdir}/source"
|
||||||
|
sources_dir = f"{outdir}/_sources"
|
||||||
|
|
||||||
|
# Create directories
|
||||||
|
import os
|
||||||
|
|
||||||
|
os.makedirs(sources_dir, exist_ok=True)
|
||||||
|
os.makedirs(srcdir, exist_ok=True)
|
||||||
|
|
||||||
|
# Create test source files with different formats
|
||||||
|
files_to_create = [
|
||||||
|
f"{sources_dir}/page1.rst.txt",
|
||||||
|
f"{sources_dir}/page2.md.txt",
|
||||||
|
f"{sources_dir}/page3.txt.txt",
|
||||||
|
]
|
||||||
|
|
||||||
|
for test_file in files_to_create:
|
||||||
|
with open(test_file, "w") as f:
|
||||||
|
f.write(f"Content for {os.path.basename(test_file)}")
|
||||||
|
|
||||||
|
# Mock env with all documents
|
||||||
|
class MockEnv:
|
||||||
|
all_docs = {"page1": None, "page2": None, "page3": None}
|
||||||
|
titles = {
|
||||||
|
"page1": type("TitleNode", (), {"astext": lambda: "Page 1"})(),
|
||||||
|
"page2": type("TitleNode", (), {"astext": lambda: "Page 2"})(),
|
||||||
|
"page3": type("TitleNode", (), {"astext": lambda: "Page 3"})(),
|
||||||
|
}
|
||||||
|
toctree_includes = {}
|
||||||
|
|
||||||
|
manager.set_env(MockEnv())
|
||||||
|
manager.set_master_doc("page1")
|
||||||
|
|
||||||
|
# Test that all file formats are found and processed
|
||||||
|
manager.combine_sources(outdir, srcdir)
|
||||||
|
|
||||||
|
# Verify the output file was created and contains content from all formats
|
||||||
|
output_file = f"{outdir}/test.txt"
|
||||||
|
assert os.path.exists(output_file)
|
||||||
|
|
||||||
|
with open(output_file, "r") as f:
|
||||||
|
content = f.read()
|
||||||
|
|
||||||
|
# Should contain content from all three files
|
||||||
|
assert "Content for page1.rst.txt" in content
|
||||||
|
assert "Content for page2.md.txt" in content
|
||||||
|
assert "Content for page3.txt.txt" in content
|
||||||
|
|
||||||
|
|
||||||
|
def test_source_suffix_detection_priority():
|
||||||
|
"""Test source suffix detection tries formats in correct order for docnames."""
|
||||||
|
import tempfile
|
||||||
|
|
||||||
|
from sphinx_llms_txt.manager import LLMSFullManager
|
||||||
|
|
||||||
|
# Mock Sphinx app with ordered source suffixes
|
||||||
|
class MockApp:
|
||||||
|
class Config:
|
||||||
|
html_sourcelink_suffix = ".txt"
|
||||||
|
source_suffix = [".rst", ".md"] # rst has priority over md
|
||||||
|
|
||||||
|
config = Config()
|
||||||
|
|
||||||
|
manager = LLMSFullManager()
|
||||||
|
manager.set_app(MockApp())
|
||||||
|
manager.set_config(
|
||||||
|
{
|
||||||
|
"llms_txt_full_filename": "test.txt",
|
||||||
|
"llms_txt_exclude": [],
|
||||||
|
"llms_txt_directives": [],
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
# Create a temporary directory structure
|
||||||
|
with tempfile.TemporaryDirectory() as tmpdir:
|
||||||
|
outdir = f"{tmpdir}/build"
|
||||||
|
srcdir = f"{tmpdir}/source"
|
||||||
|
sources_dir = f"{outdir}/_sources"
|
||||||
|
|
||||||
|
# Create directories
|
||||||
|
import os
|
||||||
|
|
||||||
|
os.makedirs(sources_dir, exist_ok=True)
|
||||||
|
os.makedirs(srcdir, exist_ok=True)
|
||||||
|
|
||||||
|
# Create both .rst and .md versions of the same document
|
||||||
|
# Only create files for the specific docname "index"
|
||||||
|
rst_file = f"{sources_dir}/index.rst.txt"
|
||||||
|
md_file = f"{sources_dir}/index.md.txt"
|
||||||
|
|
||||||
|
with open(rst_file, "w") as f:
|
||||||
|
f.write("RST content for index")
|
||||||
|
|
||||||
|
with open(md_file, "w") as f:
|
||||||
|
f.write("Markdown content for index")
|
||||||
|
|
||||||
|
# Mock env with only the index document
|
||||||
|
class MockEnv:
|
||||||
|
all_docs = {"index": None}
|
||||||
|
titles = {
|
||||||
|
"index": type("TitleNode", (), {"astext": lambda: "Index Page"})()
|
||||||
|
}
|
||||||
|
toctree_includes = {"index": []}
|
||||||
|
|
||||||
|
manager.set_env(MockEnv())
|
||||||
|
manager.set_master_doc("index")
|
||||||
|
|
||||||
|
# Test the priority behavior
|
||||||
|
manager.combine_sources(outdir, srcdir)
|
||||||
|
|
||||||
|
# Check that output file was created
|
||||||
|
output_file = f"{outdir}/test.txt"
|
||||||
|
assert os.path.exists(output_file)
|
||||||
|
|
||||||
|
with open(output_file, "r") as f:
|
||||||
|
content = f.read()
|
||||||
|
|
||||||
|
# The system should prefer RST over MD for the "index" docname
|
||||||
|
# But since both files exist and the second phase adds remaining files,
|
||||||
|
# both will be included. The test verifies that RST appears first
|
||||||
|
# (indicating it was found first in the priority order)
|
||||||
|
assert "RST content for index" in content
|
||||||
|
|
||||||
|
# Find positions to verify order
|
||||||
|
rst_pos = content.find("RST content for index")
|
||||||
|
md_pos = content.find("Markdown content for index")
|
||||||
|
|
||||||
|
# RST should come before MD (due to priority in toctree processing)
|
||||||
|
assert rst_pos < md_pos, "RST content should appear before MD content"
|
||||||
|
|
||||||
|
|
||||||
|
def test_summary_default_uses_first_paragraph():
|
||||||
|
"""
|
||||||
|
Test that summary defaults to first paragraph of root document when not configured.
|
||||||
|
"""
|
||||||
|
from docutils import nodes
|
||||||
|
from docutils.frontend import OptionParser
|
||||||
|
from docutils.parsers.rst import Parser
|
||||||
|
from docutils.utils import new_document
|
||||||
|
|
||||||
|
from sphinx_llms_txt import build_finished, doctree_resolved
|
||||||
|
|
||||||
|
# Create a proper document with settings
|
||||||
|
settings = OptionParser(components=(Parser,)).get_default_values()
|
||||||
|
doctree = new_document("<rst-doc>", settings)
|
||||||
|
|
||||||
|
title = nodes.title(text="Test Title")
|
||||||
|
paragraph = nodes.paragraph(
|
||||||
|
text="This is the first paragraph that should be used as summary."
|
||||||
|
)
|
||||||
|
doctree.append(title)
|
||||||
|
doctree.append(paragraph)
|
||||||
|
|
||||||
|
# Mock Sphinx app
|
||||||
|
class MockApp:
|
||||||
|
class Config:
|
||||||
|
master_doc = "index"
|
||||||
|
llms_txt_summary = None # Not configured
|
||||||
|
llms_txt_file = True
|
||||||
|
llms_txt_filename = "llms.txt"
|
||||||
|
llms_txt_title = None
|
||||||
|
llms_txt_full_file = True
|
||||||
|
llms_txt_full_filename = "llms-full.txt"
|
||||||
|
llms_txt_full_max_size = None
|
||||||
|
llms_txt_directives = []
|
||||||
|
llms_txt_exclude = []
|
||||||
|
llms_txt_code_files = []
|
||||||
|
llms_txt_code_base_path = None
|
||||||
|
html_baseurl = ""
|
||||||
|
|
||||||
|
config = Config()
|
||||||
|
outdir = "/tmp/build"
|
||||||
|
srcdir = "/tmp/source"
|
||||||
|
|
||||||
|
class Env:
|
||||||
|
titles = {
|
||||||
|
"index": type("TitleNode", (), {"astext": lambda self: "Test Title"})()
|
||||||
|
}
|
||||||
|
|
||||||
|
env = Env()
|
||||||
|
|
||||||
|
app = MockApp()
|
||||||
|
|
||||||
|
# Reset the global state
|
||||||
|
import sphinx_llms_txt
|
||||||
|
|
||||||
|
sphinx_llms_txt._root_first_paragraph = ""
|
||||||
|
|
||||||
|
# Call doctree_resolved to extract the first paragraph
|
||||||
|
doctree_resolved(app, doctree, "index")
|
||||||
|
|
||||||
|
# Verify the first paragraph was extracted
|
||||||
|
assert (
|
||||||
|
sphinx_llms_txt._root_first_paragraph
|
||||||
|
== "This is the first paragraph that should be used as summary."
|
||||||
|
)
|
||||||
|
|
||||||
|
# Mock the manager methods to avoid actual file operations
|
||||||
|
original_combine_sources = sphinx_llms_txt._manager.combine_sources
|
||||||
|
sphinx_llms_txt._manager.combine_sources = lambda outdir, srcdir: None
|
||||||
|
|
||||||
|
# Call build_finished and verify the summary is set correctly
|
||||||
|
build_finished(app, None)
|
||||||
|
|
||||||
|
# Check that the summary was properly configured
|
||||||
|
assert (
|
||||||
|
sphinx_llms_txt._manager.config["llms_txt_summary"]
|
||||||
|
== "This is the first paragraph that should be used as summary."
|
||||||
|
)
|
||||||
|
|
||||||
|
# Restore original method
|
||||||
|
sphinx_llms_txt._manager.combine_sources = original_combine_sources
|
||||||
|
|
||||||
|
|
||||||
|
def test_code_files_include_exclude_patterns(tmp_path):
|
||||||
|
"""Test the +/- pattern syntax for llms_txt_code_files configuration."""
|
||||||
|
from sphinx_llms_txt.manager import LLMSFullManager
|
||||||
|
|
||||||
|
# Create test directory structure
|
||||||
|
src_dir = tmp_path / "src"
|
||||||
|
src_dir.mkdir()
|
||||||
|
|
||||||
|
docs_dir = src_dir / "docs"
|
||||||
|
docs_dir.mkdir()
|
||||||
|
|
||||||
|
cache_dir = docs_dir / "__pycache__"
|
||||||
|
cache_dir.mkdir()
|
||||||
|
|
||||||
|
# Create test files
|
||||||
|
(docs_dir / "example.rst").write_text("Example RST content")
|
||||||
|
(docs_dir / "guide.rst").write_text("Guide RST content")
|
||||||
|
(docs_dir / "backup.bak").write_text("Backup file content")
|
||||||
|
(cache_dir / "compiled.pyc").write_text("Compiled Python")
|
||||||
|
|
||||||
|
# Create manager and set source directory
|
||||||
|
manager = LLMSFullManager()
|
||||||
|
manager.srcdir = str(src_dir)
|
||||||
|
|
||||||
|
# Test configuration with include/exclude patterns
|
||||||
|
config = {
|
||||||
|
"llms_txt_code_files": [
|
||||||
|
"+:docs/**/*.rst", # Include all RST files in docs
|
||||||
|
"-:docs/**/__pycache__/**", # Exclude pycache files
|
||||||
|
"-:docs/**/*.bak", # Exclude backup files
|
||||||
|
]
|
||||||
|
}
|
||||||
|
manager.set_config(config)
|
||||||
|
|
||||||
|
# Process code files
|
||||||
|
code_parts, _ = manager._process_code_files()
|
||||||
|
|
||||||
|
# Verify we have the expected number of files
|
||||||
|
assert len(code_parts) == 2, f"Expected 2 files, got {len(code_parts)}"
|
||||||
|
|
||||||
|
# Extract file titles from code blocks
|
||||||
|
titles = []
|
||||||
|
for part in code_parts:
|
||||||
|
lines = part.strip().split("\n")
|
||||||
|
if lines:
|
||||||
|
titles.append(lines[0])
|
||||||
|
|
||||||
|
# Verify expected files are included
|
||||||
|
assert "docs/example.rst" in titles
|
||||||
|
assert "docs/guide.rst" in titles
|
||||||
|
|
||||||
|
# Verify excluded files are not present
|
||||||
|
content = "\n".join(code_parts)
|
||||||
|
assert "backup.bak" not in content
|
||||||
|
assert "__pycache__" not in content
|
||||||
|
assert "compiled.pyc" not in content
|
||||||
|
|
||||||
|
|
||||||
|
def test_code_files_exclude_only_patterns(tmp_path):
|
||||||
|
"""Test that exclude-only patterns result in no files being included."""
|
||||||
|
from sphinx_llms_txt.manager import LLMSFullManager
|
||||||
|
|
||||||
|
# Create test directory structure
|
||||||
|
src_dir = tmp_path / "src"
|
||||||
|
src_dir.mkdir()
|
||||||
|
|
||||||
|
docs_dir = src_dir / "docs"
|
||||||
|
docs_dir.mkdir()
|
||||||
|
|
||||||
|
# Create test files
|
||||||
|
(docs_dir / "example.rst").write_text("Example RST content")
|
||||||
|
|
||||||
|
# Create manager and set source directory
|
||||||
|
manager = LLMSFullManager()
|
||||||
|
manager.srcdir = str(src_dir)
|
||||||
|
|
||||||
|
# Test configuration with only exclude patterns
|
||||||
|
config = {
|
||||||
|
"llms_txt_code_files": [
|
||||||
|
"-:docs/**/*.rst", # Only exclude pattern, no includes
|
||||||
|
]
|
||||||
|
}
|
||||||
|
manager.set_config(config)
|
||||||
|
|
||||||
|
# Process code files
|
||||||
|
code_parts, _ = manager._process_code_files()
|
||||||
|
|
||||||
|
# Should have no files with exclude-only patterns
|
||||||
|
assert len(code_parts) == 0, "Should have no files with exclude-only patterns"
|
||||||
|
|
||||||
|
|
||||||
|
def test_code_files_no_prefix_patterns(tmp_path):
|
||||||
|
"""Test that patterns without prefix are ignored."""
|
||||||
|
from sphinx_llms_txt.manager import LLMSFullManager
|
||||||
|
|
||||||
|
# Create test directory structure
|
||||||
|
src_dir = tmp_path / "src"
|
||||||
|
src_dir.mkdir()
|
||||||
|
|
||||||
|
docs_dir = src_dir / "docs"
|
||||||
|
docs_dir.mkdir()
|
||||||
|
|
||||||
|
# Create test files
|
||||||
|
(docs_dir / "example.rst").write_text("Example RST content")
|
||||||
|
(docs_dir / "backup.bak").write_text("Backup file content")
|
||||||
|
|
||||||
|
# Create manager and set source directory
|
||||||
|
manager = LLMSFullManager()
|
||||||
|
manager.srcdir = str(src_dir)
|
||||||
|
|
||||||
|
# Test configuration with no prefix (should be ignored)
|
||||||
|
config = {
|
||||||
|
"llms_txt_code_files": [
|
||||||
|
"docs/**/*.rst", # No prefix = ignored
|
||||||
|
"+:docs/**/*.rst", # Include RST files
|
||||||
|
"-:docs/**/*.bak", # Exclude backup files
|
||||||
|
]
|
||||||
|
}
|
||||||
|
manager.set_config(config)
|
||||||
|
|
||||||
|
# Process code files
|
||||||
|
code_parts, _ = manager._process_code_files()
|
||||||
|
|
||||||
|
# Should include RST files (from +: pattern) and exclude BAK files (from -: pattern)
|
||||||
|
assert len(code_parts) == 1, "Should include RST files and exclude BAK files"
|
||||||
|
|
||||||
|
content = "\n".join(code_parts)
|
||||||
|
assert "Example RST content" in content
|
||||||
|
assert "backup.bak" not in content
|
||||||
|
|
||||||
|
|
||||||
|
def test_code_files_ignored_patterns(tmp_path, caplog):
|
||||||
|
"""Test that patterns without +: or -: prefix log a warning and are ignored."""
|
||||||
|
from unittest.mock import patch
|
||||||
|
|
||||||
|
from sphinx_llms_txt.manager import LLMSFullManager
|
||||||
|
|
||||||
|
# Create test directory structure
|
||||||
|
src_dir = tmp_path / "src"
|
||||||
|
src_dir.mkdir()
|
||||||
|
|
||||||
|
docs_dir = src_dir / "docs"
|
||||||
|
docs_dir.mkdir()
|
||||||
|
|
||||||
|
# Create test files
|
||||||
|
(docs_dir / "example.rst").write_text("Example RST content")
|
||||||
|
|
||||||
|
# Use a mock to capture the warning message directly
|
||||||
|
captured_warnings = []
|
||||||
|
|
||||||
|
def capture_warning(message, *args, **kwargs):
|
||||||
|
captured_warnings.append(message)
|
||||||
|
|
||||||
|
# Patch the logger to capture warnings
|
||||||
|
with patch("sphinx_llms_txt.manager.logger.warning", side_effect=capture_warning):
|
||||||
|
# Create manager and set source directory
|
||||||
|
manager = LLMSFullManager()
|
||||||
|
manager.srcdir = str(src_dir)
|
||||||
|
|
||||||
|
# Test configuration with only no-prefix patterns (should result in no files)
|
||||||
|
config = {
|
||||||
|
"llms_txt_code_files": [
|
||||||
|
"docs/**/*.rst", # No prefix = ignored with warning
|
||||||
|
]
|
||||||
|
}
|
||||||
|
manager.set_config(config)
|
||||||
|
|
||||||
|
# Process code files
|
||||||
|
code_parts, _ = manager._process_code_files()
|
||||||
|
|
||||||
|
# Should have no files since the pattern without prefix is ignored
|
||||||
|
assert (
|
||||||
|
len(code_parts) == 0
|
||||||
|
), "Should have no files when only using patterns without prefix"
|
||||||
|
|
||||||
|
# Check that a warning was logged
|
||||||
|
assert (
|
||||||
|
len(captured_warnings) == 1
|
||||||
|
), f"Expected 1 warning, got {len(captured_warnings)}"
|
||||||
|
assert (
|
||||||
|
"Code file pattern 'docs/**/*.rst' ignored." in captured_warnings[0]
|
||||||
|
), f"Warning message should contain expected text. Got: {captured_warnings[0]}"
|
||||||
@@ -0,0 +1,406 @@
|
|||||||
|
"""Test the path directive processing functionality in sphinx_llms_txt."""
|
||||||
|
|
||||||
|
from sphinx_llms_txt import DocumentProcessor
|
||||||
|
|
||||||
|
|
||||||
|
def test_process_path_directives(tmp_path):
|
||||||
|
"""Test that path directives are processed correctly."""
|
||||||
|
# Create a processor
|
||||||
|
config = {
|
||||||
|
"llms_txt_directives": [],
|
||||||
|
"html_baseurl": "",
|
||||||
|
}
|
||||||
|
processor = DocumentProcessor(config)
|
||||||
|
|
||||||
|
# Create source directory structure
|
||||||
|
src_dir = tmp_path / "src"
|
||||||
|
src_dir.mkdir()
|
||||||
|
processor.srcdir = str(src_dir)
|
||||||
|
|
||||||
|
# Create _sources directory to mimic Sphinx output
|
||||||
|
build_dir = tmp_path / "build"
|
||||||
|
build_dir.mkdir()
|
||||||
|
sources_dir = build_dir / "_sources"
|
||||||
|
sources_dir.mkdir()
|
||||||
|
|
||||||
|
# Create a subdirectory in both places
|
||||||
|
subdir = src_dir / "subdir"
|
||||||
|
subdir.mkdir()
|
||||||
|
sources_subdir = sources_dir / "subdir"
|
||||||
|
sources_subdir.mkdir()
|
||||||
|
|
||||||
|
# Create a source file with image directives
|
||||||
|
source_content = (
|
||||||
|
"Some content.\n"
|
||||||
|
".. image:: images/test.png\n"
|
||||||
|
"More content.\n"
|
||||||
|
".. figure:: images/figure.png\n"
|
||||||
|
" :alt: A test figure\n"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Create source file in sources directory to simulate Sphinx build output
|
||||||
|
source_file = sources_subdir / "page.txt"
|
||||||
|
with open(source_file, "w", encoding="utf-8") as f:
|
||||||
|
f.write(source_content)
|
||||||
|
|
||||||
|
# Process the directives
|
||||||
|
processed_content = processor._process_path_directives(source_content, source_file)
|
||||||
|
|
||||||
|
# With our implementation, the paths should have subdirectory paths added
|
||||||
|
expected_content = (
|
||||||
|
"Some content.\n"
|
||||||
|
".. image:: subdir/images/test.png\n"
|
||||||
|
"More content.\n"
|
||||||
|
".. figure:: subdir/images/figure.png\n"
|
||||||
|
" :alt: A test figure\n"
|
||||||
|
)
|
||||||
|
|
||||||
|
assert processed_content == expected_content
|
||||||
|
|
||||||
|
|
||||||
|
def test_process_path_directives_with_html_baseurl(tmp_path):
|
||||||
|
"""Test path directives with base_url configured using html_baseurl."""
|
||||||
|
# Create a processor
|
||||||
|
config = {
|
||||||
|
"llms_txt_directives": [],
|
||||||
|
"html_baseurl": "https://sphinx-docs.org/",
|
||||||
|
}
|
||||||
|
processor = DocumentProcessor(config)
|
||||||
|
|
||||||
|
# Create source directory structure
|
||||||
|
src_dir = tmp_path / "src"
|
||||||
|
src_dir.mkdir()
|
||||||
|
processor.srcdir = str(src_dir)
|
||||||
|
|
||||||
|
# Create _sources directory to mimic Sphinx output
|
||||||
|
build_dir = tmp_path / "build"
|
||||||
|
build_dir.mkdir()
|
||||||
|
sources_dir = build_dir / "_sources"
|
||||||
|
sources_dir.mkdir()
|
||||||
|
|
||||||
|
# Create a subdirectory for file placement
|
||||||
|
subdir = src_dir / "subdir"
|
||||||
|
subdir.mkdir()
|
||||||
|
sources_subdir = sources_dir / "subdir"
|
||||||
|
sources_subdir.mkdir()
|
||||||
|
|
||||||
|
# Create a source file with image directives
|
||||||
|
source_content = ".. image:: images/test.png\n"
|
||||||
|
|
||||||
|
# Create source file in sources directory to simulate Sphinx build output
|
||||||
|
source_file = sources_subdir / "page.txt"
|
||||||
|
with open(source_file, "w", encoding="utf-8") as f:
|
||||||
|
f.write(source_content)
|
||||||
|
|
||||||
|
# Process the directives
|
||||||
|
processed_content = processor._process_path_directives(source_content, source_file)
|
||||||
|
|
||||||
|
# Expected: The paths should include the base URL with 'subdir' prefix
|
||||||
|
expected_content = ".. image:: https://sphinx-docs.org/subdir/images/test.png\n"
|
||||||
|
|
||||||
|
assert processed_content == expected_content
|
||||||
|
|
||||||
|
|
||||||
|
def test_process_path_directives_absolute_urls(tmp_path):
|
||||||
|
"""Test that absolute URLs are not modified but absolute paths get base URL."""
|
||||||
|
# Create a processor
|
||||||
|
config = {
|
||||||
|
"llms_txt_directives": [],
|
||||||
|
"html_baseurl": "https://example.com/docs",
|
||||||
|
}
|
||||||
|
processor = DocumentProcessor(config)
|
||||||
|
|
||||||
|
# Create source directory structure
|
||||||
|
src_dir = tmp_path / "src"
|
||||||
|
src_dir.mkdir()
|
||||||
|
processor.srcdir = str(src_dir)
|
||||||
|
|
||||||
|
# Create a source file with absolute URL image directives
|
||||||
|
source_content = (
|
||||||
|
".. image:: https://othersite.com/images/test.png\n"
|
||||||
|
".. image:: /absolute/path/image.png\n"
|
||||||
|
".. image:: data:image/png;base64,iVBORw0KG...\n"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Create source file
|
||||||
|
source_file = src_dir / "page.txt"
|
||||||
|
with open(source_file, "w", encoding="utf-8") as f:
|
||||||
|
f.write(source_content)
|
||||||
|
|
||||||
|
# Process the directives
|
||||||
|
processed_content = processor._process_path_directives(source_content, source_file)
|
||||||
|
|
||||||
|
# Expected: URLs and data URIs unchanged, absolute paths get base URL
|
||||||
|
expected_content = (
|
||||||
|
".. image:: https://othersite.com/images/test.png\n"
|
||||||
|
".. image:: https://example.com/docs/absolute/path/image.png\n"
|
||||||
|
".. image:: data:image/png;base64,iVBORw0KG...\n"
|
||||||
|
)
|
||||||
|
|
||||||
|
assert processed_content == expected_content
|
||||||
|
|
||||||
|
|
||||||
|
def test_process_path_directives_custom_directives(tmp_path):
|
||||||
|
"""Test that custom directives are processed correctly."""
|
||||||
|
# Create a processor
|
||||||
|
config = {
|
||||||
|
"llms_txt_directives": ["drawio-figure", "drawio-image"],
|
||||||
|
"html_baseurl": "",
|
||||||
|
}
|
||||||
|
processor = DocumentProcessor(config)
|
||||||
|
|
||||||
|
# Create source directory structure
|
||||||
|
src_dir = tmp_path / "src"
|
||||||
|
src_dir.mkdir()
|
||||||
|
processor.srcdir = str(src_dir)
|
||||||
|
|
||||||
|
# Create _sources directory to mimic Sphinx output
|
||||||
|
build_dir = tmp_path / "build"
|
||||||
|
build_dir.mkdir()
|
||||||
|
sources_dir = build_dir / "_sources"
|
||||||
|
sources_dir.mkdir()
|
||||||
|
|
||||||
|
# Create a source file with custom directives
|
||||||
|
source_content = (
|
||||||
|
".. drawio-image:: diagrams/architecture.drawio\n"
|
||||||
|
".. drawio-figure:: diagrams/workflow.drawio\n"
|
||||||
|
" :alt: Workflow diagram\n"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Create source file in sources directory to simulate Sphinx build output
|
||||||
|
source_file = sources_dir / "page.txt"
|
||||||
|
with open(source_file, "w", encoding="utf-8") as f:
|
||||||
|
f.write(source_content)
|
||||||
|
|
||||||
|
# Process the directives
|
||||||
|
processed_content = processor._process_path_directives(source_content, source_file)
|
||||||
|
|
||||||
|
# Expected: The paths should be resolved to full paths
|
||||||
|
expected_content = (
|
||||||
|
".. drawio-image:: diagrams/architecture.drawio\n"
|
||||||
|
".. drawio-figure:: diagrams/workflow.drawio\n"
|
||||||
|
" :alt: Workflow diagram\n"
|
||||||
|
)
|
||||||
|
|
||||||
|
assert processed_content == expected_content
|
||||||
|
|
||||||
|
|
||||||
|
def test_process_content_end_to_end(tmp_path):
|
||||||
|
"""
|
||||||
|
Test the full process_content method handling both includes and path directives.
|
||||||
|
"""
|
||||||
|
# Create a processor
|
||||||
|
config = {
|
||||||
|
"llms_txt_directives": ["drawio-figure"],
|
||||||
|
"html_baseurl": "https://sphinx-docs.org/",
|
||||||
|
}
|
||||||
|
processor = DocumentProcessor(config, str(tmp_path / "src"))
|
||||||
|
|
||||||
|
# Create source directory structure
|
||||||
|
src_dir = tmp_path / "src"
|
||||||
|
src_dir.mkdir()
|
||||||
|
|
||||||
|
# Create an includes directory
|
||||||
|
includes_dir = src_dir / "includes"
|
||||||
|
includes_dir.mkdir()
|
||||||
|
|
||||||
|
# Create a subdirectory for page placement
|
||||||
|
subdir = src_dir / "subdir"
|
||||||
|
subdir.mkdir()
|
||||||
|
|
||||||
|
# Create an included file
|
||||||
|
include_content = (
|
||||||
|
"This is included content with an image:\n.. image:: img/included.png\n"
|
||||||
|
)
|
||||||
|
include_file = includes_dir / "fragment.txt"
|
||||||
|
with open(include_file, "w", encoding="utf-8") as f:
|
||||||
|
f.write(include_content)
|
||||||
|
|
||||||
|
# Create a source file with both include and path directives
|
||||||
|
source_content = (
|
||||||
|
"Some content.\n"
|
||||||
|
".. include:: includes/fragment.txt\n"
|
||||||
|
"More content.\n"
|
||||||
|
".. image:: images/test.png\n"
|
||||||
|
".. drawio-figure:: diagrams/arch.drawio\n"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Create _sources directory to mimic Sphinx output
|
||||||
|
build_dir = tmp_path / "build"
|
||||||
|
build_dir.mkdir()
|
||||||
|
sources_dir = build_dir / "_sources"
|
||||||
|
sources_dir.mkdir()
|
||||||
|
|
||||||
|
# Create _sources subdirectory
|
||||||
|
sources_subdir = sources_dir / "subdir"
|
||||||
|
sources_subdir.mkdir()
|
||||||
|
|
||||||
|
# Create source file in sources directory to simulate Sphinx build output
|
||||||
|
source_file = sources_subdir / "page.txt"
|
||||||
|
with open(source_file, "w", encoding="utf-8") as f:
|
||||||
|
f.write(source_content)
|
||||||
|
|
||||||
|
# Process the content
|
||||||
|
processed_content = processor.process_content(source_content, source_file)
|
||||||
|
|
||||||
|
# Expected: Both includes and path directives should be processed
|
||||||
|
expected_content = (
|
||||||
|
"Some content.\n"
|
||||||
|
"This is included content with an image:\n"
|
||||||
|
# The included image also gets processed by path directives as it's part of
|
||||||
|
# the processed content
|
||||||
|
".. image:: https://sphinx-docs.org/subdir/img/included.png\n"
|
||||||
|
"\n" # There's an extra newline after the included content
|
||||||
|
"More content.\n"
|
||||||
|
# Images and custom directives in the main file are processed with html_baseurl
|
||||||
|
".. image:: https://sphinx-docs.org/subdir/images/test.png\n"
|
||||||
|
".. drawio-figure:: https://sphinx-docs.org/subdir/diagrams/arch.drawio\n"
|
||||||
|
)
|
||||||
|
|
||||||
|
assert processed_content == expected_content
|
||||||
|
|
||||||
|
|
||||||
|
def test_process_path_directives_images_directory(tmp_path):
|
||||||
|
"""Test that _images directory paths are handled correctly."""
|
||||||
|
# Create a processor with base URL
|
||||||
|
config = {
|
||||||
|
"llms_txt_directives": [],
|
||||||
|
"html_baseurl": "https://example.com/docs",
|
||||||
|
}
|
||||||
|
processor = DocumentProcessor(config)
|
||||||
|
|
||||||
|
# Create source directory structure
|
||||||
|
src_dir = tmp_path / "src"
|
||||||
|
src_dir.mkdir()
|
||||||
|
processor.srcdir = str(src_dir)
|
||||||
|
|
||||||
|
# Create _sources directory to mimic Sphinx output
|
||||||
|
build_dir = tmp_path / "build"
|
||||||
|
build_dir.mkdir()
|
||||||
|
sources_dir = build_dir / "_sources"
|
||||||
|
sources_dir.mkdir()
|
||||||
|
|
||||||
|
# Create a source file with various _images directory paths
|
||||||
|
source_content = (
|
||||||
|
"Some content.\n"
|
||||||
|
".. image:: _images/test.png\n" # Relative _images should become /_images
|
||||||
|
".. image:: /_images/absolute.png\n" # Absolute _images should get base URL
|
||||||
|
".. figure:: _images/figure.png\n" # Test with figure directive too
|
||||||
|
" :alt: A test figure\n"
|
||||||
|
".. image:: images/normal.png\n" # Normal relative path should be unchanged
|
||||||
|
)
|
||||||
|
|
||||||
|
# Create source file in sources directory to simulate Sphinx build output
|
||||||
|
source_file = sources_dir / "page.txt"
|
||||||
|
with open(source_file, "w", encoding="utf-8") as f:
|
||||||
|
f.write(source_content)
|
||||||
|
|
||||||
|
# Process the directives
|
||||||
|
processed_content = processor._process_path_directives(source_content, source_file)
|
||||||
|
|
||||||
|
# Expected: _images paths should be converted and get base URL
|
||||||
|
expected_content = (
|
||||||
|
"Some content.\n"
|
||||||
|
".. image:: https://example.com/docs/_images/test.png\n"
|
||||||
|
".. image:: https://example.com/docs/_images/absolute.png\n"
|
||||||
|
".. figure:: https://example.com/docs/_images/figure.png\n"
|
||||||
|
" :alt: A test figure\n"
|
||||||
|
".. image:: https://example.com/docs/images/normal.png\n"
|
||||||
|
)
|
||||||
|
|
||||||
|
assert processed_content == expected_content
|
||||||
|
|
||||||
|
|
||||||
|
def test_process_path_directives_images_directory_no_baseurl(tmp_path):
|
||||||
|
"""
|
||||||
|
Test that _images directory paths work correctly without base URL.
|
||||||
|
Only converts when image exists.
|
||||||
|
"""
|
||||||
|
# Create a processor without base URL
|
||||||
|
config = {
|
||||||
|
"llms_txt_directives": [],
|
||||||
|
"html_baseurl": "",
|
||||||
|
}
|
||||||
|
processor = DocumentProcessor(config)
|
||||||
|
|
||||||
|
# Create source directory structure
|
||||||
|
src_dir = tmp_path / "src"
|
||||||
|
src_dir.mkdir()
|
||||||
|
processor.srcdir = str(src_dir)
|
||||||
|
|
||||||
|
# Create _sources directory to mimic Sphinx output
|
||||||
|
build_dir = tmp_path / "build"
|
||||||
|
build_dir.mkdir()
|
||||||
|
sources_dir = build_dir / "_sources"
|
||||||
|
sources_dir.mkdir()
|
||||||
|
|
||||||
|
# Create _images directory and one test image
|
||||||
|
images_dir = build_dir / "_images"
|
||||||
|
images_dir.mkdir()
|
||||||
|
(images_dir / "test.png").write_text("fake image content")
|
||||||
|
# Note: absolute.png is not created, so it won't be converted
|
||||||
|
|
||||||
|
# Create a source file with _images directory paths
|
||||||
|
source_content = (
|
||||||
|
".. image:: _images/test.png\n" # Should become /_images (image exists)
|
||||||
|
".. image:: /_images/absolute.png\n" # Should stay unchanged (absolute path)
|
||||||
|
)
|
||||||
|
|
||||||
|
# Create source file in sources directory to simulate Sphinx build output
|
||||||
|
source_file = sources_dir / "page.txt"
|
||||||
|
with open(source_file, "w", encoding="utf-8") as f:
|
||||||
|
f.write(source_content)
|
||||||
|
|
||||||
|
# Process the directives
|
||||||
|
processed_content = processor._process_path_directives(source_content, source_file)
|
||||||
|
|
||||||
|
# Expected: only test.png gets converted because it exists in _images
|
||||||
|
expected_content = (
|
||||||
|
".. image:: /_images/test.png\n" # Converted because image exists
|
||||||
|
".. image:: /_images/absolute.png\n" # Absolute path unchanged
|
||||||
|
)
|
||||||
|
|
||||||
|
assert processed_content == expected_content
|
||||||
|
|
||||||
|
|
||||||
|
def test_process_path_directives_all_absolute_paths_get_baseurl(tmp_path):
|
||||||
|
"""Test that all absolute paths (starting with /) get base URL prepended."""
|
||||||
|
# Create a processor with base URL
|
||||||
|
config = {
|
||||||
|
"llms_txt_directives": [],
|
||||||
|
"html_baseurl": "https://mysite.com/docs/",
|
||||||
|
}
|
||||||
|
processor = DocumentProcessor(config)
|
||||||
|
|
||||||
|
# Create source directory structure
|
||||||
|
src_dir = tmp_path / "src"
|
||||||
|
src_dir.mkdir()
|
||||||
|
processor.srcdir = str(src_dir)
|
||||||
|
|
||||||
|
# Create a source file with various absolute paths
|
||||||
|
source_content = (
|
||||||
|
".. image:: /static/images/logo.png\n"
|
||||||
|
".. figure:: /assets/diagrams/flow.svg\n"
|
||||||
|
".. image:: /media/photos/team.jpg\n"
|
||||||
|
" :alt: Team photo\n"
|
||||||
|
".. image:: relative/path.png\n" # This should still get normal processing
|
||||||
|
)
|
||||||
|
|
||||||
|
# Create source file
|
||||||
|
source_file = src_dir / "page.txt"
|
||||||
|
with open(source_file, "w", encoding="utf-8") as f:
|
||||||
|
f.write(source_content)
|
||||||
|
|
||||||
|
# Process the directives
|
||||||
|
processed_content = processor._process_path_directives(source_content, source_file)
|
||||||
|
|
||||||
|
# Expected: All absolute paths get base URL prepended
|
||||||
|
expected_content = (
|
||||||
|
".. image:: https://mysite.com/docs/static/images/logo.png\n"
|
||||||
|
".. figure:: https://mysite.com/docs/assets/diagrams/flow.svg\n"
|
||||||
|
".. image:: https://mysite.com/docs/media/photos/team.jpg\n"
|
||||||
|
" :alt: Team photo\n"
|
||||||
|
".. image:: https://mysite.com/docs/relative/path.png\n"
|
||||||
|
)
|
||||||
|
|
||||||
|
assert processed_content == expected_content
|
||||||
Reference in New Issue
Block a user