Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
327ae7c2d7 | ||
|
|
10794dce65 |
@@ -10,7 +10,7 @@ jobs:
|
|||||||
pre-commit:
|
pre-commit:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v7
|
||||||
- name: Set up Python 3.10
|
- name: Set up Python 3.10
|
||||||
uses: actions/setup-python@v6
|
uses: actions/setup-python@v6
|
||||||
with:
|
with:
|
||||||
@@ -23,7 +23,7 @@ jobs:
|
|||||||
python-version: ['3.9', '3.10', '3.11', '3.12']
|
python-version: ['3.9', '3.10', '3.11', '3.12']
|
||||||
|
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v7
|
||||||
|
|
||||||
- name: Set up Python ${{ matrix.python-version }}
|
- name: Set up Python ${{ matrix.python-version }}
|
||||||
uses: actions/setup-python@v6
|
uses: actions/setup-python@v6
|
||||||
|
|||||||
@@ -1,12 +1,6 @@
|
|||||||
Changelog
|
Changelog
|
||||||
=========
|
=========
|
||||||
|
|
||||||
0.7.2
|
|
||||||
-----
|
|
||||||
|
|
||||||
- Fix encoding of Unicode characters (smart quotes, em dashes, etc.) in llms.txt output
|
|
||||||
`#65 <https://github.com/jdillard/sphinx-llms-txt/issues/65>`_
|
|
||||||
|
|
||||||
0.7.1
|
0.7.1
|
||||||
-----
|
-----
|
||||||
|
|
||||||
|
|||||||
@@ -26,6 +26,11 @@ Highlights
|
|||||||
Filter content, include source code files, or integrate with alternative output formats like Markdown for even better LLM compatibility.
|
Filter content, include source code files, or integrate with alternative output formats like Markdown for even better LLM compatibility.
|
||||||
See :doc:`getting-started` for output format options and :doc:`configuration-values` for all settings.
|
See :doc:`getting-started` for output format options and :doc:`configuration-values` for all settings.
|
||||||
|
|
||||||
|
.. seealso::
|
||||||
|
|
||||||
|
For better default output without configuration, see `sphinx-llm <https://github.com/NVIDIA/sphinx-llm>`_ from NVIDIA.
|
||||||
|
sphinx-llms-txt is best when customized with alternative output formats, content filtering, or source code inclusion.
|
||||||
|
|
||||||
.. toctree::
|
.. toctree::
|
||||||
:maxdepth: 2
|
:maxdepth: 2
|
||||||
|
|
||||||
|
|||||||
@@ -21,7 +21,7 @@ from .manager import LLMSFullManager
|
|||||||
from .processor import DocumentProcessor
|
from .processor import DocumentProcessor
|
||||||
from .writer import FileWriter
|
from .writer import FileWriter
|
||||||
|
|
||||||
__version__ = "0.7.2"
|
__version__ = "0.7.1"
|
||||||
|
|
||||||
# Export classes needed by tests
|
# Export classes needed by tests
|
||||||
__all__ = [
|
__all__ = [
|
||||||
|
|||||||
@@ -11,41 +11,6 @@ from sphinx.util import logging
|
|||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
def _fix_mojibake(text: str) -> str:
|
|
||||||
"""Fix common UTF-8/Windows-1252 mojibake in text.
|
|
||||||
|
|
||||||
Mojibake (文字化け, "character transformation") is a Japanese term for garbled text
|
|
||||||
caused by decoding bytes with the wrong character encoding.
|
|
||||||
|
|
||||||
This handles the case where UTF-8 bytes were incorrectly decoded as Windows-1252
|
|
||||||
(or Latin-1), resulting in corrupted characters like:
|
|
||||||
- ' (U+2019) becoming ’
|
|
||||||
- " (U+201C) becoming “
|
|
||||||
- " (U+201D) becoming â€
|
|
||||||
- — (U+2014) becoming â€"
|
|
||||||
- – (U+2013) becoming â€"
|
|
||||||
- … (U+2026) becoming …
|
|
||||||
|
|
||||||
Args:
|
|
||||||
text: The potentially corrupted text string
|
|
||||||
|
|
||||||
Returns:
|
|
||||||
The repaired text string, or the original if no repair was needed/possible
|
|
||||||
"""
|
|
||||||
if not text:
|
|
||||||
return text
|
|
||||||
|
|
||||||
try:
|
|
||||||
# Try to encode the text as Windows-1252 (which would succeed if it contains
|
|
||||||
# the mojibake characters) and then decode as UTF-8 (to get the original)
|
|
||||||
return text.encode("windows-1252").decode("utf-8")
|
|
||||||
except (UnicodeDecodeError, UnicodeEncodeError):
|
|
||||||
# If encoding/decoding fails, the text is either:
|
|
||||||
# - Already correct UTF-8
|
|
||||||
# - Corrupted in a different way we can't fix
|
|
||||||
return text
|
|
||||||
|
|
||||||
|
|
||||||
class FileWriter:
|
class FileWriter:
|
||||||
"""Handles writing processed content to output files."""
|
"""Handles writing processed content to output files."""
|
||||||
|
|
||||||
@@ -154,8 +119,6 @@ class FileWriter:
|
|||||||
and hasattr(self.app.config, "project")
|
and hasattr(self.app.config, "project")
|
||||||
):
|
):
|
||||||
project_name = self.app.config.project
|
project_name = self.app.config.project
|
||||||
# Fix any UTF-8/Windows-1252 mojibake in the project name
|
|
||||||
project_name = _fix_mojibake(project_name)
|
|
||||||
f.write(f"# {project_name}\n\n")
|
f.write(f"# {project_name}\n\n")
|
||||||
|
|
||||||
# Add description if available
|
# Add description if available
|
||||||
@@ -164,8 +127,6 @@ class FileWriter:
|
|||||||
# Trim leading and trailing whitespace
|
# Trim leading and trailing whitespace
|
||||||
description = description.strip()
|
description = description.strip()
|
||||||
if description:
|
if description:
|
||||||
# Fix any UTF-8/Windows-1252 mojibake in the description
|
|
||||||
description = _fix_mojibake(description)
|
|
||||||
# Only add blockquote if description is not empty
|
# Only add blockquote if description is not empty
|
||||||
# Replace newlines with newline + blockquote marker to maintain
|
# Replace newlines with newline + blockquote marker to maintain
|
||||||
# blockquote formatting
|
# blockquote formatting
|
||||||
@@ -201,8 +162,6 @@ class FileWriter:
|
|||||||
suffix = None
|
suffix = None
|
||||||
|
|
||||||
title = page_titles.get(docname, docname)
|
title = page_titles.get(docname, docname)
|
||||||
# Fix any UTF-8/Windows-1252 mojibake in the title
|
|
||||||
title = _fix_mojibake(title)
|
|
||||||
|
|
||||||
uri = uri_template.format(
|
uri = uri_template.format(
|
||||||
base_url=base_url,
|
base_url=base_url,
|
||||||
|
|||||||
@@ -1182,48 +1182,3 @@ def test_llms_txt_no_warning_when_full_file_disabled(tmp_path, caplog):
|
|||||||
# Verify llms.txt was still created
|
# Verify llms.txt was still created
|
||||||
llms_txt = outdir / "llms.txt"
|
llms_txt = outdir / "llms.txt"
|
||||||
assert llms_txt.exists()
|
assert llms_txt.exists()
|
||||||
|
|
||||||
|
|
||||||
def test_fix_mojibake():
|
|
||||||
"""
|
|
||||||
Test that the _fix_mojibake function correctly repairs UTF-8/Windows-1252 mojibake.
|
|
||||||
"""
|
|
||||||
from sphinx_llms_txt.writer import _fix_mojibake
|
|
||||||
|
|
||||||
# Test case from issue #65: smart apostrophe corrupted
|
|
||||||
# U+2019 (') encoded as UTF-8 (E2 80 99) then decoded as Windows-1252 gives ’
|
|
||||||
# In Windows-1252: E2->â(U+00E2), 80->€(U+20AC), 99->™(U+2122)
|
|
||||||
corrupted = "What\u00e2\u20ac\u2122s New" # ’
|
|
||||||
expected = "What\u2019s New" # ' = U+2019
|
|
||||||
assert _fix_mojibake(corrupted) == expected
|
|
||||||
|
|
||||||
# Test left double quote: U+201C (") -> “
|
|
||||||
# U+201C encoded as UTF-8: E2 80 9C
|
|
||||||
# In Windows-1252: E2->â(U+00E2), 80->€(U+20AC), 9C->œ(U+0153)
|
|
||||||
corrupted_ldq = "He said \u00e2\u20ac\u0153Hello"
|
|
||||||
expected_ldq = "He said \u201cHello"
|
|
||||||
assert _fix_mojibake(corrupted_ldq) == expected_ldq
|
|
||||||
|
|
||||||
# Test em dash: U+2014 (—) -> â€"
|
|
||||||
# U+2014 encoded as UTF-8: E2 80 94
|
|
||||||
# In Windows-1252: E2->â(U+00E2), 80->€(U+20AC), 94->"(U+201D)
|
|
||||||
corrupted_emdash = "one\u00e2\u20ac\u201dtwo"
|
|
||||||
expected_emdash = "one\u2014two"
|
|
||||||
assert _fix_mojibake(corrupted_emdash) == expected_emdash
|
|
||||||
|
|
||||||
# Test that already correct text is not modified
|
|
||||||
# Using Unicode escape for smart apostrophe
|
|
||||||
correct = "What\u2019s New In Our Latest Release!"
|
|
||||||
assert _fix_mojibake(correct) == correct
|
|
||||||
|
|
||||||
# Test empty string
|
|
||||||
assert _fix_mojibake("") == ""
|
|
||||||
|
|
||||||
# Test plain ASCII text passes through unchanged
|
|
||||||
plain = "Hello World"
|
|
||||||
assert _fix_mojibake(plain) == plain
|
|
||||||
|
|
||||||
# Test mixed mojibake and normal text
|
|
||||||
mixed = "Here\u00e2\u20ac\u2122s a test"
|
|
||||||
expected_mixed = "Here\u2019s a test"
|
|
||||||
assert _fix_mojibake(mixed) == expected_mixed
|
|
||||||
|
|||||||
Reference in New Issue
Block a user