Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 6 additions & 1 deletion docs-site/mkdocs.yml
Original file line number Diff line number Diff line change
Expand Up @@ -46,7 +46,12 @@ markdown_extensions:
- pymdownx.highlight

plugins:
- search
# Chinese runs no spaces, so jieba segments it at build time and marks
# each boundary with a zero-width space. The separator has to break on
# that mark: JavaScript's \s stops at \u200a and would leave every run
# as a single token.
- search:
separator: '[\s\u200b\-]+'
- i18n:
docs_structure: suffix
fallback_to_default: true
Expand Down
1 change: 1 addition & 0 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -555,6 +555,7 @@ dev = [
"ty==0.0.80",
]
docs = [
"jieba>=0.42.1",
"mkdocs-git-revision-date-localized-plugin>=1.6.0",
"mkdocs-material>=9.5",
"mkdocs-static-i18n>=1.0",
Expand Down
38 changes: 37 additions & 1 deletion tests/integration/test_docs_site_controls_e2e.py
Original file line number Diff line number Diff line change
Expand Up @@ -159,7 +159,10 @@ def test_search_results_stay_inside_the_rail_panel(page: Page, site: str) -> Non
const h = document.querySelector('h1').cloneNode(true);
const anchor = h.querySelector('.headerlink');
if (anchor) anchor.remove();
return h.textContent.trim();
// the Chinese build marks its word boundaries with a zero-width space, and
// a query carrying them matches the stored token exactly -- which is a
// thing no reader can type, so the test would prove nothing
return h.textContent.replace(/\u200b/g, '').trim();
}"""


Expand Down Expand Up @@ -333,6 +336,39 @@ def test_the_search_box_does_not_animate_between_its_two_states(page: Page, site
assert not timed, f"{state} search still animates: " + "; ".join(timed)


CHINESE_WORD = """async () => {
const response = await fetch('search/search_index.json');
const index = await response.json();
for (const entry of index.docs) {
const words = (entry.title || '').split('\u200b').filter(Boolean);
// a word from the middle of a title: typing it can only find the page if
// the run was indexed as words, not kept whole
if (words.length > 2) return {word: words[1], location: entry.location};
}
return null;
}"""


def test_a_chinese_reader_finds_a_page_by_typing_one_word(page: Page, site: str) -> None:
"""Chinese runs no spaces. The build segments it and marks each boundary,
and the theme's separator breaks on that mark; miss either and the whole
run is one token, which only answers a reader who types the run entire.
The word searched here is taken from the middle of a title, so it is a
word the segmenter found rather than anything written into this file."""
_open(page, site, "zh/")
found = page.evaluate(CHINESE_WORD)
assert found, "no Chinese title carries word boundaries: the build is not segmenting"

page.click(".md-search__input")
page.fill(".md-search__input", found["word"])
_await_results(page, found["word"])
hrefs = page.evaluate(SEARCH_RESULTS)
assert hrefs, f"a word from the middle of {found['location']!r} finds nothing"
assert any(found["location"] in href for href in hrefs), (
f"the page the word came from is not among its own results: {hrefs[:3]}"
)


TOC_FOLLOW = """() => {
const wrap = document.querySelector('.md-sidebar--secondary .md-sidebar__scrollwrap');
const links = [...wrap.querySelectorAll('a.md-nav__link')];
Expand Down
28 changes: 28 additions & 0 deletions tests/integration/test_docs_site_search_e2e.py
Original file line number Diff line number Diff line change
Expand Up @@ -15,6 +15,7 @@
from __future__ import annotations

import json
import re
import shutil
import socket
import subprocess
Expand All @@ -28,6 +29,9 @@

DOCS = Path(__file__).resolve().parents[2] / "docs-site"
ZH = "zh/"
#: What the segmenter puts between two Chinese words: a zero-width space,
#: the only word boundary the text has.
BOUNDARY = "\u200b"
READY_TIMEOUT = 120.0


Expand Down Expand Up @@ -125,6 +129,7 @@ def test_a_built_site_gives_each_language_its_own_index(built: Path) -> None:
)


@pytest.mark.production_timing # the wait is a real server starting, not a delay the suite can shorten
def test_the_preview_server_gives_each_language_its_own_index(preview: str) -> None:
english_status, english = _fetch(preview + "search/search_index.json")
assert english_status == 200, f"the English index is not served: HTTP {english_status}"
Expand All @@ -133,3 +138,26 @@ def test_the_preview_server_gives_each_language_its_own_index(preview: str) -> N
f"the preview serves no Chinese index (HTTP {chinese_status}), so Chinese pages read the English one"
)
_assert_split(_locations(english), _locations(chinese))


def _searchable(payload: bytes) -> tuple[int, str]:
index = json.loads(payload)
marked = sum(1 for e in index["docs"] if BOUNDARY in (e.get("title", "") + e.get("text", "")))
return marked, index["config"]["separator"]


def test_the_chinese_pages_are_searchable_by_word(built: Path) -> None:
"""Chinese runs no spaces, so nothing in the text says where one word ends.
Two things have to line up for a reader to find anything: the build has to
segment the text and mark the boundaries, and the theme's separator has to
treat that mark as a break. Either one alone leaves the whole run as a
single token, which only matches a reader who types the entire run."""
payload = (built / ZH / "search" / "search_index.json").read_bytes()
marked, separator = _searchable(payload)
total = len(json.loads(payload)["docs"])
assert marked == total, (
f"only {marked} of {total} Chinese entries carry word boundaries; the build is not segmenting the text"
)
assert re.search(separator, BOUNDARY), (
f"the separator {separator!r} does not break on the word boundary, so every Chinese run stays one token"
)
2 changes: 2 additions & 0 deletions uv.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

Loading