docs: preserve English heading anchors in translated pages (#4580)

This commit is contained in:
saime428
2026-08-21 18:52:33 -10:00
committed by GitHub
parent 904bc6988f
commit 60c2c4120e
2 changed files with 268 additions and 0 deletions
+139
View File
@@ -6,7 +6,12 @@ import subprocess
import sys
from collections import Counter
from pathlib import Path
from typing import Any
from openai import OpenAI
from markdown import Markdown
from markdown.blockprocessors import HashHeaderProcessor
from markdown.extensions.attr_list import AttrListTreeprocessor
from mkdocs.utils import yaml_load
from concurrent.futures import ThreadPoolExecutor
# import logging
@@ -347,6 +352,120 @@ def remove_fenced_code_blocks(markdown: str) -> str:
return "".join(parts)
# The parser's own grammars, so nothing here has to agree with them by hand.
ATX_HEADING_RE = HashHeaderProcessor.RE
ATTR_LIST_RE = AttrListTreeprocessor.HEADER_RE
# The only attribute list this script writes, and therefore the only one it rewrites.
OWN_ID_ATTR_RE = re.compile(r"^#[A-Za-z0-9_-]+$")
def mkdocs_markdown() -> Markdown:
"""Build the Markdown parser the same way mkdocs does from mkdocs.yml.
Heading ids have to come from the same parse that renders the site, so the
extension list is read from the config rather than kept in a second place.
"""
with open(REPO_ROOT / "mkdocs.yml", encoding="utf-8") as f:
config = yaml_load(f)
# mkdocs puts these in front of the configured extensions.
extensions: list[str] = ["toc", "tables", "fenced_code"]
extension_configs: dict[str, dict[str, Any]] = {}
for item in config.get("markdown_extensions", []):
if isinstance(item, dict):
for name, options in item.items():
extensions.append(name)
extension_configs[name] = options or {}
else:
extensions.append(item)
return Markdown(extensions=extensions, extension_configs=extension_configs)
def heading_ids(source_markdown: str) -> list[tuple[int, str]]:
"""Return (level, id) for every heading in document order, as the toc extension assigns them."""
parser = mkdocs_markdown()
parser.convert(source_markdown)
ids: list[tuple[int, str]] = []
def walk(tokens: list[dict[str, Any]]) -> None:
for token in tokens:
ids.append((token["level"], token["id"]))
walk(token.get("children", []))
walk(parser.toc_tokens)
return ids
def headings_outside_code(markdown_text: str) -> list[tuple[int, int, str]]:
"""Return (line index, level, text) for every ATX heading outside fenced code."""
headings: list[tuple[int, int, str]] = []
open_fence: tuple[str, int] | None = None
for index, line in enumerate(markdown_text.splitlines()):
if open_fence is None:
opening = opening_fence(line)
if opening is not None:
open_fence = opening
continue
match = ATX_HEADING_RE.match(line)
if match is not None:
headings.append((index, len(match.group("level")), match.group("header").strip()))
elif is_closing_fence(line, *open_fence):
open_fence = None
return headings
def preserve_heading_anchors(
source_markdown: str, translated_markdown: str, *, name: str = "translation"
) -> str:
"""Give each translated heading the id mkdocs derives from the English heading.
mkdocs builds a heading id from the rendered heading text, so a translated heading
gets a different id and every `#...` link written against the English page stops
resolving. With `attr_list` enabled a heading can carry an explicit `{#id}`, which
the toc extension uses instead of slugifying the text.
The contract is the shape the docs use: one ATX heading per English heading, with
no attribute list other than the `{#id}` written here. A page outside it is
reported and returned unchanged. Before the result is returned it is parsed again
and has to render exactly the English ids, so a written page is a correct page.
"""
source_headings = heading_ids(source_markdown)
translated_headings = headings_outside_code(translated_markdown)
source_levels = [level for level, _ in source_headings]
translated_levels = [level for _, level, _ in translated_headings]
if source_levels != translated_levels:
print(f"Skipping heading anchors for {name}: headings do not line up with the source.")
return translated_markdown
lines = translated_markdown.splitlines(keepends=True)
for (level, heading_id), (index, _, text) in zip(source_headings, translated_headings):
# Leave the H1 alone: mkdocs reads the page title from it.
if level == 1:
continue
attrs = ATTR_LIST_RE.search(text)
if attrs is not None:
if OWN_ID_ATTR_RE.match(attrs.group(1).strip()) is None:
print(
f"Skipping heading anchors for {name}: heading carries an attribute list "
f"this script does not manage: {text!r}"
)
return translated_markdown
text = text[: attrs.start()]
line = lines[index]
ending = line[len(line.rstrip("\r\n")) :]
hashes = "#" * level
lines[index] = f"{hashes} {text} {{#{heading_id}}}{ending}"
rewritten = "".join(lines)
# The H1 keeps its translated id; everything else has to come out as the English id.
def without_h1_ids(headings: list[tuple[int, str]]) -> list[tuple[int, str | None]]:
return [(level, heading_id if level > 1 else None) for level, heading_id in headings]
if without_h1_ids(heading_ids(rewritten)) != without_h1_ids(source_headings):
print(f"Skipping heading anchors for {name}: the rewritten page does not render the ids.")
return translated_markdown
return rewritten
def protect_fenced_code(markdown: str, *, namespace: str) -> tuple[str, list[str]]:
parts: list[str] = []
code_blocks: list[str] = []
@@ -561,6 +680,7 @@ def translate_file(file_path: str, target_path: str, lang_code: str) -> None:
f"Protected Markdown changed after 3 translation attempts for {file_path} to {lang_code}"
)
translated_text = preserve_heading_anchors(content, translated_text, name=target_path)
# FIXME: enable mkdocs search plugin to seamlessly work with i18n plugin
translated_text = SEARCH_EXCLUSION + translated_text
# Save the combined translated content
@@ -599,6 +719,24 @@ def should_translate_based_on_translation(file_path: str) -> bool:
return ja_timestamp < en_timestamp
def refresh_heading_anchors(file_path: str, relative_path: str) -> None:
"""Re-apply the English heading ids to existing translations without retranslating."""
with open(file_path, encoding="utf-8") as f:
content = f.read()
for lang_code in languages:
target_path = os.path.join(source_dir, lang_code, relative_path)
if not os.path.exists(target_path):
continue
with open(target_path, encoding="utf-8", newline="") as f:
translated_text = f.read()
updated_text = preserve_heading_anchors(content, translated_text, name=target_path)
if updated_text == translated_text:
continue
print(f"Refreshing heading anchors in {target_path}")
with open(target_path, "w", encoding="utf-8", newline="") as f:
f.write(updated_text)
def translate_single_source_file(
file_path: str, *, check_translation_outdated: bool = True
) -> None:
@@ -607,6 +745,7 @@ def translate_single_source_file(
return
if check_translation_outdated and not should_translate_based_on_translation(file_path):
print(f"Skipping {file_path}: The translated one is up-to-date.")
refresh_heading_anchors(file_path, relative_path)
return
for lang_code in languages:
+129
View File
@@ -0,0 +1,129 @@
from __future__ import annotations
import importlib.util
from pathlib import Path
from types import ModuleType
import pytest
SCRIPT_PATH = Path(__file__).resolve().parents[2] / "docs" / "scripts" / "translate_docs.py"
SOURCE = """# Agents
## Dynamic instructions
Text.
## Example
```python
# not a heading
```
## Example
"""
TRANSLATED = """# エージェント
## 動的な指示
本文。
## 例
```python
# not a heading
```
## 例
"""
@pytest.fixture
def translate_docs(monkeypatch: pytest.MonkeyPatch) -> ModuleType:
# The script builds an OpenAI client at import time; nothing here sends a request.
monkeypatch.setenv("OPENAI_API_KEY", "test-key")
spec = importlib.util.spec_from_file_location("translate_docs", SCRIPT_PATH)
assert spec is not None and spec.loader is not None
module = importlib.util.module_from_spec(spec)
spec.loader.exec_module(module)
return module
def test_translated_headings_carry_the_english_ids(translate_docs: ModuleType) -> None:
result = translate_docs.preserve_heading_anchors(SOURCE, TRANSLATED)
assert "## 動的な指示 {#dynamic-instructions}\n" in result
assert "## 例 {#example}\n" in result
assert "## 例 {#example_1}\n" in result
# The H1 is left for mkdocs to read the page title from.
assert result.startswith("# エージェント\n")
# A comment inside a fenced block is not a heading.
assert "# not a heading\n" in result
assert "# not a heading {#" not in result
def test_heading_ids_come_from_the_rendered_english_headings(translate_docs: ModuleType) -> None:
source = (
"## Using `Agent` with [tools](tools.md)\n\n"
"## [API][ref]\n\n"
"## A &amp; B\n\n"
"## <code>run</code> loop\n\n"
"[ref]: https://example.com\n"
)
translated = "## `Agent` とツール\n\n## API\n\n## A と B\n\n## 実行ループ\n"
result = translate_docs.preserve_heading_anchors(source, translated)
assert result == (
"## `Agent` とツール {#using-agent-with-tools}\n\n"
"## API {#api}\n\n"
"## A と B {#a-b}\n\n"
"## 実行ループ {#run-loop}\n"
)
def test_preserve_heading_anchors_is_idempotent(translate_docs: ModuleType) -> None:
once = translate_docs.preserve_heading_anchors(SOURCE, TRANSLATED)
assert translate_docs.preserve_heading_anchors(SOURCE, once) == once
def test_an_id_written_earlier_follows_the_english_heading(translate_docs: ModuleType) -> None:
result = translate_docs.preserve_heading_anchors("## Alpha\n", "## アルファ {#old}\n")
assert result == "## アルファ {#alpha}\n"
def test_mismatched_headings_are_left_alone(translate_docs: ModuleType) -> None:
missing_one_heading = TRANSLATED.replace("\n## 例\n", "\n", 1)
result = translate_docs.preserve_heading_anchors(SOURCE, missing_one_heading)
assert result == missing_one_heading
def test_an_english_setext_heading_still_yields_its_id(translate_docs: ModuleType) -> None:
# The English side goes through the parser, so setext is just another heading there.
source = "Alpha\n-----\n\n## Beta\n"
translated = "## アルファ\n\n## ベータ\n"
result = translate_docs.preserve_heading_anchors(source, translated)
assert result == "## アルファ {#alpha}\n\n## ベータ {#beta}\n"
def test_a_setext_heading_in_the_translation_is_outside_the_contract(
translate_docs: ModuleType,
) -> None:
source = "## Alpha\n\n## Beta\n"
translated = "アルファ\n-----\n\n## ベータ\n"
assert translate_docs.preserve_heading_anchors(source, translated) == translated
def test_a_heading_with_its_own_attribute_list_is_not_rewritten(translate_docs: ModuleType) -> None:
source = "## Alpha\n\n## Beta\n"
translated = "## アルファ {.lead}\n\n## ベータ\n"
assert translate_docs.preserve_heading_anchors(source, translated) == translated