From 58e749d07ae438beeec2da082de2be8dfd666c59 Mon Sep 17 00:00:00 2001
From: teknium1 <127238744+teknium1@users.noreply.github.com>
Date: Fri, 18 Sep 2026 00:37:21 -0700
Subject: [PATCH] fix(read_file): DOCX ruby guides (w:rt) are excluded from
paragraph text
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Same class as the XLSX rPh fix: a phonetic guide annotates the base text and
is not part of it. `_extract_docx` collected every w:t in the paragraph, so a
ruby-annotated run rendered as guide + base ("トウキョウ東京"). Skip the w:rt
subtree; w:rubyBase stays. The phonetic test now covers XLSX shared/inline/rich
and DOCX ruby through the registered read_file handler.
---
tests/tools/test_xlsx_phonetic_text.py | 38 ++++++++++++++++++++------
tools/read_extract.py | 5 +++-
2 files changed, 33 insertions(+), 10 deletions(-)
diff --git a/tests/tools/test_xlsx_phonetic_text.py b/tests/tools/test_xlsx_phonetic_text.py
index 8e926c3cd6..483bcb1d79 100644
--- a/tests/tools/test_xlsx_phonetic_text.py
+++ b/tests/tools/test_xlsx_phonetic_text.py
@@ -1,4 +1,4 @@
-"""XLSX phonetic guides annotate base text; they are not part of the cell value.
+"""Phonetic guides (XLSX ``rPh``, DOCX ``w:rt`` ruby) annotate base text; they are not the value.
See https://learn.microsoft.com/en-us/dotnet/api/documentformat.openxml.spreadsheet.phoneticrun
(rPh is permitted under both si and is).
@@ -13,6 +13,7 @@ from tools.read_extract import extract_document_text
from tools.registry import registry
S = "http://schemas.openxmlformats.org/spreadsheetml/2006/main"
+W = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
R = "http://schemas.openxmlformats.org/officeDocument/2006/relationships"
P = "http://schemas.openxmlformats.org/package/2006/relationships"
@@ -33,20 +34,39 @@ def _workbook(path, storage, text):
return path
-@pytest.mark.parametrize("storage", ["shared", "inline"])
-@pytest.mark.parametrize("rich", [False, True])
-def test_read_file_excludes_phonetic_guides(tmp_path, monkeypatch, storage, rich):
- base = '東京' if rich else '東京'
- phonetics = 'トウキョウ'
- path = _workbook(tmp_path / "cities.xlsx", storage, base + phonetics)
+def _docx(path, paragraph_xml):
+ with zipfile.ZipFile(path, "w") as package:
+ package.writestr("[Content_Types].xml", "")
+ package.writestr("word/document.xml",
+ f'{paragraph_xml}')
+ return path
+
+
+_XLSX_PHONETICS = 'トウキョウ'
+_DOCX_RUBY = ('トウキョウ'
+ '東京sentinel')
+
+
+def _document(tmp_path, kind):
+ if kind == "docx-ruby":
+ return _docx(tmp_path / "cities.docx", _DOCX_RUBY)
+ storage, rich = kind.split("-")
+ base = '東京' if rich == "rich" else '東京'
+ return _workbook(tmp_path / "cities.xlsx", storage, base + _XLSX_PHONETICS)
+
+
+@pytest.mark.parametrize("kind", ["shared-plain", "shared-rich", "inline-plain", "inline-rich", "docx-ruby"])
+def test_read_file_excludes_phonetic_guides(tmp_path, monkeypatch, kind):
+ path = _document(tmp_path, kind)
monkeypatch.setenv("TERMINAL_ENV", "local")
monkeypatch.setenv("TERMINAL_CWD", str(tmp_path))
- task_id = f"xlsx-phonetic-{storage}-{rich}"
+ task_id = f"phonetic-{kind}"
try:
result = json.loads(registry.dispatch("read_file", {"path": str(path)}, task_id=task_id))
assert not result.get("error"), result
assert result["extracted_document"] is True
- assert result["content"].splitlines()[1] == "2|東京\tsentinel"
+ row = result["content"].splitlines()[0 if kind == "docx-ruby" else 1] # XLSX line 1 is the sheet header
+ assert row.split("|", 1)[1] == "東京\tsentinel"
finally:
file_tools.clear_file_ops_cache(task_id)
diff --git a/tools/read_extract.py b/tools/read_extract.py
index 9ec26aec8e..4fe9fbfe71 100644
--- a/tools/read_extract.py
+++ b/tools/read_extract.py
@@ -468,8 +468,11 @@ def _extract_docx(path: str) -> str:
breaks = {f"{w}tab": "\t", f"{w}br": "\n", f"{w}cr": "\n"}
lines: list[str] = []
for para in root.iter(f"{w}p"):
+ # w:rt is the ruby (phonetic) guide over w:rubyBase; it annotates the text, it is not text.
+ guide = {n for rt in para.iter(f"{w}rt") for n in rt.iter()}
text = "".join(
- (n.text or "") if n.tag == f"{w}t" else breaks.get(n.tag, "") for n in para.iter())
+ (n.text or "") if n.tag == f"{w}t" else breaks.get(n.tag, "")
+ for n in para.iter() if n not in guide)
lines.extend(text.split("\n"))
return _joined(lines, "DOCX contains no extractable text")