fix(read_file): DOCX ruby guides (w:rt) are excluded from paragraph text

Same class as the XLSX rPh fix: a phonetic guide annotates the base text and
is not part of it. `_extract_docx` collected every w:t in the paragraph, so a
ruby-annotated run rendered as guide + base ("トウキョウ東京"). Skip the w:rt
subtree; w:rubyBase stays. The phonetic test now covers XLSX shared/inline/rich
and DOCX ruby through the registered read_file handler.
This commit is contained in:
teknium1
2026-09-18 00:37:21 -07:00
committed by Teknium
parent 7a79b14788
commit 58e749d07a
2 changed files with 33 additions and 10 deletions

View File

@@ -1,4 +1,4 @@
"""XLSX phonetic guides annotate base text; they are not part of the cell value.
"""Phonetic guides (XLSX ``rPh``, DOCX ``w:rt`` ruby) annotate base text; they are not the value.
See https://learn.microsoft.com/en-us/dotnet/api/documentformat.openxml.spreadsheet.phoneticrun
(rPh is permitted under both si and is).
@@ -13,6 +13,7 @@ from tools.read_extract import extract_document_text
from tools.registry import registry
S = "http://schemas.openxmlformats.org/spreadsheetml/2006/main"
W = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
R = "http://schemas.openxmlformats.org/officeDocument/2006/relationships"
P = "http://schemas.openxmlformats.org/package/2006/relationships"
@@ -33,20 +34,39 @@ def _workbook(path, storage, text):
return path
@pytest.mark.parametrize("storage", ["shared", "inline"])
@pytest.mark.parametrize("rich", [False, True])
def test_read_file_excludes_phonetic_guides(tmp_path, monkeypatch, storage, rich):
base = '<r><rPr><b/></rPr><t>東</t></r><r><t>京</t></r>' if rich else '<t>東京</t>'
phonetics = '<rPh sb="0" eb="1"><t>トウ</t></rPh><rPh sb="1" eb="2"><t>キョウ</t></rPh>'
path = _workbook(tmp_path / "cities.xlsx", storage, base + phonetics)
def _docx(path, paragraph_xml):
with zipfile.ZipFile(path, "w") as package:
package.writestr("[Content_Types].xml", "<Types/>")
package.writestr("word/document.xml",
f'<w:document xmlns:w="{W}"><w:body><w:p>{paragraph_xml}</w:p></w:body></w:document>')
return path
_XLSX_PHONETICS = '<rPh sb="0" eb="1"><t>トウ</t></rPh><rPh sb="1" eb="2"><t>キョウ</t></rPh>'
_DOCX_RUBY = ('<w:r><w:ruby><w:rt><w:r><w:t>トウキョウ</w:t></w:r></w:rt>'
'<w:rubyBase><w:r><w:t>東京</w:t></w:r></w:rubyBase></w:ruby></w:r><w:r><w:tab/><w:t>sentinel</w:t></w:r>')
def _document(tmp_path, kind):
if kind == "docx-ruby":
return _docx(tmp_path / "cities.docx", _DOCX_RUBY)
storage, rich = kind.split("-")
base = '<r><rPr><b/></rPr><t>東</t></r><r><t>京</t></r>' if rich == "rich" else '<t>東京</t>'
return _workbook(tmp_path / "cities.xlsx", storage, base + _XLSX_PHONETICS)
@pytest.mark.parametrize("kind", ["shared-plain", "shared-rich", "inline-plain", "inline-rich", "docx-ruby"])
def test_read_file_excludes_phonetic_guides(tmp_path, monkeypatch, kind):
path = _document(tmp_path, kind)
monkeypatch.setenv("TERMINAL_ENV", "local")
monkeypatch.setenv("TERMINAL_CWD", str(tmp_path))
task_id = f"xlsx-phonetic-{storage}-{rich}"
task_id = f"phonetic-{kind}"
try:
result = json.loads(registry.dispatch("read_file", {"path": str(path)}, task_id=task_id))
assert not result.get("error"), result
assert result["extracted_document"] is True
assert result["content"].splitlines()[1] == "2|東京\tsentinel"
row = result["content"].splitlines()[0 if kind == "docx-ruby" else 1] # XLSX line 1 is the sheet header
assert row.split("|", 1)[1] == "東京\tsentinel"
finally:
file_tools.clear_file_ops_cache(task_id)

View File

@@ -468,8 +468,11 @@ def _extract_docx(path: str) -> str:
breaks = {f"{w}tab": "\t", f"{w}br": "\n", f"{w}cr": "\n"}
lines: list[str] = []
for para in root.iter(f"{w}p"):
# w:rt is the ruby (phonetic) guide over w:rubyBase; it annotates the text, it is not text.
guide = {n for rt in para.iter(f"{w}rt") for n in rt.iter()}
text = "".join(
(n.text or "") if n.tag == f"{w}t" else breaks.get(n.tag, "") for n in para.iter())
(n.text or "") if n.tag == f"{w}t" else breaks.get(n.tag, "")
for n in para.iter() if n not in guide)
lines.extend(text.split("\n"))
return _joined(lines, "DOCX contains no extractable text")