diff --git a/tests/tools/test_xlsx_phonetic_text.py b/tests/tools/test_xlsx_phonetic_text.py index 8e926c3cd6..483bcb1d79 100644 --- a/tests/tools/test_xlsx_phonetic_text.py +++ b/tests/tools/test_xlsx_phonetic_text.py @@ -1,4 +1,4 @@ -"""XLSX phonetic guides annotate base text; they are not part of the cell value. +"""Phonetic guides (XLSX ``rPh``, DOCX ``w:rt`` ruby) annotate base text; they are not the value. See https://learn.microsoft.com/en-us/dotnet/api/documentformat.openxml.spreadsheet.phoneticrun (rPh is permitted under both si and is). @@ -13,6 +13,7 @@ from tools.read_extract import extract_document_text from tools.registry import registry S = "http://schemas.openxmlformats.org/spreadsheetml/2006/main" +W = "http://schemas.openxmlformats.org/wordprocessingml/2006/main" R = "http://schemas.openxmlformats.org/officeDocument/2006/relationships" P = "http://schemas.openxmlformats.org/package/2006/relationships" @@ -33,20 +34,39 @@ def _workbook(path, storage, text): return path -@pytest.mark.parametrize("storage", ["shared", "inline"]) -@pytest.mark.parametrize("rich", [False, True]) -def test_read_file_excludes_phonetic_guides(tmp_path, monkeypatch, storage, rich): - base = '東京' if rich else '東京' - phonetics = 'トウキョウ' - path = _workbook(tmp_path / "cities.xlsx", storage, base + phonetics) +def _docx(path, paragraph_xml): + with zipfile.ZipFile(path, "w") as package: + package.writestr("[Content_Types].xml", "") + package.writestr("word/document.xml", + f'{paragraph_xml}') + return path + + +_XLSX_PHONETICS = 'トウキョウ' +_DOCX_RUBY = ('トウキョウ' + '東京sentinel') + + +def _document(tmp_path, kind): + if kind == "docx-ruby": + return _docx(tmp_path / "cities.docx", _DOCX_RUBY) + storage, rich = kind.split("-") + base = '東京' if rich == "rich" else '東京' + return _workbook(tmp_path / "cities.xlsx", storage, base + _XLSX_PHONETICS) + + +@pytest.mark.parametrize("kind", ["shared-plain", "shared-rich", "inline-plain", "inline-rich", "docx-ruby"]) +def test_read_file_excludes_phonetic_guides(tmp_path, monkeypatch, kind): + path = _document(tmp_path, kind) monkeypatch.setenv("TERMINAL_ENV", "local") monkeypatch.setenv("TERMINAL_CWD", str(tmp_path)) - task_id = f"xlsx-phonetic-{storage}-{rich}" + task_id = f"phonetic-{kind}" try: result = json.loads(registry.dispatch("read_file", {"path": str(path)}, task_id=task_id)) assert not result.get("error"), result assert result["extracted_document"] is True - assert result["content"].splitlines()[1] == "2|東京\tsentinel" + row = result["content"].splitlines()[0 if kind == "docx-ruby" else 1] # XLSX line 1 is the sheet header + assert row.split("|", 1)[1] == "東京\tsentinel" finally: file_tools.clear_file_ops_cache(task_id) diff --git a/tools/read_extract.py b/tools/read_extract.py index 9ec26aec8e..4fe9fbfe71 100644 --- a/tools/read_extract.py +++ b/tools/read_extract.py @@ -468,8 +468,11 @@ def _extract_docx(path: str) -> str: breaks = {f"{w}tab": "\t", f"{w}br": "\n", f"{w}cr": "\n"} lines: list[str] = [] for para in root.iter(f"{w}p"): + # w:rt is the ruby (phonetic) guide over w:rubyBase; it annotates the text, it is not text. + guide = {n for rt in para.iter(f"{w}rt") for n in rt.iter()} text = "".join( - (n.text or "") if n.tag == f"{w}t" else breaks.get(n.tag, "") for n in para.iter()) + (n.text or "") if n.tag == f"{w}t" else breaks.get(n.tag, "") + for n in para.iter() if n not in guide) lines.extend(text.split("\n")) return _joined(lines, "DOCX contains no extractable text")