fix(google-workspace): support multi-tab Google Docs (port cloudflare/cloudflare-os#450)
Google Docs can hold a tree of tabs, each with its own body and its own 1-based character index space; the legacy top-level `body` only carries the first tab. `docs get` was silently dropping every other tab's content, and `docs append` computed its insert index from the first tab's body and sent the write with no tabId — so on a tabbed doc the append could land at a wrong offset in the wrong tab. - Requests now pass includeTabsContent=true (both gws and SDK paths). - `docs get` returns a `tabs` array (preorder flattening of the tabs/childTabs tree, nested tabs included); single-tab docs keep the `body` field so existing callers work, and `--tab <tabId>` reads one tab. Legacy no-tabs responses are unchanged. - `docs append` targets exactly one tab: the insert location carries the tabId, the end index is computed inside that tab's own body, an unknown `--tab` errors instead of falling back to the first tab, and a multi-tab doc without `--tab` errors with the tab list rather than guessing. Tabs are never merged — index spaces are independent. Adapted from cloudflare/cloudflare-os#450 (gatekeeper-google), which fixed the same provider behavior: reads must traverse Document.tabs and every write Location must carry the immutable tabId. Two pre-existing bare read_text/write_text in the touched test file gained encoding="utf-8" (windows-footgun sweep rule).
This commit is contained in:
@@ -276,8 +276,9 @@ $GAPI sheets append SHEET_ID "Sheet1!A:C" --values '[["new","row","data"]]'
|
||||
### Docs
|
||||
|
||||
```bash
|
||||
# Read
|
||||
# Read (a tabbed Doc returns a "tabs" array; single-tab and legacy Docs also return "body")
|
||||
$GAPI docs get DOC_ID
|
||||
$GAPI docs get DOC_ID --tab TAB_ID # read one tab of a tabbed Doc
|
||||
|
||||
# Create a new Doc (optionally seeded with body text)
|
||||
$GAPI docs create --title "Meeting Notes"
|
||||
@@ -285,6 +286,7 @@ $GAPI docs create --title "Draft" --body "First paragraph..."
|
||||
|
||||
# Append text to the end of an existing Doc
|
||||
$GAPI docs append DOC_ID --text "Additional content to append"
|
||||
$GAPI docs append DOC_ID --tab TAB_ID --text "..." # --tab required when the Doc has multiple tabs
|
||||
```
|
||||
|
||||
## Output Format
|
||||
|
||||
@@ -154,9 +154,9 @@ def _extract_message_body(msg: dict) -> str:
|
||||
return body
|
||||
|
||||
|
||||
def _extract_doc_text(doc: dict) -> str:
|
||||
def _extract_body_text(body: dict) -> str:
|
||||
text_parts = []
|
||||
for element in doc.get("body", {}).get("content", []):
|
||||
for element in body.get("content", []):
|
||||
paragraph = element.get("paragraph", {})
|
||||
for pe in paragraph.get("elements", []):
|
||||
text_run = pe.get("textRun", {})
|
||||
@@ -165,6 +165,70 @@ def _extract_doc_text(doc: dict) -> str:
|
||||
return "".join(text_parts)
|
||||
|
||||
|
||||
def _extract_doc_text(doc: dict) -> str:
|
||||
return _extract_body_text(doc.get("body", {}))
|
||||
|
||||
|
||||
def _flatten_doc_tabs(doc: dict) -> list[dict]:
|
||||
"""Flatten Google's recursive ``tabs``/``childTabs`` tree (preorder).
|
||||
|
||||
A tabbed Doc keeps each tab's content in its own body with an independent
|
||||
index space; the legacy top-level ``body`` only carries the first tab, so
|
||||
reads and writes that ignore ``tabs`` silently drop or mistarget content.
|
||||
Returns [] for the legacy single-body response shape (no ``tabs`` field).
|
||||
"""
|
||||
flat: list[dict] = []
|
||||
|
||||
def visit(tabs, level):
|
||||
for tab in tabs or []:
|
||||
props = tab.get("tabProperties") or {}
|
||||
doc_tab = tab.get("documentTab") or {}
|
||||
flat.append({
|
||||
"tabId": props.get("tabId", ""),
|
||||
"title": props.get("title", ""),
|
||||
"level": level,
|
||||
"body": doc_tab.get("body") or {},
|
||||
})
|
||||
visit(tab.get("childTabs"), level + 1)
|
||||
|
||||
visit(doc.get("tabs"), 0)
|
||||
return flat
|
||||
|
||||
|
||||
def _resolve_write_tab(doc: dict, tab_arg: str | None) -> tuple[str | None, dict]:
|
||||
"""Pick exactly one tab body for a write; never merge index spaces.
|
||||
|
||||
Legacy docs (no ``tabs``) return (None, body) — the write carries no tabId.
|
||||
A multi-tab doc requires an explicit --tab; an unknown ID errors instead of
|
||||
quietly falling back to the first tab.
|
||||
"""
|
||||
tabs = _flatten_doc_tabs(doc)
|
||||
if not tabs:
|
||||
return None, doc.get("body", {})
|
||||
if tab_arg:
|
||||
for tab in tabs:
|
||||
if tab["tabId"] == tab_arg:
|
||||
return tab["tabId"], tab["body"]
|
||||
print(
|
||||
json.dumps({
|
||||
"error": f"unknown tab ID {tab_arg!r}",
|
||||
"tabs": [{"tabId": t["tabId"], "title": t["title"]} for t in tabs],
|
||||
}, indent=2, ensure_ascii=False),
|
||||
file=sys.stderr,
|
||||
)
|
||||
sys.exit(1)
|
||||
if len(tabs) == 1:
|
||||
return tabs[0]["tabId"], tabs[0]["body"]
|
||||
print(
|
||||
json.dumps({
|
||||
"error": f"document has {len(tabs)} tabs; pass --tab <tabId> to pick one",
|
||||
"tabs": [{"tabId": t["tabId"], "title": t["title"]} for t in tabs],
|
||||
}, indent=2, ensure_ascii=False),
|
||||
file=sys.stderr,
|
||||
)
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def _datetime_with_timezone(value: str) -> str:
|
||||
if not value:
|
||||
return value
|
||||
@@ -953,23 +1017,41 @@ def sheets_create(args):
|
||||
|
||||
|
||||
def docs_get(args):
|
||||
tab_arg = getattr(args, "tab", None)
|
||||
params = {"documentId": args.doc_id, "includeTabsContent": True}
|
||||
if _gws_binary():
|
||||
doc = _run_gws(["docs", "documents", "get"], params={"documentId": args.doc_id})
|
||||
result = {
|
||||
"title": doc.get("title", ""),
|
||||
"documentId": doc.get("documentId", ""),
|
||||
"body": _extract_doc_text(doc),
|
||||
}
|
||||
print(json.dumps(result, indent=2, ensure_ascii=False))
|
||||
return
|
||||
doc = _run_gws(["docs", "documents", "get"], params=params)
|
||||
else:
|
||||
service = build_service("docs", "v1")
|
||||
doc = service.documents().get(
|
||||
documentId=args.doc_id, includeTabsContent=True,
|
||||
).execute()
|
||||
|
||||
service = build_service("docs", "v1")
|
||||
doc = service.documents().get(documentId=args.doc_id).execute()
|
||||
result = {
|
||||
"title": doc.get("title", ""),
|
||||
"documentId": doc.get("documentId", ""),
|
||||
"body": _extract_doc_text(doc),
|
||||
}
|
||||
tabs = _flatten_doc_tabs(doc)
|
||||
if not tabs:
|
||||
# Legacy single-body response shape.
|
||||
result["body"] = _extract_doc_text(doc)
|
||||
elif tab_arg:
|
||||
_, body = _resolve_write_tab(doc, tab_arg)
|
||||
result["tab"] = tab_arg
|
||||
result["body"] = _extract_body_text(body)
|
||||
else:
|
||||
result["tabs"] = [
|
||||
{
|
||||
"tabId": t["tabId"],
|
||||
"title": t["title"],
|
||||
"level": t["level"],
|
||||
"body": _extract_body_text(t["body"]),
|
||||
}
|
||||
for t in tabs
|
||||
]
|
||||
# Keep "body" populated for single-tab docs so existing callers work.
|
||||
if len(tabs) == 1:
|
||||
result["body"] = result["tabs"][0]["body"]
|
||||
print(json.dumps(result, indent=2, ensure_ascii=False))
|
||||
|
||||
|
||||
@@ -997,17 +1079,25 @@ def docs_create(args):
|
||||
|
||||
|
||||
def docs_append(args):
|
||||
"""Append text to the end of an existing Doc."""
|
||||
"""Append text to the end of an existing Doc (one tab of it, if tabbed)."""
|
||||
if _gws_binary():
|
||||
doc = _run_gws(["docs", "documents", "get"], params={"documentId": args.doc_id})
|
||||
doc = _run_gws(
|
||||
["docs", "documents", "get"],
|
||||
params={"documentId": args.doc_id, "includeTabsContent": True},
|
||||
)
|
||||
else:
|
||||
service = build_service("docs", "v1")
|
||||
doc = service.documents().get(documentId=args.doc_id).execute()
|
||||
doc = service.documents().get(
|
||||
documentId=args.doc_id, includeTabsContent=True,
|
||||
).execute()
|
||||
|
||||
tab_id, body = _resolve_write_tab(doc, getattr(args, "tab", None))
|
||||
|
||||
# The end-of-body index is one less than the segment endIndex of the body
|
||||
# (trailing newline is always at length-1). Docs indexes are 1-based; use
|
||||
# endIndex - 1 to insert before the final newline.
|
||||
content = doc.get("body", {}).get("content", [])
|
||||
# endIndex - 1 to insert before the final newline. Each tab has its own
|
||||
# index space, so the write location must carry the tab ID.
|
||||
content = body.get("content", [])
|
||||
end_index = 1
|
||||
for element in content:
|
||||
ei = element.get("endIndex")
|
||||
@@ -1016,21 +1106,27 @@ def docs_append(args):
|
||||
insert_index = max(end_index - 1, 1)
|
||||
|
||||
text = args.text if args.text.endswith("\n") else args.text + "\n"
|
||||
_docs_insert_text(args.doc_id, text, index=insert_index)
|
||||
_docs_insert_text(args.doc_id, text, index=insert_index, tab_id=tab_id)
|
||||
|
||||
print(json.dumps({
|
||||
result = {
|
||||
"status": "appended",
|
||||
"documentId": args.doc_id,
|
||||
"inserted_at": insert_index,
|
||||
"characters": len(text),
|
||||
}, indent=2, ensure_ascii=False))
|
||||
}
|
||||
if tab_id:
|
||||
result["tab"] = tab_id
|
||||
print(json.dumps(result, indent=2, ensure_ascii=False))
|
||||
|
||||
|
||||
def _docs_insert_text(doc_id: str, text: str, index: int) -> None:
|
||||
def _docs_insert_text(doc_id: str, text: str, index: int, tab_id: str | None = None) -> None:
|
||||
"""Send a batchUpdate with a single insertText request."""
|
||||
location: dict = {"index": index}
|
||||
if tab_id:
|
||||
location["tabId"] = tab_id
|
||||
requests = [{
|
||||
"insertText": {
|
||||
"location": {"index": index},
|
||||
"location": location,
|
||||
"text": text,
|
||||
}
|
||||
}]
|
||||
@@ -1205,6 +1301,7 @@ def main():
|
||||
|
||||
p = docs_sub.add_parser("get")
|
||||
p.add_argument("doc_id")
|
||||
p.add_argument("--tab", default=None, help="Tab ID to read (tabbed Docs)")
|
||||
p.set_defaults(func=docs_get)
|
||||
|
||||
p = docs_sub.add_parser("create")
|
||||
@@ -1215,6 +1312,7 @@ def main():
|
||||
p = docs_sub.add_parser("append")
|
||||
p.add_argument("doc_id")
|
||||
p.add_argument("--text", required=True, help="Text to append to the end of the document")
|
||||
p.add_argument("--tab", default=None, help="Tab ID to append to (required for multi-tab Docs)")
|
||||
p.set_defaults(func=docs_append)
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
@@ -65,7 +65,7 @@ def _write_token(path: Path, *, token="ya29.test", expiry=None, **extra):
|
||||
}
|
||||
if expiry is not None:
|
||||
data["expiry"] = expiry
|
||||
path.write_text(json.dumps(data))
|
||||
path.write_text(json.dumps(data), encoding="utf-8")
|
||||
|
||||
|
||||
def test_bridge_returns_valid_token(bridge_module, tmp_path):
|
||||
@@ -191,7 +191,81 @@ def test_api_get_credentials_refresh_persists_authorized_user_type(api_module, m
|
||||
|
||||
creds = api_module.get_credentials()
|
||||
|
||||
saved = json.loads(token_path.read_text())
|
||||
saved = json.loads(token_path.read_text(encoding="utf-8"))
|
||||
assert isinstance(creds, FakeCredentials)
|
||||
assert saved["token"] == "ya29.refreshed"
|
||||
assert saved["type"] == "authorized_user"
|
||||
|
||||
|
||||
def _tabbed_doc():
|
||||
"""A Doc with two tabs (one nested), as the Docs API returns with includeTabsContent."""
|
||||
def body(text):
|
||||
return {"content": [
|
||||
{"endIndex": len(text) + 2,
|
||||
"paragraph": {"elements": [{"textRun": {"content": text + "\n"}}]}},
|
||||
]}
|
||||
return {
|
||||
"title": "Tabbed",
|
||||
"documentId": "doc1",
|
||||
"tabs": [
|
||||
{
|
||||
"tabProperties": {"tabId": "t.0", "title": "First"},
|
||||
"documentTab": {"body": body("alpha")},
|
||||
"childTabs": [
|
||||
{
|
||||
"tabProperties": {"tabId": "t.0.a", "title": "Nested"},
|
||||
"documentTab": {"body": body("beta")},
|
||||
}
|
||||
],
|
||||
},
|
||||
{
|
||||
"tabProperties": {"tabId": "t.1", "title": "Second"},
|
||||
"documentTab": {"body": body("gamma")},
|
||||
},
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
def test_docs_get_returns_every_tab_of_a_tabbed_doc(api_module, monkeypatch, capsys):
|
||||
"""A multi-tab Doc must not lose tab content: reads traverse the tabs tree
|
||||
(preorder, nested tabs included) instead of only the legacy top-level body."""
|
||||
monkeypatch.setattr(
|
||||
api_module, "_run_gws",
|
||||
lambda parts, params=None, body=None: _tabbed_doc(),
|
||||
)
|
||||
args = types.SimpleNamespace(doc_id="doc1", tab=None)
|
||||
api_module.docs_get(args)
|
||||
result = json.loads(capsys.readouterr().out)
|
||||
tabs = {t["tabId"]: t for t in result["tabs"]}
|
||||
assert set(tabs) == {"t.0", "t.0.a", "t.1"}
|
||||
assert tabs["t.0.a"]["body"] == "beta\n"
|
||||
assert tabs["t.0.a"]["level"] == 1
|
||||
# Multi-tab docs have no single merged "body" — index spaces are independent.
|
||||
assert "body" not in result
|
||||
|
||||
|
||||
def test_docs_append_carries_tab_id_and_refuses_ambiguous_writes(api_module, monkeypatch, capsys):
|
||||
"""Each tab has its own index space, so a write must target exactly one tab:
|
||||
the insert location carries the tabId, and an un-targeted write against a
|
||||
multi-tab doc errors instead of silently landing in the first tab."""
|
||||
monkeypatch.setattr(
|
||||
api_module, "_run_gws",
|
||||
lambda parts, params=None, body=None: _tabbed_doc(),
|
||||
)
|
||||
sent = {}
|
||||
monkeypatch.setattr(
|
||||
api_module, "_docs_insert_text",
|
||||
lambda doc_id, text, index, tab_id=None: sent.update(
|
||||
{"doc_id": doc_id, "index": index, "tab_id": tab_id}
|
||||
),
|
||||
)
|
||||
|
||||
api_module.docs_append(types.SimpleNamespace(doc_id="doc1", text="more", tab="t.1"))
|
||||
assert sent["tab_id"] == "t.1"
|
||||
assert sent["index"] == len("gamma") + 1 # endIndex - 1 within THAT tab's space
|
||||
capsys.readouterr()
|
||||
|
||||
with pytest.raises(SystemExit):
|
||||
api_module.docs_append(types.SimpleNamespace(doc_id="doc1", text="more", tab=None))
|
||||
err = json.loads(capsys.readouterr().err)
|
||||
assert "tabs" in err and len(err["tabs"]) == 3
|
||||
|
||||
Reference in New Issue
Block a user