fix(google-workspace): support multi-tab Google Docs (port cloudflare/cloudflare-os#450)

Google Docs can hold a tree of tabs, each with its own body and its own
1-based character index space; the legacy top-level `body` only carries
the first tab. `docs get` was silently dropping every other tab's
content, and `docs append` computed its insert index from the first
tab's body and sent the write with no tabId — so on a tabbed doc the
append could land at a wrong offset in the wrong tab.

- Requests now pass includeTabsContent=true (both gws and SDK paths).
- `docs get` returns a `tabs` array (preorder flattening of the
  tabs/childTabs tree, nested tabs included); single-tab docs keep the
  `body` field so existing callers work, and `--tab <tabId>` reads one
  tab. Legacy no-tabs responses are unchanged.
- `docs append` targets exactly one tab: the insert location carries
  the tabId, the end index is computed inside that tab's own body, an
  unknown `--tab` errors instead of falling back to the first tab, and
  a multi-tab doc without `--tab` errors with the tab list rather than
  guessing. Tabs are never merged — index spaces are independent.

Adapted from cloudflare/cloudflare-os#450 (gatekeeper-google), which
fixed the same provider behavior: reads must traverse Document.tabs and
every write Location must carry the immutable tabId.

Two pre-existing bare read_text/write_text in the touched test file
gained encoding="utf-8" (windows-footgun sweep rule).
This commit is contained in:
Teknium
2026-09-09 02:10:34 -07:00
parent 11d761eed9
commit 50f17b1300
3 changed files with 200 additions and 26 deletions

View File

@@ -276,8 +276,9 @@ $GAPI sheets append SHEET_ID "Sheet1!A:C" --values '[["new","row","data"]]'
### Docs
```bash
# Read
# Read (a tabbed Doc returns a "tabs" array; single-tab and legacy Docs also return "body")
$GAPI docs get DOC_ID
$GAPI docs get DOC_ID --tab TAB_ID # read one tab of a tabbed Doc
# Create a new Doc (optionally seeded with body text)
$GAPI docs create --title "Meeting Notes"
@@ -285,6 +286,7 @@ $GAPI docs create --title "Draft" --body "First paragraph..."
# Append text to the end of an existing Doc
$GAPI docs append DOC_ID --text "Additional content to append"
$GAPI docs append DOC_ID --tab TAB_ID --text "..." # --tab required when the Doc has multiple tabs
```
## Output Format

View File

@@ -154,9 +154,9 @@ def _extract_message_body(msg: dict) -> str:
return body
def _extract_doc_text(doc: dict) -> str:
def _extract_body_text(body: dict) -> str:
text_parts = []
for element in doc.get("body", {}).get("content", []):
for element in body.get("content", []):
paragraph = element.get("paragraph", {})
for pe in paragraph.get("elements", []):
text_run = pe.get("textRun", {})
@@ -165,6 +165,70 @@ def _extract_doc_text(doc: dict) -> str:
return "".join(text_parts)
def _extract_doc_text(doc: dict) -> str:
return _extract_body_text(doc.get("body", {}))
def _flatten_doc_tabs(doc: dict) -> list[dict]:
"""Flatten Google's recursive ``tabs``/``childTabs`` tree (preorder).
A tabbed Doc keeps each tab's content in its own body with an independent
index space; the legacy top-level ``body`` only carries the first tab, so
reads and writes that ignore ``tabs`` silently drop or mistarget content.
Returns [] for the legacy single-body response shape (no ``tabs`` field).
"""
flat: list[dict] = []
def visit(tabs, level):
for tab in tabs or []:
props = tab.get("tabProperties") or {}
doc_tab = tab.get("documentTab") or {}
flat.append({
"tabId": props.get("tabId", ""),
"title": props.get("title", ""),
"level": level,
"body": doc_tab.get("body") or {},
})
visit(tab.get("childTabs"), level + 1)
visit(doc.get("tabs"), 0)
return flat
def _resolve_write_tab(doc: dict, tab_arg: str | None) -> tuple[str | None, dict]:
"""Pick exactly one tab body for a write; never merge index spaces.
Legacy docs (no ``tabs``) return (None, body) — the write carries no tabId.
A multi-tab doc requires an explicit --tab; an unknown ID errors instead of
quietly falling back to the first tab.
"""
tabs = _flatten_doc_tabs(doc)
if not tabs:
return None, doc.get("body", {})
if tab_arg:
for tab in tabs:
if tab["tabId"] == tab_arg:
return tab["tabId"], tab["body"]
print(
json.dumps({
"error": f"unknown tab ID {tab_arg!r}",
"tabs": [{"tabId": t["tabId"], "title": t["title"]} for t in tabs],
}, indent=2, ensure_ascii=False),
file=sys.stderr,
)
sys.exit(1)
if len(tabs) == 1:
return tabs[0]["tabId"], tabs[0]["body"]
print(
json.dumps({
"error": f"document has {len(tabs)} tabs; pass --tab <tabId> to pick one",
"tabs": [{"tabId": t["tabId"], "title": t["title"]} for t in tabs],
}, indent=2, ensure_ascii=False),
file=sys.stderr,
)
sys.exit(1)
def _datetime_with_timezone(value: str) -> str:
if not value:
return value
@@ -953,23 +1017,41 @@ def sheets_create(args):
def docs_get(args):
tab_arg = getattr(args, "tab", None)
params = {"documentId": args.doc_id, "includeTabsContent": True}
if _gws_binary():
doc = _run_gws(["docs", "documents", "get"], params={"documentId": args.doc_id})
result = {
"title": doc.get("title", ""),
"documentId": doc.get("documentId", ""),
"body": _extract_doc_text(doc),
}
print(json.dumps(result, indent=2, ensure_ascii=False))
return
doc = _run_gws(["docs", "documents", "get"], params=params)
else:
service = build_service("docs", "v1")
doc = service.documents().get(
documentId=args.doc_id, includeTabsContent=True,
).execute()
service = build_service("docs", "v1")
doc = service.documents().get(documentId=args.doc_id).execute()
result = {
"title": doc.get("title", ""),
"documentId": doc.get("documentId", ""),
"body": _extract_doc_text(doc),
}
tabs = _flatten_doc_tabs(doc)
if not tabs:
# Legacy single-body response shape.
result["body"] = _extract_doc_text(doc)
elif tab_arg:
_, body = _resolve_write_tab(doc, tab_arg)
result["tab"] = tab_arg
result["body"] = _extract_body_text(body)
else:
result["tabs"] = [
{
"tabId": t["tabId"],
"title": t["title"],
"level": t["level"],
"body": _extract_body_text(t["body"]),
}
for t in tabs
]
# Keep "body" populated for single-tab docs so existing callers work.
if len(tabs) == 1:
result["body"] = result["tabs"][0]["body"]
print(json.dumps(result, indent=2, ensure_ascii=False))
@@ -997,17 +1079,25 @@ def docs_create(args):
def docs_append(args):
"""Append text to the end of an existing Doc."""
"""Append text to the end of an existing Doc (one tab of it, if tabbed)."""
if _gws_binary():
doc = _run_gws(["docs", "documents", "get"], params={"documentId": args.doc_id})
doc = _run_gws(
["docs", "documents", "get"],
params={"documentId": args.doc_id, "includeTabsContent": True},
)
else:
service = build_service("docs", "v1")
doc = service.documents().get(documentId=args.doc_id).execute()
doc = service.documents().get(
documentId=args.doc_id, includeTabsContent=True,
).execute()
tab_id, body = _resolve_write_tab(doc, getattr(args, "tab", None))
# The end-of-body index is one less than the segment endIndex of the body
# (trailing newline is always at length-1). Docs indexes are 1-based; use
# endIndex - 1 to insert before the final newline.
content = doc.get("body", {}).get("content", [])
# endIndex - 1 to insert before the final newline. Each tab has its own
# index space, so the write location must carry the tab ID.
content = body.get("content", [])
end_index = 1
for element in content:
ei = element.get("endIndex")
@@ -1016,21 +1106,27 @@ def docs_append(args):
insert_index = max(end_index - 1, 1)
text = args.text if args.text.endswith("\n") else args.text + "\n"
_docs_insert_text(args.doc_id, text, index=insert_index)
_docs_insert_text(args.doc_id, text, index=insert_index, tab_id=tab_id)
print(json.dumps({
result = {
"status": "appended",
"documentId": args.doc_id,
"inserted_at": insert_index,
"characters": len(text),
}, indent=2, ensure_ascii=False))
}
if tab_id:
result["tab"] = tab_id
print(json.dumps(result, indent=2, ensure_ascii=False))
def _docs_insert_text(doc_id: str, text: str, index: int) -> None:
def _docs_insert_text(doc_id: str, text: str, index: int, tab_id: str | None = None) -> None:
"""Send a batchUpdate with a single insertText request."""
location: dict = {"index": index}
if tab_id:
location["tabId"] = tab_id
requests = [{
"insertText": {
"location": {"index": index},
"location": location,
"text": text,
}
}]
@@ -1205,6 +1301,7 @@ def main():
p = docs_sub.add_parser("get")
p.add_argument("doc_id")
p.add_argument("--tab", default=None, help="Tab ID to read (tabbed Docs)")
p.set_defaults(func=docs_get)
p = docs_sub.add_parser("create")
@@ -1215,6 +1312,7 @@ def main():
p = docs_sub.add_parser("append")
p.add_argument("doc_id")
p.add_argument("--text", required=True, help="Text to append to the end of the document")
p.add_argument("--tab", default=None, help="Tab ID to append to (required for multi-tab Docs)")
p.set_defaults(func=docs_append)
args = parser.parse_args()

View File

@@ -65,7 +65,7 @@ def _write_token(path: Path, *, token="ya29.test", expiry=None, **extra):
}
if expiry is not None:
data["expiry"] = expiry
path.write_text(json.dumps(data))
path.write_text(json.dumps(data), encoding="utf-8")
def test_bridge_returns_valid_token(bridge_module, tmp_path):
@@ -191,7 +191,81 @@ def test_api_get_credentials_refresh_persists_authorized_user_type(api_module, m
creds = api_module.get_credentials()
saved = json.loads(token_path.read_text())
saved = json.loads(token_path.read_text(encoding="utf-8"))
assert isinstance(creds, FakeCredentials)
assert saved["token"] == "ya29.refreshed"
assert saved["type"] == "authorized_user"
def _tabbed_doc():
"""A Doc with two tabs (one nested), as the Docs API returns with includeTabsContent."""
def body(text):
return {"content": [
{"endIndex": len(text) + 2,
"paragraph": {"elements": [{"textRun": {"content": text + "\n"}}]}},
]}
return {
"title": "Tabbed",
"documentId": "doc1",
"tabs": [
{
"tabProperties": {"tabId": "t.0", "title": "First"},
"documentTab": {"body": body("alpha")},
"childTabs": [
{
"tabProperties": {"tabId": "t.0.a", "title": "Nested"},
"documentTab": {"body": body("beta")},
}
],
},
{
"tabProperties": {"tabId": "t.1", "title": "Second"},
"documentTab": {"body": body("gamma")},
},
],
}
def test_docs_get_returns_every_tab_of_a_tabbed_doc(api_module, monkeypatch, capsys):
"""A multi-tab Doc must not lose tab content: reads traverse the tabs tree
(preorder, nested tabs included) instead of only the legacy top-level body."""
monkeypatch.setattr(
api_module, "_run_gws",
lambda parts, params=None, body=None: _tabbed_doc(),
)
args = types.SimpleNamespace(doc_id="doc1", tab=None)
api_module.docs_get(args)
result = json.loads(capsys.readouterr().out)
tabs = {t["tabId"]: t for t in result["tabs"]}
assert set(tabs) == {"t.0", "t.0.a", "t.1"}
assert tabs["t.0.a"]["body"] == "beta\n"
assert tabs["t.0.a"]["level"] == 1
# Multi-tab docs have no single merged "body" — index spaces are independent.
assert "body" not in result
def test_docs_append_carries_tab_id_and_refuses_ambiguous_writes(api_module, monkeypatch, capsys):
"""Each tab has its own index space, so a write must target exactly one tab:
the insert location carries the tabId, and an un-targeted write against a
multi-tab doc errors instead of silently landing in the first tab."""
monkeypatch.setattr(
api_module, "_run_gws",
lambda parts, params=None, body=None: _tabbed_doc(),
)
sent = {}
monkeypatch.setattr(
api_module, "_docs_insert_text",
lambda doc_id, text, index, tab_id=None: sent.update(
{"doc_id": doc_id, "index": index, "tab_id": tab_id}
),
)
api_module.docs_append(types.SimpleNamespace(doc_id="doc1", text="more", tab="t.1"))
assert sent["tab_id"] == "t.1"
assert sent["index"] == len("gamma") + 1 # endIndex - 1 within THAT tab's space
capsys.readouterr()
with pytest.raises(SystemExit):
api_module.docs_append(types.SimpleNamespace(doc_id="doc1", text="more", tab=None))
err = json.loads(capsys.readouterr().err)
assert "tabs" in err and len(err["tabs"]) == 3