From 50f17b13006ee94306663ecf5d63c64def63d499 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 9 Sep 2026 02:10:34 -0700 Subject: [PATCH] fix(google-workspace): support multi-tab Google Docs (port cloudflare/cloudflare-os#450) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Google Docs can hold a tree of tabs, each with its own body and its own 1-based character index space; the legacy top-level `body` only carries the first tab. `docs get` was silently dropping every other tab's content, and `docs append` computed its insert index from the first tab's body and sent the write with no tabId — so on a tabbed doc the append could land at a wrong offset in the wrong tab. - Requests now pass includeTabsContent=true (both gws and SDK paths). - `docs get` returns a `tabs` array (preorder flattening of the tabs/childTabs tree, nested tabs included); single-tab docs keep the `body` field so existing callers work, and `--tab ` reads one tab. Legacy no-tabs responses are unchanged. - `docs append` targets exactly one tab: the insert location carries the tabId, the end index is computed inside that tab's own body, an unknown `--tab` errors instead of falling back to the first tab, and a multi-tab doc without `--tab` errors with the tab list rather than guessing. Tabs are never merged — index spaces are independent. Adapted from cloudflare/cloudflare-os#450 (gatekeeper-google), which fixed the same provider behavior: reads must traverse Document.tabs and every write Location must carry the immutable tabId. Two pre-existing bare read_text/write_text in the touched test file gained encoding="utf-8" (windows-footgun sweep rule). --- skills/productivity/google-workspace/SKILL.md | 4 +- .../google-workspace/scripts/google_api.py | 144 +++++++++++++++--- tests/skills/test_google_workspace_api.py | 78 +++++++++- 3 files changed, 200 insertions(+), 26 deletions(-) diff --git a/skills/productivity/google-workspace/SKILL.md b/skills/productivity/google-workspace/SKILL.md index d27d91b342..7209418e72 100644 --- a/skills/productivity/google-workspace/SKILL.md +++ b/skills/productivity/google-workspace/SKILL.md @@ -276,8 +276,9 @@ $GAPI sheets append SHEET_ID "Sheet1!A:C" --values '[["new","row","data"]]' ### Docs ```bash -# Read +# Read (a tabbed Doc returns a "tabs" array; single-tab and legacy Docs also return "body") $GAPI docs get DOC_ID +$GAPI docs get DOC_ID --tab TAB_ID # read one tab of a tabbed Doc # Create a new Doc (optionally seeded with body text) $GAPI docs create --title "Meeting Notes" @@ -285,6 +286,7 @@ $GAPI docs create --title "Draft" --body "First paragraph..." # Append text to the end of an existing Doc $GAPI docs append DOC_ID --text "Additional content to append" +$GAPI docs append DOC_ID --tab TAB_ID --text "..." # --tab required when the Doc has multiple tabs ``` ## Output Format diff --git a/skills/productivity/google-workspace/scripts/google_api.py b/skills/productivity/google-workspace/scripts/google_api.py index 5c2e82323d..320ee5a3f3 100644 --- a/skills/productivity/google-workspace/scripts/google_api.py +++ b/skills/productivity/google-workspace/scripts/google_api.py @@ -154,9 +154,9 @@ def _extract_message_body(msg: dict) -> str: return body -def _extract_doc_text(doc: dict) -> str: +def _extract_body_text(body: dict) -> str: text_parts = [] - for element in doc.get("body", {}).get("content", []): + for element in body.get("content", []): paragraph = element.get("paragraph", {}) for pe in paragraph.get("elements", []): text_run = pe.get("textRun", {}) @@ -165,6 +165,70 @@ def _extract_doc_text(doc: dict) -> str: return "".join(text_parts) +def _extract_doc_text(doc: dict) -> str: + return _extract_body_text(doc.get("body", {})) + + +def _flatten_doc_tabs(doc: dict) -> list[dict]: + """Flatten Google's recursive ``tabs``/``childTabs`` tree (preorder). + + A tabbed Doc keeps each tab's content in its own body with an independent + index space; the legacy top-level ``body`` only carries the first tab, so + reads and writes that ignore ``tabs`` silently drop or mistarget content. + Returns [] for the legacy single-body response shape (no ``tabs`` field). + """ + flat: list[dict] = [] + + def visit(tabs, level): + for tab in tabs or []: + props = tab.get("tabProperties") or {} + doc_tab = tab.get("documentTab") or {} + flat.append({ + "tabId": props.get("tabId", ""), + "title": props.get("title", ""), + "level": level, + "body": doc_tab.get("body") or {}, + }) + visit(tab.get("childTabs"), level + 1) + + visit(doc.get("tabs"), 0) + return flat + + +def _resolve_write_tab(doc: dict, tab_arg: str | None) -> tuple[str | None, dict]: + """Pick exactly one tab body for a write; never merge index spaces. + + Legacy docs (no ``tabs``) return (None, body) — the write carries no tabId. + A multi-tab doc requires an explicit --tab; an unknown ID errors instead of + quietly falling back to the first tab. + """ + tabs = _flatten_doc_tabs(doc) + if not tabs: + return None, doc.get("body", {}) + if tab_arg: + for tab in tabs: + if tab["tabId"] == tab_arg: + return tab["tabId"], tab["body"] + print( + json.dumps({ + "error": f"unknown tab ID {tab_arg!r}", + "tabs": [{"tabId": t["tabId"], "title": t["title"]} for t in tabs], + }, indent=2, ensure_ascii=False), + file=sys.stderr, + ) + sys.exit(1) + if len(tabs) == 1: + return tabs[0]["tabId"], tabs[0]["body"] + print( + json.dumps({ + "error": f"document has {len(tabs)} tabs; pass --tab to pick one", + "tabs": [{"tabId": t["tabId"], "title": t["title"]} for t in tabs], + }, indent=2, ensure_ascii=False), + file=sys.stderr, + ) + sys.exit(1) + + def _datetime_with_timezone(value: str) -> str: if not value: return value @@ -953,23 +1017,41 @@ def sheets_create(args): def docs_get(args): + tab_arg = getattr(args, "tab", None) + params = {"documentId": args.doc_id, "includeTabsContent": True} if _gws_binary(): - doc = _run_gws(["docs", "documents", "get"], params={"documentId": args.doc_id}) - result = { - "title": doc.get("title", ""), - "documentId": doc.get("documentId", ""), - "body": _extract_doc_text(doc), - } - print(json.dumps(result, indent=2, ensure_ascii=False)) - return + doc = _run_gws(["docs", "documents", "get"], params=params) + else: + service = build_service("docs", "v1") + doc = service.documents().get( + documentId=args.doc_id, includeTabsContent=True, + ).execute() - service = build_service("docs", "v1") - doc = service.documents().get(documentId=args.doc_id).execute() result = { "title": doc.get("title", ""), "documentId": doc.get("documentId", ""), - "body": _extract_doc_text(doc), } + tabs = _flatten_doc_tabs(doc) + if not tabs: + # Legacy single-body response shape. + result["body"] = _extract_doc_text(doc) + elif tab_arg: + _, body = _resolve_write_tab(doc, tab_arg) + result["tab"] = tab_arg + result["body"] = _extract_body_text(body) + else: + result["tabs"] = [ + { + "tabId": t["tabId"], + "title": t["title"], + "level": t["level"], + "body": _extract_body_text(t["body"]), + } + for t in tabs + ] + # Keep "body" populated for single-tab docs so existing callers work. + if len(tabs) == 1: + result["body"] = result["tabs"][0]["body"] print(json.dumps(result, indent=2, ensure_ascii=False)) @@ -997,17 +1079,25 @@ def docs_create(args): def docs_append(args): - """Append text to the end of an existing Doc.""" + """Append text to the end of an existing Doc (one tab of it, if tabbed).""" if _gws_binary(): - doc = _run_gws(["docs", "documents", "get"], params={"documentId": args.doc_id}) + doc = _run_gws( + ["docs", "documents", "get"], + params={"documentId": args.doc_id, "includeTabsContent": True}, + ) else: service = build_service("docs", "v1") - doc = service.documents().get(documentId=args.doc_id).execute() + doc = service.documents().get( + documentId=args.doc_id, includeTabsContent=True, + ).execute() + + tab_id, body = _resolve_write_tab(doc, getattr(args, "tab", None)) # The end-of-body index is one less than the segment endIndex of the body # (trailing newline is always at length-1). Docs indexes are 1-based; use - # endIndex - 1 to insert before the final newline. - content = doc.get("body", {}).get("content", []) + # endIndex - 1 to insert before the final newline. Each tab has its own + # index space, so the write location must carry the tab ID. + content = body.get("content", []) end_index = 1 for element in content: ei = element.get("endIndex") @@ -1016,21 +1106,27 @@ def docs_append(args): insert_index = max(end_index - 1, 1) text = args.text if args.text.endswith("\n") else args.text + "\n" - _docs_insert_text(args.doc_id, text, index=insert_index) + _docs_insert_text(args.doc_id, text, index=insert_index, tab_id=tab_id) - print(json.dumps({ + result = { "status": "appended", "documentId": args.doc_id, "inserted_at": insert_index, "characters": len(text), - }, indent=2, ensure_ascii=False)) + } + if tab_id: + result["tab"] = tab_id + print(json.dumps(result, indent=2, ensure_ascii=False)) -def _docs_insert_text(doc_id: str, text: str, index: int) -> None: +def _docs_insert_text(doc_id: str, text: str, index: int, tab_id: str | None = None) -> None: """Send a batchUpdate with a single insertText request.""" + location: dict = {"index": index} + if tab_id: + location["tabId"] = tab_id requests = [{ "insertText": { - "location": {"index": index}, + "location": location, "text": text, } }] @@ -1205,6 +1301,7 @@ def main(): p = docs_sub.add_parser("get") p.add_argument("doc_id") + p.add_argument("--tab", default=None, help="Tab ID to read (tabbed Docs)") p.set_defaults(func=docs_get) p = docs_sub.add_parser("create") @@ -1215,6 +1312,7 @@ def main(): p = docs_sub.add_parser("append") p.add_argument("doc_id") p.add_argument("--text", required=True, help="Text to append to the end of the document") + p.add_argument("--tab", default=None, help="Tab ID to append to (required for multi-tab Docs)") p.set_defaults(func=docs_append) args = parser.parse_args() diff --git a/tests/skills/test_google_workspace_api.py b/tests/skills/test_google_workspace_api.py index 3f1f9838f2..05c31f74a8 100644 --- a/tests/skills/test_google_workspace_api.py +++ b/tests/skills/test_google_workspace_api.py @@ -65,7 +65,7 @@ def _write_token(path: Path, *, token="ya29.test", expiry=None, **extra): } if expiry is not None: data["expiry"] = expiry - path.write_text(json.dumps(data)) + path.write_text(json.dumps(data), encoding="utf-8") def test_bridge_returns_valid_token(bridge_module, tmp_path): @@ -191,7 +191,81 @@ def test_api_get_credentials_refresh_persists_authorized_user_type(api_module, m creds = api_module.get_credentials() - saved = json.loads(token_path.read_text()) + saved = json.loads(token_path.read_text(encoding="utf-8")) assert isinstance(creds, FakeCredentials) assert saved["token"] == "ya29.refreshed" assert saved["type"] == "authorized_user" + + +def _tabbed_doc(): + """A Doc with two tabs (one nested), as the Docs API returns with includeTabsContent.""" + def body(text): + return {"content": [ + {"endIndex": len(text) + 2, + "paragraph": {"elements": [{"textRun": {"content": text + "\n"}}]}}, + ]} + return { + "title": "Tabbed", + "documentId": "doc1", + "tabs": [ + { + "tabProperties": {"tabId": "t.0", "title": "First"}, + "documentTab": {"body": body("alpha")}, + "childTabs": [ + { + "tabProperties": {"tabId": "t.0.a", "title": "Nested"}, + "documentTab": {"body": body("beta")}, + } + ], + }, + { + "tabProperties": {"tabId": "t.1", "title": "Second"}, + "documentTab": {"body": body("gamma")}, + }, + ], + } + + +def test_docs_get_returns_every_tab_of_a_tabbed_doc(api_module, monkeypatch, capsys): + """A multi-tab Doc must not lose tab content: reads traverse the tabs tree + (preorder, nested tabs included) instead of only the legacy top-level body.""" + monkeypatch.setattr( + api_module, "_run_gws", + lambda parts, params=None, body=None: _tabbed_doc(), + ) + args = types.SimpleNamespace(doc_id="doc1", tab=None) + api_module.docs_get(args) + result = json.loads(capsys.readouterr().out) + tabs = {t["tabId"]: t for t in result["tabs"]} + assert set(tabs) == {"t.0", "t.0.a", "t.1"} + assert tabs["t.0.a"]["body"] == "beta\n" + assert tabs["t.0.a"]["level"] == 1 + # Multi-tab docs have no single merged "body" — index spaces are independent. + assert "body" not in result + + +def test_docs_append_carries_tab_id_and_refuses_ambiguous_writes(api_module, monkeypatch, capsys): + """Each tab has its own index space, so a write must target exactly one tab: + the insert location carries the tabId, and an un-targeted write against a + multi-tab doc errors instead of silently landing in the first tab.""" + monkeypatch.setattr( + api_module, "_run_gws", + lambda parts, params=None, body=None: _tabbed_doc(), + ) + sent = {} + monkeypatch.setattr( + api_module, "_docs_insert_text", + lambda doc_id, text, index, tab_id=None: sent.update( + {"doc_id": doc_id, "index": index, "tab_id": tab_id} + ), + ) + + api_module.docs_append(types.SimpleNamespace(doc_id="doc1", text="more", tab="t.1")) + assert sent["tab_id"] == "t.1" + assert sent["index"] == len("gamma") + 1 # endIndex - 1 within THAT tab's space + capsys.readouterr() + + with pytest.raises(SystemExit): + api_module.docs_append(types.SimpleNamespace(doc_id="doc1", text="more", tab=None)) + err = json.loads(capsys.readouterr().err) + assert "tabs" in err and len(err["tabs"]) == 3