From a45d854d7dbf9d53bbf28f3927a2dffbf45478a6 Mon Sep 17 00:00:00 2001 From: Brooklyn Nicholson Date: Wed, 19 Aug 2026 12:05:41 -0500 Subject: [PATCH] fix(docs): index every docs page in llms.txt, not a hand-picked 98 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The section list decided membership as well as order, so it drifted as the docs grew: 109 of 204 pages were absent from the index every LLM reads to learn what Hermes does — Bot Mode, the desktop app, computer use, web search, skins, Mixture of Agents, and 22 messaging platforms among them. Enumerate the docs tree instead. SECTIONS now curates only which pages lead a section; anything it does not name is absorbed under its path, and a page matching no section lands in "More" rather than falling out. This also picks up the three .mdx pages the .md-only glob never saw, points section landing pages at the directory URL Docusaurus actually serves, and drops a curated row still aimed at a guide moved to developer-guide/plugins in #59613. Tests hold both directions against the filesystem rather than the enumerator, so a page cannot go missing and a link cannot point at a page that moved. --- tests/website/test_generate_llms_txt.py | 133 +++++++++++++++ website/scripts/generate-llms-txt.py | 214 ++++++++++++++++-------- 2 files changed, 280 insertions(+), 67 deletions(-) create mode 100644 tests/website/test_generate_llms_txt.py diff --git a/tests/website/test_generate_llms_txt.py b/tests/website/test_generate_llms_txt.py new file mode 100644 index 0000000000..738ba55a09 --- /dev/null +++ b/tests/website/test_generate_llms_txt.py @@ -0,0 +1,133 @@ +"""`llms.txt` is how an LLM learns what Hermes can do. + +It is the index every model reads when pointed at our docs — including Hermes +itself, whose `hermes-agent` skill routes unknown-feature questions there. +`website/` is never packaged, so there is no shipped copy to fall back on. + +The index used to be a hand-written list of page paths, and it rotted to 53% +coverage: Bot Mode, the desktop app, computer use, web search, and 22 messaging +platforms were all absent, which is why an agent asked how to make bots talk to +each other answered that it couldn't. These tests hold the two directions of +that contract — every page reachable, every link real — so the index tracks the +docs tree instead of someone's memory of it. +""" + +from __future__ import annotations + +import importlib.util +import re +from pathlib import Path + +import pytest + +REPO_ROOT = Path(__file__).resolve().parents[2] +GENERATOR = REPO_ROOT / "website" / "scripts" / "generate-llms-txt.py" + + +@pytest.fixture(scope="module") +def gen(): + spec = importlib.util.spec_from_file_location("generate_llms_txt", GENERATOR) + assert spec is not None and spec.loader is not None + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +@pytest.fixture(scope="module") +def index(gen) -> str: + return gen.emit_llms_index() + + +def _linked(gen, index: str) -> set[str]: + return set(re.findall(rf"\]\({re.escape(gen.SITE_BASE)}/([^)]+)\)", index)) + + +def _pages_on_disk(gen) -> set[str]: + """Walk the docs tree directly, duplicating only the two documented + exclusions. + + Deliberately does not call `iter_docs()`: the index is built from that + enumeration, so checking one against the other would pass even if the + enumerator went blind to a whole tree — which is the failure being guarded. + """ + pages = set() + for path in (*gen.DOCS.rglob("*.md"), *gen.DOCS.rglob("*.mdx")): + rel = path.relative_to(gen.DOCS).with_suffix("") + slug = str(rel.parent) if rel.name == "index" else str(rel) + # The docs landing page is the index's subject; per-skill pages are + # summarized by the two catalog reference pages. + if slug == "." or slug.startswith(("user-guide/skills/bundled", "user-guide/skills/optional")): + continue + pages.add(slug) + return pages + + +def test_every_docs_page_is_indexed(gen, index): + """The regression: a page the index omits is a feature the agent denies.""" + pages = _pages_on_disk(gen) + assert len(pages) > 100, "docs root resolved wrong — the rest of this file proves nothing" + + missing = pages - _linked(gen, index) + assert not missing, ( + f"{len(missing)} docs pages missing from llms.txt: {sorted(missing)} — " + "they should have been absorbed into a section automatically" + ) + + +def test_the_enumerator_sees_the_whole_docs_tree(gen): + """Everything downstream trusts `iter_docs()`, so pin it to the filesystem.""" + assert set(gen.iter_docs()) == _pages_on_disk(gen) + + +def test_every_indexed_page_exists(gen, index): + """The other direction: a renamed page leaves the index pointing at a 404.""" + for slug in sorted(_linked(gen, index)): + assert gen.doc_path(slug) is not None, ( + f"llms.txt links {slug}, which is not in the docs tree — " + "drop the SECTIONS row and let the page be absorbed under its new path" + ) + + +def test_pages_are_listed_once(gen, index): + """Curating a page must promote it, not duplicate it.""" + entries = re.findall(rf"^- \[.*?\]\({re.escape(gen.SITE_BASE)}/([^)]+)\)", index, re.MULTILINE) + duplicated = {slug for slug in entries if entries.count(slug) > 1} + assert not duplicated, f"listed more than once in llms.txt: {sorted(duplicated)}" + + +def test_curation_orders_pages_without_gatekeeping_them(gen): + """SECTIONS decides what leads a section, never what the index contains.""" + curated = {slug for _section, items in gen.SECTIONS for slug, _t, _d in items} + pages = set(gen.iter_docs()) + + assert curated < pages, "every page is curated — absorption is no longer exercised" + assert gen.section_for("user-guide/features/some-feature-shipped-tomorrow") in dict(gen.ABSORB) + assert gen.section_for("a-tree-nobody-anticipated/page") == gen.MISC_SECTION + + +def test_section_landing_pages_resolve_to_their_directory(gen): + """`messaging/index.md` is served at `/messaging`; `/messaging/index` 404s.""" + assert gen.slug_for(gen.DOCS / "user-guide" / "messaging" / "index.md") == "user-guide/messaging" + assert "user-guide/messaging/index" not in _linked(gen, gen.emit_llms_index()) + + +def test_mdx_pages_are_indexed_without_their_imports(gen): + """MDX docs are real pages; their component imports are not prose.""" + mdx = [p for p in gen.DOCS.rglob("*.mdx") if gen.slug_for(p)] + assert mdx, "no .mdx docs — this test no longer guards anything" + assert {gen.slug_for(p) for p in mdx} <= set(gen.iter_docs()) + + _meta, body = gen.read_frontmatter(mdx[0]) + assert not re.search(r"^import\s", body, re.MULTILINE) + + +def test_per_skill_catalog_pages_stay_out(gen): + """~195 generated skill pages would bury the product docs in the index.""" + assert not [slug for slug in gen.iter_docs() if slug.startswith(gen.SKILL_CATALOG)] + assert "reference/skills-catalog" in gen.iter_docs(), "the summary page must remain" + + +def test_bot_mode_is_reachable(gen, index): + """The page behind the original complaint, and the answer it has to carry.""" + assert "user-guide/bot-mode" in _linked(gen, index) + assert "hermes peer dm" in (gen.DOCS / "user-guide" / "bot-mode.md").read_text(encoding="utf-8") diff --git a/website/scripts/generate-llms-txt.py b/website/scripts/generate-llms-txt.py index a34c57792a..91caeb4ad3 100644 --- a/website/scripts/generate-llms-txt.py +++ b/website/scripts/generate-llms-txt.py @@ -2,12 +2,20 @@ """Generate llms.txt and llms-full.txt for the Hermes docs site. Outputs: - website/static/llms.txt — short curated index of the docs, one link per page, - grouped by section. Conforms to https://llmstxt.org. - website/static/llms-full.txt — every `.md` file under `website/docs/` concatenated, + website/static/llms.txt — index of the docs, one link per page, grouped by + section. Conforms to https://llmstxt.org. + website/static/llms-full.txt — every doc under `website/docs/` concatenated, with `# ` headings and `<!-- source: … -->` comments separating files. +Both are driven by `iter_docs()`, which walks the docs tree. `SECTIONS` below +curates *order and grouping*, never membership: a page nobody curated still +gets indexed, under the section its path belongs to. That distinction is the +reason this file was rewritten — when the section list also decided membership, +it silently drifted to 53% coverage, and Bot Mode, the desktop app, computer +use, web search, and 22 messaging platforms were absent from the index every +LLM reads to learn what Hermes does. + Both publish at: https://hermes-agent.nousresearch.com/docs/llms.txt https://hermes-agent.nousresearch.com/docs/llms-full.txt @@ -33,9 +41,11 @@ STATIC = WEBSITE / "static" SITE_BASE = "https://hermes-agent.nousresearch.com/docs" -# Curated sections for llms.txt — mirrors the product story, not the filesystem. -# Each entry: (docs-relative path without .md, display title, optional short desc). -# `None` desc → pulled from frontmatter `description:` field. +# The product story: which pages lead, and in what order. Everything not named +# here is still indexed — ABSORB decides where it lands — so this list is safe +# to leave alone as the docs grow, and worth editing only to promote a page. +# Each entry: (docs-relative path without extension, display title, optional +# short desc). `None` desc → pulled from frontmatter `description:` field. SECTIONS: list[tuple[str, list[tuple[str, str, str | None]]]] = [ ("Getting Started", [ ("getting-started/installation", "Installation", None), @@ -88,7 +98,7 @@ SECTIONS: list[tuple[str, list[tuple[str, str, str | None]]]] = [ ("user-guide/features/tts", "Text-to-Speech", None), ]), ("Messaging Platforms", [ - ("user-guide/messaging/index", "Overview", None), + ("user-guide/messaging", "Overview", None), ("user-guide/messaging/telegram", "Telegram", None), ("user-guide/messaging/discord", "Discord", None), ("user-guide/messaging/slack", "Slack", None), @@ -102,7 +112,7 @@ SECTIONS: list[tuple[str, list[tuple[str, str, str | None]]]] = [ ("user-guide/messaging/webhooks", "Webhooks", None), ]), ("Integrations", [ - ("integrations/index", "Integrations Overview", None), + ("integrations", "Integrations Overview", None), ("integrations/providers", "Providers", None), ("user-guide/features/mcp", "MCP (Model Context Protocol)", None), ("user-guide/features/acp", "ACP (Agent Context Protocol)", None), @@ -121,7 +131,6 @@ SECTIONS: list[tuple[str, list[tuple[str, str, str | None]]]] = [ ("guides/use-mcp-with-hermes", "Use MCP with Hermes", None), ("guides/use-voice-mode-with-hermes", "Use Voice Mode with Hermes", None), ("guides/use-soul-with-hermes", "Use SOUL.md with Hermes", None), - ("guides/build-a-hermes-plugin", "Build a Hermes Plugin", None), ("guides/automate-with-cron", "Automate with Cron", None), ("guides/work-with-skills", "Work with Skills", None), ("guides/delegation-patterns", "Delegation Patterns", None), @@ -158,9 +167,40 @@ SECTIONS: list[tuple[str, list[tuple[str, str, str | None]]]] = [ ] +DOC_EXTS = (".md", ".mdx") + +# Per-skill pages are generated from the skill tree and summarized by the two +# catalog reference pages. Listing ~195 of them would bury the product docs in +# the index and add ~1.4 MB of duplicative material to llms-full.txt. +SKILL_CATALOG = ("user-guide/skills/bundled", "user-guide/skills/optional") + +# Where a page nobody curated goes. First match wins, so a narrower prefix must +# precede the tree containing it. Anything matching nothing lands in +# MISC_SECTION — no path can drop a page out of the index. +ABSORB: tuple[tuple[str, tuple[str, ...]], ...] = ( + ("Getting Started", ("getting-started",)), + ("Messaging Platforms", ("user-guide/messaging",)), + ("Core Features", ("user-guide/features",)), + ("Using Hermes", ("user-guide",)), + ("Integrations", ("integrations",)), + ("Guides & Tutorials", ("guides",)), + ("Developer Guide", ("developer-guide",)), + ("Reference", ("reference",)), +) +MISC_SECTION = "More" + FRONTMATTER_RE = re.compile(r"^---\s*\n(.*?)\n---\s*\n", re.DOTALL) -DESC_RE = re.compile(r"^description:\s*[\"'](.+?)[\"']\s*$", re.MULTILINE) -TITLE_RE = re.compile(r"^title:\s*[\"'](.+?)[\"']\s*$", re.MULTILINE) +DESC_RE = re.compile(r"^description:\s*(.+?)\s*$", re.MULTILINE) +TITLE_RE = re.compile(r"^title:\s*(.+?)\s*$", re.MULTILINE) +H1_RE = re.compile(r"^#\s+(.+?)\s*$", re.MULTILINE) +# MDX pages open with component imports — markup plumbing, not prose. +MDX_IMPORT_RE = re.compile(r"^(?:import|export)\s.*$\n?", re.MULTILINE) + + +def _unquote(value: str) -> str: + if len(value) >= 2 and value[0] == value[-1] and value[0] in "\"'": + return value[1:-1] + return value def read_frontmatter(path: Path) -> tuple[dict[str, str], str]: @@ -172,30 +212,86 @@ def read_frontmatter(path: Path) -> tuple[dict[str, str], str]: if m: fm = m.group(1) body = text[m.end():] - dm = DESC_RE.search(fm) - if dm: - meta["description"] = dm.group(1) - tm = TITLE_RE.search(fm) - if tm: - meta["title"] = tm.group(1) + for key, pattern in (("description", DESC_RE), ("title", TITLE_RE)): + found = pattern.search(fm) + if found: + meta[key] = _unquote(found.group(1)) + if path.suffix == ".mdx": + body = MDX_IMPORT_RE.sub("", body) return meta, body +def slug_for(path: Path) -> str: + """URL slug for a page: `user-guide/messaging/index.md` → `user-guide/messaging`.""" + rel = path.relative_to(DOCS).with_suffix("") + if rel.name == "index": + rel = rel.parent + return "" if str(rel) == "." else str(rel) + + +def doc_path(slug: str) -> Path | None: + """The file backing a slug, whether it's a page or a section landing page.""" + for ext in DOC_EXTS: + for candidate in (DOCS / f"{slug}{ext}", DOCS / slug / f"index{ext}"): + if candidate.exists(): + return candidate + return None + + +def iter_docs() -> list[str]: + """Every indexable page, as a slug — the one enumeration of the docs tree. + + llms.txt lists exactly these and llms-full.txt emits exactly these, so a + page on disk cannot be missing from either output. + """ + slugs = set() + for ext in DOC_EXTS: + for path in DOCS.rglob(f"*{ext}"): + slug = slug_for(path) + # The docs landing page is this index's subject, not an entry in it. + if slug and not slug.startswith(SKILL_CATALOG): + slugs.add(slug) + return sorted(slugs) + + +def section_for(slug: str) -> str: + for section, prefixes in ABSORB: + if any(slug == prefix or slug.startswith(f"{prefix}/") for prefix in prefixes): + return section + return MISC_SECTION + + +def resolve_meta(slug: str) -> tuple[str, str]: + """(title, description) for a page, falling back to its H1 then its slug.""" + path = doc_path(slug) + if path is None: + return slug, "" + meta, body = read_frontmatter(path) + title = meta.get("title") + if not title: + h1 = H1_RE.search(body) + title = h1.group(1) if h1 else slug.rsplit("/", 1)[-1].replace("-", " ").title() + return title, meta.get("description", "") + + def resolve_desc(slug: str, provided: str | None) -> str: """Resolve short description for llms.txt entry.""" - if provided: - return provided - path = DOCS / f"{slug}.md" - if not path.exists(): - path = DOCS / slug / "index.md" - if not path.exists(): - return "" - meta, _ = read_frontmatter(path) - return meta.get("description", "") + return provided or resolve_meta(slug)[1] + + +def _entry(slug: str, title: str, desc: str) -> str: + url = f"{SITE_BASE}/{slug}" + return f"- [{title}]({url}): {desc}" if desc else f"- [{title}]({url})" def emit_llms_index() -> str: - """Build the short llms.txt index.""" + """Build the llms.txt index: curated pages lead a section, the rest follow.""" + curated = {slug for _section, items in SECTIONS for slug, _title, _desc in items} + absorbed: dict[str, list[str]] = {} + for slug in iter_docs(): + if slug not in curated: + absorbed.setdefault(section_for(slug), []).append(slug) + lines: list[str] = [] lines.append("# Hermes Agent") lines.append("") @@ -222,23 +318,24 @@ def emit_llms_index() -> str: lines.append(f"## {section}") lines.append("") for slug, title, desc_override in items: - desc = resolve_desc(slug, desc_override) - url = f"{SITE_BASE}/{slug}" - if desc: - lines.append(f"- [{title}]({url}): {desc}") - else: - lines.append(f"- [{title}]({url})") + lines.append(_entry(slug, title, resolve_desc(slug, desc_override))) + for slug in absorbed.pop(section, []): + lines.append(_entry(slug, *resolve_meta(slug))) + lines.append("") + + # Only MISC_SECTION can survive the pops — a page whose path matched no + # section still has to appear somewhere. + for section, slugs in absorbed.items(): + lines.append(f"## {section}") + lines.append("") + for slug in slugs: + lines.append(_entry(slug, *resolve_meta(slug))) lines.append("") return "\n".join(lines).rstrip() + "\n" def emit_llms_full() -> str: - """Concatenate every doc under website/docs/ into a single markdown file. - - Order: mirrors the curated SECTIONS list first (so the most important - pages are front-loaded for agents that truncate on token budget), then - appends any remaining .md files sorted by path. - """ + """Concatenate every doc under website/docs/ into a single markdown file.""" seen: set[Path] = set() chunks: list[str] = [ "# Hermes Agent — Full Documentation\n", @@ -253,41 +350,24 @@ def emit_llms_full() -> str: "\n---\n\n", ] - def emit_file(rel: str) -> None: - path = DOCS / f"{rel}.md" - if not path.exists(): - path = DOCS / rel / "index.md" - if not path.exists() or path in seen: + def emit_file(slug: str) -> None: + path = doc_path(slug) + if path is None or path in seen: return seen.add(path) - meta, body = read_frontmatter(path) - title = meta.get("title") or rel + title, _desc = resolve_meta(slug) + _meta, body = read_frontmatter(path) chunks.append(f"<!-- source: website/docs/{path.relative_to(DOCS)} -->\n") chunks.append(f"# {title}\n\n") chunks.append(body.rstrip() + "\n\n---\n\n") - # Curated order first - for _, items in SECTIONS: - for slug, _t, _d in items: + # Curated order first, so a reader truncating on token budget keeps the + # pages that matter most; then everything else the docs tree holds. + for _section, items in SECTIONS: + for slug, _title, _desc in items: emit_file(slug) - - # Everything else (sorted, skipping already emitted and auto-gen skill pages - # — those are covered by the two catalog reference pages, emitting every - # individual skill would add ~1.4 MB of largely duplicative material). - for path in sorted(DOCS.rglob("*.md")): - if path in seen: - continue - rel = path.relative_to(DOCS) - parts = rel.parts - if len(parts) >= 3 and parts[0] == "user-guide" and parts[1] == "skills" \ - and parts[2] in {"bundled", "optional"}: - continue - seen.add(path) - meta, body = read_frontmatter(path) - title = meta.get("title") or str(rel) - chunks.append(f"<!-- source: website/docs/{rel} -->\n") - chunks.append(f"# {title}\n\n") - chunks.append(body.rstrip() + "\n\n---\n\n") + for slug in iter_docs(): + emit_file(slug) return "".join(chunks).rstrip() + "\n"