fix(docs): index every docs page in llms.txt, not a hand-picked 98

The section list decided membership as well as order, so it drifted as the
docs grew: 109 of 204 pages were absent from the index every LLM reads to
learn what Hermes does — Bot Mode, the desktop app, computer use, web search,
skins, Mixture of Agents, and 22 messaging platforms among them.

Enumerate the docs tree instead. SECTIONS now curates only which pages lead a
section; anything it does not name is absorbed under its path, and a page
matching no section lands in "More" rather than falling out. This also picks
up the three .mdx pages the .md-only glob never saw, points section landing
pages at the directory URL Docusaurus actually serves, and drops a curated row
still aimed at a guide moved to developer-guide/plugins in #59613.

Tests hold both directions against the filesystem rather than the enumerator,
so a page cannot go missing and a link cannot point at a page that moved.
This commit is contained in:
Brooklyn Nicholson
2026-08-19 12:05:41 -05:00
committed by brooklyn!
parent bdc5b1f74c
commit a45d854d7d
2 changed files with 280 additions and 67 deletions

View File

@@ -0,0 +1,133 @@
"""`llms.txt` is how an LLM learns what Hermes can do.
It is the index every model reads when pointed at our docs — including Hermes
itself, whose `hermes-agent` skill routes unknown-feature questions there.
`website/` is never packaged, so there is no shipped copy to fall back on.
The index used to be a hand-written list of page paths, and it rotted to 53%
coverage: Bot Mode, the desktop app, computer use, web search, and 22 messaging
platforms were all absent, which is why an agent asked how to make bots talk to
each other answered that it couldn't. These tests hold the two directions of
that contract — every page reachable, every link real — so the index tracks the
docs tree instead of someone's memory of it.
"""
from __future__ import annotations
import importlib.util
import re
from pathlib import Path
import pytest
REPO_ROOT = Path(__file__).resolve().parents[2]
GENERATOR = REPO_ROOT / "website" / "scripts" / "generate-llms-txt.py"
@pytest.fixture(scope="module")
def gen():
spec = importlib.util.spec_from_file_location("generate_llms_txt", GENERATOR)
assert spec is not None and spec.loader is not None
module = importlib.util.module_from_spec(spec)
spec.loader.exec_module(module)
return module
@pytest.fixture(scope="module")
def index(gen) -> str:
return gen.emit_llms_index()
def _linked(gen, index: str) -> set[str]:
return set(re.findall(rf"\]\({re.escape(gen.SITE_BASE)}/([^)]+)\)", index))
def _pages_on_disk(gen) -> set[str]:
"""Walk the docs tree directly, duplicating only the two documented
exclusions.
Deliberately does not call `iter_docs()`: the index is built from that
enumeration, so checking one against the other would pass even if the
enumerator went blind to a whole tree — which is the failure being guarded.
"""
pages = set()
for path in (*gen.DOCS.rglob("*.md"), *gen.DOCS.rglob("*.mdx")):
rel = path.relative_to(gen.DOCS).with_suffix("")
slug = str(rel.parent) if rel.name == "index" else str(rel)
# The docs landing page is the index's subject; per-skill pages are
# summarized by the two catalog reference pages.
if slug == "." or slug.startswith(("user-guide/skills/bundled", "user-guide/skills/optional")):
continue
pages.add(slug)
return pages
def test_every_docs_page_is_indexed(gen, index):
"""The regression: a page the index omits is a feature the agent denies."""
pages = _pages_on_disk(gen)
assert len(pages) > 100, "docs root resolved wrong — the rest of this file proves nothing"
missing = pages - _linked(gen, index)
assert not missing, (
f"{len(missing)} docs pages missing from llms.txt: {sorted(missing)} — "
"they should have been absorbed into a section automatically"
)
def test_the_enumerator_sees_the_whole_docs_tree(gen):
"""Everything downstream trusts `iter_docs()`, so pin it to the filesystem."""
assert set(gen.iter_docs()) == _pages_on_disk(gen)
def test_every_indexed_page_exists(gen, index):
"""The other direction: a renamed page leaves the index pointing at a 404."""
for slug in sorted(_linked(gen, index)):
assert gen.doc_path(slug) is not None, (
f"llms.txt links {slug}, which is not in the docs tree — "
"drop the SECTIONS row and let the page be absorbed under its new path"
)
def test_pages_are_listed_once(gen, index):
"""Curating a page must promote it, not duplicate it."""
entries = re.findall(rf"^- \[.*?\]\({re.escape(gen.SITE_BASE)}/([^)]+)\)", index, re.MULTILINE)
duplicated = {slug for slug in entries if entries.count(slug) > 1}
assert not duplicated, f"listed more than once in llms.txt: {sorted(duplicated)}"
def test_curation_orders_pages_without_gatekeeping_them(gen):
"""SECTIONS decides what leads a section, never what the index contains."""
curated = {slug for _section, items in gen.SECTIONS for slug, _t, _d in items}
pages = set(gen.iter_docs())
assert curated < pages, "every page is curated — absorption is no longer exercised"
assert gen.section_for("user-guide/features/some-feature-shipped-tomorrow") in dict(gen.ABSORB)
assert gen.section_for("a-tree-nobody-anticipated/page") == gen.MISC_SECTION
def test_section_landing_pages_resolve_to_their_directory(gen):
"""`messaging/index.md` is served at `/messaging`; `/messaging/index` 404s."""
assert gen.slug_for(gen.DOCS / "user-guide" / "messaging" / "index.md") == "user-guide/messaging"
assert "user-guide/messaging/index" not in _linked(gen, gen.emit_llms_index())
def test_mdx_pages_are_indexed_without_their_imports(gen):
"""MDX docs are real pages; their component imports are not prose."""
mdx = [p for p in gen.DOCS.rglob("*.mdx") if gen.slug_for(p)]
assert mdx, "no .mdx docs — this test no longer guards anything"
assert {gen.slug_for(p) for p in mdx} <= set(gen.iter_docs())
_meta, body = gen.read_frontmatter(mdx[0])
assert not re.search(r"^import\s", body, re.MULTILINE)
def test_per_skill_catalog_pages_stay_out(gen):
"""~195 generated skill pages would bury the product docs in the index."""
assert not [slug for slug in gen.iter_docs() if slug.startswith(gen.SKILL_CATALOG)]
assert "reference/skills-catalog" in gen.iter_docs(), "the summary page must remain"
def test_bot_mode_is_reachable(gen, index):
"""The page behind the original complaint, and the answer it has to carry."""
assert "user-guide/bot-mode" in _linked(gen, index)
assert "hermes peer dm" in (gen.DOCS / "user-guide" / "bot-mode.md").read_text(encoding="utf-8")

View File

@@ -2,12 +2,20 @@
"""Generate llms.txt and llms-full.txt for the Hermes docs site.
Outputs:
website/static/llms.txt — short curated index of the docs, one link per page,
grouped by section. Conforms to https://llmstxt.org.
website/static/llms-full.txt — every `.md` file under `website/docs/` concatenated,
website/static/llms.txt — index of the docs, one link per page, grouped by
section. Conforms to https://llmstxt.org.
website/static/llms-full.txt — every doc under `website/docs/` concatenated,
with `# <title>` headings and `<!-- source: … -->`
comments separating files.
Both are driven by `iter_docs()`, which walks the docs tree. `SECTIONS` below
curates *order and grouping*, never membership: a page nobody curated still
gets indexed, under the section its path belongs to. That distinction is the
reason this file was rewritten — when the section list also decided membership,
it silently drifted to 53% coverage, and Bot Mode, the desktop app, computer
use, web search, and 22 messaging platforms were absent from the index every
LLM reads to learn what Hermes does.
Both publish at:
https://hermes-agent.nousresearch.com/docs/llms.txt
https://hermes-agent.nousresearch.com/docs/llms-full.txt
@@ -33,9 +41,11 @@ STATIC = WEBSITE / "static"
SITE_BASE = "https://hermes-agent.nousresearch.com/docs"
# Curated sections for llms.txt — mirrors the product story, not the filesystem.
# Each entry: (docs-relative path without .md, display title, optional short desc).
# `None` desc → pulled from frontmatter `description:` field.
# The product story: which pages lead, and in what order. Everything not named
# here is still indexed — ABSORB decides where it lands — so this list is safe
# to leave alone as the docs grow, and worth editing only to promote a page.
# Each entry: (docs-relative path without extension, display title, optional
# short desc). `None` desc → pulled from frontmatter `description:` field.
SECTIONS: list[tuple[str, list[tuple[str, str, str | None]]]] = [
("Getting Started", [
("getting-started/installation", "Installation", None),
@@ -88,7 +98,7 @@ SECTIONS: list[tuple[str, list[tuple[str, str, str | None]]]] = [
("user-guide/features/tts", "Text-to-Speech", None),
]),
("Messaging Platforms", [
("user-guide/messaging/index", "Overview", None),
("user-guide/messaging", "Overview", None),
("user-guide/messaging/telegram", "Telegram", None),
("user-guide/messaging/discord", "Discord", None),
("user-guide/messaging/slack", "Slack", None),
@@ -102,7 +112,7 @@ SECTIONS: list[tuple[str, list[tuple[str, str, str | None]]]] = [
("user-guide/messaging/webhooks", "Webhooks", None),
]),
("Integrations", [
("integrations/index", "Integrations Overview", None),
("integrations", "Integrations Overview", None),
("integrations/providers", "Providers", None),
("user-guide/features/mcp", "MCP (Model Context Protocol)", None),
("user-guide/features/acp", "ACP (Agent Context Protocol)", None),
@@ -121,7 +131,6 @@ SECTIONS: list[tuple[str, list[tuple[str, str, str | None]]]] = [
("guides/use-mcp-with-hermes", "Use MCP with Hermes", None),
("guides/use-voice-mode-with-hermes", "Use Voice Mode with Hermes", None),
("guides/use-soul-with-hermes", "Use SOUL.md with Hermes", None),
("guides/build-a-hermes-plugin", "Build a Hermes Plugin", None),
("guides/automate-with-cron", "Automate with Cron", None),
("guides/work-with-skills", "Work with Skills", None),
("guides/delegation-patterns", "Delegation Patterns", None),
@@ -158,9 +167,40 @@ SECTIONS: list[tuple[str, list[tuple[str, str, str | None]]]] = [
]
DOC_EXTS = (".md", ".mdx")
# Per-skill pages are generated from the skill tree and summarized by the two
# catalog reference pages. Listing ~195 of them would bury the product docs in
# the index and add ~1.4 MB of duplicative material to llms-full.txt.
SKILL_CATALOG = ("user-guide/skills/bundled", "user-guide/skills/optional")
# Where a page nobody curated goes. First match wins, so a narrower prefix must
# precede the tree containing it. Anything matching nothing lands in
# MISC_SECTION — no path can drop a page out of the index.
ABSORB: tuple[tuple[str, tuple[str, ...]], ...] = (
("Getting Started", ("getting-started",)),
("Messaging Platforms", ("user-guide/messaging",)),
("Core Features", ("user-guide/features",)),
("Using Hermes", ("user-guide",)),
("Integrations", ("integrations",)),
("Guides & Tutorials", ("guides",)),
("Developer Guide", ("developer-guide",)),
("Reference", ("reference",)),
)
MISC_SECTION = "More"
FRONTMATTER_RE = re.compile(r"^---\s*\n(.*?)\n---\s*\n", re.DOTALL)
DESC_RE = re.compile(r"^description:\s*[\"'](.+?)[\"']\s*$", re.MULTILINE)
TITLE_RE = re.compile(r"^title:\s*[\"'](.+?)[\"']\s*$", re.MULTILINE)
DESC_RE = re.compile(r"^description:\s*(.+?)\s*$", re.MULTILINE)
TITLE_RE = re.compile(r"^title:\s*(.+?)\s*$", re.MULTILINE)
H1_RE = re.compile(r"^#\s+(.+?)\s*$", re.MULTILINE)
# MDX pages open with component imports — markup plumbing, not prose.
MDX_IMPORT_RE = re.compile(r"^(?:import|export)\s.*$\n?", re.MULTILINE)
def _unquote(value: str) -> str:
if len(value) >= 2 and value[0] == value[-1] and value[0] in "\"'":
return value[1:-1]
return value
def read_frontmatter(path: Path) -> tuple[dict[str, str], str]:
@@ -172,30 +212,86 @@ def read_frontmatter(path: Path) -> tuple[dict[str, str], str]:
if m:
fm = m.group(1)
body = text[m.end():]
dm = DESC_RE.search(fm)
if dm:
meta["description"] = dm.group(1)
tm = TITLE_RE.search(fm)
if tm:
meta["title"] = tm.group(1)
for key, pattern in (("description", DESC_RE), ("title", TITLE_RE)):
found = pattern.search(fm)
if found:
meta[key] = _unquote(found.group(1))
if path.suffix == ".mdx":
body = MDX_IMPORT_RE.sub("", body)
return meta, body
def slug_for(path: Path) -> str:
"""URL slug for a page: `user-guide/messaging/index.md` → `user-guide/messaging`."""
rel = path.relative_to(DOCS).with_suffix("")
if rel.name == "index":
rel = rel.parent
return "" if str(rel) == "." else str(rel)
def doc_path(slug: str) -> Path | None:
"""The file backing a slug, whether it's a page or a section landing page."""
for ext in DOC_EXTS:
for candidate in (DOCS / f"{slug}{ext}", DOCS / slug / f"index{ext}"):
if candidate.exists():
return candidate
return None
def iter_docs() -> list[str]:
"""Every indexable page, as a slug — the one enumeration of the docs tree.
llms.txt lists exactly these and llms-full.txt emits exactly these, so a
page on disk cannot be missing from either output.
"""
slugs = set()
for ext in DOC_EXTS:
for path in DOCS.rglob(f"*{ext}"):
slug = slug_for(path)
# The docs landing page is this index's subject, not an entry in it.
if slug and not slug.startswith(SKILL_CATALOG):
slugs.add(slug)
return sorted(slugs)
def section_for(slug: str) -> str:
for section, prefixes in ABSORB:
if any(slug == prefix or slug.startswith(f"{prefix}/") for prefix in prefixes):
return section
return MISC_SECTION
def resolve_meta(slug: str) -> tuple[str, str]:
"""(title, description) for a page, falling back to its H1 then its slug."""
path = doc_path(slug)
if path is None:
return slug, ""
meta, body = read_frontmatter(path)
title = meta.get("title")
if not title:
h1 = H1_RE.search(body)
title = h1.group(1) if h1 else slug.rsplit("/", 1)[-1].replace("-", " ").title()
return title, meta.get("description", "")
def resolve_desc(slug: str, provided: str | None) -> str:
"""Resolve short description for llms.txt entry."""
if provided:
return provided
path = DOCS / f"{slug}.md"
if not path.exists():
path = DOCS / slug / "index.md"
if not path.exists():
return ""
meta, _ = read_frontmatter(path)
return meta.get("description", "")
return provided or resolve_meta(slug)[1]
def _entry(slug: str, title: str, desc: str) -> str:
url = f"{SITE_BASE}/{slug}"
return f"- [{title}]({url}): {desc}" if desc else f"- [{title}]({url})"
def emit_llms_index() -> str:
"""Build the short llms.txt index."""
"""Build the llms.txt index: curated pages lead a section, the rest follow."""
curated = {slug for _section, items in SECTIONS for slug, _title, _desc in items}
absorbed: dict[str, list[str]] = {}
for slug in iter_docs():
if slug not in curated:
absorbed.setdefault(section_for(slug), []).append(slug)
lines: list[str] = []
lines.append("# Hermes Agent")
lines.append("")
@@ -222,23 +318,24 @@ def emit_llms_index() -> str:
lines.append(f"## {section}")
lines.append("")
for slug, title, desc_override in items:
desc = resolve_desc(slug, desc_override)
url = f"{SITE_BASE}/{slug}"
if desc:
lines.append(f"- [{title}]({url}): {desc}")
else:
lines.append(f"- [{title}]({url})")
lines.append(_entry(slug, title, resolve_desc(slug, desc_override)))
for slug in absorbed.pop(section, []):
lines.append(_entry(slug, *resolve_meta(slug)))
lines.append("")
# Only MISC_SECTION can survive the pops — a page whose path matched no
# section still has to appear somewhere.
for section, slugs in absorbed.items():
lines.append(f"## {section}")
lines.append("")
for slug in slugs:
lines.append(_entry(slug, *resolve_meta(slug)))
lines.append("")
return "\n".join(lines).rstrip() + "\n"
def emit_llms_full() -> str:
"""Concatenate every doc under website/docs/ into a single markdown file.
Order: mirrors the curated SECTIONS list first (so the most important
pages are front-loaded for agents that truncate on token budget), then
appends any remaining .md files sorted by path.
"""
"""Concatenate every doc under website/docs/ into a single markdown file."""
seen: set[Path] = set()
chunks: list[str] = [
"# Hermes Agent — Full Documentation\n",
@@ -253,41 +350,24 @@ def emit_llms_full() -> str:
"\n---\n\n",
]
def emit_file(rel: str) -> None:
path = DOCS / f"{rel}.md"
if not path.exists():
path = DOCS / rel / "index.md"
if not path.exists() or path in seen:
def emit_file(slug: str) -> None:
path = doc_path(slug)
if path is None or path in seen:
return
seen.add(path)
meta, body = read_frontmatter(path)
title = meta.get("title") or rel
title, _desc = resolve_meta(slug)
_meta, body = read_frontmatter(path)
chunks.append(f"<!-- source: website/docs/{path.relative_to(DOCS)} -->\n")
chunks.append(f"# {title}\n\n")
chunks.append(body.rstrip() + "\n\n---\n\n")
# Curated order first
for _, items in SECTIONS:
for slug, _t, _d in items:
# Curated order first, so a reader truncating on token budget keeps the
# pages that matter most; then everything else the docs tree holds.
for _section, items in SECTIONS:
for slug, _title, _desc in items:
emit_file(slug)
# Everything else (sorted, skipping already emitted and auto-gen skill pages
# — those are covered by the two catalog reference pages, emitting every
# individual skill would add ~1.4 MB of largely duplicative material).
for path in sorted(DOCS.rglob("*.md")):
if path in seen:
continue
rel = path.relative_to(DOCS)
parts = rel.parts
if len(parts) >= 3 and parts[0] == "user-guide" and parts[1] == "skills" \
and parts[2] in {"bundled", "optional"}:
continue
seen.add(path)
meta, body = read_frontmatter(path)
title = meta.get("title") or str(rel)
chunks.append(f"<!-- source: website/docs/{rel} -->\n")
chunks.append(f"# {title}\n\n")
chunks.append(body.rstrip() + "\n\n---\n\n")
for slug in iter_docs():
emit_file(slug)
return "".join(chunks).rstrip() + "\n"