Files
hermes-agent/tools/skills_hub_sources.py

431 lines
20 KiB
Python

"""Skills Hub adapters for well-known endpoints, direct URLs, LobeHub, and browse.sh."""
import hashlib
import json
import logging
import re
from typing import Any, Dict, List, Optional, Union
from urllib.parse import quote, urljoin, urlparse, urlunparse
from tools.skills_hub_models import (
GuardedFetchMixin, SkillBundle, SkillMeta, SkillSource, _first_matching, _get_json, _get_text,
_hermes_tags, _memo_json, _parse_frontmatter, _referenced_support_paths,
_validate_bundle_rel_path, _validate_skill_name, hub,
)
logger = logging.getLogger("tools.skills_hub")
# --- Well-known Agent Skills endpoint source adapter ------------------------
class WellKnownSkillSource(GuardedFetchMixin, SkillSource):
"""Read skills from a domain exposing /.well-known/skills/index.json."""
SOURCE_ID = "well-known"
BASE_PATH = "/.well-known/skills"
def _meta(self, parsed: dict, skill_name: str, description: str, files: Any, **extra) -> SkillMeta:
return SkillMeta(
name=skill_name, description=description, source="well-known",
identifier=self._wrap_identifier(parsed["base_url"], skill_name), trust_level="community",
path=skill_name,
extra={"index_url": parsed["index_url"], "base_url": parsed["base_url"], "files": files, **extra},
)
def search(self, query: str, limit: int = 10) -> List[SkillMeta]:
index_url = self._query_to_index_url(query)
parsed = self._parse_index(index_url) if index_url else None
if not parsed:
return []
results: List[SkillMeta] = []
for entry in parsed["skills"][:limit]:
name = entry.get("name")
if not isinstance(name, str) or not name:
continue
files = entry.get("files", ["SKILL.md"])
results.append(self._meta(parsed, name, str(entry.get("description", "")),
files if isinstance(files, list) else ["SKILL.md"]))
return results
def inspect(self, identifier: str) -> Optional[SkillMeta]:
parsed = self._parse_identifier(identifier)
entry = self._index_entry(parsed["index_url"], parsed["skill_name"]) if parsed else None
skill_md = self._fetch_text(f"{parsed['skill_url']}/SKILL.md") if entry else None
if skill_md is None:
return None
fm = _parse_frontmatter(skill_md)
return self._meta(parsed, str(fm.get("name") or parsed["skill_name"]),
str(fm.get("description") or entry.get("description") or ""),
entry.get("files", ["SKILL.md"]), endpoint=parsed["skill_url"])
def fetch(self, identifier: str) -> Optional[SkillBundle]:
parsed = self._parse_identifier(identifier)
if not parsed:
return None
try:
skill_name = _validate_skill_name(parsed["skill_name"])
except ValueError:
logger.warning("Well-known skill identifier contained unsafe skill name: %s", identifier)
return None
entry = self._index_entry(parsed["index_url"], parsed["skill_name"])
if not entry:
return None
files = entry.get("files", ["SKILL.md"])
if not isinstance(files, list) or not files:
files = ["SKILL.md"]
downloaded: Dict[str, str] = {}
for rel_path in files:
if not isinstance(rel_path, str) or not rel_path:
continue
try:
safe_rel_path = _validate_bundle_rel_path(rel_path)
except ValueError:
logger.warning("Well-known skill %s advertised unsafe file path: %r", identifier, rel_path)
return None
text = self._fetch_text(f"{parsed['skill_url']}/{safe_rel_path}")
if text is None:
return None
downloaded[safe_rel_path] = text
if "SKILL.md" not in downloaded:
return None
return SkillBundle(
name=skill_name, files=downloaded, source="well-known",
identifier=self._wrap_identifier(parsed["base_url"], skill_name), trust_level="community",
metadata={"index_url": parsed["index_url"], "base_url": parsed["base_url"],
"endpoint": parsed["skill_url"], "files": files},
)
def _query_to_index_url(self, query: str) -> Optional[str]:
query = query.strip()
if not query.startswith(("http://", "https://")):
return None
if query.endswith("/index.json"):
return query
if f"{self.BASE_PATH}/" in query:
return query.split(f"{self.BASE_PATH}/", 1)[0] + f"{self.BASE_PATH}/index.json"
return query.rstrip("/") + f"{self.BASE_PATH}/index.json"
def _parse_identifier(self, identifier: str) -> Optional[dict]:
raw = identifier[len("well-known:"):] if identifier.startswith("well-known:") else identifier
if not raw.startswith(("http://", "https://")):
return None
parsed_url = urlparse(raw)
clean_url = urlunparse(parsed_url._replace(fragment=""))
if clean_url.endswith("/index.json"):
if not parsed_url.fragment:
return None
base_url, skill_name = clean_url[:-len("/index.json")], parsed_url.fragment
skill_url = f"{base_url}/{skill_name}"
else:
skill_url = clean_url[:-len("/SKILL.md")] if clean_url.endswith("/SKILL.md") else clean_url.rstrip("/")
if f"{self.BASE_PATH}/" not in skill_url:
return None
base_url, skill_name = skill_url.rsplit("/", 1)
return {"index_url": f"{base_url}/index.json", "base_url": base_url,
"skill_name": skill_name, "skill_url": skill_url}
def _parse_index(self, index_url: str) -> Optional[dict]:
def compute():
resp = hub()._guarded_http_get(index_url, timeout=20)
if resp is None or resp.status_code != 200:
return None
try:
data = resp.json()
except json.JSONDecodeError:
return None
skills = data.get("skills", []) if isinstance(data, dict) else []
if not isinstance(skills, list):
return None
return {"index_url": index_url, "base_url": index_url[:-len("/index.json")], "skills": skills}
return _memo_json(f"well_known_index_{hashlib.md5(index_url.encode()).hexdigest()}", compute,
valid=lambda c: isinstance(c, dict) and isinstance(c.get("skills"), list))
def _index_entry(self, index_url: str, skill_name: str) -> Optional[dict]:
parsed = self._parse_index(index_url)
skills = parsed["skills"] if parsed else []
return next((e for e in skills if isinstance(e, dict) and e.get("name") == skill_name), None)
@staticmethod
def _wrap_identifier(base_url: str, skill_name: str) -> str:
return f"well-known:{base_url.rstrip('/')}/{skill_name}"
# --- Direct URL source adapter ----------------------------------------------
class UrlSource(GuardedFetchMixin, SkillSource):
"""Fetch SKILL.md plus explicitly referenced, allowlisted support files.
The identifier IS the URL (``https://example.com/path/SKILL.md``). Bare URLs cannot enumerate a
repository, so only exact references below references/templates/scripts/assets are fetched. The
skill name comes from frontmatter ``name:`` (URL-slug fallback); trust is always ``community``.
"""
SOURCE_ID = "url"
# Skill names must look like identifiers: lowercase letters/digits with optional hyphens/underscores.
# Blocks dangerous (``../evil``) AND useless (``SKILL``, ``README``, empty) candidates before they hit the disk.
_VALID_NAME_RE = re.compile(r"^[a-z][a-z0-9_-]*$")
def search(self, query: str, limit: int = 10) -> List[SkillMeta]:
return [] # search is meaningless for a direct URL
def _matches(self, identifier: str) -> bool:
"""Claim bare HTTP(S) URLs ending in ``.md``; leave wrapped identifiers
and ``/.well-known/skills/`` URLs to their own adapters."""
if not isinstance(identifier, str):
return False
ident = identifier.strip()
if (not ident.lower().startswith(("http://", "https://")) or "/.well-known/skills/" in ident
or ident.rstrip("/").endswith("/index.json")):
return False
try:
return urlparse(ident).path.lower().endswith(".md")
except ValueError:
return False
def _load(self, identifier: str):
"""``(url, text, frontmatter, resolved name)`` for a claimed identifier, else None."""
if not self._matches(identifier):
return None
url = identifier.strip()
text = self._fetch_text(url)
if text is None:
return None
fm = _parse_frontmatter(text)
return url, text, fm, self._resolve_skill_name(fm, url)
def inspect(self, identifier: str) -> Optional[SkillMeta]:
loaded = self._load(identifier)
if loaded is None:
return None
url, _text, fm, name = loaded
raw_tags = _hermes_tags(fm)
return SkillMeta(
name=name or "", description=str(fm.get("description") or ""), source="url", identifier=url,
trust_level="community", path=name or "",
tags=[str(t) for t in raw_tags] if isinstance(raw_tags, list) else [],
extra={"url": url, "awaiting_name": name is None},
)
def fetch(self, identifier: str) -> Optional[SkillBundle]:
loaded = self._load(identifier)
if loaded is None:
return None
url, text, _fm, name = loaded
referenced = _referenced_support_paths(text)
if referenced is None:
return None
files: Dict[str, Union[str, bytes]] = {"SKILL.md": text}
base_url = url.rsplit("/", 1)[0] + "/"
for rel_path in sorted(referenced):
support_url = urljoin(base_url, quote(rel_path, safe="/"))
if urlparse(support_url).netloc != urlparse(url).netloc:
return None
content = self._fetch_bytes(support_url)
if content is None: # A 404ing support file shouldn't sink the whole install.
logger.warning("URL skill %s: referenced support file %r could not be fetched from %s; skipping it",
url, rel_path, support_url)
continue
files[rel_path] = content
# When no name resolves, return the bundle with an empty name and ``awaiting_name=True``: ``do_install``
# prompts on a TTY or refuses non-interactively, without re-downloading after the user picks a name.
skill_name = ""
if name is not None:
try:
skill_name = _validate_skill_name(name)
except ValueError:
logger.warning("URL skill %s produced unsafe skill name: %r", url, name)
return None
return SkillBundle(name=skill_name, files=files, source="url", identifier=url, trust_level="community",
metadata={"url": url, "source_url": url, "awaiting_name": not skill_name})
@classmethod
def _is_valid_skill_name(cls, name: Optional[str]) -> bool:
if not isinstance(name, str):
return False
candidate = name.strip().lower()
return bool(candidate) and candidate not in {"skill", "readme", "index", "unnamed-skill"} and bool(
cls._VALID_NAME_RE.match(candidate))
@classmethod
def _resolve_skill_name(cls, fm: dict, url: str) -> Optional[str]:
"""Frontmatter ``name:`` when valid, else a URL-slug candidate (``.../<name>/SKILL.md`` -> ``<name>``,
``.../<name>.md`` -> ``<name>``). None when nothing usable — the CLI then prompts or refuses rather
than auto-naming something like ``SKILL``."""
fm_name = fm.get("name") if isinstance(fm, dict) else None
if isinstance(fm_name, str) and cls._is_valid_skill_name(fm_name):
return fm_name.strip()
try:
path = urlparse(url).path
except ValueError:
return None
parts = [p for p in path.split("/") if p]
if len(parts) >= 2 and parts[-1].lower() == "skill.md" and cls._is_valid_skill_name(parts[-2]):
return parts[-2]
candidate = re.sub(r"\.md$", "", parts[-1], flags=re.IGNORECASE) if parts else ""
return candidate if cls._is_valid_skill_name(candidate) else None
# --- LobeHub source adapter -------------------------------------------------
class LobeHubSource(SkillSource):
"""LobeHub agent marketplace (14,500+ system-prompt agents, converted to
SKILL.md on fetch). Data lives in GitHub: lobehub/lobe-chat-agents."""
SOURCE_ID = "lobehub"
INDEX_URL = "https://chat-agents.lobehub.com/index.json"
def _agents(self) -> Optional[list]:
index = self._fetch_index()
agents = (index.get("agents", index) if isinstance(index, dict) else index) if index else None
return agents if isinstance(agents, list) else None
@staticmethod
def _agent_id(identifier: str) -> str:
return identifier.split("/", 1)[-1] if identifier.startswith("lobehub/") else identifier
@staticmethod
def _agent_meta(agent: dict, name: str, description: str) -> SkillMeta:
tags = agent.get("meta", agent).get("tags", [])
return SkillMeta(name=name, description=description, source="lobehub", identifier=f"lobehub/{name}",
trust_level="community", tags=tags if isinstance(tags, list) else [])
def search(self, query: str, limit: int = 10) -> List[SkillMeta]:
agents = self._agents()
if agents is None:
return []
def fields(agent):
meta = agent.get("meta", agent)
tags = meta.get("tags", [])
return (meta.get("title", agent.get("identifier", "")), meta.get("description", ""),
tags if isinstance(tags, list) else "")
def to_meta(agent):
meta = agent.get("meta", agent)
title = meta.get("title", agent.get("identifier", ""))
identifier = agent.get("identifier", title.lower().replace(" ", "-"))
return self._agent_meta(agent, identifier, meta.get("description", "")[:200])
return _first_matching(query.lower(), agents, fields, to_meta, limit)
def fetch(self, identifier: str) -> Optional[SkillBundle]:
agent_id = self._agent_id(identifier)
agent_data = self._fetch_agent(agent_id)
if not agent_data:
return None
return SkillBundle(name=agent_id, files={"SKILL.md": self._convert_to_skill_md(agent_data)}, source="lobehub",
identifier=f"lobehub/{agent_id}", trust_level="community")
def inspect(self, identifier: str) -> Optional[SkillMeta]:
agent_id = self._agent_id(identifier)
agent = next((a for a in self._agents() or [] if a.get("identifier") == agent_id), None)
return self._agent_meta(agent, agent_id, agent.get("meta", agent).get("description", "")) if agent else None
def _fetch_index(self) -> Optional[Any]:
return _memo_json("lobehub_index", lambda: _get_json(self.INDEX_URL, timeout=30))
def _fetch_agent(self, agent_id: str) -> Optional[dict]:
return _get_json(f"https://chat-agents.lobehub.com/{agent_id}.json", timeout=15)
@staticmethod
def _convert_to_skill_md(agent_data: dict) -> str:
"""Convert a LobeHub agent JSON into SKILL.md format."""
meta = agent_data.get("meta", agent_data)
identifier = agent_data.get("identifier", "lobehub-agent")
title = meta.get("title", identifier)
description = meta.get("description", "")
tags = meta.get("tags", [])
tag_list = tags if isinstance(tags, list) else []
system_role = agent_data.get("config", {}).get("systemRole", "")
fm_lines = ["---", f"name: {identifier}", f"description: {description[:500]}", "metadata:", " hermes:",
f" tags: [{', '.join(str(t) for t in tag_list)}]", " lobehub:", " source: lobehub", "---"]
body_lines = [f"# {title}", "", description, "", "## Instructions", "",
system_role if system_role else "(No system role defined)"]
return "\n".join(fm_lines) + "\n\n" + "\n".join(body_lines) + "\n"
# --- browse.sh source adapter -----------------------------------------------
class BrowseShSource(SkillSource):
"""Browserbase's browse.sh catalog of site-specific browser-automation SKILL.md files.
The catalog is ``/api/skills``; content comes from ``/api/skills/{slug}``'s ``skillMdUrl`` (CDN blob).
The catalog's ``sourceUrl`` is a GitHub HTML URL whose repo is not always public, so it is not used for content.
"""
SOURCE_ID = "browse-sh"
CATALOG_URL = "https://browse.sh/api/skills"
SKILL_DETAIL_URL = "https://browse.sh/api/skills/{slug}"
_CACHE_KEY = "browse_sh_catalog"
def _fetch_catalog(self) -> List[Dict]:
def compute():
data = _get_json(self.CATALOG_URL)
skills = data.get("skills", []) if isinstance(data, dict) else []
return skills if isinstance(skills, list) else None
return _memo_json(self._CACHE_KEY, compute) or []
def _item_to_meta(self, item: Dict) -> Optional[SkillMeta]:
slug = item.get("slug", "")
name = item.get("name", "")
description = item.get("description", item.get("title", name))
if not slug or not name:
return None
if len(description) > 1024:
description = description[:1021] + "..."
return SkillMeta(
name=name, description=description, source="browse-sh", identifier=f"browse-sh/{slug}",
trust_level="community", tags=item.get("tags", []),
extra={"slug": slug, "hostname": item.get("hostname", ""), "category": item.get("category", ""),
"source_url": item.get("sourceUrl", ""), "recommended_method": item.get("recommendedMethod", ""),
"proxies": item.get("proxies", False), "install_count": item.get("installCount", 0)},
)
def search(self, query: str, limit: int = 10) -> List[SkillMeta]:
def fields(item):
return (item.get("name", ""), item.get("title", ""), item.get("description", ""),
item.get("hostname", ""), item.get("category", ""), item.get("tags", []))
return _first_matching(query.lower(), self._fetch_catalog(), fields, self._item_to_meta, limit)
def _catalog_item(self, identifier: str) -> Optional[Dict]:
slug = self._slug_from_identifier(identifier)
return next((i for i in self._fetch_catalog() if i.get("slug") == slug), None) if slug else None
def inspect(self, identifier: str) -> Optional[SkillMeta]:
item = self._catalog_item(identifier)
return self._item_to_meta(item) if item else None
def fetch(self, identifier: str) -> Optional[SkillBundle]:
item = self._catalog_item(identifier)
if not item:
return None
slug = item["slug"]
md_url = self._resolve_skill_md_url(slug, item)
content = _get_text(md_url, follow_redirects=True) if md_url else None
if content is None:
return None
meta = self._item_to_meta(item)
return SkillBundle(
name=meta.name if meta else slug.split("/")[-1], files={"SKILL.md": content}, source="browse-sh",
identifier=identifier, trust_level="community",
metadata={"slug": slug, "hostname": item.get("hostname", ""), "source_url": item.get("sourceUrl", ""),
"skill_md_url": md_url},
)
def _resolve_skill_md_url(self, slug: str, item: Dict) -> Optional[str]:
"""``skillMdUrl`` from ``/api/skills/{slug}``; fallback to a ``raw.githubusercontent.com`` ``sourceUrl``."""
data = _get_json(self.SKILL_DETAIL_URL.format(slug=slug), follow_redirects=True)
md_url = data.get("skillMdUrl") if isinstance(data, dict) else None
if isinstance(md_url, str) and md_url.startswith("http"):
return md_url
source_url = item.get("sourceUrl", "") if isinstance(item, dict) else ""
from utils import base_url_host_matches
return source_url if source_url and base_url_host_matches(source_url, "raw.githubusercontent.com") else None
def _slug_from_identifier(self, identifier: str) -> str:
"""'browse-sh/airbnb.com/search-listings-abc' -> 'airbnb.com/search-listings-abc'."""
return identifier[len("browse-sh/"):] if identifier.startswith("browse-sh/") else identifier