fix(langfuse): make payload capture depth configurable

This commit is contained in:
perqin
2026-09-20 12:02:51 +08:00
committed by Teknium
parent 7c4d2a812e
commit bc79fe7c75
3 changed files with 76 additions and 4 deletions

View File

@@ -47,10 +47,22 @@ HERMES_LANGFUSE_ENV=production # environment tag
HERMES_LANGFUSE_RELEASE=v1.0.0 # release tag
HERMES_LANGFUSE_SAMPLE_RATE=0.5 # sample 50% of traces
HERMES_LANGFUSE_MAX_CHARS=12000 # max chars per field (default: 12000)
HERMES_LANGFUSE_MAX_DEPTH=4 # max payload depth (default: 4)
HERMES_LANGFUSE_CAPTURE=sanitized # content capture mode (see below)
HERMES_LANGFUSE_DEBUG=true # verbose plugin logging
```
`HERMES_LANGFUSE_MAX_DEPTH` controls nested payload capture in both `sanitized`
and `full` modes, including tool arguments and JSON tool results. The root is
depth 0; each dictionary value or array element adds one level. Values beyond
the limit become `<max-depth>`, including scalars. For deeper MCP responses,
set it to a higher non-negative integer (for example, `10`) in the Hermes
process environment. Unset or blank values default to `4`; invalid or negative
values log a warning and fall back to `4`. `0` keeps only the root level.
Increasing the depth exports more content and may produce larger traces;
secret redaction, string-length limits, and the 50-item collection limit remain
unchanged. `metadata` mode still omits content.
## Capture modes
`HERMES_LANGFUSE_CAPTURE` controls how much *content* (prompts, responses,

View File

@@ -2,7 +2,7 @@
Activated via ``plugins.enabled``; hooks are inert without the ``langfuse`` SDK
and credentials. Env: HERMES_LANGFUSE_PUBLIC_KEY / SECRET_KEY (required),
BASE_URL, ENV, RELEASE, SAMPLE_RATE, MAX_CHARS (12000), DEBUG, and CAPTURE =
BASE_URL, ENV, RELEASE, SAMPLE_RATE, MAX_CHARS (12000), MAX_DEPTH (4), DEBUG, and CAPTURE =
metadata (sizes/ids/usage only) | sanitized (default: secret redaction +
truncation) | full (truncated raw content). See README.md.
"""
@@ -384,15 +384,24 @@ def _normalize_payload(value: Any, *, tool_name: str = "", args: Any = None) ->
def _safe_value(value: Any, *, max_chars: Optional[int] = None, depth: int = 0,
parse_json_strings: bool = False) -> Any:
parse_json_strings: bool = False, max_depth: Optional[int] = None) -> Any:
max_chars = max_chars if max_chars is not None else int(_env("HERMES_LANGFUSE_MAX_CHARS", "12000") or "12000")
if depth > 4:
if max_depth is None:
configured_depth = _env("HERMES_LANGFUSE_MAX_DEPTH", "4") or "4"
try:
max_depth = int(configured_depth)
if max_depth < 0:
raise ValueError
except ValueError:
logger.warning("Invalid HERMES_LANGFUSE_MAX_DEPTH=%r; use a non-negative integer. Falling back to 4.", configured_depth)
max_depth = 4
if depth > max_depth:
return "<max-depth>"
if value is None or isinstance(value, (int, float, bool)):
return value
if isinstance(value, bytes):
return {"type": "bytes", "len": len(value)}
recurse = lambda v, d: _safe_value(v, max_chars=max_chars, depth=d, parse_json_strings=parse_json_strings) # noqa: E731
recurse = lambda v, d: _safe_value(v, max_chars=max_chars, depth=d, parse_json_strings=parse_json_strings, max_depth=max_depth) # noqa: E731
if isinstance(value, str):
parsed = _maybe_parse_json_string(value) if parse_json_strings else value
return recurse(parsed, depth) if parsed is not value else _truncate_text(value, max_chars)

View File

@@ -0,0 +1,51 @@
"""Configurable Langfuse payload depth regression for #116609."""
import importlib
import json
import pytest
@pytest.mark.parametrize("mode", ["sanitized", "full"])
@pytest.mark.parametrize("configured_depth", [None, "", "0", "4", "5", " 10 "])
def test_capture_respects_configured_depth_for_tool_inputs_and_outputs(
monkeypatch, mode, configured_depth,
):
plugin = importlib.import_module("plugins.observability.langfuse")
monkeypatch.setenv("HERMES_LANGFUSE_CAPTURE", mode)
if configured_depth is None:
monkeypatch.delenv("HERMES_LANGFUSE_MAX_DEPTH", raising=False)
else:
monkeypatch.setenv("HERMES_LANGFUSE_MAX_DEPTH", configured_depth)
max_depth = int(configured_depth) if configured_depth else 4
# Each dict value/list element adds one level; scalars at the limit survive.
for leaf_depth in range(7):
payload = 7
expected = 7 if leaf_depth <= max_depth else "<max-depth>"
for level in reversed(range(leaf_depth)):
payload = {"data": payload} if level % 2 == 0 else [payload]
if level <= max_depth:
expected = {"data": expected} if level % 2 == 0 else [expected]
assert plugin._capture_content(payload) == expected
assert plugin._capture_content(
payload, tool_result_of=("example", {}),
) == expected
assert plugin._capture_content(
json.dumps(payload), tool_result_of=("example", {}),
) == (str(expected) if leaf_depth == 0 else expected)
@pytest.mark.parametrize("invalid_depth", ["nope", "1.5", "-1"])
def test_invalid_max_depth_warns_and_preserves_default_capture(monkeypatch, caplog, invalid_depth):
plugin = importlib.import_module("plugins.observability.langfuse")
monkeypatch.setenv("HERMES_LANGFUSE_CAPTURE", "full")
monkeypatch.delenv("HERMES_LANGFUSE_MAX_DEPTH", raising=False)
payload = {"result": {"data": {"results": [{"index": 0}]}}}
default_capture = plugin._capture_content(payload)
monkeypatch.setenv("HERMES_LANGFUSE_MAX_DEPTH", invalid_depth)
assert plugin._capture_content(payload) == default_capture
assert len(caplog.records) == 1
assert "HERMES_LANGFUSE_MAX_DEPTH" in caplog.text
assert "non-negative integer" in caplog.text