fix(memory): read/write .env as UTF-8 in mem0 and hindsight setup
The mem0 and hindsight memory-provider setup routines round-trip the user's ~/.hermes/.env: they read existing lines, update the keys they manage, and rewrite the whole file preserving every other line verbatim. Both used env_path.read_text() / write_text() with no encoding. read_text()/write_text() with no encoding fall back to the system locale (cp1252/GBK on Windows), so on a non-UTF-8 host the preserved lines get mangled or the call crashes on any non-ASCII value, and — because the reader never strips a BOM — a Notepad-edited .env makes the first key fail the in-place match and get duplicated instead of updated. Match the canonical .env readers in hermes_cli/config.py: read with encoding='utf-8-sig' (BOM-tolerant) and write with encoding='utf-8'. mem0/_setup.py already pins utf-8 for mem0.json, so this just aligns the .env path in the same file. Fixes both memory plugins in one class fix. Adds regression tests: a BOM'd .env updates the first key in place (locale-independent, fails without the fix) and non-ASCII existing lines survive the round-trip. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
committed by
Teknium
parent
e02170f54a
commit
75afc47baa
@@ -193,7 +193,14 @@ def _write_env(env_path: Path, env_writes: dict[str, str]) -> None:
|
||||
env_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
existing_lines: list[str] = []
|
||||
if env_path.exists():
|
||||
existing_lines = env_path.read_text().splitlines()
|
||||
# Read as UTF-8 (BOM-tolerant), matching the canonical .env readers in
|
||||
# hermes_cli/config.py. read_text() with no encoding falls back to the
|
||||
# system locale (cp1252/GBK on Windows): it mangles or crashes on
|
||||
# non-ASCII values while copying existing lines through, and a BOM'd
|
||||
# first line would fail the key match and get duplicated.
|
||||
existing_lines = env_path.read_text(
|
||||
encoding="utf-8-sig"
|
||||
).splitlines()
|
||||
|
||||
updated_keys: set[str] = set()
|
||||
new_lines: list[str] = []
|
||||
@@ -208,7 +215,7 @@ def _write_env(env_path: Path, env_writes: dict[str, str]) -> None:
|
||||
if k not in updated_keys:
|
||||
new_lines.append(f"{k}={v}")
|
||||
|
||||
env_path.write_text("\n".join(new_lines) + "\n")
|
||||
env_path.write_text("\n".join(new_lines) + "\n", encoding="utf-8")
|
||||
|
||||
|
||||
def _save_mem0_json(hermes_home: str, data: dict) -> None:
|
||||
|
||||
@@ -181,6 +181,26 @@ class TestWriteEnv:
|
||||
assert "OTHER=keep" in content
|
||||
assert "old" not in content
|
||||
|
||||
def test_preserves_non_ascii_existing_lines(self, tmp_path):
|
||||
"""Existing non-ASCII .env content must survive the read-modify-write
|
||||
as UTF-8 (the locale codec would crash/mangle it on Windows)."""
|
||||
env_path = tmp_path / ".env"
|
||||
env_path.write_bytes("PROXY_NOTE=café-zürich-完了\n".encode("utf-8"))
|
||||
_write_env(env_path, {"OPENAI_API_KEY": "sk-test"})
|
||||
content = env_path.read_text(encoding="utf-8")
|
||||
assert "PROXY_NOTE=café-zürich-完了" in content
|
||||
assert "OPENAI_API_KEY=sk-test" in content
|
||||
|
||||
def test_updates_first_key_with_bom(self, tmp_path):
|
||||
"""A Notepad-edited .env carries a BOM; the first key must still be
|
||||
matched/updated in place, not duplicated."""
|
||||
env_path = tmp_path / ".env"
|
||||
env_path.write_bytes("OPENAI_API_KEY=old\n".encode("utf-8"))
|
||||
_write_env(env_path, {"OPENAI_API_KEY": "new"})
|
||||
content = env_path.read_text(encoding="utf-8")
|
||||
assert content.count("OPENAI_API_KEY=") == 1
|
||||
assert "OPENAI_API_KEY=new" in content
|
||||
|
||||
|
||||
class TestPostSetup:
|
||||
|
||||
|
||||
Reference in New Issue
Block a user