diff --git a/agent/codex_responses_adapter.py b/agent/codex_responses_adapter.py index 8a6831144f..60e9f49ef2 100644 --- a/agent/codex_responses_adapter.py +++ b/agent/codex_responses_adapter.py @@ -165,9 +165,11 @@ def _neutralize_harmony_tokens(text: str) -> str: """Keep Harmony source readable without emitting reserved wire tokens.""" if not text or "<" not in text or "|" not in text: return text - # No ASCII code point is a Unicode format control (Cf), and str.isascii() is an - # O(1) flag check, so ASCII text skips the per-character category scan entirely. - if text.isascii() or not any(unicodedata.category(char) == "Cf" for char in text): + # No ASCII code point is a Unicode format control (Cf): str.isascii() is an O(1) flag + # check, and other text only needs each distinct non-ASCII character categorised once. + if text.isascii() or not any( + unicodedata.category(char) == "Cf" for char in set(text) if char > "\x7f" + ): return _HARMONY_CONTROL_TOKEN_RE.sub(rf"<{_FULLWIDTH_PIPE}\1{_FULLWIDTH_PIPE}>", text) # The backend strips Unicode format controls (e.g. U+200B) before its reserved-token # check, so match on the visible text and rewrite the original spans.