From 61202bf2fb2e64e9f97ae6a152161bad98f350d8 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Wed, 23 Sep 2026 19:13:37 +0530 Subject: [PATCH] perf(codex): categorise each distinct non-ASCII character once in the format-control scan No ASCII code point is a Unicode format control (Cf), so ASCII text returns on the O(1) isascii() flag and other text categorises set(text) minus ASCII instead of every character. Output is unchanged (same unicodedata predicate). Python 3.12, per call: CJK 9.2k chars 3.60 -> 1.57 ms, mixed 3.49 -> 0.30 ms, ASCII 3.16 -> ~0 ms (main -> this commit). A per-match finditer over non-ASCII characters was slower than main on CJK (11.0 ms) and is not used. --- agent/codex_responses_adapter.py | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/agent/codex_responses_adapter.py b/agent/codex_responses_adapter.py index 8a6831144f..60e9f49ef2 100644 --- a/agent/codex_responses_adapter.py +++ b/agent/codex_responses_adapter.py @@ -165,9 +165,11 @@ def _neutralize_harmony_tokens(text: str) -> str: """Keep Harmony source readable without emitting reserved wire tokens.""" if not text or "<" not in text or "|" not in text: return text - # No ASCII code point is a Unicode format control (Cf), and str.isascii() is an - # O(1) flag check, so ASCII text skips the per-character category scan entirely. - if text.isascii() or not any(unicodedata.category(char) == "Cf" for char in text): + # No ASCII code point is a Unicode format control (Cf): str.isascii() is an O(1) flag + # check, and other text only needs each distinct non-ASCII character categorised once. + if text.isascii() or not any( + unicodedata.category(char) == "Cf" for char in set(text) if char > "\x7f" + ): return _HARMONY_CONTROL_TOKEN_RE.sub(rf"<{_FULLWIDTH_PIPE}\1{_FULLWIDTH_PIPE}>", text) # The backend strips Unicode format controls (e.g. U+200B) before its reserved-token # check, so match on the visible text and rewrite the original spans.