fix(tui_gateway): project verified native image turns as display refs

Native-vision user turns persist inline base64 in history text, so the
served user text never equals the Desktop optimistic @image ref and the
optimistic bubble is re-appended below every finished reply.

Display-only projection in _history_to_messages for role=user rows:
verified local image parts become @image: refs, stored/model content
untouched. Ambiguous or missing images keep inline form.

The session store's text-only projection appends one `[screenshot]`
line per image part, so flattened rows are projected too: trailing
placeholders are peeled (one per projected ref, else unprojected) and
dropped, mirroring the Desktop's extractImageRefs.

Closes #120083
This commit is contained in:
finn763
2026-09-23 20:03:55 +08:00
committed by brooklyn!
parent 47b62eab06
commit 23de088b6a
2 changed files with 147 additions and 1 deletions

View File

@@ -0,0 +1,61 @@
"""Native image history projection regression for #120083."""
import base64
from tui_gateway import server
def test_verified_native_image_history_reconciles_without_inline_payload(tmp_path):
image = tmp_path / "sample shot.png"
image.write_bytes(b"image bytes")
url = "data:image/png;base64," + base64.b64encode(image.read_bytes()).decode()
path = str(image)
parts = [{"type": "image_url", "image_url": {"url": url}}]
legacy = [{"type": "text", "text": f"Caption\n\n[Image attached at: {path}]"}, *parts,
{"type": "text", "text": "<memory-context>private recall</memory-context>"}]
current = [{"type": "text", "text": f"Caption\n@image:`{path}`"}, *parts]
history = [{"role": "user", "content": legacy}, {"role": "assistant", "content": "Done"},
{"role": "user", "content": current}]
displayed = server._history_to_messages(history)
assert displayed == [{"role": "user", "text": f"Caption\n\n@image:`{path}`"},
{"role": "assistant", "text": "Done"},
{"role": "user", "text": f"Caption\n@image:`{path}`"}]
assert history[0]["content"] is legacy
assert history[2]["content"] is current
def test_flattened_screenshot_placeholder_still_projects(tmp_path):
"""The store's text-only projection appends one `[screenshot]` per image part; those rows must project too."""
image = tmp_path / "sample shot.png"
image.write_bytes(b"image bytes")
url = "data:image/png;base64," + base64.b64encode(image.read_bytes()).decode()
path = str(image)
parts = [{"type": "image_url", "image_url": {"url": url}}]
legacy = [{"type": "text", "text": f"Caption\n\n[Image attached at: {path}]\n[screenshot]"}, *parts]
current = [{"type": "text", "text": f"Caption\n@image:`{path}`\n[screenshot]"}, *parts]
history = [{"role": "user", "content": legacy}, {"role": "user", "content": current}]
displayed = server._history_to_messages(history)
assert displayed == [{"role": "user", "text": f"Caption\n\n@image:`{path}`"},
{"role": "user", "text": f"Caption\n@image:`{path}`"}]
def test_screenshot_placeholder_without_image_part_stays_unprojected(tmp_path):
"""Text-only rows carry no inline payload to verify against, so they keep whatever was stored."""
image = tmp_path / "shot.png"
image.write_bytes(b"image bytes")
row = {"role": "user", "content": [{"type": "text", "text": f"Caption\n@image:{image}\n[screenshot]"}]}
assert server._history_to_messages([row])[0]["text"] == f"Caption\n@image:{image}\n[screenshot]"
def test_unavailable_or_mismatched_image_keeps_inline_payload(tmp_path):
image = tmp_path / "shot.png"
image.write_bytes(b"new bytes")
url = "data:image/png;base64," + base64.b64encode(b"old bytes").decode()
parts = [{"type": "text", "text": f"Caption\n@image:{image}"},
{"type": "image_url", "image_url": {"url": url}}]
row = {"role": "user", "content": parts}
assert url in server._history_to_messages([row])[0]["text"]
image.unlink()
assert url in server._history_to_messages([row])[0]["text"]

View File

@@ -128,6 +128,89 @@ def _coerce_message_text(content: Any, *, image_urls: bool = True) -> str:
return "" if content is None else str(content)
_IMAGE_HINT_RE = re.compile(r"\[Image attached at: ([^\n\]]+)\]")
_IMAGE_DATA_RE = re.compile(r"data:image/[^;,]+;base64,([A-Za-z0-9+/=]+)\Z")
_IMAGE_REF_RE = re.compile(r"@image:(`[^`\n]+`|\"[^\"\n]+\"|'[^'\n]+'|\S+)\Z")
# The session store's text-only projection replaces each image part with this
# stand-in, so a flattened row carries one placeholder per attached image. It
# describes the same attachment the refs above describe, so the projection
# drops it (mirroring the desktop's extractImageRefs) instead of refusing.
_SCREENSHOT_PLACEHOLDER = "[screenshot]"
def _strip_trailing_screenshot_placeholders(text: str) -> tuple[str, int]:
lines = text.split("\n")
count = 0
while lines and lines[-1] == _SCREENSHOT_PLACEHOLDER:
lines.pop()
count += 1
return "\n".join(lines), count
def _user_image_display_text(content: Any) -> str | None:
"""Project verified native-vision images as refs without changing stored/model content."""
if not isinstance(content, list) or len(content) < 2 or not isinstance(content[0], dict):
return None
text_part = content[0]
if text_part.get("type") != "text" or not isinstance(text_part.get("text"), str):
return None
text = text_part["text"]
from agent.context_references import format_reference_value
if "\n\n[Image attached at: " in text:
caption, hints = text.split("\n\n[Image attached at: ", 1)
if not caption or caption == "What do you see in this image?":
return None
hints, placeholders = _strip_trailing_screenshot_placeholders(hints)
matches = [_IMAGE_HINT_RE.fullmatch(line) for line in ("[Image attached at: " + hints).split("\n")]
if not all(matches):
return None
paths = [match[1] for match in matches]
if placeholders not in (0, len(paths)):
return None
projected = caption + "\n\n" + "\n".join(f"@image:{format_reference_value(p)}" for p in paths)
else:
body, placeholders = _strip_trailing_screenshot_placeholders(text)
lines = body.splitlines()
ref_lines = [line for line in lines if line.startswith("@image:")]
matches = [_IMAGE_REF_RE.fullmatch(line) for line in ref_lines]
if not matches or not all(matches) or lines[-len(matches):] != ref_lines:
return None
paths = [match[1][1:-1] if match[1][0] in ('`', '"', "'") else match[1] for match in matches]
if any(format_reference_value(p) != match[1] for p, match in zip(paths, matches)):
return None
if placeholders not in (0, len(paths)):
return None
projected = body
import base64
import binascii
from pathlib import Path
from fastapi import HTTPException
from hermes_cli.web_routers.files import _fs_regular_file
from hermes_cli.web_server import _FS_DATA_URL_MAX_BYTES
image_parts = content[1:1 + len(paths)]
if len(image_parts) != len(paths) or len(content) not in (1 + len(paths), 2 + len(paths)):
return None
if len(content) > 1 + len(paths):
from agent.memory_manager import sanitize_context
memory = content[-1]
if not isinstance(memory, dict) or memory.get("type") != "text" or not isinstance(memory.get("text"), str) or sanitize_context(memory["text"]).strip():
return None
for path, part in zip(paths, image_parts):
if not Path(path).is_absolute() or not isinstance(part, dict) or part.get("type") != "image_url":
return None
match = _IMAGE_DATA_RE.fullmatch(_history_part_image_url(part))
if not match or len(match[1]) > ((_FS_DATA_URL_MAX_BYTES + 2) // 3) * 4:
return None
try:
target, st = _fs_regular_file(Path(path))
if st.st_size > _FS_DATA_URL_MAX_BYTES or target.read_bytes() != base64.b64decode(match[1], validate=True):
return None
except (OSError, ValueError, binascii.Error, HTTPException):
return None
return projected
def _history_text_only_part(part: dict) -> bool:
kind = part.get("type")
return kind in _HISTORY_TEXT_KINDS or (kind is None and isinstance(part.get("text"), str))
@@ -216,7 +299,9 @@ def _history_to_messages(history: list[dict], *, profile_home=None, image_urls:
# display_kind="hidden": model-facing scaffolding the "[System:" sniff does not catch.
if role not in _HISTORY_ROLES or m.get("display_kind") == "hidden":
continue
content_text = _coerce_message_text(m.get("content"), image_urls=image_urls)
content_text = _user_image_display_text(m.get("content")) if role == "user" else None
if content_text is None:
content_text = _coerce_message_text(m.get("content"), image_urls=image_urls)
if _is_display_hidden_marker(role, content_text):
continue
if role == "user":