fix(tui_gateway): project verified native image turns as display refs
Native-vision user turns persist inline base64 in history text, so the served user text never equals the Desktop optimistic @image ref and the optimistic bubble is re-appended below every finished reply. Display-only projection in _history_to_messages for role=user rows: verified local image parts become @image: refs, stored/model content untouched. Ambiguous or missing images keep inline form. The session store's text-only projection appends one `[screenshot]` line per image part, so flattened rows are projected too: trailing placeholders are peeled (one per projected ref, else unprojected) and dropped, mirroring the Desktop's extractImageRefs. Closes #120083
This commit is contained in:
61
tests/tui_gateway/test_image_history_display.py
Normal file
61
tests/tui_gateway/test_image_history_display.py
Normal file
@@ -0,0 +1,61 @@
|
||||
"""Native image history projection regression for #120083."""
|
||||
|
||||
import base64
|
||||
|
||||
from tui_gateway import server
|
||||
|
||||
|
||||
def test_verified_native_image_history_reconciles_without_inline_payload(tmp_path):
|
||||
image = tmp_path / "sample shot.png"
|
||||
image.write_bytes(b"image bytes")
|
||||
url = "data:image/png;base64," + base64.b64encode(image.read_bytes()).decode()
|
||||
path = str(image)
|
||||
parts = [{"type": "image_url", "image_url": {"url": url}}]
|
||||
legacy = [{"type": "text", "text": f"Caption\n\n[Image attached at: {path}]"}, *parts,
|
||||
{"type": "text", "text": "<memory-context>private recall</memory-context>"}]
|
||||
current = [{"type": "text", "text": f"Caption\n@image:`{path}`"}, *parts]
|
||||
history = [{"role": "user", "content": legacy}, {"role": "assistant", "content": "Done"},
|
||||
{"role": "user", "content": current}]
|
||||
|
||||
displayed = server._history_to_messages(history)
|
||||
assert displayed == [{"role": "user", "text": f"Caption\n\n@image:`{path}`"},
|
||||
{"role": "assistant", "text": "Done"},
|
||||
{"role": "user", "text": f"Caption\n@image:`{path}`"}]
|
||||
assert history[0]["content"] is legacy
|
||||
assert history[2]["content"] is current
|
||||
|
||||
|
||||
def test_flattened_screenshot_placeholder_still_projects(tmp_path):
|
||||
"""The store's text-only projection appends one `[screenshot]` per image part; those rows must project too."""
|
||||
image = tmp_path / "sample shot.png"
|
||||
image.write_bytes(b"image bytes")
|
||||
url = "data:image/png;base64," + base64.b64encode(image.read_bytes()).decode()
|
||||
path = str(image)
|
||||
parts = [{"type": "image_url", "image_url": {"url": url}}]
|
||||
legacy = [{"type": "text", "text": f"Caption\n\n[Image attached at: {path}]\n[screenshot]"}, *parts]
|
||||
current = [{"type": "text", "text": f"Caption\n@image:`{path}`\n[screenshot]"}, *parts]
|
||||
history = [{"role": "user", "content": legacy}, {"role": "user", "content": current}]
|
||||
|
||||
displayed = server._history_to_messages(history)
|
||||
assert displayed == [{"role": "user", "text": f"Caption\n\n@image:`{path}`"},
|
||||
{"role": "user", "text": f"Caption\n@image:`{path}`"}]
|
||||
|
||||
|
||||
def test_screenshot_placeholder_without_image_part_stays_unprojected(tmp_path):
|
||||
"""Text-only rows carry no inline payload to verify against, so they keep whatever was stored."""
|
||||
image = tmp_path / "shot.png"
|
||||
image.write_bytes(b"image bytes")
|
||||
row = {"role": "user", "content": [{"type": "text", "text": f"Caption\n@image:{image}\n[screenshot]"}]}
|
||||
assert server._history_to_messages([row])[0]["text"] == f"Caption\n@image:{image}\n[screenshot]"
|
||||
|
||||
|
||||
def test_unavailable_or_mismatched_image_keeps_inline_payload(tmp_path):
|
||||
image = tmp_path / "shot.png"
|
||||
image.write_bytes(b"new bytes")
|
||||
url = "data:image/png;base64," + base64.b64encode(b"old bytes").decode()
|
||||
parts = [{"type": "text", "text": f"Caption\n@image:{image}"},
|
||||
{"type": "image_url", "image_url": {"url": url}}]
|
||||
row = {"role": "user", "content": parts}
|
||||
assert url in server._history_to_messages([row])[0]["text"]
|
||||
image.unlink()
|
||||
assert url in server._history_to_messages([row])[0]["text"]
|
||||
@@ -128,6 +128,89 @@ def _coerce_message_text(content: Any, *, image_urls: bool = True) -> str:
|
||||
return "" if content is None else str(content)
|
||||
|
||||
|
||||
_IMAGE_HINT_RE = re.compile(r"\[Image attached at: ([^\n\]]+)\]")
|
||||
_IMAGE_DATA_RE = re.compile(r"data:image/[^;,]+;base64,([A-Za-z0-9+/=]+)\Z")
|
||||
_IMAGE_REF_RE = re.compile(r"@image:(`[^`\n]+`|\"[^\"\n]+\"|'[^'\n]+'|\S+)\Z")
|
||||
# The session store's text-only projection replaces each image part with this
|
||||
# stand-in, so a flattened row carries one placeholder per attached image. It
|
||||
# describes the same attachment the refs above describe, so the projection
|
||||
# drops it (mirroring the desktop's extractImageRefs) instead of refusing.
|
||||
_SCREENSHOT_PLACEHOLDER = "[screenshot]"
|
||||
|
||||
|
||||
def _strip_trailing_screenshot_placeholders(text: str) -> tuple[str, int]:
|
||||
lines = text.split("\n")
|
||||
count = 0
|
||||
while lines and lines[-1] == _SCREENSHOT_PLACEHOLDER:
|
||||
lines.pop()
|
||||
count += 1
|
||||
return "\n".join(lines), count
|
||||
|
||||
|
||||
def _user_image_display_text(content: Any) -> str | None:
|
||||
"""Project verified native-vision images as refs without changing stored/model content."""
|
||||
if not isinstance(content, list) or len(content) < 2 or not isinstance(content[0], dict):
|
||||
return None
|
||||
text_part = content[0]
|
||||
if text_part.get("type") != "text" or not isinstance(text_part.get("text"), str):
|
||||
return None
|
||||
text = text_part["text"]
|
||||
from agent.context_references import format_reference_value
|
||||
if "\n\n[Image attached at: " in text:
|
||||
caption, hints = text.split("\n\n[Image attached at: ", 1)
|
||||
if not caption or caption == "What do you see in this image?":
|
||||
return None
|
||||
hints, placeholders = _strip_trailing_screenshot_placeholders(hints)
|
||||
matches = [_IMAGE_HINT_RE.fullmatch(line) for line in ("[Image attached at: " + hints).split("\n")]
|
||||
if not all(matches):
|
||||
return None
|
||||
paths = [match[1] for match in matches]
|
||||
if placeholders not in (0, len(paths)):
|
||||
return None
|
||||
projected = caption + "\n\n" + "\n".join(f"@image:{format_reference_value(p)}" for p in paths)
|
||||
else:
|
||||
body, placeholders = _strip_trailing_screenshot_placeholders(text)
|
||||
lines = body.splitlines()
|
||||
ref_lines = [line for line in lines if line.startswith("@image:")]
|
||||
matches = [_IMAGE_REF_RE.fullmatch(line) for line in ref_lines]
|
||||
if not matches or not all(matches) or lines[-len(matches):] != ref_lines:
|
||||
return None
|
||||
paths = [match[1][1:-1] if match[1][0] in ('`', '"', "'") else match[1] for match in matches]
|
||||
if any(format_reference_value(p) != match[1] for p, match in zip(paths, matches)):
|
||||
return None
|
||||
if placeholders not in (0, len(paths)):
|
||||
return None
|
||||
projected = body
|
||||
|
||||
import base64
|
||||
import binascii
|
||||
from pathlib import Path
|
||||
from fastapi import HTTPException
|
||||
from hermes_cli.web_routers.files import _fs_regular_file
|
||||
from hermes_cli.web_server import _FS_DATA_URL_MAX_BYTES
|
||||
image_parts = content[1:1 + len(paths)]
|
||||
if len(image_parts) != len(paths) or len(content) not in (1 + len(paths), 2 + len(paths)):
|
||||
return None
|
||||
if len(content) > 1 + len(paths):
|
||||
from agent.memory_manager import sanitize_context
|
||||
memory = content[-1]
|
||||
if not isinstance(memory, dict) or memory.get("type") != "text" or not isinstance(memory.get("text"), str) or sanitize_context(memory["text"]).strip():
|
||||
return None
|
||||
for path, part in zip(paths, image_parts):
|
||||
if not Path(path).is_absolute() or not isinstance(part, dict) or part.get("type") != "image_url":
|
||||
return None
|
||||
match = _IMAGE_DATA_RE.fullmatch(_history_part_image_url(part))
|
||||
if not match or len(match[1]) > ((_FS_DATA_URL_MAX_BYTES + 2) // 3) * 4:
|
||||
return None
|
||||
try:
|
||||
target, st = _fs_regular_file(Path(path))
|
||||
if st.st_size > _FS_DATA_URL_MAX_BYTES or target.read_bytes() != base64.b64decode(match[1], validate=True):
|
||||
return None
|
||||
except (OSError, ValueError, binascii.Error, HTTPException):
|
||||
return None
|
||||
return projected
|
||||
|
||||
|
||||
def _history_text_only_part(part: dict) -> bool:
|
||||
kind = part.get("type")
|
||||
return kind in _HISTORY_TEXT_KINDS or (kind is None and isinstance(part.get("text"), str))
|
||||
@@ -216,7 +299,9 @@ def _history_to_messages(history: list[dict], *, profile_home=None, image_urls:
|
||||
# display_kind="hidden": model-facing scaffolding the "[System:" sniff does not catch.
|
||||
if role not in _HISTORY_ROLES or m.get("display_kind") == "hidden":
|
||||
continue
|
||||
content_text = _coerce_message_text(m.get("content"), image_urls=image_urls)
|
||||
content_text = _user_image_display_text(m.get("content")) if role == "user" else None
|
||||
if content_text is None:
|
||||
content_text = _coerce_message_text(m.get("content"), image_urls=image_urls)
|
||||
if _is_display_hidden_marker(role, content_text):
|
||||
continue
|
||||
if role == "user":
|
||||
|
||||
Reference in New Issue
Block a user