diff --git a/tests/tui_gateway/test_image_history_display.py b/tests/tui_gateway/test_image_history_display.py new file mode 100644 index 0000000000..4218a87725 --- /dev/null +++ b/tests/tui_gateway/test_image_history_display.py @@ -0,0 +1,61 @@ +"""Native image history projection regression for #120083.""" + +import base64 + +from tui_gateway import server + + +def test_verified_native_image_history_reconciles_without_inline_payload(tmp_path): + image = tmp_path / "sample shot.png" + image.write_bytes(b"image bytes") + url = "data:image/png;base64," + base64.b64encode(image.read_bytes()).decode() + path = str(image) + parts = [{"type": "image_url", "image_url": {"url": url}}] + legacy = [{"type": "text", "text": f"Caption\n\n[Image attached at: {path}]"}, *parts, + {"type": "text", "text": "private recall"}] + current = [{"type": "text", "text": f"Caption\n@image:`{path}`"}, *parts] + history = [{"role": "user", "content": legacy}, {"role": "assistant", "content": "Done"}, + {"role": "user", "content": current}] + + displayed = server._history_to_messages(history) + assert displayed == [{"role": "user", "text": f"Caption\n\n@image:`{path}`"}, + {"role": "assistant", "text": "Done"}, + {"role": "user", "text": f"Caption\n@image:`{path}`"}] + assert history[0]["content"] is legacy + assert history[2]["content"] is current + + +def test_flattened_screenshot_placeholder_still_projects(tmp_path): + """The store's text-only projection appends one `[screenshot]` per image part; those rows must project too.""" + image = tmp_path / "sample shot.png" + image.write_bytes(b"image bytes") + url = "data:image/png;base64," + base64.b64encode(image.read_bytes()).decode() + path = str(image) + parts = [{"type": "image_url", "image_url": {"url": url}}] + legacy = [{"type": "text", "text": f"Caption\n\n[Image attached at: {path}]\n[screenshot]"}, *parts] + current = [{"type": "text", "text": f"Caption\n@image:`{path}`\n[screenshot]"}, *parts] + history = [{"role": "user", "content": legacy}, {"role": "user", "content": current}] + + displayed = server._history_to_messages(history) + assert displayed == [{"role": "user", "text": f"Caption\n\n@image:`{path}`"}, + {"role": "user", "text": f"Caption\n@image:`{path}`"}] + + +def test_screenshot_placeholder_without_image_part_stays_unprojected(tmp_path): + """Text-only rows carry no inline payload to verify against, so they keep whatever was stored.""" + image = tmp_path / "shot.png" + image.write_bytes(b"image bytes") + row = {"role": "user", "content": [{"type": "text", "text": f"Caption\n@image:{image}\n[screenshot]"}]} + assert server._history_to_messages([row])[0]["text"] == f"Caption\n@image:{image}\n[screenshot]" + + +def test_unavailable_or_mismatched_image_keeps_inline_payload(tmp_path): + image = tmp_path / "shot.png" + image.write_bytes(b"new bytes") + url = "data:image/png;base64," + base64.b64encode(b"old bytes").decode() + parts = [{"type": "text", "text": f"Caption\n@image:{image}"}, + {"type": "image_url", "image_url": {"url": url}}] + row = {"role": "user", "content": parts} + assert url in server._history_to_messages([row])[0]["text"] + image.unlink() + assert url in server._history_to_messages([row])[0]["text"] diff --git a/tui_gateway/session_history.py b/tui_gateway/session_history.py index cac6c28873..0c61dcd1dd 100644 --- a/tui_gateway/session_history.py +++ b/tui_gateway/session_history.py @@ -128,6 +128,89 @@ def _coerce_message_text(content: Any, *, image_urls: bool = True) -> str: return "" if content is None else str(content) +_IMAGE_HINT_RE = re.compile(r"\[Image attached at: ([^\n\]]+)\]") +_IMAGE_DATA_RE = re.compile(r"data:image/[^;,]+;base64,([A-Za-z0-9+/=]+)\Z") +_IMAGE_REF_RE = re.compile(r"@image:(`[^`\n]+`|\"[^\"\n]+\"|'[^'\n]+'|\S+)\Z") +# The session store's text-only projection replaces each image part with this +# stand-in, so a flattened row carries one placeholder per attached image. It +# describes the same attachment the refs above describe, so the projection +# drops it (mirroring the desktop's extractImageRefs) instead of refusing. +_SCREENSHOT_PLACEHOLDER = "[screenshot]" + + +def _strip_trailing_screenshot_placeholders(text: str) -> tuple[str, int]: + lines = text.split("\n") + count = 0 + while lines and lines[-1] == _SCREENSHOT_PLACEHOLDER: + lines.pop() + count += 1 + return "\n".join(lines), count + + +def _user_image_display_text(content: Any) -> str | None: + """Project verified native-vision images as refs without changing stored/model content.""" + if not isinstance(content, list) or len(content) < 2 or not isinstance(content[0], dict): + return None + text_part = content[0] + if text_part.get("type") != "text" or not isinstance(text_part.get("text"), str): + return None + text = text_part["text"] + from agent.context_references import format_reference_value + if "\n\n[Image attached at: " in text: + caption, hints = text.split("\n\n[Image attached at: ", 1) + if not caption or caption == "What do you see in this image?": + return None + hints, placeholders = _strip_trailing_screenshot_placeholders(hints) + matches = [_IMAGE_HINT_RE.fullmatch(line) for line in ("[Image attached at: " + hints).split("\n")] + if not all(matches): + return None + paths = [match[1] for match in matches] + if placeholders not in (0, len(paths)): + return None + projected = caption + "\n\n" + "\n".join(f"@image:{format_reference_value(p)}" for p in paths) + else: + body, placeholders = _strip_trailing_screenshot_placeholders(text) + lines = body.splitlines() + ref_lines = [line for line in lines if line.startswith("@image:")] + matches = [_IMAGE_REF_RE.fullmatch(line) for line in ref_lines] + if not matches or not all(matches) or lines[-len(matches):] != ref_lines: + return None + paths = [match[1][1:-1] if match[1][0] in ('`', '"', "'") else match[1] for match in matches] + if any(format_reference_value(p) != match[1] for p, match in zip(paths, matches)): + return None + if placeholders not in (0, len(paths)): + return None + projected = body + + import base64 + import binascii + from pathlib import Path + from fastapi import HTTPException + from hermes_cli.web_routers.files import _fs_regular_file + from hermes_cli.web_server import _FS_DATA_URL_MAX_BYTES + image_parts = content[1:1 + len(paths)] + if len(image_parts) != len(paths) or len(content) not in (1 + len(paths), 2 + len(paths)): + return None + if len(content) > 1 + len(paths): + from agent.memory_manager import sanitize_context + memory = content[-1] + if not isinstance(memory, dict) or memory.get("type") != "text" or not isinstance(memory.get("text"), str) or sanitize_context(memory["text"]).strip(): + return None + for path, part in zip(paths, image_parts): + if not Path(path).is_absolute() or not isinstance(part, dict) or part.get("type") != "image_url": + return None + match = _IMAGE_DATA_RE.fullmatch(_history_part_image_url(part)) + if not match or len(match[1]) > ((_FS_DATA_URL_MAX_BYTES + 2) // 3) * 4: + return None + try: + target, st = _fs_regular_file(Path(path)) + if st.st_size > _FS_DATA_URL_MAX_BYTES or target.read_bytes() != base64.b64decode(match[1], validate=True): + return None + except (OSError, ValueError, binascii.Error, HTTPException): + return None + return projected + + def _history_text_only_part(part: dict) -> bool: kind = part.get("type") return kind in _HISTORY_TEXT_KINDS or (kind is None and isinstance(part.get("text"), str)) @@ -216,7 +299,9 @@ def _history_to_messages(history: list[dict], *, profile_home=None, image_urls: # display_kind="hidden": model-facing scaffolding the "[System:" sniff does not catch. if role not in _HISTORY_ROLES or m.get("display_kind") == "hidden": continue - content_text = _coerce_message_text(m.get("content"), image_urls=image_urls) + content_text = _user_image_display_text(m.get("content")) if role == "user" else None + if content_text is None: + content_text = _coerce_message_text(m.get("content"), image_urls=image_urls) if _is_display_hidden_marker(role, content_text): continue if role == "user":