Files
mempalace--mempalace/tests/test_clean_lone_surrogates.py
wehub-resource-sync c3749daf48
Tests / test-linux (3.13) (push) Failing after 0s
Tests / test-linux (3.11) (push) Failing after 1s
Tests / lint (push) Failing after 0s
Tests / test-linux (3.9) (push) Failing after 1s
Docker / build (push) Failing after 1s
Docker / build-gpu (push) Failing after 2s
Tests / test-windows (push) Has been cancelled
Tests / test-macos (push) Has been cancelled
chore: import upstream snapshot with attribution
2026-07-13 12:03:03 +08:00

265 lines
11 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Tests for lone-surrogate sanitisation (issue #1235).
Covers:
- Unit: ``strip_lone_surrogates()`` edge cases (lone surrogates, real emoji,
empty input, hashing).
- Integration: every MCP tool that takes a user string and reaches ChromaDB
must survive lone surrogates in its payload.
MCP clients (Claude Desktop, WorkBuddy) occasionally relay lone UTF-16
surrogates (U+D800U+DFFF) when proxying binary-in-Unicode or corrupted
clipboard input. Python's ``str.encode('utf-8')`` raises on these, which
crashes ChromaDB add/upsert with -32000 Internal Error.
Note on surrogate pairs in Python source: writing ``"\\ud83d\\ude00"`` in
Python creates a string with *two* lone surrogates, not the astral emoji
U+1F600. Python 3 does not silently merge them. Use ``"\\U0001f600"`` to
embed a real emoji as a single code point.
"""
import hashlib
from mempalace.config import strip_lone_surrogates
# ── Unit tests ─────────────────────────────────────────────────────────────
class TestStripLoneSurrogates:
def test_passthrough_normal(self):
assert strip_lone_surrogates("hello world") == "hello world"
assert strip_lone_surrogates("你好世界") == "你好世界"
assert strip_lone_surrogates("mixed 中 English 文") == "mixed 中 English 文"
def test_empty_string(self):
assert strip_lone_surrogates("") == ""
def test_replaces_high_surrogate(self):
assert strip_lone_surrogates("hello\udc95world") == "helloworld"
assert strip_lone_surrogates("\udcff\udc00\udcaf") == "" * 3
def test_replaces_low_surrogate(self):
assert strip_lone_surrogates("test\ud800more") == "testmore"
assert strip_lone_surrogates("\ud800\udbff") == "" * 2
def test_replaces_multiple_at_different_positions(self):
assert strip_lone_surrogates("a\udca1b\udcffc") == "abc"
def test_preserves_real_emoji(self):
"""Astral code points written with \\U have no surrogates and pass through."""
assert strip_lone_surrogates("\U0001f600") == "\U0001f600"
assert strip_lone_surrogates("\U0001f680") == "\U0001f680"
assert strip_lone_surrogates("hello \U0001f600 world") == "hello \U0001f600 world"
def test_real_emoji_with_adjacent_lone_surrogate(self):
assert strip_lone_surrogates("\U0001f600\udc95") == "\U0001f600"
assert strip_lone_surrogates("\udc95\U0001f600") == "\U0001f600"
def test_only_surrogates(self):
assert strip_lone_surrogates("\udc95\udcff") == ""
assert strip_lone_surrogates("\ud800" + "\udbff" + "\udc00" + "\udfff") == "" * 4
def test_cleaned_string_is_utf8_encodable(self):
"""The actual crash path: encode/hash must not raise after cleaning."""
dirty = "content\udc95with\ud800surrogate"
clean = strip_lone_surrogates(dirty)
h = hashlib.sha256(clean.encode("utf-8")).hexdigest()
assert len(h) == 64
def test_workbuddy_injected_surrogate(self):
"""The specific surrogate observed in WorkBuddy production logs."""
result = strip_lone_surrogates("2026-04-27\udcadworkBuddy relay")
assert result == "2026-04-27workBuddy relay"
# Must not raise — was the original failure mode.
hashlib.sha256(result.encode("utf-8")).hexdigest()
# ── Sanitizer integration ──────────────────────────────────────────────────
class TestSanitizersStripSurrogates:
"""sanitize_content and sanitize_kg_value clean inline — verifies the
architectural fix so all MCP tools route through and inherit it for free."""
def test_sanitize_content_strips_surrogates(self):
from mempalace.config import sanitize_content
assert sanitize_content("hello\udc95world") == "helloworld"
def test_sanitize_kg_value_strips_surrogates(self):
from mempalace.config import sanitize_kg_value
assert sanitize_kg_value("Alice\udc95") == "Alice"
def test_sanitize_query_strips_surrogates(self):
from mempalace.query_sanitizer import sanitize_query
result = sanitize_query("search\udc95term")
assert "\udc95" not in result["clean_query"]
assert "" in result["clean_query"]
# ── End-to-end tool integration ────────────────────────────────────────────
def _patch_mcp_server(monkeypatch, config, kg):
from mempalace import mcp_server
monkeypatch.setattr(mcp_server, "_config", config)
monkeypatch.setattr(mcp_server, "_get_kg", lambda: kg)
class TestToolsAcceptSurrogates:
"""End-to-end: every MCP tool that takes user text must not crash on
lone surrogates. Failures here surface as ChromaDB UnicodeEncodeError."""
def test_add_drawer_content(self, monkeypatch, collection, config, kg):
_patch_mcp_server(monkeypatch, config, kg)
from mempalace.mcp_server import tool_add_drawer
result = tool_add_drawer(
wing="test",
room="surrogate",
content="drawer content with \udc95 surrogate",
)
assert result["success"] is True
assert result["drawer_id"].startswith("drawer_test_surrogate_")
def test_add_drawer_metadata(self, monkeypatch, collection, config, kg):
_patch_mcp_server(monkeypatch, config, kg)
from mempalace.mcp_server import tool_add_drawer
result = tool_add_drawer(
wing="test",
room="meta",
content="content here",
source_file="path/to/\udcadfile.txt",
added_by="user\udc95agent",
)
assert result["success"] is True
def test_check_duplicate(self, monkeypatch, collection, config, kg):
_patch_mcp_server(monkeypatch, config, kg)
from mempalace.mcp_server import tool_check_duplicate
result = tool_check_duplicate(content="exact\udc95match")
assert isinstance(result, dict)
assert "is_duplicate" in result
def test_search_query(self, monkeypatch, collection, config, kg):
_patch_mcp_server(monkeypatch, config, kg)
from mempalace.mcp_server import tool_search
result = tool_search(query="search\udc95term")
assert isinstance(result, dict)
assert "error" not in result or "results" in result
def test_update_drawer(self, monkeypatch, collection, config, kg):
_patch_mcp_server(monkeypatch, config, kg)
from mempalace.mcp_server import tool_add_drawer, tool_update_drawer
add_result = tool_add_drawer(wing="test", room="update", content="original content")
assert add_result["success"] is True
update_result = tool_update_drawer(
drawer_id=add_result["drawer_id"],
content="updated\udc95content",
)
assert update_result["success"] is True
def test_diary_write(self, monkeypatch, collection, config, kg):
_patch_mcp_server(monkeypatch, config, kg)
from mempalace.mcp_server import tool_diary_write
result = tool_diary_write(
agent_name="数数",
entry="今日工作\udc95完成了,修复了Chromadb crash",
topic="log",
)
assert result["success"] is True
# ── Backend chokepoint (bulk ingest paths) ──────────────────────────────────
class _CapturingCollection:
"""Minimal chromadb.Collection stand-in that records the kwargs it receives,
so we can assert what actually reaches the chromadb client."""
def __init__(self):
self.calls = []
def add(self, **kwargs):
self.calls.append(("add", kwargs))
def upsert(self, **kwargs):
self.calls.append(("upsert", kwargs))
def update(self, **kwargs):
self.calls.append(("update", kwargs))
class TestBackendChokepointStripsDocuments:
"""#1235 sanitised the MCP write tools, but the bulk ingest paths
(miner, convo_miner, sweeper, diary_ingest) build documents without routing
through ``sanitize_content`` and reach ``ChromaCollection`` directly. A lone
surrogate in *document* text crashes the whole add/upsert batch (-32000),
silently dropping the other rows. The backend chokepoint must strip
documents, mirroring ``_sanitize_metadatas_for_chromadb``."""
@staticmethod
def _collection():
from mempalace.backends.chroma import ChromaCollection
fake = _CapturingCollection()
return fake, ChromaCollection(fake)
def test_add_strips_lone_surrogate_in_document(self):
fake, col = self._collection()
col.add(documents=["clean\udc95doc"], ids=["1"])
_, kwargs = fake.calls[0]
assert kwargs["documents"] == ["cleandoc"]
kwargs["documents"][0].encode("utf-8") # the crash path must not raise
def test_upsert_strips_lone_surrogate_in_document(self):
fake, col = self._collection()
col.upsert(documents=["a\ud800b"], ids=["1"], metadatas=[{"wing": "w"}])
_, kwargs = fake.calls[0]
assert kwargs["documents"] == ["ab"]
def test_update_strips_lone_surrogate_in_document(self):
fake, col = self._collection()
col.update(ids=["1"], documents=["x\udcffy"])
_, kwargs = fake.calls[0]
assert kwargs["documents"] == ["xy"]
def test_one_poison_message_does_not_drop_the_batch(self):
"""A single bad row used to abort the entire batch, silently dropping
the others. After sanitising, every row survives."""
fake, col = self._collection()
col.upsert(
documents=["ok one", "poison\udc95row", "ok three"],
ids=["1", "2", "3"],
)
_, kwargs = fake.calls[0]
assert kwargs["documents"] == ["ok one", "poisonrow", "ok three"]
assert len(kwargs["ids"]) == 3
for d in kwargs["documents"]:
d.encode("utf-8") # must not raise
def test_real_emoji_and_none_metadata_preserved(self):
fake, col = self._collection()
col.add(documents=["ship \U0001f680"], ids=["1"])
_, kwargs = fake.calls[0]
assert kwargs["documents"] == ["ship \U0001f680"]
def test_single_string_document_is_not_split_into_chars(self):
"""chromadb accepts a bare str as one document (OneOrMany[Document]).
The sanitiser must keep it whole and clean, not split it into
per-character documents."""
fake, col = self._collection()
col.upsert(documents="one\udc95document", ids=["1"])
_, kwargs = fake.calls[0]
assert kwargs["documents"] == "onedocument"
assert isinstance(kwargs["documents"], str)