fix(cartesia,pipecat): fall through to chunk when memory is empty

`_field` returns the first value that is not None. A v4 hybrid-search hit
shaped `{"memory": "", "chunk": "..."}` therefore resolves to the empty
string and never falls through to `chunk` — the exact fallback the call
site's comment says it is there for.

The consequences are in both directions. In `deduplicate_memories` the hit
produces an empty comparison key and is dropped, so the memory never reaches
the prompt. When such an item does survive, `format_memories_to_text` renders
it through the same `_field` call and emits a bare `- [16 hrs ago] `.

The TypeScript `getMemoryText` takes the first non-empty field instead, and
`agent-framework-python` and `openai-sdk-python` both already match it; the
two voice SDKs were the only implementations that disagreed. Add a
`_memory_text` helper mirroring the TypeScript semantics and use it on both
the dedup and the render path.

Also add tests/conftest.py to pipecat so the import stubs are installed
regardless of collection order — test_utils.py cannot import the package
on its own otherwise.
This commit is contained in:
Agnik47 2026-09-18 22:28:33 +05:30
parent b65c10bb47
commit 3d1980560e
5 changed files with 310 additions and 14 deletions

View file

@ -71,6 +71,22 @@ def _field(item: Any, *names: str, default: Any = None) -> Any:
return default
def _memory_text(item: Any) -> str:
"""First non-empty memory field, mirroring the TypeScript `getMemoryText`.
`_field` stops at the first value that is not None, so a hybrid-search hit
shaped `{"memory": "", "chunk": "..."}` resolves to the empty string and
never falls through to `chunk`.
"""
if isinstance(item, str):
return item.strip()
for name in ("memory", "chunk", "content"):
value = _field(item, name)
if isinstance(value, str) and value.strip():
return value.strip()
return ""
_MEMORY_DATE_PREFIX = re.compile(
r"^\s*(?:\[recent\]\s*)?(?:\[\d{4}-\d{2}-\d{2}\]\s*)?",
re.IGNORECASE,
@ -125,12 +141,7 @@ def deduplicate_memories(
out = []
for r in results:
# v4 search.memories/hybrid uses `memory` or `chunk`.
memory = (
r if isinstance(r, str) else _field(r, "memory", "chunk", "content", default="")
)
if not isinstance(memory, str):
memory = ""
memory = memory.strip()
memory = _memory_text(r)
key = _memory_key(memory)
if key and key not in seen:
seen.add(key)
@ -177,7 +188,7 @@ def format_memories_to_text(
lines.append(f"- {item}")
continue
memory = _field(item, "memory", "chunk", "content", default="")
memory = _memory_text(item)
updated_at = _field(item, "updatedAt", "updated_at", default="")
time_str = format_relative_time(updated_at) if updated_at else ""
if time_str:

View file

@ -114,5 +114,84 @@ class TestDeduplicateMemories(unittest.TestCase):
self.assertEqual(result["search_results"], [])
class TestSearchResultExtraction(unittest.TestCase):
"""v4 hybrid search returns chunk hits whose `memory` field is empty.
`_field` stops at the first value that is not None, so those hits used to
resolve to "" and were dropped by dedup -- or, when they survived, rendered
as an empty bullet. The TypeScript `getMemoryText` takes the first
non-empty field instead, and the other Supermemory Python SDKs match it.
"""
def test_falls_through_to_chunk_when_memory_is_empty(self) -> None:
result = deduplicate_memories(
static=[],
dynamic=[],
search_results=[{"memory": "", "chunk": "The user's dog is called Rex"}],
)
self.assertEqual(len(result["search_results"]), 1)
self.assertIn("The user's dog is called Rex", format_memories_to_text(result))
def test_falls_through_to_chunk_when_memory_is_whitespace(self) -> None:
result = deduplicate_memories(
static=[],
dynamic=[],
search_results=[{"memory": " ", "chunk": "Chunk body"}],
)
self.assertEqual(len(result["search_results"]), 1)
self.assertIn("Chunk body", format_memories_to_text(result))
def test_falls_through_on_models_too(self) -> None:
result = deduplicate_memories(
static=[],
dynamic=[],
search_results=[SimpleNamespace(memory="", chunk="Chunk body")],
)
self.assertEqual(len(result["search_results"]), 1)
self.assertIn("Chunk body", format_memories_to_text(result))
def test_never_renders_an_empty_bullet(self) -> None:
result = deduplicate_memories(
static=[],
dynamic=[],
search_results=[
{
"memory": "",
"chunk": "Chunk body",
"updatedAt": "2020-01-01T00:00:00Z",
}
],
)
rendered = format_memories_to_text(result)
self.assertIn("Chunk body", rendered)
for line in rendered.splitlines():
if line.startswith("- "):
self.assertNotRegex(line, r"^- (\[[^\]]*\] )?$")
def test_memory_still_wins_over_chunk(self) -> None:
result = deduplicate_memories(
static=[],
dynamic=[],
search_results=[{"memory": "Memory body", "chunk": "Chunk body"}],
)
rendered = format_memories_to_text(result)
self.assertIn("Memory body", rendered)
self.assertNotIn("Chunk body", rendered)
def test_entry_with_no_usable_text_is_still_dropped(self) -> None:
result = deduplicate_memories(
static=[],
dynamic=[],
search_results=[{"memory": "", "chunk": ""}, {}, None],
)
self.assertEqual(result["search_results"], [])
if __name__ == "__main__":
unittest.main()

View file

@ -90,6 +90,22 @@ def _field(item: Any, *names: str, default: Any = None) -> Any:
return default
def _memory_text(item: Any) -> str:
"""First non-empty memory field, mirroring the TypeScript `getMemoryText`.
`_field` stops at the first value that is not None, so a hybrid-search hit
shaped `{"memory": "", "chunk": "..."}` resolves to the empty string and
never falls through to `chunk`.
"""
if isinstance(item, str):
return item.strip()
for name in ("memory", "chunk", "content"):
value = _field(item, name)
if isinstance(value, str) and value.strip():
return value.strip()
return ""
def deduplicate_memories(
static: List[str],
dynamic: List[str],
@ -125,12 +141,7 @@ def deduplicate_memories(
out: List[Any] = []
for r in results:
# v4 search.memories/hybrid uses `memory` or `chunk`.
memory = (
r if isinstance(r, str) else _field(r, "memory", "chunk", "content", default="")
)
if not isinstance(memory, str):
memory = ""
memory = memory.strip()
memory = _memory_text(r)
key = comparison_key(memory)
if key and key not in seen:
seen.add(key)
@ -177,7 +188,7 @@ def format_memories_to_text(
lines.append(f"- {item}")
continue
memory = _field(item, "memory", "chunk", "content", default="")
memory = _memory_text(item)
updated_at = _field(item, "updatedAt", "updated_at", default="")
time_str = format_relative_time(updated_at) if updated_at else ""
if time_str:

View file

@ -0,0 +1,103 @@
"""Import stubs for the heavy optional runtime dependencies.
CI installs the real `pipecat-ai`, `loguru` and `pydantic`, so every stub
here is a no-op there. Locally they let the pure-Python helpers under test
import without the full voice stack, and living in conftest means any test
module gets them regardless of collection order.
"""
from __future__ import annotations
import sys
import types
def _install_test_stubs() -> None:
if "loguru" not in sys.modules:
loguru_module = types.ModuleType("loguru")
class _Logger:
def warning(self, *_args, **_kwargs):
return None
def error(self, *_args, **_kwargs):
return None
loguru_module.logger = _Logger()
sys.modules["loguru"] = loguru_module
if "pydantic" not in sys.modules:
pydantic_module = types.ModuleType("pydantic")
class BaseModel:
def __init__(self, **kwargs):
for key, value in kwargs.items():
setattr(self, key, value)
def Field(*, default=None, **_kwargs):
return default
pydantic_module.BaseModel = BaseModel
pydantic_module.Field = Field
sys.modules["pydantic"] = pydantic_module
if "pipecat" not in sys.modules:
pipecat_module = types.ModuleType("pipecat")
sys.modules["pipecat"] = pipecat_module
frames_module = types.ModuleType("pipecat.frames.frames")
class Frame: # pragma: no cover - import stub
pass
class InputAudioRawFrame: # pragma: no cover - import stub
pass
class LLMContextFrame: # pragma: no cover - import stub
pass
class LLMMessagesFrame: # pragma: no cover - import stub
pass
frames_module.Frame = Frame
frames_module.InputAudioRawFrame = InputAudioRawFrame
frames_module.LLMContextFrame = LLMContextFrame
frames_module.LLMMessagesFrame = LLMMessagesFrame
llm_context_module = types.ModuleType("pipecat.processors.aggregators.llm_context")
class LLMContext: # pragma: no cover - import stub
pass
llm_context_module.LLMContext = LLMContext
openai_context_module = types.ModuleType(
"pipecat.processors.aggregators.openai_llm_context"
)
class OpenAILLMContextFrame: # pragma: no cover - import stub
pass
openai_context_module.OpenAILLMContextFrame = OpenAILLMContextFrame
frame_processor_module = types.ModuleType("pipecat.processors.frame_processor")
class FrameDirection: # pragma: no cover - import stub
pass
class FrameProcessor:
def __init__(self, *args, **kwargs):
return None
frame_processor_module.FrameDirection = FrameDirection
frame_processor_module.FrameProcessor = FrameProcessor
sys.modules["pipecat.frames.frames"] = frames_module
sys.modules["pipecat.processors.aggregators.llm_context"] = llm_context_module
sys.modules[
"pipecat.processors.aggregators.openai_llm_context"
] = openai_context_module
sys.modules["pipecat.processors.frame_processor"] = frame_processor_module
_install_test_stubs()

View file

@ -0,0 +1,92 @@
from __future__ import annotations
import unittest
from types import SimpleNamespace
from supermemory_pipecat.utils import (
deduplicate_memories,
format_memories_to_text,
)
class TestSearchResultExtraction(unittest.TestCase):
"""v4 hybrid search returns chunk hits whose `memory` field is empty.
`_field` stops at the first value that is not None, so those hits used to
resolve to "" and were dropped by dedup -- or, when they survived, rendered
as an empty bullet. The TypeScript `getMemoryText` takes the first
non-empty field instead, and the other Supermemory Python SDKs match it.
"""
def test_falls_through_to_chunk_when_memory_is_empty(self) -> None:
result = deduplicate_memories(
static=[],
dynamic=[],
search_results=[{"memory": "", "chunk": "The user's dog is called Rex"}],
)
self.assertEqual(len(result["search_results"]), 1)
self.assertIn("The user's dog is called Rex", format_memories_to_text(result))
def test_falls_through_to_chunk_when_memory_is_whitespace(self) -> None:
result = deduplicate_memories(
static=[],
dynamic=[],
search_results=[{"memory": " ", "chunk": "Chunk body"}],
)
self.assertEqual(len(result["search_results"]), 1)
self.assertIn("Chunk body", format_memories_to_text(result))
def test_falls_through_on_models_too(self) -> None:
result = deduplicate_memories(
static=[],
dynamic=[],
search_results=[SimpleNamespace(memory="", chunk="Chunk body")],
)
self.assertEqual(len(result["search_results"]), 1)
self.assertIn("Chunk body", format_memories_to_text(result))
def test_never_renders_an_empty_bullet(self) -> None:
result = deduplicate_memories(
static=[],
dynamic=[],
search_results=[
{
"memory": "",
"chunk": "Chunk body",
"updatedAt": "2020-01-01T00:00:00Z",
}
],
)
rendered = format_memories_to_text(result)
self.assertIn("Chunk body", rendered)
for line in rendered.splitlines():
if line.startswith("- "):
self.assertNotRegex(line, r"^- (\[[^\]]*\] )?$")
def test_memory_still_wins_over_chunk(self) -> None:
result = deduplicate_memories(
static=[],
dynamic=[],
search_results=[{"memory": "Memory body", "chunk": "Chunk body"}],
)
rendered = format_memories_to_text(result)
self.assertIn("Memory body", rendered)
self.assertNotIn("Chunk body", rendered)
def test_entry_with_no_usable_text_is_still_dropped(self) -> None:
result = deduplicate_memories(
static=[],
dynamic=[],
search_results=[{"memory": "", "chunk": ""}, {}, None],
)
self.assertEqual(result["search_results"], [])
if __name__ == "__main__":
unittest.main()