Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
37 changes: 35 additions & 2 deletions hindsight-api-slim/hindsight_api/engine/retain/orchestrator.py
Original file line number Diff line number Diff line change
Expand Up @@ -53,6 +53,37 @@ def _count_delta_content_tokens(delta_contents: list["RetainContent"]) -> int:
return total


# Fixed namespace for deriving deterministic document IDs from content. NEVER
# change this value: it would change the derived IDs of every content-addressed
# document and break idempotent re-ingest against already-stored rows.
_DOCUMENT_ID_NAMESPACE = uuid.UUID("a3f2e1d0-7c4b-5a69-b8e2-1f0c9d6a4b73")


def _derive_document_id(contents_dicts: list["RetainContentDict"]) -> str:
"""Deterministic document_id derived from batch content + context namespace.

Fallback used when a caller supplies no explicit document_id. Making the id a
content hash (rather than a random uuid4) means re-ingesting identical
content+context upserts the same ``documents (id, bank_id)`` row via the
existing ON CONFLICT path instead of creating a duplicate — the random-uuid4
fallback was the source of the exact-duplicate documents found in the
BLO-9319 bank dedup.

Namespacing the hash by ``context`` keeps two intentionally-distinct
documents that happen to share identical content separate; only same-content
AND same-context re-ingests collapse. Returned as a uuid5 string so the id
keeps the 36-char shape of the historical uuid4 ids (document_id is never
parsed as a UUID downstream, but chunk ids concatenate it and the UI renders
it). Content is sanitized with the same helper used for ``content_hash`` so
whitespace/encoding noise does not perturb the id.
"""
combined_content = "\n".join(item.get("content", "") or "" for item in contents_dicts)
combined_context = "\n".join(item.get("context", "") or "" for item in contents_dicts)
sanitized_content = fact_extraction._sanitize_text(combined_content) or ""
sanitized_context = fact_extraction._sanitize_text(combined_context) or ""
return str(uuid.uuid5(_DOCUMENT_ID_NAMESPACE, f"{sanitized_context}|{sanitized_content}"))


def parse_datetime_flexible(value: Any) -> datetime:
"""
Parse a datetime value that could be either a datetime object or an ISO string.
Expand Down Expand Up @@ -478,7 +509,7 @@ async def retain_batch(
groups: dict[str, tuple[list[RetainContentDict], list[RetainContent]]] = {}
original_indices: dict[str, list[int]] = {}
for idx, (cd, c) in enumerate(zip(contents_dicts, contents)):
doc_key = cd.get("document_id") or str(uuid.uuid4())
doc_key = cd.get("document_id") or _derive_document_id([cd])
if doc_key not in groups:
groups[doc_key] = ([], [])
original_indices[doc_key] = []
Expand Down Expand Up @@ -550,7 +581,9 @@ async def retain_batch(
except Exception:
pass
if not effective_doc_id:
effective_doc_id = str(uuid.uuid4())
# Content-derived (deterministic) id so re-ingesting identical
# content+context upserts the same document instead of duplicating it.
effective_doc_id = _derive_document_id(contents_dicts)

# Record effective_doc_id on the operation (idempotent set-append). Captures
# both user-provided and generated ids so the operation shows every document
Expand Down
65 changes: 65 additions & 0 deletions hindsight-api-slim/tests/test_document_id_derivation.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,65 @@
"""
Unit tests for the content-derived document_id helper.

Pure-function tests (no DB / models / LLM) covering the deterministic id used as
the fallback when a caller supplies no explicit document_id. The end-to-end
idempotency behaviour is covered by
tests/test_document_tracking.py::test_no_document_id_is_content_derived_and_idempotent.
"""
import uuid

from hindsight_api.engine.retain.orchestrator import (
_DOCUMENT_ID_NAMESPACE,
_derive_document_id,
)


def test_same_content_and_context_yields_same_id():
a = _derive_document_id([{"content": "Alice works at Google.", "context": "sync"}])
b = _derive_document_id([{"content": "Alice works at Google.", "context": "sync"}])
assert a == b


def test_different_content_yields_different_id():
a = _derive_document_id([{"content": "Alice works at Google.", "context": "sync"}])
b = _derive_document_id([{"content": "Bob works at Microsoft.", "context": "sync"}])
assert a != b


def test_different_context_yields_different_id():
"""Namespacing: identical content under a different context is a distinct id."""
a = _derive_document_id([{"content": "Same body text.", "context": "conversation-1"}])
b = _derive_document_id([{"content": "Same body text.", "context": "conversation-2"}])
assert a != b


def test_missing_context_is_stable():
"""Absent context must not raise and must be deterministic."""
a = _derive_document_id([{"content": "No context here."}])
b = _derive_document_id([{"content": "No context here."}])
assert a == b


def test_id_is_a_uuid5_string():
"""Derived id keeps the 36-char uuid shape of the historical uuid4 ids."""
derived = _derive_document_id([{"content": "Some content.", "context": "ctx"}])
parsed = uuid.UUID(derived)
assert parsed.version == 5
assert str(parsed) == derived


def test_multi_item_batch_combines_all_items():
"""A multi-item batch hashes the combined content, distinct from one item."""
combined = _derive_document_id(
[
{"content": "First part.", "context": "ctx"},
{"content": "Second part.", "context": "ctx"},
]
)
single = _derive_document_id([{"content": "First part.", "context": "ctx"}])
assert combined != single


def test_namespace_constant_is_fixed():
"""Guard against accidental edits to the id namespace (would orphan all ids)."""
assert _DOCUMENT_ID_NAMESPACE == uuid.UUID("a3f2e1d0-7c4b-5a69-b8e2-1f0c9d6a4b73")
54 changes: 54 additions & 0 deletions hindsight-api-slim/tests/test_document_tracking.py
Original file line number Diff line number Diff line change
Expand Up @@ -86,6 +86,60 @@ async def test_document_upsert(memory, request_context):
await memory.delete_bank(bank_id, request_context=request_context)


@pytest.mark.asyncio
async def test_no_document_id_is_content_derived_and_idempotent(memory, request_context):
"""Re-ingesting identical content+context with NO explicit document_id must
upsert the same (content-derived) document, not create a duplicate.

Regression for the random-uuid4 fallback that produced the exact-duplicate
documents found in the BLO-9319 bank dedup.
"""
bank_id = f"test_content_derived_{datetime.now(timezone.utc).timestamp()}"

try:
content = "Alice works at Google. Bob works at Microsoft."
context = "Team sync notes"

# First ingest, no document_id provided.
await memory.retain_async(
bank_id=bank_id,
content=content,
context=context,
request_context=request_context,
)
docs_v1 = await memory.list_documents(bank_id, request_context=request_context)
assert docs_v1["total"] == 1

# Second ingest, byte-identical content+context, still no document_id.
await memory.retain_async(
bank_id=bank_id,
content=content,
context=context,
request_context=request_context,
)
docs_v2 = await memory.list_documents(bank_id, request_context=request_context)
assert docs_v2["total"] == 1, (
f"expected idempotent re-ingest (1 doc), got {docs_v2['total']}"
)
assert docs_v1["items"][0]["id"] == docs_v2["items"][0]["id"]

# Same content but a DIFFERENT context is a distinct document
# (namespacing keeps intentionally-separate docs apart).
await memory.retain_async(
bank_id=bank_id,
content=content,
context="A different conversation",
request_context=request_context,
)
docs_v3 = await memory.list_documents(bank_id, request_context=request_context)
assert docs_v3["total"] == 2, (
f"expected context-namespaced split (2 docs), got {docs_v3['total']}"
)

finally:
await memory.delete_bank(bank_id, request_context=request_context)


@pytest.mark.asyncio
async def test_document_deletion(memory, request_context):
"""Test that deleting a document cascades to memory units."""
Expand Down
Loading