From 91055c2f8ad4fb079aa973d667cbde14b6fa46d9 Mon Sep 17 00:00:00 2001 From: Josh Mabry Date: Thu, 27 Aug 2026 12:41:42 -0700 Subject: [PATCH 1/7] =?UTF-8?q?feat:=20typed=20memory=20schema=20=E2=80=94?= =?UTF-8?q?=20additive=20columns=20for=20memory=20classification=20(#3072)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add memory_kind, subject, review_state, and expires_at columns to the chunks table for operator-profile vs agent-note classification. Migration follows the existing additive pattern (PRAGMA table_info check, ALTER TABLE ADD COLUMN, indexed). Existing rows stay NULL. Delivery behavior is unchanged (owned by #3187). - knowledge/store.py: schema, migration, Chunk dataclass, add_chunk, add_document, search, list_chunks all updated - knowledge/hybrid_store.py: search and _vector_search pass through memory_kind/review_state filters to both FTS5 and vector rankings - knowledge/layered.py: search passes filters to both tiers - tools/lg_tools.py: memory_ingest accepts optional memory_kind, memory_recall accepts optional memory_kind filter, memory_list shows kind/review_state when present - tests/test_knowledge_typed_memory.py: 14 tests covering add, migrate, search/list filters, hybrid/layered passthrough, backward compat Co-Authored-By: Claude Opus 4.6 Claude-Session: https://claude.ai/code/session_01WEMxBi71vjtmmmziFCMcby --- changelog.d/3072.added.md | 1 + knowledge/hybrid_store.py | 20 +- knowledge/layered.py | 13 +- knowledge/store.py | 107 ++++++++- tests/test_knowledge_typed_memory.py | 328 +++++++++++++++++++++++++++ tools/lg_tools.py | 39 +++- 6 files changed, 492 insertions(+), 16 deletions(-) create mode 100644 changelog.d/3072.added.md create mode 100644 tests/test_knowledge_typed_memory.py diff --git a/changelog.d/3072.added.md b/changelog.d/3072.added.md new file mode 100644 index 000000000..28f2653b0 --- /dev/null +++ b/changelog.d/3072.added.md @@ -0,0 +1 @@ +**Typed memory schema** — additive columns (memory_kind, subject, review_state, expires_at) for operator-profile vs agent-note classification without changing delivery behavior (#3072) diff --git a/knowledge/hybrid_store.py b/knowledge/hybrid_store.py index 3710f2c1b..e7a571053 100644 --- a/knowledge/hybrid_store.py +++ b/knowledge/hybrid_store.py @@ -396,6 +396,9 @@ def _vector_search( namespace: str | list[str] | None = None, include_invalidated: bool = False, epoch: str | None = None, + *, + memory_kind: str | None = None, + review_state: str | None = None, ) -> list[int]: """Return chunk ids ranked by cosine similarity (brute force).""" db = self._get_db() @@ -411,6 +414,12 @@ def _vector_search( if epoch: where.append("c.epoch = ?") params.append(epoch) + if memory_kind: + where.append("c.memory_kind = ?") + params.append(memory_kind) + if review_state: + where.append("c.review_state = ?") + params.append(review_state) ns_sql, ns_params = _namespace_clause(namespace, col="c.namespace") if ns_sql: where.append(ns_sql) @@ -454,6 +463,8 @@ def search( namespace: str | list[str] | None = None, include_invalidated: bool = False, epoch: str | None = None, + memory_kind: str | None = None, + review_state: str | None = None, ) -> list[dict]: """RRF-fuse the FTS5 ranking with a vector ranking. @@ -465,6 +476,8 @@ def search( rankings by default; ``include_invalidated=True`` is the audit escape hatch. ``epoch`` (#1634) likewise filters BOTH rankings — an out-of-era chunk can't surface as a vector-only hit. + ``memory_kind`` / ``review_state`` (#3072) likewise filter BOTH + rankings. """ if not query or not query.strip(): return [] @@ -477,12 +490,17 @@ def search( namespace=namespace, include_invalidated=include_invalidated, epoch=epoch, + memory_kind=memory_kind, + review_state=review_state, ) query_vec = self._embed(query) if query_vec is None: return base[:k] - vec_ids = self._vector_search(query_vec, self._vector_k, domain, namespace, include_invalidated, epoch) + vec_ids = self._vector_search( + query_vec, self._vector_k, domain, namespace, include_invalidated, epoch, + memory_kind=memory_kind, review_state=review_state, + ) if not vec_ids: return base[:k] diff --git a/knowledge/layered.py b/knowledge/layered.py index 9e12f13ae..27fbd8c56 100644 --- a/knowledge/layered.py +++ b/knowledge/layered.py @@ -56,18 +56,23 @@ def search( namespace: str | list[str] | None = None, include_invalidated: bool = False, epoch: str | None = None, + memory_kind: str | None = None, + review_state: str | None = None, ) -> list[dict]: """Top-k across BOTH tiers, fused by RRF over each tier's rank, tier-tagged. A chunk promoted into the commons (same content as its private original) is de-duped — the private record wins (it's editable) but keeps the summed score. ``namespace`` (ADR 0069 D3a), ``include_invalidated`` (ADR 0069 D9 — - superseded rows are excluded by default), and ``epoch`` (#1634 — era - scoping) are passed through to both tiers.""" + superseded rows are excluded by default), ``epoch`` (#1634 — era scoping), + and ``memory_kind`` / ``review_state`` (#3072 — typed-memory classification) + are passed through to both tiers.""" priv = self._private.search( - query, k, domain=domain, namespace=namespace, include_invalidated=include_invalidated, epoch=epoch + query, k, domain=domain, namespace=namespace, include_invalidated=include_invalidated, epoch=epoch, + memory_kind=memory_kind, review_state=review_state, ) comm = self._commons.search( - query, k, domain=domain, namespace=namespace, include_invalidated=include_invalidated, epoch=epoch + query, k, domain=domain, namespace=namespace, include_invalidated=include_invalidated, epoch=epoch, + memory_kind=memory_kind, review_state=review_state, ) fused: dict[str, dict] = {} diff --git a/knowledge/store.py b/knowledge/store.py index 155a1ad84..96ea7d1c6 100644 --- a/knowledge/store.py +++ b/knowledge/store.py @@ -93,6 +93,12 @@ class Chunk: # D9, kept forever); ``_BULK_DELETE_REASON`` = a reversible bulk delete-by-source # (#1770) that the grace sweep may eventually reap. invalidation_reason: str | None = None + # Typed memory schema (#3072) — additive columns for memory classification. + # NULL on legacy/untyped rows; delivery behavior unchanged (owned by #3187). + memory_kind: str | None = None + subject: str | None = None + review_state: str | None = None + expires_at: str | None = None def as_dict(self) -> dict[str, Any]: return { @@ -108,6 +114,10 @@ def as_dict(self) -> dict[str, Any]: "created_at": self.created_at, "updated_at": self.updated_at, "invalidated_at": self.invalidated_at, + "memory_kind": self.memory_kind, + "subject": self.subject, + "review_state": self.review_state, + "expires_at": self.expires_at, } @@ -279,7 +289,11 @@ def _has_fts5(db: sqlite3.Connection) -> bool: created_at TEXT NOT NULL, updated_at TEXT NOT NULL, invalidated_at TEXT, - invalidation_reason TEXT + invalidation_reason TEXT, + memory_kind TEXT, + subject TEXT, + review_state TEXT, + expires_at TEXT ); CREATE INDEX IF NOT EXISTS idx_chunks_domain ON chunks(domain); @@ -422,6 +436,25 @@ def _init_db(self) -> None: db.execute("CREATE INDEX IF NOT EXISTS idx_chunks_epoch ON chunks(epoch)") except sqlite3.DatabaseError as exc: log.debug("[knowledge] epoch migration skipped: %s", exc) + # Migration: typed memory schema (#3072) — additive columns for + # memory classification (memory_kind, subject, review_state, + # expires_at). Same nullable pattern; existing rows stay NULL + # (classified as legacy). Delivery behavior unchanged (owned by + # #3187). + try: + cols = {r[1] for r in db.execute("PRAGMA table_info(chunks)")} + if "memory_kind" not in cols: + db.execute("ALTER TABLE chunks ADD COLUMN memory_kind TEXT") + if "subject" not in cols: + db.execute("ALTER TABLE chunks ADD COLUMN subject TEXT") + if "review_state" not in cols: + db.execute("ALTER TABLE chunks ADD COLUMN review_state TEXT") + if "expires_at" not in cols: + db.execute("ALTER TABLE chunks ADD COLUMN expires_at TEXT") + db.execute("CREATE INDEX IF NOT EXISTS idx_chunks_memory_kind ON chunks(memory_kind)") + db.execute("CREATE INDEX IF NOT EXISTS idx_chunks_review_state ON chunks(review_state)") + except sqlite3.DatabaseError as exc: + log.debug("[knowledge] typed memory schema migration skipped: %s", exc) self._fts_available = _has_fts5(db) if self._fts_available: db.executescript(_FTS_SCHEMA) @@ -495,6 +528,10 @@ def add_chunk( finding_type: str | None = None, namespace: str | None = None, epoch: str | None = None, + memory_kind: str | None = None, + subject: str | None = None, + review_state: str | None = None, + expires_at: str | None = None, ) -> int | None: """Insert a chunk. Returns the new row id, or None on failure. @@ -507,6 +544,10 @@ def add_chunk( ``epoch`` (#1634) tags the chunk with the era it was learned in (an opaque string, e.g. a reset date) so ``search(epoch=...)`` can scope retrieval to the current era of a resettable world. + + ``memory_kind`` / ``subject`` / ``review_state`` / ``expires_at`` + (#3072) are typed-memory classification columns — additive, nullable, + backward-compatible. Delivery behavior unchanged (owned by #3187). """ if not content or not content.strip(): return None @@ -521,8 +562,14 @@ def add_chunk( cur = db.execute( "INSERT INTO chunks " "(content, domain, heading, source, source_type, finding_type, " - "namespace, epoch, created_at, updated_at) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)", - (content, domain, heading, source, source_type, finding_type, namespace, epoch, now, now), + "namespace, epoch, memory_kind, subject, review_state, expires_at, " + "created_at, updated_at) " + "VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)", + ( + content, domain, heading, source, source_type, finding_type, + namespace, epoch, memory_kind, subject, review_state, expires_at, + now, now, + ), ) db.commit() chunk_id = int(cur.lastrowid) @@ -570,6 +617,10 @@ def add_document( finding_type: str | None = None, namespace: str | None = None, epoch: str | None = None, + memory_kind: str | None = None, + subject: str | None = None, + review_state: str | None = None, + expires_at: str | None = None, max_chars: int | None = None, overlap_chars: int | None = None, min_chars: int | None = None, @@ -613,6 +664,10 @@ def add_document( finding_type=finding_type, namespace=namespace, epoch=epoch, + memory_kind=memory_kind, + subject=subject, + review_state=review_state, + expires_at=expires_at, ) if cid is not None: ids.append(cid) @@ -691,6 +746,8 @@ def search( namespace: str | list[str] | None = None, include_invalidated: bool = False, epoch: str | None = None, + memory_kind: str | None = None, + review_state: str | None = None, ) -> list[dict[str, Any]]: """Top-k chunks matching ``query``. Shape matches what the ``KnowledgeMiddleware`` consumes: each result has ``table``, @@ -709,6 +766,9 @@ def search( ``epoch`` (#1634) restricts hits to chunks tagged with exactly that epoch (see :meth:`add_chunk`) — chunks from other eras, and untagged chunks, don't match. ``None`` = unfiltered (today's behavior). + + ``memory_kind`` / ``review_state`` (#3072) restrict hits to chunks + with exactly that typed-memory classification. ``None`` = unfiltered. """ if not query or not query.strip(): return [] @@ -717,9 +777,15 @@ def search( return [] try: rows = ( - self._search_fts(db, query, k, domain, namespace, include_invalidated, epoch) + self._search_fts( + db, query, k, domain, namespace, include_invalidated, epoch, + memory_kind=memory_kind, review_state=review_state, + ) if self._fts_available - else self._search_like(db, query, k, domain, namespace, include_invalidated, epoch) + else self._search_like( + db, query, k, domain, namespace, include_invalidated, epoch, + memory_kind=memory_kind, review_state=review_state, + ) ) except sqlite3.DatabaseError as exc: log.warning("[knowledge] search failed: %s", exc) @@ -748,6 +814,9 @@ def _search_fts( namespace: str | list[str] | None = None, include_invalidated: bool = False, epoch: str | None = None, + *, + memory_kind: str | None = None, + review_state: str | None = None, ) -> list[sqlite3.Row]: # Sanitize to FTS5-safe tokens; OR them so a multi-word query # matches any of the keywords (closer to LIKE behaviour). @@ -769,6 +838,12 @@ def _search_fts( if epoch: where.append("c.epoch = ?") params.append(epoch) + if memory_kind: + where.append("c.memory_kind = ?") + params.append(memory_kind) + if review_state: + where.append("c.review_state = ?") + params.append(review_state) ns_sql, ns_params = _namespace_clause(namespace, col="c.namespace") if ns_sql: where.append(ns_sql) @@ -791,6 +866,9 @@ def _search_like( namespace: str | list[str] | None = None, include_invalidated: bool = False, epoch: str | None = None, + *, + memory_kind: str | None = None, + review_state: str | None = None, ) -> list[sqlite3.Row]: tokens = [t for t in re.findall(r"[\w']+", query) if t] if not tokens: @@ -815,6 +893,12 @@ def _search_like( if epoch: sql += " AND epoch = ?" params.append(epoch) + if memory_kind: + sql += " AND memory_kind = ?" + params.append(memory_kind) + if review_state: + sql += " AND review_state = ?" + params.append(review_state) ns_sql, ns_params = _namespace_clause(namespace) if ns_sql: sql += f" AND {ns_sql}" @@ -830,6 +914,8 @@ def list_chunks( *, namespace: str | None = None, include_invalidated: bool = False, + memory_kind: str | None = None, + review_state: str | None = None, ) -> list[Chunk]: """Most-recent-first chunk listing. Used by ``memory_list`` and the fact consolidator. ``namespace`` (ADR 0021) optionally scopes to one @@ -837,7 +923,10 @@ def list_chunks( Superseded rows (ADR 0069 D9) are excluded by default — so hot-memory injection, ``memory_list``, and the fact consolidator only see valid - rows. ``include_invalidated=True`` is the audit escape hatch.""" + rows. ``include_invalidated=True`` is the audit escape hatch. + + ``memory_kind`` / ``review_state`` (#3072) optionally restrict the + listing to chunks with that typed-memory classification.""" db = self._get_db() if db is None: return [] @@ -851,6 +940,12 @@ def list_chunks( if namespace is not None: clauses.append("namespace = ?") params.append(namespace) + if memory_kind is not None: + clauses.append("memory_kind = ?") + params.append(memory_kind) + if review_state is not None: + clauses.append("review_state = ?") + params.append(review_state) where = (" WHERE " + " AND ".join(clauses)) if clauses else "" params.append(limit) try: diff --git a/tests/test_knowledge_typed_memory.py b/tests/test_knowledge_typed_memory.py new file mode 100644 index 000000000..7ac3d8b2f --- /dev/null +++ b/tests/test_knowledge_typed_memory.py @@ -0,0 +1,328 @@ +"""Typed memory schema (#3072): additive columns for memory classification. + +Tests the additive migration (memory_kind, subject, review_state, expires_at) +and the filter plumbing across all three store types without changing delivery +behavior (owned by #3187). +""" + +from __future__ import annotations + +import sqlite3 + +from knowledge.hybrid_store import HybridKnowledgeStore +from knowledge.layered import LayeredKnowledgeStore +from knowledge.store import KnowledgeStore + + +# ── basic add + retrieve ───────────────────────────────────────────────────── + + +def test_add_chunk_with_memory_kind(tmp_path): + """New typed-memory fields are stored and returned on the Chunk.""" + store = KnowledgeStore(tmp_path / "kb.db") + cid = store.add_chunk( + "The operator prefers dark mode", + domain="preferences", + memory_kind="profile", + subject="operator", + review_state="confirmed", + expires_at="2027-01-01T00:00:00+00:00", + ) + assert cid is not None + chunks = store.list_chunks(domain="preferences", limit=1) + assert len(chunks) == 1 + c = chunks[0] + assert c.id == cid + assert c.memory_kind == "profile" + assert c.subject == "operator" + assert c.review_state == "confirmed" + assert c.expires_at == "2027-01-01T00:00:00+00:00" + # as_dict includes the new fields + d = c.as_dict() + assert d["memory_kind"] == "profile" + assert d["subject"] == "operator" + assert d["review_state"] == "confirmed" + assert d["expires_at"] == "2027-01-01T00:00:00+00:00" + + +def test_add_chunk_without_kind_defaults_none(tmp_path): + """Backward compatibility: omitting typed fields leaves them NULL.""" + store = KnowledgeStore(tmp_path / "kb.db") + cid = store.add_chunk("plain fact", domain="general") + assert cid is not None + c = store.list_chunks(limit=1)[0] + assert c.memory_kind is None + assert c.subject is None + assert c.review_state is None + assert c.expires_at is None + d = c.as_dict() + assert d["memory_kind"] is None + assert d["subject"] is None + assert d["review_state"] is None + assert d["expires_at"] is None + + +# ── migration ──────────────────────────────────────────────────────────────── + + +def test_migration_adds_columns(tmp_path): + """A DB created without the typed-memory columns gets them added on next open.""" + path = tmp_path / "old.db" + # Simulate a pre-#3072 schema: chunks table without the new columns. + db = sqlite3.connect(str(path)) + db.execute( + "CREATE TABLE chunks (" + "id INTEGER PRIMARY KEY AUTOINCREMENT, content TEXT NOT NULL, " + "domain TEXT NOT NULL DEFAULT 'general', heading TEXT, source TEXT, " + "source_type TEXT, finding_type TEXT, namespace TEXT, epoch TEXT, " + "created_at TEXT NOT NULL, updated_at TEXT NOT NULL, " + "invalidated_at TEXT, invalidation_reason TEXT)" + ) + db.execute( + "INSERT INTO chunks (content, domain, created_at, updated_at) " + "VALUES ('legacy row', 'general', 'x', 'x')" + ) + db.commit() + db.close() + + # Opening the store triggers the migration. + store = KnowledgeStore(path) + store.add_chunk("new typed row", domain="general", memory_kind="fact", subject="project-x") + + rows = {c.content: c for c in store.list_chunks(limit=10)} + assert rows["legacy row"].memory_kind is None + assert rows["legacy row"].subject is None + assert rows["new typed row"].memory_kind == "fact" + assert rows["new typed row"].subject == "project-x" + + # Indexes were created. + idx_db = sqlite3.connect(str(path)) + indexes = {r[1] for r in idx_db.execute("PRAGMA index_list(chunks)")} + idx_db.close() + assert "idx_chunks_memory_kind" in indexes + assert "idx_chunks_review_state" in indexes + + +# ── search filters ─────────────────────────────────────────────────────────── + + +def test_search_filter_by_kind(tmp_path): + """search(memory_kind=...) returns only chunks of that kind.""" + store = KnowledgeStore(tmp_path / "kb.db") + store.add_chunk("operator likes coffee", domain="prefs", memory_kind="profile") + store.add_chunk("deploy on Friday", domain="prefs", memory_kind="decision") + store.add_chunk("untyped preference", domain="prefs") + + hits = store.search("preference coffee deploy Friday", k=10, memory_kind="profile") + assert len(hits) == 1 + assert hits[0]["memory_kind"] == "profile" + + # Unfiltered sees all. + all_hits = store.search("preference coffee deploy Friday", k=10) + assert len(all_hits) == 3 + + +def test_search_filter_by_kind_like_fallback(tmp_path): + """The LIKE fallback path also respects memory_kind.""" + store = KnowledgeStore(tmp_path / "kb.db") + store.add_chunk("alpha fact", domain="d", memory_kind="fact") + store.add_chunk("alpha decision", domain="d", memory_kind="decision") + store._fts_available = False # force LIKE path + + hits = store.search("alpha", k=10, memory_kind="fact") + assert len(hits) == 1 + assert hits[0]["memory_kind"] == "fact" + + +def test_search_filter_by_review_state(tmp_path): + """search(review_state=...) returns only chunks with that state.""" + store = KnowledgeStore(tmp_path / "kb.db") + store.add_chunk("confirmed insight", domain="d", memory_kind="fact", review_state="confirmed") + store.add_chunk("candidate insight", domain="d", memory_kind="fact", review_state="candidate") + + hits = store.search("insight", k=10, review_state="confirmed") + assert len(hits) == 1 + assert hits[0]["review_state"] == "confirmed" + + +# ── list filters ───────────────────────────────────────────────────────────── + + +def test_list_filter_by_kind(tmp_path): + """list_chunks(memory_kind=...) returns only chunks of that kind.""" + store = KnowledgeStore(tmp_path / "kb.db") + store.add_chunk("profile entry", domain="d", memory_kind="profile") + store.add_chunk("note entry", domain="d", memory_kind="note") + store.add_chunk("untyped entry", domain="d") + + profile_chunks = store.list_chunks(memory_kind="profile") + assert len(profile_chunks) == 1 + assert profile_chunks[0].memory_kind == "profile" + + note_chunks = store.list_chunks(memory_kind="note") + assert len(note_chunks) == 1 + assert note_chunks[0].memory_kind == "note" + + +def test_list_filter_by_review_state(tmp_path): + """list_chunks(review_state=...) returns only chunks with that state.""" + store = KnowledgeStore(tmp_path / "kb.db") + store.add_chunk("confirmed", domain="d", review_state="confirmed") + store.add_chunk("rejected", domain="d", review_state="rejected") + store.add_chunk("unreviewed", domain="d") + + confirmed = store.list_chunks(review_state="confirmed") + assert len(confirmed) == 1 + assert confirmed[0].review_state == "confirmed" + + +# ── hybrid store passthrough ───────────────────────────────────────────────── + + +def _const_embed(text: str) -> list[float]: + return [1.0, 0.0] + + +def test_hybrid_store_passthrough(tmp_path): + """HybridKnowledgeStore passes typed-memory fields through to the base.""" + store = HybridKnowledgeStore(tmp_path / "kb.db", embed_fn=_const_embed) + cid = store.add_chunk( + "hybrid typed fact", + domain="d", + memory_kind="standing", + subject="the-project", + review_state="candidate", + ) + assert cid is not None + + # Searchable with the kind filter (both FTS and vector paths). + hits = store.search("hybrid typed", k=5, memory_kind="standing") + assert any(r["id"] == cid for r in hits) + + # No hits for a different kind. + misses = store.search("hybrid typed", k=5, memory_kind="profile") + assert not any(r["id"] == cid for r in misses) + + +def test_hybrid_search_review_state_filters_both_rankings(tmp_path): + """memory_kind and review_state filter both FTS5 and vector rankings.""" + store = HybridKnowledgeStore(tmp_path / "kb.db", embed_fn=_const_embed) + store.add_chunk("gamma delta", domain="d", memory_kind="fact", review_state="confirmed") + store.add_chunk("gamma delta too", domain="d", memory_kind="fact", review_state="candidate") + + # Filter by review_state — vector-only path (no shared tokens with query). + hits = store.search("zzzzz", k=5, review_state="confirmed") + assert len(hits) == 1 + assert hits[0]["review_state"] == "confirmed" + + +# ── layered store passthrough ──────────────────────────────────────────────── + + +def test_layered_store_passthrough(tmp_path): + """LayeredKnowledgeStore passes typed-memory fields through on writes and reads.""" + private = KnowledgeStore(tmp_path / "private.db") + commons = KnowledgeStore(tmp_path / "commons.db") + layered = LayeredKnowledgeStore(private, commons) + + # Write via the layered store (targets private via __getattr__). + layered.add_chunk("layered typed fact", domain="d", memory_kind="episode", subject="session-42") + + # Verify on the private store directly. + c = private.list_chunks(domain="d", limit=1)[0] + assert c.memory_kind == "episode" + assert c.subject == "session-42" + + # Search with kind filter through the layered store. + hits = layered.search("layered typed", k=5, memory_kind="episode") + assert len(hits) == 1 + assert hits[0]["content"] == "layered typed fact" + + # No hits for a different kind. + assert layered.search("layered typed", k=5, memory_kind="profile") == [] + + +def test_layered_search_filters_both_tiers(tmp_path): + """memory_kind filter is passed through to both private and commons tiers.""" + private = KnowledgeStore(tmp_path / "private.db") + commons = KnowledgeStore(tmp_path / "commons.db") + layered = LayeredKnowledgeStore(private, commons) + + private.add_chunk("private fact", domain="d", memory_kind="fact") + commons.add_chunk("commons decision", domain="d", memory_kind="decision") + + fact_hits = layered.search("private commons fact decision", k=10, memory_kind="fact") + assert len(fact_hits) == 1 + assert fact_hits[0]["content"] == "private fact" + + decision_hits = layered.search("private commons fact decision", k=10, memory_kind="decision") + assert len(decision_hits) == 1 + assert decision_hits[0]["content"] == "commons decision" + + +# ── existing rows unaffected ───────────────────────────────────────────────── + + +def test_existing_rows_unaffected(tmp_path): + """Legacy rows (no typed-memory fields) remain fully searchable and listable.""" + store = KnowledgeStore(tmp_path / "kb.db") + # Write a legacy-style row (no typed fields). + cid = store.add_chunk( + "I remember the old days", + domain="general", + heading="nostalgia", + source="conversation", + source_type="chat", + namespace="ns1", + epoch="e1", + ) + assert cid is not None + + # All existing fields intact. + c = store.list_chunks(limit=1)[0] + assert c.content == "I remember the old days" + assert c.domain == "general" + assert c.heading == "nostalgia" + assert c.source == "conversation" + assert c.source_type == "chat" + assert c.namespace == "ns1" + assert c.epoch == "e1" + assert c.memory_kind is None + assert c.subject is None + assert c.review_state is None + assert c.expires_at is None + + # Searchable. + hits = store.search("old days", k=5) + assert len(hits) == 1 + assert hits[0]["id"] == cid + + # Unfiltered list still includes it. + all_chunks = store.list_chunks(limit=50) + assert any(ch.id == cid for ch in all_chunks) + + # Kind-filtered search excludes it (NULL != 'fact'). + typed_hits = store.search("old days", k=5, memory_kind="fact") + assert len(typed_hits) == 0 + + +# ── add_document passthrough ───────────────────────────────────────────────── + + +def test_add_document_passes_typed_fields(tmp_path): + """add_document forwards typed-memory kwargs to each chunk's add_chunk.""" + store = KnowledgeStore(tmp_path / "kb.db") + ids = store.add_document( + "doc-sized content for classification", + domain="d", + memory_kind="reference", + subject="api-docs", + review_state="confirmed", + expires_at="2027-06-01T00:00:00+00:00", + ) + assert len(ids) >= 1 + c = store.list_chunks(domain="d", limit=1)[0] + assert c.memory_kind == "reference" + assert c.subject == "api-docs" + assert c.review_state == "confirmed" + assert c.expires_at == "2027-06-01T00:00:00+00:00" diff --git a/tools/lg_tools.py b/tools/lg_tools.py index 102087e63..49e2c4918 100644 --- a/tools/lg_tools.py +++ b/tools/lg_tools.py @@ -717,6 +717,7 @@ async def memory_ingest( content: str, domain: str = "general", heading: str | None = None, + memory_kind: str | None = None, ) -> str: """Store a fact, preference, or note in long-term memory. @@ -731,6 +732,10 @@ async def memory_ingest( ``"general"``. Defaults to ``"general"``. heading: Optional short label (e.g. ``"coffee"``) used as a stable de-dupe key by the eval suite and curator. + memory_kind: Optional typed classification — one of + ``"profile"``, ``"standing"``, ``"fact"``, ``"decision"``, + ``"note"``, ``"episode"``, ``"reference"``, ``"legacy"``. + When omitted the chunk is untyped (backward-compatible). Returns ``"Stored chunk N in 'domain'."`` on success. """ @@ -751,10 +756,14 @@ async def memory_ingest( # tier (ADR 0069 D8) — this write path is model-driven, not operator-. import asyncio + kw: dict[str, Any] = {"source_type": "conversation"} + if memory_kind is not None: + kw["memory_kind"] = memory_kind + def _write(): try: - return knowledge_store.add_chunk(content, domain=domain, heading=heading, source_type="conversation") - except TypeError: # plugin backend predating the source_type kwarg + return knowledge_store.add_chunk(content, domain=domain, heading=heading, **kw) + except TypeError: # plugin backend predating the new kwargs return knowledge_store.add_chunk(content, domain=domain, heading=heading) chunk_id = await asyncio.to_thread(_write) @@ -851,7 +860,12 @@ async def knowledge_ingest( ) @tool - async def memory_recall(query: str, k: int = 5, domain: str | None = None) -> str: + async def memory_recall( + query: str, + k: int = 5, + domain: str | None = None, + memory_kind: str | None = None, + ) -> str: """Search long-term memory for chunks relevant to ``query``. Returns the top-k matches, one per line, each citing its provenance @@ -866,6 +880,10 @@ async def memory_recall(query: str, k: int = 5, domain: str | None = None) -> st agent's history), not your own actions; pass the domain you actually want (e.g. your own, or ``claude-import`` to inspect the inherited set). + ``memory_kind`` optionally restricts results to one typed-memory + classification (e.g. ``"fact"``, ``"decision"``, ``"profile"``). + Omit to search all kinds (backward-compatible default). + Returns ``"No matches."`` when the store is empty or nothing scores above the keyword threshold. """ @@ -873,7 +891,11 @@ async def memory_recall(query: str, k: int = 5, domain: str | None = None) -> st # search embeds the query over HTTP on hybrid stores — keep it off the loop. import asyncio - results = await asyncio.to_thread(knowledge_store.search, query, k=clamped_k, domain=(domain or None)) + search_kw: dict[str, Any] = {"k": clamped_k, "domain": domain or None} + if memory_kind is not None: + search_kw["memory_kind"] = memory_kind + + results = await asyncio.to_thread(knowledge_store.search, query, **search_kw) if not results: return "No matches." lines = [ @@ -980,6 +1002,7 @@ async def memory_list(domain: str | None = None, limit: int = 10) -> str: Useful when the operator asks for recent activity ("what did I log today?") or wants to inspect what the agent has stored. + Shows memory_kind and review_state when present. """ clamped_limit = max(1, min(int(limit), _MEMORY_LIST_MAX_LIMIT)) chunks = knowledge_store.list_chunks(domain=domain, limit=clamped_limit) @@ -996,7 +1019,13 @@ async def memory_list(domain: str | None = None, limit: int = 10) -> str: # created_at already leads the line, so the citation adds # src/ns/trust only. cite = _memory_citation(source=c.source, namespace=c.namespace, source_type=c.source_type) - lines.append(f"#{c.id} {c.created_at} {head} {preview}{cite}") + # Show typed-memory classification when present (#3072). + kind_tag = "" + if getattr(c, "memory_kind", None): + kind_tag += f" kind={c.memory_kind}" + if getattr(c, "review_state", None): + kind_tag += f" review={c.review_state}" + lines.append(f"#{c.id} {c.created_at} {head} {preview}{cite}{kind_tag}") return "\n".join(lines) @tool From f7f5995ff2d1581e762553310384173be27199a1 Mon Sep 17 00:00:00 2001 From: Josh Mabry Date: Thu, 27 Aug 2026 12:55:21 -0700 Subject: [PATCH 2/7] fix: add bullet prefix to changelog fragment Co-Authored-By: Claude Opus 4.6 Claude-Session: https://claude.ai/code/session_01WEMxBi71vjtmmmziFCMcby --- changelog.d/3072.added.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/changelog.d/3072.added.md b/changelog.d/3072.added.md index 28f2653b0..464d5ba24 100644 --- a/changelog.d/3072.added.md +++ b/changelog.d/3072.added.md @@ -1 +1 @@ -**Typed memory schema** — additive columns (memory_kind, subject, review_state, expires_at) for operator-profile vs agent-note classification without changing delivery behavior (#3072) +- **Typed memory schema (#3072).** Additive columns (memory_kind, subject, review_state, expires_at) for operator-profile vs agent-note classification without changing delivery behavior. From cc40047f1a368583ccb040e915480991b6d82bd5 Mon Sep 17 00:00:00 2001 From: Josh Mabry Date: Thu, 27 Aug 2026 12:59:57 -0700 Subject: [PATCH 3/7] fix(knowledge): surface subject on memory_ingest, add memory_kind filter to memory_list - memory_ingest: accept subject param and InjectedState; stamp session namespace on writes so agent-authored memories carry provenance (#3072) - memory_list: accept memory_kind filter, forwarded to list_chunks - Tests: add list_chunks filter by kind, add_chunk with subject round-trip Co-Authored-By: Claude Opus 4.6 Claude-Session: https://claude.ai/code/session_01WEMxBi71vjtmmmziFCMcby --- tests/test_knowledge_typed_memory.py | 38 ++++++++++++++++++++++++++++ tools/lg_tools.py | 26 ++++++++++++++++--- 2 files changed, 61 insertions(+), 3 deletions(-) diff --git a/tests/test_knowledge_typed_memory.py b/tests/test_knowledge_typed_memory.py index 7ac3d8b2f..1bf4d92a0 100644 --- a/tests/test_knowledge_typed_memory.py +++ b/tests/test_knowledge_typed_memory.py @@ -326,3 +326,41 @@ def test_add_document_passes_typed_fields(tmp_path): assert c.subject == "api-docs" assert c.review_state == "confirmed" assert c.expires_at == "2027-06-01T00:00:00+00:00" + + +# ── tool-level tests ──────────────────────────────────────────────────────── + + +def test_list_chunks_filter_by_memory_kind(tmp_path): + """list_chunks(memory_kind=...) returns only matching rows.""" + store = KnowledgeStore(tmp_path / "kb.db") + store.add_chunk("profile fact", domain="general", memory_kind="profile") + store.add_chunk("a note", domain="general", memory_kind="note") + store.add_chunk("untyped", domain="general") + + only_profile = store.list_chunks(memory_kind="profile") + assert len(only_profile) == 1 + assert only_profile[0].memory_kind == "profile" + + only_notes = store.list_chunks(memory_kind="note") + assert len(only_notes) == 1 + assert only_notes[0].memory_kind == "note" + + all_rows = store.list_chunks() + assert len(all_rows) == 3 + + +def test_add_chunk_with_subject(tmp_path): + """subject is stored and round-trips through list_chunks.""" + store = KnowledgeStore(tmp_path / "kb.db") + cid = store.add_chunk( + "Josh prefers dark mode", + domain="preferences", + memory_kind="profile", + subject="operator", + ) + assert cid is not None + chunks = store.list_chunks(domain="preferences") + assert len(chunks) == 1 + assert chunks[0].subject == "operator" + assert chunks[0].memory_kind == "profile" diff --git a/tools/lg_tools.py b/tools/lg_tools.py index 49e2c4918..674041ac6 100644 --- a/tools/lg_tools.py +++ b/tools/lg_tools.py @@ -718,6 +718,8 @@ async def memory_ingest( domain: str = "general", heading: str | None = None, memory_kind: str | None = None, + subject: str | None = None, + state: Annotated[Any, InjectedState] = None, ) -> str: """Store a fact, preference, or note in long-term memory. @@ -736,6 +738,8 @@ async def memory_ingest( ``"profile"``, ``"standing"``, ``"fact"``, ``"decision"``, ``"note"``, ``"episode"``, ``"reference"``, ``"legacy"``. When omitted the chunk is untyped (backward-compatible). + subject: The entity this memory describes (e.g. the operator's + name, a project, a tool). Aids recall filtering. Returns ``"Stored chunk N in 'domain'."`` on success. """ @@ -759,6 +763,11 @@ async def memory_ingest( kw: dict[str, Any] = {"source_type": "conversation"} if memory_kind is not None: kw["memory_kind"] = memory_kind + if subject is not None: + kw["subject"] = subject + sid = _session_id_from(state) if state is not None else "" + if sid: + kw["namespace"] = sid def _write(): try: @@ -997,15 +1006,26 @@ async def recall_session(session_id: str) -> str: return rendered @tool - async def memory_list(domain: str | None = None, limit: int = 10) -> str: - """List the most recent chunks. Filter by domain when given. + async def memory_list( + domain: str | None = None, + limit: int = 10, + memory_kind: str | None = None, + ) -> str: + """List the most recent chunks. Filter by domain and/or memory_kind. Useful when the operator asks for recent activity ("what did I log today?") or wants to inspect what the agent has stored. Shows memory_kind and review_state when present. + + Args: + domain: Restrict to one domain bucket. + limit: Max entries (default 10). + memory_kind: Restrict to one typed kind (e.g. ``"profile"``). """ clamped_limit = max(1, min(int(limit), _MEMORY_LIST_MAX_LIMIT)) - chunks = knowledge_store.list_chunks(domain=domain, limit=clamped_limit) + chunks = knowledge_store.list_chunks( + domain=domain, limit=clamped_limit, memory_kind=memory_kind, + ) if not chunks: return f"No chunks in {domain or 'any domain'}." lines = [] From 325e7e4833c7f1f17f2928ee3efa380103269d95 Mon Sep 17 00:00:00 2001 From: Josh Mabry Date: Thu, 27 Aug 2026 13:41:00 -0700 Subject: [PATCH 4/7] fix(knowledge): defer namespace stamping to #3185, clean up test duplicates MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit QA review correctly flagged that memory_ingest was writing namespace=, changing namespace-scoped query visibility — contradicting the PR's "no delivery changes" claim. Session/namespace provenance stamping belongs in #3185 (write lifecycle). Also removes two duplicate tests that were structurally identical to existing ones (test_list_filter_by_kind, test_add_chunk_with_memory_kind). Co-Authored-By: Claude Opus 4.6 --- tests/test_knowledge_typed_memory.py | 36 ---------------------------- tools/lg_tools.py | 4 ---- 2 files changed, 40 deletions(-) diff --git a/tests/test_knowledge_typed_memory.py b/tests/test_knowledge_typed_memory.py index 1bf4d92a0..06a6fa6f5 100644 --- a/tests/test_knowledge_typed_memory.py +++ b/tests/test_knowledge_typed_memory.py @@ -328,39 +328,3 @@ def test_add_document_passes_typed_fields(tmp_path): assert c.expires_at == "2027-06-01T00:00:00+00:00" -# ── tool-level tests ──────────────────────────────────────────────────────── - - -def test_list_chunks_filter_by_memory_kind(tmp_path): - """list_chunks(memory_kind=...) returns only matching rows.""" - store = KnowledgeStore(tmp_path / "kb.db") - store.add_chunk("profile fact", domain="general", memory_kind="profile") - store.add_chunk("a note", domain="general", memory_kind="note") - store.add_chunk("untyped", domain="general") - - only_profile = store.list_chunks(memory_kind="profile") - assert len(only_profile) == 1 - assert only_profile[0].memory_kind == "profile" - - only_notes = store.list_chunks(memory_kind="note") - assert len(only_notes) == 1 - assert only_notes[0].memory_kind == "note" - - all_rows = store.list_chunks() - assert len(all_rows) == 3 - - -def test_add_chunk_with_subject(tmp_path): - """subject is stored and round-trips through list_chunks.""" - store = KnowledgeStore(tmp_path / "kb.db") - cid = store.add_chunk( - "Josh prefers dark mode", - domain="preferences", - memory_kind="profile", - subject="operator", - ) - assert cid is not None - chunks = store.list_chunks(domain="preferences") - assert len(chunks) == 1 - assert chunks[0].subject == "operator" - assert chunks[0].memory_kind == "profile" diff --git a/tools/lg_tools.py b/tools/lg_tools.py index 674041ac6..408927e01 100644 --- a/tools/lg_tools.py +++ b/tools/lg_tools.py @@ -719,7 +719,6 @@ async def memory_ingest( heading: str | None = None, memory_kind: str | None = None, subject: str | None = None, - state: Annotated[Any, InjectedState] = None, ) -> str: """Store a fact, preference, or note in long-term memory. @@ -765,9 +764,6 @@ async def memory_ingest( kw["memory_kind"] = memory_kind if subject is not None: kw["subject"] = subject - sid = _session_id_from(state) if state is not None else "" - if sid: - kw["namespace"] = sid def _write(): try: From 2c85a3bf4dee5286cea5325bab2e1c4a8a736fb5 Mon Sep 17 00:00:00 2001 From: Josh Mabry Date: Thu, 27 Aug 2026 14:06:34 -0700 Subject: [PATCH 5/7] fix(knowledge): forward typed-memory fields in LayeredKnowledgeStore.promote MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit promote() was missing memory_kind, subject, review_state, expires_at (and epoch) when copying chunks from private to commons — promoted chunks silently lost their typed classification. Co-Authored-By: Claude Opus 4.6 Claude-Session: https://claude.ai/code/session_01WEMxBi71vjtmmmziFCMcby --- knowledge/layered.py | 5 +++++ tests/test_knowledge_typed_memory.py | 25 +++++++++++++++++++++++++ 2 files changed, 30 insertions(+) diff --git a/knowledge/layered.py b/knowledge/layered.py index 27fbd8c56..0b8beee5c 100644 --- a/knowledge/layered.py +++ b/knowledge/layered.py @@ -126,6 +126,11 @@ def promote(self, chunk_id: int) -> dict | None: source_type=chunk.get("source_type"), finding_type=chunk.get("finding_type"), namespace=chunk.get("namespace"), + epoch=chunk.get("epoch"), + memory_kind=chunk.get("memory_kind"), + subject=chunk.get("subject"), + review_state=chunk.get("review_state"), + expires_at=chunk.get("expires_at"), ) if self._commons.id_for_exact_content(content) is None: log.error("[knowledge] promote(%s): commons write did not land — is the commons writable?", chunk_id) diff --git a/tests/test_knowledge_typed_memory.py b/tests/test_knowledge_typed_memory.py index 06a6fa6f5..a6f76a30f 100644 --- a/tests/test_knowledge_typed_memory.py +++ b/tests/test_knowledge_typed_memory.py @@ -328,3 +328,28 @@ def test_add_document_passes_typed_fields(tmp_path): assert c.expires_at == "2027-06-01T00:00:00+00:00" +def test_promote_preserves_typed_fields(tmp_path): + """promote copies typed-memory fields from private to commons.""" + from knowledge.layered import LayeredKnowledgeStore + + priv = KnowledgeStore(tmp_path / "priv.db") + commons = KnowledgeStore(tmp_path / "commons.db") + store = LayeredKnowledgeStore(priv, commons) + priv.add_chunk( + "profile fact", + domain="hot", + memory_kind="profile", + subject="operator", + review_state="confirmed", + expires_at="2027-01-01T00:00:00+00:00", + ) + chunks = priv.list_chunks() + assert len(chunks) == 1 + store.promote(chunks[0].id) + commons_chunks = commons.list_chunks() + assert len(commons_chunks) == 1 + c = commons_chunks[0] + assert c.memory_kind == "profile" + assert c.subject == "operator" + assert c.review_state == "confirmed" + assert c.expires_at == "2027-01-01T00:00:00+00:00" From de35986c9f04c35ec43750f91d82bd0ef14ac9e8 Mon Sep 17 00:00:00 2001 From: Josh Mabry Date: Thu, 27 Aug 2026 14:11:00 -0700 Subject: [PATCH 6/7] fix(knowledge): forward typed-memory fields in LayeredKnowledgeStore.promote() promote() was silently dropping memory_kind, subject, review_state, and expires_at when copying chunks from private to commons tier. Co-Authored-By: Claude Opus 4.6 Claude-Session: https://claude.ai/code/session_01WEMxBi71vjtmmmziFCMcby --- tests/test_knowledge_layered.py | 24 ++++++++++++++++++++++++ 1 file changed, 24 insertions(+) diff --git a/tests/test_knowledge_layered.py b/tests/test_knowledge_layered.py index 6977e4dad..c259d15c3 100644 --- a/tests/test_knowledge_layered.py +++ b/tests/test_knowledge_layered.py @@ -97,6 +97,30 @@ def test_promote_unknown_id_returns_none(tmp_path): assert LayeredKnowledgeStore(private, commons).promote(9999) is None +def test_promote_forwards_typed_memory_fields(tmp_path): + """promote() must carry memory_kind/subject/review_state/expires_at into the commons.""" + private, commons = _stores(tmp_path) + cid = private.add_chunk( + "user prefers terse replies", + domain="general", + memory_kind="standing", + subject="operator", + review_state="approved", + expires_at="2027-01-01", + ) + layered = LayeredKnowledgeStore(private, commons) + rec = layered.promote(cid) + assert rec is not None + + commons_chunks = commons.list_chunks() + assert len(commons_chunks) == 1 + c = commons_chunks[0] + assert c["memory_kind"] == "standing" + assert c["subject"] == "operator" + assert c["review_state"] == "approved" + assert c["expires_at"] == "2027-01-01" + + def test_stats_split_by_tier(tmp_path): private, commons = _stores(tmp_path) private.add_chunk("a", domain="finding") From e0ba4cf2c668ac475e6d63d7ec46b773d89d2346 Mon Sep 17 00:00:00 2001 From: Josh Mabry Date: Thu, 27 Aug 2026 15:26:44 -0700 Subject: [PATCH 7/7] fix(tests): use attribute access on Chunk dataclass in layered test MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Same fix as the typed memory test file — list_chunks returns Chunk dataclass objects, not dicts. Co-Authored-By: Claude Opus 4.6 Claude-Session: https://claude.ai/code/session_01WEMxBi71vjtmmmziFCMcby --- tests/test_knowledge_layered.py | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/tests/test_knowledge_layered.py b/tests/test_knowledge_layered.py index c259d15c3..c79e21d27 100644 --- a/tests/test_knowledge_layered.py +++ b/tests/test_knowledge_layered.py @@ -115,10 +115,10 @@ def test_promote_forwards_typed_memory_fields(tmp_path): commons_chunks = commons.list_chunks() assert len(commons_chunks) == 1 c = commons_chunks[0] - assert c["memory_kind"] == "standing" - assert c["subject"] == "operator" - assert c["review_state"] == "approved" - assert c["expires_at"] == "2027-01-01" + assert c.memory_kind == "standing" + assert c.subject == "operator" + assert c.review_state == "approved" + assert c.expires_at == "2027-01-01" def test_stats_split_by_tier(tmp_path):