mirror of
https://github.com/langchain-ai/langgraph.git
synced 2026-09-13 05:07:51 +02:00
fix(checkpoint-postgres,checkpoint-sqlite): scope namespace matching to segment boundaries (#8478)
## Summary
Namespace scoping in the Postgres and SQLite stores matched the
dot-joined prefix with `LIKE '<path>%'`, which does not respect the `.`
separator — a search scoped to `("foo",)` also returned rows under
`("foobar",)`. Scoping now matches the namespace exactly or requires the
separator before any remainder, and pattern metacharacters in labels are
escaped.
`list_namespaces` moves to segment-aware matching for prefix and suffix
conditions, since neither `LIKE` nor `GLOB` can express "any character
except the separator".
Per-package reasoning is in the commit message.
## Compatibility
`*` in a `list_namespaces` match path now spans exactly one segment,
restoring the documented behavior (`NamespacePath` documents `("cache",
"*", "v1")` as "any cache category with v1 version") and matching
`InMemoryStore`. To match at any depth, combine both conditions, which
are ANDed: `list_namespaces(prefix=["uid"], suffix=["alice"])`.
## Test plan
- [x] `make format` / `make lint` / `make test` from
`libs/checkpoint-postgres` (224 passed) and `libs/checkpoint-sqlite`
(112 passed, 3 skipped)
This commit is contained in:
@@ -4,6 +4,7 @@ import asyncio
|
||||
import concurrent.futures
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
import threading
|
||||
from collections import defaultdict
|
||||
from collections.abc import Callable, Iterable, Iterator, Sequence
|
||||
@@ -463,8 +464,9 @@ class BasePostgresStore(Generic[C]):
|
||||
ns_condition = "TRUE"
|
||||
ns_param: Sequence[str] | None = None
|
||||
if op.namespace_prefix:
|
||||
ns_condition = "store.prefix LIKE %s"
|
||||
ns_param = (f"{_namespace_to_text(op.namespace_prefix)}%",)
|
||||
ns_condition, ns_param = _namespace_prefix_condition(
|
||||
op.namespace_prefix
|
||||
)
|
||||
else:
|
||||
ns_param = ()
|
||||
|
||||
@@ -617,15 +619,17 @@ class BasePostgresStore(Generic[C]):
|
||||
conditions.append("(expires_at IS NULL OR expires_at > NOW())")
|
||||
if op.match_conditions:
|
||||
for condition in op.match_conditions:
|
||||
if condition.match_type == "prefix":
|
||||
conditions.append("prefix LIKE %s")
|
||||
if condition.match_type in ("prefix", "suffix"):
|
||||
if not condition.path:
|
||||
# An empty path constrains nothing; skipping keeps it a
|
||||
# no-op rather than emitting a pattern that matches no
|
||||
# namespace at all.
|
||||
continue
|
||||
conditions.append("prefix ~ %s")
|
||||
params.append(
|
||||
f"{_namespace_to_text(condition.path, handle_wildcards=True)}%"
|
||||
)
|
||||
elif condition.match_type == "suffix":
|
||||
conditions.append("prefix LIKE %s")
|
||||
params.append(
|
||||
f"%{_namespace_to_text(condition.path, handle_wildcards=True)}"
|
||||
_namespace_match_pattern(
|
||||
condition.path, condition.match_type
|
||||
)
|
||||
)
|
||||
else:
|
||||
logger.warning(
|
||||
@@ -1271,15 +1275,59 @@ def _get_index_params(store: Any) -> tuple[str, dict[str, Any]]:
|
||||
return kind, sanitized
|
||||
|
||||
|
||||
def _namespace_to_text(
|
||||
namespace: tuple[str, ...], handle_wildcards: bool = False
|
||||
) -> str:
|
||||
def _namespace_to_text(namespace: tuple[str, ...]) -> str:
|
||||
"""Convert namespace tuple to text string."""
|
||||
if handle_wildcards:
|
||||
namespace = tuple("%" if val == "*" else val for val in namespace)
|
||||
return ".".join(namespace)
|
||||
|
||||
|
||||
def _escape_like_literal(text: str) -> str:
|
||||
"""Escape LIKE metacharacters so `text` is matched literally.
|
||||
|
||||
Namespace labels may contain `_` and `%`, which would otherwise act as
|
||||
wildcards: `("user_1",)` would match `("userX1",)`. Backslash is escaped
|
||||
first so it cannot escape the following character. Requires an explicit
|
||||
`ESCAPE '\\'` clause on the pattern.
|
||||
"""
|
||||
return text.replace("\\", "\\\\").replace("%", "\\%").replace("_", "\\_")
|
||||
|
||||
|
||||
def _namespace_prefix_condition(namespace_prefix: tuple[str, ...]) -> tuple[str, tuple]:
|
||||
"""Build the SQL scoping a search to a namespace and its descendants.
|
||||
|
||||
Matches the namespace exactly or requires the `.` separator before any
|
||||
remainder, so a prefix of `("foo",)` does not also match `("foobar",)`.
|
||||
|
||||
Both arms stay index-friendly: equality on the `(prefix, key)` primary key,
|
||||
the anchored LIKE on the `prefix text_pattern_ops` index.
|
||||
|
||||
Only the LIKE arm is escaped -- equality does not interpret metacharacters,
|
||||
so escaping it would stop `("user_1",)` from matching itself.
|
||||
"""
|
||||
path = _namespace_to_text(namespace_prefix)
|
||||
condition = r"(store.prefix = %s OR store.prefix LIKE %s ESCAPE '\')"
|
||||
return condition, (path, f"{_escape_like_literal(path)}.%")
|
||||
|
||||
|
||||
def _namespace_match_pattern(path: tuple[str, ...], match_type: str) -> str:
|
||||
"""Build a POSIX regex matching the dot-joined prefix on whole segments.
|
||||
|
||||
Needed because `LIKE` cannot express "any character except the separator".
|
||||
Matches how `InMemoryStore` compares namespaces element-wise.
|
||||
|
||||
`*` matches exactly one segment. Prefix matches stay open-ended but must end
|
||||
on a separator; suffix matches anchor at the end and begin on one.
|
||||
|
||||
Examples:
|
||||
prefix ("uid", "*", "alice") -> ^uid\\.[^.]+\\.alice(\\.|\\Z)
|
||||
suffix ("alice",) -> (^|\\.)alice\\Z
|
||||
"""
|
||||
segments = ("[^.]+" if part == "*" else re.escape(part) for part in path)
|
||||
body = r"\.".join(segments)
|
||||
if match_type == "suffix":
|
||||
return rf"(^|\.){body}\Z"
|
||||
return rf"^{body}(\.|\Z)"
|
||||
|
||||
|
||||
def _row_to_item(
|
||||
namespace: tuple[str, ...],
|
||||
row: Row,
|
||||
|
||||
@@ -20,6 +20,10 @@ from langgraph.store.base import (
|
||||
from psycopg import Connection
|
||||
|
||||
from langgraph.store.postgres import PostgresStore
|
||||
from langgraph.store.postgres.base import (
|
||||
_escape_like_literal,
|
||||
_namespace_match_pattern,
|
||||
)
|
||||
from tests.conftest import (
|
||||
DEFAULT_URI,
|
||||
VECTOR_TYPES,
|
||||
@@ -326,6 +330,127 @@ def test_list_namespaces(store) -> None:
|
||||
store.delete(namespace, "dummy")
|
||||
|
||||
|
||||
def test_escape_like_literal() -> None:
|
||||
assert _escape_like_literal("users.alice") == "users.alice"
|
||||
assert _escape_like_literal("user_1") == r"user\_1"
|
||||
assert _escape_like_literal("100%") == r"100\%"
|
||||
assert _escape_like_literal("a\\b") == "a\\\\b"
|
||||
assert _escape_like_literal("") == ""
|
||||
|
||||
|
||||
def test_namespace_match_pattern() -> None:
|
||||
assert _namespace_match_pattern(("foo",), "prefix") == r"^foo(\.|\Z)"
|
||||
assert _namespace_match_pattern(("uid", "users"), "prefix") == r"^uid\.users(\.|\Z)"
|
||||
assert (
|
||||
_namespace_match_pattern(("uid", "*", "alice"), "prefix")
|
||||
== r"^uid\.[^.]+\.alice(\.|\Z)"
|
||||
)
|
||||
assert _namespace_match_pattern(("alice",), "suffix") == r"(^|\.)alice\Z"
|
||||
|
||||
# Regex metacharacters in a label are quoted, not interpreted.
|
||||
pattern = _namespace_match_pattern(("a.b+c",), "prefix")
|
||||
assert re.match(pattern, "a.b+c.child")
|
||||
assert not re.match(pattern, "axbbbc")
|
||||
|
||||
|
||||
def test_search_namespace_segment_boundary(store) -> None:
|
||||
"""Prefix scoping must stop at namespace segment boundaries.
|
||||
|
||||
Namespaces are stored dot-joined, so matching the raw text also returns
|
||||
siblings sharing leading characters. Callers isolate tenants by namespace,
|
||||
so prefix-shaped ids (1 vs 12) would cross-read.
|
||||
"""
|
||||
for namespace in [
|
||||
("foo",),
|
||||
("foo", "child"),
|
||||
("foo", "child", "deep"),
|
||||
("foobar",),
|
||||
("foobar", "baz"),
|
||||
("foo2",),
|
||||
]:
|
||||
store.put(namespace, "k", {"v": 1})
|
||||
|
||||
def _namespaces(prefix: tuple[str, ...]) -> set[tuple[str, ...]]:
|
||||
return {item.namespace for item in store.search(prefix, limit=100)}
|
||||
|
||||
assert _namespaces(("foo",)) == {
|
||||
("foo",),
|
||||
("foo", "child"),
|
||||
("foo", "child", "deep"),
|
||||
}
|
||||
# The sibling scope is independent, not merely narrower.
|
||||
assert _namespaces(("foobar",)) == {("foobar",), ("foobar", "baz")}
|
||||
assert _namespaces(("foo", "child")) == {("foo", "child"), ("foo", "child", "deep")}
|
||||
assert _namespaces(("foo2",)) == {("foo2",)}
|
||||
assert _namespaces(("fo",)) == set()
|
||||
|
||||
|
||||
def test_search_empty_prefix_is_unconstrained(store) -> None:
|
||||
"""An empty prefix constrains nothing and must return every namespace."""
|
||||
for namespace in [("a",), ("b", "c"), ("d", "e", "f")]:
|
||||
store.put(namespace, "k", {"v": 1})
|
||||
|
||||
assert {item.namespace for item in store.search((), limit=100)} == {
|
||||
("a",),
|
||||
("b", "c"),
|
||||
("d", "e", "f"),
|
||||
}
|
||||
|
||||
|
||||
def test_search_namespace_like_metacharacters(store) -> None:
|
||||
"""`_` and `%` are legal namespace labels, not LIKE wildcards."""
|
||||
for namespace in [
|
||||
("user_1",),
|
||||
("user_1", "child"),
|
||||
("userX1",),
|
||||
("a%b",),
|
||||
("axxb",),
|
||||
]:
|
||||
store.put(namespace, "k", {"v": 1})
|
||||
|
||||
def _namespaces(prefix: tuple[str, ...]) -> set[tuple[str, ...]]:
|
||||
return {item.namespace for item in store.search(prefix, limit=100)}
|
||||
|
||||
# Also asserts the namespace still matches itself, which catches escaping
|
||||
# the equality arm by mistake.
|
||||
assert _namespaces(("user_1",)) == {("user_1",), ("user_1", "child")}
|
||||
assert _namespaces(("a%b",)) == {("a%b",)}
|
||||
|
||||
|
||||
def test_list_namespaces_segment_boundary(store) -> None:
|
||||
for namespace in [
|
||||
("foo",),
|
||||
("foo", "child"),
|
||||
("foobar",),
|
||||
("foobar", "baz"),
|
||||
("uid", "users", "alice"),
|
||||
("uid", "users", "malice"),
|
||||
("uid", "a", "b", "alice"),
|
||||
]:
|
||||
store.put(namespace, "k", {"v": 1})
|
||||
|
||||
assert set(store.list_namespaces(prefix=["foo"], limit=100)) == {
|
||||
("foo",),
|
||||
("foo", "child"),
|
||||
}
|
||||
# Suffix must align to a segment: "malice" does not end with the "alice"
|
||||
# segment.
|
||||
assert set(store.list_namespaces(suffix=["alice"], limit=100)) == {
|
||||
("uid", "users", "alice"),
|
||||
("uid", "a", "b", "alice"),
|
||||
}
|
||||
# "*" spans exactly one segment.
|
||||
assert set(store.list_namespaces(prefix=["uid", "*", "alice"], limit=100)) == {
|
||||
("uid", "users", "alice"),
|
||||
}
|
||||
# Prefix matching stays open-ended across depth.
|
||||
assert set(store.list_namespaces(prefix=["uid"], limit=100)) == {
|
||||
("uid", "users", "alice"),
|
||||
("uid", "users", "malice"),
|
||||
("uid", "a", "b", "alice"),
|
||||
}
|
||||
|
||||
|
||||
def test_search(store) -> None:
|
||||
# Create test data
|
||||
test_data = [
|
||||
@@ -1026,3 +1151,16 @@ def test_non_ascii(
|
||||
assert result3[0].key == "3"
|
||||
assert result4[0].key == "4"
|
||||
assert result5[0].key == "5"
|
||||
|
||||
|
||||
def test_namespace_labels_with_trailing_newline(store) -> None:
|
||||
"""Labels may contain newlines, and must not match a differently-named label."""
|
||||
store.put(("users", "alice"), "k", {"v": 1})
|
||||
store.put(("users", "alice\n"), "k", {"v": 2})
|
||||
|
||||
assert set(store.list_namespaces(suffix=["alice"], limit=100)) == {
|
||||
("users", "alice"),
|
||||
}
|
||||
assert set(store.list_namespaces(prefix=["users", "alice"], limit=100)) == {
|
||||
("users", "alice"),
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user