mirror of
https://github.com/bytedance/deer-flow.git
synced 2026-08-04 03:49:25 +00:00
* feat(persistence): support custom postgres schema * fix(persistence): address CI lint/test failures and review feedback - Map missing psycopg import to actionable POSTGRES_INSTALL guidance in sync/async schema-creation helpers - Accept SQLAlchemy compound DSN schemes (postgresql+asyncpg) when injecting search_path, normalizing to a libpq-consumable DSN - Guard keyword-DSN tests with importorskip so they skip without psycopg - Set database=None in sync checkpointer none-fix test to avoid MagicMock backend resolution - Apply ruff import sort and format * fix(persistence): address pg-schema review feedback - Restrict postgres_schema regex to lowercase-only so the quoted CREATE SCHEMA matches the unquoted search_path (PG case-folds it), fixing the mixed-case bug where tables silently fell back to public. - Replace shlex.join/split with libpq-correct backslash escaping for the options parameter so values containing spaces survive intact. - Add normalize_libpq_dsn() and route the async checkpointer pool through dsn_with_search_path() so a +asyncpg suffix is stripped and existing DSN options (e.g. statement_timeout) are merged instead of overridden. - Extract shared ensure_postgres_schema()/ensure_postgres_schema_async() helpers (mapping missing psycopg to the install hint) used by all four provider sites. - Tests: reject mixed-case schemas, preserve space-containing libpq option, cover normalize_libpq_dsn, and assert pool search_path via DSN. * fix(persistence): align pg-schema test with merged store API The main merge moved the sync Store factory to the single-path _resolve_store_config/_sync_store_cm design, dropping the PR's _sync_store_from_database helper. The integration test still imported the removed symbol, breaking test collection (backend-unit-tests). Resolve the store config from a DatabaseConfig and drive it through _sync_store_cm instead. * fix(persistence): address pg-schema review feedback - reject trailing/leading whitespace in postgres_schema via re.fullmatch (a $-anchored re.match let "deerflow\n" through, silently landing tables in public) - re-escape all whitespace (TAB/CR/LF) when re-joining libpq options so a caller's pre-existing options value round-trips losslessly - re-validate the identifier inside create_schema_sql as defense-in-depth at the SQL-emitting boundary - accept the postgres:// short scheme in the alembic search_path injection - close the sync psycopg connection explicitly (psycopg3 __exit__ does not close()), mirroring the async path - drop the partial checkpointer/store reset on a database config change; database is restart-required and the ORM engine is not rebuilt, so a partial reset would half-migrate the deployment * docs(config): complete the postgres_schema migration checklist Address PR review (P1): the documented `public`->schema migration only moved runs, run_events, threads_meta, feedback, and users. That strands every other DeerFlow-owned table -- the four channel_* tables, both scheduled_* tables, agents, and (critically) alembic_version -- in `public`. On restart bootstrap treats the partially-populated target schema as unversioned, re-baselines it, and replays migrations while the real rows stay invisible in `public`. List the full owned set explicitly, call out alembic_version as required, and keep the "discover the rest" query for version-drift safety. * refactor(checkpointer): drop test-only _sync_checkpointer_from_database Address PR review: the helper was only reached by the env-gated integration test and re-implemented the DatabaseConfig->CheckpointerConfig backend resolution that _resolve_checkpointer_config already owns, so a future backend added there would silently miss this path. Mirror the store side of the same test, which reuses the production path directly: _resolve_checkpointer_config(...) + _sync_checkpointer_cm(...).
233 lines
9.1 KiB
Python
233 lines
9.1 KiB
Python
"""Sync checkpointer factory.
|
|
|
|
Provides a **sync singleton** and a **sync context manager** for LangGraph
|
|
graph compilation and CLI tools.
|
|
|
|
Supported backends: memory, sqlite, postgres.
|
|
|
|
Usage::
|
|
|
|
from deerflow.runtime.checkpointer.provider import get_checkpointer, checkpointer_context
|
|
|
|
# Singleton — reused across calls, closed on process exit
|
|
cp = get_checkpointer()
|
|
|
|
# One-shot — fresh connection, closed on block exit
|
|
with checkpointer_context() as cp:
|
|
graph.invoke(input, config={"configurable": {"thread_id": "1"}})
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import contextlib
|
|
import logging
|
|
import threading
|
|
from collections.abc import Iterator
|
|
|
|
from langgraph.types import Checkpointer
|
|
|
|
from deerflow.config.app_config import AppConfig, get_app_config
|
|
from deerflow.config.checkpointer_config import CheckpointerConfig, ensure_config_loaded, get_checkpointer_config
|
|
from deerflow.persistence.postgres_schema import dsn_with_search_path, ensure_postgres_schema
|
|
from deerflow.runtime.store._sqlite_utils import ensure_sqlite_parent_dir, resolve_sqlite_conn_str
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Error message constants — imported by aio.provider too
|
|
# ---------------------------------------------------------------------------
|
|
|
|
SQLITE_INSTALL = "langgraph-checkpoint-sqlite is required for the SQLite checkpointer. Install it with: uv add langgraph-checkpoint-sqlite"
|
|
POSTGRES_INSTALL = (
|
|
"langgraph-checkpoint-postgres is required for the PostgreSQL checkpointer. Install the package extra with: pip install 'deerflow-harness[postgres]' (or use: uv sync --all-packages --extra postgres when developing locally)"
|
|
)
|
|
POSTGRES_CONN_REQUIRED = "checkpointer.connection_string is required for the postgres backend"
|
|
|
|
|
|
def _ensure_postgres_schema(conn_string: str, schema: str) -> None:
|
|
"""Create the configured schema before LangGraph creates its tables."""
|
|
ensure_postgres_schema(conn_string, schema, install_hint=POSTGRES_INSTALL)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Config resolution
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def _resolve_checkpointer_config(app_config: AppConfig) -> CheckpointerConfig:
|
|
"""Resolve the checkpointer backend from legacy or unified application config.
|
|
|
|
The legacy ``checkpointer`` section remains authoritative when present so
|
|
Checkpointer and Store keep using the same backend. Otherwise the unified
|
|
``database`` section drives the checkpointer, matching the async
|
|
:func:`~deerflow.runtime.checkpointer.async_provider.make_checkpointer`
|
|
factory and the sync Store provider's ``_resolve_store_config``.
|
|
"""
|
|
if app_config.checkpointer is not None:
|
|
return app_config.checkpointer
|
|
|
|
database = app_config.database
|
|
if database is None or database.backend == "memory":
|
|
return CheckpointerConfig(type="memory")
|
|
if database.backend == "sqlite":
|
|
return CheckpointerConfig(type="sqlite", connection_string=database.checkpointer_sqlite_path)
|
|
if database.backend == "postgres":
|
|
if not database.postgres_url:
|
|
raise ValueError("database.postgres_url is required for the postgres backend")
|
|
return CheckpointerConfig(type="postgres", connection_string=database.postgres_url, postgres_schema=database.postgres_schema)
|
|
raise ValueError(f"Unknown database backend: {database.backend!r}")
|
|
|
|
|
|
def _get_checkpointer_config() -> CheckpointerConfig:
|
|
"""Load checkpointer config without holding the provider singleton lock."""
|
|
ensure_config_loaded()
|
|
|
|
# Preserve callers that initialise the legacy config singleton directly.
|
|
legacy_config = get_checkpointer_config()
|
|
if legacy_config is not None:
|
|
return legacy_config
|
|
try:
|
|
app_config = get_app_config()
|
|
except FileNotFoundError:
|
|
return CheckpointerConfig(type="memory")
|
|
return _resolve_checkpointer_config(app_config)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Sync factory
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@contextlib.contextmanager
|
|
def _sync_checkpointer_cm(config: CheckpointerConfig) -> Iterator[Checkpointer]:
|
|
"""Context manager that creates and tears down a sync checkpointer.
|
|
|
|
Returns a configured ``Checkpointer`` instance. Resource cleanup for any
|
|
underlying connections or pools is handled by higher-level helpers in
|
|
this module (such as the singleton factory or context manager); this
|
|
function does not return a separate cleanup callback.
|
|
"""
|
|
if config.type == "memory":
|
|
from langgraph.checkpoint.memory import InMemorySaver
|
|
|
|
logger.info("Checkpointer: using InMemorySaver (in-process, not persistent)")
|
|
yield InMemorySaver()
|
|
return
|
|
|
|
if config.type == "sqlite":
|
|
try:
|
|
from langgraph.checkpoint.sqlite import SqliteSaver
|
|
except ImportError as exc:
|
|
raise ImportError(SQLITE_INSTALL) from exc
|
|
|
|
conn_str = resolve_sqlite_conn_str(config.connection_string or "store.db")
|
|
ensure_sqlite_parent_dir(conn_str)
|
|
with SqliteSaver.from_conn_string(conn_str) as saver:
|
|
saver.setup()
|
|
logger.info("Checkpointer: using SqliteSaver (%s)", conn_str)
|
|
yield saver
|
|
return
|
|
|
|
if config.type == "postgres":
|
|
try:
|
|
from langgraph.checkpoint.postgres import PostgresSaver
|
|
except ImportError as exc:
|
|
raise ImportError(POSTGRES_INSTALL) from exc
|
|
|
|
if not config.connection_string:
|
|
raise ValueError(POSTGRES_CONN_REQUIRED)
|
|
|
|
_ensure_postgres_schema(config.connection_string, config.postgres_schema)
|
|
conn_string = dsn_with_search_path(config.connection_string, config.postgres_schema)
|
|
with PostgresSaver.from_conn_string(conn_string) as saver:
|
|
saver.setup()
|
|
logger.info("Checkpointer: using PostgresSaver")
|
|
yield saver
|
|
return
|
|
|
|
raise ValueError(f"Unknown checkpointer type: {config.type!r}")
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Sync singleton
|
|
# ---------------------------------------------------------------------------
|
|
|
|
_checkpointer: Checkpointer | None = None
|
|
_checkpointer_ctx = None # open context manager keeping the connection alive
|
|
_checkpointer_lock = threading.Lock()
|
|
|
|
|
|
def get_checkpointer() -> Checkpointer:
|
|
"""Return the global sync checkpointer singleton, creating it on first call.
|
|
|
|
The legacy ``checkpointer`` section takes precedence when configured;
|
|
otherwise the unified ``database`` section selects the backend. Returns an
|
|
``InMemorySaver`` when neither selects a persistent backend.
|
|
|
|
Raises:
|
|
ImportError: If the required package for the configured backend is not installed.
|
|
ValueError: If ``connection_string`` is missing for a backend that requires it.
|
|
"""
|
|
global _checkpointer, _checkpointer_ctx
|
|
|
|
if _checkpointer is not None:
|
|
return _checkpointer
|
|
|
|
# Config loading can reset both persistence singletons. Resolve the full
|
|
# config outside this provider lock to avoid cross-provider lock-order inversion.
|
|
config = _get_checkpointer_config()
|
|
|
|
with _checkpointer_lock:
|
|
if _checkpointer is not None:
|
|
return _checkpointer
|
|
|
|
checkpointer_ctx = _sync_checkpointer_cm(config)
|
|
checkpointer = checkpointer_ctx.__enter__()
|
|
_checkpointer_ctx = checkpointer_ctx
|
|
_checkpointer = checkpointer
|
|
|
|
return _checkpointer
|
|
|
|
|
|
def reset_checkpointer() -> None:
|
|
"""Reset the sync singleton, forcing recreation on the next call.
|
|
|
|
Closes any open backend connections and clears the cached instance.
|
|
Useful in tests or after a configuration change.
|
|
"""
|
|
global _checkpointer, _checkpointer_ctx
|
|
with _checkpointer_lock:
|
|
if _checkpointer_ctx is not None:
|
|
try:
|
|
_checkpointer_ctx.__exit__(None, None, None)
|
|
except Exception:
|
|
logger.warning("Error during checkpointer cleanup", exc_info=True)
|
|
_checkpointer_ctx = None
|
|
_checkpointer = None
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Sync context manager
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@contextlib.contextmanager
|
|
def checkpointer_context() -> Iterator[Checkpointer]:
|
|
"""Sync context manager that yields a checkpointer and cleans up on exit.
|
|
|
|
Unlike :func:`get_checkpointer`, this does **not** cache the instance —
|
|
each ``with`` block creates and destroys its own connection. Use it in
|
|
CLI scripts or tests where you want deterministic cleanup::
|
|
|
|
with checkpointer_context() as cp:
|
|
graph.invoke(input, config={"configurable": {"thread_id": "1"}})
|
|
|
|
The legacy ``checkpointer`` section takes precedence when configured;
|
|
otherwise the unified ``database`` section selects the backend. Yields an
|
|
``InMemorySaver`` when neither selects a persistent backend.
|
|
"""
|
|
|
|
config = _resolve_checkpointer_config(get_app_config())
|
|
with _sync_checkpointer_cm(config) as saver:
|
|
yield saver
|