mirror of
https://github.com/bytedance/deer-flow.git
synced 2026-08-15 17:28:40 +00:00
* feat(agents): database-backed storage for custom agent definitions Add an agent_storage.backend switch (default file, behaviour-unchanged) with a db backend that stores each custom agent as a row in the shared SQL persistence layer, so a multi-instance deployment sees the same agents on every node (#4331, #4357). Introduces an AgentStore interface routing all read/write surfaces, an agents table + migration 0006, startup validation, and a file->db importer. Follows the thread_meta store / run_events backend-switch / 0003_scheduled_tasks migration patterns; no new dependency. * fix(agents): make db storage path production-ready (review round 1) Addresses review feedback on the db/sync agent-storage path: - sql.py: mirror the async engine's per-connection SQLite PRAGMAs on the sync engine (busy_timeout=30000, synchronous=NORMAL, foreign_keys=ON, WAL) so both engines behave identically against the shared DB; guard the engine cache with a lock (double-checked) so concurrent first-touch cannot build duplicate engines or register the connect listener twice. - routers/agents.py + routers/assistants_compat.py: offload the sync-store reads that ran on the event loop (list/get/check, update's pre-read + legacy guard + refresh, and assistants_compat's four list routes) via asyncio.to_thread — on db+postgres each was a network round trip stalling the loop. Writes were already offloaded. - file.py: translate the create() mkdir(exist_ok=False) race FileExistsError into AgentExistsError (router 409, matching SqlAgentStore's IntegrityError path); correct the _write docstring — per-file atomic replace, two commits sequential not transactional. Tests: sync-engine PRAGMA + engine-cache reuse assertions; file create-race -> AgentExistsError; strict Blockbuster anchor over the read endpoints so a regression back onto the loop fails CI. * fix(agents): address round-2 review on the db store path - update_agent tool: align the docstring/inline comment with FileAgentStore._write. Cross-field write atomicity is db-only; the file backend commits config then soul via two sequential os.replace (a crash between them can leave a fresh config.yaml beside a stale SOUL.md). The dropped partial-write *reporting* is an intentional tradeoff — the stage-then-replace safety is preserved (test_update_agent_soul_failure_does_not_replace_config still holds). - SqlAgentStore.update(): true upsert. Catch IntegrityError on the insert-on-missing branch, re-fetch and apply, so two concurrent first-time writes (e.g. two setup_agent handshakes) converge instead of surfacing a raw UNIQUE(user_id, name) violation as a 500. Symmetric with create(). - get_agent_store(): document the graph-subprocess config-resolution invariant (the except->file fallback is a genuine no-config path, not a mask for a misconfigured graph process) and pin it with two tests driving the real get_app_config() file resolution: db resolves from an on-disk config.yaml, file fallback when config is unresolvable. * test(agents): cover SqlAgentStore.update() write-race upsert recovery Mandatory-TDD test for the round-2 fix in 0680340a: two concurrent first-time update()s where the loser's insert hits UNIQUE(user_id, name). Deterministically forces the IntegrityError recovery path by making the first _row probe miss the committed winner, and asserts last-writer-wins instead of a surfaced 500.
152 lines
6.1 KiB
Python
152 lines
6.1 KiB
Python
"""Unified database backend configuration.
|
|
|
|
Controls BOTH the LangGraph checkpointer and the DeerFlow application
|
|
persistence layer (runs, threads metadata, users, etc.). The user
|
|
configures one backend; the system handles physical separation details.
|
|
|
|
SQLite mode: checkpointer and app share a single .db file
|
|
({sqlite_dir}/deerflow.db) with WAL journal mode enabled on every
|
|
connection. WAL allows concurrent readers and a single writer without
|
|
blocking, making a unified file safe for both workloads. Writers
|
|
that contend for the lock wait via the default 5-second sqlite3
|
|
busy timeout rather than failing immediately.
|
|
|
|
Postgres mode: both use the same database URL but maintain independent
|
|
connection pools with different lifecycles.
|
|
|
|
Memory mode: checkpointer uses MemorySaver, app uses in-memory stores.
|
|
No database is initialized.
|
|
|
|
Sensitive values (postgres_url) should use $VAR syntax in config.yaml
|
|
to reference environment variables from .env:
|
|
|
|
database:
|
|
backend: postgres
|
|
postgres_url: $DATABASE_URL
|
|
|
|
The $VAR resolution is handled by AppConfig.resolve_env_variables()
|
|
before this config is instantiated -- DatabaseConfig itself does not
|
|
need to do any environment variable processing.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import os
|
|
from typing import Literal
|
|
|
|
from pydantic import BaseModel, Field
|
|
|
|
CheckpointChannelMode = Literal["full", "delta"]
|
|
|
|
|
|
class DatabaseConfig(BaseModel):
|
|
backend: Literal["memory", "sqlite", "postgres"] = Field(
|
|
default="memory",
|
|
description=("Storage backend for both checkpointer and application data. 'memory' for development (no persistence across restarts), 'sqlite' for single-node deployment, 'postgres' for production multi-node deployment."),
|
|
)
|
|
checkpoint_channel_mode: CheckpointChannelMode = Field(
|
|
default="full",
|
|
description=(
|
|
"Checkpoint representation for accumulating channels. "
|
|
"'full' preserves full-value message checkpoints; 'delta' uses "
|
|
"LangGraph DeltaChannel for messages. Restart is required, and all "
|
|
"processes sharing one checkpoint database must use the same value."
|
|
),
|
|
)
|
|
sqlite_dir: str = Field(
|
|
default=".deer-flow/data",
|
|
description=("Directory for the SQLite database file. Both checkpointer and application data share {sqlite_dir}/deerflow.db."),
|
|
)
|
|
postgres_url: str = Field(
|
|
default="",
|
|
description=(
|
|
"PostgreSQL connection URL, shared by checkpointer and app. "
|
|
"Use $DATABASE_URL in config.yaml to reference .env. "
|
|
"Example: postgresql://user:pass@host:5432/deerflow "
|
|
"(the +asyncpg driver suffix is added automatically where needed)."
|
|
),
|
|
)
|
|
echo_sql: bool = Field(
|
|
default=False,
|
|
description="Echo all SQL statements to log (debug only).",
|
|
)
|
|
pool_size: int = Field(
|
|
default=5,
|
|
description="Connection pool size for the app ORM engine (postgres only).",
|
|
)
|
|
pool_recycle: int = Field(
|
|
default=300,
|
|
gt=0,
|
|
description="Seconds before app ORM PostgreSQL connections are recycled.",
|
|
)
|
|
command_timeout: float | None = Field(
|
|
default=30,
|
|
gt=0,
|
|
description="Timeout in seconds for app ORM PostgreSQL commands. Set to null to disable the command timeout.",
|
|
)
|
|
|
|
# -- Derived helpers (not user-configured) --
|
|
|
|
@property
|
|
def _resolved_sqlite_dir(self) -> str:
|
|
"""Resolve sqlite_dir to an absolute path (relative to CWD)."""
|
|
from pathlib import Path
|
|
|
|
return str(Path(self.sqlite_dir).resolve())
|
|
|
|
@property
|
|
def sqlite_path(self) -> str:
|
|
"""Unified SQLite file path shared by checkpointer and app."""
|
|
return os.path.join(self._resolved_sqlite_dir, "deerflow.db")
|
|
|
|
# Backward-compatible aliases
|
|
@property
|
|
def checkpointer_sqlite_path(self) -> str:
|
|
"""SQLite file path for the LangGraph checkpointer (alias for sqlite_path)."""
|
|
return self.sqlite_path
|
|
|
|
@property
|
|
def app_sqlite_path(self) -> str:
|
|
"""SQLite file path for application ORM data (alias for sqlite_path)."""
|
|
return self.sqlite_path
|
|
|
|
@property
|
|
def app_sqlalchemy_url(self) -> str:
|
|
"""SQLAlchemy async URL for the application ORM engine."""
|
|
if self.backend == "sqlite":
|
|
return f"sqlite+aiosqlite:///{self.sqlite_path}"
|
|
if self.backend == "postgres":
|
|
url = self.postgres_url
|
|
if url.startswith("postgresql://"):
|
|
url = url.replace("postgresql://", "postgresql+asyncpg://", 1)
|
|
elif url.startswith("postgres://"):
|
|
# libpq's short alias: accepted by the psycopg checkpointer, but not a SQLAlchemy dialect.
|
|
url = url.replace("postgres://", "postgresql+asyncpg://", 1)
|
|
return url
|
|
raise ValueError(f"No SQLAlchemy URL for backend={self.backend!r}")
|
|
|
|
@property
|
|
def app_sync_sqlalchemy_url(self) -> str:
|
|
"""SQLAlchemy *synchronous* URL for the application ORM data.
|
|
|
|
Used by the ``agent_storage.backend: db`` store, whose consumers (the
|
|
LangGraph graph factory, the setup/update tools) are synchronous and may
|
|
run on the event loop or in a separate process from the gateway, where an
|
|
async engine cannot be driven. Points at the same database file/server as
|
|
:meth:`app_sqlalchemy_url`; only the driver differs (both drivers —
|
|
stdlib sqlite3 and psycopg — ship with the app, so this adds no
|
|
dependency).
|
|
"""
|
|
if self.backend == "sqlite":
|
|
return f"sqlite:///{self.sqlite_path}"
|
|
if self.backend == "postgres":
|
|
url = self.postgres_url
|
|
if url.startswith("postgresql+asyncpg://"):
|
|
url = url.replace("postgresql+asyncpg://", "postgresql+psycopg://", 1)
|
|
elif url.startswith("postgresql://"):
|
|
url = url.replace("postgresql://", "postgresql+psycopg://", 1)
|
|
elif url.startswith("postgres://"):
|
|
url = url.replace("postgres://", "postgresql+psycopg://", 1)
|
|
return url
|
|
raise ValueError(f"No SQLAlchemy URL for backend={self.backend!r}")
|