mirror of
https://github.com/bytedance/deer-flow.git
synced 2026-08-12 15:59:04 +00:00
* feat(checkpoint-cache): delta-mode checkpoint history cache with recursive compose
Read-only, invalidation-free cache for LangGraph delta-channel history
({writes, seed}) at the get_delta_channel_history choke point:
- database.checkpoint_cache config (memory|redis; max_entries 0=disabled;
redis bounded by TTL, Gateway/async only)
- memory LRU backend (copy-on-read, zero-serde hit path) and redis backend
(lazy import, degrades to all-miss on outage)
- CachedHistorySaver: recursive composition from the nearest warm ancestor
(depth budget 8), caching each level; depth-0 cold chains delegate one
inner fast-path walk. Entries keyed by immutable
(db, thread, ns, checkpoint_id, channel) — no invalidation, coherent
across workers
- provider wiring: wraps in delta mode only (async + sync), full mode
untouched; sync path is memory-only
- bench opt-in: DEERFLOW_CHECKPOINT_BENCH_HISTORY_CACHE=1
sqlite bench (500 updates, payload 2KB): write phase 2.28x at f=250,
1.32x at f=10; one delegated walk per thread cold start.
* chore(config): bump config_version to 32 for database.checkpoint_cache
The checkpoint history cache feature added the database.checkpoint_cache
section to config.example.yaml; bump the schema version so existing
deployments get the outdated-config warning and can run make config-upgrade.
* chore(helm): bump config_version to 32 in chart values and README
* fix(checkpoint-cache): purge thread history entries on delete paths
Addresses review on #4638: delete_thread/prune removed source-of-truth
checkpoints but left the thread's materialized history payloads in the
cache (memory: until LRU eviction; redis: until TTL, default 1 day) — a
data-lifecycle gap for tenant offboarding / GDPR-style erasure.
- Cache contract gains thread-scoped adelete_thread/delete_thread
(lifecycle purge, not invalidation; entries remain immutable)
- Memory backend: stem scan over the LRU map; redis: SCAN MATCH + UNLINK,
outage degrades to TTL-bounded retention without raising
- CachedHistorySaver purges on delete_thread/adelete_thread and
prune/aprune (prune rewrites chains, so pre-prune histories must go);
delete_for_runs stays delegation-only (run->thread mapping unavailable,
no in-tree callers), documented in code
- ttl_seconds description documents the residual-retention window
- Tests: thread-scoped purge on both backends, saver-level delete/prune
purge, prefix-safety (t1 vs t10), redis outage degradation, and the
pinned no-purge behavior of delete_for_runs
* fix(checkpoint-cache): stable db identity, prefix-aware sync singleton, explicit zero TTL
Addresses Copilot review on #4638:
- checkpoint_cache_db_hash now hashes the credential-free postgres
identity (host:port/database + schema): credential rotation no longer
changes the cache namespace (cold cache + orphaned keys until TTL).
Unparseable URLs fall back to the raw string.
- The sync-path memory cache singleton is also keyed by its key_prefix:
a namespace change (db identity change or operator override) recreates
the cache instead of leaving stale-prefix entries unreachable and
unpurgeable.
- ttl_seconds=0 is now an explicit, documented opt-out of redis expiry
(SET without EX; redis maxmemory policy only) instead of a silent
'ttl_seconds or None' coercion.
Tests: credential-rotation hash stability, unparseable-URL fallback,
prefix-change singleton recreation, same-prefix singleton reuse, and
zero-TTL wire behavior (ex=None).
---------
Co-authored-by: Willem Jiang <willem.jiang@gmail.com>
80 lines
2.7 KiB
Python
80 lines
2.7 KiB
Python
"""Process-local LRU backend. Zero serialization on the hit path."""
|
|
|
|
from __future__ import annotations
|
|
|
|
from collections import OrderedDict
|
|
from typing import Any
|
|
|
|
from deerflow.runtime.checkpoint_cache.base import CheckpointCacheStats, thread_key_stem
|
|
|
|
|
|
def _copy_entry(entry: dict[str, Any]) -> dict[str, Any]:
|
|
"""Copy-on-read/write: fresh writes list; seed shared (never mutated in place)."""
|
|
copied: dict[str, Any] = {"writes": list(entry["writes"])}
|
|
if "seed" in entry:
|
|
copied["seed"] = entry["seed"]
|
|
return copied
|
|
|
|
|
|
class MemoryCheckpointHistoryCache:
|
|
def __init__(self, max_entries: int = 128) -> None:
|
|
if max_entries < 0:
|
|
raise ValueError("max_entries must be >= 0")
|
|
self._max_entries = max_entries
|
|
self._data: OrderedDict[str, dict[str, Any]] = OrderedDict()
|
|
self._hits = 0
|
|
self._misses = 0
|
|
self._evictions = 0
|
|
|
|
@property
|
|
def enabled(self) -> bool:
|
|
return self._max_entries > 0
|
|
|
|
def get_many(self, keys: list[str]) -> dict[str, dict[str, Any]]:
|
|
found: dict[str, dict[str, Any]] = {}
|
|
for key in keys:
|
|
entry = self._data.get(key)
|
|
if entry is None:
|
|
self._misses += 1
|
|
continue
|
|
self._data.move_to_end(key)
|
|
self._hits += 1
|
|
found[key] = _copy_entry(entry)
|
|
return found
|
|
|
|
def set_many(self, entries: dict[str, dict[str, Any]]) -> None:
|
|
if not self.enabled:
|
|
return
|
|
for key, entry in entries.items():
|
|
self._data[key] = _copy_entry(entry)
|
|
self._data.move_to_end(key)
|
|
while len(self._data) > self._max_entries:
|
|
self._data.popitem(last=False)
|
|
self._evictions += 1
|
|
|
|
async def aget_many(self, keys: list[str]) -> dict[str, dict[str, Any]]:
|
|
return self.get_many(keys)
|
|
|
|
async def aset_many(self, entries: dict[str, dict[str, Any]]) -> None:
|
|
self.set_many(entries)
|
|
|
|
def delete_thread(self, key_prefix: str, thread_id: str) -> None:
|
|
"""Purge every entry of one thread (lifecycle, not invalidation)."""
|
|
stem = thread_key_stem(key_prefix, thread_id)
|
|
for key in [k for k in self._data if k.startswith(stem)]:
|
|
del self._data[key]
|
|
|
|
async def adelete_thread(self, key_prefix: str, thread_id: str) -> None:
|
|
self.delete_thread(key_prefix, thread_id)
|
|
|
|
def stats(self) -> CheckpointCacheStats:
|
|
return CheckpointCacheStats(
|
|
hits=self._hits,
|
|
misses=self._misses,
|
|
evictions=self._evictions,
|
|
entries=len(self._data),
|
|
)
|
|
|
|
async def aclose(self) -> None:
|
|
self._data.clear()
|