mirror of
https://github.com/bytedance/deer-flow.git
synced 2026-08-08 05:48:53 +00:00
* feat(extensions): add middleware plugin foundation * fix(extensions): stop config resolution from masking extension loading `create_app()` resolved the configured plugin list inside the fail-open guard around `load_extensions()`. CI has no `config.yaml` (gitignored and never generated by the workflow), so `get_app_config()` raised `FileNotFoundError` there and was swallowed as an extension failure -- `load_extensions()` never ran at all, and the four `create_app()` tests in `test_extension_app_loading.py` passed locally but failed on every runner. Resolve the plugin list before the guard. Only an absent `config.yaml` is tolerated, mirroring `_resolve_trace_enabled_for_app_construction()`: `create_app()` runs at import time, and lifespan still performs strict config loading before serving. A `config.yaml` that exists but fails to parse or validate now propagates instead of being reported as an extension failure -- reporting it as the latter silently dropped a `required: true` extension rather than failing the boot. Make the tests config-independent with an autouse `stub_app_config` fixture, following the existing pattern in `test_gateway_lifespan_shutdown.py`, and cover both new branches of the config-resolution boundary. * fix(extensions): bind the run's extension snapshot through subagent delegation The lead-agent path resolves one immutable loaded-extension snapshot per run and binds it through task-store allocation and graph construction, but the subagent path re-read the process-wide singleton at execution time. In production both are the same object, yet a `set_loaded_extensions()` between the lead run's start and a subagent's execution (test teardown, a future hot-reload path) would let one run mix two extension generations — exactly what the documented invariant exists to prevent. The graph-build binding is a ContextVar scoped to synchronous construction, so it has already exited by the time a tool delegates; the snapshot has to travel through runtime context instead. The run worker publishes it under the host-internal `EXTENSION_SNAPSHOT_CONTEXT_KEY` (written after the caller merge, popped when the run has none, so a caller-supplied value is never authoritative), `task_tool` reads it back through the type-checking `resolve_run_extensions()`, and `SubagentExecutor` binds it at construction. Callers outside the Gateway run path — embedded `DeerFlowClient`, standalone LangGraph Server — install no snapshot and keep the existing `get_loaded_extensions()` fallback. * refactor(extensions): defer the ordering table by call, not by a lying tuple `CORE_ORDERING_CONSTRAINTS` was a `tuple` subclass that overrode only `__iter__` and resolved into a class-level `_resolved` side channel. A tuple cannot populate its own storage after construction, so the instance stayed the empty tuple it was built as: `len()` was 0, `bool()` was False, `in` was always False, indexing raised, slicing and `reversed()` came back empty, and it compared unequal to the plain tuples tests substitute for it — all while iteration yielded the real constraints. Only `assert_ordering` consumed it, and only by iterating, so the split went unnoticed. The sibling `_AnchorTable(dict)` uses the same idea soundly because dict is mutable: `self.update()` fills the real storage, making every inherited operation correct. That trick does not survive the port to an immutable type. Replace it with `core_ordering_constraints()`, matching how `stack.py` defers the same kind of table via `_anchors()`. The deferral is kept — it is about dependency direction, not just cycles: `extensions/` is the layer the middleware layer calls into, so a module-scope `agents.middlewares` import here points the dependency backwards and closes a cycle as soon as any middleware imports something under `extensions/` at module level. Resolution stays at `assert_ordering` time, which already runs inside the middleware builder. Tests pin both halves: the returned value is a plain tuple whose len/bool/ membership/indexing/reversal/equality agree with iteration, and a subprocess probe asserts importing `extensions.ordering` does not load the middleware layer while calling the function does.
114 lines
3.2 KiB
Python
114 lines
3.2 KiB
Python
"""Tests for ExtensionData, the per-scope typed store handed to extensions."""
|
|
|
|
from __future__ import annotations
|
|
|
|
from dataclasses import dataclass
|
|
from threading import Thread
|
|
|
|
from deerflow_extension_api import ExtensionData
|
|
|
|
|
|
@dataclass
|
|
class _Counter:
|
|
value: int = 0
|
|
|
|
|
|
@dataclass
|
|
class _Other:
|
|
name: str = ""
|
|
|
|
|
|
def test_get_returns_none_when_absent():
|
|
store = ExtensionData("task-1")
|
|
assert store.get(_Counter) is None
|
|
|
|
|
|
def test_set_then_get_roundtrips():
|
|
store = ExtensionData("task-1")
|
|
store.set(_Counter(value=7))
|
|
got = store.get(_Counter)
|
|
assert got is not None
|
|
assert got.value == 7
|
|
|
|
|
|
def test_get_or_init_creates_once():
|
|
store = ExtensionData("task-1")
|
|
calls = []
|
|
|
|
def _init() -> _Counter:
|
|
calls.append(1)
|
|
return _Counter(value=1)
|
|
|
|
first = store.get_or_init(_Counter, _init)
|
|
second = store.get_or_init(_Counter, _init)
|
|
assert first is second
|
|
assert calls == [1]
|
|
|
|
|
|
def test_get_or_init_allows_initializer_to_use_the_same_store():
|
|
"""Extension initializers may compose other extension-local state."""
|
|
store = ExtensionData("task-1")
|
|
completed: list[_Counter] = []
|
|
|
|
def _init_counter() -> _Counter:
|
|
store.set(_Other(name="nested"))
|
|
return _Counter(value=2)
|
|
|
|
def _initialize() -> None:
|
|
completed.append(store.get_or_init(_Counter, _init_counter))
|
|
|
|
thread = Thread(target=_initialize, daemon=True)
|
|
thread.start()
|
|
thread.join(timeout=0.5)
|
|
|
|
assert not thread.is_alive(), "nested store access deadlocked"
|
|
assert completed == [_Counter(value=2)]
|
|
assert store.get(_Other) == _Other(name="nested")
|
|
|
|
|
|
def test_types_are_isolated():
|
|
store = ExtensionData("task-1")
|
|
store.set(_Counter(value=1))
|
|
store.set(_Other(name="x"))
|
|
assert store.get(_Counter).value == 1
|
|
assert store.get(_Other).name == "x"
|
|
|
|
|
|
def test_remove_returns_and_clears():
|
|
store = ExtensionData("task-1")
|
|
store.set(_Counter(value=3))
|
|
removed = store.remove(_Counter)
|
|
assert removed.value == 3
|
|
assert store.get(_Counter) is None
|
|
|
|
|
|
def test_scope_id_is_exposed():
|
|
store = ExtensionData("run-42")
|
|
assert store.scope_id == "run-42"
|
|
|
|
|
|
def test_stores_are_independent():
|
|
a = ExtensionData("task-a")
|
|
b = ExtensionData("task-b")
|
|
a.set(_Counter(value=1))
|
|
assert b.get(_Counter) is None
|
|
|
|
|
|
def test_api_package_does_not_import_deerflow():
|
|
"""The API package must stay independent of the host so extensions can
|
|
depend on it alone. A `deerflow` import here would silently couple every
|
|
extension to the harness release cadence."""
|
|
import pathlib
|
|
|
|
import deerflow_extension_api
|
|
|
|
root = pathlib.Path(deerflow_extension_api.__file__).parent
|
|
offenders = []
|
|
for path in root.rglob("*.py"):
|
|
text = path.read_text(encoding="utf-8")
|
|
for lineno, line in enumerate(text.splitlines(), start=1):
|
|
stripped = line.strip()
|
|
if stripped.startswith(("import deerflow", "from deerflow")) and not stripped.startswith(("import deerflow_extension_api", "from deerflow_extension_api")):
|
|
offenders.append(f"{path.name}:{lineno}: {stripped}")
|
|
assert offenders == [], "deerflow-extension-api must not import deerflow: " + "; ".join(offenders)
|