mirror of
https://github.com/bytedance/deer-flow.git
synced 2026-08-14 08:49:00 +00:00
* feat(extensions): add middleware plugin foundation * fix(extensions): stop config resolution from masking extension loading `create_app()` resolved the configured plugin list inside the fail-open guard around `load_extensions()`. CI has no `config.yaml` (gitignored and never generated by the workflow), so `get_app_config()` raised `FileNotFoundError` there and was swallowed as an extension failure -- `load_extensions()` never ran at all, and the four `create_app()` tests in `test_extension_app_loading.py` passed locally but failed on every runner. Resolve the plugin list before the guard. Only an absent `config.yaml` is tolerated, mirroring `_resolve_trace_enabled_for_app_construction()`: `create_app()` runs at import time, and lifespan still performs strict config loading before serving. A `config.yaml` that exists but fails to parse or validate now propagates instead of being reported as an extension failure -- reporting it as the latter silently dropped a `required: true` extension rather than failing the boot. Make the tests config-independent with an autouse `stub_app_config` fixture, following the existing pattern in `test_gateway_lifespan_shutdown.py`, and cover both new branches of the config-resolution boundary. * fix(extensions): bind the run's extension snapshot through subagent delegation The lead-agent path resolves one immutable loaded-extension snapshot per run and binds it through task-store allocation and graph construction, but the subagent path re-read the process-wide singleton at execution time. In production both are the same object, yet a `set_loaded_extensions()` between the lead run's start and a subagent's execution (test teardown, a future hot-reload path) would let one run mix two extension generations — exactly what the documented invariant exists to prevent. The graph-build binding is a ContextVar scoped to synchronous construction, so it has already exited by the time a tool delegates; the snapshot has to travel through runtime context instead. The run worker publishes it under the host-internal `EXTENSION_SNAPSHOT_CONTEXT_KEY` (written after the caller merge, popped when the run has none, so a caller-supplied value is never authoritative), `task_tool` reads it back through the type-checking `resolve_run_extensions()`, and `SubagentExecutor` binds it at construction. Callers outside the Gateway run path — embedded `DeerFlowClient`, standalone LangGraph Server — install no snapshot and keep the existing `get_loaded_extensions()` fallback. * refactor(extensions): defer the ordering table by call, not by a lying tuple `CORE_ORDERING_CONSTRAINTS` was a `tuple` subclass that overrode only `__iter__` and resolved into a class-level `_resolved` side channel. A tuple cannot populate its own storage after construction, so the instance stayed the empty tuple it was built as: `len()` was 0, `bool()` was False, `in` was always False, indexing raised, slicing and `reversed()` came back empty, and it compared unequal to the plain tuples tests substitute for it — all while iteration yielded the real constraints. Only `assert_ordering` consumed it, and only by iterating, so the split went unnoticed. The sibling `_AnchorTable(dict)` uses the same idea soundly because dict is mutable: `self.update()` fills the real storage, making every inherited operation correct. That trick does not survive the port to an immutable type. Replace it with `core_ordering_constraints()`, matching how `stack.py` defers the same kind of table via `_anchors()`. The deferral is kept — it is about dependency direction, not just cycles: `extensions/` is the layer the middleware layer calls into, so a module-scope `agents.middlewares` import here points the dependency backwards and closes a cycle as soon as any middleware imports something under `extensions/` at module level. Resolution stays at `assert_ordering` time, which already runs inside the middleware builder. Tests pin both halves: the returned value is a plain tuple whose len/bool/ membership/indexing/reversal/equality agree with iteration, and a subprocess probe asserts importing `extensions.ordering` does not load the middleware layer while calling the function does.
143 lines
4.8 KiB
Python
143 lines
4.8 KiB
Python
"""Tests for extension configuration and the process-wide singleton."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import pytest
|
|
from pydantic import ValidationError
|
|
|
|
from deerflow.config.app_config import AppConfig
|
|
from deerflow.config.reload_boundary import STARTUP_ONLY_FIELDS
|
|
from deerflow.extensions import (
|
|
EMPTY_EXTENSIONS,
|
|
ExtensionRegistry,
|
|
get_loaded_extensions,
|
|
reset_loaded_extensions,
|
|
set_loaded_extensions,
|
|
)
|
|
|
|
|
|
@pytest.fixture(autouse=True)
|
|
def _reset_singleton():
|
|
reset_loaded_extensions()
|
|
yield
|
|
reset_loaded_extensions()
|
|
|
|
|
|
class _Marker:
|
|
"""Sentinel written into an app_store to detect state leaking across resets."""
|
|
|
|
def __init__(self, tag: str = "") -> None:
|
|
self.tag = tag
|
|
|
|
|
|
# AppConfig.sandbox has no default (see app_config.py's
|
|
# `_drop_null_config_sections`: "Required sections without a default
|
|
# (sandbox) intentionally still error when null"), so every AppConfig
|
|
# construction below supplies it, matching the pattern already used in
|
|
# test_app_config_reload.py.
|
|
_SANDBOX = {"sandbox": {"use": "deerflow.sandbox.local:LocalSandboxProvider"}}
|
|
|
|
|
|
def test_app_config_defaults_to_no_plugins():
|
|
assert AppConfig.model_validate(_SANDBOX).plugins == []
|
|
|
|
|
|
def test_app_config_parses_plugin_entries():
|
|
config = AppConfig.model_validate(
|
|
{
|
|
**_SANDBOX,
|
|
"plugins": [
|
|
{"use": "acme_observability:install", "config": {"enabled": True}},
|
|
{"use": "acme_policy:install", "required": True},
|
|
],
|
|
}
|
|
)
|
|
assert [e.use for e in config.plugins] == ["acme_observability:install", "acme_policy:install"]
|
|
assert config.plugins[0].config == {"enabled": True}
|
|
assert config.plugins[0].required is False
|
|
assert config.plugins[1].required is True
|
|
|
|
|
|
def test_plugin_entries_reject_unknown_fields_instead_of_weakening_required():
|
|
with pytest.raises(ValidationError, match="require"):
|
|
AppConfig.model_validate(
|
|
{
|
|
**_SANDBOX,
|
|
"plugins": [
|
|
{
|
|
"use": "acme_policy:install",
|
|
"require": True,
|
|
}
|
|
],
|
|
}
|
|
)
|
|
|
|
|
|
def test_new_field_does_not_disturb_the_existing_extensions_field():
|
|
"""AppConfig.extensions is a pre-existing, unrelated field (MCP servers,
|
|
skills, config-declared middlewares) backed by extensions_config.json.
|
|
The plugin list is deliberately a separate top-level key: that file is
|
|
writable through an HTTP endpoint, and a code-loading list must not be."""
|
|
config = AppConfig.model_validate({**_SANDBOX, "plugins": [{"use": "a:install"}]})
|
|
assert config.plugins[0].use == "a:install"
|
|
assert hasattr(config.extensions, "mcp_servers")
|
|
assert hasattr(config.extensions, "middlewares")
|
|
|
|
|
|
def test_plugins_is_registered_as_startup_only():
|
|
"""Plugins load once in create_app(); a config.yaml edit needs a restart.
|
|
Registering here is what surfaces that to operators."""
|
|
assert "plugins" in STARTUP_ONLY_FIELDS
|
|
assert "restart" in STARTUP_ONLY_FIELDS["plugins"].lower()
|
|
|
|
|
|
def test_singleton_defaults_to_empty():
|
|
loaded = get_loaded_extensions()
|
|
assert loaded.has_middleware_contributors is False
|
|
assert loaded.needs_task_store is False
|
|
|
|
|
|
def test_singleton_roundtrips():
|
|
loaded = ExtensionRegistry().build()
|
|
set_loaded_extensions(loaded)
|
|
assert get_loaded_extensions() is loaded
|
|
|
|
|
|
def test_reset_gives_a_fresh_instance():
|
|
"""Reset must not hand back a shared object. EMPTY_EXTENSIONS owns a
|
|
mutable app_store, so resetting to it would carry writes forward."""
|
|
populated = ExtensionRegistry().build()
|
|
set_loaded_extensions(populated)
|
|
reset_loaded_extensions()
|
|
after = get_loaded_extensions()
|
|
assert after is not populated
|
|
assert after is not EMPTY_EXTENSIONS
|
|
assert after.has_middleware_contributors is False
|
|
|
|
|
|
def test_reset_does_not_leak_app_store_writes():
|
|
"""The regression the fresh-build reset exists to prevent."""
|
|
reset_loaded_extensions()
|
|
get_loaded_extensions().app_store.set(_Marker("dirty"))
|
|
reset_loaded_extensions()
|
|
assert get_loaded_extensions().app_store.get(_Marker) is None
|
|
|
|
|
|
def test_runtime_diagnostics_are_bounded_without_replacing_the_live_list(monkeypatch):
|
|
import deerflow.extensions as extensions_module
|
|
|
|
monkeypatch.setattr(extensions_module, "_MAX_RUNTIME_DIAGNOSTICS", 3)
|
|
extensions_module.reset_runtime_diagnostics()
|
|
live = extensions_module.initialize_runtime_diagnostics([])
|
|
try:
|
|
for index in range(5):
|
|
extensions_module.record_runtime_diagnostic(extensions_module.Diagnostic.error("demo:install", f"error-{index}"))
|
|
|
|
assert [diagnostic.message for diagnostic in live] == [
|
|
"error-2",
|
|
"error-3",
|
|
"error-4",
|
|
]
|
|
finally:
|
|
extensions_module.reset_runtime_diagnostics()
|