deer-flow/backend/pyproject.toml
spud fe379c4486
feat(ci): split backend unit tests into parallel shards (#5137)
* feat(ci): split backend unit tests into parallel CI shards

Split the single offline backend `make test` job into four GitHub Actions
matrix shards (SPLITS=4, GROUP=1..4) via pytest-split, so the ~12k-test suite
runs in parallel instead of in one 15-minute job. Each shard runs on its own
runner with its own Postgres/Redis services; fail-fast: false lets a failing
shard report its owned tests without cancelling its peers.

`make test` stays the canonical full-suite entry point; CI now calls the new
`make test-shard SPLITS=4 GROUP=N`. tests/blocking_io remains owned solely by
the dedicated blocking-I/O workflow (excluded via --ignore), extending #5105.

Fixes #5088

* test(ci): make backend test shards duration-aware and pin the contract

Make `make test-shard` an explicit least_duration split that READS
backend/.test_durations (read-only for shards, so concurrent CI jobs never
race writes on it), and add `make test-shard-durations` to regenerate that
file from the full offline suite. Update the CI unit-test workflow contract to
call `make test-shard SPLITS=4 GROUP=<n>` and assert the shard command carries
--splits 4, --group 2, -m "not live", --ignore=tests/blocking_io and
--splitting-algorithm least_duration. Verified on the real 13,140-test normal
suite that the four shards are pairwise disjoint and their union equals the
unsplit suite.

Refs #5088

* test(ci): fail fast when the duration baseline is missing

`make test-shard` now requires backend/.test_durations and exits with a clear
error instead of letting pytest-split silently degrade to an even (count-based)
split. Harden the CI contract test to pin `--durations-path=.test_durations` and
to assert the repo ships the committed duration baseline.

Refs #5088

* docs: trim backend/AGENTS.md within guidance budget

* test(ci): add backend test duration baseline

Add the duration baseline generated by a full offline backend run on a
GitHub-hosted ubuntu-latest runner (the same runner type the shards use), so
`make test-shard` balances the four matrix shards by real wall-clock cost.

Refs #5088

* test(ci): make the duration writer honor DURATIONS_FILE

`test-shard-durations` now writes `--durations-path=$(DURATIONS_FILE)` instead of a
hard-coded .test_durations, so the reader and writer stay consistent when the
path is overridden.

Refs #5088

* test(ci): address sharding review feedback

* test: isolate subagent execution capacity state

---------

Co-authored-by: Willem Jiang <willem.jiang@gmail.com>
2026-09-02 11:54:40 +08:00

113 lines
5.0 KiB
TOML

[project]
name = "deer-flow"
version = "2.1.0"
description = "LangGraph-based AI agent system with sandbox execution capabilities"
readme = "README.md"
requires-python = ">=3.12"
dependencies = [
"deerflow-harness",
# Direct dependency on purpose, even though deerflow-harness already pulls
# it in: app/gateway/app.py and app/gateway/services.py import the public
# contract package themselves (the extension principal resolver, and the
# provenance key set the state route strips). Those are the app layer's own
# imports of a package whose whole point is a stable public surface, so the
# app declares them rather than relying on the harness to keep supplying it.
"deerflow-extension-api",
"fastapi>=0.115.0",
"httpx>=0.28.0",
"python-multipart>=0.0.31",
"sse-starlette>=2.1.0",
# Direct dependency on purpose, even though FastAPI already pulls it in:
# app/gateway/request_path.py imports the private
# `starlette._utils.get_route_path` so the auth and CSRF predicates
# classify the exact string Starlette's router matches on. A private
# import is the safest option here because it fails loudly (ImportError at
# startup) instead of silently drifting from the dispatcher at a security
# boundary -- but it does mean a Starlette bump is a security-relevant
# change. Declaring and bounding it here makes that bump visible in the
# diff; tests/test_gateway_request_path.py pins the agreement itself.
"starlette>=1.3.1,<2",
"uvicorn[standard]>=0.34.0",
"lark-oapi>=1.4.0",
"slack-sdk>=3.33.0",
"python-telegram-bot>=21.0",
"langgraph-sdk>=0.1.51",
"markdown-to-mrkdwn>=0.3.1",
"wecom-aibot-python-sdk>=0.1.6",
"dingtalk-stream>=0.24.3",
"bcrypt>=4.0.0",
"pyjwt>=2.13.0",
"email-validator>=2.0.0",
"e2b-code-interpreter>=2.8.1",
]
[project.optional-dependencies]
postgres = ["deerflow-harness[postgres]"]
redis = ["deerflow-harness[redis]"]
discord = ["discord.py>=2.7.0"]
buzz = ["coincurve>=20.0.0"]
monocle = ["deerflow-harness[monocle]"]
browser = ["deerflow-harness[browser]"]
memory-zh = ["deerflow-harness[memory-zh]"]
[dependency-groups]
# Managed extension packages are added here by `deerflow extensions install`.
# Keeping them separate from development tooling lets every startup mode sync
# the same locked runtime set without promoting those packages to core deps.
extensions = []
dev = [
"blockbuster>=1.5.26,<1.6",
"hypothesis>=6.100,<7",
"jsonschema>=4.26.0",
"prompt-toolkit>=3.0.0",
"pytest>=9.0.3",
"pytest-asyncio>=1.3.0",
"pytest-split>=0.11.0",
"ruff>=0.14.11",
# Monocle tracer (also the deerflow-harness[monocle] extra); kept in the dev
# group so the tracing tests can import it without forcing it onto installs.
"monocle_apptrace>=0.8.8",
# redis is an optional runtime extra (deerflow-harness[redis]); pin it in the
# dev group so the stream-bridge tests can always import/exercise the redis
# bridge without forcing it onto production installs.
"redis>=5.0.0",
# TUI runtime dep (also declared as the deerflow-harness[tui] extra); kept in
# the dev group so the terminal workbench can be run and tested locally / in CI.
"textual>=0.80",
]
[tool.pytest.ini_options]
markers = [
"no_auto_user: disable the conftest autouse contextvar fixture for this test",
"allow_blocking_io: opt out of the strict Blockbuster gate in tests/blocking_io/",
"integration: tests that require an external service (e.g. Redis); skipped when unavailable",
"live: tests that call real external APIs and require explicit opt-in",
]
[tool.uv]
index-url = "https://pypi.org/simple"
default-groups = ["dev", "extensions"]
# langgraph-sdk 0.4.2 (pulled in by langgraph 1.2.9 for DeltaChannel) pins
# `websockets<16,>=14`, silently downgrading websockets 16.0 -> 15.0.1. The
# pin is not grounded in any API incompatibility: websockets 16's only
# breaking change is requiring Python >=3.10 (we require >=3.12), the sdk
# only imports `websockets.asyncio.client`/`websockets.exceptions` (both
# 16-compatible), and DeerFlow never uses the sdk's WebSocket transport
# (httpx/SSE only). DeerFlow does have one direct consumer of its own: the Buzz
# channel (`app/channels/buzz.py`) imports the top-level `websockets` package and
# calls `websockets.connect()`, which in 16.0 is the same asyncio client the sdk
# uses, re-exported at the package root -- so it is covered by the same
# compatibility argument. Pin the exact pre-upgrade 16.0 for that plus the IM
# channel integrations (dingtalk-stream, python-telegram-bot, etc.) that ran on
# it before. Remove once langgraph-sdk relaxes the pin upstream. Note: enabling
# the `openai[realtime]` or `slack-sdk[optional]` extras would conflict (they
# also cap websockets<16).
override-dependencies = ["websockets==16.0"]
[tool.uv.workspace]
members = ["packages/harness", "packages/extension-api"]
[tool.uv.sources]
deerflow-harness = { workspace = true }
deerflow-extension-api = { workspace = true }