mirror of
https://github.com/bytedance/deer-flow.git
synced 2026-09-15 00:19:14 +00:00
* feat: add opt-in task notes and compacted history recall * fix: validate task continuity state and preserve user answers Honor explicit opt-out, preserve clarification replies and capture failure statuses, validate notebook writes, and clear branch archive references. Update the config version and audit optional LLM credentials, with regression and integration evidence. * fix: align Helm config version with task continuity schema * fix: preserve mixed task history and declare continuity policies * fix: recover malformed history and evict archives atomically
149 lines
8.6 KiB
JSON
149 lines
8.6 KiB
JSON
{
|
|
"base_commit": "4501c76b0f44cc55af6332d65ac2e7f5311f71fd",
|
|
"reviewed_commit": "aee9a537be53a411632dc5130c29ba0ef96f86ae",
|
|
"format": "passed",
|
|
"lint": "passed",
|
|
"focused": {
|
|
"passed": 370,
|
|
"seconds": 9.56
|
|
},
|
|
"full_branch": {
|
|
"passed": 15474,
|
|
"failed": 15,
|
|
"skipped": 182,
|
|
"deselected": 3,
|
|
"seconds": 394.61
|
|
},
|
|
"full_base": {
|
|
"passed": 15427,
|
|
"failed": 15,
|
|
"skipped": 182,
|
|
"deselected": 3,
|
|
"seconds": 467.39
|
|
},
|
|
"branch_only_failures": [],
|
|
"base_only_failures": [],
|
|
"shared_failure_ids": [
|
|
"tests/test_browser_automation.py::TestBrowserTools::test_navigate_emits_screenshot_artifact_and_browser_view",
|
|
"tests/test_browser_automation.py::TestBrowserTools::test_navigate_returns_snapshot",
|
|
"tests/test_browser_automation.py::TestBrowserTools::test_navigate_screenshot_failure_does_not_break_action",
|
|
"tests/test_browser_router.py::test_validate_browser_url_rejects_private_and_non_http",
|
|
"tests/test_browserless_client.py::TestBrowserlessTools::test_web_fetch_and_web_capture_tools_agree_on_target_error_warning",
|
|
"tests/test_browserless_client.py::TestBrowserlessTools::test_web_fetch_tool_no_warning_for_normal_target_status",
|
|
"tests/test_browserless_client.py::TestBrowserlessTools::test_web_fetch_tool_success",
|
|
"tests/test_browserless_client.py::TestBrowserlessTools::test_web_fetch_tool_warns_on_target_error_status",
|
|
"tests/test_crawl4ai_tools.py::TestCrawl4AiTools::test_web_fetch_tool_invalid_filter_falls_back_to_fit",
|
|
"tests/test_crawl4ai_tools.py::TestCrawl4AiTools::test_web_fetch_tool_passes_configured_filter",
|
|
"tests/test_crawl4ai_tools.py::TestCrawl4AiTools::test_web_fetch_tool_success",
|
|
"tests/test_crawl4ai_tools.py::TestCrawl4AiTools::test_web_fetch_tool_truncates_to_4096",
|
|
"tests/test_fastcrw_tools.py::TestWebFetchTool::test_fetch_returns_error_string_on_exception",
|
|
"tests/test_fastcrw_tools.py::TestWebFetchTool::test_fetch_returns_error_when_no_content",
|
|
"tests/test_fastcrw_tools.py::TestWebFetchTool::test_fetch_uses_web_fetch_config"
|
|
],
|
|
"review_regressions": {
|
|
"before_fix": {
|
|
"backend_failed": 22,
|
|
"backend_control_passed": 1,
|
|
"audit_failed": 2
|
|
},
|
|
"first_full_review_run": {
|
|
"passed": 15472,
|
|
"failed": 16,
|
|
"skipped": 182,
|
|
"deselected": 3,
|
|
"seconds": 470.45
|
|
},
|
|
"intermediate_branch_before_channel_validation": {
|
|
"passed": 15473,
|
|
"failed": 15,
|
|
"skipped": 182,
|
|
"deselected": 3,
|
|
"seconds": 396.36
|
|
},
|
|
"additional_channel_boundary_regressions_failed_before_fix": 2,
|
|
"first_full_review_run_note": "The additional failure was the existing reducer-field contract test, whose expected field list needed task_notes. Final rerun has only the same 15 base failures.",
|
|
"scenarios": [
|
|
"hidden clarification text/option responses survive compaction and source reads",
|
|
"explicit disabled middleware does not archive, sync and async",
|
|
"capture failure status with no scope, empty matching scope, old readable sources, and mismatched scope",
|
|
"notebook bounds and model-report shape on initial writes, Overwrite and reducer updates through the shared state channel, plus defensive rendering",
|
|
"normal run note deletions still work",
|
|
"state replacement via introspection and fallback in full/delta modes",
|
|
"branches clear archive references/status but preserve ordinary notes in full/delta modes",
|
|
"optional credential scanning including absent/null/empty values"
|
|
]
|
|
},
|
|
"config_upgrade": {
|
|
"from_version": 41,
|
|
"to_version": 42,
|
|
"default_disabled": true,
|
|
"explicit_enabled_preserved": true
|
|
},
|
|
"published_prototype_unit_tests": {
|
|
"passed": 14,
|
|
"seconds": 0.58
|
|
},
|
|
"published_score_consistency": "all five tables matched unchanged per-case metadata",
|
|
"live_integration": {
|
|
"passed": 3,
|
|
"total": 3,
|
|
"results": "integration/review-network.json",
|
|
"protocol": "integration/protocol.json"
|
|
},
|
|
"validation_conditions": [
|
|
"Locked Python dependencies installed separately for branch and clean base.",
|
|
"Both full suites had local server and dependency access available.",
|
|
"Successful historical A/B/C/D model samples were not rerun; the 3-case production-middleware integration was rerun.",
|
|
"Replay tests use pinned original local fixtures in a temporary copy; two new audit regressions raise the script test count from 12 to 14."
|
|
],
|
|
"source_sha256": {
|
|
"backend/app/gateway/AGENTS.md": "2928ffea0560997963b0b086fc08f00dfc37f80bec1837e4f98e0fa44d06786f",
|
|
"backend/app/gateway/routers/threads.py": "5a5263ec9d8ad297f1928e1665f27bb925ead9f578505350646e144328ef65cb",
|
|
"backend/packages/harness/deerflow/agents/lead_agent/agent.py": "7c55780374bc732e56a2ff8843a357daf8e525e3002e2342115fad7b9049b0be",
|
|
"backend/packages/harness/deerflow/agents/middlewares/durable_context_middleware.py": "2adfe86b877fa207635d63a365711d5b5a87bfc0cf7e932fc7d4daec6f8f0833",
|
|
"backend/packages/harness/deerflow/agents/middlewares/summarization_middleware.py": "bf5e12d4ca44eae77e030fc7743ffeec3f1bf2538f9ee4cc606d7adcebe94325",
|
|
"backend/packages/harness/deerflow/agents/middlewares/tool_error_handling_middleware.py": "4cdec0c2f614bdfb2b95d28065c142602beeff793aa6b43ba55695479519475c",
|
|
"backend/packages/harness/deerflow/agents/task_continuity/__init__.py": "8fba61f70e526bc45fe426f15746eb31b2d42e9d19b2852942787266a68bcd30",
|
|
"backend/packages/harness/deerflow/agents/task_continuity/archive.py": "ba821a6aa8d406d5fa2aeebe2108aee38f35b681412268bc7518b9c0ec61f428",
|
|
"backend/packages/harness/deerflow/agents/task_continuity/state.py": "feb780e202d88a14d74d62d80067bcd7a3a9c119e83a050c9e564e59aabffc6e",
|
|
"backend/packages/harness/deerflow/agents/task_continuity/tools.py": "d86570525feb805a15c862a1f4032f7e204282e3fd0e65966a36dbf5e20057f4",
|
|
"backend/packages/harness/deerflow/agents/thread_state.py": "04e10d910f721f6ff9996b0b8bdef09c9560ed4234678712c4ce3d514d7b796f",
|
|
"backend/packages/harness/deerflow/client.py": "fd4264fddd86a21329ca79bbf9c0b4994d56fbe96ed6fbc67289770f8d1478dd",
|
|
"backend/packages/harness/deerflow/config/app_config.py": "fcae3bf34177bb52820f1469382c857f6845f1a46d5bfa8eb3333b18a7148eaa",
|
|
"backend/packages/harness/deerflow/config/task_continuity_config.py": "bdb062a43fe9d1ba6ac35818afe847ac049cc62cbac39a3ec7489c8fd22078da",
|
|
"backend/packages/harness/deerflow/runtime/context_compaction.py": "e1332bff82b68e3edb23859e7833bcd53696998a7b5541edd43392320277e09a",
|
|
"backend/scripts/manual_task_continuity_check.py": "763fdccdffd3bd9c2566638b5f199bb806a01951c48014ea9c63031b7765f548",
|
|
"backend/tests/test_authorization_enforcement.py": "89e349781a73aa212a973257d6ff4b59edc66bebc374e784aebe10007fb895ae",
|
|
"backend/tests/test_client.py": "2310b72f08db2f8d620bd41e213227290a6116e4998531912599d9aa6511b16a",
|
|
"backend/tests/test_task_continuity.py": "46fa16fcbcf372c6a50f863e9f4d341a4c67b7ab36df64a834dc1848f77a2afc",
|
|
"backend/tests/test_thread_state_reducers.py": "e51de12ac62d732d8d8462e5ddf0afa818611b114ca574675b2429c4cfa25f59",
|
|
"backend/tests/test_threads_router.py": "501126bdacfbd6239289e6f65960c5a6228eaecdab86a0c274d179ea5726731e",
|
|
"backend/tests/test_tool_error_handling_middleware.py": "fd4c3683028901e4d520abd615694f1bbbf44d8c244b8e520c15ad86d3fe03fb"
|
|
},
|
|
"log_sha256": {
|
|
"full_branch": "10969fcac135691134fcfc5f710bc3ecd99c7d92e41bdc61525fd161d6305e76",
|
|
"full_base": "533d31aec15acbb365204e7572dded9859abf4a84c4c3a2f2f003b5c76b0520b",
|
|
"first_review_branch": "ffb43536e3ba4c81fef9a9d7f44cb1a475f0969a056ffef5043a6e6a18df7b91",
|
|
"focused": "ae9a3add0f93f86124a71dc0545213e60a4d64bb645d2be7a031d565182fa091",
|
|
"replay": "3d982959a6d0ad327e4c2ff7cfedf579823ee4d0c00b0aff1195763fb428a921",
|
|
"red_backend": "c15c90854ec15f06f6abbba3adb818a990201dbf6f6087674e3ce34d7da38b1e",
|
|
"red_audit": "901d91ea3bed702437826b7ba60d55a23e0464c9ce8eaf54116deecab10ed855",
|
|
"channel_red": "f97b5ead21ee616225c2d37daf7014107749754454f7d67514ae60eb55366358",
|
|
"intermediate_branch": "a68e6487284a428998f67eaf98753086df73b0bdeeb326ace827bb77d7612cd3"
|
|
},
|
|
"chart_validation": {
|
|
"initial_remote_commit": "2c884ab98ffdc40d34099abc9ec1a5eda98b39e6",
|
|
"initial_remote_failure": "config_version drift: example 42, chart 41",
|
|
"local_checks": {
|
|
"helm_lint": "passed",
|
|
"helm_template": "passed",
|
|
"sandbox_service_gating": "passed",
|
|
"skill_upload_ingress": "passed",
|
|
"config_version_drift": "passed"
|
|
},
|
|
"config_version": 42,
|
|
"rendered_task_continuity_default_enabled": false,
|
|
"scope": "Chart values and README version alignment only; backend code and its validated source fingerprints are unchanged."
|
|
}
|
|
}
|