mirror of
https://github.com/bytedance/deer-flow.git
synced 2026-08-26 14:48:48 +00:00
* feat: add lark cli integration * fix: polish lark integration actions * feat: support lark incremental permissions * fix: detect lark authorization completion * fix: harden lark integration install * feat: expand lark auth scopes and reuse host auth in sandbox Default lark auth to least-privilege (recommend=false, base sign-in only) and expose the full set of lark-cli --domain business domains as native --domain grants instead of a 4-domain read-only mapping. Resolve the skill pack from the latest larksuite/cli GitHub release at install time with content-hash integrity, and surface version/runtime drift in status. Share the per-user lark-cli config/data profile between the Gateway Settings auth flow and agent conversations by mounting the integration dirs into the AIO sandbox and injecting the matching env for lark-cli commands, with an allowlisted extra_mounts path in the provisioner/K8s backend and traversal guards on integration paths. * style: fix lint issues from ruff and prettier Sort imports in the provisioner PVC test and re-wrap two long i18n description strings to satisfy backend ruff and frontend prettier CI. * fix(lark): address managed integration review feedback * fix(frontend): stabilize integrations settings e2e * test(sandbox): isolate remote backend legacy visibility check * test: fix backend unit failures after merge * Harden Lark integration review fixes * Format Lark integration E2E test * fix(lark): harden sandbox credential exposure and status disclosure Address willem_bd's security review on PR #3971: - Mount the per-user lark-cli config dir (long-lived appSecret) read-only into the AIO sandbox; only the refreshable-token data dir stays writable. - Redact host filesystem paths (install_path, cli.path) from GET /lark/status and the config/auth complete responses for non-admin callers, fail-closed on any auth error. - Document the npm postinstall trade-off (--ignore-scripts is not viable because @larksuite/cli fetches its platform binary in postinstall). - Document the sandbox credential trust boundary in AGENTS.md and README, pointing at the sidecar-broker follow-up (#4338). --------- Co-authored-by: Willem Jiang <willem.jiang@gmail.com>
157 lines
5.2 KiB
Python
157 lines
5.2 KiB
Python
"""Abstract base class for sandbox provisioning backends."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import asyncio
|
|
import logging
|
|
import time
|
|
from abc import ABC, abstractmethod
|
|
|
|
import httpx
|
|
import requests
|
|
|
|
from .sandbox_info import SandboxInfo
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
def wait_for_sandbox_ready(sandbox_url: str, timeout: int = 30) -> bool:
|
|
"""Poll sandbox health endpoint until ready or timeout.
|
|
|
|
Args:
|
|
sandbox_url: URL of the sandbox (e.g. http://k3s:30001).
|
|
timeout: Maximum time to wait in seconds.
|
|
|
|
Returns:
|
|
True if sandbox is ready, False otherwise.
|
|
"""
|
|
start_time = time.time()
|
|
while time.time() - start_time < timeout:
|
|
try:
|
|
response = requests.get(f"{sandbox_url}/v1/sandbox", timeout=5)
|
|
if response.status_code == 200:
|
|
return True
|
|
except requests.exceptions.RequestException:
|
|
pass
|
|
time.sleep(1)
|
|
return False
|
|
|
|
|
|
async def wait_for_sandbox_ready_async(sandbox_url: str, timeout: int = 30, poll_interval: float = 1.0) -> bool:
|
|
"""Async variant of sandbox readiness polling.
|
|
|
|
Use this from async runtime paths so sandbox startup waits do not block the
|
|
event loop. The synchronous ``wait_for_sandbox_ready`` function remains for
|
|
existing synchronous backend/provider call sites.
|
|
"""
|
|
loop = asyncio.get_running_loop()
|
|
deadline = loop.time() + timeout
|
|
|
|
async with httpx.AsyncClient(timeout=5) as client:
|
|
while True:
|
|
remaining = deadline - loop.time()
|
|
if remaining <= 0:
|
|
break
|
|
try:
|
|
response = await client.get(f"{sandbox_url}/v1/sandbox", timeout=min(5.0, remaining))
|
|
if response.status_code == 200:
|
|
return True
|
|
except httpx.RequestError:
|
|
pass
|
|
remaining = deadline - loop.time()
|
|
if remaining <= 0:
|
|
break
|
|
await asyncio.sleep(min(poll_interval, remaining))
|
|
return False
|
|
|
|
|
|
class SandboxBackend(ABC):
|
|
"""Abstract base for sandbox provisioning backends.
|
|
|
|
Two implementations:
|
|
- LocalContainerBackend: starts Docker/Apple Container locally, manages ports
|
|
- RemoteSandboxBackend: connects to a pre-existing URL (K8s service, external)
|
|
"""
|
|
|
|
@abstractmethod
|
|
def create(
|
|
self,
|
|
thread_id: str | None,
|
|
sandbox_id: str,
|
|
extra_mounts: list[tuple[str, str, bool]] | None = None,
|
|
*,
|
|
user_id: str | None = None,
|
|
provision_lark_cli_runtime: bool = False,
|
|
) -> SandboxInfo:
|
|
"""Create/provision a new sandbox.
|
|
|
|
Args:
|
|
thread_id: Thread ID for which the sandbox is being created. Useful for backends that want to organize sandboxes by thread.
|
|
sandbox_id: Deterministic sandbox identifier.
|
|
extra_mounts: Additional volume mounts as (host_path, container_path, read_only) tuples.
|
|
Ignored by backends that don't manage containers (e.g., remote).
|
|
user_id: User bucket that the sandbox should mount or provision for.
|
|
provision_lark_cli_runtime: Ask the backend to provision the sandbox
|
|
lark-cli runtime via its native mechanism (e.g. the provisioner's
|
|
init container + emptyDir). Backends that can't do this ignore it.
|
|
|
|
Returns:
|
|
SandboxInfo with connection details.
|
|
"""
|
|
...
|
|
|
|
@abstractmethod
|
|
def destroy(self, info: SandboxInfo) -> None:
|
|
"""Destroy/cleanup a sandbox and release its resources.
|
|
|
|
Args:
|
|
info: The sandbox metadata to destroy.
|
|
"""
|
|
...
|
|
|
|
@abstractmethod
|
|
def is_alive(self, info: SandboxInfo) -> bool:
|
|
"""Quick check whether a sandbox is still alive.
|
|
|
|
This should be a lightweight check (e.g., container inspect)
|
|
rather than a full health check.
|
|
|
|
Args:
|
|
info: The sandbox metadata to check.
|
|
|
|
Returns:
|
|
True if the sandbox appears to be alive.
|
|
"""
|
|
...
|
|
|
|
@abstractmethod
|
|
def discover(self, sandbox_id: str) -> SandboxInfo | None:
|
|
"""Try to discover an existing sandbox by its deterministic ID.
|
|
|
|
Used for cross-process recovery: when another process started a sandbox,
|
|
this process can discover it by the deterministic container name or URL.
|
|
|
|
Args:
|
|
sandbox_id: The deterministic sandbox ID to look for.
|
|
|
|
Returns:
|
|
SandboxInfo if found and healthy, None otherwise.
|
|
"""
|
|
...
|
|
|
|
def list_running(self) -> list[SandboxInfo]:
|
|
"""Enumerate all running sandboxes managed by this backend.
|
|
|
|
Used for startup reconciliation: when the process restarts, it needs
|
|
to discover containers started by previous processes so they can be
|
|
adopted into the warm pool or destroyed if idle too long.
|
|
|
|
The default implementation returns an empty list, which is correct
|
|
for backends that don't manage local containers (e.g., RemoteSandboxBackend
|
|
delegates lifecycle to the provisioner which handles its own cleanup).
|
|
|
|
Returns:
|
|
A list of SandboxInfo for all currently running sandboxes.
|
|
"""
|
|
return []
|