mirror of
https://github.com/bytedance/deer-flow.git
synced 2026-08-15 09:19:02 +00:00
* fix(sandbox): project enabled skills into sandbox views * fix(skills): keep projection mutations consistent * fix(skills): fail closed on projection errors * fix(skills): isolate per-scope failures during boot projection rebuild rebuild_all_skill_projections() propagated any exception from the public rebuild or from a single user's rebuild straight out of the gateway lifespan startup, uncaught. A single broken user directory (bad permissions, corrupted _skill_states.json, unreadable content) would therefore abort gateway boot for every user, not just that one - _rebuild_*_locked already fails closed internally (clears the view and re-raises), so the boot loop only needed to stop treating that re-raise as fatal. Each scope's rebuild now fails closed independently and boot continues; a scope left empty by a boot failure self-heals on the next sandbox acquire via ensure_skill_projections(). Also patches deerflow.skills.projection.rebuild_all_skill_projections in the memory-flush lifespan test fixture, matching the two sibling fixtures in the same file — this call is now on the lifespan startup path and the fixture's minimal SimpleNamespace config predates it. * test(skills): update authz test for the projection-aware public toggle _persist_shared_skill_state (introduced earlier in this branch) reads the shared extensions_config.json fresh from disk under the projection lock instead of through the cached get_extensions_config() singleton - that's the whole point of the fix (stale worker caches must not clobber another worker's concurrent update). The name no longer exists on the skills router module, so the test's monkeypatch of it started raising AttributeError instead of exercising the endpoint. The mock storage in this test isn't a real LocalSkillStorage instance, so _persist_shared_skill_state's projection-mutation branch is already skipped (nullcontext) and it falls back to a fresh ExtensionsConfig() for the nonexistent tmp config_path - no replacement monkeypatch needed. * fix(sandbox): make skill projection ensure best-effort in acquire acquire() called _ensure_skills_projection() directly, outside any try/except, in both LocalSandboxProvider and AioSandboxProvider. Every other skill-mount setup path in these providers has always caught exceptions and logged a warning rather than failing sandbox acquire outright (e.g. when config.yaml can't be resolved) - these two new call sites broke that contract, so any projection failure (including simply not having a config.yaml, as in CI's test environment) now failed acquire() itself instead of just leaving skill mounts off. _ensure_skills_projection now catches its own exceptions and returns None; both providers' callers already tolerate that (a None projection skips the skill-specific mounts, matching the existing degrade path) after making _append_public_skill_mapping and the custom/legacy mount block in LocalSandboxProvider explicitly None-safe. Caught by running the full suite with config.yaml removed, matching CI's environment - not caught locally because a real config.yaml was present, masking the failure. * fix(sandbox): make E2B skill projection mounts best-effort _skill_projection_mounts called ensure_skill_projections with no guard, unlike Local/AIO's _ensure_skills_projection. A raise propagated out of _apply_mounts before the configured-mounts loop ran, so a skills projection failure dropped the operator's own configured mounts too - only caught by create()'s outer warning, with nothing applied at all. Swallow here and return an empty mount list on failure, matching the Local/AIO pattern: still fail-closed for skills, but no longer widens the blast radius to unrelated configured mounts. Review feedback from PR #4178. * docs(skills): document projection trade-offs flagged in review - _update_tree_digest: note the metadata-only (not content) hashing trade-off and why runtime writes through this codebase are still covered regardless (rebuild-under-lock + rename always changes inode). - LocalSandboxProvider.acquire: note the acquire-time self-heal cost (cheap on a fresh manifest, ~400ms rebuild under lock on stale/drift). - skill_projection_mutation: drop the no-op except-Exception-then-raise; a raise from the mutation already propagates past the yield with the view left cleared, no explicit re-raise needed. - provisioner README: spell out that hostPath skills volumes require the gateway and K8s node to share DEER_FLOW_HOST_BASE_DIR (single-node or shared storage), and that the custom/legacy volumes' hostPath type Directory (not DirectoryOrCreate) makes a violation of that assumption a visible Pod-creation failure instead of a silent empty mount. Review feedback from PR #4178. * fix(skills): lazily repair user projections * fix(skills): close projection review gaps * fix(skills): refresh user projection enable state * fix(skills): close projection review follow-ups * fix(skills): preserve state across projection writes --------- Co-authored-by: Willem Jiang <willem.jiang@gmail.com>
483 lines
21 KiB
Python
483 lines
21 KiB
Python
import hashlib
|
|
import logging
|
|
import os
|
|
import re
|
|
import shutil
|
|
from pathlib import Path, PureWindowsPath
|
|
|
|
from deerflow.config.runtime_paths import runtime_home
|
|
|
|
# Virtual path prefix seen by agents inside the sandbox
|
|
VIRTUAL_PATH_PREFIX = "/mnt/user-data"
|
|
|
|
_SAFE_THREAD_ID_RE = re.compile(r"^[A-Za-z0-9_\-]+$")
|
|
_SAFE_USER_ID_RE = re.compile(r"^[A-Za-z0-9_\-]+$")
|
|
_SAFE_INTEGRATION_ID_RE = re.compile(r"^[A-Za-z0-9_.\-]+$")
|
|
_UNSAFE_USER_ID_CHAR_RE = re.compile(r"[^A-Za-z0-9_\-]")
|
|
_SAFE_USER_ID_DIGEST_HEX_LEN = 16
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
def _default_local_base_dir() -> Path:
|
|
"""Return the caller project's writable DeerFlow state directory."""
|
|
return runtime_home()
|
|
|
|
|
|
def _validate_thread_id(thread_id: str) -> str:
|
|
"""Validate a thread ID before using it in filesystem paths."""
|
|
if not _SAFE_THREAD_ID_RE.match(thread_id):
|
|
raise ValueError(f"Invalid thread_id {thread_id!r}: only alphanumeric characters, hyphens, and underscores are allowed.")
|
|
return thread_id
|
|
|
|
|
|
def _validate_user_id(user_id: str) -> str:
|
|
"""Validate a user ID before using it in filesystem paths."""
|
|
if not _SAFE_USER_ID_RE.match(user_id):
|
|
raise ValueError(f"Invalid user_id {user_id!r}: only alphanumeric characters, hyphens, and underscores are allowed.")
|
|
return user_id
|
|
|
|
|
|
def _validate_integration_id(integration_id: str) -> str:
|
|
"""Validate an integration ID before using it in filesystem paths."""
|
|
if not _SAFE_INTEGRATION_ID_RE.match(integration_id):
|
|
raise ValueError(f"Invalid integration_id {integration_id!r}: only alphanumeric characters, dots, hyphens, and underscores are allowed.")
|
|
# The charset allows dots for names like ``some.integration``; reject the
|
|
# bare ``.``/``..`` path components so a future caller cannot escape the
|
|
# per-integration namespace via ``_join_host_path(..., integration_id, ...)``.
|
|
if integration_id in {".", ".."}:
|
|
raise ValueError(f"Invalid integration_id {integration_id!r}: '.' and '..' are not allowed.")
|
|
return integration_id
|
|
|
|
|
|
def make_safe_user_id(raw: str) -> str:
|
|
"""Normalize an external identity into the user-id charset (``[A-Za-z0-9_-]``).
|
|
|
|
IM channel ids (Feishu/Slack/Telegram) may contain characters that
|
|
:func:`_validate_user_id` rejects. Already-safe ids pass through unchanged;
|
|
lossy ones get a short digest suffix so two distinct inputs never share a
|
|
storage bucket.
|
|
"""
|
|
if not raw:
|
|
raise ValueError("user_id must be a non-empty string.")
|
|
sanitized = _UNSAFE_USER_ID_CHAR_RE.sub("-", raw)
|
|
if sanitized == raw:
|
|
return raw
|
|
digest = hashlib.sha256(raw.encode("utf-8")).hexdigest()[:_SAFE_USER_ID_DIGEST_HEX_LEN]
|
|
return f"{sanitized}-{digest}"
|
|
|
|
|
|
def _legacy_safe_user_id(raw: str, sanitized: str) -> str:
|
|
"""Bucket name produced by the previous (SHA-1) digest revision for ``raw``."""
|
|
digest = hashlib.sha1(raw.encode("utf-8"), usedforsecurity=False).hexdigest()[:_SAFE_USER_ID_DIGEST_HEX_LEN]
|
|
return f"{sanitized}-{digest}"
|
|
|
|
|
|
def _join_host_path(base: str, *parts: str) -> str:
|
|
"""Join host filesystem path segments while preserving native style.
|
|
|
|
Docker Desktop on Windows expects bind mount sources to stay in Windows
|
|
path form (for example ``C:\\repo\\backend\\.deer-flow``). Using
|
|
``Path(base) / ...`` on a POSIX host can accidentally rewrite those paths
|
|
with mixed separators, so this helper preserves the original style.
|
|
"""
|
|
if not parts:
|
|
return base
|
|
|
|
if re.match(r"^[A-Za-z]:[\\/]", base) or base.startswith("\\\\") or "\\" in base:
|
|
result = PureWindowsPath(base)
|
|
for part in parts:
|
|
result /= part
|
|
return str(result)
|
|
|
|
result = Path(base)
|
|
for part in parts:
|
|
result /= part
|
|
return str(result)
|
|
|
|
|
|
def join_host_path(base: str, *parts: str) -> str:
|
|
"""Join host filesystem path segments while preserving native style."""
|
|
return _join_host_path(base, *parts)
|
|
|
|
|
|
class Paths:
|
|
"""
|
|
Centralized path configuration for DeerFlow application data.
|
|
|
|
Directory layout (host side):
|
|
{base_dir}/
|
|
├── memory.json
|
|
├── USER.md <-- global user profile (injected into all agents)
|
|
├── agents/
|
|
│ └── {agent_name}/
|
|
│ ├── config.yaml
|
|
│ ├── SOUL.md <-- agent personality/identity (injected alongside lead prompt)
|
|
│ └── memory.json
|
|
└── threads/
|
|
└── {thread_id}/
|
|
└── user-data/ <-- mounted as /mnt/user-data/ inside sandbox
|
|
├── workspace/ <-- /mnt/user-data/workspace/
|
|
├── uploads/ <-- /mnt/user-data/uploads/
|
|
└── outputs/ <-- /mnt/user-data/outputs/
|
|
|
|
BaseDir resolution (in priority order):
|
|
1. Constructor argument `base_dir`
|
|
2. DEER_FLOW_HOME environment variable
|
|
3. Caller project fallback: `{project_root}/.deer-flow`
|
|
"""
|
|
|
|
def __init__(self, base_dir: str | Path | None = None) -> None:
|
|
self._base_dir = Path(base_dir).resolve() if base_dir is not None else None
|
|
|
|
@property
|
|
def host_base_dir(self) -> Path:
|
|
"""Host-visible base dir for Docker volume mount sources.
|
|
|
|
When running inside Docker with a mounted Docker socket (DooD), the Docker
|
|
daemon runs on the host and resolves mount paths against the host filesystem.
|
|
Set DEER_FLOW_HOST_BASE_DIR to the host-side path that corresponds to this
|
|
container's base_dir so that sandbox container volume mounts work correctly.
|
|
|
|
Falls back to base_dir when the env var is not set (native/local execution).
|
|
"""
|
|
if env := os.getenv("DEER_FLOW_HOST_BASE_DIR"):
|
|
return Path(env)
|
|
return self.base_dir
|
|
|
|
def _host_base_dir_str(self) -> str:
|
|
"""Return the host base dir as a raw string for bind mounts."""
|
|
if env := os.getenv("DEER_FLOW_HOST_BASE_DIR"):
|
|
return env
|
|
return str(self.base_dir)
|
|
|
|
@property
|
|
def base_dir(self) -> Path:
|
|
"""Root directory for all application data."""
|
|
if self._base_dir is not None:
|
|
return self._base_dir
|
|
|
|
if env_home := os.getenv("DEER_FLOW_HOME"):
|
|
return Path(env_home).resolve()
|
|
|
|
return _default_local_base_dir()
|
|
|
|
@property
|
|
def memory_file(self) -> Path:
|
|
"""Path to the persisted memory file: `{base_dir}/memory.json`."""
|
|
return self.base_dir / "memory.json"
|
|
|
|
@property
|
|
def user_md_file(self) -> Path:
|
|
"""Path to the global user profile file: `{base_dir}/USER.md`."""
|
|
return self.base_dir / "USER.md"
|
|
|
|
@property
|
|
def agents_dir(self) -> Path:
|
|
"""Legacy root for shared (pre user-isolation) custom agents: `{base_dir}/agents/`.
|
|
|
|
New code should use :meth:`user_agents_dir` instead. This property remains
|
|
only as a read-side fallback for installations that have not yet run the
|
|
``migrate_user_isolation.py`` script.
|
|
"""
|
|
return self.base_dir / "agents"
|
|
|
|
def agent_dir(self, name: str) -> Path:
|
|
"""Legacy per-agent directory (no user isolation): `{base_dir}/agents/{name}/`."""
|
|
return self.agents_dir / name.lower()
|
|
|
|
def agent_memory_file(self, name: str) -> Path:
|
|
"""Legacy per-agent memory file: `{base_dir}/agents/{name}/memory.json`."""
|
|
return self.agent_dir(name) / "memory.json"
|
|
|
|
def user_dir(self, user_id: str) -> Path:
|
|
"""Directory for a specific user: `{base_dir}/users/{user_id}/`."""
|
|
return self.base_dir / "users" / _validate_user_id(user_id)
|
|
|
|
def prepare_user_dir_for_raw_id(self, raw_user_id: str) -> str:
|
|
"""Return the safe user ID and migrate this ID's legacy unsafe-id bucket.
|
|
|
|
A previous branch revision used SHA-1 for unsafe external user IDs.
|
|
New IDs use SHA-256; the legacy bucket name is recomputed from the same
|
|
raw ID, so only this user's own old bucket can ever be moved — a
|
|
different raw ID sharing the sanitized prefix produces a different
|
|
legacy digest and is never touched.
|
|
"""
|
|
safe_user_id = make_safe_user_id(raw_user_id)
|
|
sanitized = _UNSAFE_USER_ID_CHAR_RE.sub("-", raw_user_id)
|
|
if safe_user_id == raw_user_id:
|
|
return safe_user_id
|
|
|
|
users_dir = self.base_dir / "users"
|
|
target_dir = users_dir / safe_user_id
|
|
legacy_dir = users_dir / _legacy_safe_user_id(raw_user_id, sanitized)
|
|
try:
|
|
if target_dir.exists() or not legacy_dir.is_dir():
|
|
return safe_user_id
|
|
legacy_dir.rename(target_dir)
|
|
logger.info("Migrated legacy unsafe-id user directory to the current digest format")
|
|
except OSError:
|
|
logger.exception("Failed to migrate legacy unsafe-id user directory")
|
|
return safe_user_id
|
|
|
|
def user_memory_file(self, user_id: str) -> Path:
|
|
"""Per-user memory file: `{base_dir}/users/{user_id}/memory.json`."""
|
|
return self.user_dir(user_id) / "memory.json"
|
|
|
|
def user_agents_dir(self, user_id: str) -> Path:
|
|
"""Per-user root for that user's custom agents: `{base_dir}/users/{user_id}/agents/`."""
|
|
return self.user_dir(user_id) / "agents"
|
|
|
|
def user_agent_dir(self, user_id: str, agent_name: str) -> Path:
|
|
"""Per-user per-agent directory: `{base_dir}/users/{user_id}/agents/{name}/`."""
|
|
return self.user_agents_dir(user_id) / agent_name.lower()
|
|
|
|
def user_agent_memory_file(self, user_id: str, agent_name: str) -> Path:
|
|
"""Per-user per-agent memory: `{base_dir}/users/{user_id}/agents/{name}/memory.json`."""
|
|
return self.user_agent_dir(user_id, agent_name) / "memory.json"
|
|
|
|
def user_skills_dir(self, user_id: str) -> Path:
|
|
"""Per-user root for that user's custom skills: `{base_dir}/users/{user_id}/skills/`."""
|
|
return self.user_dir(user_id) / "skills"
|
|
|
|
def user_custom_skills_dir(self, user_id: str) -> Path:
|
|
"""Per-user custom skills directory: `{base_dir}/users/{user_id}/skills/custom/`.
|
|
|
|
This is the user-scoped replacement for the global ``{base_dir}/skills/custom/``
|
|
directory. Custom skills are written here; public skills remain under the
|
|
global ``{base_dir}/skills/public/`` (read-only).
|
|
"""
|
|
return self.user_skills_dir(user_id) / "custom"
|
|
|
|
def integration_skills_dir(self) -> Path:
|
|
"""Globally installed managed integration skills.
|
|
|
|
Layout: ``{base_dir}/integrations/skills/{provider}/{skill}/``. The
|
|
package contents are shared and read-only; credentials and enabled
|
|
state remain user-scoped elsewhere under ``users/{user_id}``.
|
|
"""
|
|
return self.base_dir / "integrations" / "skills"
|
|
|
|
@property
|
|
def skills_view_dir(self) -> Path:
|
|
"""Global sandbox-visible skills projection: ``{base_dir}/skills_view/``."""
|
|
return self.base_dir / "skills_view"
|
|
|
|
@property
|
|
def public_skills_view_dir(self) -> Path:
|
|
"""Enabled public skills exposed to sandboxes."""
|
|
return self.skills_view_dir / "public"
|
|
|
|
def user_skills_view_dir(self, user_id: str) -> Path:
|
|
"""Per-user sandbox-visible skills projection root."""
|
|
return self.user_dir(user_id) / "skills_view"
|
|
|
|
def user_custom_skills_view_dir(self, user_id: str) -> Path:
|
|
"""Enabled custom skills exposed to one user's sandboxes."""
|
|
return self.user_skills_view_dir(user_id) / "custom"
|
|
|
|
def user_legacy_skills_view_dir(self, user_id: str) -> Path:
|
|
"""Enabled legacy skills exposed to one user's sandboxes."""
|
|
return self.user_skills_view_dir(user_id) / "legacy"
|
|
|
|
def user_integration_skills_view_dir(self, user_id: str) -> Path:
|
|
"""Enabled managed integration skills exposed to one user's sandboxes."""
|
|
return self.user_skills_view_dir(user_id) / "integrations"
|
|
|
|
def thread_dir(self, thread_id: str, *, user_id: str | None = None) -> Path:
|
|
"""
|
|
Host path for a thread's data.
|
|
|
|
When *user_id* is provided:
|
|
`{base_dir}/users/{user_id}/threads/{thread_id}/`
|
|
Otherwise (legacy layout):
|
|
`{base_dir}/threads/{thread_id}/`
|
|
|
|
This directory contains a `user-data/` subdirectory that is mounted
|
|
as `/mnt/user-data/` inside the sandbox.
|
|
|
|
Raises:
|
|
ValueError: If `thread_id` or `user_id` contains unsafe characters (path
|
|
separators or `..`) that could cause directory traversal.
|
|
"""
|
|
if user_id is not None:
|
|
return self.user_dir(user_id) / "threads" / _validate_thread_id(thread_id)
|
|
return self.base_dir / "threads" / _validate_thread_id(thread_id)
|
|
|
|
def sandbox_work_dir(self, thread_id: str, *, user_id: str | None = None) -> Path:
|
|
"""
|
|
Host path for the agent's workspace directory.
|
|
Host: `{base_dir}/threads/{thread_id}/user-data/workspace/`
|
|
Sandbox: `/mnt/user-data/workspace/`
|
|
"""
|
|
return self.thread_dir(thread_id, user_id=user_id) / "user-data" / "workspace"
|
|
|
|
def sandbox_uploads_dir(self, thread_id: str, *, user_id: str | None = None) -> Path:
|
|
"""
|
|
Host path for user-uploaded files.
|
|
Host: `{base_dir}/threads/{thread_id}/user-data/uploads/`
|
|
Sandbox: `/mnt/user-data/uploads/`
|
|
"""
|
|
return self.thread_dir(thread_id, user_id=user_id) / "user-data" / "uploads"
|
|
|
|
def sandbox_outputs_dir(self, thread_id: str, *, user_id: str | None = None) -> Path:
|
|
"""
|
|
Host path for agent-generated artifacts.
|
|
Host: `{base_dir}/threads/{thread_id}/user-data/outputs/`
|
|
Sandbox: `/mnt/user-data/outputs/`
|
|
"""
|
|
return self.thread_dir(thread_id, user_id=user_id) / "user-data" / "outputs"
|
|
|
|
def acp_workspace_dir(self, thread_id: str, *, user_id: str | None = None) -> Path:
|
|
"""
|
|
Host path for the ACP workspace of a specific thread.
|
|
Host: `{base_dir}/threads/{thread_id}/acp-workspace/`
|
|
Sandbox: `/mnt/acp-workspace/`
|
|
|
|
Each thread gets its own isolated ACP workspace so that concurrent
|
|
sessions cannot read each other's ACP agent outputs.
|
|
"""
|
|
return self.thread_dir(thread_id, user_id=user_id) / "acp-workspace"
|
|
|
|
def sandbox_user_data_dir(self, thread_id: str, *, user_id: str | None = None) -> Path:
|
|
"""
|
|
Host path for the user-data root.
|
|
Host: `{base_dir}/threads/{thread_id}/user-data/`
|
|
Sandbox: `/mnt/user-data/`
|
|
"""
|
|
return self.thread_dir(thread_id, user_id=user_id) / "user-data"
|
|
|
|
def host_thread_dir(self, thread_id: str, *, user_id: str | None = None) -> str:
|
|
"""Host path for a thread directory, preserving Windows path syntax."""
|
|
if user_id is not None:
|
|
return _join_host_path(self._host_base_dir_str(), "users", _validate_user_id(user_id), "threads", _validate_thread_id(thread_id))
|
|
return _join_host_path(self._host_base_dir_str(), "threads", _validate_thread_id(thread_id))
|
|
|
|
def host_sandbox_user_data_dir(self, thread_id: str, *, user_id: str | None = None) -> str:
|
|
"""Host path for a thread's user-data root."""
|
|
return _join_host_path(self.host_thread_dir(thread_id, user_id=user_id), "user-data")
|
|
|
|
def host_sandbox_work_dir(self, thread_id: str, *, user_id: str | None = None) -> str:
|
|
"""Host path for the workspace mount source."""
|
|
return _join_host_path(self.host_sandbox_user_data_dir(thread_id, user_id=user_id), "workspace")
|
|
|
|
def host_sandbox_uploads_dir(self, thread_id: str, *, user_id: str | None = None) -> str:
|
|
"""Host path for the uploads mount source."""
|
|
return _join_host_path(self.host_sandbox_user_data_dir(thread_id, user_id=user_id), "uploads")
|
|
|
|
def host_sandbox_outputs_dir(self, thread_id: str, *, user_id: str | None = None) -> str:
|
|
"""Host path for the outputs mount source."""
|
|
return _join_host_path(self.host_sandbox_user_data_dir(thread_id, user_id=user_id), "outputs")
|
|
|
|
def host_acp_workspace_dir(self, thread_id: str, *, user_id: str | None = None) -> str:
|
|
"""Host path for the ACP workspace mount source."""
|
|
return _join_host_path(self.host_thread_dir(thread_id, user_id=user_id), "acp-workspace")
|
|
|
|
def host_user_custom_skills_dir(self, user_id: str) -> str:
|
|
"""Host path for a user's custom skills directory, preserving Windows path syntax."""
|
|
return _join_host_path(self._host_base_dir_str(), "users", _validate_user_id(user_id), "skills", "custom")
|
|
|
|
def host_integration_skills_dir(self) -> str:
|
|
"""Host path for globally installed managed integration skills."""
|
|
return _join_host_path(self._host_base_dir_str(), "integrations", "skills")
|
|
|
|
def host_user_integration_config_dir(self, user_id: str, integration_id: str) -> str:
|
|
"""Host path for a user's managed integration runtime config directory."""
|
|
return _join_host_path(self._host_base_dir_str(), "users", _validate_user_id(user_id), "integrations", _validate_integration_id(integration_id), "config")
|
|
|
|
def host_user_integration_data_dir(self, user_id: str, integration_id: str) -> str:
|
|
"""Host path for a user's managed integration runtime data directory."""
|
|
return _join_host_path(self._host_base_dir_str(), "users", _validate_user_id(user_id), "integrations", _validate_integration_id(integration_id), "data")
|
|
|
|
def ensure_thread_dirs(self, thread_id: str, *, user_id: str | None = None) -> None:
|
|
"""Create all standard sandbox directories for a thread.
|
|
|
|
Directories are created with mode 0o777 so that sandbox containers
|
|
(which may run as a different UID than the host backend process) can
|
|
write to the volume-mounted paths without "Permission denied" errors.
|
|
The explicit chmod() call is necessary because Path.mkdir(mode=...) is
|
|
subject to the process umask and may not yield the intended permissions.
|
|
|
|
Includes the ACP workspace directory so it can be volume-mounted into
|
|
the sandbox container at ``/mnt/acp-workspace`` even before the first
|
|
ACP agent invocation.
|
|
"""
|
|
for d in [
|
|
self.sandbox_work_dir(thread_id, user_id=user_id),
|
|
self.sandbox_uploads_dir(thread_id, user_id=user_id),
|
|
self.sandbox_outputs_dir(thread_id, user_id=user_id),
|
|
self.acp_workspace_dir(thread_id, user_id=user_id),
|
|
]:
|
|
d.mkdir(parents=True, exist_ok=True)
|
|
d.chmod(0o777)
|
|
|
|
def delete_thread_dir(self, thread_id: str, *, user_id: str | None = None) -> None:
|
|
"""Delete all persisted data for a thread.
|
|
|
|
The operation is idempotent: missing thread directories are ignored.
|
|
"""
|
|
thread_dir = self.thread_dir(thread_id, user_id=user_id)
|
|
if thread_dir.exists():
|
|
shutil.rmtree(thread_dir)
|
|
|
|
def resolve_virtual_path(self, thread_id: str, virtual_path: str, *, user_id: str | None = None) -> Path:
|
|
"""Resolve a sandbox virtual path to the actual host filesystem path.
|
|
|
|
Args:
|
|
thread_id: The thread ID.
|
|
virtual_path: Virtual path as seen inside the sandbox, e.g.
|
|
``/mnt/user-data/outputs/report.pdf``.
|
|
Leading slashes are stripped before matching.
|
|
user_id: Optional user ID for user-scoped path resolution.
|
|
|
|
Returns:
|
|
The resolved absolute host filesystem path.
|
|
|
|
Raises:
|
|
ValueError: If the path does not start with the expected virtual
|
|
prefix or a path-traversal attempt is detected.
|
|
"""
|
|
stripped = virtual_path.lstrip("/")
|
|
prefix = VIRTUAL_PATH_PREFIX.lstrip("/")
|
|
|
|
# Require an exact segment-boundary match to avoid prefix confusion
|
|
# (e.g. reject paths like "mnt/user-dataX/...").
|
|
if stripped != prefix and not stripped.startswith(prefix + "/"):
|
|
raise ValueError(f"Path must start with /{prefix}")
|
|
|
|
relative = stripped[len(prefix) :].lstrip("/")
|
|
base = self.sandbox_user_data_dir(thread_id, user_id=user_id).resolve()
|
|
actual = (base / relative).resolve()
|
|
|
|
try:
|
|
actual.relative_to(base)
|
|
except ValueError:
|
|
raise ValueError("Access denied: path traversal detected")
|
|
|
|
return actual
|
|
|
|
|
|
# ── Singleton ────────────────────────────────────────────────────────────
|
|
|
|
_paths: Paths | None = None
|
|
|
|
|
|
def get_paths() -> Paths:
|
|
"""Return the global Paths singleton (lazy-initialized)."""
|
|
global _paths
|
|
if _paths is None:
|
|
_paths = Paths()
|
|
return _paths
|
|
|
|
|
|
def resolve_path(path: str) -> Path:
|
|
"""Resolve *path* to an absolute ``Path``.
|
|
|
|
Relative paths are resolved relative to the application base directory.
|
|
Absolute paths are returned as-is (after normalisation).
|
|
"""
|
|
p = Path(path)
|
|
if not p.is_absolute():
|
|
p = get_paths().base_dir / path
|
|
return p.resolve()
|