Shxiao 48bbea6df3
fix(workspace-changes): record symlink targets without the verbatim prefix (#5250)
* fix(workspace-changes): record symlink targets without the verbatim prefix

os.readlink on Windows reports absolute targets in extended-length
form (\?\C:\... or \?\UNC\server\share). The scanner stored that raw
spelling, so workspace-change events showed \?\-prefixed targets that
do not match ordinary Windows paths. Strip the prefix when recording;
POSIX readlink output is unchanged. Skills projection/review readlink
sites are untouched — they have no user-facing contract pinned on the
spelling.

* fix(workspace-changes): gate symlink target normalization to Windows

Review follow-up on #5250:

- Gate _normalize_symlink_target on os.name == "nt". readlink(2) on
  POSIX returns the literal string the link was created with, and
  backslash is a valid filename byte on Linux, so a target that starts
  with the extended-length prefix there must be recorded verbatim. The
  docstring's POSIX claim is now provably true.
- Commit the unit checks the PR body previously described as ad-hoc:
  drive and UNC prefix stripping, relative and plain POSIX targets,
  mid-string prefix left verbatim, and an off-Windows identity case, so
  the new branch has real coverage on every platform instead of relying
  on a Windows host with symlink privilege.

* test(workspace-changes): force Windows platform in prefix-strip unit tests

The os.name gate added in the previous commit makes
_normalize_symlink_target a verbatim identity off-Windows, so the two
prefix-strip assertions failed on the ubuntu-only unit CI. Force
os.name to "nt" via monkeypatch in both, mirroring the off-Windows
identity test, so every case pins exactly one platform's contract and
the suite is green on every host.

* fix(workspace-changes): strip only extended drive-letter prefixes

Review follow-up on #5250: the catch-all branch also stripped the
extended-length prefix from volume-GUID targets (\?\Volume{...}\...),
leaving a relative-looking path that loses the target's namespace.
Restrict the branch to extended drive-letter paths (letter, colon,
separator) and keep every other \?\ namespace form verbatim; add the
volume-GUID regression plus degenerate-prefix cases.
2026-09-08 14:51:57 +08:00

379 lines
12 KiB
Python

from __future__ import annotations
import fnmatch
import hashlib
import os
from codecs import BOM_UTF16_BE, BOM_UTF16_LE, getincrementaldecoder
from pathlib import Path
from deerflow.constants import BROWSER_FRAMES_DIRNAME, MCP_INTERNAL_DIRNAME, TOOL_RESULTS_DIRNAME
from .types import (
DiffUnavailableReason,
FileSnapshot,
WorkspaceChangeLimits,
WorkspaceRoot,
WorkspaceSnapshot,
)
EXCLUDED_DIR_NAMES = {
".git",
".hg",
".svn",
".cache",
# Stdio MCP subprocess temp/debug files live below this server-owned
# namespace. They remain addressable when a tool returns their path, but
# they are not user-authored workspace deliverables or workspace changes.
MCP_INTERNAL_DIRNAME,
".next",
".venv",
# Transient per-step browser screenshots: live progress feedback surfaced in
# the browser panel + inline thumbnails, not workspace deliverables. Shared
# constant with the browser tools so the name cannot drift out of sync.
BROWSER_FRAMES_DIRNAME,
# Externalized oversized tool outputs (the tool-output budget middleware's
# default storage_subdir): process feedback the model reads back via
# read_file, not workspace deliverables — same intent as the browser frames
# exclusion above. Without this, a run that externalizes any tool output
# would trip run delivery verification (produced output never presented)
# and fail as an error. Custom storage_subdir values are passed through
# ``extra_excluded_dir_names`` instead.
TOOL_RESULTS_DIRNAME,
"__pycache__",
"build",
"dist",
"node_modules",
}
BINARY_EXTENSIONS = {
".7z",
".avif",
".bmp",
".class",
".db",
".dll",
".dmg",
".doc",
".docx",
".exe",
".gif",
".gz",
".ico",
".jar",
".jpeg",
".jpg",
".mov",
".mp3",
".mp4",
".o",
".pdf",
".png",
".pyc",
".so",
".tar",
".webp",
".xls",
".xlsx",
".zip",
}
SENSITIVE_PATH_PATTERNS = (
".env",
".env.*",
"*api_key*",
"*apikey*",
"*.key",
"*.pem",
"*credential*",
"*password*",
"*private_key*",
"*secret*",
"*token*",
)
SAMPLE_BYTES = 4096
_UTF16_BOMS = (BOM_UTF16_LE, BOM_UTF16_BE)
def is_sensitive_workspace_path(path: str) -> bool:
normalized = path.lower()
parts = [part.lower() for part in Path(path).parts]
basename = parts[-1] if parts else normalized
for pattern in SENSITIVE_PATH_PATTERNS:
if fnmatch.fnmatch(basename, pattern) or fnmatch.fnmatch(normalized, pattern):
return True
if any(fnmatch.fnmatch(part, pattern) for part in parts):
return True
return False
def scan_workspace_roots(
roots: list[WorkspaceRoot],
*,
limits: WorkspaceChangeLimits | None = None,
include_text: bool = True,
text_paths: set[str] | None = None,
text_cache_dir: Path | None = None,
extra_excluded_dir_names: frozenset[str] | None = None,
) -> WorkspaceSnapshot:
resolved_limits = limits or WorkspaceChangeLimits()
cache_dir = Path(text_cache_dir) if text_cache_dir is not None else None
if cache_dir is not None:
cache_dir.mkdir(parents=True, exist_ok=True)
# Operator-customized tool_output.storage_subdir values arrive here; the
# default name is already part of EXCLUDED_DIR_NAMES, so merging is safe.
# Only single-segment directory names are meaningful: os.walk yields
# one-segment dirnames, so a nested value like "cache/tool-results" would
# never match. ToolOutputConfig enforces the single-segment contract, so a
# multi-segment value is a caller error, not a silent no-op.
excluded_dir_names = EXCLUDED_DIR_NAMES | extra_excluded_dir_names if extra_excluded_dir_names else EXCLUDED_DIR_NAMES
files: dict[str, FileSnapshot] = {}
scanned = 0
truncated = False
for root in roots:
if not root.host_path.exists():
continue
for dirpath, dirnames, filenames in os.walk(root.host_path, followlinks=False):
dirnames[:] = [dirname for dirname in dirnames if dirname not in excluded_dir_names and not (Path(dirpath) / dirname).is_symlink()]
for filename in sorted(filenames):
if scanned >= resolved_limits.max_scanned_files:
truncated = True
return WorkspaceSnapshot(
files=files,
truncated=truncated,
text_cache_dir=str(cache_dir) if cache_dir is not None else None,
)
host_file = Path(dirpath) / filename
if host_file.is_symlink():
# A symlink must never be followed for stat/content purposes: its
# target can point anywhere on the host (including outside the
# scanned root), so it is recorded as a metadata-only stub -
# mirroring how binary/large/sensitive-looking files are handled
# below - instead of being silently omitted from the snapshot.
symlink_snapshot = _snapshot_symlink(root, host_file)
if symlink_snapshot is not None:
files[symlink_snapshot.path] = symlink_snapshot
scanned += 1
continue
if not host_file.is_file():
continue
snapshot = _snapshot_file(
root,
host_file,
limits=resolved_limits,
include_text=include_text,
text_paths=text_paths,
text_cache_dir=cache_dir,
)
if snapshot is not None:
files[snapshot.path] = snapshot
scanned += 1
return WorkspaceSnapshot(
files=files,
truncated=truncated,
text_cache_dir=str(cache_dir) if cache_dir is not None else None,
)
def _snapshot_file(
root: WorkspaceRoot,
host_file: Path,
*,
limits: WorkspaceChangeLimits,
include_text: bool,
text_paths: set[str] | None,
text_cache_dir: Path | None,
) -> FileSnapshot | None:
try:
stat = host_file.stat()
size = stat.st_size
mtime_ns = stat.st_mtime_ns
relative = host_file.relative_to(root.host_path).as_posix()
virtual_path = f"{root.virtual_prefix}/{relative}"
sensitive = is_sensitive_workspace_path(virtual_path)
except OSError:
return None
if sensitive:
return FileSnapshot(
path=virtual_path,
root=root.name,
size=size,
mtime_ns=mtime_ns,
sha256=None,
binary=False,
sensitive=True,
text=None,
content_unavailable_reason="sensitive",
)
try:
sample = host_file.read_bytes()[:SAMPLE_BYTES] if size <= SAMPLE_BYTES else _read_sample(host_file)
except OSError:
return None
binary = host_file.suffix.lower() in BINARY_EXTENSIONS or _looks_binary(sample)
sha256 = _sha256_file(host_file) if size <= limits.max_file_bytes_for_diff else None
text: str | None = None
text_path: str | None = None
reason: DiffUnavailableReason | None = None
should_include_text = include_text and (text_paths is None or virtual_path in text_paths)
if binary:
reason = "binary"
elif size > limits.max_file_bytes_for_diff:
reason = "large"
elif not should_include_text:
text = None
else:
try:
raw = host_file.read_bytes()
except OSError:
return None
decoded = _decode_text_bytes(raw)
if decoded is None:
binary = True
reason = "binary"
elif text_cache_dir is not None:
text_path = str(_cache_text_file(decoded, virtual_path, text_cache_dir))
else:
text = decoded
return FileSnapshot(
path=virtual_path,
root=root.name,
size=size,
mtime_ns=mtime_ns,
sha256=sha256,
binary=binary,
sensitive=sensitive,
text=text,
text_path=text_path,
content_unavailable_reason=reason,
)
def _normalize_symlink_target(target: str) -> str:
"""Strip the Windows extended-length prefix from a symlink target.
``os.readlink`` on Windows reports absolute targets in extended-length
form (``\\\\?\\C:\\...`` or ``\\\\?\\UNC\\server\\share``). Recorded targets
are surfaced in workspace-change events and compared against ordinary
paths, so keep the plain spelling.
On POSIX this is a provable identity: the strip only applies on Windows
hosts. ``readlink(2)`` returns the literal string the link was created
with, and backslash is a valid filename byte on Linux — a target string
that merely starts with ``\\\\?\\`` there must be recorded verbatim.
On Windows, only extended *drive-letter* paths are stripped. Other
``\\\\?\\`` namespace forms (volume-GUID paths, device paths) are kept
verbatim: stripping them would leave a relative-looking remainder that
no longer names the target's namespace.
"""
if os.name != "nt":
return target
if target.startswith("\\\\?\\UNC\\"):
return "\\\\" + target[len("\\\\?\\UNC\\") :]
if target.startswith("\\\\?\\") and len(target) >= 7 and target[4].isascii() and target[4].isalpha() and target[5] == ":" and target[6] in "\\/":
return target[4:]
return target
def _snapshot_symlink(root: WorkspaceRoot, host_file: Path) -> FileSnapshot | None:
# Deliberately never follows the link (no read_bytes()/open() on the target):
# the target may point anywhere on the host, including outside the scanned
# root, so stat'ing or reading through it here would risk exposing arbitrary
# host file content/metadata as if it belonged to the workspace.
try:
stat = host_file.lstat()
size = stat.st_size
mtime_ns = stat.st_mtime_ns
relative = host_file.relative_to(root.host_path).as_posix()
virtual_path = f"{root.virtual_prefix}/{relative}"
sensitive = is_sensitive_workspace_path(virtual_path)
except OSError:
return None
try:
target = os.readlink(host_file)
except OSError:
target = None
else:
target = _normalize_symlink_target(target)
return FileSnapshot(
path=virtual_path,
root=root.name,
size=size,
mtime_ns=mtime_ns,
sha256=None,
binary=False,
sensitive=sensitive,
text=None,
content_unavailable_reason="symlink",
symlink=True,
symlink_target=target,
)
def _cache_text_file(text: str, virtual_path: str, cache_dir: Path) -> Path:
cache_name = hashlib.sha256(virtual_path.encode("utf-8")).hexdigest()
target = cache_dir / cache_name
target.write_text(text, encoding="utf-8")
return target
def _read_sample(path: Path) -> bytes:
with path.open("rb") as file:
return file.read(SAMPLE_BYTES)
def _sha256_file(path: Path) -> str:
digest = hashlib.sha256()
with path.open("rb") as file:
for chunk in iter(lambda: file.read(1024 * 1024), b""):
digest.update(chunk)
return digest.hexdigest()
def _decode_text_bytes(data: bytes) -> str | None:
for encoding in ("utf-8-sig", "utf-8"):
try:
return data.decode(encoding)
except UnicodeDecodeError:
continue
if data.startswith(_UTF16_BOMS):
try:
return data.decode("utf-16")
except UnicodeDecodeError:
return None
return None
def _sample_decodes_as_text(sample: bytes, encoding: str) -> bool:
try:
decoder = getincrementaldecoder(encoding)()
decoder.decode(sample, final=False)
except UnicodeDecodeError:
return False
return True
def _looks_binary(sample: bytes) -> bool:
if sample.startswith(_UTF16_BOMS) and _sample_decodes_as_text(sample, "utf-16"):
return False
if b"\x00" in sample:
return True
if _sample_decodes_as_text(sample, "utf-8"):
return False
return True