2026-06-09 19:05:36 +05:30
|
|
|
import asyncio
|
|
|
|
|
import json
|
|
|
|
|
import os
|
2026-06-19 03:02:29 +07:00
|
|
|
import re
|
2026-06-09 19:05:36 +05:30
|
|
|
import difflib
|
|
|
|
|
import fnmatch
|
|
|
|
|
import shutil
|
|
|
|
|
from typing import Optional, Dict, Any, Tuple
|
|
|
|
|
|
|
|
|
|
from src.constants import MAX_READ_CHARS, MAX_DIFF_LINES, MAX_OUTPUT_CHARS
|
|
|
|
|
|
|
|
|
|
_CODENAV_SKIP_DIRS = frozenset({
|
|
|
|
|
".git", ".hg", ".svn", "node_modules", "venv", ".venv", "__pycache__",
|
|
|
|
|
".mypy_cache", ".pytest_cache", ".ruff_cache", "dist", "build",
|
|
|
|
|
".next", ".cache", "site-packages", ".idea", ".tox",
|
|
|
|
|
})
|
|
|
|
|
_CODENAV_MAX_HITS = 200
|
|
|
|
|
_CODENAV_MAX_LINE = 400
|
|
|
|
|
|
2026-06-19 03:02:29 +07:00
|
|
|
|
|
|
|
|
def _glob_to_regex(pat: str) -> "re.Pattern":
|
|
|
|
|
"""Translate a forward-slash glob (**, *, ?) into a compiled regex.
|
|
|
|
|
`**/` matches zero or more complete directories.
|
|
|
|
|
`*` matches within a single path segment (does not cross /).
|
|
|
|
|
"""
|
|
|
|
|
i, n, out = 0, len(pat), []
|
|
|
|
|
while i < n:
|
|
|
|
|
if pat[i : i + 3] == "**/":
|
|
|
|
|
out.append("(?:[^/]+/)*")
|
|
|
|
|
i += 3
|
|
|
|
|
elif pat[i : i + 2] == "**":
|
|
|
|
|
out.append(".*")
|
|
|
|
|
i += 2
|
|
|
|
|
elif pat[i] == "*":
|
|
|
|
|
out.append("[^/]*")
|
|
|
|
|
i += 1
|
|
|
|
|
elif pat[i] == "?":
|
|
|
|
|
out.append("[^/]")
|
|
|
|
|
i += 1
|
|
|
|
|
else:
|
|
|
|
|
out.append(re.escape(pat[i]))
|
|
|
|
|
i += 1
|
|
|
|
|
return re.compile("".join(out))
|
|
|
|
|
|
2026-06-09 19:05:36 +05:30
|
|
|
def _unified_diff(old: str, new: str, path: str) -> Optional[Dict[str, Any]]:
|
|
|
|
|
if old == new:
|
|
|
|
|
return None
|
|
|
|
|
old_lines = old.splitlines()
|
|
|
|
|
new_lines = new.splitlines()
|
|
|
|
|
label = path or "file"
|
|
|
|
|
diff_lines = list(difflib.unified_diff(
|
|
|
|
|
old_lines, new_lines,
|
|
|
|
|
fromfile=f"a/{label}", tofile=f"b/{label}",
|
|
|
|
|
lineterm="",
|
|
|
|
|
))
|
|
|
|
|
added = sum(1 for line in diff_lines if line.startswith("+") and not line.startswith("+++"))
|
|
|
|
|
removed = sum(1 for line in diff_lines if line.startswith("-") and not line.startswith("---"))
|
|
|
|
|
truncated = False
|
|
|
|
|
if len(diff_lines) > MAX_DIFF_LINES:
|
|
|
|
|
diff_lines = diff_lines[:MAX_DIFF_LINES]
|
|
|
|
|
truncated = True
|
|
|
|
|
text = "\n".join(diff_lines)
|
|
|
|
|
if truncated:
|
|
|
|
|
text += f"\n… diff truncated at {MAX_DIFF_LINES} lines"
|
|
|
|
|
return {
|
|
|
|
|
"text": text,
|
|
|
|
|
"added": added,
|
|
|
|
|
"removed": removed,
|
|
|
|
|
"new_file": old == "",
|
|
|
|
|
"file": os.path.basename(path) or (path or "file"),
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
class EditFileTool:
|
|
|
|
|
async def execute(self, content: str, ctx: dict) -> dict:
|
feat(agent): confine agent file/shell tools to a selectable workspace (#3665)
* feat(agent): workspace confinement via context-local binding + get_workspace tool
Bind the per-turn workspace once in execute_tool_block; the shared path
resolvers (_resolve_tool_path / _resolve_search_root) and the subprocess cwd
helper (agent_cwd) read it, so file tools + bash/python are confined centrally
and a new tool that uses the shared helpers cannot accidentally bypass it.
Adds the admin-gated /api/workspace/browse picker, a workspace pill + directory
modal (reusing existing modal/button CSS), the /workspace slash command, and a
get_workspace tool (replaces a system-prompt block). Confinement is OS-agnostic
(realpath/normcase/commonpath) and docker-safe (container paths, no host
assumptions). Reopens #2023.
* ux(workspace): clarify workspace is not a sandbox
Picker modal note + pill tooltip + get_workspace tool/output wording now state
plainly: read_file/write_file/edit_file/grep/glob/ls are confined to the folder,
but bash/python only start there (cwd) and are not sandboxed. Modal note reuses
the existing .muted class.
* fix(agent): treat an active workspace as file-work intent
A vague low-signal message (e.g. "look at the local project") matches no
domain keywords, so tool retrieval is skipped and only always-available tools
are offered — leaving the agent with no file access even though a workspace is
set. When a workspace is active, include the file/code tools (incl.
get_workspace) on low-signal turns so the agent can act on the folder.
Also requires the tool index (ChromaDB) to be reachable for normal retrieval;
that is an environment dependency, not part of this change.
* ux(workspace): hide pill + overflow entry in chat mode
Workspace only scopes the agent's file/shell tools, so the pill and the
overflow 'Workspace' entry are agent-only now — hidden in chat mode like the
bash toggle. Mode read from the DOM in syncWorkspaceIndicator; applyMode() is
called from the agent/chat setMode handler.
* prompt(tools): steer bash/python to defer to the dedicated file tools
bash/python schema descriptions (what native-tool-calling models read) were
bare and gave no steer, so models would do file ops via the shell (e.g. writing
SVG/HTML, which then dumps raw markup into the tool preview). Tell bash/python
in the schema + tool-index + prompt section to prefer read_file/write_file/
edit_file/grep/glob/ls and only be used for what those do not cover.
* prompt(tools): keep bash/python deferral generic (no hardcoded tool names)
Reference 'a dedicated tool' rather than listing read_file/write_file/grep/etc.
by name, so the guidance does not go stale if those tools are renamed.
* style(workspace): drop em-dashes from added code comments/strings
* ux(workspace): terser non-sandbox note in picker (no tool-name list)
* ux(workspace): mirror terse non-sandbox wording in pill tooltip
* chore: untrack local venv symlink (run-only, not part of the feature)
* prompt(workspace): keep get_workspace text generic (no hardcoded tool names)
* fix(agent): low-signal + workspace surfaces only read-only file tools
Intersect the files tool group with PLAN_MODE_READONLY_TOOLS so a vague message
in a workspace exposes read_file/grep/glob/ls/get_workspace for exploration, but
not write_file/edit_file/bash/python -- those wait for a request that actually
calls for them (RAG retrieval still adds them on a real ask).
* feat(workspace): cap browse listing at 500 dirs with a truncated hint
Mirror the filesystem_tools._CODENAV_MAX_HITS pattern with a module-local
_MAX_BROWSE_DIRS so a directory with thousands of children does not dump every
row into the picker; the response carries a truncated flag and the modal tells
the user to type a path to jump in.
* chore: untrack local venv symlink (run-only artifact)
* fix(workspace): vet the workspace root against the sensitive-path deny list at bind time
The in-workspace resolver deny-lists sensitive paths inside the workspace,
but the empty-path search root is the workspace itself, so a workspace of
~/.ssh could be listed via ls with no path. vet_workspace() (public, in
tool_execution next to the resolvers) rejects non-directories and sensitive
roots before the path is ever bound; chat_routes uses it instead of its
inline isdir check.
* fix(workspace): reject filesystem roots and stop showing rejected workspaces as active
Review findings from #3665:
P2: vet_workspace accepted / (and would accept drive/UNC roots), which makes
every absolute path 'inside' the workspace and collapses confinement into
host-wide file access. A root is its own dirname, so reject when
dirname(resolved) == resolved; the browse response now carries a selectable
flag and the picker disables 'Use this folder' on unselectable dirs.
P3: /workspace set stored any string client-side and the chat route silently
dropped rejected values, so the pill could claim a confinement that was not
in effect. New admin-gated /api/workspace/vet validates manual paths before
they persist (canonical path returned), and when a posted workspace is
rejected at send time the stream emits workspace_rejected so the client
clears the stored value and toasts instead of continuing silently.
* fix(workspace): check caller privilege before vetting the posted workspace
Review finding: /api/chat_stream called vet_workspace() on the posted value
for every caller and emitted workspace_rejected on failure, so a non-admin
who can chat but cannot use file/shell tools could distinguish existing
directories from missing/file/sensitive/root paths by whether the event
appeared. The resolution now lives in _resolve_request_workspace, which
drops the submitted value uniformly for non-admin callers, with no vetting
and no event, before the path ever touches the filesystem. Admin and
single-user behavior is unchanged. Test pins that valid and invalid paths
are indistinguishable for a non-admin and that vet_workspace is never
invoked for them.
2026-06-11 18:17:54 +02:00
|
|
|
from src.tool_execution import _resolve_tool_path, _resolve_search_root, _truncate
|
2026-06-09 19:05:36 +05:30
|
|
|
try:
|
|
|
|
|
args = json.loads(content) if content.strip().startswith("{") else {}
|
|
|
|
|
except (json.JSONDecodeError, TypeError):
|
|
|
|
|
args = {}
|
|
|
|
|
raw_path = (args.get("path") or "").strip()
|
|
|
|
|
old = args.get("old_string", "")
|
|
|
|
|
new = args.get("new_string", "")
|
|
|
|
|
replace_all = bool(args.get("replace_all", False))
|
|
|
|
|
if not raw_path:
|
|
|
|
|
return {"error": "edit_file: path required", "exit_code": 1}
|
|
|
|
|
try:
|
feat(agent): confine agent file/shell tools to a selectable workspace (#3665)
* feat(agent): workspace confinement via context-local binding + get_workspace tool
Bind the per-turn workspace once in execute_tool_block; the shared path
resolvers (_resolve_tool_path / _resolve_search_root) and the subprocess cwd
helper (agent_cwd) read it, so file tools + bash/python are confined centrally
and a new tool that uses the shared helpers cannot accidentally bypass it.
Adds the admin-gated /api/workspace/browse picker, a workspace pill + directory
modal (reusing existing modal/button CSS), the /workspace slash command, and a
get_workspace tool (replaces a system-prompt block). Confinement is OS-agnostic
(realpath/normcase/commonpath) and docker-safe (container paths, no host
assumptions). Reopens #2023.
* ux(workspace): clarify workspace is not a sandbox
Picker modal note + pill tooltip + get_workspace tool/output wording now state
plainly: read_file/write_file/edit_file/grep/glob/ls are confined to the folder,
but bash/python only start there (cwd) and are not sandboxed. Modal note reuses
the existing .muted class.
* fix(agent): treat an active workspace as file-work intent
A vague low-signal message (e.g. "look at the local project") matches no
domain keywords, so tool retrieval is skipped and only always-available tools
are offered — leaving the agent with no file access even though a workspace is
set. When a workspace is active, include the file/code tools (incl.
get_workspace) on low-signal turns so the agent can act on the folder.
Also requires the tool index (ChromaDB) to be reachable for normal retrieval;
that is an environment dependency, not part of this change.
* ux(workspace): hide pill + overflow entry in chat mode
Workspace only scopes the agent's file/shell tools, so the pill and the
overflow 'Workspace' entry are agent-only now — hidden in chat mode like the
bash toggle. Mode read from the DOM in syncWorkspaceIndicator; applyMode() is
called from the agent/chat setMode handler.
* prompt(tools): steer bash/python to defer to the dedicated file tools
bash/python schema descriptions (what native-tool-calling models read) were
bare and gave no steer, so models would do file ops via the shell (e.g. writing
SVG/HTML, which then dumps raw markup into the tool preview). Tell bash/python
in the schema + tool-index + prompt section to prefer read_file/write_file/
edit_file/grep/glob/ls and only be used for what those do not cover.
* prompt(tools): keep bash/python deferral generic (no hardcoded tool names)
Reference 'a dedicated tool' rather than listing read_file/write_file/grep/etc.
by name, so the guidance does not go stale if those tools are renamed.
* style(workspace): drop em-dashes from added code comments/strings
* ux(workspace): terser non-sandbox note in picker (no tool-name list)
* ux(workspace): mirror terse non-sandbox wording in pill tooltip
* chore: untrack local venv symlink (run-only, not part of the feature)
* prompt(workspace): keep get_workspace text generic (no hardcoded tool names)
* fix(agent): low-signal + workspace surfaces only read-only file tools
Intersect the files tool group with PLAN_MODE_READONLY_TOOLS so a vague message
in a workspace exposes read_file/grep/glob/ls/get_workspace for exploration, but
not write_file/edit_file/bash/python -- those wait for a request that actually
calls for them (RAG retrieval still adds them on a real ask).
* feat(workspace): cap browse listing at 500 dirs with a truncated hint
Mirror the filesystem_tools._CODENAV_MAX_HITS pattern with a module-local
_MAX_BROWSE_DIRS so a directory with thousands of children does not dump every
row into the picker; the response carries a truncated flag and the modal tells
the user to type a path to jump in.
* chore: untrack local venv symlink (run-only artifact)
* fix(workspace): vet the workspace root against the sensitive-path deny list at bind time
The in-workspace resolver deny-lists sensitive paths inside the workspace,
but the empty-path search root is the workspace itself, so a workspace of
~/.ssh could be listed via ls with no path. vet_workspace() (public, in
tool_execution next to the resolvers) rejects non-directories and sensitive
roots before the path is ever bound; chat_routes uses it instead of its
inline isdir check.
* fix(workspace): reject filesystem roots and stop showing rejected workspaces as active
Review findings from #3665:
P2: vet_workspace accepted / (and would accept drive/UNC roots), which makes
every absolute path 'inside' the workspace and collapses confinement into
host-wide file access. A root is its own dirname, so reject when
dirname(resolved) == resolved; the browse response now carries a selectable
flag and the picker disables 'Use this folder' on unselectable dirs.
P3: /workspace set stored any string client-side and the chat route silently
dropped rejected values, so the pill could claim a confinement that was not
in effect. New admin-gated /api/workspace/vet validates manual paths before
they persist (canonical path returned), and when a posted workspace is
rejected at send time the stream emits workspace_rejected so the client
clears the stored value and toasts instead of continuing silently.
* fix(workspace): check caller privilege before vetting the posted workspace
Review finding: /api/chat_stream called vet_workspace() on the posted value
for every caller and emitted workspace_rejected on failure, so a non-admin
who can chat but cannot use file/shell tools could distinguish existing
directories from missing/file/sensitive/root paths by whether the event
appeared. The resolution now lives in _resolve_request_workspace, which
drops the submitted value uniformly for non-admin callers, with no vetting
and no event, before the path ever touches the filesystem. Admin and
single-user behavior is unchanged. Test pins that valid and invalid paths
are indistinguishable for a non-admin and that vet_workspace is never
invoked for them.
2026-06-11 18:17:54 +02:00
|
|
|
path = _resolve_tool_path(raw_path)
|
2026-06-09 19:05:36 +05:30
|
|
|
except ValueError as e:
|
|
|
|
|
return {"error": f"edit_file: {e}", "exit_code": 1}
|
|
|
|
|
if old == "":
|
|
|
|
|
return {"error": "edit_file: old_string required (use write_file to create a file)", "exit_code": 1}
|
|
|
|
|
if old == new:
|
|
|
|
|
return {"error": "edit_file: old_string and new_string are identical", "exit_code": 1}
|
|
|
|
|
|
|
|
|
|
def _apply():
|
|
|
|
|
"""Helper function that performs the actual string replacement and file writing logic."""
|
|
|
|
|
with open(path, "r", encoding="utf-8") as f:
|
|
|
|
|
original = f.read()
|
|
|
|
|
count = original.count(old)
|
|
|
|
|
if count == 0:
|
|
|
|
|
return original, None, "not_found"
|
|
|
|
|
if count > 1 and not replace_all:
|
|
|
|
|
return original, None, f"not_unique:{count}"
|
|
|
|
|
updated = original.replace(old, new) if replace_all else original.replace(old, new, 1)
|
|
|
|
|
with open(path, "w", encoding="utf-8") as f:
|
|
|
|
|
f.write(updated)
|
|
|
|
|
return original, updated, "ok"
|
|
|
|
|
|
|
|
|
|
try:
|
|
|
|
|
original, updated, status = await asyncio.to_thread(_apply)
|
|
|
|
|
except FileNotFoundError:
|
|
|
|
|
return {"error": f"edit_file: {path}: not found (use write_file to create it)", "exit_code": 1}
|
|
|
|
|
except (IsADirectoryError, UnicodeDecodeError):
|
|
|
|
|
return {"error": f"edit_file: {path}: not an editable text file", "exit_code": 1}
|
|
|
|
|
except PermissionError:
|
|
|
|
|
return {"error": f"edit_file: {path}: permission denied", "exit_code": 1}
|
|
|
|
|
except OSError as e:
|
|
|
|
|
return {"error": f"edit_file: {path}: {e}", "exit_code": 1}
|
|
|
|
|
|
|
|
|
|
if status == "not_found":
|
|
|
|
|
return {"error": f"edit_file: old_string not found in {path}. Read the file and match it exactly.", "exit_code": 1}
|
|
|
|
|
if status.startswith("not_unique"):
|
|
|
|
|
n = status.split(":", 1)[1]
|
|
|
|
|
return {"error": f"edit_file: old_string is not unique in {path} ({n} matches). Add surrounding context or set replace_all=true.", "exit_code": 1}
|
|
|
|
|
|
|
|
|
|
n = original.count(old)
|
|
|
|
|
result = {"output": f"Edited {path} ({n} replacement{'s' if n != 1 else ''})", "exit_code": 0}
|
|
|
|
|
diff = _unified_diff(original, updated, path)
|
|
|
|
|
if diff:
|
|
|
|
|
result["diff"] = diff
|
|
|
|
|
return result
|
|
|
|
|
|
|
|
|
|
class ReadFileTool:
|
|
|
|
|
async def execute(self, content: str, ctx: dict) -> dict:
|
feat(agent): confine agent file/shell tools to a selectable workspace (#3665)
* feat(agent): workspace confinement via context-local binding + get_workspace tool
Bind the per-turn workspace once in execute_tool_block; the shared path
resolvers (_resolve_tool_path / _resolve_search_root) and the subprocess cwd
helper (agent_cwd) read it, so file tools + bash/python are confined centrally
and a new tool that uses the shared helpers cannot accidentally bypass it.
Adds the admin-gated /api/workspace/browse picker, a workspace pill + directory
modal (reusing existing modal/button CSS), the /workspace slash command, and a
get_workspace tool (replaces a system-prompt block). Confinement is OS-agnostic
(realpath/normcase/commonpath) and docker-safe (container paths, no host
assumptions). Reopens #2023.
* ux(workspace): clarify workspace is not a sandbox
Picker modal note + pill tooltip + get_workspace tool/output wording now state
plainly: read_file/write_file/edit_file/grep/glob/ls are confined to the folder,
but bash/python only start there (cwd) and are not sandboxed. Modal note reuses
the existing .muted class.
* fix(agent): treat an active workspace as file-work intent
A vague low-signal message (e.g. "look at the local project") matches no
domain keywords, so tool retrieval is skipped and only always-available tools
are offered — leaving the agent with no file access even though a workspace is
set. When a workspace is active, include the file/code tools (incl.
get_workspace) on low-signal turns so the agent can act on the folder.
Also requires the tool index (ChromaDB) to be reachable for normal retrieval;
that is an environment dependency, not part of this change.
* ux(workspace): hide pill + overflow entry in chat mode
Workspace only scopes the agent's file/shell tools, so the pill and the
overflow 'Workspace' entry are agent-only now — hidden in chat mode like the
bash toggle. Mode read from the DOM in syncWorkspaceIndicator; applyMode() is
called from the agent/chat setMode handler.
* prompt(tools): steer bash/python to defer to the dedicated file tools
bash/python schema descriptions (what native-tool-calling models read) were
bare and gave no steer, so models would do file ops via the shell (e.g. writing
SVG/HTML, which then dumps raw markup into the tool preview). Tell bash/python
in the schema + tool-index + prompt section to prefer read_file/write_file/
edit_file/grep/glob/ls and only be used for what those do not cover.
* prompt(tools): keep bash/python deferral generic (no hardcoded tool names)
Reference 'a dedicated tool' rather than listing read_file/write_file/grep/etc.
by name, so the guidance does not go stale if those tools are renamed.
* style(workspace): drop em-dashes from added code comments/strings
* ux(workspace): terser non-sandbox note in picker (no tool-name list)
* ux(workspace): mirror terse non-sandbox wording in pill tooltip
* chore: untrack local venv symlink (run-only, not part of the feature)
* prompt(workspace): keep get_workspace text generic (no hardcoded tool names)
* fix(agent): low-signal + workspace surfaces only read-only file tools
Intersect the files tool group with PLAN_MODE_READONLY_TOOLS so a vague message
in a workspace exposes read_file/grep/glob/ls/get_workspace for exploration, but
not write_file/edit_file/bash/python -- those wait for a request that actually
calls for them (RAG retrieval still adds them on a real ask).
* feat(workspace): cap browse listing at 500 dirs with a truncated hint
Mirror the filesystem_tools._CODENAV_MAX_HITS pattern with a module-local
_MAX_BROWSE_DIRS so a directory with thousands of children does not dump every
row into the picker; the response carries a truncated flag and the modal tells
the user to type a path to jump in.
* chore: untrack local venv symlink (run-only artifact)
* fix(workspace): vet the workspace root against the sensitive-path deny list at bind time
The in-workspace resolver deny-lists sensitive paths inside the workspace,
but the empty-path search root is the workspace itself, so a workspace of
~/.ssh could be listed via ls with no path. vet_workspace() (public, in
tool_execution next to the resolvers) rejects non-directories and sensitive
roots before the path is ever bound; chat_routes uses it instead of its
inline isdir check.
* fix(workspace): reject filesystem roots and stop showing rejected workspaces as active
Review findings from #3665:
P2: vet_workspace accepted / (and would accept drive/UNC roots), which makes
every absolute path 'inside' the workspace and collapses confinement into
host-wide file access. A root is its own dirname, so reject when
dirname(resolved) == resolved; the browse response now carries a selectable
flag and the picker disables 'Use this folder' on unselectable dirs.
P3: /workspace set stored any string client-side and the chat route silently
dropped rejected values, so the pill could claim a confinement that was not
in effect. New admin-gated /api/workspace/vet validates manual paths before
they persist (canonical path returned), and when a posted workspace is
rejected at send time the stream emits workspace_rejected so the client
clears the stored value and toasts instead of continuing silently.
* fix(workspace): check caller privilege before vetting the posted workspace
Review finding: /api/chat_stream called vet_workspace() on the posted value
for every caller and emitted workspace_rejected on failure, so a non-admin
who can chat but cannot use file/shell tools could distinguish existing
directories from missing/file/sensitive/root paths by whether the event
appeared. The resolution now lives in _resolve_request_workspace, which
drops the submitted value uniformly for non-admin callers, with no vetting
and no event, before the path ever touches the filesystem. Admin and
single-user behavior is unchanged. Test pins that valid and invalid paths
are indistinguishable for a non-admin and that vet_workspace is never
invoked for them.
2026-06-11 18:17:54 +02:00
|
|
|
from src.tool_execution import _resolve_tool_path, _resolve_search_root, _truncate
|
2026-06-09 19:05:36 +05:30
|
|
|
raw_path, offset, limit = content.split("\n", 1)[0].strip(), 0, 0
|
|
|
|
|
_stripped = content.strip()
|
|
|
|
|
if _stripped.startswith("{"):
|
|
|
|
|
try:
|
|
|
|
|
_a = json.loads(_stripped)
|
|
|
|
|
raw_path = str(_a.get("path", "")).strip()
|
|
|
|
|
offset = int(_a.get("offset") or 0)
|
|
|
|
|
limit = int(_a.get("limit") or 0)
|
|
|
|
|
except (json.JSONDecodeError, TypeError, ValueError):
|
|
|
|
|
pass
|
|
|
|
|
try:
|
feat(agent): confine agent file/shell tools to a selectable workspace (#3665)
* feat(agent): workspace confinement via context-local binding + get_workspace tool
Bind the per-turn workspace once in execute_tool_block; the shared path
resolvers (_resolve_tool_path / _resolve_search_root) and the subprocess cwd
helper (agent_cwd) read it, so file tools + bash/python are confined centrally
and a new tool that uses the shared helpers cannot accidentally bypass it.
Adds the admin-gated /api/workspace/browse picker, a workspace pill + directory
modal (reusing existing modal/button CSS), the /workspace slash command, and a
get_workspace tool (replaces a system-prompt block). Confinement is OS-agnostic
(realpath/normcase/commonpath) and docker-safe (container paths, no host
assumptions). Reopens #2023.
* ux(workspace): clarify workspace is not a sandbox
Picker modal note + pill tooltip + get_workspace tool/output wording now state
plainly: read_file/write_file/edit_file/grep/glob/ls are confined to the folder,
but bash/python only start there (cwd) and are not sandboxed. Modal note reuses
the existing .muted class.
* fix(agent): treat an active workspace as file-work intent
A vague low-signal message (e.g. "look at the local project") matches no
domain keywords, so tool retrieval is skipped and only always-available tools
are offered — leaving the agent with no file access even though a workspace is
set. When a workspace is active, include the file/code tools (incl.
get_workspace) on low-signal turns so the agent can act on the folder.
Also requires the tool index (ChromaDB) to be reachable for normal retrieval;
that is an environment dependency, not part of this change.
* ux(workspace): hide pill + overflow entry in chat mode
Workspace only scopes the agent's file/shell tools, so the pill and the
overflow 'Workspace' entry are agent-only now — hidden in chat mode like the
bash toggle. Mode read from the DOM in syncWorkspaceIndicator; applyMode() is
called from the agent/chat setMode handler.
* prompt(tools): steer bash/python to defer to the dedicated file tools
bash/python schema descriptions (what native-tool-calling models read) were
bare and gave no steer, so models would do file ops via the shell (e.g. writing
SVG/HTML, which then dumps raw markup into the tool preview). Tell bash/python
in the schema + tool-index + prompt section to prefer read_file/write_file/
edit_file/grep/glob/ls and only be used for what those do not cover.
* prompt(tools): keep bash/python deferral generic (no hardcoded tool names)
Reference 'a dedicated tool' rather than listing read_file/write_file/grep/etc.
by name, so the guidance does not go stale if those tools are renamed.
* style(workspace): drop em-dashes from added code comments/strings
* ux(workspace): terser non-sandbox note in picker (no tool-name list)
* ux(workspace): mirror terse non-sandbox wording in pill tooltip
* chore: untrack local venv symlink (run-only, not part of the feature)
* prompt(workspace): keep get_workspace text generic (no hardcoded tool names)
* fix(agent): low-signal + workspace surfaces only read-only file tools
Intersect the files tool group with PLAN_MODE_READONLY_TOOLS so a vague message
in a workspace exposes read_file/grep/glob/ls/get_workspace for exploration, but
not write_file/edit_file/bash/python -- those wait for a request that actually
calls for them (RAG retrieval still adds them on a real ask).
* feat(workspace): cap browse listing at 500 dirs with a truncated hint
Mirror the filesystem_tools._CODENAV_MAX_HITS pattern with a module-local
_MAX_BROWSE_DIRS so a directory with thousands of children does not dump every
row into the picker; the response carries a truncated flag and the modal tells
the user to type a path to jump in.
* chore: untrack local venv symlink (run-only artifact)
* fix(workspace): vet the workspace root against the sensitive-path deny list at bind time
The in-workspace resolver deny-lists sensitive paths inside the workspace,
but the empty-path search root is the workspace itself, so a workspace of
~/.ssh could be listed via ls with no path. vet_workspace() (public, in
tool_execution next to the resolvers) rejects non-directories and sensitive
roots before the path is ever bound; chat_routes uses it instead of its
inline isdir check.
* fix(workspace): reject filesystem roots and stop showing rejected workspaces as active
Review findings from #3665:
P2: vet_workspace accepted / (and would accept drive/UNC roots), which makes
every absolute path 'inside' the workspace and collapses confinement into
host-wide file access. A root is its own dirname, so reject when
dirname(resolved) == resolved; the browse response now carries a selectable
flag and the picker disables 'Use this folder' on unselectable dirs.
P3: /workspace set stored any string client-side and the chat route silently
dropped rejected values, so the pill could claim a confinement that was not
in effect. New admin-gated /api/workspace/vet validates manual paths before
they persist (canonical path returned), and when a posted workspace is
rejected at send time the stream emits workspace_rejected so the client
clears the stored value and toasts instead of continuing silently.
* fix(workspace): check caller privilege before vetting the posted workspace
Review finding: /api/chat_stream called vet_workspace() on the posted value
for every caller and emitted workspace_rejected on failure, so a non-admin
who can chat but cannot use file/shell tools could distinguish existing
directories from missing/file/sensitive/root paths by whether the event
appeared. The resolution now lives in _resolve_request_workspace, which
drops the submitted value uniformly for non-admin callers, with no vetting
and no event, before the path ever touches the filesystem. Admin and
single-user behavior is unchanged. Test pins that valid and invalid paths
are indistinguishable for a non-admin and that vet_workspace is never
invoked for them.
2026-06-11 18:17:54 +02:00
|
|
|
path = _resolve_tool_path(raw_path)
|
2026-06-09 19:05:36 +05:30
|
|
|
except ValueError as e:
|
|
|
|
|
return {"error": f"read_file: {e}", "exit_code": 1}
|
|
|
|
|
try:
|
|
|
|
|
def _read():
|
|
|
|
|
if offset > 0 or limit > 0:
|
|
|
|
|
start = max(offset, 1)
|
|
|
|
|
out, n, budget = [], 0, MAX_READ_CHARS
|
|
|
|
|
with open(path, "r", encoding="utf-8", errors="replace") as f:
|
|
|
|
|
for i, line in enumerate(f, 1):
|
|
|
|
|
if i < start:
|
|
|
|
|
continue
|
|
|
|
|
if limit > 0 and n >= limit:
|
|
|
|
|
break
|
|
|
|
|
out.append(line)
|
|
|
|
|
n += 1
|
|
|
|
|
budget -= len(line)
|
|
|
|
|
if budget <= 0:
|
|
|
|
|
out.append(f"\n... [truncated at {MAX_READ_CHARS} chars]")
|
|
|
|
|
break
|
|
|
|
|
return "".join(out)
|
|
|
|
|
with open(path, "r", encoding="utf-8", errors="replace") as f:
|
|
|
|
|
return f.read(MAX_READ_CHARS + 1)
|
|
|
|
|
data = await asyncio.to_thread(_read)
|
|
|
|
|
except FileNotFoundError:
|
|
|
|
|
return {"error": f"read_file: {path}: not found", "exit_code": 1}
|
|
|
|
|
except PermissionError:
|
|
|
|
|
return {"error": f"read_file: {path}: permission denied", "exit_code": 1}
|
|
|
|
|
except IsADirectoryError:
|
|
|
|
|
return {"error": f"read_file: {path}: is a directory (use ls)", "exit_code": 1}
|
|
|
|
|
except OSError as e:
|
|
|
|
|
return {"error": f"read_file: {path}: {e}", "exit_code": 1}
|
|
|
|
|
if not (offset > 0 or limit > 0) and len(data) > MAX_READ_CHARS:
|
|
|
|
|
data = data[:MAX_READ_CHARS] + f"\n... [truncated at {MAX_READ_CHARS} chars]"
|
|
|
|
|
return {"output": data, "exit_code": 0}
|
|
|
|
|
|
|
|
|
|
class WriteFileTool:
|
|
|
|
|
async def execute(self, content: str, ctx: dict) -> dict:
|
feat(agent): confine agent file/shell tools to a selectable workspace (#3665)
* feat(agent): workspace confinement via context-local binding + get_workspace tool
Bind the per-turn workspace once in execute_tool_block; the shared path
resolvers (_resolve_tool_path / _resolve_search_root) and the subprocess cwd
helper (agent_cwd) read it, so file tools + bash/python are confined centrally
and a new tool that uses the shared helpers cannot accidentally bypass it.
Adds the admin-gated /api/workspace/browse picker, a workspace pill + directory
modal (reusing existing modal/button CSS), the /workspace slash command, and a
get_workspace tool (replaces a system-prompt block). Confinement is OS-agnostic
(realpath/normcase/commonpath) and docker-safe (container paths, no host
assumptions). Reopens #2023.
* ux(workspace): clarify workspace is not a sandbox
Picker modal note + pill tooltip + get_workspace tool/output wording now state
plainly: read_file/write_file/edit_file/grep/glob/ls are confined to the folder,
but bash/python only start there (cwd) and are not sandboxed. Modal note reuses
the existing .muted class.
* fix(agent): treat an active workspace as file-work intent
A vague low-signal message (e.g. "look at the local project") matches no
domain keywords, so tool retrieval is skipped and only always-available tools
are offered — leaving the agent with no file access even though a workspace is
set. When a workspace is active, include the file/code tools (incl.
get_workspace) on low-signal turns so the agent can act on the folder.
Also requires the tool index (ChromaDB) to be reachable for normal retrieval;
that is an environment dependency, not part of this change.
* ux(workspace): hide pill + overflow entry in chat mode
Workspace only scopes the agent's file/shell tools, so the pill and the
overflow 'Workspace' entry are agent-only now — hidden in chat mode like the
bash toggle. Mode read from the DOM in syncWorkspaceIndicator; applyMode() is
called from the agent/chat setMode handler.
* prompt(tools): steer bash/python to defer to the dedicated file tools
bash/python schema descriptions (what native-tool-calling models read) were
bare and gave no steer, so models would do file ops via the shell (e.g. writing
SVG/HTML, which then dumps raw markup into the tool preview). Tell bash/python
in the schema + tool-index + prompt section to prefer read_file/write_file/
edit_file/grep/glob/ls and only be used for what those do not cover.
* prompt(tools): keep bash/python deferral generic (no hardcoded tool names)
Reference 'a dedicated tool' rather than listing read_file/write_file/grep/etc.
by name, so the guidance does not go stale if those tools are renamed.
* style(workspace): drop em-dashes from added code comments/strings
* ux(workspace): terser non-sandbox note in picker (no tool-name list)
* ux(workspace): mirror terse non-sandbox wording in pill tooltip
* chore: untrack local venv symlink (run-only, not part of the feature)
* prompt(workspace): keep get_workspace text generic (no hardcoded tool names)
* fix(agent): low-signal + workspace surfaces only read-only file tools
Intersect the files tool group with PLAN_MODE_READONLY_TOOLS so a vague message
in a workspace exposes read_file/grep/glob/ls/get_workspace for exploration, but
not write_file/edit_file/bash/python -- those wait for a request that actually
calls for them (RAG retrieval still adds them on a real ask).
* feat(workspace): cap browse listing at 500 dirs with a truncated hint
Mirror the filesystem_tools._CODENAV_MAX_HITS pattern with a module-local
_MAX_BROWSE_DIRS so a directory with thousands of children does not dump every
row into the picker; the response carries a truncated flag and the modal tells
the user to type a path to jump in.
* chore: untrack local venv symlink (run-only artifact)
* fix(workspace): vet the workspace root against the sensitive-path deny list at bind time
The in-workspace resolver deny-lists sensitive paths inside the workspace,
but the empty-path search root is the workspace itself, so a workspace of
~/.ssh could be listed via ls with no path. vet_workspace() (public, in
tool_execution next to the resolvers) rejects non-directories and sensitive
roots before the path is ever bound; chat_routes uses it instead of its
inline isdir check.
* fix(workspace): reject filesystem roots and stop showing rejected workspaces as active
Review findings from #3665:
P2: vet_workspace accepted / (and would accept drive/UNC roots), which makes
every absolute path 'inside' the workspace and collapses confinement into
host-wide file access. A root is its own dirname, so reject when
dirname(resolved) == resolved; the browse response now carries a selectable
flag and the picker disables 'Use this folder' on unselectable dirs.
P3: /workspace set stored any string client-side and the chat route silently
dropped rejected values, so the pill could claim a confinement that was not
in effect. New admin-gated /api/workspace/vet validates manual paths before
they persist (canonical path returned), and when a posted workspace is
rejected at send time the stream emits workspace_rejected so the client
clears the stored value and toasts instead of continuing silently.
* fix(workspace): check caller privilege before vetting the posted workspace
Review finding: /api/chat_stream called vet_workspace() on the posted value
for every caller and emitted workspace_rejected on failure, so a non-admin
who can chat but cannot use file/shell tools could distinguish existing
directories from missing/file/sensitive/root paths by whether the event
appeared. The resolution now lives in _resolve_request_workspace, which
drops the submitted value uniformly for non-admin callers, with no vetting
and no event, before the path ever touches the filesystem. Admin and
single-user behavior is unchanged. Test pins that valid and invalid paths
are indistinguishable for a non-admin and that vet_workspace is never
invoked for them.
2026-06-11 18:17:54 +02:00
|
|
|
from src.tool_execution import _resolve_tool_path, _resolve_search_root, _truncate
|
2026-06-09 19:05:36 +05:30
|
|
|
lines = content.split("\n", 1)
|
|
|
|
|
raw_path = lines[0].strip()
|
|
|
|
|
body = lines[1] if len(lines) > 1 else ""
|
fix(agent): execute fenced tool calls with inline args and route bare email tool names (#3681)
* fix(agent): execute fenced tool calls with inline args and bare email tool names
Two bugs made local (Ollama) models unable to use email tools, leaving
raw fences like ```list_email_accounts {}``` in the chat:
1. _TOOL_BLOCK_RE required a newline right after the fence tag, so a
tool call with args on the same line ("```list_email_accounts {}")
never matched and was never executed. The fence now matches with
optional spaces/newline after the tag.
2. Even when parsed, bare email tool names had no dispatch branch in
tool_execution.py and fell through to "Unknown tool type". They now
route to the email MCP server as mcp__email__<name>, matching how
function_call_to_tool_block already maps them for native callers.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* fix(security): block all bare email tool names for non-admins; harden fence-tag regex
Review follow-up on #3681 (thanks @vgalin):
1. Routing bare email names made 10 of the 14 email tools executable by
non-admin owners — is_public_blocked_tool() runs on the bare name
before dispatch, and NON_ADMIN_BLOCKED_TOOLS only listed 4. Define the
full email tool set once (BUILTIN_EMAIL_TOOLS in tool_security.py) and
derive the blocklist, the fence tags (TOOL_TAGS), the bare-name
dispatch, and the native-call mapping from it so they can't drift.
This also fixes 4 tools (search_emails, draft_email, draft_email_reply,
ai_draft_email_reply) that were missing from the old tool_schemas copy
and therefore unreachable even for native function-calling models.
2. The relaxed fence regex from the previous commit could prefix-match
longer fence tags: ```python3 parsed as tool "python" with content
"3\nprint(...)" and executed as code. Add a (?![\w-]) boundary after
the tag.
Tests: test_public_agent_policy_blocks_sensitive_tools now covers all 14
bare email names + the mcp__email__ form; new tests/test_fenced_inline_args.py
pins inline-args parsing, the python3/hyphenated-tag non-matches, and
strip/parse display mirroring.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
* fix(security): gate bare and mcp-qualified email names together; stop executing Markdown info strings
Review follow-up on #3681 (thanks @RaresKeY):
1. P1: execute_tool_block() checked disabled_tools / the turn ToolPolicy
only against the incoming block name, then the bare-email branch
qualified it to mcp__email__<name> and called the MCP manager. Plan
mode and the MCP settings toggle write the QUALIFIED name into the
denylist, so a bare fence like ```list_emails``` sailed past a
mcp__email__list_emails entry. Both gates now match on both
spellings (bare <-> mcp__email__-qualified), in either direction.
2. P2: the relaxed fence regex accepted arbitrary same-line text after
a recognized tag, which made ordinary Markdown info strings
executable: ```python title="example.py" ran as a python tool call.
Same-line content now only counts as tool input when it starts with
{ or [ (JSON args); anything else leaves the fence as display text,
and strip_tool_blocks mirrors that (the fence stays visible).
Tests: disabled-tools alias regression (qualified entry blocks bare
name and vice versa, never reaching the MCP manager), ToolPolicy alias
regression, python/bash title="..." non-execution + display retention,
and inline JSON-array args still parsing.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
* fix(security): reject brace-style fence metadata; cover the full email set in the friendly toggle
Review follow-up round 3 on #3681 (thanks @RaresKeY):
1. Brace-style fence metadata no longer executes. The previous narrowing
still treated any same-line {/[ after a recognized tag as tool input,
so ```bash {title="setup"} ran as a bash call. The fence header is now
captured separately and judged by one predicate shared between
parse_tool_blocks and strip_tool_blocks (_fenced_tool_call), so the
execute and display decisions can't disagree: same-line content only
counts as inline args when the tag is NOT a code tag (bash/python
never take same-line args — that text is Markdown fence attributes)
AND the inline text (plus any continuation lines) parses as standalone
JSON. ```bash {title="setup"}, ```python {"title":"example.py"} and
```list_emails {title="x"} all stay visible and inert.
2. The friendly `disable_tool email` toggle covered 3 of the 14 email
tools (mcp__email__{list_emails,read_email,send_email}); the other
bare aliases this PR routes stayed executable after an operator
disabled email. The alias now derives from BUILTIN_EMAIL_TOOLS in
BOTH spellings — bare (function-schema hiding, bare-fence dispatch)
and mcp__email__* (MCP schema hiding, qualified runtime blocks) —
so the toggle and the runtime gate can't drift apart.
Tests: brace/bracket metadata regressions for parse and strip symmetry
(code tags, invalid-JSON inline on a JSON tool, multi-line inline JSON
still parsing), and disable_tool/enable_tool email covering all 14 names
in both spellings.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
* fix(email): close remaining email-tool registry drift; classify every email tool for plan mode
Deep self-review follow-up on #3681. Three review rounds each found another
hand-maintained copy of the email tool list that had drifted; this commit
hunts down ALL remaining copies and pins them to BUILTIN_EMAIL_TOOLS.
The same 5 tools (search_emails, draft_email, draft_email_reply,
ai_draft_email_reply, download_attachment) were missing from every
advertising surface, so they were dispatchable but never offered:
- FUNCTION_TOOL_SCHEMAS: native function-calling models never saw them
(the round-1 fix covered dispatch only); schemas added, mirroring the
email server's inputSchema definitions.
- TOOL_SECTIONS: fenced-block models were never told about them; prompt
sections added.
- tool_index: absent from the RAG embedding registry (never retrievable),
the email keyword hints, and the scheduled assistant's always-available
set — the latter two now derive from BUILTIN_EMAIL_TOOLS.
- agent_loop._DOMAIN_TOOL_MAP["email"], tool_policy._COMMON_TOOL_NAMES,
the assistant tool-selector UI groups (assistant.js), and the default
Assistant crew seed (task_scheduler) now derive from / cover the set.
Plan mode now classifies every email tool explicitly:
- list_email_accounts and search_emails join PLAN_MODE_READONLY_TOOLS.
Without this, list_email_accounts sat in the plan-mode bare denylist
(schema-derived) while its qualified form passed the MCP read-only
filter — and the round-2 bare/qualified alias gate would have blocked
the qualified call too, regressing read-only email discovery in plan
mode.
- draft_email, draft_email_reply, ai_draft_email_reply, and
download_attachment join the fail-closed mutator backstop (drafts
create documents; download_attachment writes to disk).
Tests: tests/test_email_registry_sync.py pins every registry (including
the email server source and assistant.js) to BUILTIN_EMAIL_TOOLS and
asserts the plan-mode partition, so the next email tool can't drift; a
parse/strip mirror grid covers 192 fence shapes (tag x header x body)
asserting executed <=> stripped.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
* refactor: move the email alias rule into tool_security; extract the assistant seed constant
Code-quality pass over the PR's own changes:
- The bare<->qualified email aliasing rule lived inline in the generic
dispatcher (_execute_tool_block_impl). It is policy knowledge, so it
moves next to BUILTIN_EMAIL_TOOLS as email_tool_policy_names(); the
dispatcher just consumes it, and the rule gets its own unit test
(including the mcp__email__<not-a-tool> and mcp__other__ non-alias
cases).
- The default Assistant's enabled_tools list was an inline literal
inside the CrewMember seed, and its registry-sync test asserted a
source-code substring. Extracted to DEFAULT_ASSISTANT_ENABLED_TOOLS
so the test imports and checks the actual value.
- _fenced_tool_call return type tightened to Optional[Tuple[str, str]].
No behavior change; suite green (3295 passed).
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
* revert: move the email registry consolidation to a follow-up PR
Per review feedback on scope, this PR stays narrow: fenced inline-args
parsing, bare email tool routing, and the directly required safety
gates. This commit reverts the registry/advertising consolidation from
db29046 and 016ce47 (native schemas, prompt sections, RAG description
index + keyword hints, assistant always-available set, guide-only
known-names union, frontend tool-selector groups, default assistant
seed, and their sync tests) — all of that moves to a dedicated
follow-up PR together with the _EMAIL_TOOL_HINTS finding.
Kept here because the narrow scope needs them:
- email_tool_policy_names() in tool_security + its use in the
execute_tool_block gates and its unit test (refactor of this PR's own
round-2 alias fix),
- list_email_accounts in PLAN_MODE_READONLY_TOOLS (the alias gate works
both ways, and the schema-derived plan-mode bare denylist would
otherwise block the qualified read-only call too),
- the parse/strip mirror grid test (parser scope),
- the narrow registry sync tests (email server <-> BUILTIN_EMAIL_TOOLS
match, fence-tag coverage, non-admin blocklist coverage).
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
* fix(email): execute empty email fences with empty args; reject non-object JSON args
Two gaps found by replaying captured local-model traffic against the
narrowed branch:
1. ```list_email_accounts``` with NO body — a shape gemma really emits
for no-arg tools — was silently dropped (parse skips empty content),
so the model concluded email was broken: the original #337 symptom
through a different door. Empty fences whose tag is a built-in email
tool now dispatch with {} args and the tool's own validation answers
(e.g. an empty send_email returns "to is required" instead of
silence). Empty bash/python/other fences keep skipping, and strip
stays mirrored (the fence was executed, so it is removed).
2. The fence parser accepts JSON arrays as inline args, but the email
dispatch parsed only objects — an array silently became {} args.
Non-object JSON now returns a correctable "arguments must be a JSON
object" error before reaching the MCP server (same class as #3966).
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
* fix(security): classify all email tools for plan mode statically; reject invalid email JSON bodies
Review follow-up round 5 on #3681 (thanks @RaresKeY):
1. This PR makes every BUILTIN_EMAIL_TOOLS name fence-taggable, so each
one must be explicitly classified for plan mode — the draft tools and
download_attachment were in neither the read-only allowlist nor the
static denylist, leaving their bare-alias plan-mode safety dependent
on the MCP read-only inventory being present and current.
search_emails joins PLAN_MODE_READONLY_TOOLS (explicit, not
allowed-by-omission); draft_email, draft_email_reply,
ai_draft_email_reply, and download_attachment join the fail-closed
_PLAN_MODE_KNOWN_MUTATORS backstop. (Moved back from the #4053 split:
the partition is directly required for this PR to merge
independently.)
2. The classic tag/body fence form reaches execution unvalidated (only
INLINE args are JSON-checked by the parser), so a body like
{account: "work"} silently became {} args and read the DEFAULT
mailbox instead of the intended one. JSON-looking bodies that fail to
parse now return a correctable "not valid JSON" error before reaching
the MCP server.
Tests: a partition invariant (every email tool is explicitly read-only
or plan-mode-denied), a mutating-alias probe that uses only the static
denylist with a fake MCP manager (no inventory layer), and the
body-form invalid-JSON regression.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
* fix(tool-dispatch): decode inline JSON args for legacy MCP tools; reject all non-object email bodies
Review follow-up round 6 on #3681 (thanks @RaresKeY) — both pre-existing
on this branch, surfaced by the relaxed inline-args parser:
1. The relaxed parser accepts inline JSON for every non-code tag, but
the legacy line-based arg builders (web_search/web_fetch/read_file/
write_file/generate_image/manage_memory) wrapped the whole JSON
string as the query/url/path/prompt — so `web_search {"query": "x"}`
executed as a search for the literal string `{"query": "x"}`.
_build_mcp_args now uses a fenced JSON object directly when it carries
the tool's primary arg key (query/url/path/prompt/action). Keyed off
membership so it can't drift; an object without the primary key (e.g.
a freeform JSON query, or bare object content for write_file) falls
through to the line parser unchanged. Also fixes the same corruption
for the classic newline-JSON form.
2. The bare-email dispatch only rejected bodies starting with { or [, so
a non-empty non-JSON body like `account: work` still fell through to
{} args and silently read the DEFAULT mailbox. Now ANY non-empty body
must decode to a JSON object or it returns a correctable error; only a
truly empty body keeps the no-arg path (```list_email_accounts```).
Tests: inline-JSON arg decoding for the five legacy tools plus the
freeform and missing-primary-key fallbacks; the email body rejection
extended to cover the brace-looking and bare `key: value` shapes.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
* fix(tool-dispatch): drop dead manage_memory JSON-decode entry; pin the live-path invariant
Self-audit catch on the round-6 fix. manage_memory was added to
_MCP_JSON_PRIMARY_KEYS, but _build_mcp_args is only reached via
_call_mcp_tool, which only runs for _MCP_TOOL_MAP tools — and
manage_memory isn't one (its tag routes through dispatch_ai_tool ->
do_manage_memory, which line-parses). So the round-6 decode for
manage_memory was dead code: the unit test exercising _build_mcp_args
passed while a real `manage_memory {"action": ...}` fence still parsed
the whole JSON blob as the action.
Remove the dead entry and add test_mcp_json_primary_keys_are_all_live,
which asserts every JSON-primary tool is in _MCP_TOOL_MAP so a dead
decode can't be added again. The same inline-JSON corruption for
manage_memory and the other tools that route through positional
dispatchers (create_session, ui_control, send_to_session, search_chats,
the document tools, etc.) is pre-existing (dev corrupts their newline
JSON form too) and tracked separately; the proper fix there is to route
fenced JSON through function_call_to_tool_block.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
* fix(tool-dispatch): decode inline JSON in WriteFileTool (its live path); round-6 fix was on the dead MCP path
Self-audit: round 6 claimed to fix inline JSON args for write_file via
_build_mcp_args, but there is no filesystem MCP server, so write_file
always runs through _direct_fallback -> WriteFileTool, never through
_build_mcp_args. WriteFileTool — unlike its siblings ReadFileTool /
WebSearchTool / WebFetchTool, which all decode JSON — took lines[0] as
the path, so `write_file {"path": "/tmp/x", "content": "y"}` wrote to a
file literally named with the JSON blob. The round-6 _build_mcp_args
entry decoded correctly but on a path that never executes (same class
as the manage_memory dead entry), and the round-6 unit test passed on
that dead path.
WriteFileTool now decodes a JSON object carrying "path" (matching
ReadFileTool directly above it), and the comment on _MCP_JSON_PRIMARY_KEYS
records that only generate_image has a live MCP server today — the other
entries are defense-in-depth for the MCP path; the live fix for each
server-less tool is in its handler.
Test: test_write_file_inline_json_args drives the LIVE path
(execute_tool_block with no MCP) and asserts the intended path is used —
verified to fail without the handler fix. web_search/web_fetch/read_file
were already correct (their handlers decode); write_file was the gap.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
* test(strip-fence): derive the live-strip TOOL_TAGS from the real set
Semantic conflict from the dev merge that textual auto-merge didn't flag:
dev added test_live_strip_email_tool_fences.py whose _tool_tags() helper
source-scrapes only the TOOL_TAGS literal `{...}`, which worked on dev
because the email tool names were listed inline there. This branch makes
TOOL_TAGS the single source — `{...} | BUILTIN_EMAIL_TOOLS` — so the email
names are no longer in the literal and the scraper missed them, leaving the
email-fence strip assertions failing even though TOOL_TAGS does contain them
at runtime.
Import the real TOOL_TAGS instead of scraping source, so the test mirrors
exactly what GET /api/tools serves (sorted(TOOL_TAGS)) and the live
EXEC_FENCE_RE derives from — robust to however the set is composed. The
source-level frontend/route guards in the same file are unchanged.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
---------
Co-authored-by: botinate <285686135+botinate@users.noreply.github.com>
Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-30 17:50:32 +02:00
|
|
|
# Decode JSON-object args (the fenced inline-args shape
|
|
|
|
|
# ```write_file {"path": "...", "content": "..."}```), matching
|
|
|
|
|
# ReadFileTool above. Without this the whole JSON string becomes the
|
|
|
|
|
# path and the file is written under a garbage name. This is the live
|
|
|
|
|
# path: there is no filesystem MCP server, so write_file always runs
|
|
|
|
|
# here via _direct_fallback, not through _build_mcp_args.
|
|
|
|
|
_stripped = content.strip()
|
|
|
|
|
if _stripped.startswith("{"):
|
|
|
|
|
try:
|
|
|
|
|
_a = json.loads(_stripped)
|
|
|
|
|
if isinstance(_a, dict) and "path" in _a:
|
|
|
|
|
raw_path = str(_a.get("path", "")).strip()
|
|
|
|
|
body = str(_a.get("content", ""))
|
|
|
|
|
except (json.JSONDecodeError, TypeError, ValueError):
|
|
|
|
|
pass
|
2026-06-09 19:05:36 +05:30
|
|
|
try:
|
feat(agent): confine agent file/shell tools to a selectable workspace (#3665)
* feat(agent): workspace confinement via context-local binding + get_workspace tool
Bind the per-turn workspace once in execute_tool_block; the shared path
resolvers (_resolve_tool_path / _resolve_search_root) and the subprocess cwd
helper (agent_cwd) read it, so file tools + bash/python are confined centrally
and a new tool that uses the shared helpers cannot accidentally bypass it.
Adds the admin-gated /api/workspace/browse picker, a workspace pill + directory
modal (reusing existing modal/button CSS), the /workspace slash command, and a
get_workspace tool (replaces a system-prompt block). Confinement is OS-agnostic
(realpath/normcase/commonpath) and docker-safe (container paths, no host
assumptions). Reopens #2023.
* ux(workspace): clarify workspace is not a sandbox
Picker modal note + pill tooltip + get_workspace tool/output wording now state
plainly: read_file/write_file/edit_file/grep/glob/ls are confined to the folder,
but bash/python only start there (cwd) and are not sandboxed. Modal note reuses
the existing .muted class.
* fix(agent): treat an active workspace as file-work intent
A vague low-signal message (e.g. "look at the local project") matches no
domain keywords, so tool retrieval is skipped and only always-available tools
are offered — leaving the agent with no file access even though a workspace is
set. When a workspace is active, include the file/code tools (incl.
get_workspace) on low-signal turns so the agent can act on the folder.
Also requires the tool index (ChromaDB) to be reachable for normal retrieval;
that is an environment dependency, not part of this change.
* ux(workspace): hide pill + overflow entry in chat mode
Workspace only scopes the agent's file/shell tools, so the pill and the
overflow 'Workspace' entry are agent-only now — hidden in chat mode like the
bash toggle. Mode read from the DOM in syncWorkspaceIndicator; applyMode() is
called from the agent/chat setMode handler.
* prompt(tools): steer bash/python to defer to the dedicated file tools
bash/python schema descriptions (what native-tool-calling models read) were
bare and gave no steer, so models would do file ops via the shell (e.g. writing
SVG/HTML, which then dumps raw markup into the tool preview). Tell bash/python
in the schema + tool-index + prompt section to prefer read_file/write_file/
edit_file/grep/glob/ls and only be used for what those do not cover.
* prompt(tools): keep bash/python deferral generic (no hardcoded tool names)
Reference 'a dedicated tool' rather than listing read_file/write_file/grep/etc.
by name, so the guidance does not go stale if those tools are renamed.
* style(workspace): drop em-dashes from added code comments/strings
* ux(workspace): terser non-sandbox note in picker (no tool-name list)
* ux(workspace): mirror terse non-sandbox wording in pill tooltip
* chore: untrack local venv symlink (run-only, not part of the feature)
* prompt(workspace): keep get_workspace text generic (no hardcoded tool names)
* fix(agent): low-signal + workspace surfaces only read-only file tools
Intersect the files tool group with PLAN_MODE_READONLY_TOOLS so a vague message
in a workspace exposes read_file/grep/glob/ls/get_workspace for exploration, but
not write_file/edit_file/bash/python -- those wait for a request that actually
calls for them (RAG retrieval still adds them on a real ask).
* feat(workspace): cap browse listing at 500 dirs with a truncated hint
Mirror the filesystem_tools._CODENAV_MAX_HITS pattern with a module-local
_MAX_BROWSE_DIRS so a directory with thousands of children does not dump every
row into the picker; the response carries a truncated flag and the modal tells
the user to type a path to jump in.
* chore: untrack local venv symlink (run-only artifact)
* fix(workspace): vet the workspace root against the sensitive-path deny list at bind time
The in-workspace resolver deny-lists sensitive paths inside the workspace,
but the empty-path search root is the workspace itself, so a workspace of
~/.ssh could be listed via ls with no path. vet_workspace() (public, in
tool_execution next to the resolvers) rejects non-directories and sensitive
roots before the path is ever bound; chat_routes uses it instead of its
inline isdir check.
* fix(workspace): reject filesystem roots and stop showing rejected workspaces as active
Review findings from #3665:
P2: vet_workspace accepted / (and would accept drive/UNC roots), which makes
every absolute path 'inside' the workspace and collapses confinement into
host-wide file access. A root is its own dirname, so reject when
dirname(resolved) == resolved; the browse response now carries a selectable
flag and the picker disables 'Use this folder' on unselectable dirs.
P3: /workspace set stored any string client-side and the chat route silently
dropped rejected values, so the pill could claim a confinement that was not
in effect. New admin-gated /api/workspace/vet validates manual paths before
they persist (canonical path returned), and when a posted workspace is
rejected at send time the stream emits workspace_rejected so the client
clears the stored value and toasts instead of continuing silently.
* fix(workspace): check caller privilege before vetting the posted workspace
Review finding: /api/chat_stream called vet_workspace() on the posted value
for every caller and emitted workspace_rejected on failure, so a non-admin
who can chat but cannot use file/shell tools could distinguish existing
directories from missing/file/sensitive/root paths by whether the event
appeared. The resolution now lives in _resolve_request_workspace, which
drops the submitted value uniformly for non-admin callers, with no vetting
and no event, before the path ever touches the filesystem. Admin and
single-user behavior is unchanged. Test pins that valid and invalid paths
are indistinguishable for a non-admin and that vet_workspace is never
invoked for them.
2026-06-11 18:17:54 +02:00
|
|
|
path = _resolve_tool_path(raw_path)
|
2026-06-09 19:05:36 +05:30
|
|
|
except ValueError as e:
|
|
|
|
|
return {"error": f"write_file: {e}", "exit_code": 1}
|
|
|
|
|
try:
|
|
|
|
|
def _write():
|
|
|
|
|
old = ""
|
|
|
|
|
try:
|
|
|
|
|
with open(path, "r", encoding="utf-8") as f:
|
|
|
|
|
old = f.read()
|
|
|
|
|
except (FileNotFoundError, IsADirectoryError, UnicodeDecodeError, OSError):
|
|
|
|
|
old = ""
|
|
|
|
|
d = os.path.dirname(path)
|
|
|
|
|
if d:
|
|
|
|
|
os.makedirs(d, exist_ok=True)
|
|
|
|
|
with open(path, "w", encoding="utf-8") as f:
|
|
|
|
|
f.write(body)
|
|
|
|
|
return old, len(body)
|
|
|
|
|
old_content, size = await asyncio.to_thread(_write)
|
|
|
|
|
except PermissionError:
|
|
|
|
|
return {"error": f"write_file: {path}: permission denied", "exit_code": 1}
|
|
|
|
|
except OSError as e:
|
|
|
|
|
return {"error": f"write_file: {path}: {e}", "exit_code": 1}
|
|
|
|
|
diff = _unified_diff(old_content, body, path)
|
|
|
|
|
result = {"output": f"Wrote {size} bytes to {path}", "exit_code": 0}
|
|
|
|
|
if diff:
|
|
|
|
|
result["diff"] = diff
|
|
|
|
|
return result
|
|
|
|
|
|
|
|
|
|
class LsTool:
|
|
|
|
|
async def execute(self, content: str, ctx: dict) -> dict:
|
feat(agent): confine agent file/shell tools to a selectable workspace (#3665)
* feat(agent): workspace confinement via context-local binding + get_workspace tool
Bind the per-turn workspace once in execute_tool_block; the shared path
resolvers (_resolve_tool_path / _resolve_search_root) and the subprocess cwd
helper (agent_cwd) read it, so file tools + bash/python are confined centrally
and a new tool that uses the shared helpers cannot accidentally bypass it.
Adds the admin-gated /api/workspace/browse picker, a workspace pill + directory
modal (reusing existing modal/button CSS), the /workspace slash command, and a
get_workspace tool (replaces a system-prompt block). Confinement is OS-agnostic
(realpath/normcase/commonpath) and docker-safe (container paths, no host
assumptions). Reopens #2023.
* ux(workspace): clarify workspace is not a sandbox
Picker modal note + pill tooltip + get_workspace tool/output wording now state
plainly: read_file/write_file/edit_file/grep/glob/ls are confined to the folder,
but bash/python only start there (cwd) and are not sandboxed. Modal note reuses
the existing .muted class.
* fix(agent): treat an active workspace as file-work intent
A vague low-signal message (e.g. "look at the local project") matches no
domain keywords, so tool retrieval is skipped and only always-available tools
are offered — leaving the agent with no file access even though a workspace is
set. When a workspace is active, include the file/code tools (incl.
get_workspace) on low-signal turns so the agent can act on the folder.
Also requires the tool index (ChromaDB) to be reachable for normal retrieval;
that is an environment dependency, not part of this change.
* ux(workspace): hide pill + overflow entry in chat mode
Workspace only scopes the agent's file/shell tools, so the pill and the
overflow 'Workspace' entry are agent-only now — hidden in chat mode like the
bash toggle. Mode read from the DOM in syncWorkspaceIndicator; applyMode() is
called from the agent/chat setMode handler.
* prompt(tools): steer bash/python to defer to the dedicated file tools
bash/python schema descriptions (what native-tool-calling models read) were
bare and gave no steer, so models would do file ops via the shell (e.g. writing
SVG/HTML, which then dumps raw markup into the tool preview). Tell bash/python
in the schema + tool-index + prompt section to prefer read_file/write_file/
edit_file/grep/glob/ls and only be used for what those do not cover.
* prompt(tools): keep bash/python deferral generic (no hardcoded tool names)
Reference 'a dedicated tool' rather than listing read_file/write_file/grep/etc.
by name, so the guidance does not go stale if those tools are renamed.
* style(workspace): drop em-dashes from added code comments/strings
* ux(workspace): terser non-sandbox note in picker (no tool-name list)
* ux(workspace): mirror terse non-sandbox wording in pill tooltip
* chore: untrack local venv symlink (run-only, not part of the feature)
* prompt(workspace): keep get_workspace text generic (no hardcoded tool names)
* fix(agent): low-signal + workspace surfaces only read-only file tools
Intersect the files tool group with PLAN_MODE_READONLY_TOOLS so a vague message
in a workspace exposes read_file/grep/glob/ls/get_workspace for exploration, but
not write_file/edit_file/bash/python -- those wait for a request that actually
calls for them (RAG retrieval still adds them on a real ask).
* feat(workspace): cap browse listing at 500 dirs with a truncated hint
Mirror the filesystem_tools._CODENAV_MAX_HITS pattern with a module-local
_MAX_BROWSE_DIRS so a directory with thousands of children does not dump every
row into the picker; the response carries a truncated flag and the modal tells
the user to type a path to jump in.
* chore: untrack local venv symlink (run-only artifact)
* fix(workspace): vet the workspace root against the sensitive-path deny list at bind time
The in-workspace resolver deny-lists sensitive paths inside the workspace,
but the empty-path search root is the workspace itself, so a workspace of
~/.ssh could be listed via ls with no path. vet_workspace() (public, in
tool_execution next to the resolvers) rejects non-directories and sensitive
roots before the path is ever bound; chat_routes uses it instead of its
inline isdir check.
* fix(workspace): reject filesystem roots and stop showing rejected workspaces as active
Review findings from #3665:
P2: vet_workspace accepted / (and would accept drive/UNC roots), which makes
every absolute path 'inside' the workspace and collapses confinement into
host-wide file access. A root is its own dirname, so reject when
dirname(resolved) == resolved; the browse response now carries a selectable
flag and the picker disables 'Use this folder' on unselectable dirs.
P3: /workspace set stored any string client-side and the chat route silently
dropped rejected values, so the pill could claim a confinement that was not
in effect. New admin-gated /api/workspace/vet validates manual paths before
they persist (canonical path returned), and when a posted workspace is
rejected at send time the stream emits workspace_rejected so the client
clears the stored value and toasts instead of continuing silently.
* fix(workspace): check caller privilege before vetting the posted workspace
Review finding: /api/chat_stream called vet_workspace() on the posted value
for every caller and emitted workspace_rejected on failure, so a non-admin
who can chat but cannot use file/shell tools could distinguish existing
directories from missing/file/sensitive/root paths by whether the event
appeared. The resolution now lives in _resolve_request_workspace, which
drops the submitted value uniformly for non-admin callers, with no vetting
and no event, before the path ever touches the filesystem. Admin and
single-user behavior is unchanged. Test pins that valid and invalid paths
are indistinguishable for a non-admin and that vet_workspace is never
invoked for them.
2026-06-11 18:17:54 +02:00
|
|
|
from src.tool_execution import _resolve_tool_path, _resolve_search_root, _truncate
|
2026-06-09 19:05:36 +05:30
|
|
|
raw_path = ""
|
|
|
|
|
_s = (content or "").strip()
|
|
|
|
|
if _s.startswith("{"):
|
|
|
|
|
try:
|
|
|
|
|
raw_path = str(json.loads(_s).get("path", "")).strip()
|
|
|
|
|
except json.JSONDecodeError:
|
|
|
|
|
raw_path = ""
|
|
|
|
|
else:
|
|
|
|
|
raw_path = _s.split("\n", 1)[0].strip()
|
|
|
|
|
try:
|
|
|
|
|
root = _resolve_search_root(raw_path)
|
|
|
|
|
except ValueError as e:
|
|
|
|
|
return {"error": f"ls: {e}", "exit_code": 1}
|
|
|
|
|
|
|
|
|
|
def _ls():
|
|
|
|
|
if not os.path.isdir(root):
|
|
|
|
|
return None, f"ls: {root}: not a directory"
|
|
|
|
|
rows = []
|
|
|
|
|
try:
|
|
|
|
|
with os.scandir(root) as it:
|
|
|
|
|
for entry in it:
|
|
|
|
|
if entry.name.startswith("."):
|
|
|
|
|
continue
|
|
|
|
|
try:
|
|
|
|
|
is_dir = entry.is_dir(follow_symlinks=False)
|
|
|
|
|
size = entry.stat(follow_symlinks=False).st_size if not is_dir else 0
|
|
|
|
|
except OSError:
|
|
|
|
|
continue
|
|
|
|
|
rows.append((is_dir, entry.name, size))
|
|
|
|
|
except (PermissionError, OSError) as _e:
|
|
|
|
|
return None, f"ls: {_e}"
|
|
|
|
|
rows.sort(key=lambda r: (not r[0], r[1].lower()))
|
|
|
|
|
lines = [f"{root}:"]
|
|
|
|
|
for is_dir, name, size in rows[:_CODENAV_MAX_HITS]:
|
|
|
|
|
lines.append(f" {name}/" if is_dir else f" {name} ({size} B)")
|
|
|
|
|
if len(rows) > _CODENAV_MAX_HITS:
|
|
|
|
|
lines.append(f" ... [{len(rows) - _CODENAV_MAX_HITS} more]")
|
|
|
|
|
if not rows:
|
|
|
|
|
lines.append(" (empty)")
|
|
|
|
|
return "\n".join(lines), None
|
|
|
|
|
|
|
|
|
|
out, err = await asyncio.to_thread(_ls)
|
|
|
|
|
if err:
|
|
|
|
|
return {"error": err, "exit_code": 1}
|
|
|
|
|
return {"output": _truncate(out), "exit_code": 0}
|
|
|
|
|
|
|
|
|
|
class GlobTool:
|
|
|
|
|
async def execute(self, content: str, ctx: dict) -> dict:
|
2026-07-02 14:58:33 +05:30
|
|
|
from src.tool_execution import (
|
|
|
|
|
_SENSITIVE_BASENAMES,
|
|
|
|
|
_is_sensitive_path,
|
|
|
|
|
_resolve_tool_path,
|
|
|
|
|
_resolve_search_root,
|
|
|
|
|
_truncate,
|
|
|
|
|
)
|
2026-06-09 19:05:36 +05:30
|
|
|
args = {}
|
|
|
|
|
_s = (content or "").strip()
|
|
|
|
|
if _s.startswith("{"):
|
|
|
|
|
try:
|
|
|
|
|
args = json.loads(_s)
|
|
|
|
|
except json.JSONDecodeError:
|
|
|
|
|
args = {}
|
|
|
|
|
else:
|
|
|
|
|
args = {"pattern": _s}
|
|
|
|
|
pattern = str(args.get("pattern", "")).strip()
|
|
|
|
|
if not pattern:
|
|
|
|
|
return {"error": "glob: pattern is required", "exit_code": 1}
|
|
|
|
|
try:
|
|
|
|
|
root = _resolve_search_root(str(args.get("path", "")))
|
|
|
|
|
except ValueError as e:
|
|
|
|
|
return {"error": f"glob: {e}", "exit_code": 1}
|
|
|
|
|
|
|
|
|
|
def _glob():
|
2026-06-19 03:02:29 +07:00
|
|
|
base = os.path.abspath(root)
|
|
|
|
|
if not os.path.isdir(base):
|
2026-06-09 19:05:36 +05:30
|
|
|
return None, f"glob: {root}: not a directory"
|
2026-06-30 22:19:53 +05:30
|
|
|
rbase = os.path.realpath(base)
|
2026-06-19 03:02:29 +07:00
|
|
|
norm_pat = pattern.replace("\\", "/")
|
|
|
|
|
# Fast path: literal pattern (no wildcards) → direct path lookup.
|
|
|
|
|
if not any(c in norm_pat for c in "*?["):
|
2026-06-30 22:19:53 +05:30
|
|
|
cand = os.path.realpath(os.path.join(base, norm_pat))
|
|
|
|
|
# Keep the literal lookup inside the search root. os.path.join
|
|
|
|
|
# lets an absolute pattern (or one containing ../) escape `base`,
|
|
|
|
|
# which would turn glob into an existence/path oracle for
|
|
|
|
|
# arbitrary host files — bypassing the workspace/allowlist
|
|
|
|
|
# confinement that _resolve_search_root applies to the root.
|
|
|
|
|
# An escaping literal falls through to the walk, which only ever
|
|
|
|
|
# yields paths under base.
|
|
|
|
|
nbase = os.path.normcase(rbase)
|
|
|
|
|
try:
|
|
|
|
|
inside = cand == rbase or os.path.commonpath(
|
|
|
|
|
[os.path.normcase(cand), nbase]
|
|
|
|
|
) == nbase
|
|
|
|
|
except ValueError:
|
|
|
|
|
inside = False
|
2026-07-02 14:58:33 +05:30
|
|
|
# A literal that names a deny-listed sensitive file (.env,
|
|
|
|
|
# .ssh/id_rsa, …) falls through to the walk, which skips it —
|
|
|
|
|
# otherwise glob would surface secret paths that read_file /
|
|
|
|
|
# grep already refuse to touch.
|
|
|
|
|
if inside and os.path.exists(cand) and not _is_sensitive_path(cand):
|
2026-06-19 03:02:29 +07:00
|
|
|
return [cand], None
|
|
|
|
|
# Literal not at exact path — fall through to walk so
|
|
|
|
|
# e.g. "foo.py" still matches at any depth (like rglob).
|
|
|
|
|
# Compile glob to regex: * stays within one segment, **/ spans dirs.
|
|
|
|
|
regex = _glob_to_regex(norm_pat)
|
2026-06-09 19:05:36 +05:30
|
|
|
matched = []
|
2026-06-19 03:02:29 +07:00
|
|
|
cap = _CODENAV_MAX_HITS * 5
|
2026-06-09 19:05:36 +05:30
|
|
|
try:
|
2026-06-19 03:02:29 +07:00
|
|
|
for dp, dns, fns in os.walk(base):
|
|
|
|
|
# Prune skipped dirs before descending (unlike rglob which
|
|
|
|
|
# descends first then filters — fatal on large node_modules).
|
2026-07-02 14:58:33 +05:30
|
|
|
# Sensitive dirs (.ssh, .gnupg, …) are pruned too so glob
|
|
|
|
|
# never enumerates the keys/tokens inside them.
|
|
|
|
|
dns[:] = [
|
|
|
|
|
d for d in dns
|
|
|
|
|
if d not in _CODENAV_SKIP_DIRS and d not in _SENSITIVE_BASENAMES
|
|
|
|
|
]
|
2026-06-19 03:02:29 +07:00
|
|
|
for name in fns + dns:
|
|
|
|
|
full = os.path.join(dp, name)
|
|
|
|
|
rel = os.path.relpath(full, base).replace(os.sep, "/")
|
|
|
|
|
if regex.fullmatch(rel) or regex.fullmatch(name):
|
2026-07-02 14:58:33 +05:30
|
|
|
# Skip deny-listed sensitive files (.env, id_rsa,
|
|
|
|
|
# known_hosts, …) the same way grep does.
|
|
|
|
|
if _is_sensitive_path(os.path.realpath(full)):
|
|
|
|
|
continue
|
2026-06-19 03:02:29 +07:00
|
|
|
try:
|
|
|
|
|
mtime = os.stat(full).st_mtime
|
|
|
|
|
except OSError:
|
|
|
|
|
mtime = 0
|
|
|
|
|
matched.append((mtime, full))
|
|
|
|
|
if len(matched) > cap:
|
2026-06-09 19:05:36 +05:30
|
|
|
break
|
2026-06-19 03:02:29 +07:00
|
|
|
except OSError as _e:
|
2026-06-09 19:05:36 +05:30
|
|
|
return None, f"glob: {_e}"
|
|
|
|
|
matched.sort(key=lambda t: t[0], reverse=True)
|
|
|
|
|
return [pth for _, pth in matched[:_CODENAV_MAX_HITS]], None
|
|
|
|
|
|
|
|
|
|
paths, err = await asyncio.to_thread(_glob)
|
|
|
|
|
if err:
|
|
|
|
|
return {"error": err, "exit_code": 1}
|
|
|
|
|
if not paths:
|
|
|
|
|
return {"output": f"No files matching {pattern!r} under {root}", "exit_code": 0}
|
|
|
|
|
out = "\n".join(paths)
|
|
|
|
|
if len(paths) >= _CODENAV_MAX_HITS:
|
|
|
|
|
out += f"\n... [capped at {_CODENAV_MAX_HITS} files]"
|
|
|
|
|
return {"output": _truncate(out), "exit_code": 0}
|
|
|
|
|
|
|
|
|
|
class GrepTool:
|
|
|
|
|
async def execute(self, content: str, ctx: dict) -> dict:
|
2026-06-30 23:44:45 +07:00
|
|
|
from src.tool_execution import (
|
|
|
|
|
_SENSITIVE_FILE_PATTERNS,
|
|
|
|
|
_is_sensitive_path,
|
|
|
|
|
_resolve_tool_path,
|
|
|
|
|
_resolve_search_root,
|
|
|
|
|
_truncate,
|
|
|
|
|
)
|
2026-06-09 19:05:36 +05:30
|
|
|
args: Dict[str, Any] = {}
|
|
|
|
|
_s = (content or "").strip()
|
|
|
|
|
if _s.startswith("{"):
|
|
|
|
|
try:
|
|
|
|
|
args = json.loads(_s)
|
|
|
|
|
except json.JSONDecodeError:
|
|
|
|
|
args = {}
|
|
|
|
|
else:
|
|
|
|
|
args = {"pattern": _s}
|
|
|
|
|
pattern = str(args.get("pattern", "")).strip()
|
|
|
|
|
if not pattern:
|
|
|
|
|
return {"error": "grep: pattern is required", "exit_code": 1}
|
|
|
|
|
ignore_case = bool(args.get("ignore_case"))
|
|
|
|
|
glob_pat = str(args.get("glob", "") or "").strip()
|
|
|
|
|
try:
|
|
|
|
|
max_hits = int(args.get("max_results") or _CODENAV_MAX_HITS)
|
|
|
|
|
except (TypeError, ValueError):
|
|
|
|
|
max_hits = _CODENAV_MAX_HITS
|
|
|
|
|
max_hits = max(1, min(max_hits, _CODENAV_MAX_HITS))
|
|
|
|
|
try:
|
|
|
|
|
root = _resolve_search_root(str(args.get("path", "")))
|
|
|
|
|
except ValueError as e:
|
|
|
|
|
return {"error": f"grep: {e}", "exit_code": 1}
|
|
|
|
|
|
|
|
|
|
def _grep():
|
|
|
|
|
import re as _re
|
|
|
|
|
import shutil
|
|
|
|
|
rg = shutil.which("rg")
|
|
|
|
|
if rg:
|
|
|
|
|
cmd = [rg, "--line-number", "--no-heading", "--color=never",
|
|
|
|
|
"--max-count", str(max_hits)]
|
|
|
|
|
if ignore_case:
|
|
|
|
|
cmd.append("--ignore-case")
|
|
|
|
|
if glob_pat:
|
|
|
|
|
cmd += ["--glob", glob_pat]
|
2026-06-30 23:44:45 +07:00
|
|
|
for _pat in _SENSITIVE_FILE_PATTERNS:
|
|
|
|
|
cmd += ["--glob", f"!*{_pat}*"]
|
2026-06-09 19:05:36 +05:30
|
|
|
for _d in _CODENAV_SKIP_DIRS:
|
|
|
|
|
cmd += ["--glob", f"!**/{_d}/**"]
|
|
|
|
|
cmd += ["--regexp", pattern, root]
|
|
|
|
|
try:
|
|
|
|
|
import subprocess
|
|
|
|
|
p = subprocess.run(cmd, capture_output=True, text=True, timeout=20)
|
|
|
|
|
lines = [ln for ln in (p.stdout or "").splitlines() if ln][:max_hits]
|
|
|
|
|
return lines, None
|
|
|
|
|
except subprocess.TimeoutExpired:
|
|
|
|
|
return None, "grep: timed out"
|
|
|
|
|
except Exception as _e:
|
|
|
|
|
return None, f"grep: {_e}"
|
|
|
|
|
try:
|
|
|
|
|
rx = _re.compile(pattern, _re.IGNORECASE if ignore_case else 0)
|
|
|
|
|
except _re.error as _e:
|
|
|
|
|
return None, f"grep: bad pattern: {_e}"
|
|
|
|
|
hits = []
|
|
|
|
|
if os.path.isfile(root):
|
|
|
|
|
file_iter = [root]
|
|
|
|
|
else:
|
|
|
|
|
file_iter = []
|
|
|
|
|
for dp, dns, fns in os.walk(root):
|
|
|
|
|
dns[:] = [d for d in dns if d not in _CODENAV_SKIP_DIRS]
|
|
|
|
|
for fn in fns:
|
|
|
|
|
if glob_pat and not fnmatch.fnmatch(fn, glob_pat):
|
|
|
|
|
continue
|
|
|
|
|
file_iter.append(os.path.join(dp, fn))
|
|
|
|
|
for fp in file_iter:
|
|
|
|
|
if len(hits) >= max_hits:
|
|
|
|
|
break
|
2026-06-30 23:44:45 +07:00
|
|
|
if _is_sensitive_path(os.path.realpath(fp)):
|
|
|
|
|
continue
|
2026-06-09 19:05:36 +05:30
|
|
|
try:
|
|
|
|
|
with open(fp, "r", encoding="utf-8", errors="strict") as f:
|
|
|
|
|
for i, line in enumerate(f, 1):
|
|
|
|
|
if rx.search(line):
|
|
|
|
|
hits.append(f"{fp}:{i}:{line.rstrip()[:_CODENAV_MAX_LINE]}")
|
|
|
|
|
if len(hits) >= max_hits:
|
|
|
|
|
break
|
|
|
|
|
except (UnicodeDecodeError, OSError):
|
|
|
|
|
continue
|
|
|
|
|
return hits, None
|
|
|
|
|
|
|
|
|
|
lines, err = await asyncio.to_thread(_grep)
|
|
|
|
|
if err:
|
|
|
|
|
return {"error": err, "exit_code": 1}
|
|
|
|
|
if not lines:
|
|
|
|
|
return {"output": f"No matches for {pattern!r} under {root}", "exit_code": 0}
|
|
|
|
|
out = "\n".join(ln[:_CODENAV_MAX_LINE] for ln in lines)
|
|
|
|
|
if len(lines) >= max_hits:
|
|
|
|
|
out += f"\n... [capped at {max_hits} matches]"
|
|
|
|
|
return {"output": _truncate(out), "exit_code": 0}
|
feat(agent): confine agent file/shell tools to a selectable workspace (#3665)
* feat(agent): workspace confinement via context-local binding + get_workspace tool
Bind the per-turn workspace once in execute_tool_block; the shared path
resolvers (_resolve_tool_path / _resolve_search_root) and the subprocess cwd
helper (agent_cwd) read it, so file tools + bash/python are confined centrally
and a new tool that uses the shared helpers cannot accidentally bypass it.
Adds the admin-gated /api/workspace/browse picker, a workspace pill + directory
modal (reusing existing modal/button CSS), the /workspace slash command, and a
get_workspace tool (replaces a system-prompt block). Confinement is OS-agnostic
(realpath/normcase/commonpath) and docker-safe (container paths, no host
assumptions). Reopens #2023.
* ux(workspace): clarify workspace is not a sandbox
Picker modal note + pill tooltip + get_workspace tool/output wording now state
plainly: read_file/write_file/edit_file/grep/glob/ls are confined to the folder,
but bash/python only start there (cwd) and are not sandboxed. Modal note reuses
the existing .muted class.
* fix(agent): treat an active workspace as file-work intent
A vague low-signal message (e.g. "look at the local project") matches no
domain keywords, so tool retrieval is skipped and only always-available tools
are offered — leaving the agent with no file access even though a workspace is
set. When a workspace is active, include the file/code tools (incl.
get_workspace) on low-signal turns so the agent can act on the folder.
Also requires the tool index (ChromaDB) to be reachable for normal retrieval;
that is an environment dependency, not part of this change.
* ux(workspace): hide pill + overflow entry in chat mode
Workspace only scopes the agent's file/shell tools, so the pill and the
overflow 'Workspace' entry are agent-only now — hidden in chat mode like the
bash toggle. Mode read from the DOM in syncWorkspaceIndicator; applyMode() is
called from the agent/chat setMode handler.
* prompt(tools): steer bash/python to defer to the dedicated file tools
bash/python schema descriptions (what native-tool-calling models read) were
bare and gave no steer, so models would do file ops via the shell (e.g. writing
SVG/HTML, which then dumps raw markup into the tool preview). Tell bash/python
in the schema + tool-index + prompt section to prefer read_file/write_file/
edit_file/grep/glob/ls and only be used for what those do not cover.
* prompt(tools): keep bash/python deferral generic (no hardcoded tool names)
Reference 'a dedicated tool' rather than listing read_file/write_file/grep/etc.
by name, so the guidance does not go stale if those tools are renamed.
* style(workspace): drop em-dashes from added code comments/strings
* ux(workspace): terser non-sandbox note in picker (no tool-name list)
* ux(workspace): mirror terse non-sandbox wording in pill tooltip
* chore: untrack local venv symlink (run-only, not part of the feature)
* prompt(workspace): keep get_workspace text generic (no hardcoded tool names)
* fix(agent): low-signal + workspace surfaces only read-only file tools
Intersect the files tool group with PLAN_MODE_READONLY_TOOLS so a vague message
in a workspace exposes read_file/grep/glob/ls/get_workspace for exploration, but
not write_file/edit_file/bash/python -- those wait for a request that actually
calls for them (RAG retrieval still adds them on a real ask).
* feat(workspace): cap browse listing at 500 dirs with a truncated hint
Mirror the filesystem_tools._CODENAV_MAX_HITS pattern with a module-local
_MAX_BROWSE_DIRS so a directory with thousands of children does not dump every
row into the picker; the response carries a truncated flag and the modal tells
the user to type a path to jump in.
* chore: untrack local venv symlink (run-only artifact)
* fix(workspace): vet the workspace root against the sensitive-path deny list at bind time
The in-workspace resolver deny-lists sensitive paths inside the workspace,
but the empty-path search root is the workspace itself, so a workspace of
~/.ssh could be listed via ls with no path. vet_workspace() (public, in
tool_execution next to the resolvers) rejects non-directories and sensitive
roots before the path is ever bound; chat_routes uses it instead of its
inline isdir check.
* fix(workspace): reject filesystem roots and stop showing rejected workspaces as active
Review findings from #3665:
P2: vet_workspace accepted / (and would accept drive/UNC roots), which makes
every absolute path 'inside' the workspace and collapses confinement into
host-wide file access. A root is its own dirname, so reject when
dirname(resolved) == resolved; the browse response now carries a selectable
flag and the picker disables 'Use this folder' on unselectable dirs.
P3: /workspace set stored any string client-side and the chat route silently
dropped rejected values, so the pill could claim a confinement that was not
in effect. New admin-gated /api/workspace/vet validates manual paths before
they persist (canonical path returned), and when a posted workspace is
rejected at send time the stream emits workspace_rejected so the client
clears the stored value and toasts instead of continuing silently.
* fix(workspace): check caller privilege before vetting the posted workspace
Review finding: /api/chat_stream called vet_workspace() on the posted value
for every caller and emitted workspace_rejected on failure, so a non-admin
who can chat but cannot use file/shell tools could distinguish existing
directories from missing/file/sensitive/root paths by whether the event
appeared. The resolution now lives in _resolve_request_workspace, which
drops the submitted value uniformly for non-admin callers, with no vetting
and no event, before the path ever touches the filesystem. Admin and
single-user behavior is unchanged. Test pins that valid and invalid paths
are indistinguishable for a non-admin and that vet_workspace is never
invoked for them.
2026-06-11 18:17:54 +02:00
|
|
|
|
|
|
|
|
class GetWorkspaceTool:
|
|
|
|
|
"""Report the active workspace folder (no args). File tools are confined to
|
|
|
|
|
it; the shell starts there (cwd) but is NOT sandboxed."""
|
|
|
|
|
async def execute(self, content: str, ctx: dict) -> dict:
|
|
|
|
|
from src.tool_execution import get_active_workspace
|
|
|
|
|
ws = get_active_workspace()
|
|
|
|
|
if ws:
|
|
|
|
|
return {
|
|
|
|
|
"output": f"{ws}\n(File tools are confined to this folder; the shell starts "
|
|
|
|
|
f"here but is not sandboxed and can reach outside it.)",
|
|
|
|
|
"exit_code": 0,
|
|
|
|
|
}
|
|
|
|
|
return {
|
|
|
|
|
"output": "No workspace is set. File tools use the default allowed roots; "
|
|
|
|
|
"resolve paths from the user or use absolute paths.",
|
|
|
|
|
"exit_code": 0,
|
|
|
|
|
}
|