2026-05-31 23:58:26 +09:00
""" Chat routes — /api/chat, /api/chat_stream, /api/inject_context, /api/search. """
import asyncio
import json
2026-06-05 00:06:37 +02:00
import os
2026-07-07 00:50:07 +00:00
import re
2026-05-31 23:58:26 +09:00
import time
import logging
2026-06-01 10:00:15 +09:00
from datetime import datetime
Open email context for agent, email search across All Mail, cookbook serve polish
- Agent: pass the open email reader (uid/folder/account/from/subject/body
preview) on every chat submit so 'reply to this' / 'write email saying
hi' route to ui_control open_email_reply with the right UID instead of
inventing a new .md draft. Code-level enforcement (chat_routes strips
create_document + send_email when active_email is set); cross-session
active_doc_id is now trusted instead of being silently dropped.
set_active_email/clear_active_email tool-layer helpers in
tool_implementations.
- ui_control open_email_reply: optional body argument so the agent can
open-and-write in one call; envelope now forwards uid/folder/account/
body/panel through tool_output. Tool description sharpened and the
parser rejects empty bodies on reply/reply-all (forces the agent to
write rather than open an empty draft).
- Email library: search now runs against [Gmail]/All Mail when the
current folder is INBOX (archived emails surface). Whirlpool spinner
+ 'Searching…' placeholder while in flight. Each search result is
stamped with its source folder so clicks open the right email instead
of whatever shares its UID in INBOX. Search no longer re-applies the
same text pill locally (which only checks subject/from/snippet, never
body) so body-only matches don't get dropped after IMAP returns them.
Initial inbox load bumped 100→500.
- Email favorites: 'Favorite (pin to top)' / 'Unfavorite' in both the
card menu and the open-reader more menu, backed by a new
/api/email/flag/{uid}?on=true|false endpoint. Flagged emails always
bubble to the top of the grid regardless of active sort.
- AI reply in doc editor: never overwrites existing draft text or the
quoted history. AI suggestion is prepended; AI-generated 'On …
wrote:' re-quotes are stripped so the original quote isn't visually
edited.
- Cookbook serve: pre-launch GPU driver / has_gpu / install / version-
floor checks (vllm minimax_m2 needs 0.10.0+, deepseek_r1 needs 0.7.0
etc.) before the launch chain starts. Detect 'another model already
running on this host' and offer Stop & launch (with graceful then
force tmux kill helpers, port release wait). Per-vendor deep-link
buttons (vLLM recipe / SGLang cookbook) with hardware hash. Backend
picker is now a custom dropdown with accent-coloured logos for vLLM,
SGLang, llama.cpp, Ollama, Diffusers; same glyphs added next to
package names in Dependencies. Runtime-readiness note moved inside
the panel (green when ready, red when missing) with an × dismiss.
Esc collapses the expanded card; expanded card scrolls when it
overflows; Trust Remote / Auto Tool / Reasoning Parser / Enforce
Eager / Prefix Caching / Expert Parallel / Speculative / MoE Env on
one row (Reasoning Parser auto-detected per model family).
Dtype→Row 1, GPUs→Row 2 (rightmost). Removed redundant GPU 'auto'
input — command builders read from the GPU button strip. Default
cookbook open is Download tab.
- Cookbook hwfit: 'Model (latest)' / 'Model (oldest)' header sorts by
release_date; release dates can be backfilled with the new
scripts/backfill_model_release_dates.py and recipe metadata pulled
with scripts/import_from_vllm_recipes.py against the upstream
vllm-project/recipes catalog (vllm_recipe + min_vllm_version stamped
on entries).
- Calendar: Quick add hint cycles a random Odysseus-themed example per
open (wooden horse Friday, crew muster 10am daily, council on
Ithaca, …). Typing a time like '11pm' in the event title updates
the hero clock live.
- Doc editor: email-mode Reply button (sparkle icon, accent) opens the
same Fast/Full + context popover the email reader uses; Ctrl+Alt+M
toggles markdown preview.
- Memories panel: custom sort picker with per-option icons, default
'Latest', visible Enabled/Disabled toggle text matching the section
description style.
2026-06-15 20:47:51 +09:00
from typing import Dict , Any , AsyncGenerator , List , Optional
2026-05-31 23:58:26 +09:00
from fastapi import APIRouter , Request , HTTPException , Form , Query
from fastapi . responses import StreamingResponse
from pydantic import ValidationError
from core . models import ChatMessage
from src . request_models import ChatRequest
from src . llm_core import llm_call_async , stream_llm , stream_llm_with_fallback
from src . agent_loop import stream_agent_loop
from src import agent_runs
from src . model_context import estimate_tokens
from src . chat_helpers import coerce_message_and_session
2026-06-01 10:00:15 +09:00
from src . endpoint_resolver import normalize_base as _normalize_base , build_chat_url
2026-06-05 18:08:31 -06:00
from src . session_search import search_session_messages
2026-05-31 23:58:26 +09:00
from src . prompt_security import untrusted_context_message
from core . exceptions import SessionNotFoundError
fix(api): attribute bearer-token actions to the token owner on owner-scoped routes (#4054)
* fix(api): attribute bearer-token actions to the token owner on owner-scoped routes
Owner-scoped chat, session, and upload routes called
get_current_user(), which resolves a bearer ody_ API token to the
sandboxed "api" pseudo-user. A paired API-token client (companion, CLI,
IDE extension) therefore saw and created a separate "api"-owned silo
instead of the owner's data.
effective_user() already exists for exactly this: it attributes a token's
actions to request.state.api_token_owner, is identical to
get_current_user() for cookie sessions, and falls back safely when a
token has no owner. session_routes.py was already migrated; this
completes the migration for the remaining owner-scoped routes:
- chat_helpers.py: chat-privilege enforcement, message attribution, prefs/context
- chat_routes.py: orphaned-endpoint owner, session-auth owner, message search
- upload_routes.py: upload owner attribution + access checks
The /api/models swap is intentionally omitted: #4292 already migrated it
to effective_user (plus the chat-scope gate and ownerless-token 403), so
this PR keeps dev's version of routes/model_routes.py unchanged.
chat_routes.py keeps importing get_current_user for the workspace owner
gate; session_routes.py drops the now-unused import.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* test: target effective_user in auth monkeypatches and owner-scope assertion
The owner-scoped routes now call effective_user() instead of
get_current_user(), so the tests that stubbed get_current_user (or
asserted on it) follow suit:
- test_chat_helpers.py, test_review_regressions.py,
test_kv_cache_invalidation_2927.py: monkeypatch effective_user
- test_session_endpoint_owner_scope.py: assert the owner-scope guard uses
effective_user(request)
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
---------
Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-15 22:56:22 +01:00
from src . auth_helpers import effective_user , get_current_user
2026-05-31 23:58:26 +09:00
from routes . session_routes import _verify_session_owner
2026-06-02 04:46:40 +02:00
from routes . document_helpers import _owner_session_filter
fix: stop leaking DB connections when persisting session mode (#64)
chat_routes.py persisted a session's "mode" in three best-effort spots —
reading the current mode, writing the effective mode, and setting
research_pending on the stream path. Each opened a session with SessionLocal()
and called .close() as the LAST statement inside a try/except, so if anything
before close() raised (e.g. a SQLite "database is locked" under concurrent chat
streams) the except only logged and the connection was never returned to the
pool.
DATABASE_URL defaults to file-backed SQLite, whose engine uses SQLAlchemy's
default QueuePool (5 connections + 10 overflow). Repeated leaks on these hot
paths exhaust the pool; later requests then block for pool_timeout and fail
with "QueuePool limit ... reached", taking the app down until restart.
Move the logic into two best-effort helpers in core.database, next to the
existing session helpers (update_session_last_accessed, get_session_by_id):
- get_session_mode(session_id) -> Optional[str]
- set_session_mode(session_id, mode) -> bool
Both route through the existing get_db_session() context manager, which commits
on success, rolls back on error, and always closes in a finally, so the
connection is returned to the pool on every path. chat_routes.py now calls
these instead of hand-rolling sessions, also removing three copies of the same
try/except.
Add tests/test_session_mode_helpers.py: the helpers commit+close on success
and, on a mid-operation DB error, swallow + roll back + close (no leak). The
error-path tests fail against the old close()-inside-try pattern.
2026-06-01 00:57:48 -04:00
from core . database import SessionLocal , get_session_mode , set_session_mode
2026-05-31 23:58:26 +09:00
from core . database import Session as DBSession , ChatMessage as DBChatMessage
from core . database import Document as DBDocument , ModelEndpoint
fix(security): redact credential-bearing URLs and PII from logs (#4750)
* fix(security): redact credential-bearing URLs and PII from logs
Several log statements emitted sensitive data in clear text:
- model_routes / chat_routes / contacts_routes logged endpoint URLs raw.
Admin-configured URLs can embed credentials in userinfo or query
(e.g. https://user:pass@host, ?api_key=...). Route them through a
shared core.log_safety.redact_url() that drops userinfo/query/fragment.
- note_routes / task_scheduler logged operator email addresses (smtp_user,
recipient). Replaced with presence booleans, which keeps the diagnostic
("why didn't this send") without writing PII to logs.
model_routes already had a local redactor on its HTTPStatusError branch;
the generic except branch was missed, so reuse the existing helper there.
Clears CodeQL py/clear-text-logging-sensitive-data alerts 264, 317, 324,
325, 343, 344, 528.
* fix(security): re-bracket IPv6 hosts and single-source the URL redactor
Address review on #4750:
- redact_url now re-brackets IPv6 literals so host:port stays
unambiguous (https://[2001:db8::1]:8443/v1, not the bracket-less
ambiguous form).
- point model_routes._redact_url_for_log at the shared helper so the
two redactors are single-sourced (also picks up the IPv6 fix).
2026-06-22 14:12:39 -07:00
from core . log_safety import redact_url
2026-05-31 23:58:26 +09:00
from routes . research_routes import _resolve_research_endpoint
2026-06-02 13:37:14 +02:00
from routes . model_routes import _visible_models
2026-05-31 23:58:26 +09:00
from routes . chat_helpers import (
resolve_session_auth ,
build_chat_context ,
save_assistant_response ,
run_post_response_tasks ,
clean_thinking_for_save ,
_enforce_chat_privileges ,
)
2026-07-07 00:50:07 +00:00
from src . action_intents import ToolIntent , classify_tool_intent as _classify_tool_intent
2026-06-06 18:48:24 -06:00
from src . tool_policy import build_effective_tool_policy
2026-05-31 23:58:26 +09:00
logger = logging . getLogger ( __name__ )
# Track active streams for partial-save safety net
_active_streams : Dict [ str , dict ] = { }
2026-06-02 06:52:03 -05:00
_IMAGE_MODEL_PREFIXES = ( " gpt-image " , " dall-e " , " chatgpt-image " )
2026-05-31 23:58:26 +09:00
def _stream_set ( session_id : str , * * fields ) - > None :
""" Update fields on the active-stream entry for `session_id`, or
no - op if the entry has already been popped . Using . get ( ) avoids a
KeyError race between ` if x in d ` and ` d [ x ] [ " k " ] = v ` if a sibling
finally pops the key in between ( which becomes possible the moment
a coroutine cancellation reaches an inner cleanup before the
outermost cleanup runs ) . """
rec = _active_streams . get ( session_id )
if rec is None :
return
rec . update ( fields )
2026-07-07 00:50:07 +00:00
def _message_plain_text ( content : Any ) - > str :
if isinstance ( content , list ) :
parts : List [ str ] = [ ]
for block in content :
if isinstance ( block , dict ) :
text = block . get ( " text " )
if isinstance ( text , str ) :
parts . append ( text )
elif isinstance ( block , str ) :
parts . append ( block )
return " " . join ( parts )
return str ( content or " " )
def _last_user_plain_text ( messages : List [ Dict [ str , Any ] ] ) - > str :
for msg in reversed ( messages or [ ] ) :
if msg . get ( " role " ) == " user " :
return _message_plain_text ( msg . get ( " content " ) )
return " "
def _ensure_current_request_is_latest_user ( messages : List [ Dict [ str , Any ] ] , current_message : str ) - > List [ Dict [ str , Any ] ] :
""" Defensively keep detached streams grounded on the request that created them. """
current = str ( current_message or " " ) . strip ( )
if not current :
return messages
latest = _last_user_plain_text ( messages ) . strip ( )
if latest == current or current in latest or latest in current :
return messages
logger . warning (
" [chat_stream] latest user context mismatch; appending current request for model call. latest= %r current= %r " ,
latest [ : 120 ] ,
current [ : 120 ] ,
)
repaired = list ( messages or [ ] )
repaired . append ( { " role " : " user " , " content " : current } )
return repaired
_WEB_FOLLOWUP_RE = re . compile (
r " ^ \ s*(?:(?:can|could|would|will) \ s+you \ s+)? "
r " (?:check|try \ s+again|look(?: \ s+now| \ s+it \ s+up)?|search(?: \ s+now| \ s+online| \ s+it)?| "
r " do \ s+it|again) \ ?? \ s*$ " ,
re . I ,
)
_RECENT_WEB_CONTEXT_RE = re . compile (
r " \ b(?:weather|forecast|rain|raining|hourly|news|headlines|rate|exchange|currency| "
r " price|current|latest|search|look \ s+up|online) \ b " ,
re . I ,
)
def _recent_session_text ( sess , limit : int = 8 , max_chars : int = 2000 ) - > str :
history = getattr ( sess , " history " , None ) or getattr ( sess , " _history " , None ) or [ ]
chunks : List [ str ] = [ ]
for msg in history [ - limit : ] :
content = getattr ( msg , " content " , None )
if content is None and isinstance ( msg , dict ) :
content = msg . get ( " content " )
text = _message_plain_text ( content ) . strip ( )
if text :
chunks . append ( text )
return " " . join ( chunks ) [ - max_chars : ]
def _is_contextual_web_followup ( message : str , sess ) - > bool :
""" Treat short retry/check replies as web lookups when recent context was web. """
if not message or not _WEB_FOLLOWUP_RE . search ( message ) :
return False
return bool ( _RECENT_WEB_CONTEXT_RE . search ( _recent_session_text ( sess ) ) )
feat(agent): confine agent file/shell tools to a selectable workspace (#3665)
* feat(agent): workspace confinement via context-local binding + get_workspace tool
Bind the per-turn workspace once in execute_tool_block; the shared path
resolvers (_resolve_tool_path / _resolve_search_root) and the subprocess cwd
helper (agent_cwd) read it, so file tools + bash/python are confined centrally
and a new tool that uses the shared helpers cannot accidentally bypass it.
Adds the admin-gated /api/workspace/browse picker, a workspace pill + directory
modal (reusing existing modal/button CSS), the /workspace slash command, and a
get_workspace tool (replaces a system-prompt block). Confinement is OS-agnostic
(realpath/normcase/commonpath) and docker-safe (container paths, no host
assumptions). Reopens #2023.
* ux(workspace): clarify workspace is not a sandbox
Picker modal note + pill tooltip + get_workspace tool/output wording now state
plainly: read_file/write_file/edit_file/grep/glob/ls are confined to the folder,
but bash/python only start there (cwd) and are not sandboxed. Modal note reuses
the existing .muted class.
* fix(agent): treat an active workspace as file-work intent
A vague low-signal message (e.g. "look at the local project") matches no
domain keywords, so tool retrieval is skipped and only always-available tools
are offered — leaving the agent with no file access even though a workspace is
set. When a workspace is active, include the file/code tools (incl.
get_workspace) on low-signal turns so the agent can act on the folder.
Also requires the tool index (ChromaDB) to be reachable for normal retrieval;
that is an environment dependency, not part of this change.
* ux(workspace): hide pill + overflow entry in chat mode
Workspace only scopes the agent's file/shell tools, so the pill and the
overflow 'Workspace' entry are agent-only now — hidden in chat mode like the
bash toggle. Mode read from the DOM in syncWorkspaceIndicator; applyMode() is
called from the agent/chat setMode handler.
* prompt(tools): steer bash/python to defer to the dedicated file tools
bash/python schema descriptions (what native-tool-calling models read) were
bare and gave no steer, so models would do file ops via the shell (e.g. writing
SVG/HTML, which then dumps raw markup into the tool preview). Tell bash/python
in the schema + tool-index + prompt section to prefer read_file/write_file/
edit_file/grep/glob/ls and only be used for what those do not cover.
* prompt(tools): keep bash/python deferral generic (no hardcoded tool names)
Reference 'a dedicated tool' rather than listing read_file/write_file/grep/etc.
by name, so the guidance does not go stale if those tools are renamed.
* style(workspace): drop em-dashes from added code comments/strings
* ux(workspace): terser non-sandbox note in picker (no tool-name list)
* ux(workspace): mirror terse non-sandbox wording in pill tooltip
* chore: untrack local venv symlink (run-only, not part of the feature)
* prompt(workspace): keep get_workspace text generic (no hardcoded tool names)
* fix(agent): low-signal + workspace surfaces only read-only file tools
Intersect the files tool group with PLAN_MODE_READONLY_TOOLS so a vague message
in a workspace exposes read_file/grep/glob/ls/get_workspace for exploration, but
not write_file/edit_file/bash/python -- those wait for a request that actually
calls for them (RAG retrieval still adds them on a real ask).
* feat(workspace): cap browse listing at 500 dirs with a truncated hint
Mirror the filesystem_tools._CODENAV_MAX_HITS pattern with a module-local
_MAX_BROWSE_DIRS so a directory with thousands of children does not dump every
row into the picker; the response carries a truncated flag and the modal tells
the user to type a path to jump in.
* chore: untrack local venv symlink (run-only artifact)
* fix(workspace): vet the workspace root against the sensitive-path deny list at bind time
The in-workspace resolver deny-lists sensitive paths inside the workspace,
but the empty-path search root is the workspace itself, so a workspace of
~/.ssh could be listed via ls with no path. vet_workspace() (public, in
tool_execution next to the resolvers) rejects non-directories and sensitive
roots before the path is ever bound; chat_routes uses it instead of its
inline isdir check.
* fix(workspace): reject filesystem roots and stop showing rejected workspaces as active
Review findings from #3665:
P2: vet_workspace accepted / (and would accept drive/UNC roots), which makes
every absolute path 'inside' the workspace and collapses confinement into
host-wide file access. A root is its own dirname, so reject when
dirname(resolved) == resolved; the browse response now carries a selectable
flag and the picker disables 'Use this folder' on unselectable dirs.
P3: /workspace set stored any string client-side and the chat route silently
dropped rejected values, so the pill could claim a confinement that was not
in effect. New admin-gated /api/workspace/vet validates manual paths before
they persist (canonical path returned), and when a posted workspace is
rejected at send time the stream emits workspace_rejected so the client
clears the stored value and toasts instead of continuing silently.
* fix(workspace): check caller privilege before vetting the posted workspace
Review finding: /api/chat_stream called vet_workspace() on the posted value
for every caller and emitted workspace_rejected on failure, so a non-admin
who can chat but cannot use file/shell tools could distinguish existing
directories from missing/file/sensitive/root paths by whether the event
appeared. The resolution now lives in _resolve_request_workspace, which
drops the submitted value uniformly for non-admin callers, with no vetting
and no event, before the path ever touches the filesystem. Admin and
single-user behavior is unchanged. Test pins that valid and invalid paths
are indistinguishable for a non-admin and that vet_workspace is never
invoked for them.
2026-06-11 18:17:54 +02:00
def _resolve_request_workspace ( request , raw_value ) - > tuple :
""" Resolve the posted workspace for this request: (workspace, rejected).
Privilege is checked BEFORE the path ever touches the filesystem . Only
admin / single - user callers can use the workspace - backed file / shell tools ,
so only they get vet_workspace ( ) and the workspace_rejected signal . For
any other caller the submitted value is dropped uniformly , with no vetting
and no event : otherwise the presence / absence of workspace_rejected would
let a non - admin chat caller probe which host paths exist .
vet_workspace rejects non - directories , sensitive roots ( . ssh , . gnupg ,
. . . ) , and filesystem roots ; on rejection there is no confinement and the
default tool - path allowlist applies . The rejected value is surfaced so the
stream can tell an admin client ( which believes a workspace is active )
that it was dropped .
"""
requested = ( raw_value or " " ) . strip ( )
if not requested :
return " " , " "
from src . tool_security import owner_is_admin_or_single_user
if not owner_is_admin_or_single_user ( get_current_user ( request ) ) :
return " " , " "
from src . tool_execution import vet_workspace
workspace = vet_workspace ( requested ) or " "
return workspace , ( requested if not workspace else " " )
2026-06-01 10:00:15 +09:00
def _session_url_matches_endpoint ( session_url : str , endpoint_base : str ) - > bool :
if not session_url or not endpoint_base :
return False
sess = session_url . rstrip ( " / " )
base = _normalize_base ( endpoint_base ) . rstrip ( " / " )
variants = {
base ,
base + " /chat/completions " ,
build_chat_url ( base ) . rstrip ( " / " ) ,
}
return sess in variants or sess . startswith ( base + " / " )
2026-06-02 19:40:22 +02:00
def _clear_orphaned_session_endpoint ( sess , owner : str | None = None ) - > bool :
2026-06-01 10:00:15 +09:00
""" Clear a session model if its endpoint was deleted from ModelEndpoint. """
if not getattr ( sess , " endpoint_url " , " " ) :
return False
db = SessionLocal ( )
try :
2026-06-02 19:40:22 +02:00
q = db . query ( ModelEndpoint ) . filter ( ModelEndpoint . is_enabled == True )
if owner :
from src . auth_helpers import owner_filter
q = owner_filter ( q , ModelEndpoint , owner )
endpoints = q . all ( )
2026-06-01 10:00:15 +09:00
for ep in endpoints :
if _session_url_matches_endpoint ( sess . endpoint_url or " " , ep . base_url or " " ) :
return False
db_session = db . query ( DBSession ) . filter ( DBSession . id == sess . id ) . first ( )
if db_session :
db_session . endpoint_url = " "
db_session . model = " "
db_session . updated_at = datetime . utcnow ( )
db . commit ( )
sess . endpoint_url = " "
sess . model = " "
sess . headers = { }
return True
2026-06-15 13:49:27 -03:00
except Exception as e :
logger . warning ( " Failed to clear orphaned session endpoint " , exc_info = e )
2026-06-01 10:00:15 +09:00
db . rollback ( )
return False
finally :
db . close ( )
2026-06-02 06:52:03 -05:00
def _endpoint_cache_contains_model ( endpoint , model : str ) - > bool :
""" Return True when a populated endpoint model cache includes ``model``.
Empty / malformed caches are treated as unknown rather than a negative match
so older image endpoints without cached models still work .
"""
raw = getattr ( endpoint , " cached_models " , None )
if not raw :
return True
try :
models = json . loads ( raw ) if isinstance ( raw , str ) else raw
2026-06-15 13:49:27 -03:00
except Exception as e :
logger . warning ( " Failed to parse cached models list, treating as containing model " , exc_info = e )
2026-06-02 06:52:03 -05:00
return True
if not isinstance ( models , list ) or not models :
return True
wanted = ( model or " " ) . strip ( )
return wanted in { str ( item ) . strip ( ) for item in models }
2026-06-02 19:40:22 +02:00
def _is_image_generation_session ( sess , owner : str | None = None ) - > bool :
2026-06-02 06:52:03 -05:00
""" Whether this chat session should bypass text chat and generate images.
Model - name prefixes are explicit image models . Endpoint type is only used
when the current session endpoint actually matches that image endpoint , and
when a populated endpoint model cache includes the selected model . This
prevents an image endpoint on the same host from misrouting ordinary text
models into the image - generation path .
"""
model = ( getattr ( sess , " model " , " " ) or " " ) . strip ( )
if any ( model . lower ( ) . startswith ( prefix ) for prefix in _IMAGE_MODEL_PREFIXES ) :
return True
endpoint_url = ( getattr ( sess , " endpoint_url " , " " ) or " " ) . strip ( )
if not endpoint_url :
return False
db = SessionLocal ( )
try :
2026-06-02 19:40:22 +02:00
q = db . query ( ModelEndpoint ) . filter ( ModelEndpoint . is_enabled == True )
if owner :
from src . auth_helpers import owner_filter
q = owner_filter ( q , ModelEndpoint , owner )
endpoints = q . all ( )
2026-06-02 06:52:03 -05:00
for endpoint in endpoints :
if ( getattr ( endpoint , " model_type " , None ) or " llm " ) != " image " :
continue
if not _session_url_matches_endpoint ( endpoint_url , getattr ( endpoint , " base_url " , " " ) or " " ) :
continue
if _endpoint_cache_contains_model ( endpoint , model ) :
return True
except Exception :
return False
finally :
db . close ( )
return False
2026-06-02 19:40:22 +02:00
def _recover_empty_session_model ( sess , session_id : str , owner : str | None = None ) - > bool :
2026-06-02 07:56:38 +05:30
""" Re-populate sess.model from the matching endpoint ' s cached models.
Covers the window between endpoint setup and the first chat send : the
picker showed a model in the dropdown but the session record never got
written ( Issue #587 — UI uses the cached endpoint list, not s.model).
feat: add ChatGPT Subscription provider (#2876)
* feat: Add ChatGPT Subscription support and related features
- Introduced a new provider option for ChatGPT Subscription in the endpoint selection UI.
- Implemented OAuth flow for ChatGPT Subscription sign-in, including polling for authorization status.
- Updated admin interface to handle ChatGPT Subscription, including disabling API key input and providing user guidance.
- Enhanced cost tracking logic to differentiate between subscription and non-subscription endpoints.
- Added new slash commands for managing skills, including listing, searching, and invoking skills.
- Implemented caching for skill catalog to optimize performance.
- Updated tests to cover new ChatGPT Subscription functionality and ensure proper endpoint probing.
- Refactored existing code to accommodate new features and improve maintainability.
* refactor: share provider device-flow setup
- reuse one device-flow backend for Copilot and ChatGPT Subscription
- add one frontend device-flow helper for Settings and /setup
- put GitHub Copilot back into Add Models, now as a dropdown option
- make provider selection just select; clicking Add starts sign-in
- stop ChatGPT Subscription setup from opening auth tabs automatically
- make /setup copilot and /setup chatgpt-subscription work from chat
- show ChatGPT Subscription in the /setup suggestions
- show the real error message when setup fails
- add focused tests for the shared flow and setup UI
* feat(chatgpt-subscription): harden credential lifecycle and streamline auth UX
Backend:
- Resolve runtime bearer for provider-auth endpoints at probe time via a
shared _resolve_probe_key() that delegates to resolve_endpoint_runtime,
applied across all probe/refresh call sites.
- Skip live completion probes and health pings for discovery-only providers
(centralized behind _is_discovery_only_provider) — the Codex/Responses API
has no such endpoints, so status is derived from cached models.
- Never persist the short lived ChatGPT bearer to the plaintext sessions
table; proactively clear any stale bearer left by an earlier code path.
- Revoke orphaned ProviderAuthSession credentials when the last endpoint
backing them is deleted (_delete_orphaned_provider_auth), surfaced via
cleared_provider_auth in the delete response.
Frontend (admin.js):
- Auto-start the device-auth flow on provider selection so the authorization
panel (code + Authorize) shows immediately instead of behind a "Sign in" click.
- Remove the redundant top button for device auth providers, move retry
into the panel via an inline "Try again".
- Drop the self-evident hint text and add an execCommand clipboard fallback so
Copy works in non-secure (HTTP/LAN) contexts.
* fix: harden chatgpt subscription provider
* chore: remove PR media from branch
* Fix chatgpt subscription recovery and token handling
---------
Co-authored-by: 5p00kyy <admin@5p00ky.dev>
2026-06-08 18:19:18 +10:00
For ChatGPT Subscription , also repairs stale OpenAI API model names such as
` ` gpt - 5 ` ` that are not accepted by the Codex - backed ChatGPT account route .
2026-06-02 07:56:38 +05:30
"""
feat: add ChatGPT Subscription provider (#2876)
* feat: Add ChatGPT Subscription support and related features
- Introduced a new provider option for ChatGPT Subscription in the endpoint selection UI.
- Implemented OAuth flow for ChatGPT Subscription sign-in, including polling for authorization status.
- Updated admin interface to handle ChatGPT Subscription, including disabling API key input and providing user guidance.
- Enhanced cost tracking logic to differentiate between subscription and non-subscription endpoints.
- Added new slash commands for managing skills, including listing, searching, and invoking skills.
- Implemented caching for skill catalog to optimize performance.
- Updated tests to cover new ChatGPT Subscription functionality and ensure proper endpoint probing.
- Refactored existing code to accommodate new features and improve maintainability.
* refactor: share provider device-flow setup
- reuse one device-flow backend for Copilot and ChatGPT Subscription
- add one frontend device-flow helper for Settings and /setup
- put GitHub Copilot back into Add Models, now as a dropdown option
- make provider selection just select; clicking Add starts sign-in
- stop ChatGPT Subscription setup from opening auth tabs automatically
- make /setup copilot and /setup chatgpt-subscription work from chat
- show ChatGPT Subscription in the /setup suggestions
- show the real error message when setup fails
- add focused tests for the shared flow and setup UI
* feat(chatgpt-subscription): harden credential lifecycle and streamline auth UX
Backend:
- Resolve runtime bearer for provider-auth endpoints at probe time via a
shared _resolve_probe_key() that delegates to resolve_endpoint_runtime,
applied across all probe/refresh call sites.
- Skip live completion probes and health pings for discovery-only providers
(centralized behind _is_discovery_only_provider) — the Codex/Responses API
has no such endpoints, so status is derived from cached models.
- Never persist the short lived ChatGPT bearer to the plaintext sessions
table; proactively clear any stale bearer left by an earlier code path.
- Revoke orphaned ProviderAuthSession credentials when the last endpoint
backing them is deleted (_delete_orphaned_provider_auth), surfaced via
cleared_provider_auth in the delete response.
Frontend (admin.js):
- Auto-start the device-auth flow on provider selection so the authorization
panel (code + Authorize) shows immediately instead of behind a "Sign in" click.
- Remove the redundant top button for device auth providers, move retry
into the panel via an inline "Try again".
- Drop the self-evident hint text and add an execCommand clipboard fallback so
Copy works in non-secure (HTTP/LAN) contexts.
* fix: harden chatgpt subscription provider
* chore: remove PR media from branch
* Fix chatgpt subscription recovery and token handling
---------
Co-authored-by: 5p00kyy <admin@5p00ky.dev>
2026-06-08 18:19:18 +10:00
current_model = ( getattr ( sess , " model " , " " ) or " " ) . strip ( )
endpoint_url = ( getattr ( sess , " endpoint_url " , " " ) or " " ) . strip ( )
is_chatgpt_subscription = False
if current_model :
try :
from src . chatgpt_subscription import is_chatgpt_subscription_base
is_chatgpt_subscription = is_chatgpt_subscription_base ( endpoint_url )
if not is_chatgpt_subscription :
return False
except Exception :
return False
2026-06-02 07:56:38 +05:30
db = SessionLocal ( )
try :
# Prefer the endpoint whose base URL matches the session — we know the
# user already pointed this session at that endpoint, so its first
# cached model is the most defensible default.
ep = None
if getattr ( sess , " endpoint_url " , " " ) :
2026-06-02 19:40:22 +02:00
q = db . query ( ModelEndpoint ) . filter ( ModelEndpoint . is_enabled == True )
if owner :
from src . auth_helpers import owner_filter
q = owner_filter ( q , ModelEndpoint , owner )
endpoints = q . all ( )
2026-06-02 07:56:38 +05:30
for cand in endpoints :
if _session_url_matches_endpoint ( sess . endpoint_url or " " , cand . base_url or " " ) :
ep = cand
break
if not ep :
return False
feat: add ChatGPT Subscription provider (#2876)
* feat: Add ChatGPT Subscription support and related features
- Introduced a new provider option for ChatGPT Subscription in the endpoint selection UI.
- Implemented OAuth flow for ChatGPT Subscription sign-in, including polling for authorization status.
- Updated admin interface to handle ChatGPT Subscription, including disabling API key input and providing user guidance.
- Enhanced cost tracking logic to differentiate between subscription and non-subscription endpoints.
- Added new slash commands for managing skills, including listing, searching, and invoking skills.
- Implemented caching for skill catalog to optimize performance.
- Updated tests to cover new ChatGPT Subscription functionality and ensure proper endpoint probing.
- Refactored existing code to accommodate new features and improve maintainability.
* refactor: share provider device-flow setup
- reuse one device-flow backend for Copilot and ChatGPT Subscription
- add one frontend device-flow helper for Settings and /setup
- put GitHub Copilot back into Add Models, now as a dropdown option
- make provider selection just select; clicking Add starts sign-in
- stop ChatGPT Subscription setup from opening auth tabs automatically
- make /setup copilot and /setup chatgpt-subscription work from chat
- show ChatGPT Subscription in the /setup suggestions
- show the real error message when setup fails
- add focused tests for the shared flow and setup UI
* feat(chatgpt-subscription): harden credential lifecycle and streamline auth UX
Backend:
- Resolve runtime bearer for provider-auth endpoints at probe time via a
shared _resolve_probe_key() that delegates to resolve_endpoint_runtime,
applied across all probe/refresh call sites.
- Skip live completion probes and health pings for discovery-only providers
(centralized behind _is_discovery_only_provider) — the Codex/Responses API
has no such endpoints, so status is derived from cached models.
- Never persist the short lived ChatGPT bearer to the plaintext sessions
table; proactively clear any stale bearer left by an earlier code path.
- Revoke orphaned ProviderAuthSession credentials when the last endpoint
backing them is deleted (_delete_orphaned_provider_auth), surfaced via
cleared_provider_auth in the delete response.
Frontend (admin.js):
- Auto-start the device-auth flow on provider selection so the authorization
panel (code + Authorize) shows immediately instead of behind a "Sign in" click.
- Remove the redundant top button for device auth providers, move retry
into the panel via an inline "Try again".
- Drop the self-evident hint text and add an execCommand clipboard fallback so
Copy works in non-secure (HTTP/LAN) contexts.
* fix: harden chatgpt subscription provider
* chore: remove PR media from branch
* Fix chatgpt subscription recovery and token handling
---------
Co-authored-by: 5p00kyy <admin@5p00ky.dev>
2026-06-08 18:19:18 +10:00
if not is_chatgpt_subscription :
try :
from src . chatgpt_subscription import is_chatgpt_subscription_base
is_chatgpt_subscription = is_chatgpt_subscription_base ( getattr ( ep , " base_url " , " " ) or endpoint_url )
except Exception :
is_chatgpt_subscription = False
2026-06-02 07:56:38 +05:30
try :
cached = json . loads ( ep . cached_models ) if isinstance ( ep . cached_models , str ) else ( ep . cached_models or [ ] )
2026-06-15 13:49:27 -03:00
except Exception as e :
logger . warning ( " Failed to parse cached_models for endpoint %r " , getattr ( ep , " id " , " ? " ) , exc_info = e )
2026-06-02 07:56:38 +05:30
cached = [ ]
if not cached :
feat: add ChatGPT Subscription provider (#2876)
* feat: Add ChatGPT Subscription support and related features
- Introduced a new provider option for ChatGPT Subscription in the endpoint selection UI.
- Implemented OAuth flow for ChatGPT Subscription sign-in, including polling for authorization status.
- Updated admin interface to handle ChatGPT Subscription, including disabling API key input and providing user guidance.
- Enhanced cost tracking logic to differentiate between subscription and non-subscription endpoints.
- Added new slash commands for managing skills, including listing, searching, and invoking skills.
- Implemented caching for skill catalog to optimize performance.
- Updated tests to cover new ChatGPT Subscription functionality and ensure proper endpoint probing.
- Refactored existing code to accommodate new features and improve maintainability.
* refactor: share provider device-flow setup
- reuse one device-flow backend for Copilot and ChatGPT Subscription
- add one frontend device-flow helper for Settings and /setup
- put GitHub Copilot back into Add Models, now as a dropdown option
- make provider selection just select; clicking Add starts sign-in
- stop ChatGPT Subscription setup from opening auth tabs automatically
- make /setup copilot and /setup chatgpt-subscription work from chat
- show ChatGPT Subscription in the /setup suggestions
- show the real error message when setup fails
- add focused tests for the shared flow and setup UI
* feat(chatgpt-subscription): harden credential lifecycle and streamline auth UX
Backend:
- Resolve runtime bearer for provider-auth endpoints at probe time via a
shared _resolve_probe_key() that delegates to resolve_endpoint_runtime,
applied across all probe/refresh call sites.
- Skip live completion probes and health pings for discovery-only providers
(centralized behind _is_discovery_only_provider) — the Codex/Responses API
has no such endpoints, so status is derived from cached models.
- Never persist the short lived ChatGPT bearer to the plaintext sessions
table; proactively clear any stale bearer left by an earlier code path.
- Revoke orphaned ProviderAuthSession credentials when the last endpoint
backing them is deleted (_delete_orphaned_provider_auth), surfaced via
cleared_provider_auth in the delete response.
Frontend (admin.js):
- Auto-start the device-auth flow on provider selection so the authorization
panel (code + Authorize) shows immediately instead of behind a "Sign in" click.
- Remove the redundant top button for device auth providers, move retry
into the panel via an inline "Try again".
- Drop the self-evident hint text and add an execCommand clipboard fallback so
Copy works in non-secure (HTTP/LAN) contexts.
* fix: harden chatgpt subscription provider
* chore: remove PR media from branch
* Fix chatgpt subscription recovery and token handling
---------
Co-authored-by: 5p00kyy <admin@5p00ky.dev>
2026-06-08 18:19:18 +10:00
visible = [ ]
else :
try :
visible = _visible_models ( cached , getattr ( ep , " hidden_models " , None ) )
except Exception :
visible = cached
if current_model and current_model in { str ( item ) . strip ( ) for item in visible } :
2026-06-02 07:56:38 +05:30
return False
feat: add ChatGPT Subscription provider (#2876)
* feat: Add ChatGPT Subscription support and related features
- Introduced a new provider option for ChatGPT Subscription in the endpoint selection UI.
- Implemented OAuth flow for ChatGPT Subscription sign-in, including polling for authorization status.
- Updated admin interface to handle ChatGPT Subscription, including disabling API key input and providing user guidance.
- Enhanced cost tracking logic to differentiate between subscription and non-subscription endpoints.
- Added new slash commands for managing skills, including listing, searching, and invoking skills.
- Implemented caching for skill catalog to optimize performance.
- Updated tests to cover new ChatGPT Subscription functionality and ensure proper endpoint probing.
- Refactored existing code to accommodate new features and improve maintainability.
* refactor: share provider device-flow setup
- reuse one device-flow backend for Copilot and ChatGPT Subscription
- add one frontend device-flow helper for Settings and /setup
- put GitHub Copilot back into Add Models, now as a dropdown option
- make provider selection just select; clicking Add starts sign-in
- stop ChatGPT Subscription setup from opening auth tabs automatically
- make /setup copilot and /setup chatgpt-subscription work from chat
- show ChatGPT Subscription in the /setup suggestions
- show the real error message when setup fails
- add focused tests for the shared flow and setup UI
* feat(chatgpt-subscription): harden credential lifecycle and streamline auth UX
Backend:
- Resolve runtime bearer for provider-auth endpoints at probe time via a
shared _resolve_probe_key() that delegates to resolve_endpoint_runtime,
applied across all probe/refresh call sites.
- Skip live completion probes and health pings for discovery-only providers
(centralized behind _is_discovery_only_provider) — the Codex/Responses API
has no such endpoints, so status is derived from cached models.
- Never persist the short lived ChatGPT bearer to the plaintext sessions
table; proactively clear any stale bearer left by an earlier code path.
- Revoke orphaned ProviderAuthSession credentials when the last endpoint
backing them is deleted (_delete_orphaned_provider_auth), surfaced via
cleared_provider_auth in the delete response.
Frontend (admin.js):
- Auto-start the device-auth flow on provider selection so the authorization
panel (code + Authorize) shows immediately instead of behind a "Sign in" click.
- Remove the redundant top button for device auth providers, move retry
into the panel via an inline "Try again".
- Drop the self-evident hint text and add an execCommand clipboard fallback so
Copy works in non-secure (HTTP/LAN) contexts.
* fix: harden chatgpt subscription provider
* chore: remove PR media from branch
* Fix chatgpt subscription recovery and token handling
---------
Co-authored-by: 5p00kyy <admin@5p00ky.dev>
2026-06-08 18:19:18 +10:00
if is_chatgpt_subscription :
live_models = [ ]
if getattr ( ep , " provider_auth_id " , None ) :
try :
from src . chatgpt_subscription import fetch_available_models
from src . endpoint_resolver import resolve_endpoint_runtime
_base , api_key = resolve_endpoint_runtime ( ep , owner = owner )
if api_key :
live_models = fetch_available_models ( api_key )
if live_models :
ep . cached_models = json . dumps ( live_models )
db . commit ( )
except Exception :
live_models = [ ]
# ChatGPT Subscription recovery must use the live Codex catalog.
# Cached rows are only trusted above to avoid revalidating a model
# that is already present in the visible picker list.
cached = live_models
if not cached :
return False
try :
visible = _visible_models ( cached , getattr ( ep , " hidden_models " , None ) )
except Exception :
visible = cached
if current_model and current_model in { str ( item ) . strip ( ) for item in visible } :
return False
2026-06-02 13:37:14 +02:00
if not visible :
return False
model = visible [ 0 ]
2026-06-02 07:56:38 +05:30
if not isinstance ( model , str ) or not model . strip ( ) :
return False
model = model . strip ( )
# Persist so the next request, websocket reconnect, or page reload
# picks up the same model (we'd otherwise re-pick on every send
# and silently switch on the user if the cached order shifts).
feat: add ChatGPT Subscription provider (#2876)
* feat: Add ChatGPT Subscription support and related features
- Introduced a new provider option for ChatGPT Subscription in the endpoint selection UI.
- Implemented OAuth flow for ChatGPT Subscription sign-in, including polling for authorization status.
- Updated admin interface to handle ChatGPT Subscription, including disabling API key input and providing user guidance.
- Enhanced cost tracking logic to differentiate between subscription and non-subscription endpoints.
- Added new slash commands for managing skills, including listing, searching, and invoking skills.
- Implemented caching for skill catalog to optimize performance.
- Updated tests to cover new ChatGPT Subscription functionality and ensure proper endpoint probing.
- Refactored existing code to accommodate new features and improve maintainability.
* refactor: share provider device-flow setup
- reuse one device-flow backend for Copilot and ChatGPT Subscription
- add one frontend device-flow helper for Settings and /setup
- put GitHub Copilot back into Add Models, now as a dropdown option
- make provider selection just select; clicking Add starts sign-in
- stop ChatGPT Subscription setup from opening auth tabs automatically
- make /setup copilot and /setup chatgpt-subscription work from chat
- show ChatGPT Subscription in the /setup suggestions
- show the real error message when setup fails
- add focused tests for the shared flow and setup UI
* feat(chatgpt-subscription): harden credential lifecycle and streamline auth UX
Backend:
- Resolve runtime bearer for provider-auth endpoints at probe time via a
shared _resolve_probe_key() that delegates to resolve_endpoint_runtime,
applied across all probe/refresh call sites.
- Skip live completion probes and health pings for discovery-only providers
(centralized behind _is_discovery_only_provider) — the Codex/Responses API
has no such endpoints, so status is derived from cached models.
- Never persist the short lived ChatGPT bearer to the plaintext sessions
table; proactively clear any stale bearer left by an earlier code path.
- Revoke orphaned ProviderAuthSession credentials when the last endpoint
backing them is deleted (_delete_orphaned_provider_auth), surfaced via
cleared_provider_auth in the delete response.
Frontend (admin.js):
- Auto-start the device-auth flow on provider selection so the authorization
panel (code + Authorize) shows immediately instead of behind a "Sign in" click.
- Remove the redundant top button for device auth providers, move retry
into the panel via an inline "Try again".
- Drop the self-evident hint text and add an execCommand clipboard fallback so
Copy works in non-secure (HTTP/LAN) contexts.
* fix: harden chatgpt subscription provider
* chore: remove PR media from branch
* Fix chatgpt subscription recovery and token handling
---------
Co-authored-by: 5p00kyy <admin@5p00ky.dev>
2026-06-08 18:19:18 +10:00
db_session_q = db . query ( DBSession ) . filter ( DBSession . id == session_id )
if owner :
db_session_q = db_session_q . filter ( DBSession . owner == owner )
db_session = db_session_q . first ( )
2026-06-02 07:56:38 +05:30
if db_session :
db_session . model = model
db_session . updated_at = datetime . utcnow ( )
db . commit ( )
sess . model = model
logger . info (
feat: add ChatGPT Subscription provider (#2876)
* feat: Add ChatGPT Subscription support and related features
- Introduced a new provider option for ChatGPT Subscription in the endpoint selection UI.
- Implemented OAuth flow for ChatGPT Subscription sign-in, including polling for authorization status.
- Updated admin interface to handle ChatGPT Subscription, including disabling API key input and providing user guidance.
- Enhanced cost tracking logic to differentiate between subscription and non-subscription endpoints.
- Added new slash commands for managing skills, including listing, searching, and invoking skills.
- Implemented caching for skill catalog to optimize performance.
- Updated tests to cover new ChatGPT Subscription functionality and ensure proper endpoint probing.
- Refactored existing code to accommodate new features and improve maintainability.
* refactor: share provider device-flow setup
- reuse one device-flow backend for Copilot and ChatGPT Subscription
- add one frontend device-flow helper for Settings and /setup
- put GitHub Copilot back into Add Models, now as a dropdown option
- make provider selection just select; clicking Add starts sign-in
- stop ChatGPT Subscription setup from opening auth tabs automatically
- make /setup copilot and /setup chatgpt-subscription work from chat
- show ChatGPT Subscription in the /setup suggestions
- show the real error message when setup fails
- add focused tests for the shared flow and setup UI
* feat(chatgpt-subscription): harden credential lifecycle and streamline auth UX
Backend:
- Resolve runtime bearer for provider-auth endpoints at probe time via a
shared _resolve_probe_key() that delegates to resolve_endpoint_runtime,
applied across all probe/refresh call sites.
- Skip live completion probes and health pings for discovery-only providers
(centralized behind _is_discovery_only_provider) — the Codex/Responses API
has no such endpoints, so status is derived from cached models.
- Never persist the short lived ChatGPT bearer to the plaintext sessions
table; proactively clear any stale bearer left by an earlier code path.
- Revoke orphaned ProviderAuthSession credentials when the last endpoint
backing them is deleted (_delete_orphaned_provider_auth), surfaced via
cleared_provider_auth in the delete response.
Frontend (admin.js):
- Auto-start the device-auth flow on provider selection so the authorization
panel (code + Authorize) shows immediately instead of behind a "Sign in" click.
- Remove the redundant top button for device auth providers, move retry
into the panel via an inline "Try again".
- Drop the self-evident hint text and add an execCommand clipboard fallback so
Copy works in non-secure (HTTP/LAN) contexts.
* fix: harden chatgpt subscription provider
* chore: remove PR media from branch
* Fix chatgpt subscription recovery and token handling
---------
Co-authored-by: 5p00kyy <admin@5p00ky.dev>
2026-06-08 18:19:18 +10:00
" Recovered session model for %s — picked %r from endpoint %s " ,
2026-06-02 07:56:38 +05:30
session_id , model , ep . id ,
)
return True
except Exception as e :
db . rollback ( )
logger . warning ( " Failed to recover empty session model for %s : %s " , session_id , e )
return False
finally :
db . close ( )
2026-06-04 22:20:04 +10:00
def _set_user_time_from_request ( request : Request ) - > None :
""" Copy browser timezone headers into the per-request context.
This is intentionally ephemeral : it is used only while building prompts
and running tools for this request . It is not persisted or logged .
"""
try :
tz_offset = request . headers . get ( " x-tz-offset " )
tz_name = request . headers . get ( " x-tz-name " )
from src . user_time import clear_user_time_context , set_user_tz_name , set_user_tz_offset
clear_user_time_context ( )
if tz_offset is not None :
set_user_tz_offset ( tz_offset )
if tz_name :
set_user_tz_name ( tz_name )
except Exception :
pass
2026-05-31 23:58:26 +09:00
def setup_chat_routes (
session_manager ,
chat_handler ,
chat_processor ,
memory_manager ,
research_handler ,
upload_handler ,
memory_vector = None ,
webhook_manager = None ,
skills_manager = None ,
) - > APIRouter :
router = APIRouter ( tags = [ " chat " ] )
# ------------------------------------------------------------------ #
# POST /api/chat (non-streaming)
# ------------------------------------------------------------------ #
@router.post ( " /api/chat " , response_model = Dict [ str , str ] )
async def chat_endpoint ( request : Request , chat_request : ChatRequest ) - > Dict [ str , str ] :
2026-06-04 22:20:04 +10:00
_set_user_time_from_request ( request )
2026-05-31 23:58:26 +09:00
message = chat_request . message
session = chat_request . session
att_ids = chat_request . attachments or [ ]
use_web = chat_request . use_web
use_research = chat_request . use_research
time_filter = chat_request . time_filter
preset_id = chat_request . preset_id
# Verify the caller owns this session before loading it.
# Without this, any authenticated user can post into another user's chat.
_verify_session_owner ( request , session )
try :
sess = session_manager . get_session ( session )
except KeyError :
raise HTTPException ( 404 , f " Session ' { session } ' not found " )
fix(api): attribute bearer-token actions to the token owner on owner-scoped routes (#4054)
* fix(api): attribute bearer-token actions to the token owner on owner-scoped routes
Owner-scoped chat, session, and upload routes called
get_current_user(), which resolves a bearer ody_ API token to the
sandboxed "api" pseudo-user. A paired API-token client (companion, CLI,
IDE extension) therefore saw and created a separate "api"-owned silo
instead of the owner's data.
effective_user() already exists for exactly this: it attributes a token's
actions to request.state.api_token_owner, is identical to
get_current_user() for cookie sessions, and falls back safely when a
token has no owner. session_routes.py was already migrated; this
completes the migration for the remaining owner-scoped routes:
- chat_helpers.py: chat-privilege enforcement, message attribution, prefs/context
- chat_routes.py: orphaned-endpoint owner, session-auth owner, message search
- upload_routes.py: upload owner attribution + access checks
The /api/models swap is intentionally omitted: #4292 already migrated it
to effective_user (plus the chat-scope gate and ownerless-token 403), so
this PR keeps dev's version of routes/model_routes.py unchanged.
chat_routes.py keeps importing get_current_user for the workspace owner
gate; session_routes.py drops the now-unused import.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* test: target effective_user in auth monkeypatches and owner-scope assertion
The owner-scoped routes now call effective_user() instead of
get_current_user(), so the tests that stubbed get_current_user (or
asserted on it) follow suit:
- test_chat_helpers.py, test_review_regressions.py,
test_kv_cache_invalidation_2927.py: monkeypatch effective_user
- test_session_endpoint_owner_scope.py: assert the owner-scope guard uses
effective_user(request)
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
---------
Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-15 22:56:22 +01:00
owner = effective_user ( request )
2026-06-02 19:40:22 +02:00
if _clear_orphaned_session_endpoint ( sess , owner = owner ) :
2026-06-01 10:00:15 +09:00
raise HTTPException ( 400 , " Selected model endpoint was removed. Pick another model in Settings. " )
2026-05-31 23:58:26 +09:00
2026-06-02 07:56:38 +05:30
# Empty model + live endpoint = setup race (Issue #587). Repair from
# the endpoint's cached model list before privilege checks, which
# otherwise see "" and behave inconsistently with the allowlist.
2026-06-02 19:40:22 +02:00
_recover_empty_session_model ( sess , session , owner = owner )
2026-06-02 07:56:38 +05:30
if not getattr ( sess , " model " , " " ) . strip ( ) :
raise HTTPException (
400 ,
" No model selected for this chat. Open the model picker and choose one before sending. " ,
)
2026-05-31 23:58:26 +09:00
# Same allowed_models + daily-cap gate as chat_stream (mirror so the
# non-streaming path can't be used to bypass).
_enforce_chat_privileges ( request , sess )
2026-06-06 18:48:24 -06:00
tool_policy = build_effective_tool_policy ( last_user_message = message )
allow_tool_preprocessing = not tool_policy . block_all_tool_calls
2026-05-31 23:58:26 +09:00
# Inline memory command
2026-06-06 18:48:24 -06:00
memory_response = None
if not tool_policy . blocks ( " manage_memory " ) :
memory_response = await chat_handler . handle_memory_command ( sess , message )
2026-05-31 23:58:26 +09:00
if memory_response :
return { " response " : memory_response }
# Build shared context (preset, preprocess, preface, compact)
ctx = await build_chat_context (
sess , request , chat_handler , chat_processor ,
message = message ,
session_id = session ,
preset_id = preset_id ,
att_ids = att_ids ,
use_web = use_web ,
time_filter = time_filter ,
webhook_manager = webhook_manager ,
2026-06-06 18:48:24 -06:00
allow_tool_preprocessing = allow_tool_preprocessing ,
2026-05-31 23:58:26 +09:00
)
# Research injection
2026-06-06 18:48:24 -06:00
research_blocked_by_policy = (
tool_policy . blocks ( " trigger_research " )
or tool_policy . blocks ( " manage_research " )
)
if use_research and not research_blocked_by_policy :
2026-05-31 23:58:26 +09:00
try :
_r_ep , _r_model , _r_headers = _resolve_research_endpoint ( sess )
research_ctx = await research_handler . call_research_service (
message , _r_ep , _r_model , llm_headers = _r_headers
)
ctx . messages . insert (
len ( ctx . preface ) ,
untrusted_context_message ( " research context " , research_ctx ) ,
)
except Exception as e :
logger . error ( f " Research failed: { e } " )
reply = await llm_call_async (
sess . endpoint_url ,
sess . model ,
ctx . messages ,
headers = sess . headers ,
temperature = ctx . preset . temperature ,
max_tokens = ctx . preset . max_tokens ,
prompt_type = preset_id ,
fix(chat): stabilize system prompt, sequence memory extraction, and send stable session id to preserve KV cache (#3360)
* fix(chat): stabilize system prompt, sequence memory extraction, send stable session id to preserve KV cache
Fixes #2927. As diagnosed in the issue, three things in Odysseus's request
pattern actively destroyed local backends' (llama.cpp / LM Studio) KV-cache
continuity, forcing a full prompt re-evaluation (15-30s+) on every turn:
1. Dynamic content folded into the system prompt every turn. Both the chat
preface (ChatProcessor.build_context_preface) and the agent system prompt
(_build_system_prompt) injected current_datetime_prompt() — text that
changes every minute — directly into system-role messages, which llm_core
then concatenates into the single system message sent as the cached
prefix. Any byte difference there invalidates the entire cache. Moved this
to a new current_datetime_context_message() helper that returns a
standalone user-role message, inserted near the end of the array (right
before the latest user turn) instead of mixed into the system prompt. The
static system prefix (preset prompt + safety policy + agent base prompt)
now stays byte-identical across turns of the same session.
2. Memory/skill extraction side-requests competed with the main completion.
run_post_response_tasks fired extract_and_store / maybe_extract_skill via
asyncio.create_task — fire-and-forget coroutines that could overlap the
next turn's main request and steal llama.cpp's limited processing slots,
evicting the cached checkpoint. They're now queued through a new
_run_extraction_jobs_sequentially helper that waits for the session's
stream to go idle and runs the jobs strictly one at a time.
3. No stable session identifier was sent to local backends, so llama.cpp
assigned a new processing slot via LRU every turn ("session_id=<empty>
server-selected (LCP/LRU)"), losing slot affinity. Added
_apply_local_cache_affinity() in llm_core, which sets session_id and
cache_prompt: true on outgoing payloads — gated to self-hosted
OpenAI-compatible endpoints only (never api.openai.com or other cloud
providers, which reject unrecognized request fields with a 400). Threaded
session_id through stream_llm / llm_call_async / stream_agent_loop from
the existing Odysseus session id.
Tests in tests/test_kv_cache_invalidation_2927.py exercise the real payload-
assembly and scheduling code paths: byte-identical system prefix across two
turns of the same session (with a regression check that genuinely changed
instructions DO still change it), the dynamic time block landing as a
user-role message, extraction jobs waiting for the stream to go idle and
running sequentially, and the outgoing payload carrying a stable session_id
(same across turns of one session, different across sessions) only for
self-hosted endpoints. Updated tests/test_user_time.py for the new message
placement.
* fix(tests): accept owner= kwarg in normalize_model_id monkeypatch
The upstream normalize_model_id signature now takes an owner= keyword
argument, and chat_helpers.py passes owner=getattr(sess, "owner", None)
at the call site. Update the test stub lambda to **kwargs so it handles
the new argument without breaking, and update chat_helpers.py to forward
the owner parameter consistently.
---------
Co-authored-by: Alexandre Teixeira <111787685+alteixeira20@users.noreply.github.com>
2026-06-09 18:46:54 -03:00
session_id = session ,
2026-05-31 23:58:26 +09:00
)
_clean_reply , _clean_md = clean_thinking_for_save ( reply , { " model " : sess . model } )
sess . add_message ( ChatMessage ( " assistant " , _clean_reply , metadata = _clean_md ) )
from core . database import update_session_last_accessed
update_session_last_accessed ( session )
session_manager . save_sessions ( )
# Background tasks (memory, webhook, auto-name)
run_post_response_tasks (
sess , session_manager , session , message , reply , None ,
ctx . uprefs , memory_manager , memory_vector , webhook_manager ,
character_name = ctx . preset . character_name ,
owner = ctx . user ,
2026-06-06 18:48:24 -06:00
allow_background_extraction = not tool_policy . block_all_tool_calls ,
2026-05-31 23:58:26 +09:00
)
return { " response " : reply }
# ------------------------------------------------------------------ #
# POST /api/chat_stream
# ------------------------------------------------------------------ #
@router.post ( " /api/chat_stream " )
async def chat_stream ( request : Request ) - > StreamingResponse :
body = None
try :
if request . headers . get ( " content-type " , " " ) . startswith ( " application/json " ) :
try :
body = await request . json ( )
except json . JSONDecodeError as e :
raise HTTPException ( 400 , f " Invalid JSON: { e } " )
except HTTPException :
raise
except Exception as e :
raise HTTPException ( 400 , f " Request parsing error: { e } " )
2026-06-04 22:20:04 +10:00
_set_user_time_from_request ( request )
2026-05-31 23:58:26 +09:00
form_data = await request . form ( )
message = form_data . get ( " message " )
session = form_data . get ( " session " )
attachments = form_data . get ( " attachments " )
use_web = form_data . get ( " use_web " )
use_research = form_data . get ( " use_research " )
time_filter = form_data . get ( " time_filter " )
preset_id = form_data . get ( " preset_id " )
2026-06-12 01:14:41 +07:00
# Issue #3229: API callers send JSON, not FormData. Read from the
# JSON body as fallback so callers who send {"allow_bash": true}
# actually get bash enabled.
allow_bash = form_data . get ( " allow_bash " ) or ( body or { } ) . get ( " allow_bash " )
allow_web_search = form_data . get ( " allow_web_search " ) or ( body or { } ) . get ( " allow_web_search " )
2026-05-31 23:58:26 +09:00
use_rag = form_data . get ( " use_rag " )
search_context = form_data . get ( " search_context " ) # pre-fetched web search results (compare mode)
compare_mode = str ( form_data . get ( " compare_mode " , " " ) ) . lower ( ) == " true "
incognito = str ( form_data . get ( " incognito " , " " ) ) . lower ( ) == " true "
2026-06-09 09:40:20 +09:00
# Plan mode is not part of the merge-ready UI. Ignore stale clients or
# manual form posts that still send plan_mode=true.
plan_mode = False
2026-05-31 23:58:26 +09:00
chat_mode = str ( form_data . get ( " mode " , " " ) ) . lower ( ) # 'chat' or 'agent'
feat(agent): confine agent file/shell tools to a selectable workspace (#3665)
* feat(agent): workspace confinement via context-local binding + get_workspace tool
Bind the per-turn workspace once in execute_tool_block; the shared path
resolvers (_resolve_tool_path / _resolve_search_root) and the subprocess cwd
helper (agent_cwd) read it, so file tools + bash/python are confined centrally
and a new tool that uses the shared helpers cannot accidentally bypass it.
Adds the admin-gated /api/workspace/browse picker, a workspace pill + directory
modal (reusing existing modal/button CSS), the /workspace slash command, and a
get_workspace tool (replaces a system-prompt block). Confinement is OS-agnostic
(realpath/normcase/commonpath) and docker-safe (container paths, no host
assumptions). Reopens #2023.
* ux(workspace): clarify workspace is not a sandbox
Picker modal note + pill tooltip + get_workspace tool/output wording now state
plainly: read_file/write_file/edit_file/grep/glob/ls are confined to the folder,
but bash/python only start there (cwd) and are not sandboxed. Modal note reuses
the existing .muted class.
* fix(agent): treat an active workspace as file-work intent
A vague low-signal message (e.g. "look at the local project") matches no
domain keywords, so tool retrieval is skipped and only always-available tools
are offered — leaving the agent with no file access even though a workspace is
set. When a workspace is active, include the file/code tools (incl.
get_workspace) on low-signal turns so the agent can act on the folder.
Also requires the tool index (ChromaDB) to be reachable for normal retrieval;
that is an environment dependency, not part of this change.
* ux(workspace): hide pill + overflow entry in chat mode
Workspace only scopes the agent's file/shell tools, so the pill and the
overflow 'Workspace' entry are agent-only now — hidden in chat mode like the
bash toggle. Mode read from the DOM in syncWorkspaceIndicator; applyMode() is
called from the agent/chat setMode handler.
* prompt(tools): steer bash/python to defer to the dedicated file tools
bash/python schema descriptions (what native-tool-calling models read) were
bare and gave no steer, so models would do file ops via the shell (e.g. writing
SVG/HTML, which then dumps raw markup into the tool preview). Tell bash/python
in the schema + tool-index + prompt section to prefer read_file/write_file/
edit_file/grep/glob/ls and only be used for what those do not cover.
* prompt(tools): keep bash/python deferral generic (no hardcoded tool names)
Reference 'a dedicated tool' rather than listing read_file/write_file/grep/etc.
by name, so the guidance does not go stale if those tools are renamed.
* style(workspace): drop em-dashes from added code comments/strings
* ux(workspace): terser non-sandbox note in picker (no tool-name list)
* ux(workspace): mirror terse non-sandbox wording in pill tooltip
* chore: untrack local venv symlink (run-only, not part of the feature)
* prompt(workspace): keep get_workspace text generic (no hardcoded tool names)
* fix(agent): low-signal + workspace surfaces only read-only file tools
Intersect the files tool group with PLAN_MODE_READONLY_TOOLS so a vague message
in a workspace exposes read_file/grep/glob/ls/get_workspace for exploration, but
not write_file/edit_file/bash/python -- those wait for a request that actually
calls for them (RAG retrieval still adds them on a real ask).
* feat(workspace): cap browse listing at 500 dirs with a truncated hint
Mirror the filesystem_tools._CODENAV_MAX_HITS pattern with a module-local
_MAX_BROWSE_DIRS so a directory with thousands of children does not dump every
row into the picker; the response carries a truncated flag and the modal tells
the user to type a path to jump in.
* chore: untrack local venv symlink (run-only artifact)
* fix(workspace): vet the workspace root against the sensitive-path deny list at bind time
The in-workspace resolver deny-lists sensitive paths inside the workspace,
but the empty-path search root is the workspace itself, so a workspace of
~/.ssh could be listed via ls with no path. vet_workspace() (public, in
tool_execution next to the resolvers) rejects non-directories and sensitive
roots before the path is ever bound; chat_routes uses it instead of its
inline isdir check.
* fix(workspace): reject filesystem roots and stop showing rejected workspaces as active
Review findings from #3665:
P2: vet_workspace accepted / (and would accept drive/UNC roots), which makes
every absolute path 'inside' the workspace and collapses confinement into
host-wide file access. A root is its own dirname, so reject when
dirname(resolved) == resolved; the browse response now carries a selectable
flag and the picker disables 'Use this folder' on unselectable dirs.
P3: /workspace set stored any string client-side and the chat route silently
dropped rejected values, so the pill could claim a confinement that was not
in effect. New admin-gated /api/workspace/vet validates manual paths before
they persist (canonical path returned), and when a posted workspace is
rejected at send time the stream emits workspace_rejected so the client
clears the stored value and toasts instead of continuing silently.
* fix(workspace): check caller privilege before vetting the posted workspace
Review finding: /api/chat_stream called vet_workspace() on the posted value
for every caller and emitted workspace_rejected on failure, so a non-admin
who can chat but cannot use file/shell tools could distinguish existing
directories from missing/file/sensitive/root paths by whether the event
appeared. The resolution now lives in _resolve_request_workspace, which
drops the submitted value uniformly for non-admin callers, with no vetting
and no event, before the path ever touches the filesystem. Admin and
single-user behavior is unchanged. Test pins that valid and invalid paths
are indistinguishable for a non-admin and that vet_workspace is never
invoked for them.
2026-06-11 18:17:54 +02:00
# Workspace: confine the agent's file/shell tools to this folder.
workspace , workspace_rejected = _resolve_request_workspace (
request , form_data . get ( " workspace " )
)
feat: Add plan mode to the chat agent (#638)
* feat: Add plan mode to the chat agent
Adds a plan mode: the agent investigates read-only, proposes a checklist, and
waits for approval before changing anything. On approval it runs with full
tools and checks items off as it goes. Enforcement reuses the existing
disabled_tools gate.
Includes a slash command: `/plan [on|off]` (and `/toggle plan`) to flip the
plan toggle from the chat input.
- src/tool_security.py, src/mcp_manager.py: read-only allowlist (tools + MCP).
- src/agent_loop.py, routes/chat_routes.py: union the disabled set, prepend the
plan directive, force agent mode.
- static/: plan toggle pill, Approve & Run, dockable plan window, task-list
checkboxes, and the /plan slash command.
- tests/test_plan_mode.py.
* Plan mode: persistent re-referenceable plan + agent write-back
Three improvements so a long plan survives a weak model and stays in reach:
1. Re-reference the plan (out-of-context fix). On the execution turn the frontend
sends the approved checklist back (`approved_plan`); the backend pins it as a
top-of-context `## ACTIVE PLAN` system note (kept by the context trimmer), so
the agent can always re-read the plan instead of losing the thread on a long
run. New `build_active_plan_note()` (unit-tested).
2. Re-open / dock the plan anytime. The plan checklist is stored per-session
(localStorage). When a plan exists, the plan-mode button opens a small menu
("Show plan" / "Plan mode: On/Off") that re-opens the side-dockable plan
window — so it can stay docked while the agent works. The window live-refreshes
as the plan changes.
3. Agent write-back: new `update_plan` tool. The agent calls it to tick steps
`- [x]` after finishing them, or to revise steps when the user asks. Marker
tool (no I/O) → `plan_update` SSE event → the stored plan + docked window
update live. The ACTIVE PLAN note instructs the agent to use it.
Backend: src/agent_loop.py (param + pin + note builder + emit + prompt blurb),
src/tool_execution.py (update_plan handler), routes/chat_routes.py (parse
`approved_plan`, relay `plan_update`), registration in tool_schemas / agent_tools
/ tool_index (always-available, not admin-gated).
Frontend: static/js/chat.js (plan store, send `approved_plan`, handle
`plan_update`, capture restated checklists), static/app.js (plan-button menu),
static/js/planWindow.js (`isPlanWindowOpen`), static/js/storage.js (PLAN key).
Tests: tests/test_plan_mode.py (plan-note), tests/test_update_plan_tool.py.
* Plan mode: drop bash/python, rely on read-only discovery tools
Shell can mutate (write files, hit the network) and can't be constrained to
read-only at the tool layer, so plan mode no longer relies on a prompt to keep
it well-behaved — bash/python are removed from the read-only allowlist and added
to the fail-closed block set. Discovery is covered by the dedicated read-only
tools (read_file, grep, glob, ls) instead.
Rewrites the plan-mode directive to state shell is disabled and lists the
available read-only tools positively. Addresses review feedback on #638.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Comment: note _MCP_READONLY_VERBS are prefixes not whole words
Clarifies that entries like "summar" are intentional stems matched via
startswith (covers summarise/summarize/summary), not typos. Addresses review
feedback on #638.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Plan mode: clarify why gating inverts the allowlist into a denylist
Rename _PLAN_MODE_FALLBACK_BLOCK -> _PLAN_MODE_KNOWN_MUTATORS and rewrite the
comments. The tool gate is a denylist (disabled_tools); plan mode's policy is an
allowlist, so it returns the inverse (all known tool names minus the allowlist).
The static mutator set is a backstop for the schema-derived name list, which
misses XML-only tools and can fail to import. Addresses review feedback on #638.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Plan mode: stop hardcoding the read-only tool list in the directive
The model is already shown its available (read-only) tools by _assemble_prompt,
which removes every disabled tool. Enumerating them again in the directive only
duplicated that list and would drift as tools change. Point at the tools listed
below instead. Addresses review feedback on #638.
2026-06-05 16:32:25 +02:00
# Plan mode is a modifier on agent mode — it only makes sense with tools.
if plan_mode :
chat_mode = " agent "
# An approved plan being EXECUTED: the frontend sends the checklist back
# on each turn so we can pin it in context. This way a long plan on a
# weak model survives history truncation — the agent can always re-read
# the plan. Ignored while still proposing (plan_mode on). Capped so a
# huge plan can't blow the prompt.
approved_plan = " "
if not plan_mode :
approved_plan = ( form_data . get ( " approved_plan " ) or " " ) . strip ( ) [ : 8192 ]
2026-05-31 23:58:26 +09:00
# Did the USER explicitly pick agent mode? (vs. us auto-escalating
# below). Skill extraction should only learn from real agent sessions,
# not chats we quietly promoted for a notes/calendar intent.
user_requested_agent = ( chat_mode == " agent " )
2026-07-07 00:50:07 +00:00
_search_enabled = (
str ( allow_web_search ) . lower ( ) == " true "
or str ( use_web ) . lower ( ) == " true "
)
2026-05-31 23:58:26 +09:00
# Intent auto-escalation: if the user is clearly asking the assistant
# to create a todo, reminder, or calendar event, promote chat → agent
# for this turn so the LLM has access to manage_notes / manage_calendar.
# This is a LIGHT promotion — see the disabled_tools block below, which
# withholds shell/code/file tools so the model doesn't try to `bash`
# its way through a plain chat request (and fail, especially with the
# shell disabled).
auto_escalated = False
2026-06-04 22:20:04 +10:00
_tool_intent = _classify_tool_intent ( message ) if isinstance ( message , str ) else None
if chat_mode == " chat " and _tool_intent and _tool_intent . needs_tools :
2026-05-31 23:58:26 +09:00
chat_mode = " agent "
auto_escalated = True
2026-06-04 22:20:04 +10:00
logger . info (
" chat→agent auto-escalation: category= %s reason= %s " ,
_tool_intent . category ,
_tool_intent . reason ,
)
2026-07-07 00:50:07 +00:00
elif chat_mode == " chat " and _search_enabled :
chat_mode = " agent "
auto_escalated = True
logger . info ( " chat→agent auto-escalation: search enabled " )
2026-05-31 23:58:26 +09:00
active_doc_id = form_data . get ( " active_doc_id " , " " ) . strip ( )
logger . info ( f " [doc-inject] chat_mode= { chat_mode } , active_doc_id= { active_doc_id !r} " )
Open email context for agent, email search across All Mail, cookbook serve polish
- Agent: pass the open email reader (uid/folder/account/from/subject/body
preview) on every chat submit so 'reply to this' / 'write email saying
hi' route to ui_control open_email_reply with the right UID instead of
inventing a new .md draft. Code-level enforcement (chat_routes strips
create_document + send_email when active_email is set); cross-session
active_doc_id is now trusted instead of being silently dropped.
set_active_email/clear_active_email tool-layer helpers in
tool_implementations.
- ui_control open_email_reply: optional body argument so the agent can
open-and-write in one call; envelope now forwards uid/folder/account/
body/panel through tool_output. Tool description sharpened and the
parser rejects empty bodies on reply/reply-all (forces the agent to
write rather than open an empty draft).
- Email library: search now runs against [Gmail]/All Mail when the
current folder is INBOX (archived emails surface). Whirlpool spinner
+ 'Searching…' placeholder while in flight. Each search result is
stamped with its source folder so clicks open the right email instead
of whatever shares its UID in INBOX. Search no longer re-applies the
same text pill locally (which only checks subject/from/snippet, never
body) so body-only matches don't get dropped after IMAP returns them.
Initial inbox load bumped 100→500.
- Email favorites: 'Favorite (pin to top)' / 'Unfavorite' in both the
card menu and the open-reader more menu, backed by a new
/api/email/flag/{uid}?on=true|false endpoint. Flagged emails always
bubble to the top of the grid regardless of active sort.
- AI reply in doc editor: never overwrites existing draft text or the
quoted history. AI suggestion is prepended; AI-generated 'On …
wrote:' re-quotes are stripped so the original quote isn't visually
edited.
- Cookbook serve: pre-launch GPU driver / has_gpu / install / version-
floor checks (vllm minimax_m2 needs 0.10.0+, deepseek_r1 needs 0.7.0
etc.) before the launch chain starts. Detect 'another model already
running on this host' and offer Stop & launch (with graceful then
force tmux kill helpers, port release wait). Per-vendor deep-link
buttons (vLLM recipe / SGLang cookbook) with hardware hash. Backend
picker is now a custom dropdown with accent-coloured logos for vLLM,
SGLang, llama.cpp, Ollama, Diffusers; same glyphs added next to
package names in Dependencies. Runtime-readiness note moved inside
the panel (green when ready, red when missing) with an × dismiss.
Esc collapses the expanded card; expanded card scrolls when it
overflows; Trust Remote / Auto Tool / Reasoning Parser / Enforce
Eager / Prefix Caching / Expert Parallel / Speculative / MoE Env on
one row (Reasoning Parser auto-detected per model family).
Dtype→Row 1, GPUs→Row 2 (rightmost). Removed redundant GPU 'auto'
input — command builders read from the GPU button strip. Default
cookbook open is Download tab.
- Cookbook hwfit: 'Model (latest)' / 'Model (oldest)' header sorts by
release_date; release dates can be backfilled with the new
scripts/backfill_model_release_dates.py and recipe metadata pulled
with scripts/import_from_vllm_recipes.py against the upstream
vllm-project/recipes catalog (vllm_recipe + min_vllm_version stamped
on entries).
- Calendar: Quick add hint cycles a random Odysseus-themed example per
open (wooden horse Friday, crew muster 10am daily, council on
Ithaca, …). Typing a time like '11pm' in the event title updates
the hero clock live.
- Doc editor: email-mode Reply button (sparkle icon, accent) opens the
same Fast/Full + context popover the email reader uses; Ctrl+Alt+M
toggles markdown preview.
- Memories panel: custom sort picker with per-option icons, default
'Latest', visible Enabled/Disabled toggle text matching the section
description style.
2026-06-15 20:47:51 +09:00
# Active email reader — when the user has an email open in the UI, the
# frontend passes its uid/folder/account so "reply", "summarize this",
# etc. resolve to the real email instead of the agent inventing a
# fake markdown draft.
active_email_uid = form_data . get ( " active_email_uid " , " " ) . strip ( )
active_email_folder = form_data . get ( " active_email_folder " , " INBOX " ) . strip ( ) or " INBOX "
active_email_account = form_data . get ( " active_email_account " , " " ) . strip ( )
active_email_ctx : Optional [ Dict [ str , str ] ] = None
# Always reset between requests so a stale active-email pointer from
# a previous turn (different reader closed, different account, etc.)
# can't leak in when the user has no email open this turn.
try :
from src . tool_implementations import clear_active_email
clear_active_email ( )
except Exception :
pass
if active_email_uid :
active_email_ctx = {
" uid " : active_email_uid ,
" folder " : active_email_folder ,
" account " : active_email_account ,
}
# Try to enrich with subject + from so the agent's system prompt
# block can quote them. Best-effort: a stale cache is fine, a
# missing email just means we pass uid/folder/account only.
try :
from routes . email_routes import _read_cache_get , _read_cache_key
_ck = _read_cache_key ( active_email_account or None , active_email_folder , active_email_uid , owner = get_current_user ( request ) )
_cached_email = _read_cache_get ( _ck )
if _cached_email and isinstance ( _cached_email , dict ) :
active_email_ctx [ " subject " ] = str ( _cached_email . get ( " subject " ) or " " )
active_email_ctx [ " from " ] = str (
_cached_email . get ( " from_address " )
or _cached_email . get ( " from " )
or _cached_email . get ( " from_name " )
or " "
)
_body_preview = ( _cached_email . get ( " body " ) or " " ) [ : 2000 ]
if _body_preview :
active_email_ctx [ " body_preview " ] = _body_preview
except Exception as _e :
logger . debug ( f " [email-inject] cache enrich skipped: { _e } " )
# Stash so email tools can resolve "this email" without UID guessing.
try :
from src . tool_implementations import set_active_email
set_active_email (
uid = active_email_uid ,
folder = active_email_folder ,
account = active_email_account or None ,
subject = active_email_ctx . get ( " subject " ) ,
sender = active_email_ctx . get ( " from " ) ,
)
except Exception as _e :
logger . debug ( f " [email-inject] set_active_email failed: { _e } " )
logger . info (
" [email-inject] active_email uid= %s folder= %s account= %s subject= %r " ,
active_email_uid , active_email_folder , active_email_account or " (default) " ,
active_email_ctx . get ( " subject " , " " ) ,
)
2026-05-31 23:58:26 +09:00
try :
# Attachment-only sends: skip the message-required check when the
# user has attached one or more files (the attachment IS the action).
_has_atts = (
bool ( body and isinstance ( body . get ( " attachments " ) , list ) and body [ " attachments " ] )
or bool ( form_data . get ( " attachments " ) )
)
message , session = coerce_message_and_session (
body , message , session , session_manager , allow_empty = _has_atts ,
)
# Verify ownership AFTER coerce (which may resolve a default session)
# but BEFORE loading. Prevents cross-user session hijack.
_verify_session_owner ( request , session )
sess = session_manager . get_session ( session )
fix(api): attribute bearer-token actions to the token owner on owner-scoped routes (#4054)
* fix(api): attribute bearer-token actions to the token owner on owner-scoped routes
Owner-scoped chat, session, and upload routes called
get_current_user(), which resolves a bearer ody_ API token to the
sandboxed "api" pseudo-user. A paired API-token client (companion, CLI,
IDE extension) therefore saw and created a separate "api"-owned silo
instead of the owner's data.
effective_user() already exists for exactly this: it attributes a token's
actions to request.state.api_token_owner, is identical to
get_current_user() for cookie sessions, and falls back safely when a
token has no owner. session_routes.py was already migrated; this
completes the migration for the remaining owner-scoped routes:
- chat_helpers.py: chat-privilege enforcement, message attribution, prefs/context
- chat_routes.py: orphaned-endpoint owner, session-auth owner, message search
- upload_routes.py: upload owner attribution + access checks
The /api/models swap is intentionally omitted: #4292 already migrated it
to effective_user (plus the chat-scope gate and ownerless-token 403), so
this PR keeps dev's version of routes/model_routes.py unchanged.
chat_routes.py keeps importing get_current_user for the workspace owner
gate; session_routes.py drops the now-unused import.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* test: target effective_user in auth monkeypatches and owner-scope assertion
The owner-scoped routes now call effective_user() instead of
get_current_user(), so the tests that stubbed get_current_user (or
asserted on it) follow suit:
- test_chat_helpers.py, test_review_regressions.py,
test_kv_cache_invalidation_2927.py: monkeypatch effective_user
- test_session_endpoint_owner_scope.py: assert the owner-scope guard uses
effective_user(request)
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
---------
Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-15 22:56:22 +01:00
owner = effective_user ( request )
2026-06-02 19:40:22 +02:00
if _clear_orphaned_session_endpoint ( sess , owner = owner ) :
2026-06-01 10:00:15 +09:00
raise HTTPException ( 400 , " Selected model endpoint was removed. Pick another model in Settings. " )
2026-06-02 07:56:38 +05:30
# Issue #587: picker shows a model from the endpoint cache but
# s.model never made it onto the DB row (first-send race after
# endpoint setup, or a previous endpoint delete/recreate). Pull
# the first cached model off the matching endpoint so the
# upstream isn't called with model="" (which surfaces as a
# generic 401/503).
2026-06-02 19:40:22 +02:00
_recover_empty_session_model ( sess , session , owner = owner )
2026-06-02 07:56:38 +05:30
if not getattr ( sess , " model " , " " ) . strip ( ) :
raise HTTPException (
400 ,
" No model selected for this chat. Open the model picker and choose one before sending. " ,
)
2026-07-07 00:50:07 +00:00
if (
chat_mode == " chat "
and isinstance ( message , str )
and ( not _tool_intent or not _tool_intent . needs_tools )
and _is_contextual_web_followup ( message , sess )
) :
_tool_intent = ToolIntent ( True , " web " , " contextual web lookup follow-up " )
chat_mode = " agent "
auto_escalated = True
logger . info (
" chat→agent auto-escalation: category= %s reason= %s " ,
_tool_intent . category ,
_tool_intent . reason ,
)
2026-05-31 23:58:26 +09:00
except SessionNotFoundError as e :
raise HTTPException ( 404 , str ( e ) )
except ( ValueError , ValidationError ) :
raise HTTPException ( 400 , " Invalid request parameters " )
# ------------------------------------------------------------------ #
# Privilege gates that must fire BEFORE any LLM work / token spend.
# 1. allowed_models — reject if session.model isn't in the user's
# configured allowlist (empty list = "no restriction").
# 2. max_messages_per_day — count user-role ChatMessage rows owned
# by this user in the last UTC day; 429 if at/over the cap.
# Admins always have full privileges via get_privileges (returns
# ADMIN_PRIVILEGES wholesale) so this is a no-op for them.
_enforce_chat_privileges ( request , sess )
# Ensure session has auth headers
fix(api): attribute bearer-token actions to the token owner on owner-scoped routes (#4054)
* fix(api): attribute bearer-token actions to the token owner on owner-scoped routes
Owner-scoped chat, session, and upload routes called
get_current_user(), which resolves a bearer ody_ API token to the
sandboxed "api" pseudo-user. A paired API-token client (companion, CLI,
IDE extension) therefore saw and created a separate "api"-owned silo
instead of the owner's data.
effective_user() already exists for exactly this: it attributes a token's
actions to request.state.api_token_owner, is identical to
get_current_user() for cookie sessions, and falls back safely when a
token has no owner. session_routes.py was already migrated; this
completes the migration for the remaining owner-scoped routes:
- chat_helpers.py: chat-privilege enforcement, message attribution, prefs/context
- chat_routes.py: orphaned-endpoint owner, session-auth owner, message search
- upload_routes.py: upload owner attribution + access checks
The /api/models swap is intentionally omitted: #4292 already migrated it
to effective_user (plus the chat-scope gate and ownerless-token 403), so
this PR keeps dev's version of routes/model_routes.py unchanged.
chat_routes.py keeps importing get_current_user for the workspace owner
gate; session_routes.py drops the now-unused import.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* test: target effective_user in auth monkeypatches and owner-scope assertion
The owner-scoped routes now call effective_user() instead of
get_current_user(), so the tests that stubbed get_current_user (or
asserted on it) follow suit:
- test_chat_helpers.py, test_review_regressions.py,
test_kv_cache_invalidation_2927.py: monkeypatch effective_user
- test_session_endpoint_owner_scope.py: assert the owner-scope guard uses
effective_user(request)
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
---------
Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-15 22:56:22 +01:00
resolve_session_auth ( sess , session , owner = effective_user ( request ) )
2026-05-31 23:58:26 +09:00
# Check for research_pending BEFORE mode persist overwrites it
do_research = str ( use_research ) . lower ( ) == " true "
if not do_research :
fix: stop leaking DB connections when persisting session mode (#64)
chat_routes.py persisted a session's "mode" in three best-effort spots —
reading the current mode, writing the effective mode, and setting
research_pending on the stream path. Each opened a session with SessionLocal()
and called .close() as the LAST statement inside a try/except, so if anything
before close() raised (e.g. a SQLite "database is locked" under concurrent chat
streams) the except only logged and the connection was never returned to the
pool.
DATABASE_URL defaults to file-backed SQLite, whose engine uses SQLAlchemy's
default QueuePool (5 connections + 10 overflow). Repeated leaks on these hot
paths exhaust the pool; later requests then block for pool_timeout and fail
with "QueuePool limit ... reached", taking the app down until restart.
Move the logic into two best-effort helpers in core.database, next to the
existing session helpers (update_session_last_accessed, get_session_by_id):
- get_session_mode(session_id) -> Optional[str]
- set_session_mode(session_id, mode) -> bool
Both route through the existing get_db_session() context manager, which commits
on success, rolls back on error, and always closes in a finally, so the
connection is returned to the pool on every path. chat_routes.py now calls
these instead of hand-rolling sessions, also removing three copies of the same
try/except.
Add tests/test_session_mode_helpers.py: the helpers commit+close on success
and, on a mid-operation DB error, swallow + roll back + close (no leak). The
error-path tests fail against the old close()-inside-try pattern.
2026-06-01 00:57:48 -04:00
if get_session_mode ( session ) == ' research_pending ' :
do_research = True
logger . info ( f " Session { session } in research_pending — auto-triggering research " )
2026-05-31 23:58:26 +09:00
att_ids = [ ]
if body and isinstance ( body . get ( " attachments " ) , list ) :
att_ids = [ str ( x ) for x in body [ " attachments " ] ]
elif attachments :
try :
att_ids = [ str ( x ) for x in json . loads ( attachments ) ]
2026-06-15 13:49:27 -03:00
except Exception as e :
logger . warning ( " Failed to parse attachments JSON, ignoring attachments " , exc_info = e )
2026-05-31 23:58:26 +09:00
no_memory = str ( form_data . get ( " no_memory " , " " ) ) . lower ( ) == " true "
2026-06-06 18:48:24 -06:00
pre_context_tool_policy = build_effective_tool_policy (
last_user_message = message ,
)
allow_tool_preprocessing = not pre_context_tool_policy . block_all_tool_calls
2026-05-31 23:58:26 +09:00
# Build shared context (stream path uses enhanced_message for context preface)
ctx = await build_chat_context (
sess , request , chat_handler , chat_processor ,
message = message ,
session_id = session ,
preset_id = preset_id ,
att_ids = att_ids ,
use_web = use_web ,
use_rag = use_rag ,
time_filter = time_filter ,
incognito = incognito ,
no_memory = no_memory ,
search_context = search_context ,
compare_mode = compare_mode ,
webhook_manager = webhook_manager ,
use_enhanced_message = True ,
# Skills index only ships when the model can actually call
# manage_skills (agent mode). In plain chat or incognito the
# index would be useless / unwanted noise.
agent_mode = ( chat_mode == " agent " ) ,
2026-06-06 18:48:24 -06:00
allow_tool_preprocessing = allow_tool_preprocessing ,
2026-05-31 23:58:26 +09:00
)
_research_flags = { " do " : do_research } # Mutable container for generator scope
# Query active document — prefer explicit ID from frontend, fall back to session lookup
active_doc = None
_doc_db = SessionLocal ( )
try :
if active_doc_id :
logger . info ( f " [doc-inject] active_doc_id from frontend: { active_doc_id } " )
2026-06-02 04:46:40 +02:00
# Scope to the caller's documents. The session and in-memory
# fallbacks below are already owner/session-bound; this
# explicit-id path looked up by id alone, so a user could
# inject another user's document by passing its id.
_doc_q = _doc_db . query ( DBDocument ) . filter ( DBDocument . id == active_doc_id )
active_doc = _owner_session_filter ( _doc_q , ctx . user ) . first ( )
2026-05-31 23:58:26 +09:00
if active_doc :
2026-06-04 17:27:46 +05:00
doc_session = active_doc . session_id
doc_owner = getattr ( active_doc , " owner " , None )
if doc_owner and ctx . user and doc_owner != ctx . user :
logger . warning (
" [doc-inject] ignoring active_doc_id %s owned by another user " ,
active_doc_id ,
)
active_doc = None
else :
Open email context for agent, email search across All Mail, cookbook serve polish
- Agent: pass the open email reader (uid/folder/account/from/subject/body
preview) on every chat submit so 'reply to this' / 'write email saying
hi' route to ui_control open_email_reply with the right UID instead of
inventing a new .md draft. Code-level enforcement (chat_routes strips
create_document + send_email when active_email is set); cross-session
active_doc_id is now trusted instead of being silently dropped.
set_active_email/clear_active_email tool-layer helpers in
tool_implementations.
- ui_control open_email_reply: optional body argument so the agent can
open-and-write in one call; envelope now forwards uid/folder/account/
body/panel through tool_output. Tool description sharpened and the
parser rejects empty bodies on reply/reply-all (forces the agent to
write rather than open an empty draft).
- Email library: search now runs against [Gmail]/All Mail when the
current folder is INBOX (archived emails surface). Whirlpool spinner
+ 'Searching…' placeholder while in flight. Each search result is
stamped with its source folder so clicks open the right email instead
of whatever shares its UID in INBOX. Search no longer re-applies the
same text pill locally (which only checks subject/from/snippet, never
body) so body-only matches don't get dropped after IMAP returns them.
Initial inbox load bumped 100→500.
- Email favorites: 'Favorite (pin to top)' / 'Unfavorite' in both the
card menu and the open-reader more menu, backed by a new
/api/email/flag/{uid}?on=true|false endpoint. Flagged emails always
bubble to the top of the grid regardless of active sort.
- AI reply in doc editor: never overwrites existing draft text or the
quoted history. AI suggestion is prepended; AI-generated 'On …
wrote:' re-quotes are stripped so the original quote isn't visually
edited.
- Cookbook serve: pre-launch GPU driver / has_gpu / install / version-
floor checks (vllm minimax_m2 needs 0.10.0+, deepseek_r1 needs 0.7.0
etc.) before the launch chain starts. Detect 'another model already
running on this host' and offer Stop & launch (with graceful then
force tmux kill helpers, port release wait). Per-vendor deep-link
buttons (vLLM recipe / SGLang cookbook) with hardware hash. Backend
picker is now a custom dropdown with accent-coloured logos for vLLM,
SGLang, llama.cpp, Ollama, Diffusers; same glyphs added next to
package names in Dependencies. Runtime-readiness note moved inside
the panel (green when ready, red when missing) with an × dismiss.
Esc collapses the expanded card; expanded card scrolls when it
overflows; Trust Remote / Auto Tool / Reasoning Parser / Enforce
Eager / Prefix Caching / Expert Parallel / Speculative / MoE Env on
one row (Reasoning Parser auto-detected per model family).
Dtype→Row 1, GPUs→Row 2 (rightmost). Removed redundant GPU 'auto'
input — command builders read from the GPU button strip. Default
cookbook open is Download tab.
- Cookbook hwfit: 'Model (latest)' / 'Model (oldest)' header sorts by
release_date; release dates can be backfilled with the new
scripts/backfill_model_release_dates.py and recipe metadata pulled
with scripts/import_from_vllm_recipes.py against the upstream
vllm-project/recipes catalog (vllm_recipe + min_vllm_version stamped
on entries).
- Calendar: Quick add hint cycles a random Odysseus-themed example per
open (wooden horse Friday, crew muster 10am daily, council on
Ithaca, …). Typing a time like '11pm' in the event title updates
the hero clock live.
- Doc editor: email-mode Reply button (sparkle icon, accent) opens the
same Fast/Full + context popover the email reader uses; Ctrl+Alt+M
toggles markdown preview.
- Memories panel: custom sort picker with per-option icons, default
'Latest', visible Enabled/Disabled toggle text matching the section
description style.
2026-06-15 20:47:51 +09:00
# NOTE: previously dropped the doc when doc.session_id
# != current chat session — but that broke the common
# case of "open an email draft from one chat, ask a
# different chat to write into it". The frontend only
# sends active_doc_id for docs currently visible in
# the UI, and we already owner-checked above, so trust
# the explicit signal. We just log the mismatch and
# re-bind the doc to the current session so future
# turns find it via the session-fallback path too.
if doc_session and doc_session != session :
logger . info (
" [doc-inject] cross-session active_doc_id %s (was session %s , now %s ) — accepting and rebinding " ,
active_doc_id , doc_session , session ,
)
try :
active_doc . session_id = session
_doc_db . commit ( )
except Exception as _e :
_doc_db . rollback ( )
logger . warning ( f " [doc-inject] session rebind failed: { _e } " )
2026-06-04 17:27:46 +05:00
logger . info ( f " [doc-inject] found by ID: title= { active_doc . title !r} , lang= { active_doc . language !r} , is_active= { active_doc . is_active } , content_len= { len ( active_doc . current_content or ' ' ) } " )
2026-05-31 23:58:26 +09:00
else :
logger . warning ( f " [doc-inject] NOT FOUND by ID { active_doc_id } " )
2026-06-29 03:10:42 +00:00
if not active_doc :
_email_doc_q = _doc_db . query ( DBDocument ) . filter (
DBDocument . session_id == session ,
DBDocument . is_active == True ,
DBDocument . language == " email " ,
)
active_doc = _owner_session_filter ( _email_doc_q , ctx . user ) . order_by ( DBDocument . updated_at . desc ( ) ) . first ( )
if active_doc :
logger . info ( f " [doc-inject] found email draft by session fallback: title= { active_doc . title !r} " )
2026-05-31 23:58:26 +09:00
if not active_doc :
2026-06-02 06:29:27 -05:00
_session_doc_q = _doc_db . query ( DBDocument ) . filter (
2026-05-31 23:58:26 +09:00
DBDocument . session_id == session ,
DBDocument . is_active == True
2026-06-02 06:29:27 -05:00
)
active_doc = _owner_session_filter ( _session_doc_q , ctx . user ) . order_by ( DBDocument . updated_at . desc ( ) ) . first ( )
2026-05-31 23:58:26 +09:00
if active_doc :
logger . info ( f " [doc-inject] found by session fallback: title= { active_doc . title !r} " )
# Last resort: the document the agent itself just created/edited
# (tracked in-memory by the tool layer). This rescues docs that
# got orphaned from their session (session_id NULL) — otherwise
# neither lookup above can associate them with this conversation,
# so the agent never sees what it just wrote. Guarded so we never
# leak a doc that belongs to a DIFFERENT session.
if not active_doc :
try :
2026-06-10 09:41:52 +01:00
from src . agent_tools . document_tools import get_active_document
2026-05-31 23:58:26 +09:00
_mem_id = get_active_document ( )
if _mem_id :
2026-06-02 06:29:27 -05:00
_mem_q = _doc_db . query ( DBDocument ) . filter ( DBDocument . id == _mem_id )
cand = _owner_session_filter ( _mem_q , ctx . user ) . first ( )
2026-05-31 23:58:26 +09:00
if cand and ( not cand . session_id or cand . session_id == session ) :
active_doc = cand
logger . info ( f " [doc-inject] found by in-memory active id: title= { active_doc . title !r} (session_id= { cand . session_id !r} ) " )
except Exception as _e :
logger . debug ( f " [doc-inject] in-memory fallback failed: { _e } " )
if not active_doc :
logger . info ( f " [doc-inject] no active doc for session { session } " )
if active_doc :
_doc_db . expunge ( active_doc )
except Exception as e :
logger . warning ( f " Failed to query active document: { e } " )
finally :
_doc_db . close ( )
# Build disabled-tools set from frontend toggles + user privileges
disabled_tools = set ( )
2026-06-12 01:14:41 +07:00
# Only disable bash/web_search when the caller *explicitly* set them
# to a falsy value. When unset (None), defer to per-user privilege
# checks below — this lets admins with can_use_bash=True use bash
# by default without having to send allow_bash in every request.
if allow_bash is not None and str ( allow_bash ) . lower ( ) != " true " :
2026-05-31 23:58:26 +09:00
disabled_tools . add ( " bash " )
2026-06-15 02:02:10 -04:00
if (
allow_web_search is not None
and str ( allow_web_search ) . lower ( ) != " true "
) :
2026-05-31 23:58:26 +09:00
disabled_tools . add ( " web_search " )
2026-06-01 14:57:28 +07:00
disabled_tools . add ( " web_fetch " )
2026-07-07 00:50:07 +00:00
if _explicit_web_intent :
# A direct lookup/search request should not drift into personal
# tools or shell fallbacks. We still keep web_search/web_fetch
# available even when the frontend toggle is stale/falsy because
# the user's words are the stronger signal.
disabled_tools . update ( {
" bash " , " python " ,
" search_chats " , " manage_skills " , " manage_memory " ,
" read_file " , " write_file " , " edit_file " ,
" create_document " , " edit_document " , " update_document " ,
" send_email " , " reply_to_email " ,
" manage_notes " , " manage_calendar " , " manage_tasks " ,
" api_call " , " builtin_browser " ,
} )
disabled_tools . discard ( " web_search " )
disabled_tools . discard ( " web_fetch " )
elif _search_enabled :
disabled_tools . discard ( " web_search " )
disabled_tools . discard ( " web_fetch " )
2026-05-31 23:58:26 +09:00
# Nobody/incognito mode: deny tools that would expose the user's
# persistent memory, past chats, or other identity-linked data.
if incognito :
disabled_tools . update ( {
" manage_memory " , # persistent memory store
" search_chats " , # past chat history
" manage_skills " , # skill presets tied to user
} )
2026-06-27 13:05:44 +00:00
# Active email reader open → strip the tools that let the agent drift
# away from the visible email or skip review. The only allowed compose
# path is ui_control open_email_reply, which opens the same draft editor
# as the Reply button with the generated body pre-filled. This prevents
# the model from falling back to direct SMTP when it botches a draft
# call, and prevents fake email-shaped documents.
Open email context for agent, email search across All Mail, cookbook serve polish
- Agent: pass the open email reader (uid/folder/account/from/subject/body
preview) on every chat submit so 'reply to this' / 'write email saying
hi' route to ui_control open_email_reply with the right UID instead of
inventing a new .md draft. Code-level enforcement (chat_routes strips
create_document + send_email when active_email is set); cross-session
active_doc_id is now trusted instead of being silently dropped.
set_active_email/clear_active_email tool-layer helpers in
tool_implementations.
- ui_control open_email_reply: optional body argument so the agent can
open-and-write in one call; envelope now forwards uid/folder/account/
body/panel through tool_output. Tool description sharpened and the
parser rejects empty bodies on reply/reply-all (forces the agent to
write rather than open an empty draft).
- Email library: search now runs against [Gmail]/All Mail when the
current folder is INBOX (archived emails surface). Whirlpool spinner
+ 'Searching…' placeholder while in flight. Each search result is
stamped with its source folder so clicks open the right email instead
of whatever shares its UID in INBOX. Search no longer re-applies the
same text pill locally (which only checks subject/from/snippet, never
body) so body-only matches don't get dropped after IMAP returns them.
Initial inbox load bumped 100→500.
- Email favorites: 'Favorite (pin to top)' / 'Unfavorite' in both the
card menu and the open-reader more menu, backed by a new
/api/email/flag/{uid}?on=true|false endpoint. Flagged emails always
bubble to the top of the grid regardless of active sort.
- AI reply in doc editor: never overwrites existing draft text or the
quoted history. AI suggestion is prepended; AI-generated 'On …
wrote:' re-quotes are stripped so the original quote isn't visually
edited.
- Cookbook serve: pre-launch GPU driver / has_gpu / install / version-
floor checks (vllm minimax_m2 needs 0.10.0+, deepseek_r1 needs 0.7.0
etc.) before the launch chain starts. Detect 'another model already
running on this host' and offer Stop & launch (with graceful then
force tmux kill helpers, port release wait). Per-vendor deep-link
buttons (vLLM recipe / SGLang cookbook) with hardware hash. Backend
picker is now a custom dropdown with accent-coloured logos for vLLM,
SGLang, llama.cpp, Ollama, Diffusers; same glyphs added next to
package names in Dependencies. Runtime-readiness note moved inside
the panel (green when ready, red when missing) with an × dismiss.
Esc collapses the expanded card; expanded card scrolls when it
overflows; Trust Remote / Auto Tool / Reasoning Parser / Enforce
Eager / Prefix Caching / Expert Parallel / Speculative / MoE Env on
one row (Reasoning Parser auto-detected per model family).
Dtype→Row 1, GPUs→Row 2 (rightmost). Removed redundant GPU 'auto'
input — command builders read from the GPU button strip. Default
cookbook open is Download tab.
- Cookbook hwfit: 'Model (latest)' / 'Model (oldest)' header sorts by
release_date; release dates can be backfilled with the new
scripts/backfill_model_release_dates.py and recipe metadata pulled
with scripts/import_from_vllm_recipes.py against the upstream
vllm-project/recipes catalog (vllm_recipe + min_vllm_version stamped
on entries).
- Calendar: Quick add hint cycles a random Odysseus-themed example per
open (wooden horse Friday, crew muster 10am daily, council on
Ithaca, …). Typing a time like '11pm' in the event title updates
the hero clock live.
- Doc editor: email-mode Reply button (sparkle icon, accent) opens the
same Fast/Full + context popover the email reader uses; Ctrl+Alt+M
toggles markdown preview.
- Memories panel: custom sort picker with per-option icons, default
'Latest', visible Enabled/Disabled toggle text matching the section
description style.
2026-06-15 20:47:51 +09:00
if active_email_ctx and active_email_ctx . get ( " uid " ) :
disabled_tools . update ( {
" create_document " ,
" send_email " ,
2026-06-27 13:05:44 +00:00
" reply_to_email " ,
Open email context for agent, email search across All Mail, cookbook serve polish
- Agent: pass the open email reader (uid/folder/account/from/subject/body
preview) on every chat submit so 'reply to this' / 'write email saying
hi' route to ui_control open_email_reply with the right UID instead of
inventing a new .md draft. Code-level enforcement (chat_routes strips
create_document + send_email when active_email is set); cross-session
active_doc_id is now trusted instead of being silently dropped.
set_active_email/clear_active_email tool-layer helpers in
tool_implementations.
- ui_control open_email_reply: optional body argument so the agent can
open-and-write in one call; envelope now forwards uid/folder/account/
body/panel through tool_output. Tool description sharpened and the
parser rejects empty bodies on reply/reply-all (forces the agent to
write rather than open an empty draft).
- Email library: search now runs against [Gmail]/All Mail when the
current folder is INBOX (archived emails surface). Whirlpool spinner
+ 'Searching…' placeholder while in flight. Each search result is
stamped with its source folder so clicks open the right email instead
of whatever shares its UID in INBOX. Search no longer re-applies the
same text pill locally (which only checks subject/from/snippet, never
body) so body-only matches don't get dropped after IMAP returns them.
Initial inbox load bumped 100→500.
- Email favorites: 'Favorite (pin to top)' / 'Unfavorite' in both the
card menu and the open-reader more menu, backed by a new
/api/email/flag/{uid}?on=true|false endpoint. Flagged emails always
bubble to the top of the grid regardless of active sort.
- AI reply in doc editor: never overwrites existing draft text or the
quoted history. AI suggestion is prepended; AI-generated 'On …
wrote:' re-quotes are stripped so the original quote isn't visually
edited.
- Cookbook serve: pre-launch GPU driver / has_gpu / install / version-
floor checks (vllm minimax_m2 needs 0.10.0+, deepseek_r1 needs 0.7.0
etc.) before the launch chain starts. Detect 'another model already
running on this host' and offer Stop & launch (with graceful then
force tmux kill helpers, port release wait). Per-vendor deep-link
buttons (vLLM recipe / SGLang cookbook) with hardware hash. Backend
picker is now a custom dropdown with accent-coloured logos for vLLM,
SGLang, llama.cpp, Ollama, Diffusers; same glyphs added next to
package names in Dependencies. Runtime-readiness note moved inside
the panel (green when ready, red when missing) with an × dismiss.
Esc collapses the expanded card; expanded card scrolls when it
overflows; Trust Remote / Auto Tool / Reasoning Parser / Enforce
Eager / Prefix Caching / Expert Parallel / Speculative / MoE Env on
one row (Reasoning Parser auto-detected per model family).
Dtype→Row 1, GPUs→Row 2 (rightmost). Removed redundant GPU 'auto'
input — command builders read from the GPU button strip. Default
cookbook open is Download tab.
- Cookbook hwfit: 'Model (latest)' / 'Model (oldest)' header sorts by
release_date; release dates can be backfilled with the new
scripts/backfill_model_release_dates.py and recipe metadata pulled
with scripts/import_from_vllm_recipes.py against the upstream
vllm-project/recipes catalog (vllm_recipe + min_vllm_version stamped
on entries).
- Calendar: Quick add hint cycles a random Odysseus-themed example per
open (wooden horse Friday, crew muster 10am daily, council on
Ithaca, …). Typing a time like '11pm' in the event title updates
the hero clock live.
- Doc editor: email-mode Reply button (sparkle icon, accent) opens the
same Fast/Full + context popover the email reader uses; Ctrl+Alt+M
toggles markdown preview.
- Memories panel: custom sort picker with per-option icons, default
'Latest', visible Enabled/Disabled toggle text matching the section
description style.
2026-06-15 20:47:51 +09:00
" mcp__email__send_email " ,
2026-06-27 13:05:44 +00:00
" mcp__email__reply_to_email " ,
Open email context for agent, email search across All Mail, cookbook serve polish
- Agent: pass the open email reader (uid/folder/account/from/subject/body
preview) on every chat submit so 'reply to this' / 'write email saying
hi' route to ui_control open_email_reply with the right UID instead of
inventing a new .md draft. Code-level enforcement (chat_routes strips
create_document + send_email when active_email is set); cross-session
active_doc_id is now trusted instead of being silently dropped.
set_active_email/clear_active_email tool-layer helpers in
tool_implementations.
- ui_control open_email_reply: optional body argument so the agent can
open-and-write in one call; envelope now forwards uid/folder/account/
body/panel through tool_output. Tool description sharpened and the
parser rejects empty bodies on reply/reply-all (forces the agent to
write rather than open an empty draft).
- Email library: search now runs against [Gmail]/All Mail when the
current folder is INBOX (archived emails surface). Whirlpool spinner
+ 'Searching…' placeholder while in flight. Each search result is
stamped with its source folder so clicks open the right email instead
of whatever shares its UID in INBOX. Search no longer re-applies the
same text pill locally (which only checks subject/from/snippet, never
body) so body-only matches don't get dropped after IMAP returns them.
Initial inbox load bumped 100→500.
- Email favorites: 'Favorite (pin to top)' / 'Unfavorite' in both the
card menu and the open-reader more menu, backed by a new
/api/email/flag/{uid}?on=true|false endpoint. Flagged emails always
bubble to the top of the grid regardless of active sort.
- AI reply in doc editor: never overwrites existing draft text or the
quoted history. AI suggestion is prepended; AI-generated 'On …
wrote:' re-quotes are stripped so the original quote isn't visually
edited.
- Cookbook serve: pre-launch GPU driver / has_gpu / install / version-
floor checks (vllm minimax_m2 needs 0.10.0+, deepseek_r1 needs 0.7.0
etc.) before the launch chain starts. Detect 'another model already
running on this host' and offer Stop & launch (with graceful then
force tmux kill helpers, port release wait). Per-vendor deep-link
buttons (vLLM recipe / SGLang cookbook) with hardware hash. Backend
picker is now a custom dropdown with accent-coloured logos for vLLM,
SGLang, llama.cpp, Ollama, Diffusers; same glyphs added next to
package names in Dependencies. Runtime-readiness note moved inside
the panel (green when ready, red when missing) with an × dismiss.
Esc collapses the expanded card; expanded card scrolls when it
overflows; Trust Remote / Auto Tool / Reasoning Parser / Enforce
Eager / Prefix Caching / Expert Parallel / Speculative / MoE Env on
one row (Reasoning Parser auto-detected per model family).
Dtype→Row 1, GPUs→Row 2 (rightmost). Removed redundant GPU 'auto'
input — command builders read from the GPU button strip. Default
cookbook open is Download tab.
- Cookbook hwfit: 'Model (latest)' / 'Model (oldest)' header sorts by
release_date; release dates can be backfilled with the new
scripts/backfill_model_release_dates.py and recipe metadata pulled
with scripts/import_from_vllm_recipes.py against the upstream
vllm-project/recipes catalog (vllm_recipe + min_vllm_version stamped
on entries).
- Calendar: Quick add hint cycles a random Odysseus-themed example per
open (wooden horse Friday, crew muster 10am daily, council on
Ithaca, …). Typing a time like '11pm' in the event title updates
the hero clock live.
- Doc editor: email-mode Reply button (sparkle icon, accent) opens the
same Fast/Full + context popover the email reader uses; Ctrl+Alt+M
toggles markdown preview.
- Memories panel: custom sort picker with per-option icons, default
'Latest', visible Enabled/Disabled toggle text matching the section
description style.
2026-06-15 20:47:51 +09:00
} )
2026-05-31 23:58:26 +09:00
# Enforce per-user privileges
_privs = { }
_user = ctx . user
if _user and hasattr ( request . app . state , ' auth_manager ' ) and request . app . state . auth_manager :
_privs = request . app . state . auth_manager . get_privileges ( _user )
if _privs :
if not _privs . get ( " can_use_bash " , True ) :
disabled_tools . update ( { " bash " , " python " , " read_file " , " write_file " } )
if not _privs . get ( " can_use_browser " , True ) :
disabled_tools . add ( " builtin_browser " )
if not _privs . get ( " can_use_documents " , True ) :
disabled_tools . update ( { " create_document " , " edit_document " , " update_document " , " suggest_document " } )
if not _privs . get ( " can_generate_images " , True ) :
disabled_tools . add ( " generate_image " )
if not _privs . get ( " can_manage_memory " , True ) :
disabled_tools . update ( { " manage_memory " , " manage_skills " } )
if not _privs . get ( " can_use_research " , True ) :
_research_flags [ " do " ] = False
if not _privs . get ( " can_use_agent " , True ) :
_effective_mode = ' chat '
chat_mode = ' chat '
# Global admin disabled tools
from src . settings import get_setting
_global_disabled = get_setting ( " disabled_tools " , [ ] )
if _global_disabled and isinstance ( _global_disabled , list ) :
2026-07-07 00:50:07 +00:00
explicit_web_allowed = (
_explicit_web_intent
or ( allow_web_search is not None and str ( allow_web_search ) . lower ( ) == " true " )
)
2026-06-21 11:02:35 +00:00
if explicit_web_allowed :
disabled_tools . update ( t for t in _global_disabled if t not in { " web_search " , " web_fetch " } )
else :
disabled_tools . update ( _global_disabled )
2026-05-31 23:58:26 +09:00
# Light auto-escalation: the user is in chat mode and just expressed a
# notes/calendar/email intent. Grant the relevant managers but withhold
# the heavy "do things on the computer" tools — otherwise the model
# tries to shell out for a request that never needed it, then fails
# (and looks broken when the shell is disabled).
if auto_escalated :
disabled_tools . update ( {
" bash " , " python " , " read_file " , " write_file " , " builtin_browser " ,
} )
# Disable document tools in compare sessions — they break the pane UI
if sess . name and sess . name . startswith ( " [CMP] " ) :
disabled_tools . update ( { " create_document " , " edit_document " , " update_document " } )
# Compare mode: disable tools based on compare type
if compare_mode :
_compare_strip = {
" create_document " , " edit_document " , " update_document " ,
" chat_with_model " , " create_session " , " list_sessions " ,
" send_to_session " ,
" pipeline " , " manage_session " , " manage_memory " , " list_models " ,
" generate_image " , " ui_control " ,
}
disabled_tools . update ( _compare_strip )
# In chat mode compare, disable ALL agent tools (no bash, python, file ops)
if chat_mode == ' chat ' :
2026-06-01 14:57:28 +07:00
disabled_tools . update ( { " bash " , " python " , " read_file " , " write_file " , " web_search " , " web_fetch " , " search_chats " , " manage_tasks " } )
2026-05-31 23:58:26 +09:00
feat: Add plan mode to the chat agent (#638)
* feat: Add plan mode to the chat agent
Adds a plan mode: the agent investigates read-only, proposes a checklist, and
waits for approval before changing anything. On approval it runs with full
tools and checks items off as it goes. Enforcement reuses the existing
disabled_tools gate.
Includes a slash command: `/plan [on|off]` (and `/toggle plan`) to flip the
plan toggle from the chat input.
- src/tool_security.py, src/mcp_manager.py: read-only allowlist (tools + MCP).
- src/agent_loop.py, routes/chat_routes.py: union the disabled set, prepend the
plan directive, force agent mode.
- static/: plan toggle pill, Approve & Run, dockable plan window, task-list
checkboxes, and the /plan slash command.
- tests/test_plan_mode.py.
* Plan mode: persistent re-referenceable plan + agent write-back
Three improvements so a long plan survives a weak model and stays in reach:
1. Re-reference the plan (out-of-context fix). On the execution turn the frontend
sends the approved checklist back (`approved_plan`); the backend pins it as a
top-of-context `## ACTIVE PLAN` system note (kept by the context trimmer), so
the agent can always re-read the plan instead of losing the thread on a long
run. New `build_active_plan_note()` (unit-tested).
2. Re-open / dock the plan anytime. The plan checklist is stored per-session
(localStorage). When a plan exists, the plan-mode button opens a small menu
("Show plan" / "Plan mode: On/Off") that re-opens the side-dockable plan
window — so it can stay docked while the agent works. The window live-refreshes
as the plan changes.
3. Agent write-back: new `update_plan` tool. The agent calls it to tick steps
`- [x]` after finishing them, or to revise steps when the user asks. Marker
tool (no I/O) → `plan_update` SSE event → the stored plan + docked window
update live. The ACTIVE PLAN note instructs the agent to use it.
Backend: src/agent_loop.py (param + pin + note builder + emit + prompt blurb),
src/tool_execution.py (update_plan handler), routes/chat_routes.py (parse
`approved_plan`, relay `plan_update`), registration in tool_schemas / agent_tools
/ tool_index (always-available, not admin-gated).
Frontend: static/js/chat.js (plan store, send `approved_plan`, handle
`plan_update`, capture restated checklists), static/app.js (plan-button menu),
static/js/planWindow.js (`isPlanWindowOpen`), static/js/storage.js (PLAN key).
Tests: tests/test_plan_mode.py (plan-note), tests/test_update_plan_tool.py.
* Plan mode: drop bash/python, rely on read-only discovery tools
Shell can mutate (write files, hit the network) and can't be constrained to
read-only at the tool layer, so plan mode no longer relies on a prompt to keep
it well-behaved — bash/python are removed from the read-only allowlist and added
to the fail-closed block set. Discovery is covered by the dedicated read-only
tools (read_file, grep, glob, ls) instead.
Rewrites the plan-mode directive to state shell is disabled and lists the
available read-only tools positively. Addresses review feedback on #638.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Comment: note _MCP_READONLY_VERBS are prefixes not whole words
Clarifies that entries like "summar" are intentional stems matched via
startswith (covers summarise/summarize/summary), not typos. Addresses review
feedback on #638.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Plan mode: clarify why gating inverts the allowlist into a denylist
Rename _PLAN_MODE_FALLBACK_BLOCK -> _PLAN_MODE_KNOWN_MUTATORS and rewrite the
comments. The tool gate is a denylist (disabled_tools); plan mode's policy is an
allowlist, so it returns the inverse (all known tool names minus the allowlist).
The static mutator set is a backstop for the schema-derived name list, which
misses XML-only tools and can fail to import. Addresses review feedback on #638.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Plan mode: stop hardcoding the read-only tool list in the directive
The model is already shown its available (read-only) tools by _assemble_prompt,
which removes every disabled tool. Enumerating them again in the directive only
duplicated that list and would drift as tools change. Point at the tools listed
below instead. Addresses review feedback on #638.
2026-06-05 16:32:25 +02:00
# Plan mode: investigate read-only, propose a plan, don't mutate. Block
# every tool not on the read-only allowlist. (stream_agent_loop enforces
# this again + drops MCP, so this is belt-and-suspenders.)
if plan_mode :
from src . tool_security import plan_mode_disabled_tools
disabled_tools . update ( plan_mode_disabled_tools ( ) )
2026-06-06 18:48:24 -06:00
tool_policy = build_effective_tool_policy (
disabled_tools = disabled_tools ,
last_user_message = message ,
)
disabled_tools = tool_policy . all_disabled_names ( )
research_blocked_by_policy = bool (
tool_policy . blocks ( " trigger_research " )
or tool_policy . blocks ( " manage_research " )
)
effective_do_research = bool (
do_research and _research_flags [ " do " ] and not research_blocked_by_policy
)
# Persist session mode after policy/privilege gates so blocked research
# turns remain ordinary chat/agent streams and saved messages.
_effective_mode = ' research ' if effective_do_research else ( chat_mode or ' chat ' )
if _effective_mode in ( ' agent ' , ' research ' , ' chat ' ) :
set_session_mode ( session , _effective_mode )
2026-05-31 23:58:26 +09:00
async def stream_with_save ( ) - > AsyncGenerator [ str , None ] :
# _effective_mode is read-only here; closure captures it from
# the outer scope. (Was `nonlocal` but never reassigned.)
research_sources = None
web_sources = ctx . web_sources
# Register active stream for partial-save safety net
2026-06-06 18:48:24 -06:00
_active_streams [ session ] = { " status " : " streaming " , " partial " : " " , " query " : message , " is_research " : effective_do_research , " mode " : _effective_mode }
2026-05-31 23:58:26 +09:00
feat(agent): confine agent file/shell tools to a selectable workspace (#3665)
* feat(agent): workspace confinement via context-local binding + get_workspace tool
Bind the per-turn workspace once in execute_tool_block; the shared path
resolvers (_resolve_tool_path / _resolve_search_root) and the subprocess cwd
helper (agent_cwd) read it, so file tools + bash/python are confined centrally
and a new tool that uses the shared helpers cannot accidentally bypass it.
Adds the admin-gated /api/workspace/browse picker, a workspace pill + directory
modal (reusing existing modal/button CSS), the /workspace slash command, and a
get_workspace tool (replaces a system-prompt block). Confinement is OS-agnostic
(realpath/normcase/commonpath) and docker-safe (container paths, no host
assumptions). Reopens #2023.
* ux(workspace): clarify workspace is not a sandbox
Picker modal note + pill tooltip + get_workspace tool/output wording now state
plainly: read_file/write_file/edit_file/grep/glob/ls are confined to the folder,
but bash/python only start there (cwd) and are not sandboxed. Modal note reuses
the existing .muted class.
* fix(agent): treat an active workspace as file-work intent
A vague low-signal message (e.g. "look at the local project") matches no
domain keywords, so tool retrieval is skipped and only always-available tools
are offered — leaving the agent with no file access even though a workspace is
set. When a workspace is active, include the file/code tools (incl.
get_workspace) on low-signal turns so the agent can act on the folder.
Also requires the tool index (ChromaDB) to be reachable for normal retrieval;
that is an environment dependency, not part of this change.
* ux(workspace): hide pill + overflow entry in chat mode
Workspace only scopes the agent's file/shell tools, so the pill and the
overflow 'Workspace' entry are agent-only now — hidden in chat mode like the
bash toggle. Mode read from the DOM in syncWorkspaceIndicator; applyMode() is
called from the agent/chat setMode handler.
* prompt(tools): steer bash/python to defer to the dedicated file tools
bash/python schema descriptions (what native-tool-calling models read) were
bare and gave no steer, so models would do file ops via the shell (e.g. writing
SVG/HTML, which then dumps raw markup into the tool preview). Tell bash/python
in the schema + tool-index + prompt section to prefer read_file/write_file/
edit_file/grep/glob/ls and only be used for what those do not cover.
* prompt(tools): keep bash/python deferral generic (no hardcoded tool names)
Reference 'a dedicated tool' rather than listing read_file/write_file/grep/etc.
by name, so the guidance does not go stale if those tools are renamed.
* style(workspace): drop em-dashes from added code comments/strings
* ux(workspace): terser non-sandbox note in picker (no tool-name list)
* ux(workspace): mirror terse non-sandbox wording in pill tooltip
* chore: untrack local venv symlink (run-only, not part of the feature)
* prompt(workspace): keep get_workspace text generic (no hardcoded tool names)
* fix(agent): low-signal + workspace surfaces only read-only file tools
Intersect the files tool group with PLAN_MODE_READONLY_TOOLS so a vague message
in a workspace exposes read_file/grep/glob/ls/get_workspace for exploration, but
not write_file/edit_file/bash/python -- those wait for a request that actually
calls for them (RAG retrieval still adds them on a real ask).
* feat(workspace): cap browse listing at 500 dirs with a truncated hint
Mirror the filesystem_tools._CODENAV_MAX_HITS pattern with a module-local
_MAX_BROWSE_DIRS so a directory with thousands of children does not dump every
row into the picker; the response carries a truncated flag and the modal tells
the user to type a path to jump in.
* chore: untrack local venv symlink (run-only artifact)
* fix(workspace): vet the workspace root against the sensitive-path deny list at bind time
The in-workspace resolver deny-lists sensitive paths inside the workspace,
but the empty-path search root is the workspace itself, so a workspace of
~/.ssh could be listed via ls with no path. vet_workspace() (public, in
tool_execution next to the resolvers) rejects non-directories and sensitive
roots before the path is ever bound; chat_routes uses it instead of its
inline isdir check.
* fix(workspace): reject filesystem roots and stop showing rejected workspaces as active
Review findings from #3665:
P2: vet_workspace accepted / (and would accept drive/UNC roots), which makes
every absolute path 'inside' the workspace and collapses confinement into
host-wide file access. A root is its own dirname, so reject when
dirname(resolved) == resolved; the browse response now carries a selectable
flag and the picker disables 'Use this folder' on unselectable dirs.
P3: /workspace set stored any string client-side and the chat route silently
dropped rejected values, so the pill could claim a confinement that was not
in effect. New admin-gated /api/workspace/vet validates manual paths before
they persist (canonical path returned), and when a posted workspace is
rejected at send time the stream emits workspace_rejected so the client
clears the stored value and toasts instead of continuing silently.
* fix(workspace): check caller privilege before vetting the posted workspace
Review finding: /api/chat_stream called vet_workspace() on the posted value
for every caller and emitted workspace_rejected on failure, so a non-admin
who can chat but cannot use file/shell tools could distinguish existing
directories from missing/file/sensitive/root paths by whether the event
appeared. The resolution now lives in _resolve_request_workspace, which
drops the submitted value uniformly for non-admin callers, with no vetting
and no event, before the path ever touches the filesystem. Admin and
single-user behavior is unchanged. Test pins that valid and invalid paths
are indistinguishable for a non-admin and that vet_workspace is never
invoked for them.
2026-06-11 18:17:54 +02:00
# The client sent a workspace the server refused to bind (deleted
# folder, file path, sensitive dir, filesystem root). Tell it up
# front so the UI can clear the pill instead of displaying a
# confinement that is not actually in effect.
if workspace_rejected :
yield f " data: { json . dumps ( { ' type ' : ' workspace_rejected ' , ' data ' : { ' path ' : workspace_rejected } } ) } \n \n "
2026-05-31 23:58:26 +09:00
if ctx . preprocessed . attachment_meta :
yield f " data: { json . dumps ( { ' type ' : ' attachments ' , ' data ' : ctx . preprocessed . attachment_meta } ) } \n \n "
# Announce any docs auto-created during preprocess (e.g. fillable
# PDF → editable markdown) so the editor pane switches to them
# before the model starts streaming.
for _opened in ctx . auto_opened_docs :
yield (
f ' data: { json . dumps ( { " type " : " doc_update " , * * _opened } ) } \n \n '
)
if ctx . rag_sources :
yield f " data: { json . dumps ( { ' type ' : ' rag_sources ' , ' data ' : ctx . rag_sources } ) } \n \n "
if web_sources :
yield f " data: { json . dumps ( { ' type ' : ' web_sources ' , ' data ' : web_sources } ) } \n \n "
# Emit which memories were injected into context (captured before stream)
if ctx . used_memories :
yield f " data: { json . dumps ( { ' type ' : ' memories_used ' , ' data ' : ctx . used_memories } ) } \n \n "
# Run research as a background task (survives page refresh)
2026-06-06 18:48:24 -06:00
if effective_do_research :
2026-05-31 23:58:26 +09:00
_r_ep , _r_model , _r_headers = _resolve_research_endpoint ( sess )
_auth_keys = list ( _r_headers . keys ( ) ) if _r_headers else [ ]
fix(security): redact credential-bearing URLs and PII from logs (#4750)
* fix(security): redact credential-bearing URLs and PII from logs
Several log statements emitted sensitive data in clear text:
- model_routes / chat_routes / contacts_routes logged endpoint URLs raw.
Admin-configured URLs can embed credentials in userinfo or query
(e.g. https://user:pass@host, ?api_key=...). Route them through a
shared core.log_safety.redact_url() that drops userinfo/query/fragment.
- note_routes / task_scheduler logged operator email addresses (smtp_user,
recipient). Replaced with presence booleans, which keeps the diagnostic
("why didn't this send") without writing PII to logs.
model_routes already had a local redactor on its HTTPStatusError branch;
the generic except branch was missed, so reuse the existing helper there.
Clears CodeQL py/clear-text-logging-sensitive-data alerts 264, 317, 324,
325, 343, 344, 528.
* fix(security): re-bracket IPv6 hosts and single-source the URL redactor
Address review on #4750:
- redact_url now re-brackets IPv6 literals so host:port stays
unambiguous (https://[2001:db8::1]:8443/v1, not the bracket-less
ambiguous form).
- point model_routes._redact_url_for_log at the shared helper so the
two redactors are single-sourced (also picks up the IPv6 fix).
2026-06-22 14:12:39 -07:00
logger . info ( f " Research endpoint resolved: model= { _r_model } , endpoint= { redact_url ( _r_ep ) } , auth_keys= { _auth_keys } , sess_headers_keys= { list ( sess . headers . keys ( ) ) if isinstance ( sess . headers , dict ) else type ( sess . headers ) } " )
2026-05-31 23:58:26 +09:00
# Clarification round: only for very short/vague queries on first research message.
# Skip in compare mode — each pane is a fresh session, so every one would
# ask clarifying questions and the user would have to answer each pane
# separately, breaking the parallel comparison.
_prior_json = research_handler . _get_session_json ( session )
_history_len = len ( sess . history ) if hasattr ( sess , ' history ' ) else 0
_is_first_research = not _prior_json and _history_len < = 2 and not compare_mode
if _is_first_research :
logger . info ( f " First research message — asking clarifying questions for: { message [ : 60 ] } " )
yield f ' data: { json . dumps ( { " type " : " model_info " , " model " : sess . model , " suffix " : " Research " } ) } \n \n '
# Set DB mode to research_pending so the NEXT message auto-triggers research
fix: stop leaking DB connections when persisting session mode (#64)
chat_routes.py persisted a session's "mode" in three best-effort spots —
reading the current mode, writing the effective mode, and setting
research_pending on the stream path. Each opened a session with SessionLocal()
and called .close() as the LAST statement inside a try/except, so if anything
before close() raised (e.g. a SQLite "database is locked" under concurrent chat
streams) the except only logged and the connection was never returned to the
pool.
DATABASE_URL defaults to file-backed SQLite, whose engine uses SQLAlchemy's
default QueuePool (5 connections + 10 overflow). Repeated leaks on these hot
paths exhaust the pool; later requests then block for pool_timeout and fail
with "QueuePool limit ... reached", taking the app down until restart.
Move the logic into two best-effort helpers in core.database, next to the
existing session helpers (update_session_last_accessed, get_session_by_id):
- get_session_mode(session_id) -> Optional[str]
- set_session_mode(session_id, mode) -> bool
Both route through the existing get_db_session() context manager, which commits
on success, rolls back on error, and always closes in a finally, so the
connection is returned to the pool on every path. chat_routes.py now calls
these instead of hand-rolling sessions, also removing three copies of the same
try/except.
Add tests/test_session_mode_helpers.py: the helpers commit+close on success
and, on a mid-operation DB error, swallow + roll back + close (no leak). The
error-path tests fail against the old close()-inside-try pattern.
2026-06-01 00:57:48 -04:00
set_session_mode ( session , " research_pending " )
2026-05-31 23:58:26 +09:00
ctx . messages . insert ( 0 , { " role " : " system " , " content " :
" The user wants to start deep web research. Before searching, ask 2-3 brief "
" clarifying questions to understand exactly what they want to know. For example: "
" what aspects matter most, are they comparing to something, what ' s their context "
" (moving, traveling, curiosity). Be conversational. Keep it short. "
} )
_skip_research = True
else :
_skip_research = False
if not _skip_research :
# Phase 2: Start actual research
def _on_research_done ( _sid , _result , _sources , _findings ) :
""" Persist research to DB when background task finishes. """
if incognito :
return
try :
_s = session_manager . get_session ( _sid )
if not _s :
logger . warning ( f " Session { _sid } expired before research completed " )
return
_md = { " research " : True , " model " : _s . model }
if _sources :
_md [ " research_sources " ] = _sources
if _findings :
_md [ " research_findings " ] = _findings
_clean_res , _md = clean_thinking_for_save ( _result , _md )
_s . add_message ( ChatMessage ( " assistant " , _clean_res , metadata = _md ) )
session_manager . save_sessions ( )
logger . info ( f " Research result persisted to DB for session { _sid } " )
except Exception as _e :
logger . error ( f " Failed to persist research to DB: { _e } " )
# Check for prior research to continue from
_prior_report = " "
_prior_findings = None
_prior_urls = None
_prior_json = research_handler . _get_session_json ( session )
if _prior_json :
_prior_report = _prior_json . get ( " raw_report " , " " )
_prior_findings = _prior_json . get ( " raw_findings " )
_src_urls = { s . get ( " url " , " " ) for s in ( _prior_json . get ( " sources " ) or [ ] ) if s . get ( " url " ) }
_prior_urls = _src_urls if _src_urls else None
if _prior_report :
logger . info ( f " Continuing research for session { session } with { len ( _src_urls ) } prior URLs " )
# Synthesize conversation into a focused research query
_research_query = await research_handler . synthesize_query (
sess , message , _r_ep , _r_model , _r_headers ,
)
logger . info ( f " Research query: { _research_query [ : 120 ] } " )
research_handler . start_research (
session , _research_query , _r_ep , _r_model ,
llm_headers = _r_headers ,
prior_report = _prior_report ,
prior_findings = _prior_findings ,
prior_urls = _prior_urls ,
on_complete = _on_research_done ,
2026-06-02 18:32:38 +01:00
owner = _user ,
2026-05-31 23:58:26 +09:00
)
_heartbeat_counter = 0
_last_progress = { }
_sent_avg = False
while True :
status = research_handler . get_status ( session )
if not status or status [ " status " ] != " running " :
break
progress = status . get ( " progress " , { } )
if progress and progress != _last_progress :
_last_progress = progress
if not _sent_avg :
_sent_avg = True
progress = dict ( progress )
progress [ " started_at " ] = status . get ( " started_at " )
avg = status . get ( " avg_duration " )
if avg :
progress [ " avg_duration " ] = avg
yield f " data: { json . dumps ( { ' type ' : ' research_progress ' , ' data ' : progress } ) } \n \n "
_heartbeat_counter = 0
else :
_heartbeat_counter + = 1
yield f " : heartbeat { _heartbeat_counter } \n \n "
await asyncio . sleep ( 1.0 )
research_sources = research_handler . get_sources ( session )
if research_sources :
yield f " data: { json . dumps ( { ' type ' : ' research_sources ' , ' data ' : research_sources } ) } \n \n "
research_findings = research_handler . get_raw_findings ( session )
if research_findings :
yield f " data: { json . dumps ( { ' type ' : ' research_findings ' , ' data ' : research_findings } ) } \n \n "
# Signal frontend to fetch and render the research result
yield f " data: { json . dumps ( { ' type ' : ' research_done ' , ' data ' : { ' session_id ' : session } } ) } \n \n "
yield " data: [DONE] \n \n "
research_handler . clear_result ( session )
_stream_set ( session , status = " done " )
_active_streams . pop ( session , None )
return
2026-07-07 00:50:07 +00:00
messages = _ensure_current_request_is_latest_user ( ctx . messages , message )
2026-05-31 23:58:26 +09:00
# Auto-compact notification
if ctx . was_compacted :
yield f " data: { json . dumps ( { ' type ' : ' compacted ' , ' context_length ' : ctx . context_length } ) } \n \n "
2026-07-07 00:50:07 +00:00
if ctx . context_trimmed and not ctx . was_compacted :
yield f " data: { json . dumps ( { ' type ' : ' context_trimmed ' , ' data ' : { ' context_length ' : ctx . context_length , ' messages_before ' : ctx . context_messages_before_trim , ' messages_after ' : ctx . context_messages_after_trim , ' tokens_before ' : ctx . context_tokens_before_trim , ' tokens_after ' : ctx . context_tokens_after_trim } } ) } \n \n "
2026-05-31 23:58:26 +09:00
full_response = " "
2026-07-07 00:50:07 +00:00
thinking_response = " "
2026-05-31 23:58:26 +09:00
last_metrics = None
# Configured fallback chain for the default chat model. Tried in
# order if the session's primary model fails before producing
# output. Resolved once per request.
try :
from src . endpoint_resolver import resolve_chat_fallback_candidates
2026-06-02 19:40:22 +02:00
_fallback_candidates = resolve_chat_fallback_candidates ( owner = _user )
2026-05-31 23:58:26 +09:00
except Exception :
_fallback_candidates = [ ]
# Send model name early so the frontend can show it during streaming
2026-06-06 18:48:24 -06:00
_model_suffix = " Research " if effective_do_research else None
2026-05-31 23:58:26 +09:00
_model_info = { " type " : " model_info " , " model " : sess . model }
if _model_suffix :
_model_info [ " suffix " ] = _model_suffix
if ctx . preset . character_name :
_model_info [ " character_name " ] = ctx . preset . character_name
yield f ' data: { json . dumps ( _model_info ) } \n \n '
2026-06-02 19:40:22 +02:00
if _is_image_generation_session ( sess , owner = _user ) :
2026-05-31 23:58:26 +09:00
from src . settings import get_setting
2026-06-06 18:48:24 -06:00
if tool_policy . blocks ( " generate_image " ) :
_blocked_msg = tool_policy . reason_for ( " generate_image " )
yield f ' data: { json . dumps ( { " delta " : _blocked_msg } ) } \n \n '
yield " data: [DONE] \n \n "
_active_streams . pop ( session , None )
return
2026-05-31 23:58:26 +09:00
if not get_setting ( " image_gen_enabled " , True ) :
yield f ' data: { json . dumps ( { " delta " : " Image generation is disabled by the administrator. " } ) } \n \n '
yield " data: [DONE] \n \n "
_active_streams . pop ( session , None )
return
from src . ai_interaction import do_generate_image
_user_msg = message or " "
yield f ' data: { json . dumps ( { " type " : " tool_start " , " tool " : " generate_image " , " command " : _user_msg [ : 100 ] } ) } \n \n '
yield " : heartbeat \n \n "
2026-06-02 19:40:22 +02:00
_img_result = await do_generate_image ( f " { _user_msg } \n { sess . model } " , session , owner = _user )
2026-05-31 23:58:26 +09:00
_img_output = _img_result . get ( " results " , _img_result . get ( " error " , " " ) )
_img_tool_data = { " type " : " tool_output " , " tool " : " generate_image " , " command " : _user_msg [ : 100 ] , " output " : _img_output , " exit_code " : 0 if " error " not in _img_result else 1 }
for _k in ( " image_url " , " image_id " , " image_prompt " , " image_model " , " image_size " , " image_quality " ) :
if _k in _img_result :
_img_tool_data [ _k ] = _img_result [ _k ]
yield f ' data: { json . dumps ( _img_tool_data ) } \n \n '
_desc = _img_result . get ( " results " , _img_result . get ( " error " , " Image generation complete " ) )
full_response = _desc
yield f ' data: { json . dumps ( { " delta " : _desc } ) } \n \n '
# Save to session history
if not incognito :
_ev = { " round " : 1 , " tool " : " generate_image " , " command " : _user_msg [ : 100 ] , " output " : _img_output , " exit_code " : 0 if " error " not in _img_result else 1 }
for _ek in ( " image_url " , " image_id " , " image_prompt " , " image_model " , " image_size " , " image_quality " ) :
if _img_result . get ( _ek ) :
_ev [ _ek ] = _img_result [ _ek ]
sess . add_message ( ChatMessage ( " assistant " , full_response , metadata = { " tool_events " : [ _ev ] , " model " : sess . model } ) )
session_manager . save_sessions ( )
yield f ' data: { json . dumps ( { " type " : " metrics " , " data " : { " total_time " : 0 } } ) } \n \n '
yield " data: [DONE] \n \n "
_active_streams . pop ( session , None )
return
elif chat_mode == " chat " :
_chat_start = time . time ( )
2026-06-02 04:37:25 +02:00
_answered_by = None # set if the selected model failed and a fallback answered
2026-06-06 14:30:16 +04:00
_requested_model = sess . model
_actual_model = None
2026-05-31 23:58:26 +09:00
# ── Chat mode: call stream_llm directly, NO tools, NO document access ──
try :
_chat_candidates = [ ( sess . endpoint_url , sess . model , sess . headers ) ] + _fallback_candidates
async for chunk in stream_llm_with_fallback (
_chat_candidates ,
messages ,
temperature = ctx . preset . temperature ,
# Respect the preset; 0/unset = let the server decide (no
# cap), matching agent mode. The old hard 4096 fallback
# truncated reasoning models mid-<think> — they'd burn the
# whole budget thinking and never emit the answer (seen in
# Compare on heavy generation prompts).
max_tokens = ctx . preset . max_tokens ,
prompt_type = preset_id ,
tools = None ,
fix(chat): stabilize system prompt, sequence memory extraction, and send stable session id to preserve KV cache (#3360)
* fix(chat): stabilize system prompt, sequence memory extraction, send stable session id to preserve KV cache
Fixes #2927. As diagnosed in the issue, three things in Odysseus's request
pattern actively destroyed local backends' (llama.cpp / LM Studio) KV-cache
continuity, forcing a full prompt re-evaluation (15-30s+) on every turn:
1. Dynamic content folded into the system prompt every turn. Both the chat
preface (ChatProcessor.build_context_preface) and the agent system prompt
(_build_system_prompt) injected current_datetime_prompt() — text that
changes every minute — directly into system-role messages, which llm_core
then concatenates into the single system message sent as the cached
prefix. Any byte difference there invalidates the entire cache. Moved this
to a new current_datetime_context_message() helper that returns a
standalone user-role message, inserted near the end of the array (right
before the latest user turn) instead of mixed into the system prompt. The
static system prefix (preset prompt + safety policy + agent base prompt)
now stays byte-identical across turns of the same session.
2. Memory/skill extraction side-requests competed with the main completion.
run_post_response_tasks fired extract_and_store / maybe_extract_skill via
asyncio.create_task — fire-and-forget coroutines that could overlap the
next turn's main request and steal llama.cpp's limited processing slots,
evicting the cached checkpoint. They're now queued through a new
_run_extraction_jobs_sequentially helper that waits for the session's
stream to go idle and runs the jobs strictly one at a time.
3. No stable session identifier was sent to local backends, so llama.cpp
assigned a new processing slot via LRU every turn ("session_id=<empty>
server-selected (LCP/LRU)"), losing slot affinity. Added
_apply_local_cache_affinity() in llm_core, which sets session_id and
cache_prompt: true on outgoing payloads — gated to self-hosted
OpenAI-compatible endpoints only (never api.openai.com or other cloud
providers, which reject unrecognized request fields with a 400). Threaded
session_id through stream_llm / llm_call_async / stream_agent_loop from
the existing Odysseus session id.
Tests in tests/test_kv_cache_invalidation_2927.py exercise the real payload-
assembly and scheduling code paths: byte-identical system prefix across two
turns of the same session (with a regression check that genuinely changed
instructions DO still change it), the dynamic time block landing as a
user-role message, extraction jobs waiting for the stream to go idle and
running sequentially, and the outgoing payload carrying a stable session_id
(same across turns of one session, different across sessions) only for
self-hosted endpoints. Updated tests/test_user_time.py for the new message
placement.
* fix(tests): accept owner= kwarg in normalize_model_id monkeypatch
The upstream normalize_model_id signature now takes an owner= keyword
argument, and chat_helpers.py passes owner=getattr(sess, "owner", None)
at the call site. Update the test stub lambda to **kwargs so it handles
the new argument without breaking, and update chat_helpers.py to forward
the owner parameter consistently.
---------
Co-authored-by: Alexandre Teixeira <111787685+alteixeira20@users.noreply.github.com>
2026-06-09 18:46:54 -03:00
session_id = session ,
2026-05-31 23:58:26 +09:00
) :
if chunk . startswith ( " data: " ) and not chunk . startswith ( " data: [DONE] " ) :
try :
data = json . loads ( chunk [ 6 : ] )
if " delta " in data :
2026-06-02 06:17:41 +04:00
# Reasoning tokens arrive flagged thinking:true.
# Forward them so the client can show a thinking
# indicator, but don't fold them into the saved
# reply (mirrors the rewrite path below).
2026-07-07 00:50:07 +00:00
if data . get ( " thinking " ) :
thinking_response + = data [ " delta " ]
else :
2026-06-02 06:17:41 +04:00
full_response + = data [ " delta " ]
_stream_set ( session , partial = full_response )
2026-05-31 23:58:26 +09:00
yield chunk
2026-06-02 04:37:25 +02:00
elif data . get ( " type " ) == " fallback " :
# Selected model failed; a fallback answered.
# Forward the notice and remember the real model.
_answered_by = data . get ( " answered_by " ) or _answered_by
2026-06-06 14:30:16 +04:00
_actual_model = _actual_model or _answered_by
data [ " selected_model " ] = data . get ( " selected_model " ) or _requested_model
2026-06-02 04:37:25 +02:00
yield chunk
2026-06-06 14:30:16 +04:00
elif data . get ( " type " ) == " model_actual " :
_actual_model = data . get ( " model " ) or _actual_model
data [ " requested_model " ] = _requested_model
yield f ' data: { json . dumps ( data ) } \n \n '
2026-05-31 23:58:26 +09:00
elif data . get ( " type " ) == " usage " :
last_metrics = data . get ( " data " , { } )
2026-06-06 14:30:16 +04:00
_reported_model = last_metrics . get ( " model " )
last_metrics [ " requested_model " ] = _requested_model
last_metrics [ " model " ] = _reported_model or _actual_model or _answered_by or _requested_model
2026-07-07 00:50:07 +00:00
if ctx . context_trimmed :
last_metrics [ " context_trimmed " ] = True
last_metrics [ " context_messages_before_trim " ] = ctx . context_messages_before_trim
last_metrics [ " context_messages_after_trim " ] = ctx . context_messages_after_trim
last_metrics [ " context_tokens_before_trim " ] = ctx . context_tokens_before_trim
last_metrics [ " context_tokens_after_trim " ] = ctx . context_tokens_after_trim
2026-05-31 23:58:26 +09:00
if ctx . context_length and last_metrics . get ( " input_tokens " ) :
pct = min ( round ( ( last_metrics [ " input_tokens " ] / ctx . context_length ) * 100 , 1 ) , 100.0 )
last_metrics [ " context_percent " ] = pct
last_metrics [ " context_length " ] = ctx . context_length
Chat metrics: surface backend generation speed
* Chat metrics: show backend's true generation t/s, not tokens÷wall-clock
The per-message tokens/sec read low and felt wrong because it was computed as
output_tokens / total_duration, where total_duration is wall-clock including
prefill, tool calls, and network — not pure decode time. llama.cpp already
reports the correct gen speed in its stream (timings.predicted_per_second), but
it was being dropped.
- llm_core.py: when parsing the OpenAI-compatible usage chunk, also read the
sibling `timings` block llama.cpp includes — pass predicted_per_second through
as gen_tps and prompt_per_second as prefill_tps on the usage event.
- agent_loop.py: capture backend_gen_tps/backend_prefill_tps from usage events;
in _compute_final_metrics prefer backend_gen_tps over the wall-clock division
when present (fall back to computed for cloud APIs that omit timings). Tag the
result with tps_source ("backend" vs "computed") and surface prefill_tps.
Result: the displayed t/s now matches the model's real decode speed and is
stable regardless of prompt length (a long prefill no longer deflates it).
Checks: py_compile passes; verified extraction against a real llama.cpp final
chunk (gen 79 t/s surfaced vs the deflated wall-clock figure shown before).
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
* Chat metrics: surface true t/s on the direct-chat path too
Follow-up to the gen-tps work: the non-agent direct-chat stream path in
chat_routes turned the raw `usage` event straight into a metrics event but only
copied token counts — it never set tokens_per_second or response_time. So simple
(non-tool) replies showed "Speed: n/a" / "Time: undefineds" and the chip fell
back to a bare token count ("27 tok") instead of t/s.
Map the usage event's gen_tps (llama.cpp timings.predicted_per_second, added in
the prior commit) into tokens_per_second here too, tag tps_source=backend, and
set response_time from wall-clock for the stats popup.
Checks: py_compile passes; verified llama.cpp emits usage+timings on the final
stream chunk (gen ~90 t/s) that this path consumes.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
* Tests: backend gen/prefill t/s passthrough and preference
Cover the two pieces of the true-t/s metric so it can be reviewed on its own:
- stream_llm surfaces llama.cpp's timings.predicted_per_second /
prompt_per_second as gen_tps / prefill_tps on the usage event (captured
llama.cpp final-chunk fixture), and omits them when the backend reports no
timings.
- _compute_final_metrics prefers backend_gen_tps over output/wall-clock,
tags tps_source ("backend" vs "computed"), and surfaces prefill_tps.
Reuses the fake-client stream harness from test_llm_core_streaming.py.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
---------
Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-02 13:52:08 +02:00
# The frontend reads `tokens_per_second`; the raw usage event
# carries the backend's true gen speed as `gen_tps` (llama.cpp
# timings). Map it through so this direct-chat path shows real
# t/s instead of "n/a" → falling back to a bare token count.
if last_metrics . get ( " gen_tps " ) and not last_metrics . get ( " tokens_per_second " ) :
last_metrics [ " tokens_per_second " ] = last_metrics [ " gen_tps " ]
last_metrics [ " tps_source " ] = " backend "
# Wall-clock response time for the stats popup ("Time").
last_metrics . setdefault ( " response_time " , round ( time . time ( ) - _chat_start , 2 ) )
2026-05-31 23:58:26 +09:00
yield f ' data: { json . dumps ( { " type " : " metrics " , " data " : last_metrics } ) } \n \n '
except json . JSONDecodeError :
yield chunk
elif chunk . startswith ( " event: error " ) :
logger . warning ( f " Stream error for { sess . model } on { sess . endpoint_url } : { chunk !r} " )
yield chunk
elif chunk . startswith ( " event: " ) :
yield chunk
elif chunk == " data: [DONE] \n \n " :
# Generate fallback metrics if LLM didn't send usage
if not last_metrics and full_response :
_elapsed = time . time ( ) - _chat_start
_est_in = estimate_tokens ( messages )
_est_out = len ( full_response ) / / 4
_tps = round ( _est_out / _elapsed , 2 ) if _elapsed > 0 else 0
_ctx_pct = min ( round ( ( _est_in / ctx . context_length ) * 100 , 1 ) , 100.0 ) if ctx . context_length else 0
last_metrics = {
" response_time " : round ( _elapsed , 2 ) ,
" input_tokens " : _est_in ,
" output_tokens " : _est_out ,
" tokens_per_second " : _tps ,
" context_percent " : _ctx_pct ,
" context_length " : ctx . context_length ,
2026-06-06 14:30:16 +04:00
" model " : _actual_model or _answered_by or _requested_model ,
" requested_model " : _requested_model ,
2026-05-31 23:58:26 +09:00
" usage_source " : " estimated " ,
}
yield f ' data: { json . dumps ( { " type " : " metrics " , " data " : last_metrics } ) } \n \n '
if full_response :
2026-07-07 00:50:07 +00:00
_metrics_to_save = dict ( last_metrics or { } )
if thinking_response . strip ( ) and not _metrics_to_save . get ( " thinking " ) :
_metrics_to_save [ " thinking " ] = thinking_response . strip ( )
2026-05-31 23:58:26 +09:00
_saved_id = save_assistant_response (
2026-07-07 00:50:07 +00:00
sess , session_manager , session , full_response , _metrics_to_save ,
2026-05-31 23:58:26 +09:00
character_name = ctx . preset . character_name ,
web_sources = web_sources ,
rag_sources = ctx . rag_sources ,
research_sources = research_sources ,
used_memories = ctx . used_memories ,
2026-06-06 18:48:24 -06:00
do_research = effective_do_research ,
2026-05-31 23:58:26 +09:00
incognito = incognito ,
)
if _saved_id :
yield f ' data: { json . dumps ( { " type " : " message_saved " , " id " : _saved_id } ) } \n \n '
run_post_response_tasks (
sess , session_manager , session , message , full_response ,
2026-07-07 00:50:07 +00:00
_metrics_to_save , ctx . uprefs , memory_manager , memory_vector , webhook_manager ,
2026-05-31 23:58:26 +09:00
incognito = incognito , compare_mode = compare_mode ,
character_name = ctx . preset . character_name ,
2026-06-06 18:48:24 -06:00
owner = _user ,
allow_background_extraction = not tool_policy . block_all_tool_calls ,
2026-05-31 23:58:26 +09:00
)
_stream_set ( session , status = " done " )
yield chunk
except ( asyncio . CancelledError , GeneratorExit ) :
if full_response :
logger . info ( " Client disconnected mid-stream (chat mode) for session %s , saving partial ( %d chars) " , session , len ( full_response ) )
2026-06-06 14:30:16 +04:00
_stopped_content , _stopped_md = clean_thinking_for_save (
full_response ,
{
" stopped " : True ,
" model " : _actual_model or _answered_by or _requested_model ,
" requested_model " : _requested_model ,
} ,
)
2026-05-31 23:58:26 +09:00
sess . add_message ( ChatMessage ( " assistant " , _stopped_content , metadata = _stopped_md ) )
if not incognito :
session_manager . save_sessions ( )
raise
finally :
_active_streams . pop ( session , None )
else :
# ── Agent mode: full agent loop with tools ──
_agent_rounds = 0
_agent_tool_calls = 0
2026-06-02 04:37:25 +02:00
_answered_by = None # set if the selected model failed and a fallback answered
2026-06-06 14:30:16 +04:00
_requested_model = sess . model
_actual_model = None
2026-05-31 23:58:26 +09:00
try :
from src . settings import get_setting
2026-06-04 22:36:05 +02:00
from src . agent_tools import MAX_AGENT_ROUNDS as _DEFAULT_ROUNDS
2026-06-27 23:50:48 +05:30
# Per-message tool budget from settings; guard defensively in
# case settings.json was hand-edited to a non-numeric value
# (the HTTP admin endpoint validates, but direct edits bypass
# it). 0 = unlimited, matching auth_routes set_settings().
try :
_tool_budget = int ( get_setting ( " agent_max_tool_calls " , 0 ) )
except ( TypeError , ValueError ) :
_tool_budget = 0
2026-06-04 22:36:05 +02:00
# Per-message round cap from settings; clamp defensively in
# case settings.json was hand-edited to a bad value.
try :
_max_rounds = int ( get_setting ( " agent_max_rounds " , _DEFAULT_ROUNDS ) or _DEFAULT_ROUNDS )
except ( TypeError , ValueError ) :
_max_rounds = _DEFAULT_ROUNDS
_max_rounds = max ( 1 , min ( _max_rounds , 200 ) )
2026-05-31 23:58:26 +09:00
2026-06-21 11:02:35 +00:00
_forced_tools = None
2026-07-07 00:50:07 +00:00
if _explicit_web_intent :
_forced_tools = { " web_search " , " web_fetch " }
elif _search_enabled :
2026-06-21 11:02:35 +00:00
_forced_tools = { " web_search " , " web_fetch " }
2026-05-31 23:58:26 +09:00
async for chunk in stream_agent_loop (
sess . endpoint_url ,
sess . model ,
messages ,
headers = sess . headers ,
temperature = ctx . preset . temperature ,
max_tokens = ctx . preset . max_tokens ,
prompt_type = preset_id ,
max_tool_calls = _tool_budget ,
2026-06-04 22:36:05 +02:00
max_rounds = _max_rounds ,
2026-05-31 23:58:26 +09:00
context_length = ctx . context_length ,
active_document = active_doc ,
Open email context for agent, email search across All Mail, cookbook serve polish
- Agent: pass the open email reader (uid/folder/account/from/subject/body
preview) on every chat submit so 'reply to this' / 'write email saying
hi' route to ui_control open_email_reply with the right UID instead of
inventing a new .md draft. Code-level enforcement (chat_routes strips
create_document + send_email when active_email is set); cross-session
active_doc_id is now trusted instead of being silently dropped.
set_active_email/clear_active_email tool-layer helpers in
tool_implementations.
- ui_control open_email_reply: optional body argument so the agent can
open-and-write in one call; envelope now forwards uid/folder/account/
body/panel through tool_output. Tool description sharpened and the
parser rejects empty bodies on reply/reply-all (forces the agent to
write rather than open an empty draft).
- Email library: search now runs against [Gmail]/All Mail when the
current folder is INBOX (archived emails surface). Whirlpool spinner
+ 'Searching…' placeholder while in flight. Each search result is
stamped with its source folder so clicks open the right email instead
of whatever shares its UID in INBOX. Search no longer re-applies the
same text pill locally (which only checks subject/from/snippet, never
body) so body-only matches don't get dropped after IMAP returns them.
Initial inbox load bumped 100→500.
- Email favorites: 'Favorite (pin to top)' / 'Unfavorite' in both the
card menu and the open-reader more menu, backed by a new
/api/email/flag/{uid}?on=true|false endpoint. Flagged emails always
bubble to the top of the grid regardless of active sort.
- AI reply in doc editor: never overwrites existing draft text or the
quoted history. AI suggestion is prepended; AI-generated 'On …
wrote:' re-quotes are stripped so the original quote isn't visually
edited.
- Cookbook serve: pre-launch GPU driver / has_gpu / install / version-
floor checks (vllm minimax_m2 needs 0.10.0+, deepseek_r1 needs 0.7.0
etc.) before the launch chain starts. Detect 'another model already
running on this host' and offer Stop & launch (with graceful then
force tmux kill helpers, port release wait). Per-vendor deep-link
buttons (vLLM recipe / SGLang cookbook) with hardware hash. Backend
picker is now a custom dropdown with accent-coloured logos for vLLM,
SGLang, llama.cpp, Ollama, Diffusers; same glyphs added next to
package names in Dependencies. Runtime-readiness note moved inside
the panel (green when ready, red when missing) with an × dismiss.
Esc collapses the expanded card; expanded card scrolls when it
overflows; Trust Remote / Auto Tool / Reasoning Parser / Enforce
Eager / Prefix Caching / Expert Parallel / Speculative / MoE Env on
one row (Reasoning Parser auto-detected per model family).
Dtype→Row 1, GPUs→Row 2 (rightmost). Removed redundant GPU 'auto'
input — command builders read from the GPU button strip. Default
cookbook open is Download tab.
- Cookbook hwfit: 'Model (latest)' / 'Model (oldest)' header sorts by
release_date; release dates can be backfilled with the new
scripts/backfill_model_release_dates.py and recipe metadata pulled
with scripts/import_from_vllm_recipes.py against the upstream
vllm-project/recipes catalog (vllm_recipe + min_vllm_version stamped
on entries).
- Calendar: Quick add hint cycles a random Odysseus-themed example per
open (wooden horse Friday, crew muster 10am daily, council on
Ithaca, …). Typing a time like '11pm' in the event title updates
the hero clock live.
- Doc editor: email-mode Reply button (sparkle icon, accent) opens the
same Fast/Full + context popover the email reader uses; Ctrl+Alt+M
toggles markdown preview.
- Memories panel: custom sort picker with per-option icons, default
'Latest', visible Enabled/Disabled toggle text matching the section
description style.
2026-06-15 20:47:51 +09:00
active_email = active_email_ctx ,
2026-05-31 23:58:26 +09:00
session_id = session ,
disabled_tools = disabled_tools if disabled_tools else None ,
2026-06-06 18:48:24 -06:00
tool_policy = tool_policy ,
2026-05-31 23:58:26 +09:00
owner = _user ,
fallbacks = _fallback_candidates ,
feat: Add plan mode to the chat agent (#638)
* feat: Add plan mode to the chat agent
Adds a plan mode: the agent investigates read-only, proposes a checklist, and
waits for approval before changing anything. On approval it runs with full
tools and checks items off as it goes. Enforcement reuses the existing
disabled_tools gate.
Includes a slash command: `/plan [on|off]` (and `/toggle plan`) to flip the
plan toggle from the chat input.
- src/tool_security.py, src/mcp_manager.py: read-only allowlist (tools + MCP).
- src/agent_loop.py, routes/chat_routes.py: union the disabled set, prepend the
plan directive, force agent mode.
- static/: plan toggle pill, Approve & Run, dockable plan window, task-list
checkboxes, and the /plan slash command.
- tests/test_plan_mode.py.
* Plan mode: persistent re-referenceable plan + agent write-back
Three improvements so a long plan survives a weak model and stays in reach:
1. Re-reference the plan (out-of-context fix). On the execution turn the frontend
sends the approved checklist back (`approved_plan`); the backend pins it as a
top-of-context `## ACTIVE PLAN` system note (kept by the context trimmer), so
the agent can always re-read the plan instead of losing the thread on a long
run. New `build_active_plan_note()` (unit-tested).
2. Re-open / dock the plan anytime. The plan checklist is stored per-session
(localStorage). When a plan exists, the plan-mode button opens a small menu
("Show plan" / "Plan mode: On/Off") that re-opens the side-dockable plan
window — so it can stay docked while the agent works. The window live-refreshes
as the plan changes.
3. Agent write-back: new `update_plan` tool. The agent calls it to tick steps
`- [x]` after finishing them, or to revise steps when the user asks. Marker
tool (no I/O) → `plan_update` SSE event → the stored plan + docked window
update live. The ACTIVE PLAN note instructs the agent to use it.
Backend: src/agent_loop.py (param + pin + note builder + emit + prompt blurb),
src/tool_execution.py (update_plan handler), routes/chat_routes.py (parse
`approved_plan`, relay `plan_update`), registration in tool_schemas / agent_tools
/ tool_index (always-available, not admin-gated).
Frontend: static/js/chat.js (plan store, send `approved_plan`, handle
`plan_update`, capture restated checklists), static/app.js (plan-button menu),
static/js/planWindow.js (`isPlanWindowOpen`), static/js/storage.js (PLAN key).
Tests: tests/test_plan_mode.py (plan-note), tests/test_update_plan_tool.py.
* Plan mode: drop bash/python, rely on read-only discovery tools
Shell can mutate (write files, hit the network) and can't be constrained to
read-only at the tool layer, so plan mode no longer relies on a prompt to keep
it well-behaved — bash/python are removed from the read-only allowlist and added
to the fail-closed block set. Discovery is covered by the dedicated read-only
tools (read_file, grep, glob, ls) instead.
Rewrites the plan-mode directive to state shell is disabled and lists the
available read-only tools positively. Addresses review feedback on #638.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Comment: note _MCP_READONLY_VERBS are prefixes not whole words
Clarifies that entries like "summar" are intentional stems matched via
startswith (covers summarise/summarize/summary), not typos. Addresses review
feedback on #638.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Plan mode: clarify why gating inverts the allowlist into a denylist
Rename _PLAN_MODE_FALLBACK_BLOCK -> _PLAN_MODE_KNOWN_MUTATORS and rewrite the
comments. The tool gate is a denylist (disabled_tools); plan mode's policy is an
allowlist, so it returns the inverse (all known tool names minus the allowlist).
The static mutator set is a backstop for the schema-derived name list, which
misses XML-only tools and can fail to import. Addresses review feedback on #638.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Plan mode: stop hardcoding the read-only tool list in the directive
The model is already shown its available (read-only) tools by _assemble_prompt,
which removes every disabled tool. Enumerating them again in the directive only
duplicated that list and would drift as tools change. Point at the tools listed
below instead. Addresses review feedback on #638.
2026-06-05 16:32:25 +02:00
plan_mode = plan_mode ,
approved_plan = approved_plan or None ,
feat(agent): confine agent file/shell tools to a selectable workspace (#3665)
* feat(agent): workspace confinement via context-local binding + get_workspace tool
Bind the per-turn workspace once in execute_tool_block; the shared path
resolvers (_resolve_tool_path / _resolve_search_root) and the subprocess cwd
helper (agent_cwd) read it, so file tools + bash/python are confined centrally
and a new tool that uses the shared helpers cannot accidentally bypass it.
Adds the admin-gated /api/workspace/browse picker, a workspace pill + directory
modal (reusing existing modal/button CSS), the /workspace slash command, and a
get_workspace tool (replaces a system-prompt block). Confinement is OS-agnostic
(realpath/normcase/commonpath) and docker-safe (container paths, no host
assumptions). Reopens #2023.
* ux(workspace): clarify workspace is not a sandbox
Picker modal note + pill tooltip + get_workspace tool/output wording now state
plainly: read_file/write_file/edit_file/grep/glob/ls are confined to the folder,
but bash/python only start there (cwd) and are not sandboxed. Modal note reuses
the existing .muted class.
* fix(agent): treat an active workspace as file-work intent
A vague low-signal message (e.g. "look at the local project") matches no
domain keywords, so tool retrieval is skipped and only always-available tools
are offered — leaving the agent with no file access even though a workspace is
set. When a workspace is active, include the file/code tools (incl.
get_workspace) on low-signal turns so the agent can act on the folder.
Also requires the tool index (ChromaDB) to be reachable for normal retrieval;
that is an environment dependency, not part of this change.
* ux(workspace): hide pill + overflow entry in chat mode
Workspace only scopes the agent's file/shell tools, so the pill and the
overflow 'Workspace' entry are agent-only now — hidden in chat mode like the
bash toggle. Mode read from the DOM in syncWorkspaceIndicator; applyMode() is
called from the agent/chat setMode handler.
* prompt(tools): steer bash/python to defer to the dedicated file tools
bash/python schema descriptions (what native-tool-calling models read) were
bare and gave no steer, so models would do file ops via the shell (e.g. writing
SVG/HTML, which then dumps raw markup into the tool preview). Tell bash/python
in the schema + tool-index + prompt section to prefer read_file/write_file/
edit_file/grep/glob/ls and only be used for what those do not cover.
* prompt(tools): keep bash/python deferral generic (no hardcoded tool names)
Reference 'a dedicated tool' rather than listing read_file/write_file/grep/etc.
by name, so the guidance does not go stale if those tools are renamed.
* style(workspace): drop em-dashes from added code comments/strings
* ux(workspace): terser non-sandbox note in picker (no tool-name list)
* ux(workspace): mirror terse non-sandbox wording in pill tooltip
* chore: untrack local venv symlink (run-only, not part of the feature)
* prompt(workspace): keep get_workspace text generic (no hardcoded tool names)
* fix(agent): low-signal + workspace surfaces only read-only file tools
Intersect the files tool group with PLAN_MODE_READONLY_TOOLS so a vague message
in a workspace exposes read_file/grep/glob/ls/get_workspace for exploration, but
not write_file/edit_file/bash/python -- those wait for a request that actually
calls for them (RAG retrieval still adds them on a real ask).
* feat(workspace): cap browse listing at 500 dirs with a truncated hint
Mirror the filesystem_tools._CODENAV_MAX_HITS pattern with a module-local
_MAX_BROWSE_DIRS so a directory with thousands of children does not dump every
row into the picker; the response carries a truncated flag and the modal tells
the user to type a path to jump in.
* chore: untrack local venv symlink (run-only artifact)
* fix(workspace): vet the workspace root against the sensitive-path deny list at bind time
The in-workspace resolver deny-lists sensitive paths inside the workspace,
but the empty-path search root is the workspace itself, so a workspace of
~/.ssh could be listed via ls with no path. vet_workspace() (public, in
tool_execution next to the resolvers) rejects non-directories and sensitive
roots before the path is ever bound; chat_routes uses it instead of its
inline isdir check.
* fix(workspace): reject filesystem roots and stop showing rejected workspaces as active
Review findings from #3665:
P2: vet_workspace accepted / (and would accept drive/UNC roots), which makes
every absolute path 'inside' the workspace and collapses confinement into
host-wide file access. A root is its own dirname, so reject when
dirname(resolved) == resolved; the browse response now carries a selectable
flag and the picker disables 'Use this folder' on unselectable dirs.
P3: /workspace set stored any string client-side and the chat route silently
dropped rejected values, so the pill could claim a confinement that was not
in effect. New admin-gated /api/workspace/vet validates manual paths before
they persist (canonical path returned), and when a posted workspace is
rejected at send time the stream emits workspace_rejected so the client
clears the stored value and toasts instead of continuing silently.
* fix(workspace): check caller privilege before vetting the posted workspace
Review finding: /api/chat_stream called vet_workspace() on the posted value
for every caller and emitted workspace_rejected on failure, so a non-admin
who can chat but cannot use file/shell tools could distinguish existing
directories from missing/file/sensitive/root paths by whether the event
appeared. The resolution now lives in _resolve_request_workspace, which
drops the submitted value uniformly for non-admin callers, with no vetting
and no event, before the path ever touches the filesystem. Admin and
single-user behavior is unchanged. Test pins that valid and invalid paths
are indistinguishable for a non-admin and that vet_workspace is never
invoked for them.
2026-06-11 18:17:54 +02:00
workspace = workspace or None ,
2026-06-21 11:02:35 +00:00
forced_tools = _forced_tools ,
2026-06-27 21:24:17 +03:00
uploaded_files = ctx . uploaded_files ,
2026-05-31 23:58:26 +09:00
) :
if chunk . startswith ( " data: " ) and not chunk . startswith ( " data: [DONE] " ) :
try :
data = json . loads ( chunk [ 6 : ] )
if " delta " in data :
2026-06-02 06:17:41 +04:00
# Reasoning tokens arrive flagged thinking:true.
# Forward them for the live indicator, but keep
# them out of the saved reply (same as chat mode).
2026-07-07 00:50:07 +00:00
if data . get ( " thinking " ) :
thinking_response + = data [ " delta " ]
else :
2026-06-02 06:17:41 +04:00
full_response + = data [ " delta " ]
_stream_set ( session , partial = full_response )
2026-05-31 23:58:26 +09:00
yield chunk
elif data . get ( " type " ) == " web_sources " :
web_sources = data . get ( " data " , [ ] )
yield chunk
elif data . get ( " type " ) in (
" tool_start " , " tool_output " , " agent_step " ,
" doc_stream_open " , " doc_stream_delta " ,
" doc_update " , " doc_suggestions " , " ui_control " ,
2026-06-04 22:36:05 +02:00
" rounds_exhausted " ,
Add ask_user tool: agent-posed multiple-choice questions (#2111)
Let the agent pause and ask the user a multiple-choice question when a
task is genuinely ambiguous and the answer changes what it does next —
choosing between approaches, confirming an assumption, picking a target —
instead of guessing.
Modeled on the existing `ui_control` marker pattern: the `ask_user` tool
returns an `ask_user` payload that the agent loop emits as an SSE event
and then ends the turn. The frontend renders the question with clickable
option buttons, a free-text "Other" input, and an x to dismiss; the user's
choice is sent as the next message and the agent resumes with it in
context.
- src/tool_execution.py: `ask_user` handler — pure UI marker, no I/O.
Validates a non-empty question + 2..6 options, normalizes string/object
options, returns the payload.
- src/agent_loop.py: emit the `ask_user` event and break the round loop so
the turn ends and waits for the user's selection. Stream the question as
assistant text so it persists/replays (prevents a re-ask loop).
- Registration: TOOL_TAGS, ALWAYS_AVAILABLE, BUILTIN_TOOL_DESCRIPTIONS,
FUNCTION_TOOL_SCHEMAS, the system-prompt blurb. Not admin-gated (any
user can be asked); the structured args serialize via the default
json.dumps path.
- routes/chat_routes.py: relay the `ask_user` event to the client.
- static/js/chat.js + static/style.css: render the question card (options +
free-text Other + dismiss x; removed once answered). Reuses CSS vars and
the .modal-close button; emoji go through the monochrome-SVG pipeline.
Bump chat.js cache pin.
- tests/test_ask_user_tool.py: payload, multi flag, string options, option
cap, validation errors, serializer round-trip, registration.
2026-06-05 11:49:11 +02:00
" ask_user " ,
feat: Add plan mode to the chat agent (#638)
* feat: Add plan mode to the chat agent
Adds a plan mode: the agent investigates read-only, proposes a checklist, and
waits for approval before changing anything. On approval it runs with full
tools and checks items off as it goes. Enforcement reuses the existing
disabled_tools gate.
Includes a slash command: `/plan [on|off]` (and `/toggle plan`) to flip the
plan toggle from the chat input.
- src/tool_security.py, src/mcp_manager.py: read-only allowlist (tools + MCP).
- src/agent_loop.py, routes/chat_routes.py: union the disabled set, prepend the
plan directive, force agent mode.
- static/: plan toggle pill, Approve & Run, dockable plan window, task-list
checkboxes, and the /plan slash command.
- tests/test_plan_mode.py.
* Plan mode: persistent re-referenceable plan + agent write-back
Three improvements so a long plan survives a weak model and stays in reach:
1. Re-reference the plan (out-of-context fix). On the execution turn the frontend
sends the approved checklist back (`approved_plan`); the backend pins it as a
top-of-context `## ACTIVE PLAN` system note (kept by the context trimmer), so
the agent can always re-read the plan instead of losing the thread on a long
run. New `build_active_plan_note()` (unit-tested).
2. Re-open / dock the plan anytime. The plan checklist is stored per-session
(localStorage). When a plan exists, the plan-mode button opens a small menu
("Show plan" / "Plan mode: On/Off") that re-opens the side-dockable plan
window — so it can stay docked while the agent works. The window live-refreshes
as the plan changes.
3. Agent write-back: new `update_plan` tool. The agent calls it to tick steps
`- [x]` after finishing them, or to revise steps when the user asks. Marker
tool (no I/O) → `plan_update` SSE event → the stored plan + docked window
update live. The ACTIVE PLAN note instructs the agent to use it.
Backend: src/agent_loop.py (param + pin + note builder + emit + prompt blurb),
src/tool_execution.py (update_plan handler), routes/chat_routes.py (parse
`approved_plan`, relay `plan_update`), registration in tool_schemas / agent_tools
/ tool_index (always-available, not admin-gated).
Frontend: static/js/chat.js (plan store, send `approved_plan`, handle
`plan_update`, capture restated checklists), static/app.js (plan-button menu),
static/js/planWindow.js (`isPlanWindowOpen`), static/js/storage.js (PLAN key).
Tests: tests/test_plan_mode.py (plan-note), tests/test_update_plan_tool.py.
* Plan mode: drop bash/python, rely on read-only discovery tools
Shell can mutate (write files, hit the network) and can't be constrained to
read-only at the tool layer, so plan mode no longer relies on a prompt to keep
it well-behaved — bash/python are removed from the read-only allowlist and added
to the fail-closed block set. Discovery is covered by the dedicated read-only
tools (read_file, grep, glob, ls) instead.
Rewrites the plan-mode directive to state shell is disabled and lists the
available read-only tools positively. Addresses review feedback on #638.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Comment: note _MCP_READONLY_VERBS are prefixes not whole words
Clarifies that entries like "summar" are intentional stems matched via
startswith (covers summarise/summarize/summary), not typos. Addresses review
feedback on #638.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Plan mode: clarify why gating inverts the allowlist into a denylist
Rename _PLAN_MODE_FALLBACK_BLOCK -> _PLAN_MODE_KNOWN_MUTATORS and rewrite the
comments. The tool gate is a denylist (disabled_tools); plan mode's policy is an
allowlist, so it returns the inverse (all known tool names minus the allowlist).
The static mutator set is a backstop for the schema-derived name list, which
misses XML-only tools and can fail to import. Addresses review feedback on #638.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Plan mode: stop hardcoding the read-only tool list in the directive
The model is already shown its available (read-only) tools by _assemble_prompt,
which removes every disabled tool. Enumerating them again in the directive only
duplicated that list and would drift as tools change. Point at the tools listed
below instead. Addresses review feedback on #638.
2026-06-05 16:32:25 +02:00
" plan_update " ,
2026-05-31 23:58:26 +09:00
) :
if data . get ( " type " ) == " agent_step " :
_agent_rounds = max ( _agent_rounds , data . get ( " round " , 1 ) )
elif data . get ( " type " ) == " tool_start " :
_agent_tool_calls + = 1
yield chunk
2026-06-02 04:37:25 +02:00
elif data . get ( " type " ) == " fallback " :
# Selected model failed; a fallback answered.
# Forward the notice and remember the real
# model so metrics reflect it, not the masked
# selected model.
_answered_by = data . get ( " answered_by " ) or _answered_by
2026-06-06 14:30:16 +04:00
_actual_model = _actual_model or _answered_by
data [ " selected_model " ] = data . get ( " selected_model " ) or _requested_model
2026-06-02 04:37:25 +02:00
yield chunk
2026-06-06 14:30:16 +04:00
elif data . get ( " type " ) == " model_actual " :
_actual_model = data . get ( " model " ) or _actual_model
data [ " requested_model " ] = _requested_model
yield f ' data: { json . dumps ( data ) } \n \n '
2026-05-31 23:58:26 +09:00
elif data . get ( " type " ) == " metrics " :
last_metrics = data . get ( " data " , { } )
2026-06-06 14:30:16 +04:00
_reported_model = last_metrics . get ( " model " )
last_metrics [ " requested_model " ] = last_metrics . get ( " requested_model " ) or _requested_model
last_metrics [ " model " ] = _reported_model or _actual_model or _answered_by or _requested_model
2026-07-07 00:50:07 +00:00
if ctx . context_trimmed :
last_metrics [ " context_trimmed " ] = True
last_metrics [ " context_messages_before_trim " ] = ctx . context_messages_before_trim
last_metrics [ " context_messages_after_trim " ] = ctx . context_messages_after_trim
last_metrics [ " context_tokens_before_trim " ] = ctx . context_tokens_before_trim
last_metrics [ " context_tokens_after_trim " ] = ctx . context_tokens_after_trim
2026-05-31 23:58:26 +09:00
yield f ' data: { json . dumps ( { " type " : " metrics " , " data " : last_metrics } ) } \n \n '
except json . JSONDecodeError :
yield chunk
elif chunk . startswith ( " event: " ) :
yield chunk
elif chunk == " data: [DONE] \n \n " :
2026-07-01 11:12:55 +00:00
_has_tool_events = bool ( ( last_metrics or { } ) . get ( " tool_events " ) )
if full_response or _has_tool_events :
_response_to_save = full_response or " Done. "
2026-07-07 00:50:07 +00:00
_metrics_to_save = dict ( last_metrics or { } )
if thinking_response . strip ( ) and not _metrics_to_save . get ( " thinking " ) :
_metrics_to_save [ " thinking " ] = thinking_response . strip ( )
2026-05-31 23:58:26 +09:00
_saved_id = save_assistant_response (
2026-07-07 00:50:07 +00:00
sess , session_manager , session , _response_to_save , _metrics_to_save ,
2026-05-31 23:58:26 +09:00
character_name = ctx . preset . character_name ,
web_sources = web_sources ,
rag_sources = ctx . rag_sources ,
used_memories = ctx . used_memories ,
incognito = incognito ,
)
if _saved_id :
yield f ' data: { json . dumps ( { " type " : " message_saved " , " id " : _saved_id } ) } \n \n '
run_post_response_tasks (
2026-07-01 11:12:55 +00:00
sess , session_manager , session , message , _response_to_save ,
2026-07-07 00:50:07 +00:00
_metrics_to_save , ctx . uprefs , memory_manager , memory_vector , webhook_manager ,
2026-05-31 23:58:26 +09:00
incognito = incognito , compare_mode = compare_mode ,
character_name = ctx . preset . character_name ,
agent_rounds = _agent_rounds ,
agent_tool_calls = _agent_tool_calls ,
skills_manager = skills_manager ,
owner = _user ,
extract_skills = user_requested_agent ,
2026-06-06 18:48:24 -06:00
allow_background_extraction = not tool_policy . block_all_tool_calls ,
2026-05-31 23:58:26 +09:00
)
_stream_set ( session , status = " done " )
yield chunk
except ( asyncio . CancelledError , GeneratorExit ) :
# Client disconnected — save partial response. Wrap
# the save in its own try so an exception inside
# add_message / save_sessions doesn't mask the
# original CancelledError (which prevented the
# outer finally from running and left _active_streams
# with a stale entry).
try :
if full_response :
logger . info ( " Client disconnected mid-stream for session %s , saving partial response ( %d chars) " , session , len ( full_response ) )
2026-06-06 14:30:16 +04:00
_stopped_content2 , _stopped_md2 = clean_thinking_for_save (
full_response ,
{
" stopped " : True ,
" model " : _actual_model or _answered_by or _requested_model ,
" requested_model " : _requested_model ,
} ,
)
2026-05-31 23:58:26 +09:00
sess . add_message ( ChatMessage ( " assistant " , _stopped_content2 , metadata = _stopped_md2 ) )
if not incognito :
session_manager . save_sessions ( )
except Exception :
logger . exception ( " Failed to save partial response on disconnect (session %s ) " , session )
raise
finally :
_active_streams . pop ( session , None )
async def _safe_stream ( ) - > AsyncGenerator [ str , None ] :
""" Wrapper that guarantees _active_streams cleanup even if stream_with_save
raises before reaching a mode - specific finally block . """
try :
async for chunk in stream_with_save ( ) :
yield chunk
finally :
_active_streams . pop ( session , None )
2026-06-07 21:13:45 -03:00
# Compare panes are short-lived, single-shot generations whose sessions
# exist only to drive that one pane — there's nothing to "resume" and
# the user expects the pane's Stop button (which aborts the fetch,
# closing this SSE) to promptly cancel the upstream LLM call. Detaching
# them would keep burning upstream tokens/compute after the pane is
# stopped or the comparison is abandoned, and would surface a stale
# "still streaming" /resume target for a session nobody will revisit.
#
# So: stream them directly (no agent_runs wrapping). Starlette cancels
# the underlying async generator (raising CancelledError/GeneratorExit
# inside it) as soon as it notices the client disconnected — which the
# mode-specific except blocks above already handle by saving the
# partial response exactly once. This stops the upstream call promptly
# without waiting on the next streamed chunk.
#
# Normal chat/agent streams keep the DETACHED behavior below: they
2026-06-09 09:48:59 +09:00
# survive the client closing the tab / navigating away. The SSE response just subscribes (replay
2026-06-07 21:13:45 -03:00
# buffered output + live); dropping the SSE only removes a subscriber —
# the run keeps going and saves the assistant message on completion
# regardless. Reconnect via /api/chat/resume.
if compare_mode :
return StreamingResponse ( _safe_stream ( ) , media_type = " text/event-stream " )
2026-05-31 23:58:26 +09:00
agent_runs . start ( session , _safe_stream ( ) )
return StreamingResponse ( agent_runs . subscribe ( session ) , media_type = " text/event-stream " )
# ------------------------------------------------------------------ #
# GET /api/chat/resume — reconnect to a detached run that's still going
# (e.g. after reopening a session whose agent kept running in the background)
# ------------------------------------------------------------------ #
@router.get ( " /api/chat/resume/ {session_id} " )
async def chat_resume ( request : Request , session_id : str ) - > StreamingResponse :
_verify_session_owner ( request , session_id )
if not agent_runs . is_active ( session_id ) :
raise HTTPException ( 404 , " No active run for this session " )
return StreamingResponse ( agent_runs . subscribe ( session_id ) , media_type = " text/event-stream " )
# ------------------------------------------------------------------ #
# POST /api/chat/stop — cancel a detached run (Stop button). Closing the SSE
# no longer stops it (it's detached), so the Stop button must call this.
# ------------------------------------------------------------------ #
@router.post ( " /api/chat/stop/ {session_id} " )
async def chat_stop ( request : Request , session_id : str ) - > Dict [ str , Any ] :
_verify_session_owner ( request , session_id )
stopped = agent_runs . stop ( session_id )
return { " stopped " : stopped }
# ------------------------------------------------------------------ #
# GET /api/chat/stream_status — check if a stream is active for a session
# ------------------------------------------------------------------ #
@router.get ( " /api/chat/stream_status/ {session_id} " )
async def chat_stream_status ( request : Request , session_id : str ) - > Dict [ str , Any ] :
_verify_session_owner ( request , session_id )
# A detached run can still be going even if _active_streams was popped;
# report it as active so the client knows to reconnect via /resume.
2026-06-02 03:02:30 +05:30
# Read once via .get() to avoid a KeyError race between the membership
# check and the indexed read if a sibling stream's finally pops the
# entry in between (same pattern _stream_set already uses).
rec = _active_streams . get ( session_id )
if rec is None :
2026-05-31 23:58:26 +09:00
if agent_runs . is_active ( session_id ) :
return { " status " : " streaming " , " detached " : True }
raise HTTPException ( 404 , " No active stream for this session " )
2026-06-02 03:02:30 +05:30
return rec
2026-05-31 23:58:26 +09:00
# ------------------------------------------------------------------ #
# POST /api/inject_context
# ------------------------------------------------------------------ #
@router.post ( " /api/inject_context/ {session_id} " )
async def inject_context ( request : Request , session_id : str , context : str = Form ( . . . ) ) - > Dict [ str , str ] :
_verify_session_owner ( request , session_id )
try :
sess = session_manager . get_session ( session_id )
msg = untrusted_context_message ( " injected research context " , f " Research Context: { context } " )
sess . add_message ( ChatMessage ( msg [ " role " ] , msg [ " content " ] , metadata = msg . get ( " metadata " ) ) )
session_manager . save_sessions ( )
return { " status " : " context_injected " }
except KeyError :
raise HTTPException ( 404 , " Session not found " )
# ------------------------------------------------------------------ #
# GET /api/search — search across chat messages
# ------------------------------------------------------------------ #
@router.get ( " /api/search " )
async def search_messages (
request : Request ,
q : str = Query ( " " , min_length = 0 ) ,
limit : int = Query ( 20 , ge = 1 , le = 100 ) ,
) - > List [ Dict [ str , Any ] ] :
if not q or not q . strip ( ) :
return [ ]
fix(api): attribute bearer-token actions to the token owner on owner-scoped routes (#4054)
* fix(api): attribute bearer-token actions to the token owner on owner-scoped routes
Owner-scoped chat, session, and upload routes called
get_current_user(), which resolves a bearer ody_ API token to the
sandboxed "api" pseudo-user. A paired API-token client (companion, CLI,
IDE extension) therefore saw and created a separate "api"-owned silo
instead of the owner's data.
effective_user() already exists for exactly this: it attributes a token's
actions to request.state.api_token_owner, is identical to
get_current_user() for cookie sessions, and falls back safely when a
token has no owner. session_routes.py was already migrated; this
completes the migration for the remaining owner-scoped routes:
- chat_helpers.py: chat-privilege enforcement, message attribution, prefs/context
- chat_routes.py: orphaned-endpoint owner, session-auth owner, message search
- upload_routes.py: upload owner attribution + access checks
The /api/models swap is intentionally omitted: #4292 already migrated it
to effective_user (plus the chat-scope gate and ownerless-token 403), so
this PR keeps dev's version of routes/model_routes.py unchanged.
chat_routes.py keeps importing get_current_user for the workspace owner
gate; session_routes.py drops the now-unused import.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* test: target effective_user in auth monkeypatches and owner-scope assertion
The owner-scoped routes now call effective_user() instead of
get_current_user(), so the tests that stubbed get_current_user (or
asserted on it) follow suit:
- test_chat_helpers.py, test_review_regressions.py,
test_kv_cache_invalidation_2927.py: monkeypatch effective_user
- test_session_endpoint_owner_scope.py: assert the owner-scope guard uses
effective_user(request)
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
---------
Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-15 22:56:22 +01:00
_user = effective_user ( request )
2026-06-05 18:08:31 -06:00
return [
result . to_dict ( )
for result in search_session_messages (
q ,
limit = limit ,
owner = _user ,
restrict_owner = _user is not None ,
include_legacy_owner = False ,
2026-05-31 23:58:26 +09:00
)
2026-06-05 18:08:31 -06:00
]
2026-05-31 23:58:26 +09:00
# ------------------------------------------------------------------ #
# POST /api/rewrite — lightweight rewrite of last AI message (no tools)
# ------------------------------------------------------------------ #
@router.post ( " /api/rewrite " )
async def rewrite_message ( request : Request ) - > StreamingResponse :
""" Rewrite the last AI message with an instruction (shorter/simpler/etc).
Unlike the full chat pipeline , this does NOT run the agent loop or tools .
It just asks the LLM to rewrite the given text .
"""
try :
body = await request . json ( )
except Exception :
raise HTTPException ( 400 , " Invalid JSON " )
session_id = body . get ( " session_id " )
original_text = body . get ( " original_text " , " " )
instruction = body . get ( " instruction " , " " )
if not session_id or not original_text or not instruction :
raise HTTPException ( 400 , " session_id, original_text, and instruction are required " )
_verify_session_owner ( request , session_id )
try :
sess = session_manager . get_session ( session_id )
except ( KeyError , SessionNotFoundError ) :
raise HTTPException ( 404 , " Session not found " )
messages = [
{ " role " : " system " , " content " : (
" You are rewriting a previous response. Follow the instruction exactly. "
" Output ONLY the rewritten text — no preamble, no explanation, no meta-commentary. "
" Preserve any formatting (markdown, code blocks, lists) from the original. "
) } ,
{ " role " : " user " , " content " : (
f " Here is the original response: \n \n { original_text } \n \n "
f " Instruction: { instruction } "
) } ,
]
async def stream_rewrite ( ) - > AsyncGenerator [ str , None ] :
full_response = " "
try :
async for chunk in stream_llm (
sess . endpoint_url ,
sess . model ,
messages ,
headers = sess . headers ,
temperature = 0.7 ,
# 0 = let the server decide (no cap). A hardcoded 4096 made
# local reasoning models (Qwen3 / R1) burn the whole budget
# inside <think> and emit no rewrite — the bubble just hung
# on "Rewriting...". Same fix as the chat max_tokens cap.
max_tokens = 0 ,
tools = None ,
) :
if chunk . startswith ( " data: " ) and not chunk . startswith ( " data: [DONE] " ) :
try :
data = json . loads ( chunk [ 6 : ] )
if " delta " in data :
# Forward the chunk (so the client can show a
# thinking indicator) but DON'T fold reasoning
# tokens into the saved rewrite — only real
# content. reasoning_content arrives flagged
# with thinking:true.
if not data . get ( " thinking " ) :
full_response + = data [ " delta " ]
yield chunk
except json . JSONDecodeError :
yield chunk
elif chunk . startswith ( " event: " ) :
yield chunk
elif chunk == " data: [DONE] \n \n " :
# Update the last assistant message in session history.
# Strip reasoning-model <think> blocks so the persisted
# rewrite is just the rewritten text, not its scratchpad.
from src . research_utils import strip_thinking
full_response = strip_thinking ( full_response ) . strip ( ) or full_response
if full_response :
for msg in reversed ( sess . history ) :
if ( isinstance ( msg , ChatMessage ) and msg . role == ' assistant ' ) or \
( isinstance ( msg , dict ) and msg . get ( ' role ' ) == ' assistant ' ) :
if isinstance ( msg , ChatMessage ) :
msg . content = full_response
else :
msg [ ' content ' ] = full_response
break
# Update in DB too
db = SessionLocal ( )
try :
db_msg = (
db . query ( DBChatMessage )
. filter ( DBChatMessage . session_id == session_id , DBChatMessage . role == ' assistant ' )
2026-06-03 05:31:19 +01:00
. order_by ( DBChatMessage . timestamp . desc ( ) )
2026-05-31 23:58:26 +09:00
. first ( )
)
if db_msg :
db_msg . content = full_response
db . commit ( )
except Exception as e :
logger . warning ( " Failed to update rewritten message in DB: %s " , e )
db . rollback ( )
finally :
db . close ( )
session_manager . save_sessions ( )
yield chunk
except Exception as e :
logger . error ( " Rewrite stream error: %s " , e )
yield f ' event: error \n data: { json . dumps ( { " error " : str ( e ) , " status " : 500 } ) } \n \n '
return StreamingResponse ( stream_rewrite ( ) , media_type = " text/event-stream " )
return router