2026-05-31 23:58:26 +09:00
""" cookbook_helpers.py — validators + small helpers shared by the cookbook routes.
Extracted from cookbook_routes . py ; the routes module imports the symbols it needs . """
2026-06-11 23:53:16 +03:00
import json
2026-05-31 23:58:26 +09:00
import logging
2026-06-04 09:00:01 +05:30
import ntpath
2026-05-31 23:58:26 +09:00
import os
2026-06-01 21:08:39 +07:00
import posixpath
2026-05-31 23:58:26 +09:00
import re
import shlex
2026-06-11 23:53:16 +03:00
from pathlib import Path
2026-05-31 23:58:26 +09:00
from fastapi import HTTPException
from pydantic import BaseModel
2026-06-11 01:43:49 +03:00
from routes . _validators import validate_remote_host , validate_ssh_port
2026-06-08 00:33:50 +02:00
from core . platform_compat import _ssh_exec_argv
2026-05-31 23:58:26 +09:00
logger = logging . getLogger ( __name__ )
# HuggingFace repo IDs are <org>/<name>, both alphanumerics plus ._-
# Rejecting anything else up front closes off shell-interpolation vectors.
_REPO_ID_RE = re . compile ( r " ^[A-Za-z0-9][A-Za-z0-9._-]*/[A-Za-z0-9][A-Za-z0-9._-]*$ " )
2026-06-01 10:10:08 -04:00
# Cached models scanned from a custom/local model dir are keyed by their leaf
# folder name (no slash), e.g. `DeepSeek-R1-UD-IQ4_XS`. The serve command uses
# the real on-disk path separately; this identifier is only for UI/task
# bookkeeping, so serving should accept the same safe glyph set as repo IDs.
_LOCAL_MODEL_ID_RE = re . compile ( r " ^[A-Za-z0-9][A-Za-z0-9._-]*$ " )
2026-06-02 07:14:59 +09:00
# Ollama model names include tags, e.g. `qwen2.5:0.5b` or `llama3.2:latest`.
# Some registries also use a namespace path. Keep this shell-safe: no spaces,
# quotes, `$`, `;`, `&`, pipes, or redirects.
_OLLAMA_MODEL_ID_RE = re . compile ( r " ^[A-Za-z0-9][A-Za-z0-9._:/-] { 0,200}$ " )
2026-05-31 23:58:26 +09:00
# Include pattern is a glob: allow typical safe glyphs only.
_INCLUDE_RE = re . compile ( r " ^[A-Za-z0-9._ \ -*?/ \ [ \ ]]+$ " )
# HF tokens and API tokens are url-safe base64-like.
_TOKEN_RE = re . compile ( r " ^[A-Za-z0-9._~+/=-]+$ " )
# Session IDs we mint look like "cookbook-deadbeef" or "serve-deadbeef".
# Anything beyond plain alphanumerics + dash + underscore could break out
# of the shell/PowerShell contexts the value lands in.
_SESSION_ID_RE = re . compile ( r " ^[A-Za-z0-9_-] { 1,64}$ " )
_GPU_LIST_RE = re . compile ( r " ^ \ d+(?:, \ d+)*$ " )
# A download target directory. Absolute or ~-relative path; safe path glyphs
2026-06-09 07:58:38 +02:00
# only (no quotes or shell metacharacters). Spaces are allowed because command
# builders pass the value through quoted shell/Python contexts. The character
# class uses ``\w`` — Unicode word characters under Python 3's default str
# matching — so non-ASCII folder names pass validation too: Cyrillic, accented
# Latin, CJK, e.g. ``/Volumes/Модели`` or ``D:\AI Models\Модели``. This stays
# shell-safe: none of ``; & | ` $ '' "" () {}`` newlines etc. are in ``[\w. -]``,
# so injection vectors remain rejected. A leading ~ is expanded to $HOME at
# command-build time. (Drive letters stay ASCII: ``[A-Za-z]:``.)
_LOCAL_DIR_RE = re . compile ( r " ^~?(?:/[ \ w. -]*)+$|^~$ " )
_WINDOWS_LOCAL_DIR_RE = re . compile ( r " ^[A-Za-z]:[ \\ /](?:[ \ w. -]+(?:[ \\ /][ \ w. -]+)*[ \\ /]?)?$ " )
2026-06-04 09:00:01 +05:30
_WINDOWS_DRIVE_PATH_RE = re . compile ( r " ^[A-Za-z]:[ \\ /] " )
def _git_bash_path ( path : str ) - > str :
m = re . match ( r " ^([A-Za-z]):[ \\ /](.*)$ " , path )
if not m :
return path
drive , rest = m . groups ( )
return f " / { drive . lower ( ) } / { rest . replace ( chr ( 92 ) , ' / ' ) } "
2026-05-31 23:58:26 +09:00
def _validate_repo_id ( v : str | None ) - > str :
if not v or not _REPO_ID_RE . match ( v ) :
raise HTTPException ( 400 , " Invalid repo_id — must be <org>/<name> using [A-Za-z0-9._-] " )
return v
2026-06-01 10:10:08 -04:00
def _validate_serve_model_id ( v : str | None ) - > str :
if not v :
raise HTTPException ( 400 , " repo_id is required " )
2026-06-02 07:14:59 +09:00
if _REPO_ID_RE . match ( v ) or _LOCAL_MODEL_ID_RE . match ( v ) or _OLLAMA_MODEL_ID_RE . match ( v ) :
2026-06-01 10:10:08 -04:00
return v
2026-06-02 07:14:59 +09:00
raise HTTPException ( 400 , " Invalid repo_id — must be <org>/<name>, an Ollama name:tag, or a cached local model id " )
2026-06-01 10:10:08 -04:00
2026-05-31 23:58:26 +09:00
def _validate_include ( v : str | None ) - > str | None :
if v is None or v == " " :
return None
if not _INCLUDE_RE . match ( v ) :
raise HTTPException ( 400 , " Invalid include pattern " )
return v
def _validate_token ( v : str | None ) - > str | None :
if v is None or v == " " :
return None
if not _TOKEN_RE . match ( v ) :
raise HTTPException ( 400 , " Invalid token characters " )
return v
2026-06-11 23:53:16 +03:00
def load_stored_hf_token ( * , state_path : Path | str | None = None ) - > str :
""" Return the decrypted HF token from cookbook_state.json, else env fallback. """
path = Path ( state_path ) if state_path else Path ( os . environ . get ( " DATA_DIR " , " data " ) ) / " cookbook_state.json "
token = " "
if path . exists ( ) :
try :
state = json . loads ( path . read_text ( encoding = " utf-8 " ) )
env = state . get ( " env " ) if isinstance ( state , dict ) else { }
if isinstance ( env , dict ) and env . get ( " hfToken " ) :
from src . secret_storage import decrypt
token = decrypt ( env . get ( " hfToken " ) or " " )
except Exception :
token = " "
if not token :
token = ( os . environ . get ( " HF_TOKEN " ) or os . environ . get ( " HUGGING_FACE_HUB_TOKEN " ) or " " ) . strip ( )
return token
2026-05-31 23:58:26 +09:00
def _validate_local_dir ( v : str | None ) - > str | None :
if v is None or v == " " :
return None
2026-06-09 07:58:38 +02:00
if len ( v ) > = 2 and v [ 0 ] == v [ - 1 ] and v [ 0 ] in { " ' " , ' " ' } :
v = v [ 1 : - 1 ]
2026-05-31 23:58:26 +09:00
v = v . rstrip ( " / " ) or " / "
2026-06-09 07:58:38 +02:00
if not ( _LOCAL_DIR_RE . match ( v ) or _WINDOWS_LOCAL_DIR_RE . match ( v ) ) :
raise HTTPException ( 400 , " Invalid local_dir — must be an absolute or ~ path with no shell metacharacters " )
# Reject path segments that start with '-' (option injection). '-' is in the
# allowlist, so a dir like ``/models/-rf`` or ``D:\models\-rf`` could be read
# as a CLI flag by hf/etc. — and quoting does NOT stop a value from being
# parsed as an option. This is the one residual that command-build-time
# quoting can't cover, so the guard lives here, keeping the safety wholly
# inside the validator rather than relying on consumers.
if any ( seg . startswith ( " - " ) for seg in re . split ( r " [ \\ /] " , v ) if seg ) :
raise HTTPException ( 400 , " Invalid local_dir — path segments cannot start with ' - ' " )
2026-05-31 23:58:26 +09:00
return v
def _validate_gpus ( v : str | None ) - > str | None :
if v is None or v == " " :
return None
if not _GPU_LIST_RE . fullmatch ( str ( v ) ) :
raise HTTPException ( 400 , " Invalid gpus — expected comma-separated GPU indexes " )
return str ( v )
def _shell_path ( p : str ) - > str :
""" Render a validated path for a double-quoted shell context, expanding a
leading ~ to $ HOME ( single quotes wouldn ' t expand it). Safe because
2026-06-09 07:58:38 +02:00
_validate_local_dir already rejects quotes and shell metacharacters . """
2026-05-31 23:58:26 +09:00
if p == " ~ " :
return ' " $HOME " '
if p . startswith ( " ~/ " ) :
return ' " $HOME/ ' + p [ 2 : ] + ' " '
return ' " ' + p + ' " '
Add macOS Apple Silicon Cookbook support
* Add Apple Silicon (Metal) GPU detection and unified-memory fit tuning
hardware.py detects Apple Silicon locally and over SSH, reporting
backend=metal, the chip name, and a RAM-scaled fraction of unified
memory as the usable GPU budget. fit.py gains an M1-M4 memory-bandwidth
table for realistic tok/s and drops vLLM-only formats (AWQ/GPTQ/FP8)
that can't be served on Metal.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 32ac81dbc680361463a088dae867d555d5a79c3b)
* Generate macOS/Metal serve commands and surface the Metal GPU
cookbook_routes.py adds a macOS serve path (Ollama, Metal-aware
llama.cpp build using `sysctl hw.ncpu` instead of `nproc`, and a clear
error if vLLM is attempted). The frontend defaults Metal serving to
llama.cpp and offers llama.cpp/Ollama instead of vLLM/SGLang. The
odysseus-cookbook CLI's `gpus` command reports the Metal GPU via
sysctl/vm_stat.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 4ba01ce25d256ae032029898f361c824a34fcd4b)
* Add launchd LaunchAgent for macOS (systemd equivalent)
com.odysseus.ui.plist + install-service-macos.sh run Odysseus at login
and restart on crash, the macOS counterpart to odysseus-ui.service. The
installer auto-fills paths from the venv, so there's no hand-editing.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 3d4b6b2c7b8b31af32201ed278115df9a559dea9)
* Document macOS install (brew, Ollama, AirPlay port, launchd)
README + setup.py cover the Homebrew / Apple Silicon path: brew install
python@3.11 tmux ollama, Metal serving via Ollama/llama.cpp, the launchd
service, and the macOS AirPlay Receiver conflict on ports 7000/5000.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 8dc9a3578a1726f070ed9f75c0958ae291a6d966)
* Add downloadable macOS launcher app builder
build-macos-app.sh generates dist/Odysseus.app and a drag-to-Applications
dist/Odysseus.dmg. The app starts the local server from this repo's venv and
opens the UI in a chrome-less app window (Chromium --app mode, falling back to
the default browser). It's a launcher wrapper — it drives the venv rather than
bundling Python — so the install path is baked in at build time.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 7927940c3810ee34640803b198d334a6ac93474d)
* Harden macOS Cookbook support: hide MLX, fix Metal build cache
Builds on the adopted PR #213 macOS/Metal work with two fixes and tests:
- fit.py: always drop MLX-quantized models. Odysseus only generates serve
commands for llama.cpp/Ollama (Metal) and vLLM/SGLang (CUDA); MLX needs the
mlx_lm runtime and the catalog's MLX repos ship no GGUF alternative, so they
were surfaced on Apple Silicon but could never be served.
- cookbook_routes.py (macOS branch only): `rm -rf build` before configure so a
poisoned CMakeCache from a prior failed CUDA attempt can't make every later
build fail; explicit -DCMAKE_BUILD_TYPE=Release; a clear "brew install cmake"
hint if cmake is missing. Linux/CUDA path unchanged.
- tests/test_hwfit_macos.py: MLX hidden on metal, MLX still hidden on CUDA
(regression guard), Metal detection on Apple Silicon, and skipped on
Linux/Intel (proves non-macOS detection is untouched).
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Propagate unified_memory flag and document macOS GPU/Docker caveat
- hardware.py: detect_system now carries the unified_memory flag from GPU
detection into the system dict (it was set by _detect_apple_silicon / AMD-APU
detection but dropped during result assembly, so the API always reported
null). Lets callers distinguish unified from discrete VRAM.
- README: prominent warning that Docker on Apple Silicon can't reach the Metal
GPU (runs a Linux VM) — Cookbook must run natively for GPU serving; fix stale
text that said Cookbook recommends MLX models (now hidden as unservable).
- test: detect_system propagates unified_memory.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Put Odysseus's venv bin on PATH for cookbook runners
Native (non-Docker) installs run from a virtualenv whose bin holds the `hf` CLI
and `python3` the cookbook download/serve tmux scripts shell out to. Those
scripts start in a fresh login shell with the venv NOT activated, so on a native
macOS install `hf download` failed with "hf: command not found" — and the
`pip --user` self-heal missed because macOS has no bare `pip` command.
- cookbook_helpers.py: _local_tooling_path_export() — pure helper returning a
PATH export for the running interpreter's bin dir (escaped for double quotes).
- cookbook_routes.py: download + serve runners prepend that dir on local runs
(gated off SSH/Windows); swap the `pip` install fallbacks to `python3 -m pip`.
- tests: helper output for normal and spaced paths.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Document macOS llama.cpp serving prerequisites
Clarify the two serving paths on Apple Silicon: the recommended zero-build
route (brew install llama.cpp ships a Metal llama-server Cookbook finds on PATH),
and the from-source fallback, which requires cmake + Xcode Command Line Tools.
Without those the build is skipped and serving silently degrades to a slow CPU
build, so new users now know to install them (or use the prebuilt) up front.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Recommend only GGUF-servable models on Metal
Apple Silicon's only serving engines are llama.cpp and Ollama, both GGUF-only
(vLLM/SGLang are CUDA/ROCm and don't run on macOS). The catalog tags raw
safetensors repos with a default Q4_K_M quant, so the fit-ranking was
recommending ~397/501 models that have no GGUF and fail to serve on Metal with
"No GGUF found" (e.g. microsoft/Phi-mini-MoE-instruct).
Drop any model without a real GGUF (is_gguf/gguf_sources) on Apple Silicon —
subsumes the previous AWQ/GPTQ/FP8 special-case into one rule. On CUDA these
stay visible since vLLM serves safetensors directly. Metal recommendations go
501 -> 104, all actually servable.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Remove macOS launchd LaunchAgent (cherry-picked extra)
Drop the launchd service from the PR #213 cherry-picks: the
install-service-macos.sh installer, the com.odysseus.ui.plist template, and the
README section documenting them. Tangential to the core Cookbook/Metal support
and not wanted. The build-macos-app.sh launcher is kept.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Add one-command macOS quick start (start-macos.sh)
Running Odysseus natively on a Mac previously meant ~7 manual terminal steps
(brew deps, venv, activate, pip, setup.py, uvicorn with the right port) — not
friendly for a generic macOS user, and the native run is required because Docker
on macOS can't reach the Metal GPU.
- start-macos.sh: installs Homebrew deps (python@3.11, tmux, prebuilt Metal
llama.cpp), creates the venv, installs requirements, runs setup, and launches
on a non-AirPlay port (7860). Idempotent; re-run to start again.
- README: the Apple Silicon section now leads with this one-command quick start
and the clickable .app, with engine/port/manual details folded into a
collapsible block. Added a pointer at the top of the manual-install section.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* macOS quick start: auto-open browser when ready
The "open this URL" line scrolled out of view as uvicorn kept logging after it,
so users missed it. Now start-macos.sh waits (in the background) until the
server accepts connections, prints a boxed "ready" banner at that point (i.e.
after the startup burst, not before), and opens the URL in the default browser
automatically. Skippable with ODYSSEUS_NO_OPEN=1 for headless/SSH use.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Don't assume/force a specific Python version on macOS
The README claimed "system Python is 3.9" — a machine-specific generalization
that's often wrong (macOS ships no recent Python by default; many users already
have 3.11+). Make it generic, and make start-macos.sh detect an existing
Python 3.11+ and use it, only installing python@3.11 when none is found instead
of forcing it on top of the user's Python.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Align start-macos.sh venv path with build-macos-app.sh
start-macos.sh created the environment in .venv/, but build-macos-app.sh and
the manual install steps use venv/ — so the clickable .app wouldn't reuse the
quick-start's environment and would rebuild a second one. Use venv/ everywhere.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* README: state clearly that MLX is unsupported on Apple Silicon
Odysseus has no mlx_lm runtime; it serves GGUF (llama.cpp/Ollama) and CUDA
(vLLM/SGLang) only. MLX-only models can't run on a Mac and are hidden from
Cookbook — make that explicit in both the quick start and the details.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* start-macos.sh: build the venv with an arm64 Python on Apple Silicon
A clean-room run surfaced this: with a universal2/x86 Python (e.g. the
python.org installer under /usr/local), the venv's compiled extensions install
as arm64 but get loaded as x86_64 when launched from the .app bundle, so it
crashes with "incompatible architecture (have arm64, need x86_64)". The terminal
run happened to work only because a universal binary defaults to arm64 there.
On Apple Silicon, look only under /opt/homebrew (arm64-only) for the build
Python, and install Homebrew's python@3.11 if none is present — so the venv is
arm64-only and launches correctly from both the terminal and the .app. Intel
and non-mac paths are unchanged. Verified end-to-end in a clean clone: .app now
boots on Metal with no arch error.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Address dev-exp review: macOS setup robustness + doc/UX fixes
From the voltagent dev-exp review of the branch:
- README: fix broken anchor links (the em-dash heading produced a slug the links
didn't match); simplify the heading to a stable slug.
- cookbook_routes.py: add /opt/homebrew/bin and /usr/local/bin to the serve PATH
so a brew-installed llama-server/ollama is found instead of falling back to a
slow source build.
- start-macos.sh: guard against an empty Python path; fail fast with a clear
message on port-in-use; ERR trap with a "safe to re-run" message; show pip
progress (drop --quiet on the slow requirements install); stop the background
browser-opener cleanly on exit/Ctrl+C (no orphaned poller).
- setup.py: bind hint to 127.0.0.1; suppress the manual run-hint when launched
by start-macos.sh (ODYSSEUS_SKIP_RUN_HINT) so the URL isn't contradictory.
- build-macos-app.sh: the .app only opens the browser once the server is
actually ready (not after the readiness timeout).
- cookbookServe.js: drop "Diffusers" from the Metal backend picker —
diffusion_server.py is CUDA-only, so it was an unservable option on macOS.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
---------
Co-authored-by: yunggilja <yunggilja@gmail.com>
Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-01 15:29:19 +09:30
def _local_tooling_path_export ( executable : str ) - > str :
""" Bash line prepending the running interpreter ' s bin dir to PATH.
When Odysseus runs from a virtualenv , that bin dir holds the tools the
cookbook runners shell out to ( ` hf ` , ` python ` ) . tmux runners start from a
fresh login shell with the venv NOT activated , so without this they can ' t
find ` hf ` and downloads fail with " hf: command not found " — notably on
macOS , where the ` pip - - user ` self - heal also misses ( ` pip ` isn ' t a command,
only ` pip3 ` / ` python3 - m pip ` ) . Local runs only ; meaningless over SSH .
"""
2026-06-01 21:08:39 +07:00
# This builds a bash snippet, so an explicit POSIX absolute path should keep
# POSIX semantics even when the app/tests run on Windows. Otherwise
# os.path.abspath("/opt/...") would incorrectly turn it into "D:\\opt\\...".
if executable . startswith ( " / " ) :
bin_dir = posixpath . dirname ( executable )
2026-06-04 09:00:01 +05:30
elif _WINDOWS_DRIVE_PATH_RE . match ( executable ) :
bin_dir = ntpath . dirname ( executable )
2026-06-01 21:08:39 +07:00
else :
bin_dir = os . path . dirname ( os . path . abspath ( executable ) )
2026-06-04 09:00:01 +05:30
bin_dir = _git_bash_path ( bin_dir )
Add macOS Apple Silicon Cookbook support
* Add Apple Silicon (Metal) GPU detection and unified-memory fit tuning
hardware.py detects Apple Silicon locally and over SSH, reporting
backend=metal, the chip name, and a RAM-scaled fraction of unified
memory as the usable GPU budget. fit.py gains an M1-M4 memory-bandwidth
table for realistic tok/s and drops vLLM-only formats (AWQ/GPTQ/FP8)
that can't be served on Metal.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 32ac81dbc680361463a088dae867d555d5a79c3b)
* Generate macOS/Metal serve commands and surface the Metal GPU
cookbook_routes.py adds a macOS serve path (Ollama, Metal-aware
llama.cpp build using `sysctl hw.ncpu` instead of `nproc`, and a clear
error if vLLM is attempted). The frontend defaults Metal serving to
llama.cpp and offers llama.cpp/Ollama instead of vLLM/SGLang. The
odysseus-cookbook CLI's `gpus` command reports the Metal GPU via
sysctl/vm_stat.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 4ba01ce25d256ae032029898f361c824a34fcd4b)
* Add launchd LaunchAgent for macOS (systemd equivalent)
com.odysseus.ui.plist + install-service-macos.sh run Odysseus at login
and restart on crash, the macOS counterpart to odysseus-ui.service. The
installer auto-fills paths from the venv, so there's no hand-editing.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 3d4b6b2c7b8b31af32201ed278115df9a559dea9)
* Document macOS install (brew, Ollama, AirPlay port, launchd)
README + setup.py cover the Homebrew / Apple Silicon path: brew install
python@3.11 tmux ollama, Metal serving via Ollama/llama.cpp, the launchd
service, and the macOS AirPlay Receiver conflict on ports 7000/5000.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 8dc9a3578a1726f070ed9f75c0958ae291a6d966)
* Add downloadable macOS launcher app builder
build-macos-app.sh generates dist/Odysseus.app and a drag-to-Applications
dist/Odysseus.dmg. The app starts the local server from this repo's venv and
opens the UI in a chrome-less app window (Chromium --app mode, falling back to
the default browser). It's a launcher wrapper — it drives the venv rather than
bundling Python — so the install path is baked in at build time.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 7927940c3810ee34640803b198d334a6ac93474d)
* Harden macOS Cookbook support: hide MLX, fix Metal build cache
Builds on the adopted PR #213 macOS/Metal work with two fixes and tests:
- fit.py: always drop MLX-quantized models. Odysseus only generates serve
commands for llama.cpp/Ollama (Metal) and vLLM/SGLang (CUDA); MLX needs the
mlx_lm runtime and the catalog's MLX repos ship no GGUF alternative, so they
were surfaced on Apple Silicon but could never be served.
- cookbook_routes.py (macOS branch only): `rm -rf build` before configure so a
poisoned CMakeCache from a prior failed CUDA attempt can't make every later
build fail; explicit -DCMAKE_BUILD_TYPE=Release; a clear "brew install cmake"
hint if cmake is missing. Linux/CUDA path unchanged.
- tests/test_hwfit_macos.py: MLX hidden on metal, MLX still hidden on CUDA
(regression guard), Metal detection on Apple Silicon, and skipped on
Linux/Intel (proves non-macOS detection is untouched).
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Propagate unified_memory flag and document macOS GPU/Docker caveat
- hardware.py: detect_system now carries the unified_memory flag from GPU
detection into the system dict (it was set by _detect_apple_silicon / AMD-APU
detection but dropped during result assembly, so the API always reported
null). Lets callers distinguish unified from discrete VRAM.
- README: prominent warning that Docker on Apple Silicon can't reach the Metal
GPU (runs a Linux VM) — Cookbook must run natively for GPU serving; fix stale
text that said Cookbook recommends MLX models (now hidden as unservable).
- test: detect_system propagates unified_memory.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Put Odysseus's venv bin on PATH for cookbook runners
Native (non-Docker) installs run from a virtualenv whose bin holds the `hf` CLI
and `python3` the cookbook download/serve tmux scripts shell out to. Those
scripts start in a fresh login shell with the venv NOT activated, so on a native
macOS install `hf download` failed with "hf: command not found" — and the
`pip --user` self-heal missed because macOS has no bare `pip` command.
- cookbook_helpers.py: _local_tooling_path_export() — pure helper returning a
PATH export for the running interpreter's bin dir (escaped for double quotes).
- cookbook_routes.py: download + serve runners prepend that dir on local runs
(gated off SSH/Windows); swap the `pip` install fallbacks to `python3 -m pip`.
- tests: helper output for normal and spaced paths.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Document macOS llama.cpp serving prerequisites
Clarify the two serving paths on Apple Silicon: the recommended zero-build
route (brew install llama.cpp ships a Metal llama-server Cookbook finds on PATH),
and the from-source fallback, which requires cmake + Xcode Command Line Tools.
Without those the build is skipped and serving silently degrades to a slow CPU
build, so new users now know to install them (or use the prebuilt) up front.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Recommend only GGUF-servable models on Metal
Apple Silicon's only serving engines are llama.cpp and Ollama, both GGUF-only
(vLLM/SGLang are CUDA/ROCm and don't run on macOS). The catalog tags raw
safetensors repos with a default Q4_K_M quant, so the fit-ranking was
recommending ~397/501 models that have no GGUF and fail to serve on Metal with
"No GGUF found" (e.g. microsoft/Phi-mini-MoE-instruct).
Drop any model without a real GGUF (is_gguf/gguf_sources) on Apple Silicon —
subsumes the previous AWQ/GPTQ/FP8 special-case into one rule. On CUDA these
stay visible since vLLM serves safetensors directly. Metal recommendations go
501 -> 104, all actually servable.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Remove macOS launchd LaunchAgent (cherry-picked extra)
Drop the launchd service from the PR #213 cherry-picks: the
install-service-macos.sh installer, the com.odysseus.ui.plist template, and the
README section documenting them. Tangential to the core Cookbook/Metal support
and not wanted. The build-macos-app.sh launcher is kept.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Add one-command macOS quick start (start-macos.sh)
Running Odysseus natively on a Mac previously meant ~7 manual terminal steps
(brew deps, venv, activate, pip, setup.py, uvicorn with the right port) — not
friendly for a generic macOS user, and the native run is required because Docker
on macOS can't reach the Metal GPU.
- start-macos.sh: installs Homebrew deps (python@3.11, tmux, prebuilt Metal
llama.cpp), creates the venv, installs requirements, runs setup, and launches
on a non-AirPlay port (7860). Idempotent; re-run to start again.
- README: the Apple Silicon section now leads with this one-command quick start
and the clickable .app, with engine/port/manual details folded into a
collapsible block. Added a pointer at the top of the manual-install section.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* macOS quick start: auto-open browser when ready
The "open this URL" line scrolled out of view as uvicorn kept logging after it,
so users missed it. Now start-macos.sh waits (in the background) until the
server accepts connections, prints a boxed "ready" banner at that point (i.e.
after the startup burst, not before), and opens the URL in the default browser
automatically. Skippable with ODYSSEUS_NO_OPEN=1 for headless/SSH use.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Don't assume/force a specific Python version on macOS
The README claimed "system Python is 3.9" — a machine-specific generalization
that's often wrong (macOS ships no recent Python by default; many users already
have 3.11+). Make it generic, and make start-macos.sh detect an existing
Python 3.11+ and use it, only installing python@3.11 when none is found instead
of forcing it on top of the user's Python.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Align start-macos.sh venv path with build-macos-app.sh
start-macos.sh created the environment in .venv/, but build-macos-app.sh and
the manual install steps use venv/ — so the clickable .app wouldn't reuse the
quick-start's environment and would rebuild a second one. Use venv/ everywhere.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* README: state clearly that MLX is unsupported on Apple Silicon
Odysseus has no mlx_lm runtime; it serves GGUF (llama.cpp/Ollama) and CUDA
(vLLM/SGLang) only. MLX-only models can't run on a Mac and are hidden from
Cookbook — make that explicit in both the quick start and the details.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* start-macos.sh: build the venv with an arm64 Python on Apple Silicon
A clean-room run surfaced this: with a universal2/x86 Python (e.g. the
python.org installer under /usr/local), the venv's compiled extensions install
as arm64 but get loaded as x86_64 when launched from the .app bundle, so it
crashes with "incompatible architecture (have arm64, need x86_64)". The terminal
run happened to work only because a universal binary defaults to arm64 there.
On Apple Silicon, look only under /opt/homebrew (arm64-only) for the build
Python, and install Homebrew's python@3.11 if none is present — so the venv is
arm64-only and launches correctly from both the terminal and the .app. Intel
and non-mac paths are unchanged. Verified end-to-end in a clean clone: .app now
boots on Metal with no arch error.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Address dev-exp review: macOS setup robustness + doc/UX fixes
From the voltagent dev-exp review of the branch:
- README: fix broken anchor links (the em-dash heading produced a slug the links
didn't match); simplify the heading to a stable slug.
- cookbook_routes.py: add /opt/homebrew/bin and /usr/local/bin to the serve PATH
so a brew-installed llama-server/ollama is found instead of falling back to a
slow source build.
- start-macos.sh: guard against an empty Python path; fail fast with a clear
message on port-in-use; ERR trap with a "safe to re-run" message; show pip
progress (drop --quiet on the slow requirements install); stop the background
browser-opener cleanly on exit/Ctrl+C (no orphaned poller).
- setup.py: bind hint to 127.0.0.1; suppress the manual run-hint when launched
by start-macos.sh (ODYSSEUS_SKIP_RUN_HINT) so the URL isn't contradictory.
- build-macos-app.sh: the .app only opens the browser once the server is
actually ready (not after the readiness timeout).
- cookbookServe.js: drop "Diffusers" from the Metal backend picker —
diffusion_server.py is CUDA-only, so it was an unservable option on macOS.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
---------
Co-authored-by: yunggilja <yunggilja@gmail.com>
Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-01 15:29:19 +09:30
# Escape for a double-quoted context: $PATH must still expand, but spaces
# and shell metacharacters in the path must be preserved literally.
esc = (
bin_dir . replace ( " \\ " , " \\ \\ " )
. replace ( ' " ' , ' \\ " ' )
. replace ( " $ " , " \\ $ " )
. replace ( " ` " , " \\ ` " )
)
return f ' export PATH= " { esc } :$PATH " '
2026-06-03 13:23:49 +08:00
def _pip_install_no_cache ( cmd : str ) - > str :
""" Add ``--no-cache-dir`` to a pip install command.
Cookbook dependency installs ( vLLM , llama - cpp - python , … ) build large wheels ;
pip ' s default cache lives under ``$HOME/.cache/pip`` and these builds can fill
a small home filesystem with ` ` [ Errno 28 ] No space left on device ` ` mid - build
( issue #1219), leaving the dependency "installed" but unusable (#1459).
Disabling the cache for these one - off installs keeps them off the home disk
( the maintainer ' s suggested ``PIP_CACHE_DIR=`` workaround, made the default).
Idempotent ; leaves non - pip - install commands untouched . """
if not cmd or " pip install " not in cmd or " --no-cache-dir " in cmd :
return cmd
return cmd . replace ( " pip install " , " pip install --no-cache-dir " , 1 )
2026-06-02 12:34:52 +01:00
def _pip_install_attempt ( pip_cmd : str ) - > str :
""" Wrap a single pip install command so its exit status survives the
fallback chain and its stderr is visible in the tmux log on failure .
Without this wrapper , ` pip … 2 > & 1 | tail - 5 ` returns ` ` tail ` ` ' s exit
code ( 0 ) , masking pip ' s real failure and preventing the next fallback
from running . The generated snippet captures all output to a temp
file , prints the last 5 lines on failure ( so the Cookbook log panel
shows useful diagnostics ) , cleans up , and exits with pip ' s original
status .
"""
return (
" bash -c ' "
f ' _out=$(mktemp) && { pip_cmd } > " $_out " 2>&1; _rc=$?; '
' tail -5 " $_out " ; rm -f " $_out " ; exit $_rc '
" ' "
)
2026-06-09 00:10:20 +03:00
def _pip_command ( python_cmd : str ) - > str :
""" Return a pip command for either a pip executable or a Python executable. """
cmd = python_cmd . strip ( )
if " -m pip " in cmd or cmd in { " pip " , " pip3 " } :
return python_cmd
if cmd in { " python " , " python3 " , " python.exe " } or cmd . endswith ( ( " /python " , " /python3 " , " \\ python.exe " ) ) :
return f " { python_cmd } -m pip "
return python_cmd
def _pip_break_system_packages_check ( pip_cmd : str ) - > str :
return f " { pip_cmd } install --help 2>/dev/null | grep -q -- --break-system-packages "
2026-06-02 08:01:59 +05:30
def _pip_install_fallback_chain ( package : str , * , python_cmd : str = " python3 -m pip " , upgrade : bool = False ) - > str :
2026-06-02 12:34:52 +01:00
""" Build a bash pip install fallback chain that surfaces errors.
Try the active interpreter / environment first . ` ` - - user ` ` is invalid
inside many venvs , so only attempt the ` ` - - user ` ` fallback when NOT
inside a venv .
2026-06-02 08:01:59 +05:30
2026-06-02 12:34:52 +01:00
Each attempt is wrapped via : func : ` _pip_install_attempt ` so pip ' s real
exit code is preserved ( no ` ` | tail ` ` masking ) and the last 5 lines of
pip output appear in the Cookbook log on failure .
2026-06-02 08:01:59 +05:30
"""
2026-06-05 14:41:07 +02:00
from core . platform_compat import IS_WINDOWS
2026-06-02 08:01:59 +05:30
upgrade_flag = " -U " if upgrade else " "
2026-06-03 01:24:26 -04:00
# Shell-quote the package spec: an extras spec like ``llama-cpp-python[server]``
# contains brackets that bash would treat as a glob, so it must be quoted
# before being embedded in the install command. Plain names (e.g.
# ``huggingface_hub``) are returned unchanged by ``shlex.quote``.
pkg = shlex . quote ( package )
2026-06-08 00:33:50 +02:00
# llama-cpp-python source builds are brittle on older distro pip/packaging
# stacks (common on WSL images). Prefer the prebuilt wheel index whenever
# this package is requested so dependency-install tasks are reliable.
if " llama-cpp-python " in package :
2026-06-05 14:41:07 +02:00
pkg + = " --extra-index-url https://abetlen.github.io/llama-cpp-python/whl/cpu "
2026-06-09 00:10:20 +03:00
pip_cmd = _pip_command ( python_cmd )
base = _pip_install_attempt ( f " { pip_cmd } install -q { upgrade_flag } { pkg } " )
user = _pip_install_attempt ( f " { pip_cmd } install --user -q { upgrade_flag } { pkg } " )
user_break_system = _pip_install_attempt ( f " { pip_cmd } install --user --break-system-packages -q { upgrade_flag } { pkg } " )
user_fallback = f " ( { user } || {{ { _pip_break_system_packages_check ( pip_cmd ) } && { user_break_system } ; }} ) "
2026-06-02 10:23:20 +07:00
# Derive the python executable for the venv detection check.
# Must use the same interpreter that pip belongs to; hardcoding
# python3 breaks when pip lives in a venv that only has "python".
2026-06-09 00:10:20 +03:00
if " -m pip " in pip_cmd :
python_exe = pip_cmd . replace ( " -m pip " , " " )
elif pip_cmd . strip ( ) == " pip " :
2026-06-02 10:23:20 +07:00
python_exe = " python "
2026-06-09 00:10:20 +03:00
elif pip_cmd . strip ( ) == " pip3 " :
2026-06-02 10:23:20 +07:00
python_exe = " python3 "
else :
python_exe = " python3 "
venv_check = f ' { python_exe } -c " import sys; sys.exit(0 if sys.prefix != sys.base_prefix else 1) " '
2026-06-09 00:10:20 +03:00
# Negated: `! venv_check` succeeds (exit 0) when NOT in a venv -> `&&` tries
# --user. When IN a venv `! venv_check` fails -> `&&` skips --user and the
2026-06-02 12:34:52 +01:00
# group exits non-zero, propagating the base-install failure instead of
# masking it as success (the `|| { venv_check || … }` shape from #903
# swallowed the exit code because venv_check's exit-0 became the group's
2026-06-09 00:10:20 +03:00
# result). `--break-system-packages` is only attempted when the active pip
# supports it; older pip versions abort with "no such option" otherwise.
return f " { base } || {{ ! { venv_check } && { user_fallback } ; }} "
2026-06-02 08:01:59 +05:30
2026-06-02 16:39:02 +03:00
def _venv_safe_local_pip_install_cmd ( cmd : str , * , local : bool , in_venv : bool ) - > str :
""" Drop pip user-install flags that are invalid for local venv installs.
Cookbook dependency installs run through the model - serve task path so users
can watch progress in the same log UI . For local POSIX runs , that task
prepends Odysseus ' own interpreter directory to PATH. If Odysseus itself is
running from a venv , ` python3 ` resolves to the venv Python and pip rejects
` - - user ` with " User site-packages are not visible in this virtualenv " .
Keep remote and non - venv installs unchanged : remotes may intentionally use
system Python , and Docker / non - venv installs still need user - site fallback .
"""
if not local or not in_venv :
return cmd
if " pip install " not in ( cmd or " " ) :
return cmd
try :
parts = shlex . split ( cmd )
except ValueError :
return cmd
stripped = [
part
for part in parts
if part not in { " --user " , " --break-system-packages " }
]
return shlex . join ( stripped )
2026-06-09 00:10:20 +03:00
def _pip_install_command_without_break_system_packages ( cmd : str ) - > str :
try :
parts = shlex . split ( cmd )
except ValueError :
return cmd
stripped = [ part for part in parts if part != " --break-system-packages " ]
return shlex . join ( stripped )
def _pip_install_help_check_from_cmd ( cmd : str ) - > str | None :
try :
parts = shlex . split ( cmd )
except ValueError :
return None
try :
install_index = parts . index ( " install " )
except ValueError :
return None
if install_index < = 0 :
return None
pip_prefix = parts [ : install_index ]
return f " { shlex . join ( pip_prefix + [ ' install ' , ' --help ' ] ) } 2>/dev/null | grep -q -- --break-system-packages "
def _append_pip_install_runner_lines ( runner_lines : list [ str ] , cmd : str ) - > None :
""" Append a pip install command, guarding --break-system-packages support.
The Dependencies UI may submit ` ` python3 - m pip install - - user
- - break - system - packages . . . ` ` for non - venv installs . That flag is useful on
PEP - 668 - locked distros , but older pip ( including Ubuntu 22.04 ' s apt pip in
the NVIDIA CUDA base image ) aborts with " no such option " . Branch at runner
time so stale browser JS and remote targets are handled by the server too .
"""
if " --break-system-packages " not in ( cmd or " " ) :
runner_lines . append ( cmd )
return
help_check = _pip_install_help_check_from_cmd ( cmd )
without_break = _pip_install_command_without_break_system_packages ( cmd )
if not help_check or without_break == cmd :
runner_lines . append ( cmd )
return
runner_lines . append ( f " if { help_check } ; then " )
runner_lines . append ( f " { cmd } " )
runner_lines . append ( " else " )
runner_lines . append ( ' echo " [odysseus] pip does not support --break-system-packages; installing without it. " ' )
runner_lines . append ( f " { without_break } " )
runner_lines . append ( " fi " )
2026-06-04 09:00:01 +05:30
def _user_shell_path_bootstrap ( ) - > list [ str ] :
return [
' ODYSSEUS_USER_SHELL= " $ {SHELL:-} " ' ,
' if [ -n " $ODYSSEUS_USER_SHELL " ] && [ -x " $ODYSSEUS_USER_SHELL " ]; then ' ,
' ODYSSEUS_USER_PATH= " $( " $ODYSSEUS_USER_SHELL " -ic \' printf " __ODYSSEUS_PATH__ %s \\ n " " $PATH " \' 2>/dev/null | sed -n \' s/^__ODYSSEUS_PATH__//p \' | tail -n 1 || true) " ' ,
' if [ -n " $ODYSSEUS_USER_PATH " ]; then export PATH= " $ODYSSEUS_USER_PATH:$PATH " ; fi ' ,
' fi ' ,
2026-06-15 07:25:30 -04:00
# Windows can expose python3 as a Microsoft Store App Execution Alias
# under WindowsApps. Git Bash sees that stub as present, but it exits
# before running Python. A Windows venv usually has python.exe, not
# python3.exe, so treat a missing or WindowsApps python3 as absent.
' _odys_py3= " $(command -v python3 2>/dev/null || true) " ' ,
' case " $_odys_py3 " in " " |*[Ww]indows[Aa]pps*) python3() { python " $@ " ; } ;; esac ' ,
2026-06-08 00:33:50 +02:00
' command -v python >/dev/null 2>&1 || python() { python3 " $@ " ; } ' ,
2026-06-04 09:00:01 +05:30
]
2026-06-08 00:33:50 +02:00
def _cached_model_scan_script ( model_dirs : list [ str ] | None = None , add_hf_cache : str | None = None ) - > str :
""" Build the standalone Python scanner used by /api/model/cached.
Allows for an additional HuggingFace cache path to be scanned ( i . e . Windows HF cache for local WSL envs . )
"""
2026-06-01 22:46:54 +09:00
lines = [
2026-06-02 07:14:59 +09:00
" import json, os, re, shutil, subprocess, urllib.request " ,
2026-06-01 22:46:54 +09:00
" models = [] " ,
" seen = set() " ,
" BLOCKED_ROOTS = ( ' /sys ' , ' /proc ' , ' /dev ' , ' /run ' , ' /var/run ' ) " ,
" def safe_path(p): " ,
" try: " ,
" rp = os.path.realpath(os.path.expanduser(p)) " ,
" return not any(rp == b or rp.startswith(b + os.sep) for b in BLOCKED_ROOTS) " ,
" except Exception: " ,
" return False " ,
" def safe_walk(top): " ,
" if not safe_path(top): return " ,
" for root, dirs, fns in os.walk(top, followlinks=False): " ,
" dirs[:] = [d for d in dirs if not os.path.islink(os.path.join(root, d)) and safe_path(os.path.join(root, d))] " ,
" yield root, dirs, fns " ,
2026-06-02 13:32:40 +10:00
" def gguf_role(name): " ,
" n = name.lower() " ,
" if n.startswith( ' mmproj ' ) or ' mmproj ' in n: return ' projector ' " ,
" return ' model ' " ,
" def gguf_quant(name): " ,
" m = re.search(r ' (?i)(UD-)?(IQ[0-9]_[A-Z0-9_]+|Q[0-9](?:_[A-Z0-9]+)+|BF16|F16|FP16|F32|Q8_0) ' , name) " ,
" return m.group(0).upper() if m else ' ' " ,
" def collect_ggufs(base): " ,
" files = [] " ,
" split_groups = {} " ,
" if not os.path.isdir(base) or not safe_path(base): return files " ,
" for root, dirs, fns in safe_walk(base): " ,
" for fn in sorted(fns): " ,
" if not fn.lower().endswith( ' .gguf ' ): continue " ,
2026-06-09 07:58:38 +02:00
" if fn.startswith( ' ._ ' ): continue # macOS AppleDouble sidecar, not a real GGUF " ,
2026-06-02 13:32:40 +10:00
" fp = os.path.join(root, fn) " ,
" try: size = os.path.getsize(fp) " ,
" except Exception: size = 0 " ,
" try: rel = os.path.relpath(fp, base).replace(os.sep, ' / ' ) " ,
" except Exception: rel = fn " ,
" sm = re.match(r ' (?i)^(.+)-( \\ d+)-of-( \\ d+) \\ .gguf$ ' , fn) " ,
" if sm: " ,
" prefix, part_s, total_s = sm.group(1), sm.group(2), sm.group(3) " ,
" key = (root, prefix, total_s) " ,
" g = split_groups.setdefault(key, { ' name ' :fn, ' rel_path ' :rel, ' size_bytes ' :0, ' role ' :gguf_role(fn), ' quant ' :gguf_quant(fn), ' parts ' :int(total_s), ' split ' :True}) " ,
" g[ ' size_bytes ' ] += size " ,
" if int(part_s) == 1: " ,
" g.update( { ' name ' :fn, ' rel_path ' :rel, ' role ' :gguf_role(fn), ' quant ' :gguf_quant(fn)}) " ,
" continue " ,
" files.append( { ' name ' :fn, ' rel_path ' :rel, ' size_bytes ' :size, ' role ' :gguf_role(fn), ' quant ' :gguf_quant(fn)}) " ,
" files.extend(split_groups.values()) " ,
" files.sort(key=lambda f: (f.get( ' role ' ) != ' model ' , f.get( ' rel_path ' , ' ' ))) " ,
" return files " ,
2026-06-01 22:46:54 +09:00
" def scan_hf(cache): " ,
" if not os.path.isdir(cache): return " ,
" for d in sorted(os.listdir(cache)): " ,
" if not d.startswith( ' models-- ' ): continue " ,
" rid = d.replace( ' models-- ' , ' ' ).replace( ' -- ' , ' / ' ) " ,
" if rid in seen: continue " ,
" seen.add(rid) " ,
" blobs = os.path.join(cache, d, ' blobs ' ) " ,
" sz, nf, ic = 0, 0, False " ,
" if os.path.isdir(blobs): " ,
" for f in os.scandir(blobs): " ,
" if f.is_file(): nf += 1; sz += f.stat().st_size " ,
" if f.name.endswith( ' .incomplete ' ): ic = True " ,
" snap = os.path.join(cache, d, ' snapshots ' ) " ,
2026-06-05 13:53:33 +01:00
" # Windows HF cache stores files directly in snapshots/; blobs/ may be empty. " ,
" # Fallback: scan snapshots for real files when blobs yielded nothing. " ,
" if sz == 0 and os.path.isdir(snap): " ,
" for sd in os.listdir(snap): " ,
" sf = os.path.join(snap, sd) " ,
" if not os.path.isdir(sf): continue " ,
" for f in os.scandir(sf): " ,
" if f.is_file(): nf += 1; sz += f.stat().st_size " ,
" if f.name.endswith( ' .incomplete ' ): ic = True " ,
2026-06-02 13:32:40 +10:00
" is_diffusion = False; gguf_files = [] " ,
2026-06-01 22:46:54 +09:00
" if os.path.isdir(snap): " ,
" for sd in os.listdir(snap): " ,
" sf = os.path.join(snap, sd) " ,
" if not os.path.isdir(sf): continue " ,
" if os.path.exists(os.path.join(sf, ' model_index.json ' )): is_diffusion = True " ,
2026-06-02 13:32:40 +10:00
" for f in collect_ggufs(sf): f[ ' rel_path ' ] = sd + ' / ' + f[ ' rel_path ' ]; gguf_files.append(f) " ,
" models.append( { ' repo_id ' :rid, ' size_bytes ' :sz, ' nb_files ' :nf, ' has_incomplete ' :ic, ' path ' :cache, ' is_diffusion ' :is_diffusion, ' is_gguf ' :bool(gguf_files), ' gguf_files ' :gguf_files}) " ,
2026-06-07 12:19:47 -04:00
" def hf_cache_paths(): " ,
" candidates = [] " ,
" def add(p): " ,
" if not p: return " ,
" p = os.path.expanduser(p) " ,
" if p not in candidates: candidates.append(p) " ,
" add(os.environ.get( ' HUGGINGFACE_HUB_CACHE ' )) " ,
" hf_home = os.environ.get( ' HF_HOME ' ) " ,
" if hf_home: add(os.path.join(hf_home, ' hub ' )) " ,
" add( ' ~/.cache/huggingface/hub ' ) " ,
" # Docker images mount ./data/huggingface at /app/.cache/huggingface. " ,
" # When HOME is /root, expanduser() misses that persisted cache. " ,
" add( ' /app/.cache/huggingface/hub ' ) " ,
2026-06-08 00:33:50 +02:00
f " add( { add_hf_cache !r} ) " if add_hf_cache else " " ,
2026-06-07 12:19:47 -04:00
" return candidates " ,
2026-06-01 22:46:54 +09:00
" def scan_dir(p): " ,
" if not os.path.isdir(p) or not safe_path(p): return " ,
" for d in sorted(os.listdir(p)): " ,
" if d.startswith( ' . ' ): continue " ,
" if d.startswith( ' models-- ' ): continue " ,
" fp = os.path.join(p, d) " ,
" if not os.path.isdir(fp) or os.path.islink(fp) or not safe_path(fp): continue " ,
" if d in seen: continue " ,
2026-06-02 13:32:40 +10:00
" is_model = False; gguf_files = [] " ,
2026-06-01 22:46:54 +09:00
" for root, dirs, fns in safe_walk(fp): " ,
" for fn in fns: " ,
2026-06-02 13:32:40 +10:00
" if fn.lower().endswith( ' .gguf ' ): is_model = True " ,
2026-06-01 22:46:54 +09:00
" elif fn == ' config.json ' or fn.endswith( ' .safetensors ' ) or fn.endswith( ' .bin ' ): is_model = True " ,
" if is_model: break " ,
" if not is_model: continue " ,
2026-06-02 13:32:40 +10:00
" gguf_files = collect_ggufs(fp) " ,
2026-06-01 22:46:54 +09:00
" seen.add(d) " ,
" sz, nf = 0, 0 " ,
" for dp, _, fns in safe_walk(fp): " ,
" for fn in fns: " ,
" try: nf += 1; sz += os.path.getsize(os.path.join(dp, fn)) " ,
" except Exception: pass " ,
" is_diff = os.path.exists(os.path.join(fp, ' model_index.json ' )) " ,
2026-06-02 13:32:40 +10:00
" models.append( { ' repo_id ' :d, ' size_bytes ' :sz, ' nb_files ' :nf, ' has_incomplete ' :False, ' path ' :p, ' is_local_dir ' :True, ' is_diffusion ' :is_diff, ' is_gguf ' :bool(gguf_files), ' gguf_files ' :gguf_files}) " ,
2026-06-02 07:14:59 +09:00
" def parse_size(num, unit): " ,
" try: n = float(num) " ,
" except Exception: return 0 " ,
" u = (unit or ' ' ).upper() " ,
" if u.startswith( ' TB ' ): return int(n * 1024 ** 4) " ,
" if u.startswith( ' GB ' ): return int(n * 1024 ** 3) " ,
" if u.startswith( ' MB ' ): return int(n * 1024 ** 2) " ,
" if u.startswith( ' KB ' ): return int(n * 1024) " ,
" return int(n) " ,
" def scan_ollama(): " ,
" if not shutil.which( ' ollama ' ): return " ,
" try: " ,
" p = subprocess.run([ ' ollama ' , ' list ' ], stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, text=True, timeout=6) " ,
" except Exception: " ,
" return " ,
" if p.returncode != 0: return " ,
" for line in (p.stdout or ' ' ).splitlines()[1:]: " ,
" parts = line.split() " ,
" if len(parts) < 4: continue " ,
" name = parts[0] " ,
" if not name or name in seen: continue " ,
" size_bytes = parse_size(parts[2], parts[3]) " ,
" seen.add(name) " ,
" models.append( { ' repo_id ' :name, ' size_bytes ' :size_bytes, ' nb_files ' :1, ' has_incomplete ' :False, ' path ' : ' ollama ' , ' backend ' : ' ollama ' , ' is_ollama ' :True}) " ,
" def scan_ollama_api(): " ,
" urls = [ ' http://127.0.0.1:11434/api/tags ' , ' http://localhost:11434/api/tags ' , ' http://host.docker.internal:11434/api/tags ' ] " ,
" for url in urls: " ,
" try: " ,
" with urllib.request.urlopen(url, timeout=2) as r: " ,
" data = json.loads(r.read().decode( ' utf-8 ' , ' replace ' )) " ,
" except Exception: " ,
" continue " ,
" for item in data.get( ' models ' , []): " ,
" name = item.get( ' name ' ) or item.get( ' model ' ) " ,
" if not name or name in seen: continue " ,
" size_bytes = int(item.get( ' size ' ) or item.get( ' size_bytes ' ) or 0) " ,
" seen.add(name) " ,
" models.append( { ' repo_id ' :name, ' size_bytes ' :size_bytes, ' nb_files ' :1, ' has_incomplete ' :False, ' path ' : ' ollama ' , ' backend ' : ' ollama ' , ' is_ollama ' :True}) " ,
" return " ,
2026-06-07 12:19:47 -04:00
" for _hf_cache in hf_cache_paths(): scan_hf(_hf_cache) " ,
2026-06-02 07:14:59 +09:00
" scan_ollama() " ,
" scan_ollama_api() " ,
2026-06-01 22:46:54 +09:00
]
for model_dir in model_dirs or [ ] :
lines . append ( f " scan_dir(os.path.expanduser( { model_dir !r} )) " )
lines . append ( " print(json.dumps(models)) " )
return " \n " . join ( lines ) + " \n "
2026-05-31 23:58:26 +09:00
def _ps_squote ( v : str ) - > str :
""" Escape a value for PowerShell single-quoted string interpolation.
Belt - and - suspenders on top of _validate_token ' s regex — if the regex
is ever loosened , this still keeps the heredoc shell - safe . """
return v . replace ( " ' " , " ' ' " )
def _bash_squote ( v : str ) - > str :
""" Escape a value for bash/sh single-quoted string interpolation. """
return v . replace ( " ' " , " ' \\ ' ' " )
# Allow-list of binaries permitted as the leading token of `req.cmd` for /api/model/serve.
# Anything else is rejected before the cmd is interpolated into a tmux/PowerShell wrapper.
_SERVE_CMD_ALLOWLIST = {
" vllm " , " llama-server " , " llama_server " , " llama.cpp " , " ollama " ,
" python " , " python3 " ,
" sglang " , " lmdeploy " ,
" node " , " npx " ,
}
# The llama.cpp GGUF launcher (static/js/cookbook.js) emits a fixed-shape
# prelude that resolves the cached .gguf on the target host before serving:
# MODEL_FILE=$( { find …; find …; } | head -1 ) && { [ -n "$MODEL_FILE" ] && \
# [ -f "$MODEL_FILE" ]; } || { echo "ERROR…"; exit 1; } && <serve> || <serve>
# That legitimately needs $(...)/&&/||, so we recognise this exact shape and
# validate the serve binaries it guards rather than rejecting it wholesale.
_GGUF_PRELUDE_RE = re . compile (
r ' ^MODEL_FILE= \ $ \ ([^ \ n]*? \ ) \ s*&& \ s* \ { [^ {} ]* \ } \ s* \ | \ | \ s* \ { [^ {} ]* \ } \ s*&& \ s* '
)
2026-06-01 21:27:04 -06:00
_OLLAMA_HOST_ASSIGNMENT_RE = re . compile ( r " (?:^| \ s)OLLAMA_HOST=([^ \ s]+) " )
_OLLAMA_BIND_RE = re . compile ( r " ^ \ [([^ \ ]]+) \ ]:( \ d+)$|^([^:]+):( \ d+)$ " )
_OLLAMA_BIND_HOST_RE = re . compile ( r " ^[A-Za-z0-9._:-]+$ " )
2026-06-15 02:12:18 -04:00
_LLAMA_CPP_PYTHON_GGML_TYPES = {
" f32 " : " 0 " ,
" f16 " : " 1 " ,
" q4_0 " : " 2 " ,
" q4_1 " : " 3 " ,
" q5_0 " : " 6 " ,
" q5_1 " : " 7 " ,
" q8_0 " : " 8 " ,
" q8_1 " : " 9 " ,
" q2_k " : " 10 " ,
" q3_k " : " 11 " ,
" q4_k " : " 12 " ,
" q5_k " : " 13 " ,
" q6_k " : " 14 " ,
" q8_k " : " 15 " ,
" iq2_xxs " : " 16 " ,
" iq2_xs " : " 17 " ,
" iq3_xxs " : " 18 " ,
" iq1_s " : " 19 " ,
" iq4_nl " : " 20 " ,
" iq3_s " : " 21 " ,
" iq2_s " : " 22 " ,
" iq4_xs " : " 23 " ,
" mxfp4 " : " 39 " ,
" nvfp4 " : " 40 " ,
" q1_0 " : " 41 " ,
}
_LLAMA_CPP_PYTHON_TYPE_FLAG_RE = re . compile (
r " (?P<flag>--type_[kv])(?P<sep> \ s+|=)(?P<quote>[ ' \" ]?)(?P<value>[A-Za-z0-9_]+)(?P=quote) "
)
2026-06-01 21:27:04 -06:00
def _ollama_bind_from_cmd ( cmd : str | None , * , default_host : str = " 127.0.0.1 " ) - > tuple [ str , str ] :
""" Return the Ollama bind host/port requested by a serve command.
Plain local ` ollama serve ` defaults to loopback . Remote callers can pass a
wider default host so the resulting API is reachable by Odysseus .
"""
if not cmd :
return default_host , " 11434 "
match = _OLLAMA_HOST_ASSIGNMENT_RE . search ( cmd )
if not match :
return default_host , " 11434 "
value = match . group ( 1 ) . strip ( " ' \" " )
bind_match = _OLLAMA_BIND_RE . match ( value )
if not bind_match :
return " 127.0.0.1 " , " 11434 "
bracketed_host = bind_match . group ( 1 )
host = bracketed_host or bind_match . group ( 3 ) or " 127.0.0.1 "
port = bind_match . group ( 2 ) or bind_match . group ( 4 ) or " 11434 "
if not _OLLAMA_BIND_HOST_RE . match ( host ) :
return " 127.0.0.1 " , " 11434 "
try :
port_num = int ( port , 10 )
except ValueError :
return " 127.0.0.1 " , " 11434 "
if port_num < 1 or port_num > 65535 :
return " 127.0.0.1 " , " 11434 "
return f " [ { host } ] " if bracketed_host else host , port
2026-05-31 23:58:26 +09:00
2026-06-15 02:12:18 -04:00
def _normalize_llama_cpp_python_cache_types ( cmd : str | None ) - > str | None :
""" Map llama.cpp KV cache type names to llama-cpp-python ' s integer enum. """
if not cmd or " llama_cpp.server " not in cmd :
return cmd
def repl ( match : re . Match [ str ] ) - > str :
value = match . group ( " value " )
mapped = _LLAMA_CPP_PYTHON_GGML_TYPES . get ( value . lower ( ) )
if not mapped :
return match . group ( 0 )
quote = match . group ( " quote " )
return f " { match . group ( ' flag ' ) } { match . group ( ' sep ' ) } { quote } { mapped } { quote } "
return _LLAMA_CPP_PYTHON_TYPE_FLAG_RE . sub ( repl , cmd )
2026-05-31 23:58:26 +09:00
def _check_serve_binary ( seg : str ) - > None :
""" Validate that a single command segment starts with an allowlisted binary
( after skipping leading env - var assignments like ` CUDA_VISIBLE_DEVICES = 0 ` ) . """
try :
tokens = shlex . split ( seg ) if seg . strip ( ) else [ ]
except ValueError :
raise HTTPException ( 400 , " Invalid cmd — could not parse " )
if not tokens :
return
env_re = re . compile ( r " ^[A-Za-z_][A-Za-z0-9_]*= " )
first = next ( ( t for t in tokens if not env_re . match ( t ) ) , " " )
base = os . path . basename ( first )
if base not in _SERVE_CMD_ALLOWLIST :
raise HTTPException (
400 ,
f " cmd binary ' { base or ' (empty) ' } ' is not allowed. Must start with one of: "
f " { ' , ' . join ( sorted ( _SERVE_CMD_ALLOWLIST ) ) } " ,
)
def _validate_serve_cmd ( v : str | None ) - > str | None :
""" Reject serve commands that aren ' t in the allowlist or contain shell metachars.
` req . cmd ` is dropped verbatim into a bash / PowerShell wrapper script and
executed in a tmux session . Without this gate , an admin ( or anyone in the
pre - fix world ) could pass arbitrary shell payloads .
Leading env - var assignments ( e . g . ` CUDA_VISIBLE_DEVICES = 0 python3 . . . ` )
are stripped before checking the binary — several of our cmd builders
prepend them , and they shouldn ' t trip the allowlist.
"""
if v is None or v == " " :
return None
# Collapse backslash-newline line continuations into single spaces. Serve
# commands (vLLM especially) are routinely pasted multi-line with trailing
# `\` — that's a safe shell/shlex continuation, so the command stays ONE
# logical invocation and the leading-token allowlist below still governs.
v = re . sub ( r " \\ [ \ t]* \ r? \ n[ \ t]* " , " " , v ) . strip ( )
# Backticks and raw newlines are never legitimate here.
if any ( c in v for c in ( " ` " , " \n " , " \r " ) ) :
raise HTTPException ( 400 , " Invalid characters in cmd " )
2026-06-05 14:41:07 +02:00
2026-05-31 23:58:26 +09:00
# Known GGUF launcher prelude → validate the serve invocation(s) it guards.
m = _GGUF_PRELUDE_RE . match ( v )
if m :
rest = v [ m . end ( ) : ]
# rest is `[ENV=…] python3 -m llama_cpp.server … || [ENV=…] llama-server …`
for part in rest . split ( " || " ) :
_check_serve_binary ( part . strip ( ) )
return v
2026-06-05 14:41:07 +02:00
2026-05-31 23:58:26 +09:00
# Otherwise: a single invocation — no shell metacharacters allowed.
2026-06-05 14:41:07 +02:00
# Temporarily replace safe $(printf %s ...) expressions with a placeholder
# to avoid triggering the metacharacter/command-injection checks.
cleaned_v = v
printf_matches = list ( re . finditer ( r " \ $ \ ( \ s*printf \ s+ %s \ s+([^ \ n()]*?) \ ) " , v ) )
for match in printf_matches :
inner = match . group ( 1 )
if not any ( c in inner for c in ( " ; " , " && " , " || " , " $( " , " ` " ) ) :
cleaned_v = cleaned_v . replace ( match . group ( 0 ) , " /placeholder/safe/path.gguf " )
2026-05-31 23:58:26 +09:00
# (`$(` was the original intent; bare `$` is fine for shell-safe paths.)
2026-06-05 14:41:07 +02:00
if any ( c in cleaned_v for c in ( " ; " , " && " , " || " , " $( " ) ) :
2026-05-31 23:58:26 +09:00
raise HTTPException ( 400 , " Invalid characters in cmd " )
_check_serve_binary ( v )
return v
2026-06-01 23:40:06 +10:00
def _append_serve_preflight_exit_lines ( runner_lines : list [ str ] , * , keep_shell_open : bool ) - > None :
""" Append serve-runner lines that surface preflight failures before exit. """
runner_lines . append ( ' if [ -n " $ODYSSEUS_PREFLIGHT_EXIT " ]; then ' )
runner_lines . append ( ' echo " " ; echo " === Process exited with code $ODYSSEUS_PREFLIGHT_EXIT === " ' )
if keep_shell_open :
cookbook agent debug loop: persistent log files, auto-adopt orphan tmux, Codex/Claude skill parity
Three converging fixes so the chat agent + external Codex/Claude skills can actually debug a crashed serve instead of staring at a post-crash neofetch banner:
* Serves now `tee` to /tmp/odysseus-tmux/SESSION.log on the host running them. Runner saves fds 3/4 before the tee and restores them right before `exec ${SHELL}`, so the post-crash interactive zsh banner does NOT pollute the log file.
* `tail_serve_output` (chat agent) and `/api/codex/cookbook/output/{sid}` (Codex+Claude skills) both prefer the persistent log file over the tmux pane. Pane is fallback for sessions predating the tee runner. Default tail bumped 150 -> 400.
* `list_served_models` "recent log" snippet seeks to the Traceback line instead of showing the last 6 lines (which was always the bash prompt).
Cookbook auto-adoption sweep on `/api/cookbook/tasks/status`: every 20s (rate-limited) the cookbook SSHes each configured server, finds `serve-*` / `cookbook-*` tmux sessions running an actual model process (vllm/python/llama-server/etc., filtered via `pane_current_command`), and writes them into state.tasks. So when the agent falls back to raw ssh+tmux, the session appears in the Cookbook UI on the next poll.
`serve_model` error path now reads `data["detail"]` in addition to `data["error"]` so the FastAPI HTTPException message ("Invalid characters in cmd") actually reaches the agent instead of being swallowed as a generic "Serve failed". Tool description updated to warn against `cd …`/`source …`/`&&` prefixes.
Intent-without-action supervisor in agent_loop: when the model writes "Let me tail the output" / "I'll check the logs" / "Let me investigate" and ends the turn without emitting a tool call, the loop injects a sharp system nudge ("You said you would X — DO IT NOW") and continues. Capped at 2 nudges per chat so a model that genuinely cannot use the tool does not pin the loop.
Codex/Claude skill parity: adds `/cookbook/cached`, `/cookbook/presets`, `/cookbook/preset/{name}`, `/cookbook/adopt` so external agents have the same surface as the chat agent. SKILL.md docs + odysseus_api.py wrapper updated for both bundles.
`adopt_served_model` promoted to the always-on tool set so the agent has a documented fallback when serve_model rejects a cmd.
Also various cookbook UI tweaks accumulated alongside the above (cookbook.js, cookbookRunning.js, cookbookServe.js, cookbook-diagnosis.js, settings.js, style.css).
2026-06-04 23:27:18 +09:00
# Decouple the post-crash interactive shell from the persistent log
# file. fds 3/4 were saved BEFORE the tee redirect at the top of
# the runner; restoring them here means the neofetch banner the
# user's .zshrc prints lands on the tmux pane only, not in the
# log file the agent's tail_serve_output reads.
runner_lines . append ( ' exec 1>&3 2>&4 3>&- 4>&- 2>/dev/null || true ' )
runner_lines . append ( ' sleep 0.2 # let tee child flush + exit ' )
2026-06-01 23:40:06 +10:00
runner_lines . append ( ' exec " $ { SHELL:-/bin/bash} " ' )
else :
runner_lines . append ( ' exit " $ODYSSEUS_PREFLIGHT_EXIT " ' )
runner_lines . append ( ' fi ' )
2026-06-05 20:03:04 +10:00
def _append_vllm_linux_preflight_lines ( runner_lines : list [ str ] ) - > None :
""" Append Linux vLLM readiness lines that identify the runtime being used. """
# Keep the user install bin visible for Odysseus-managed `pip install --user`
# installs, but then report the actual CLI path so external runtimes are clear.
runner_lines . append ( ' export PATH= " $HOME/.local/bin:$PATH " ' )
runner_lines . append ( ' ODYSSEUS_VLLM_BIN= " $(command -v vllm 2>/dev/null || true) " ' )
runner_lines . append ( ' if [ -z " $ODYSSEUS_VLLM_BIN " ]; then ' )
runner_lines . append ( ' echo " ERROR: vLLM is not installed. " ' )
runner_lines . append ( ' ODYSSEUS_PREFLIGHT_EXIT=127 ' )
runner_lines . append ( ' else ' )
runner_lines . append ( ' echo " [odysseus] vLLM CLI: $ODYSSEUS_VLLM_BIN " ' )
runner_lines . append ( ' ODYSSEUS_VLLM_VERSION= " $( " $ODYSSEUS_VLLM_BIN " --version 2>&1 | head -n 1 || true) " ' )
runner_lines . append ( ' if [ -n " $ODYSSEUS_VLLM_VERSION " ]; then echo " [odysseus] vLLM version: $ODYSSEUS_VLLM_VERSION " ; fi ' )
runner_lines . append ( ' fi ' )
2026-06-04 09:00:01 +05:30
def _append_serve_exit_code_lines (
runner_lines : list [ str ] ,
* ,
keep_shell_open : bool ,
is_pip_install : bool = False ,
) - > None :
2026-06-01 22:41:25 +09:00
""" Append serve-runner lines that preserve and report the command exit code. """
runner_lines . append ( ' ODYSSEUS_CMD_EXIT=$? ' )
2026-06-04 09:00:01 +05:30
if is_pip_install :
runner_lines . append ( ' if [ $ODYSSEUS_CMD_EXIT -eq 0 ]; then echo " " ; echo " DOWNLOAD_OK " ; fi ' )
2026-06-01 22:41:25 +09:00
if keep_shell_open :
cookbook agent debug loop: persistent log files, auto-adopt orphan tmux, Codex/Claude skill parity
Three converging fixes so the chat agent + external Codex/Claude skills can actually debug a crashed serve instead of staring at a post-crash neofetch banner:
* Serves now `tee` to /tmp/odysseus-tmux/SESSION.log on the host running them. Runner saves fds 3/4 before the tee and restores them right before `exec ${SHELL}`, so the post-crash interactive zsh banner does NOT pollute the log file.
* `tail_serve_output` (chat agent) and `/api/codex/cookbook/output/{sid}` (Codex+Claude skills) both prefer the persistent log file over the tmux pane. Pane is fallback for sessions predating the tee runner. Default tail bumped 150 -> 400.
* `list_served_models` "recent log" snippet seeks to the Traceback line instead of showing the last 6 lines (which was always the bash prompt).
Cookbook auto-adoption sweep on `/api/cookbook/tasks/status`: every 20s (rate-limited) the cookbook SSHes each configured server, finds `serve-*` / `cookbook-*` tmux sessions running an actual model process (vllm/python/llama-server/etc., filtered via `pane_current_command`), and writes them into state.tasks. So when the agent falls back to raw ssh+tmux, the session appears in the Cookbook UI on the next poll.
`serve_model` error path now reads `data["detail"]` in addition to `data["error"]` so the FastAPI HTTPException message ("Invalid characters in cmd") actually reaches the agent instead of being swallowed as a generic "Serve failed". Tool description updated to warn against `cd …`/`source …`/`&&` prefixes.
Intent-without-action supervisor in agent_loop: when the model writes "Let me tail the output" / "I'll check the logs" / "Let me investigate" and ends the turn without emitting a tool call, the loop injects a sharp system nudge ("You said you would X — DO IT NOW") and continues. Capped at 2 nudges per chat so a model that genuinely cannot use the tool does not pin the loop.
Codex/Claude skill parity: adds `/cookbook/cached`, `/cookbook/presets`, `/cookbook/preset/{name}`, `/cookbook/adopt` so external agents have the same surface as the chat agent. SKILL.md docs + odysseus_api.py wrapper updated for both bundles.
`adopt_served_model` promoted to the always-on tool set so the agent has a documented fallback when serve_model rejects a cmd.
Also various cookbook UI tweaks accumulated alongside the above (cookbook.js, cookbookRunning.js, cookbookServe.js, cookbook-diagnosis.js, settings.js, style.css).
2026-06-04 23:27:18 +09:00
runner_lines . append ( ' echo " " ; echo " === Process exited with code $ODYSSEUS_CMD_EXIT === " ' )
# See preflight branch above for the rationale on restoring fds 3/4.
runner_lines . append ( ' exec 1>&3 2>&4 3>&- 4>&- 2>/dev/null || true ' )
runner_lines . append ( ' sleep 0.2 # let tee child flush + exit ' )
runner_lines . append ( ' exec " $ { SHELL:-/bin/bash} " ' )
2026-06-01 22:41:25 +09:00
else :
runner_lines . append ( ' echo " " ; echo " === Process exited with code $ODYSSEUS_CMD_EXIT === " ' )
2026-06-02 07:59:44 -04:00
runner_lines . append ( ' exit " $ODYSSEUS_CMD_EXIT " ' )
2026-06-01 22:41:25 +09:00
2026-06-02 07:59:44 -04:00
def _append_llama_cpp_linux_accel_build_lines ( runner_lines : list [ str ] ) - > None :
""" Append Linux llama.cpp build lines that prefer ROCm/HIP when available.
Cookbook already detects AMD GPUs elsewhere , but the llama . cpp bootstrap used
to hard - wire CUDA on Linux . That made ROCm hosts attempt a CUDA configure and
fail with " CUDA Toolkit not found " instead of building with HIP .
"""
2026-06-19 00:33:07 +00:00
# Try a prebuilt binary from llama.cpp's GitHub releases FIRST — no
# cmake/build-essential/git/CUDA-headers needed at all. The from-source
# build below stays as a fallback (custom flags, esoteric arch, no
# internet, etc). 30 seconds vs 5+ minutes of compile, and removes
# every OS-package dep from the launch path. Sets _odysseus_have_prebuilt=1
# on success; the existing build-tier if/elif chain below is gated on
# that variable so we never compile twice or shadow the prebuilt symlink.
runner_lines . append ( ' _odysseus_have_prebuilt= " " ' )
runner_lines . append ( ' _odysseus_arch= " $(uname -m) " ' )
runner_lines . append ( ' _odysseus_prebuilt_url= " " ' )
runner_lines . append ( ' if command -v curl >/dev/null 2>&1 && [ " $_odysseus_arch " = " x86_64 " ]; then ' )
runner_lines . append ( ' _odysseus_pat= " " ' )
runner_lines . append ( ' _odysseus_has_nv_inline() { command -v nvidia-smi >/dev/null 2>&1 && nvidia-smi -L 2>/dev/null | grep -q " GPU " ; } ' )
runner_lines . append ( ' _odysseus_has_vk_inline() { ldconfig -p 2>/dev/null | grep -q " libvulkan \\ .so " || command -v vulkaninfo >/dev/null 2>&1 || [ -e /usr/lib/x86_64-linux-gnu/libvulkan.so.1 ]; } ' )
runner_lines . append ( ' _odysseus_has_vkdev_inline() { ls /dev/dri/renderD* >/dev/null 2>&1 || (lspci 2>/dev/null | grep -Ei \' VGA|3D|Display \' | grep -Eiq \' AMD|ATI|Radeon \' ); } ' )
runner_lines . append ( ' if _odysseus_has_nv_inline; then ' )
runner_lines . append ( ' _odysseus_pat= " ubuntu.*cuda " ' )
runner_lines . append ( ' elif _odysseus_has_vkdev_inline && _odysseus_has_vk_inline; then ' )
runner_lines . append ( ' _odysseus_pat= " ubuntu.*vulkan " ' )
runner_lines . append ( ' else ' )
runner_lines . append ( ' _odysseus_pat= " ubuntu-x64 \\ \\ .zip " ' )
runner_lines . append ( ' fi ' )
runner_lines . append ( ' _odysseus_prebuilt_url= " $(curl -fsSL --max-time 15 https://api.github.com/repos/ggml-org/llama.cpp/releases/latest 2>/dev/null | grep \' " browser_download_url " \' | cut -d \' " \' -f4 | grep -iE " $_odysseus_pat " | grep -iv " arm \\ |aarch64 " | head -1) " ' )
runner_lines . append ( ' fi ' )
# Accept any of unzip / bsdtar / python3 -m zipfile as the extractor.
# python3 is essentially always present on modern Linux, so this lets
# the prebuilt path work on minimal Ubuntu installs that lack `unzip`.
runner_lines . append ( ' if [ -n " $_odysseus_prebuilt_url " ] && (command -v unzip >/dev/null 2>&1 || command -v bsdtar >/dev/null 2>&1 || command -v python3 >/dev/null 2>&1); then ' )
runner_lines . append ( ' echo " [odysseus] Found prebuilt llama-server: $_odysseus_prebuilt_url " ' )
runner_lines . append ( ' mkdir -p ~/bin " $HOME/.cache/odysseus/llama-cpp-prebuilt " && cd " $HOME/.cache/odysseus/llama-cpp-prebuilt " ' )
runner_lines . append ( ' rm -f llama-cpp.zip ' )
runner_lines . append ( ' if curl -fsSL --max-time 120 " $_odysseus_prebuilt_url " -o llama-cpp.zip && [ -s llama-cpp.zip ]; then ' )
runner_lines . append ( ' rm -rf build && mkdir -p build ' )
runner_lines . append ( ' if command -v unzip >/dev/null 2>&1; then unzip -qq -o llama-cpp.zip -d build; elif command -v bsdtar >/dev/null 2>&1; then bsdtar -xf llama-cpp.zip -C build; else python3 -c " import zipfile; zipfile.ZipFile( \\ " llama-cpp.zip \\ " ).extractall( \\ " build \\ " ) " ; fi ' )
runner_lines . append ( ' _odysseus_extracted= " $(find build -type f -name llama-server 2>/dev/null | head -1) " ' )
runner_lines . append ( ' if [ -n " $_odysseus_extracted " ]; then ' )
runner_lines . append ( ' chmod +x " $_odysseus_extracted " ' )
runner_lines . append ( ' ln -sf " $_odysseus_extracted " ~/bin/llama-server ' )
runner_lines . append ( ' _odysseus_libdir= " $(dirname " $_odysseus_extracted " ) " ' )
runner_lines . append ( ' mkdir -p ~/.config && echo " export LD_LIBRARY_PATH= \\ " $_odysseus_libdir: \\ $ {LD_LIBRARY_PATH:-} \\ " " > ~/.config/odysseus-llama-cpp-env ' )
runner_lines . append ( ' _odysseus_have_prebuilt=1 ' )
runner_lines . append ( ' echo " [odysseus] Prebuilt llama-server installed at $_odysseus_extracted " ' )
runner_lines . append ( ' fi ' )
runner_lines . append ( ' fi ' )
runner_lines . append ( ' [ -z " $_odysseus_have_prebuilt " ] && echo " [odysseus] Prebuilt download/extract failed — falling back to from-source build. " ' )
runner_lines . append ( ' elif [ -z " $_odysseus_prebuilt_url " ]; then ' )
runner_lines . append ( ' echo " [odysseus] No matching prebuilt llama-server for this host (arch=$_odysseus_arch) — will build from source. " ' )
runner_lines . append ( ' fi ' )
runner_lines . append ( ' if [ -z " $_odysseus_have_prebuilt " ]; then ' )
2026-06-02 07:59:44 -04:00
# Detect pip-installed nvcc (from vLLM/nvidia CUDA wheels) and put it on PATH
2026-06-19 00:33:07 +00:00
# so cmake's CUDA configure can find it — BUT only when actual NVIDIA
# hardware is present. On AMD/Intel hosts the pip nvcc is a misleading
# leftover (no libcudart, no GPU it could target) and would otherwise
# send the build down the CUDA branch and fail with "CUDA Toolkit not
# found" instead of trying Vulkan.
runner_lines . append ( ' _odysseus_has_nvidia_hw() { ' )
runner_lines . append ( ' command -v nvidia-smi >/dev/null 2>&1 && nvidia-smi -L 2>/dev/null | grep -q " GPU " && return 0 ' )
runner_lines . append ( ' ls /dev/nvidia* >/dev/null 2>&1 && return 0 ' )
runner_lines . append ( ' lspci 2>/dev/null | grep -iE \' VGA|3D|Display \' | grep -iq nvidia && return 0 ' )
runner_lines . append ( ' return 1 ' )
runner_lines . append ( ' } ' )
runner_lines . append ( ' if _odysseus_has_nvidia_hw; then ' )
runner_lines . append ( ' for _cudir in ~/.local/lib/python*/site-packages/nvidia/cu13 ~/.local/lib/python*/site-packages/nvidia/cu12 ~/.local/lib/python*/site-packages/nvidia/cuda_nvcc; do ' )
runner_lines . append ( ' [ -x " $_cudir/bin/nvcc " ] && export CUDA_HOME= " $_cudir " && export PATH= " $_cudir/bin:$PATH " && break ' )
runner_lines . append ( ' done ' )
runner_lines . append ( ' fi ' )
2026-06-02 07:59:44 -04:00
# rm -rf build so a prior poisoned CMakeCache.txt (e.g. from a failed CUDA
# or HIP attempt) doesn't cause the next configure to reuse stale settings.
2026-06-15 02:03:55 -04:00
runner_lines . append ( ' mkdir -p ~/bin ' )
2026-06-19 00:33:07 +00:00
# Try to install cmake / build-essential / git automatically before the
# build, but ONLY via passwordless sudo (`sudo -n`) — interactive sudo
# would hang a tmux-backgrounded serve task waiting for a password. If
# sudo asks for a password the install is skipped silently and the
# diagnosis pattern (cookbook_routes.py / cookbook_helpers.py) surfaces
# an explicit "install cmake" suggestion in the Cookbook diagnosis
# toolbar after the inevitable build failure.
runner_lines . append ( ' _odysseus_apt_bootstrap() { ' )
runner_lines . append ( ' local _missing= " " ' )
runner_lines . append ( ' command -v cmake >/dev/null 2>&1 || _missing= " $_missing cmake " ' )
runner_lines . append ( ' command -v g++ >/dev/null 2>&1 || command -v gcc >/dev/null 2>&1 || _missing= " $_missing build-essential " ' )
runner_lines . append ( ' command -v git >/dev/null 2>&1 || _missing= " $_missing git " ' )
runner_lines . append ( ' [ -z " $_missing " ] && return 0 ' )
runner_lines . append ( ' if command -v apt-get >/dev/null 2>&1 && sudo -n true 2>/dev/null; then ' )
runner_lines . append ( ' echo " [odysseus] Auto-installing missing build deps via apt:$_missing " ' )
runner_lines . append ( ' sudo -n env DEBIAN_FRONTEND=noninteractive apt-get update -qq 2>&1 | tail -3 ' )
runner_lines . append ( ' sudo -n env DEBIAN_FRONTEND=noninteractive apt-get install -y --no-install-recommends $_missing 2>&1 | tail -5 || true ' )
runner_lines . append ( ' elif command -v pacman >/dev/null 2>&1 && sudo -n true 2>/dev/null; then ' )
runner_lines . append ( ' echo " [odysseus] Auto-installing missing build deps via pacman:$_missing " ' )
runner_lines . append ( ' local _pacpkgs= " $(echo " $_missing " | sed -e \' s/build-essential/base-devel/g \' ) " ' )
runner_lines . append ( ' sudo -n pacman -Sy --needed --noconfirm $_pacpkgs 2>&1 | tail -5 || true ' )
runner_lines . append ( ' elif command -v dnf >/dev/null 2>&1 && sudo -n true 2>/dev/null; then ' )
runner_lines . append ( ' echo " [odysseus] Auto-installing missing build deps via dnf:$_missing " ' )
runner_lines . append ( ' local _dnfpkgs= " $(echo " $_missing " | sed -e \' s/build-essential/gcc gcc-c++ make/g \' ) " ' )
runner_lines . append ( ' sudo -n dnf install -y $_dnfpkgs 2>&1 | tail -5 || true ' )
runner_lines . append ( ' else ' )
runner_lines . append ( ' echo " [odysseus] WARNING: missing build deps ($_missing) — passwordless sudo is unavailable, cannot auto-install. Cookbook Diagnosis will explain the fix after the build fails. " ' )
runner_lines . append ( ' fi ' )
runner_lines . append ( ' } ' )
runner_lines . append ( ' _odysseus_apt_bootstrap ' )
runner_lines . append ( ' _odysseus_missing_build_deps= " " ' )
runner_lines . append ( ' command -v cmake >/dev/null 2>&1 || _odysseus_missing_build_deps= " $_odysseus_missing_build_deps cmake " ' )
runner_lines . append ( ' command -v git >/dev/null 2>&1 || _odysseus_missing_build_deps= " $_odysseus_missing_build_deps git " ' )
runner_lines . append ( ' command -v g++ >/dev/null 2>&1 || command -v gcc >/dev/null 2>&1 || _odysseus_missing_build_deps= " $_odysseus_missing_build_deps build-essential " ' )
runner_lines . append ( ' if [ -n " $_odysseus_missing_build_deps " ]; then ' )
runner_lines . append ( ' echo " ERROR: llama.cpp source build needs missing packages:$_odysseus_missing_build_deps " ' )
runner_lines . append ( ' if command -v apt-get >/dev/null 2>&1; then ' )
runner_lines . append ( ' echo " Install on this host: sudo apt-get update && sudo apt-get install -y cmake build-essential git " ' )
runner_lines . append ( ' elif command -v pacman >/dev/null 2>&1; then ' )
runner_lines . append ( ' echo " Install on this host: sudo pacman -Sy --needed cmake base-devel git " ' )
runner_lines . append ( ' elif command -v dnf >/dev/null 2>&1; then ' )
runner_lines . append ( ' echo " Install on this host: sudo dnf install -y cmake gcc gcc-c++ make git " ' )
runner_lines . append ( ' fi ' )
runner_lines . append ( ' echo " Alternative: install a native llama-server on PATH, then relaunch. " ' )
runner_lines . append ( ' ODYSSEUS_PREFLIGHT_EXIT=127 ' )
runner_lines . append ( ' fi ' )
runner_lines . append ( ' cd ~/llama.cpp ' )
runner_lines . append ( ' _odysseus_has_vulkan() { ' )
runner_lines . append ( ' ldconfig -p 2>/dev/null | grep -q \' libvulkan \\ .so \' && return 0 ' )
runner_lines . append ( ' [ -e /usr/lib/libvulkan.so.1 ] && return 0 ' )
runner_lines . append ( ' [ -e /usr/lib/x86_64-linux-gnu/libvulkan.so.1 ] && return 0 ' )
runner_lines . append ( ' command -v vulkaninfo >/dev/null 2>&1 && return 0 ' )
runner_lines . append ( ' return 1 ' )
runner_lines . append ( ' } ' )
runner_lines . append ( ' _odysseus_has_vulkan_device() { ' )
runner_lines . append ( ' ls /dev/dri/renderD* >/dev/null 2>&1 && return 0 ' )
runner_lines . append ( ' lspci 2>/dev/null | grep -Ei \' VGA|3D|Display \' | grep -Eiq \' AMD|ATI|Radeon \' && return 0 ' )
runner_lines . append ( ' return 1 ' )
runner_lines . append ( ' } ' )
# Backend preference: native ROCm/HIP > native CUDA > Vulkan > CPU.
# Vulkan is a portable fallback that works on AMD when ROCm isn't
# installed (e.g. Strix Halo) and on any vendor's discrete GPU, but
# it's ~30-40% slower than native HIP/CUDA for LLM inference — only
# pick it when no native toolchain is present.
2026-06-02 07:59:44 -04:00
runner_lines . append ( ' if command -v hipconfig &>/dev/null || [ -d /opt/rocm ] || [ -n " $ROCM_PATH " ] || [ -n " $HIP_PATH " ]; then ' )
2026-06-19 00:33:07 +00:00
runner_lines . append ( ' rm -rf build ' )
2026-06-02 07:59:44 -04:00
runner_lines . append ( ' if command -v hipconfig &>/dev/null; then ' )
runner_lines . append ( ' export HIPCXX= " $ { HIPCXX:-$(hipconfig -l)/clang} " ' )
runner_lines . append ( ' export HIP_PATH= " $ { HIP_PATH:-$(hipconfig -R)} " ' )
runner_lines . append ( ' fi ' )
runner_lines . append ( ' echo " [odysseus] ROCm/HIP detected — building llama-server with HIP support... " ' )
2026-06-02 19:08:09 +02:00
runner_lines . append ( ' cmake -B build -DCMAKE_BUILD_TYPE=Release -DGGML_HIP=ON && cmake --build build -j " $NPROC " --target llama-server && ln -sf ~/llama.cpp/build/bin/llama-server ~/bin/llama-server ' )
2026-06-19 00:33:07 +00:00
runner_lines . append ( ' elif command -v nvcc &>/dev/null && _odysseus_has_nvidia_hw; then ' )
runner_lines . append ( ' rm -rf build ' )
2026-06-03 06:23:55 +01:00
# nvcc alone is not sufficient — pip-installed CUDA wheels or incomplete
# tooling can expose nvcc without shipping libcudart, causing cmake to fail
# mid-build with "CUDA runtime library not found". Check cudart explicitly
# via a small helper so the guard stays readable.
runner_lines . append ( ' _odysseus_has_cudart() { ' )
runner_lines . append ( ' ldconfig -p 2>/dev/null | grep -q \' libcudart \\ .so \' && return 0 ' )
runner_lines . append ( ' local _cuh= " $ { CUDA_HOME:-/usr/local/cuda} " ' )
runner_lines . append ( ' ls " $_cuh/lib64/libcudart.so " * &>/dev/null && return 0 ' )
runner_lines . append ( ' ls " $_cuh/lib/libcudart.so " * &>/dev/null && return 0 ' )
runner_lines . append ( ' ls /usr/local/cuda/lib64/libcudart.so* &>/dev/null && return 0 ' )
runner_lines . append ( ' ls /usr/local/cuda/lib/libcudart.so* &>/dev/null && return 0 ' )
runner_lines . append ( ' ls " $ { _cuh % /cuda_nvcc}/cuda_runtime/lib/libcudart.so " * &>/dev/null && return 0 ' )
runner_lines . append ( ' return 1 ' )
runner_lines . append ( ' } ' )
runner_lines . append ( ' if _odysseus_has_cudart; then ' )
runner_lines . append ( ' echo " [odysseus] CUDA nvcc + cudart found — building llama-server with CUDA (GPU) support... " ' )
runner_lines . append ( ' cmake -B build -DCMAKE_BUILD_TYPE=Release -DGGML_CUDA=ON && cmake --build build -j " $NPROC " --target llama-server && ln -sf ~/llama.cpp/build/bin/llama-server ~/bin/llama-server ' )
runner_lines . append ( ' else ' )
runner_lines . append ( ' echo " [odysseus] WARNING: nvcc found but CUDA runtime (libcudart.so) is not visible — building llama-server for CPU only. " ' )
runner_lines . append ( ' echo " [odysseus] GPU inference will not be available for this llama.cpp build. " ' )
runner_lines . append ( ' echo " [odysseus] Ensure libcudart is installed (e.g. cuda-runtime package) and visible via ldconfig or CUDA_HOME. " ' )
runner_lines . append ( ' cmake -B build -DCMAKE_BUILD_TYPE=Release && cmake --build build -j " $NPROC " --target llama-server && ln -sf ~/llama.cpp/build/bin/llama-server ~/bin/llama-server ' )
runner_lines . append ( ' fi ' )
2026-06-19 00:33:07 +00:00
runner_lines . append ( ' elif _odysseus_has_vulkan_device && _odysseus_has_vulkan; then ' )
runner_lines . append ( ' echo " [odysseus] Vulkan-capable GPU detected (no ROCm/CUDA toolchain installed) — building llama-server with Vulkan support... " ' )
runner_lines . append ( ' rm -rf build-vulkan ' )
runner_lines . append ( ' cmake -B build-vulkan -DCMAKE_BUILD_TYPE=Release -DGGML_VULKAN=ON && cmake --build build-vulkan -j " $NPROC " --target llama-server && ln -sf ~/llama.cpp/build-vulkan/bin/llama-server ~/bin/llama-server ' )
2026-06-02 07:59:44 -04:00
runner_lines . append ( ' else ' )
2026-06-19 00:33:07 +00:00
runner_lines . append ( ' echo " [odysseus] WARNING: no HIP/CUDA/Vulkan toolchain found — building llama-server for CPU only. " ' )
2026-06-02 07:59:44 -04:00
runner_lines . append ( ' echo " [odysseus] GPU inference will not be available for this llama.cpp build. " ' )
2026-06-19 00:33:07 +00:00
runner_lines . append ( ' echo " [odysseus] Install Vulkan (libvulkan-dev) / ROCm for AMD GPUs or CUDA tooling for NVIDIA, then re-launch this serve task. " ' )
runner_lines . append ( ' rm -rf build ' )
2026-06-02 19:08:09 +02:00
runner_lines . append ( ' cmake -B build -DCMAKE_BUILD_TYPE=Release && cmake --build build -j " $NPROC " --target llama-server && ln -sf ~/llama.cpp/build/bin/llama-server ~/bin/llama-server ' )
2026-06-02 07:59:44 -04:00
runner_lines . append ( ' fi ' )
2026-06-19 00:33:07 +00:00
runner_lines . append ( ' fi # end _odysseus_have_prebuilt guard ' )
2026-06-02 07:59:44 -04:00
2026-06-02 23:28:19 -05:00
def _llama_cpp_rebuild_cmd ( ) - > str :
""" Shell command that clears the Cookbook-managed llama.cpp build.
2026-06-19 00:33:07 +00:00
Removes the cached ` ` llama - server ` ` symlink and the ` ` ~ / llama . cpp / build * ` `
2026-06-02 23:28:19 -05:00
directory so the next llama . cpp serve recompiles from source , picking up a
CUDA or HIP toolchain if one is now available . The serve bootstrap only
builds when ` ` llama - server ` ` is missing from PATH , so without this an
existing CPU - only build is reused forever . It deliberately installs and
downloads nothing ; the rebuild itself happens on the next serve .
"""
return (
' mkdir -p " $HOME/bin " && '
' rm -f " $HOME/bin/llama-server " && '
2026-06-19 00:33:07 +00:00
' rm -rf " $HOME/llama.cpp/build " " $HOME/llama.cpp/build-vulkan " && '
2026-06-02 23:28:19 -05:00
' echo " [odysseus] Cleared the cached llama.cpp build. '
' Re-launch the serve task to rebuild llama-server from source '
2026-06-19 00:33:07 +00:00
' (Vulkan, HIP, or CUDA will be used if a matching toolchain is now available). " '
2026-06-02 23:28:19 -05:00
)
2026-05-31 23:58:26 +09:00
class ModelDownloadRequest ( BaseModel ) :
repo_id : str
Cookbook UI: Ollama browser, advanced serve fold, API tokens form, diagnosis toolbar, polish
Surface a lot of accumulated cookbook + UI work as a single non-agent
commit so the agent rework lands cleanly.
Highlights:
- Ollama as a first-class backend in the Cookbook:
* Download input accepts ollama-style names (name:tag) → backend=ollama
* /api/cookbook/ollama/library (cached scrape of ollama.com + curated
fallback so classic models like qwen2.5 stay reachable)
* "Browse Ollama library" toggle below Download with size chips
* Engine=Ollama in hwfit toolbar merges the Ollama library into the
main scan list as per-tag rows with the same Fit/Param/Quant/VRAM
columns; click → fills Download input
- API Tokens form added to Integrations panel (matching wired
loadTokens()/initTokenForm() that had no HTML)
- Serve panel polish: Advanced fold tightening (-8px nudges on vLLM
checks, Extra args, Spec row), n_cpu_moe + Split Mode controls
pulled up 8px to align with the row's checkboxes, GGUF File dropdown
exposed for Ollama backend, GPU re-render on Edit serve restore,
_forceBackend flag so saved serveState wins over backend detection,
cookbook:servers-changed CustomEvent so panels don't need refresh
- Models page redesign: Add Models row (URL + hidden API key reveal +
Type select + Scan/Ollama/Key/Test/Add icon buttons), Probe All +
Clear-offline buttons in Added Models toolbar, offline-pill removed
(opacity already conveys state), Engine dropdown gains Ollama option
- _ping_endpoint probes /v1/models then base, accepts 4xx as
reachable (vLLM returns 404 on bare /v1, fully working endpoints
were showing offline)
- Diagnosis card: × dismiss + Copy bundle buttons restored on the
serve error feedback card
- Orphan tmux sweep re-enabled behind a 60s rate-limit + background
Thread (off the main event loop) so dead serves get discovered
- cookbook_routes auto-register watchdog: drops the endpoint if the
serve session exits non-zero within the first ~3min
- ollama-rocm sidecar awareness in download wrapper (`docker exec
ollama-rocm ollama pull` when host ollama isn't installed)
- Skill extractor sets initial_status="published" when
auto_approve_skills pref is on (audit demotes later)
- Skill list / model list / cookbook scan misc polish
2026-06-08 22:38:49 +09:00
backend : str | None = None # "hf" (default) or "ollama"
2026-05-31 23:58:26 +09:00
include : str | None = None # glob pattern e.g. "*Q4_K_M*"
hf_token : str | None = None
env_prefix : str | None = None # e.g. "source ~/venv/bin/activate"
remote_host : str | None = None # e.g. "gpu-box" — run download on this host via SSH
ssh_port : str | None = None # e.g. "8022" for Termux
platform : str | None = None # "linux", "termux", or "windows"
local_dir : str | None = None # base dir to download into (a per-model subfolder is created under it); None = default HF cache
disable_hf_transfer : bool = False # skip the Rust hf_transfer downloader — slower but far more reliable on large files (used by retries)
class ServeRequest ( BaseModel ) :
repo_id : str
cmd : str
remote_host : str | None = None
ssh_port : str | None = None
env_prefix : str | None = None
hf_token : str | None = None
gpus : str | None = None
platform : str | None = None # "linux", "termux", or "windows"
def _parse_serve_phase ( snapshot : str , task_type : str = " serve " ) - > dict :
""" Parse a tmux snapshot of a serve task into structured phase info.
Single source of truth for serve task status detection . Returns :
{ " phase " : str , " status " : " ready " | " running " | " " , " tps " : float | None ,
" reqs " : int | None , " pct " : int | None }
"""
import re
if task_type != " serve " or not snapshot :
return { }
# Strip newlines so tmux line-wrapping doesn't break regex matching
flat = re . sub ( r ' \ s+ ' , ' ' , snapshot )
load_matches = re . findall ( r ' Loading safetensors.*?( \ d+) % ' , flat )
# Prefer "Downloading (incomplete total...)" (real aggregate bytes) over
# "Fetching N files" (whole-file count, lags with hf_transfer's chunked pulls).
downloading_matches = re . findall ( r ' Downloading.*?( \ d+) % ' , flat )
fetching_matches = re . findall ( r ' Fetching.*?( \ d+) % ' , flat )
dl_matches = downloading_matches if downloading_matches else fetching_matches
# Match "Avg generation throughput: X tokens/s, Running: N reqs" (with line-wrap tolerance)
tps_matches = re . findall (
r ' (?:Avg )?generation throughput: \ s*([ \ d.]+) \ s*tokens/s.*?Running: \ s*( \ d+) \ s*reqs ' ,
flat ,
)
# Check throughput FIRST — the throughput log line contains "GPU KV cache usage"
# which would otherwise false-match the warmup check
if tps_matches :
tps_str , reqs_str = tps_matches [ - 1 ]
tps = float ( tps_str )
reqs = int ( reqs_str )
return {
" phase " : f " { tps_str } tok/s " if reqs > 0 else " idle " ,
" status " : " ready " ,
" tps " : tps ,
" reqs " : reqs ,
}
if " Application startup complete " in flat :
return { " phase " : " ready " , " status " : " ready " }
2026-06-02 09:36:03 +09:00
if re . search ( r ' Ollama API ready on port \ s+ \ d+ ' , flat , re . I ) :
return { " phase " : " ready " , " status " : " ready " }
2026-05-31 23:58:26 +09:00
# HTTP access logs (e.g. GET /v1/models 200 OK) mean the server is up and serving
if re . search ( r ' (?:GET|POST) \ s+/[^ \ s]* \ s+HTTP/[ \ d.]+ " \ s* \ d {3} ' , flat ) :
return { " phase " : " idle " , " status " : " ready " }
if " Loading weights took " in flat :
return { " phase " : " initializing " , " status " : " running " }
# "GPU KV cache" alone (during allocation) — not "GPU KV cache usage" (runtime log)
if " GPU KV cache " in flat and " GPU KV cache usage " not in flat :
return { " phase " : " warming up " , " status " : " running " }
if load_matches :
pct = int ( load_matches [ - 1 ] )
return { " phase " : f " loading { pct } % " , " status " : " running " , " pct " : pct }
if dl_matches :
pct = int ( dl_matches [ - 1 ] )
return { " phase " : f " downloading { pct } % " , " status " : " running " , " pct " : pct }
return { }
def _ssh ( host , cmd , port = None ) :
""" Build SSH command string with optional port. """
pf = f " -p { port } " if port and port != " 22 " else " "
return f " ssh { pf } { host } ' { cmd } ' "
def _safe_env_prefix ( ep : str | None ) - > str | None :
""" Rewrite a `source <path>` env_prefix so it no-ops if the path is missing.
Prevents ` line N : < path > : No such file or directory ` errors when a serve
task is launched against a host that doesn ' t have the expected venv.
Also rewrites leading ` ~ / ` → ` $ HOME / ` so the path expands inside double
quotes ( bash only tilde - expands unquoted tokens at word start ) . """
if not ep :
return ep
import shlex
try :
parts = shlex . split ( ep , posix = True )
except ValueError :
raise HTTPException ( 400 , " Invalid env_prefix " )
if len ( parts ) != 2 or parts [ 0 ] not in { " source " , " . " } :
# Bash conda activation emitted by the frontend:
# eval "$(conda shell.bash hook)" && conda activate ENV
m = re . fullmatch ( r ' eval " \ $ \ (conda shell \ .bash hook \ ) " && conda activate (.+) ' , ep )
if m :
env = m . group ( 1 ) . strip ( )
try :
env_parts = shlex . split ( env , posix = True )
except ValueError :
raise HTTPException ( 400 , " Invalid env_prefix " )
if len ( env_parts ) != 1 :
raise HTTPException ( 400 , " Invalid env_prefix " )
return ' eval " $(conda shell.bash hook) " && conda activate ' + shlex . quote ( env_parts [ 0 ] )
# Plain conda activation, used by Windows/PowerShell and some manual callers.
if len ( parts ) == 3 and parts [ 0 ] == " conda " and parts [ 1 ] == " activate " :
return " conda activate " + shlex . quote ( parts [ 2 ] )
# PowerShell venv activation emitted by the frontend:
# & 'C:\path\Scripts\Activate.ps1'
if len ( parts ) == 2 and parts [ 0 ] == " & " :
path = parts [ 1 ]
if any ( c in path for c in " \r \n ;&|`$<> " ) :
raise HTTPException ( 400 , " Invalid env_prefix " )
return " & ' " + path . replace ( " ' " , " ' ' " ) + " ' "
raise HTTPException ( 400 , " Invalid env_prefix " )
path = parts [ 1 ]
if any ( c in path for c in " \r \n ;&|`$<> " ) :
raise HTTPException ( 400 , " Invalid env_prefix " )
# Replace a leading "~/" with "$HOME/" so it survives quoting
if path . startswith ( " ~/ " ) :
path = " $HOME/ " + path [ 2 : ]
elif path == " ~ " :
path = " $HOME "
path = path . replace ( ' " ' , ' \\ " ' )
return f ' [ -f " { path } " ] && source " { path } " || true '
def _ssh_ps ( host , script_path , port = None ) :
""" Build SSH command to run a PowerShell script on a Windows remote. """
pf = f " -p { port } " if port and port != " 22 " else " "
return f ' ssh { pf } { host } " powershell -ExecutionPolicy Bypass -File { script_path } " '
# Windows session dir — stored in user's temp on the remote
WIN_SESSION_DIR = " $env:TEMP \\ \\ odysseus-sessions "
2026-06-05 05:52:07 -03:00
def _diagnose_serve_output ( text : str ) - > dict | None :
""" Server-side mirror of the Cookbook UI ' s common serve diagnoses.
The browser uses cookbook - diagnosis . js for clickable fixes . This gives
the agent / tool path the same structured signal so it can retry with an
adjusted command instead of guessing from raw tmux output .
"""
if not text :
return None
tail = text [ - 6000 : ]
patterns = [
(
r " No available memory for the cache blocks|Available KV cache memory:.*- " ,
" No GPU memory left for KV cache after loading model. " ,
[
{ " label " : " retry with GPU memory utilization 0.95 " , " op " : " replace " , " flag " : " --gpu-memory-utilization " , " value " : " 0.95 " } ,
{ " label " : " retry with context 2048 " , " op " : " replace " , " flag " : " --max-model-len " , " value " : " 2048 " } ,
] ,
) ,
(
r " CUDA out of memory|torch \ .cuda \ .OutOfMemoryError|CUDA error: out of memory|warming up sampler|max_num_seqs.*gpu_memory_utilization " ,
" GPU ran out of memory during startup or warmup. " ,
[
{ " label " : " retry with context 4096 " , " op " : " replace " , " flag " : " --max-model-len " , " value " : " 4096 " } ,
{ " label " : " retry with GPU memory utilization 0.80 " , " op " : " replace " , " flag " : " --gpu-memory-utilization " , " value " : " 0.80 " } ,
{ " label " : " retry with --enforce-eager " , " op " : " append " , " arg " : " --enforce-eager " } ,
] ,
) ,
(
r " not divisib|must be divisible|attention heads.*divisible " ,
" Tensor parallel size is incompatible with the model. " ,
[
{ " label " : " retry with tensor parallel size 1 " , " op " : " replace " , " flag " : " --tensor-parallel-size " , " value " : " 1 " } ,
{ " label " : " retry with tensor parallel size 2 " , " op " : " replace " , " flag " : " --tensor-parallel-size " , " value " : " 2 " } ,
] ,
) ,
(
r " KV cache.*too (small|large)|max_model_len.*exceeds|maximum.*context " ,
" Context length is too large for available GPU memory. " ,
[
{ " label " : " retry with context 8192 " , " op " : " replace " , " flag " : " --max-model-len " , " value " : " 8192 " } ,
{ " label " : " retry with context 4096 " , " op " : " replace " , " flag " : " --max-model-len " , " value " : " 4096 " } ,
] ,
) ,
(
r " enable-auto-tool-choice requires --tool-call-parser " ,
" Auto tool choice requires an explicit tool call parser. " ,
[ { " label " : " retry with Hermes tool parser " , " op " : " append " , " arg " : " --tool-call-parser hermes " } ] ,
) ,
(
r " Please pass.*trust.remote.code=True|contains custom code which must be executed to correctly load|does not recognize this architecture|model type.*but Transformers does not " ,
" Model requires custom code or newer model support. " ,
[ { " label " : " retry with --trust-remote-code " , " op " : " append " , " arg " : " --trust-remote-code " } ] ,
) ,
2026-06-05 20:03:04 +10:00
(
r " There is no module or parameter named [ ' \" ]lm_head \ .input_scale[ ' \" ]|lm_head \ .input_scale|weight_scale_2 " ,
" vLLM cannot load this ModelOpt LM-head quantized checkpoint with the current runtime. " ,
[
{
" label " : " upgrade vLLM through the environment that provides this CLI, or use a compatible checkpoint " ,
" op " : " manual " ,
}
] ,
) ,
2026-06-05 05:52:07 -03:00
(
r " Either a revision or a version must be specified|transformers \ .integrations \ .hub_kernels|kernels/layer " ,
" vLLM/Transformers kernel package mismatch. " ,
[ { " label " : " update vLLM, Transformers, and kernels on this server " , " op " : " dependency " , " package " : " vllm transformers kernels " } ] ,
) ,
(
r " Address already in use|bind.*address.*in use " ,
" Port is already in use. " ,
[ { " label " : " retry on port 8001 " , " op " : " replace " , " flag " : " --port " , " value " : " 8001 " } ] ,
) ,
(
r " No CUDA GPUs are available|no GPU.*found|CUDA_VISIBLE_DEVICES.*invalid " ,
" No GPUs are visible to the serve process. " ,
[ { " label " : " clear Cookbook GPU selection or choose available GPUs " , " op " : " settings " , " field " : " gpus " , " value " : " " } ] ,
) ,
(
r " Failed to infer device type|NVML Shared Library Not Found|No module named ' amdsmi ' |platform is not available " ,
" vLLM could not find a supported GPU (CUDA or ROCm). "
" This machine may have integrated or unsupported graphics only. " ,
[
{ " label " : " switch to llama.cpp (CPU/Metal, works without a discrete GPU) " , " op " : " manual " } ,
{ " label " : " switch to Ollama (CPU/Metal, works without a discrete GPU) " , " op " : " manual " } ,
] ,
) ,
(
r " vllm.*command not found|No module named vllm|ERROR: vLLM is not installed " ,
" vLLM is not installed or not in PATH on this server. " ,
[ { " label " : " install vLLM in Cookbook Dependencies " , " op " : " dependency " , " package " : " vllm " } ] ,
) ,
2026-06-15 14:14:37 +08:00
(
r " sgl_kernel[ \ s \ S]*(Python \ .h|libnuma \ .so \ .1|common_ops)| "
r " (Python \ .h|libnuma \ .so \ .1|common_ops)[ \ s \ S]*sgl_kernel| "
r " Please ensure sgl_kernel is properly installed " ,
" SGLang native dependencies are missing on this server. " ,
[
{ " label " : " install OS packages: libnuma-dev python3.12-dev build-essential " , " op " : " manual " } ,
{ " label " : " upgrade sglang-kernel after OS packages are installed " , " op " : " manual " } ,
] ,
) ,
2026-06-05 05:52:07 -03:00
(
r " sglang.*command not found|No module named sglang|SGLang is not installed " ,
" SGLang is not installed or not in PATH on this server. " ,
[ { " label " : " install SGLang in Cookbook Dependencies " , " op " : " dependency " , " package " : " sglang[all] " } ] ,
) ,
2026-06-19 00:33:07 +00:00
# System build deps come BEFORE the generic llama.cpp catch-all so
# cmake / build-essential / git missing → a specific OS-package
# remediation instead of "install llama-cpp-python[server]" (which
# itself fails to compile when cmake is absent).
(
r " cmake: command not found|cmake.*not found.*[Cc]ould not " ,
" cmake is required to build llama.cpp from source but isn ' t installed on this server. " ,
[ { " label " : " install build deps for llama.cpp (apt: cmake build-essential git / pacman: cmake base-devel git / dnf: cmake gcc-c++ make git / brew: cmake git) " , " op " : " dependency " , " package " : " llama-cpp-python[server] " } ] ,
) ,
(
r " ^(make|g \ + \ +|gcc): command not found|Could not find C \ + \ + compiler " ,
" A C/C++ compiler (build-essential) is required to build llama.cpp from source. " ,
[ { " label " : " install build deps for llama.cpp on this server " , " op " : " dependency " , " package " : " llama-cpp-python[server] " } ] ,
) ,
(
r " ^git: command not found " ,
" git is required to clone the llama.cpp source tree. " ,
[ { " label " : " install build deps for llama.cpp on this server " , " op " : " dependency " , " package " : " llama-cpp-python[server] " } ] ,
) ,
2026-06-05 05:52:07 -03:00
(
2026-06-19 00:33:07 +00:00
r " llama-server.*command not found|llama \ .cpp.*not found|No module named.*llama_cpp|No module named ' starlette_context ' " ,
2026-06-05 05:52:07 -03:00
" llama.cpp / llama-cpp-python dependencies are missing. " ,
[ { " label " : " install llama.cpp dependencies or llama-cpp-python[server] " , " op " : " dependency " , " package " : " llama-cpp-python[server] " } ] ,
) ,
(
r " No GGUF found on this host|no \ .gguf file|No GGUF file found " ,
" No GGUF file found for this model on this host. The llama.cpp backend needs a .gguf file. " ,
[ { " label " : " download a GGUF build of this model (repo name usually ends in -GGUF, file like Q4_K_M.gguf) " , " op " : " manual " } ] ,
) ,
(
r " No module named ' torch ' |No module named torch|No module named ' diffusers ' |No module named diffusers " ,
" Diffusion serving requires PyTorch and diffusers. " ,
[ { " label " : " install diffusers[torch] in Cookbook Dependencies " , " op " : " dependency " , " package " : " diffusers[torch] " } ] ,
) ,
(
r " 403 Forbidden|401 Unauthorized|Access to model.*is restricted|gated repo|not in the authorized list|awaiting a review " ,
" Model access is gated or unauthorized. " ,
[ { " label " : " set HF token and request model access on HuggingFace " , " op " : " manual " } ] ,
) ,
]
for pattern , message , suggestions in patterns :
if re . search ( pattern , tail , re . I ) :
return { " message " : message , " suggestions " : suggestions }
if re . search ( r " Traceback \ (most recent call last \ ) " , tail , re . I ) and not re . search (
r " Application startup complete|GET /v1/|Uvicorn running on " , tail , re . I
) :
return {
" message " : " Python traceback detected during serve startup. " ,
" suggestions " : [ { " label " : " inspect traceback and retry with adjusted backend/settings " , " op " : " manual " } ] ,
}
return None
2026-06-08 00:33:50 +02:00
async def run_ssh_command_async (
remote : str ,
ssh_port : str | None ,
remote_cmd : str ,
* ,
timeout : float ,
connect_timeout : int | None = None ,
strict_host_key_checking : bool | None = None ,
stdin_data : bytes | None = None ,
) - > tuple [ int , bytes , bytes ] :
""" Run an ssh command with centralized timeout and stderr/stdout capture.
Async version of core . platform_compat . run_ssh_command_sync .
"""
import asyncio
proc = await asyncio . create_subprocess_exec (
* _ssh_exec_argv (
remote ,
ssh_port ,
remote_cmd = remote_cmd ,
connect_timeout = connect_timeout ,
strict_host_key_checking = strict_host_key_checking ,
) ,
stdin = asyncio . subprocess . PIPE if stdin_data is not None else None ,
stdout = asyncio . subprocess . PIPE ,
stderr = asyncio . subprocess . PIPE ,
)
try :
stdout , stderr = await asyncio . wait_for (
proc . communicate ( input = stdin_data ) , timeout = timeout
)
except asyncio . TimeoutError :
proc . kill ( )
await proc . communicate ( )
raise
2026-06-09 00:10:20 +03:00
return proc . returncode or 0 , stdout , stderr