2026-05-31 23:58:26 +09:00
// ============================================
// COOKBOOK MODULE (v2 — simplified)
// What Fits? + Saved presets, inline action panels
// ============================================
import uiModule from './ui.js' ;
import spinnerModule from './spinner.js' ;
import { providerLogo } from './providers.js' ;
import { makeWindowDraggable } from './windowDrag.js' ;
import { _diagnose , _showDiagnosis , _clearDiagnosis , _runQuickCmd , ERROR _PATTERNS } from './cookbook-diagnosis.js' ;
import { _hwfitCache , _hwfitDebounce , _hwfitFetch , _hwfitInit , _hwfitRenderList , _hwfitRenderHw , _renderGpuToggles , _expandModelRow , _fitColors , _hwfitColumns , _cachedModelIds , _gpuToggleTotal , _resetGpuToggleState } from './cookbook-hwfit.js' ;
// Sub-modules
import {
initRunning ,
_loadTasks , _saveTasks , _addTask , _removeTask ,
_tmuxCmd , _renderRunningTab , _clearCookbookNotif ,
_launchServeTask , _serveAutoFix , _serveAutoRetry , _serveAutoRetryReplace , _serveAutoRetryRemove ,
_startBackgroundMonitor , _syncFromServer ,
_retryDownload , _nextAvailablePort , _processQueue ,
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
_selfHealStaleTasks ,
2026-05-31 23:58:26 +09:00
} from './cookbookRunning.js' ;
import {
initDownload ,
_setPanelField , _setPanelCheckbox ,
_wirePanelEvents , _runPanelCmd , _runModelDownload , _buildDownloadCmd ,
} from './cookbookDownload.js' ;
import {
initServe ,
_fetchCachedModels , _cachedAllModels , _filterCachedList , _rerenderCachedModels , _deleteCachedModel ,
} from './cookbookServe.js' ;
const STORAGE _KEY = 'cookbook-presets' ;
const LAST _STATE _KEY = 'cookbook-last-state' ;
const SERVE _STATE _KEY = 'cookbook-serve-state' ;
// Global, once: tag chip rows (.doclib-lang-chips) scroll horizontally on mobile.
// Stop their touch events (capture phase, before any ancestor sees them) so a
// sideways tag scroll never triggers a swipe-to-change-tab / swipe-dismiss
// gesture in ANY modal (cookbook, document library, etc.). We don't preventDefault,
// so the browser's native horizontal scroll of the chips still works.
if ( typeof window !== 'undefined' && ! window . _tagScrollGuardWired ) {
window . _tagScrollGuardWired = true ;
[ 'touchstart' , 'touchmove' ] . forEach ( evt => {
document . addEventListener ( evt , ( e ) => {
const t = e . target ;
if ( t && t . closest && t . closest ( '.doclib-lang-chips' ) ) e . stopPropagation ( ) ;
} , true ) ;
} ) ;
}
// Radio-style check marking which model directory is a server's download target.
// OFF = hollow circle (pickable); ON = checked circle (accent-tinted via CSS).
export const _MODELDIR _CHECK _OFF = '<svg width="13" height="13" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round"><circle cx="12" cy="12" r="9"/></svg>' ;
export const _MODELDIR _CHECK _ON = '<svg width="13" height="13" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2.6" stroke-linecap="round" stroke-linejoin="round"><circle cx="12" cy="12" r="9"/><polyline points="8 12 11 15 16 9"/></svg>' ;
// Monochrome platform glyphs (currentColor) for a server's OS tag: a penguin for
// Linux, the four-pane logo for Windows, an Android robot for Termux/Android.
function _platformIcon ( platform ) {
const k = ( platform || '' ) . toLowerCase ( ) ;
if ( k === 'windows' ) {
return '<svg viewBox="0 0 24 24" width="12" height="12" fill="currentColor" aria-hidden="true"><path d="M3 4.6l8-1.2v8.1H3V4.6zm9-1.3L21 2v9.5h-9V3.3zM3 12.5h8v8.1l-8-1.2v-6.9zm9 0h9V22l-9-1.3v-8.2z"/></svg>' ;
}
if ( k === 'termux' || k === 'android' ) {
return '<svg viewBox="0 0 24 24" width="12" height="12" fill="currentColor" aria-hidden="true"><path d="M7 9h10v6.6a1 1 0 0 1-1 1h-.7v2.6a1.15 1.15 0 1 1-2.3 0V16.6h-1.5v2.6a1.15 1.15 0 1 1-2.3 0V16.6H8a1 1 0 0 1-1-1V9zM4.3 9.1a1.15 1.15 0 0 1 2.3 0v4.6a1.15 1.15 0 1 1-2.3 0V9.1zm13.1 0a1.15 1.15 0 0 1 2.3 0v4.6a1.15 1.15 0 1 1-2.3 0V9.1zM8 8a4 4 0 0 1 8 0H8zm1.7-2.6-.8-1.2a.28.28 0 0 1 .47-.3l.83 1.25a4.8 4.8 0 0 1 3.66 0l.83-1.25a.28.28 0 0 1 .47.3L14.3 5.4M9.8 6.6a.62.62 0 1 0 0-1.24.62.62 0 0 0 0 1.24zm4.4 0a.62.62 0 1 0 0-1.24.62.62 0 0 0 0 1.24z"/></svg>' ;
}
if ( k === 'linux' || k === 'termux-linux' ) {
return '<svg viewBox="0 0 24 24" width="12" height="12" fill="currentColor" aria-hidden="true"><path d="M12 2a4 4 0 0 0-4 4v4.7c0 .9-.4 1.7-1 2.4-1.2 1.4-2 3-2 4.5C5 20.4 8.1 22 12 22s7-1.6 7-4.4c0-1.5-.8-3.1-2-4.5-.6-.7-1-1.5-1-2.4V6a4 4 0 0 0-4-4zm-1.7 4.8a1 1 0 1 1 0 2 1 1 0 0 1 0-2zm3.4 0a1 1 0 1 1 0 2 1 1 0 0 1 0-2zM12 9.4c.75 0 1.4.45 1.7 1.1h-3.4c.3-.65.95-1.1 1.7-1.1z"/></svg>' ;
}
return '' ;
}
export let _envState = { env : 'none' , envPath : '' , hfToken : '' , hfTokenConfigured : false , hfTokenMasked : '' , gpus : '' , remoteHost : '' , servers : [ ] , modelPaths : [ ] , platform : '' , defaultServer : '' } ;
let _lastCacheHostVal = null ;
let _cookbookOpeningSpinners = [ ] ;
export function _lastCacheHost ( ) { return _lastCacheHostVal ; }
export function _setLastCacheHost ( v ) { _lastCacheHostVal = v ; }
function _setCookbookOpening ( on ) {
// Sidebar (tool-cookbook-btn) deliberately excluded — the inline
// whirlpool on the sidebar row read as "the click didn't register"
// rather than "loading", which made users (rightly) think clicks
// were being eaten. Keep only the icon-rail spinner since the
// rail is narrow enough that an obvious loading state still helps.
const targets = [
document . getElementById ( 'rail-cookbook' ) ,
] . filter ( Boolean ) ;
if ( ! on ) {
_cookbookOpeningSpinners . forEach ( ( { spinner , wrap , target } ) => {
try { spinner ? . stop ? . ( ) ; } catch { }
try { wrap ? . remove ? . ( ) ; } catch { }
target ? . classList ? . remove ( 'cookbook-opening' ) ;
} ) ;
_cookbookOpeningSpinners = [ ] ;
return ;
}
if ( _cookbookOpeningSpinners . length ) return ;
targets . forEach ( target => {
const spinner = spinnerModule . create ( '' , 'clean' , 'whirlpool' ) ;
spinner . _wpSize = target . id === 'rail-cookbook' ? 12 : 13 ;
const wrap = document . createElement ( 'span' ) ;
wrap . className = 'cookbook-open-loading' ;
wrap . appendChild ( spinner . createElement ( ) ) ;
target . appendChild ( wrap ) ;
target . classList . add ( 'cookbook-opening' ) ;
spinner . start ( ) ;
_cookbookOpeningSpinners . push ( { spinner , wrap , target } ) ;
} ) ;
}
/** Build server <option> HTML from _envState.servers. excludeLocal skips local-only entries. */
// True for the local server entry (empty / "local" / "localhost" host).
function _isLocalEntry ( s ) { return ! s || ! s . host || s . host === 'local' || s . host . toLowerCase ( ) === 'localhost' ; }
// Resolve a dropdown option value to a server entry. Option values are the
// stable HOST string ('local' for the local box) — NOT array indices — because
// `_envState.servers` gets deduped/reordered, which made index-based selection
// silently resolve to the wrong (or local) server. Accepts a numeric index too
// for backwards-compat with any stale value.
function _serverByVal ( val ) {
if ( val == null || val === 'local' || val === '' ) return null ;
let s = _envState . servers . find ( x => x . host === val ) ;
if ( ! s && /^\d+$/ . test ( String ( val ) ) ) s = _envState . servers [ parseInt ( val ) ] ;
return s || null ;
}
function _buildServerOpts ( excludeLocal = false ) {
// The local server is ALWAYS represented by the synthetic value="local" option
// (showing its custom name from the "server name" feature). We must therefore
// skip that same entry in the loop below — otherwise it appeared twice.
const _localIdx = _envState . servers . findIndex ( _isLocalEntry ) ;
const _localSrv = _localIdx >= 0 ? _envState . servers [ _localIdx ] : null ;
const _localLabel = ( _localSrv && _localSrv . name ) ? _localSrv . name : 'Local' ;
let html = ` <option value="local" ${ ! _envState . remoteHost ? ' selected' : '' } > ${ esc ( _localLabel ) } </option> ` ;
for ( let i = 0 ; i < _envState . servers . length ; i ++ ) {
const s = _envState . servers [ i ] ;
if ( i === _localIdx ) continue ; // already the synthetic "local" option
if ( excludeLocal && _isLocalEntry ( s ) ) continue ;
const label = s . name || s . host || ` Server ${ i + 1 } ` ;
const selected = _envState . remoteHost === s . host ? ' selected' : '' ;
html += ` <option value=" ${ esc ( s . host ) } " ${ selected } > ${ esc ( label ) } </option> ` ;
}
return html ;
}
/** Wrap a command in SSH for a remote host, with proper single-quote escaping. */
export function _sshCmd ( host , cmd , port ) {
const portFlag = port && port !== '22' ? ` -p ${ port } ` : '' ;
return ` ssh ${ portFlag } ${ host } ' ${ cmd . replace ( /'/g , "'\\''" ) } ' ` ;
}
/** Get SSH port for a given host (or task object) */
function _getPort ( hostOrTask ) {
if ( ! hostOrTask ) return '' ;
if ( typeof hostOrTask === 'object' ) return hostOrTask . sshPort || _getPort ( hostOrTask . remoteHost ) ;
const srv = _envState . servers . find ( s => s . host === hostOrTask ) ;
return srv ? . port || '' ;
}
/** Get platform for a given host (or task object). Returns 'windows', 'termux', 'linux', or '' */
export function _getPlatform ( hostOrTask ) {
if ( ! hostOrTask ) return _envState . platform || '' ;
if ( typeof hostOrTask === 'object' ) return hostOrTask . platform || _getPlatform ( hostOrTask . remoteHost ) ;
const srv = _envState . servers . find ( s => s . host === hostOrTask ) ;
return srv ? . platform || '' ;
}
/** Check if the current active server is Windows */
export function _isWindows ( hostOrTask ) {
return _getPlatform ( hostOrTask ) === 'windows' ;
}
Add macOS Apple Silicon Cookbook support
* Add Apple Silicon (Metal) GPU detection and unified-memory fit tuning
hardware.py detects Apple Silicon locally and over SSH, reporting
backend=metal, the chip name, and a RAM-scaled fraction of unified
memory as the usable GPU budget. fit.py gains an M1-M4 memory-bandwidth
table for realistic tok/s and drops vLLM-only formats (AWQ/GPTQ/FP8)
that can't be served on Metal.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 32ac81dbc680361463a088dae867d555d5a79c3b)
* Generate macOS/Metal serve commands and surface the Metal GPU
cookbook_routes.py adds a macOS serve path (Ollama, Metal-aware
llama.cpp build using `sysctl hw.ncpu` instead of `nproc`, and a clear
error if vLLM is attempted). The frontend defaults Metal serving to
llama.cpp and offers llama.cpp/Ollama instead of vLLM/SGLang. The
odysseus-cookbook CLI's `gpus` command reports the Metal GPU via
sysctl/vm_stat.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 4ba01ce25d256ae032029898f361c824a34fcd4b)
* Add launchd LaunchAgent for macOS (systemd equivalent)
com.odysseus.ui.plist + install-service-macos.sh run Odysseus at login
and restart on crash, the macOS counterpart to odysseus-ui.service. The
installer auto-fills paths from the venv, so there's no hand-editing.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 3d4b6b2c7b8b31af32201ed278115df9a559dea9)
* Document macOS install (brew, Ollama, AirPlay port, launchd)
README + setup.py cover the Homebrew / Apple Silicon path: brew install
python@3.11 tmux ollama, Metal serving via Ollama/llama.cpp, the launchd
service, and the macOS AirPlay Receiver conflict on ports 7000/5000.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 8dc9a3578a1726f070ed9f75c0958ae291a6d966)
* Add downloadable macOS launcher app builder
build-macos-app.sh generates dist/Odysseus.app and a drag-to-Applications
dist/Odysseus.dmg. The app starts the local server from this repo's venv and
opens the UI in a chrome-less app window (Chromium --app mode, falling back to
the default browser). It's a launcher wrapper — it drives the venv rather than
bundling Python — so the install path is baked in at build time.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 7927940c3810ee34640803b198d334a6ac93474d)
* Harden macOS Cookbook support: hide MLX, fix Metal build cache
Builds on the adopted PR #213 macOS/Metal work with two fixes and tests:
- fit.py: always drop MLX-quantized models. Odysseus only generates serve
commands for llama.cpp/Ollama (Metal) and vLLM/SGLang (CUDA); MLX needs the
mlx_lm runtime and the catalog's MLX repos ship no GGUF alternative, so they
were surfaced on Apple Silicon but could never be served.
- cookbook_routes.py (macOS branch only): `rm -rf build` before configure so a
poisoned CMakeCache from a prior failed CUDA attempt can't make every later
build fail; explicit -DCMAKE_BUILD_TYPE=Release; a clear "brew install cmake"
hint if cmake is missing. Linux/CUDA path unchanged.
- tests/test_hwfit_macos.py: MLX hidden on metal, MLX still hidden on CUDA
(regression guard), Metal detection on Apple Silicon, and skipped on
Linux/Intel (proves non-macOS detection is untouched).
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Propagate unified_memory flag and document macOS GPU/Docker caveat
- hardware.py: detect_system now carries the unified_memory flag from GPU
detection into the system dict (it was set by _detect_apple_silicon / AMD-APU
detection but dropped during result assembly, so the API always reported
null). Lets callers distinguish unified from discrete VRAM.
- README: prominent warning that Docker on Apple Silicon can't reach the Metal
GPU (runs a Linux VM) — Cookbook must run natively for GPU serving; fix stale
text that said Cookbook recommends MLX models (now hidden as unservable).
- test: detect_system propagates unified_memory.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Put Odysseus's venv bin on PATH for cookbook runners
Native (non-Docker) installs run from a virtualenv whose bin holds the `hf` CLI
and `python3` the cookbook download/serve tmux scripts shell out to. Those
scripts start in a fresh login shell with the venv NOT activated, so on a native
macOS install `hf download` failed with "hf: command not found" — and the
`pip --user` self-heal missed because macOS has no bare `pip` command.
- cookbook_helpers.py: _local_tooling_path_export() — pure helper returning a
PATH export for the running interpreter's bin dir (escaped for double quotes).
- cookbook_routes.py: download + serve runners prepend that dir on local runs
(gated off SSH/Windows); swap the `pip` install fallbacks to `python3 -m pip`.
- tests: helper output for normal and spaced paths.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Document macOS llama.cpp serving prerequisites
Clarify the two serving paths on Apple Silicon: the recommended zero-build
route (brew install llama.cpp ships a Metal llama-server Cookbook finds on PATH),
and the from-source fallback, which requires cmake + Xcode Command Line Tools.
Without those the build is skipped and serving silently degrades to a slow CPU
build, so new users now know to install them (or use the prebuilt) up front.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Recommend only GGUF-servable models on Metal
Apple Silicon's only serving engines are llama.cpp and Ollama, both GGUF-only
(vLLM/SGLang are CUDA/ROCm and don't run on macOS). The catalog tags raw
safetensors repos with a default Q4_K_M quant, so the fit-ranking was
recommending ~397/501 models that have no GGUF and fail to serve on Metal with
"No GGUF found" (e.g. microsoft/Phi-mini-MoE-instruct).
Drop any model without a real GGUF (is_gguf/gguf_sources) on Apple Silicon —
subsumes the previous AWQ/GPTQ/FP8 special-case into one rule. On CUDA these
stay visible since vLLM serves safetensors directly. Metal recommendations go
501 -> 104, all actually servable.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Remove macOS launchd LaunchAgent (cherry-picked extra)
Drop the launchd service from the PR #213 cherry-picks: the
install-service-macos.sh installer, the com.odysseus.ui.plist template, and the
README section documenting them. Tangential to the core Cookbook/Metal support
and not wanted. The build-macos-app.sh launcher is kept.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Add one-command macOS quick start (start-macos.sh)
Running Odysseus natively on a Mac previously meant ~7 manual terminal steps
(brew deps, venv, activate, pip, setup.py, uvicorn with the right port) — not
friendly for a generic macOS user, and the native run is required because Docker
on macOS can't reach the Metal GPU.
- start-macos.sh: installs Homebrew deps (python@3.11, tmux, prebuilt Metal
llama.cpp), creates the venv, installs requirements, runs setup, and launches
on a non-AirPlay port (7860). Idempotent; re-run to start again.
- README: the Apple Silicon section now leads with this one-command quick start
and the clickable .app, with engine/port/manual details folded into a
collapsible block. Added a pointer at the top of the manual-install section.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* macOS quick start: auto-open browser when ready
The "open this URL" line scrolled out of view as uvicorn kept logging after it,
so users missed it. Now start-macos.sh waits (in the background) until the
server accepts connections, prints a boxed "ready" banner at that point (i.e.
after the startup burst, not before), and opens the URL in the default browser
automatically. Skippable with ODYSSEUS_NO_OPEN=1 for headless/SSH use.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Don't assume/force a specific Python version on macOS
The README claimed "system Python is 3.9" — a machine-specific generalization
that's often wrong (macOS ships no recent Python by default; many users already
have 3.11+). Make it generic, and make start-macos.sh detect an existing
Python 3.11+ and use it, only installing python@3.11 when none is found instead
of forcing it on top of the user's Python.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Align start-macos.sh venv path with build-macos-app.sh
start-macos.sh created the environment in .venv/, but build-macos-app.sh and
the manual install steps use venv/ — so the clickable .app wouldn't reuse the
quick-start's environment and would rebuild a second one. Use venv/ everywhere.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* README: state clearly that MLX is unsupported on Apple Silicon
Odysseus has no mlx_lm runtime; it serves GGUF (llama.cpp/Ollama) and CUDA
(vLLM/SGLang) only. MLX-only models can't run on a Mac and are hidden from
Cookbook — make that explicit in both the quick start and the details.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* start-macos.sh: build the venv with an arm64 Python on Apple Silicon
A clean-room run surfaced this: with a universal2/x86 Python (e.g. the
python.org installer under /usr/local), the venv's compiled extensions install
as arm64 but get loaded as x86_64 when launched from the .app bundle, so it
crashes with "incompatible architecture (have arm64, need x86_64)". The terminal
run happened to work only because a universal binary defaults to arm64 there.
On Apple Silicon, look only under /opt/homebrew (arm64-only) for the build
Python, and install Homebrew's python@3.11 if none is present — so the venv is
arm64-only and launches correctly from both the terminal and the .app. Intel
and non-mac paths are unchanged. Verified end-to-end in a clean clone: .app now
boots on Metal with no arch error.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Address dev-exp review: macOS setup robustness + doc/UX fixes
From the voltagent dev-exp review of the branch:
- README: fix broken anchor links (the em-dash heading produced a slug the links
didn't match); simplify the heading to a stable slug.
- cookbook_routes.py: add /opt/homebrew/bin and /usr/local/bin to the serve PATH
so a brew-installed llama-server/ollama is found instead of falling back to a
slow source build.
- start-macos.sh: guard against an empty Python path; fail fast with a clear
message on port-in-use; ERR trap with a "safe to re-run" message; show pip
progress (drop --quiet on the slow requirements install); stop the background
browser-opener cleanly on exit/Ctrl+C (no orphaned poller).
- setup.py: bind hint to 127.0.0.1; suppress the manual run-hint when launched
by start-macos.sh (ODYSSEUS_SKIP_RUN_HINT) so the URL isn't contradictory.
- build-macos-app.sh: the .app only opens the browser once the server is
actually ready (not after the readiness timeout).
- cookbookServe.js: drop "Diffusers" from the Metal backend picker —
diffusion_server.py is CUDA-only, so it was an unservable option on macOS.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
---------
Co-authored-by: yunggilja <yunggilja@gmail.com>
Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-01 15:29:19 +09:30
/** Check if the detected (local) hardware is Apple Silicon / Metal . Keys off the
* hardware probe ' s backend rather than a platform string , since a local Mac
* reports no platform but does report backend : "metal" . * /
export function _isMetal ( ) {
return [ 'metal' , 'mps' , 'apple' ] . includes ( String ( _hwfitCache ? . system ? . backend || '' ) . toLowerCase ( ) ) ;
}
2026-05-31 23:58:26 +09:00
/** Detect model-specific vLLM optimizations */
function _detectModelOptimizations ( modelName ) {
const n = ( modelName || '' ) . toLowerCase ( ) ;
const opts = { envVars : [ ] , flags : [ ] , tips : [ ] } ;
// Qwen3.5 MoE models
if ( n . includes ( 'qwen3.5' ) || n . includes ( 'qwen3-' ) && ( n . includes ( 'a10b' ) || n . includes ( 'a22b' ) || n . includes ( 'a3b' ) ) ) {
opts . envVars . push ( 'VLLM_USE_DEEP_GEMM=0' , 'VLLM_USE_FLASHINFER_MOE_FP16=1' , 'VLLM_USE_FLASHINFER_SAMPLER=0' , 'OMP_NUM_THREADS=4' ) ;
opts . flags . push ( '--enable-expert-parallel' , '--reasoning-parser qwen3' ) ;
opts . tips . push ( 'MoE optimizations: expert parallel + flashinfer MoE kernels' ) ;
}
// Qwen3 MoE (non-3.5)
else if ( n . includes ( 'qwen3' ) && ( n . includes ( 'a10b' ) || n . includes ( 'a22b' ) || n . includes ( 'a3b' ) ) ) {
opts . envVars . push ( 'VLLM_USE_DEEP_GEMM=0' , 'VLLM_USE_FLASHINFER_MOE_FP16=1' ) ;
opts . flags . push ( '--enable-expert-parallel' , '--reasoning-parser qwen3' ) ;
opts . tips . push ( 'MoE optimizations: expert parallel' ) ;
}
// DeepSeek MoE
else if ( n . includes ( 'deepseek' ) && ( n . includes ( 'v3' ) || n . includes ( 'r1' ) ) ) {
opts . flags . push ( '--enable-expert-parallel' ) ;
opts . tips . push ( 'MoE expert parallel for DeepSeek' ) ;
}
// Speculative decoding — pick the right MTP method per model family.
// opts.spec.{method,tokens} seed the UI dropdown/input; the actual flag is
// assembled by the command builder so the user can edit before launching.
let specDefault = null ;
if ( n . includes ( 'qwen3-next' ) || ( n . includes ( 'qwen3.5' ) && ( n . includes ( 'a10b' ) || n . includes ( 'a22b' ) ) ) ) {
specDefault = { method : 'qwen3_next_mtp' , tokens : 2 } ;
} else if (
( n . includes ( 'deepseek' ) && ( n . includes ( 'v3' ) || n . includes ( 'v3.1' ) || n . includes ( 'r1' ) ) ) ||
n . includes ( 'kimi-k2' ) || n . includes ( 'kimi_k2' ) ||
n . includes ( 'glm-4.5' ) || n . includes ( 'glm4.5' ) ||
n . includes ( 'minimax-m1' ) || n . includes ( 'minimax_m1' )
) {
specDefault = { method : 'mtp' , tokens : 3 } ;
}
if ( specDefault ) {
opts . spec = specDefault ;
opts . flags . push ( ` --speculative-config '{"method":" ${ specDefault . method } ","num_speculative_tokens": ${ specDefault . tokens } }' ` ) ;
opts . tips . push ( ` Speculative decoding ( ${ specDefault . method } , ${ specDefault . tokens } tokens): ~1.5-2x faster generation ` ) ;
}
return opts ;
}
2026-06-02 12:47:15 +10:00
/ * * D e t e c t t h e r i g h t v L L M t o o l - c a l l - p a r s e r b a s e d o n m o d e l n a m e .
* Qwen tool - call formats split by generation :
* - Qwen3 - Coder → qwen3 _coder ( XML < tool _call > with named params )
* - Qwen3 ( non - coder ) → qwen3 _xml ( reasoning / instruct , XML wrapper )
* - Qwen2 . 5 / Qwen2 / 1.5 → hermes ( Qwen2 . 5 was trained on Hermes format )
* Catching "qwen" first and labelling everything qwen3 _xml breaks tool
* calls on the Qwen2 . 5 line ( the model emits hermes - style which the
* qwen3 _xml parser doesn ' t recognise , so the call leaks through as text ) .
* /
2026-05-31 23:58:26 +09:00
export function _detectToolParser ( modelName ) {
const n = ( modelName || '' ) . toLowerCase ( ) ;
if ( n . includes ( 'qwen3' ) && n . includes ( 'coder' ) ) return 'qwen3_coder' ;
2026-06-02 12:47:15 +10:00
if ( n . includes ( 'qwen3' ) ) return 'qwen3_xml' ;
if ( n . includes ( 'qwen' ) ) return 'hermes' ; // Qwen2.5 / Qwen2 / Qwen1.5
2026-05-31 23:58:26 +09:00
if ( n . includes ( 'llama-4' ) || n . includes ( 'llama4' ) ) return 'llama4_json' ;
if ( n . includes ( 'llama' ) || n . includes ( 'nemotron' ) ) return 'llama3_json' ;
if ( n . includes ( 'mistral' ) || n . includes ( 'mixtral' ) ) return 'mistral' ;
if ( n . includes ( 'deepseek-v3' ) ) return 'deepseek_v3' ;
if ( n . includes ( 'deepseek' ) ) return 'deepseek_v3' ;
if ( n . includes ( 'minimax' ) && n . includes ( 'm2' ) ) return 'minimax_m2' ;
if ( n . includes ( 'minimax' ) ) return 'minimax' ;
if ( n . includes ( 'gemma' ) ) return 'pythonic' ;
if ( n . includes ( 'glm-4' ) ) return 'glm45' ;
if ( n . includes ( 'internlm' ) ) return 'internlm' ;
if ( n . includes ( 'granite' ) ) return 'granite' ;
return 'hermes' ; // default fallback
}
// ── Backend detection ──
export function _detectBackend ( model ) {
2026-06-02 07:14:59 +09:00
if ( model ? . backend === 'ollama' || model ? . is _ollama ) {
return { backend : 'ollama' , label : 'Ollama' } ;
}
2026-05-31 23:58:26 +09:00
const q = ( model . quant || '' ) . toUpperCase ( ) ;
const sysBackend = String ( _hwfitCache ? . system ? . backend || '' ) . toLowerCase ( ) ;
const isRocm = sysBackend === 'rocm' ;
2026-06-02 12:15:41 +09:00
const isAppleSilicon = [ 'metal' , 'mps' , 'apple' ] . includes ( sysBackend ) ;
const _nm = ` ${ model . repo _id || '' } ${ model . path || '' } ${ model . name || '' } ` . toLowerCase ( ) ;
2026-06-02 14:07:20 +10:00
if ( /\bmlx\b|mlx-|_mlx/i . test ( _nm ) || q . startsWith ( 'MLX' ) ) {
2026-06-02 12:15:41 +09:00
return { backend : 'unsupported' , label : 'Unsupported' } ;
}
2026-06-02 14:07:20 +10:00
const isAwqLike = /^AWQ|^GPTQ|^NVFP4/ . test ( q ) || [ 'FP8' , 'FP4' , 'MXFP4' , 'NF4' , 'INT4' , 'INT8' , 'W4A16' , 'W8A8' , 'W8A16' ] . includes ( q ) || /\b(awq|gptq|fp8|fp4|nvfp4|mxfp4|nf4|int4|int8|w4a16|w8a8|w8a16)\b/i . test ( _nm ) ;
2026-06-02 12:15:41 +09:00
const isGgufLike = model . is _gguf || /^Q[2-8]/ . test ( q ) || /^IQ/ . test ( q ) || q === 'GGUF' || _nm . includes ( 'gguf' ) ;
2026-05-31 23:58:26 +09:00
// Image gen models → diffusers
if ( model . is _image _gen || model . is _diffusion || model . _tag === 'image' ) {
return { backend : 'diffusers' , label : 'Diffusers' } ;
}
2026-06-02 12:15:41 +09:00
// AWQ / GPTQ / FP8 are safetensors GPU-serving formats. Never route them
// through llama.cpp/Ollama just because the host is Mac/Windows; those engines
// need GGUF. The UI will warn/block on Metal where vLLM/SGLang aren't viable.
if ( isAwqLike ) {
return { backend : 'vllm' , label : 'vLLM' } ;
}
// GGUF → llama.cpp/Ollama-compatible.
if ( isGgufLike ) {
return { backend : 'llamacpp' , label : 'llama.cpp' } ;
}
2026-05-31 23:58:26 +09:00
// Windows → default to llama.cpp (no vLLM support on Windows)
if ( _isWindows ( ) ) {
return { backend : 'llamacpp' , label : 'llama.cpp' } ;
}
Add macOS Apple Silicon Cookbook support
* Add Apple Silicon (Metal) GPU detection and unified-memory fit tuning
hardware.py detects Apple Silicon locally and over SSH, reporting
backend=metal, the chip name, and a RAM-scaled fraction of unified
memory as the usable GPU budget. fit.py gains an M1-M4 memory-bandwidth
table for realistic tok/s and drops vLLM-only formats (AWQ/GPTQ/FP8)
that can't be served on Metal.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 32ac81dbc680361463a088dae867d555d5a79c3b)
* Generate macOS/Metal serve commands and surface the Metal GPU
cookbook_routes.py adds a macOS serve path (Ollama, Metal-aware
llama.cpp build using `sysctl hw.ncpu` instead of `nproc`, and a clear
error if vLLM is attempted). The frontend defaults Metal serving to
llama.cpp and offers llama.cpp/Ollama instead of vLLM/SGLang. The
odysseus-cookbook CLI's `gpus` command reports the Metal GPU via
sysctl/vm_stat.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 4ba01ce25d256ae032029898f361c824a34fcd4b)
* Add launchd LaunchAgent for macOS (systemd equivalent)
com.odysseus.ui.plist + install-service-macos.sh run Odysseus at login
and restart on crash, the macOS counterpart to odysseus-ui.service. The
installer auto-fills paths from the venv, so there's no hand-editing.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 3d4b6b2c7b8b31af32201ed278115df9a559dea9)
* Document macOS install (brew, Ollama, AirPlay port, launchd)
README + setup.py cover the Homebrew / Apple Silicon path: brew install
python@3.11 tmux ollama, Metal serving via Ollama/llama.cpp, the launchd
service, and the macOS AirPlay Receiver conflict on ports 7000/5000.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 8dc9a3578a1726f070ed9f75c0958ae291a6d966)
* Add downloadable macOS launcher app builder
build-macos-app.sh generates dist/Odysseus.app and a drag-to-Applications
dist/Odysseus.dmg. The app starts the local server from this repo's venv and
opens the UI in a chrome-less app window (Chromium --app mode, falling back to
the default browser). It's a launcher wrapper — it drives the venv rather than
bundling Python — so the install path is baked in at build time.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 7927940c3810ee34640803b198d334a6ac93474d)
* Harden macOS Cookbook support: hide MLX, fix Metal build cache
Builds on the adopted PR #213 macOS/Metal work with two fixes and tests:
- fit.py: always drop MLX-quantized models. Odysseus only generates serve
commands for llama.cpp/Ollama (Metal) and vLLM/SGLang (CUDA); MLX needs the
mlx_lm runtime and the catalog's MLX repos ship no GGUF alternative, so they
were surfaced on Apple Silicon but could never be served.
- cookbook_routes.py (macOS branch only): `rm -rf build` before configure so a
poisoned CMakeCache from a prior failed CUDA attempt can't make every later
build fail; explicit -DCMAKE_BUILD_TYPE=Release; a clear "brew install cmake"
hint if cmake is missing. Linux/CUDA path unchanged.
- tests/test_hwfit_macos.py: MLX hidden on metal, MLX still hidden on CUDA
(regression guard), Metal detection on Apple Silicon, and skipped on
Linux/Intel (proves non-macOS detection is untouched).
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Propagate unified_memory flag and document macOS GPU/Docker caveat
- hardware.py: detect_system now carries the unified_memory flag from GPU
detection into the system dict (it was set by _detect_apple_silicon / AMD-APU
detection but dropped during result assembly, so the API always reported
null). Lets callers distinguish unified from discrete VRAM.
- README: prominent warning that Docker on Apple Silicon can't reach the Metal
GPU (runs a Linux VM) — Cookbook must run natively for GPU serving; fix stale
text that said Cookbook recommends MLX models (now hidden as unservable).
- test: detect_system propagates unified_memory.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Put Odysseus's venv bin on PATH for cookbook runners
Native (non-Docker) installs run from a virtualenv whose bin holds the `hf` CLI
and `python3` the cookbook download/serve tmux scripts shell out to. Those
scripts start in a fresh login shell with the venv NOT activated, so on a native
macOS install `hf download` failed with "hf: command not found" — and the
`pip --user` self-heal missed because macOS has no bare `pip` command.
- cookbook_helpers.py: _local_tooling_path_export() — pure helper returning a
PATH export for the running interpreter's bin dir (escaped for double quotes).
- cookbook_routes.py: download + serve runners prepend that dir on local runs
(gated off SSH/Windows); swap the `pip` install fallbacks to `python3 -m pip`.
- tests: helper output for normal and spaced paths.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Document macOS llama.cpp serving prerequisites
Clarify the two serving paths on Apple Silicon: the recommended zero-build
route (brew install llama.cpp ships a Metal llama-server Cookbook finds on PATH),
and the from-source fallback, which requires cmake + Xcode Command Line Tools.
Without those the build is skipped and serving silently degrades to a slow CPU
build, so new users now know to install them (or use the prebuilt) up front.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Recommend only GGUF-servable models on Metal
Apple Silicon's only serving engines are llama.cpp and Ollama, both GGUF-only
(vLLM/SGLang are CUDA/ROCm and don't run on macOS). The catalog tags raw
safetensors repos with a default Q4_K_M quant, so the fit-ranking was
recommending ~397/501 models that have no GGUF and fail to serve on Metal with
"No GGUF found" (e.g. microsoft/Phi-mini-MoE-instruct).
Drop any model without a real GGUF (is_gguf/gguf_sources) on Apple Silicon —
subsumes the previous AWQ/GPTQ/FP8 special-case into one rule. On CUDA these
stay visible since vLLM serves safetensors directly. Metal recommendations go
501 -> 104, all actually servable.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Remove macOS launchd LaunchAgent (cherry-picked extra)
Drop the launchd service from the PR #213 cherry-picks: the
install-service-macos.sh installer, the com.odysseus.ui.plist template, and the
README section documenting them. Tangential to the core Cookbook/Metal support
and not wanted. The build-macos-app.sh launcher is kept.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Add one-command macOS quick start (start-macos.sh)
Running Odysseus natively on a Mac previously meant ~7 manual terminal steps
(brew deps, venv, activate, pip, setup.py, uvicorn with the right port) — not
friendly for a generic macOS user, and the native run is required because Docker
on macOS can't reach the Metal GPU.
- start-macos.sh: installs Homebrew deps (python@3.11, tmux, prebuilt Metal
llama.cpp), creates the venv, installs requirements, runs setup, and launches
on a non-AirPlay port (7860). Idempotent; re-run to start again.
- README: the Apple Silicon section now leads with this one-command quick start
and the clickable .app, with engine/port/manual details folded into a
collapsible block. Added a pointer at the top of the manual-install section.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* macOS quick start: auto-open browser when ready
The "open this URL" line scrolled out of view as uvicorn kept logging after it,
so users missed it. Now start-macos.sh waits (in the background) until the
server accepts connections, prints a boxed "ready" banner at that point (i.e.
after the startup burst, not before), and opens the URL in the default browser
automatically. Skippable with ODYSSEUS_NO_OPEN=1 for headless/SSH use.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Don't assume/force a specific Python version on macOS
The README claimed "system Python is 3.9" — a machine-specific generalization
that's often wrong (macOS ships no recent Python by default; many users already
have 3.11+). Make it generic, and make start-macos.sh detect an existing
Python 3.11+ and use it, only installing python@3.11 when none is found instead
of forcing it on top of the user's Python.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Align start-macos.sh venv path with build-macos-app.sh
start-macos.sh created the environment in .venv/, but build-macos-app.sh and
the manual install steps use venv/ — so the clickable .app wouldn't reuse the
quick-start's environment and would rebuild a second one. Use venv/ everywhere.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* README: state clearly that MLX is unsupported on Apple Silicon
Odysseus has no mlx_lm runtime; it serves GGUF (llama.cpp/Ollama) and CUDA
(vLLM/SGLang) only. MLX-only models can't run on a Mac and are hidden from
Cookbook — make that explicit in both the quick start and the details.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* start-macos.sh: build the venv with an arm64 Python on Apple Silicon
A clean-room run surfaced this: with a universal2/x86 Python (e.g. the
python.org installer under /usr/local), the venv's compiled extensions install
as arm64 but get loaded as x86_64 when launched from the .app bundle, so it
crashes with "incompatible architecture (have arm64, need x86_64)". The terminal
run happened to work only because a universal binary defaults to arm64 there.
On Apple Silicon, look only under /opt/homebrew (arm64-only) for the build
Python, and install Homebrew's python@3.11 if none is present — so the venv is
arm64-only and launches correctly from both the terminal and the .app. Intel
and non-mac paths are unchanged. Verified end-to-end in a clean clone: .app now
boots on Metal with no arch error.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Address dev-exp review: macOS setup robustness + doc/UX fixes
From the voltagent dev-exp review of the branch:
- README: fix broken anchor links (the em-dash heading produced a slug the links
didn't match); simplify the heading to a stable slug.
- cookbook_routes.py: add /opt/homebrew/bin and /usr/local/bin to the serve PATH
so a brew-installed llama-server/ollama is found instead of falling back to a
slow source build.
- start-macos.sh: guard against an empty Python path; fail fast with a clear
message on port-in-use; ERR trap with a "safe to re-run" message; show pip
progress (drop --quiet on the slow requirements install); stop the background
browser-opener cleanly on exit/Ctrl+C (no orphaned poller).
- setup.py: bind hint to 127.0.0.1; suppress the manual run-hint when launched
by start-macos.sh (ODYSSEUS_SKIP_RUN_HINT) so the URL isn't contradictory.
- build-macos-app.sh: the .app only opens the browser once the server is
actually ready (not after the readiness timeout).
- cookbookServe.js: drop "Diffusers" from the Metal backend picker —
diffusion_server.py is CUDA-only, so it was an unservable option on macOS.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
---------
Co-authored-by: yunggilja <yunggilja@gmail.com>
Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-01 15:29:19 +09:30
// Apple Silicon (Metal) → llama.cpp (GGUF). vLLM/SGLang are CUDA/ROCm-only and
2026-06-02 14:07:20 +10:00
// don't run on macOS; vLLM-native quantized models are already filtered out
Add macOS Apple Silicon Cookbook support
* Add Apple Silicon (Metal) GPU detection and unified-memory fit tuning
hardware.py detects Apple Silicon locally and over SSH, reporting
backend=metal, the chip name, and a RAM-scaled fraction of unified
memory as the usable GPU budget. fit.py gains an M1-M4 memory-bandwidth
table for realistic tok/s and drops vLLM-only formats (AWQ/GPTQ/FP8)
that can't be served on Metal.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 32ac81dbc680361463a088dae867d555d5a79c3b)
* Generate macOS/Metal serve commands and surface the Metal GPU
cookbook_routes.py adds a macOS serve path (Ollama, Metal-aware
llama.cpp build using `sysctl hw.ncpu` instead of `nproc`, and a clear
error if vLLM is attempted). The frontend defaults Metal serving to
llama.cpp and offers llama.cpp/Ollama instead of vLLM/SGLang. The
odysseus-cookbook CLI's `gpus` command reports the Metal GPU via
sysctl/vm_stat.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 4ba01ce25d256ae032029898f361c824a34fcd4b)
* Add launchd LaunchAgent for macOS (systemd equivalent)
com.odysseus.ui.plist + install-service-macos.sh run Odysseus at login
and restart on crash, the macOS counterpart to odysseus-ui.service. The
installer auto-fills paths from the venv, so there's no hand-editing.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 3d4b6b2c7b8b31af32201ed278115df9a559dea9)
* Document macOS install (brew, Ollama, AirPlay port, launchd)
README + setup.py cover the Homebrew / Apple Silicon path: brew install
python@3.11 tmux ollama, Metal serving via Ollama/llama.cpp, the launchd
service, and the macOS AirPlay Receiver conflict on ports 7000/5000.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 8dc9a3578a1726f070ed9f75c0958ae291a6d966)
* Add downloadable macOS launcher app builder
build-macos-app.sh generates dist/Odysseus.app and a drag-to-Applications
dist/Odysseus.dmg. The app starts the local server from this repo's venv and
opens the UI in a chrome-less app window (Chromium --app mode, falling back to
the default browser). It's a launcher wrapper — it drives the venv rather than
bundling Python — so the install path is baked in at build time.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 7927940c3810ee34640803b198d334a6ac93474d)
* Harden macOS Cookbook support: hide MLX, fix Metal build cache
Builds on the adopted PR #213 macOS/Metal work with two fixes and tests:
- fit.py: always drop MLX-quantized models. Odysseus only generates serve
commands for llama.cpp/Ollama (Metal) and vLLM/SGLang (CUDA); MLX needs the
mlx_lm runtime and the catalog's MLX repos ship no GGUF alternative, so they
were surfaced on Apple Silicon but could never be served.
- cookbook_routes.py (macOS branch only): `rm -rf build` before configure so a
poisoned CMakeCache from a prior failed CUDA attempt can't make every later
build fail; explicit -DCMAKE_BUILD_TYPE=Release; a clear "brew install cmake"
hint if cmake is missing. Linux/CUDA path unchanged.
- tests/test_hwfit_macos.py: MLX hidden on metal, MLX still hidden on CUDA
(regression guard), Metal detection on Apple Silicon, and skipped on
Linux/Intel (proves non-macOS detection is untouched).
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Propagate unified_memory flag and document macOS GPU/Docker caveat
- hardware.py: detect_system now carries the unified_memory flag from GPU
detection into the system dict (it was set by _detect_apple_silicon / AMD-APU
detection but dropped during result assembly, so the API always reported
null). Lets callers distinguish unified from discrete VRAM.
- README: prominent warning that Docker on Apple Silicon can't reach the Metal
GPU (runs a Linux VM) — Cookbook must run natively for GPU serving; fix stale
text that said Cookbook recommends MLX models (now hidden as unservable).
- test: detect_system propagates unified_memory.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Put Odysseus's venv bin on PATH for cookbook runners
Native (non-Docker) installs run from a virtualenv whose bin holds the `hf` CLI
and `python3` the cookbook download/serve tmux scripts shell out to. Those
scripts start in a fresh login shell with the venv NOT activated, so on a native
macOS install `hf download` failed with "hf: command not found" — and the
`pip --user` self-heal missed because macOS has no bare `pip` command.
- cookbook_helpers.py: _local_tooling_path_export() — pure helper returning a
PATH export for the running interpreter's bin dir (escaped for double quotes).
- cookbook_routes.py: download + serve runners prepend that dir on local runs
(gated off SSH/Windows); swap the `pip` install fallbacks to `python3 -m pip`.
- tests: helper output for normal and spaced paths.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Document macOS llama.cpp serving prerequisites
Clarify the two serving paths on Apple Silicon: the recommended zero-build
route (brew install llama.cpp ships a Metal llama-server Cookbook finds on PATH),
and the from-source fallback, which requires cmake + Xcode Command Line Tools.
Without those the build is skipped and serving silently degrades to a slow CPU
build, so new users now know to install them (or use the prebuilt) up front.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Recommend only GGUF-servable models on Metal
Apple Silicon's only serving engines are llama.cpp and Ollama, both GGUF-only
(vLLM/SGLang are CUDA/ROCm and don't run on macOS). The catalog tags raw
safetensors repos with a default Q4_K_M quant, so the fit-ranking was
recommending ~397/501 models that have no GGUF and fail to serve on Metal with
"No GGUF found" (e.g. microsoft/Phi-mini-MoE-instruct).
Drop any model without a real GGUF (is_gguf/gguf_sources) on Apple Silicon —
subsumes the previous AWQ/GPTQ/FP8 special-case into one rule. On CUDA these
stay visible since vLLM serves safetensors directly. Metal recommendations go
501 -> 104, all actually servable.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Remove macOS launchd LaunchAgent (cherry-picked extra)
Drop the launchd service from the PR #213 cherry-picks: the
install-service-macos.sh installer, the com.odysseus.ui.plist template, and the
README section documenting them. Tangential to the core Cookbook/Metal support
and not wanted. The build-macos-app.sh launcher is kept.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Add one-command macOS quick start (start-macos.sh)
Running Odysseus natively on a Mac previously meant ~7 manual terminal steps
(brew deps, venv, activate, pip, setup.py, uvicorn with the right port) — not
friendly for a generic macOS user, and the native run is required because Docker
on macOS can't reach the Metal GPU.
- start-macos.sh: installs Homebrew deps (python@3.11, tmux, prebuilt Metal
llama.cpp), creates the venv, installs requirements, runs setup, and launches
on a non-AirPlay port (7860). Idempotent; re-run to start again.
- README: the Apple Silicon section now leads with this one-command quick start
and the clickable .app, with engine/port/manual details folded into a
collapsible block. Added a pointer at the top of the manual-install section.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* macOS quick start: auto-open browser when ready
The "open this URL" line scrolled out of view as uvicorn kept logging after it,
so users missed it. Now start-macos.sh waits (in the background) until the
server accepts connections, prints a boxed "ready" banner at that point (i.e.
after the startup burst, not before), and opens the URL in the default browser
automatically. Skippable with ODYSSEUS_NO_OPEN=1 for headless/SSH use.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Don't assume/force a specific Python version on macOS
The README claimed "system Python is 3.9" — a machine-specific generalization
that's often wrong (macOS ships no recent Python by default; many users already
have 3.11+). Make it generic, and make start-macos.sh detect an existing
Python 3.11+ and use it, only installing python@3.11 when none is found instead
of forcing it on top of the user's Python.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Align start-macos.sh venv path with build-macos-app.sh
start-macos.sh created the environment in .venv/, but build-macos-app.sh and
the manual install steps use venv/ — so the clickable .app wouldn't reuse the
quick-start's environment and would rebuild a second one. Use venv/ everywhere.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* README: state clearly that MLX is unsupported on Apple Silicon
Odysseus has no mlx_lm runtime; it serves GGUF (llama.cpp/Ollama) and CUDA
(vLLM/SGLang) only. MLX-only models can't run on a Mac and are hidden from
Cookbook — make that explicit in both the quick start and the details.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* start-macos.sh: build the venv with an arm64 Python on Apple Silicon
A clean-room run surfaced this: with a universal2/x86 Python (e.g. the
python.org installer under /usr/local), the venv's compiled extensions install
as arm64 but get loaded as x86_64 when launched from the .app bundle, so it
crashes with "incompatible architecture (have arm64, need x86_64)". The terminal
run happened to work only because a universal binary defaults to arm64 there.
On Apple Silicon, look only under /opt/homebrew (arm64-only) for the build
Python, and install Homebrew's python@3.11 if none is present — so the venv is
arm64-only and launches correctly from both the terminal and the .app. Intel
and non-mac paths are unchanged. Verified end-to-end in a clean clone: .app now
boots on Metal with no arch error.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Address dev-exp review: macOS setup robustness + doc/UX fixes
From the voltagent dev-exp review of the branch:
- README: fix broken anchor links (the em-dash heading produced a slug the links
didn't match); simplify the heading to a stable slug.
- cookbook_routes.py: add /opt/homebrew/bin and /usr/local/bin to the serve PATH
so a brew-installed llama-server/ollama is found instead of falling back to a
slow source build.
- start-macos.sh: guard against an empty Python path; fail fast with a clear
message on port-in-use; ERR trap with a "safe to re-run" message; show pip
progress (drop --quiet on the slow requirements install); stop the background
browser-opener cleanly on exit/Ctrl+C (no orphaned poller).
- setup.py: bind hint to 127.0.0.1; suppress the manual run-hint when launched
by start-macos.sh (ODYSSEUS_SKIP_RUN_HINT) so the URL isn't contradictory.
- build-macos-app.sh: the .app only opens the browser once the server is
actually ready (not after the readiness timeout).
- cookbookServe.js: drop "Diffusers" from the Metal backend picker —
diffusion_server.py is CUDA-only, so it was an unservable option on macOS.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
---------
Co-authored-by: yunggilja <yunggilja@gmail.com>
Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-01 15:29:19 +09:30
// of metal Cookbook results, so llama.cpp is always the right engine here.
if ( [ 'metal' , 'mps' , 'apple' ] . includes ( sysBackend ) ) {
return { backend : 'llamacpp' , label : 'llama.cpp' } ;
}
2026-05-31 23:58:26 +09:00
// ROCm/AMD machines should not blindly default HF safetensors models to
// vLLM. SGLang is the safer OpenAI-compatible default for plain HF text
// repos there; llama.cpp still wins above whenever the model is GGUF.
if ( isRocm ) {
return { backend : 'sglang' , label : 'SGLang' } ;
}
// Unquantized / BF16 / F16 → vLLM
return { backend : 'vllm' , label : 'vLLM' } ;
}
// ── Command builders ──
export function _shellQuote ( value ) {
return "'" + String ( value ? ? '' ) . replace ( /'/g , "'\\''" ) + "'" ;
}
export function _psQuote ( value ) {
return "'" + String ( value ? ? '' ) . replace ( /'/g , "''" ) + "'" ;
}
export function _buildEnvPrefix ( ) {
if ( _isWindows ( ) ) return _buildEnvPrefixWindows ( ) ;
let parts = [ ] ;
if ( _envState . env === 'venv' && _envState . envPath ) {
const p = _envState . envPath ;
const activate = p . endsWith ( '/bin/activate' ) ? p : p + '/bin/activate' ;
parts . push ( 'source ' + _shellQuote ( activate ) ) ;
} else if ( _envState . env === 'conda' && _envState . envPath ) {
parts . push ( 'eval "$(conda shell.bash hook)" && conda activate ' + _shellQuote ( _envState . envPath ) ) ;
}
let envVars = [ ] ;
if ( _envState . hfToken ) envVars . push ( 'export HF_TOKEN=' + _shellQuote ( _envState . hfToken ) ) ;
if ( _envState . gpus ) envVars . push ( 'export CUDA_VISIBLE_DEVICES=' + _shellQuote ( _envState . gpus ) ) ;
if ( envVars . length ) parts . push ( envVars . join ( ' && ' ) ) ;
if ( parts . length === 0 ) return '' ;
return parts . join ( ' && ' ) + ' &&' ;
}
function _buildEnvPrefixWindows ( ) {
let parts = [ ] ;
if ( _envState . env === 'venv' && _envState . envPath ) {
const p = _envState . envPath ;
const activate = p . endsWith ( '\\Scripts\\Activate.ps1' ) ? p : p + '\\Scripts\\Activate.ps1' ;
parts . push ( '& ' + _psQuote ( activate ) ) ;
} else if ( _envState . env === 'conda' && _envState . envPath ) {
parts . push ( 'conda activate ' + _psQuote ( _envState . envPath ) ) ;
}
if ( _envState . hfToken ) parts . push ( '$env:HF_TOKEN=' + _psQuote ( _envState . hfToken ) ) ;
if ( _envState . gpus ) parts . push ( '$env:CUDA_VISIBLE_DEVICES=' + _psQuote ( _envState . gpus ) ) ;
if ( parts . length === 0 ) return '' ;
return parts . join ( '; ' ) + ';' ;
}
export function _buildServeCmd ( f , modelName , backend ) {
let cmd = '' ;
if ( backend === 'vllm' ) {
const gpuId = f . gpu _id ? . trim ( ) || '' ;
if ( gpuId ) cmd += ` CUDA_VISIBLE_DEVICES= ${ gpuId } ` ;
if ( f . moe _env ) {
const _opts = _detectModelOptimizations ( modelName ) ;
if ( _opts . envVars . length ) cmd += _opts . envVars . join ( ' ' ) + ' ' ;
}
cmd += ` vllm serve ${ modelName } --host 0.0.0.0 --port ${ f . port || '8000' } ` ;
cmd += ` --tensor-parallel-size ${ f . tp || '1' } ` ;
cmd += ` --max-model-len ${ f . ctx || '8192' } ` ;
cmd += ` --gpu-memory-utilization ${ f . gpu _mem || '0.90' } ` ;
if ( f . swap && f . swap !== '0' ) cmd += ` --swap-space ${ f . swap } ` ;
cmd += ` --dtype ${ f . dtype || 'auto' } ` ;
2026-06-03 00:17:16 +10:00
const _kv = ( f . vllm _kv _cache _dtype ? ? '' ) . toString ( ) . trim ( ) ;
if ( _kv === 'fp8' ) cmd += ' --kv-cache-dtype fp8' ;
2026-05-31 23:58:26 +09:00
if ( f . max _seqs && f . max _seqs . toString ( ) . trim ( ) ) cmd += ` --max-num-seqs ${ f . max _seqs . toString ( ) . trim ( ) } ` ;
if ( f . enforce _eager ) cmd += ' --enforce-eager' ;
if ( f . trust _remote ) cmd += ' --trust-remote-code' ;
if ( f . prefix _cache ) cmd += ' --enable-prefix-caching' ;
if ( f . auto _tool ) cmd += ` --enable-auto-tool-choice --tool-call-parser ${ _detectToolParser ( modelName ) } ` ;
if ( f . expert _parallel ) cmd += ' --enable-expert-parallel' ;
if ( f . reasoning _parser ) {
const rp = typeof f . reasoning _parser === 'string' && f . reasoning _parser !== 'true'
? f . reasoning _parser : ( f . _reasoning _parser _value || 'qwen3' ) ;
cmd += ` --reasoning-parser ${ rp } ` ;
}
if ( f . speculative ) {
const _specMethod = ( f . spec _method || 'mtp' ) . trim ( ) || 'mtp' ;
const _specToksRaw = parseInt ( f . spec _tokens , 10 ) ;
const _specToks = ( Number . isFinite ( _specToksRaw ) && _specToksRaw > 0 ) ? _specToksRaw : 3 ;
cmd += ` --speculative-config '{"method":" ${ _specMethod } ","num_speculative_tokens": ${ _specToks } }' ` ;
}
} else if ( backend === 'sglang' ) {
const gpuId = f . gpu _id ? . trim ( ) || '' ;
if ( gpuId ) cmd += ` CUDA_VISIBLE_DEVICES= ${ gpuId } ` ;
cmd += ` python3 -m sglang.launch_server --model-path ${ modelName } --host 0.0.0.0 --port ${ f . port || '30000' } ` ;
if ( f . tp && f . tp !== '1' ) cmd += ` --tp ${ f . tp } ` ;
if ( f . ctx ) cmd += ` --context-length ${ f . ctx } ` ;
if ( f . gpu _mem && f . gpu _mem !== '0.90' ) cmd += ` --mem-fraction-static ${ f . gpu _mem } ` ;
if ( f . dtype && f . dtype !== 'auto' ) cmd += ` --dtype ${ f . dtype } ` ;
if ( f . max _seqs && f . max _seqs . toString ( ) . trim ( ) ) cmd += ` --max-running-requests ${ f . max _seqs . toString ( ) . trim ( ) } ` ;
if ( f . trust _remote ) cmd += ' --trust-remote-code' ;
if ( ! f . prefix _cache ) cmd += ' --disable-radix-cache' ;
if ( f . enforce _eager ) cmd += ' --disable-cuda-graph' ;
} else if ( backend === 'llamacpp' ) {
const ggufPath = f . _gguf _path || 'model.gguf' ;
const gpuId = f . gpu _id ? . trim ( ) || '' ;
const py = _isWindows ( ) ? 'python' : 'python3' ;
2026-06-03 03:26:15 +08:00
// CPU-only serve (-ngl 0): drop the GPU-only flags, otherwise the command
// mixes "zero GPU layers" with CUDA unified-memory + flash-attn and fails to
// start (issue #1291). Only affects the ngl=0 path; GPU serving is unchanged.
const _cpuOnly = String ( f . ngl ) . trim ( ) === '0' ;
2026-05-31 23:58:26 +09:00
const lcPrefix = ( ( ) => {
let p = '' ;
2026-06-03 03:26:15 +08:00
if ( f . unified _mem && ! _cpuOnly && ! _isWindows ( ) ) p += ` GGML_CUDA_ENABLE_UNIFIED_MEMORY=1 ` ;
2026-05-31 23:58:26 +09:00
if ( gpuId && ! _isWindows ( ) ) p += ` CUDA_VISIBLE_DEVICES= ${ gpuId } ` ;
return p ;
} ) ( ) ;
2026-06-03 03:26:15 +08:00
if ( f . unified _mem && ! _cpuOnly && _isWindows ( ) ) cmd += ` $ env:GGML_CUDA_ENABLE_UNIFIED_MEMORY="1"; ` ;
2026-05-31 23:58:26 +09:00
if ( gpuId && _isWindows ( ) ) cmd += ` $ env:CUDA_VISIBLE_DEVICES=" ${ gpuId } "; ` ;
if ( ! _isWindows ( ) ) {
// Resolve GGUF path once, fail loudly if nothing matched (prevents
// `--model ""` which causes confusing downstream errors).
cmd += ` MODEL_FILE= ${ ggufPath } && { [ -n " $ MODEL_FILE" ] && [ -f " $ MODEL_FILE" ]; } || { echo "ERROR: No GGUF found on this host. Either download the model here, or switch to the server where it's cached."; exit 1; } && ` ;
}
const modelArg = _isWindows ( ) ? ` " ${ ggufPath } " ` : ` " $ MODEL_FILE" ` ;
// Prefer the native llama-server binary on Linux — its minja templating
// renders modern GGUF chat templates that the Python bindings' Jinja2
// rejects (do_tojson ensure_ascii). Fall back to llama_cpp.server.
// Don't suppress stderr — surface real errors (missing file, lib, OOM).
Cookbook serve profiles and engine filter
* Cookbook: Engine filter + intelligent hardware-computed serve profiles
Two related Cookbook serving improvements for accurate, hardware-aware model
serving (especially on consumer GPUs that can only run GGUF/llama.cpp).
Engine filter
- New "Engine" dropdown (All / llama.cpp / vLLM / SGLang) beside the quant
picker. Pure client-side view filter over the fetched list via the same
_detectBackend() the serve commands use, so what you filter to is exactly what
would launch. Re-renders from cache (no refetch). Empty-state message + the
instant-cache-paint path account for it too.
Intelligent serve profiles (Quality / Balanced / Speed)
- services/hwfit/profiles.py: compute_serve_profiles() turns detected VRAM +
model size into concrete llama.cpp flags (n_gpu_layers, n_cpu_moe, cache-type,
context). Encodes the by-hand tuning: a too-big MoE offloads experts to CPU
instead of failing; a model that fits stays fully on GPU; quant tracks profile
intent; vision models keep image-encoder headroom. Reuses models.py VRAM math
so filtering and serving agree on what fits. Pure/deterministic (no t/s claims
— partial-offload speed isn't reliably predictable; fit is what's computed).
- /api/hwfit/profiles endpoint returns the profiles + the model's trained
context limit, with loose name matching (strips org/ prefix, -GGUF suffix,
quant tag) so a local GGUF folder name resolves to its catalog entry.
- _buildServeCmd (llama.cpp) now emits --n-cpu-moe / --flash-attn /
--cache-type-k/v when set, with llama-cpp-python fallback equivalents. It
previously only set -ngl/-c, which is why it OOM'd or ran slow.
- Serve panel: profile chips that fill the fields on click, plus CPU-MoE / KV
Cache / Flash Attn fields. Context is clamped to the model's trained limit
(and an absolute 1M sanity ceiling) on type/blur/profile-load and at launch —
fixes a crash where a stale 256k/16M preset + quantized KV cache caused an
amdgpu ErrorDeviceLost.
Tests: tests/test_serve_profiles.py (7) — offload vs full-GPU fit, never exceed
VRAM, context cap, launchable flags, vision headroom, no-GPU empty.
Checks: py_compile + node --check pass; pytest test_serve_profiles + test_hwfit_amd
green; verified live on an RDNA4 box (gfx1200) — Balanced lands ~ncm18 q4 128k,
matching hand-tuning.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
* Cookbook: make column-header sorting discoverable (incl. Newest)
Sorting in Cookbook is via clickable column headers (pewds' design), but the
headers had no visual cue that they're interactive — so sorting in general, and
the Newest sort on the Model header specifically, was undiscoverable.
- Style sortable headers as interactive: pointer cursor, hover underline, and
the active sort column bolded/highlighted. There was no CSS for
.hwfit-sortable / .hwfit-sort-active at all; this helps every existing sort,
not just Newest.
- The Model column header sorts by release_date (newest first), reusing the
existing header-click sort wiring and the "newest" SORT_KEY.
No new sort control — uses the existing column-header paradigm.
Checks: node --check passes.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
* Cookbook serve profiles: keep the on-disk file's quant fixed (don't propose Q6/Q2)
In the Serve tab the model is a specific GGUF file already on disk, so its quant
can't change — but the profiles were suggesting "Quality · Q6_K" / "Speed · Q2_K"
as if you could re-quantize it. That's meaningless when serving a fixed file.
- compute_serve_profiles gains serve_weights_gb / serve_quant. When set (SERVE
mode), the quant is locked to the file's and profiles differ only in the real
serving knobs — n_cpu_moe, KV-cache type, context. _weights_gb / _cpu_moe_for_budget
use the file's actual size instead of a quant-derived estimate. DOWNLOAD mode
(no override) still varies the quant to show download options.
- /api/hwfit/profiles accepts serve_weights_gb & serve_quant.
- The Serve panel parses the file's size (from m.size "20.6 GB") and quant (from
the repo/file name) and passes them, so profiles match what's actually served.
Result for a 20.6 GB Q4_K_M file: all three profiles stay Q4_K_M and differ by
KV/ctx/offload (Quality q8 KV 128k ncm21, Balanced q4 128k ncm17, Speed q4 32k
ncm15) — no nonsensical quant changes.
Tests: test_serve_mode_keeps_fixed_quant. Full serve-profile suite green (9).
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
* Cookbook serve: Vision toggle (auto-find mmproj) + live VRAM/RAM-spillover monitor
Two serve-panel additions:
1. **Vision toggle.** A "Vision" checkbox that serves the model with its
multimodal projector so it can read images. The mmproj path is resolved at
runtime (find mmproj-*.gguf next to the model), so dropping an mmproj file in
the model folder makes the toggle just work; `--mmproj … --image-max-tokens
1024` (native) / `--clip_model_path` (llama-cpp-python) only when on + found.
2. **Live GPU-memory monitor.** A readout that polls /api/cookbook/gpus every 4s
while the panel is open and shows VRAM used/total/%, free, and — crucially on
a discrete card — **RAM spillover** (AMD gtt_used_mb), with a plain-language
health hint: green/healthy, amber/tight, red/"spilled to RAM — slow (raise
CPU MoE or lower context)". Surfaces gtt_used_mb from the gpus endpoint
(previously read for total only and discarded for 'used').
Lets you see at a glance whether a config fits VRAM (fast) or is paging to system
RAM over PCIe (slow) instead of guessing.
Checks: node --check + py_compile pass.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
---------
Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-02 05:34:42 +02:00
// Optional perf/fit flags from a hardware profile (see services/hwfit/
// profiles.py). n_cpu_moe offloads MoE expert layers to CPU when the model
// is bigger than VRAM; flash-attn + a quantized KV cache cut KV memory and
// speed things up. Only emitted when set, so manual/older flows are unchanged.
const _ncm = ( f . n _cpu _moe ? ? '' ) . toString ( ) . trim ( ) ;
const _kv = ( f . cache _type ? ? '' ) . toString ( ) . trim ( ) ;
2026-06-02 13:46:16 +10:00
const _llamaNum = ( v ) => {
const s = String ( v || '' ) . trim ( ) ;
return /^\d+$/ . test ( s ) ? s : '' ;
} ;
const _llamaCsv = ( v ) => {
const s = String ( v || '' ) . replace ( /\s+/g , '' ) ;
return /^\d+(?:\.\d+)?(?:,\d+(?:\.\d+)?)*$/ . test ( s ) ? s : '' ;
} ;
Cookbook serve profiles and engine filter
* Cookbook: Engine filter + intelligent hardware-computed serve profiles
Two related Cookbook serving improvements for accurate, hardware-aware model
serving (especially on consumer GPUs that can only run GGUF/llama.cpp).
Engine filter
- New "Engine" dropdown (All / llama.cpp / vLLM / SGLang) beside the quant
picker. Pure client-side view filter over the fetched list via the same
_detectBackend() the serve commands use, so what you filter to is exactly what
would launch. Re-renders from cache (no refetch). Empty-state message + the
instant-cache-paint path account for it too.
Intelligent serve profiles (Quality / Balanced / Speed)
- services/hwfit/profiles.py: compute_serve_profiles() turns detected VRAM +
model size into concrete llama.cpp flags (n_gpu_layers, n_cpu_moe, cache-type,
context). Encodes the by-hand tuning: a too-big MoE offloads experts to CPU
instead of failing; a model that fits stays fully on GPU; quant tracks profile
intent; vision models keep image-encoder headroom. Reuses models.py VRAM math
so filtering and serving agree on what fits. Pure/deterministic (no t/s claims
— partial-offload speed isn't reliably predictable; fit is what's computed).
- /api/hwfit/profiles endpoint returns the profiles + the model's trained
context limit, with loose name matching (strips org/ prefix, -GGUF suffix,
quant tag) so a local GGUF folder name resolves to its catalog entry.
- _buildServeCmd (llama.cpp) now emits --n-cpu-moe / --flash-attn /
--cache-type-k/v when set, with llama-cpp-python fallback equivalents. It
previously only set -ngl/-c, which is why it OOM'd or ran slow.
- Serve panel: profile chips that fill the fields on click, plus CPU-MoE / KV
Cache / Flash Attn fields. Context is clamped to the model's trained limit
(and an absolute 1M sanity ceiling) on type/blur/profile-load and at launch —
fixes a crash where a stale 256k/16M preset + quantized KV cache caused an
amdgpu ErrorDeviceLost.
Tests: tests/test_serve_profiles.py (7) — offload vs full-GPU fit, never exceed
VRAM, context cap, launchable flags, vision headroom, no-GPU empty.
Checks: py_compile + node --check pass; pytest test_serve_profiles + test_hwfit_amd
green; verified live on an RDNA4 box (gfx1200) — Balanced lands ~ncm18 q4 128k,
matching hand-tuning.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
* Cookbook: make column-header sorting discoverable (incl. Newest)
Sorting in Cookbook is via clickable column headers (pewds' design), but the
headers had no visual cue that they're interactive — so sorting in general, and
the Newest sort on the Model header specifically, was undiscoverable.
- Style sortable headers as interactive: pointer cursor, hover underline, and
the active sort column bolded/highlighted. There was no CSS for
.hwfit-sortable / .hwfit-sort-active at all; this helps every existing sort,
not just Newest.
- The Model column header sorts by release_date (newest first), reusing the
existing header-click sort wiring and the "newest" SORT_KEY.
No new sort control — uses the existing column-header paradigm.
Checks: node --check passes.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
* Cookbook serve profiles: keep the on-disk file's quant fixed (don't propose Q6/Q2)
In the Serve tab the model is a specific GGUF file already on disk, so its quant
can't change — but the profiles were suggesting "Quality · Q6_K" / "Speed · Q2_K"
as if you could re-quantize it. That's meaningless when serving a fixed file.
- compute_serve_profiles gains serve_weights_gb / serve_quant. When set (SERVE
mode), the quant is locked to the file's and profiles differ only in the real
serving knobs — n_cpu_moe, KV-cache type, context. _weights_gb / _cpu_moe_for_budget
use the file's actual size instead of a quant-derived estimate. DOWNLOAD mode
(no override) still varies the quant to show download options.
- /api/hwfit/profiles accepts serve_weights_gb & serve_quant.
- The Serve panel parses the file's size (from m.size "20.6 GB") and quant (from
the repo/file name) and passes them, so profiles match what's actually served.
Result for a 20.6 GB Q4_K_M file: all three profiles stay Q4_K_M and differ by
KV/ctx/offload (Quality q8 KV 128k ncm21, Balanced q4 128k ncm17, Speed q4 32k
ncm15) — no nonsensical quant changes.
Tests: test_serve_mode_keeps_fixed_quant. Full serve-profile suite green (9).
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
* Cookbook serve: Vision toggle (auto-find mmproj) + live VRAM/RAM-spillover monitor
Two serve-panel additions:
1. **Vision toggle.** A "Vision" checkbox that serves the model with its
multimodal projector so it can read images. The mmproj path is resolved at
runtime (find mmproj-*.gguf next to the model), so dropping an mmproj file in
the model folder makes the toggle just work; `--mmproj … --image-max-tokens
1024` (native) / `--clip_model_path` (llama-cpp-python) only when on + found.
2. **Live GPU-memory monitor.** A readout that polls /api/cookbook/gpus every 4s
while the panel is open and shows VRAM used/total/%, free, and — crucially on
a discrete card — **RAM spillover** (AMD gtt_used_mb), with a plain-language
health hint: green/healthy, amber/tight, red/"spilled to RAM — slow (raise
CPU MoE or lower context)". Surfaces gtt_used_mb from the gpus endpoint
(previously read for total only and discarded for 'used').
Lets you see at a glance whether a config fits VRAM (fast) or is paging to system
RAM over PCIe (slow) instead of guessing.
Checks: node --check + py_compile pass.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
---------
Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-02 05:34:42 +02:00
let _lcExtra = '' ;
let _lcpExtra = '' ;
if ( _ncm !== '' && Number ( _ncm ) > 0 ) {
_lcExtra += ` --n-cpu-moe ${ _ncm } ` ;
_lcpExtra += ` --n_cpu_moe ${ _ncm } ` ; // llama-cpp-python uses underscores
}
2026-06-03 03:26:15 +08:00
if ( f . flash _attn && ! _cpuOnly ) {
Cookbook serve profiles and engine filter
* Cookbook: Engine filter + intelligent hardware-computed serve profiles
Two related Cookbook serving improvements for accurate, hardware-aware model
serving (especially on consumer GPUs that can only run GGUF/llama.cpp).
Engine filter
- New "Engine" dropdown (All / llama.cpp / vLLM / SGLang) beside the quant
picker. Pure client-side view filter over the fetched list via the same
_detectBackend() the serve commands use, so what you filter to is exactly what
would launch. Re-renders from cache (no refetch). Empty-state message + the
instant-cache-paint path account for it too.
Intelligent serve profiles (Quality / Balanced / Speed)
- services/hwfit/profiles.py: compute_serve_profiles() turns detected VRAM +
model size into concrete llama.cpp flags (n_gpu_layers, n_cpu_moe, cache-type,
context). Encodes the by-hand tuning: a too-big MoE offloads experts to CPU
instead of failing; a model that fits stays fully on GPU; quant tracks profile
intent; vision models keep image-encoder headroom. Reuses models.py VRAM math
so filtering and serving agree on what fits. Pure/deterministic (no t/s claims
— partial-offload speed isn't reliably predictable; fit is what's computed).
- /api/hwfit/profiles endpoint returns the profiles + the model's trained
context limit, with loose name matching (strips org/ prefix, -GGUF suffix,
quant tag) so a local GGUF folder name resolves to its catalog entry.
- _buildServeCmd (llama.cpp) now emits --n-cpu-moe / --flash-attn /
--cache-type-k/v when set, with llama-cpp-python fallback equivalents. It
previously only set -ngl/-c, which is why it OOM'd or ran slow.
- Serve panel: profile chips that fill the fields on click, plus CPU-MoE / KV
Cache / Flash Attn fields. Context is clamped to the model's trained limit
(and an absolute 1M sanity ceiling) on type/blur/profile-load and at launch —
fixes a crash where a stale 256k/16M preset + quantized KV cache caused an
amdgpu ErrorDeviceLost.
Tests: tests/test_serve_profiles.py (7) — offload vs full-GPU fit, never exceed
VRAM, context cap, launchable flags, vision headroom, no-GPU empty.
Checks: py_compile + node --check pass; pytest test_serve_profiles + test_hwfit_amd
green; verified live on an RDNA4 box (gfx1200) — Balanced lands ~ncm18 q4 128k,
matching hand-tuning.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
* Cookbook: make column-header sorting discoverable (incl. Newest)
Sorting in Cookbook is via clickable column headers (pewds' design), but the
headers had no visual cue that they're interactive — so sorting in general, and
the Newest sort on the Model header specifically, was undiscoverable.
- Style sortable headers as interactive: pointer cursor, hover underline, and
the active sort column bolded/highlighted. There was no CSS for
.hwfit-sortable / .hwfit-sort-active at all; this helps every existing sort,
not just Newest.
- The Model column header sorts by release_date (newest first), reusing the
existing header-click sort wiring and the "newest" SORT_KEY.
No new sort control — uses the existing column-header paradigm.
Checks: node --check passes.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
* Cookbook serve profiles: keep the on-disk file's quant fixed (don't propose Q6/Q2)
In the Serve tab the model is a specific GGUF file already on disk, so its quant
can't change — but the profiles were suggesting "Quality · Q6_K" / "Speed · Q2_K"
as if you could re-quantize it. That's meaningless when serving a fixed file.
- compute_serve_profiles gains serve_weights_gb / serve_quant. When set (SERVE
mode), the quant is locked to the file's and profiles differ only in the real
serving knobs — n_cpu_moe, KV-cache type, context. _weights_gb / _cpu_moe_for_budget
use the file's actual size instead of a quant-derived estimate. DOWNLOAD mode
(no override) still varies the quant to show download options.
- /api/hwfit/profiles accepts serve_weights_gb & serve_quant.
- The Serve panel parses the file's size (from m.size "20.6 GB") and quant (from
the repo/file name) and passes them, so profiles match what's actually served.
Result for a 20.6 GB Q4_K_M file: all three profiles stay Q4_K_M and differ by
KV/ctx/offload (Quality q8 KV 128k ncm21, Balanced q4 128k ncm17, Speed q4 32k
ncm15) — no nonsensical quant changes.
Tests: test_serve_mode_keeps_fixed_quant. Full serve-profile suite green (9).
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
* Cookbook serve: Vision toggle (auto-find mmproj) + live VRAM/RAM-spillover monitor
Two serve-panel additions:
1. **Vision toggle.** A "Vision" checkbox that serves the model with its
multimodal projector so it can read images. The mmproj path is resolved at
runtime (find mmproj-*.gguf next to the model), so dropping an mmproj file in
the model folder makes the toggle just work; `--mmproj … --image-max-tokens
1024` (native) / `--clip_model_path` (llama-cpp-python) only when on + found.
2. **Live GPU-memory monitor.** A readout that polls /api/cookbook/gpus every 4s
while the panel is open and shows VRAM used/total/%, free, and — crucially on
a discrete card — **RAM spillover** (AMD gtt_used_mb), with a plain-language
health hint: green/healthy, amber/tight, red/"spilled to RAM — slow (raise
CPU MoE or lower context)". Surfaces gtt_used_mb from the gpus endpoint
(previously read for total only and discarded for 'used').
Lets you see at a glance whether a config fits VRAM (fast) or is paging to system
RAM over PCIe (slow) instead of guessing.
Checks: node --check + py_compile pass.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
---------
Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-02 05:34:42 +02:00
_lcExtra += ' --flash-attn on' ;
_lcpExtra += ' --flash_attn true' ;
}
if ( _kv ) {
_lcExtra += ` --cache-type-k ${ _kv } --cache-type-v ${ _kv } ` ;
// llama-cpp-python exposes these as type_k/type_v; pass through best-effort.
_lcpExtra += ` --type_k ${ _kv } --type_v ${ _kv } ` ;
}
2026-06-02 13:46:16 +10:00
const _llamaFit = String ( f . llama _fit || '' ) . trim ( ) ;
if ( [ 'on' , 'off' ] . includes ( _llamaFit ) ) _lcExtra += ` --fit ${ _llamaFit } ` ;
if ( f . llama _no _mmap ) _lcExtra += ' --no-mmap' ;
if ( f . llama _no _warmup ) _lcExtra += ' --no-warmup' ;
const _llamaSplitMode = String ( f . llama _split _mode || '' ) . trim ( ) ;
if ( [ 'none' , 'layer' , 'row' , 'tensor' ] . includes ( _llamaSplitMode ) ) _lcExtra += ` --split-mode ${ _llamaSplitMode } ` ;
const _llamaTensorSplit = _llamaCsv ( f . llama _tensor _split ) ;
if ( _llamaTensorSplit ) _lcExtra += ` --tensor-split ${ _llamaTensorSplit } ` ;
const _llamaMainGpu = _llamaNum ( f . llama _main _gpu ) ;
if ( _llamaMainGpu ) _lcExtra += ` --main-gpu ${ _llamaMainGpu } ` ;
const _llamaParallel = _llamaNum ( f . llama _parallel ) ;
if ( _llamaParallel ) _lcExtra += ` --parallel ${ _llamaParallel } ` ;
const _llamaBatch = _llamaNum ( f . llama _batch _size ) ;
if ( _llamaBatch ) _lcExtra += ` --batch-size ${ _llamaBatch } ` ;
const _llamaUBatch = _llamaNum ( f . llama _ubatch _size ) ;
if ( _llamaUBatch ) _lcExtra += ` --ubatch-size ${ _llamaUBatch } ` ;
if ( f . llama _speculative _mtp ) {
const specTokens = parseInt ( f . llama _spec _tokens , 10 ) ;
const specN = Number . isFinite ( specTokens ) && specTokens > 0 ? specTokens : 3 ;
_lcExtra += ` --spec-type draft-mtp --spec-draft-n-max ${ specN } ` ;
}
Cookbook serve profiles and engine filter
* Cookbook: Engine filter + intelligent hardware-computed serve profiles
Two related Cookbook serving improvements for accurate, hardware-aware model
serving (especially on consumer GPUs that can only run GGUF/llama.cpp).
Engine filter
- New "Engine" dropdown (All / llama.cpp / vLLM / SGLang) beside the quant
picker. Pure client-side view filter over the fetched list via the same
_detectBackend() the serve commands use, so what you filter to is exactly what
would launch. Re-renders from cache (no refetch). Empty-state message + the
instant-cache-paint path account for it too.
Intelligent serve profiles (Quality / Balanced / Speed)
- services/hwfit/profiles.py: compute_serve_profiles() turns detected VRAM +
model size into concrete llama.cpp flags (n_gpu_layers, n_cpu_moe, cache-type,
context). Encodes the by-hand tuning: a too-big MoE offloads experts to CPU
instead of failing; a model that fits stays fully on GPU; quant tracks profile
intent; vision models keep image-encoder headroom. Reuses models.py VRAM math
so filtering and serving agree on what fits. Pure/deterministic (no t/s claims
— partial-offload speed isn't reliably predictable; fit is what's computed).
- /api/hwfit/profiles endpoint returns the profiles + the model's trained
context limit, with loose name matching (strips org/ prefix, -GGUF suffix,
quant tag) so a local GGUF folder name resolves to its catalog entry.
- _buildServeCmd (llama.cpp) now emits --n-cpu-moe / --flash-attn /
--cache-type-k/v when set, with llama-cpp-python fallback equivalents. It
previously only set -ngl/-c, which is why it OOM'd or ran slow.
- Serve panel: profile chips that fill the fields on click, plus CPU-MoE / KV
Cache / Flash Attn fields. Context is clamped to the model's trained limit
(and an absolute 1M sanity ceiling) on type/blur/profile-load and at launch —
fixes a crash where a stale 256k/16M preset + quantized KV cache caused an
amdgpu ErrorDeviceLost.
Tests: tests/test_serve_profiles.py (7) — offload vs full-GPU fit, never exceed
VRAM, context cap, launchable flags, vision headroom, no-GPU empty.
Checks: py_compile + node --check pass; pytest test_serve_profiles + test_hwfit_amd
green; verified live on an RDNA4 box (gfx1200) — Balanced lands ~ncm18 q4 128k,
matching hand-tuning.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
* Cookbook: make column-header sorting discoverable (incl. Newest)
Sorting in Cookbook is via clickable column headers (pewds' design), but the
headers had no visual cue that they're interactive — so sorting in general, and
the Newest sort on the Model header specifically, was undiscoverable.
- Style sortable headers as interactive: pointer cursor, hover underline, and
the active sort column bolded/highlighted. There was no CSS for
.hwfit-sortable / .hwfit-sort-active at all; this helps every existing sort,
not just Newest.
- The Model column header sorts by release_date (newest first), reusing the
existing header-click sort wiring and the "newest" SORT_KEY.
No new sort control — uses the existing column-header paradigm.
Checks: node --check passes.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
* Cookbook serve profiles: keep the on-disk file's quant fixed (don't propose Q6/Q2)
In the Serve tab the model is a specific GGUF file already on disk, so its quant
can't change — but the profiles were suggesting "Quality · Q6_K" / "Speed · Q2_K"
as if you could re-quantize it. That's meaningless when serving a fixed file.
- compute_serve_profiles gains serve_weights_gb / serve_quant. When set (SERVE
mode), the quant is locked to the file's and profiles differ only in the real
serving knobs — n_cpu_moe, KV-cache type, context. _weights_gb / _cpu_moe_for_budget
use the file's actual size instead of a quant-derived estimate. DOWNLOAD mode
(no override) still varies the quant to show download options.
- /api/hwfit/profiles accepts serve_weights_gb & serve_quant.
- The Serve panel parses the file's size (from m.size "20.6 GB") and quant (from
the repo/file name) and passes them, so profiles match what's actually served.
Result for a 20.6 GB Q4_K_M file: all three profiles stay Q4_K_M and differ by
KV/ctx/offload (Quality q8 KV 128k ncm21, Balanced q4 128k ncm17, Speed q4 32k
ncm15) — no nonsensical quant changes.
Tests: test_serve_mode_keeps_fixed_quant. Full serve-profile suite green (9).
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
* Cookbook serve: Vision toggle (auto-find mmproj) + live VRAM/RAM-spillover monitor
Two serve-panel additions:
1. **Vision toggle.** A "Vision" checkbox that serves the model with its
multimodal projector so it can read images. The mmproj path is resolved at
runtime (find mmproj-*.gguf next to the model), so dropping an mmproj file in
the model folder makes the toggle just work; `--mmproj … --image-max-tokens
1024` (native) / `--clip_model_path` (llama-cpp-python) only when on + found.
2. **Live GPU-memory monitor.** A readout that polls /api/cookbook/gpus every 4s
while the panel is open and shows VRAM used/total/%, free, and — crucially on
a discrete card — **RAM spillover** (AMD gtt_used_mb), with a plain-language
health hint: green/healthy, amber/tight, red/"spilled to RAM — slow (raise
CPU MoE or lower context)". Surfaces gtt_used_mb from the gpus endpoint
(previously read for total only and discarded for 'used').
Lets you see at a glance whether a config fits VRAM (fast) or is paging to system
RAM over PCIe (slow) instead of guessing.
Checks: node --check + py_compile pass.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
---------
Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-02 05:34:42 +02:00
// Vision: serve the multimodal projector so the model can read images. The
// mmproj path is resolved at runtime (find mmproj-*.gguf next to the model);
// only emitted when the Vision toggle is on AND a projector was found.
if ( f . vision && f . _mmproj _path ) {
_lcExtra += ` --mmproj " ${ f . _mmproj _path } " --image-max-tokens 1024 ` ;
// llama-cpp-python takes the projector via --clip_model_path.
_lcpExtra += ` --clip_model_path " ${ f . _mmproj _path } " ` ;
}
const _lcpServer = ` ${ lcPrefix } ${ py } -m llama_cpp.server --model ${ modelArg } --host 0.0.0.0 --port ${ f . port || '8080' } --n_gpu_layers ${ f . ngl || '99' } --n_ctx ${ f . ctx || '8192' } ${ _lcpExtra } ` ;
2026-05-31 23:58:26 +09:00
if ( _isWindows ( ) ) {
cmd += _lcpServer ;
} else {
Cookbook serve profiles and engine filter
* Cookbook: Engine filter + intelligent hardware-computed serve profiles
Two related Cookbook serving improvements for accurate, hardware-aware model
serving (especially on consumer GPUs that can only run GGUF/llama.cpp).
Engine filter
- New "Engine" dropdown (All / llama.cpp / vLLM / SGLang) beside the quant
picker. Pure client-side view filter over the fetched list via the same
_detectBackend() the serve commands use, so what you filter to is exactly what
would launch. Re-renders from cache (no refetch). Empty-state message + the
instant-cache-paint path account for it too.
Intelligent serve profiles (Quality / Balanced / Speed)
- services/hwfit/profiles.py: compute_serve_profiles() turns detected VRAM +
model size into concrete llama.cpp flags (n_gpu_layers, n_cpu_moe, cache-type,
context). Encodes the by-hand tuning: a too-big MoE offloads experts to CPU
instead of failing; a model that fits stays fully on GPU; quant tracks profile
intent; vision models keep image-encoder headroom. Reuses models.py VRAM math
so filtering and serving agree on what fits. Pure/deterministic (no t/s claims
— partial-offload speed isn't reliably predictable; fit is what's computed).
- /api/hwfit/profiles endpoint returns the profiles + the model's trained
context limit, with loose name matching (strips org/ prefix, -GGUF suffix,
quant tag) so a local GGUF folder name resolves to its catalog entry.
- _buildServeCmd (llama.cpp) now emits --n-cpu-moe / --flash-attn /
--cache-type-k/v when set, with llama-cpp-python fallback equivalents. It
previously only set -ngl/-c, which is why it OOM'd or ran slow.
- Serve panel: profile chips that fill the fields on click, plus CPU-MoE / KV
Cache / Flash Attn fields. Context is clamped to the model's trained limit
(and an absolute 1M sanity ceiling) on type/blur/profile-load and at launch —
fixes a crash where a stale 256k/16M preset + quantized KV cache caused an
amdgpu ErrorDeviceLost.
Tests: tests/test_serve_profiles.py (7) — offload vs full-GPU fit, never exceed
VRAM, context cap, launchable flags, vision headroom, no-GPU empty.
Checks: py_compile + node --check pass; pytest test_serve_profiles + test_hwfit_amd
green; verified live on an RDNA4 box (gfx1200) — Balanced lands ~ncm18 q4 128k,
matching hand-tuning.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
* Cookbook: make column-header sorting discoverable (incl. Newest)
Sorting in Cookbook is via clickable column headers (pewds' design), but the
headers had no visual cue that they're interactive — so sorting in general, and
the Newest sort on the Model header specifically, was undiscoverable.
- Style sortable headers as interactive: pointer cursor, hover underline, and
the active sort column bolded/highlighted. There was no CSS for
.hwfit-sortable / .hwfit-sort-active at all; this helps every existing sort,
not just Newest.
- The Model column header sorts by release_date (newest first), reusing the
existing header-click sort wiring and the "newest" SORT_KEY.
No new sort control — uses the existing column-header paradigm.
Checks: node --check passes.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
* Cookbook serve profiles: keep the on-disk file's quant fixed (don't propose Q6/Q2)
In the Serve tab the model is a specific GGUF file already on disk, so its quant
can't change — but the profiles were suggesting "Quality · Q6_K" / "Speed · Q2_K"
as if you could re-quantize it. That's meaningless when serving a fixed file.
- compute_serve_profiles gains serve_weights_gb / serve_quant. When set (SERVE
mode), the quant is locked to the file's and profiles differ only in the real
serving knobs — n_cpu_moe, KV-cache type, context. _weights_gb / _cpu_moe_for_budget
use the file's actual size instead of a quant-derived estimate. DOWNLOAD mode
(no override) still varies the quant to show download options.
- /api/hwfit/profiles accepts serve_weights_gb & serve_quant.
- The Serve panel parses the file's size (from m.size "20.6 GB") and quant (from
the repo/file name) and passes them, so profiles match what's actually served.
Result for a 20.6 GB Q4_K_M file: all three profiles stay Q4_K_M and differ by
KV/ctx/offload (Quality q8 KV 128k ncm21, Balanced q4 128k ncm17, Speed q4 32k
ncm15) — no nonsensical quant changes.
Tests: test_serve_mode_keeps_fixed_quant. Full serve-profile suite green (9).
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
* Cookbook serve: Vision toggle (auto-find mmproj) + live VRAM/RAM-spillover monitor
Two serve-panel additions:
1. **Vision toggle.** A "Vision" checkbox that serves the model with its
multimodal projector so it can read images. The mmproj path is resolved at
runtime (find mmproj-*.gguf next to the model), so dropping an mmproj file in
the model folder makes the toggle just work; `--mmproj … --image-max-tokens
1024` (native) / `--clip_model_path` (llama-cpp-python) only when on + found.
2. **Live GPU-memory monitor.** A readout that polls /api/cookbook/gpus every 4s
while the panel is open and shows VRAM used/total/%, free, and — crucially on
a discrete card — **RAM spillover** (AMD gtt_used_mb), with a plain-language
health hint: green/healthy, amber/tight, red/"spilled to RAM — slow (raise
CPU MoE or lower context)". Surfaces gtt_used_mb from the gpus endpoint
(previously read for total only and discarded for 'used').
Lets you see at a glance whether a config fits VRAM (fast) or is paging to system
RAM over PCIe (slow) instead of guessing.
Checks: node --check + py_compile pass.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
---------
Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-02 05:34:42 +02:00
cmd += ` ${ lcPrefix } llama-server --model ${ modelArg } --host 0.0.0.0 --port ${ f . port || '8080' } -ngl ${ f . ngl || '99' } -c ${ f . ctx || '8192' } ${ _lcExtra } ` ;
2026-05-31 23:58:26 +09:00
cmd += ` || ${ _lcpServer } ` ;
}
} else if ( backend === 'ollama' ) {
const ollamaPort = f . port || '11434' ;
2026-06-01 21:27:04 -06:00
const bindHost = _envState . remoteHost ? '0.0.0.0' : '127.0.0.1' ;
const hostEnv = ollamaPort !== '11434' ? ` OLLAMA_HOST= ${ bindHost } : ${ ollamaPort } ` : '' ;
2026-06-02 07:14:59 +09:00
cmd = ` ${ hostEnv } ollama serve ` ;
2026-05-31 23:58:26 +09:00
} else if ( backend === 'diffusers' ) {
const gpuStr = f . gpus ? . trim ( ) ;
if ( gpuStr ) cmd += ` CUDA_VISIBLE_DEVICES= ${ gpuStr } ` ;
cmd += ` python3 scripts/diffusion_server.py --model ${ modelName } --port ${ f . port || '8100' } ` ;
if ( f . diff _dtype && f . diff _dtype !== 'bfloat16' ) cmd += ` --dtype ${ f . diff _dtype } ` ;
if ( f . diff _device _map && f . diff _device _map !== 'balanced' ) cmd += ` --device-map ${ f . diff _device _map } ` ;
if ( f . diff _steps ) cmd += ` --steps ${ f . diff _steps } ` ;
if ( f . diff _width ) cmd += ` --width ${ f . diff _width } ` ;
if ( f . diff _height ) cmd += ` --height ${ f . diff _height } ` ;
if ( f . diff _offload ) cmd += ' --cpu-offload' ;
if ( f . diff _attention _slicing ) cmd += ' --attention-slicing' ;
if ( f . diff _vae _slicing ) cmd += ' --vae-slicing' ;
if ( f . diff _harmonize _gpu ) cmd += ` --harmonize-gpu ${ f . diff _harmonize _gpu } ` ;
}
return cmd ;
}
/** Get inline logo HTML for a model name/repo_id */
export function modelLogo ( name ) {
const logo = providerLogo ( name ) ;
const svg = logo || '<svg viewBox="0 0 24 24" fill="currentColor"><circle cx="12" cy="12" r="4"/></svg>' ;
return ` <span style="width:12px;height:12px;display:inline-flex;align-items:center;vertical-align:-2px;margin-right:3px;opacity: ${ logo ? '0.5' : '0.2' } ;"> ${ svg } </span> ` ;
}
// Use shared esc() from ui module
export const esc = uiModule . esc ;
// ── Clipboard ──
export function _copyText ( text ) {
if ( navigator . clipboard && navigator . clipboard . writeText ) {
return navigator . clipboard . writeText ( text ) . catch ( ( ) => _fallbackCopy ( text ) ) ;
}
return _fallbackCopy ( text ) ;
}
function _fallbackCopy ( text ) {
const ta = document . createElement ( 'textarea' ) ;
ta . value = text ;
ta . style . cssText = 'position:fixed;left:-9999px;top:-9999px' ;
document . body . appendChild ( ta ) ;
ta . select ( ) ;
try { document . execCommand ( 'copy' ) ; } catch ( _ ) { }
document . body . removeChild ( ta ) ;
return Promise . resolve ( ) ;
}
// ── Presets (server-synced; localStorage is offline cache) ──
// Presets sync to/from cookbook_state.json via _syncToServer / _syncFromServer.
// _loadPresets reads the cache (which gets refreshed at app boot and on modal open).
export function _loadPresets ( ) {
try { return JSON . parse ( localStorage . getItem ( STORAGE _KEY ) ) || [ ] ; }
catch { return [ ] ; }
}
export function _savePresets ( presets ) {
localStorage . setItem ( STORAGE _KEY , JSON . stringify ( presets ) ) ;
// Trigger sync to server (via running module's _syncToServer debounce)
_saveTasks ( _loadTasks ( ) ) ;
}
function _envStateForStorage ( ) {
const { hfToken , ... safeState } = _envState ;
return safeState ;
}
function _readStoredEnvState ( ) {
const stored = JSON . parse ( localStorage . getItem ( LAST _STATE _KEY ) || '{}' ) ;
delete stored . hfToken ;
return stored ;
}
export function _persistEnvState ( ) {
try { localStorage . setItem ( LAST _STATE _KEY , JSON . stringify ( _envStateForStorage ( ) ) ) ; }
catch ( _ ) { }
_saveTasks ( _loadTasks ( ) ) ;
}
// ── Dependencies ──
// Category colors removed — using theme CSS classes instead
async function _fetchDependencies ( ) {
const list = document . getElementById ( 'cookbook-deps-list' ) ;
if ( ! list ) return ;
// Use the shared whirlpool spinner so the user sees the request is in
// flight (the package list takes a few seconds to enumerate on slow links).
list . innerHTML = '' ;
let _spin = null ;
try {
const sp = ( await import ( './spinner.js' ) ) . default ;
_spin = sp . createWhirlpool ( 28 ) ;
_spin . element . style . cssText = 'margin:24px auto 0;display:block;' ;
list . appendChild ( _spin . element ) ;
const label = document . createElement ( 'div' ) ;
label . className = 'hwfit-loading' ;
label . textContent = 'Loading packages…' ;
label . style . cssText = 'text-align:center;opacity:0.5;font-size:11px;margin-top:6px;' ;
list . appendChild ( label ) ;
} catch {
list . innerHTML = '<div class="hwfit-loading">Loading packages...</div>' ;
}
try {
// Resolve the target server from the deps dropdown so remote-target
// packages are checked on THAT server's venv (not just the local host).
let _depHost = '' , _depPort = '' , _depVenv = '' ;
const _dsel = document . getElementById ( 'hwfit-deps-server' ) ;
const _depSrv = _dsel && _dsel . value !== 'local' ? _serverByVal ( _dsel . value ) : null ;
if ( _depSrv ) {
_depHost = _depSrv . host || '' ; _depPort = _depSrv . port || '' ; _depVenv = _depSrv . envPath || '' ;
} else if ( _envState . remoteHost ) {
_depHost = _envState . remoteHost ; _depPort = _getPort ( _envState . remoteHost ) || '' ; _depVenv = _envState . envPath || '' ;
}
const _pkgParams = new URLSearchParams ( ) ;
if ( _depHost ) {
_pkgParams . set ( 'host' , _depHost ) ;
if ( _depPort ) _pkgParams . set ( 'ssh_port' , _depPort ) ;
if ( _depVenv ) _pkgParams . set ( 'venv' , _depVenv ) ;
}
const resp = await fetch ( '/api/cookbook/packages' + ( _pkgParams . toString ( ) ? '?' + _pkgParams . toString ( ) : '' ) ) ;
const data = await resp . json ( ) ;
const pkgs = data . packages || [ ] ;
if ( ! pkgs . length ) { list . innerHTML = '<div class="hwfit-loading">No packages found</div>' ; return ; }
const _winUnsupported = new Set ( [ 'diffusers' , 'hf_transfer' , 'vllm' , 'rembg' , 'gfpgan' ] ) ;
2026-06-01 04:50:50 +02:00
const _statusTag = ( pkg , isLocal , isSystemDep , winBlocked ) => {
if ( winBlocked ) return ` <span class="cookbook-dep-tag cookbook-dep-na">N/A</span> ` ;
if ( pkg . installed && isSystemDep ) return ` <span class="cookbook-dep-tag cookbook-dep-installed" title="Found on selected server">Installed</span> ` ;
2026-06-03 00:20:00 +10:00
if ( pkg . installed && pkg . pip _update _available === false ) {
const tip = esc ( pkg . update _note || pkg . status _note || 'Found externally; update outside Odysseus.' ) ;
return ` <span class="cookbook-dep-tag cookbook-dep-installed" title=" ${ tip } ">Installed</span> ` ;
}
2026-06-01 04:50:50 +02:00
if ( pkg . installed ) return ` <button class="cookbook-dep-tag cookbook-dep-installed cookbook-dep-installed-btn" title="Installed — click for actions"><span class="cookbook-dep-installed-label">Installed</span><span class="cookbook-dep-caret">▾</span></button> ` ;
2026-06-01 09:56:42 +02:00
if ( isSystemDep ) {
const depTip = esc ( pkg . install _hint || 'Install this OS package on the selected server.' ) ;
const depLabel = pkg . applicable === false ? 'N/A ?' : 'Missing' ;
return ` <span class="cookbook-dep-tag cookbook-dep-na" title=" ${ depTip } "> ${ depLabel } </span> ` ;
}
2026-06-01 04:50:50 +02:00
return ` <button class="cookbook-dep-tag cookbook-dep-install" data-dep-pip=" ${ esc ( pkg . pip ) } " data-dep-target=" ${ isLocal ? 'local' : 'remote' } ">Install</button> ` ;
} ;
const _depRow = ( pkg ) => {
2026-05-31 23:58:26 +09:00
const isLocal = pkg . target === 'local' ;
const isSystemDep = pkg . kind === 'system' ;
2026-06-01 04:50:50 +02:00
const winBlocked = ! isLocal && _isWindows ( ) && _winUnsupported . has ( pkg . name ) ;
2026-06-01 23:39:36 +10:00
const note = pkg . status _note ? ` <div class="memory-item-meta" style="font-size:10px;opacity:0.65;margin-top:3px;"> ${ esc ( pkg . status _note ) } </div> ` : '' ;
2026-06-03 00:20:00 +10:00
const updateNote = pkg . installed && pkg . pip _update _available === false && pkg . update _note ? ` <div class="memory-item-meta" style="font-size:10px;opacity:0.55;margin-top:3px;"> ${ esc ( pkg . update _note ) } </div> ` : '' ;
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
// Inline "Rebuild" tag for the llama_cpp row only. Styled as a
// .cookbook-dep-tag so it matches the LLM category tag's pill look,
// and lives to the LEFT of the category tag (clear affordance before
// the row "value").
const _rebuildBtn = ( pkg . name === 'llama_cpp' )
? ` <button type="button" class="cookbook-dep-tag cookbook-dep-rebuild" id="cookbook-rebuild-engine" title="Clear the cached llama.cpp build so the next serve recompiles from source (use after installing a CUDA/ROCm toolkit to turn a CPU-only build into a GPU build).">Rebuild</button> `
: '' ;
2026-06-01 04:50:50 +02:00
return ` <div class="cookbook-dep-row ${ winBlocked ? ' cookbook-dep-blocked' : '' } " data-pkg-name=" ${ esc ( pkg . name ) } " data-dep-pip=" ${ esc ( pkg . pip || '' ) } " data-dep-target=" ${ isLocal ? 'local' : 'remote' } " data-dep-kind=" ${ esc ( pkg . kind || 'python' ) } "> `
+ ` <div class="cookbook-dep-info"> `
+ ` <div class="memory-item-title"> ${ esc ( pkg . name ) } </div> `
+ ` <div class="memory-item-meta" style="font-size:10px;opacity:0.5;margin-top:2px;"> ${ esc ( pkg . desc ) } </div> `
2026-06-01 23:39:36 +10:00
+ note
2026-06-03 00:20:00 +10:00
+ updateNote
2026-06-01 04:50:50 +02:00
+ ` </div> `
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
+ _rebuildBtn
2026-06-01 04:50:50 +02:00
+ ` <span class="cookbook-dep-tag cookbook-dep-cat"> ${ esc ( pkg . category ) } </span> `
+ _statusTag ( pkg , isLocal , isSystemDep , winBlocked )
+ ` </div> ` ;
} ;
const _section = ( title , note , items ) =>
items . length
? ` <div class="cookbook-dep-section"><span class="cookbook-dep-section-title"> ${ title } </span><span class="cookbook-dep-section-note"> ${ note } </span></div> ` + items . map ( _depRow ) . join ( '' )
: '' ;
const _viewingRemote = ! ! ( _dsel && _dsel . value && _dsel . value !== 'local' ) ;
const _appDeps = pkgs . filter ( p => p . target === 'local' ) ;
const _serverDeps = pkgs . filter ( p => p . target !== 'local' ) ;
list . innerHTML = [
_viewingRemote ? '' : _section ( 'Odysseus app' , 'Run inside the Odysseus app itself.' , _appDeps ) ,
_section ( 'Server' , 'Run on the server chosen above (Local, or a remote box over SSH).' , _serverDeps ) ,
] . join ( '' ) ;
2026-05-31 23:58:26 +09:00
// Shared install/update routine — used by the Install button and the
// "Update" item in an installed package's ⋮ menu. `upgrade` adds pip -U;
// `statusEl`, when given, shows "Installing…/Updating…" and is disabled.
async function _installDep ( pipName , pkgName , isLocalOnly , upgrade , statusEl ) {
if ( isLocalOnly ) {
_envState . remoteHost = '' ;
_envState . env = 'none' ;
_envState . envPath = '' ;
} else {
const depsServerSel = document . getElementById ( 'hwfit-deps-server' ) ;
if ( depsServerSel ) _applyServerSelection ( depsServerSel . value ) ;
}
const targetHost = isLocalOnly ? 'this server' : ( _envState . remoteHost || 'local' ) ;
// Always go through `python -m pip` so the leading token is `python`
// — matches the /api/model/serve allow-list (bare `pip` is blocked).
// Inside a venv/conda env, `--user` is invalid (pip refuses), so we
// only add `--user --break-system-packages` when there's no env —
// for PEP-668-locked system pythons (Arch, newer Debian).
const _inEnv = _envState . env === 'venv' || _envState . env === 'conda' ;
const _pipFlags = ( ! _isWindows ( ) && ! _inEnv ) ? ' --user --break-system-packages' : '' ;
const _py = _isWindows ( ) ? 'python' : 'python3' ;
const cmd = ` ${ _py } -m pip install ${ upgrade ? ' -U' : '' } ${ _pipFlags } " ${ pipName } " ` ;
let envPrefix = '' ;
if ( _isWindows ( ) ) {
if ( _envState . env === 'venv' && _envState . envPath ) {
envPrefix = '& ' + _psQuote ( _envState . envPath . endsWith ( '\\Scripts\\Activate.ps1' ) ? _envState . envPath : _envState . envPath + '\\Scripts\\Activate.ps1' ) ;
} else if ( _envState . env === 'conda' && _envState . envPath ) {
envPrefix = 'conda activate ' + _psQuote ( _envState . envPath ) ;
}
} else {
if ( _envState . env === 'venv' && _envState . envPath ) {
const p = _envState . envPath ;
envPrefix = 'source ' + _shellQuote ( p . endsWith ( '/bin/activate' ) ? p : p + '/bin/activate' ) ;
} else if ( _envState . env === 'conda' && _envState . envPath ) {
envPrefix = 'eval "$(conda shell.bash hook)" && conda activate ' + _shellQuote ( _envState . envPath ) ;
}
}
try {
const reqBody = {
repo _id : pipName ,
cmd : cmd ,
remote _host : _envState . remoteHost || undefined ,
ssh _port : _getPort ( _envState . remoteHost ) || undefined ,
env _prefix : envPrefix || undefined ,
platform : _envState . platform || undefined ,
} ;
const res = await fetch ( '/api/model/serve' , {
method : 'POST' , credentials : 'same-origin' ,
headers : { 'Content-Type' : 'application/json' } ,
body : JSON . stringify ( reqBody ) ,
} ) ;
const data = await res . json ( ) . catch ( ( ) => ( { } ) ) ;
if ( ! res . ok || ! data . ok ) {
// FastAPI HTTPException returns {detail: …}; the route's own
// path returns {ok:false, error:…}. Surface whichever we get.
const reason = data . detail || data . error || ` HTTP ${ res . status } ` ;
uiModule . showToast ( 'Install failed: ' + String ( reason ) . slice ( 0 , 200 ) ) ;
return ;
}
// _dep flags this as a pip dependency/driver install (not a servable
// model) so the running-task card doesn't offer a "Serve →" button.
2026-06-01 22:59:29 -05:00
const payload = { repo _id : pipName , _cmd : cmd , remote _host : _envState . remoteHost || '' , _dep : true , env _path : _envState . envPath || '' } ;
2026-05-31 23:58:26 +09:00
_addTask ( data . session _id , 'pip ' + pkgName , 'download' , payload ) ;
if ( statusEl ) { statusEl . textContent = upgrade ? 'Updating...' : 'Installing...' ; statusEl . disabled = true ; }
uiModule . showToast ( ` ${ upgrade ? 'Updating' : 'Installing' } ${ pkgName } on ${ targetHost } ... ` ) ;
} catch ( err ) {
uiModule . showToast ( 'Install failed: ' + err . message ) ;
}
}
// Wire install buttons (not-installed packages)
list . querySelectorAll ( '.cookbook-dep-install' ) . forEach ( btn => {
btn . addEventListener ( 'click' , async ( e ) => {
e . stopPropagation ( ) ;
const pipName = btn . dataset . depPip ;
const pkgName = btn . closest ( '.cookbook-dep-row' ) ? . querySelector ( '.memory-item-title' ) ? . textContent || pipName ;
await _installDep ( pipName , pkgName , btn . dataset . depTarget === 'local' , ! ! btn . dataset . upgrade , btn ) ;
} ) ;
} ) ;
// Wire the ⋮ menu on installed packages — currently just "Update".
function _showDepMenu ( anchor ) {
document . querySelectorAll ( '.cookbook-dep-menu' ) . forEach ( d => d . remove ( ) ) ;
const row = anchor . closest ( '.cookbook-dep-row' ) ;
if ( ! row ) return ;
const pipName = row . dataset . depPip ;
const pkgName = row . querySelector ( '.memory-item-title' ) ? . textContent || pipName ;
const isLocalOnly = row . dataset . depTarget === 'local' ;
const dropdown = document . createElement ( 'div' ) ;
dropdown . className = 'dropdown cookbook-dep-menu' ;
const rect = anchor . getBoundingClientRect ( ) ;
const minW = 150 ;
let left = Math . min ( rect . right - minW , window . innerWidth - minW - 8 ) ;
left = Math . max ( 8 , left ) ;
dropdown . style . cssText = ` position:fixed;display:block;z-index:10001;top: ${ rect . bottom + 6 } px;left: ${ left } px;right:auto;min-width: ${ minW } px;max-width:calc(100vw - 16px);background:var(--panel,var(--bg));border:1px solid var(--border);border-radius:10px;box-shadow:0 8px 24px rgba(0,0,0,0.3);padding:6px;font-size:11px; ` ;
const upIco = '<svg width="13" height="13" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round"><path d="M21 2v6h-6"/><path d="M3 12a9 9 0 0 1 15-6.7L21 8"/><path d="M3 22v-6h6"/><path d="M21 12a9 9 0 0 1-15 6.7L3 16"/></svg>' ;
const it = document . createElement ( 'div' ) ;
it . className = 'dropdown-item-compact' ;
it . innerHTML = ` <span class="dropdown-icon"> ${ upIco } </span><span>Update</span> ` ;
it . title = ` Update ${ pkgName } to the latest version (pip install -U) ` ;
it . addEventListener ( 'click' , async ( e ) => {
e . stopPropagation ( ) ;
dropdown . remove ( ) ;
await _installDep ( pipName , pkgName , isLocalOnly , true , null ) ;
} ) ;
dropdown . appendChild ( it ) ;
document . body . appendChild ( dropdown ) ;
const close = ( ev ) => {
if ( ! dropdown . contains ( ev . target ) && ev . target !== anchor && ! anchor . contains ( ev . target ) ) {
dropdown . remove ( ) ;
document . removeEventListener ( 'click' , close , true ) ;
}
} ;
setTimeout ( ( ) => document . addEventListener ( 'click' , close , true ) , 10 ) ;
}
list . querySelectorAll ( '.cookbook-dep-installed-btn' ) . forEach ( btn => {
btn . addEventListener ( 'click' , ( e ) => {
e . stopPropagation ( ) ;
if ( document . querySelector ( '.cookbook-dep-menu' ) ) {
document . querySelectorAll ( '.cookbook-dep-menu' ) . forEach ( d => d . remove ( ) ) ;
return ;
}
_showDepMenu ( btn ) ;
} ) ;
} ) ;
} catch ( err ) {
list . innerHTML = ` <div class="hwfit-loading">Error loading packages: ${ esc ( err . message ) } </div> ` ;
}
}
// ── Tab wiring ──
function _applyServerSelection ( val ) {
if ( val === 'local' ) {
_envState . remoteHost = '' ;
_envState . env = 'none' ;
_envState . envPath = '' ;
_envState . platform = '' ;
} else {
const s = _serverByVal ( val ) ;
if ( s ) {
_envState . remoteHost = s . host ;
_envState . env = s . env || 'none' ;
_envState . envPath = s . envPath || '' ;
_envState . platform = s . platform || '' ;
}
}
// Persist + keep every server dropdown in sync, so the choice sticks across
// re-renders and the scan/download all target the SAME host (this was the
// bug: the Download/Cache/Deps dropdowns set the host but never saved it, so
// it silently reverted and downloads/scans hit the wrong server).
_persistEnvState ( ) ;
const _want = _envState . remoteHost || 'local' ;
document . querySelectorAll ( '#hwfit-server-select, #hwfit-dl-server, #hwfit-cache-server, #hwfit-deps-server' ) . forEach ( sel => {
if ( ! sel || sel . tagName !== 'SELECT' ) return ;
// Option values are host strings now ('local' for the local box).
sel . value = _want ;
// If the host isn't among this select's current options (stale options after
// the server list changed), the browser leaves the box BLANK/grey even though
// the value is "set". Rebuild the options so the chosen host has an entry, then
// re-apply; fall back to 'local' only if it's genuinely gone.
if ( sel . selectedIndex < 0 ) {
sel . innerHTML = _buildServerOpts ( sel . id === 'hwfit-dl-server' ) ;
sel . value = _want ;
if ( sel . selectedIndex < 0 ) sel . value = 'local' ;
}
} ) ;
}
function _wireTabEvents ( body ) {
// Tab switching
body . querySelectorAll ( '.cookbook-tab' ) . forEach ( tab => {
tab . addEventListener ( 'click' , ( ) => {
body . querySelectorAll ( '.cookbook-tab' ) . forEach ( t => t . classList . remove ( 'active' ) ) ;
tab . classList . add ( 'active' ) ;
const backend = tab . dataset . backend ;
body . querySelectorAll ( '.cookbook-group' ) . forEach ( g => {
g . classList . toggle ( 'hidden' , g . dataset . backendGroup !== backend ) ;
} ) ;
if ( backend === 'Search' ) {
_hwfitInit ( ) ;
_hwfitFetch ( ) ;
}
if ( backend === 'Serve' ) {
_fetchCachedModels ( ) ;
}
if ( backend === 'Dependencies' ) {
_fetchDependencies ( ) ;
}
} ) ;
} ) ;
// Mobile: swipe left/right anywhere in the body to move to the next/previous
// tab. Guarded so it ignores vertical scrolls, tiny moves, and form fields.
if ( ! body . _swipeWired ) {
body . _swipeWired = true ;
let _sx = null , _sy = null ;
body . addEventListener ( 'touchstart' , ( e ) => {
// Ignore swipes that start in a horizontally-scrollable tag row — those
// should scroll the chips, not flip the tab.
if ( window . innerWidth > 768 || e . touches . length !== 1
|| e . target . closest ( 'input, textarea, select, .doclib-lang-chips' ) ) { _sx = null ; return ; }
_sx = e . touches [ 0 ] . clientX ; _sy = e . touches [ 0 ] . clientY ;
} , { passive : true } ) ;
body . addEventListener ( 'touchend' , ( e ) => {
if ( _sx === null ) return ;
const dx = e . changedTouches [ 0 ] . clientX - _sx ;
const dy = e . changedTouches [ 0 ] . clientY - _sy ;
_sx = null ;
// Require a clear horizontal swipe (>60px and mostly horizontal).
if ( Math . abs ( dx ) < 60 || Math . abs ( dx ) < Math . abs ( dy ) * 1.5 ) return ;
const tabs = [ ... body . querySelectorAll ( '.cookbook-tab' ) ] ;
const idx = tabs . findIndex ( t => t . classList . contains ( 'active' ) ) ;
if ( idx < 0 ) return ;
const next = dx < 0 ? idx + 1 : idx - 1 ; // swipe left → next tab
if ( next >= 0 && next < tabs . length ) tabs [ next ] . click ( ) ;
} , { passive : true } ) ;
}
// Sync server form DOM → _envState.servers
function _syncServers ( ) {
const entries = document . querySelectorAll ( '.cookbook-server-entry' ) ;
const servers = [ ] ;
entries . forEach ( entry => {
const name = entry . querySelector ( '.cookbook-srv-name' ) ? . value ? . trim ( ) || '' ;
const host = entry . querySelector ( '.cookbook-srv-host' ) ? . value ? . trim ( ) || '' ;
const port = entry . querySelector ( '.cookbook-srv-port' ) ? . value ? . trim ( ) || '' ;
const env = entry . querySelector ( '.cookbook-srv-env' ) ? . value || 'none' ;
const envPath = entry . querySelector ( '.cookbook-srv-path' ) ? . value ? . trim ( ) || '' ;
const platform = entry . dataset . platform || '' ;
const dirs = [ ] ;
entry . querySelectorAll ( '.cookbook-modeldir-tag' ) . forEach ( tag => {
// Read from data attribute (authoritative) — never parse displayed text
const d = ( tag . dataset . dir || '' ) . replaceAll ( '✕' , '' ) . replaceAll ( '✖' , '' ) . trim ( ) ;
if ( d ) dirs . push ( d ) ;
} ) ;
// Directory flagged as the download target ('' = default HF cache).
const dlEl = entry . querySelector ( '.cookbook-modeldir-dl.active' ) ;
const downloadDir = dlEl ? ( dlEl . dataset . dlDir || '' ) : '' ;
servers . push ( { name , host , port , env , envPath , modelDirs : dirs , downloadDir , platform } ) ;
} ) ;
_envState . servers = servers ;
// Auto-default: when the user has configured EXACTLY ONE remote server
// and hasn't picked one yet, select it. Without this, the dropdown
// stays on "Local" so the eventual serve/scan/launch resolves to no
// remote host and the backend rejects the call with 403 (Forbidden),
// which read to the user as a permission bug.
if ( ! _envState . remoteHost ) {
const remotes = servers . filter ( s => ! _isLocalEntry ( s ) ) ;
if ( remotes . length === 1 ) {
_envState . remoteHost = remotes [ 0 ] . host ;
_envState . env = remotes [ 0 ] . env || 'none' ;
_envState . envPath = remotes [ 0 ] . envPath || '' ;
}
}
const activeSrv = servers . find ( s => s . host === _envState . remoteHost ) ;
_envState . platform = activeSrv ? . platform || '' ;
localStorage . setItem ( 'cookbook-last-state' , JSON . stringify ( _envStateForStorage ( ) ) ) ;
_saveTasks ( _loadTasks ( ) ) ;
// Reflect the auto-default selection into every server dropdown so the
// UI matches the resolved host. Done in a microtask so the dropdowns
// exist by the time we set their .value.
Promise . resolve ( ) . then ( ( ) => {
const _want = _envState . remoteHost || 'local' ;
document . querySelectorAll ( '#hwfit-server-select, #hwfit-dl-server, #hwfit-cache-server, #hwfit-deps-server' ) . forEach ( sel => {
if ( sel && sel . tagName === 'SELECT' ) sel . value = _want ;
} ) ;
} ) ;
}
// Wire server form inputs
document . querySelectorAll ( '.cookbook-srv-name, .cookbook-srv-host, .cookbook-srv-port, .cookbook-srv-path' ) . forEach ( el => {
el . addEventListener ( 'change' , _syncServers ) ;
} ) ;
document . querySelectorAll ( '.cookbook-srv-env' ) . forEach ( el => {
el . addEventListener ( 'change' , _syncServers ) ;
} ) ;
// Server selector — the server is global, so switching it here re-scans the
// main Scan/Download list (#hwfit-list) for the new server's hardware too.
// (The trending sublist reloads via its own handler in the HF-latest wiring.)
const dlServer = document . getElementById ( 'hwfit-dl-server' ) ;
if ( dlServer ) {
dlServer . addEventListener ( 'change' , ( ) => {
_applyServerSelection ( dlServer . value ) ;
// Reset toggle state (no flicker) so the new server's hardware re-renders.
_resetGpuToggleState ( ) ;
_hwfitFetch ( ) ;
} ) ;
}
// Add server link — switch to Settings tab
const addServerLink = document . querySelector ( '.cookbook-dl-add-server' ) ;
if ( addServerLink ) {
addServerLink . addEventListener ( 'click' , ( ) => {
const settingsTab = body . querySelector ( '.cookbook-tab[data-backend="Settings"]' ) ;
if ( settingsTab ) settingsTab . click ( ) ;
} ) ;
}
// Cache server selector
const cacheServer = document . getElementById ( 'hwfit-cache-server' ) ;
const cacheDirEl = document . getElementById ( 'hwfit-cache-dir' ) ;
if ( cacheServer ) {
cacheServer . addEventListener ( 'change' , ( ) => {
_applyServerSelection ( cacheServer . value ) ;
const val = cacheServer . value ;
let srv ;
if ( val === 'local' ) {
srv = _envState . servers . find ( _isLocalEntry ) || _envState . servers [ 0 ] || { } ;
} else {
srv = _serverByVal ( val ) || { } ;
}
if ( cacheDirEl ) cacheDirEl . value = srv . modelDir || '~/.cache/huggingface/hub' ;
const dirsEl = document . querySelector ( '.cookbook-serve-dirs' ) ;
if ( dirsEl ) {
const dirs = ( Array . isArray ( srv . modelDirs ) ? srv . modelDirs : [ srv . modelDir || '~/.cache/huggingface/hub' ] ) . map ( d => d . replaceAll ( '✕' , '' ) . replaceAll ( '✖' , '' ) . trim ( ) ) . filter ( Boolean ) ;
dirsEl . innerHTML = dirs . map ( d => ` <span class="cookbook-serve-dir-pill"> ${ esc ( d ) } </span> ` ) . join ( '' ) +
'<span class="cookbook-serve-dir-edit" title="Edit in Settings">edit</span>' ;
dirsEl . querySelector ( '.cookbook-serve-dir-edit' ) ? . addEventListener ( 'click' , ( ) => {
const settingsTab = body . querySelector ( '.cookbook-tab[data-backend="Settings"]' ) ;
if ( settingsTab ) settingsTab . click ( ) ;
} ) ;
}
_fetchCachedModels ( ) ;
} ) ;
}
const scanBtn = document . getElementById ( 'hwfit-cache-scan' ) ;
if ( scanBtn ) {
scanBtn . addEventListener ( 'click' , ( ) => _fetchCachedModels ( ) ) ;
}
const editDirsLink = document . querySelector ( '.cookbook-serve-dir-edit' ) ;
if ( editDirsLink ) {
editDirsLink . addEventListener ( 'click' , ( ) => {
const settingsTab = body . querySelector ( '.cookbook-tab[data-backend="Settings"]' ) ;
if ( settingsTab ) settingsTab . click ( ) ;
} ) ;
}
const depsServer = document . getElementById ( 'hwfit-deps-server' ) ;
if ( depsServer ) {
depsServer . addEventListener ( 'change' , ( ) => {
_applyServerSelection ( depsServer . value ) ;
// Re-fetch the package list for the newly selected server — the installed
// status is per-server, so the list must refresh on a server switch.
_fetchDependencies ( ) ;
} ) ;
}
2026-06-02 23:28:19 -05:00
// "Rebuild llama.cpp" clears the cached build so the next serve recompiles.
// The serve bootstrap only builds llama-server when it is missing from PATH,
// so a host that first built CPU-only (no nvcc at build time) keeps reusing
// that binary forever; this is the lever to force a fresh GPU build after a
// CUDA/ROCm toolkit is installed.
const rebuildBtn = document . getElementById ( 'cookbook-rebuild-engine' ) ;
if ( rebuildBtn && ! rebuildBtn . _wired ) {
rebuildBtn . _wired = true ;
rebuildBtn . addEventListener ( 'click' , async ( ) => {
// Match _installDep: honor the Dependencies server selector so the clear
// runs on the same host the build runs on.
const sel = document . getElementById ( 'hwfit-deps-server' ) ;
if ( sel ) _applyServerSelection ( sel . value ) ;
const host = _envState . remoteHost || '' ;
const where = host || 'this server' ;
if ( ! confirm ( ` Rebuild the llama.cpp engine on ${ where } ? \n \n This clears the cached llama-server build so the next serve recompiles from source (with CUDA/HIP if a toolchain is present). It does not download or install anything. ` ) ) return ;
const _label = rebuildBtn . textContent ;
rebuildBtn . disabled = true ;
rebuildBtn . textContent = 'Clearing...' ;
try {
const res = await fetch ( '/api/cookbook/rebuild-engine' , {
method : 'POST' , credentials : 'same-origin' ,
headers : { 'Content-Type' : 'application/json' } ,
body : JSON . stringify ( {
engine : 'llamacpp' ,
remote _host : host || undefined ,
ssh _port : _getPort ( host ) || undefined ,
} ) ,
} ) ;
const data = await res . json ( ) . catch ( ( ) => ( { } ) ) ;
if ( ! res . ok || ! data . ok ) {
const reason = data . detail || data . error || ` HTTP ${ res . status } ` ;
uiModule . showToast ( 'Rebuild failed: ' + String ( reason ) . slice ( 0 , 200 ) ) ;
} else {
uiModule . showToast ( ` Cleared llama.cpp build on ${ where } . Re-launch the serve task to rebuild with GPU support. ` ) ;
}
} catch ( err ) {
uiModule . showToast ( 'Rebuild failed: ' + err . message ) ;
} finally {
rebuildBtn . disabled = false ;
rebuildBtn . textContent = _label ;
}
} ) ;
}
2026-05-31 23:58:26 +09:00
// Serve sort
const serveSort = document . getElementById ( 'serve-sort' ) ;
if ( serveSort ) {
serveSort . addEventListener ( 'change' , ( ) => {
if ( _cachedAllModels . length ) _rerenderCachedModels ( ) ;
} ) ;
}
// Serve search
const serveSearch = document . getElementById ( 'serve-search' ) ;
if ( serveSearch ) {
let _srvDebounce = null ;
serveSearch . addEventListener ( 'input' , ( ) => {
clearTimeout ( _srvDebounce ) ;
_srvDebounce = setTimeout ( ( ) => _filterCachedList ( ) , 200 ) ;
} ) ;
}
// Select mode — bulk actions
const selectBtn = document . getElementById ( 'hwfit-cache-select' ) ;
const bulkBar = document . getElementById ( 'serve-bulk-bar' ) ;
if ( selectBtn && bulkBar ) {
selectBtn . addEventListener ( 'click' , ( ) => {
const active = selectBtn . classList . toggle ( 'active' ) ;
selectBtn . textContent = active ? 'Cancel' : 'Select' ;
bulkBar . classList . toggle ( 'hidden' , ! active ) ;
document . querySelectorAll ( '.serve-select-cb' ) . forEach ( dot => {
dot . style . display = active ? '' : 'none' ;
dot . classList . remove ( 'selected' ) ;
} ) ;
_updateBulkCount ( ) ;
} ) ;
document . getElementById ( 'hwfit-cached-list' ) ? . addEventListener ( 'click' , ( e ) => {
if ( ! selectBtn . classList . contains ( 'active' ) ) return ;
const item = e . target . closest ( '.memory-item[data-repo]' ) ;
if ( ! item ) return ;
if ( e . target . closest ( 'a, .hwfit-cached-menu-btn, .memory-item-btn, .hwfit-serve-panel' ) ) return ;
const dot = item . querySelector ( '.serve-select-cb' ) ;
if ( dot ) {
dot . classList . toggle ( 'selected' ) ;
_updateBulkCount ( ) ;
}
} ) ;
function _updateBulkCount ( ) {
const count = document . querySelectorAll ( '.serve-select-cb.selected' ) . length ;
const countEl = document . getElementById ( 'serve-bulk-count' ) ;
if ( countEl ) countEl . textContent = count + ' selected' ;
}
document . getElementById ( 'serve-bulk-cancel' ) ? . addEventListener ( 'click' , ( ) => {
selectBtn . classList . remove ( 'active' ) ;
2026-06-03 16:49:10 +09:00
selectBtn . textContent = 'Select' ; // reset label so the button doesn't stay reading "Cancel" after exit
2026-05-31 23:58:26 +09:00
bulkBar . classList . add ( 'hidden' ) ;
document . querySelectorAll ( '.serve-select-cb' ) . forEach ( dot => { dot . style . display = 'none' ; dot . classList . remove ( 'selected' ) ; } ) ;
} ) ;
document . getElementById ( 'serve-bulk-delete' ) ? . addEventListener ( 'click' , async ( ) => {
const checked = document . querySelectorAll ( '.serve-select-cb.selected' ) ;
if ( ! checked . length ) return ;
const repos = [ ] ;
checked . forEach ( dot => {
const item = dot . closest ( '.memory-item[data-repo]' ) ;
if ( item ? . dataset . repo ) repos . push ( item . dataset . repo ) ;
} ) ;
if ( ! ( await uiModule . styledConfirm ( ` Delete ${ repos . length } model(s)? This removes cached files. ` , { confirmText : 'Delete' , danger : true } ) ) ) return ;
for ( const repo of repos ) {
const item = document . querySelector ( ` .memory-item[data-repo=" ${ repo } "] ` ) ;
if ( item ) await _deleteCachedModel ( repo , item , true ) ;
}
selectBtn . classList . remove ( 'active' ) ;
2026-06-03 16:49:10 +09:00
selectBtn . textContent = 'Select' ; // same reset as bulk-cancel
2026-05-31 23:58:26 +09:00
bulkBar . classList . add ( 'hidden' ) ;
document . querySelectorAll ( '.serve-select-cb' ) . forEach ( dot => { dot . style . display = 'none' ; dot . classList . remove ( 'selected' ) ; } ) ;
} ) ;
}
// Download input
const dlBtn = document . getElementById ( 'cookbook-dl-btn' ) ;
const dlInput = document . getElementById ( 'cookbook-dl-repo' ) ;
2026-06-02 12:15:41 +09:00
const dlCardToggle = document . getElementById ( 'cookbook-download-card-toggle' ) ;
const dlCardBody = document . getElementById ( 'cookbook-download-card-body' ) ;
const dlCardArrow = document . getElementById ( 'cookbook-download-card-arrow' ) ;
if ( dlCardToggle && dlCardBody ) {
dlCardToggle . addEventListener ( 'click' , ( ) => {
const isOpen = dlCardBody . style . display !== 'none' ;
dlCardBody . style . display = isOpen ? 'none' : 'block' ;
if ( dlCardArrow ) dlCardArrow . style . transform = isOpen ? 'rotate(0deg)' : 'rotate(90deg)' ;
} ) ;
}
2026-05-31 23:58:26 +09:00
if ( dlBtn && dlInput ) {
function _stripHfUrl ( input ) {
let repo = input . trim ( ) ;
// Strip Ollama-style "hf.co/" prefix if present (e.g. hf.co/unsloth/...:tag)
repo = repo . replace ( /^hf\.co\// , '' ) ;
const hfMatch = repo . match ( /^https?:\/\/huggingface\.co\/([^/]+\/[^/?#]+(?::[^/?#\s]+)?)/ ) ;
if ( hfMatch ) repo = hfMatch [ 1 ] ;
return repo ;
}
// Split `org/repo:tag` (Ollama/llama.cpp style) into repo + include-glob.
// The `:tag` picks a specific GGUF quantization file from the repo.
function _splitRepoTag ( raw ) {
const m = raw . match ( /^([^\s/:]+\/[^\s/:]+):([^\s/]+)$/ ) ;
if ( ! m ) return { repo : raw , include : null } ;
return { repo : m [ 1 ] , include : ` * ${ m [ 2 ] } * ` } ;
}
const triggerDownload = ( ) => {
const rawRepo = _stripHfUrl ( dlInput . value ) ;
if ( ! rawRepo ) return ;
const { repo , include : autoInclude } = _splitRepoTag ( rawRepo ) ;
// HuggingFace repo IDs must be `org/model`. A bare model name would 404
// at snapshot_download time with a raw traceback, so reject it up front.
if ( ! /^[^\s/]+\/[^\s/]+$/ . test ( repo ) ) {
uiModule . showToast ( 'Enter a full HuggingFace repo ID like "org/model-name" (or paste the full HF URL).' ) ;
dlInput . focus ( ) ;
return ;
}
// Resolve the host straight from THIS window's server dropdown, by index
// into the (consistent) servers list. We deliberately don't use
// _envState.remoteHost — there can be multiple copies of the cookbook
// state in memory and they disagree on the active host, which is what sent
// downloads to the wrong server. The dropdown the user sees is the truth.
const dlSrv = document . getElementById ( 'hwfit-dl-server' ) ;
const srvVal = dlSrv ? dlSrv . value : 'local' ;
let host = '' ;
if ( srvVal !== 'local' ) {
host = _serverByVal ( srvVal ) ? . host || '' ;
}
const _hsrv = _envState . servers . find ( sv => sv . host === host ) || { } ;
let env = host ? ( _hsrv . env || 'none' ) : _envState . env ;
let envPath = host ? ( _hsrv . envPath || '' ) : _envState . envPath ;
const payload = { repo _id : repo } ;
if ( autoInclude ) payload . include = autoInclude ;
if ( _envState . hfToken ) payload . hf _token = _envState . hfToken ;
if ( host ) { payload . remote _host = host ; const _sp3 = _getPort ( host ) ; if ( _sp3 ) payload . ssh _port = _sp3 ; }
const srvPlatform = _getPlatform ( host ) ;
if ( srvPlatform ) payload . platform = srvPlatform ;
if ( srvPlatform === 'windows' ) {
if ( env === 'venv' && envPath ) {
payload . env _prefix = '& ' + _psQuote ( envPath . endsWith ( '\\Scripts\\Activate.ps1' ) ? envPath : envPath + '\\Scripts\\Activate.ps1' ) ;
} else if ( env === 'conda' && envPath ) {
payload . env _prefix = 'conda activate ' + _psQuote ( envPath ) ;
}
} else {
if ( env === 'venv' && envPath ) {
const p = envPath ;
payload . env _prefix = 'source ' + _shellQuote ( p . endsWith ( '/bin/activate' ) ? p : p + '/bin/activate' ) ;
} else if ( env === 'conda' && envPath ) {
payload . env _prefix = 'eval "$(conda shell.bash hook)" && conda activate ' + _shellQuote ( envPath ) ;
}
}
const shortName = repo . split ( '/' ) . pop ( ) ;
_retryDownload ( shortName , payload ) ;
dlInput . value = '' ;
} ;
dlBtn . addEventListener ( 'click' , triggerDownload ) ;
dlInput . addEventListener ( 'keydown' , ( e ) => {
if ( e . key === 'Enter' ) triggerDownload ( ) ;
} ) ;
}
// Latest HF models that fit — collapsible card list
Cookbook: scoring fixes, UI polish, false-finished + stale-state bug fixes
Backend (services/hwfit + routes):
- rank_models picks visible set by REQUESTED column, not always score —
sorting by Param now shows highest-param models PERIOD (incl. too_tight).
- New fit_only param. Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang
cannot serve them); default non-prequantized to BF16 on 2+ GPUs.
- AWQ / GPTQ-8bit get a -1.0 quality penalty (was 0.0, tied with FP8), so
FP8 wins when both fit.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above
M2.5 on equal composite score; >=100B integers not misread as versions.
- /api/cookbook/hf-latest no longer drops models without an "NB" pattern in
the repo id (MiniMax-M2.7, DeepSeek-V4-Pro etc. were silently filtered).
- Cached-model scan: atexit flushes models JSON even if the script is
killed mid-walk; each scan_dir wrapped in try/except; timeout 60s -> 180s.
- KB granularity for sub-MB sizes (was "0 MB" for 12 KB shells). New
"stalled" status for shells <1 MB with no .incomplete files.
- /api/cookbook/state POST guard: rejects "done" download tasks lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stops stale tabs from poisoning persisted state.
- hf_models.json: add zai-org/GLM-5.1; flip zai-org/GLM-5 quantization
Q4_K_M -> BF16 (it is the native base, not a quant).
Frontend (static/js):
- Scan/Download toolbar: quant defaults to All; ctx slider (8k/16k/32k/
50k/128k/Max) ported from origin/main with sort=fit on drag, sort=score
on Max. GPU toggle commits _activeCount to maxGpu on initial render. Fit
column header tagged with active budget (RAM / GPU / N GPU).
- Foldable Download admin-card: the Download h2 is the chevron trigger;
state persists in localStorage.
- Download card surfaces destination dir (Dir: <path>). Same dir on running
task row, font/color matched to uptime (9px Fira Code muted, opacity .4).
- Serve panel ctx text input always resets to model max on open. Sub-MB
cached models show with red "download stalled" badge.
- Bulk-select Cancel + Delete reset the Select button label on exit.
- Cookbook running: false-finished bug fixed — DOWNLOAD_OK or /snapshots/
required; bare "Download complete" no longer marks the task done after
the first config file. Clear button now sends tmux kill-session too.
True overall % for multi-shard downloads: ((N-1)+frac)/total instead of
hf_transfer per-shard aggregate.
- Diagnosis card simplified: removed fold toggle, copy button, dismiss X.
Suggestion font matches message body (12px).
- HF token field flashes green check + "Saved" on save.
- Cached scan no longer counts stalled rows as downloaded in Scan/Download.
CSS:
- dep Install button width pinned to 76px to match Installed split.
- task-sub row +1px; task-status badge gets margin-right 8px.
- Ctx slider styled like gallery editor sliders (thin pill rail, red thumb).
- Bulk-select cancel button top -3px -> -5px.
2026-06-03 16:32:20 +09:00
// Foldable Download admin-card — h2 "Download" doubles as the chevron
// toggle; collapses the entire card body (description + input + HF list).
// State persisted to localStorage so the fold sticks across reloads.
const dlFold = document . getElementById ( 'cookbook-dl-tab-fold' ) ;
const dlFoldBody = document . getElementById ( 'cookbook-dl-tab-fold-body' ) ;
const dlFoldChevron = document . getElementById ( 'cookbook-dl-tab-chevron' ) ;
if ( dlFold && dlFoldBody && dlFoldChevron ) {
dlFold . addEventListener ( 'click' , ( ) => {
const folded = dlFoldBody . style . display === 'none' ;
dlFoldBody . style . display = folded ? '' : 'none' ;
dlFoldChevron . textContent = folded ? '▾' : '▸' ;
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
// Toggle is-folded class on the h2 so the line under it only shows when
// the section is collapsed (the body's content normally provides
// separation; with no body visible, the line gives the h2 definition).
dlFold . classList . toggle ( 'is-folded' , ! folded ) ;
Cookbook: scoring fixes, UI polish, false-finished + stale-state bug fixes
Backend (services/hwfit + routes):
- rank_models picks visible set by REQUESTED column, not always score —
sorting by Param now shows highest-param models PERIOD (incl. too_tight).
- New fit_only param. Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang
cannot serve them); default non-prequantized to BF16 on 2+ GPUs.
- AWQ / GPTQ-8bit get a -1.0 quality penalty (was 0.0, tied with FP8), so
FP8 wins when both fit.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above
M2.5 on equal composite score; >=100B integers not misread as versions.
- /api/cookbook/hf-latest no longer drops models without an "NB" pattern in
the repo id (MiniMax-M2.7, DeepSeek-V4-Pro etc. were silently filtered).
- Cached-model scan: atexit flushes models JSON even if the script is
killed mid-walk; each scan_dir wrapped in try/except; timeout 60s -> 180s.
- KB granularity for sub-MB sizes (was "0 MB" for 12 KB shells). New
"stalled" status for shells <1 MB with no .incomplete files.
- /api/cookbook/state POST guard: rejects "done" download tasks lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stops stale tabs from poisoning persisted state.
- hf_models.json: add zai-org/GLM-5.1; flip zai-org/GLM-5 quantization
Q4_K_M -> BF16 (it is the native base, not a quant).
Frontend (static/js):
- Scan/Download toolbar: quant defaults to All; ctx slider (8k/16k/32k/
50k/128k/Max) ported from origin/main with sort=fit on drag, sort=score
on Max. GPU toggle commits _activeCount to maxGpu on initial render. Fit
column header tagged with active budget (RAM / GPU / N GPU).
- Foldable Download admin-card: the Download h2 is the chevron trigger;
state persists in localStorage.
- Download card surfaces destination dir (Dir: <path>). Same dir on running
task row, font/color matched to uptime (9px Fira Code muted, opacity .4).
- Serve panel ctx text input always resets to model max on open. Sub-MB
cached models show with red "download stalled" badge.
- Bulk-select Cancel + Delete reset the Select button label on exit.
- Cookbook running: false-finished bug fixed — DOWNLOAD_OK or /snapshots/
required; bare "Download complete" no longer marks the task done after
the first config file. Clear button now sends tmux kill-session too.
True overall % for multi-shard downloads: ((N-1)+frac)/total instead of
hf_transfer per-shard aggregate.
- Diagnosis card simplified: removed fold toggle, copy button, dismiss X.
Suggestion font matches message body (12px).
- HF token field flashes green check + "Saved" on save.
- Cached scan no longer counts stalled rows as downloaded in Scan/Download.
CSS:
- dep Install button width pinned to 76px to match Installed split.
- task-sub row +1px; task-status badge gets margin-right 8px.
- Ctx slider styled like gallery editor sliders (thin pill rail, red thumb).
- Bulk-select cancel button top -3px -> -5px.
2026-06-03 16:32:20 +09:00
try { localStorage . setItem ( 'cookbook_dl_tab_folded_v1' , folded ? '0' : '1' ) ; } catch { }
} ) ;
}
2026-05-31 23:58:26 +09:00
const hfToggle = document . getElementById ( 'cookbook-hf-latest-toggle' ) ;
const hfArrow = document . getElementById ( 'cookbook-hf-latest-arrow' ) ;
const hfList = document . getElementById ( 'cookbook-hf-latest-list' ) ;
const hfRefresh = document . getElementById ( 'cookbook-hf-latest-refresh' ) ;
if ( hfToggle && hfList ) {
let _loaded = false ;
// Per-server VRAM cache so we don't re-probe on every expand
2026-06-02 12:15:41 +09:00
const _hwCache = { } ;
function _hfModelLooksAwqLike ( m ) {
const text = ` ${ m ? . repo _id || '' } ${ ( m ? . tags || [ ] ) . join ( ' ' ) } ` . toLowerCase ( ) ;
return /\b(awq|gptq|fp8|4bit|int4)\b/ . test ( text ) ;
}
async function _getSelectedServerHw ( ) {
2026-05-31 23:58:26 +09:00
// Prefer the "What Fits" dropdown (the main control that shows hardware);
// fall back to the download dropdown. This is the server the list ranks for.
const dlSrv = document . getElementById ( 'hwfit-server-select' ) || document . getElementById ( 'hwfit-dl-server' ) ;
const val = dlSrv ? . value || 'local' ;
let host = '' ;
let sshPort = '' ;
let platform = '' ;
if ( val !== 'local' ) {
const s = _serverByVal ( val ) ;
if ( s ) {
host = s . host || '' ;
sshPort = s . port || '' ;
platform = s . platform || '' ;
}
}
const cacheKey = host || 'local' ;
2026-06-02 12:15:41 +09:00
if ( _hwCache [ cacheKey ] ) return _hwCache [ cacheKey ] ;
2026-05-31 23:58:26 +09:00
// Fetch system info for this server from hwfit
try {
const qp = new URLSearchParams ( ) ;
if ( host ) qp . set ( 'host' , host ) ;
if ( sshPort ) qp . set ( 'ssh_port' , sshPort ) ;
if ( platform ) qp . set ( 'platform' , platform ) ;
const r = await fetch ( ` /api/hwfit/system? ${ qp } ` ) ;
if ( r . ok ) {
const sys = await r . json ( ) ;
2026-06-02 12:15:41 +09:00
const hw = { vram : sys ? . gpu _vram _gb || 0 , backend : String ( sys ? . backend || '' ) . toLowerCase ( ) } ;
_hwCache [ cacheKey ] = hw ;
return hw ;
2026-05-31 23:58:26 +09:00
}
} catch { }
2026-06-02 12:15:41 +09:00
_hwCache [ cacheKey ] = { vram : 0 , backend : '' } ;
return _hwCache [ cacheKey ] ;
2026-05-31 23:58:26 +09:00
}
async function _loadLatest ( ) {
// Match the Dependencies loader: whirlpool spinner + text label so the
// user gets immediate feedback while the scan runs.
hfList . innerHTML = '' ;
try {
const sp = ( await import ( './spinner.js' ) ) . default ;
const _spin = sp . createWhirlpool ( 28 ) ;
_spin . element . style . cssText = 'margin:24px auto 0;display:block;' ;
hfList . appendChild ( _spin . element ) ;
const lbl = document . createElement ( 'div' ) ;
lbl . className = 'hwfit-loading' ;
lbl . textContent = 'Scanning models…' ;
lbl . style . cssText = 'text-align:center;opacity:0.5;font-size:11px;margin-top:6px;' ;
hfList . appendChild ( lbl ) ;
} catch {
hfList . innerHTML = '<div class="hwfit-loading">Scanning models…</div>' ;
}
2026-06-02 12:15:41 +09:00
const hwInfo = await _getSelectedServerHw ( ) ;
const vram = hwInfo . vram || 0 ;
2026-05-31 23:58:26 +09:00
try {
let lastErr = '' ;
const _fetchLatest = async ( v ) => {
const res = await fetch ( ` /api/cookbook/hf-latest?vram_gb= ${ v } &limit=10 ` ) ;
const data = await res . json ( ) ;
if ( data . error ) lastErr = data . error ; // HF API timeout/rate-limit etc.
return data . models || [ ] ;
} ;
let models = await _fetchLatest ( vram ) ;
// If the VRAM filter wiped everything out (often a flaky/zero hardware
// probe for a remote server — a huge-VRAM box should fit MORE, not
// fewer), fall back to the unfiltered trending list so something shows.
if ( ! models . length && vram > 0 ) {
models = await _fetchLatest ( 0 ) ;
}
2026-06-02 12:15:41 +09:00
if ( [ 'rocm' , 'metal' , 'mps' , 'apple' , 'generic' , 'cpu' ] . includes ( hwInfo . backend ) ) {
models = models . filter ( m => ! _hfModelLooksAwqLike ( m ) ) ;
}
2026-05-31 23:58:26 +09:00
if ( ! models . length ) {
// Distinguish "the HF API failed" from "nothing matched" so an outage
// doesn't masquerade as no-fitting-models.
const msg = lastErr
? ` Couldn't load trending models ( ${ esc ( lastErr ) } ) `
: 'No trending models found' ;
hfList . innerHTML = ` <div class="hwfit-loading"> ${ msg } </div> ` ;
return ;
}
let html = '' ;
for ( const m of models ) {
const shortName = m . repo _id . split ( '/' ) . pop ( ) || m . repo _id ;
const org = m . repo _id . includes ( '/' ) ? m . repo _id . split ( '/' ) [ 0 ] : '' ;
const meta = [ ] ;
if ( org ) meta . push ( esc ( org ) ) ;
if ( m . needed _vram _gb ) meta . push ( ` ~ ${ m . needed _vram _gb } GB ` ) ;
if ( m . downloads ) meta . push ( ` ${ m . downloads . toLocaleString ( ) } downloads ` ) ;
const date = m . createdAt ? new Date ( m . createdAt ) . toISOString ( ) . slice ( 0 , 10 ) : '' ;
if ( date ) meta . push ( date ) ;
html += ` <div class="doclib-card memory-item cookbook-hf-latest-card" data-repo=" ${ esc ( m . repo _id ) } " style="cursor:pointer;"> ` ;
html += ` <div style="flex:1;min-width:0;"> ` ;
html += ` <div class="memory-item-title"> ${ esc ( shortName ) } <a href="https://huggingface.co/ ${ esc ( m . repo _id ) } " target="_blank" rel="noopener" class="cookbook-hf-link">HF \u 2197</a></div> ` ;
html += ` <div class="memory-item-meta" style="font-size:10px;opacity:0.5;margin-top:2px;"> ${ meta . join ( ' \u00b7 ' ) } </div> ` ;
html += ` </div> ` ;
html += ` </div> ` ;
}
hfList . innerHTML = html ;
// Wire card clicks → fill download input
hfList . querySelectorAll ( '.cookbook-hf-latest-card' ) . forEach ( card => {
card . addEventListener ( 'click' , ( e ) => {
if ( e . target . closest ( 'a' ) ) return ;
if ( dlInput ) {
dlInput . value = card . dataset . repo ;
dlInput . focus ( ) ;
}
} ) ;
} ) ;
} catch ( e ) {
hfList . innerHTML = '<div class="hwfit-loading">Failed to load</div>' ;
}
}
hfToggle . addEventListener ( 'click' , ( ) => {
const isOpen = hfList . style . display !== 'none' ;
hfList . style . display = isOpen ? 'none' : 'flex' ;
if ( hfArrow ) hfArrow . style . transform = isOpen ? 'rotate(0deg)' : 'rotate(90deg)' ;
if ( ! isOpen && ! _loaded ) {
_loaded = true ;
_loadLatest ( ) ;
}
} ) ;
if ( hfRefresh ) hfRefresh . addEventListener ( 'click' , ( e ) => {
e . stopPropagation ( ) ;
_loaded = true ;
_loadLatest ( ) ;
// If list is hidden, open it
if ( hfList . style . display === 'none' ) {
hfList . style . display = 'flex' ;
if ( hfArrow ) hfArrow . style . transform = 'rotate(90deg)' ;
}
} ) ;
// Re-fetch when a server dropdown changes — different server = different
// hardware/VRAM. Mark the list stale so it reloads for the new server even
// if it's currently collapsed (otherwise reopening showed the old server's
// models); reload immediately when it's open.
const _onServerChange = ( ) => {
_loaded = false ;
if ( hfList . style . display !== 'none' ) { _loaded = true ; _loadLatest ( ) ; }
} ;
document . getElementById ( 'hwfit-dl-server' ) ? . addEventListener ( 'change' , _onServerChange ) ;
document . getElementById ( 'hwfit-server-select' ) ? . addEventListener ( 'change' , _onServerChange ) ;
}
// Server add button, row removal, model-dir add/remove, and per-row wiring
// are ALL owned by cookbook-hwfit.js's _hwfitInit / _wireServerEntry.
// A duplicate add handler used to live here and fired alongside the hwfit
// one, appending two rows per click — removed.
// HF token — save on change
const hfInput = document . getElementById ( 'hwfit-hftoken' ) ;
if ( hfInput ) {
2026-06-03 16:49:10 +09:00
hfInput . addEventListener ( 'change' , async ( ) => {
const val = hfInput . value . trim ( ) ;
_envState . hfToken = val ;
try { await _persistEnvState ( ) ; } catch { }
if ( val ) {
_envState . hfTokenConfigured = true ;
const masked = val . length > 6 ? val . slice ( 0 , 3 ) + '…' + val . slice ( - 3 ) : '••••' ;
_envState . hfTokenMasked = masked ;
hfInput . placeholder = ` Stored ( ${ masked } ) - enter a new token to replace ` ;
hfInput . value = '' ;
let check = hfInput . parentNode . querySelector ( '.hwfit-hf-check' ) ;
if ( ! check ) {
check = document . createElement ( 'span' ) ;
check . className = 'hwfit-hf-check' ;
check . title = 'Token stored' ;
check . textContent = '✓' ;
check . style . cssText = 'font-weight:800;color:var(--green,#50fa7b);font-size:15px;line-height:1;flex-shrink:0;position:relative;top:2px;' ;
hfInput . parentNode . insertBefore ( check , hfInput ) ;
}
const flash = document . createElement ( 'span' ) ;
flash . textContent = 'Saved' ;
flash . style . cssText = 'margin-left:8px;font-size:11px;color:var(--green,#50fa7b);opacity:0;transition:opacity 0.18s;flex-shrink:0;position:relative;top:1px;' ;
hfInput . parentNode . appendChild ( flash ) ;
requestAnimationFrame ( ( ) => { flash . style . opacity = '1' ; } ) ;
setTimeout ( ( ) => { flash . style . opacity = '0' ; setTimeout ( ( ) => flash . remove ( ) , 220 ) ; } , 1400 ) ;
}
2026-05-31 23:58:26 +09:00
} ) ;
}
}
// ── Main render ──
// Build one server entry's HTML — shared by the Settings render loop AND the
// "+ Add server" handler, so a freshly-added server has the IDENTICAL layout
// (Model Directory header, default-server checkmark, trash delete, platform icon).
// forceRemote renders an editable remote entry even before a host is typed
// (a new server's host is empty, which would otherwise read as "Local").
export function _serverEntryHtml ( s , i , defaultServer , forceRemote , isNew ) {
const isLocal = ( forceRemote || isNew ) ? false : ( ! s . host || s . host === 'local' ) ;
const envOpts = [ 'none' , 'venv' ] . map ( e => ` <option value=" ${ e } " ${ s . env === e ? ' selected' : '' } > ${ e === 'none' ? 'None' : e } </option> ` ) . join ( '' ) ;
let html = '' ;
html += ` <div class="cookbook-server-entry" data-idx=" ${ i } " data-platform=" ${ esc ( s . platform || '' ) } "> ` ;
const _srvTitle = s . name || ( isLocal ? 'Local' : ( s . host || ` Server ${ i + 1 } ` ) ) ;
const _srvKey = isLocal ? 'local' : ( s . host || '' ) ;
const _isDefaultSrv = ( defaultServer || '' ) === _srvKey ;
const _pIco = _platformIcon ( s . platform ) ;
const _keyBtn = ` <button class="cookbook-server-key-btn" title="Set up SSH key for this server" style="height:22px;box-sizing:border-box;display:inline-flex;align-items:center;position:relative;top:-2px;"><svg width="11" height="11" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" style="margin-right:4px;flex-shrink:0;"><circle cx="7.5" cy="15.5" r="5.5"/><path d="M12 11l8-8"/><path d="M17 6l3 3"/></svg>Key</button> ` ;
const _checkBtn = ` <button class="cookbook-server-check-btn" title="Check SSH connection" style="height:22px;box-sizing:border-box;display:inline-flex;align-items:center;position:relative;top:-2px;"><svg width="11" height="11" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2.2" stroke-linecap="round" stroke-linejoin="round" style="margin-right:4px;flex-shrink:0;"><polyline points="20 6 9 17 4 12"/></svg>Check</button> ` ;
html += ` <span class="cookbook-server-title" style="display:flex;align-items:center;gap:6px;width:100%;font-size:13px;font-weight:600;margin-bottom:4px;"> ` ;
html += ` ${ esc ( _srvTitle ) } ` ;
html += _pIco ? ` <span class="cookbook-srv-platform" title=" ${ esc ( s . platform || '' ) } " style="display:inline-flex;align-items:center;opacity:0.55;"> ${ _pIco } </span> ` : '' ;
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
html += ` <span class="cookbook-srv-test-msg" style="font-size:10px;font-weight:400;opacity:0.55;max-width:160px;white-space:nowrap;overflow:hidden;text-overflow:ellipsis;position:relative;top:1px;"></span> ` ;
2026-05-31 23:58:26 +09:00
if ( isNew ) {
// New server: Cancel (discard) sits top-right; the default toggle only makes
// sense once the server is saved.
html += ` <span style="margin-left:auto;display:inline-flex;gap:4px;align-items:center;"> ${ _checkBtn } ${ _keyBtn } <button class="cookbook-server-cancel-btn" title="Discard this new server" style="height:22px;box-sizing:border-box;display:inline-flex;align-items:center;position:relative;top:-2px;"><svg width="11" height="11" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" style="margin-right:4px;flex-shrink:0;"><line x1="18" y1="6" x2="6" y2="18"/><line x1="6" y1="6" x2="18" y2="18"/></svg>Cancel</button></span> ` ;
} else {
html += ` <span style="margin-left:auto;display:inline-flex;gap:4px;align-items:center;"> ${ ! isLocal ? _checkBtn + _keyBtn : '' } <span class="cookbook-srv-default ${ _isDefaultSrv ? ' active' : '' } " title=" ${ _isDefaultSrv ? 'Default server — Cookbook opens here' : 'Make this the default server' } " data-srv-key=" ${ esc ( _srvKey ) } "> ${ _isDefaultSrv ? _MODELDIR _CHECK _ON : _MODELDIR _CHECK _OFF } <span class="cookbook-srv-default-label">default</span></span></span> ` ;
}
html += ` </span> ` ;
html += ` <div class="cookbook-server-row"> ` ;
html += ` <input type="text" class="hwfit-sf cookbook-srv-name" value=" ${ esc ( s . name || ( isLocal ? 'Local' : '' ) ) } " placeholder="Name (optional)" style="width:92px;flex-shrink:0;" /> ` ;
html += ` <input type="text" class="hwfit-sf cookbook-srv-host" value=" ${ isLocal ? '' : esc ( s . host || '' ) } " placeholder="e.g. user@ip" style="width:214.5px;flex-shrink:0;box-sizing:border-box;" ${ isLocal ? 'readonly' : '' } /> ` ;
html += ` <input type="text" class="hwfit-sf cookbook-srv-port" value=" ${ esc ( s . port || '' ) } " placeholder="Port" title="SSH port (default 22)" style="width:48px;flex-shrink:0;" ${ isLocal ? 'readonly' : '' } /> ` ;
html += ` <select class="hwfit-sf cookbook-srv-env"> ${ envOpts } </select> ` ;
html += ` <input type="text" class="hwfit-sf cookbook-srv-path" value=" ${ esc ( s . envPath || '' ) } " placeholder=" ${ s . platform === 'windows' ? 'venv path' : '~/venv' } " /> ` ;
html += ` <span class="cookbook-dep-tag cookbook-dep-target" style="font-size:8px;flex-shrink:0;min-width:46px;text-align:center;visibility:hidden;">placeholder</span> ` ;
html += ` <span class="cookbook-srv-actions" style="display:inline-flex;gap:4px;align-items:center;width:78px;flex-shrink:0;justify-content:flex-end;"></span> ` ;
html += ` </div> ` ;
const modelDirs = Array . isArray ( s . modelDirs ) && s . modelDirs . length ? s . modelDirs : [ '~/.cache/huggingface/hub' ] ;
const activeDlDir = s . downloadDir || '' ;
html += ` <div class="cookbook-modeldirs" style="margin:2px 0 0 0;display:flex;flex-wrap:wrap;gap:4px;align-items:center;"> ` ;
html += ` <span style="width:100%;font-size:13px;font-weight:600;margin-bottom:3px;">Model Directory <span style="font-weight:400;opacity:0.5;font-size:11px;">— check the one downloads should go to</span></span> ` ;
for ( let j = 0 ; j < modelDirs . length ; j ++ ) {
const isDefault = modelDirs [ j ] === '~/.cache/huggingface/hub' ;
const dirVal = isDefault ? '' : modelDirs [ j ] ;
const isTarget = activeDlDir === dirVal ;
const dlBtn = ` <span class="cookbook-modeldir-dl ${ isTarget ? ' active' : '' } " title=" ${ isTarget ? 'Downloads go here' : 'Send downloads here' } " data-dl-dir=" ${ esc ( dirVal ) } "> ${ isTarget ? _MODELDIR _CHECK _ON : _MODELDIR _CHECK _OFF } </span> ` ;
const rmBtn = isDefault ? '' : ' <span class="cookbook-modeldir-rm" title="Remove">✖</span>' ;
html += ` <span class="cookbook-modeldir-tag ${ isDefault ? ' cookbook-modeldir-default' : '' } ${ isTarget ? ' cookbook-modeldir-target' : '' } " data-dir-idx=" ${ j } " data-dir=" ${ esc ( modelDirs [ j ] ) } "> ${ dlBtn } ${ esc ( modelDirs [ j ] ) } ${ rmBtn } </span> ` ;
}
html += ` <button class="cookbook-modeldir-add" title="Add model directory">+ Add</button> ` ;
const _btnStyle = 'margin-left:auto;position:relative;top:-2px;height:22px;box-sizing:border-box;display:inline-flex;align-items:center;' ;
if ( isNew ) {
// A brand-new server: Save (confirm) sits where Delete would be; Cancel is
// top-right in the title. Save confirms with a checkmark (auto-saves on edit too).
html += ` <button class="cookbook-server-save-btn" title="Save this server" style=" ${ _btnStyle } "><svg width="11" height="11" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" style="margin-right:4px;flex-shrink:0;"><path d="M19 21H5a2 2 0 0 1-2-2V5a2 2 0 0 1 2-2h11l5 5v11a2 2 0 0 1-2 2z"/><polyline points="17 21 17 13 7 13 7 21"/><polyline points="7 3 7 8 15 8"/></svg>Save</button> ` ;
} else if ( ! isLocal ) {
html += ` <button class="cookbook-server-rm cookbook-server-rm-btn" title="Delete this server" style=" ${ _btnStyle } "><svg width="11" height="11" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" style="margin-right:4px;flex-shrink:0;"><path d="M3 6h18"/><path d="M8 6V4a2 2 0 0 1 2-2h4a2 2 0 0 1 2 2v2"/><path d="M19 6v14a2 2 0 0 1-2 2H7a2 2 0 0 1-2-2V6"/></svg>Delete</button> ` ;
}
html += ` </div> ` ;
if ( ! isLocal ) {
html += ` <div class="cookbook-server-key-panel hidden" style="margin-top:6px;flex-direction:column;gap:5px;"> ` ;
html += ` <div style="display:flex;gap:4px;align-items:center;"> ` ;
html += ` <button type="button" class="memory-toolbar-btn cookbook-server-key-gen" style="height:23px;">Generate key</button> ` ;
html += ` <button type="button" class="memory-toolbar-btn cookbook-server-key-copy" style="height:23px;" disabled>Copy command</button> ` ;
html += ` <span style="font-size:10px;opacity:0.55;line-height:1.25;">Docker: run this command in your terminal once.</span> ` ;
html += ` </div> ` ;
html += ` <textarea class="memory-search-input cookbook-server-key-command" readonly rows="3" style="min-height:58px;resize:vertical;font-family:var(--mono,monospace);font-size:10px;line-height:1.35;">Enter user@host, then generate the key.</textarea> ` ;
html += ` </div> ` ;
}
html += ` </div> ` ;
return html ;
}
function _renderRecipes ( ) {
const body = document . querySelector ( '#cookbook-modal .cookbook-body' ) ;
if ( ! body ) return ;
const presets = _loadPresets ( ) ;
const hasSaved = presets . length > 0 ;
let html = '' ;
// Tabs
html += '<div class="cookbook-tabs">' ;
html += '<button class="cookbook-tab active" data-backend="Search"><svg width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" style="vertical-align:-1px;margin-right:3px;"><polyline points="7 14 12 19 17 14"/><line x1="12" y1="19" x2="12" y2="5"/><line x1="5" y1="21" x2="19" y2="21"/></svg>Download</button>' ;
html += '<button class="cookbook-tab" data-backend="Serve"><svg width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" style="vertical-align:-1px;margin-right:3px;"><rect x="2" y="2" width="20" height="8" rx="2"/><rect x="2" y="14" width="20" height="8" rx="2"/><circle cx="6" cy="6" r="1"/><circle cx="6" cy="18" r="1"/></svg>Serve</button>' ;
html += '<button class="cookbook-tab" data-backend="Dependencies"><svg width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" style="vertical-align:-1px;margin-right:3px;"><path d="M21 16V8a2 2 0 0 0-1-1.73l-7-4a2 2 0 0 0-2 0l-7 4A2 2 0 0 0 3 8v8a2 2 0 0 0 1 1.73l7 4a2 2 0 0 0 2 0l7-4A2 2 0 0 0 21 16z"/><polyline points="3.27 6.96 12 12.01 20.73 6.96"/><line x1="12" y1="22.08" x2="12" y2="12"/></svg>Dependencies</button>' ;
html += '<button class="cookbook-tab" data-backend="Settings"><svg width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" style="vertical-align:-1px;margin-right:3px;"><circle cx="12" cy="12" r="3"/><path d="M19.4 15a1.65 1.65 0 0 0 .33 1.82l.06.06a2 2 0 1 1-2.83 2.83l-.06-.06a1.65 1.65 0 0 0-1.82-.33 1.65 1.65 0 0 0-1 1.51V21a2 2 0 0 1-4 0v-.09A1.65 1.65 0 0 0 9 19.4a1.65 1.65 0 0 0-1.82.33l-.06.06a2 2 0 1 1-2.83-2.83l.06-.06A1.65 1.65 0 0 0 4.68 15a1.65 1.65 0 0 0-1.51-1H3a2 2 0 0 1 0-4h.09A1.65 1.65 0 0 0 4.6 9a1.65 1.65 0 0 0-.33-1.82l-.06-.06a2 2 0 1 1 2.83-2.83l.06.06A1.65 1.65 0 0 0 9 4.68a1.65 1.65 0 0 0 1-1.51V3a2 2 0 0 1 4 0v.09a1.65 1.65 0 0 0 1 1.51 1.65 1.65 0 0 0 1.82-.33l.06-.06a2 2 0 1 1 2.83 2.83l-.06.06A1.65 1.65 0 0 0 19.4 9a1.65 1.65 0 0 0 1.51 1H21a2 2 0 0 1 0 4h-.09a1.65 1.65 0 0 0-1.51 1z"/></svg>Settings</button>' ;
html += '</div>' ;
// Search group
html += '<div class="cookbook-group" data-backend-group="Search" style="flex:0 0 auto;">' ;
html += '<div class="admin-card" style="display:flex;flex-direction:column;overflow:hidden;">' ;
Cookbook: scoring fixes, UI polish, false-finished + stale-state bug fixes
Backend (services/hwfit + routes):
- rank_models picks visible set by REQUESTED column, not always score —
sorting by Param now shows highest-param models PERIOD (incl. too_tight).
- New fit_only param. Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang
cannot serve them); default non-prequantized to BF16 on 2+ GPUs.
- AWQ / GPTQ-8bit get a -1.0 quality penalty (was 0.0, tied with FP8), so
FP8 wins when both fit.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above
M2.5 on equal composite score; >=100B integers not misread as versions.
- /api/cookbook/hf-latest no longer drops models without an "NB" pattern in
the repo id (MiniMax-M2.7, DeepSeek-V4-Pro etc. were silently filtered).
- Cached-model scan: atexit flushes models JSON even if the script is
killed mid-walk; each scan_dir wrapped in try/except; timeout 60s -> 180s.
- KB granularity for sub-MB sizes (was "0 MB" for 12 KB shells). New
"stalled" status for shells <1 MB with no .incomplete files.
- /api/cookbook/state POST guard: rejects "done" download tasks lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stops stale tabs from poisoning persisted state.
- hf_models.json: add zai-org/GLM-5.1; flip zai-org/GLM-5 quantization
Q4_K_M -> BF16 (it is the native base, not a quant).
Frontend (static/js):
- Scan/Download toolbar: quant defaults to All; ctx slider (8k/16k/32k/
50k/128k/Max) ported from origin/main with sort=fit on drag, sort=score
on Max. GPU toggle commits _activeCount to maxGpu on initial render. Fit
column header tagged with active budget (RAM / GPU / N GPU).
- Foldable Download admin-card: the Download h2 is the chevron trigger;
state persists in localStorage.
- Download card surfaces destination dir (Dir: <path>). Same dir on running
task row, font/color matched to uptime (9px Fira Code muted, opacity .4).
- Serve panel ctx text input always resets to model max on open. Sub-MB
cached models show with red "download stalled" badge.
- Bulk-select Cancel + Delete reset the Select button label on exit.
- Cookbook running: false-finished bug fixed — DOWNLOAD_OK or /snapshots/
required; bare "Download complete" no longer marks the task done after
the first config file. Clear button now sends tmux kill-session too.
True overall % for multi-shard downloads: ((N-1)+frac)/total instead of
hf_transfer per-shard aggregate.
- Diagnosis card simplified: removed fold toggle, copy button, dismiss X.
Suggestion font matches message body (12px).
- HF token field flashes green check + "Saved" on save.
- Cached scan no longer counts stalled rows as downloaded in Scan/Download.
CSS:
- dep Install button width pinned to 76px to match Installed split.
- task-sub row +1px; task-status badge gets margin-right 8px.
- Ctx slider styled like gallery editor sliders (thin pill rail, red thumb).
- Bulk-select cancel button top -3px -> -5px.
2026-06-03 16:32:20 +09:00
// Foldable Download admin-card: clicking the h2 header collapses the
// entire card body (description + download input + HF latest section).
// State persisted to localStorage so the fold survives reloads.
const _dlTabFolded = ( ( ) => { try { return localStorage . getItem ( 'cookbook_dl_tab_folded_v1' ) === '1' ; } catch { return false ; } } ) ( ) ;
html += '<div style="display:flex;align-items:center;gap:8px;margin-bottom:2px;">' ;
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
html += ` <h2 id="cookbook-dl-tab-fold" class=" ${ _dlTabFolded ? 'is-folded' : '' } " style="margin:0;padding:0;line-height:1;cursor:pointer;display:flex;align-items:center;justify-content:space-between;user-select:none;flex:1;">Download<span id="cookbook-dl-tab-chevron" style="display:inline-block;transition:transform 0.15s;font-size:1.1em;margin-left:8px;opacity:0.85;"> ${ _dlTabFolded ? '▸' : '▾' } </span></h2> ` ;
2026-05-31 23:58:26 +09:00
html += '</div>' ;
Cookbook: scoring fixes, UI polish, false-finished + stale-state bug fixes
Backend (services/hwfit + routes):
- rank_models picks visible set by REQUESTED column, not always score —
sorting by Param now shows highest-param models PERIOD (incl. too_tight).
- New fit_only param. Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang
cannot serve them); default non-prequantized to BF16 on 2+ GPUs.
- AWQ / GPTQ-8bit get a -1.0 quality penalty (was 0.0, tied with FP8), so
FP8 wins when both fit.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above
M2.5 on equal composite score; >=100B integers not misread as versions.
- /api/cookbook/hf-latest no longer drops models without an "NB" pattern in
the repo id (MiniMax-M2.7, DeepSeek-V4-Pro etc. were silently filtered).
- Cached-model scan: atexit flushes models JSON even if the script is
killed mid-walk; each scan_dir wrapped in try/except; timeout 60s -> 180s.
- KB granularity for sub-MB sizes (was "0 MB" for 12 KB shells). New
"stalled" status for shells <1 MB with no .incomplete files.
- /api/cookbook/state POST guard: rejects "done" download tasks lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stops stale tabs from poisoning persisted state.
- hf_models.json: add zai-org/GLM-5.1; flip zai-org/GLM-5 quantization
Q4_K_M -> BF16 (it is the native base, not a quant).
Frontend (static/js):
- Scan/Download toolbar: quant defaults to All; ctx slider (8k/16k/32k/
50k/128k/Max) ported from origin/main with sort=fit on drag, sort=score
on Max. GPU toggle commits _activeCount to maxGpu on initial render. Fit
column header tagged with active budget (RAM / GPU / N GPU).
- Foldable Download admin-card: the Download h2 is the chevron trigger;
state persists in localStorage.
- Download card surfaces destination dir (Dir: <path>). Same dir on running
task row, font/color matched to uptime (9px Fira Code muted, opacity .4).
- Serve panel ctx text input always resets to model max on open. Sub-MB
cached models show with red "download stalled" badge.
- Bulk-select Cancel + Delete reset the Select button label on exit.
- Cookbook running: false-finished bug fixed — DOWNLOAD_OK or /snapshots/
required; bare "Download complete" no longer marks the task done after
the first config file. Clear button now sends tmux kill-session too.
True overall % for multi-shard downloads: ((N-1)+frac)/total instead of
hf_transfer per-shard aggregate.
- Diagnosis card simplified: removed fold toggle, copy button, dismiss X.
Suggestion font matches message body (12px).
- HF token field flashes green check + "Saved" on save.
- Cached scan no longer counts stalled rows as downloaded in Scan/Download.
CSS:
- dep Install button width pinned to 76px to match Installed split.
- task-sub row +1px; task-status badge gets margin-right 8px.
- Ctx slider styled like gallery editor sliders (thin pill rail, red thumb).
- Bulk-select cancel button top -3px -> -5px.
2026-06-03 16:32:20 +09:00
html += ` <div id="cookbook-dl-tab-fold-body" style=" ${ _dlTabFolded ? 'display:none;' : '' } "> ` ;
2026-05-31 23:58:26 +09:00
html += '<p class="memory-desc doclib-desc" style="margin-top:6px;">Download from <a href="https://huggingface.co/models" target="_blank" rel="noopener" style="color:var(--accent,var(--red));text-decoration:none;"><svg width="10" height="10" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" style="vertical-align:-1px;margin-right:1px;"><path d="M18 13v6a2 2 0 0 1-2 2H5a2 2 0 0 1-2-2V8a2 2 0 0 1 2-2h6"/><polyline points="15 3 21 3 21 9"/><line x1="10" y1="14" x2="21" y2="3"/></svg>HuggingFace</a> by pasting model link, or download directly in the Scan section below.</p>' ;
html += '<div class="hwfit-container" id="hwfit-container">' ;
// Section 1: Settings
const _es = _envState ;
if ( ! _es . servers ) _es . servers = [ ] ;
let _localSeen = false ;
_es . servers = _es . servers . filter ( s => {
const isLocal = ! s . host || s . host . toLowerCase ( ) === 'local' ;
if ( isLocal ) {
s . host = '' ;
if ( _localSeen ) return false ;
_localSeen = true ;
}
return true ;
} ) ;
if ( ! _localSeen ) {
_es . servers . unshift ( { host : '' , env : _es . env || 'none' , envPath : _es . envPath || '' , modelDir : '~/.cache/huggingface/hub' } ) ;
}
if ( _es . remoteHost && ! _es . servers . some ( s => s . host === _es . remoteHost ) ) {
_es . servers . push ( { host : _es . remoteHost , env : _es . env || 'none' , envPath : _es . envPath || '' , modelDir : '~/.cache/huggingface/hub' } ) ;
_persistEnvState ( ) ;
}
// NOTE: deliberately do NOT auto-pick the first remote server when no host is
// selected. That fallback turned any momentarily-empty remoteHost (a clobber,
// a render before the user's pick registered) into the first saved server,
// silently sending downloads to the wrong server. An empty selection means Local; the user
// chooses a remote server explicitly via the dropdown.
2026-06-02 12:15:41 +09:00
// Manual download input
2026-05-31 23:58:26 +09:00
html += ` <div style="margin-top:7px;margin-bottom:2px;display:flex;gap:4px;align-items:center;"> ` ;
if ( _es . servers . length > 1 ) {
html += ` <select class="cookbook-field-input hwfit-dl-server" id="hwfit-dl-server" style="height:28px;position:relative;top:0px;"> ` ;
html += _buildServerOpts ( true ) ;
html += ` </select> ` ;
} else {
html += ` <input type="hidden" id="hwfit-dl-server" value="local" /> ` ;
}
html += ` <button class="memory-toolbar-btn cookbook-dl-add-server" title="Add server in Settings" style="height:28px;">add server</button> ` ;
html += ` </div> ` ;
html += ` <div class="cookbook-dl-input" style="margin-top:0;"> ` ;
html += ` <input type="text" class="cookbook-dl-repo" id="cookbook-dl-repo" placeholder="org/model-name, HF URL, or org/model:QUANT_TAG" /> ` ;
html += ` <button class="cookbook-btn cookbook-dl-btn" id="cookbook-dl-btn">Download</button> ` ;
html += ` </div> ` ;
// Latest HF models that fit — collapsible card list
2026-06-02 12:15:41 +09:00
html += ` <div style="margin-top:5px;position:relative;top:-3px;"> ` ;
2026-05-31 23:58:26 +09:00
html += ` <div style="display:flex;gap:4px;align-items:center;"> ` ;
html += ` <button type="button" class="memory-toolbar-btn" id="cookbook-hf-latest-toggle" style="flex:1;text-align:left;height:26px;display:flex;align-items:center;gap:6px;border-radius:4px;"> ` ;
html += ` <span id="cookbook-hf-latest-arrow" style="display:inline-block;transition:transform 0.15s;pointer-events:none;"> \u 25B8</span> ` ;
html += ` <span style="pointer-events:none;">Trending models that fit your hardware</span> ` ;
html += ` </button> ` ;
html += ` <button type="button" class="memory-toolbar-btn" id="cookbook-hf-latest-refresh" title="Refresh" style="height:26px;width:26px;padding:0;border-radius:4px;"> \u 21BB</button> ` ;
html += ` </div> ` ;
html += ` <div id="cookbook-hf-latest-list" style="display:none;margin-top:4px;max-height:320px;overflow-y:auto;flex-direction:column;gap:4px;"></div> ` ;
html += ` </div> ` ;
Cookbook: scoring fixes, UI polish, false-finished + stale-state bug fixes
Backend (services/hwfit + routes):
- rank_models picks visible set by REQUESTED column, not always score —
sorting by Param now shows highest-param models PERIOD (incl. too_tight).
- New fit_only param. Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang
cannot serve them); default non-prequantized to BF16 on 2+ GPUs.
- AWQ / GPTQ-8bit get a -1.0 quality penalty (was 0.0, tied with FP8), so
FP8 wins when both fit.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above
M2.5 on equal composite score; >=100B integers not misread as versions.
- /api/cookbook/hf-latest no longer drops models without an "NB" pattern in
the repo id (MiniMax-M2.7, DeepSeek-V4-Pro etc. were silently filtered).
- Cached-model scan: atexit flushes models JSON even if the script is
killed mid-walk; each scan_dir wrapped in try/except; timeout 60s -> 180s.
- KB granularity for sub-MB sizes (was "0 MB" for 12 KB shells). New
"stalled" status for shells <1 MB with no .incomplete files.
- /api/cookbook/state POST guard: rejects "done" download tasks lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stops stale tabs from poisoning persisted state.
- hf_models.json: add zai-org/GLM-5.1; flip zai-org/GLM-5 quantization
Q4_K_M -> BF16 (it is the native base, not a quant).
Frontend (static/js):
- Scan/Download toolbar: quant defaults to All; ctx slider (8k/16k/32k/
50k/128k/Max) ported from origin/main with sort=fit on drag, sort=score
on Max. GPU toggle commits _activeCount to maxGpu on initial render. Fit
column header tagged with active budget (RAM / GPU / N GPU).
- Foldable Download admin-card: the Download h2 is the chevron trigger;
state persists in localStorage.
- Download card surfaces destination dir (Dir: <path>). Same dir on running
task row, font/color matched to uptime (9px Fira Code muted, opacity .4).
- Serve panel ctx text input always resets to model max on open. Sub-MB
cached models show with red "download stalled" badge.
- Bulk-select Cancel + Delete reset the Select button label on exit.
- Cookbook running: false-finished bug fixed — DOWNLOAD_OK or /snapshots/
required; bare "Download complete" no longer marks the task done after
the first config file. Clear button now sends tmux kill-session too.
True overall % for multi-shard downloads: ((N-1)+frac)/total instead of
hf_transfer per-shard aggregate.
- Diagnosis card simplified: removed fold toggle, copy button, dismiss X.
Suggestion font matches message body (12px).
- HF token field flashes green check + "Saved" on save.
- Cached scan no longer counts stalled rows as downloaded in Scan/Download.
CSS:
- dep Install button width pinned to 76px to match Installed split.
- task-sub row +1px; task-status badge gets margin-right 8px.
- Ctx slider styled like gallery editor sliders (thin pill rail, red thumb).
- Bulk-select cancel button top -3px -> -5px.
2026-06-03 16:32:20 +09:00
html += ` </div> ` ; // /#cookbook-dl-tab-fold-body (whole Download card body)
2026-05-31 23:58:26 +09:00
// Search section
2026-06-02 12:15:41 +09:00
html += '</div></div></div></div>' ;
2026-05-31 23:58:26 +09:00
html += '<div class="cookbook-group" data-backend-group="Search">' ;
html += '<div class="admin-card" style="flex:1;display:flex;flex-direction:column;overflow:hidden;">' ;
html += '<div style="display:flex;align-items:baseline;gap:8px;margin-bottom:2px;">' ;
html += '<h2 style="margin:0;padding:0;line-height:1;">Scan / Download</h2>' ;
html += '</div>' ;
html += '<p class="memory-desc doclib-desc" style="margin-top:6px;">Scans your hardware for what models you can run. Hardware is cached; hit the scan button to re-probe after changing GPUs.</p>' ;
html += '<div class="hwfit-toolbar" style="margin-top:9px;">' ;
html += '<select class="cookbook-field-input hwfit-usecase" id="hwfit-usecase" style="height:28px;">' ;
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
html += '<option value="general" selected>Standard</option><option value="coding">Coding</option>' ;
2026-05-31 23:58:26 +09:00
html += '<option value="reasoning">Reasoning</option><option value="chat">Chat</option>' ;
// Image tab removed — text→image gen is gone from this build (only inpaint
// remains, which uses its own settings panel). Vision (multimodal) stays.
html += '<option value="multimodal">Vision</option></select>' ;
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
// Engine sits next to the type filter so the "what category / which serving
// path" filters live together; Quant + Context are storage-format and budget
// levers, grouped to the right.
html += '<span class="hwfit-engine-wrap">' ;
Cookbook serve profiles and engine filter
* Cookbook: Engine filter + intelligent hardware-computed serve profiles
Two related Cookbook serving improvements for accurate, hardware-aware model
serving (especially on consumer GPUs that can only run GGUF/llama.cpp).
Engine filter
- New "Engine" dropdown (All / llama.cpp / vLLM / SGLang) beside the quant
picker. Pure client-side view filter over the fetched list via the same
_detectBackend() the serve commands use, so what you filter to is exactly what
would launch. Re-renders from cache (no refetch). Empty-state message + the
instant-cache-paint path account for it too.
Intelligent serve profiles (Quality / Balanced / Speed)
- services/hwfit/profiles.py: compute_serve_profiles() turns detected VRAM +
model size into concrete llama.cpp flags (n_gpu_layers, n_cpu_moe, cache-type,
context). Encodes the by-hand tuning: a too-big MoE offloads experts to CPU
instead of failing; a model that fits stays fully on GPU; quant tracks profile
intent; vision models keep image-encoder headroom. Reuses models.py VRAM math
so filtering and serving agree on what fits. Pure/deterministic (no t/s claims
— partial-offload speed isn't reliably predictable; fit is what's computed).
- /api/hwfit/profiles endpoint returns the profiles + the model's trained
context limit, with loose name matching (strips org/ prefix, -GGUF suffix,
quant tag) so a local GGUF folder name resolves to its catalog entry.
- _buildServeCmd (llama.cpp) now emits --n-cpu-moe / --flash-attn /
--cache-type-k/v when set, with llama-cpp-python fallback equivalents. It
previously only set -ngl/-c, which is why it OOM'd or ran slow.
- Serve panel: profile chips that fill the fields on click, plus CPU-MoE / KV
Cache / Flash Attn fields. Context is clamped to the model's trained limit
(and an absolute 1M sanity ceiling) on type/blur/profile-load and at launch —
fixes a crash where a stale 256k/16M preset + quantized KV cache caused an
amdgpu ErrorDeviceLost.
Tests: tests/test_serve_profiles.py (7) — offload vs full-GPU fit, never exceed
VRAM, context cap, launchable flags, vision headroom, no-GPU empty.
Checks: py_compile + node --check pass; pytest test_serve_profiles + test_hwfit_amd
green; verified live on an RDNA4 box (gfx1200) — Balanced lands ~ncm18 q4 128k,
matching hand-tuning.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
* Cookbook: make column-header sorting discoverable (incl. Newest)
Sorting in Cookbook is via clickable column headers (pewds' design), but the
headers had no visual cue that they're interactive — so sorting in general, and
the Newest sort on the Model header specifically, was undiscoverable.
- Style sortable headers as interactive: pointer cursor, hover underline, and
the active sort column bolded/highlighted. There was no CSS for
.hwfit-sortable / .hwfit-sort-active at all; this helps every existing sort,
not just Newest.
- The Model column header sorts by release_date (newest first), reusing the
existing header-click sort wiring and the "newest" SORT_KEY.
No new sort control — uses the existing column-header paradigm.
Checks: node --check passes.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
* Cookbook serve profiles: keep the on-disk file's quant fixed (don't propose Q6/Q2)
In the Serve tab the model is a specific GGUF file already on disk, so its quant
can't change — but the profiles were suggesting "Quality · Q6_K" / "Speed · Q2_K"
as if you could re-quantize it. That's meaningless when serving a fixed file.
- compute_serve_profiles gains serve_weights_gb / serve_quant. When set (SERVE
mode), the quant is locked to the file's and profiles differ only in the real
serving knobs — n_cpu_moe, KV-cache type, context. _weights_gb / _cpu_moe_for_budget
use the file's actual size instead of a quant-derived estimate. DOWNLOAD mode
(no override) still varies the quant to show download options.
- /api/hwfit/profiles accepts serve_weights_gb & serve_quant.
- The Serve panel parses the file's size (from m.size "20.6 GB") and quant (from
the repo/file name) and passes them, so profiles match what's actually served.
Result for a 20.6 GB Q4_K_M file: all three profiles stay Q4_K_M and differ by
KV/ctx/offload (Quality q8 KV 128k ncm21, Balanced q4 128k ncm17, Speed q4 32k
ncm15) — no nonsensical quant changes.
Tests: test_serve_mode_keeps_fixed_quant. Full serve-profile suite green (9).
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
* Cookbook serve: Vision toggle (auto-find mmproj) + live VRAM/RAM-spillover monitor
Two serve-panel additions:
1. **Vision toggle.** A "Vision" checkbox that serves the model with its
multimodal projector so it can read images. The mmproj path is resolved at
runtime (find mmproj-*.gguf next to the model), so dropping an mmproj file in
the model folder makes the toggle just work; `--mmproj … --image-max-tokens
1024` (native) / `--clip_model_path` (llama-cpp-python) only when on + found.
2. **Live GPU-memory monitor.** A readout that polls /api/cookbook/gpus every 4s
while the panel is open and shows VRAM used/total/%, free, and — crucially on
a discrete card — **RAM spillover** (AMD gtt_used_mb), with a plain-language
health hint: green/healthy, amber/tight, red/"spilled to RAM — slow (raise
CPU MoE or lower context)". Surfaces gtt_used_mb from the gpus endpoint
(previously read for total only and discarded for 'used').
Lets you see at a glance whether a config fits VRAM (fast) or is paging to system
RAM over PCIe (slow) instead of guessing.
Checks: node --check + py_compile pass.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
---------
Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-02 05:34:42 +02:00
html += '<select class="cookbook-field-input hwfit-engine" id="hwfit-engine" style="height:28px;" title="Filter by serving engine">' ;
html += '<option value="">Engine</option>' ;
html += '<option value="llamacpp">llama.cpp</option>' ;
html += '<option value="vllm">vLLM</option>' ;
html += '<option value="sglang">SGLang</option>' ;
html += '</select>' ;
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
html += '<span class="hwfit-help-chip hwfit-help-chip-inline hwfit-engine-help" title="Rule of thumb: GGUF on single GPU / CPU+RAM → llama.cpp (or Ollama). Safetensors on multi-GPU NVIDIA → vLLM. SGLang is a vLLM-class alternative, sometimes faster on big-MoE / long-context.">?</span>' ;
html += '</span>' ;
// Quant (Q4/Q8/…). Default is "All" so the list shows the best-scoring
// quant for every model instead of silently filtering to Q4.
html += '<span class="hwfit-quant-wrap">' ;
html += '<select class="cookbook-field-input hwfit-quant" id="hwfit-quant" style="height:28px;">' ;
html += '<option value="" selected>Quant: All</option>' ;
html += '<option value="Q4_K_M">Q4</option><option value="Q8_0">Q8</option>' ;
html += '<option value="Q6_K">Q6</option><option value="Q5_K_M">Q5</option>' ;
html += '<option value="Q3_K_M">Q3</option><option value="Q2_K">Q2</option>' ;
html += '<option value="AWQ-4bit">AWQ</option><option value="FP8">FP8</option><option value="FP4">FP4</option><option value="NVFP4">NVFP4</option></select>' ;
html += '<span class="hwfit-help-chip hwfit-help-chip-inline hwfit-quant-help" title="Lower quant tiers (Q2/Q3/Q4 / AWQ-4bit) are smaller, faster, and cheaper to run, at some quality loss. Higher tiers (Q8 / FP8 / FP16 / BF16) preserve more quality but need more VRAM. “All” shows the best-scoring quant per model — pick a specific one to filter.">?</span>' ;
html += '</span>' ;
2026-06-03 16:49:10 +09:00
// Ctx slider — lets you target a context length for fit estimates; the
// hwfit ranking uses _ctxValue() to factor that into VRAM math, so
// dragging this re-sorts the list toward models that fit your chosen ctx.
Cookbook: scoring fixes, UI polish, false-finished + stale-state bug fixes
Backend (services/hwfit + routes):
- rank_models picks visible set by REQUESTED column, not always score —
sorting by Param now shows highest-param models PERIOD (incl. too_tight).
- New fit_only param. Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang
cannot serve them); default non-prequantized to BF16 on 2+ GPUs.
- AWQ / GPTQ-8bit get a -1.0 quality penalty (was 0.0, tied with FP8), so
FP8 wins when both fit.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above
M2.5 on equal composite score; >=100B integers not misread as versions.
- /api/cookbook/hf-latest no longer drops models without an "NB" pattern in
the repo id (MiniMax-M2.7, DeepSeek-V4-Pro etc. were silently filtered).
- Cached-model scan: atexit flushes models JSON even if the script is
killed mid-walk; each scan_dir wrapped in try/except; timeout 60s -> 180s.
- KB granularity for sub-MB sizes (was "0 MB" for 12 KB shells). New
"stalled" status for shells <1 MB with no .incomplete files.
- /api/cookbook/state POST guard: rejects "done" download tasks lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stops stale tabs from poisoning persisted state.
- hf_models.json: add zai-org/GLM-5.1; flip zai-org/GLM-5 quantization
Q4_K_M -> BF16 (it is the native base, not a quant).
Frontend (static/js):
- Scan/Download toolbar: quant defaults to All; ctx slider (8k/16k/32k/
50k/128k/Max) ported from origin/main with sort=fit on drag, sort=score
on Max. GPU toggle commits _activeCount to maxGpu on initial render. Fit
column header tagged with active budget (RAM / GPU / N GPU).
- Foldable Download admin-card: the Download h2 is the chevron trigger;
state persists in localStorage.
- Download card surfaces destination dir (Dir: <path>). Same dir on running
task row, font/color matched to uptime (9px Fira Code muted, opacity .4).
- Serve panel ctx text input always resets to model max on open. Sub-MB
cached models show with red "download stalled" badge.
- Bulk-select Cancel + Delete reset the Select button label on exit.
- Cookbook running: false-finished bug fixed — DOWNLOAD_OK or /snapshots/
required; bare "Download complete" no longer marks the task done after
the first config file. Clear button now sends tmux kill-session too.
True overall % for multi-shard downloads: ((N-1)+frac)/total instead of
hf_transfer per-shard aggregate.
- Diagnosis card simplified: removed fold toggle, copy button, dismiss X.
Suggestion font matches message body (12px).
- HF token field flashes green check + "Saved" on save.
- Cached scan no longer counts stalled rows as downloaded in Scan/Download.
CSS:
- dep Install button width pinned to 76px to match Installed split.
- task-sub row +1px; task-status badge gets margin-right 8px.
- Ctx slider styled like gallery editor sliders (thin pill rail, red thumb).
- Bulk-select cancel button top -3px -> -5px.
2026-06-03 16:32:20 +09:00
html += '<label class="hwfit-ctx-control" title="Context length for fit estimates. Lower it to find more models that could fit your hardware.">' ;
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
html += '<span>Context</span><span class="hwfit-help-chip hwfit-help-chip-inline" title="Context length. Lower it to find more models that could fit your hardware; raise it when you need longer chats or documents.">?</span><input type="range" id="hwfit-context" min="0" max="5" step="1" value="3" />' ;
Cookbook: scoring fixes, UI polish, false-finished + stale-state bug fixes
Backend (services/hwfit + routes):
- rank_models picks visible set by REQUESTED column, not always score —
sorting by Param now shows highest-param models PERIOD (incl. too_tight).
- New fit_only param. Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang
cannot serve them); default non-prequantized to BF16 on 2+ GPUs.
- AWQ / GPTQ-8bit get a -1.0 quality penalty (was 0.0, tied with FP8), so
FP8 wins when both fit.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above
M2.5 on equal composite score; >=100B integers not misread as versions.
- /api/cookbook/hf-latest no longer drops models without an "NB" pattern in
the repo id (MiniMax-M2.7, DeepSeek-V4-Pro etc. were silently filtered).
- Cached-model scan: atexit flushes models JSON even if the script is
killed mid-walk; each scan_dir wrapped in try/except; timeout 60s -> 180s.
- KB granularity for sub-MB sizes (was "0 MB" for 12 KB shells). New
"stalled" status for shells <1 MB with no .incomplete files.
- /api/cookbook/state POST guard: rejects "done" download tasks lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stops stale tabs from poisoning persisted state.
- hf_models.json: add zai-org/GLM-5.1; flip zai-org/GLM-5 quantization
Q4_K_M -> BF16 (it is the native base, not a quant).
Frontend (static/js):
- Scan/Download toolbar: quant defaults to All; ctx slider (8k/16k/32k/
50k/128k/Max) ported from origin/main with sort=fit on drag, sort=score
on Max. GPU toggle commits _activeCount to maxGpu on initial render. Fit
column header tagged with active budget (RAM / GPU / N GPU).
- Foldable Download admin-card: the Download h2 is the chevron trigger;
state persists in localStorage.
- Download card surfaces destination dir (Dir: <path>). Same dir on running
task row, font/color matched to uptime (9px Fira Code muted, opacity .4).
- Serve panel ctx text input always resets to model max on open. Sub-MB
cached models show with red "download stalled" badge.
- Bulk-select Cancel + Delete reset the Select button label on exit.
- Cookbook running: false-finished bug fixed — DOWNLOAD_OK or /snapshots/
required; bare "Download complete" no longer marks the task done after
the first config file. Clear button now sends tmux kill-session too.
True overall % for multi-shard downloads: ((N-1)+frac)/total instead of
hf_transfer per-shard aggregate.
- Diagnosis card simplified: removed fold toggle, copy button, dismiss X.
Suggestion font matches message body (12px).
- HF token field flashes green check + "Saved" on save.
- Cached scan no longer counts stalled rows as downloaded in Scan/Download.
CSS:
- dep Install button width pinned to 76px to match Installed split.
- task-sub row +1px; task-status badge gets margin-right 8px.
- Ctx slider styled like gallery editor sliders (thin pill rail, red thumb).
- Bulk-select cancel button top -3px -> -5px.
2026-06-03 16:32:20 +09:00
html += '<output id="hwfit-context-label">50k</output></label>' ;
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
// Search lives at the far right of the toolbar so the controls (Type/Quant/
// Engine/Context) read as a row of compact filters followed by free-text.
html += '<input type="text" class="cookbook-field-input hwfit-search" id="hwfit-search" placeholder="Search models..." style="flex:1;" />' ;
2026-05-31 23:58:26 +09:00
html += '</div>' ;
html += '<div class="hwfit-toolbar" style="margin-top:7px;">' ;
html += '<select class="cookbook-field-input hwfit-server-select" id="hwfit-server-select" style="height:28px;min-width:88px;position:relative;top:0px;">' ;
html += _buildServerOpts ( false ) ;
html += '</select>' ;
html += '<div class="hwfit-gpu-toggles" id="hwfit-gpu-toggles"></div>' ;
// Scan/refresh button (icon-only) where the quant dropdown used to sit.
html += '<button type="button" class="hwfit-gpu-btn" id="hwfit-rescan" title="Re-scan hardware" style="flex-shrink:0;position:relative;top:-3px;left:-1px;">↻ RESCAN</button>' ;
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
html += '<button type="button" class="hwfit-gpu-btn hwfit-hw-manual-btn" id="hwfit-hw-manual-btn" title="Set hardware manually" style="flex-shrink:0;position:relative;top:-3px;left:-1px;display:inline-flex;align-items:center;gap:3px;"><svg width="10" height="10" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2.2" stroke-linecap="round" stroke-linejoin="round" style="flex-shrink:0;"><path d="M12 20h9"/><path d="M16.5 3.5a2.121 2.121 0 0 1 3 3L7 19l-4 1 1-4Z"/></svg>EDIT</button>' ;
Cookbook serve profiles and engine filter
* Cookbook: Engine filter + intelligent hardware-computed serve profiles
Two related Cookbook serving improvements for accurate, hardware-aware model
serving (especially on consumer GPUs that can only run GGUF/llama.cpp).
Engine filter
- New "Engine" dropdown (All / llama.cpp / vLLM / SGLang) beside the quant
picker. Pure client-side view filter over the fetched list via the same
_detectBackend() the serve commands use, so what you filter to is exactly what
would launch. Re-renders from cache (no refetch). Empty-state message + the
instant-cache-paint path account for it too.
Intelligent serve profiles (Quality / Balanced / Speed)
- services/hwfit/profiles.py: compute_serve_profiles() turns detected VRAM +
model size into concrete llama.cpp flags (n_gpu_layers, n_cpu_moe, cache-type,
context). Encodes the by-hand tuning: a too-big MoE offloads experts to CPU
instead of failing; a model that fits stays fully on GPU; quant tracks profile
intent; vision models keep image-encoder headroom. Reuses models.py VRAM math
so filtering and serving agree on what fits. Pure/deterministic (no t/s claims
— partial-offload speed isn't reliably predictable; fit is what's computed).
- /api/hwfit/profiles endpoint returns the profiles + the model's trained
context limit, with loose name matching (strips org/ prefix, -GGUF suffix,
quant tag) so a local GGUF folder name resolves to its catalog entry.
- _buildServeCmd (llama.cpp) now emits --n-cpu-moe / --flash-attn /
--cache-type-k/v when set, with llama-cpp-python fallback equivalents. It
previously only set -ngl/-c, which is why it OOM'd or ran slow.
- Serve panel: profile chips that fill the fields on click, plus CPU-MoE / KV
Cache / Flash Attn fields. Context is clamped to the model's trained limit
(and an absolute 1M sanity ceiling) on type/blur/profile-load and at launch —
fixes a crash where a stale 256k/16M preset + quantized KV cache caused an
amdgpu ErrorDeviceLost.
Tests: tests/test_serve_profiles.py (7) — offload vs full-GPU fit, never exceed
VRAM, context cap, launchable flags, vision headroom, no-GPU empty.
Checks: py_compile + node --check pass; pytest test_serve_profiles + test_hwfit_amd
green; verified live on an RDNA4 box (gfx1200) — Balanced lands ~ncm18 q4 128k,
matching hand-tuning.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
* Cookbook: make column-header sorting discoverable (incl. Newest)
Sorting in Cookbook is via clickable column headers (pewds' design), but the
headers had no visual cue that they're interactive — so sorting in general, and
the Newest sort on the Model header specifically, was undiscoverable.
- Style sortable headers as interactive: pointer cursor, hover underline, and
the active sort column bolded/highlighted. There was no CSS for
.hwfit-sortable / .hwfit-sort-active at all; this helps every existing sort,
not just Newest.
- The Model column header sorts by release_date (newest first), reusing the
existing header-click sort wiring and the "newest" SORT_KEY.
No new sort control — uses the existing column-header paradigm.
Checks: node --check passes.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
* Cookbook serve profiles: keep the on-disk file's quant fixed (don't propose Q6/Q2)
In the Serve tab the model is a specific GGUF file already on disk, so its quant
can't change — but the profiles were suggesting "Quality · Q6_K" / "Speed · Q2_K"
as if you could re-quantize it. That's meaningless when serving a fixed file.
- compute_serve_profiles gains serve_weights_gb / serve_quant. When set (SERVE
mode), the quant is locked to the file's and profiles differ only in the real
serving knobs — n_cpu_moe, KV-cache type, context. _weights_gb / _cpu_moe_for_budget
use the file's actual size instead of a quant-derived estimate. DOWNLOAD mode
(no override) still varies the quant to show download options.
- /api/hwfit/profiles accepts serve_weights_gb & serve_quant.
- The Serve panel parses the file's size (from m.size "20.6 GB") and quant (from
the repo/file name) and passes them, so profiles match what's actually served.
Result for a 20.6 GB Q4_K_M file: all three profiles stay Q4_K_M and differ by
KV/ctx/offload (Quality q8 KV 128k ncm21, Balanced q4 128k ncm17, Speed q4 32k
ncm15) — no nonsensical quant changes.
Tests: test_serve_mode_keeps_fixed_quant. Full serve-profile suite green (9).
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
* Cookbook serve: Vision toggle (auto-find mmproj) + live VRAM/RAM-spillover monitor
Two serve-panel additions:
1. **Vision toggle.** A "Vision" checkbox that serves the model with its
multimodal projector so it can read images. The mmproj path is resolved at
runtime (find mmproj-*.gguf next to the model), so dropping an mmproj file in
the model folder makes the toggle just work; `--mmproj … --image-max-tokens
1024` (native) / `--clip_model_path` (llama-cpp-python) only when on + found.
2. **Live GPU-memory monitor.** A readout that polls /api/cookbook/gpus every 4s
while the panel is open and shows VRAM used/total/%, free, and — crucially on
a discrete card — **RAM spillover** (AMD gtt_used_mb), with a plain-language
health hint: green/healthy, amber/tight, red/"spilled to RAM — slow (raise
CPU MoE or lower context)". Surfaces gtt_used_mb from the gpus endpoint
(previously read for total only and discarded for 'used').
Lets you see at a glance whether a config fits VRAM (fast) or is paging to system
RAM over PCIe (slow) instead of guessing.
Checks: node --check + py_compile pass.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
---------
Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-02 05:34:42 +02:00
// Sort state — the clickable column headers read/write this (pewds' original
// sort paradigm). Newest is reachable by clicking the Model column header.
2026-05-31 23:58:26 +09:00
html += '<select class="cookbook-field-input hwfit-sort" id="hwfit-sort" style="display:none">' ;
2026-06-02 09:09:18 +07:00
html += '<option value="fit">Fit</option><option value="score">Score</option><option value="vram">VRAM</option>' ;
2026-05-31 23:58:26 +09:00
html += '<option value="speed">Speed</option><option value="params">Params</option>' ;
html += '<option value="context">Context</option></select>' ;
html += '</div>' ;
html += '<div class="hwfit-manual-panel hidden" id="hwfit-manual-panel">' ;
html += '<span class="hwfit-manual-note" style="font-size:10px;opacity:0.6;width:100%;margin-bottom:2px;">Simulator — these values REPLACE detected hardware.</span>' ;
html += '<select class="hwfit-manual-mode"><option value="gpu">GPU</option><option value="ram">RAM</option></select>' ;
html += '<label>GPUs<input class="hwfit-manual-gpus" type="text" inputmode="numeric" placeholder="1"></label>' ;
html += '<label>VRAM per GPU<input class="hwfit-manual-vram" type="text" inputmode="decimal" placeholder="8 GB"></label>' ;
html += '<label>Total RAM<input class="hwfit-manual-ram" type="text" inputmode="decimal" placeholder="32 GB"></label>' ;
html += '<select class="hwfit-manual-backend"><option value="cuda">CUDA</option><option value="rocm">ROCm</option></select>' ;
html += '<button type="button" class="hwfit-hw-manual-save">✓ Apply</button>' ;
html += '<button type="button" class="hwfit-hw-manual-clear">× Clear</button>' ;
html += '</div>' ;
html += '<div id="hwfit-hw-row" style="display:none;align-items:center;gap:4px;margin-top:3px;padding-top:2px;"><span style="font-size:10px;padding:2px 8px;border-radius:10px;background:color-mix(in srgb, var(--fg) 8%, transparent);color:var(--fg);opacity:0.7;white-space:nowrap;flex-shrink:0;position:relative;top:-1px;">Detected hardware</span><div class="hwfit-hw" id="hwfit-hw" style="flex:1;"></div></div>' ;
html += '<div class="hwfit-list" id="hwfit-list"></div>' ;
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
// Footer: link to the public discussion where users can request additions
// to the curated model list. Sits below the list so it reads as a callout
// after browsing, not a header.
html += '<div class="hwfit-list-footer" style="margin-top:8px;padding-top:6px;border-top:1px solid color-mix(in srgb, var(--border) 50%, transparent);font-size:9.5px;opacity:0.65;text-align:right;">'
+ 'Don\'t see a model? '
+ '<a href="https://github.com/pewdiepie-archdaemon/odysseus/discussions/1962" target="_blank" rel="noopener" style="color:var(--accent,var(--red));text-decoration:none;display:inline-flex;align-items:center;gap:4px;vertical-align:middle;">'
+ 'Request it →'
+ '<svg width="11" height="11" viewBox="0 0 16 16" fill="currentColor" aria-hidden="true" style="flex-shrink:0;"><path d="M8 0C3.58 0 0 3.58 0 8a8 8 0 0 0 5.47 7.59c.4.07.55-.17.55-.38 0-.19-.01-.82-.01-1.49-2.01.37-2.53-.49-2.69-.94-.09-.23-.48-.94-.82-1.13-.28-.15-.68-.52-.01-.53.63-.01 1.08.58 1.23.82.72 1.21 1.87.87 2.33.66.07-.52.28-.87.51-1.07-1.78-.2-3.64-.89-3.64-3.95 0-.87.31-1.59.82-2.15-.08-.2-.36-1.02.08-2.12 0 0 .67-.21 2.2.82.64-.18 1.32-.27 2-.27.68 0 1.36.09 2 .27 1.53-1.04 2.2-.82 2.2-.82.44 1.1.16 1.92.08 2.12.51.56.82 1.27.82 2.15 0 3.07-1.87 3.75-3.65 3.95.29.25.54.73.54 1.48 0 1.07-.01 1.93-.01 2.2 0 .21.15.46.55.38A8.013 8.013 0 0 0 16 8c0-4.42-3.58-8-8-8z"/></svg>'
+ '</a>'
+ '</div>' ;
2026-05-31 23:58:26 +09:00
html += '</div></div>' ;
// Serve group
html += '<div class="cookbook-group hidden" data-backend-group="Serve">' ;
html += '<div class="admin-card" style="flex:1;display:flex;flex-direction:column;overflow:hidden;">' ;
html += '<div style="display:flex;align-items:baseline;gap:8px;margin-bottom:2px;">' ;
html += '<h2 style="margin:0;padding:0;line-height:1;">Serve <span id="serve-stats" class="memory-count" style="font-size:0.6em;opacity:0.6;font-weight:normal"></span></h2>' ;
html += '</div>' ;
const _selSrv = _es . servers . find ( s => s . host === _es . remoteHost ) || _es . servers [ 0 ] || { } ;
const _srvDirs = ( Array . isArray ( _selSrv . modelDirs ) ? _selSrv . modelDirs : [ _selSrv . modelDir || '~/.cache/huggingface/hub' ] ) . map ( d => d . replaceAll ( '✕' , '' ) . replaceAll ( '✖' , '' ) . trim ( ) ) . filter ( Boolean ) ;
html += '<div class="cookbook-serve-dirs" style="margin-top:6px;">' ;
html += _srvDirs . map ( d => ` <span class="cookbook-serve-dir-pill"> ${ esc ( d ) } </span> ` ) . join ( '' ) ;
html += '<span class="cookbook-serve-dir-edit" title="Edit in Settings">edit</span>' ;
html += '</div>' ;
html += '<div style="display:flex;gap:4px;align-items:center;margin-top:4px;">' ;
html += '<select class="memory-sort-select" id="hwfit-cache-server" style="height:24px;">' + _buildServerOpts ( true ) + '</select>' ;
html += '<select class="memory-sort-select" id="serve-sort" style="height:24px;">' ;
html += '<option value="name">Name</option><option value="size-desc">Size \u2193</option><option value="size-asc">Size \u2191</option><option value="recent">Recent</option>' ;
html += '</select>' ;
html += '</div>' ;
html += '<div class="memory-toolbar" style="margin-top:8px;">' ;
html += '<div class="memory-category-filters">' ;
html += '<input type="text" class="memory-search-input" id="serve-search" placeholder="Search cached models\u2026" style="flex:1;min-width:120px;" />' ;
html += '<button class="memory-toolbar-btn" id="hwfit-cache-select">Select</button>' ;
html += '</div>' ;
html += '<div class="doclib-lang-chips" id="serve-tags"></div>' ;
html += '</div>' ;
html += '<div class="memory-bulk-bar hidden" id="serve-bulk-bar">' ;
html += '<label class="memory-bulk-check-all"><input type="checkbox" id="serve-select-all"> All</label>' ;
html += '<span id="serve-bulk-count" style="font-size:10px;opacity:0.5;">0 selected</span>' ;
html += '<button class="memory-toolbar-btn danger" id="serve-bulk-delete" style="position:relative;top:-3px;"><svg width="11" height="11" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" style="vertical-align:-1px;margin-right:3px;"><polyline points="3 6 5 6 21 6"/><path d="M19 6l-1 14a2 2 0 0 1-2 2H8a2 2 0 0 1-2-2L5 6"/><path d="M10 11v6"/><path d="M14 11v6"/></svg>Delete</button>' ;
html += '<button class="memory-toolbar-btn" id="serve-bulk-cancel" title="Cancel (Esc)" style="margin-left:4px;padding:3px 6px;position:relative;top:-3px;"><svg width="11" height="11" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2.5" stroke-linecap="round"><line x1="18" y1="6" x2="6" y2="18"/><line x1="6" y1="6" x2="18" y2="18"/></svg></button>' ;
html += '</div>' ;
html += '<div class="doclib-grid hwfit-cached-list" id="hwfit-cached-list"></div>' ;
html += '</div></div>' ;
// Dependencies tab
html += '<div class="cookbook-group hidden" data-backend-group="Dependencies">' ;
html += '<div class="admin-card" style="flex:1;display:flex;flex-direction:column;overflow:hidden;">' ;
html += '<div style="display:flex;align-items:center;gap:8px;margin-bottom:4px;">' ;
html += '<h2 style="margin:0;padding:0;line-height:1;">Dependencies</h2>' ;
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
// Rebuild llama.cpp button moved into the llama_cpp dep row (see _depRow);
// having it in the title polluted the section header.
2026-05-31 23:58:26 +09:00
html += '<span style="font-size:10px;opacity:0.5;margin-left:auto;">Server</span>' ;
html += '<select class="cookbook-field-input" id="hwfit-deps-server" style="height:28px;min-width:70px;">' ;
html += _buildServerOpts ( false ) ;
html += '</select>' ;
html += '</div>' ;
2026-06-01 04:50:50 +02:00
html += '<p class="memory-desc doclib-desc">Optional packages that extend Odysseus capabilities.</p>' ;
2026-05-31 23:58:26 +09:00
html += '<div class="doclib-grid" id="cookbook-deps-list"></div>' ;
html += '</div></div>' ;
// Settings tab
// Settings tab — split into two separate `.admin-card` blocks so the
// HF Token and Server config look like distinct panels (matches the
// Download tab's block-per-section layout).
html += '<div class="cookbook-group hidden cookbook-settings-stack" data-backend-group="Settings">' ;
// ── HuggingFace Token block ─────────────────────────────────────────
html += '<div class="admin-card" style="flex:0 0 auto;display:flex;flex-direction:column;">' ;
html += '<div style="display:flex;align-items:baseline;gap:8px;margin-bottom:2px;">' ;
html += '<h2 style="margin:0;padding:0;line-height:1;">HuggingFace Token</h2>' ;
html += '</div>' ;
html += '<p class="memory-desc doclib-desc">Personal access token for downloading gated and private models.</p>' ;
html += '<div class="memory-toolbar">' ;
html += ` <div style="display:flex;gap:4px;align-items:center;"> ` ;
// Bold green check shown when a token is stored (a placeholder can't style a
// single glyph, so it's its own element next to the input).
if ( _es . hfTokenConfigured ) {
html += ` <span class="hwfit-hf-check" title="Token stored" style="font-weight:800;color:var(--green,#50fa7b);font-size:15px;line-height:1;flex-shrink:0;position:relative;top:2px;">✓</span> ` ;
}
const hfPlaceholder = _es . hfTokenConfigured
? ` Stored ( ${ esc ( _es . hfTokenMasked || 'configured' ) } ) - enter a new token to replace `
: 'hf_...' ;
html += ` <input type="password" class="memory-search-input" id="hwfit-hftoken" value=" ${ esc ( _es . hfToken || '' ) } " placeholder=" ${ hfPlaceholder } " style="flex:1;" /> ` ;
html += ` </div> ` ;
html += '</div>' ;
html += '</div>' ;
// ── Servers block ───────────────────────────────────────────────────
html += '<div class="admin-card" style="flex:0 0 auto;display:flex;flex-direction:column;">' ;
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
html += '<div style="display:flex;align-items:baseline;gap:8px;margin-bottom:2px;margin-top:-4px;">' ;
2026-05-31 23:58:26 +09:00
html += '<h2 style="margin:0;padding:0;line-height:1;">Servers</h2>' ;
// Reuse the calendar +New pill: spinning plus, label fades in idea uses
// the same `.cal-add-btn-text` rules, so styling stays consistent.
html += '<button class="cal-add-btn cal-add-btn-text" id="cookbook-server-add" title="Add server" style="margin-left:auto;"><span class="cal-add-plus">+</span><span class="cal-add-label">Add</span></button>' ;
html += '</div>' ;
html += '<p class="memory-desc doclib-desc">Configure SSH servers, install Odysseus keys, choose model directories, and set the default server. Local is this machine.</p>' ;
html += '<div class="memory-toolbar cookbook-servers-toolbar" style="margin-top:4px;">' ;
html += ` <div id="cookbook-servers-list"> ` ;
for ( let i = 0 ; i < _es . servers . length ; i ++ ) {
html += _serverEntryHtml ( _es . servers [ i ] , i , _es . defaultServer || '' , false ) ;
}
html += ` </div> ` ;
html += '</div>' ;
html += '</div></div>' ;
body . innerHTML = html ;
_wireTabEvents ( body ) ;
// Auto-init What Fits
_hwfitInit ( ) ;
_hwfitFetch ( ) ;
}
// ── Public API ──
import * as Modals from './modalManager.js' ;
let _rendered = false ;
let _closeGen = 0 ;
// ESC while a Serve card is expanded should collapse just that card, not
// close the whole Cookbook modal. Capture-phase so we run before the
// modal manager's global ESC-to-close handler and can stop it.
if ( typeof window !== 'undefined' && ! window . _cookbookServeEscBound ) {
window . _cookbookServeEscBound = true ;
document . addEventListener ( 'keydown' , ( e ) => {
if ( e . key !== 'Escape' ) return ;
const modal = document . getElementById ( 'cookbook-modal' ) ;
if ( ! modal || modal . classList . contains ( 'hidden' ) ) return ;
// Layer 1: a model row in the scan/download list is highlighted —
// deselect it before doing anything else.
const activeRow = modal . querySelector ( '.hwfit-row-active' ) ;
if ( activeRow ) {
e . stopImmediatePropagation ( ) ;
e . preventDefault ( ) ;
activeRow . classList . remove ( 'hwfit-row-active' ) ;
return ;
}
const expanded = modal . querySelector ( '.memory-item.doclib-card-expanded' ) ;
if ( ! expanded ) return ; // nothing expanded — let the modal close normally
e . stopImmediatePropagation ( ) ;
e . preventDefault ( ) ;
// Collapse the card (mirror the toggle-close path in cookbookServe.js).
expanded . querySelector ( '.hwfit-serve-panel' ) ? . remove ( ) ;
expanded . classList . remove ( 'doclib-card-expanded' ) ;
expanded . style . flexDirection = '' ;
expanded . style . alignItems = '' ;
const list = expanded . closest ( '.hwfit-cached-list' ) || document . getElementById ( 'hwfit-cached-list' ) ;
if ( list ) { list . style . minHeight = '' ; list . style . maxHeight = '' ; }
} , true ) ; // capture
}
export async function open ( opts ) {
const modal = document . getElementById ( 'cookbook-modal' ) ;
if ( ! modal ) return ;
// Run any post-open intent (switch tab, prefill search, etc) after the
// current render pass so the target elements exist.
const _applyIntent = ( ) => {
if ( ! opts ) return ;
if ( opts . tab ) {
const t = modal . querySelector ( ` .cookbook-tab[data-backend=" ${ opts . tab } "] ` ) ;
if ( t && ! t . classList . contains ( 'active' ) ) t . click ( ) ;
}
if ( opts . usecase ) {
const u = document . getElementById ( 'hwfit-usecase' ) ;
if ( u && u . value !== opts . usecase ) { u . value = opts . usecase ; u . dispatchEvent ( new Event ( 'change' , { bubbles : true } ) ) ; }
}
if ( opts . serveSearch ) {
const s = document . getElementById ( 'serve-search' ) ;
if ( s ) { s . value = opts . serveSearch ; s . dispatchEvent ( new Event ( 'input' , { bubbles : true } ) ) ; }
}
} ;
// If minimized, restore in place — preserve all state
if ( Modals . isMinimized ( 'cookbook-modal' ) ) {
Modals . restore ( 'cookbook-modal' ) ;
_renderRunningTab ( ) ;
setTimeout ( _applyIntent , 0 ) ;
return ;
}
// If already visible, no-op (but still honour the intent)
if ( ! modal . classList . contains ( 'hidden' ) ) {
setTimeout ( _applyIntent , 0 ) ;
return ;
}
_setCookbookOpening ( true ) ;
try {
// Invalidate any pending close() animation handlers so they won't re-hide us
_closeGen ++ ;
// Clear any leftover inline styles from a previous swipe-dismiss or close animation
const _content = modal . querySelector ( '.modal-content' ) ;
if ( _content ) {
_content . classList . remove ( 'modal-closing' , 'sheet-ready' , 'cookbook-modal-entering' ) ;
_content . style . transform = '' ;
_content . style . transition = '' ;
_content . style . animation = '' ;
_content . style . opacity = '' ;
}
modal . style . display = '' ;
Modals . register ( 'cookbook-modal' , {
railBtnId : 'rail-cookbook' ,
sidebarBtnId : 'tool-cookbook-btn' ,
closeFn : ( ) => _doClose ( ) ,
restoreFn : ( ) => { _renderRunningTab ( ) ; } ,
} ) ;
_wireCookbookDrag ( modal ) ;
await _syncFromServer ( ) ;
// `_syncFromServer` lives in cookbookRunning.js and populates *its* _envState
// (a different object reference than this module's), then mirrors the merged
// state to localStorage. So ALWAYS hydrate our _envState from that mirror —
// on a successful sync it holds the freshly-fetched servers; on failure it
// holds the last-known state. Gating this on `!synced` left the render's
// _envState empty whenever sync succeeded → "servers don't show".
try { Object . assign ( _envState , _readStoredEnvState ( ) ) ; } catch { }
// Honour a user-set default server: always land on it when Cookbook opens, so
// every dropdown (scan/download/serve/cache/deps) starts on the same machine.
if ( _envState . defaultServer ) {
const _dk = _envState . defaultServer ;
if ( _dk === 'local' ) {
_envState . remoteHost = '' ; _envState . env = 'none' ; _envState . envPath = '' ; _envState . platform = '' ;
} else {
const _ds = ( _envState . servers || [ ] ) . find ( s => s . host === _dk ) ;
if ( _ds ) { _envState . remoteHost = _ds . host ; _envState . env = _ds . env || 'none' ; _envState . envPath = _ds . envPath || '' ; _envState . platform = _ds . platform || '' ; }
}
}
// Re-render on every open AFTER sync so the freshly-fetched state (servers,
// HF token, presets) is always reflected. Gating this to once-per-page used
// to freeze a stale/empty servers list whenever the first sync raced or
// returned before hydration — and since close/reopen doesn't reset the page,
// only a full reload recovered it. Re-rendering is cheap and the in-progress
// Running tab is rendered separately just below.
_renderRecipes ( ) ;
_rendered = true ;
_clearCookbookNotif ( ) ;
_renderRunningTab ( ) ;
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
// Self-heal: revive any download tasks whose tmux session is still alive
// but were persisted as done/error (covers the "restarted server while a
// big multi-shard download was in flight" case — the task survived in
// tmux, the cookbook just lost track of it).
try { _selfHealStaleTasks ( { oneShot : true } ) ; } catch { }
2026-05-31 23:58:26 +09:00
if ( _content ) {
// Put the panel in its entering state before it becomes visible. On
// mobile, showing first and adding the class a frame later can paint the
// sheet at its final position, which makes the slide-up look like a snap.
_content . classList . add ( 'cookbook-modal-entering' ) ;
}
modal . classList . remove ( 'hidden' ) ;
if ( _content ) {
void _content . offsetWidth ;
_content . addEventListener ( 'animationend' , ( ) => {
_content . classList . remove ( 'cookbook-modal-entering' ) ;
} , { once : true } ) ;
}
setTimeout ( _applyIntent , 0 ) ;
} finally {
_setCookbookOpening ( false ) ;
}
}
// Make the Cookbook modal draggable (it had no drag wiring at all). We do
// NOT supply a fsClass fullscreen here — that would cover the whole viewport
// incl. the sidebar. Instead tileManager.js handles maximize/tiling (its
// safe-rect sits the window NEXT TO the sidebar), same as tasks/gallery/etc.
let _cookbookDragWired = false ;
function _wireCookbookDrag ( modal ) {
if ( _cookbookDragWired || ! modal ) return ;
const content = modal . querySelector ( '.modal-content' ) ;
const header = modal . querySelector ( '.modal-header' ) ;
if ( ! content || ! header ) return ;
_cookbookDragWired = true ;
makeWindowDraggable ( modal , {
content , header ,
skipSelector : '.close-btn, .modal-close' ,
// Keep only the "close to the edge" dock gesture for Cookbook. The
// tileManager side snap is suppressed for this modal so there isn't a
// second, tighter edge state fighting the working one.
enableDock : true ,
} ) ;
}
function _doClose ( ) {
const modal = document . getElementById ( 'cookbook-modal' ) ;
if ( ! modal ) return ;
const content = modal . querySelector ( '.modal-content' ) ;
const myGen = ++ _closeGen ;
if ( content && ! content . classList . contains ( 'modal-closing' ) ) {
content . classList . add ( 'modal-closing' ) ;
content . addEventListener ( 'animationend' , ( ) => {
if ( myGen !== _closeGen ) return ;
modal . classList . add ( 'hidden' ) ;
content . classList . remove ( 'modal-closing' ) ;
} , { once : true } ) ;
setTimeout ( ( ) => {
if ( myGen !== _closeGen ) return ;
if ( ! modal . classList . contains ( 'hidden' ) ) { modal . classList . add ( 'hidden' ) ; content . classList . remove ( 'modal-closing' ) ; }
} , 250 ) ;
} else {
modal . classList . add ( 'hidden' ) ;
}
}
export function close ( ) {
// Full close — fires registered closeFn, removes badge, unregisters
if ( Modals . isRegistered ( 'cookbook-modal' ) ) {
Modals . close ( 'cookbook-modal' ) ;
} else {
_doClose ( ) ;
}
}
export function isVisible ( ) {
const modal = document . getElementById ( 'cookbook-modal' ) ;
if ( ! modal ) return false ;
if ( Modals . isMinimized ( 'cookbook-modal' ) ) return false ;
return ! modal . classList . contains ( 'hidden' ) ;
}
// Close button
document . addEventListener ( 'DOMContentLoaded' , ( ) => {
const closeBtn = document . getElementById ( 'close-cookbook-modal' ) ;
if ( closeBtn ) closeBtn . addEventListener ( 'click' , close ) ;
const modal = document . getElementById ( 'cookbook-modal' ) ;
if ( modal ) {
modal . addEventListener ( 'click' , ( e ) => {
if ( uiModule . isTouchInsideModal ( ) ) return ;
if ( e . target === modal ) close ( ) ;
} ) ;
}
} ) ;
// ── Initialize sub-modules ──
// Shared SSH-port resolver — sub-modules use this via the shared bundle
// instead of redefining it. Kept here as the single source of truth.
function _sshPrefix ( port ) {
return port && port !== '22' ? ` -p ${ port } ` : '' ;
}
const shared = {
_envState ,
_sshCmd ,
_getPort ,
_sshPrefix ,
_getPlatform ,
_isWindows ,
Add macOS Apple Silicon Cookbook support
* Add Apple Silicon (Metal) GPU detection and unified-memory fit tuning
hardware.py detects Apple Silicon locally and over SSH, reporting
backend=metal, the chip name, and a RAM-scaled fraction of unified
memory as the usable GPU budget. fit.py gains an M1-M4 memory-bandwidth
table for realistic tok/s and drops vLLM-only formats (AWQ/GPTQ/FP8)
that can't be served on Metal.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 32ac81dbc680361463a088dae867d555d5a79c3b)
* Generate macOS/Metal serve commands and surface the Metal GPU
cookbook_routes.py adds a macOS serve path (Ollama, Metal-aware
llama.cpp build using `sysctl hw.ncpu` instead of `nproc`, and a clear
error if vLLM is attempted). The frontend defaults Metal serving to
llama.cpp and offers llama.cpp/Ollama instead of vLLM/SGLang. The
odysseus-cookbook CLI's `gpus` command reports the Metal GPU via
sysctl/vm_stat.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 4ba01ce25d256ae032029898f361c824a34fcd4b)
* Add launchd LaunchAgent for macOS (systemd equivalent)
com.odysseus.ui.plist + install-service-macos.sh run Odysseus at login
and restart on crash, the macOS counterpart to odysseus-ui.service. The
installer auto-fills paths from the venv, so there's no hand-editing.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 3d4b6b2c7b8b31af32201ed278115df9a559dea9)
* Document macOS install (brew, Ollama, AirPlay port, launchd)
README + setup.py cover the Homebrew / Apple Silicon path: brew install
python@3.11 tmux ollama, Metal serving via Ollama/llama.cpp, the launchd
service, and the macOS AirPlay Receiver conflict on ports 7000/5000.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 8dc9a3578a1726f070ed9f75c0958ae291a6d966)
* Add downloadable macOS launcher app builder
build-macos-app.sh generates dist/Odysseus.app and a drag-to-Applications
dist/Odysseus.dmg. The app starts the local server from this repo's venv and
opens the UI in a chrome-less app window (Chromium --app mode, falling back to
the default browser). It's a launcher wrapper — it drives the venv rather than
bundling Python — so the install path is baked in at build time.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
(cherry picked from commit 7927940c3810ee34640803b198d334a6ac93474d)
* Harden macOS Cookbook support: hide MLX, fix Metal build cache
Builds on the adopted PR #213 macOS/Metal work with two fixes and tests:
- fit.py: always drop MLX-quantized models. Odysseus only generates serve
commands for llama.cpp/Ollama (Metal) and vLLM/SGLang (CUDA); MLX needs the
mlx_lm runtime and the catalog's MLX repos ship no GGUF alternative, so they
were surfaced on Apple Silicon but could never be served.
- cookbook_routes.py (macOS branch only): `rm -rf build` before configure so a
poisoned CMakeCache from a prior failed CUDA attempt can't make every later
build fail; explicit -DCMAKE_BUILD_TYPE=Release; a clear "brew install cmake"
hint if cmake is missing. Linux/CUDA path unchanged.
- tests/test_hwfit_macos.py: MLX hidden on metal, MLX still hidden on CUDA
(regression guard), Metal detection on Apple Silicon, and skipped on
Linux/Intel (proves non-macOS detection is untouched).
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Propagate unified_memory flag and document macOS GPU/Docker caveat
- hardware.py: detect_system now carries the unified_memory flag from GPU
detection into the system dict (it was set by _detect_apple_silicon / AMD-APU
detection but dropped during result assembly, so the API always reported
null). Lets callers distinguish unified from discrete VRAM.
- README: prominent warning that Docker on Apple Silicon can't reach the Metal
GPU (runs a Linux VM) — Cookbook must run natively for GPU serving; fix stale
text that said Cookbook recommends MLX models (now hidden as unservable).
- test: detect_system propagates unified_memory.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Put Odysseus's venv bin on PATH for cookbook runners
Native (non-Docker) installs run from a virtualenv whose bin holds the `hf` CLI
and `python3` the cookbook download/serve tmux scripts shell out to. Those
scripts start in a fresh login shell with the venv NOT activated, so on a native
macOS install `hf download` failed with "hf: command not found" — and the
`pip --user` self-heal missed because macOS has no bare `pip` command.
- cookbook_helpers.py: _local_tooling_path_export() — pure helper returning a
PATH export for the running interpreter's bin dir (escaped for double quotes).
- cookbook_routes.py: download + serve runners prepend that dir on local runs
(gated off SSH/Windows); swap the `pip` install fallbacks to `python3 -m pip`.
- tests: helper output for normal and spaced paths.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Document macOS llama.cpp serving prerequisites
Clarify the two serving paths on Apple Silicon: the recommended zero-build
route (brew install llama.cpp ships a Metal llama-server Cookbook finds on PATH),
and the from-source fallback, which requires cmake + Xcode Command Line Tools.
Without those the build is skipped and serving silently degrades to a slow CPU
build, so new users now know to install them (or use the prebuilt) up front.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Recommend only GGUF-servable models on Metal
Apple Silicon's only serving engines are llama.cpp and Ollama, both GGUF-only
(vLLM/SGLang are CUDA/ROCm and don't run on macOS). The catalog tags raw
safetensors repos with a default Q4_K_M quant, so the fit-ranking was
recommending ~397/501 models that have no GGUF and fail to serve on Metal with
"No GGUF found" (e.g. microsoft/Phi-mini-MoE-instruct).
Drop any model without a real GGUF (is_gguf/gguf_sources) on Apple Silicon —
subsumes the previous AWQ/GPTQ/FP8 special-case into one rule. On CUDA these
stay visible since vLLM serves safetensors directly. Metal recommendations go
501 -> 104, all actually servable.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Remove macOS launchd LaunchAgent (cherry-picked extra)
Drop the launchd service from the PR #213 cherry-picks: the
install-service-macos.sh installer, the com.odysseus.ui.plist template, and the
README section documenting them. Tangential to the core Cookbook/Metal support
and not wanted. The build-macos-app.sh launcher is kept.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Add one-command macOS quick start (start-macos.sh)
Running Odysseus natively on a Mac previously meant ~7 manual terminal steps
(brew deps, venv, activate, pip, setup.py, uvicorn with the right port) — not
friendly for a generic macOS user, and the native run is required because Docker
on macOS can't reach the Metal GPU.
- start-macos.sh: installs Homebrew deps (python@3.11, tmux, prebuilt Metal
llama.cpp), creates the venv, installs requirements, runs setup, and launches
on a non-AirPlay port (7860). Idempotent; re-run to start again.
- README: the Apple Silicon section now leads with this one-command quick start
and the clickable .app, with engine/port/manual details folded into a
collapsible block. Added a pointer at the top of the manual-install section.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* macOS quick start: auto-open browser when ready
The "open this URL" line scrolled out of view as uvicorn kept logging after it,
so users missed it. Now start-macos.sh waits (in the background) until the
server accepts connections, prints a boxed "ready" banner at that point (i.e.
after the startup burst, not before), and opens the URL in the default browser
automatically. Skippable with ODYSSEUS_NO_OPEN=1 for headless/SSH use.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Don't assume/force a specific Python version on macOS
The README claimed "system Python is 3.9" — a machine-specific generalization
that's often wrong (macOS ships no recent Python by default; many users already
have 3.11+). Make it generic, and make start-macos.sh detect an existing
Python 3.11+ and use it, only installing python@3.11 when none is found instead
of forcing it on top of the user's Python.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Align start-macos.sh venv path with build-macos-app.sh
start-macos.sh created the environment in .venv/, but build-macos-app.sh and
the manual install steps use venv/ — so the clickable .app wouldn't reuse the
quick-start's environment and would rebuild a second one. Use venv/ everywhere.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* README: state clearly that MLX is unsupported on Apple Silicon
Odysseus has no mlx_lm runtime; it serves GGUF (llama.cpp/Ollama) and CUDA
(vLLM/SGLang) only. MLX-only models can't run on a Mac and are hidden from
Cookbook — make that explicit in both the quick start and the details.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* start-macos.sh: build the venv with an arm64 Python on Apple Silicon
A clean-room run surfaced this: with a universal2/x86 Python (e.g. the
python.org installer under /usr/local), the venv's compiled extensions install
as arm64 but get loaded as x86_64 when launched from the .app bundle, so it
crashes with "incompatible architecture (have arm64, need x86_64)". The terminal
run happened to work only because a universal binary defaults to arm64 there.
On Apple Silicon, look only under /opt/homebrew (arm64-only) for the build
Python, and install Homebrew's python@3.11 if none is present — so the venv is
arm64-only and launches correctly from both the terminal and the .app. Intel
and non-mac paths are unchanged. Verified end-to-end in a clean clone: .app now
boots on Metal with no arch error.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* Address dev-exp review: macOS setup robustness + doc/UX fixes
From the voltagent dev-exp review of the branch:
- README: fix broken anchor links (the em-dash heading produced a slug the links
didn't match); simplify the heading to a stable slug.
- cookbook_routes.py: add /opt/homebrew/bin and /usr/local/bin to the serve PATH
so a brew-installed llama-server/ollama is found instead of falling back to a
slow source build.
- start-macos.sh: guard against an empty Python path; fail fast with a clear
message on port-in-use; ERR trap with a "safe to re-run" message; show pip
progress (drop --quiet on the slow requirements install); stop the background
browser-opener cleanly on exit/Ctrl+C (no orphaned poller).
- setup.py: bind hint to 127.0.0.1; suppress the manual run-hint when launched
by start-macos.sh (ODYSSEUS_SKIP_RUN_HINT) so the URL isn't contradictory.
- build-macos-app.sh: the .app only opens the browser once the server is
actually ready (not after the readiness timeout).
- cookbookServe.js: drop "Diffusers" from the Metal backend picker —
diffusion_server.py is CUDA-only, so it was an unservable option on macOS.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
---------
Co-authored-by: yunggilja <yunggilja@gmail.com>
Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-01 15:29:19 +09:30
_isMetal ,
2026-05-31 23:58:26 +09:00
_buildEnvPrefix ,
_buildServeCmd ,
_shellQuote ,
_psQuote ,
_detectBackend ,
_detectToolParser ,
_detectModelOptimizations ,
_loadPresets ,
_savePresets ,
_copyText ,
_persistEnvState ,
2026-06-01 18:58:06 +05:30
_refreshDependencies : _fetchDependencies ,
2026-05-31 23:58:26 +09:00
_getGpuToggleTotal : ( ) => _gpuToggleTotal ,
modelLogo ,
esc ,
} ;
// Init running module (adds task management, auto-fix, launch, background monitor)
initRunning ( {
... shared ,
} ) ;
// Init download module (adds SSE, panel rendering, download commands)
initDownload ( {
... shared ,
_addTask ,
_renderRunningTab ,
_loadTasks ,
_saveTasks ,
} ) ;
// Init serve module (adds cached models, serve panels, launch)
initServe ( {
... shared ,
_launchServeTask ,
_retryDownload ,
_nextAvailablePort ,
} ) ;
// ── Re-exports for cookbook-diagnosis.js and cookbook-hwfit.js ──
// These modules import from cookbook.js, so we re-export what they need
export {
_loadTasks , _saveTasks , _addTask , _removeTask ,
_tmuxCmd , _renderRunningTab ,
_launchServeTask , _serveAutoFix , _serveAutoRetry , _serveAutoRetryReplace , _serveAutoRetryRemove ,
_startBackgroundMonitor ,
_setPanelField , _setPanelCheckbox ,
_wirePanelEvents , _runPanelCmd , _runModelDownload , _buildDownloadCmd ,
_serverByVal , _isLocalEntry ,
} ;
const cookbookModule = { open , close , isVisible , startBackgroundMonitor : _startBackgroundMonitor } ;
export default cookbookModule ;