2026-05-31 23:58:26 +09:00
// ============================================
// COOKBOOK RUNNING SUB-MODULE
// Running tasks tab: task cards, status monitoring,
// stop/restart, diagnosis, auto-fix, background monitor
// ============================================
import uiModule from './ui.js' ;
import { _diagnose , _showDiagnosis , _clearDiagnosis } from './cookbook-diagnosis.js' ;
fix: make transient dropdown/popup menus close on Escape
The global Escape arbiter in ui.js only sees `.modal` elements, so the many
ad-hoc dropdowns and context popups that are built on the fly and appended to
<body> ignored Escape entirely: document-library card/chat menus, chat
context/stats/overflow popups, cookbook serve & running menus, calendar event
menus, and compare pane menus.
Add a small DOM-free dismissal registry (static/js/escMenuStack.js). Menus
register a dismiss callback while open, and the arbiter closes the
most-recently-opened one first, so a menu opened over a modal closes before the
modal. bindMenuDismiss() wires the ubiquitous "append-to-body, close on outside
click" idiom to both the outside-click listener and the Escape stack in one
call, and dismissOrRemove() lets the pre-existing bulk removers (scroll/swipe/
modal-dismiss cleanup, reopen sweeps) tear a menu down through its real teardown
instead of orphaning its stack entry.
Covers ~14 menus across documentLibrary, chatRenderer, cookbookServe,
cookbookRunning, calendar, and compare/panes. Every teardown path — item click,
outside click, swipe, toggle, rebuild, bulk cleanup — routes through the
registry so no entry is ever stranded.
tests/test_esc_menu_stack_js.py pins the registry's LIFO and
exactly-one-per-press guarantees (node-driven; skips when node is absent).
2026-06-01 14:23:22 -04:00
import { registerMenuDismiss } from './escMenuStack.js' ;
2026-06-03 12:23:35 +08:00
import { computeProgressSignal } from './cookbookProgressSignal.js' ;
fix(cookbook): only block model launch on real port collisions (#4760)
* Fix #4507: only block model launch on real port collisions
Quick-run hardcoded port 8000 and never called _nextAvailablePort(), so
every launch collided. Both pre-launch guards (serve panel + quick-run)
were count-based and fired regardless of port.
- quick-run now auto-assigns a free port (8080 for llama.cpp)
- both guards parse the new port and only prompt on a real overlap,
stopping only the colliding serve
- dialog reports the actual port instead of a hardcoded 8000
* refactor(cookbook): share _taskPort for port parsing; auto-assign llama.cpp port
Addresses review on #4760:
- _taskPort regex now matches --port= as well as --port (space)
- _nextAvailablePort and both launch guards reuse _taskPort instead of inline regex
- quick-run llama.cpp no longer pins 8080, so two can run concurrently
* fix(cookbook): _taskPort also parses -p; add port-parsing tests
Addresses review on #4760:
- _taskPort now matches -p <n> too, so it's the complete single reader
(was missing the short flag that other readers already handle)
- add tests/test_cookbook_port_parsing_js.py covering the port forms,
shared-reader reuse, and llama.cpp auto-assign
* test(cookbook): extract pure port helpers and test behavior
Addresses review on #4760: the prior tests only asserted source strings.
- extract portOf() and nextFreePort() into static/js/cookbookPorts.js
- cookbookRunning.js imports them; _taskPort and _nextAvailablePort delegate
- tests run the helpers via node and assert real behavior: all port forms
(--port, --port=, -p, -p=), next-free-port skipping taken ports, and the
same-port-clash / different-port-coexist outcome
---------
Co-authored-by: samy <samy@odysseus.boukouro.com>
2026-06-24 13:44:09 -04:00
import { portOf , nextFreePort } from './cookbookPorts.js' ;
2026-05-31 23:58:26 +09:00
// Human-friendly badge label for a task's internal status. Avoids surfacing
// the word "error" in the sidebar — a server the user stopped or one that
// quit cleanly reads as "stopped", not "error".
function _statusLabel ( status , type ) {
if ( status === 'running' && type === 'download' ) return 'downloading' ;
if ( status === 'done' && type === 'download' ) return 'finished' ;
if ( status === 'error' ) return 'stopped' ;
return status || '' ;
}
// Single source of truth for what a task's status badge shows + its style class.
// Crucially, a serve task that's still coming up shows its live phase
// ("loading 45%", "warming up", …) rather than the generic "running" — they're
// the same state, so the badge shouldn't flip between two different labels on
// every re-render. Returns { text, cls } where cls is appended after
// "cookbook-task-status" ('' = the neutral loading style).
function _taskBadge ( task ) {
if ( task . _unreachable && task . status === 'running' ) return { text : 'unreachable' , cls : 'cookbook-task-error' } ;
2026-06-21 11:02:35 +00:00
if ( task . type === 'download' && task . status === 'running' ) {
2026-06-29 03:02:58 +00:00
const progress = String ( task . progress || '' ) . trim ( ) ;
return { text : progress || _statusLabel ( task . status , task . type ) , cls : 'cookbook-task-downloading' } ;
2026-06-21 11:02:35 +00:00
}
2026-05-31 23:58:26 +09:00
if ( task . type === 'serve' && task . status === 'running' && task . progress ) {
// Same green "running" pill — just with dynamic phase text, so it doesn't
// read as a different status while the server is coming up.
return { text : task . progress , cls : 'cookbook-task-running' } ;
}
return { text : _statusLabel ( task . status , task . type ) , cls : 'cookbook-task-' + task . status } ;
}
2026-06-22 01:49:15 +00:00
function _ggufDisplayPartFromPath ( path ) {
const parts = String ( path || '' ) . split ( '/' ) . filter ( Boolean ) ;
const file = parts [ parts . length - 1 ] || '' ;
const dir = parts . length > 1 ? parts [ parts . length - 2 ] : '' ;
const text = ` ${ dir } ${ file } ` ;
const quant = text . match ( /\b(?:UD-)?(?:IQ[1-8]_[A-Z0-9]+|Q[2-8]_K_[MLS]|Q[2-8]_[0-9A-Z]+|Q[2-8])\b/i ) ;
if ( quant ) return quant [ 0 ] . toUpperCase ( ) . replace ( /^UD-/ , '' ) ;
return file . replace ( /\.gguf$/i , '' ) . replace ( /-\d{5}-of-\d{5}$/i , '' ) ;
}
function _downloadDisplayName ( name , task ) {
const include = task ? . payload ? . include || '' ;
if ( ! include || String ( name || '' ) . includes ( ' · ' ) ) return name ;
const part = _ggufDisplayPartFromPath ( include . replace ( /\*/g , '' ) ) ;
return part ? ` ${ name } · ${ part } ` : name ;
}
2026-07-03 00:45:43 +00:00
function _downloadNameFromPayload ( name , payload ) {
const rawName = String ( name || '' ) . trim ( ) ;
// Defensive: failed/restarted downloads can inherit the wrapper executable
// name if older state was saved from a command preview. The row title should
// always be the model/repo, never "bash" or "python".
const looksLikeLauncher = /^(?:bash|sh|zsh|python|python3|pwsh|powershell|cmd|tmux)$/i . test ( rawName ) ;
const base = ( ! rawName || looksLikeLauncher )
? String ( payload ? . repo _id || payload ? . repo || '' ) . split ( '/' ) . pop ( )
: rawName ;
const include = payload ? . include || '' ;
if ( ! include || String ( base || '' ) . includes ( ' · ' ) ) return base || rawName || 'download' ;
const part = _ggufDisplayPartFromPath ( String ( include ) . replace ( /\*/g , '' ) ) ;
return part ? ` ${ base } · ${ part } ` : ( base || rawName || 'download' ) ;
}
2026-06-22 01:49:15 +00:00
function _taskDisplayName ( task ) {
const name = String ( task ? . name || '' ) . trim ( ) ;
2026-07-03 00:45:43 +00:00
if ( task ? . type === 'download' ) return _downloadDisplayName ( _downloadNameFromPayload ( name , task ? . payload ) , task ) ;
2026-06-22 01:49:15 +00:00
if ( task ? . type !== 'serve' ) return name ;
const gguf = task ? . payload ? . _fields ? . gguf _file || task ? . payload ? . gguf _file || '' ;
if ( ! gguf || name . includes ( ' · ' ) ) return name ;
const part = _ggufDisplayPartFromPath ( gguf ) ;
return part ? ` ${ name } · ${ part } ` : name ;
}
function _canLaunchDownloadedTask ( task ) {
return task ? . type === 'download' && [ 'done' , 'completed' ] . includes ( task . status || '' ) && ! ! ( task . payload ? . repo _id || task . name ) ;
}
function _downloadServeFields ( task ) {
const include = String ( task ? . payload ? . include || '' ) . trim ( ) ;
if ( ! include ) return null ;
return {
backend : 'llamacpp' ,
_forceBackend : true ,
_preferredGgufInclude : include ,
} ;
}
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
// A download task whose tmux output still shows an active per-shard line
// (e.g. "model-00012-of-00082.safetensors: 56%|") is NOT actually finished —
// the cookbook just lost track. The clear pill becomes a "reconnect" affordance
// in that case (click → revive the row + reattach the poll loop).
function _downloadOutputLooksActive ( task ) {
if ( ! task || task . type !== 'download' ) return false ;
const out = task . output || '' ;
if ( ! out ) return false ;
if ( out . includes ( 'DOWNLOAD_OK' ) || out . includes ( 'DOWNLOAD_FAILED' ) ) return false ;
// An active shard line: filename + a colon + a percentage that isn't 100%.
// We catch any in-flight shard or "Downloading 'X' to ..." line (no %).
return /model-\d+-of-\d+\.[a-z]+:\s+(?!100%)\d+%/i . test ( out )
|| /Downloading\s+'[^']+'\s+to\s+'[^']*\.incomplete'/i . test ( out ) ;
}
2026-06-02 12:15:41 +09:00
function _canClearTask ( task ) {
if ( ! task || task . status === 'running' ) return false ;
2026-07-07 00:50:07 +00:00
if ( task . type === 'serve' && ( task . status === 'ready' || ( ! [ 'error' , 'crashed' , 'failed' , 'completed' ] . includes ( task . status ) && _serveOutputLooksReady ( task ) ) ) ) return false ;
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
// If the tmux output still shows an in-flight download, the task isn't
// actually finished — hide the clear/check pill so it doesn't show on a
// task that's still doing work. (The next render will reflect this and
// ideally the self-heal flips status back to running.)
if ( _downloadOutputLooksActive ( task ) ) return false ;
2026-06-21 11:02:35 +00:00
return [ 'done' , 'completed' , 'stopped' , 'error' , 'crashed' , 'failed' ] . includes ( task . status ) ;
2026-06-02 12:15:41 +09:00
}
function _clearPillLabel ( task ) {
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
if ( _downloadOutputLooksActive ( task ) ) return 'reconnect' ;
2026-06-02 12:15:41 +09:00
return 'clear' ;
}
2026-06-21 11:02:35 +00:00
function _venvRootFromPath ( path ) {
let p = ( path || '' ) . toString ( ) . trim ( ) . replace ( /\/+$/ , '' ) ;
if ( ! p ) return '' ;
p = p . replace ( /\/bin\/(?:activate|python(?:3(?:\.\d+)?)?|vllm|pip(?:3)?)$/i , '' ) ;
return p ;
}
2026-06-04 17:25:06 +05:30
// A pip dependency/driver install (payload._dep) reports success with the
// runner's "=== Process exited with code 0 ===" sentinel and pip's
// "Successfully installed" line — never the HuggingFace download markers
// (DONE / 100% / /snapshots/ / DOWNLOAD_OK) that the download heuristics look
// for. Without this, a clean install whose tmux pane has already gone away is
// misread as crashed/stopped even though pip exited 0. Prefer the authoritative
// exit-code sentinel; fall back to pip's success line when no sentinel was
// captured (and there's no install error in the same output).
function _depInstallSucceeded ( output ) {
const text = String ( output || '' ) ;
if ( ! text ) return false ;
const exitMatch = text . match ( /=== Process exited with code (-?\d+) ===/ ) ;
if ( exitMatch ) return Number ( exitMatch [ 1 ] ) === 0 ;
return /\b(?:Successfully installed|Requirement already satisfied)\b/ . test ( text )
&& ! /\bERROR\b|No matching distribution|Could not find a version|Traceback \(most recent call last\)/ . test ( text ) ;
}
2026-06-01 09:12:35 -05:00
function _shouldOfferCrashReport ( task ) {
if ( ! task ) return false ;
if ( task . _unreachable && task . type === 'serve' ) return true ;
return [ 'error' , 'crashed' , 'failed' ] . includes ( task . status ) ;
}
2026-06-02 12:15:41 +09:00
function _serveTaskLooksAwqOnLocalBackend ( task , outputText = '' ) {
const repo = ` ${ task ? . payload ? . repo _id || '' } ${ task ? . name || '' } ` . toLowerCase ( ) ;
const cmd = ` ${ task ? . payload ? . _cmd || '' } ${ outputText || '' } ` . toLowerCase ( ) ;
return /\b(awq|gptq|fp8)\b/ . test ( repo ) && /(llama-server|llama_cpp\.server|ollama|ggml_cuda_enable_unified_memory)/ . test ( cmd ) ;
}
function _serveTaskLooksAwqWithoutUsableAccelerator ( task , outputText = '' ) {
const repo = ` ${ task ? . payload ? . repo _id || '' } ${ task ? . name || '' } ` . toLowerCase ( ) ;
const out = String ( outputText || '' ) . toLowerCase ( ) ;
return /\b(awq|gptq|fp8)\b/ . test ( repo )
&& /(no accelerator|no cuda runtime|failed to infer device type|triton is not supported|0 active driver)/i . test ( out ) ;
}
async function _openDownloadForGgufTask ( task ) {
const raw = task ? . payload ? . repo _id || task ? . name || '' ;
const modelName = String ( raw )
. split ( '/' ) . pop ( )
. replace ( /[-_](?:AWQ|GPTQ|FP8|4bit|8bit|Int4|Int8).*$/i , '' )
. replace ( /[-_]+$/g , '' )
|| String ( raw ) . split ( '/' ) . pop ( )
|| raw ;
const cookbook = window . cookbookModule ;
if ( cookbook && typeof cookbook . open === 'function' ) {
cookbook . open ( { tab : 'Search' } ) ;
} else {
document . getElementById ( 'tool-cookbook-btn' ) ? . click ( ) ;
}
setTimeout ( async ( ) => {
const modal = document . getElementById ( 'cookbook-modal' ) ;
const tab = modal ? . querySelector ( '.cookbook-tab[data-backend="Search"]' ) ;
if ( tab && ! tab . classList . contains ( 'active' ) ) tab . click ( ) ;
const search = document . getElementById ( 'hwfit-search' ) ;
if ( search ) {
search . value = modelName ;
search . dispatchEvent ( new Event ( 'input' , { bubbles : true } ) ) ;
search . focus ( ) ;
}
const quant = document . getElementById ( 'hwfit-quant' ) ;
if ( quant ) {
quant . value = 'Q4_K_M' ;
quant . dispatchEvent ( new Event ( 'change' , { bubbles : true } ) ) ;
}
try {
const hwfit = await import ( './cookbook-hwfit.js' ) ;
if ( typeof hwfit . _hwfitFetch === 'function' ) hwfit . _hwfitFetch ( true ) ;
} catch { }
} , 80 ) ;
}
function _terminalServeDiagnosis ( task , outputText ) {
const out = String ( outputText || task ? . output || '' ) ;
if ( ! task || task . type !== 'serve' || ! [ 'stopped' , 'error' , 'crashed' , 'failed' ] . includes ( task . status ) || ! out . trim ( ) ) return null ;
2026-06-19 00:33:48 +00:00
// Suppress the crash diagnosis when the output proves the server
// actually became reachable — e.g. an early `exit 127` from a failed
// build attempt was followed by the shim/Python fallback successfully
// starting Uvicorn. Without this, the user sees a confusing "build
// stopped before the server became reachable" toast while the server
// is right there serving requests.
if ( _serveOutputLooksReady ( task ) ) return null ;
cookbook agent debug loop: persistent log files, auto-adopt orphan tmux, Codex/Claude skill parity
Three converging fixes so the chat agent + external Codex/Claude skills can actually debug a crashed serve instead of staring at a post-crash neofetch banner:
* Serves now `tee` to /tmp/odysseus-tmux/SESSION.log on the host running them. Runner saves fds 3/4 before the tee and restores them right before `exec ${SHELL}`, so the post-crash interactive zsh banner does NOT pollute the log file.
* `tail_serve_output` (chat agent) and `/api/codex/cookbook/output/{sid}` (Codex+Claude skills) both prefer the persistent log file over the tmux pane. Pane is fallback for sessions predating the tee runner. Default tail bumped 150 -> 400.
* `list_served_models` "recent log" snippet seeks to the Traceback line instead of showing the last 6 lines (which was always the bash prompt).
Cookbook auto-adoption sweep on `/api/cookbook/tasks/status`: every 20s (rate-limited) the cookbook SSHes each configured server, finds `serve-*` / `cookbook-*` tmux sessions running an actual model process (vllm/python/llama-server/etc., filtered via `pane_current_command`), and writes them into state.tasks. So when the agent falls back to raw ssh+tmux, the session appears in the Cookbook UI on the next poll.
`serve_model` error path now reads `data["detail"]` in addition to `data["error"]` so the FastAPI HTTPException message ("Invalid characters in cmd") actually reaches the agent instead of being swallowed as a generic "Serve failed". Tool description updated to warn against `cd …`/`source …`/`&&` prefixes.
Intent-without-action supervisor in agent_loop: when the model writes "Let me tail the output" / "I'll check the logs" / "Let me investigate" and ends the turn without emitting a tool call, the loop injects a sharp system nudge ("You said you would X — DO IT NOW") and continues. Capped at 2 nudges per chat so a model that genuinely cannot use the tool does not pin the loop.
Codex/Claude skill parity: adds `/cookbook/cached`, `/cookbook/presets`, `/cookbook/preset/{name}`, `/cookbook/adopt` so external agents have the same surface as the chat agent. SKILL.md docs + odysseus_api.py wrapper updated for both bundles.
`adopt_served_model` promoted to the always-on tool set so the agent has a documented fallback when serve_model rejects a cmd.
Also various cookbook UI tweaks accumulated alongside the above (cookbook.js, cookbookRunning.js, cookbookServe.js, cookbook-diagnosis.js, settings.js, style.css).
2026-06-04 23:27:18 +09:00
// Pip tasks (Reinstall vLLM, Upgrade torch, etc.) ride on the serve task
// type so they get a tmux session + show up in Running tab — but they are
// NOT serve invocations. Their output is pip's own; the generic
// "Serve stopped before the model became reachable" message + Edit-serve
// fix make no sense. Bail so the panel just shows pip's output.
const _isPipTask = ( ( task . payload ? . repo _id || '' ) . startsWith ( 'pip-' ) )
|| /python3? -m pip\b/ . test ( task . payload ? . _cmd || '' ) ;
if ( _isPipTask ) return null ;
2026-06-02 12:15:41 +09:00
if ( _serveTaskLooksAwqOnLocalBackend ( task , out ) ) {
return {
message : 'AWQ/GPTQ/FP8 cannot be served through llama.cpp/Ollama unified-memory mode.' ,
suggestion : 'Suggested action: use vLLM/SGLang on a compatible CUDA/ROCm GPU server, or download a GGUF version for llama.cpp/Ollama/unified-memory serving.' ,
fixes : [
{ label : 'Find GGUF download' , action : ( ) => _openDownloadForGgufTask ( task ) } ,
{ label : 'Edit serve' , action : ( panel ) => _openServeEditForTask ( task ) } ,
] ,
} ;
}
if ( _serveTaskLooksAwqWithoutUsableAccelerator ( task , out ) ) {
return {
message : 'AWQ/GPTQ/FP8 needs a working vLLM/SGLang accelerator path; this server did not expose one.' ,
suggestion : 'Suggested action: choose a CUDA/ROCm server where vLLM/SGLang can see the GPU, or download a GGUF version and serve it with llama.cpp/Ollama.' ,
fixes : [
{ label : 'Find GGUF download' , action : ( ) => _openDownloadForGgufTask ( task ) } ,
{ label : 'Edit serve' , action : ( panel ) => _openServeEditForTask ( task ) } ,
] ,
} ;
}
return _diagnose ( out ) || {
message : /Native llama-server not found|building llama-server|llama\.cpp/i . test ( out )
? 'llama.cpp build stopped before the server became reachable.'
: 'Serve stopped before the model became reachable.' ,
suggestion : /Native llama-server not found|building llama-server|llama\.cpp/i . test ( out )
? 'Suggested action: copy the troubleshooting bundle, then edit serve settings. For the quickest local/CPU path, use Ollama or a prebuilt llama-server; source builds can take several minutes and fail if build dependencies are incomplete.'
: 'Suggested action: copy the troubleshooting bundle, then edit serve settings or relaunch with a CPU/backend fallback.' ,
fixes : [ { label : 'Edit serve' , action : ( panel ) => _openServeEditForTask ( task ) } ] ,
} ;
}
2026-06-01 09:12:35 -05:00
function _redactCrashReportText ( text ) {
if ( ! text ) return '' ;
return String ( text )
. replace ( /\b(Bearer\s+)[A-Za-z0-9._~+/=-]{12,}/gi , '$1[redacted]' )
. replace ( /\b(hf_[A-Za-z0-9]{16,})\b/g , '[redacted-hf-token]' )
. replace ( /\b(sk-[A-Za-z0-9_-]{16,})\b/g , '[redacted-api-key]' )
. replace ( /\b(xox[baprs]-[A-Za-z0-9-]{16,})\b/g , '[redacted-slack-token]' )
. replace ( /\b(AIza[0-9A-Za-z_-]{20,})\b/g , '[redacted-google-key]' )
. replace ( /\b((?:HF_TOKEN|HUGGING_FACE_HUB_TOKEN|OPENAI_API_KEY|ANTHROPIC_API_KEY|BRAVE_API_KEY|TAVILY_API_KEY|SERPER_API_KEY|GOOGLE_API_KEY|API_KEY|TOKEN|PASSWORD)\s*=\s*)(['"]?)[^\s'"\\]+/gi , '$1$2[redacted]' )
. replace ( /\b(--(?:api-key|token|hf-token|password)\s+)([^\s]+)/gi , '$1[redacted]' ) ;
}
function _lastLines ( text , count = 160 ) {
const clean = _redactCrashReportText ( text || '' ) . trimEnd ( ) ;
if ( ! clean ) return '(no captured output)' ;
return clean . split ( '\n' ) . slice ( - count ) . join ( '\n' ) ;
}
function _codeFence ( text ) {
return String ( text || '' ) . replace ( /```/g , '` ` `' ) ;
}
function _taskHostLabel ( task ) {
if ( ! task ? . remoteHost ) return 'local' ;
return task . remoteHost + ( task . sshPort ? ` : ${ task . sshPort } ` : '' ) ;
}
function _taskPort ( task ) {
fix(cookbook): only block model launch on real port collisions (#4760)
* Fix #4507: only block model launch on real port collisions
Quick-run hardcoded port 8000 and never called _nextAvailablePort(), so
every launch collided. Both pre-launch guards (serve panel + quick-run)
were count-based and fired regardless of port.
- quick-run now auto-assigns a free port (8080 for llama.cpp)
- both guards parse the new port and only prompt on a real overlap,
stopping only the colliding serve
- dialog reports the actual port instead of a hardcoded 8000
* refactor(cookbook): share _taskPort for port parsing; auto-assign llama.cpp port
Addresses review on #4760:
- _taskPort regex now matches --port= as well as --port (space)
- _nextAvailablePort and both launch guards reuse _taskPort instead of inline regex
- quick-run llama.cpp no longer pins 8080, so two can run concurrently
* fix(cookbook): _taskPort also parses -p; add port-parsing tests
Addresses review on #4760:
- _taskPort now matches -p <n> too, so it's the complete single reader
(was missing the short flag that other readers already handle)
- add tests/test_cookbook_port_parsing_js.py covering the port forms,
shared-reader reuse, and llama.cpp auto-assign
* test(cookbook): extract pure port helpers and test behavior
Addresses review on #4760: the prior tests only asserted source strings.
- extract portOf() and nextFreePort() into static/js/cookbookPorts.js
- cookbookRunning.js imports them; _taskPort and _nextAvailablePort delegate
- tests run the helpers via node and assert real behavior: all port forms
(--port, --port=, -p, -p=), next-free-port skipping taken ports, and the
same-port-clash / different-port-coexist outcome
---------
Co-authored-by: samy <samy@odysseus.boukouro.com>
2026-06-24 13:44:09 -04:00
return portOf ( task ? . payload ? . _cmd || '' ) ;
2026-06-01 09:12:35 -05:00
}
function _buildCrashReport ( task , outputText ) {
const capturedOutput = outputText || task ? . output || '' ;
const cmd = _redactCrashReportText ( task ? . payload ? . _cmd || '' ) ;
const diag = _diagnose ( capturedOutput ) ;
const started = task ? . ts ? new Date ( task . ts ) . toISOString ( ) : '' ;
const report = [
'## Odysseus Cookbook crash report' ,
'' ,
'Please review this report for secrets before posting it publicly.' ,
'' ,
'### Task' ,
` - ID: \` ${ task ? . sessionId || task ? . id || 'unknown' } \` ` ,
` - Type: \` ${ task ? . type || 'unknown' } \` ` ,
` - Status: \` ${ task ? . _unreachable ? 'unreachable' : ( task ? . status || 'unknown' ) } \` ` ,
` - Model/repo: \` ${ task ? . payload ? . repo _id || task ? . name || 'unknown' } \` ` ,
` - Host: \` ${ _taskHostLabel ( task ) } \` ` ,
] ;
if ( task ? . platform ) report . push ( ` - Platform: \` ${ task . platform } \` ` ) ;
if ( started ) report . push ( ` - Started: \` ${ started } \` ` ) ;
const port = _taskPort ( task ) ;
if ( port ) report . push ( ` - Port: \` ${ port } \` ` ) ;
if ( diag ? . message ) report . push ( ` - Diagnosis: ${ diag . message } ` ) ;
if ( cmd ) {
report . push ( '' , '### Command' , '```bash' , _codeFence ( cmd ) , '```' ) ;
}
report . push ( '' , '### Last captured output' , '```text' , _codeFence ( _lastLines ( capturedOutput ) ) , '```' ) ;
return report . join ( '\n' ) ;
}
2026-05-31 23:58:26 +09:00
// Shared state/functions injected by init()
let _envState ;
let _sshCmd ;
let _getPort ;
let _sshPrefix ;
let _getPlatform ;
let _isWindows ;
let _buildEnvPrefix ;
let _loadPresets ;
let _savePresets ;
let _copyText ;
let _persistEnvState ;
2026-06-01 18:58:06 +05:30
let _refreshDependencies ;
2026-06-08 18:36:10 -04:00
let _serverByVal ;
2026-06-21 11:02:35 +00:00
let _serverKey ;
2026-06-08 18:36:10 -04:00
let _selectedServer ;
2026-05-31 23:58:26 +09:00
let modelLogo ;
let esc ;
let _detectBackend ;
let _detectToolParser ;
let _detectModelOptimizations ;
let _buildServeCmd ;
2026-06-22 01:49:15 +00:00
function _taskServerSelection ( task ) {
const host = task ? . remoteHost || task ? . payload ? . remote _host || '' ;
const savedKey = task ? . remoteServerKey || task ? . payload ? . remote _server _key || '' ;
const server = ( savedKey ? _serverByVal ( savedKey ) : null )
|| ( host ? _serverByVal ( host ) : null )
|| ( host ? _envState . servers . find ( s => s . host === host ) : null )
|| null ;
const key = server ? ( _serverKey ? _serverKey ( server ) : savedKey ) : ( savedKey || ( host || 'local' ) ) ;
return { host , server , key } ;
}
2026-07-07 00:50:07 +00:00
function _serverColorForTaskGroup ( key , tasks ) {
const firstTask = Array . isArray ( tasks ) ? tasks [ 0 ] : null ;
const host = firstTask ? . remoteHost || firstTask ? . payload ? . remote _host || '' ;
const savedKey = firstTask ? . remoteServerKey || firstTask ? . payload ? . remote _server _key || key || '' ;
const server = ( savedKey ? _serverByVal ? . ( savedKey ) : null )
|| ( key ? _serverByVal ? . ( key ) : null )
|| ( host ? _serverByVal ? . ( host ) : null )
|| ( key === 'local' || ! key ? ( _envState ? . servers || [ ] ) . find ( s => ! s . host || String ( s . host ) . toLowerCase ( ) === 'local' ) : null )
|| null ;
const color = String ( server ? . color || '' ) . trim ( ) ;
return /^#[0-9a-fA-F]{6}$/ . test ( color ) ? color : '' ;
}
function _serverHeaderStyle ( color ) {
if ( ! color ) return '' ;
const c = color . toLowerCase ( ) ;
const accent = ( c === '#ffffff' || c === '#f8fafc' ) ? '#cbd5e1'
: ( c === '#111827' || c === '#000000' ) ? '#64748b'
: color ;
return ` style="--cookbook-server-color: ${ esc ( color ) } ;--cookbook-server-accent: ${ esc ( accent ) } ;" ` ;
}
function _shouldAutoExpandTaskOutput ( task ) {
return task ? . type === 'download'
&& ! task ? . payload ? . _dep
&& [ 'running' , 'queued' , 'error' , 'crashed' ] . includes ( String ( task ? . status || '' ) ) ;
}
2026-06-22 01:49:15 +00:00
function _selectTaskServer ( task ) {
const { host , server , key } = _taskServerSelection ( task ) ;
_envState . remoteHost = host ;
_envState . remoteServerKey = key === 'local' ? '' : key ;
if ( server ) {
_envState . env = server . env || 'none' ;
_envState . envPath = server . envPath || '' ;
_envState . platform = server . platform || '' ;
} else if ( ! host ) {
_envState . env = 'none' ;
_envState . envPath = '' ;
_envState . platform = '' ;
}
document . querySelectorAll ( '#hwfit-server-select, #hwfit-dl-server, #hwfit-cache-server, #hwfit-deps-server' ) . forEach ( sel => {
if ( ! sel || sel . tagName !== 'SELECT' ) return ;
const wanted = key || ( host || 'local' ) ;
if ( [ ... sel . options ] . some ( o => o . value === wanted ) ) sel . value = wanted ;
else if ( host && [ ... sel . options ] . some ( o => o . value === host ) ) sel . value = host ;
else sel . value = host ? wanted : 'local' ;
} ) ;
return { host , server , key } ;
}
2026-05-31 23:58:26 +09:00
// When a new action is started (download / dependency / serve), this holds the
// new task's id so the next render collapses every other card and leaves only
// the new one open. Consumed (cleared) by _renderRunningTab.
let _soloExpandTaskId = null ;
// Storage keys
const TASKS _KEY = 'cookbook-tasks' ;
const STORAGE _KEY = 'cookbook-presets' ;
const SERVE _STATE _KEY = 'cookbook-serve-state' ;
2026-07-07 00:50:07 +00:00
const SERVE _FAVORITES _KEY = 'cookbook-serve-favorite-models' ;
2026-05-31 23:58:26 +09:00
// Polling / timeout intervals
const TASK _POLL _INTERVAL _MS = 3000 ; // delay between reconnect-loop iterations
2026-07-01 10:09:25 +00:00
const BG _MONITOR _INTERVAL _MS = 10000 ; // background task status poll
2026-05-31 23:58:26 +09:00
const STALE _PROGRESS _MS = 5 * 60 * 1000 ; // download with no progress this long = stale
2026-06-01 22:42:59 -04:00
const STARTUP _STALE _PROGRESS _MS = 45 * 1000 ; // 0%-forever startup stall: retry much sooner
2026-05-31 23:58:26 +09:00
// ── Phase detection (mirrors Python _parse_serve_phase in cookbook_routes.py) ──
// Single source of truth for serve task status. KEEP IN SYNC with the Python version.
export function _parseServePhase ( snapshot ) {
if ( ! snapshot ) return { } ;
// Strip newlines so tmux line-wrapping doesn't break regex matching
const flat = snapshot . replace ( /\s+/g , ' ' ) ;
const loadMatches = [ ... flat . matchAll ( /Loading safetensors.*?(\d+)%/g ) ] ;
// "Downloading (incomplete total...)" tracks real aggregate bytes; prefer it
// over "Fetching N files" which only counts fully-closed files and lags badly
// with hf_transfer's parallel-chunk strategy (often sits at 0/N for most of the run).
const downloadingMatches = [ ... flat . matchAll ( /Downloading.*?(\d+)%/g ) ] ;
const fetchingMatches = [ ... flat . matchAll ( /Fetching.*?(\d+)%/g ) ] ;
const dlMatches = downloadingMatches . length ? downloadingMatches : fetchingMatches ;
// "Avg generation throughput: X tokens/s, Running: N reqs"
const tpsMatches = [ ... flat . matchAll ( /(?:Avg )?generation throughput:\s*([\d.]+)\s*tokens\/s.*?Running:\s*(\d+)\s*reqs/g ) ] ;
// Throughput FIRST — its log line contains "GPU KV cache usage" which would
// otherwise false-match the warmup check
if ( tpsMatches . length ) {
const m = tpsMatches [ tpsMatches . length - 1 ] ;
const tps = parseFloat ( m [ 1 ] ) ;
const reqs = parseInt ( m [ 2 ] ) ;
return {
phase : reqs > 0 ? ` ${ m [ 1 ] } tok/s ` : 'idle' ,
status : 'ready' ,
tps ,
reqs ,
} ;
}
if ( flat . includes ( 'Application startup complete' ) ) {
return { phase : 'ready' , status : 'ready' } ;
}
Cookbook: scoring fixes, UI polish, false-finished + stale-state bug fixes
Backend (services/hwfit + routes):
- rank_models picks visible set by REQUESTED column, not always score —
sorting by Param now shows highest-param models PERIOD (incl. too_tight).
- New fit_only param. Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang
cannot serve them); default non-prequantized to BF16 on 2+ GPUs.
- AWQ / GPTQ-8bit get a -1.0 quality penalty (was 0.0, tied with FP8), so
FP8 wins when both fit.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above
M2.5 on equal composite score; >=100B integers not misread as versions.
- /api/cookbook/hf-latest no longer drops models without an "NB" pattern in
the repo id (MiniMax-M2.7, DeepSeek-V4-Pro etc. were silently filtered).
- Cached-model scan: atexit flushes models JSON even if the script is
killed mid-walk; each scan_dir wrapped in try/except; timeout 60s -> 180s.
- KB granularity for sub-MB sizes (was "0 MB" for 12 KB shells). New
"stalled" status for shells <1 MB with no .incomplete files.
- /api/cookbook/state POST guard: rejects "done" download tasks lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stops stale tabs from poisoning persisted state.
- hf_models.json: add zai-org/GLM-5.1; flip zai-org/GLM-5 quantization
Q4_K_M -> BF16 (it is the native base, not a quant).
Frontend (static/js):
- Scan/Download toolbar: quant defaults to All; ctx slider (8k/16k/32k/
50k/128k/Max) ported from origin/main with sort=fit on drag, sort=score
on Max. GPU toggle commits _activeCount to maxGpu on initial render. Fit
column header tagged with active budget (RAM / GPU / N GPU).
- Foldable Download admin-card: the Download h2 is the chevron trigger;
state persists in localStorage.
- Download card surfaces destination dir (Dir: <path>). Same dir on running
task row, font/color matched to uptime (9px Fira Code muted, opacity .4).
- Serve panel ctx text input always resets to model max on open. Sub-MB
cached models show with red "download stalled" badge.
- Bulk-select Cancel + Delete reset the Select button label on exit.
- Cookbook running: false-finished bug fixed — DOWNLOAD_OK or /snapshots/
required; bare "Download complete" no longer marks the task done after
the first config file. Clear button now sends tmux kill-session too.
True overall % for multi-shard downloads: ((N-1)+frac)/total instead of
hf_transfer per-shard aggregate.
- Diagnosis card simplified: removed fold toggle, copy button, dismiss X.
Suggestion font matches message body (12px).
- HF token field flashes green check + "Saved" on save.
- Cached scan no longer counts stalled rows as downloaded in Scan/Download.
CSS:
- dep Install button width pinned to 76px to match Installed split.
- task-sub row +1px; task-status badge gets margin-right 8px.
- Ctx slider styled like gallery editor sliders (thin pill rail, red thumb).
- Bulk-select cancel button top -3px -> -5px.
2026-06-03 16:32:20 +09:00
if ( /Ollama API ready on port\s+\d+/i . test ( flat ) ) {
return { phase : 'ready' , status : 'ready' } ;
}
2026-06-02 12:15:41 +09:00
const llamaBuildMatches = [ ... flat . matchAll ( /\[\s*(\d{1,3})%\]\s*(?:Building|Linking)/gi ) ] ;
if ( llamaBuildMatches . length ) {
const pct = Math . min ( 100 , parseInt ( llamaBuildMatches [ llamaBuildMatches . length - 1 ] [ 1 ] , 10 ) ) ;
return { phase : ` building llama.cpp ${ pct } % ` , status : 'running' , pct } ;
}
if ( /Native llama-server not found|building from source/i . test ( flat ) ) {
if ( /Cloning into ['"]?llama\.cpp/i . test ( flat ) && ! /Receiving objects:\s*100%/i . test ( flat ) ) {
return { phase : 'cloning llama.cpp' , status : 'running' } ;
}
if ( /Configuring incomplete|CMake Error/i . test ( flat ) ) {
return { } ;
}
if ( /CMAKE_BUILD_TYPE|Detecting CXX|Found Threads|Including CPU backend|CUDA nvcc found|building llama-server/i . test ( flat ) ) {
return { phase : 'configuring llama.cpp' , status : 'running' } ;
}
return { phase : 'building llama.cpp' , status : 'running' } ;
}
2026-05-31 23:58:26 +09:00
// HTTP access logs (e.g. GET /v1/models 200 OK) mean the server is up
if ( /(?:GET|POST)\s+\/[^\s]*\s+HTTP\/[\d.]+"\s*\d{3}/ . test ( flat ) ) {
return { phase : 'idle' , status : 'ready' } ;
}
if ( flat . includes ( 'Loading weights took' ) ) {
return { phase : 'initializing' , status : 'running' } ;
}
// "GPU KV cache" alone (during allocation) — not "GPU KV cache usage" (runtime log)
if ( flat . includes ( 'GPU KV cache' ) && ! flat . includes ( 'GPU KV cache usage' ) ) {
return { phase : 'warming up' , status : 'running' } ;
}
if ( loadMatches . length ) {
const pct = parseInt ( loadMatches [ loadMatches . length - 1 ] [ 1 ] ) ;
return { phase : ` loading ${ pct } % ` , status : 'running' , pct } ;
}
if ( dlMatches . length ) {
const pct = parseInt ( dlMatches [ dlMatches . length - 1 ] [ 1 ] ) ;
return { phase : ` downloading ${ pct } % ` , status : 'running' , pct } ;
}
return { } ;
}
// ── Port auto-increment ──
function _nextAvailablePort ( ) {
const tasks = _loadTasks ( ) ;
const presets = _loadPresets ( ) ;
const usedPorts = new Set ( ) ;
tasks . forEach ( t => {
if ( t . type === 'serve' && ( t . status === 'running' || t . status === 'queued' ) ) {
fix(cookbook): only block model launch on real port collisions (#4760)
* Fix #4507: only block model launch on real port collisions
Quick-run hardcoded port 8000 and never called _nextAvailablePort(), so
every launch collided. Both pre-launch guards (serve panel + quick-run)
were count-based and fired regardless of port.
- quick-run now auto-assigns a free port (8080 for llama.cpp)
- both guards parse the new port and only prompt on a real overlap,
stopping only the colliding serve
- dialog reports the actual port instead of a hardcoded 8000
* refactor(cookbook): share _taskPort for port parsing; auto-assign llama.cpp port
Addresses review on #4760:
- _taskPort regex now matches --port= as well as --port (space)
- _nextAvailablePort and both launch guards reuse _taskPort instead of inline regex
- quick-run llama.cpp no longer pins 8080, so two can run concurrently
* fix(cookbook): _taskPort also parses -p; add port-parsing tests
Addresses review on #4760:
- _taskPort now matches -p <n> too, so it's the complete single reader
(was missing the short flag that other readers already handle)
- add tests/test_cookbook_port_parsing_js.py covering the port forms,
shared-reader reuse, and llama.cpp auto-assign
* test(cookbook): extract pure port helpers and test behavior
Addresses review on #4760: the prior tests only asserted source strings.
- extract portOf() and nextFreePort() into static/js/cookbookPorts.js
- cookbookRunning.js imports them; _taskPort and _nextAvailablePort delegate
- tests run the helpers via node and assert real behavior: all port forms
(--port, --port=, -p, -p=), next-free-port skipping taken ports, and the
same-port-clash / different-port-coexist outcome
---------
Co-authored-by: samy <samy@odysseus.boukouro.com>
2026-06-24 13:44:09 -04:00
const p = _taskPort ( t ) ;
if ( p ) usedPorts . add ( parseInt ( p ) ) ;
2026-05-31 23:58:26 +09:00
}
} ) ;
presets . forEach ( p => {
if ( p . port ) usedPorts . add ( parseInt ( p . port ) ) ;
} ) ;
fix(cookbook): only block model launch on real port collisions (#4760)
* Fix #4507: only block model launch on real port collisions
Quick-run hardcoded port 8000 and never called _nextAvailablePort(), so
every launch collided. Both pre-launch guards (serve panel + quick-run)
were count-based and fired regardless of port.
- quick-run now auto-assigns a free port (8080 for llama.cpp)
- both guards parse the new port and only prompt on a real overlap,
stopping only the colliding serve
- dialog reports the actual port instead of a hardcoded 8000
* refactor(cookbook): share _taskPort for port parsing; auto-assign llama.cpp port
Addresses review on #4760:
- _taskPort regex now matches --port= as well as --port (space)
- _nextAvailablePort and both launch guards reuse _taskPort instead of inline regex
- quick-run llama.cpp no longer pins 8080, so two can run concurrently
* fix(cookbook): _taskPort also parses -p; add port-parsing tests
Addresses review on #4760:
- _taskPort now matches -p <n> too, so it's the complete single reader
(was missing the short flag that other readers already handle)
- add tests/test_cookbook_port_parsing_js.py covering the port forms,
shared-reader reuse, and llama.cpp auto-assign
* test(cookbook): extract pure port helpers and test behavior
Addresses review on #4760: the prior tests only asserted source strings.
- extract portOf() and nextFreePort() into static/js/cookbookPorts.js
- cookbookRunning.js imports them; _taskPort and _nextAvailablePort delegate
- tests run the helpers via node and assert real behavior: all port forms
(--port, --port=, -p, -p=), next-free-port skipping taken ports, and the
same-port-clash / different-port-coexist outcome
---------
Co-authored-by: samy <samy@odysseus.boukouro.com>
2026-06-24 13:44:09 -04:00
return nextFreePort ( usedPorts ) ;
2026-05-31 23:58:26 +09:00
}
// ── Endpoint cleanup ──
async function _removeEndpointByUrl ( baseUrl ) {
try {
const res = await fetch ( '/api/model-endpoints' , { credentials : 'same-origin' } ) ;
if ( ! res . ok ) return ;
const endpoints = await res . json ( ) ;
const hostPort = baseUrl . replace ( /^https?:\/\// , '' ) . replace ( /\/.*$/ , '' ) ;
const ep = endpoints . find ( e => e . base _url === baseUrl )
|| endpoints . find ( e => e . base _url . includes ( hostPort ) ) ;
if ( ep ) {
await fetch ( ` /api/model-endpoints/ ${ ep . id } ` , { method : 'DELETE' , credentials : 'same-origin' } ) ;
_refreshModelsAfterEndpointChange ( ) ;
}
} catch { }
}
function _refreshModelsAfterEndpointChange ( ) {
const pickerLabel = document . getElementById ( 'model-picker-label' ) ;
if ( pickerLabel ) {
pickerLabel . dataset . prevHtml = pickerLabel . innerHTML ;
pickerLabel . innerHTML = '<span style="opacity:0.4;">refreshing…</span>' ;
}
if ( window . modelsModule && window . modelsModule . refreshModels ) {
2026-07-07 00:50:07 +00:00
window . modelsModule . refreshModels ( false ) ;
2026-05-31 23:58:26 +09:00
}
setTimeout ( ( ) => {
if ( ! window . sessionModule ) return ;
const currentModel = window . sessionModule . getCurrentModel ? window . sessionModule . getCurrentModel ( ) : null ;
if ( currentModel ) {
const items = ( window . modelsModule && window . modelsModule . getCachedItems ) ? window . modelsModule . getCachedItems ( ) : [ ] ;
const allModels = [ ] ;
items . forEach ( item => {
if ( item . offline ) return ;
( item . models || [ ] ) . concat ( item . models _extra || [ ] ) . forEach ( m => allModels . push ( { mid : m , url : item . url , endpointId : item . endpoint _id } ) ) ;
} ) ;
const stillExists = allModels . some ( m => m . mid === currentModel ) ;
if ( ! stillExists && allModels . length > 0 ) {
const fallback = allModels [ 0 ] ;
if ( window . sessionModule . createDirectChat ) {
window . sessionModule . createDirectChat ( fallback . url , fallback . mid , fallback . endpointId ) ;
}
}
}
if ( window . sessionModule . updateModelPicker ) {
window . sessionModule . updateModelPicker ( ) ;
}
} , 1500 ) ;
}
2026-06-02 10:09:48 -05:00
function _appendCookbookEndpointScope ( fd , remoteHost ) {
const host = String ( remoteHost || '' ) . trim ( ) ;
if ( ! host || host === 'local' || host === 'localhost' || host === '127.0.0.1' ) {
fd . append ( 'container_local' , 'true' ) ;
}
}
function _connectHostFromRemote ( remoteHost , fallback = 'localhost' ) {
const host = String ( remoteHost || '' ) . trim ( ) ;
if ( ! host || host === 'local' ) return fallback ;
return host . includes ( '@' ) ? host . split ( '@' ) . pop ( ) : host ;
}
function _isAnyBindHost ( host ) {
const h = String ( host || '' ) . trim ( ) . toLowerCase ( ) ;
return h === '0.0.0.0' || h === '::' || h === '[::]' ;
}
function _endpointFromAdvertisedUrl ( rawUrl , currentHost , fallbackPort = '11434' ) {
try {
const u = new URL ( rawUrl ) ;
const host = _isAnyBindHost ( u . hostname ) ? currentHost : ( u . hostname || currentHost ) ;
const port = u . port || fallbackPort ;
const bracketedHost = host . includes ( ':' ) && ! host . startsWith ( '[' ) ? ` [ ${ host } ] ` : host ;
return { host , port , baseUrl : ` ${ u . protocol } // ${ bracketedHost } ${ port ? ` : ${ port } ` : '' } /v1 ` } ;
} catch {
return null ;
}
}
2026-07-07 00:50:07 +00:00
function _serveExpectedModel ( task ) {
const fields = task ? . payload ? . _fields || { } ;
return String (
fields . served _model _name ||
fields . model _path ||
task ? . payload ? . repo _id ||
task ? . model ||
task ? . name ||
''
) . trim ( ) ;
}
function _modelIdMatchesExpected ( modelId , expected ) {
const got = String ( modelId || '' ) . trim ( ) . toLowerCase ( ) ;
const want = String ( expected || '' ) . trim ( ) . toLowerCase ( ) ;
if ( ! got || ! want ) return true ;
if ( got === want ) return true ;
const gotBase = got . split ( '/' ) . pop ( ) ;
const wantBase = want . split ( '/' ) . pop ( ) ;
return gotBase === wantBase || got . includes ( wantBase ) || want . includes ( gotBase ) ;
}
function _endpointMatchesServe ( ep , task ) {
const expected = _serveExpectedModel ( task ) ;
const models = [ ... ( ep ? . models || [ ] ) , ... ( ep ? . pinned _models || [ ] ) ] ;
if ( ! models . length ) return true ;
return models . some ( mid => _modelIdMatchesExpected ( mid , expected ) ) ;
}
function _markServeEndpointMismatch ( task , ep , host , port ) {
const expected = _serveExpectedModel ( task ) ;
const actual = ( ep ? . models || [ ] ) . join ( ', ' ) || 'no models' ;
const msg = ` Port ${ host } : ${ port } answered, but it is serving ${ actual } , not ${ expected || task ? . name || 'the launched model' } . The new serve likely failed or the port is occupied by an older server. ` ;
_updateTask ( task . sessionId || task . session _id , {
status : 'error' ,
_serveReady : false ,
_endpointAdded : false ,
output : ` ${ task . output || '' } \n \n ${ msg } ` . trim ( ) ,
} ) ;
uiModule . showError ( msg ) ;
}
function _appendPinnedServeModel ( fd , task ) {
const expected = _serveExpectedModel ( task ) ;
if ( expected ) fd . append ( 'pinned_models' , expected ) ;
}
2026-05-31 23:58:26 +09:00
// ── Download queue — runs one at a time per server ──
function _processQueue ( ) {
2026-06-02 22:38:55 +09:00
const tasks = _loadPrunedTasks ( ) ;
2026-05-31 23:58:26 +09:00
const running = tasks . filter ( t => t . type === 'download' && t . status === 'running' ) ;
const queued = tasks . filter ( t => t . type === 'download' && t . status === 'queued' ) ;
if ( ! queued . length ) return ;
const busyHosts = new Set ( running . map ( t => t . remoteHost || 'local' ) ) ;
for ( const task of queued ) {
const host = task . remoteHost || 'local' ;
if ( busyHosts . has ( host ) ) continue ;
busyHosts . add ( host ) ;
_startQueuedDownload ( task ) ;
}
}
async function _startQueuedDownload ( task ) {
if ( ! task . payload ) {
_updateTask ( task . sessionId , { status : 'error' , output : 'No payload' } ) ;
_renderRunningTab ( ) ;
return ;
}
// Flip to 'running' SYNCHRONOUSLY (before the async POST) so a concurrent
// _processQueue — or a second "Start now" — can't see it as still 'queued' and
// launch the same download a second time. Without this, finishing another
// download mid-POST re-queued this one into a duplicate task.
{
const _pre = _loadTasks ( ) ;
const _pt = _pre . find ( t => t . sessionId === task . sessionId ) ;
if ( _pt ) {
if ( _pt . status === 'running' && _pt . _startLaunched ) return ; // already being started
_pt . status = 'running' ;
_pt . _startLaunched = true ;
_saveTasks ( _pre ) ;
}
}
try {
const res = await fetch ( '/api/model/download' , {
method : 'POST' , credentials : 'same-origin' ,
headers : { 'Content-Type' : 'application/json' } ,
body : JSON . stringify ( task . payload ) ,
} ) ;
if ( ! res . ok ) {
const errText = await res . text ( ) . catch ( ( ) => '' ) ;
_updateTask ( task . sessionId , { status : 'error' , output : ` HTTP ${ res . status } : ${ errText . slice ( 0 , 200 ) } ` } ) ;
_renderRunningTab ( ) ;
return ;
}
const data = await res . json ( ) ;
if ( ! data . ok ) {
_updateTask ( task . sessionId , { status : 'error' , output : data . error || 'Unknown error' } ) ;
_renderRunningTab ( ) ;
return ;
}
const oldId = task . sessionId ;
2026-06-02 22:38:55 +09:00
const launchedTask = { ... task , sessionId : data . session _id , id : data . session _id , status : 'running' } ;
const key = _downloadDedupeKey ( launchedTask ) ;
let found = false ;
const tasks = _loadTasks ( ) . filter ( t => {
if ( t . sessionId === oldId ) {
found = true ;
t . sessionId = data . session _id ;
t . id = data . session _id ;
t . status = 'running' ;
t . _startLaunched = true ;
return true ;
}
if ( t . sessionId === data . session _id ) return false ;
return ! ( key && t . type === 'download' && t . status === 'queued' && _downloadDedupeKey ( t ) === key ) ;
} ) ;
2026-06-22 02:39:18 +00:00
if ( ! found ) tasks . push ( _redactTaskForStorage ( launchedTask ) ) ;
2026-06-02 22:38:55 +09:00
_saveTasks ( tasks ) ;
_renderRunningTab ( ) ;
2026-05-31 23:58:26 +09:00
_startBackgroundMonitor ( ) ;
await new Promise ( r => setTimeout ( r , 2000 ) ) ;
_renderRunningTab ( ) ;
} catch ( e ) {
_updateTask ( task . sessionId , { status : 'error' , output : e . message || 'Network error' } ) ;
_renderRunningTab ( ) ;
}
}
// ── Task CRUD ──
2026-06-02 12:15:41 +09:00
function _serveOutputLooksReady ( task ) {
const out = String ( task ? . output || '' ) ;
return ! ! task ? . _serveReady
|| /Application startup complete/i . test ( out )
|| /Ollama API ready on port\s+\d+/i . test ( out )
|| /(?:GET|POST)\s+\/[^\s]*\s+HTTP\/[\d.]+"\s*2\d\d/i . test ( out ) ;
}
function _normalizeTaskForDisplay ( task ) {
if ( ! task || typeof task !== 'object' ) return task ;
cookbook agent debug loop: persistent log files, auto-adopt orphan tmux, Codex/Claude skill parity
Three converging fixes so the chat agent + external Codex/Claude skills can actually debug a crashed serve instead of staring at a post-crash neofetch banner:
* Serves now `tee` to /tmp/odysseus-tmux/SESSION.log on the host running them. Runner saves fds 3/4 before the tee and restores them right before `exec ${SHELL}`, so the post-crash interactive zsh banner does NOT pollute the log file.
* `tail_serve_output` (chat agent) and `/api/codex/cookbook/output/{sid}` (Codex+Claude skills) both prefer the persistent log file over the tmux pane. Pane is fallback for sessions predating the tee runner. Default tail bumped 150 -> 400.
* `list_served_models` "recent log" snippet seeks to the Traceback line instead of showing the last 6 lines (which was always the bash prompt).
Cookbook auto-adoption sweep on `/api/cookbook/tasks/status`: every 20s (rate-limited) the cookbook SSHes each configured server, finds `serve-*` / `cookbook-*` tmux sessions running an actual model process (vllm/python/llama-server/etc., filtered via `pane_current_command`), and writes them into state.tasks. So when the agent falls back to raw ssh+tmux, the session appears in the Cookbook UI on the next poll.
`serve_model` error path now reads `data["detail"]` in addition to `data["error"]` so the FastAPI HTTPException message ("Invalid characters in cmd") actually reaches the agent instead of being swallowed as a generic "Serve failed". Tool description updated to warn against `cd …`/`source …`/`&&` prefixes.
Intent-without-action supervisor in agent_loop: when the model writes "Let me tail the output" / "I'll check the logs" / "Let me investigate" and ends the turn without emitting a tool call, the loop injects a sharp system nudge ("You said you would X — DO IT NOW") and continues. Capped at 2 nudges per chat so a model that genuinely cannot use the tool does not pin the loop.
Codex/Claude skill parity: adds `/cookbook/cached`, `/cookbook/presets`, `/cookbook/preset/{name}`, `/cookbook/adopt` so external agents have the same surface as the chat agent. SKILL.md docs + odysseus_api.py wrapper updated for both bundles.
`adopt_served_model` promoted to the always-on tool set so the agent has a documented fallback when serve_model rejects a cmd.
Also various cookbook UI tweaks accumulated alongside the above (cookbook.js, cookbookRunning.js, cookbookServe.js, cookbook-diagnosis.js, settings.js, style.css).
2026-06-04 23:27:18 +09:00
// Pip tasks (Reinstall vLLM / Upgrade torch / etc.) ride on the serve task
// type so they get tmux + the Running tab. They are NOT serves — their
// "ready" markers are pip's `Successfully installed` / `Requirement already
// satisfied`, not "Application startup complete".
const _isPipTask = ( ( task . payload ? . repo _id || '' ) . startsWith ( 'pip-' ) )
|| /python3? -m pip\b/ . test ( task . payload ? . _cmd || '' ) ;
if ( _isPipTask ) {
// Override stale status: any pip task whose output carries pip's own
// success markers gets displayed as `done` regardless of what's in
// localStorage. Old pre-fix runs landed in error/stopped state and
// stuck there even after we taught the rest of the flow about pip
// tasks — this is the catch-all that flips them to Finished on render.
const out = String ( task . output || '' ) ;
const ranOk = /Successfully installed|Requirement already (?:satisfied|up-to-date)/i . test ( out )
&& ! /error:|ERROR:/ . test ( out . slice ( - 1024 ) ) ;
if ( ranOk && task . status !== 'done' && task . status !== 'running' ) {
return { ... task , status : 'done' } ;
}
return task ;
}
2026-06-02 12:15:41 +09:00
if ( task . type === 'serve' && task . status === 'done' && ! _serveOutputLooksReady ( task ) ) {
return { ... task , status : 'error' } ;
}
return task ;
}
2026-05-31 23:58:26 +09:00
export function _loadTasks ( ) {
2026-06-02 12:15:41 +09:00
try { return ( JSON . parse ( localStorage . getItem ( TASKS _KEY ) ) || [ ] ) . map ( _normalizeTaskForDisplay ) ; }
2026-05-31 23:58:26 +09:00
catch { return [ ] ; }
}
2026-06-02 22:38:55 +09:00
function _downloadRepoKey ( task ) {
return String ( task ? . payload ? . repo _id || task ? . repo _id || task ? . repo || task ? . name || '' ) . trim ( ) ;
}
function _downloadHostKey ( task ) {
return String ( task ? . remoteHost || task ? . payload ? . remote _host || 'local' ) . trim ( ) || 'local' ;
}
function _downloadDedupeKey ( task ) {
if ( ! task || task . type !== 'download' ) return '' ;
const repo = _downloadRepoKey ( task ) ;
if ( ! repo ) return '' ;
return ` ${ _downloadHostKey ( task ) } \n ${ repo } ` ;
}
function _pruneQueuedDownloadDuplicates ( tasks ) {
if ( ! Array . isArray ( tasks ) || ! tasks . length ) return tasks || [ ] ;
const launched = new Set ( ) ;
for ( const task of tasks ) {
if ( task ? . type !== 'download' || task . status === 'queued' ) continue ;
const key = _downloadDedupeKey ( task ) ;
if ( key ) launched . add ( key ) ;
}
let changed = false ;
const seenQueued = new Set ( ) ;
const next = tasks . filter ( task => {
if ( task ? . type !== 'download' || task . status !== 'queued' ) return true ;
const key = _downloadDedupeKey ( task ) ;
if ( ! key ) return true ;
if ( launched . has ( key ) || seenQueued . has ( key ) ) {
changed = true ;
return false ;
}
seenQueued . add ( key ) ;
return true ;
} ) ;
return changed ? next : tasks ;
}
function _loadPrunedTasks ( ) {
const tasks = _loadTasks ( ) ;
const pruned = _pruneQueuedDownloadDuplicates ( tasks ) ;
if ( pruned !== tasks ) _saveTasks ( pruned ) ;
return pruned ;
}
2026-05-31 23:58:26 +09:00
// Tombstones for removed tasks. Without these, removing a task only deletes it
// locally — but the server still has it (its own POST guard even re-preserves
// recently-added ones), so the next sync/poll merges it right back ("I removed
// it and it came back"). A tombstone makes the removal stick: merges skip any
// id the user removed, until the entry expires.
const _REMOVED _KEY = 'cookbook-removed-tasks' ;
const _TOMBSTONE _TTL _MS = 24 * 3600 * 1000 ;
function _loadTombstones ( ) {
2026-06-22 01:49:15 +00:00
try {
const tomb = JSON . parse ( localStorage . getItem ( _REMOVED _KEY ) ) || { } ;
const now = Date . now ( ) ;
let changed = false ;
for ( const k in tomb ) {
if ( now - tomb [ k ] > _TOMBSTONE _TTL _MS ) {
delete tomb [ k ] ;
changed = true ;
}
}
if ( changed ) localStorage . setItem ( _REMOVED _KEY , JSON . stringify ( tomb ) ) ;
return tomb ;
}
2026-05-31 23:58:26 +09:00
catch { return { } ; }
}
2026-06-22 01:49:15 +00:00
function _saveTombstones ( tomb ) {
localStorage . setItem ( _REMOVED _KEY , JSON . stringify ( tomb || { } ) ) ;
}
2026-05-31 23:58:26 +09:00
function _tombstoneTask ( id ) {
if ( ! id ) return ;
const tomb = _loadTombstones ( ) ;
const now = Date . now ( ) ;
tomb [ id ] = now ;
for ( const k in tomb ) { if ( now - tomb [ k ] > _TOMBSTONE _TTL _MS ) delete tomb [ k ] ; }
2026-06-22 01:49:15 +00:00
_saveTombstones ( tomb ) ;
2026-05-31 23:58:26 +09:00
}
function _isTombstoned ( id ) {
const ts = _loadTombstones ( ) [ id ] ;
return ts != null && ( Date . now ( ) - ts ) <= _TOMBSTONE _TTL _MS ;
}
2026-06-22 02:39:18 +00:00
function _redactStoredText ( value ) {
return String ( value || '' )
. replace ( /hf_[A-Za-z0-9]{20,}/g , '[redacted-token]' )
. replace ( /((?:api[_-]?key|token|authorization|password|passwd|secret)\s*[=:]\s*)(["']?)[^\s"']+/gi , '$1$2[redacted]' ) ;
}
2026-06-30 01:47:48 +00:00
function _isServeOutputPlaceholder ( value ) {
const text = String ( value || '' ) . trim ( ) ;
return ! text || /^Launched via agent\s+—\s+waiting for tmux output/i . test ( text ) ;
}
2026-06-22 02:39:18 +00:00
function _redactTaskForStorage ( task ) {
2026-05-31 23:58:26 +09:00
if ( ! task || typeof task !== 'object' ) return task ;
const safe = { ... task } ;
2026-06-22 02:39:18 +00:00
if ( typeof safe . output === 'string' ) safe . output = _redactStoredText ( safe . output ) ;
2026-05-31 23:58:26 +09:00
if ( safe . payload && typeof safe . payload === 'object' ) {
safe . payload = { ... safe . payload } ;
delete safe . payload . hf _token ;
2026-06-22 02:39:18 +00:00
delete safe . payload . hfToken ;
if ( typeof safe . payload . _cmd === 'string' ) safe . payload . _cmd = _redactStoredText ( safe . payload . _cmd ) ;
if ( typeof safe . payload . cmd === 'string' ) safe . payload . cmd = _redactStoredText ( safe . payload . cmd ) ;
2026-05-31 23:58:26 +09:00
}
return safe ;
}
function _stripStateSecrets ( state ) {
const safe = { ... state } ;
if ( safe . env && typeof safe . env === 'object' ) {
const { hfToken , ... env } = safe . env ;
2026-06-26 08:13:01 -04:00
delete env . hostPlatform ;
2026-05-31 23:58:26 +09:00
safe . env = env ;
}
2026-06-22 02:39:18 +00:00
if ( Array . isArray ( safe . tasks ) ) safe . tasks = safe . tasks . map ( _redactTaskForStorage ) ;
2026-05-31 23:58:26 +09:00
return safe ;
}
export function _saveTasks ( tasks ) {
2026-06-22 02:39:18 +00:00
localStorage . setItem ( TASKS _KEY , JSON . stringify ( ( tasks || [ ] ) . map ( _redactTaskForStorage ) ) ) ;
2026-05-31 23:58:26 +09:00
_syncToServer ( ) ;
}
export function _addTask ( sessionId , name , type , payload ) {
let tasks = _loadTasks ( ) ;
const remoteHost = ( payload && payload . remote _host ) || _envState . remoteHost || '' ;
2026-06-21 11:02:35 +00:00
const remoteServerKey = ( payload && payload . remote _server _key ) || '' ;
const remoteServerName = ( payload && payload . remote _server _name ) || '' ;
const sshPort = ( payload && payload . ssh _port ) || _getPort ( remoteServerKey || remoteHost ) || '' ;
const platform = ( payload && payload . platform ) || _getPlatform ( remoteServerKey || remoteHost ) || '' ;
2026-05-31 23:58:26 +09:00
// Serving a model supersedes its finished download — clear the matching
// finished download card (covers serving directly from the Serve tab, not just
// via the download card's "Serve →" button).
if ( type === 'serve' && payload && payload . repo _id ) {
const _repoId = payload . repo _id ;
tasks = tasks . filter ( t => ! ( t . type === 'download' && t . status === 'done' && t . payload && t . payload . repo _id === _repoId ) ) ;
}
2026-06-02 22:38:55 +09:00
if ( type === 'download' && payload && payload . repo _id ) {
const key = _downloadDedupeKey ( { type : 'download' , payload , remoteHost } ) ;
tasks = tasks . filter ( t => {
if ( t . sessionId === sessionId ) return false ;
return ! ( key && t . type === 'download' && t . status === 'queued' && _downloadDedupeKey ( t ) === key ) ;
} ) ;
}
2026-06-22 02:39:18 +00:00
const task = _redactTaskForStorage ( { id : sessionId , sessionId , name , type , status : 'running' , output : '' , ts : Date . now ( ) , payload : payload || null , remoteHost , remoteServerKey , remoteServerName , sshPort , platform } ) ;
2026-05-31 23:58:26 +09:00
tasks . push ( task ) ;
_saveTasks ( tasks ) ;
// New action → collapse all other cards, leave only this one open.
_soloExpandTaskId = sessionId ;
_renderRunningTab ( ) ;
// Always start the background monitor when a task is added — works even
// when modal is closed and ensures the sidebar shows live status immediately
_startBackgroundMonitor ( ) ;
// Switch to Running tab
const body = document . querySelector ( '#cookbook-modal .cookbook-body' ) ;
if ( body ) {
const tab = body . querySelector ( '.cookbook-tab[data-backend="Running"]' ) ;
if ( tab ) tab . click ( ) ;
}
return task ;
}
function _updateTask ( sessionId , updates ) {
const tasks = _loadTasks ( ) ;
const task = tasks . find ( t => t . sessionId === sessionId ) ;
if ( task ) {
Object . assign ( task , updates ) ;
_saveTasks ( tasks ) ;
}
if ( 'status' in updates || '_unreachable' in updates ) {
_refreshServerDots ( ) ;
}
if ( updates . status && updates . status !== 'running' ) {
const el = document . querySelector ( ` .cookbook-task[data-task-id=" ${ sessionId } "] ` ) ;
if ( el ) {
if ( el . _uptimeInterval ) { clearInterval ( el . _uptimeInterval ) ; el . _uptimeInterval = null ; }
const wave = el . querySelector ( '.cookbook-task-wave' ) ;
if ( wave ) wave . style . display = 'none' ;
const uptime = el . querySelector ( '.cookbook-task-uptime' ) ;
if ( uptime ) uptime . style . display = 'none' ;
}
}
}
2026-06-01 18:58:06 +05:30
function _refreshDepsAfterInstall ( task ) {
if ( ! task || task . type !== 'download' || ! task . payload ? . _dep ) return ;
try {
_refreshDependencies ? . ( { host : task . remoteHost || '' , port : task . sshPort || '' , venv : task . payload ? . env _path || '' } ) ;
} catch { }
}
2026-05-31 23:58:26 +09:00
export function _removeTask ( sessionId ) {
_tombstoneTask ( sessionId ) ; // so sync/poll can't resurrect it
const tasks = _loadTasks ( ) . filter ( t => t . sessionId !== sessionId ) ;
_saveTasks ( tasks ) ;
_renderRunningTab ( ) ;
}
// Fade/slide the task card out, then remove it — so the smooth exit is the same
// whether a task auto-stops or the user removes/kills it manually.
function _animateOutThenRemove ( el , sessionId ) {
if ( ! el || ! el . style ) { _removeTask ( sessionId ) ; return ; }
if ( el . _abort ) el . _abort . abort ( ) ;
el . style . transition = 'opacity 0.35s ease, transform 0.35s ease' ;
el . style . opacity = '0' ;
el . style . transform = 'translateX(-10px)' ;
setTimeout ( ( ) => _removeTask ( sessionId ) , 360 ) ;
}
// ── tmux / Windows session commands ──
2026-06-30 01:47:48 +00:00
function _taskRemoteHost ( task ) {
return task ? . remoteHost || task ? . payload ? . remote _host || '' ;
}
2026-07-07 00:50:07 +00:00
function _remoteTmuxPrefix ( ) {
return 'PATH="$HOME/.local/bin:$HOME/bin:/opt/homebrew/bin:/usr/local/bin:$PATH"; ' ;
}
2026-05-31 23:58:26 +09:00
export function _tmuxCmd ( task , tmuxArgs ) {
if ( _isWindows ( task ) ) {
return _winSessionCmd ( task , tmuxArgs ) ;
}
2026-06-30 01:47:48 +00:00
const host = _taskRemoteHost ( task ) ;
if ( host ) {
2026-07-07 00:50:07 +00:00
return ` ssh ${ _sshPrefix ( _getPort ( task ) ) } ${ host } ' ${ _remoteTmuxPrefix ( ) } tmux ${ tmuxArgs } ' 2>/dev/null ` ;
2026-05-31 23:58:26 +09:00
}
return ` tmux ${ tmuxArgs } 2>/dev/null ` ;
}
function _winSessionCmd ( task , tmuxArgs ) {
2026-06-30 01:47:48 +00:00
const host = _taskRemoteHost ( task ) ;
2026-06-04 09:00:01 +05:30
const sd = host ? '$env:TEMP\\odysseus-sessions' : '$env:TEMP\\odysseus-tmux' ;
2026-05-31 23:58:26 +09:00
const sid = task . sessionId ;
const pf = _sshPrefix ( _getPort ( task ) ) ;
if ( tmuxArgs . includes ( 'capture-pane' ) ) {
const lines = tmuxArgs . match ( /-S\s*-?(\d+)/ ) ? . [ 1 ] || '200' ;
2026-06-04 09:00:01 +05:30
const ps = host
? ` Get-Content ' ${ sd } \\ ${ sid } .log' -Tail ${ lines } -ErrorAction SilentlyContinue `
: ` Get-Content (Join-Path $ env:TEMP 'odysseus-tmux \\ ${ sid } .log') -Tail ${ lines } -ErrorAction SilentlyContinue ` ;
2026-06-19 09:28:25 +02:00
return _winPowerShellCmd ( task , ps ) ;
2026-05-31 23:58:26 +09:00
}
if ( tmuxArgs . includes ( 'has-session' ) ) {
2026-06-04 09:00:01 +05:30
const ps = host
? ` $ p = Get-Content ' ${ sd } \\ ${ sid } .pid' -ErrorAction SilentlyContinue; if ( $ p) { Get-Process -Id $ p -ErrorAction SilentlyContinue | Out-Null; if ( $ ?) { exit 0 } else { exit 1 } } else { exit 1 } `
: ` $ p = Get-Content (Join-Path $ env:TEMP 'odysseus-tmux \\ ${ sid } .pid') -ErrorAction SilentlyContinue; if ( $ p) { Get-Process -Id $ p -ErrorAction SilentlyContinue | Out-Null; if ( $ ?) { exit 0 } else { exit 1 } } else { exit 1 } ` ;
2026-06-19 09:28:25 +02:00
return _winPowerShellCmd ( task , ps ) ;
2026-05-31 23:58:26 +09:00
}
if ( tmuxArgs . includes ( 'kill-session' ) ) {
2026-06-19 09:28:25 +02:00
const ps = _winSessionStopTreePs ( task ) ;
return _winPowerShellCmd ( task , ps ) ;
2026-05-31 23:58:26 +09:00
}
if ( tmuxArgs . includes ( 'send-keys' ) && tmuxArgs . includes ( 'C-c' ) ) {
2026-06-04 09:00:01 +05:30
const ps = host
? ` $ p = Get-Content ' ${ sd } \\ ${ sid } .pid' -ErrorAction SilentlyContinue; if ( $ p) { Stop-Process -Id $ p -ErrorAction SilentlyContinue } `
: ` $ p = Get-Content (Join-Path $ env:TEMP 'odysseus-tmux \\ ${ sid } .pid') -ErrorAction SilentlyContinue; if ( $ p) { Stop-Process -Id $ p -ErrorAction SilentlyContinue } ` ;
2026-06-19 09:28:25 +02:00
return _winPowerShellCmd ( task , ps ) ;
2026-05-31 23:58:26 +09:00
}
2026-07-07 00:50:07 +00:00
return host ? ` ssh ${ pf } ${ host } ' ${ _remoteTmuxPrefix ( ) } tmux ${ tmuxArgs } ' 2>/dev/null ` : ` tmux ${ tmuxArgs } 2>/dev/null ` ;
2026-05-31 23:58:26 +09:00
}
2026-06-19 09:28:25 +02:00
function _winPowerShellCmd ( task , ps ) {
const command = ` powershell -Command " ${ ps } " ` ;
2026-06-30 01:47:48 +00:00
const host = _taskRemoteHost ( task ) ;
if ( ! host ) return command ;
return ` ssh ${ _sshPrefix ( _getPort ( task ) ) } ${ host } ${ _shQuote ( command ) } ` ;
2026-06-19 09:28:25 +02:00
}
function _winSessionStopTreePs ( task ) {
2026-06-30 01:47:48 +00:00
const host = _taskRemoteHost ( task ) ;
2026-06-19 09:28:25 +02:00
const sd = host ? '$env:TEMP\\odysseus-sessions' : '$env:TEMP\\odysseus-tmux' ;
const sid = task . sessionId ;
const stopTree = ` function Stop-Tree([int] $ Id) { Get-CimInstance Win32_Process -Filter ('ParentProcessId = ' + $ Id) -ErrorAction SilentlyContinue | ForEach-Object { Stop-Tree ([int] $ _.ProcessId) }; Stop-Process -Id $ Id -Force -ErrorAction SilentlyContinue } ` ;
return host
? ` ${ stopTree } ; $ p = Get-Content ' ${ sd } \\ ${ sid } .pid' -ErrorAction SilentlyContinue; if ( $ p -match '^ \\ d+ $ ') { Stop-Tree ([int] $ p) }; Remove-Item ' ${ sd } \\ ${ sid } .*' -Force -ErrorAction SilentlyContinue `
: ` ${ stopTree } ; $ p = Get-Content (Join-Path $ env:TEMP 'odysseus-tmux \\ ${ sid } .pid') -ErrorAction SilentlyContinue; if ( $ p -match '^ \\ d+ $ ') { Stop-Tree ([int] $ p) }; Remove-Item (Join-Path $ env:TEMP 'odysseus-tmux \\ ${ sid } .*') -Force -ErrorAction SilentlyContinue ` ;
}
Open email context for agent, email search across All Mail, cookbook serve polish
- Agent: pass the open email reader (uid/folder/account/from/subject/body
preview) on every chat submit so 'reply to this' / 'write email saying
hi' route to ui_control open_email_reply with the right UID instead of
inventing a new .md draft. Code-level enforcement (chat_routes strips
create_document + send_email when active_email is set); cross-session
active_doc_id is now trusted instead of being silently dropped.
set_active_email/clear_active_email tool-layer helpers in
tool_implementations.
- ui_control open_email_reply: optional body argument so the agent can
open-and-write in one call; envelope now forwards uid/folder/account/
body/panel through tool_output. Tool description sharpened and the
parser rejects empty bodies on reply/reply-all (forces the agent to
write rather than open an empty draft).
- Email library: search now runs against [Gmail]/All Mail when the
current folder is INBOX (archived emails surface). Whirlpool spinner
+ 'Searching…' placeholder while in flight. Each search result is
stamped with its source folder so clicks open the right email instead
of whatever shares its UID in INBOX. Search no longer re-applies the
same text pill locally (which only checks subject/from/snippet, never
body) so body-only matches don't get dropped after IMAP returns them.
Initial inbox load bumped 100→500.
- Email favorites: 'Favorite (pin to top)' / 'Unfavorite' in both the
card menu and the open-reader more menu, backed by a new
/api/email/flag/{uid}?on=true|false endpoint. Flagged emails always
bubble to the top of the grid regardless of active sort.
- AI reply in doc editor: never overwrites existing draft text or the
quoted history. AI suggestion is prepended; AI-generated 'On …
wrote:' re-quotes are stripped so the original quote isn't visually
edited.
- Cookbook serve: pre-launch GPU driver / has_gpu / install / version-
floor checks (vllm minimax_m2 needs 0.10.0+, deepseek_r1 needs 0.7.0
etc.) before the launch chain starts. Detect 'another model already
running on this host' and offer Stop & launch (with graceful then
force tmux kill helpers, port release wait). Per-vendor deep-link
buttons (vLLM recipe / SGLang cookbook) with hardware hash. Backend
picker is now a custom dropdown with accent-coloured logos for vLLM,
SGLang, llama.cpp, Ollama, Diffusers; same glyphs added next to
package names in Dependencies. Runtime-readiness note moved inside
the panel (green when ready, red when missing) with an × dismiss.
Esc collapses the expanded card; expanded card scrolls when it
overflows; Trust Remote / Auto Tool / Reasoning Parser / Enforce
Eager / Prefix Caching / Expert Parallel / Speculative / MoE Env on
one row (Reasoning Parser auto-detected per model family).
Dtype→Row 1, GPUs→Row 2 (rightmost). Removed redundant GPU 'auto'
input — command builders read from the GPU button strip. Default
cookbook open is Download tab.
- Cookbook hwfit: 'Model (latest)' / 'Model (oldest)' header sorts by
release_date; release dates can be backfilled with the new
scripts/backfill_model_release_dates.py and recipe metadata pulled
with scripts/import_from_vllm_recipes.py against the upstream
vllm-project/recipes catalog (vllm_recipe + min_vllm_version stamped
on entries).
- Calendar: Quick add hint cycles a random Odysseus-themed example per
open (wooden horse Friday, crew muster 10am daily, council on
Ithaca, …). Typing a time like '11pm' in the event title updates
the hero clock live.
- Doc editor: email-mode Reply button (sparkle icon, accent) opens the
same Fast/Full + context popover the email reader uses; Ctrl+Alt+M
toggles markdown preview.
- Memories panel: custom sort picker with per-option icons, default
'Latest', visible Enabled/Disabled toggle text matching the section
description style.
2026-06-15 20:47:51 +09:00
export function _tmuxGracefulKill ( task ) {
2026-05-31 23:58:26 +09:00
if ( _isWindows ( task ) ) {
2026-06-19 09:28:25 +02:00
const ps = _winSessionStopTreePs ( task ) ;
return _winPowerShellCmd ( task , ps ) ;
2026-05-31 23:58:26 +09:00
}
2026-06-30 01:47:48 +00:00
const host = _taskRemoteHost ( task ) ;
if ( host ) {
2026-07-07 00:50:07 +00:00
return ` ssh ${ _sshPrefix ( _getPort ( task ) ) } ${ host } ' ${ _remoteTmuxPrefix ( ) } tmux send-keys -t ${ task . sessionId } C-c 2>/dev/null; sleep 2; tmux kill-session -t ${ task . sessionId } 2>/dev/null' ` ;
2026-05-31 23:58:26 +09:00
}
return ` tmux send-keys -t ${ task . sessionId } C-c 2>/dev/null; sleep 2; tmux kill-session -t ${ task . sessionId } 2>/dev/null ` ;
}
Open email context for agent, email search across All Mail, cookbook serve polish
- Agent: pass the open email reader (uid/folder/account/from/subject/body
preview) on every chat submit so 'reply to this' / 'write email saying
hi' route to ui_control open_email_reply with the right UID instead of
inventing a new .md draft. Code-level enforcement (chat_routes strips
create_document + send_email when active_email is set); cross-session
active_doc_id is now trusted instead of being silently dropped.
set_active_email/clear_active_email tool-layer helpers in
tool_implementations.
- ui_control open_email_reply: optional body argument so the agent can
open-and-write in one call; envelope now forwards uid/folder/account/
body/panel through tool_output. Tool description sharpened and the
parser rejects empty bodies on reply/reply-all (forces the agent to
write rather than open an empty draft).
- Email library: search now runs against [Gmail]/All Mail when the
current folder is INBOX (archived emails surface). Whirlpool spinner
+ 'Searching…' placeholder while in flight. Each search result is
stamped with its source folder so clicks open the right email instead
of whatever shares its UID in INBOX. Search no longer re-applies the
same text pill locally (which only checks subject/from/snippet, never
body) so body-only matches don't get dropped after IMAP returns them.
Initial inbox load bumped 100→500.
- Email favorites: 'Favorite (pin to top)' / 'Unfavorite' in both the
card menu and the open-reader more menu, backed by a new
/api/email/flag/{uid}?on=true|false endpoint. Flagged emails always
bubble to the top of the grid regardless of active sort.
- AI reply in doc editor: never overwrites existing draft text or the
quoted history. AI suggestion is prepended; AI-generated 'On …
wrote:' re-quotes are stripped so the original quote isn't visually
edited.
- Cookbook serve: pre-launch GPU driver / has_gpu / install / version-
floor checks (vllm minimax_m2 needs 0.10.0+, deepseek_r1 needs 0.7.0
etc.) before the launch chain starts. Detect 'another model already
running on this host' and offer Stop & launch (with graceful then
force tmux kill helpers, port release wait). Per-vendor deep-link
buttons (vLLM recipe / SGLang cookbook) with hardware hash. Backend
picker is now a custom dropdown with accent-coloured logos for vLLM,
SGLang, llama.cpp, Ollama, Diffusers; same glyphs added next to
package names in Dependencies. Runtime-readiness note moved inside
the panel (green when ready, red when missing) with an × dismiss.
Esc collapses the expanded card; expanded card scrolls when it
overflows; Trust Remote / Auto Tool / Reasoning Parser / Enforce
Eager / Prefix Caching / Expert Parallel / Speculative / MoE Env on
one row (Reasoning Parser auto-detected per model family).
Dtype→Row 1, GPUs→Row 2 (rightmost). Removed redundant GPU 'auto'
input — command builders read from the GPU button strip. Default
cookbook open is Download tab.
- Cookbook hwfit: 'Model (latest)' / 'Model (oldest)' header sorts by
release_date; release dates can be backfilled with the new
scripts/backfill_model_release_dates.py and recipe metadata pulled
with scripts/import_from_vllm_recipes.py against the upstream
vllm-project/recipes catalog (vllm_recipe + min_vllm_version stamped
on entries).
- Calendar: Quick add hint cycles a random Odysseus-themed example per
open (wooden horse Friday, crew muster 10am daily, council on
Ithaca, …). Typing a time like '11pm' in the event title updates
the hero clock live.
- Doc editor: email-mode Reply button (sparkle icon, accent) opens the
same Fast/Full + context popover the email reader uses; Ctrl+Alt+M
toggles markdown preview.
- Memories panel: custom sort picker with per-option icons, default
'Latest', visible Enabled/Disabled toggle text matching the section
description style.
2026-06-15 20:47:51 +09:00
// Force-kill escalation: SIGKILL the tmux pane's owning PID and any children,
// then nuke the session. Use AFTER the graceful kill when the process is
// still detected — vLLM sometimes ignores SIGINT during model init, and a
// stuck CUDA context can survive `tmux kill-session` alone.
export function _tmuxForceKill ( task ) {
if ( _isWindows ( task ) ) {
// Windows graceful path already does Stop-Process -Force, so the same
// command serves as the "force" variant.
return _tmuxGracefulKill ( task ) ;
}
const sid = task . sessionId ;
const inner =
` PIDS= $ (tmux list-panes -t ${ sid } -F "#{pane_pid}" 2>/dev/null); ` +
` if [ -n " $ PIDS" ]; then ` +
` for P in $ PIDS; do ` +
` pkill -KILL -P " $ P" 2>/dev/null; ` +
` kill -9 " $ P" 2>/dev/null; ` +
` done; ` +
` fi; ` +
` tmux kill-session -t ${ sid } 2>/dev/null ` ;
2026-06-30 01:47:48 +00:00
const host = _taskRemoteHost ( task ) ;
if ( host ) {
2026-07-07 00:50:07 +00:00
return ` ssh ${ _sshPrefix ( _getPort ( task ) ) } ${ host } ${ _shQuote ( _remoteTmuxPrefix ( ) + inner ) } ` ;
Open email context for agent, email search across All Mail, cookbook serve polish
- Agent: pass the open email reader (uid/folder/account/from/subject/body
preview) on every chat submit so 'reply to this' / 'write email saying
hi' route to ui_control open_email_reply with the right UID instead of
inventing a new .md draft. Code-level enforcement (chat_routes strips
create_document + send_email when active_email is set); cross-session
active_doc_id is now trusted instead of being silently dropped.
set_active_email/clear_active_email tool-layer helpers in
tool_implementations.
- ui_control open_email_reply: optional body argument so the agent can
open-and-write in one call; envelope now forwards uid/folder/account/
body/panel through tool_output. Tool description sharpened and the
parser rejects empty bodies on reply/reply-all (forces the agent to
write rather than open an empty draft).
- Email library: search now runs against [Gmail]/All Mail when the
current folder is INBOX (archived emails surface). Whirlpool spinner
+ 'Searching…' placeholder while in flight. Each search result is
stamped with its source folder so clicks open the right email instead
of whatever shares its UID in INBOX. Search no longer re-applies the
same text pill locally (which only checks subject/from/snippet, never
body) so body-only matches don't get dropped after IMAP returns them.
Initial inbox load bumped 100→500.
- Email favorites: 'Favorite (pin to top)' / 'Unfavorite' in both the
card menu and the open-reader more menu, backed by a new
/api/email/flag/{uid}?on=true|false endpoint. Flagged emails always
bubble to the top of the grid regardless of active sort.
- AI reply in doc editor: never overwrites existing draft text or the
quoted history. AI suggestion is prepended; AI-generated 'On …
wrote:' re-quotes are stripped so the original quote isn't visually
edited.
- Cookbook serve: pre-launch GPU driver / has_gpu / install / version-
floor checks (vllm minimax_m2 needs 0.10.0+, deepseek_r1 needs 0.7.0
etc.) before the launch chain starts. Detect 'another model already
running on this host' and offer Stop & launch (with graceful then
force tmux kill helpers, port release wait). Per-vendor deep-link
buttons (vLLM recipe / SGLang cookbook) with hardware hash. Backend
picker is now a custom dropdown with accent-coloured logos for vLLM,
SGLang, llama.cpp, Ollama, Diffusers; same glyphs added next to
package names in Dependencies. Runtime-readiness note moved inside
the panel (green when ready, red when missing) with an × dismiss.
Esc collapses the expanded card; expanded card scrolls when it
overflows; Trust Remote / Auto Tool / Reasoning Parser / Enforce
Eager / Prefix Caching / Expert Parallel / Speculative / MoE Env on
one row (Reasoning Parser auto-detected per model family).
Dtype→Row 1, GPUs→Row 2 (rightmost). Removed redundant GPU 'auto'
input — command builders read from the GPU button strip. Default
cookbook open is Download tab.
- Cookbook hwfit: 'Model (latest)' / 'Model (oldest)' header sorts by
release_date; release dates can be backfilled with the new
scripts/backfill_model_release_dates.py and recipe metadata pulled
with scripts/import_from_vllm_recipes.py against the upstream
vllm-project/recipes catalog (vllm_recipe + min_vllm_version stamped
on entries).
- Calendar: Quick add hint cycles a random Odysseus-themed example per
open (wooden horse Friday, crew muster 10am daily, council on
Ithaca, …). Typing a time like '11pm' in the event title updates
the hero clock live.
- Doc editor: email-mode Reply button (sparkle icon, accent) opens the
same Fast/Full + context popover the email reader uses; Ctrl+Alt+M
toggles markdown preview.
- Memories panel: custom sort picker with per-option icons, default
'Latest', visible Enabled/Disabled toggle text matching the section
description style.
2026-06-15 20:47:51 +09:00
}
return inner ;
}
// Returns a shell snippet that prints "ALIVE" if the tmux session still
// exists (or its main PID is still listed in /proc), "DEAD" otherwise.
// Used by the Stop-all escalation to decide whether to force-kill.
export function _tmuxIsAliveCheck ( task ) {
if ( _isWindows ( task ) ) {
// Skip the check on Windows — the graceful path already force-kills.
return null ;
}
const sid = task . sessionId ;
const inner = ` if tmux has-session -t ${ sid } 2>/dev/null; then echo ALIVE; else echo DEAD; fi ` ;
2026-06-30 01:47:48 +00:00
const host = _taskRemoteHost ( task ) ;
if ( host ) {
2026-07-07 00:50:07 +00:00
return ` ssh ${ _sshPrefix ( _getPort ( task ) ) } ${ host } ${ _shQuote ( _remoteTmuxPrefix ( ) + inner ) } ` ;
Open email context for agent, email search across All Mail, cookbook serve polish
- Agent: pass the open email reader (uid/folder/account/from/subject/body
preview) on every chat submit so 'reply to this' / 'write email saying
hi' route to ui_control open_email_reply with the right UID instead of
inventing a new .md draft. Code-level enforcement (chat_routes strips
create_document + send_email when active_email is set); cross-session
active_doc_id is now trusted instead of being silently dropped.
set_active_email/clear_active_email tool-layer helpers in
tool_implementations.
- ui_control open_email_reply: optional body argument so the agent can
open-and-write in one call; envelope now forwards uid/folder/account/
body/panel through tool_output. Tool description sharpened and the
parser rejects empty bodies on reply/reply-all (forces the agent to
write rather than open an empty draft).
- Email library: search now runs against [Gmail]/All Mail when the
current folder is INBOX (archived emails surface). Whirlpool spinner
+ 'Searching…' placeholder while in flight. Each search result is
stamped with its source folder so clicks open the right email instead
of whatever shares its UID in INBOX. Search no longer re-applies the
same text pill locally (which only checks subject/from/snippet, never
body) so body-only matches don't get dropped after IMAP returns them.
Initial inbox load bumped 100→500.
- Email favorites: 'Favorite (pin to top)' / 'Unfavorite' in both the
card menu and the open-reader more menu, backed by a new
/api/email/flag/{uid}?on=true|false endpoint. Flagged emails always
bubble to the top of the grid regardless of active sort.
- AI reply in doc editor: never overwrites existing draft text or the
quoted history. AI suggestion is prepended; AI-generated 'On …
wrote:' re-quotes are stripped so the original quote isn't visually
edited.
- Cookbook serve: pre-launch GPU driver / has_gpu / install / version-
floor checks (vllm minimax_m2 needs 0.10.0+, deepseek_r1 needs 0.7.0
etc.) before the launch chain starts. Detect 'another model already
running on this host' and offer Stop & launch (with graceful then
force tmux kill helpers, port release wait). Per-vendor deep-link
buttons (vLLM recipe / SGLang cookbook) with hardware hash. Backend
picker is now a custom dropdown with accent-coloured logos for vLLM,
SGLang, llama.cpp, Ollama, Diffusers; same glyphs added next to
package names in Dependencies. Runtime-readiness note moved inside
the panel (green when ready, red when missing) with an × dismiss.
Esc collapses the expanded card; expanded card scrolls when it
overflows; Trust Remote / Auto Tool / Reasoning Parser / Enforce
Eager / Prefix Caching / Expert Parallel / Speculative / MoE Env on
one row (Reasoning Parser auto-detected per model family).
Dtype→Row 1, GPUs→Row 2 (rightmost). Removed redundant GPU 'auto'
input — command builders read from the GPU button strip. Default
cookbook open is Download tab.
- Cookbook hwfit: 'Model (latest)' / 'Model (oldest)' header sorts by
release_date; release dates can be backfilled with the new
scripts/backfill_model_release_dates.py and recipe metadata pulled
with scripts/import_from_vllm_recipes.py against the upstream
vllm-project/recipes catalog (vllm_recipe + min_vllm_version stamped
on entries).
- Calendar: Quick add hint cycles a random Odysseus-themed example per
open (wooden horse Friday, crew muster 10am daily, council on
Ithaca, …). Typing a time like '11pm' in the event title updates
the hero clock live.
- Doc editor: email-mode Reply button (sparkle icon, accent) opens the
same Fast/Full + context popover the email reader uses; Ctrl+Alt+M
toggles markdown preview.
- Memories panel: custom sort picker with per-option icons, default
'Latest', visible Enabled/Disabled toggle text matching the section
description style.
2026-06-15 20:47:51 +09:00
}
return inner ;
}
2026-06-02 22:38:55 +09:00
function _shQuote ( value ) {
return "'" + String ( value ? ? '' ) . replace ( /'/g , "'\\''" ) + "'" ;
}
function _taskLooksOllama ( task , outputText = '' ) {
const haystack = ` ${ task ? . payload ? . backend || '' } ${ task ? . payload ? . _cmd || '' } ${ task ? . payload ? . _fields ? . backend || '' } ${ outputText || '' } ` ;
return /\bollama\b/i . test ( haystack ) || /Ollama API ready on port\s+\d+/i . test ( haystack ) ;
}
function _ollamaBaseUrlForTask ( task , outputText = '' ) {
const out = String ( outputText || '' ) ;
const ready = out . match ( /Ollama API ready on port\s+\d+:\s*(http:\/\/[^\s]+)/i ) ;
if ( ready ) return ready [ 1 ] . replace ( /\/+$/ , '' ) ;
const cmd = String ( task ? . payload ? . _cmd || '' ) ;
const host = cmd . match ( /OLLAMA_HOST=([^\s]+)/ ) ? . [ 1 ] || '' ;
const port = host . match ( /:(\d+)$/ ) ? . [ 1 ] || '11434' ;
return ` http://127.0.0.1: ${ port } ` ;
}
function _ollamaModelForTask ( task ) {
return String ( task ? . payload ? . model || task ? . payload ? . repo _id || task ? . name || '' ) . trim ( ) ;
}
function _ollamaUnloadCommand ( task , outputText = '' ) {
if ( ! _taskLooksOllama ( task , outputText ) ) return '' ;
const model = _ollamaModelForTask ( task ) ;
if ( ! model ) return '' ;
const base = _ollamaBaseUrlForTask ( task , outputText ) ;
const body = JSON . stringify ( { model , prompt : '' , keep _alive : 0 , stream : false } ) ;
const inner = ` curl -sf -X POST ${ _shQuote ( base + '/api/generate' ) } -H 'Content-Type: application/json' -d ${ _shQuote ( body ) } >/dev/null 2>&1 || true ` ;
2026-06-30 01:47:48 +00:00
const host = _taskRemoteHost ( task ) ;
if ( host ) {
return ` ssh ${ _sshPrefix ( _getPort ( task ) ) } ${ host } ${ _shQuote ( inner ) } ` ;
2026-06-02 22:38:55 +09:00
}
return inner ;
}
function _endpointUrlForTask ( task , outputText = '' ) {
if ( _taskLooksOllama ( task , outputText ) ) {
return _ollamaBaseUrlForTask ( task , outputText ) + '/v1' ;
}
2026-06-30 01:47:48 +00:00
const host = _connectHostFromRemote ( _taskRemoteHost ( task ) ) ;
2026-06-02 22:38:55 +09:00
const portMatch = task . payload ? . _cmd ? . match ( /--port\s+(\d+)/ ) ;
const port = portMatch ? portMatch [ 1 ] : '8000' ;
return ` http:// ${ host } : ${ port } /v1 ` ;
}
2026-05-31 23:58:26 +09:00
// ── Wave animation ──
const _waveFrames = [ '▁▂▃' , '▂▃▄' , '▃▄▅' , '▄▅▆' , '▅▆▅' , '▆▅▄' , '▅▄▃' , '▄▃▂' , '▃▂▁' ] ;
let _waveIdx = 0 ;
let _waveTimer = null ;
const _waveEls = new Set ( ) ;
function _startWaveSync ( ) {
if ( _waveTimer ) return ;
_waveTimer = setInterval ( ( ) => {
_waveIdx = ( _waveIdx + 1 ) % _waveFrames . length ;
for ( const el of _waveEls ) {
if ( ! el . isConnected ) { _waveEls . delete ( el ) ; continue ; }
if ( el . style . display !== 'none' ) el . textContent = _waveFrames [ _waveIdx ] ;
}
if ( ! _waveEls . size ) { clearInterval ( _waveTimer ) ; _waveTimer = null ; }
} , 200 ) ;
}
function _registerWaveEl ( el ) { _waveEls . add ( el ) ; _startWaveSync ( ) ; }
// ── Notifications ──
function _showCookbookNotif ( isError = false ) {
const dot = document . getElementById ( 'cookbook-notif-dot' ) ;
if ( dot ) {
dot . style . display = '' ;
dot . classList . toggle ( 'cookbook-notif-error' , isError ) ;
}
const btn = document . getElementById ( 'tool-cookbook-btn' ) ;
if ( btn ) { btn . style . opacity = '1' ; btn . classList . add ( 'cookbook-notif-active' ) ; }
const railBtn = document . getElementById ( 'rail-cookbook' ) ;
if ( railBtn ) {
railBtn . classList . remove ( 'rail-notify-success' , 'rail-notify-error' ) ;
railBtn . classList . add ( 'rail-notify' , isError ? 'rail-notify-error' : 'rail-notify-success' , 'cookbook-notif-active' ) ;
}
if ( window . _syncRailDynamic ) window . _syncRailDynamic ( ) ;
}
export function _clearCookbookNotif ( ) {
const dot = document . getElementById ( 'cookbook-notif-dot' ) ;
if ( dot ) dot . style . display = 'none' ;
const btn = document . getElementById ( 'tool-cookbook-btn' ) ;
if ( btn ) { btn . style . opacity = '' ; btn . classList . remove ( 'cookbook-notif-active' ) ; }
const railBtn = document . getElementById ( 'rail-cookbook' ) ;
if ( railBtn ) {
railBtn . classList . remove ( 'rail-notify' , 'rail-notify-success' , 'cookbook-notif-active' ) ;
}
if ( window . _syncRailDynamic ) window . _syncRailDynamic ( ) ;
}
// ── Presets helper (for save-from-task) ──
// A preset must carry the venv + activated GPUs, not just the command — without
// them a relaunch has no environment activated and no GPU pinning, so a config
// that worked when saved fails on reload. Pull them from the launch payload
// (_env/_envPath/_gpus, captured by _launchServeTask) and fold them into the
// serve-form `fields` the Serve panel restores from.
function _presetEnvFields ( task ) {
const p = task . payload || { } ;
const fields = { ... ( p . _fields || { } ) } ;
// The Serve panel's venv field is a path; conda/venv both activate from it.
if ( p . _envPath && ( p . _env === 'venv' || p . _env === 'conda' ) ) fields . venv = fields . venv || p . _envPath ;
if ( p . _gpus ) fields . gpus = p . _gpus ;
return {
fields : Object . keys ( fields ) . length ? fields : undefined ,
env : p . _env || '' ,
envPath : p . _envPath || '' ,
gpus : p . _gpus || '' ,
} ;
}
2026-06-22 02:39:18 +00:00
function _redactPresetForStorage ( preset ) {
if ( ! preset || typeof preset !== 'object' ) return preset ;
const safe = { ... preset } ;
if ( typeof safe . cmd === 'string' ) safe . cmd = _redactStoredText ( safe . cmd ) ;
if ( typeof safe . command === 'string' ) safe . command = _redactStoredText ( safe . command ) ;
delete safe . hf _token ;
delete safe . hfToken ;
return safe ;
}
2026-05-31 23:58:26 +09:00
function _saveTaskAsPreset ( task , label ) {
const host = task . remoteHost || 'localhost' ;
const portMatch = task . payload ? . _cmd ? . match ( /--port\s+(\d+)/ ) ;
const port = portMatch ? portMatch [ 1 ] : '8000' ;
const presets = _loadPresets ( ) ;
if ( presets . some ( p => p . cmd === task . payload . _cmd ) ) return false ;
2026-06-22 02:39:18 +00:00
presets . push ( _redactPresetForStorage ( { name : task . name , model : task . payload . repo _id , backend : 'vllm' , host , port , cmd : task . payload . _cmd , remoteHost : task . remoteHost || '' , label : label || task . name , ... _presetEnvFields ( task ) } ) ) ;
_savePresets ( presets . map ( _redactPresetForStorage ) ) ;
2026-05-31 23:58:26 +09:00
return true ;
}
// Same model-matching as cookbookServe's _presetsForModel, so the auto-save cap
// counts the exact slots the Serve tab shows for this model.
function _presetsForModelLocal ( presets , repo ) {
const short = ( repo || '' ) . split ( '/' ) . pop ( ) ;
return presets . filter ( p => {
const pm = p . model || '' , pn = p . name || '' ;
return pm === repo || pn === repo || pm . split ( '/' ) . pop ( ) === short || pn === short ;
} ) ;
}
// Build a short auto-label from the launched command so an auto-saved config is
// recognizable in the Saved dropdown (e.g. "TP2 · 16k ctx · AWQ").
function _autoConfigLabel ( task ) {
const cmd = task . payload ? . _cmd || '' ;
const bits = [ ] ;
const tp = cmd . match ( /--tensor-parallel-size[=\s]+(\d+)/ ) ;
if ( tp && tp [ 1 ] !== '1' ) bits . push ( 'TP' + tp [ 1 ] ) ;
const ml = cmd . match ( /--max-model-len[=\s]+(\d+)/ ) ;
if ( ml ) { const n = parseInt ( ml [ 1 ] ) ; bits . push ( ( n >= 1024 ? Math . round ( n / 1024 ) + 'k' : n ) + ' ctx' ) ; }
const q = ( task . name || '' ) . match ( /AWQ|GPTQ|FP8|Q4|Q5|Q6|Q8|INT8|INT4/i ) ;
if ( q ) bits . push ( q [ 0 ] . toUpperCase ( ) ) ;
return bits . length ? bits . join ( ' · ' ) : 'working' ;
}
// Auto-save a serve config the moment its endpoint registers successfully, and
// flag it confirmed-working. Dedups by exact command: if the same settings are
// already saved we just upgrade that slot's badge instead of duplicating it.
// Runs at most once per task.
function _autoSaveWorkingConfig ( task ) {
if ( ! task || task . type !== 'serve' || ! task . payload ? . _cmd ) return ;
if ( task . _autoSaved ) return ;
const cmd = task . payload . _cmd ;
// Diffusion/image servers aren't vLLM presets — skip them.
if ( cmd . includes ( 'diffusion_server' ) ) { task . _autoSaved = true ; return ; }
const model = task . payload . repo _id || task . name ;
const presets = _loadPresets ( ) ;
const existing = presets . find ( p => p . cmd === cmd ) ;
if ( existing ) {
task . _autoSaved = true ;
2026-06-22 02:39:18 +00:00
if ( ! existing . confirmedWorking ) { existing . confirmedWorking = true ; _savePresets ( presets . map ( _redactPresetForStorage ) ) ; }
2026-05-31 23:58:26 +09:00
return ; // already saved → just confirm it, no duplicate, no toast
}
// Respect the per-model cap the manual save flow uses (max 5).
if ( _presetsForModelLocal ( presets , model ) . length >= 5 ) { task . _autoSaved = true ; return ; }
const host = task . remoteHost || 'localhost' ;
const portMatch = cmd . match ( /--port[=\s]+(\d+)/ ) ;
const port = portMatch ? portMatch [ 1 ] : '8000' ;
2026-06-22 02:39:18 +00:00
presets . push ( _redactPresetForStorage ( {
2026-05-31 23:58:26 +09:00
name : task . name , model , backend : 'vllm' , host , port ,
cmd , remoteHost : task . remoteHost || '' ,
label : _autoConfigLabel ( task ) , confirmedWorking : true , autoSaved : true ,
... _presetEnvFields ( task ) ,
2026-06-22 02:39:18 +00:00
} ) ) ;
_savePresets ( presets . map ( _redactPresetForStorage ) ) ;
2026-05-31 23:58:26 +09:00
task . _autoSaved = true ;
uiModule . showToast ( 'Saved working config' ) ;
}
// ── Cross-device sync ──
let _syncTimer = null ;
function _syncToServer ( ) {
// Debounce to coalesce bursts of writes, but keep latency low so the server
// is effectively authoritative across devices
clearTimeout ( _syncTimer ) ;
_syncTimer = setTimeout ( async ( ) => {
try {
// Don't push a not-yet-hydrated state. A legit state always has at
// least the "Local" server, so an empty servers list means we loaded
// before GET /state populated _envState — syncing it would wipe the
// saved servers. (The server has an anti-wipe guard too; this avoids
// the needless round-trip.)
if ( ! _envState || ! Array . isArray ( _envState . servers ) || _envState . servers . length === 0 ) return ;
const state = {
tasks : _loadTasks ( ) ,
2026-06-22 01:49:15 +00:00
removedTasks : _loadTombstones ( ) ,
2026-05-31 23:58:26 +09:00
presets : _loadPresets ( ) ,
env : _envState ,
serveState : null ,
2026-07-07 00:50:07 +00:00
serveFavorites : [ ] ,
2026-05-31 23:58:26 +09:00
} ;
try { state . serveState = JSON . parse ( localStorage . getItem ( SERVE _STATE _KEY ) ) ; } catch { }
2026-07-07 00:50:07 +00:00
try {
const favorites = JSON . parse ( localStorage . getItem ( SERVE _FAVORITES _KEY ) || '[]' ) ;
state . serveFavorites = Array . isArray ( favorites ) ? favorites . filter ( Boolean ) . map ( String ) : [ ] ;
} catch { }
2026-05-31 23:58:26 +09:00
await fetch ( '/api/cookbook/state' , {
method : 'POST' , credentials : 'same-origin' ,
headers : { 'Content-Type' : 'application/json' } ,
body : JSON . stringify ( _stripStateSecrets ( state ) ) ,
} ) ;
} catch { }
} , 400 ) ;
}
2026-07-07 00:50:07 +00:00
document . addEventListener ( 'cookbook:state-dirty' , ( ) => {
_syncToServer ( ) ;
} ) ;
2026-05-31 23:58:26 +09:00
// Normalize state from server: collapse legacy duplicate keys to canonical form.
// - server.modelDir (singular) → server.modelDirs[0] (canonical)
// - strip ✕/✖ pollution from modelDirs
// - dedupe modelDirs
function _normalizeState ( state ) {
if ( ! state || typeof state !== 'object' ) return state ;
if ( state . env && Array . isArray ( state . env . servers ) ) {
for ( const s of state . env . servers ) {
// Collapse legacy modelDir → modelDirs
let dirs = Array . isArray ( s . modelDirs ) ? s . modelDirs : [ ] ;
if ( s . modelDir && ! dirs . includes ( s . modelDir ) ) dirs . push ( s . modelDir ) ;
dirs = dirs
. map ( d => ( d || '' ) . replaceAll ( '\u2715' , '' ) . replaceAll ( '\u2716' , '' ) . trim ( ) )
. filter ( Boolean ) ;
if ( ! dirs . includes ( '~/.cache/huggingface/hub' ) ) dirs . unshift ( '~/.cache/huggingface/hub' ) ;
s . modelDirs = [ ... new Set ( dirs ) ] ;
delete s . modelDir ; // Drop the legacy singular form
// A download target that's no longer in the dir list falls back to the
// default HF cache (empty) so we never download into an unscanned dir.
if ( s . downloadDir && ! s . modelDirs . includes ( s . downloadDir ) ) s . downloadDir = '' ;
}
}
return state ;
}
export async function _syncFromServer ( ) {
try {
const res = await fetch ( '/api/cookbook/state' , { credentials : 'same-origin' } ) ;
if ( ! res . ok ) return false ;
const state = _normalizeState ( await res . json ( ) ) ;
if ( ! state || ! state . env ) return false ;
const localTasks = _loadTasks ( ) ;
const serverTasks = state . tasks || [ ] ;
2026-06-22 01:49:15 +00:00
const serverTombstones = ( state . removedTasks && typeof state . removedTasks === 'object' ) ? state . removedTasks : { } ;
const localTombstones = _loadTombstones ( ) ;
const mergedTombstones = { ... serverTombstones , ... localTombstones } ;
for ( const [ id , ts ] of Object . entries ( serverTombstones ) ) {
if ( localTombstones [ id ] == null || Number ( ts ) > Number ( localTombstones [ id ] ) ) mergedTombstones [ id ] = ts ;
}
_saveTombstones ( mergedTombstones ) ;
2026-05-31 23:58:26 +09:00
const localIds = new Set ( localTasks . map ( t => t . sessionId ) ) ;
2026-06-22 01:49:15 +00:00
const merged = localTasks . filter ( t => ! _isTombstoned ( t . sessionId ) ) ;
2026-05-31 23:58:26 +09:00
for ( const t of serverTasks ) {
if ( ! localIds . has ( t . sessionId ) && ! _isTombstoned ( t . sessionId ) ) {
merged . push ( t ) ;
}
}
2026-06-22 02:39:18 +00:00
localStorage . setItem ( TASKS _KEY , JSON . stringify ( merged . map ( _redactTaskForStorage ) ) ) ;
2026-05-31 23:58:26 +09:00
if ( state . env ) {
// The active server selection (remoteHost + its env/path/platform) is a
// per-device, live choice. NEVER let the server's stored copy overwrite
// it here — doing so silently snapped the active host back to whatever was
// saved server-side, so downloads/scans ignored what the user just
// picked. Sync only the shared non-secret settings (servers list, gpus, paths).
const { remoteHost : _rh , env : _e , envPath : _ep , platform : _pf , ... settings } = state . env ;
delete settings . hfToken ;
Object . assign ( _envState , settings ) ;
2026-06-22 01:49:15 +00:00
const selected = ( _envState . remoteServerKey && _serverByVal ? . ( _envState . remoteServerKey ) )
|| ( _envState . remoteHost ? ( _envState . servers || [ ] ) . find ( s => s . host === _envState . remoteHost ) : null ) ;
if ( selected ) {
_envState . env = selected . env || 'none' ;
_envState . envPath = selected . envPath || '' ;
_envState . platform = selected . platform || '' ;
} else if ( ! _envState . remoteHost ) {
const local = ( _envState . servers || [ ] ) . find ( s => ! s . host || s . host === 'local' ) ;
_envState . env = local ? . env || 'none' ;
_envState . envPath = local ? . envPath || '' ;
_envState . platform = local ? . platform || '' ;
}
2026-05-31 23:58:26 +09:00
const { hfToken , ... safeState } = _envState ;
localStorage . setItem ( 'cookbook-last-state' , JSON . stringify ( safeState ) ) ;
}
if ( state . presets ) {
localStorage . setItem ( STORAGE _KEY , JSON . stringify ( state . presets ) ) ;
}
if ( state . serveState ) {
localStorage . setItem ( SERVE _STATE _KEY , JSON . stringify ( state . serveState ) ) ;
}
2026-07-07 00:50:07 +00:00
if ( Array . isArray ( state . serveFavorites ) ) {
localStorage . setItem ( SERVE _FAVORITES _KEY , JSON . stringify ( state . serveFavorites . filter ( Boolean ) . map ( String ) ) ) ;
}
2026-06-22 01:49:15 +00:00
document . dispatchEvent ( new CustomEvent ( 'cookbook:state-synced' , { detail : state } ) ) ;
2026-05-31 23:58:26 +09:00
return true ;
} catch { return false ; }
}
// ── Retry download ──
// Bounded auto-retry counter for downloads, keyed by model — network blips on
// big multi-file downloads are common and HF resumes from the .incomplete parts.
const _dlRetryCount = new Map ( ) ;
const _DL _MAX _AUTO _RETRY = 2 ;
// Kill + relaunch a task (download or serve). Shared by the ⋮ → Restart action
// and the click-to-retry on a stalled download badge.
async function _retryTask ( el , task ) {
if ( el && el . _abort ) el . _abort . abort ( ) ;
const badge = el ? . querySelector ( '.cookbook-task-status' ) ;
if ( badge ) { badge . textContent = 'restarting...' ; badge . className = 'cookbook-task-status' ; }
try {
await fetch ( '/api/shell/exec' , {
method : 'POST' , credentials : 'same-origin' ,
headers : { 'Content-Type' : 'application/json' } ,
body : JSON . stringify ( { command : _tmuxGracefulKill ( task ) } ) ,
} ) ;
} catch { }
if ( task . payload ) {
if ( task . type === 'serve' && task . payload . _cmd ) {
2026-06-02 22:38:55 +09:00
_removeTask ( task . sessionId ) ;
2026-05-31 23:58:26 +09:00
_launchServeTask ( task . name , task . payload . repo _id , task . payload . _cmd , task . payload . _fields , task . remoteHost || '' ) ;
} else {
2026-06-02 22:38:55 +09:00
uiModule . showToast ( 'Retrying download — progress may look reset while HuggingFace checks cached files, then it should resume.' , 7000 ) ;
_updateTask ( task . sessionId , {
status : 'running' ,
output : ` ${ task . output || '' } \n \n [odysseus] Retrying download. Progress may briefly look like a fresh download while HuggingFace checks cached/incomplete files; cached partial files will be reused when available. ` . trim ( ) ,
_retrying : true ,
} ) ;
_retryDownload ( task . name , task . payload , task . sessionId ) ;
2026-05-31 23:58:26 +09:00
}
}
}
2026-06-02 22:38:55 +09:00
async function _retryDownload ( name , payload , replaceSessionId = '' ) {
2026-05-31 23:58:26 +09:00
try {
// A retry means the fast hf_transfer path already failed once — fall back to
// the plain, reliable downloader for this and any further attempt (it resumes
// from the cached .incomplete files, so no progress is lost).
const _payload = { ... ( payload || { } ) , disable _hf _transfer : true } ;
const res = await fetch ( '/api/model/download' , {
method : 'POST' , credentials : 'same-origin' ,
headers : { 'Content-Type' : 'application/json' } ,
body : JSON . stringify ( _payload ) ,
} ) ;
if ( ! res . ok ) {
uiModule . showToast ( 'Download failed: HTTP ' + res . status ) ;
2026-06-02 22:38:55 +09:00
if ( replaceSessionId ) _updateTask ( replaceSessionId , { status : 'crashed' , _retrying : false } ) ;
2026-05-31 23:58:26 +09:00
return ;
}
const data = await res . json ( ) ;
if ( ! data . ok ) {
uiModule . showToast ( 'Download failed: ' + ( data . error || '' ) ) ;
2026-06-02 22:38:55 +09:00
if ( replaceSessionId ) _updateTask ( replaceSessionId , { status : 'crashed' , _retrying : false } ) ;
2026-05-31 23:58:26 +09:00
return ;
}
2026-06-02 22:38:55 +09:00
if ( replaceSessionId ) {
const tasks = _loadTasks ( ) ;
const task = tasks . find ( t => t . sessionId === replaceSessionId ) ;
if ( task ) {
2026-07-03 00:45:43 +00:00
task . name = _downloadNameFromPayload ( name || task . name , _payload ) ;
2026-06-02 22:38:55 +09:00
task . id = data . session _id ;
task . sessionId = data . session _id ;
task . status = 'running' ;
task . output = '' ;
task . ts = Date . now ( ) ;
task . payload = _payload ;
task . _retrying = false ;
_saveTasks ( tasks ) ;
_soloExpandTaskId = data . session _id ;
_renderRunningTab ( ) ;
_startBackgroundMonitor ( ) ;
} else {
_addTask ( data . session _id , name , 'download' , _payload ) ;
}
} else {
_addTask ( data . session _id , name , 'download' , _payload ) ;
}
2026-05-31 23:58:26 +09:00
uiModule . showToast ( ` Downloading ${ name } ... ` ) ;
} catch ( e ) {
uiModule . showToast ( 'Download failed: ' + e . message ) ;
2026-06-02 22:38:55 +09:00
if ( replaceSessionId ) _updateTask ( replaceSessionId , { status : 'crashed' , _retrying : false } ) ;
2026-05-31 23:58:26 +09:00
}
}
// ── Serve auto-fix (kill + relaunch with env var) ──
// Block stacked retries: once any "Retry with X" is clicked for a task, ignore
// every further retry click for it. Each retry fires its own _launchServeTask,
// so clicking several options — or one repeatedly during the fade-out / while a
// relaunch was loading — used to stack up multiple servers (e.g. 6 launches).
// The flag rides on the card element (removed right after), so it can't re-arm.
function _guardServeRetry ( panel , taskEl ) {
if ( ! taskEl || taskEl . dataset . retrying ) return false ;
taskEl . dataset . retrying = '1' ;
panel . querySelectorAll ( 'button' ) . forEach ( b => {
b . disabled = true ;
b . style . opacity = '0.5' ;
b . style . pointerEvents = 'none' ;
} ) ;
return true ;
}
export async function _serveAutoFix ( panel , envVar ) {
const taskEl = panel . closest ( '.cookbook-task' ) ;
if ( ! taskEl ) return ;
const taskId = taskEl . dataset . taskId ;
const tasks = _loadTasks ( ) ;
const task = tasks . find ( t => t . sessionId === taskId ) ;
if ( ! task || ! task . payload ) return ;
if ( ! _guardServeRetry ( panel , taskEl ) ) return ;
const killCmd = _tmuxCmd ( task , ` kill-session -t ${ taskId } ` ) ;
try {
await fetch ( '/api/shell/exec' , {
method : 'POST' , credentials : 'same-origin' ,
headers : { 'Content-Type' : 'application/json' } ,
body : JSON . stringify ( { command : killCmd } ) ,
} ) ;
} catch { }
_animateOutThenRemove ( taskEl , taskId ) ;
const origCmd = task . payload . _cmd || '' ;
const newCmd = ` export ${ envVar } && ${ origCmd } ` ;
const origHost = _envState . remoteHost ;
if ( task . remoteHost ) _envState . remoteHost = task . remoteHost ;
try {
uiModule . showToast ( ` Retrying with ${ envVar } ... ` ) ;
await _launchServeTask ( task . name , task . payload . repo _id , newCmd ) ;
} finally {
// Always restore — otherwise a thrown launch leaves the global host stuck
// on this serve task, so later downloads/scans hit it.
_envState . remoteHost = origHost ;
}
}
// Open the Serve panel pre-filled for a task — the same flow as the task's
// Edit button, but optionally with a modified command (used by the diagnosis
// "Retry with X" buttons so a retry lands in the editable Serve panel with the
// adjusted setting, instead of blindly relaunching).
2026-06-02 12:15:41 +09:00
async function _openServeEditForTask ( task , cmdOverride , fieldOverrides = null ) {
2026-05-31 23:58:26 +09:00
const repo = task . payload ? . repo _id ;
if ( ! repo ) { uiModule . showToast ( 'No model info on this task' ) ; return ; }
const cmd = cmdOverride || task . payload ? . _cmd ;
// A modified cmd must be re-parsed; otherwise prefer the exact launch fields.
let fields = cmdOverride
? _parseServeCmdToFields ( cmd )
: ( task . payload ? . _fields || ( cmd ? _parseServeCmdToFields ( cmd ) : null ) ) ;
2026-06-02 12:15:41 +09:00
if ( fieldOverrides && typeof fieldOverrides === 'object' ) {
fields = { ... ( fields || { } ) , ... fieldOverrides } ;
}
2026-06-22 01:49:15 +00:00
fields = { ... ( fields || { } ) , _replaceTaskId : task . sessionId } ;
// Switch the active server to the exact profile this serve ran on. The
// dropdown stores stable srv: keys, not raw host strings, so preserving only
// task.remoteHost can relaunch against the local container by accident.
_selectTaskServer ( task ) ;
2026-05-31 23:58:26 +09:00
try {
const { openServePanelForRepo } = await import ( './cookbookServe.js' ) ;
await openServePanelForRepo ( repo , fields ) ;
} catch ( err ) {
console . error ( '[cookbook] open serve panel failed' , err ) ;
uiModule . showToast ( 'Could not open serve panel' ) ;
}
}
export async function _serveAutoRetryReplace ( panel , flag , value ) {
const taskEl = panel . closest ( '.cookbook-task' ) ;
if ( ! taskEl ) return ;
const taskId = taskEl . dataset . taskId ;
const tasks = _loadTasks ( ) ;
const task = tasks . find ( t => t . sessionId === taskId ) ;
if ( ! task || ! task . payload || ! task . payload . _cmd ) return ;
if ( ! _guardServeRetry ( panel , taskEl ) ) return ;
try {
await fetch ( '/api/shell/exec' , {
method : 'POST' , credentials : 'same-origin' ,
headers : { 'Content-Type' : 'application/json' } ,
body : JSON . stringify ( { command : _tmuxCmd ( task , ` kill-session -t ${ taskId } ` ) } ) ,
} ) ;
} catch { }
_animateOutThenRemove ( taskEl , taskId ) ;
let newCmd = task . payload . _cmd ;
2026-07-07 00:50:07 +00:00
if ( flag === '--cuda-graph-backend-decode' ) {
newCmd = newCmd . replace ( /\s+--cuda-graph-max-bs-decode(?:\s+\S+|=\S+)/g , '' ) ;
} else if ( flag === '--cuda-graph-max-bs-decode' ) {
newCmd = newCmd . replace ( /\s+--cuda-graph-backend-decode(?:\s+\S+|=\S+)/g , '' ) ;
}
2026-05-31 23:58:26 +09:00
const re = new RegExp ( flag . replace ( /[.*+?^${}()|[\]\\]/g , '\\$&' ) + '\\s+\\S+' ) ;
if ( re . test ( newCmd ) ) {
newCmd = newCmd . replace ( re , ` ${ flag } ${ value } ` ) ;
} else {
newCmd += ` ${ flag } ${ value } ` ;
}
const origHost = _envState . remoteHost ;
if ( task . remoteHost ) _envState . remoteHost = task . remoteHost ;
try {
uiModule . showToast ( ` Retrying with ${ flag } ${ value } ... ` ) ;
await _launchServeTask ( task . name , task . payload . repo _id , newCmd ) ;
} finally {
_envState . remoteHost = origHost ;
}
}
export async function _serveAutoRetryRemove ( panel , flag ) {
const taskEl = panel . closest ( '.cookbook-task' ) ;
if ( ! taskEl ) return ;
const taskId = taskEl . dataset . taskId ;
const tasks = _loadTasks ( ) ;
const task = tasks . find ( t => t . sessionId === taskId ) ;
if ( ! task || ! task . payload || ! task . payload . _cmd ) return ;
if ( ! _guardServeRetry ( panel , taskEl ) ) return ;
try {
await fetch ( '/api/shell/exec' , {
method : 'POST' , credentials : 'same-origin' ,
headers : { 'Content-Type' : 'application/json' } ,
body : JSON . stringify ( { command : _tmuxCmd ( task , ` kill-session -t ${ taskId } ` ) } ) ,
} ) ;
} catch { }
_animateOutThenRemove ( taskEl , taskId ) ;
let newCmd = task . payload . _cmd ;
const re = new RegExp ( flag . replace ( /[.*+?^${}()|[\]\\]/g , '\\$&' ) + '\\s+\\S+' ) ;
newCmd = newCmd . replace ( re , '' ) . replace ( /\s{2,}/g , ' ' ) . trim ( ) ;
const origHost = _envState . remoteHost ;
if ( task . remoteHost ) _envState . remoteHost = task . remoteHost ;
try {
uiModule . showToast ( ` Retrying without ${ flag } ... ` ) ;
await _launchServeTask ( task . name , task . payload . repo _id , newCmd ) ;
} finally {
_envState . remoteHost = origHost ;
}
}
export async function _serveAutoRetry ( panel , flag ) {
const taskEl = panel . closest ( '.cookbook-task' ) ;
if ( ! taskEl ) return ;
const taskId = taskEl . dataset . taskId ;
const tasks = _loadTasks ( ) ;
const task = tasks . find ( t => t . sessionId === taskId ) ;
if ( ! task || ! task . payload || ! task . payload . _cmd ) return ;
if ( ! _guardServeRetry ( panel , taskEl ) ) return ;
try {
await fetch ( '/api/shell/exec' , {
method : 'POST' , credentials : 'same-origin' ,
headers : { 'Content-Type' : 'application/json' } ,
body : JSON . stringify ( { command : _tmuxCmd ( task , ` kill-session -t ${ taskId } ` ) } ) ,
} ) ;
} catch { }
_animateOutThenRemove ( taskEl , taskId ) ;
let newCmd = task . payload . _cmd ;
if ( ! newCmd . includes ( flag ) ) {
newCmd += ' ' + flag ;
}
const origHost = _envState . remoteHost ;
if ( task . remoteHost ) _envState . remoteHost = task . remoteHost ;
try {
uiModule . showToast ( ` Retrying with ${ flag } ... ` ) ;
await _launchServeTask ( task . name , task . payload . repo _id , newCmd ) ;
} finally {
_envState . remoteHost = origHost ;
}
}
// ── Edit-command prompt ──
// Shows a small modal with a textarea pre-filled with the current serve cmd.
// Resolves to the edited string on Save, or null on Cancel.
function _promptEditServeCmd ( currentCmd ) {
return new Promise ( ( resolve ) => {
const overlay = document . createElement ( 'div' ) ;
overlay . className = 'cookbook-edit-overlay' ;
overlay . innerHTML = `
< div class = "cookbook-edit-modal" >
< div class = "cookbook-edit-title" > Edit serve command < / d i v >
< textarea class = "cookbook-edit-textarea" spellcheck = "false" > < / t e x t a r e a >
< div class = "cookbook-edit-actions" >
< button class = "cookbook-edit-cancel memory-toolbar-btn" > Cancel < / b u t t o n >
< button class = "cookbook-edit-save memory-toolbar-btn" > Save & amp ; relaunch < / b u t t o n >
< / d i v >
< / d i v > ` ;
const ta = overlay . querySelector ( '.cookbook-edit-textarea' ) ;
ta . value = currentCmd || '' ;
document . body . appendChild ( overlay ) ;
setTimeout ( ( ) => { ta . focus ( ) ; ta . setSelectionRange ( ta . value . length , ta . value . length ) ; } , 0 ) ;
const close = ( result ) => {
overlay . remove ( ) ;
document . removeEventListener ( 'keydown' , onKey ) ;
resolve ( result ) ;
} ;
const onKey = ( e ) => {
if ( e . key === 'Escape' ) close ( null ) ;
else if ( e . key === 'Enter' && ( e . ctrlKey || e . metaKey ) ) close ( ta . value . trim ( ) || null ) ;
} ;
overlay . querySelector ( '.cookbook-edit-cancel' ) . addEventListener ( 'click' , ( ) => close ( null ) ) ;
overlay . querySelector ( '.cookbook-edit-save' ) . addEventListener ( 'click' , ( ) => close ( ta . value . trim ( ) || null ) ) ;
overlay . addEventListener ( 'click' , ( e ) => { if ( e . target === overlay ) close ( null ) ; } ) ;
document . addEventListener ( 'keydown' , onKey ) ;
} ) ;
}
// ── Launch serve task ──
// Best-effort reconstruction of serve-form field values from a raw launch
// command. Fallback for tasks created before _fields capture existed.
// Mirrors the regex parser in cookbookServe.js's _loadSlotIntoPanel.
function _parseServeCmdToFields ( cmd ) {
if ( ! cmd ) return null ;
const ex = ( re ) => { const m = cmd . match ( re ) ; return m ? m [ 1 ] : '' ; } ;
const fields = {
backend : cmd . includes ( 'llama_cpp' ) || cmd . includes ( 'llama-server' ) ? 'llamacpp'
2026-07-07 00:50:07 +00:00
: cmd . includes ( 'mlx_lm.server' ) ? 'mlx'
2026-05-31 23:58:26 +09:00
: cmd . includes ( 'diffusion_server' ) ? 'diffusers'
: cmd . includes ( 'sglang' ) ? 'sglang'
: cmd . includes ( 'ollama' ) ? 'ollama' : 'vllm' ,
port : ex ( /--port\s+(\d+)/ ) || '8000' ,
tp : ex ( /--tensor-parallel-size\s+(\d+)/ ) || '1' ,
ctx : ex ( /--max-model-len\s+(\d+)/ ) || ex ( /--n_ctx\s+(\d+)/ ) || ex ( /-c\s+(\d+)/ ) || '8192' ,
gpu _mem : ex ( /--gpu-memory-utilization\s+([\d.]+)/ ) || '0.90' ,
swap : ex ( /--swap-space\s+(\d+)/ ) || '' ,
dtype : ex ( /--dtype\s+(\w+)/ ) || 'auto' ,
2026-06-03 00:17:16 +10:00
vllm _kv _cache _dtype : ex ( /--kv-cache-dtype\s+([\w.-]+)/ ) || 'auto' ,
2026-05-31 23:58:26 +09:00
max _seqs : ex ( /--max-num-seqs\s+(\d+)/ ) || '' ,
gpus : ex ( /CUDA_VISIBLE_DEVICES=(\S+)/ ) || '' ,
2026-06-02 13:46:16 +10:00
cache _type : ex ( /(?:--cache-type-k|-ctk)\s+(\S+)/ ) || '' ,
llama _fit : ex ( /(?:--fit|-fit)\s+(on|off)/ ) || '' ,
llama _split _mode : ex ( /(?:--split-mode|-sm)\s+(none|layer|row|tensor)/ ) || '' ,
llama _tensor _split : ex ( /(?:--tensor-split|-ts)\s+([0-9.,]+)/ ) || '' ,
llama _main _gpu : ex ( /(?:--main-gpu|-mg)\s+(\d+)/ ) || '' ,
llama _parallel : ex ( /(?:--parallel|-np)\s+(\d+)/ ) || '' ,
llama _batch _size : ex ( /(?:--batch-size|-b)\s+(\d+)/ ) || '' ,
llama _ubatch _size : ex ( /(?:--ubatch-size|-ub)\s+(\d+)/ ) || '' ,
llama _spec _tokens : ex ( /--spec-draft-n-max\s+(\d+)/ ) || '3' ,
2026-05-31 23:58:26 +09:00
enforce _eager : cmd . includes ( '--enforce-eager' ) ,
trust _remote : cmd . includes ( '--trust-remote-code' ) ,
prefix _cache : cmd . includes ( '--enable-prefix-caching' ) ,
auto _tool : cmd . includes ( '--enable-auto-tool-choice' ) ,
2026-06-02 13:46:16 +10:00
flash _attn : /--flash-attn\s+on\b/ . test ( cmd ) ,
unified _mem : /GGML_CUDA_ENABLE_UNIFIED_MEMORY=1/ . test ( cmd ) ,
llama _no _mmap : /--no-mmap\b/ . test ( cmd ) ,
llama _no _warmup : /--no-warmup\b/ . test ( cmd ) ,
llama _speculative _mtp : /--spec-type\s+\S*draft-mtp/ . test ( cmd ) ,
2026-05-31 23:58:26 +09:00
speculative : cmd . includes ( '--speculative-config' ) ,
} ;
const spec = cmd . match ( /--speculative-config\s+'?\{[^}]*"method"\s*:\s*"([^"]+)"[^}]*"num_speculative_tokens"\s*:\s*(\d+)/ ) ;
if ( spec ) { fields . spec _method = spec [ 1 ] ; fields . spec _tokens = spec [ 2 ] ; }
return fields ;
}
2026-07-07 00:50:07 +00:00
function _serveCmdNeedsGpuPreflight ( cmd , repo ) {
const c = String ( cmd || '' ) . toLowerCase ( ) ;
const r = String ( repo || '' ) . toLowerCase ( ) ;
if ( ! c || /gpu-cleanup|sglang-kernel|mlx-lm|pip\s+install|python\d*\s+-m\s+pip/ . test ( ` ${ r } ${ c } ` ) ) return false ;
return /\b(vllm\s+serve|sglang(?:\.launch_server|\s+serve)|mlx_lm\.server|llama-server|llama_cpp\.server|text-generation-launcher|aphrodite|ollama\s+(?:serve|run))\b/ . test ( c ) ;
}
function _selectedGpuIndexes ( gpus ) {
const raw = String ( gpus || '' ) . trim ( ) ;
if ( ! raw ) return null ;
const out = new Set ( ) ;
raw . split ( ',' ) . forEach ( part => {
const p = part . trim ( ) ;
const range = p . match ( /^(\d+)\s*-\s*(\d+)$/ ) ;
if ( range ) {
const a = parseInt ( range [ 1 ] , 10 ) ;
const b = parseInt ( range [ 2 ] , 10 ) ;
for ( let i = Math . min ( a , b ) ; i <= Math . max ( a , b ) ; i ++ ) out . add ( i ) ;
return ;
}
const n = parseInt ( p , 10 ) ;
if ( Number . isFinite ( n ) ) out . add ( n ) ;
} ) ;
return out . size ? out : null ;
}
function _gbFromMb ( mb ) {
const n = Number ( mb || 0 ) ;
if ( ! Number . isFinite ( n ) || n <= 0 ) return '' ;
return n >= 1024 ? ` ${ ( n / 1024 ) . toFixed ( n >= 10240 ? 0 : 1 ) } G ` : ` ${ Math . round ( n ) } M ` ;
}
function _gpuPreflightIssues ( data , selected ) {
const backend = String ( data ? . backend || data ? . source || '' ) . toLowerCase ( ) ;
const isCuda = backend . includes ( 'cuda' ) || String ( data ? . source || '' ) . toLowerCase ( ) . includes ( 'nvidia' ) ;
const rows = Array . isArray ( data ? . gpus ) ? data . gpus : [ ] ;
const issues = [ ] ;
rows . forEach ( g => {
const idx = Number ( g ? . index ) ;
if ( selected && ! selected . has ( idx ) ) return ;
const procs = Array . isArray ( g ? . processes ) ? g . processes : [ ] ;
if ( procs . length ) {
procs . slice ( 0 , 3 ) . forEach ( p => {
const name = String ( p ? . name || 'process' ) . split ( /[\\/]/ ) . pop ( ) ;
const used = _gbFromMb ( p ? . used _mb ) ;
issues . push ( ` GPU ${ idx } : ${ name } ${ p ? . pid ? ` # ${ p . pid } ` : '' } ${ used ? ` ( ${ used } ) ` : '' } ` ) ;
} ) ;
if ( procs . length > 3 ) issues . push ( ` GPU ${ idx } : + ${ procs . length - 3 } more process ${ procs . length - 3 === 1 ? '' : 'es' } ` ) ;
return ;
}
const total = Number ( g ? . total _mb || 0 ) ;
const free = Number ( g ? . free _mb || 0 ) ;
const used = Number ( g ? . used _mb || 0 ) ;
const freeRatio = total > 0 ? free / total : 1 ;
// CUDA can have display/runtime crumbs; warn only for meaningful occupied memory.
if ( isCuda && used > 4096 && freeRatio < 0.9 ) {
issues . push ( ` GPU ${ idx } : ${ _gbFromMb ( used ) } already used ( ${ _gbFromMb ( free ) } free) ` ) ;
} else if ( ! isCuda && total > 0 && freeRatio < 0.2 ) {
issues . push ( ` ${ g ? . name || ` GPU ${ idx } ` } : low free memory ( ${ _gbFromMb ( free ) } free of ${ _gbFromMb ( total ) } ) ` ) ;
} else if ( ! isCuda && g ? . busy && total <= 0 ) {
issues . push ( ` ${ g ? . name || ` GPU ${ idx } ` } : GPU device is busy ` ) ;
}
} ) ;
return issues ;
}
async function _confirmGpuPreflight ( reqBody , shortName , repo , cmd ) {
if ( ! _serveCmdNeedsGpuPreflight ( cmd , repo ) ) return true ;
const params = new URLSearchParams ( ) ;
if ( reqBody . remote _host ) params . set ( 'host' , reqBody . remote _host ) ;
if ( reqBody . ssh _port ) params . set ( 'ssh_port' , reqBody . ssh _port ) ;
try {
const res = await fetch ( ` /api/cookbook/gpus ${ params . toString ( ) ? ` ? ${ params . toString ( ) } ` : '' } ` , {
method : 'GET' ,
credentials : 'same-origin' ,
} ) ;
const data = await res . json ( ) . catch ( ( ) => null ) ;
if ( ! res . ok || ! data ? . ok ) return true ;
const selected = _selectedGpuIndexes ( reqBody . gpus ) ;
const issues = _gpuPreflightIssues ( data , selected ) ;
if ( ! issues . length ) return true ;
const where = reqBody . remote _host || 'local' ;
const list = issues . slice ( 0 , 6 ) . join ( '; ' ) ;
const more = issues . length > 6 ? ` ; + ${ issues . length - 6 } more ` : '' ;
const msg = ` GPU preflight found existing load on ${ where } : ${ list } ${ more } . Launch ${ shortName || 'model' } anyway? ` ;
const confirm = window . styledConfirm || uiModule ? . styledConfirm ;
if ( confirm ) return await confirm ( msg , { confirmText : 'Launch anyway' , cancelText : 'Cancel' } ) ;
return window . confirm ? window . confirm ( msg ) : true ;
} catch ( e ) {
console . warn ( '[cookbook] GPU preflight failed; allowing launch' , e ) ;
return true ;
}
}
2026-06-21 11:02:35 +00:00
export async function _launchServeTask ( shortName , repo , cmd , fields , hostOverride , targetMeta = null ) {
2026-05-31 23:58:26 +09:00
// Host resolution mirrors the download path: when the caller passes an explicit
// host (resolved from the dropdown the user actually picked), use it and look
// up that server's port/platform from the shared servers list. Only fall back
// to _envState.remoteHost for legacy callers (diagnosis/pip-update).
const _host = ( hostOverride !== undefined ) ? ( hostOverride || '' ) : ( _envState . remoteHost || '' ) ;
2026-06-21 11:02:35 +00:00
const _targetKey = targetMeta ? . serverKey || '' ;
const _hsrv = ( _targetKey && _targetKey !== 'local' ? _serverByVal ( _targetKey ) : null )
|| ( hostOverride === undefined ? _serverByVal ( _envState . remoteServerKey || _host ) : null )
2026-06-08 18:36:10 -04:00
|| _envState . servers . find ( s => s . host === _host ) || { } ;
2026-06-21 11:02:35 +00:00
const _serverMetaKey = _targetKey || ( _hsrv && _serverKey ? _serverKey ( _hsrv ) : '' ) || ( _host || 'local' ) ;
const _serverMetaName = targetMeta ? . serverName || _hsrv . name || ( _host ? _host : 'Local' ) ;
2026-06-26 08:13:01 -04:00
const _hplatform = _host ? ( _hsrv . platform || '' ) : ( _envState . hostPlatform || '' ) ;
2026-06-22 01:49:15 +00:00
const _replaceTaskId = fields ? . _replaceTaskId || '' ;
if ( _replaceTaskId ) {
try {
const _old = _loadTasks ( ) . find ( t => t . sessionId === _replaceTaskId ) ;
if ( _old && _old . type === 'serve' ) {
await fetch ( '/api/shell/exec' , {
method : 'POST' , credentials : 'same-origin' ,
headers : { 'Content-Type' : 'application/json' } ,
body : JSON . stringify ( { command : _tmuxGracefulKill ( _old ) } ) ,
} ) ;
_removeTask ( _old . sessionId ) ;
}
} catch { }
}
2026-05-31 23:58:26 +09:00
// Replace any serve already targeting this same host:port — you can't run two
// servers on one port, so re-serving (or retrying) should stop & remove the
// old one instead of leaving a dead duplicate behind. (The retry buttons
// already removed their own task, so this is a no-op for them.)
try {
const _pm = cmd . match ( /--port[=\s]+(\d+)/ ) || cmd . match ( /(?:^|\s)-p[=\s]+(\d+)/ ) ;
const _newPort = _pm ? _pm [ 1 ] : '' ;
if ( _newPort ) {
for ( const _t of _loadTasks ( ) ) {
if ( _t . type !== 'serve' || ! _t . payload || ! _t . payload . _cmd ) continue ;
const _tm = _t . payload . _cmd . match ( /--port[=\s]+(\d+)/ ) || _t . payload . _cmd . match ( /(?:^|\s)-p[=\s]+(\d+)/ ) ;
if ( ( _tm ? _tm [ 1 ] : '' ) === _newPort && ( _t . remoteHost || '' ) === _host ) {
try {
await fetch ( '/api/shell/exec' , {
method : 'POST' , credentials : 'same-origin' ,
headers : { 'Content-Type' : 'application/json' } ,
body : JSON . stringify ( { command : _tmuxGracefulKill ( _t ) } ) ,
} ) ;
} catch { }
_removeTask ( _t . sessionId ) ;
}
}
}
} catch { }
// Capture the env + GPU pin used for THIS launch BEFORE building the request.
// The serve panel sets _envState.env/envPath/gpus, calls us, then restores them
// synchronously — and our payload is built after an `await`, so reading
// _envState there would see the restored (wrong) values. Persisting these lets
// a saved preset relaunch with the same venv + GPUs (otherwise a confirmed
// working config fails: no venv activation, no GPU pinning).
const _usedEnv = _envState . env ;
const _usedEnvPath = _envState . envPath ;
const _usedGpus = _envState . gpus || '' ;
let envPrefix = '' ;
if ( _isWindows ( ) ) {
if ( _envState . env === 'venv' && _envState . envPath ) {
envPrefix = '& ' + ( _envState . envPath . endsWith ( '\\Scripts\\Activate.ps1' ) ? _envState . envPath : _envState . envPath + '\\Scripts\\Activate.ps1' ) ;
} else if ( _envState . env === 'conda' && _envState . envPath ) {
envPrefix = 'conda activate ' + _envState . envPath ;
}
} else {
if ( _envState . env === 'venv' && _envState . envPath ) {
2026-06-21 11:02:35 +00:00
const p = _venvRootFromPath ( _envState . envPath ) ;
2026-05-31 23:58:26 +09:00
envPrefix = 'source ' + ( p . endsWith ( '/bin/activate' ) ? p : p + '/bin/activate' ) ;
} else if ( _envState . env === 'conda' && _envState . envPath ) {
envPrefix = 'eval "$(conda shell.bash hook)" && conda activate ' + _envState . envPath ;
}
}
const reqBody = {
repo _id : repo ,
cmd : cmd ,
remote _host : _host || undefined ,
2026-06-21 11:02:35 +00:00
ssh _port : _getPort ( _serverMetaKey || _host ) || undefined ,
2026-05-31 23:58:26 +09:00
env _prefix : envPrefix || undefined ,
hf _token : _envState . hfToken || undefined ,
2026-07-07 00:50:07 +00:00
gpus : _usedGpus || undefined ,
2026-05-31 23:58:26 +09:00
platform : _hplatform || undefined ,
} ;
try {
2026-07-07 00:50:07 +00:00
const _preflightOk = await _confirmGpuPreflight ( reqBody , shortName , repo , cmd ) ;
if ( ! _preflightOk ) {
uiModule . showToast ( 'Launch cancelled — GPU is already in use' ) ;
return ;
}
2026-05-31 23:58:26 +09:00
const res = await fetch ( '/api/model/serve' , {
method : 'POST' , credentials : 'same-origin' ,
headers : { 'Content-Type' : 'application/json' } ,
body : JSON . stringify ( reqBody ) ,
} ) ;
const data = await res . json ( ) ;
if ( ! data . ok ) {
// Two error shapes: `{ok:false, error}` (tmux launch failed) or
// `{detail}` (FastAPI HTTPException). Show whichever is present
// + log full payload so the user can copy the error.
const err = data . error || data . detail || res . statusText || 'unknown' ;
console . error ( '[cookbook] /api/model/serve failed' , { status : res . status , body : data } ) ;
uiModule . showToast ( 'Failed to start: ' + String ( err ) . slice ( 0 , 200 ) , 9000 ) ;
return ;
}
2026-06-21 11:02:35 +00:00
const _sp = _getPort ( _serverMetaKey || _host ) ;
2026-05-31 23:58:26 +09:00
// _fields = the exact structured serve-form values used for this launch,
// so the "Edit / relaunch" button can re-open the Serve panel pre-filled
// with these precise settings (not just the last-used-for-repo state).
2026-06-21 11:02:35 +00:00
const payload = { repo _id : repo , remote _host : _host || undefined , remote _server _key : _serverMetaKey || undefined , remote _server _name : _serverMetaName || undefined , ssh _port : _sp || undefined , _cmd : cmd , _fields : fields || undefined , _env : _usedEnv , _envPath : _usedEnvPath , _gpus : _usedGpus } ;
2026-05-31 23:58:26 +09:00
_addTask ( data . session _id , shortName , 'serve' , payload ) ;
uiModule . showToast ( ` Serving ${ shortName } ... ` ) ;
Cookbook UI: Ollama browser, advanced serve fold, API tokens form, diagnosis toolbar, polish
Surface a lot of accumulated cookbook + UI work as a single non-agent
commit so the agent rework lands cleanly.
Highlights:
- Ollama as a first-class backend in the Cookbook:
* Download input accepts ollama-style names (name:tag) → backend=ollama
* /api/cookbook/ollama/library (cached scrape of ollama.com + curated
fallback so classic models like qwen2.5 stay reachable)
* "Browse Ollama library" toggle below Download with size chips
* Engine=Ollama in hwfit toolbar merges the Ollama library into the
main scan list as per-tag rows with the same Fit/Param/Quant/VRAM
columns; click → fills Download input
- API Tokens form added to Integrations panel (matching wired
loadTokens()/initTokenForm() that had no HTML)
- Serve panel polish: Advanced fold tightening (-8px nudges on vLLM
checks, Extra args, Spec row), n_cpu_moe + Split Mode controls
pulled up 8px to align with the row's checkboxes, GGUF File dropdown
exposed for Ollama backend, GPU re-render on Edit serve restore,
_forceBackend flag so saved serveState wins over backend detection,
cookbook:servers-changed CustomEvent so panels don't need refresh
- Models page redesign: Add Models row (URL + hidden API key reveal +
Type select + Scan/Ollama/Key/Test/Add icon buttons), Probe All +
Clear-offline buttons in Added Models toolbar, offline-pill removed
(opacity already conveys state), Engine dropdown gains Ollama option
- _ping_endpoint probes /v1/models then base, accepts 4xx as
reachable (vLLM returns 404 on bare /v1, fully working endpoints
were showing offline)
- Diagnosis card: × dismiss + Copy bundle buttons restored on the
serve error feedback card
- Orphan tmux sweep re-enabled behind a 60s rate-limit + background
Thread (off the main event loop) so dead serves get discovered
- cookbook_routes auto-register watchdog: drops the endpoint if the
serve session exits non-zero within the first ~3min
- ollama-rocm sidecar awareness in download wrapper (`docker exec
ollama-rocm ollama pull` when host ollama isn't installed)
- Skill extractor sets initial_status="published" when
auto_approve_skills pref is on (audit demotes later)
- Skill list / model list / cookbook scan misc polish
2026-06-08 22:38:49 +09:00
// Auto-register may have enabled an existing (offline) endpoint for this
// host:port. Refresh the picker so the row is no longer dimmed, and the
// user doesn't see "offline" on a serve they just started.
try { _refreshModelsAfterEndpointChange ( ) ; } catch ( _ ) { }
2026-05-31 23:58:26 +09:00
} catch ( e ) {
uiModule . showToast ( 'Failed: ' + e . message ) ;
}
}
// ── Render Running tab ──
export function _renderRunningTab ( ) {
// Auto-clear the sidebar notif (the bright-icon highlight) when no tasks
// are actively running or errored. _showCookbookNotif fires on each task
// event but the matching clear only ran on modal-open, so the highlight
// persisted indefinitely after tasks finished in the background.
try {
2026-06-02 22:38:55 +09:00
const _activeTasks = _loadPrunedTasks ( ) . filter ( t => t . status === 'running' || t . status === 'queued' || t . status === 'error' ) ;
2026-05-31 23:58:26 +09:00
if ( ! _activeTasks . length ) _clearCookbookNotif ( ) ;
} catch { }
const body = document . querySelector ( '#cookbook-modal .cookbook-body' ) ;
if ( ! body ) return ;
// Capture expansion state so re-renders don't collapse whatever the user
// had open. Task output: presence of .cookbook-task-collapsed means collapsed.
// Section body: inline display:none means collapsed.
const _collapsedTaskIds = new Set ( ) ;
const _expandedTaskIds = new Set ( ) ; // mobile: tasks the user explicitly opened
body . querySelectorAll ( '.cookbook-task' ) . forEach ( tEl => {
const id = tEl . dataset . taskId ;
if ( ! id ) return ;
const wrap = tEl . querySelector ( '.cookbook-output-wrap' ) ;
if ( ! wrap ) return ;
if ( wrap . classList . contains ( 'cookbook-task-collapsed' ) ) _collapsedTaskIds . add ( id ) ;
else _expandedTaskIds . add ( id ) ;
} ) ;
// A new action was just started — collapse every existing card and open only
// the new one (works on both desktop and the mobile collapse-by-default path).
if ( _soloExpandTaskId ) {
const _allIds = new Set ( [ ... _collapsedTaskIds , ... _expandedTaskIds ] ) ;
_collapsedTaskIds . clear ( ) ;
_expandedTaskIds . clear ( ) ;
_allIds . forEach ( id => { if ( id !== _soloExpandTaskId ) _collapsedTaskIds . add ( id ) ; } ) ;
_expandedTaskIds . add ( _soloExpandTaskId ) ;
_soloExpandTaskId = null ;
}
// On mobile, task outputs start COLLAPSED — having every running window
// expanded on entry meant a lot of tapping to collapse them. User-expanded
// ones are re-opened from _expandedTaskIds below.
const _mobileCollapseDefault = window . innerWidth <= 768 ;
const _collapsedSectionIds = new Set ( ) ;
body . querySelectorAll ( '.cookbook-section-body' ) . forEach ( sb => {
if ( sb . style . display === 'none' && sb . id ) _collapsedSectionIds . add ( sb . id ) ;
} ) ;
const tasks = _loadTasks ( ) ;
const hasContent = tasks . length > 0 ;
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
// Count anything that's really active: explicit 'running'/'queued' status,
// OR a download whose tmux output is still showing live shard progress.
// Without the output check, a task whose status got stuck at 'done' /
// 'crashed' (before auto-reconnect catches it) would read as "Running 0"
// even when the model is actively downloading on the host.
const activeCount = tasks . filter ( t =>
t . status === 'running'
|| t . status === 'queued'
|| _downloadOutputLooksActive ( t )
) . length ;
2026-06-01 22:59:29 -05:00
const activeCountHtml = activeCount ? ` <span class="cookbook-tab-count"> ${ activeCount } </span> ` : '' ;
2026-05-31 23:58:26 +09:00
let tabBar = body . querySelector ( '.cookbook-tabs' ) ;
if ( ! tabBar ) return ;
let runTab = tabBar . querySelector ( '.cookbook-tab[data-backend="Running"]' ) ;
if ( hasContent && ! runTab ) {
runTab = document . createElement ( 'button' ) ;
runTab . className = 'cookbook-tab' ;
runTab . dataset . backend = 'Running' ;
const _errCount = tasks . filter ( t => t . status === 'error' || t . status === 'crashed' ) . length ;
2026-06-13 22:11:45 +09:00
runTab . innerHTML = ` Active ${ activeCountHtml } ${ _errCount ? ` <span class="cookbook-tab-error-dot"></span> ` : '' } ` ;
2026-05-31 23:58:26 +09:00
tabBar . insertBefore ( runTab , tabBar . firstChild ) ;
runTab . addEventListener ( 'click' , ( ) => {
tabBar . querySelectorAll ( '.cookbook-tab' ) . forEach ( t => t . classList . remove ( 'active' ) ) ;
runTab . classList . add ( 'active' ) ;
body . querySelectorAll ( '.cookbook-group' ) . forEach ( g => {
g . classList . toggle ( 'hidden' , g . dataset . backendGroup !== 'Running' ) ;
} ) ;
2026-06-27 13:50:21 +00:00
setTimeout ( ( ) => _renderRunningTab ( ) , 0 ) ;
2026-05-31 23:58:26 +09:00
} ) ;
} else if ( runTab ) {
const _errCount2 = tasks . filter ( t => t . status === 'error' || t . status === 'crashed' ) . length ;
2026-06-13 22:11:45 +09:00
runTab . innerHTML = tasks . length ? ` Active ${ activeCountHtml } ${ _errCount2 ? '<span class="cookbook-tab-error-dot"></span>' : '' } ` : 'Active' ;
2026-05-31 23:58:26 +09:00
if ( ! hasContent ) {
if ( runTab . classList . contains ( 'active' ) ) {
const wfTab = tabBar . querySelector ( '.cookbook-tab[data-backend="Search"]' ) ;
if ( wfTab ) wfTab . click ( ) ;
}
runTab . remove ( ) ;
}
}
let group = body . querySelector ( '.cookbook-group[data-backend-group="Running"]' ) ;
if ( hasContent && ! group ) {
group = document . createElement ( 'div' ) ;
group . className = 'cookbook-group hidden' ;
group . dataset . backendGroup = 'Running' ;
Open email context for agent, email search across All Mail, cookbook serve polish
- Agent: pass the open email reader (uid/folder/account/from/subject/body
preview) on every chat submit so 'reply to this' / 'write email saying
hi' route to ui_control open_email_reply with the right UID instead of
inventing a new .md draft. Code-level enforcement (chat_routes strips
create_document + send_email when active_email is set); cross-session
active_doc_id is now trusted instead of being silently dropped.
set_active_email/clear_active_email tool-layer helpers in
tool_implementations.
- ui_control open_email_reply: optional body argument so the agent can
open-and-write in one call; envelope now forwards uid/folder/account/
body/panel through tool_output. Tool description sharpened and the
parser rejects empty bodies on reply/reply-all (forces the agent to
write rather than open an empty draft).
- Email library: search now runs against [Gmail]/All Mail when the
current folder is INBOX (archived emails surface). Whirlpool spinner
+ 'Searching…' placeholder while in flight. Each search result is
stamped with its source folder so clicks open the right email instead
of whatever shares its UID in INBOX. Search no longer re-applies the
same text pill locally (which only checks subject/from/snippet, never
body) so body-only matches don't get dropped after IMAP returns them.
Initial inbox load bumped 100→500.
- Email favorites: 'Favorite (pin to top)' / 'Unfavorite' in both the
card menu and the open-reader more menu, backed by a new
/api/email/flag/{uid}?on=true|false endpoint. Flagged emails always
bubble to the top of the grid regardless of active sort.
- AI reply in doc editor: never overwrites existing draft text or the
quoted history. AI suggestion is prepended; AI-generated 'On …
wrote:' re-quotes are stripped so the original quote isn't visually
edited.
- Cookbook serve: pre-launch GPU driver / has_gpu / install / version-
floor checks (vllm minimax_m2 needs 0.10.0+, deepseek_r1 needs 0.7.0
etc.) before the launch chain starts. Detect 'another model already
running on this host' and offer Stop & launch (with graceful then
force tmux kill helpers, port release wait). Per-vendor deep-link
buttons (vLLM recipe / SGLang cookbook) with hardware hash. Backend
picker is now a custom dropdown with accent-coloured logos for vLLM,
SGLang, llama.cpp, Ollama, Diffusers; same glyphs added next to
package names in Dependencies. Runtime-readiness note moved inside
the panel (green when ready, red when missing) with an × dismiss.
Esc collapses the expanded card; expanded card scrolls when it
overflows; Trust Remote / Auto Tool / Reasoning Parser / Enforce
Eager / Prefix Caching / Expert Parallel / Speculative / MoE Env on
one row (Reasoning Parser auto-detected per model family).
Dtype→Row 1, GPUs→Row 2 (rightmost). Removed redundant GPU 'auto'
input — command builders read from the GPU button strip. Default
cookbook open is Download tab.
- Cookbook hwfit: 'Model (latest)' / 'Model (oldest)' header sorts by
release_date; release dates can be backfilled with the new
scripts/backfill_model_release_dates.py and recipe metadata pulled
with scripts/import_from_vllm_recipes.py against the upstream
vllm-project/recipes catalog (vllm_recipe + min_vllm_version stamped
on entries).
- Calendar: Quick add hint cycles a random Odysseus-themed example per
open (wooden horse Friday, crew muster 10am daily, council on
Ithaca, …). Typing a time like '11pm' in the event title updates
the hero clock live.
- Doc editor: email-mode Reply button (sparkle icon, accent) opens the
same Fast/Full + context popover the email reader uses; Ctrl+Alt+M
toggles markdown preview.
- Memories panel: custom sort picker with per-option icons, default
'Latest', visible Enabled/Disabled toggle text matching the section
description style.
2026-06-15 20:47:51 +09:00
// No `flex:1` on the card — with overflow:visible (forced via #cookbook-modal
// .cookbook-group > .admin-card), flex:1 collapsed the card to body height
// and the body's scrollHeight stopped tracking the overflowing children.
// Sized-to-content means cookbook-body's overflow-y:auto kicks in naturally.
group . innerHTML = '<div class="admin-card" style="display:flex;flex-direction:column;">' +
2026-05-31 23:58:26 +09:00
'<div style="display:flex;align-items:baseline;gap:8px;margin-bottom:2px;">' +
2026-06-13 22:11:45 +09:00
'<h2 style="margin:0;padding:0;line-height:1;">Active <span id="running-count" class="memory-count" style="font-size:0.6em;opacity:0.6;font-weight:normal">' + activeCount + '</span></h2>' +
2026-05-31 23:58:26 +09:00
'</div>' +
2026-06-22 01:49:15 +00:00
'<p class="memory-desc doclib-desc" style="margin-top:6px;">Active downloads, installs and model launches.</p>' +
2026-05-31 23:58:26 +09:00
'</div>' ;
const firstGroup = body . querySelector ( '.cookbook-group' ) ;
if ( firstGroup ) body . insertBefore ( group , firstGroup ) ;
else body . appendChild ( group ) ;
}
if ( ! group ) return ;
const countEl = group . querySelector ( '#running-count' ) ;
2026-06-01 22:59:29 -05:00
if ( countEl ) countEl . textContent = activeCount ;
2026-05-31 23:58:26 +09:00
if ( ! hasContent ) {
group . remove ( ) ;
return ;
}
const _adminCard = group . querySelector ( '.admin-card' ) ;
function _ensureSection ( cls , label , items ) {
let sec = group . querySelector ( '.' + cls ) ;
if ( ! sec ) {
sec = document . createElement ( 'div' ) ;
sec . className = cls ;
( _adminCard || group ) . appendChild ( sec ) ;
}
if ( ! items || ! items . length ) {
sec . style . display = 'none' ;
return sec ;
}
sec . style . display = '' ;
return sec ;
}
// Group tasks by server
2026-06-21 11:02:35 +00:00
const _taskServerKey = ( task ) => task ? . remoteServerKey || task ? . remoteHost || '' ;
const _serverName = ( keyOrTask ) => {
if ( keyOrTask && typeof keyOrTask === 'object' ) {
const task = keyOrTask ;
if ( task . remoteServerName ) return task . remoteServerName ;
const srv = task . remoteServerKey ? _serverByVal ( task . remoteServerKey ) : null ;
if ( srv ? . name ) return srv . name ;
if ( ! task . remoteHost ) return 'Local' ;
return ( _envState . servers . find ( s => s . host === task . remoteHost ) ? . name ) || task . remoteHost ;
}
const key = keyOrTask || '' ;
if ( ! key || key === 'local' ) return 'Local' ;
const srv = _serverByVal ( key ) ;
return srv ? . name || key ;
2026-05-31 23:58:26 +09:00
} ;
const serverGroups = { } ;
for ( const t of tasks ) {
2026-06-21 11:02:35 +00:00
const key = _taskServerKey ( t ) ;
if ( ! serverGroups [ key ] ) serverGroups [ key ] = { name : _serverName ( t ) , serve : [ ] , download : [ ] } ;
2026-05-31 23:58:26 +09:00
serverGroups [ key ] [ t . type === 'serve' ? 'serve' : 'download' ] . push ( t ) ;
}
// ── Server-grouped sections ──
group . querySelectorAll ( '.cookbook-serve-section, .cookbook-dl-section' ) . forEach ( el => el . remove ( ) ) ;
const serverKeys = Object . keys ( serverGroups ) . sort ( ( a , b ) => {
if ( ! a ) return - 1 ; if ( ! b ) return 1 ;
return serverGroups [ a ] . name . localeCompare ( serverGroups [ b ] . name ) ;
} ) ;
// Prune stale server sections: a server that no longer has ANY tasks isn't in
// serverKeys, so its section header/dropdown would otherwise linger until the
// user manually cleared it. Drop those automatically on each render.
const _liveSafeKeys = new Set ( serverKeys . map ( k => ( k || 'local' ) . replace ( /[^a-zA-Z0-9-]/g , '_' ) ) ) ;
( _adminCard || group ) . querySelectorAll ( '[class*="cookbook-server-section-"]' ) . forEach ( el => {
const cls = [ ... el . classList ] . find ( c => c . startsWith ( 'cookbook-server-section-' ) ) ;
if ( cls && ! _liveSafeKeys . has ( cls . replace ( 'cookbook-server-section-' , '' ) ) ) el . remove ( ) ;
} ) ;
for ( const key of serverKeys ) {
const sg = serverGroups [ key ] ;
const allTasks = [ ... sg . serve , ... sg . download ] ;
const safeKey = ( key || 'local' ) . replace ( /[^a-zA-Z0-9-]/g , '_' ) ;
const sectionCls = ` cookbook-server-section- ${ safeKey } ` ;
const bodyId = ` server-body- ${ safeKey } ` ;
let sec = _ensureSection ( sectionCls , sg . name , allTasks ) ;
if ( allTasks . length && ! sec . querySelector ( '.cookbook-section-header' ) ) {
const clearId = ` clear-server- ${ key || 'local' } ` ;
// Glowy status dot next to the server name (like the Settings server card):
// green when reachable, red if any serve task on it is crashed/unreachable.
const _secDot = ( key && allTasks . some ( _serveTaskFailed ) ) ? 'fail' : 'ok' ;
const _dotTitle = key ? ( _secDot === 'fail' ? 'Server not responding' : 'Reachable' ) : 'Local (this machine)' ;
2026-07-07 00:50:07 +00:00
const _srvColor = _serverColorForTaskGroup ( key || 'local' , allTasks ) ;
sec . insertAdjacentHTML ( 'afterbegin' , ` <div class="cookbook-section-header ${ _srvColor ? ' has-server-color' : '' } " data-collapse=" ${ bodyId } " ${ _serverHeaderStyle ( _srvColor ) } ><svg class="cookbook-section-chevron" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2.5" stroke-linecap="round"><polyline points="6 9 12 15 18 9"/></svg><span class="cookbook-srv-status ${ _secDot } " title=" ${ _dotTitle } " style="flex-shrink:0;position:relative;top:0px;"></span><span class="cookbook-section-title" style="margin:0;"> ${ esc ( sg . name ) } </span><button class="cookbook-btn cookbook-stop-all-btn" data-stop-server=" ${ esc ( key ) } " title="Stop all running servers"><svg width="11" height="11" viewBox="0 0 24 24" fill="currentColor" stroke="none" aria-hidden="true" style="vertical-align:-1px;margin-right:4px;"><rect x="5" y="5" width="14" height="14" rx="1.5"/></svg>Stop all</button><button class="cookbook-btn cookbook-clear-btn" data-clear-server=" ${ esc ( key ) } " title="Clear finished tasks"><svg width="11" height="11" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" style="vertical-align:-1px;margin-right:4px;"><polyline points="3 6 5 6 21 6"/><path d="M19 6l-1 14a2 2 0 0 1-2 2H8a2 2 0 0 1-2-2L5 6"/><path d="M10 11v6"/><path d="M14 11v6"/></svg>Clear finished</button></div><div id=" ${ bodyId } " class="cookbook-section-body"></div> ` ) ;
2026-05-31 23:58:26 +09:00
}
}
// Wire clear all buttons
group . querySelectorAll ( '[data-clear-server]' ) . forEach ( btn => {
if ( btn . _bound ) return ;
btn . _bound = true ;
btn . addEventListener ( 'click' , async ( e ) => {
e . stopPropagation ( ) ; // don't toggle the section collapse (was an inline onclick, blocked by CSP)
const host = btn . dataset . clearServer ;
const allTasks = _loadTasks ( ) ;
2026-06-21 11:02:35 +00:00
const toRemove = allTasks . filter ( t => _taskServerKey ( t ) === host && _canClearTask ( t ) ) ;
Open email context for agent, email search across All Mail, cookbook serve polish
- Agent: pass the open email reader (uid/folder/account/from/subject/body
preview) on every chat submit so 'reply to this' / 'write email saying
hi' route to ui_control open_email_reply with the right UID instead of
inventing a new .md draft. Code-level enforcement (chat_routes strips
create_document + send_email when active_email is set); cross-session
active_doc_id is now trusted instead of being silently dropped.
set_active_email/clear_active_email tool-layer helpers in
tool_implementations.
- ui_control open_email_reply: optional body argument so the agent can
open-and-write in one call; envelope now forwards uid/folder/account/
body/panel through tool_output. Tool description sharpened and the
parser rejects empty bodies on reply/reply-all (forces the agent to
write rather than open an empty draft).
- Email library: search now runs against [Gmail]/All Mail when the
current folder is INBOX (archived emails surface). Whirlpool spinner
+ 'Searching…' placeholder while in flight. Each search result is
stamped with its source folder so clicks open the right email instead
of whatever shares its UID in INBOX. Search no longer re-applies the
same text pill locally (which only checks subject/from/snippet, never
body) so body-only matches don't get dropped after IMAP returns them.
Initial inbox load bumped 100→500.
- Email favorites: 'Favorite (pin to top)' / 'Unfavorite' in both the
card menu and the open-reader more menu, backed by a new
/api/email/flag/{uid}?on=true|false endpoint. Flagged emails always
bubble to the top of the grid regardless of active sort.
- AI reply in doc editor: never overwrites existing draft text or the
quoted history. AI suggestion is prepended; AI-generated 'On …
wrote:' re-quotes are stripped so the original quote isn't visually
edited.
- Cookbook serve: pre-launch GPU driver / has_gpu / install / version-
floor checks (vllm minimax_m2 needs 0.10.0+, deepseek_r1 needs 0.7.0
etc.) before the launch chain starts. Detect 'another model already
running on this host' and offer Stop & launch (with graceful then
force tmux kill helpers, port release wait). Per-vendor deep-link
buttons (vLLM recipe / SGLang cookbook) with hardware hash. Backend
picker is now a custom dropdown with accent-coloured logos for vLLM,
SGLang, llama.cpp, Ollama, Diffusers; same glyphs added next to
package names in Dependencies. Runtime-readiness note moved inside
the panel (green when ready, red when missing) with an × dismiss.
Esc collapses the expanded card; expanded card scrolls when it
overflows; Trust Remote / Auto Tool / Reasoning Parser / Enforce
Eager / Prefix Caching / Expert Parallel / Speculative / MoE Env on
one row (Reasoning Parser auto-detected per model family).
Dtype→Row 1, GPUs→Row 2 (rightmost). Removed redundant GPU 'auto'
input — command builders read from the GPU button strip. Default
cookbook open is Download tab.
- Cookbook hwfit: 'Model (latest)' / 'Model (oldest)' header sorts by
release_date; release dates can be backfilled with the new
scripts/backfill_model_release_dates.py and recipe metadata pulled
with scripts/import_from_vllm_recipes.py against the upstream
vllm-project/recipes catalog (vllm_recipe + min_vllm_version stamped
on entries).
- Calendar: Quick add hint cycles a random Odysseus-themed example per
open (wooden horse Friday, crew muster 10am daily, council on
Ithaca, …). Typing a time like '11pm' in the event title updates
the hero clock live.
- Doc editor: email-mode Reply button (sparkle icon, accent) opens the
same Fast/Full + context popover the email reader uses; Ctrl+Alt+M
toggles markdown preview.
- Memories panel: custom sort picker with per-option icons, default
'Latest', visible Enabled/Disabled toggle text matching the section
description style.
2026-06-15 20:47:51 +09:00
// Bail with a clear message instead of silently doing nothing when
// every task on this server is still running (nothing finished to
// clear yet) — the previous behavior looked like the button was dead.
if ( ! toRemove . length ) {
2026-06-21 11:02:35 +00:00
const stillRunning = allTasks . filter ( t => _taskServerKey ( t ) === host && t . status === 'running' ) . length ;
Open email context for agent, email search across All Mail, cookbook serve polish
- Agent: pass the open email reader (uid/folder/account/from/subject/body
preview) on every chat submit so 'reply to this' / 'write email saying
hi' route to ui_control open_email_reply with the right UID instead of
inventing a new .md draft. Code-level enforcement (chat_routes strips
create_document + send_email when active_email is set); cross-session
active_doc_id is now trusted instead of being silently dropped.
set_active_email/clear_active_email tool-layer helpers in
tool_implementations.
- ui_control open_email_reply: optional body argument so the agent can
open-and-write in one call; envelope now forwards uid/folder/account/
body/panel through tool_output. Tool description sharpened and the
parser rejects empty bodies on reply/reply-all (forces the agent to
write rather than open an empty draft).
- Email library: search now runs against [Gmail]/All Mail when the
current folder is INBOX (archived emails surface). Whirlpool spinner
+ 'Searching…' placeholder while in flight. Each search result is
stamped with its source folder so clicks open the right email instead
of whatever shares its UID in INBOX. Search no longer re-applies the
same text pill locally (which only checks subject/from/snippet, never
body) so body-only matches don't get dropped after IMAP returns them.
Initial inbox load bumped 100→500.
- Email favorites: 'Favorite (pin to top)' / 'Unfavorite' in both the
card menu and the open-reader more menu, backed by a new
/api/email/flag/{uid}?on=true|false endpoint. Flagged emails always
bubble to the top of the grid regardless of active sort.
- AI reply in doc editor: never overwrites existing draft text or the
quoted history. AI suggestion is prepended; AI-generated 'On …
wrote:' re-quotes are stripped so the original quote isn't visually
edited.
- Cookbook serve: pre-launch GPU driver / has_gpu / install / version-
floor checks (vllm minimax_m2 needs 0.10.0+, deepseek_r1 needs 0.7.0
etc.) before the launch chain starts. Detect 'another model already
running on this host' and offer Stop & launch (with graceful then
force tmux kill helpers, port release wait). Per-vendor deep-link
buttons (vLLM recipe / SGLang cookbook) with hardware hash. Backend
picker is now a custom dropdown with accent-coloured logos for vLLM,
SGLang, llama.cpp, Ollama, Diffusers; same glyphs added next to
package names in Dependencies. Runtime-readiness note moved inside
the panel (green when ready, red when missing) with an × dismiss.
Esc collapses the expanded card; expanded card scrolls when it
overflows; Trust Remote / Auto Tool / Reasoning Parser / Enforce
Eager / Prefix Caching / Expert Parallel / Speculative / MoE Env on
one row (Reasoning Parser auto-detected per model family).
Dtype→Row 1, GPUs→Row 2 (rightmost). Removed redundant GPU 'auto'
input — command builders read from the GPU button strip. Default
cookbook open is Download tab.
- Cookbook hwfit: 'Model (latest)' / 'Model (oldest)' header sorts by
release_date; release dates can be backfilled with the new
scripts/backfill_model_release_dates.py and recipe metadata pulled
with scripts/import_from_vllm_recipes.py against the upstream
vllm-project/recipes catalog (vllm_recipe + min_vllm_version stamped
on entries).
- Calendar: Quick add hint cycles a random Odysseus-themed example per
open (wooden horse Friday, crew muster 10am daily, council on
Ithaca, …). Typing a time like '11pm' in the event title updates
the hero clock live.
- Doc editor: email-mode Reply button (sparkle icon, accent) opens the
same Fast/Full + context popover the email reader uses; Ctrl+Alt+M
toggles markdown preview.
- Memories panel: custom sort picker with per-option icons, default
'Latest', visible Enabled/Disabled toggle text matching the section
description style.
2026-06-15 20:47:51 +09:00
const _msg = stillRunning
? ` No finished tasks on ${ _serverName ( host ) } — ${ stillRunning } still running. Stop them first to clear. `
: ` No finished tasks on ${ _serverName ( host ) } . ` ;
if ( window . uiModule ? . showToast ) window . uiModule . showToast ( _msg ) ;
else alert ( _msg ) ;
return ;
}
if ( ! await window . styledConfirm ( ` Clear ${ toRemove . length } finished task ${ toRemove . length === 1 ? '' : 's' } on ${ _serverName ( host ) } ? ` , { confirmText : 'Clear' } ) ) return ;
2026-06-22 01:49:15 +00:00
toRemove . forEach ( t => _tombstoneTask ( t . sessionId ) ) ;
2026-06-21 11:02:35 +00:00
const remaining = allTasks . filter ( t => _taskServerKey ( t ) !== host || ! _canClearTask ( t ) ) ;
2026-05-31 23:58:26 +09:00
_saveTasks ( remaining ) ;
// Fade/slide each finished card out (same exit as the per-card clear)
// instead of yanking them instantly.
toRemove . forEach ( t => {
const el = document . querySelector ( ` .cookbook-task[data-task-id=" ${ t . sessionId } "] ` ) ;
if ( el ) {
if ( el . _abort ) el . _abort . abort ( ) ;
if ( el . _uptimeInterval ) clearInterval ( el . _uptimeInterval ) ;
el . style . transition = 'opacity 0.35s ease, transform 0.35s ease' ;
el . style . opacity = '0' ;
el . style . transform = 'translateX(-10px)' ;
}
} ) ;
// After the animation, remove the cards and tidy up the now-empty section.
setTimeout ( ( ) => {
toRemove . forEach ( t => document . querySelector ( ` .cookbook-task[data-task-id=" ${ t . sessionId } "] ` ) ? . remove ( ) ) ;
// If this server's section is now empty (only finished tasks lived here),
// remove the whole section so its header/title doesn't linger.
const _sk = ( host || 'local' ) . replace ( /[^a-zA-Z0-9-]/g , '_' ) ;
const _sec = group . querySelector ( ` .cookbook-server-section- ${ _sk } ` ) ;
if ( _sec && ! _sec . querySelector ( '.cookbook-task' ) ) _sec . remove ( ) ;
if ( ! remaining . length ) _renderRunningTab ( ) ;
} , 360 ) ;
} ) ;
} ) ;
// Wire "Stop all" buttons — stop every running task on that server.
group . querySelectorAll ( '[data-stop-server]' ) . forEach ( btn => {
if ( btn . _bound ) return ;
btn . _bound = true ;
btn . addEventListener ( 'click' , async ( e ) => {
e . stopPropagation ( ) ; // don't toggle the section collapse
const host = btn . dataset . stopServer ;
2026-06-21 11:02:35 +00:00
const running = _loadTasks ( ) . filter ( t => _taskServerKey ( t ) === host && t . status === 'running' ) ;
2026-05-31 23:58:26 +09:00
if ( ! running . length ) { uiModule . showToast ( ` Nothing running on ${ _serverName ( host ) } ` ) ; return ; }
if ( ! await window . styledConfirm ( ` Stop ${ running . length } running task ${ running . length > 1 ? 's' : '' } on ${ _serverName ( host ) } ? ` , { confirmText : 'Stop all' } ) ) return ;
2026-06-03 01:22:39 -03:00
// Mark every task as user-stopped BEFORE firing the kills so that the
// download auto-retry logic never restarts a task the user just stopped.
running . forEach ( t => _updateTask ( t . sessionId , { _userStopped : true } ) ) ;
2026-05-31 23:58:26 +09:00
// Reuse each task's own Stop action so it does the full teardown
// (send C-c, drop the endpoint, mark stopped) consistently.
running . forEach ( t => {
const el = document . querySelector ( ` .cookbook-task[data-task-id=" ${ t . sessionId } "] ` ) ;
el ? . querySelector ( '.cookbook-task-action-stop' ) ? . click ( ) ;
} ) ;
uiModule . showToast ( ` Stopped ${ running . length } task ${ running . length > 1 ? 's' : '' } on ${ _serverName ( host ) } ` ) ;
} ) ;
} ) ;
// Section collapse/expand
group . querySelectorAll ( '.cookbook-section-header[data-collapse]' ) . forEach ( hdr => {
if ( hdr . _bound ) return ;
hdr . _bound = true ;
hdr . addEventListener ( 'click' , ( ) => {
const bodyId = hdr . dataset . collapse ;
const body = document . getElementById ( bodyId ) ;
if ( ! body ) return ;
const isHidden = body . style . display === 'none' ;
body . style . display = isHidden ? '' : 'none' ;
const chevron = hdr . querySelector ( '.cookbook-section-chevron' ) ;
if ( chevron ) {
// Collapsed → point right (▶, click to expand); expanded → down (▼).
chevron . style . transform = isHidden ? '' : 'rotate(-90deg)' ;
chevron . style . opacity = '' ;
}
} ) ;
} ) ;
// Only add new tasks or update existing ones
const existingIds = new Set ( ) ;
group . querySelectorAll ( '.cookbook-task' ) . forEach ( el => {
const id = el . dataset . taskId ;
existingIds . add ( id ) ;
const task = tasks . find ( t => t . sessionId === id ) ;
if ( task ) {
el . dataset . status = task . status ;
const isDone = task . status === 'done' ;
// Type chip doubles as the "finished" badge once a task completes — both
// download and serve show the same green FINISHED chip.
const typeChip = el . querySelector ( '.cookbook-task-type' ) ;
if ( typeChip ) {
// Only DOWNLOAD tasks flip to "finished" when done — serve tasks keep
// saying "serve" because the model is still running on that port.
const isDoneDl = isDone && task . type === 'download' ;
typeChip . textContent = isDoneDl ? 'finished' : task . type ;
typeChip . classList . toggle ( 'cookbook-task-type-done' , isDoneDl ) ;
}
const badge = el . querySelector ( '.cookbook-task-status' ) ;
if ( badge ) {
const _bdg = _taskBadge ( task ) ;
badge . textContent = _bdg . text ;
badge . className = 'cookbook-task-status' + ( _bdg . cls ? ' ' + _bdg . cls : '' ) ;
2026-06-02 12:15:41 +09:00
badge . style . display = '' ;
2026-05-31 23:58:26 +09:00
}
// Indicator: spinning wave while running, green check when finished.
const wave = el . querySelector ( '.cookbook-task-wave' ) ;
if ( wave ) wave . style . display = task . status === 'running' ? '' : 'none' ;
const check = el . querySelector ( '.cookbook-task-check' ) ;
2026-06-02 12:15:41 +09:00
if ( check ) {
check . style . display = _canClearTask ( task ) ? '' : 'none' ;
const label = check . querySelector ( '.cookbook-task-done-label' ) ;
if ( label ) label . textContent = _clearPillLabel ( task ) ;
}
2026-06-02 22:38:55 +09:00
const startNow = el . querySelector ( '.cookbook-task-start-now' ) ;
if ( startNow ) startNow . style . display = ( task . type === 'download' && task . status === 'queued' ) ? '' : 'none' ;
2026-06-30 01:47:48 +00:00
const pre = el . querySelector ( '.cookbook-output-pre' ) ;
if ( pre && typeof task . output === 'string' && task . output && pre . textContent !== task . output ) {
const atBottom = ( pre . scrollHeight - pre . scrollTop - pre . clientHeight ) < 40 ;
pre . textContent = task . output ;
if ( atBottom ) pre . scrollTop = pre . scrollHeight ;
}
2026-06-02 12:15:41 +09:00
const terminalDiag = _terminalServeDiagnosis ( task , el . querySelector ( '.cookbook-output-pre' ) ? . textContent || task . output || '' ) ;
2026-06-05 13:53:33 +01:00
if ( terminalDiag ) {
_showDiagnosis ( el , terminalDiag , el . querySelector ( '.cookbook-output-pre' ) ? . textContent || task . output || '' ) ;
} else {
const existingDiag = el . querySelector ( '.cookbook-diagnosis' ) ;
// Keep diagnosis for failed tasks even if output was cleared and we
// can no longer re-derive the exact message — removing it would hide
// the crash reason from the user.
if ( existingDiag && ! [ 'stopped' , 'error' , 'crashed' , 'failed' ] . includes ( task . status ) ) {
existingDiag . remove ( ) ;
}
}
2026-05-31 23:58:26 +09:00
}
if ( ! task ) {
if ( el . _uptimeInterval ) { clearInterval ( el . _uptimeInterval ) ; el . _uptimeInterval = null ; }
el . remove ( ) ;
}
} ) ;
// Add new task entries
for ( const task of tasks ) {
if ( existingIds . has ( task . sessionId ) ) continue ;
const el = document . createElement ( 'div' ) ;
el . className = 'cookbook-task' + ( task . _unreachable && task . status === 'running' ? ' cookbook-task-unreachable' : '' ) ;
el . dataset . taskId = task . sessionId ;
el . dataset . status = task . status ;
el . dataset . type = task . type || '' ;
const _bdg = _taskBadge ( task ) ;
const _bdgTitle = ( task . _unreachable && task . status === 'running' ) ? ' title="Server not responding — it may have crashed"' : '' ;
2026-06-22 01:49:15 +00:00
const displayName = _taskDisplayName ( task ) ;
2026-05-31 23:58:26 +09:00
el . innerHTML = `
< div class = "cookbook-task-header" >
< span class = "cookbook-task-type${(task.status === 'done' && task.type === 'download') ? ' cookbook-task-type-done' : ''}" data - type = "${esc(task.type)}" > $ { esc ( ( task . status === 'done' && task . type === 'download' ) ? 'finished' : task . type ) } < / s p a n >
2026-06-22 01:49:15 +00:00
< span class = "cookbook-task-name" > $ { modelLogo ( task . name ) } $ { esc ( displayName ) } < / s p a n >
< span class = "cookbook-task-indicator" > < span class = "cookbook-task-wave" style = "display:${task.status === 'running' ? '' : 'none'}" > < /span>${_canLaunchDownloadedTask(task) ? '<button type="button" class="cookbook-task-serve-btn" title="Open in Launch"><svg width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2.4" stroke-linecap="round" stroke-linejoin="round"><polygon points="13 2 3 14 12 14 11 22 21 10 12 10 13 2"/ > < / s v g > < s p a n > L a u n c h < / s p a n > < / b u t t o n > ' : ' ' } < s p a n c l a s s = " c o o k b o o k - t a s k - c h e c k " t i t l e = " C l e a r " s t y l e = " d i s p l a y : $ { _ c a n C l e a r T a s k ( t a s k ) ? ' ' : ' n o n e ' } " > < s v g c l a s s = " c o o k b o o k - t a s k - c h e c k - i c o " w i d t h = " 1 2 " h e i g h t = " 1 2 " v i e w B o x = " 0 0 2 4 2 4 " f i l l = " n o n e " s t r o k e = " # 5 0 f a 7 b " s t r o k e - w i d t h = " 3 " s t r o k e - l i n e c a p = " r o u n d " s t r o k e - l i n e j o i n = " r o u n d " > < p o l y l i n e p o i n t s = " 2 0 6 9 1 7 4 1 2 " / > < / s v g > < s v g c l a s s = " c o o k b o o k - t a s k - c l e a r - i c o " w i d t h = " 1 2 " h e i g h t = " 1 2 " v i e w B o x = " 0 0 2 4 2 4 " f i l l = " n o n e " s t r o k e = " c u r r e n t C o l o r " s t r o k e - w i d t h = " 3 " s t r o k e - l i n e c a p = " r o u n d " s t r o k e - l i n e j o i n = " r o u n d " > < l i n e x 1 = " 1 8 " y 1 = " 6 " x 2 = " 6 " y 2 = " 1 8 " / > < l i n e x 1 = " 6 " y 1 = " 6 " x 2 = " 1 8 " y 2 = " 1 8 " / > < / s v g > < s p a n c l a s s = " c o o k b o o k - t a s k - d o n e - l a b e l " > $ { e s c ( _ c l e a r P i l l L a b e l ( t a s k ) ) } < / s p a n > < s p a n c l a s s = " c o o k b o o k - t a s k - c l e a r - l a b e l " > c l e a r < / s p a n > < / s p a n > < / s p a n >
2026-06-02 22:38:55 +09:00
< button type = "button" class = "cookbook-task-start-now" title = "Start this queued download now" style = "display:${(task.type === 'download' && task.status === 'queued') ? '' : 'none'}" > < svg width = "11" height = "11" viewBox = "0 0 24 24" fill = "currentColor" aria - hidden = "true" > < polygon points = "8 5 19 12 8 19 8 5" / > < / s v g > < s p a n > s t a r t n o w < / s p a n > < / b u t t o n >
2026-06-02 12:15:41 +09:00
< span class = "cookbook-task-status ${_bdg.cls}" $ { _bdgTitle } > $ { esc ( _bdg . text ) } < / s p a n >
2026-05-31 23:58:26 +09:00
< button class = "cookbook-task-menu-btn" title = "Actions" > & # 8942 ; < / b u t t o n >
< / d i v >
2026-06-03 16:49:10 +09:00
< div class = "cookbook-task-sub" > < span class = "cookbook-task-session" > $ { esc ( task . sessionId ) } < / s p a n > < s p a n c l a s s = " c o o k b o o k - t a s k - u p t i m e " s t y l e = " d i s p l a y : $ { ( ( t a s k . t y p e = = = ' s e r v e ' | | t a s k . t y p e = = = ' d o w n l o a d ' ) & & t a s k . s t a t u s = = = ' r u n n i n g ' ) ? ' ' : ' n o n e ' } " > < / s p a n > $ { ( t a s k . t y p e = = = ' d o w n l o a d ' ) ? ` < s p a n c l a s s = " c o o k b o o k - t a s k - d l d i r " t i t l e = " D o w n l o a d d e s t i n a t i o n " s t y l e = " f o n t - s i z e : 9 p x ; c o l o r : v a r ( - - f g - m u t e d ) ; f o n t - f a m i l y : ' F i r a C o d e ' , m o n o s p a c e ; o p a c i t y : 0 . 4 ; w h i t e - s p a c e : n o w r a p ; o v e r f l o w : h i d d e n ; t e x t - o v e r f l o w : e l l i p s i s ; m a x - w i d t h : 4 0 c h ; " > D i r : $ { e s c ( t a s k . p a y l o a d ? . l o c a l _ d i r | | ' ~ / . c a c h e / h u g g i n g f a c e / h u b ' ) } < / s p a n > ` : ' ' } < / d i v >
2026-07-07 00:50:07 +00:00
< div class = "cookbook-output-wrap cookbook-task-collapsible${(_mobileCollapseDefault && !_shouldAutoExpandTaskOutput(task)) ? ' cookbook-task-collapsed' : ''}" > < pre class = "cookbook-output-pre" > $ { esc ( task . output || '' ) } < /pre><button type="button" class="copy-code cookbook-output-copy"><svg width="14" height="14" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round"><rect x="9" y="9" width="13" height="13" rx="2"/ > < path d = "M5 15H4a2 2 0 0 1-2-2V4a2 2 0 0 1 2-2h9a2 2 0 0 1 2 2v1" / > < / s v g > < / b u t t o n > < / d i v >
2026-05-31 23:58:26 +09:00
` ;
const _waveEl = el . querySelector ( '.cookbook-task-wave' ) ;
if ( _waveEl && task . status === 'running' ) _registerWaveEl ( _waveEl ) ;
2026-06-02 12:15:41 +09:00
const terminalDiag = _terminalServeDiagnosis ( task , task . output || '' ) ;
if ( terminalDiag ) _showDiagnosis ( el , terminalDiag , task . output || '' ) ;
2026-06-05 05:52:07 -03:00
if ( ! terminalDiag && ( task . status === 'error' || task . status === 'crashed' ) && task . _backendDiagnosis ) {
_showDiagnosis ( el , task . _backendDiagnosis , task . output || '' ) ;
}
2026-06-02 12:15:41 +09:00
2026-05-31 23:58:26 +09:00
const _uptimeEl = el . querySelector ( '.cookbook-task-uptime' ) ;
if ( _uptimeEl && ( task . type === 'serve' || task . type === 'download' ) && task . status === 'running' ) {
const _startedAt = task . ts || Date . now ( ) ;
const _prefix = task . type === 'download' ? 'downloading' : 'uptime' ;
el . _uptimeInterval = setInterval ( ( ) => {
const secs = Math . floor ( ( Date . now ( ) - _startedAt ) / 1000 ) ;
const h = Math . floor ( secs / 3600 ) ;
const m = Math . floor ( ( secs % 3600 ) / 60 ) ;
const s = secs % 60 ;
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
const _timer = h > 0
2026-05-31 23:58:26 +09:00
? ` ${ _prefix } : ${ h } h ${ String ( m ) . padStart ( 2 , '0' ) } m `
: ` ${ _prefix } : ${ m } m ${ String ( s ) . padStart ( 2 , '0' ) } s ` ;
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
// ETA — only for downloads, only when we have a meaningful overall %.
// Reads the badge text (which already shows the true overall % we
// compute in the live-polling block) and back-derives a remaining-time
// estimate from elapsed/done. Hidden until pct >= 3% so the early-job
// wild estimates don't show.
let _eta = '' ;
if ( task . type === 'download' ) {
const _badge = el . querySelector ( '.cookbook-task-status' ) ;
const _m = _badge && /^(\d+)%/ . exec ( _badge . textContent || '' ) ;
const _pct = _m ? parseInt ( _m [ 1 ] , 10 ) : 0 ;
if ( _pct >= 3 && _pct < 100 && secs > 5 ) {
const _totalSec = Math . round ( secs * ( 100 / _pct ) ) ;
const _remain = Math . max ( 0 , _totalSec - secs ) ;
const _eh = Math . floor ( _remain / 3600 ) ;
const _em = Math . floor ( ( _remain % 3600 ) / 60 ) ;
const _es = _remain % 60 ;
_eta = _eh > 0
? ` · ETA ${ _eh } h ${ String ( _em ) . padStart ( 2 , '0' ) } m `
: ( _em > 0 ? ` · ETA ${ _em } m ${ String ( _es ) . padStart ( 2 , '0' ) } s ` : ` · ETA ${ _es } s ` ) ;
}
}
_uptimeEl . textContent = _timer + _eta ;
2026-05-31 23:58:26 +09:00
} , 1000 ) ;
}
// Re-open the Serve panel for this model, pre-filled with the EXACT
2026-06-02 12:15:41 +09:00
// settings this instance launched with, and on the SERVER it runs on.
2026-05-31 23:58:26 +09:00
const _openEdit = ( ) => _openServeEditForTask ( task ) ;
2026-06-02 12:15:41 +09:00
el . addEventListener ( 'cookbook:edit-serve' , ( e ) => {
e . stopPropagation ( ) ;
_openServeEditForTask ( task , null , e . detail ? . fields || null ) ;
} ) ;
2026-05-31 23:58:26 +09:00
// Finished download → an explicit "Serve →" button jumps straight to the
// Serve tab with this model pre-selected (on the server it downloaded to).
if ( task . type === 'download' ) {
const _serveBtn = el . querySelector ( '.cookbook-task-serve-btn' ) ;
if ( _serveBtn ) {
_serveBtn . addEventListener ( 'click' , async ( e ) => {
e . stopPropagation ( ) ;
const repo = task . payload ? . repo _id || task . name ;
if ( ! repo ) { uiModule . showToast ( 'No model info on this task' ) ; return ; }
2026-06-22 01:49:15 +00:00
// Point the active server at the exact profile it downloaded to.
_selectTaskServer ( task ) ;
2026-05-31 23:58:26 +09:00
try {
const { openServePanelForRepo } = await import ( './cookbookServe.js' ) ;
2026-06-22 01:49:15 +00:00
await openServePanelForRepo ( repo , _downloadServeFields ( task ) ) ;
2026-05-31 23:58:26 +09:00
// Serving it supersedes the finished download — clear the card from
// the Running tab (smooth exit) now that we've jumped to Serve.
_animateOutThenRemove ( el , task . sessionId ) ;
} catch ( err ) { uiModule . showToast ( 'Could not open Serve: ' + err . message ) ; }
} ) ;
}
}
// Finished tasks show a green check — make it click-to-clear so the user can
// dismiss a completed download/update (we no longer auto-remove them). It
// morphs to a red ✕ on hover (see CSS).
const _clearChk = el . querySelector ( '.cookbook-task-check' ) ;
if ( _clearChk ) {
_clearChk . addEventListener ( 'click' , ( e ) => {
e . stopPropagation ( ) ;
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
// If the output still shows an active shard line, the task isn't
// actually finished — clicking is "reconnect" (flip back to running
// + let _reconnectTask reattach to the live tmux session), not
// "clear". The pill label already reflects this via _clearPillLabel.
if ( _downloadOutputLooksActive ( task ) ) {
const _fresh = _loadTasks ( ) ;
const _ft = _fresh . find ( t => t . sessionId === task . sessionId ) ;
if ( _ft ) {
_ft . status = 'running' ;
_ft . _selfHealed = true ;
_saveTasks ( _fresh ) ;
}
// Visually flip without waiting for a full re-render — same path the
// self-heal uses on cookbook open.
const _chk = el . querySelector ( '.cookbook-task-check' ) ;
if ( _chk ) _chk . style . display = 'none' ;
const _wave = el . querySelector ( '.cookbook-task-wave' ) ;
if ( _wave ) _wave . style . display = '' ;
const _up = el . querySelector ( '.cookbook-task-uptime' ) ;
if ( _up ) _up . style . display = '' ;
el . dataset . status = 'running' ;
_renderRunningTab ( ) ;
return ;
}
// Otherwise: real clear. Kill the tmux session as belt-and-suspenders,
// then animate out + remove the row.
2026-06-03 16:49:10 +09:00
try {
fetch ( '/api/shell/exec' , {
method : 'POST' , credentials : 'same-origin' ,
headers : { 'Content-Type' : 'application/json' } ,
body : JSON . stringify ( { command : _tmuxCmd ( task , ` kill-session -t ${ task . sessionId } ` ) } ) ,
} ) . catch ( ( ) => { } ) ;
} catch { }
2026-05-31 23:58:26 +09:00
_animateOutThenRemove ( el , task . sessionId ) ;
} ) ;
}
2026-06-02 22:38:55 +09:00
const _startNowBtn = el . querySelector ( '.cookbook-task-start-now' ) ;
if ( _startNowBtn ) {
_startNowBtn . addEventListener ( 'click' , ( e ) => {
e . stopPropagation ( ) ;
_startQueuedDownload ( task ) ;
} ) ;
}
2026-05-31 23:58:26 +09:00
// Wire header click to collapse/expand output
el . querySelector ( '.cookbook-task-header' ) . addEventListener ( 'click' , ( e ) => {
if ( e . target . closest ( 'button' ) ) return ;
const wrap = el . querySelector ( '.cookbook-output-wrap' ) ;
2026-07-07 00:50:07 +00:00
if ( ! wrap ) return ;
const isOpening = wrap . classList . contains ( 'cookbook-task-collapsed' ) ;
wrap . classList . toggle ( 'cookbook-task-collapsed' ) ;
if ( isOpening ) {
_expandedTaskIds . add ( task . sessionId ) ;
_collapsedTaskIds . delete ( task . sessionId ) ;
if ( task . sessionId && [ 'serve' , 'download' ] . includes ( task . type || '' ) ) {
_reconnectTask ( el , task ) ;
}
} else {
_collapsedTaskIds . add ( task . sessionId ) ;
_expandedTaskIds . delete ( task . sessionId ) ;
if ( el . _abort ) {
try { el . _abort . abort ( ) ; } catch { }
el . _abort = null ;
}
}
2026-05-31 23:58:26 +09:00
} ) ;
// Wire menu button (also fire from a long-press anywhere on the card so
// mobile users don't have to hit the small ⋮ target precisely).
const menuBtn = el . querySelector ( '.cookbook-task-menu-btn' ) ;
if ( menuBtn ) {
// Long-press detection on the card: ~500ms hold without scroll movement
// re-uses the menu button's click path (so we don't duplicate logic).
let _lpTimer = null ;
let _lpStartY = 0 ;
let _lpCanceled = false ;
const _lpStart = ( e ) => {
_lpCanceled = false ;
_lpStartY = ( e . touches ? . [ 0 ] ? . clientY ) ? ? 0 ;
_lpTimer = setTimeout ( ( ) => {
if ( _lpCanceled ) return ;
_lpCanceled = true ; // suppress the subsequent click-through
try { menuBtn . click ( ) ; } catch { }
} , 500 ) ;
} ;
const _lpCancel = ( ) => {
if ( _lpTimer ) { clearTimeout ( _lpTimer ) ; _lpTimer = null ; }
} ;
const _lpMove = ( e ) => {
const y = ( e . touches ? . [ 0 ] ? . clientY ) ? ? 0 ;
if ( Math . abs ( y - _lpStartY ) > 8 ) _lpCancel ( ) ;
} ;
el . addEventListener ( 'touchstart' , ( e ) => {
// Skip if the user is starting touch on a button / link inside the
// card — those already have their own tap handlers.
if ( e . target . closest ( 'button, a, input, textarea, .cookbook-task-dropdown' ) ) return ;
_lpStart ( e ) ;
} , { passive : true } ) ;
el . addEventListener ( 'touchmove' , _lpMove , { passive : true } ) ;
el . addEventListener ( 'touchend' , _lpCancel , { passive : true } ) ;
el . addEventListener ( 'touchcancel' , _lpCancel , { passive : true } ) ;
menuBtn . addEventListener ( 'click' , ( e ) => {
e . stopPropagation ( ) ;
2026-07-03 00:45:43 +00:00
const existing = document . querySelector ( '.cookbook-task-dropdown' ) ;
if ( existing && existing . _anchor === menuBtn ) {
if ( typeof existing . _dismiss === 'function' ) existing . _dismiss ( ) ;
else existing . remove ( ) ;
return ;
}
fix: make transient dropdown/popup menus close on Escape
The global Escape arbiter in ui.js only sees `.modal` elements, so the many
ad-hoc dropdowns and context popups that are built on the fly and appended to
<body> ignored Escape entirely: document-library card/chat menus, chat
context/stats/overflow popups, cookbook serve & running menus, calendar event
menus, and compare pane menus.
Add a small DOM-free dismissal registry (static/js/escMenuStack.js). Menus
register a dismiss callback while open, and the arbiter closes the
most-recently-opened one first, so a menu opened over a modal closes before the
modal. bindMenuDismiss() wires the ubiquitous "append-to-body, close on outside
click" idiom to both the outside-click listener and the Escape stack in one
call, and dismissOrRemove() lets the pre-existing bulk removers (scroll/swipe/
modal-dismiss cleanup, reopen sweeps) tear a menu down through its real teardown
instead of orphaning its stack entry.
Covers ~14 menus across documentLibrary, chatRenderer, cookbookServe,
cookbookRunning, calendar, and compare/panes. Every teardown path — item click,
outside click, swipe, toggle, rebuild, bulk cleanup — routes through the
registry so no entry is ever stranded.
tests/test_esc_menu_stack_js.py pins the registry's LIFO and
exactly-one-per-press guarantees (node-driven; skips when node is absent).
2026-06-01 14:23:22 -04:00
document . querySelectorAll ( '.cookbook-task-dropdown' ) . forEach ( d => { if ( typeof d . _dismiss === 'function' ) d . _dismiss ( ) ; else d . remove ( ) ; } ) ;
2026-05-31 23:58:26 +09:00
const dropdown = document . createElement ( 'div' ) ;
dropdown . className = 'cookbook-task-dropdown' ;
2026-07-03 00:45:43 +00:00
dropdown . _anchor = menuBtn ;
menuBtn . classList . add ( 'cookbook-menu-active' ) ;
2026-05-31 23:58:26 +09:00
const items = [ ] ;
2026-06-13 22:18:13 +09:00
// ── Run section ─────────────────────────────────────────────
2026-05-31 23:58:26 +09:00
// Queued download: let the user jump the queue and start it immediately
// (downloads otherwise run one-at-a-time per server).
if ( task . type === 'download' && task . status === 'queued' ) {
2026-06-13 22:18:13 +09:00
items . push ( { group : 'run' , label : 'Start now' , action : 'start-now' , custom : ( ) => {
2026-05-31 23:58:26 +09:00
_startQueuedDownload ( task ) ;
_renderRunningTab ( ) ;
} } ) ;
}
if ( task . status !== 'running' && task . status !== 'queued' ) {
2026-06-13 22:18:13 +09:00
items . push ( { group : 'run' , label : 'Reconnect tmux' , action : 'reconnect' } ) ;
2026-05-31 23:58:26 +09:00
}
2026-06-13 22:18:13 +09:00
items . push ( { group : 'run' , label : 'Restart' , action : 'retry' } ) ;
// ── Edit section ────────────────────────────────────────────
// Merged "Edit & relaunch" — opens the structured serve panel
// pre-filled with this task's config. The old standalone "Edit
// cmd & relaunch" raw-text dialog is now reachable from inside
// that panel (Show command). Single entry-point per task.
2026-05-31 23:58:26 +09:00
if ( task . type === 'serve' && task . payload ? . repo _id ) {
2026-06-13 22:18:13 +09:00
items . push ( { group : 'edit' , label : 'Edit & relaunch' , action : 'edit-panel' , tooltip : 'Open the Serve config panel pre-filled with this task — pick a different backend, change GPUs, edit env vars or the raw cmd, then Launch.' , custom : ( ) => _openEdit ( ) } ) ;
2026-05-31 23:58:26 +09:00
}
if ( task . type === 'serve' && task . payload ? . _cmd ) {
2026-06-13 22:18:13 +09:00
items . push ( { group : 'edit' , label : 'Save serve' , action : 'save' , custom : ( ) => {
2026-05-31 23:58:26 +09:00
if ( ! _saveTaskAsPreset ( task ) ) { uiModule . showToast ( 'Already saved' ) ; return ; }
uiModule . showToast ( 'Saved to presets' ) ;
_renderRunningTab ( ) ;
} } ) ;
}
2026-06-13 22:18:13 +09:00
// ── Endpoint section ────────────────────────────────────────
2026-05-31 23:58:26 +09:00
// Manual endpoint registration — fallback for when auto-add fails
// (e.g. probe timeout on a remote that's slow). Forces adding this
// serve to the model-endpoints list regardless of prior flag state.
if ( task . type === 'serve' && task . payload ? . _cmd ) {
2026-06-13 22:18:13 +09:00
items . push ( { group : 'endpoint' , label : 'Register endpoint' , action : 'register-endpoint' , custom : async ( ) => {
2026-06-02 10:09:48 -05:00
const host = _connectHostFromRemote ( task . remoteHost ) ;
2026-05-31 23:58:26 +09:00
const portMatch = task . payload ? . _cmd ? . match ( /--port\s+(\d+)/ ) ;
const port = portMatch ? portMatch [ 1 ] : '8000' ;
const baseUrl = ` http:// ${ host } : ${ port } /v1 ` ;
try {
// Check existing first — offer to overwrite if present
const eps = await ( await fetch ( '/api/model-endpoints' , { credentials : 'same-origin' } ) ) . json ( ) ;
const existing = eps . find ( e => e . base _url === baseUrl ) ;
if ( existing ) {
uiModule . showToast ( ` Already registered as " ${ existing . name } " ` ) ;
task . _endpointAdded = true ;
_updateTask ( task . sessionId , { _endpointAdded : true } ) ;
_refreshModelsAfterEndpointChange ( ) ;
// If it's still offline (registered before the server finished
// loading), keep probing until it answers instead of leaving it
// stuck offline until a manual delete/re-add.
if ( existing . id && ! ( existing . models || [ ] ) . length ) _probeEndpointUntilOnline ( existing . id , host , port ) ;
return ;
}
const fd = new FormData ( ) ;
fd . append ( 'base_url' , baseUrl ) ;
fd . append ( 'name' , task . name ) ;
fd . append ( 'skip_probe' , 'true' ) ;
2026-06-02 10:09:48 -05:00
_appendCookbookEndpointScope ( fd , task . remoteHost || '' ) ;
2026-05-31 23:58:26 +09:00
if ( task . payload ? . _cmd ? . includes ( 'diffusion_server' ) ) fd . append ( 'model_type' , 'image' ) ;
const res = await fetch ( '/api/model-endpoints' , { method : 'POST' , credentials : 'same-origin' , body : fd } ) ;
if ( res . ok ) {
task . _endpointAdded = true ;
_updateTask ( task . sessionId , { _endpointAdded : true } ) ;
uiModule . showToast ( ` Endpoint registered: ${ host } : ${ port } ` ) ;
_refreshModelsAfterEndpointChange ( ) ;
// Added with skip_probe → probe until the (possibly still
// warming) server answers, so it flips online on its own.
const _ep = await res . json ( ) . catch ( ( ) => ( { } ) ) ;
if ( _ep && _ep . id ) _probeEndpointUntilOnline ( _ep . id , host , port ) ;
} else {
const body = await res . text ( ) . catch ( ( ) => '' ) ;
uiModule . showError ( ` Register failed: ${ res . status } ${ body . slice ( 0 , 140 ) } ` ) ;
}
} catch ( e ) {
uiModule . showError ( ` Register failed: ${ e . message || e } ` ) ;
}
} } ) ;
}
2026-06-13 22:18:13 +09:00
// ── Copy section ────────────────────────────────────────────
2026-05-31 23:58:26 +09:00
if ( _isWindows ( task ) ) {
2026-06-04 09:00:01 +05:30
const host = task . remoteHost ;
const sd = host ? '$env:TEMP\\odysseus-sessions' : '$env:TEMP\\odysseus-tmux' ;
const logCmd = host
? ` ssh ${ _sshPrefix ( _getPort ( task ) ) } ${ host } "powershell -Command \\ "Get-Content ' ${ sd } \\ ${ task . sessionId } .log' -Wait \\ "" `
: ` powershell -Command "Get-Content (Join-Path $ env:TEMP 'odysseus-tmux \\ ${ task . sessionId } .log') -Wait" ` ;
2026-06-13 22:18:13 +09:00
items . push ( { group : 'copy' , label : 'Copy log cmd' , action : 'copy-tmux' , custom : ( ) => {
2026-05-31 23:58:26 +09:00
_copyText ( logCmd ) ;
} } ) ;
} else {
// Just the tmux command itself — no ssh wrapper.
const tmuxAttach = ` tmux attach -t ${ task . sessionId } ` ;
2026-06-13 22:18:13 +09:00
items . push ( { group : 'copy' , label : 'Copy tmux' , action : 'copy-tmux' , custom : ( ) => {
2026-05-31 23:58:26 +09:00
_copyText ( tmuxAttach ) ;
} } ) ;
}
2026-06-01 09:12:35 -05:00
if ( _shouldOfferCrashReport ( task ) ) {
2026-06-13 22:18:13 +09:00
items . push ( { group : 'copy' , label : 'Copy crash report' , action : 'copy-crash-report' , custom : ( ) => {
2026-06-01 09:12:35 -05:00
const out = ( el . querySelector ( '.cookbook-output-pre' ) ? . textContent || task . output || '' ) ;
_copyText ( _buildCrashReport ( task , out ) ) ;
uiModule . showToast ( 'Copied crash report' ) ;
} } ) ;
}
2026-05-31 23:58:26 +09:00
// Copy the last 50 lines of the task's output/log.
2026-06-13 22:18:13 +09:00
items . push ( { group : 'copy' , label : 'Copy last 50 lines' , action : 'copy-log' , custom : ( ) => {
2026-05-31 23:58:26 +09:00
const out = ( el . querySelector ( '.cookbook-output-pre' ) ? . textContent || task . output || '' ) ;
const last = out . split ( '\n' ) . slice ( - 50 ) . join ( '\n' ) ;
2026-06-05 13:53:33 +01:00
if ( ! last . trim ( ) ) {
uiModule . showToast ( 'No log content available yet' ) ;
return ;
}
2026-05-31 23:58:26 +09:00
_copyText ( last ) ;
uiModule . showToast ( 'Copied last 50 lines' ) ;
} } ) ;
Cookbook scheduler + serve: schedule via Tasks, Stop verifies kill, Ollama auto port-pick
- Schedule cookbook serves through the existing ScheduledTask system: the
serve preset gets a ^ button next to Launch that opens a daily/hourly/
weekly form mirroring the admin-switch style; the schedule action runs
action_cookbook_serve, which delegates to /api/model/serve and stamps
the resulting task with _scheduledStopAtMs. A background
cookbook_serve_lifecycle loop ticks every 60s and kills any serve
whose window has ended, also dropping the auto-registered endpoint
so the model picker doesn't keep pointing at a dead server.
- Stop and remove on a Running serve now awaits the SSH/tmux kill,
re-checks tmux has-session, and surfaces an error toast (leaving the
row) when the kill failed. Previously fire-and-forget, so a failed
SSH/tmux call silently left the live serve running while the row
vanished from the UI.
- Cookbook tasks/status orphan-adoption sweep no longer requires the
serve-/cookbook- session-id prefix; any tmux session whose pane is
running a known model-server process gets auto-pulled into Running.
Without this loosening, a cookbook-launched serve whose tmux id
fell back to a bare number was invisible — you couldn't see it,
let alone stop it.
- Ollama serve always launches a fresh process under cookbook's tmux
(no more monitor-mode reattach to a systemd/Docker ollama Stop can't
reach). The handler pre-picks a free port by probing the target
host over SSH and mutates req.cmd's OLLAMA_HOST so the runner script
AND the auto-registered endpoint agree on the same bind port.
- Auto-register uses host.docker.internal (when running inside Docker)
instead of localhost, matching the URL /setup adds for Ollama by
hand. Local cookbook serves now produce a chat-reachable endpoint
on first launch.
- Cascade-delete: removing a scheduled cookbook task also deletes any
linked calendar event (cookbook_task_id marker in the description).
- Tasks list groups cookbook_serve under a "Cookbook" category that
sorts above the rest, so scheduler-launched serves are easy to find.
2026-06-05 14:41:43 +09:00
// Label matches behavior — the kill handler ALWAYS first kills
// the live tmux session and (for serve tasks) deletes the
// matching model-endpoint, THEN animates the task card out.
// Just "Remove" hid that it stops the live serve too.
2026-06-13 22:18:13 +09:00
// ── Danger section ──────────────────────────────────────────
Cookbook scheduler + serve: schedule via Tasks, Stop verifies kill, Ollama auto port-pick
- Schedule cookbook serves through the existing ScheduledTask system: the
serve preset gets a ^ button next to Launch that opens a daily/hourly/
weekly form mirroring the admin-switch style; the schedule action runs
action_cookbook_serve, which delegates to /api/model/serve and stamps
the resulting task with _scheduledStopAtMs. A background
cookbook_serve_lifecycle loop ticks every 60s and kills any serve
whose window has ended, also dropping the auto-registered endpoint
so the model picker doesn't keep pointing at a dead server.
- Stop and remove on a Running serve now awaits the SSH/tmux kill,
re-checks tmux has-session, and surfaces an error toast (leaving the
row) when the kill failed. Previously fire-and-forget, so a failed
SSH/tmux call silently left the live serve running while the row
vanished from the UI.
- Cookbook tasks/status orphan-adoption sweep no longer requires the
serve-/cookbook- session-id prefix; any tmux session whose pane is
running a known model-server process gets auto-pulled into Running.
Without this loosening, a cookbook-launched serve whose tmux id
fell back to a bare number was invisible — you couldn't see it,
let alone stop it.
- Ollama serve always launches a fresh process under cookbook's tmux
(no more monitor-mode reattach to a systemd/Docker ollama Stop can't
reach). The handler pre-picks a free port by probing the target
host over SSH and mutates req.cmd's OLLAMA_HOST so the runner script
AND the auto-registered endpoint agree on the same bind port.
- Auto-register uses host.docker.internal (when running inside Docker)
instead of localhost, matching the URL /setup adds for Ollama by
hand. Local cookbook serves now produce a chat-reachable endpoint
on first launch.
- Cascade-delete: removing a scheduled cookbook task also deletes any
linked calendar event (cookbook_task_id marker in the description).
- Tasks list groups cookbook_serve under a "Cookbook" category that
sorts above the rest, so scheduler-launched serves are easy to find.
2026-06-05 14:41:43 +09:00
const _isLive = task . type === 'serve' && [ 'running' , 'ready' , 'loading' , 'warming' , 'starting' ] . includes ( task . status || '' ) ;
items . push ( {
2026-06-13 22:18:13 +09:00
group : 'danger' ,
Cookbook scheduler + serve: schedule via Tasks, Stop verifies kill, Ollama auto port-pick
- Schedule cookbook serves through the existing ScheduledTask system: the
serve preset gets a ^ button next to Launch that opens a daily/hourly/
weekly form mirroring the admin-switch style; the schedule action runs
action_cookbook_serve, which delegates to /api/model/serve and stamps
the resulting task with _scheduledStopAtMs. A background
cookbook_serve_lifecycle loop ticks every 60s and kills any serve
whose window has ended, also dropping the auto-registered endpoint
so the model picker doesn't keep pointing at a dead server.
- Stop and remove on a Running serve now awaits the SSH/tmux kill,
re-checks tmux has-session, and surfaces an error toast (leaving the
row) when the kill failed. Previously fire-and-forget, so a failed
SSH/tmux call silently left the live serve running while the row
vanished from the UI.
- Cookbook tasks/status orphan-adoption sweep no longer requires the
serve-/cookbook- session-id prefix; any tmux session whose pane is
running a known model-server process gets auto-pulled into Running.
Without this loosening, a cookbook-launched serve whose tmux id
fell back to a bare number was invisible — you couldn't see it,
let alone stop it.
- Ollama serve always launches a fresh process under cookbook's tmux
(no more monitor-mode reattach to a systemd/Docker ollama Stop can't
reach). The handler pre-picks a free port by probing the target
host over SSH and mutates req.cmd's OLLAMA_HOST so the runner script
AND the auto-registered endpoint agree on the same bind port.
- Auto-register uses host.docker.internal (when running inside Docker)
instead of localhost, matching the URL /setup adds for Ollama by
hand. Local cookbook serves now produce a chat-reachable endpoint
on first launch.
- Cascade-delete: removing a scheduled cookbook task also deletes any
linked calendar event (cookbook_task_id marker in the description).
- Tasks list groups cookbook_serve under a "Cookbook" category that
sorts above the rest, so scheduler-launched serves are easy to find.
2026-06-05 14:41:43 +09:00
label : _isLive ? 'Stop and remove' : 'Remove' ,
action : 'kill' ,
tooltip : _isLive
? 'Kill the live tmux session, deregister the chat endpoint, and remove this row'
: 'Remove this row' ,
danger : true ,
} ) ;
2026-06-13 22:18:13 +09:00
// Cancel = mobile-only dismiss item. Same pattern as the email kebab.
items . push ( { group : 'danger' , label : 'Cancel' , action : 'cancel' , mobileOnly : true , custom : ( ) => { } } ) ;
2026-05-31 23:58:26 +09:00
const _MENU _ICONS = {
'start-now' : '<polygon points="6 4 20 12 6 20 6 4"/>' ,
reconnect : '<path d="M1 4v6h6"/><path d="M3.5 15a9 9 0 1 0 2.1-9.4L1 10"/>' ,
retry : '<path d="M1 4v6h6"/><path d="M3.5 15a9 9 0 1 0 2.1-9.4L1 10"/>' ,
stop : '<rect x="6" y="6" width="12" height="12" rx="1"/>' ,
edit : '<path d="M11 4H4a2 2 0 0 0-2 2v14a2 2 0 0 0 2 2h14a2 2 0 0 0 2-2v-7"/><path d="M18.5 2.5a2.12 2.12 0 0 1 3 3L12 15l-4 1 1-4z"/>' ,
'edit-panel' : '<path d="M11 4H4a2 2 0 0 0-2 2v14a2 2 0 0 0 2 2h14a2 2 0 0 0 2-2v-7"/><path d="M18.5 2.5a2.12 2.12 0 0 1 3 3L12 15l-4 1 1-4z"/>' ,
'register-endpoint' : '<circle cx="12" cy="12" r="9"/><path d="M12 8v8M8 12h8"/>' ,
save : '<path d="M19 21H5a2 2 0 0 1-2-2V5a2 2 0 0 1 2-2h11l5 5v11a2 2 0 0 1-2 2z"/><path d="M17 21v-8H7v8M7 3v5h8"/>' ,
'copy-tmux' : '<rect x="9" y="9" width="13" height="13" rx="2"/><path d="M5 15H4a2 2 0 0 1-2-2V4a2 2 0 0 1 2-2h9a2 2 0 0 1 2 2v1"/>' ,
2026-06-01 09:12:35 -05:00
'copy-crash-report' : '<path d="M10.3 2.3 1.8 17a2 2 0 0 0 1.7 3h17a2 2 0 0 0 1.7-3L13.7 2.3a2 2 0 0 0-3.4 0z"/><path d="M12 8v5M12 17h.01"/>' ,
2026-05-31 23:58:26 +09:00
'copy-log' : '<rect x="9" y="9" width="13" height="13" rx="2"/><path d="M5 15H4a2 2 0 0 1-2-2V4a2 2 0 0 1 2-2h9a2 2 0 0 1 2 2v1"/>' ,
kill : '<path d="M3 6h18"/><path d="M19 6v14a2 2 0 0 1-2 2H7a2 2 0 0 1-2-2V6m3 0V4a2 2 0 0 1 2-2h4a2 2 0 0 1 2 2v2"/>' ,
cancel : '<line x1="18" y1="6" x2="6" y2="18"/><line x1="6" y1="6" x2="18" y2="18"/>' ,
} ;
2026-06-13 22:18:13 +09:00
let _lastGroup = null ;
2026-05-31 23:58:26 +09:00
for ( const item of items ) {
2026-06-13 22:18:13 +09:00
// Insert a thin divider whenever the group changes, so the
// user can visually scan Run / Edit / Endpoint / Copy / Danger
// blocks instead of one long undifferentiated list.
if ( item . group && _lastGroup && item . group !== _lastGroup ) {
const sep = document . createElement ( 'div' ) ;
sep . className = 'cookbook-dropdown-divider' ;
sep . style . cssText = 'height:1px;margin:4px 6px;background:color-mix(in srgb, var(--fg) 12%, transparent);pointer-events:none;' ;
dropdown . appendChild ( sep ) ;
}
_lastGroup = item . group || _lastGroup ;
2026-05-31 23:58:26 +09:00
const div = document . createElement ( 'div' ) ;
div . className = 'dropdown-item-compact'
+ ( item . danger ? ' cookbook-dropdown-danger' : '' )
+ ( item . mobileOnly ? ' dropdown-cancel-mobile' : '' ) ;
div . style . cssText = 'display:flex;align-items:center;gap:8px;' ;
Cookbook scheduler + serve: schedule via Tasks, Stop verifies kill, Ollama auto port-pick
- Schedule cookbook serves through the existing ScheduledTask system: the
serve preset gets a ^ button next to Launch that opens a daily/hourly/
weekly form mirroring the admin-switch style; the schedule action runs
action_cookbook_serve, which delegates to /api/model/serve and stamps
the resulting task with _scheduledStopAtMs. A background
cookbook_serve_lifecycle loop ticks every 60s and kills any serve
whose window has ended, also dropping the auto-registered endpoint
so the model picker doesn't keep pointing at a dead server.
- Stop and remove on a Running serve now awaits the SSH/tmux kill,
re-checks tmux has-session, and surfaces an error toast (leaving the
row) when the kill failed. Previously fire-and-forget, so a failed
SSH/tmux call silently left the live serve running while the row
vanished from the UI.
- Cookbook tasks/status orphan-adoption sweep no longer requires the
serve-/cookbook- session-id prefix; any tmux session whose pane is
running a known model-server process gets auto-pulled into Running.
Without this loosening, a cookbook-launched serve whose tmux id
fell back to a bare number was invisible — you couldn't see it,
let alone stop it.
- Ollama serve always launches a fresh process under cookbook's tmux
(no more monitor-mode reattach to a systemd/Docker ollama Stop can't
reach). The handler pre-picks a free port by probing the target
host over SSH and mutates req.cmd's OLLAMA_HOST so the runner script
AND the auto-registered endpoint agree on the same bind port.
- Auto-register uses host.docker.internal (when running inside Docker)
instead of localhost, matching the URL /setup adds for Ollama by
hand. Local cookbook serves now produce a chat-reachable endpoint
on first launch.
- Cascade-delete: removing a scheduled cookbook task also deletes any
linked calendar event (cookbook_task_id marker in the description).
- Tasks list groups cookbook_serve under a "Cookbook" category that
sorts above the rest, so scheduler-launched serves are easy to find.
2026-06-05 14:41:43 +09:00
if ( item . tooltip ) div . title = item . tooltip ;
2026-05-31 23:58:26 +09:00
const ic = _MENU _ICONS [ item . action ] || '' ;
div . innerHTML = ` <span style="display:inline-flex;flex-shrink:0;opacity:0.7;"><svg width="14" height="14" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round"> ${ ic } </svg></span><span> ${ item . label } </span> ` ;
div . addEventListener ( 'click' , ( ) => {
fix: make transient dropdown/popup menus close on Escape
The global Escape arbiter in ui.js only sees `.modal` elements, so the many
ad-hoc dropdowns and context popups that are built on the fly and appended to
<body> ignored Escape entirely: document-library card/chat menus, chat
context/stats/overflow popups, cookbook serve & running menus, calendar event
menus, and compare pane menus.
Add a small DOM-free dismissal registry (static/js/escMenuStack.js). Menus
register a dismiss callback while open, and the arbiter closes the
most-recently-opened one first, so a menu opened over a modal closes before the
modal. bindMenuDismiss() wires the ubiquitous "append-to-body, close on outside
click" idiom to both the outside-click listener and the Escape stack in one
call, and dismissOrRemove() lets the pre-existing bulk removers (scroll/swipe/
modal-dismiss cleanup, reopen sweeps) tear a menu down through its real teardown
instead of orphaning its stack entry.
Covers ~14 menus across documentLibrary, chatRenderer, cookbookServe,
cookbookRunning, calendar, and compare/panes. Every teardown path — item click,
outside click, swipe, toggle, rebuild, bulk cleanup — routes through the
registry so no entry is ever stranded.
tests/test_esc_menu_stack_js.py pins the registry's LIFO and
exactly-one-per-press guarantees (node-driven; skips when node is absent).
2026-06-01 14:23:22 -04:00
_cleanup ( ) ;
2026-05-31 23:58:26 +09:00
if ( item . custom ) { item . custom ( ) ; return ; }
el . querySelector ( '.cookbook-task-action-' + item . action ) ? . click ( ) ;
} ) ;
dropdown . appendChild ( div ) ;
}
const rect = menuBtn . getBoundingClientRect ( ) ;
dropdown . style . position = 'fixed' ;
dropdown . style . top = rect . bottom + 2 + 'px' ;
dropdown . style . right = ( window . innerWidth - rect . right ) + 'px' ;
document . body . appendChild ( dropdown ) ;
// Clamp into the *visible* area. On mobile (esp. Firefox) window.innerHeight
// includes the strip hidden under the dynamic toolbar, so a menu that "fits"
// by innerHeight still lands off-screen at the bottom. visualViewport gives
// the real visible region. Flip above the button if there's no room below,
// else clamp to the bottom edge.
{
const vv = window . visualViewport ;
const viewTop = vv ? vv . offsetTop : 0 ;
const viewBottom = vv ? vv . offsetTop + vv . height : window . innerHeight ;
const dh = dropdown . offsetHeight ;
const m = 8 ;
let top = rect . bottom + 2 ;
if ( top + dh > viewBottom - m ) {
const above = rect . top - 2 - dh ;
top = above >= viewTop + m ? above : Math . max ( viewTop + m , viewBottom - dh - m ) ;
}
dropdown . style . top = top + 'px' ;
}
const closeHandler = ( ev ) => {
2026-07-03 00:45:43 +00:00
if ( ! dropdown . contains ( ev . target ) && ev . target !== menuBtn && ! menuBtn . contains ( ev . target ) ) {
2026-05-31 23:58:26 +09:00
_cleanup ( ) ;
}
} ;
// Close on scroll too — once the page scrolls, the dropdown's
// fixed position no longer matches the originating ⋮ button, so
// it visually drifts. Matches the email kebab behaviour.
const scrollClose = ( ) => _cleanup ( ) ;
fix: make transient dropdown/popup menus close on Escape
The global Escape arbiter in ui.js only sees `.modal` elements, so the many
ad-hoc dropdowns and context popups that are built on the fly and appended to
<body> ignored Escape entirely: document-library card/chat menus, chat
context/stats/overflow popups, cookbook serve & running menus, calendar event
menus, and compare pane menus.
Add a small DOM-free dismissal registry (static/js/escMenuStack.js). Menus
register a dismiss callback while open, and the arbiter closes the
most-recently-opened one first, so a menu opened over a modal closes before the
modal. bindMenuDismiss() wires the ubiquitous "append-to-body, close on outside
click" idiom to both the outside-click listener and the Escape stack in one
call, and dismissOrRemove() lets the pre-existing bulk removers (scroll/swipe/
modal-dismiss cleanup, reopen sweeps) tear a menu down through its real teardown
instead of orphaning its stack entry.
Covers ~14 menus across documentLibrary, chatRenderer, cookbookServe,
cookbookRunning, calendar, and compare/panes. Every teardown path — item click,
outside click, swipe, toggle, rebuild, bulk cleanup — routes through the
registry so no entry is ever stranded.
tests/test_esc_menu_stack_js.py pins the registry's LIFO and
exactly-one-per-press guarantees (node-driven; skips when node is absent).
2026-06-01 14:23:22 -04:00
let _unreg = ( ) => { } ;
2026-05-31 23:58:26 +09:00
const _cleanup = ( ) => {
fix: make transient dropdown/popup menus close on Escape
The global Escape arbiter in ui.js only sees `.modal` elements, so the many
ad-hoc dropdowns and context popups that are built on the fly and appended to
<body> ignored Escape entirely: document-library card/chat menus, chat
context/stats/overflow popups, cookbook serve & running menus, calendar event
menus, and compare pane menus.
Add a small DOM-free dismissal registry (static/js/escMenuStack.js). Menus
register a dismiss callback while open, and the arbiter closes the
most-recently-opened one first, so a menu opened over a modal closes before the
modal. bindMenuDismiss() wires the ubiquitous "append-to-body, close on outside
click" idiom to both the outside-click listener and the Escape stack in one
call, and dismissOrRemove() lets the pre-existing bulk removers (scroll/swipe/
modal-dismiss cleanup, reopen sweeps) tear a menu down through its real teardown
instead of orphaning its stack entry.
Covers ~14 menus across documentLibrary, chatRenderer, cookbookServe,
cookbookRunning, calendar, and compare/panes. Every teardown path — item click,
outside click, swipe, toggle, rebuild, bulk cleanup — routes through the
registry so no entry is ever stranded.
tests/test_esc_menu_stack_js.py pins the registry's LIFO and
exactly-one-per-press guarantees (node-driven; skips when node is absent).
2026-06-01 14:23:22 -04:00
_unreg ( ) ; _unreg = ( ) => { } ;
2026-05-31 23:58:26 +09:00
dropdown . remove ( ) ;
2026-07-03 00:45:43 +00:00
menuBtn . classList . remove ( 'cookbook-menu-active' ) ;
2026-05-31 23:58:26 +09:00
document . removeEventListener ( 'click' , closeHandler ) ;
window . removeEventListener ( 'scroll' , scrollClose , true ) ;
window . visualViewport ? . removeEventListener ( 'scroll' , scrollClose ) ;
} ;
fix: make transient dropdown/popup menus close on Escape
The global Escape arbiter in ui.js only sees `.modal` elements, so the many
ad-hoc dropdowns and context popups that are built on the fly and appended to
<body> ignored Escape entirely: document-library card/chat menus, chat
context/stats/overflow popups, cookbook serve & running menus, calendar event
menus, and compare pane menus.
Add a small DOM-free dismissal registry (static/js/escMenuStack.js). Menus
register a dismiss callback while open, and the arbiter closes the
most-recently-opened one first, so a menu opened over a modal closes before the
modal. bindMenuDismiss() wires the ubiquitous "append-to-body, close on outside
click" idiom to both the outside-click listener and the Escape stack in one
call, and dismissOrRemove() lets the pre-existing bulk removers (scroll/swipe/
modal-dismiss cleanup, reopen sweeps) tear a menu down through its real teardown
instead of orphaning its stack entry.
Covers ~14 menus across documentLibrary, chatRenderer, cookbookServe,
cookbookRunning, calendar, and compare/panes. Every teardown path — item click,
outside click, swipe, toggle, rebuild, bulk cleanup — routes through the
registry so no entry is ever stranded.
tests/test_esc_menu_stack_js.py pins the registry's LIFO and
exactly-one-per-press guarantees (node-driven; skips when node is absent).
2026-06-01 14:23:22 -04:00
dropdown . _dismiss = _cleanup ;
2026-05-31 23:58:26 +09:00
setTimeout ( ( ) => {
document . addEventListener ( 'click' , closeHandler ) ;
window . addEventListener ( 'scroll' , scrollClose , true ) ;
window . visualViewport ? . addEventListener ( 'scroll' , scrollClose ) ;
} , 0 ) ;
fix: make transient dropdown/popup menus close on Escape
The global Escape arbiter in ui.js only sees `.modal` elements, so the many
ad-hoc dropdowns and context popups that are built on the fly and appended to
<body> ignored Escape entirely: document-library card/chat menus, chat
context/stats/overflow popups, cookbook serve & running menus, calendar event
menus, and compare pane menus.
Add a small DOM-free dismissal registry (static/js/escMenuStack.js). Menus
register a dismiss callback while open, and the arbiter closes the
most-recently-opened one first, so a menu opened over a modal closes before the
modal. bindMenuDismiss() wires the ubiquitous "append-to-body, close on outside
click" idiom to both the outside-click listener and the Escape stack in one
call, and dismissOrRemove() lets the pre-existing bulk removers (scroll/swipe/
modal-dismiss cleanup, reopen sweeps) tear a menu down through its real teardown
instead of orphaning its stack entry.
Covers ~14 menus across documentLibrary, chatRenderer, cookbookServe,
cookbookRunning, calendar, and compare/panes. Every teardown path — item click,
outside click, swipe, toggle, rebuild, bulk cleanup — routes through the
registry so no entry is ever stranded.
tests/test_esc_menu_stack_js.py pins the registry's LIFO and
exactly-one-per-press guarantees (node-driven; skips when node is absent).
2026-06-01 14:23:22 -04:00
_unreg = registerMenuDismiss ( _cleanup ) ;
2026-05-31 23:58:26 +09:00
} ) ;
}
// Hidden action buttons for menu dispatch
const _actionBtns = document . createElement ( 'div' ) ;
_actionBtns . style . display = 'none' ;
_actionBtns . innerHTML = `
< button class = "cookbook-task-action-reconnect" > < / b u t t o n >
< button class = "cookbook-task-action-retry" > < / b u t t o n >
< button class = "cookbook-task-action-stop" > < / b u t t o n >
< button class = "cookbook-task-action-kill" > < / b u t t o n >
` ;
el . appendChild ( _actionBtns ) ;
// Wire reconnect
el . querySelector ( '.cookbook-task-action-reconnect' ) . addEventListener ( 'click' , ( ) => {
_updateTask ( task . sessionId , { status : 'running' } ) ;
el . dataset . status = 'running' ;
const badge = el . querySelector ( '.cookbook-task-status' ) ;
if ( badge ) { badge . textContent = _statusLabel ( 'running' , task . type ) ; badge . className = 'cookbook-task-status cookbook-task-running' ; }
_reconnectTask ( el , task ) ;
} ) ;
// Wire stop
el . querySelector ( '.cookbook-task-action-stop' ) . addEventListener ( 'click' , async ( ) => {
fix(cookbook): stop-all no longer auto-retries interrupted HF downloads fixes (#1474)
* fix(cookbook): stop-all no longer auto-retries interrupted HF downloads
When C-c was sent to a running download, the bash wrapper printed
DOWNLOAD_FAILED on non-zero exit (SIGINT = 130). The reconnect polling
loop was still running at that point, saw the failure marker, and
silently relaunched the download — making "Stop all" appear to have no
effect while the UI showed the toast as if it succeeded.
Fix: abort the reconnect controller immediately when the stop button is
clicked (before the kill command is dispatched), and guard the
auto-retry condition with !controller.signal.aborted so that any
in-flight poll that completes after abort cannot trigger a retry.
Fixes #1458
Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
* Fix Edge/Chromium sidebar section-title clipping (#1420)
Sidebar section titles were vertically clipped in Chromium/Edge (fine in
Firefox). Raise line-height 1 → 1.3, mirroring the existing .list-item fix.
The titles are flex-centred in a fixed-height (29px) header, so this adds
glyph headroom without any reflow.
* Drop GPU-only flags from the CPU-only (-ngl 0) serve command (#1433)
A CPU-only llama.cpp serve config still emitted --flash-attn on and exported
GGML_CUDA_ENABLE_UNIFIED_MEMORY=1 (independent toggles, often left on by an Auto
profile), so the command mixed "zero GPU layers" with CUDA/flash-attn and failed
to start (issue #1291). Gate both on a _cpuOnly check (ngl == 0). GPU serving is
unchanged — the gate only affects the ngl=0 path.
Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
* fix: APIKeyManager.load crashes app startup on a corrupt/wrong-shape api_keys.json (#1565)
* Don't lose deep-research findings when synthesis times out (#1551) (#1562)
Two problems made deep research report "No information could be gathered" even
after it had extracted findings, on slow local models (reporter served a 20B
via LM Studio):
- _synthesize hard-capped its LLM call at timeout=60, while extraction uses the
user's extraction_timeout (300s here) and the final report uses 180s. The slow
model needed >60s to synthesize the round's findings, so synthesis timed out
after 3 attempts. Raised it to 180s to match the final-report call.
- When synthesis produced no report (it returns the unchanged, still-empty
report on failure during round 1), the run hit
`if not report: return "No information could be gathered…"` and discarded the
findings it had already gathered. Now it falls back to a compiled report built
from those findings (_fallback_report) so the user keeps the gathered material.
Tests stub the LLM (no live model/DB), pin the synthesis timeout >= 180, that the
fallback surfaces the findings rather than the give-up message, and that a failed
synthesis preserves the previous report.
Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
* fix: return sorted model list on first call in group chat (#1484)
Both _getModels() and getAllModels() store the sorted copy in a cache
variable but return the original unsorted array on first invocation.
Subsequent calls return the cache (sorted), causing inconsistent
model picker ordering on first render.
* fix: guard sp.destroy() in _loadScheduled against null spinner (#1495)
When the scheduled folder is opened with cached data, sp is null
(the loading spinner is skipped). _loadScheduled receives null and
calls sp.destroy() unconditionally, crashing with TypeError.
* fix: capture download exit code before test consumes it (#1497)
The shell pattern 'if [ $? -eq 0 ]; ... else ... echo DOWNLOAD_FAILED (exit $?)' always reports 'exit 1' because $? inside the else branch is the exit code of the [ test command, not the download. Capture into _ec first.
* fix: guard uid.decode() in auto-classify warning log against str UIDs (#1472)
Every other uid.decode() call in this function uses
'uid.decode() if isinstance(uid, bytes) else str(uid)' but the
warning at line 832 does bare uid.decode(), crashing with
AttributeError when uid is already a string.
* fix: guard AI tidy verdict against non-string LLM output (#1486)
The AI document-tidy endpoint parses verdicts from LLM JSON output
and calls .lower().strip() directly. If the model returns null or a
non-string element, this crashes with AttributeError. Coerce to str
so malformed output is treated as 'keep' instead of crashing.
* fix: rename local url-quote import to avoid shadowing module-level _q (#1471)
The 'from urllib.parse import quote as _q' at line 734 shadows the
module-level _q (istrstrstrstrstrstrIMAPutility) imported from email_helpers, causing
UnboundLocalError at lines 191 and 278 where _q is used before the
local import executes. This silently breaks the entire auto-summarize
pass.
* fix(ui): add missing Escape key handlers for email-lib-modal, model-picker-menu, and sort dropdowns (#1487)
CONTEXT: Several interactive elements lacked Escape key handlers: the email library modal was not in dynamicModals, the model-picker popup had no Escape close, and the session/model sort dropdowns only closed on outside click.
CHANGE: Adds email-lib-modal to the dynamicModals array in the Escape handler so it gets dismissed via dismissModal. Adds a check for model-picker-menu.open before the modal chain to close the dropdown on Escape. Adds checks for session-sort-dropdown and model-sort-dropdown display=block before the document panel minimize fallback.
WHY: Users expect consistent Escape-to-close behavior across all modals, overlays, and popups. These four were the only interactive containers in the app that ignored the Escape key entirely.
IMPACT: Pressing Escape now closes the email library modal, model picker popup, session sort dropdown, and model sort dropdown -- matching user expectations and the behavior of every other modal in the app.
* fix: mcp CLI _serialize crashes when stored env JSON is a list (#1609)
* fix: validate_caldav_url crashes with TypeError on a non-string URL (#1608)
* fix: _sanitize_export_filename crashes on a non-string session name (#1607)
* fix: shared MCP truncate() crashes on None/non-string tool output (#1605)
* fix: search query helpers crash on a non-string query (#1604)
* fix: rag_server add/remove_directory crashes on a non-string directory arg (#1614)
* fix: gallery CLI image serialization crashes on a non-string prompt (#1598)
* fix: research CLI summary crashes on a non-string query (#1596)
* fix: skills CLI summary crashes on a non-string description (#1595)
* fix(cookbook): set UTF-8 encoding for detached download/serve subprocesses (#1599)
On Windows, Python defaults to the active code page (cp1252) for
subprocess I/O. HuggingFace CLI outputs U+2713 (✓) when validating
tokens, which cp1252 cannot encode, crashing the download process.
Set PYTHONUTF8=1 and PYTHONIOENCODING=utf-8 in the subprocess
environment so Unicode output from hf/pip/llama-server is handled
correctly.
Fixes #1543
Co-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
* docs: clarify host Ollama with Docker (#1594)
* fix(ui): stop welcome-screen tip from clipping on narrow phones (#1612)
The empty-state tip ("Add an AI endpoint from Settings...") shares a 60px
max-height ceiling with the one-line .welcome-sub / .welcome-version. On
narrow phones the welcome block shrink-wraps and the tip wraps to 4-5 lines
(~67px), so the shared ceiling clipped its last line ("...key into the
chat.") - the only setup hint a first-run user gets.
Give .welcome-tip its own taller max-height (120px), placed above the
@media (max-height: 650px) block so that rule's max-height:0 still collapses
the tip on short viewports. .welcome-sub / .welcome-version are untouched,
and desktop is unchanged (the tip is ~50px there, well under the ceiling).
* Save only string personal doc paths (#1566)
* Reject backup output inside data dir (#1587)
* Parse all AMD GPU check args (#1586)
* Require runnable dispatcher subcommands (#1585)
* Require runnable dispatcher subcommands
* Use modern dispatcher test loader
* Remove duplicate update database body (#1584)
* Skip invalid research service sources (#1583)
* Reject CalDAV writeback events without uid (#1582)
* Reject empty mail CLI recipients (#1581)
* Reject empty mail CLI recipients
* Keep mail CLI test imports isolated
* Validate signature CLI PNG data (#1580)
* Validate signature CLI PNG data
* Keep signature CLI test imports isolated
* Reject invalid preset CLI entries (#1579)
* Reject invalid preset CLI entries
* Use modern preset CLI test loader
* Normalize session CLI counters (#1578)
* Normalize session CLI counters
* Keep sessions CLI test imports isolated
* fix: monthly schedule label shows 21th/22th/31th (ordinal suffix for days >20) (#1577)
* fix: split_chunks emits a duplicate trailing chunk for text over size-overlap (#1573)
* fix: builtin_actions heuristics crash on a truthy non-string input (#1639)
* fix: skill test-task / precision helpers crash on a non-dict skill (#1638)
* fix: logs CLI _resolve crashes on a non-string name (#1631)
* fix: _extract_skill_json crashes on a truthy non-string teacher response (#1630)
* fix: tool-block parsing crashes on a non-string input (#1628)
* fix: check_outbound_url crashes on a truthy non-string URL (#1623)
* fix: document_actions title/content helpers crash on non-string input (#1621)
* fix: inside_base_dir raises TypeError on a non-string path instead of failing closed (#1619)
* fix: is_markitdown_format crashes on a non-string path (#1618)
* Close app_api blocklist gap for bare /api/tokens and /api/users
The blocklist prefixes had trailing slashes, so path.startswith() only
matched /api/tokens/{id} but not /api/tokens itself — the bare GET (list)
and POST (mint) endpoints were reachable via app_api. Same gap on
/api/users (list/create/delete). Drop trailing slashes so both bare and
sub-resource forms are blocked. /api/auth and /api/admin had no bare
endpoints today but get the same treatment to prevent future drift.
Caught by #1462.
* Decrypt CalDAV password before write-back (#1731)
writeback_event read cfg["password"] (the encrypted blob) and passed it
straight to DAVClient, so every local create/edit/delete authenticated
with the literal ciphertext, the remote rejected it, and the change
never reached the server — the exact silent-write-loss this module was
built to prevent. The pull path src/caldav_sync.py already decrypts;
mirror that. decrypt() is a no-op on legacy plaintext.
Caught by #1731.
* Memory MCP delete: match exact id, not prefix (#1303)
The delete action looked up the target with startswith() to capture
full_id, but then re-applied startswith() to filter the list — so a
short or ambiguous memory_id silently deleted every memory whose id
shared the prefix, while the success message reported only the first
match. The edit action used the first match and stopped, so the two
actions disagreed on multi-match behaviour. Use full_id for both.
Caught by #1303.
* Rebuild memory vector index from the full saved set, not just the audited owner (#1747)
audit_memories saves final_entries merged with other owners' entries
(correct), but then rebuilt the shared vector collection from
final_entries alone — wiping every other owner from semantic search
until they happened to run their own audit. Keyword fallback masked
it, so it degraded silently. Capture saved_entries once and rebuild
from that.
Caught by #1747.
* Owner-scope RAG doc ids so identical chunks across users don't collide (#1738, #1760)
_generate_doc_id hashed only text. add_document / add_documents_batch
early-return when the id exists, so the second owner indexing a
byte-identical chunk hit the first owner's id, was silently dropped,
and never stored under their owner — their owner-filtered search then
quietly omitted it. Hash owner + text; empty owner reproduces the
legacy id, so the unowned/base index keeps existing ids and isn't
re-churned. Same-owner identical chunks still dedupe.
Caught by #1738 and #1760 (independent reports of the same bug).
* Removed duplicate definition of _preview_text()
---------
Co-authored-by: Claude Sonnet 4.6 <noreply@anthropic.com>
Co-authored-by: Zeus-Deus <100132710+Zeus-Deus@users.noreply.github.com>
Co-authored-by: lekt8 <lewistham9x@gmail.com>
Co-authored-by: Afonso Coutinho <afonso@omelhorsite.pt>
Co-authored-by: Paulo Victor Cordeiro <146781332+pvcordeiro@users.noreply.github.com>
Co-authored-by: Zarl-prog <asimjunaidi5u@gmail.com>
Co-authored-by: Wes Huber <wesleybaxterhuber@gmail.com>
Co-authored-by: .bulat <its.bulat@icloud.com>
Co-authored-by: Mahdi Salmanzade <mahdisalmanzadehasl@gmail.com>
Co-authored-by: red person <redpersoncoding@gmail.com>
Co-authored-by: pewdiepie-archdaemon <pewdiepie-archdaemon@users.noreply.github.com>
2026-06-04 16:18:39 +05:30
// Abort the reconnect loop before sending kill so that a DOWNLOAD_FAILED
// marker written by the shell wrapper (on SIGINT/non-zero exit) cannot
// trigger an auto-retry after a manual stop.
if ( el . _abort ) el . _abort . abort ( ) ;
2026-05-31 23:58:26 +09:00
const badge = el . querySelector ( '.cookbook-task-status' ) ;
if ( badge ) { badge . textContent = 'stopping...' ; badge . className = 'cookbook-task-status cookbook-task-stopping' ; }
el . dataset . status = 'stopped' ;
2026-06-03 01:22:39 -03:00
_updateTask ( task . sessionId , { _userStopped : true } ) ;
2026-06-02 22:38:55 +09:00
const outputText = el . querySelector ( '.cookbook-output-pre' ) ? . textContent || task . output || '' ;
2026-05-31 23:58:26 +09:00
// Drop the model endpoint so the picker stops listing it.
if ( task . type === 'serve' && task . payload ) {
2026-06-02 22:38:55 +09:00
_removeEndpointByUrl ( _endpointUrlForTask ( task , outputText ) ) ;
}
const ollamaUnload = _ollamaUnloadCommand ( task , outputText ) ;
if ( ollamaUnload ) {
try {
await fetch ( '/api/shell/exec' , {
method : 'POST' , credentials : 'same-origin' ,
headers : { 'Content-Type' : 'application/json' } ,
body : JSON . stringify ( { command : ollamaUnload } ) ,
} ) ;
} catch { }
2026-05-31 23:58:26 +09:00
}
// Gracefully stop (C-c, then kill the session) so it's fully down...
try {
await fetch ( '/api/shell/exec' , {
method : 'POST' , credentials : 'same-origin' ,
headers : { 'Content-Type' : 'application/json' } ,
body : JSON . stringify ( { command : _tmuxGracefulKill ( task ) } ) ,
} ) ;
} catch { }
// ...then smoothly fade/slide the card out and auto-remove it — no manual
// ⋮ → Remove needed.
_animateOutThenRemove ( el , task . sessionId ) ;
} ) ;
Cookbook scheduler + serve: schedule via Tasks, Stop verifies kill, Ollama auto port-pick
- Schedule cookbook serves through the existing ScheduledTask system: the
serve preset gets a ^ button next to Launch that opens a daily/hourly/
weekly form mirroring the admin-switch style; the schedule action runs
action_cookbook_serve, which delegates to /api/model/serve and stamps
the resulting task with _scheduledStopAtMs. A background
cookbook_serve_lifecycle loop ticks every 60s and kills any serve
whose window has ended, also dropping the auto-registered endpoint
so the model picker doesn't keep pointing at a dead server.
- Stop and remove on a Running serve now awaits the SSH/tmux kill,
re-checks tmux has-session, and surfaces an error toast (leaving the
row) when the kill failed. Previously fire-and-forget, so a failed
SSH/tmux call silently left the live serve running while the row
vanished from the UI.
- Cookbook tasks/status orphan-adoption sweep no longer requires the
serve-/cookbook- session-id prefix; any tmux session whose pane is
running a known model-server process gets auto-pulled into Running.
Without this loosening, a cookbook-launched serve whose tmux id
fell back to a bare number was invisible — you couldn't see it,
let alone stop it.
- Ollama serve always launches a fresh process under cookbook's tmux
(no more monitor-mode reattach to a systemd/Docker ollama Stop can't
reach). The handler pre-picks a free port by probing the target
host over SSH and mutates req.cmd's OLLAMA_HOST so the runner script
AND the auto-registered endpoint agree on the same bind port.
- Auto-register uses host.docker.internal (when running inside Docker)
instead of localhost, matching the URL /setup adds for Ollama by
hand. Local cookbook serves now produce a chat-reachable endpoint
on first launch.
- Cascade-delete: removing a scheduled cookbook task also deletes any
linked calendar event (cookbook_task_id marker in the description).
- Tasks list groups cookbook_serve under a "Cookbook" category that
sorts above the rest, so scheduler-launched serves are easy to find.
2026-06-05 14:41:43 +09:00
// Wire kill — awaits the SSH/tmux kill and verifies the session is
// actually gone before removing the row. Previously fire-and-forget,
// which meant a failed kill (wrong remoteHost, SSH error, tmux server
// already exited) silently left the live serve running while the
// row disappeared from the UI.
el . querySelector ( '.cookbook-task-action-kill' ) . addEventListener ( 'click' , async ( ) => {
2026-06-02 22:38:55 +09:00
const outputText = el . querySelector ( '.cookbook-output-pre' ) ? . textContent || task . output || '' ;
Cookbook scheduler + serve: schedule via Tasks, Stop verifies kill, Ollama auto port-pick
- Schedule cookbook serves through the existing ScheduledTask system: the
serve preset gets a ^ button next to Launch that opens a daily/hourly/
weekly form mirroring the admin-switch style; the schedule action runs
action_cookbook_serve, which delegates to /api/model/serve and stamps
the resulting task with _scheduledStopAtMs. A background
cookbook_serve_lifecycle loop ticks every 60s and kills any serve
whose window has ended, also dropping the auto-registered endpoint
so the model picker doesn't keep pointing at a dead server.
- Stop and remove on a Running serve now awaits the SSH/tmux kill,
re-checks tmux has-session, and surfaces an error toast (leaving the
row) when the kill failed. Previously fire-and-forget, so a failed
SSH/tmux call silently left the live serve running while the row
vanished from the UI.
- Cookbook tasks/status orphan-adoption sweep no longer requires the
serve-/cookbook- session-id prefix; any tmux session whose pane is
running a known model-server process gets auto-pulled into Running.
Without this loosening, a cookbook-launched serve whose tmux id
fell back to a bare number was invisible — you couldn't see it,
let alone stop it.
- Ollama serve always launches a fresh process under cookbook's tmux
(no more monitor-mode reattach to a systemd/Docker ollama Stop can't
reach). The handler pre-picks a free port by probing the target
host over SSH and mutates req.cmd's OLLAMA_HOST so the runner script
AND the auto-registered endpoint agree on the same bind port.
- Auto-register uses host.docker.internal (when running inside Docker)
instead of localhost, matching the URL /setup adds for Ollama by
hand. Local cookbook serves now produce a chat-reachable endpoint
on first launch.
- Cascade-delete: removing a scheduled cookbook task also deletes any
linked calendar event (cookbook_task_id marker in the description).
- Tasks list groups cookbook_serve under a "Cookbook" category that
sorts above the rest, so scheduler-launched serves are easy to find.
2026-06-05 14:41:43 +09:00
const isLive = task . type === 'serve' && [ 'running' , 'ready' , 'loading' , 'warming' , 'starting' ] . includes ( task . status || '' ) ;
2026-06-02 22:38:55 +09:00
const ollamaUnload = _ollamaUnloadCommand ( task , outputText ) ;
if ( ollamaUnload ) {
Cookbook scheduler + serve: schedule via Tasks, Stop verifies kill, Ollama auto port-pick
- Schedule cookbook serves through the existing ScheduledTask system: the
serve preset gets a ^ button next to Launch that opens a daily/hourly/
weekly form mirroring the admin-switch style; the schedule action runs
action_cookbook_serve, which delegates to /api/model/serve and stamps
the resulting task with _scheduledStopAtMs. A background
cookbook_serve_lifecycle loop ticks every 60s and kills any serve
whose window has ended, also dropping the auto-registered endpoint
so the model picker doesn't keep pointing at a dead server.
- Stop and remove on a Running serve now awaits the SSH/tmux kill,
re-checks tmux has-session, and surfaces an error toast (leaving the
row) when the kill failed. Previously fire-and-forget, so a failed
SSH/tmux call silently left the live serve running while the row
vanished from the UI.
- Cookbook tasks/status orphan-adoption sweep no longer requires the
serve-/cookbook- session-id prefix; any tmux session whose pane is
running a known model-server process gets auto-pulled into Running.
Without this loosening, a cookbook-launched serve whose tmux id
fell back to a bare number was invisible — you couldn't see it,
let alone stop it.
- Ollama serve always launches a fresh process under cookbook's tmux
(no more monitor-mode reattach to a systemd/Docker ollama Stop can't
reach). The handler pre-picks a free port by probing the target
host over SSH and mutates req.cmd's OLLAMA_HOST so the runner script
AND the auto-registered endpoint agree on the same bind port.
- Auto-register uses host.docker.internal (when running inside Docker)
instead of localhost, matching the URL /setup adds for Ollama by
hand. Local cookbook serves now produce a chat-reachable endpoint
on first launch.
- Cascade-delete: removing a scheduled cookbook task also deletes any
linked calendar event (cookbook_task_id marker in the description).
- Tasks list groups cookbook_serve under a "Cookbook" category that
sorts above the rest, so scheduler-launched serves are easy to find.
2026-06-05 14:41:43 +09:00
try {
await fetch ( '/api/shell/exec' , {
method : 'POST' , credentials : 'same-origin' ,
headers : { 'Content-Type' : 'application/json' } ,
body : JSON . stringify ( { command : ollamaUnload } ) ,
} ) ;
} catch ( _ ) { /* unload best-effort */ }
}
let killOk = true ;
try {
const r = await fetch ( '/api/shell/exec' , {
2026-06-02 22:38:55 +09:00
method : 'POST' , credentials : 'same-origin' ,
headers : { 'Content-Type' : 'application/json' } ,
Cookbook scheduler + serve: schedule via Tasks, Stop verifies kill, Ollama auto port-pick
- Schedule cookbook serves through the existing ScheduledTask system: the
serve preset gets a ^ button next to Launch that opens a daily/hourly/
weekly form mirroring the admin-switch style; the schedule action runs
action_cookbook_serve, which delegates to /api/model/serve and stamps
the resulting task with _scheduledStopAtMs. A background
cookbook_serve_lifecycle loop ticks every 60s and kills any serve
whose window has ended, also dropping the auto-registered endpoint
so the model picker doesn't keep pointing at a dead server.
- Stop and remove on a Running serve now awaits the SSH/tmux kill,
re-checks tmux has-session, and surfaces an error toast (leaving the
row) when the kill failed. Previously fire-and-forget, so a failed
SSH/tmux call silently left the live serve running while the row
vanished from the UI.
- Cookbook tasks/status orphan-adoption sweep no longer requires the
serve-/cookbook- session-id prefix; any tmux session whose pane is
running a known model-server process gets auto-pulled into Running.
Without this loosening, a cookbook-launched serve whose tmux id
fell back to a bare number was invisible — you couldn't see it,
let alone stop it.
- Ollama serve always launches a fresh process under cookbook's tmux
(no more monitor-mode reattach to a systemd/Docker ollama Stop can't
reach). The handler pre-picks a free port by probing the target
host over SSH and mutates req.cmd's OLLAMA_HOST so the runner script
AND the auto-registered endpoint agree on the same bind port.
- Auto-register uses host.docker.internal (when running inside Docker)
instead of localhost, matching the URL /setup adds for Ollama by
hand. Local cookbook serves now produce a chat-reachable endpoint
on first launch.
- Cascade-delete: removing a scheduled cookbook task also deletes any
linked calendar event (cookbook_task_id marker in the description).
- Tasks list groups cookbook_serve under a "Cookbook" category that
sorts above the rest, so scheduler-launched serves are easy to find.
2026-06-05 14:41:43 +09:00
body : JSON . stringify ( { command : _tmuxGracefulKill ( task ) } ) ,
} ) ;
if ( r . ok ) {
const out = await r . json ( ) ;
// Don't trust exit_code alone — tmux kill returns 0 even when
// there was nothing to kill. Verify the session is actually gone.
if ( task . sessionId && isLive ) {
try {
const probe = await fetch ( '/api/shell/exec' , {
method : 'POST' , credentials : 'same-origin' ,
headers : { 'Content-Type' : 'application/json' } ,
body : JSON . stringify ( { command : _tmuxCmd ( task , ` has-session -t ${ task . sessionId } ` ) } ) ,
} ) ;
if ( probe . ok ) {
const pj = await probe . json ( ) ;
// has-session exits 0 when session STILL exists; non-zero = gone.
if ( ( pj . exit _code || 0 ) === 0 ) killOk = false ;
}
} catch ( _ ) { /* probe best-effort; trust kill */ }
}
} else {
killOk = false ;
}
} catch ( _ ) { killOk = false ; }
if ( ! killOk ) {
try { uiModule . showToast ( 'Kill failed — session may still be running. Check `tmux ls` on the server.' , 'error' ) ; } catch ( _ ) { }
return ; // leave the row so the user can retry
2026-06-02 22:38:55 +09:00
}
2026-05-31 23:58:26 +09:00
if ( task . type === 'serve' && task . payload ) {
2026-06-02 22:38:55 +09:00
const endpointUrl = _endpointUrlForTask ( task , outputText ) ;
_removeEndpointByUrl ( endpointUrl ) ;
2026-05-31 23:58:26 +09:00
const modelName = task . payload . model || task . name || '' ;
if ( modelName ) {
fetch ( '/api/model-endpoints' , { credentials : 'same-origin' } )
. then ( r => r . json ( ) )
. then ( eps => {
2026-06-02 22:38:55 +09:00
const ep = eps . find ( e => e . name === modelName || e . base _url === endpointUrl ) ;
2026-05-31 23:58:26 +09:00
if ( ep ) fetch ( ` /api/model-endpoints/ ${ ep . id } ` , { method : 'DELETE' , credentials : 'same-origin' } ) . then ( ( ) => _refreshModelsAfterEndpointChange ( ) ) ;
} ) . catch ( ( ) => { } ) ;
}
}
_animateOutThenRemove ( el , task . sessionId ) ;
} ) ;
// Wire retry
el . querySelector ( '.cookbook-task-action-retry' ) . addEventListener ( 'click' , ( ) => _retryTask ( el , task ) ) ;
// Wire copy button
el . querySelector ( '.cookbook-output-copy' ) . addEventListener ( 'click' , ( e ) => {
e . stopPropagation ( ) ;
const text = el . querySelector ( '.cookbook-output-pre' ) ? . textContent || '' ;
2026-06-05 13:53:33 +01:00
if ( ! text . trim ( ) ) {
uiModule . showToast ( 'No log content available yet' ) ;
return ;
}
2026-05-31 23:58:26 +09:00
_copyText ( text ) . then ( ( ) => {
const btn = el . querySelector ( '.cookbook-output-copy' ) ;
const origHTML = btn . innerHTML ;
btn . innerHTML = '<svg width="14" height="14" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2.5" stroke-linecap="round" stroke-linejoin="round"><polyline points="20 6 9 17 4 12"/></svg>' ;
btn . classList . add ( 'copied' ) ;
setTimeout ( ( ) => { btn . innerHTML = origHTML ; btn . classList . remove ( 'copied' ) ; } , 1500 ) ;
} ) ;
} ) ;
// Route to the right server section body
2026-06-21 11:02:35 +00:00
const serverBodyId = ` server-body- ${ ( _taskServerKey ( task ) || 'local' ) . replace ( /[^a-zA-Z0-9-]/g , '_' ) } ` ;
2026-05-31 23:58:26 +09:00
const targetBody = document . getElementById ( serverBodyId ) ;
if ( targetBody ) targetBody . appendChild ( el ) ;
else group . appendChild ( el ) ;
Cookbook scheduler + serve: schedule via Tasks, Stop verifies kill, Ollama auto port-pick
- Schedule cookbook serves through the existing ScheduledTask system: the
serve preset gets a ^ button next to Launch that opens a daily/hourly/
weekly form mirroring the admin-switch style; the schedule action runs
action_cookbook_serve, which delegates to /api/model/serve and stamps
the resulting task with _scheduledStopAtMs. A background
cookbook_serve_lifecycle loop ticks every 60s and kills any serve
whose window has ended, also dropping the auto-registered endpoint
so the model picker doesn't keep pointing at a dead server.
- Stop and remove on a Running serve now awaits the SSH/tmux kill,
re-checks tmux has-session, and surfaces an error toast (leaving the
row) when the kill failed. Previously fire-and-forget, so a failed
SSH/tmux call silently left the live serve running while the row
vanished from the UI.
- Cookbook tasks/status orphan-adoption sweep no longer requires the
serve-/cookbook- session-id prefix; any tmux session whose pane is
running a known model-server process gets auto-pulled into Running.
Without this loosening, a cookbook-launched serve whose tmux id
fell back to a bare number was invisible — you couldn't see it,
let alone stop it.
- Ollama serve always launches a fresh process under cookbook's tmux
(no more monitor-mode reattach to a systemd/Docker ollama Stop can't
reach). The handler pre-picks a free port by probing the target
host over SSH and mutates req.cmd's OLLAMA_HOST so the runner script
AND the auto-registered endpoint agree on the same bind port.
- Auto-register uses host.docker.internal (when running inside Docker)
instead of localhost, matching the URL /setup adds for Ollama by
hand. Local cookbook serves now produce a chat-reachable endpoint
on first launch.
- Cascade-delete: removing a scheduled cookbook task also deletes any
linked calendar event (cookbook_task_id marker in the description).
- Tasks list groups cookbook_serve under a "Cookbook" category that
sorts above the rest, so scheduler-launched serves are easy to find.
2026-06-05 14:41:43 +09:00
// Auto-attach the tmux output stream for any task whose underlying
// session could still be alive — not just 'running'. Scheduler-
// launched serves transition to 'ready' as soon as /v1/models
// responds; without this, the user opens the Running tab and sees
// only the placeholder ("Launched by scheduled task …") because
// _reconnectTask never fires for status 'ready'/'loading'/'warming'.
2026-07-07 00:50:07 +00:00
const _wrapForStream = el . querySelector ( '.cookbook-output-wrap' ) ;
const _streamExpanded = _wrapForStream && ! _wrapForStream . classList . contains ( 'cookbook-task-collapsed' ) ;
if ( _isRunningTabVisible ( ) && _streamExpanded && task . sessionId && [ 'serve' , 'download' ] . includes ( task . type || '' ) ) {
2026-05-31 23:58:26 +09:00
_reconnectTask ( el , task ) ;
}
}
if ( tasks . some ( t => t . status === 'running' ) ) _startWaveSync ( ) ;
// Re-apply captured expansion state so re-renders don't fold open tasks/sections.
_collapsedTaskIds . forEach ( ( id ) => {
const wrap = body . querySelector ( ` .cookbook-task[data-task-id=" ${ id } "] .cookbook-output-wrap ` ) ;
if ( wrap ) wrap . classList . add ( 'cookbook-task-collapsed' ) ;
} ) ;
// Mobile defaults to collapsed (above), so re-open whatever the user had
// explicitly expanded before this re-render.
if ( _mobileCollapseDefault ) {
_expandedTaskIds . forEach ( ( id ) => {
const wrap = body . querySelector ( ` .cookbook-task[data-task-id=" ${ id } "] .cookbook-output-wrap ` ) ;
if ( wrap ) wrap . classList . remove ( 'cookbook-task-collapsed' ) ;
} ) ;
}
_collapsedSectionIds . forEach ( ( sid ) => {
const sb = document . getElementById ( sid ) ;
if ( sb ) sb . style . display = 'none' ;
const hdr = body . querySelector ( ` .cookbook-section-header[data-collapse=" ${ sid } "] ` ) ;
const chevron = hdr ? . querySelector ( '.cookbook-section-chevron' ) ;
if ( chevron ) { chevron . style . transform = 'rotate(-90deg)' ; chevron . style . opacity = '' ; }
} ) ;
}
// ── Reconnect task (polling loop) ──
async function _reconnectTask ( el , task ) {
2026-07-07 00:50:07 +00:00
if ( ! el || ! task ) return ;
const wrap = el . querySelector ( '.cookbook-output-wrap' ) ;
if ( ! _isRunningTabVisible ( ) || ! wrap || wrap . classList . contains ( 'cookbook-task-collapsed' ) ) return ;
if ( el . _abort && ! el . _abort . signal ? . aborted ) return ;
2026-05-31 23:58:26 +09:00
const output = el . querySelector ( '.cookbook-output-pre' ) ;
2026-07-07 00:50:07 +00:00
if ( ! output ) return ;
2026-05-31 23:58:26 +09:00
const controller = new AbortController ( ) ;
el . _abort = controller ;
let failCount = 0 ;
while ( ! controller . signal . aborted ) {
2026-07-07 00:50:07 +00:00
const liveWrap = el . querySelector ( '.cookbook-output-wrap' ) ;
if ( ! el . isConnected || ! _isRunningTabVisible ( ) || ! liveWrap || liveWrap . classList . contains ( 'cookbook-task-collapsed' ) ) {
2026-05-31 23:58:26 +09:00
controller . abort ( ) ;
break ;
}
try {
const res = await fetch ( '/api/shell/exec' , {
method : 'POST' , credentials : 'same-origin' ,
headers : { 'Content-Type' : 'application/json' } ,
2026-06-30 01:47:48 +00:00
body : JSON . stringify ( { command : _tmuxCmd ( task , ` capture-pane -t ${ task . sessionId } -p -S -500 ` ) , timeout : 15 } ) ,
2026-05-31 23:58:26 +09:00
} ) ;
const data = await res . json ( ) ;
if ( data . exit _code !== 0 ) {
failCount ++ ;
if ( failCount < 5 ) {
cookbook agent debug loop: persistent log files, auto-adopt orphan tmux, Codex/Claude skill parity
Three converging fixes so the chat agent + external Codex/Claude skills can actually debug a crashed serve instead of staring at a post-crash neofetch banner:
* Serves now `tee` to /tmp/odysseus-tmux/SESSION.log on the host running them. Runner saves fds 3/4 before the tee and restores them right before `exec ${SHELL}`, so the post-crash interactive zsh banner does NOT pollute the log file.
* `tail_serve_output` (chat agent) and `/api/codex/cookbook/output/{sid}` (Codex+Claude skills) both prefer the persistent log file over the tmux pane. Pane is fallback for sessions predating the tee runner. Default tail bumped 150 -> 400.
* `list_served_models` "recent log" snippet seeks to the Traceback line instead of showing the last 6 lines (which was always the bash prompt).
Cookbook auto-adoption sweep on `/api/cookbook/tasks/status`: every 20s (rate-limited) the cookbook SSHes each configured server, finds `serve-*` / `cookbook-*` tmux sessions running an actual model process (vllm/python/llama-server/etc., filtered via `pane_current_command`), and writes them into state.tasks. So when the agent falls back to raw ssh+tmux, the session appears in the Cookbook UI on the next poll.
`serve_model` error path now reads `data["detail"]` in addition to `data["error"]` so the FastAPI HTTPException message ("Invalid characters in cmd") actually reaches the agent instead of being swallowed as a generic "Serve failed". Tool description updated to warn against `cd …`/`source …`/`&&` prefixes.
Intent-without-action supervisor in agent_loop: when the model writes "Let me tail the output" / "I'll check the logs" / "Let me investigate" and ends the turn without emitting a tool call, the loop injects a sharp system nudge ("You said you would X — DO IT NOW") and continues. Capped at 2 nudges per chat so a model that genuinely cannot use the tool does not pin the loop.
Codex/Claude skill parity: adds `/cookbook/cached`, `/cookbook/presets`, `/cookbook/preset/{name}`, `/cookbook/adopt` so external agents have the same surface as the chat agent. SKILL.md docs + odysseus_api.py wrapper updated for both bundles.
`adopt_served_model` promoted to the always-on tool set so the agent has a documented fallback when serve_model rejects a cmd.
Also various cookbook UI tweaks accumulated alongside the above (cookbook.js, cookbookRunning.js, cookbookServe.js, cookbook-diagnosis.js, settings.js, style.css).
2026-06-04 23:27:18 +09:00
await new Promise ( r => setTimeout ( r , 3000 ) ) ;
2026-05-31 23:58:26 +09:00
continue ;
}
try {
const verify = await fetch ( '/api/shell/exec' , {
method : 'POST' , credentials : 'same-origin' ,
headers : { 'Content-Type' : 'application/json' } ,
body : JSON . stringify ( { command : _tmuxCmd ( task , ` has-session -t ${ task . sessionId } ` ) } ) ,
} ) ;
const vData = await verify . json ( ) ;
if ( vData . exit _code === 0 ) {
failCount = 0 ;
await new Promise ( r => setTimeout ( r , 5000 ) ) ;
continue ;
}
} catch {
await new Promise ( r => setTimeout ( r , 10000 ) ) ;
continue ;
}
const lastOutput = output . textContent || '' ;
cookbook agent debug loop: persistent log files, auto-adopt orphan tmux, Codex/Claude skill parity
Three converging fixes so the chat agent + external Codex/Claude skills can actually debug a crashed serve instead of staring at a post-crash neofetch banner:
* Serves now `tee` to /tmp/odysseus-tmux/SESSION.log on the host running them. Runner saves fds 3/4 before the tee and restores them right before `exec ${SHELL}`, so the post-crash interactive zsh banner does NOT pollute the log file.
* `tail_serve_output` (chat agent) and `/api/codex/cookbook/output/{sid}` (Codex+Claude skills) both prefer the persistent log file over the tmux pane. Pane is fallback for sessions predating the tee runner. Default tail bumped 150 -> 400.
* `list_served_models` "recent log" snippet seeks to the Traceback line instead of showing the last 6 lines (which was always the bash prompt).
Cookbook auto-adoption sweep on `/api/cookbook/tasks/status`: every 20s (rate-limited) the cookbook SSHes each configured server, finds `serve-*` / `cookbook-*` tmux sessions running an actual model process (vllm/python/llama-server/etc., filtered via `pane_current_command`), and writes them into state.tasks. So when the agent falls back to raw ssh+tmux, the session appears in the Cookbook UI on the next poll.
`serve_model` error path now reads `data["detail"]` in addition to `data["error"]` so the FastAPI HTTPException message ("Invalid characters in cmd") actually reaches the agent instead of being swallowed as a generic "Serve failed". Tool description updated to warn against `cd …`/`source …`/`&&` prefixes.
Intent-without-action supervisor in agent_loop: when the model writes "Let me tail the output" / "I'll check the logs" / "Let me investigate" and ends the turn without emitting a tool call, the loop injects a sharp system nudge ("You said you would X — DO IT NOW") and continues. Capped at 2 nudges per chat so a model that genuinely cannot use the tool does not pin the loop.
Codex/Claude skill parity: adds `/cookbook/cached`, `/cookbook/presets`, `/cookbook/preset/{name}`, `/cookbook/adopt` so external agents have the same surface as the chat agent. SKILL.md docs + odysseus_api.py wrapper updated for both bundles.
`adopt_served_model` promoted to the always-on tool set so the agent has a documented fallback when serve_model rejects a cmd.
Also various cookbook UI tweaks accumulated alongside the above (cookbook.js, cookbookRunning.js, cookbookServe.js, cookbook-diagnosis.js, settings.js, style.css).
2026-06-04 23:27:18 +09:00
// Pip tasks (Reinstall vLLM / Upgrade torch / etc.) must skip the
// generic serve `_diagnose` step. Their output is pip's own and the
// error patterns there (torch ABI traceback, "No module named torch",
// etc.) are routinely matched against the previous tmux scrollback,
// tagging a clean pip success as a crashed serve. Detection is the
// same shape as the looksSuccessful branch below.
const _isPipTaskDiag = ( ( task . payload ? . repo _id || '' ) . startsWith ( 'pip-' ) )
|| /python3? -m pip\b/ . test ( task . payload ? . _cmd || '' ) ;
const diag = _isPipTaskDiag ? null : _diagnose ( lastOutput ) ;
2026-05-31 23:58:26 +09:00
if ( diag ) {
let diagEl = el . querySelector ( '.cookbook-diagnosis' ) ;
if ( ! diagEl ) {
diagEl = document . createElement ( 'div' ) ;
diagEl . className = 'cookbook-diagnosis' ;
el . appendChild ( diagEl ) ;
}
_showDiagnosis ( el , diag , lastOutput ) ;
_updateTask ( task . sessionId , { status : 'error' } ) ;
el . dataset . status = 'error' ;
const badge = el . querySelector ( '.cookbook-task-status' ) ;
if ( badge ) { badge . textContent = _statusLabel ( 'error' , task . type ) ; badge . className = 'cookbook-task-status cookbook-task-error' ; }
_showCookbookNotif ( true ) ;
} else {
2026-06-02 12:15:41 +09:00
const downloadLooksSuccessful = ! lastOutput . includes ( 'DOWNLOAD_FAILED' )
&& ( lastOutput . includes ( 'DONE' ) || lastOutput . includes ( '100%' ) || lastOutput . includes ( '/snapshots/' ) || lastOutput . includes ( 'Download complete' ) || lastOutput . includes ( 'DOWNLOAD_OK' ) ) ;
cookbook agent debug loop: persistent log files, auto-adopt orphan tmux, Codex/Claude skill parity
Three converging fixes so the chat agent + external Codex/Claude skills can actually debug a crashed serve instead of staring at a post-crash neofetch banner:
* Serves now `tee` to /tmp/odysseus-tmux/SESSION.log on the host running them. Runner saves fds 3/4 before the tee and restores them right before `exec ${SHELL}`, so the post-crash interactive zsh banner does NOT pollute the log file.
* `tail_serve_output` (chat agent) and `/api/codex/cookbook/output/{sid}` (Codex+Claude skills) both prefer the persistent log file over the tmux pane. Pane is fallback for sessions predating the tee runner. Default tail bumped 150 -> 400.
* `list_served_models` "recent log" snippet seeks to the Traceback line instead of showing the last 6 lines (which was always the bash prompt).
Cookbook auto-adoption sweep on `/api/cookbook/tasks/status`: every 20s (rate-limited) the cookbook SSHes each configured server, finds `serve-*` / `cookbook-*` tmux sessions running an actual model process (vllm/python/llama-server/etc., filtered via `pane_current_command`), and writes them into state.tasks. So when the agent falls back to raw ssh+tmux, the session appears in the Cookbook UI on the next poll.
`serve_model` error path now reads `data["detail"]` in addition to `data["error"]` so the FastAPI HTTPException message ("Invalid characters in cmd") actually reaches the agent instead of being swallowed as a generic "Serve failed". Tool description updated to warn against `cd …`/`source …`/`&&` prefixes.
Intent-without-action supervisor in agent_loop: when the model writes "Let me tail the output" / "I'll check the logs" / "Let me investigate" and ends the turn without emitting a tool call, the loop injects a sharp system nudge ("You said you would X — DO IT NOW") and continues. Capped at 2 nudges per chat so a model that genuinely cannot use the tool does not pin the loop.
Codex/Claude skill parity: adds `/cookbook/cached`, `/cookbook/presets`, `/cookbook/preset/{name}`, `/cookbook/adopt` so external agents have the same surface as the chat agent. SKILL.md docs + odysseus_api.py wrapper updated for both bundles.
`adopt_served_model` promoted to the always-on tool set so the agent has a documented fallback when serve_model rejects a cmd.
Also various cookbook UI tweaks accumulated alongside the above (cookbook.js, cookbookRunning.js, cookbookServe.js, cookbook-diagnosis.js, settings.js, style.css).
2026-06-04 23:27:18 +09:00
// Pip install / reinstall tasks are launched via _launchServeTask (so
// they show up in the Running tab + use tmux) but they aren't real
// serves — the cmd is `python3 -m pip ...` and the success markers
// are pip's own. Without this branch, a successful reinstall ends
// with no "Uvicorn running on" line and gets mis-flagged as a crashed
// serve.
const _isPipTask = ( ( task . payload ? . repo _id || '' ) . startsWith ( 'pip-' ) )
|| /python3? -m pip\b/ . test ( task . payload ? . _cmd || '' ) ;
const pipLooksSuccessful = _isPipTask
&& /Successfully installed|Requirement already (?:satisfied|up-to-date)/i . test ( lastOutput )
&& ! /error:|ERROR:/ . test ( lastOutput . slice ( - 1024 ) ) ;
2026-06-02 12:15:41 +09:00
const serveLooksReady = task . type === 'serve' && _serveOutputLooksReady ( { ... task , output : lastOutput } ) ;
2026-06-04 17:25:06 +05:30
// Dependency installs are tracked as download tasks but finish with a
2026-06-05 11:23:15 +09:00
// pip exit-0 sentinel, not HF download markers — check that too.
// Standalone pip-* serves finish with pip's own success line, not
// HF or "Uvicorn running on".
2026-06-04 17:25:06 +05:30
const depInstallSucceeded = ! ! task . payload ? . _dep && _depInstallSucceeded ( lastOutput ) ;
2026-06-05 11:23:15 +09:00
const looksSuccessful = depInstallSucceeded
|| ( task . type === 'download'
? downloadLooksSuccessful
: ( _isPipTask ? pipLooksSuccessful : serveLooksReady ) ) ;
2026-06-02 12:15:41 +09:00
if ( ! lastOutput . trim ( ) || ! looksSuccessful ) {
2026-05-31 23:58:26 +09:00
_updateTask ( task . sessionId , { status : 'crashed' } ) ;
el . dataset . status = 'crashed' ;
const badge = el . querySelector ( '.cookbook-task-status' ) ;
if ( badge ) { badge . textContent = _statusLabel ( 'crashed' , task . type ) ; badge . className = 'cookbook-task-status cookbook-task-crashed' ; }
cookbook agent debug loop: persistent log files, auto-adopt orphan tmux, Codex/Claude skill parity
Three converging fixes so the chat agent + external Codex/Claude skills can actually debug a crashed serve instead of staring at a post-crash neofetch banner:
* Serves now `tee` to /tmp/odysseus-tmux/SESSION.log on the host running them. Runner saves fds 3/4 before the tee and restores them right before `exec ${SHELL}`, so the post-crash interactive zsh banner does NOT pollute the log file.
* `tail_serve_output` (chat agent) and `/api/codex/cookbook/output/{sid}` (Codex+Claude skills) both prefer the persistent log file over the tmux pane. Pane is fallback for sessions predating the tee runner. Default tail bumped 150 -> 400.
* `list_served_models` "recent log" snippet seeks to the Traceback line instead of showing the last 6 lines (which was always the bash prompt).
Cookbook auto-adoption sweep on `/api/cookbook/tasks/status`: every 20s (rate-limited) the cookbook SSHes each configured server, finds `serve-*` / `cookbook-*` tmux sessions running an actual model process (vllm/python/llama-server/etc., filtered via `pane_current_command`), and writes them into state.tasks. So when the agent falls back to raw ssh+tmux, the session appears in the Cookbook UI on the next poll.
`serve_model` error path now reads `data["detail"]` in addition to `data["error"]` so the FastAPI HTTPException message ("Invalid characters in cmd") actually reaches the agent instead of being swallowed as a generic "Serve failed". Tool description updated to warn against `cd …`/`source …`/`&&` prefixes.
Intent-without-action supervisor in agent_loop: when the model writes "Let me tail the output" / "I'll check the logs" / "Let me investigate" and ends the turn without emitting a tool call, the loop injects a sharp system nudge ("You said you would X — DO IT NOW") and continues. Capped at 2 nudges per chat so a model that genuinely cannot use the tool does not pin the loop.
Codex/Claude skill parity: adds `/cookbook/cached`, `/cookbook/presets`, `/cookbook/preset/{name}`, `/cookbook/adopt` so external agents have the same surface as the chat agent. SKILL.md docs + odysseus_api.py wrapper updated for both bundles.
`adopt_served_model` promoted to the always-on tool set so the agent has a documented fallback when serve_model rejects a cmd.
Also various cookbook UI tweaks accumulated alongside the above (cookbook.js, cookbookRunning.js, cookbookServe.js, cookbook-diagnosis.js, settings.js, style.css).
2026-06-04 23:27:18 +09:00
if ( _isPipTask ) {
// Pip tasks: don't run the serve diagnosis (which would yell
// "Serve stopped before the model became reachable"). Show a
// pip-tailored message; the user can read pip's own error output
// directly above.
const _ranOk = /Successfully installed|Requirement already (?:satisfied|up-to-date)/i . test ( lastOutput ) ;
if ( ! _ranOk ) {
_showDiagnosis ( el , {
message : 'Pip install did not finish with a success marker. Check the output for the underlying error.' ,
suggestion : 'Suggested action: copy the troubleshooting bundle. Common causes: missing build deps, network blip, mismatched torch ABI.' ,
fixes : [ ] ,
} , lastOutput ) ;
}
} else if ( task . type === 'serve' ) {
2026-06-02 12:15:41 +09:00
const diag = _diagnose ( lastOutput ) || {
message : _serveTaskLooksAwqOnLocalBackend ( task , lastOutput )
? 'AWQ/GPTQ/FP8 cannot be served through llama.cpp/Ollama unified-memory mode.'
: /Native llama-server not found|building llama-server|llama\.cpp/i . test ( lastOutput )
? 'llama.cpp build stopped before the server became reachable.'
: 'Serve stopped before the model became reachable.' ,
suggestion : _serveTaskLooksAwqOnLocalBackend ( task , lastOutput )
? 'Suggested action: use vLLM/SGLang on a compatible CUDA/ROCm GPU server, or download a GGUF version for llama.cpp/Ollama/unified-memory serving.'
: /Native llama-server not found|building llama-server|llama\.cpp/i . test ( lastOutput )
? 'Suggested action: copy the troubleshooting bundle, then edit serve settings. For the quickest local/CPU path, use Ollama or a prebuilt llama-server; source builds can take several minutes and fail if build dependencies are incomplete.'
: 'Suggested action: copy the troubleshooting bundle, then edit serve settings or relaunch with a CPU/backend fallback.' ,
fixes : [ { label : 'Edit serve' , action : ( panel ) => _openServeEditForTask ( task ) } ] ,
} ;
_showDiagnosis ( el , diag , lastOutput ) ;
2026-06-02 22:38:55 +09:00
} else if ( task . type === 'download' ) {
const isDisk = /no space left|disk quota|enospc/i . test ( lastOutput ) ;
const isNetwork = /connection|timeout|timed out|incompleteread|chunkedencoding|reset by peer|protocolerror|all connection attempts failed/i . test ( lastOutput ) ;
const progressMatch = String ( lastOutput || '' ) . match ( /(\d+)%\|/ ) ;
const nearDone = progressMatch && Number ( progressMatch [ 1 ] ) >= 80 ;
feat: Claude Agent integration + cookbook reconnect + UI polish
- Claude Agent integration: AGENT_CONFIGS.claude, INTG_TYPES.claude,
setup_claude_routes + integrations/claude/ skill bundle. Wired in
app.py alongside the existing Codex integration; same scope-gated
/api/codex/* backend; agent form has new description so users know
it's setup for an external CLI, not an agent streamed inside Odysseus.
- Remove mark_email_boundaries action: not good enough yet. Stripped
from task UI, scheduler defaults, registry, tool schema, clear-cache
route. Added to RETIRED_HOUSEKEEPING_ACTIONS so existing rows + their
task_runs auto-purge on startup.
- Cookbook download reliability: "Reconnect" fix button in the crash
diagnosis runs _reconnectTask after probing has-session. 30s confirm
window before marking a download "done" — kills the Finished/Downloading
flicker when tmux briefly drops between captures.
- Mobile UX: tap anywhere on a note card body opens the editor;
Update button morphs to Archive when no text was edited; bell icon
accent-colored; chip-trashing notif pills fade so only the icon
rotates into the trash zone.
- Settings integrations: SVG-per-provider in email + API preset
dropdowns, custom drop-up-aware menus, accent sub-header icons
(IMAP/SMTP), consistent card styling between list + edit, contacts
Edit/Delete icons, agent form description copy.
2026-06-04 08:27:26 +09:00
// Reconnect: most "crashed" downloads near the end are actually
// finished — we just missed the DOWNLOAD_OK / /snapshots/ marker
// because output rolled over, or the tmux session ended a tick
// before we polled. Probing has-session and re-attaching to
// capture-pane lets the existing _reconnectTask flow pick up
// the real state (running, finished, or truly dead).
const _reconnectFix = {
2026-06-13 22:11:45 +09:00
label : 'Reconnect tmux' ,
feat: Claude Agent integration + cookbook reconnect + UI polish
- Claude Agent integration: AGENT_CONFIGS.claude, INTG_TYPES.claude,
setup_claude_routes + integrations/claude/ skill bundle. Wired in
app.py alongside the existing Codex integration; same scope-gated
/api/codex/* backend; agent form has new description so users know
it's setup for an external CLI, not an agent streamed inside Odysseus.
- Remove mark_email_boundaries action: not good enough yet. Stripped
from task UI, scheduler defaults, registry, tool schema, clear-cache
route. Added to RETIRED_HOUSEKEEPING_ACTIONS so existing rows + their
task_runs auto-purge on startup.
- Cookbook download reliability: "Reconnect" fix button in the crash
diagnosis runs _reconnectTask after probing has-session. 30s confirm
window before marking a download "done" — kills the Finished/Downloading
flicker when tmux briefly drops between captures.
- Mobile UX: tap anywhere on a note card body opens the editor;
Update button morphs to Archive when no text was edited; bell icon
accent-colored; chip-trashing notif pills fade so only the icon
rotates into the trash zone.
- Settings integrations: SVG-per-provider in email + API preset
dropdowns, custom drop-up-aware menus, accent sub-header icons
(IMAP/SMTP), consistent card styling between list + edit, contacts
Edit/Delete icons, agent form description copy.
2026-06-04 08:27:26 +09:00
action : ( ) => {
_updateTask ( task . sessionId , { status : 'running' } ) ;
el . dataset . status = 'running' ;
const badge2 = el . querySelector ( '.cookbook-task-status' ) ;
if ( badge2 ) { badge2 . textContent = _statusLabel ( 'running' , task . type ) ; badge2 . className = 'cookbook-task-status' ; }
const _diagEl = el . querySelector ( '.cookbook-diagnosis' ) ;
if ( _diagEl ) _diagEl . remove ( ) ;
const _wave = el . querySelector ( '.cookbook-task-wave' ) ; if ( _wave ) _wave . style . display = '' ;
const _up = el . querySelector ( '.cookbook-task-uptime' ) ; if ( _up ) _up . style . display = '' ;
_reconnectTask ( el , task ) ;
} ,
} ;
2026-06-02 22:38:55 +09:00
const diag = {
message : isDisk
? 'Download stopped because this server ran out of disk space.'
: isNetwork
? 'Download stopped after the HuggingFace connection was interrupted.'
: nearDone
? 'Download stopped near the end before the final completion marker was captured.'
: 'Download stopped before HuggingFace reported completion.' ,
suggestion : isDisk
? 'Suggested action: free disk space, then retry the download. HuggingFace resumes incomplete files when possible.'
: nearDone
feat: Claude Agent integration + cookbook reconnect + UI polish
- Claude Agent integration: AGENT_CONFIGS.claude, INTG_TYPES.claude,
setup_claude_routes + integrations/claude/ skill bundle. Wired in
app.py alongside the existing Codex integration; same scope-gated
/api/codex/* backend; agent form has new description so users know
it's setup for an external CLI, not an agent streamed inside Odysseus.
- Remove mark_email_boundaries action: not good enough yet. Stripped
from task UI, scheduler defaults, registry, tool schema, clear-cache
route. Added to RETIRED_HOUSEKEEPING_ACTIONS so existing rows + their
task_runs auto-purge on startup.
- Cookbook download reliability: "Reconnect" fix button in the crash
diagnosis runs _reconnectTask after probing has-session. 30s confirm
window before marking a download "done" — kills the Finished/Downloading
flicker when tmux briefly drops between captures.
- Mobile UX: tap anywhere on a note card body opens the editor;
Update button morphs to Archive when no text was edited; bell icon
accent-colored; chip-trashing notif pills fade so only the icon
rotates into the trash zone.
- Settings integrations: SVG-per-provider in email + API preset
dropdowns, custom drop-up-aware menus, accent sub-header icons
(IMAP/SMTP), consistent card styling between list + edit, contacts
Edit/Delete icons, agent form description copy.
2026-06-04 08:27:26 +09:00
? 'Suggested action: hit Reconnect first — the download may have finished after the output buffer rolled over. Retry only if reconnect cannot recover.'
: 'Suggested action: hit Reconnect to re-attach to the tmux session. If that fails, retry — HuggingFace resumes incomplete files when possible.' ,
fixes : isDisk
? [
{ label : 'Retry download' , action : ( ) => _retryTask ( el , task ) } ,
{ label : 'Copy last 50 lines' , action : ( ) => {
const last = String ( lastOutput || '' ) . split ( '\n' ) . slice ( - 50 ) . join ( '\n' ) ;
_copyText ( last || 'No download log available.' ) ;
} } ,
]
: [
_reconnectFix ,
{ label : 'Retry download' , action : ( ) => _retryTask ( el , task ) } ,
{ label : 'Copy last 50 lines' , action : ( ) => {
const last = String ( lastOutput || '' ) . split ( '\n' ) . slice ( - 50 ) . join ( '\n' ) ;
_copyText ( last || 'No download log available.' ) ;
} } ,
] ,
2026-06-02 22:38:55 +09:00
} ;
_showDiagnosis ( el , diag , lastOutput ) ;
feat: Claude Agent integration + cookbook reconnect + UI polish
- Claude Agent integration: AGENT_CONFIGS.claude, INTG_TYPES.claude,
setup_claude_routes + integrations/claude/ skill bundle. Wired in
app.py alongside the existing Codex integration; same scope-gated
/api/codex/* backend; agent form has new description so users know
it's setup for an external CLI, not an agent streamed inside Odysseus.
- Remove mark_email_boundaries action: not good enough yet. Stripped
from task UI, scheduler defaults, registry, tool schema, clear-cache
route. Added to RETIRED_HOUSEKEEPING_ACTIONS so existing rows + their
task_runs auto-purge on startup.
- Cookbook download reliability: "Reconnect" fix button in the crash
diagnosis runs _reconnectTask after probing has-session. 30s confirm
window before marking a download "done" — kills the Finished/Downloading
flicker when tmux briefly drops between captures.
- Mobile UX: tap anywhere on a note card body opens the editor;
Update button morphs to Archive when no text was edited; bell icon
accent-colored; chip-trashing notif pills fade so only the icon
rotates into the trash zone.
- Settings integrations: SVG-per-provider in email + API preset
dropdowns, custom drop-up-aware menus, accent sub-header icons
(IMAP/SMTP), consistent card styling between list + edit, contacts
Edit/Delete icons, agent form description copy.
2026-06-04 08:27:26 +09:00
// Auto-probe: if the tmux session is still alive (download
// genuinely still in progress), _selfHealStaleTasks flips the
// task back to running and the diagnosis disappears without
// the user needing to click Reconnect.
if ( nearDone ) setTimeout ( ( ) => { _selfHealStaleTasks ( ) . catch ( ( ) => { } ) ; } , 1200 ) ;
2026-06-02 12:15:41 +09:00
}
2026-05-31 23:58:26 +09:00
_showCookbookNotif ( true ) ;
} else {
cookbook agent debug loop: persistent log files, auto-adopt orphan tmux, Codex/Claude skill parity
Three converging fixes so the chat agent + external Codex/Claude skills can actually debug a crashed serve instead of staring at a post-crash neofetch banner:
* Serves now `tee` to /tmp/odysseus-tmux/SESSION.log on the host running them. Runner saves fds 3/4 before the tee and restores them right before `exec ${SHELL}`, so the post-crash interactive zsh banner does NOT pollute the log file.
* `tail_serve_output` (chat agent) and `/api/codex/cookbook/output/{sid}` (Codex+Claude skills) both prefer the persistent log file over the tmux pane. Pane is fallback for sessions predating the tee runner. Default tail bumped 150 -> 400.
* `list_served_models` "recent log" snippet seeks to the Traceback line instead of showing the last 6 lines (which was always the bash prompt).
Cookbook auto-adoption sweep on `/api/cookbook/tasks/status`: every 20s (rate-limited) the cookbook SSHes each configured server, finds `serve-*` / `cookbook-*` tmux sessions running an actual model process (vllm/python/llama-server/etc., filtered via `pane_current_command`), and writes them into state.tasks. So when the agent falls back to raw ssh+tmux, the session appears in the Cookbook UI on the next poll.
`serve_model` error path now reads `data["detail"]` in addition to `data["error"]` so the FastAPI HTTPException message ("Invalid characters in cmd") actually reaches the agent instead of being swallowed as a generic "Serve failed". Tool description updated to warn against `cd …`/`source …`/`&&` prefixes.
Intent-without-action supervisor in agent_loop: when the model writes "Let me tail the output" / "I'll check the logs" / "Let me investigate" and ends the turn without emitting a tool call, the loop injects a sharp system nudge ("You said you would X — DO IT NOW") and continues. Capped at 2 nudges per chat so a model that genuinely cannot use the tool does not pin the loop.
Codex/Claude skill parity: adds `/cookbook/cached`, `/cookbook/presets`, `/cookbook/preset/{name}`, `/cookbook/adopt` so external agents have the same surface as the chat agent. SKILL.md docs + odysseus_api.py wrapper updated for both bundles.
`adopt_served_model` promoted to the always-on tool set so the agent has a documented fallback when serve_model rejects a cmd.
Also various cookbook UI tweaks accumulated alongside the above (cookbook.js, cookbookRunning.js, cookbookServe.js, cookbook-diagnosis.js, settings.js, style.css).
2026-06-04 23:27:18 +09:00
// Strong completion markers — `DOWNLOAD_OK` is emitted by our
// downloader wrapper AFTER the model snapshot is on disk, and
// `/snapshots/` only appears once HF has resolved the cached
// tree. Either is conclusive. Finalize as done immediately, skip
// the 30s debounce — the debounce only exists to guard against
// ambiguous markers (bare "100%" / "Download complete") which can
// appear mid-stream during multi-file downloads.
const _strongDone = task . type === 'download'
&& ( lastOutput . includes ( 'DOWNLOAD_OK' ) || lastOutput . includes ( '/snapshots/' ) ) ;
if ( _strongDone ) {
_updateTask ( task . sessionId , { status : 'done' , _doneConfirmAt : null , _lastStatusFlipAt : Date . now ( ) } ) ;
el . dataset . status = 'done' ;
const badge = el . querySelector ( '.cookbook-task-status' ) ;
if ( badge ) { badge . textContent = _statusLabel ( 'done' , task . type ) ; badge . className = 'cookbook-task-status cookbook-task-done' ; }
const _chk = el . querySelector ( '.cookbook-task-check' ) ; if ( _chk ) _chk . style . display = '' ;
const _sb = el . querySelector ( '.cookbook-task-serve-btn' ) ; if ( _sb ) _sb . style . display = '' ;
_showCookbookNotif ( ) ;
_refreshDepsAfterInstall ( task ) ;
_renderRunningTab ( ) ;
_processQueue ( ) ;
break ;
}
feat: Claude Agent integration + cookbook reconnect + UI polish
- Claude Agent integration: AGENT_CONFIGS.claude, INTG_TYPES.claude,
setup_claude_routes + integrations/claude/ skill bundle. Wired in
app.py alongside the existing Codex integration; same scope-gated
/api/codex/* backend; agent form has new description so users know
it's setup for an external CLI, not an agent streamed inside Odysseus.
- Remove mark_email_boundaries action: not good enough yet. Stripped
from task UI, scheduler defaults, registry, tool schema, clear-cache
route. Added to RETIRED_HOUSEKEEPING_ACTIONS so existing rows + their
task_runs auto-purge on startup.
- Cookbook download reliability: "Reconnect" fix button in the crash
diagnosis runs _reconnectTask after probing has-session. 30s confirm
window before marking a download "done" — kills the Finished/Downloading
flicker when tmux briefly drops between captures.
- Mobile UX: tap anywhere on a note card body opens the editor;
Update button morphs to Archive when no text was edited; bell icon
accent-colored; chip-trashing notif pills fade so only the icon
rotates into the trash zone.
- Settings integrations: SVG-per-provider in email + API preset
dropdowns, custom drop-up-aware menus, accent sub-header icons
(IMAP/SMTP), consistent card styling between list + edit, contacts
Edit/Delete icons, agent form description copy.
2026-06-04 08:27:26 +09:00
// Debounce the done flip. Tmux capture-pane can fail transiently
// (network blip, ssh reconnect), and the verify has-session right
// above can briefly report dead even when the session is in the
// middle of finalizing. Marking done immediately + the periodic
// _selfHealStaleTasks then flipping back to running causes the
// status badge to oscillate between Finished and Downloading.
// Wait 30s and re-probe: only finalize as done if tmux is STILL
// gone. If the session resurfaces, restart _reconnectTask so live
// capture resumes without the user seeing a fake "done" first.
if ( ! task . _doneConfirmAt ) {
_updateTask ( task . sessionId , { _doneConfirmAt : Date . now ( ) + 30000 } ) ;
setTimeout ( async ( ) => {
try {
const fresh = _loadTasks ( ) . find ( t => t . sessionId === task . sessionId ) ;
if ( ! fresh ) return ;
let stillAlive = false ;
try {
const probe = await fetch ( '/api/shell/exec' , {
method : 'POST' , credentials : 'same-origin' ,
headers : { 'Content-Type' : 'application/json' } ,
body : JSON . stringify ( { command : _tmuxCmd ( task , ` has-session -t ${ task . sessionId } ` ) , timeout : 5 } ) ,
} ) ;
const pData = await probe . json ( ) ;
stillAlive = pData . exit _code === 0 ;
} catch { /* network blip — treat as inconclusive, prefer running */ stillAlive = true ; }
if ( stillAlive ) {
cookbook agent debug loop: persistent log files, auto-adopt orphan tmux, Codex/Claude skill parity
Three converging fixes so the chat agent + external Codex/Claude skills can actually debug a crashed serve instead of staring at a post-crash neofetch banner:
* Serves now `tee` to /tmp/odysseus-tmux/SESSION.log on the host running them. Runner saves fds 3/4 before the tee and restores them right before `exec ${SHELL}`, so the post-crash interactive zsh banner does NOT pollute the log file.
* `tail_serve_output` (chat agent) and `/api/codex/cookbook/output/{sid}` (Codex+Claude skills) both prefer the persistent log file over the tmux pane. Pane is fallback for sessions predating the tee runner. Default tail bumped 150 -> 400.
* `list_served_models` "recent log" snippet seeks to the Traceback line instead of showing the last 6 lines (which was always the bash prompt).
Cookbook auto-adoption sweep on `/api/cookbook/tasks/status`: every 20s (rate-limited) the cookbook SSHes each configured server, finds `serve-*` / `cookbook-*` tmux sessions running an actual model process (vllm/python/llama-server/etc., filtered via `pane_current_command`), and writes them into state.tasks. So when the agent falls back to raw ssh+tmux, the session appears in the Cookbook UI on the next poll.
`serve_model` error path now reads `data["detail"]` in addition to `data["error"]` so the FastAPI HTTPException message ("Invalid characters in cmd") actually reaches the agent instead of being swallowed as a generic "Serve failed". Tool description updated to warn against `cd …`/`source …`/`&&` prefixes.
Intent-without-action supervisor in agent_loop: when the model writes "Let me tail the output" / "I'll check the logs" / "Let me investigate" and ends the turn without emitting a tool call, the loop injects a sharp system nudge ("You said you would X — DO IT NOW") and continues. Capped at 2 nudges per chat so a model that genuinely cannot use the tool does not pin the loop.
Codex/Claude skill parity: adds `/cookbook/cached`, `/cookbook/presets`, `/cookbook/preset/{name}`, `/cookbook/adopt` so external agents have the same surface as the chat agent. SKILL.md docs + odysseus_api.py wrapper updated for both bundles.
`adopt_served_model` promoted to the always-on tool set so the agent has a documented fallback when serve_model rejects a cmd.
Also various cookbook UI tweaks accumulated alongside the above (cookbook.js, cookbookRunning.js, cookbookServe.js, cookbook-diagnosis.js, settings.js, style.css).
2026-06-04 23:27:18 +09:00
_updateTask ( task . sessionId , { status : 'running' , _doneConfirmAt : null , _lastStatusFlipAt : Date . now ( ) } ) ;
feat: Claude Agent integration + cookbook reconnect + UI polish
- Claude Agent integration: AGENT_CONFIGS.claude, INTG_TYPES.claude,
setup_claude_routes + integrations/claude/ skill bundle. Wired in
app.py alongside the existing Codex integration; same scope-gated
/api/codex/* backend; agent form has new description so users know
it's setup for an external CLI, not an agent streamed inside Odysseus.
- Remove mark_email_boundaries action: not good enough yet. Stripped
from task UI, scheduler defaults, registry, tool schema, clear-cache
route. Added to RETIRED_HOUSEKEEPING_ACTIONS so existing rows + their
task_runs auto-purge on startup.
- Cookbook download reliability: "Reconnect" fix button in the crash
diagnosis runs _reconnectTask after probing has-session. 30s confirm
window before marking a download "done" — kills the Finished/Downloading
flicker when tmux briefly drops between captures.
- Mobile UX: tap anywhere on a note card body opens the editor;
Update button morphs to Archive when no text was edited; bell icon
accent-colored; chip-trashing notif pills fade so only the icon
rotates into the trash zone.
- Settings integrations: SVG-per-provider in email + API preset
dropdowns, custom drop-up-aware menus, accent sub-header icons
(IMAP/SMTP), consistent card styling between list + edit, contacts
Edit/Delete icons, agent form description copy.
2026-06-04 08:27:26 +09:00
const _el = document . querySelector ( ` .cookbook-task[data-task-id=" ${ task . sessionId } "] ` ) ;
if ( _el ) {
_el . dataset . status = 'running' ;
const _badge = _el . querySelector ( '.cookbook-task-status' ) ;
if ( _badge ) { _badge . textContent = _statusLabel ( 'running' , task . type ) ; _badge . className = 'cookbook-task-status' ; }
const _wave = _el . querySelector ( '.cookbook-task-wave' ) ; if ( _wave ) _wave . style . display = '' ;
const _up = _el . querySelector ( '.cookbook-task-uptime' ) ; if ( _up ) _up . style . display = '' ;
_reconnectTask ( _el , _loadTasks ( ) . find ( t => t . sessionId === task . sessionId ) ) ;
}
return ;
}
cookbook agent debug loop: persistent log files, auto-adopt orphan tmux, Codex/Claude skill parity
Three converging fixes so the chat agent + external Codex/Claude skills can actually debug a crashed serve instead of staring at a post-crash neofetch banner:
* Serves now `tee` to /tmp/odysseus-tmux/SESSION.log on the host running them. Runner saves fds 3/4 before the tee and restores them right before `exec ${SHELL}`, so the post-crash interactive zsh banner does NOT pollute the log file.
* `tail_serve_output` (chat agent) and `/api/codex/cookbook/output/{sid}` (Codex+Claude skills) both prefer the persistent log file over the tmux pane. Pane is fallback for sessions predating the tee runner. Default tail bumped 150 -> 400.
* `list_served_models` "recent log" snippet seeks to the Traceback line instead of showing the last 6 lines (which was always the bash prompt).
Cookbook auto-adoption sweep on `/api/cookbook/tasks/status`: every 20s (rate-limited) the cookbook SSHes each configured server, finds `serve-*` / `cookbook-*` tmux sessions running an actual model process (vllm/python/llama-server/etc., filtered via `pane_current_command`), and writes them into state.tasks. So when the agent falls back to raw ssh+tmux, the session appears in the Cookbook UI on the next poll.
`serve_model` error path now reads `data["detail"]` in addition to `data["error"]` so the FastAPI HTTPException message ("Invalid characters in cmd") actually reaches the agent instead of being swallowed as a generic "Serve failed". Tool description updated to warn against `cd …`/`source …`/`&&` prefixes.
Intent-without-action supervisor in agent_loop: when the model writes "Let me tail the output" / "I'll check the logs" / "Let me investigate" and ends the turn without emitting a tool call, the loop injects a sharp system nudge ("You said you would X — DO IT NOW") and continues. Capped at 2 nudges per chat so a model that genuinely cannot use the tool does not pin the loop.
Codex/Claude skill parity: adds `/cookbook/cached`, `/cookbook/presets`, `/cookbook/preset/{name}`, `/cookbook/adopt` so external agents have the same surface as the chat agent. SKILL.md docs + odysseus_api.py wrapper updated for both bundles.
`adopt_served_model` promoted to the always-on tool set so the agent has a documented fallback when serve_model rejects a cmd.
Also various cookbook UI tweaks accumulated alongside the above (cookbook.js, cookbookRunning.js, cookbookServe.js, cookbook-diagnosis.js, settings.js, style.css).
2026-06-04 23:27:18 +09:00
_updateTask ( task . sessionId , { status : 'done' , _doneConfirmAt : null , _lastStatusFlipAt : Date . now ( ) } ) ;
feat: Claude Agent integration + cookbook reconnect + UI polish
- Claude Agent integration: AGENT_CONFIGS.claude, INTG_TYPES.claude,
setup_claude_routes + integrations/claude/ skill bundle. Wired in
app.py alongside the existing Codex integration; same scope-gated
/api/codex/* backend; agent form has new description so users know
it's setup for an external CLI, not an agent streamed inside Odysseus.
- Remove mark_email_boundaries action: not good enough yet. Stripped
from task UI, scheduler defaults, registry, tool schema, clear-cache
route. Added to RETIRED_HOUSEKEEPING_ACTIONS so existing rows + their
task_runs auto-purge on startup.
- Cookbook download reliability: "Reconnect" fix button in the crash
diagnosis runs _reconnectTask after probing has-session. 30s confirm
window before marking a download "done" — kills the Finished/Downloading
flicker when tmux briefly drops between captures.
- Mobile UX: tap anywhere on a note card body opens the editor;
Update button morphs to Archive when no text was edited; bell icon
accent-colored; chip-trashing notif pills fade so only the icon
rotates into the trash zone.
- Settings integrations: SVG-per-provider in email + API preset
dropdowns, custom drop-up-aware menus, accent sub-header icons
(IMAP/SMTP), consistent card styling between list + edit, contacts
Edit/Delete icons, agent form description copy.
2026-06-04 08:27:26 +09:00
const _el = document . querySelector ( ` .cookbook-task[data-task-id=" ${ task . sessionId } "] ` ) ;
if ( _el ) {
2026-06-05 14:41:07 +02:00
_clearDiagnosis ( _el ) ;
feat: Claude Agent integration + cookbook reconnect + UI polish
- Claude Agent integration: AGENT_CONFIGS.claude, INTG_TYPES.claude,
setup_claude_routes + integrations/claude/ skill bundle. Wired in
app.py alongside the existing Codex integration; same scope-gated
/api/codex/* backend; agent form has new description so users know
it's setup for an external CLI, not an agent streamed inside Odysseus.
- Remove mark_email_boundaries action: not good enough yet. Stripped
from task UI, scheduler defaults, registry, tool schema, clear-cache
route. Added to RETIRED_HOUSEKEEPING_ACTIONS so existing rows + their
task_runs auto-purge on startup.
- Cookbook download reliability: "Reconnect" fix button in the crash
diagnosis runs _reconnectTask after probing has-session. 30s confirm
window before marking a download "done" — kills the Finished/Downloading
flicker when tmux briefly drops between captures.
- Mobile UX: tap anywhere on a note card body opens the editor;
Update button morphs to Archive when no text was edited; bell icon
accent-colored; chip-trashing notif pills fade so only the icon
rotates into the trash zone.
- Settings integrations: SVG-per-provider in email + API preset
dropdowns, custom drop-up-aware menus, accent sub-header icons
(IMAP/SMTP), consistent card styling between list + edit, contacts
Edit/Delete icons, agent form description copy.
2026-06-04 08:27:26 +09:00
_el . dataset . status = 'done' ;
const _badge = _el . querySelector ( '.cookbook-task-status' ) ;
if ( _badge ) { _badge . textContent = _statusLabel ( 'done' , task . type ) ; _badge . className = 'cookbook-task-status cookbook-task-done' ; }
const _chk = _el . querySelector ( '.cookbook-task-check' ) ; if ( _chk ) _chk . style . display = '' ;
const _sb = _el . querySelector ( '.cookbook-task-serve-btn' ) ; if ( _sb ) _sb . style . display = '' ;
}
_showCookbookNotif ( ) ;
_refreshDepsAfterInstall ( task ) ;
_renderRunningTab ( ) ;
_processQueue ( ) ;
} catch { /* swallow — next polling cycle will retry */ }
} , 30000 ) ;
}
2026-05-31 23:58:26 +09:00
}
}
_renderRunningTab ( ) ;
_processQueue ( ) ;
break ;
}
const snapshot = ( data . stdout || '' ) . trim ( ) ;
if ( snapshot ) {
cookbook agent debug loop: persistent log files, auto-adopt orphan tmux, Codex/Claude skill parity
Three converging fixes so the chat agent + external Codex/Claude skills can actually debug a crashed serve instead of staring at a post-crash neofetch banner:
* Serves now `tee` to /tmp/odysseus-tmux/SESSION.log on the host running them. Runner saves fds 3/4 before the tee and restores them right before `exec ${SHELL}`, so the post-crash interactive zsh banner does NOT pollute the log file.
* `tail_serve_output` (chat agent) and `/api/codex/cookbook/output/{sid}` (Codex+Claude skills) both prefer the persistent log file over the tmux pane. Pane is fallback for sessions predating the tee runner. Default tail bumped 150 -> 400.
* `list_served_models` "recent log" snippet seeks to the Traceback line instead of showing the last 6 lines (which was always the bash prompt).
Cookbook auto-adoption sweep on `/api/cookbook/tasks/status`: every 20s (rate-limited) the cookbook SSHes each configured server, finds `serve-*` / `cookbook-*` tmux sessions running an actual model process (vllm/python/llama-server/etc., filtered via `pane_current_command`), and writes them into state.tasks. So when the agent falls back to raw ssh+tmux, the session appears in the Cookbook UI on the next poll.
`serve_model` error path now reads `data["detail"]` in addition to `data["error"]` so the FastAPI HTTPException message ("Invalid characters in cmd") actually reaches the agent instead of being swallowed as a generic "Serve failed". Tool description updated to warn against `cd …`/`source …`/`&&` prefixes.
Intent-without-action supervisor in agent_loop: when the model writes "Let me tail the output" / "I'll check the logs" / "Let me investigate" and ends the turn without emitting a tool call, the loop injects a sharp system nudge ("You said you would X — DO IT NOW") and continues. Capped at 2 nudges per chat so a model that genuinely cannot use the tool does not pin the loop.
Codex/Claude skill parity: adds `/cookbook/cached`, `/cookbook/presets`, `/cookbook/preset/{name}`, `/cookbook/adopt` so external agents have the same surface as the chat agent. SKILL.md docs + odysseus_api.py wrapper updated for both bundles.
`adopt_served_model` promoted to the always-on tool set so the agent has a documented fallback when serve_model rejects a cmd.
Also various cookbook UI tweaks accumulated alongside the above (cookbook.js, cookbookRunning.js, cookbookServe.js, cookbook-diagnosis.js, settings.js, style.css).
2026-06-04 23:27:18 +09:00
// Only auto-scroll to bottom if the user was already there. When
// they've scrolled up to read earlier output, leave their position
// alone so a fresh snapshot doesn't yank them back to the tail.
// 40px tolerance covers sub-pixel rounding + the moment between
// releasing the scrollbar and the next poll arriving.
const _atBottom = ( output . scrollHeight - output . scrollTop - output . clientHeight ) < 40 ;
2026-05-31 23:58:26 +09:00
output . textContent = snapshot ;
cookbook agent debug loop: persistent log files, auto-adopt orphan tmux, Codex/Claude skill parity
Three converging fixes so the chat agent + external Codex/Claude skills can actually debug a crashed serve instead of staring at a post-crash neofetch banner:
* Serves now `tee` to /tmp/odysseus-tmux/SESSION.log on the host running them. Runner saves fds 3/4 before the tee and restores them right before `exec ${SHELL}`, so the post-crash interactive zsh banner does NOT pollute the log file.
* `tail_serve_output` (chat agent) and `/api/codex/cookbook/output/{sid}` (Codex+Claude skills) both prefer the persistent log file over the tmux pane. Pane is fallback for sessions predating the tee runner. Default tail bumped 150 -> 400.
* `list_served_models` "recent log" snippet seeks to the Traceback line instead of showing the last 6 lines (which was always the bash prompt).
Cookbook auto-adoption sweep on `/api/cookbook/tasks/status`: every 20s (rate-limited) the cookbook SSHes each configured server, finds `serve-*` / `cookbook-*` tmux sessions running an actual model process (vllm/python/llama-server/etc., filtered via `pane_current_command`), and writes them into state.tasks. So when the agent falls back to raw ssh+tmux, the session appears in the Cookbook UI on the next poll.
`serve_model` error path now reads `data["detail"]` in addition to `data["error"]` so the FastAPI HTTPException message ("Invalid characters in cmd") actually reaches the agent instead of being swallowed as a generic "Serve failed". Tool description updated to warn against `cd …`/`source …`/`&&` prefixes.
Intent-without-action supervisor in agent_loop: when the model writes "Let me tail the output" / "I'll check the logs" / "Let me investigate" and ends the turn without emitting a tool call, the loop injects a sharp system nudge ("You said you would X — DO IT NOW") and continues. Capped at 2 nudges per chat so a model that genuinely cannot use the tool does not pin the loop.
Codex/Claude skill parity: adds `/cookbook/cached`, `/cookbook/presets`, `/cookbook/preset/{name}`, `/cookbook/adopt` so external agents have the same surface as the chat agent. SKILL.md docs + odysseus_api.py wrapper updated for both bundles.
`adopt_served_model` promoted to the always-on tool set so the agent has a documented fallback when serve_model rejects a cmd.
Also various cookbook UI tweaks accumulated alongside the above (cookbook.js, cookbookRunning.js, cookbookServe.js, cookbook-diagnosis.js, settings.js, style.css).
2026-06-04 23:27:18 +09:00
if ( _atBottom ) output . scrollTop = output . scrollHeight ;
2026-05-31 23:58:26 +09:00
// Live status parsing for download tasks
if ( task . type === 'download' ) {
const badge = el . querySelector ( '.cookbook-task-status' ) ;
if ( badge ) {
const completed = ( snapshot . match ( /Download complete/g ) || [ ] ) . length ;
const downloading = snapshot . match ( /Downloading '([^']+)'/g ) || [ ] ;
const totalFiles = downloading . length ;
const pctMatches = [ ... snapshot . matchAll ( /(\d+)%\|/g ) ] ;
const lastPct = pctMatches . length ? pctMatches [ pctMatches . length - 1 ] [ 1 ] : null ;
const speedMatch = [ ... snapshot . matchAll ( /([\d.]+)(?:MB|GB)\/s/g ) ] ;
const lastSpeed = speedMatch . length ? speedMatch [ speedMatch . length - 1 ] [ 0 ] : null ;
// hf_transfer prints "Downloading (incomplete total...): 73% | 1.81G/2.49G"
// — the real aggregate byte progress. The "Fetching N files" line (often
// last in the output) sits at 0%, so lastPct/_fetchPct can read 0 even at
// 73% done. Prefer this aggregate when present.
const _dlAggMatches = [ ... snapshot . matchAll ( /Downloading\s*\(incomplete[^)]*\):\s*(\d+)%/g ) ] ;
const _dlAgg = _dlAggMatches . length ? parseInt ( _dlAggMatches [ _dlAggMatches . length - 1 ] [ 1 ] ) : null ;
// Stale download detection.
// Use the DOWNLOADED-BYTE count ("1.81G" from "1.81G/2.49G") as the
// progress signal: it climbs continuously while transferring (even when
// the % plateaus during a big hf_transfer chunk) and FREEZES when stuck.
// The % alone plateaus (false stall), and a frozen frame still shows a
// stale speed/ETA — so keying off speed masked real stalls (that's why a
// 97%-stuck download went undetected). Bytes are the honest signal; fall
// back to %/aggregate only when no byte counter is present.
const _byteMatches = [ ... snapshot . matchAll ( /([\d.]+\s?[KMGT])B?\s*\/\s*[\d.]+\s?[KMGT]B?/gi ) ] ;
const _bytes = _byteMatches . length ? _byteMatches [ _byteMatches . length - 1 ] [ 1 ] . replace ( /\s/g , '' ) : null ;
2026-06-03 12:23:35 +08:00
// When there's no byte counter (pip resolve / native build phase of a
// dependency install), key off the output tail so new build lines count
// as progress — otherwise a long quiet build is falsely declared stale
// and restarted mid-build, looping forever (#1568).
const curProgress = computeProgressSignal ( _bytes , _dlAgg , lastPct , snapshot ) ;
2026-06-01 22:42:59 -04:00
const _fetchPctMatches = [ ... snapshot . matchAll ( /Fetching\s+\d+\s+files:\s*(\d+)%/g ) ] ;
const _fetchPct = _fetchPctMatches . length ? parseInt ( _fetchPctMatches [ _fetchPctMatches . length - 1 ] [ 1 ] ) : null ;
2026-06-05 14:41:07 +02:00
const isPipDep = ! ! ( task . payload && task . payload . _dep ) ;
2026-06-01 22:42:59 -04:00
const _startupStalled = ! _bytes && ( ( _dlAgg === 0 ) || ( _fetchPct === 0 ) ) && curProgress === '0' ;
const _STALE _TIMEOUT = _startupStalled ? STARTUP _STALE _PROGRESS _MS : STALE _PROGRESS _MS ;
2026-05-31 23:58:26 +09:00
if ( ! el . _lastProgress ) { el . _lastProgress = curProgress ; el . _lastProgressTime = Date . now ( ) ; }
if ( curProgress !== el . _lastProgress ) {
el . _lastProgress = curProgress ;
el . _lastProgressTime = Date . now ( ) ;
2026-06-05 14:41:07 +02:00
} else if ( ! isPipDep && Date . now ( ) - ( el . _lastProgressTime || 0 ) > _STALE _TIMEOUT && task . _autoRestarted ) {
2026-05-31 23:58:26 +09:00
const mins = Math . floor ( ( Date . now ( ) - ( el . _lastProgressTime || 0 ) ) / 60000 ) ;
// Already auto-restarted once and stalled again — make the badge a
// one-click retry (resumes from the cached partial files) so the
// user doesn't have to dig into the ⋮ menu.
badge . textContent = ` stalled ${ mins } m ↻ ` ;
badge . className = 'cookbook-task-status cookbook-task-error' ;
badge . title = 'Click to retry — resumes where it stopped' ;
badge . style . cursor = 'pointer' ;
if ( ! badge . _retryBound ) {
badge . _retryBound = true ;
badge . addEventListener ( 'click' , ( e ) => { e . stopPropagation ( ) ; _retryTask ( el , task ) ; } ) ;
}
2026-06-05 14:41:07 +02:00
} else if ( ! isPipDep && Date . now ( ) - ( el . _lastProgressTime || 0 ) > _STALE _TIMEOUT && ! task . _autoRestarted ) {
2026-05-31 23:58:26 +09:00
task . _autoRestarted = true ;
_updateTask ( task . sessionId , { _autoRestarted : true } ) ;
2026-06-01 22:42:59 -04:00
badge . textContent = _startupStalled ? '0% stall — retrying' : 'stale — restarting' ;
2026-05-31 23:58:26 +09:00
badge . className = 'cookbook-task-status cookbook-task-error' ;
_showCookbookNotif ( true ) ;
try {
await fetch ( '/api/shell/exec' , {
method : 'POST' , credentials : 'same-origin' ,
headers : { 'Content-Type' : 'application/json' } ,
body : JSON . stringify ( { command : _tmuxCmd ( task , ` kill-session -t ${ task . sessionId } ` ) } ) ,
} ) ;
} catch { }
try {
// Reuse original payload so the full repo_id (e.g. "Qwen/Qwen3.5-...")
// is preserved — rebuilding from task.repo/task.name drops the org prefix.
const dlPayload = task . payload
? { ... task . payload }
: { repo _id : task . repo || task . name , remote _host : task . remoteHost || '' } ;
if ( _envState . hfToken ) dlPayload . hf _token = _envState . hfToken ;
// Stalled with hf_transfer — restart on the reliable downloader.
dlPayload . disable _hf _transfer = true ;
// Don't overwrite env_prefix — task.payload already has the correct
// "source <path>" form. The bare envPath would miss the `source` and
// the venv never activates (so hf CLI falls off PATH).
const res = await fetch ( '/api/model/download' , {
method : 'POST' , credentials : 'same-origin' ,
headers : { 'Content-Type' : 'application/json' } ,
body : JSON . stringify ( dlPayload ) ,
} ) ;
const data = await res . json ( ) ;
if ( data . ok && data . session _id ) {
_updateTask ( task . sessionId , { sessionId : data . session _id , status : 'running' , output : '' } ) ;
task . sessionId = data . session _id ;
el . _lastProgress = null ;
el . _lastProgressTime = Date . now ( ) ;
badge . textContent = 'restarted' ;
badge . className = 'cookbook-task-status cookbook-task-running' ;
continue ;
}
} catch { }
badge . textContent = 'stale — restart failed' ;
badge . className = 'cookbook-task-status cookbook-task-error' ;
_showCookbookNotif ( true ) ;
break ;
}
2026-06-03 16:49:10 +09:00
// When the snapshot includes a shard-of-N marker (e.g.
// "model-00006-of-00082.safetensors"), TRUE overall progress is
// ((shard-1) + currentShardFraction) / totalShards. Before, _dlAgg
// (hf_transfer's per-current-shard aggregate, e.g. 53% of shard 6)
// was treated as overall and the row read "53%" while only 5 of
// 82 shards were actually done.
const _shardPat = [ ... snapshot . matchAll ( /model-(\d+)-of-(\d+)\.(?:safetensors|bin)/g ) ] ;
const _lastShard = _shardPat . length ? _shardPat [ _shardPat . length - 1 ] : null ;
const _curShardNum = _lastShard ? parseInt ( _lastShard [ 1 ] , 10 ) : null ;
const _totalShards = _lastShard ? parseInt ( _lastShard [ 2 ] , 10 ) : null ;
const _useShardAgg = _curShardNum && _totalShards && _totalShards > 1 ;
2026-05-31 23:58:26 +09:00
// HF's own "Fetching N files: X%" aggregate counts ALL files,
// including ones already finished in a previous session (resume) —
// so on a resumed download it reflects the true overall progress,
// whereas completed/totalFiles only see this session's files (→ 0%).
// Take the higher of the two so resume doesn't read as 0%.
2026-06-03 16:49:10 +09:00
if ( _useShardAgg ) {
// Multi-shard download: compute TRUE overall as completed shards
// plus the current shard's fraction. _dlAgg / lastPct represent
// *this shard's* progress, not the whole download.
const curShardFrac = ( _dlAgg != null )
? _dlAgg / 100
: ( lastPct ? parseInt ( lastPct , 10 ) / 100 : 0 ) ;
let overallPct = Math . round ( ( ( ( _curShardNum - 1 ) + curShardFrac ) / _totalShards ) * 100 ) ;
if ( _fetchPct != null ) overallPct = Math . max ( overallPct , _fetchPct ) ;
let text = ` ${ overallPct } % ` ;
if ( lastSpeed ) text += ` · ${ lastSpeed } ` ;
badge . textContent = text ;
badge . className = 'cookbook-task-status cookbook-task-running' ;
} else if ( _dlAgg != null ) {
2026-05-31 23:58:26 +09:00
// Real aggregate byte progress — most accurate; take the max of all signals.
let pct = _dlAgg ;
if ( _fetchPct != null ) pct = Math . max ( pct , _fetchPct ) ;
let text = ` ${ pct } % ` ;
if ( lastSpeed ) text += ` · ${ lastSpeed } ` ;
badge . textContent = text ;
badge . className = 'cookbook-task-status cookbook-task-running' ;
} else if ( totalFiles > 0 && completed < totalFiles ) {
const curFilePct = lastPct ? parseInt ( lastPct ) / 100 : 0 ;
let overallPct = Math . round ( ( ( completed + curFilePct ) / totalFiles ) * 100 ) ;
if ( _fetchPct != null ) overallPct = Math . max ( overallPct , _fetchPct ) ;
let text = ` ${ overallPct } % ` ;
if ( lastSpeed ) text += ` · ${ lastSpeed } ` ;
badge . textContent = text ;
badge . className = 'cookbook-task-status cookbook-task-running' ;
} else if ( _fetchPct != null && _fetchPct < 100 ) {
// Resume start: only the aggregate is meaningful yet.
let text = ` ${ _fetchPct } % ` ;
if ( lastSpeed ) text += ` · ${ lastSpeed } ` ;
badge . textContent = text ;
badge . className = 'cookbook-task-status cookbook-task-running' ;
} else if ( completed > 0 && completed >= totalFiles ) {
badge . textContent = 'finishing' ;
badge . className = 'cookbook-task-status cookbook-task-running' ;
}
if ( snapshot . includes ( 'DOWNLOAD_FAILED' ) ) {
// The wrapper prints DOWNLOAD_FAILED but exits 0, and per-file
// "Download complete"/"100%" lines make it look successful — so
// catch the explicit failure marker and handle it.
// A gated/auth failure can NEVER be fixed by retrying (the HF token
// is sent, but its account isn't approved for this repo) — skip the
// auto-retries and surface the gated diagnosis straight away.
const _accessDenied = /Access to model.*is restricted|gated repo|GatedRepoError|401 Unauthorized|403 Forbidden|not in the authorized list|awaiting a review|must (?:be authenticated|have access)/i . test ( snapshot ) ;
const _dlKey = task . payload ? . repo _id || task . name ;
const _dlN = _dlRetryCount . get ( _dlKey ) || 0 ;
fix(cookbook): stop-all no longer auto-retries interrupted HF downloads fixes (#1474)
* fix(cookbook): stop-all no longer auto-retries interrupted HF downloads
When C-c was sent to a running download, the bash wrapper printed
DOWNLOAD_FAILED on non-zero exit (SIGINT = 130). The reconnect polling
loop was still running at that point, saw the failure marker, and
silently relaunched the download — making "Stop all" appear to have no
effect while the UI showed the toast as if it succeeded.
Fix: abort the reconnect controller immediately when the stop button is
clicked (before the kill command is dispatched), and guard the
auto-retry condition with !controller.signal.aborted so that any
in-flight poll that completes after abort cannot trigger a retry.
Fixes #1458
Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
* Fix Edge/Chromium sidebar section-title clipping (#1420)
Sidebar section titles were vertically clipped in Chromium/Edge (fine in
Firefox). Raise line-height 1 → 1.3, mirroring the existing .list-item fix.
The titles are flex-centred in a fixed-height (29px) header, so this adds
glyph headroom without any reflow.
* Drop GPU-only flags from the CPU-only (-ngl 0) serve command (#1433)
A CPU-only llama.cpp serve config still emitted --flash-attn on and exported
GGML_CUDA_ENABLE_UNIFIED_MEMORY=1 (independent toggles, often left on by an Auto
profile), so the command mixed "zero GPU layers" with CUDA/flash-attn and failed
to start (issue #1291). Gate both on a _cpuOnly check (ngl == 0). GPU serving is
unchanged — the gate only affects the ngl=0 path.
Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
* fix: APIKeyManager.load crashes app startup on a corrupt/wrong-shape api_keys.json (#1565)
* Don't lose deep-research findings when synthesis times out (#1551) (#1562)
Two problems made deep research report "No information could be gathered" even
after it had extracted findings, on slow local models (reporter served a 20B
via LM Studio):
- _synthesize hard-capped its LLM call at timeout=60, while extraction uses the
user's extraction_timeout (300s here) and the final report uses 180s. The slow
model needed >60s to synthesize the round's findings, so synthesis timed out
after 3 attempts. Raised it to 180s to match the final-report call.
- When synthesis produced no report (it returns the unchanged, still-empty
report on failure during round 1), the run hit
`if not report: return "No information could be gathered…"` and discarded the
findings it had already gathered. Now it falls back to a compiled report built
from those findings (_fallback_report) so the user keeps the gathered material.
Tests stub the LLM (no live model/DB), pin the synthesis timeout >= 180, that the
fallback surfaces the findings rather than the give-up message, and that a failed
synthesis preserves the previous report.
Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
* fix: return sorted model list on first call in group chat (#1484)
Both _getModels() and getAllModels() store the sorted copy in a cache
variable but return the original unsorted array on first invocation.
Subsequent calls return the cache (sorted), causing inconsistent
model picker ordering on first render.
* fix: guard sp.destroy() in _loadScheduled against null spinner (#1495)
When the scheduled folder is opened with cached data, sp is null
(the loading spinner is skipped). _loadScheduled receives null and
calls sp.destroy() unconditionally, crashing with TypeError.
* fix: capture download exit code before test consumes it (#1497)
The shell pattern 'if [ $? -eq 0 ]; ... else ... echo DOWNLOAD_FAILED (exit $?)' always reports 'exit 1' because $? inside the else branch is the exit code of the [ test command, not the download. Capture into _ec first.
* fix: guard uid.decode() in auto-classify warning log against str UIDs (#1472)
Every other uid.decode() call in this function uses
'uid.decode() if isinstance(uid, bytes) else str(uid)' but the
warning at line 832 does bare uid.decode(), crashing with
AttributeError when uid is already a string.
* fix: guard AI tidy verdict against non-string LLM output (#1486)
The AI document-tidy endpoint parses verdicts from LLM JSON output
and calls .lower().strip() directly. If the model returns null or a
non-string element, this crashes with AttributeError. Coerce to str
so malformed output is treated as 'keep' instead of crashing.
* fix: rename local url-quote import to avoid shadowing module-level _q (#1471)
The 'from urllib.parse import quote as _q' at line 734 shadows the
module-level _q (istrstrstrstrstrstrIMAPutility) imported from email_helpers, causing
UnboundLocalError at lines 191 and 278 where _q is used before the
local import executes. This silently breaks the entire auto-summarize
pass.
* fix(ui): add missing Escape key handlers for email-lib-modal, model-picker-menu, and sort dropdowns (#1487)
CONTEXT: Several interactive elements lacked Escape key handlers: the email library modal was not in dynamicModals, the model-picker popup had no Escape close, and the session/model sort dropdowns only closed on outside click.
CHANGE: Adds email-lib-modal to the dynamicModals array in the Escape handler so it gets dismissed via dismissModal. Adds a check for model-picker-menu.open before the modal chain to close the dropdown on Escape. Adds checks for session-sort-dropdown and model-sort-dropdown display=block before the document panel minimize fallback.
WHY: Users expect consistent Escape-to-close behavior across all modals, overlays, and popups. These four were the only interactive containers in the app that ignored the Escape key entirely.
IMPACT: Pressing Escape now closes the email library modal, model picker popup, session sort dropdown, and model sort dropdown -- matching user expectations and the behavior of every other modal in the app.
* fix: mcp CLI _serialize crashes when stored env JSON is a list (#1609)
* fix: validate_caldav_url crashes with TypeError on a non-string URL (#1608)
* fix: _sanitize_export_filename crashes on a non-string session name (#1607)
* fix: shared MCP truncate() crashes on None/non-string tool output (#1605)
* fix: search query helpers crash on a non-string query (#1604)
* fix: rag_server add/remove_directory crashes on a non-string directory arg (#1614)
* fix: gallery CLI image serialization crashes on a non-string prompt (#1598)
* fix: research CLI summary crashes on a non-string query (#1596)
* fix: skills CLI summary crashes on a non-string description (#1595)
* fix(cookbook): set UTF-8 encoding for detached download/serve subprocesses (#1599)
On Windows, Python defaults to the active code page (cp1252) for
subprocess I/O. HuggingFace CLI outputs U+2713 (✓) when validating
tokens, which cp1252 cannot encode, crashing the download process.
Set PYTHONUTF8=1 and PYTHONIOENCODING=utf-8 in the subprocess
environment so Unicode output from hf/pip/llama-server is handled
correctly.
Fixes #1543
Co-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
* docs: clarify host Ollama with Docker (#1594)
* fix(ui): stop welcome-screen tip from clipping on narrow phones (#1612)
The empty-state tip ("Add an AI endpoint from Settings...") shares a 60px
max-height ceiling with the one-line .welcome-sub / .welcome-version. On
narrow phones the welcome block shrink-wraps and the tip wraps to 4-5 lines
(~67px), so the shared ceiling clipped its last line ("...key into the
chat.") - the only setup hint a first-run user gets.
Give .welcome-tip its own taller max-height (120px), placed above the
@media (max-height: 650px) block so that rule's max-height:0 still collapses
the tip on short viewports. .welcome-sub / .welcome-version are untouched,
and desktop is unchanged (the tip is ~50px there, well under the ceiling).
* Save only string personal doc paths (#1566)
* Reject backup output inside data dir (#1587)
* Parse all AMD GPU check args (#1586)
* Require runnable dispatcher subcommands (#1585)
* Require runnable dispatcher subcommands
* Use modern dispatcher test loader
* Remove duplicate update database body (#1584)
* Skip invalid research service sources (#1583)
* Reject CalDAV writeback events without uid (#1582)
* Reject empty mail CLI recipients (#1581)
* Reject empty mail CLI recipients
* Keep mail CLI test imports isolated
* Validate signature CLI PNG data (#1580)
* Validate signature CLI PNG data
* Keep signature CLI test imports isolated
* Reject invalid preset CLI entries (#1579)
* Reject invalid preset CLI entries
* Use modern preset CLI test loader
* Normalize session CLI counters (#1578)
* Normalize session CLI counters
* Keep sessions CLI test imports isolated
* fix: monthly schedule label shows 21th/22th/31th (ordinal suffix for days >20) (#1577)
* fix: split_chunks emits a duplicate trailing chunk for text over size-overlap (#1573)
* fix: builtin_actions heuristics crash on a truthy non-string input (#1639)
* fix: skill test-task / precision helpers crash on a non-dict skill (#1638)
* fix: logs CLI _resolve crashes on a non-string name (#1631)
* fix: _extract_skill_json crashes on a truthy non-string teacher response (#1630)
* fix: tool-block parsing crashes on a non-string input (#1628)
* fix: check_outbound_url crashes on a truthy non-string URL (#1623)
* fix: document_actions title/content helpers crash on non-string input (#1621)
* fix: inside_base_dir raises TypeError on a non-string path instead of failing closed (#1619)
* fix: is_markitdown_format crashes on a non-string path (#1618)
* Close app_api blocklist gap for bare /api/tokens and /api/users
The blocklist prefixes had trailing slashes, so path.startswith() only
matched /api/tokens/{id} but not /api/tokens itself — the bare GET (list)
and POST (mint) endpoints were reachable via app_api. Same gap on
/api/users (list/create/delete). Drop trailing slashes so both bare and
sub-resource forms are blocked. /api/auth and /api/admin had no bare
endpoints today but get the same treatment to prevent future drift.
Caught by #1462.
* Decrypt CalDAV password before write-back (#1731)
writeback_event read cfg["password"] (the encrypted blob) and passed it
straight to DAVClient, so every local create/edit/delete authenticated
with the literal ciphertext, the remote rejected it, and the change
never reached the server — the exact silent-write-loss this module was
built to prevent. The pull path src/caldav_sync.py already decrypts;
mirror that. decrypt() is a no-op on legacy plaintext.
Caught by #1731.
* Memory MCP delete: match exact id, not prefix (#1303)
The delete action looked up the target with startswith() to capture
full_id, but then re-applied startswith() to filter the list — so a
short or ambiguous memory_id silently deleted every memory whose id
shared the prefix, while the success message reported only the first
match. The edit action used the first match and stopped, so the two
actions disagreed on multi-match behaviour. Use full_id for both.
Caught by #1303.
* Rebuild memory vector index from the full saved set, not just the audited owner (#1747)
audit_memories saves final_entries merged with other owners' entries
(correct), but then rebuilt the shared vector collection from
final_entries alone — wiping every other owner from semantic search
until they happened to run their own audit. Keyword fallback masked
it, so it degraded silently. Capture saved_entries once and rebuild
from that.
Caught by #1747.
* Owner-scope RAG doc ids so identical chunks across users don't collide (#1738, #1760)
_generate_doc_id hashed only text. add_document / add_documents_batch
early-return when the id exists, so the second owner indexing a
byte-identical chunk hit the first owner's id, was silently dropped,
and never stored under their owner — their owner-filtered search then
quietly omitted it. Hash owner + text; empty owner reproduces the
legacy id, so the unowned/base index keeps existing ids and isn't
re-churned. Same-owner identical chunks still dedupe.
Caught by #1738 and #1760 (independent reports of the same bug).
* Removed duplicate definition of _preview_text()
---------
Co-authored-by: Claude Sonnet 4.6 <noreply@anthropic.com>
Co-authored-by: Zeus-Deus <100132710+Zeus-Deus@users.noreply.github.com>
Co-authored-by: lekt8 <lewistham9x@gmail.com>
Co-authored-by: Afonso Coutinho <afonso@omelhorsite.pt>
Co-authored-by: Paulo Victor Cordeiro <146781332+pvcordeiro@users.noreply.github.com>
Co-authored-by: Zarl-prog <asimjunaidi5u@gmail.com>
Co-authored-by: Wes Huber <wesleybaxterhuber@gmail.com>
Co-authored-by: .bulat <its.bulat@icloud.com>
Co-authored-by: Mahdi Salmanzade <mahdisalmanzadehasl@gmail.com>
Co-authored-by: red person <redpersoncoding@gmail.com>
Co-authored-by: pewdiepie-archdaemon <pewdiepie-archdaemon@users.noreply.github.com>
2026-06-04 16:18:39 +05:30
if ( ! controller . signal . aborted && ! _accessDenied && task . type === 'download' && task . payload && _dlN < _DL _MAX _AUTO _RETRY ) {
2026-05-31 23:58:26 +09:00
// Auto-retry: kill the dead session and re-launch (resumes from
// the cached .incomplete files) after a short delay.
_dlRetryCount . set ( _dlKey , _dlN + 1 ) ;
badge . textContent = ` retrying ( ${ _dlN + 1 } / ${ _DL _MAX _AUTO _RETRY } )… ` ;
badge . className = 'cookbook-task-status cookbook-task-running' ;
uiModule . showToast ( ` Download interrupted — retrying ( ${ _dlN + 1 } / ${ _DL _MAX _AUTO _RETRY } ), resumes where it stopped… ` , 6000 ) ;
const _p = task . payload , _nm = task . name ;
try {
await fetch ( '/api/shell/exec' , {
method : 'POST' , credentials : 'same-origin' ,
headers : { 'Content-Type' : 'application/json' } ,
body : JSON . stringify ( { command : _tmuxCmd ( task , ` kill-session -t ${ task . sessionId } ` ) } ) ,
} ) ;
} catch { }
_removeTask ( task . sessionId ) ;
setTimeout ( ( ) => { _retryDownload ( _nm , _p ) ; } , 8000 ) ;
break ;
}
// Out of auto-retries (or not a download) — surface the error; the
// card's Retry button stays available to resume manually.
badge . textContent = _statusLabel ( 'error' , task . type ) ;
badge . className = 'cookbook-task-status cookbook-task-error' ;
_updateTask ( task . sessionId , { status : 'error' } ) ;
el . dataset . status = 'error' ;
// Explain a gated/access failure with actionable buttons (request
// access on HF, check token) — otherwise it's just raw red text.
if ( _accessDenied ) {
const _diag = _diagnose ( snapshot ) ;
if ( _diag ) {
let diagEl = el . querySelector ( '.cookbook-diagnosis' ) ;
if ( ! diagEl ) { diagEl = document . createElement ( 'div' ) ; diagEl . className = 'cookbook-diagnosis' ; el . appendChild ( diagEl ) ; }
_showDiagnosis ( el , _diag , snapshot ) ;
}
}
_showCookbookNotif ( true ) ;
break ;
}
if ( snapshot . includes ( 'DOWNLOAD_OK' ) || ( snapshot . includes ( '/snapshots/' ) && completed >= totalFiles && totalFiles > 0 ) ) {
2026-06-05 14:41:07 +02:00
_clearDiagnosis ( el ) ;
2026-05-31 23:58:26 +09:00
_dlRetryCount . delete ( task . payload ? . repo _id || task . name ) ;
badge . textContent = _statusLabel ( 'done' , task . type ) ;
badge . className = 'cookbook-task-status cookbook-task-done' ;
// Flip the type chip from "download" to the green "finished"
// badge so the header reads as completed without a stale label.
const _typeChip = el . querySelector ( '.cookbook-task-type' ) ;
if ( _typeChip ) { _typeChip . textContent = 'finished' ; _typeChip . classList . add ( 'cookbook-task-type-done' ) ; }
_updateTask ( task . sessionId , { status : 'done' } ) ;
const _sb2 = el . querySelector ( '.cookbook-task-serve-btn' ) ; if ( _sb2 ) _sb2 . style . display = '' ;
_showCookbookNotif ( ) ;
2026-06-01 18:58:06 +05:30
_refreshDepsAfterInstall ( task ) ;
2026-05-31 23:58:26 +09:00
fetch ( '/api/shell/exec' , {
method : 'POST' , credentials : 'same-origin' ,
headers : { 'Content-Type' : 'application/json' } ,
body : JSON . stringify ( { command : _tmuxCmd ( task , ` kill-session -t ${ task . sessionId } ` ) } ) ,
} ) . catch ( ( ) => { } ) ;
_processQueue ( ) ;
break ;
}
}
}
// Live status parsing for serve tasks — uses shared _parseServePhase
if ( task . type === 'serve' ) {
const badge = el . querySelector ( '.cookbook-task-status' ) ;
if ( badge ) {
const info = _parseServePhase ( snapshot ) ;
if ( info . status === 'ready' && ! task . _serveReady ) {
task . _serveReady = true ;
_updateTask ( task . sessionId , { _serveReady : true } ) ;
Cookbook UI: Ollama browser, advanced serve fold, API tokens form, diagnosis toolbar, polish
Surface a lot of accumulated cookbook + UI work as a single non-agent
commit so the agent rework lands cleanly.
Highlights:
- Ollama as a first-class backend in the Cookbook:
* Download input accepts ollama-style names (name:tag) → backend=ollama
* /api/cookbook/ollama/library (cached scrape of ollama.com + curated
fallback so classic models like qwen2.5 stay reachable)
* "Browse Ollama library" toggle below Download with size chips
* Engine=Ollama in hwfit toolbar merges the Ollama library into the
main scan list as per-tag rows with the same Fit/Param/Quant/VRAM
columns; click → fills Download input
- API Tokens form added to Integrations panel (matching wired
loadTokens()/initTokenForm() that had no HTML)
- Serve panel polish: Advanced fold tightening (-8px nudges on vLLM
checks, Extra args, Spec row), n_cpu_moe + Split Mode controls
pulled up 8px to align with the row's checkboxes, GGUF File dropdown
exposed for Ollama backend, GPU re-render on Edit serve restore,
_forceBackend flag so saved serveState wins over backend detection,
cookbook:servers-changed CustomEvent so panels don't need refresh
- Models page redesign: Add Models row (URL + hidden API key reveal +
Type select + Scan/Ollama/Key/Test/Add icon buttons), Probe All +
Clear-offline buttons in Added Models toolbar, offline-pill removed
(opacity already conveys state), Engine dropdown gains Ollama option
- _ping_endpoint probes /v1/models then base, accepts 4xx as
reachable (vLLM returns 404 on bare /v1, fully working endpoints
were showing offline)
- Diagnosis card: × dismiss + Copy bundle buttons restored on the
serve error feedback card
- Orphan tmux sweep re-enabled behind a 60s rate-limit + background
Thread (off the main event loop) so dead serves get discovered
- cookbook_routes auto-register watchdog: drops the endpoint if the
serve session exits non-zero within the first ~3min
- ollama-rocm sidecar awareness in download wrapper (`docker exec
ollama-rocm ollama pull` when host ollama isn't installed)
- Skill extractor sets initial_status="published" when
auto_approve_skills pref is on (audit demotes later)
- Skill list / model list / cookbook scan misc polish
2026-06-08 22:38:49 +09:00
// The auto-registered endpoint was marked offline while the
// server was coming up. Now that it's reachable, nudge the
// picker to re-probe so the offline pill clears without the
// user having to reopen Settings or refresh the page.
try { _refreshModelsAfterEndpointChange ( ) ; } catch ( _ ) { }
2026-05-31 23:58:26 +09:00
}
if ( info . phase ) {
badge . textContent = info . phase ;
// Always the green "running" style — loading/warming is the same
// state, just with dynamic text (don't switch to a neutral style).
badge . className = 'cookbook-task-status cookbook-task-running' ;
// Live output reporting 'ready' is direct proof the server is up —
// clear a stale "unreachable" flag here too. The HTTP probe can lag,
// miss a remote endpoint, or cache a down result, leaving the card
// stuck red even after the server recovered ("doesn't recheck").
if ( info . status === 'ready' && task . _unreachable ) {
task . _unreachable = false ;
_updateTask ( task . sessionId , { _unreachable : false } ) ;
el . classList . remove ( 'cookbook-task-unreachable' ) ;
_refreshServerDots ( ) ;
}
// Persist the loading phase so a re-render keeps showing "loading 45%"
// instead of resetting the badge to the generic "running". Clear it
// once ready so the badge falls back to "running".
if ( info . status !== 'ready' ) {
if ( task . progress !== info . phase ) _updateTask ( task . sessionId , { progress : info . phase } ) ;
} else if ( task . progress ) {
_updateTask ( task . sessionId , { progress : '' } ) ;
}
}
}
}
// Run error diagnosis on serve tasks
const diag = _diagnose ( snapshot ) ;
if ( diag ) {
let diagEl = el . querySelector ( '.cookbook-diagnosis' ) ;
if ( ! diagEl ) {
diagEl = document . createElement ( 'div' ) ;
diagEl . className = 'cookbook-diagnosis' ;
el . appendChild ( diagEl ) ;
}
_showDiagnosis ( el , diag , snapshot ) ;
}
// Detect serve ready — auto-add to model endpoints. Don't flip
// `_endpointAdded` until the POST succeeds; otherwise a transient
// error silently prevents any future retry. An in-flight guard
// prevents a second poll from firing a duplicate POST before the
// first one's dedup check can observe the newly-added row.
if ( task . type === 'serve' && ! task . _endpointAdded && ! task . _endpointAddInFlight && task . _serveReady ) {
task . _endpointAddInFlight = true ;
2026-06-02 10:09:48 -05:00
let host = _connectHostFromRemote ( task . remoteHost ) ;
2026-05-31 23:58:26 +09:00
const portMatch = task . payload ? . _cmd ? . match ( /--port[=\s]+(\d+)/ )
|| task . payload ? . _cmd ? . match ( /(?:^|\s)-p[=\s]+(\d+)/ )
|| snapshot . match ( /Uvicorn running on\D*?:(\d+)/i )
|| snapshot . match ( /running on\D*?:(\d+)/i )
|| snapshot . match ( /listening on\D*?:(\d+)/i )
|| snapshot . match ( /port[:=\s]+(\d+)/i ) ;
Cookbook: scoring fixes, UI polish, false-finished + stale-state bug fixes
Backend (services/hwfit + routes):
- rank_models picks visible set by REQUESTED column, not always score —
sorting by Param now shows highest-param models PERIOD (incl. too_tight).
- New fit_only param. Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang
cannot serve them); default non-prequantized to BF16 on 2+ GPUs.
- AWQ / GPTQ-8bit get a -1.0 quality penalty (was 0.0, tied with FP8), so
FP8 wins when both fit.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above
M2.5 on equal composite score; >=100B integers not misread as versions.
- /api/cookbook/hf-latest no longer drops models without an "NB" pattern in
the repo id (MiniMax-M2.7, DeepSeek-V4-Pro etc. were silently filtered).
- Cached-model scan: atexit flushes models JSON even if the script is
killed mid-walk; each scan_dir wrapped in try/except; timeout 60s -> 180s.
- KB granularity for sub-MB sizes (was "0 MB" for 12 KB shells). New
"stalled" status for shells <1 MB with no .incomplete files.
- /api/cookbook/state POST guard: rejects "done" download tasks lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stops stale tabs from poisoning persisted state.
- hf_models.json: add zai-org/GLM-5.1; flip zai-org/GLM-5 quantization
Q4_K_M -> BF16 (it is the native base, not a quant).
Frontend (static/js):
- Scan/Download toolbar: quant defaults to All; ctx slider (8k/16k/32k/
50k/128k/Max) ported from origin/main with sort=fit on drag, sort=score
on Max. GPU toggle commits _activeCount to maxGpu on initial render. Fit
column header tagged with active budget (RAM / GPU / N GPU).
- Foldable Download admin-card: the Download h2 is the chevron trigger;
state persists in localStorage.
- Download card surfaces destination dir (Dir: <path>). Same dir on running
task row, font/color matched to uptime (9px Fira Code muted, opacity .4).
- Serve panel ctx text input always resets to model max on open. Sub-MB
cached models show with red "download stalled" badge.
- Bulk-select Cancel + Delete reset the Select button label on exit.
- Cookbook running: false-finished bug fixed — DOWNLOAD_OK or /snapshots/
required; bare "Download complete" no longer marks the task done after
the first config file. Clear button now sends tmux kill-session too.
True overall % for multi-shard downloads: ((N-1)+frac)/total instead of
hf_transfer per-shard aggregate.
- Diagnosis card simplified: removed fold toggle, copy button, dismiss X.
Suggestion font matches message body (12px).
- HF token field flashes green check + "Saved" on save.
- Cached scan no longer counts stalled rows as downloaded in Scan/Download.
CSS:
- dep Install button width pinned to 76px to match Installed split.
- task-sub row +1px; task-status badge gets margin-right 8px.
- Ctx slider styled like gallery editor sliders (thin pill rail, red thumb).
- Bulk-select cancel button top -3px -> -5px.
2026-06-03 16:32:20 +09:00
let port = portMatch ? portMatch [ 1 ] : '8000' ;
let baseUrl = ` http:// ${ host } : ${ port } /v1 ` ;
const ollamaUrlMatch = snapshot . match ( /Ollama API ready on port\s+\d+:\s*(http:\/\/[^\s]+)/i ) ;
if ( ollamaUrlMatch ) {
2026-06-02 10:09:48 -05:00
const endpoint = _endpointFromAdvertisedUrl ( ollamaUrlMatch [ 1 ] , host , '11434' ) ;
if ( endpoint ) ( { host , port , baseUrl } = endpoint ) ;
Cookbook: scoring fixes, UI polish, false-finished + stale-state bug fixes
Backend (services/hwfit + routes):
- rank_models picks visible set by REQUESTED column, not always score —
sorting by Param now shows highest-param models PERIOD (incl. too_tight).
- New fit_only param. Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang
cannot serve them); default non-prequantized to BF16 on 2+ GPUs.
- AWQ / GPTQ-8bit get a -1.0 quality penalty (was 0.0, tied with FP8), so
FP8 wins when both fit.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above
M2.5 on equal composite score; >=100B integers not misread as versions.
- /api/cookbook/hf-latest no longer drops models without an "NB" pattern in
the repo id (MiniMax-M2.7, DeepSeek-V4-Pro etc. were silently filtered).
- Cached-model scan: atexit flushes models JSON even if the script is
killed mid-walk; each scan_dir wrapped in try/except; timeout 60s -> 180s.
- KB granularity for sub-MB sizes (was "0 MB" for 12 KB shells). New
"stalled" status for shells <1 MB with no .incomplete files.
- /api/cookbook/state POST guard: rejects "done" download tasks lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stops stale tabs from poisoning persisted state.
- hf_models.json: add zai-org/GLM-5.1; flip zai-org/GLM-5 quantization
Q4_K_M -> BF16 (it is the native base, not a quant).
Frontend (static/js):
- Scan/Download toolbar: quant defaults to All; ctx slider (8k/16k/32k/
50k/128k/Max) ported from origin/main with sort=fit on drag, sort=score
on Max. GPU toggle commits _activeCount to maxGpu on initial render. Fit
column header tagged with active budget (RAM / GPU / N GPU).
- Foldable Download admin-card: the Download h2 is the chevron trigger;
state persists in localStorage.
- Download card surfaces destination dir (Dir: <path>). Same dir on running
task row, font/color matched to uptime (9px Fira Code muted, opacity .4).
- Serve panel ctx text input always resets to model max on open. Sub-MB
cached models show with red "download stalled" badge.
- Bulk-select Cancel + Delete reset the Select button label on exit.
- Cookbook running: false-finished bug fixed — DOWNLOAD_OK or /snapshots/
required; bare "Download complete" no longer marks the task done after
the first config file. Clear button now sends tmux kill-session too.
True overall % for multi-shard downloads: ((N-1)+frac)/total instead of
hf_transfer per-shard aggregate.
- Diagnosis card simplified: removed fold toggle, copy button, dismiss X.
Suggestion font matches message body (12px).
- HF token field flashes green check + "Saved" on save.
- Cached scan no longer counts stalled rows as downloaded in Scan/Download.
CSS:
- dep Install button width pinned to 76px to match Installed split.
- task-sub row +1px; task-status badge gets margin-right 8px.
- Ctx slider styled like gallery editor sliders (thin pill rail, red thumb).
- Bulk-select cancel button top -3px -> -5px.
2026-06-03 16:32:20 +09:00
}
2026-05-31 23:58:26 +09:00
fetch ( '/api/model-endpoints' , { credentials : 'same-origin' } )
. then ( r => r . json ( ) )
. then ( async ( eps ) => {
// Match only exact base_url — don't dedup by friendly name,
// because other endpoints may happen to share a model name.
const exists = eps . some ( e => e . base _url === baseUrl ) ;
if ( exists ) {
// Already registered — e.g. the backend pre-registers diffusion
// endpoints server-side. Mark so we don't retry, but STILL
// refresh the picker (and probe until online) so the new model
// shows up without the user having to manually refresh.
2026-07-07 00:50:07 +00:00
const _ex = eps . find ( e => e . base _url === baseUrl ) ;
if ( _ex && ! _endpointMatchesServe ( _ex , task ) ) {
_markServeEndpointMismatch ( task , _ex , host , port ) ;
return null ;
}
2026-05-31 23:58:26 +09:00
task . _endpointAdded = true ;
_updateTask ( task . sessionId , { _endpointAdded : true } ) ;
_autoSaveWorkingConfig ( task ) ; // endpoint live → remember these settings
2026-07-07 00:50:07 +00:00
if ( window . modelsModule ? . refreshModels ) await window . modelsModule . refreshModels ( false ) ;
2026-05-31 23:58:26 +09:00
if ( window . sessionModule ? . updateModelPicker ) window . sessionModule . updateModelPicker ( ) ;
window . dispatchEvent ( new CustomEvent ( 'ge:model-endpoints-updated' , { detail : { baseUrl , host , port , model : task . name } } ) ) ;
if ( _ex && _ex . id && ! ( _ex . models || [ ] ) . length ) _probeEndpointUntilOnline ( _ex . id , host , port ) ;
return null ;
}
const _isDiffusion = task . payload ? . _cmd ? . includes ( 'diffusion_server' ) ;
const fd = new FormData ( ) ;
fd . append ( 'base_url' , baseUrl ) ;
fd . append ( 'name' , task . name ) ;
fd . append ( 'skip_probe' , 'true' ) ;
2026-06-02 10:09:48 -05:00
_appendCookbookEndpointScope ( fd , task . remoteHost || '' ) ;
2026-07-07 00:50:07 +00:00
_appendPinnedServeModel ( fd , task ) ;
2026-05-31 23:58:26 +09:00
if ( _isDiffusion ) fd . append ( 'model_type' , 'image' ) ;
return fetch ( '/api/model-endpoints' , { method : 'POST' , credentials : 'same-origin' , body : fd } ) ;
} )
. then ( async ( res ) => {
if ( res && res . ok ) {
// Flip the flag only on confirmed success
task . _endpointAdded = true ;
_updateTask ( task . sessionId , { _endpointAdded : true } ) ;
_autoSaveWorkingConfig ( task ) ; // endpoint live → remember these settings
uiModule . showToast ( ` Model endpoint added: ${ host } : ${ port } ` ) ;
// Retry-probe until the warming server answers, so it
// flips online without a manual enable/disable toggle.
const _epData = await res . json ( ) . catch ( ( ) => ( { } ) ) ;
if ( _epData && _epData . id && ! ( _epData . models || [ ] ) . length ) {
_probeEndpointUntilOnline ( _epData . id , host , port ) ;
}
window . dispatchEvent ( new CustomEvent ( 'ge:model-endpoints-updated' , { detail : { baseUrl , host , port , model : task . name } } ) ) ;
const _trySelectModel = async ( attempt ) => {
2026-07-07 00:50:07 +00:00
if ( window . modelsModule ? . refreshModels ) await window . modelsModule . refreshModels ( false ) ;
2026-05-31 23:58:26 +09:00
const items = window . modelsModule ? . getCachedItems ? . ( ) || [ ] ;
for ( const item of items ) {
if ( item . offline ) continue ;
const url = item . url || '' ;
if ( url . includes ( host ) || url . includes ( port ) ) {
const mid = ( item . models || [ ] ) [ 0 ] ;
if ( mid && window . sessionModule ? . createDirectChat ) {
window . sessionModule . createDirectChat ( url , mid , item . endpoint _id ) ;
if ( window . sessionModule ? . updateModelPicker ) window . sessionModule . updateModelPicker ( ) ;
uiModule . showToast ( ` Switched to ${ mid . split ( '/' ) . pop ( ) } ` ) ;
return ;
}
}
}
if ( attempt < 3 ) setTimeout ( ( ) => _trySelectModel ( attempt + 1 ) , 2000 ) ;
else if ( window . sessionModule ? . updateModelPicker ) window . sessionModule . updateModelPicker ( ) ;
} ;
setTimeout ( ( ) => _trySelectModel ( 0 ) , 1000 ) ;
} else if ( res && ! res . ok ) {
const body = await res . text ( ) . catch ( ( ) => '' ) ;
console . warn ( 'Endpoint auto-add failed' , res . status , body ) ;
uiModule . showError ( ` Auto-register endpoint failed ( ${ res . status } ). Use ⋮ → Register endpoint to retry. ` ) ;
}
} )
. catch ( ( e ) => {
console . warn ( 'Endpoint auto-add error' , e ) ;
uiModule . showError ( ` Auto-register endpoint error: ${ e . message || e } . Use ⋮ → Register endpoint to retry. ` ) ;
} )
. finally ( ( ) => { task . _endpointAddInFlight = false ; } ) ;
_updateTask ( task . sessionId , { status : 'running' } ) ;
const badge = el . querySelector ( '.cookbook-task-status' ) ;
if ( badge ) { badge . textContent = 'running' ; badge . className = 'cookbook-task-status cookbook-task-running' ; }
_showCookbookNotif ( ) ;
}
// Detect process exit
if ( snapshot . includes ( '=== Process exited with code' ) ) {
const codeMatch = snapshot . match ( /=== Process exited with code (\d+)/ ) ;
const code = codeMatch ? parseInt ( codeMatch [ 1 ] ) : - 1 ;
// Serve tasks that exit without reaching ready state are always errors —
// a serve process should run indefinitely
const status = ( task . type === 'serve' && ! task . _serveReady ) ? 'error'
: ( code === 0 ? 'done' : 'error' ) ;
_updateTask ( task . sessionId , { status } ) ;
const badge = el . querySelector ( '.cookbook-task-status' ) ;
if ( badge ) { badge . textContent = status ; badge . className = ` cookbook-task-status cookbook-task- ${ status } ` ; }
_renderRunningTab ( ) ;
}
_updateTask ( task . sessionId , { output : snapshot . slice ( - 5000 ) } ) ;
}
} catch {
failCount ++ ;
if ( failCount > 10 ) break ;
await new Promise ( r => setTimeout ( r , 10000 ) ) ;
continue ;
}
failCount = 0 ;
await new Promise ( r => setTimeout ( r , TASK _POLL _INTERVAL _MS ) ) ;
}
}
// ── Background monitor ──
let _bgMonitorInterval = null ;
2026-07-01 10:09:25 +00:00
let _bgPollInFlight = false ;
const BG _LEADER _KEY = 'odysseus-cookbook-bg-leader' ;
const BG _LEADER _ID = ` ${ Date . now ( ) } - ${ Math . random ( ) . toString ( 36 ) . slice ( 2 ) } ` ;
const BG _LEADER _TTL _MS = 15000 ;
2026-05-31 23:58:26 +09:00
2026-06-27 13:50:21 +00:00
function _hasLiveTasks ( tasks = null ) {
const list = tasks || _loadTasks ( ) ;
return list . some ( t =>
t . status === 'running'
|| t . status === 'queued'
|| t . status === 'ready'
|| _downloadOutputLooksActive ( t )
) ;
}
function _isRunningTabVisible ( ) {
const modal = document . getElementById ( 'cookbook-modal' ) ;
if ( ! modal || modal . classList . contains ( 'hidden' ) ) return false ;
const activeTab = modal . querySelector ( '.cookbook-tab.active' ) ? . dataset ? . backend || '' ;
return activeTab === 'Running' ;
}
2026-07-07 00:50:07 +00:00
function _isCookbookVisible ( ) {
try {
if ( window . cookbookModule && typeof window . cookbookModule . isVisible === 'function' ) {
return ! ! window . cookbookModule . isVisible ( ) ;
}
} catch ( _ ) { }
const modal = document . getElementById ( 'cookbook-modal' ) ;
return ! ! modal && ! modal . classList . contains ( 'hidden' ) ;
}
2026-07-01 10:09:25 +00:00
function _foregroundChatBusy ( ) {
try {
return ! ! window . _ _odysseusChatBusy || Date . now ( ) < ( window . _ _odysseusChatBusyUntil || 0 ) ;
} catch {
return false ;
}
}
function _claimBackgroundLeader ( ) {
if ( document . visibilityState !== 'visible' ) return false ;
const now = Date . now ( ) ;
try {
const raw = localStorage . getItem ( BG _LEADER _KEY ) ;
const current = raw ? JSON . parse ( raw ) : null ;
if (
! current
|| ! current . id
|| current . id === BG _LEADER _ID
|| now - Number ( current . ts || 0 ) > BG _LEADER _TTL _MS
) {
localStorage . setItem ( BG _LEADER _KEY , JSON . stringify ( { id : BG _LEADER _ID , ts : now } ) ) ;
return true ;
}
return current . id === BG _LEADER _ID ;
} catch ( _ ) {
return true ;
}
}
function _canBackgroundPoll ( ) {
if ( _foregroundChatBusy ( ) ) return false ;
if ( document . visibilityState !== 'visible' ) return false ;
return _claimBackgroundLeader ( ) ;
}
2026-05-31 23:58:26 +09:00
// Reachability check for running serve tasks. The tmux pane can stay alive
// while the model server inside it has crashed (so no "Process exited" line
// ever appears) — leaving the card showing "running" forever. So we actively
// probe the registered endpoint (same /probe-local the model picker uses) and
// flag the card "unreachable" (red) when the server stops answering.
2026-07-01 10:09:25 +00:00
let _serveReachabilityInFlight = false ;
let _serveReachabilityLastAt = 0 ;
2026-05-31 23:58:26 +09:00
async function _checkServeReachability ( ) {
2026-07-01 10:09:25 +00:00
// This reaches out to local model servers. Keep it out of the normal chat
// path unless the user is actively looking at the Running tab.
if ( _foregroundChatBusy ( ) ) return ;
if ( ! _isRunningTabVisible ( ) ) return ;
const now = Date . now ( ) ;
if ( _serveReachabilityInFlight || now - _serveReachabilityLastAt < 10000 ) return ;
_serveReachabilityInFlight = true ;
_serveReachabilityLastAt = now ;
2026-05-31 23:58:26 +09:00
let serveTasks ;
try {
serveTasks = _loadTasks ( ) . filter ( t => t . type === 'serve' && t . status === 'running' ) ;
2026-07-01 10:09:25 +00:00
} catch {
_serveReachabilityInFlight = false ;
return ;
}
if ( ! serveTasks . length ) {
_serveReachabilityInFlight = false ;
return ;
}
2026-05-31 23:58:26 +09:00
let eps = [ ] , probe = { } ;
try {
[ eps , probe ] = await Promise . all ( [
fetch ( '/api/model-endpoints' , { credentials : 'same-origin' } ) . then ( r => r . json ( ) ) . catch ( ( ) => [ ] ) ,
fetch ( '/api/model-endpoints/probe-local' , { credentials : 'same-origin' } ) . then ( r => r . json ( ) ) . catch ( ( ) => ( { } ) ) ,
] ) ;
2026-07-01 10:09:25 +00:00
for ( const task of serveTasks ) {
const host = _connectHostFromRemote ( task . remoteHost ) ;
const portMatch = task . payload ? . _cmd ? . match ( /--port\s+(\d+)/ ) ;
const port = portMatch ? portMatch [ 1 ] : '8000' ;
const baseUrl = ` http:// ${ host } : ${ port } /v1 ` ;
const ep = ( eps || [ ] ) . find ( e => e . base _url === baseUrl ) ;
if ( ! ep ) continue ; // not registered yet — can't judge
const pr = probe [ ep . id ] ;
if ( ! pr || pr . alive === undefined ) continue ; // not probed (non-local) — skip
// Record the first time it actually answers. Until then the server is still
// LOADING/warming (the endpoint can get registered on the 300s timeout for a
// big model that hasn't finished loading), and a not-yet-answering server is
// not "unreachable" — flagging it as such while you're launching is a false
// alarm. Only treat it as unreachable once it has been reachable at least once.
if ( pr . alive === true && ! task . _everReachable ) {
task . _everReachable = true ;
_updateTask ( task . sessionId , { _everReachable : true } ) ;
}
const unreachable = pr . alive === false ;
if ( unreachable && ! task . _everReachable ) continue ; // still coming up, not crashed
if ( ! ! task . _unreachable !== unreachable ) {
_updateTask ( task . sessionId , { _unreachable : unreachable } ) ;
}
const el = document . querySelector ( ` .cookbook-task[data-task-id=" ${ task . sessionId } "] ` ) ;
if ( el ) {
el . classList . toggle ( 'cookbook-task-unreachable' , unreachable ) ;
const badge = el . querySelector ( '.cookbook-task-status' ) ;
if ( badge ) {
if ( unreachable ) {
badge . textContent = 'unreachable' ;
badge . className = 'cookbook-task-status cookbook-task-error' ;
badge . title = pr . error || 'Server not responding — it may have crashed' ;
} else if ( badge . textContent === 'unreachable' ) {
// Recovered — restore the normal running label.
badge . textContent = _statusLabel ( 'running' , task . type ) ;
badge . className = 'cookbook-task-status cookbook-task-running' ;
badge . title = '' ;
}
2026-05-31 23:58:26 +09:00
}
}
2026-07-01 10:09:25 +00:00
if ( unreachable ) _showCookbookNotif ( true ) ;
2026-05-31 23:58:26 +09:00
}
2026-07-01 10:09:25 +00:00
_refreshServerDots ( ) ;
} catch {
// Non-fatal: the normal task status poll continues separately.
} finally {
_serveReachabilityInFlight = false ;
2026-05-31 23:58:26 +09:00
}
}
function _serveTaskFailed ( task ) {
if ( ! task || task . type !== 'serve' ) return false ;
return ! ! task . _unreachable || [ 'error' , 'crashed' , 'failed' ] . includes ( task . status ) ;
}
function _setServerDot ( dot , failed , title ) {
if ( ! dot ) return ;
dot . classList . toggle ( 'fail' , ! ! failed ) ;
dot . classList . toggle ( 'ok' , ! failed ) ;
dot . title = title ;
}
function _syncSettingsServerDots ( byKey ) {
document . querySelectorAll ( '.cookbook-server-entry' ) . forEach ( entry => {
const hostEl = entry . querySelector ( '.cookbook-srv-host' ) ;
const dot = entry . querySelector ( '.cookbook-srv-status' ) ;
const msg = entry . querySelector ( '.cookbook-srv-test-msg' ) ;
if ( ! hostEl || ! dot ) return ;
const host = hostEl . value ? . trim ( ) || '' ;
if ( ! host || hostEl . readOnly || hostEl . disabled ) {
_setServerDot ( dot , false , 'Local (this machine)' ) ;
return ;
}
const list = byKey [ host ] || [ ] ;
if ( ! list . length ) return ;
const failed = list . some ( _serveTaskFailed ) ;
_setServerDot ( dot , failed , failed ? 'Server not responding - running serve may have crashed' : 'Reachable' ) ;
if ( ! msg ) return ;
if ( failed ) {
msg . textContent = 'Server not responding' ;
msg . title = 'Server not responding - running serve may have crashed' ;
msg . style . color = 'var(--red,#e06c75)' ;
msg . style . opacity = '0.75' ;
} else if ( /failed|crashed|not responding|unreachable/i . test ( msg . textContent || '' ) ) {
msg . textContent = 'Reachable' ;
msg . title = 'Reachable' ;
msg . style . color = 'var(--green,#50fa7b)' ;
msg . style . opacity = '0.75' ;
}
} ) ;
}
// Keep each server section's status dot (green ↔ red) in sync with the live
// health of its SERVE tasks. The header dot is only built once, so without
// this it got stuck on its first value. Downloads never count because they have
// no endpoint to be "unreachable".
function _refreshServerDots ( ) {
let tasks ;
try { tasks = _loadTasks ( ) ; } catch { return ; }
const byKey = { } ;
2026-06-21 11:02:35 +00:00
const _taskServerKeyForDot = ( task ) => task ? . remoteServerKey || task ? . remoteHost || '' ;
for ( const t of tasks ) { ( byKey [ _taskServerKeyForDot ( t ) ] = byKey [ _taskServerKeyForDot ( t ) ] || [ ] ) . push ( t ) ; }
2026-05-31 23:58:26 +09:00
document . querySelectorAll ( '.cookbook-section-header' ) . forEach ( header => {
const dot = header . querySelector ( '.cookbook-srv-status' ) ;
if ( ! dot ) return ;
const key = header . querySelector ( '[data-stop-server]' ) ? . dataset . stopServer || '' ;
const list = byKey [ key ] || [ ] ;
const fail = ! ! key && list . some ( _serveTaskFailed ) ;
_setServerDot ( dot , fail , key ? ( fail ? 'Server not responding' : 'Reachable' ) : 'Local (this machine)' ) ;
} ) ;
_syncSettingsServerDots ( byKey ) ;
}
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
// Self-heal: scan persisted download tasks marked done/error/crashed and
// check whether their tmux session is still alive on the host. If yes —
// the task isn't actually finished, the cookbook just lost the in-flight
// status during restart — flip status back to 'running' so _reconnectTask
// picks it up. The one-shot guard is enforced by callers (open path) or
// time-throttled inside (background-monitor path).
let _selfHealRan = false ;
let _selfHealLastTs = 0 ;
export async function _selfHealStaleTasks ( opts = { } ) {
// Open-path call: one-shot per page load.
if ( opts . oneShot ) {
if ( _selfHealRan ) return ;
_selfHealRan = true ;
} else {
// Background-monitor call: throttle to once every 8s (the bg monitor
// itself fires every 10s, so this almost always fires too, but the
// guard keeps a fast manual call from doubling up).
const now = Date . now ( ) ;
cookbook agent debug loop: persistent log files, auto-adopt orphan tmux, Codex/Claude skill parity
Three converging fixes so the chat agent + external Codex/Claude skills can actually debug a crashed serve instead of staring at a post-crash neofetch banner:
* Serves now `tee` to /tmp/odysseus-tmux/SESSION.log on the host running them. Runner saves fds 3/4 before the tee and restores them right before `exec ${SHELL}`, so the post-crash interactive zsh banner does NOT pollute the log file.
* `tail_serve_output` (chat agent) and `/api/codex/cookbook/output/{sid}` (Codex+Claude skills) both prefer the persistent log file over the tmux pane. Pane is fallback for sessions predating the tee runner. Default tail bumped 150 -> 400.
* `list_served_models` "recent log" snippet seeks to the Traceback line instead of showing the last 6 lines (which was always the bash prompt).
Cookbook auto-adoption sweep on `/api/cookbook/tasks/status`: every 20s (rate-limited) the cookbook SSHes each configured server, finds `serve-*` / `cookbook-*` tmux sessions running an actual model process (vllm/python/llama-server/etc., filtered via `pane_current_command`), and writes them into state.tasks. So when the agent falls back to raw ssh+tmux, the session appears in the Cookbook UI on the next poll.
`serve_model` error path now reads `data["detail"]` in addition to `data["error"]` so the FastAPI HTTPException message ("Invalid characters in cmd") actually reaches the agent instead of being swallowed as a generic "Serve failed". Tool description updated to warn against `cd …`/`source …`/`&&` prefixes.
Intent-without-action supervisor in agent_loop: when the model writes "Let me tail the output" / "I'll check the logs" / "Let me investigate" and ends the turn without emitting a tool call, the loop injects a sharp system nudge ("You said you would X — DO IT NOW") and continues. Capped at 2 nudges per chat so a model that genuinely cannot use the tool does not pin the loop.
Codex/Claude skill parity: adds `/cookbook/cached`, `/cookbook/presets`, `/cookbook/preset/{name}`, `/cookbook/adopt` so external agents have the same surface as the chat agent. SKILL.md docs + odysseus_api.py wrapper updated for both bundles.
`adopt_served_model` promoted to the always-on tool set so the agent has a documented fallback when serve_model rejects a cmd.
Also various cookbook UI tweaks accumulated alongside the above (cookbook.js, cookbookRunning.js, cookbookServe.js, cookbook-diagnosis.js, settings.js, style.css).
2026-06-04 23:27:18 +09:00
if ( now - _selfHealLastTs < 4000 ) return ;
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
_selfHealLastTs = now ;
}
const tasks = _loadTasks ( ) ;
cookbook agent debug loop: persistent log files, auto-adopt orphan tmux, Codex/Claude skill parity
Three converging fixes so the chat agent + external Codex/Claude skills can actually debug a crashed serve instead of staring at a post-crash neofetch banner:
* Serves now `tee` to /tmp/odysseus-tmux/SESSION.log on the host running them. Runner saves fds 3/4 before the tee and restores them right before `exec ${SHELL}`, so the post-crash interactive zsh banner does NOT pollute the log file.
* `tail_serve_output` (chat agent) and `/api/codex/cookbook/output/{sid}` (Codex+Claude skills) both prefer the persistent log file over the tmux pane. Pane is fallback for sessions predating the tee runner. Default tail bumped 150 -> 400.
* `list_served_models` "recent log" snippet seeks to the Traceback line instead of showing the last 6 lines (which was always the bash prompt).
Cookbook auto-adoption sweep on `/api/cookbook/tasks/status`: every 20s (rate-limited) the cookbook SSHes each configured server, finds `serve-*` / `cookbook-*` tmux sessions running an actual model process (vllm/python/llama-server/etc., filtered via `pane_current_command`), and writes them into state.tasks. So when the agent falls back to raw ssh+tmux, the session appears in the Cookbook UI on the next poll.
`serve_model` error path now reads `data["detail"]` in addition to `data["error"]` so the FastAPI HTTPException message ("Invalid characters in cmd") actually reaches the agent instead of being swallowed as a generic "Serve failed". Tool description updated to warn against `cd …`/`source …`/`&&` prefixes.
Intent-without-action supervisor in agent_loop: when the model writes "Let me tail the output" / "I'll check the logs" / "Let me investigate" and ends the turn without emitting a tool call, the loop injects a sharp system nudge ("You said you would X — DO IT NOW") and continues. Capped at 2 nudges per chat so a model that genuinely cannot use the tool does not pin the loop.
Codex/Claude skill parity: adds `/cookbook/cached`, `/cookbook/presets`, `/cookbook/preset/{name}`, `/cookbook/adopt` so external agents have the same surface as the chat agent. SKILL.md docs + odysseus_api.py wrapper updated for both bundles.
`adopt_served_model` promoted to the always-on tool set so the agent has a documented fallback when serve_model rejects a cmd.
Also various cookbook UI tweaks accumulated alongside the above (cookbook.js, cookbookRunning.js, cookbookServe.js, cookbook-diagnosis.js, settings.js, style.css).
2026-06-04 23:27:18 +09:00
const candidates = tasks . filter ( t => {
if ( t . type !== 'download' ) return false ;
if ( ! [ 'done' , 'error' , 'crashed' , 'stopped' ] . includes ( t . status ) ) return false ;
if ( ! t . sessionId || String ( t . sessionId ) . startsWith ( 'queue-' ) ) return false ;
// Finished downloads with strong completion markers (DOWNLOAD_OK or HF
// /snapshots/ resolution) are demonstrably done — do not flip them back
// to running just because the tmux session is still alive (e.g., a
// long-lived shell that hosted the download or a flapping SSH that
// reports the session as up). This was the main source of finished↔
// downloading oscillation on a flaky connection.
if ( t . status === 'done' && /DOWNLOAD_OK|\/snapshots\// . test ( t . output || '' ) ) return false ;
// Cooldown: never flip the same task more than once every 45s. A flapping
// SSH connection used to drive the badge back-and-forth on every probe
// cycle; this enforces a stable view between flaps.
if ( t . _lastStatusFlipAt && ( Date . now ( ) - t . _lastStatusFlipAt < 45000 ) ) return false ;
return true ;
} ) ;
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
if ( ! candidates . length ) return ;
let flipped = 0 ;
for ( const t of candidates ) {
try {
const res = await fetch ( '/api/shell/exec' , {
method : 'POST' , credentials : 'same-origin' ,
headers : { 'Content-Type' : 'application/json' } ,
body : JSON . stringify ( { command : _tmuxCmd ( t , ` has-session -t ${ t . sessionId } ` ) , timeout : 5 } ) ,
} ) ;
const data = await res . json ( ) ;
if ( data . exit _code === 0 ) {
// Session still alive → the task is actually still running.
const fresh = _loadTasks ( ) ;
const ft = fresh . find ( x => x . sessionId === t . sessionId ) ;
if ( ft && ft . status !== 'running' ) {
ft . status = 'running' ;
ft . _selfHealed = true ;
cookbook agent debug loop: persistent log files, auto-adopt orphan tmux, Codex/Claude skill parity
Three converging fixes so the chat agent + external Codex/Claude skills can actually debug a crashed serve instead of staring at a post-crash neofetch banner:
* Serves now `tee` to /tmp/odysseus-tmux/SESSION.log on the host running them. Runner saves fds 3/4 before the tee and restores them right before `exec ${SHELL}`, so the post-crash interactive zsh banner does NOT pollute the log file.
* `tail_serve_output` (chat agent) and `/api/codex/cookbook/output/{sid}` (Codex+Claude skills) both prefer the persistent log file over the tmux pane. Pane is fallback for sessions predating the tee runner. Default tail bumped 150 -> 400.
* `list_served_models` "recent log" snippet seeks to the Traceback line instead of showing the last 6 lines (which was always the bash prompt).
Cookbook auto-adoption sweep on `/api/cookbook/tasks/status`: every 20s (rate-limited) the cookbook SSHes each configured server, finds `serve-*` / `cookbook-*` tmux sessions running an actual model process (vllm/python/llama-server/etc., filtered via `pane_current_command`), and writes them into state.tasks. So when the agent falls back to raw ssh+tmux, the session appears in the Cookbook UI on the next poll.
`serve_model` error path now reads `data["detail"]` in addition to `data["error"]` so the FastAPI HTTPException message ("Invalid characters in cmd") actually reaches the agent instead of being swallowed as a generic "Serve failed". Tool description updated to warn against `cd …`/`source …`/`&&` prefixes.
Intent-without-action supervisor in agent_loop: when the model writes "Let me tail the output" / "I'll check the logs" / "Let me investigate" and ends the turn without emitting a tool call, the loop injects a sharp system nudge ("You said you would X — DO IT NOW") and continues. Capped at 2 nudges per chat so a model that genuinely cannot use the tool does not pin the loop.
Codex/Claude skill parity: adds `/cookbook/cached`, `/cookbook/presets`, `/cookbook/preset/{name}`, `/cookbook/adopt` so external agents have the same surface as the chat agent. SKILL.md docs + odysseus_api.py wrapper updated for both bundles.
`adopt_served_model` promoted to the always-on tool set so the agent has a documented fallback when serve_model rejects a cmd.
Also various cookbook UI tweaks accumulated alongside the above (cookbook.js, cookbookRunning.js, cookbookServe.js, cookbook-diagnosis.js, settings.js, style.css).
2026-06-04 23:27:18 +09:00
ft . _lastStatusFlipAt = Date . now ( ) ;
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
_saveTasks ( fresh ) ;
flipped ++ ;
const _el = document . querySelector ( ` .cookbook-task[data-task-id=" ${ t . sessionId } "] ` ) ;
if ( _el ) {
const _chk = _el . querySelector ( '.cookbook-task-check' ) ;
if ( _chk ) _chk . style . display = 'none' ;
const _wave = _el . querySelector ( '.cookbook-task-wave' ) ;
if ( _wave ) _wave . style . display = '' ;
const _up = _el . querySelector ( '.cookbook-task-uptime' ) ;
if ( _up ) _up . style . display = '' ;
_el . dataset . status = 'running' ;
}
}
}
} catch { /* network blip — skip this one */ }
}
if ( flipped ) {
console . log ( ` [cookbook] auto-reconnect: revived ${ flipped } task(s) whose tmux session was still alive ` ) ;
_renderRunningTab ( ) ;
}
}
2026-05-31 23:58:26 +09:00
export function _startBackgroundMonitor ( ) {
if ( _bgMonitorInterval ) return ;
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
_bgMonitorInterval = setInterval ( ( ) => {
2026-07-01 10:09:25 +00:00
if ( ! _canBackgroundPoll ( ) ) return ;
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
_pollBackgroundStatus ( ) ;
_checkServeReachability ( ) ;
// Auto-reconnect: every cycle, look for download tasks marked finished/
// crashed/etc. whose tmux session is actually still running, and flip
// them back to running. Internally throttled to 8s so a manual call from
// the open path or a fast invocation doesn't double up.
2026-06-27 13:50:21 +00:00
if ( _hasLiveTasks ( ) || _isRunningTabVisible ( ) ) {
_selfHealStaleTasks ( ) . catch ( ( ) => { } ) ;
}
Cookbook polish: auto-reconnect, ctx slider fixes, scoring, lots of UI
Backend (services/hwfit + routes):
- VRAM column sort now shows global highest first (was special-cased to
ascending then truncated top-N, which made "highest VRAM" mathematically
unreachable). Every column path uses reverse=True for the truncation.
- Hardware probe cache TTL 30min -> 24h so changing filters doesn't keep
re-probing the rig during a session; Rescan button still forces fresh.
- Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang can't serve them);
default non-prequantized to BF16 on 2+ GPUs.
- AWQ / AWQ-8bit / GPTQ-8bit get a -1.0 quality penalty so FP8 wins ties.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above M2.5.
- hf_models.json: zai-org/GLM-5.1 added; zai-org/GLM-5 quantization flipped
Q4_K_M -> BF16. DeepSeek-V4-Flash / -Pro + their -Base variants registered
with new FP4-MoE-Mixed / FP8-Mixed quant keys (calibrated BPP from the
actual 156 GB / 284 GB disk footprints).
- New FP4-MoE-Mixed + FP8-Mixed entries in QUANT_BPP / QUANT_SPEED_MULT /
QUANT_QUALITY_PENALTY / QUANT_BYTES_PER_PARAM / PREQUANTIZED_PREFIXES.
Frontend — Scan/Download:
- Engine + Quant swapped in the toolbar; Quant defaults to "All".
- Ctx (range slider) ported from origin/main: 8k/16k/32k/50k/128k/Max. Drag
re-sorts by vram ascending (smallest fitting first); back to Max → score.
- Ctx slider rail now visible — was background:transparent in a duplicate
later-cascade rule. Hardcoded grey + !important.
- Search input moved to the far right of the toolbar.
- Type/Standard default; "Context" not uppercased; Search placeholder dimmed.
- Engine "?" + Quant "?" inline help chips inside their dropdown boxes.
- Fit-column dot toggles fit-only filter; un-toggling re-sorts by VRAM desc.
- Quant column truncates to 9 chars + ellipsis ("FP4-MoE-M..."), full in
tooltip. Smart title-suffix strips the parts already in the repo name
(QuantTrio/MiniMax-M2-AWQ + quant AWQ-4bit -> just "(4bit)").
- Conditional warning for safetensors models on non-GPU rigs only.
- Dependency Install / Installed / Installed▾ / N/A all 75.85px wide.
- Rebuild llama.cpp moved into the llama_cpp dep row, styled as a tag.
- Foldable Download admin-card (h2 chevron); line under h2 only when folded.
- HF token save gets a green ✓ + "Saved" flash.
- Cached scan no longer counts stalled rows as downloaded.
- Footer: "Request it →" link with GitHub mark to the public discussion
(#1962) for model-add requests.
Frontend — Running tab:
- Strict download-finish check (DOWNLOAD_OK or /snapshots/, not bare
"Download complete"). True overall % for multi-shard downloads:
((N-1)+frac)/total instead of hf_transfer's per-shard aggregate.
- ETA in the uptime ticker: "downloading: 12m 34s · ETA 1h 23m".
- Clear button kills the tmux session too; if the output still shows a
live shard line, the pill is hidden + relabels as "reconnect" + revives
on click.
- Self-heal: on cookbook open AND every bg-monitor cycle (10s, throttled
to 8s), scan persisted done/error/crashed downloads and probe their
tmux session — if alive, flip status back to running and reattach.
- Per-launch zombie probe: clicking Download on a model whose persisted
state is done but tmux is still alive revives the existing task and
refuses to start a duplicate.
- Pre-launch GPU probe: vllm / sglang / diffusers serve check
/api/cookbook/gpus first; warns + confirms if no GPU is visible.
- Server-side state guard: rejects "done" POSTs for downloads lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stale tabs can't poison persisted state any more.
- Running count includes tasks whose output looks active even if persisted
status got stuck. Dir text on the running row, font matched to uptime.
Serve panel:
- Ctx text input always resets to model max on open (default 20000 when
metadata is missing).
- Max Seqs default 8 -> 4. KV Cache dtype select 32px tall.
- Lightning icon on Launch (same as Action toggle).
- Diagnosis card simplified (no fold/copy/dismiss), suggestion font
matches body; action buttons get icons on the left (Retry/Copy/Edit/
Install/Kill/Switch/etc.).
- Incomplete-download serve warning when model status is
downloading / stalled / has_incomplete.
- MTP "?" tooltip ("supported on a few model families … up to ~3× faster").
2026-06-03 20:25:25 +09:00
} , BG _MONITOR _INTERVAL _MS ) ;
2026-07-01 10:09:25 +00:00
if ( _canBackgroundPoll ( ) ) {
_pollBackgroundStatus ( ) ;
_checkServeReachability ( ) ;
}
2026-05-31 23:58:26 +09:00
}
function _stopBackgroundMonitor ( ) {
if ( _bgMonitorInterval ) {
clearInterval ( _bgMonitorInterval ) ;
_bgMonitorInterval = null ;
}
const statusEl = document . getElementById ( 'cookbook-bg-status' ) ;
if ( statusEl ) statusEl . style . display = 'none' ;
}
// Retry-probe a freshly-added endpoint until its model server answers.
// A model that just reached "ready" in the cookbook often can't satisfy
// the 1s add-time probe (remote, weights still mmap-ing), so it's added
// offline. This polls the per-endpoint /probe (which uses a longer
// server-side timeout + persists cached_models) every few seconds until
// the endpoint reports models, then refreshes the picker. Bounded so a
// genuinely-dead server doesn't poll forever.
async function _probeEndpointUntilOnline ( epId , host , port ) {
2026-07-07 00:50:07 +00:00
if ( ! _isCookbookVisible ( ) || _foregroundChatBusy ( ) ) return ;
2026-05-31 23:58:26 +09:00
if ( ! epId ) return ;
// Big models (e.g. 70B+) can take several minutes to load weights before
// the server answers /v1/models. Probe for up to ~5 min, easing the
// interval out so we're not hammering during a long warmup.
const MAX _TRIES = 40 ;
for ( let i = 0 ; i < MAX _TRIES ; i ++ ) {
const interval = i < 12 ? 5000 : 10000 ; // 5s for the first minute, then 10s
await new Promise ( r => setTimeout ( r , interval ) ) ;
2026-07-07 00:50:07 +00:00
if ( ! _isCookbookVisible ( ) || _foregroundChatBusy ( ) ) return ;
2026-05-31 23:58:26 +09:00
try {
// Hit the probe endpoint — it re-probes server-side and updates
// cached_models. We consume (and discard) the SSE stream.
2026-06-22 01:49:15 +00:00
const probeRes = await fetch ( ` /api/model-endpoints/ ${ epId } /probe ` , { credentials : 'same-origin' } ) . catch ( ( ) => null ) ;
if ( probeRes && probeRes . status === 404 ) return ;
if ( probeRes ) await probeRes . text ( ) . catch ( ( ) => { } ) ;
2026-05-31 23:58:26 +09:00
const eps = await fetch ( '/api/model-endpoints' , { credentials : 'same-origin' } ) . then ( r => r . json ( ) ) . catch ( ( ) => [ ] ) ;
const ep = ( eps || [ ] ) . find ( e => e . id === epId ) ;
if ( ep && ( ep . models || [ ] ) . length ) {
2026-07-07 00:50:07 +00:00
if ( window . modelsModule ? . refreshModels ) await window . modelsModule . refreshModels ( false ) ;
2026-05-31 23:58:26 +09:00
if ( window . sessionModule ? . updateModelPicker ) window . sessionModule . updateModelPicker ( ) ;
window . dispatchEvent ( new CustomEvent ( 'ge:model-endpoints-updated' , {
detail : { baseUrl : ep . base _url || ` http:// ${ host } : ${ port } /v1 ` , host , port , model : ( ep . models || [ ] ) [ 0 ] || '' } ,
} ) ) ;
uiModule . showToast ( ` ${ host } : ${ port } is online ` ) ;
return ;
}
} catch ( _ ) { /* keep retrying */ }
}
}
async function _pollBackgroundStatus ( ) {
2026-07-01 10:09:25 +00:00
if ( ! _canBackgroundPoll ( ) || _bgPollInFlight ) return ;
_bgPollInFlight = true ;
2026-05-31 23:58:26 +09:00
try {
// Pull any tasks the server knows about that aren't in localStorage
// yet (e.g. agent-spawned downloads/serves). Without this merge,
// _syncToServer keeps clobbering server-added tasks on every poll.
try {
const stateRes = await fetch ( '/api/cookbook/state' , { credentials : 'same-origin' } ) ;
if ( stateRes . ok ) {
const serverState = await stateRes . json ( ) ;
const serverTasks = ( serverState && Array . isArray ( serverState . tasks ) ) ? serverState . tasks : [ ] ;
if ( serverTasks . length ) {
const localTasks = _loadTasks ( ) ;
const localIds = new Set ( localTasks . map ( t => t . sessionId ) ) ;
const merged = [ ... localTasks ] ;
let added = 0 ;
for ( const t of serverTasks ) {
if ( t && t . sessionId && ! localIds . has ( t . sessionId ) && ! _isTombstoned ( t . sessionId ) ) {
merged . push ( t ) ;
added ++ ;
}
}
if ( added > 0 ) {
2026-06-22 02:39:18 +00:00
localStorage . setItem ( TASKS _KEY , JSON . stringify ( merged . map ( _redactTaskForStorage ) ) ) ;
2026-05-31 23:58:26 +09:00
_renderRunningTab ( ) ;
}
}
}
} catch ( _ ) { /* non-fatal */ }
const res = await fetch ( '/api/cookbook/tasks/status' , { credentials : 'same-origin' } ) ;
if ( ! res . ok ) return ;
const data = await res . json ( ) ;
const tasks = data . tasks || [ ] ;
2026-06-01 22:59:29 -05:00
// Reconcile the authoritative tmux/process status back into the persisted
// client task list. The Running-tab reconnect loop also does this, but it
// only exists while cards are rendered; after a page refresh or closed modal
// dependency installs could finish server-side while localStorage stayed
// stuck at "running".
try {
const statusById = new Map ( tasks . map ( t => [ t . session _id , t ] ) ) ;
const localTasks = _loadTasks ( ) ;
let changed = false ;
const completedDeps = [ ] ;
2026-06-29 03:02:58 +00:00
const localIds = new Set ( localTasks . map ( t => t . sessionId ) . filter ( Boolean ) ) ;
for ( const live of tasks ) {
const sid = live ? . session _id ;
if ( ! sid || localIds . has ( sid ) || _isTombstoned ( sid ) ) continue ;
const liveType = live . type || 'download' ;
const liveStatus = live . status === 'completed' ? 'done' : ( live . status || 'running' ) ;
const name = live . model || sid ;
const remoteHost = live . remote && live . remote !== 'local' ? live . remote : '' ;
localTasks . push ( _redactTaskForStorage ( {
id : sid ,
sessionId : sid ,
name ,
type : liveType ,
status : liveStatus ,
progress : live . progress || '' ,
output : live . output _tail || '' ,
ts : Date . now ( ) ,
payload : {
repo _id : name ,
remote _host : remoteHost ,
_cmd : live . cmd || '(adopted from live tmux status)' ,
} ,
remoteHost ,
_adoptedExternally : true ,
} ) ) ;
localIds . add ( sid ) ;
changed = true ;
}
2026-06-01 22:59:29 -05:00
for ( const task of localTasks ) {
const live = statusById . get ( task . sessionId ) ;
if ( ! live ) continue ;
const updates = { } ;
2026-06-04 17:25:06 +05:30
// A finished dependency install whose tmux pane is gone is reported
// "stopped" by the backend (its pip package is never in the HF cache the
// dead-session check inspects). Recover "done" from the retained output's
// exit-0 sentinel so a clean install isn't downgraded to crashed.
2026-07-07 00:50:07 +00:00
const combinedOutput = ` ${ task . output || '' } \n ${ live . output _tail || '' } ` ;
const depDone = ! ! task . payload ? . _dep && _depInstallSucceeded ( combinedOutput ) ;
2026-06-15 14:36:39 +08:00
// A finished model download whose tmux pane is gone is also reported
// "stopped" (the dead-session check can miss the landed snapshot).
// Recover "done" from the terminal `DOWNLOAD_OK` sentinel — emitted
// only after the runner exits 0 — so a completed download isn't
// downgraded to crashed. This background poll runs blind (no live
// stream to debounce against), so unlike the reconnect loop it keys
// off the conclusive exit sentinel only, never the `/snapshots/` path,
// which can be printed mid-stream for multi-file downloads.
const downloadDone = task . type === 'download'
2026-07-07 00:50:07 +00:00
&& String ( combinedOutput || '' ) . includes ( 'DOWNLOAD_OK' ) ;
const serveReady = task . type === 'serve'
&& ( live . status === 'ready' || _serveOutputLooksReady ( { ... task , output : live . output _tail || task . output || '' } ) ) ;
const completedByOutput = depDone || downloadDone ;
const nextStatus = completedByOutput
? 'done'
: ( serveReady
? 'ready'
: ( live . status === 'completed'
2026-06-01 22:59:29 -05:00
? 'done'
2026-06-02 22:38:55 +09:00
: ( live . status === 'error'
? 'error'
2026-06-04 17:25:06 +05:30
: ( live . status === 'stopped'
2026-06-15 14:36:39 +08:00
? ( ( depDone || downloadDone ) ? 'done' : ( task . type === 'download' ? 'crashed' : 'stopped' ) )
2026-07-07 00:50:07 +00:00
: null ) ) ) ) ;
2026-06-01 22:59:29 -05:00
if ( nextStatus && task . status !== nextStatus ) {
updates . status = nextStatus ;
if ( nextStatus === 'done' && task . payload ? . _dep ) completedDeps . push ( task ) ;
}
2026-07-07 00:50:07 +00:00
if ( serveReady && ! task . _serveReady ) {
updates . _serveReady = true ;
}
if ( ( live . status === 'running' || live . status === 'ready' ) && task . status !== live . status && ! serveReady && ! completedByOutput ) {
2026-06-02 22:38:55 +09:00
updates . status = live . status === 'ready' ? 'ready' : 'running' ;
}
2026-06-01 22:59:29 -05:00
if ( live . progress && live . progress !== task . progress ) updates . progress = live . progress ;
2026-06-11 18:55:33 +02:00
if ( live . exit _code != null && live . exit _code !== task . exit _code ) updates . exit _code = live . exit _code ;
2026-06-02 22:38:55 +09:00
if ( live . output _tail ) {
const previous = String ( task . output || '' ) ;
const tail = String ( live . output _tail || '' ) ;
if ( tail && ! previous . endsWith ( tail ) ) {
2026-06-30 01:47:48 +00:00
updates . output = _isServeOutputPlaceholder ( previous )
? tail . slice ( - 5000 )
: ` ${ previous ? ` ${ previous } \n ` : '' } ${ tail } ` . slice ( - 5000 ) ;
2026-06-02 22:38:55 +09:00
}
}
2026-06-05 05:52:07 -03:00
if ( live . diagnosis && ! task . _diagnosisDismissed ) {
updates . _backendDiagnosis = live . diagnosis ;
}
if ( live . cmd && ! task . payload ? . _cmd ) {
updates . payload = { ... ( task . payload || { } ) , _cmd : live . cmd } ;
}
2026-06-01 22:59:29 -05:00
if ( Object . keys ( updates ) . length ) {
Object . assign ( task , updates ) ;
changed = true ;
}
}
if ( changed ) {
_saveTasks ( localTasks ) ;
_renderRunningTab ( ) ;
2026-06-05 05:52:07 -03:00
for ( const task of localTasks ) {
if ( ! task . _backendDiagnosis ) continue ;
const el = document . querySelector ( ` [data-session-id=" ${ CSS . escape ( task . sessionId ) } "] ` ) ;
if ( ! el || el . querySelector ( '.cookbook-diagnosis' ) ) continue ;
_showDiagnosis ( el , task . _backendDiagnosis , task . output || '' ) ;
}
2026-06-01 22:59:29 -05:00
completedDeps . forEach ( t => _refreshDepsAfterInstall ( t ) ) ;
}
} catch ( _ ) { /* non-fatal: background status should never break polling */ }
2026-05-31 23:58:26 +09:00
const statusEl = document . getElementById ( 'cookbook-bg-status' ) ;
const activeTasks = tasks . filter ( t => t . status === 'running' || t . status === 'ready' ) ;
const errorTasks = tasks . filter ( t => t . status === 'error' ) ;
const completedTasks = tasks . filter ( t => t . status === 'completed' ) ;
// Auto-add serve endpoints that became ready (works even when modal is closed)
const readyServes = tasks . filter ( t => t . type === 'serve' && t . status === 'ready' ) ;
for ( const t of readyServes ) {
const localTasks = _loadTasks ( ) ;
const localTask = localTasks . find ( lt => lt . sessionId === t . session _id ) ;
if ( localTask && localTask . _endpointAdded ) continue ;
2026-06-02 10:09:48 -05:00
let host = _connectHostFromRemote ( localTask ? . remoteHost || t . remote ) ;
Cookbook: scoring fixes, UI polish, false-finished + stale-state bug fixes
Backend (services/hwfit + routes):
- rank_models picks visible set by REQUESTED column, not always score —
sorting by Param now shows highest-param models PERIOD (incl. too_tight).
- New fit_only param. Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang
cannot serve them); default non-prequantized to BF16 on 2+ GPUs.
- AWQ / GPTQ-8bit get a -1.0 quality penalty (was 0.0, tied with FP8), so
FP8 wins when both fit.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above
M2.5 on equal composite score; >=100B integers not misread as versions.
- /api/cookbook/hf-latest no longer drops models without an "NB" pattern in
the repo id (MiniMax-M2.7, DeepSeek-V4-Pro etc. were silently filtered).
- Cached-model scan: atexit flushes models JSON even if the script is
killed mid-walk; each scan_dir wrapped in try/except; timeout 60s -> 180s.
- KB granularity for sub-MB sizes (was "0 MB" for 12 KB shells). New
"stalled" status for shells <1 MB with no .incomplete files.
- /api/cookbook/state POST guard: rejects "done" download tasks lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stops stale tabs from poisoning persisted state.
- hf_models.json: add zai-org/GLM-5.1; flip zai-org/GLM-5 quantization
Q4_K_M -> BF16 (it is the native base, not a quant).
Frontend (static/js):
- Scan/Download toolbar: quant defaults to All; ctx slider (8k/16k/32k/
50k/128k/Max) ported from origin/main with sort=fit on drag, sort=score
on Max. GPU toggle commits _activeCount to maxGpu on initial render. Fit
column header tagged with active budget (RAM / GPU / N GPU).
- Foldable Download admin-card: the Download h2 is the chevron trigger;
state persists in localStorage.
- Download card surfaces destination dir (Dir: <path>). Same dir on running
task row, font/color matched to uptime (9px Fira Code muted, opacity .4).
- Serve panel ctx text input always resets to model max on open. Sub-MB
cached models show with red "download stalled" badge.
- Bulk-select Cancel + Delete reset the Select button label on exit.
- Cookbook running: false-finished bug fixed — DOWNLOAD_OK or /snapshots/
required; bare "Download complete" no longer marks the task done after
the first config file. Clear button now sends tmux kill-session too.
True overall % for multi-shard downloads: ((N-1)+frac)/total instead of
hf_transfer per-shard aggregate.
- Diagnosis card simplified: removed fold toggle, copy button, dismiss X.
Suggestion font matches message body (12px).
- HF token field flashes green check + "Saved" on save.
- Cached scan no longer counts stalled rows as downloaded in Scan/Download.
CSS:
- dep Install button width pinned to 76px to match Installed split.
- task-sub row +1px; task-status badge gets margin-right 8px.
- Ctx slider styled like gallery editor sliders (thin pill rail, red thumb).
- Bulk-select cancel button top -3px -> -5px.
2026-06-03 16:32:20 +09:00
const portMatch = localTask ? . payload ? . _cmd ? . match ( /--port\s+(\d+)/ )
|| localTask ? . payload ? . _cmd ? . match ( /OLLAMA_HOST=[^\s:]+:(\d+)/ ) ;
let port = portMatch ? portMatch [ 1 ] : '8000' ;
let baseUrl = ` http:// ${ host } : ${ port } /v1 ` ;
const snapshot = t . output || localTask ? . output || '' ;
const ollamaUrlMatch = snapshot . match ( /Ollama API ready on port\s+\d+:\s*(http:\/\/[^\s]+)/i ) ;
if ( ollamaUrlMatch ) {
2026-06-02 10:09:48 -05:00
const endpoint = _endpointFromAdvertisedUrl ( ollamaUrlMatch [ 1 ] , host , '11434' ) ;
if ( endpoint ) ( { host , port , baseUrl } = endpoint ) ;
Cookbook: scoring fixes, UI polish, false-finished + stale-state bug fixes
Backend (services/hwfit + routes):
- rank_models picks visible set by REQUESTED column, not always score —
sorting by Param now shows highest-param models PERIOD (incl. too_tight).
- New fit_only param. Multi-GPU rigs filter GGUF Q*/IQ quants (vLLM/SGLang
cannot serve them); default non-prequantized to BF16 on 2+ GPUs.
- AWQ / GPTQ-8bit get a -1.0 quality penalty (was 0.0, tied with FP8), so
FP8 wins when both fit.
- Version-aware tiebreaker (parse Mn.n / Vn) — MiniMax-M2.7 ranks above
M2.5 on equal composite score; >=100B integers not misread as versions.
- /api/cookbook/hf-latest no longer drops models without an "NB" pattern in
the repo id (MiniMax-M2.7, DeepSeek-V4-Pro etc. were silently filtered).
- Cached-model scan: atexit flushes models JSON even if the script is
killed mid-walk; each scan_dir wrapped in try/except; timeout 60s -> 180s.
- KB granularity for sub-MB sizes (was "0 MB" for 12 KB shells). New
"stalled" status for shells <1 MB with no .incomplete files.
- /api/cookbook/state POST guard: rejects "done" download tasks lacking
DOWNLOAD_OK / DOWNLOAD_FAILED / /snapshots/ when the last-mentioned
shard is N<total — stops stale tabs from poisoning persisted state.
- hf_models.json: add zai-org/GLM-5.1; flip zai-org/GLM-5 quantization
Q4_K_M -> BF16 (it is the native base, not a quant).
Frontend (static/js):
- Scan/Download toolbar: quant defaults to All; ctx slider (8k/16k/32k/
50k/128k/Max) ported from origin/main with sort=fit on drag, sort=score
on Max. GPU toggle commits _activeCount to maxGpu on initial render. Fit
column header tagged with active budget (RAM / GPU / N GPU).
- Foldable Download admin-card: the Download h2 is the chevron trigger;
state persists in localStorage.
- Download card surfaces destination dir (Dir: <path>). Same dir on running
task row, font/color matched to uptime (9px Fira Code muted, opacity .4).
- Serve panel ctx text input always resets to model max on open. Sub-MB
cached models show with red "download stalled" badge.
- Bulk-select Cancel + Delete reset the Select button label on exit.
- Cookbook running: false-finished bug fixed — DOWNLOAD_OK or /snapshots/
required; bare "Download complete" no longer marks the task done after
the first config file. Clear button now sends tmux kill-session too.
True overall % for multi-shard downloads: ((N-1)+frac)/total instead of
hf_transfer per-shard aggregate.
- Diagnosis card simplified: removed fold toggle, copy button, dismiss X.
Suggestion font matches message body (12px).
- HF token field flashes green check + "Saved" on save.
- Cached scan no longer counts stalled rows as downloaded in Scan/Download.
CSS:
- dep Install button width pinned to 76px to match Installed split.
- task-sub row +1px; task-status badge gets margin-right 8px.
- Ctx slider styled like gallery editor sliders (thin pill rail, red thumb).
- Bulk-select cancel button top -3px -> -5px.
2026-06-03 16:32:20 +09:00
}
2026-05-31 23:58:26 +09:00
const _isDiffusion = localTask ? . payload ? . _cmd ? . includes ( 'diffusion_server' ) ;
_updateTask ( t . session _id , { _serveReady : true , _endpointAdded : true } ) ;
if ( localTask ) _autoSaveWorkingConfig ( localTask ) ; // remember working settings (modal may be closed)
// Auto-detect function-calling support from the serve cmd.
// vLLM emits OpenAI-style tool_calls only when launched with
// `--enable-auto-tool-choice`; local-only models otherwise
// hallucinate a fake [TOOL_CALL]...[/TOOL_CALL] text format
// the backend can't parse.
const _cmd = localTask ? . payload ? . _cmd || '' ;
const _supportsTools = _cmd . includes ( '--enable-auto-tool-choice' ) || _isDiffusion === false && /(?:^|\s)(?:deepseek|gpt-[45o]|claude|gemini|qwen3|qwen2\.5|mixtral|llama-[34]|minimax|kimi|hermes|glm-4)/i . test ( t . model ) ;
fetch ( '/api/model-endpoints' , { credentials : 'same-origin' } )
. then ( r => r . json ( ) )
. then ( eps => {
const hostPort = ` ${ host } : ${ port } ` ;
const existing = eps . find ( e => e . base _url === baseUrl || e . base _url . includes ( hostPort ) || e . name === t . model ) ;
if ( existing ) {
2026-07-07 00:50:07 +00:00
const taskForMatch = localTask || { sessionId : t . session _id , name : t . model , model : t . model , payload : { repo _id : t . model , _cmd } } ;
if ( ! _endpointMatchesServe ( existing , taskForMatch ) ) {
_markServeEndpointMismatch ( taskForMatch , existing , host , port ) ;
return null ;
}
2026-05-31 23:58:26 +09:00
// Already registered — but it may be showing offline because
// it was added while the server was still warming. Kick a
// re-probe so it flips online without manual toggle.
if ( ! ( existing . models || [ ] ) . length ) _probeEndpointUntilOnline ( existing . id , host , port ) ;
return null ;
}
const fd = new FormData ( ) ;
fd . append ( 'base_url' , baseUrl ) ;
fd . append ( 'name' , t . model ) ;
fd . append ( 'skip_probe' , 'true' ) ;
2026-06-02 10:09:48 -05:00
_appendCookbookEndpointScope ( fd , localTask ? . remoteHost || t . remote || '' ) ;
2026-07-07 00:50:07 +00:00
_appendPinnedServeModel ( fd , localTask || { name : t . model , model : t . model , payload : { repo _id : t . model , _cmd } } ) ;
2026-05-31 23:58:26 +09:00
if ( _isDiffusion ) fd . append ( 'model_type' , 'image' ) ;
if ( _supportsTools ) fd . append ( 'supports_tools' , 'true' ) ;
return fetch ( '/api/model-endpoints' , { method : 'POST' , credentials : 'same-origin' , body : fd } ) ;
} )
. then ( async ( res ) => {
if ( res && res . ok ) {
uiModule . showToast ( ` Model endpoint added: ${ host } : ${ port } ` ) ;
const data = await res . json ( ) . catch ( ( ) => ( { } ) ) ;
// A just-started server often can't answer the 1s add-time
// probe, so it lands "offline". Retry-probe in the background
// until /v1/models responds — no manual enable/disable needed.
if ( data && data . id ) _probeEndpointUntilOnline ( data . id , host , port ) ;
2026-07-07 00:50:07 +00:00
if ( window . modelsModule ? . refreshModels ) await window . modelsModule . refreshModels ( false ) ;
2026-05-31 23:58:26 +09:00
if ( window . sessionModule ? . updateModelPicker ) window . sessionModule . updateModelPicker ( ) ;
}
} )
. catch ( ( ) => { } ) ;
}
if ( errorTasks . length > 0 ) {
_showCookbookNotif ( true ) ;
} else if ( completedTasks . length > 0 ) {
_showCookbookNotif ( false ) ;
} else if ( activeTasks . length > 0 ) {
_showCookbookNotif ( false ) ;
} else {
_clearCookbookNotif ( ) ;
_stopBackgroundMonitor ( ) ;
}
if ( statusEl ) {
if ( activeTasks . length > 0 ) {
const t = activeTasks [ 0 ] ;
if ( t . type === 'serve' ) {
if ( t . progress ) {
// Show serve phase from backend (e.g. "loading 45%", "warming up", "idle", "12.5 tok/s")
statusEl . textContent = t . progress ;
} else if ( t . status === 'ready' ) {
statusEl . textContent = 'ready' ;
} else {
statusEl . textContent = 'cooking' ;
}
} else {
var _dlProgress = '' ;
if ( t . progress ) {
var _pctMatch = t . progress . match ( /(\d+)%/ ) ;
_dlProgress = _pctMatch ? ` ${ _pctMatch [ 0 ] } ` : '' ;
}
statusEl . textContent = ` downloading ${ _dlProgress } ` ;
}
statusEl . style . display = '' ;
} else if ( errorTasks . length > 0 ) {
statusEl . textContent = 'error' ;
statusEl . style . display = '' ;
statusEl . style . color = 'var(--color-error, #f44)' ;
} else if ( completedTasks . length > 0 ) {
statusEl . textContent = 'done' ;
statusEl . style . display = '' ;
statusEl . style . color = 'var(--color-success, #4caf50)' ;
} else {
statusEl . style . display = 'none' ;
statusEl . style . color = '' ;
}
}
// Also clear the sidebar/rail icon highlight when no tasks are alive.
// Without this, the cookbook icon stays at full opacity ("highlighted")
// indefinitely once any task fires the notif, because the modal-open
// clear only runs when the user actually reopens Cookbook.
if ( ! activeTasks . length && ! errorTasks . length ) {
_clearCookbookNotif ( ) ;
}
} catch ( e ) {
// Silent fail
2026-07-01 10:09:25 +00:00
} finally {
_bgPollInFlight = false ;
2026-05-31 23:58:26 +09:00
}
}
// ── Init: receive shared state/functions ──
export function initRunning ( shared ) {
_envState = shared . _envState ;
_sshCmd = shared . _sshCmd ;
_getPort = shared . _getPort ;
_sshPrefix = shared . _sshPrefix ;
_getPlatform = shared . _getPlatform ;
_isWindows = shared . _isWindows ;
_buildEnvPrefix = shared . _buildEnvPrefix ;
_loadPresets = shared . _loadPresets ;
_savePresets = shared . _savePresets ;
_copyText = shared . _copyText ;
_persistEnvState = shared . _persistEnvState ;
2026-06-01 18:58:06 +05:30
_refreshDependencies = shared . _refreshDependencies ;
2026-06-08 18:36:10 -04:00
_serverByVal = shared . _serverByVal ;
2026-06-21 11:02:35 +00:00
_serverKey = shared . _serverKey ;
2026-06-08 18:36:10 -04:00
_selectedServer = shared . _selectedServer ;
2026-05-31 23:58:26 +09:00
modelLogo = shared . modelLogo ;
esc = shared . esc ;
_detectBackend = shared . _detectBackend ;
_detectToolParser = shared . _detectToolParser ;
_detectModelOptimizations = shared . _detectModelOptimizations ;
_buildServeCmd = shared . _buildServeCmd ;
2026-06-27 13:50:21 +00:00
// App boot: pull authoritative state from server, but don't start the
// running-task monitor unless there is real work to watch. Starting it
// unconditionally made a plain Cookbook open keep probing stale tmux/SSH
// sessions, which is expensive when a saved remote host is unreachable.
2026-05-31 23:58:26 +09:00
( async ( ) => {
try {
await _syncFromServer ( ) ;
} catch { }
2026-06-27 13:50:21 +00:00
if ( _hasLiveTasks ( ) ) _startBackgroundMonitor ( ) ;
2026-05-31 23:58:26 +09:00
} ) ( ) ;
}
// Also export _retryDownload and _nextAvailablePort for use by other modules
fix(cookbook): only block model launch on real port collisions (#4760)
* Fix #4507: only block model launch on real port collisions
Quick-run hardcoded port 8000 and never called _nextAvailablePort(), so
every launch collided. Both pre-launch guards (serve panel + quick-run)
were count-based and fired regardless of port.
- quick-run now auto-assigns a free port (8080 for llama.cpp)
- both guards parse the new port and only prompt on a real overlap,
stopping only the colliding serve
- dialog reports the actual port instead of a hardcoded 8000
* refactor(cookbook): share _taskPort for port parsing; auto-assign llama.cpp port
Addresses review on #4760:
- _taskPort regex now matches --port= as well as --port (space)
- _nextAvailablePort and both launch guards reuse _taskPort instead of inline regex
- quick-run llama.cpp no longer pins 8080, so two can run concurrently
* fix(cookbook): _taskPort also parses -p; add port-parsing tests
Addresses review on #4760:
- _taskPort now matches -p <n> too, so it's the complete single reader
(was missing the short flag that other readers already handle)
- add tests/test_cookbook_port_parsing_js.py covering the port forms,
shared-reader reuse, and llama.cpp auto-assign
* test(cookbook): extract pure port helpers and test behavior
Addresses review on #4760: the prior tests only asserted source strings.
- extract portOf() and nextFreePort() into static/js/cookbookPorts.js
- cookbookRunning.js imports them; _taskPort and _nextAvailablePort delegate
- tests run the helpers via node and assert real behavior: all port forms
(--port, --port=, -p, -p=), next-free-port skipping taken ports, and the
same-port-clash / different-port-coexist outcome
---------
Co-authored-by: samy <samy@odysseus.boukouro.com>
2026-06-24 13:44:09 -04:00
export { _retryDownload , _nextAvailablePort , _processQueue , _taskPort } ;