src / services / LlamaServerRouterManager.ts
src / services / LlamaServerRouterManager.ts
/**
* Lifecycle manager for a standalone llama.cpp router-mode `llama-server` process, used by the
* VISION_API="llama-server" strand (see VisionAnalyzer.ts). Router mode auto-discovers GGUF
* models from the local HF cache and loads/unloads them on demand per request, entirely isolated
* from any agent-model process (LM Studio Bionic, Unsloth Studio) -- live-verified 2026-10-01 that
* loading/using a vision model through a separate router-mode process never evicts an agent model
* loaded elsewhere, which is the whole reason this strand exists. See
* made-for-bionic/planning/made-for-preset-rollout-plan.md for the root-cause writeup of the
* eviction behavior this works around.
*/
import fs from "fs";
import os from "os";
import path from "path";
import { spawn } from "child_process";
import { getLogsDir } from "../core-bundle.mjs";
const HEALTH_POLL_INTERVAL_MS = 1_000;
const HEALTH_POLL_TIMEOUT_MS = 30_000;
const HEALTH_FETCH_TIMEOUT_MS = 2_000;
let activePort: number | null = null;
let idleStopTimer: NodeJS.Timeout | null = null;
function cancelIdleStop(): void {
if (idleStopTimer !== null) {
clearTimeout(idleStopTimer);
idleStopTimer = null;
}
}
function normalizePath(value: string): string {
const trimmed = value.trim();
if (!trimmed) return "";
return trimmed.startsWith("~") ? path.join(os.homedir(), trimmed.slice(1)) : trimmed;
}
function pidFilePath(): string {
return path.join(getLogsDir(), "llama-server-router.pid");
}
function logFilePath(): string {
return path.join(getLogsDir(), "llama-server-router.log");
}
function presetFilePath(): string {
return path.join(getLogsDir(), "llama-server-router-preset.ini");
}
function isProcessAlive(pid: number): boolean {
try {
process.kill(pid, 0);
return true;
} catch {
return false;
}
}
function readPid(): number | null {
try {
const raw = fs.readFileSync(pidFilePath(), "utf-8").trim();
const pid = parseInt(raw, 10);
return Number.isFinite(pid) && pid > 0 ? pid : null;
} catch {
return null;
}
}
function fetchWithTimeout(url: string, timeoutMs: number): Promise<Response> {
const controller = new AbortController();
const timer = setTimeout(() => controller.abort(), timeoutMs);
return fetch(url, { signal: controller.signal }).finally(() => clearTimeout(timer));
}
async function isHealthy(host: string, port: number): Promise<boolean> {
try {
const res = await fetchWithTimeout(`http://${host}:${port}/v1/models`, HEALTH_FETCH_TIMEOUT_MS);
return res.ok;
} catch {
return false;
}
}
async function pollHealth(host: string, port: number): Promise<boolean> {
const deadline = Date.now() + HEALTH_POLL_TIMEOUT_MS;
while (Date.now() < deadline) {
if (await isHealthy(host, port)) return true;
await new Promise<void>((resolve) => setTimeout(resolve, HEALTH_POLL_INTERVAL_MS));
}
return false;
}
export interface LlamaServerRouterConfig {
host: string;
port: number;
binaryPath: string;
/**
* --ctx-size applied to every router-spawned child model via a `[*]` preset section. Without
* this, llama.cpp router mode defaults each auto-loaded model to its own native context length
* -- live-verified 2026-10-01: Qwen3-VL-8B-Instruct's native 262144 context blew up to ~45GB
* VRAM with no cap; capped at 8192 it used ~7.3GB for the same model. Configured via
* src/config.ts's llamaServerCtxSize / mcp/config.ts's LLAMA_SERVER_CTX_SIZE, never hardcoded here.
*/
ctxSize: number;
/** --models-max passed to the router server. Configured via src/config.ts's llamaServerModelsMax / mcp/config.ts's LLAMA_SERVER_MODELS_MAX. */
modelsMax: number;
/** Minutes of inactivity before an owned router process is stopped again, 0 disables idle shutdown. Configured via src/config.ts's llamaServerIdleTtlMinutes / mcp/config.ts's LLAMA_SERVER_IDLE_TTL_MINUTES. */
idleTtlMinutes: number;
/**
* Registers an absolute-path GGUF model (not an hf-repo id) as a named custom preset section,
* so the router can serve it without relying on HF-cache auto-discovery. Only set when
* VISION_API=llama-server's modelKey is an absolute .gguf file path (see ensureLlamaServerVisionModelReady).
*/
localModel?: {
name: string;
modelPath: string;
mmprojPath: string;
};
}
function buildPresetIni(ctxSize: number, localModel: LlamaServerRouterConfig["localModel"]): string {
let ini = `[*]\nctx-size = ${ctxSize}\n`;
if (localModel) {
ini += `\n[${localModel.name}]\nmodel = ${localModel.modelPath}\nmmproj = ${localModel.mmprojPath}\n`;
}
return ini;
}
async function listRouterModelIds(host: string, port: number): Promise<string[] | null> {
try {
const res = await fetchWithTimeout(`http://${host}:${port}/v1/models`, HEALTH_FETCH_TIMEOUT_MS);
if (!res.ok) return null;
const data = (await res.json()) as any;
const models: any[] = Array.isArray(data?.data) ? data.data : [];
return models.map((m) => String(m?.id ?? "").trim()).filter(Boolean);
} catch {
return null;
}
}
/**
* Ensures a router-mode llama-server is listening at host:port, spawning it if nothing answers
* there yet. Idempotent and safe to call before every request -- a already-healthy endpoint
* (ours or adopted) is trusted as-is, since router mode's own model management is independent of
* which process started the server. Reschedules this process's own idle-shutdown timer on every
* call (see releaseIdleLlamaServerRouter()).
*/
export async function ensureLlamaServerRouterRunning(config: LlamaServerRouterConfig): Promise<void> {
const { host, port, ctxSize, modelsMax, idleTtlMinutes, localModel } = config;
cancelIdleStop();
activePort = port;
const binaryPath = normalizePath(config.binaryPath);
if (!binaryPath) {
throw new Error("LLAMA_SERVER_BINARY is not configured.");
}
if (!fs.existsSync(binaryPath)) {
throw new Error(`llama-server binary does not exist: ${binaryPath}`);
}
if (await isHealthy(host, port)) {
if (localModel) {
const registeredIds = await listRouterModelIds(host, port);
const alreadyRegistered = registeredIds !== null && registeredIds.includes(localModel.name);
if (!alreadyRegistered) {
// The running router's preset doesn't know this local model yet -- router mode has no
// hot-reload endpoint for --models-preset, so the only way to register a new local model
// is to rewrite the ini and restart the process we ourselves own.
await stopLlamaServerRouter(port);
}
}
if (await isHealthy(host, port)) {
scheduleIdleStop(port, idleTtlMinutes);
return;
}
}
const stalePid = readPid();
if (stalePid !== null) {
if (isProcessAlive(stalePid)) {
throw new Error(
`Port ${port} is not responding, but a previously spawned llama-server router (PID ${stalePid}) is still alive. Check logs: ${logFilePath()}`
);
}
try {
fs.unlinkSync(pidFilePath());
} catch {}
}
fs.mkdirSync(getLogsDir(), { recursive: true });
fs.writeFileSync(presetFilePath(), buildPresetIni(ctxSize, localModel));
const args = [
"--host", host,
"--port", String(port),
"--models-max", String(modelsMax),
"--models-preset", presetFilePath(),
];
const logStream = fs.createWriteStream(logFilePath(), { flags: "a" });
const child = spawn(binaryPath, args, {
detached: true,
stdio: ["ignore", "pipe", "pipe"],
env: { ...process.env, HOME: os.homedir() },
});
child.stdout?.pipe(logStream);
child.stderr?.pipe(logStream);
child.once("error", (error) => {
try {
fs.appendFileSync(logFilePath(), `[llama-server-router] process error: ${error.message}\n`);
} catch {}
});
child.unref();
if (!child.pid) {
throw new Error(`Failed to spawn llama-server router process. Check logs: ${logFilePath()}`);
}
fs.writeFileSync(pidFilePath(), String(child.pid));
const ready = await pollHealth(host, port);
if (!ready) {
throw new Error(
`llama-server router did not become healthy within ${HEALTH_POLL_TIMEOUT_MS / 1000}s on ${host}:${port}. Check logs: ${logFilePath()}`
);
}
scheduleIdleStop(port, idleTtlMinutes);
}
/** 0 disables idle shutdown entirely (process stays until the host process itself exits). */
function scheduleIdleStop(port: number, idleTtlMinutes: number): void {
cancelIdleStop();
if (idleTtlMinutes <= 0) return;
idleStopTimer = setTimeout(() => {
idleStopTimer = null;
if (activePort !== port) return;
void stopLlamaServerRouter(port).catch((error) => {
try {
fs.appendFileSync(logFilePath(), `[llama-server-router] failed to stop idle router: ${error instanceof Error ? error.message : String(error)}\n`);
} catch {}
});
}, idleTtlMinutes * 60_000);
idleStopTimer.unref();
}
/** Stops the router process this plugin itself spawned (tracked via the PID file); a no-op if nothing is running. */
export async function stopLlamaServerRouter(port: number): Promise<void> {
cancelIdleStop();
const pid = readPid();
if (pid === null) return;
if (isProcessAlive(pid)) {
try {
process.kill(pid, "SIGTERM");
} catch {}
}
try {
fs.unlinkSync(pidFilePath());
} catch {}
if (activePort === port) activePort = null;
}
/**
* Lifecycle manager for a standalone llama.cpp router-mode `llama-server` process, used by the
* VISION_API="llama-server" strand (see VisionAnalyzer.ts). Router mode auto-discovers GGUF
* models from the local HF cache and loads/unloads them on demand per request, entirely isolated
* from any agent-model process (LM Studio Bionic, Unsloth Studio) -- live-verified 2026-10-01 that
* loading/using a vision model through a separate router-mode process never evicts an agent model
* loaded elsewhere, which is the whole reason this strand exists. See
* made-for-bionic/planning/made-for-preset-rollout-plan.md for the root-cause writeup of the
* eviction behavior this works around.
*/
import fs from "fs";
import os from "os";
import path from "path";
import { spawn } from "child_process";
import { getLogsDir } from "../core-bundle.mjs";
const HEALTH_POLL_INTERVAL_MS = 1_000;
const HEALTH_POLL_TIMEOUT_MS = 30_000;
const HEALTH_FETCH_TIMEOUT_MS = 2_000;
let activePort: number | null = null;
let idleStopTimer: NodeJS.Timeout | null = null;
function cancelIdleStop(): void {
if (idleStopTimer !== null) {
clearTimeout(idleStopTimer);
idleStopTimer = null;
}
}
function normalizePath(value: string): string {
const trimmed = value.trim();
if (!trimmed) return "";
return trimmed.startsWith("~") ? path.join(os.homedir(), trimmed.slice(1)) : trimmed;
}
function pidFilePath(): string {
return path.join(getLogsDir(), "llama-server-router.pid");
}
function logFilePath(): string {
return path.join(getLogsDir(), "llama-server-router.log");
}
function presetFilePath(): string {
return path.join(getLogsDir(), "llama-server-router-preset.ini");
}
function isProcessAlive(pid: number): boolean {
try {
process.kill(pid, 0);
return true;
} catch {
return false;
}
}
function readPid(): number | null {
try {
const raw = fs.readFileSync(pidFilePath(), "utf-8").trim();
const pid = parseInt(raw, 10);
return Number.isFinite(pid) && pid > 0 ? pid : null;
} catch {
return null;
}
}
function fetchWithTimeout(url: string, timeoutMs: number): Promise<Response> {
const controller = new AbortController();
const timer = setTimeout(() => controller.abort(), timeoutMs);
return fetch(url, { signal: controller.signal }).finally(() => clearTimeout(timer));
}
async function isHealthy(host: string, port: number): Promise<boolean> {
try {
const res = await fetchWithTimeout(`http://${host}:${port}/v1/models`, HEALTH_FETCH_TIMEOUT_MS);
return res.ok;
} catch {
return false;
}
}
async function pollHealth(host: string, port: number): Promise<boolean> {
const deadline = Date.now() + HEALTH_POLL_TIMEOUT_MS;
while (Date.now() < deadline) {
if (await isHealthy(host, port)) return true;
await new Promise<void>((resolve) => setTimeout(resolve, HEALTH_POLL_INTERVAL_MS));
}
return false;
}
export interface LlamaServerRouterConfig {
host: string;
port: number;
binaryPath: string;
/**
* --ctx-size applied to every router-spawned child model via a `[*]` preset section. Without
* this, llama.cpp router mode defaults each auto-loaded model to its own native context length
* -- live-verified 2026-10-01: Qwen3-VL-8B-Instruct's native 262144 context blew up to ~45GB
* VRAM with no cap; capped at 8192 it used ~7.3GB for the same model. Configured via
* src/config.ts's llamaServerCtxSize / mcp/config.ts's LLAMA_SERVER_CTX_SIZE, never hardcoded here.
*/
ctxSize: number;
/** --models-max passed to the router server. Configured via src/config.ts's llamaServerModelsMax / mcp/config.ts's LLAMA_SERVER_MODELS_MAX. */
modelsMax: number;
/** Minutes of inactivity before an owned router process is stopped again, 0 disables idle shutdown. Configured via src/config.ts's llamaServerIdleTtlMinutes / mcp/config.ts's LLAMA_SERVER_IDLE_TTL_MINUTES. */
idleTtlMinutes: number;
/**
* Registers an absolute-path GGUF model (not an hf-repo id) as a named custom preset section,
* so the router can serve it without relying on HF-cache auto-discovery. Only set when
* VISION_API=llama-server's modelKey is an absolute .gguf file path (see ensureLlamaServerVisionModelReady).
*/
localModel?: {
name: string;
modelPath: string;
mmprojPath: string;
};
}
function buildPresetIni(ctxSize: number, localModel: LlamaServerRouterConfig["localModel"]): string {
let ini = `[*]\nctx-size = ${ctxSize}\n`;
if (localModel) {
ini += `\n[${localModel.name}]\nmodel = ${localModel.modelPath}\nmmproj = ${localModel.mmprojPath}\n`;
}
return ini;
}
async function listRouterModelIds(host: string, port: number): Promise<string[] | null> {
try {
const res = await fetchWithTimeout(`http://${host}:${port}/v1/models`, HEALTH_FETCH_TIMEOUT_MS);
if (!res.ok) return null;
const data = (await res.json()) as any;
const models: any[] = Array.isArray(data?.data) ? data.data : [];
return models.map((m) => String(m?.id ?? "").trim()).filter(Boolean);
} catch {
return null;
}
}
/**
* Ensures a router-mode llama-server is listening at host:port, spawning it if nothing answers
* there yet. Idempotent and safe to call before every request -- a already-healthy endpoint
* (ours or adopted) is trusted as-is, since router mode's own model management is independent of
* which process started the server. Reschedules this process's own idle-shutdown timer on every
* call (see releaseIdleLlamaServerRouter()).
*/
export async function ensureLlamaServerRouterRunning(config: LlamaServerRouterConfig): Promise<void> {
const { host, port, ctxSize, modelsMax, idleTtlMinutes, localModel } = config;
cancelIdleStop();
activePort = port;
const binaryPath = normalizePath(config.binaryPath);
if (!binaryPath) {
throw new Error("LLAMA_SERVER_BINARY is not configured.");
}
if (!fs.existsSync(binaryPath)) {
throw new Error(`llama-server binary does not exist: ${binaryPath}`);
}
if (await isHealthy(host, port)) {
if (localModel) {
const registeredIds = await listRouterModelIds(host, port);
const alreadyRegistered = registeredIds !== null && registeredIds.includes(localModel.name);
if (!alreadyRegistered) {
// The running router's preset doesn't know this local model yet -- router mode has no
// hot-reload endpoint for --models-preset, so the only way to register a new local model
// is to rewrite the ini and restart the process we ourselves own.
await stopLlamaServerRouter(port);
}
}
if (await isHealthy(host, port)) {
scheduleIdleStop(port, idleTtlMinutes);
return;
}
}
const stalePid = readPid();
if (stalePid !== null) {
if (isProcessAlive(stalePid)) {
throw new Error(
`Port ${port} is not responding, but a previously spawned llama-server router (PID ${stalePid}) is still alive. Check logs: ${logFilePath()}`
);
}
try {
fs.unlinkSync(pidFilePath());
} catch {}
}
fs.mkdirSync(getLogsDir(), { recursive: true });
fs.writeFileSync(presetFilePath(), buildPresetIni(ctxSize, localModel));
const args = [
"--host", host,
"--port", String(port),
"--models-max", String(modelsMax),
"--models-preset", presetFilePath(),
];
const logStream = fs.createWriteStream(logFilePath(), { flags: "a" });
const child = spawn(binaryPath, args, {
detached: true,
stdio: ["ignore", "pipe", "pipe"],
env: { ...process.env, HOME: os.homedir() },
});
child.stdout?.pipe(logStream);
child.stderr?.pipe(logStream);
child.once("error", (error) => {
try {
fs.appendFileSync(logFilePath(), `[llama-server-router] process error: ${error.message}\n`);
} catch {}
});
child.unref();
if (!child.pid) {
throw new Error(`Failed to spawn llama-server router process. Check logs: ${logFilePath()}`);
}
fs.writeFileSync(pidFilePath(), String(child.pid));
const ready = await pollHealth(host, port);
if (!ready) {
throw new Error(
`llama-server router did not become healthy within ${HEALTH_POLL_TIMEOUT_MS / 1000}s on ${host}:${port}. Check logs: ${logFilePath()}`
);
}
scheduleIdleStop(port, idleTtlMinutes);
}
/** 0 disables idle shutdown entirely (process stays until the host process itself exits). */
function scheduleIdleStop(port: number, idleTtlMinutes: number): void {
cancelIdleStop();
if (idleTtlMinutes <= 0) return;
idleStopTimer = setTimeout(() => {
idleStopTimer = null;
if (activePort !== port) return;
void stopLlamaServerRouter(port).catch((error) => {
try {
fs.appendFileSync(logFilePath(), `[llama-server-router] failed to stop idle router: ${error instanceof Error ? error.message : String(error)}\n`);
} catch {}
});
}, idleTtlMinutes * 60_000);
idleStopTimer.unref();
}
/** Stops the router process this plugin itself spawned (tracked via the PID file); a no-op if nothing is running. */
export async function stopLlamaServerRouter(port: number): Promise<void> {
cancelIdleStop();
const pid = readPid();
if (pid === null) return;
if (isProcessAlive(pid)) {
try {
process.kill(pid, "SIGTERM");
} catch {}
}
try {
fs.unlinkSync(pidFilePath());
} catch {}
if (activePort === port) activePort = null;
}