dist-mcp / services / LlamaServerRouterManager.js
dist-mcp / services / LlamaServerRouterManager.js
"use strict";
var __importDefault = (this && this.__importDefault) || function (mod) {
return (mod && mod.__esModule) ? mod : { "default": mod };
};
Object.defineProperty(exports, "__esModule", { value: true });
exports.ensureLlamaServerRouterRunning = ensureLlamaServerRouterRunning;
exports.stopLlamaServerRouter = stopLlamaServerRouter;
/**
* Lifecycle manager for a standalone llama.cpp router-mode `llama-server` process, used by the
* VISION_API="llama-server" strand (see VisionAnalyzer.ts). Router mode auto-discovers GGUF
* models from the local HF cache and loads/unloads them on demand per request, entirely isolated
* from any agent-model process (LM Studio Bionic, Unsloth Studio) -- live-verified 2026-10-01 that
* loading/using a vision model through a separate router-mode process never evicts an agent model
* loaded elsewhere, which is the whole reason this strand exists. See
* made-for-bionic/planning/made-for-preset-rollout-plan.md for the root-cause writeup of the
* eviction behavior this works around.
*/
const fs_1 = __importDefault(require("fs"));
const os_1 = __importDefault(require("os"));
const path_1 = __importDefault(require("path"));
const child_process_1 = require("child_process");
const core_bundle_mjs_1 = require("../core-bundle.mjs");
const HEALTH_POLL_INTERVAL_MS = 1_000;
const HEALTH_POLL_TIMEOUT_MS = 30_000;
const HEALTH_FETCH_TIMEOUT_MS = 2_000;
let activePort = null;
let idleStopTimer = null;
function cancelIdleStop() {
if (idleStopTimer !== null) {
clearTimeout(idleStopTimer);
idleStopTimer = null;
}
}
function normalizePath(value) {
const trimmed = value.trim();
if (!trimmed)
return "";
return trimmed.startsWith("~") ? path_1.default.join(os_1.default.homedir(), trimmed.slice(1)) : trimmed;
}
function pidFilePath() {
return path_1.default.join((0, core_bundle_mjs_1.getLogsDir)(), "llama-server-router.pid");
}
function logFilePath() {
return path_1.default.join((0, core_bundle_mjs_1.getLogsDir)(), "llama-server-router.log");
}
function presetFilePath() {
return path_1.default.join((0, core_bundle_mjs_1.getLogsDir)(), "llama-server-router-preset.ini");
}
function isProcessAlive(pid) {
try {
process.kill(pid, 0);
return true;
}
catch {
return false;
}
}
function readPid() {
try {
const raw = fs_1.default.readFileSync(pidFilePath(), "utf-8").trim();
const pid = parseInt(raw, 10);
return Number.isFinite(pid) && pid > 0 ? pid : null;
}
catch {
return null;
}
}
function fetchWithTimeout(url, timeoutMs) {
const controller = new AbortController();
const timer = setTimeout(() => controller.abort(), timeoutMs);
return fetch(url, { signal: controller.signal }).finally(() => clearTimeout(timer));
}
async function isHealthy(host, port) {
try {
const res = await fetchWithTimeout(`http://${host}:${port}/v1/models`, HEALTH_FETCH_TIMEOUT_MS);
return res.ok;
}
catch {
return false;
}
}
async function pollHealth(host, port) {
const deadline = Date.now() + HEALTH_POLL_TIMEOUT_MS;
while (Date.now() < deadline) {
if (await isHealthy(host, port))
return true;
await new Promise((resolve) => setTimeout(resolve, HEALTH_POLL_INTERVAL_MS));
}
return false;
}
function buildPresetIni(ctxSize, localModel) {
let ini = `[*]\nctx-size = ${ctxSize}\n`;
if (localModel) {
ini += `\n[${localModel.name}]\nmodel = ${localModel.modelPath}\nmmproj = ${localModel.mmprojPath}\n`;
}
return ini;
}
async function listRouterModelIds(host, port) {
try {
const res = await fetchWithTimeout(`http://${host}:${port}/v1/models`, HEALTH_FETCH_TIMEOUT_MS);
if (!res.ok)
return null;
const data = (await res.json());
const models = Array.isArray(data?.data) ? data.data : [];
return models.map((m) => String(m?.id ?? "").trim()).filter(Boolean);
}
catch {
return null;
}
}
/**
* Ensures a router-mode llama-server is listening at host:port, spawning it if nothing answers
* there yet. Idempotent and safe to call before every request -- a already-healthy endpoint
* (ours or adopted) is trusted as-is, since router mode's own model management is independent of
* which process started the server. Reschedules this process's own idle-shutdown timer on every
* call (see releaseIdleLlamaServerRouter()).
*/
async function ensureLlamaServerRouterRunning(config) {
const { host, port, ctxSize, modelsMax, idleTtlMinutes, localModel } = config;
cancelIdleStop();
activePort = port;
const binaryPath = normalizePath(config.binaryPath);
if (!binaryPath) {
throw new Error("LLAMA_SERVER_BINARY is not configured.");
}
if (!fs_1.default.existsSync(binaryPath)) {
throw new Error(`llama-server binary does not exist: ${binaryPath}`);
}
if (await isHealthy(host, port)) {
if (localModel) {
const registeredIds = await listRouterModelIds(host, port);
const alreadyRegistered = registeredIds !== null && registeredIds.includes(localModel.name);
if (!alreadyRegistered) {
// The running router's preset doesn't know this local model yet -- router mode has no
// hot-reload endpoint for --models-preset, so the only way to register a new local model
// is to rewrite the ini and restart the process we ourselves own.
await stopLlamaServerRouter(port);
}
}
if (await isHealthy(host, port)) {
scheduleIdleStop(port, idleTtlMinutes);
return;
}
}
const stalePid = readPid();
if (stalePid !== null) {
if (isProcessAlive(stalePid)) {
throw new Error(`Port ${port} is not responding, but a previously spawned llama-server router (PID ${stalePid}) is still alive. Check logs: ${logFilePath()}`);
}
try {
fs_1.default.unlinkSync(pidFilePath());
}
catch { }
}
fs_1.default.mkdirSync((0, core_bundle_mjs_1.getLogsDir)(), { recursive: true });
fs_1.default.writeFileSync(presetFilePath(), buildPresetIni(ctxSize, localModel));
const args = [
"--host", host,
"--port", String(port),
"--models-max", String(modelsMax),
"--models-preset", presetFilePath(),
];
const logStream = fs_1.default.createWriteStream(logFilePath(), { flags: "a" });
const child = (0, child_process_1.spawn)(binaryPath, args, {
detached: true,
stdio: ["ignore", "pipe", "pipe"],
env: { ...process.env, HOME: os_1.default.homedir() },
});
child.stdout?.pipe(logStream);
child.stderr?.pipe(logStream);
child.once("error", (error) => {
try {
fs_1.default.appendFileSync(logFilePath(), `[llama-server-router] process error: ${error.message}\n`);
}
catch { }
});
child.unref();
if (!child.pid) {
throw new Error(`Failed to spawn llama-server router process. Check logs: ${logFilePath()}`);
}
fs_1.default.writeFileSync(pidFilePath(), String(child.pid));
const ready = await pollHealth(host, port);
if (!ready) {
throw new Error(`llama-server router did not become healthy within ${HEALTH_POLL_TIMEOUT_MS / 1000}s on ${host}:${port}. Check logs: ${logFilePath()}`);
}
scheduleIdleStop(port, idleTtlMinutes);
}
/** 0 disables idle shutdown entirely (process stays until the host process itself exits). */
function scheduleIdleStop(port, idleTtlMinutes) {
cancelIdleStop();
if (idleTtlMinutes <= 0)
return;
idleStopTimer = setTimeout(() => {
idleStopTimer = null;
if (activePort !== port)
return;
void stopLlamaServerRouter(port).catch((error) => {
try {
fs_1.default.appendFileSync(logFilePath(), `[llama-server-router] failed to stop idle router: ${error instanceof Error ? error.message : String(error)}\n`);
}
catch { }
});
}, idleTtlMinutes * 60_000);
idleStopTimer.unref();
}
/** Stops the router process this plugin itself spawned (tracked via the PID file); a no-op if nothing is running. */
async function stopLlamaServerRouter(port) {
cancelIdleStop();
const pid = readPid();
if (pid === null)
return;
if (isProcessAlive(pid)) {
try {
process.kill(pid, "SIGTERM");
}
catch { }
}
try {
fs_1.default.unlinkSync(pidFilePath());
}
catch { }
if (activePort === port)
activePort = null;
}
"use strict";
var __importDefault = (this && this.__importDefault) || function (mod) {
return (mod && mod.__esModule) ? mod : { "default": mod };
};
Object.defineProperty(exports, "__esModule", { value: true });
exports.ensureLlamaServerRouterRunning = ensureLlamaServerRouterRunning;
exports.stopLlamaServerRouter = stopLlamaServerRouter;
/**
* Lifecycle manager for a standalone llama.cpp router-mode `llama-server` process, used by the
* VISION_API="llama-server" strand (see VisionAnalyzer.ts). Router mode auto-discovers GGUF
* models from the local HF cache and loads/unloads them on demand per request, entirely isolated
* from any agent-model process (LM Studio Bionic, Unsloth Studio) -- live-verified 2026-10-01 that
* loading/using a vision model through a separate router-mode process never evicts an agent model
* loaded elsewhere, which is the whole reason this strand exists. See
* made-for-bionic/planning/made-for-preset-rollout-plan.md for the root-cause writeup of the
* eviction behavior this works around.
*/
const fs_1 = __importDefault(require("fs"));
const os_1 = __importDefault(require("os"));
const path_1 = __importDefault(require("path"));
const child_process_1 = require("child_process");
const core_bundle_mjs_1 = require("../core-bundle.mjs");
const HEALTH_POLL_INTERVAL_MS = 1_000;
const HEALTH_POLL_TIMEOUT_MS = 30_000;
const HEALTH_FETCH_TIMEOUT_MS = 2_000;
let activePort = null;
let idleStopTimer = null;
function cancelIdleStop() {
if (idleStopTimer !== null) {
clearTimeout(idleStopTimer);
idleStopTimer = null;
}
}
function normalizePath(value) {
const trimmed = value.trim();
if (!trimmed)
return "";
return trimmed.startsWith("~") ? path_1.default.join(os_1.default.homedir(), trimmed.slice(1)) : trimmed;
}
function pidFilePath() {
return path_1.default.join((0, core_bundle_mjs_1.getLogsDir)(), "llama-server-router.pid");
}
function logFilePath() {
return path_1.default.join((0, core_bundle_mjs_1.getLogsDir)(), "llama-server-router.log");
}
function presetFilePath() {
return path_1.default.join((0, core_bundle_mjs_1.getLogsDir)(), "llama-server-router-preset.ini");
}
function isProcessAlive(pid) {
try {
process.kill(pid, 0);
return true;
}
catch {
return false;
}
}
function readPid() {
try {
const raw = fs_1.default.readFileSync(pidFilePath(), "utf-8").trim();
const pid = parseInt(raw, 10);
return Number.isFinite(pid) && pid > 0 ? pid : null;
}
catch {
return null;
}
}
function fetchWithTimeout(url, timeoutMs) {
const controller = new AbortController();
const timer = setTimeout(() => controller.abort(), timeoutMs);
return fetch(url, { signal: controller.signal }).finally(() => clearTimeout(timer));
}
async function isHealthy(host, port) {
try {
const res = await fetchWithTimeout(`http://${host}:${port}/v1/models`, HEALTH_FETCH_TIMEOUT_MS);
return res.ok;
}
catch {
return false;
}
}
async function pollHealth(host, port) {
const deadline = Date.now() + HEALTH_POLL_TIMEOUT_MS;
while (Date.now() < deadline) {
if (await isHealthy(host, port))
return true;
await new Promise((resolve) => setTimeout(resolve, HEALTH_POLL_INTERVAL_MS));
}
return false;
}
function buildPresetIni(ctxSize, localModel) {
let ini = `[*]\nctx-size = ${ctxSize}\n`;
if (localModel) {
ini += `\n[${localModel.name}]\nmodel = ${localModel.modelPath}\nmmproj = ${localModel.mmprojPath}\n`;
}
return ini;
}
async function listRouterModelIds(host, port) {
try {
const res = await fetchWithTimeout(`http://${host}:${port}/v1/models`, HEALTH_FETCH_TIMEOUT_MS);
if (!res.ok)
return null;
const data = (await res.json());
const models = Array.isArray(data?.data) ? data.data : [];
return models.map((m) => String(m?.id ?? "").trim()).filter(Boolean);
}
catch {
return null;
}
}
/**
* Ensures a router-mode llama-server is listening at host:port, spawning it if nothing answers
* there yet. Idempotent and safe to call before every request -- a already-healthy endpoint
* (ours or adopted) is trusted as-is, since router mode's own model management is independent of
* which process started the server. Reschedules this process's own idle-shutdown timer on every
* call (see releaseIdleLlamaServerRouter()).
*/
async function ensureLlamaServerRouterRunning(config) {
const { host, port, ctxSize, modelsMax, idleTtlMinutes, localModel } = config;
cancelIdleStop();
activePort = port;
const binaryPath = normalizePath(config.binaryPath);
if (!binaryPath) {
throw new Error("LLAMA_SERVER_BINARY is not configured.");
}
if (!fs_1.default.existsSync(binaryPath)) {
throw new Error(`llama-server binary does not exist: ${binaryPath}`);
}
if (await isHealthy(host, port)) {
if (localModel) {
const registeredIds = await listRouterModelIds(host, port);
const alreadyRegistered = registeredIds !== null && registeredIds.includes(localModel.name);
if (!alreadyRegistered) {
// The running router's preset doesn't know this local model yet -- router mode has no
// hot-reload endpoint for --models-preset, so the only way to register a new local model
// is to rewrite the ini and restart the process we ourselves own.
await stopLlamaServerRouter(port);
}
}
if (await isHealthy(host, port)) {
scheduleIdleStop(port, idleTtlMinutes);
return;
}
}
const stalePid = readPid();
if (stalePid !== null) {
if (isProcessAlive(stalePid)) {
throw new Error(`Port ${port} is not responding, but a previously spawned llama-server router (PID ${stalePid}) is still alive. Check logs: ${logFilePath()}`);
}
try {
fs_1.default.unlinkSync(pidFilePath());
}
catch { }
}
fs_1.default.mkdirSync((0, core_bundle_mjs_1.getLogsDir)(), { recursive: true });
fs_1.default.writeFileSync(presetFilePath(), buildPresetIni(ctxSize, localModel));
const args = [
"--host", host,
"--port", String(port),
"--models-max", String(modelsMax),
"--models-preset", presetFilePath(),
];
const logStream = fs_1.default.createWriteStream(logFilePath(), { flags: "a" });
const child = (0, child_process_1.spawn)(binaryPath, args, {
detached: true,
stdio: ["ignore", "pipe", "pipe"],
env: { ...process.env, HOME: os_1.default.homedir() },
});
child.stdout?.pipe(logStream);
child.stderr?.pipe(logStream);
child.once("error", (error) => {
try {
fs_1.default.appendFileSync(logFilePath(), `[llama-server-router] process error: ${error.message}\n`);
}
catch { }
});
child.unref();
if (!child.pid) {
throw new Error(`Failed to spawn llama-server router process. Check logs: ${logFilePath()}`);
}
fs_1.default.writeFileSync(pidFilePath(), String(child.pid));
const ready = await pollHealth(host, port);
if (!ready) {
throw new Error(`llama-server router did not become healthy within ${HEALTH_POLL_TIMEOUT_MS / 1000}s on ${host}:${port}. Check logs: ${logFilePath()}`);
}
scheduleIdleStop(port, idleTtlMinutes);
}
/** 0 disables idle shutdown entirely (process stays until the host process itself exits). */
function scheduleIdleStop(port, idleTtlMinutes) {
cancelIdleStop();
if (idleTtlMinutes <= 0)
return;
idleStopTimer = setTimeout(() => {
idleStopTimer = null;
if (activePort !== port)
return;
void stopLlamaServerRouter(port).catch((error) => {
try {
fs_1.default.appendFileSync(logFilePath(), `[llama-server-router] failed to stop idle router: ${error instanceof Error ? error.message : String(error)}\n`);
}
catch { }
});
}, idleTtlMinutes * 60_000);
idleStopTimer.unref();
}
/** Stops the router process this plugin itself spawned (tracked via the PID file); a no-op if nothing is running. */
async function stopLlamaServerRouter(port) {
cancelIdleStop();
const pid = readPid();
if (pid === null)
return;
if (isProcessAlive(pid)) {
try {
process.kill(pid, "SIGTERM");
}
catch { }
}
try {
fs_1.default.unlinkSync(pidFilePath());
}
catch { }
if (activePort === port)
activePort = null;
}