dist-mcp / services / VisionAnalyzer.js
dist-mcp / services / VisionAnalyzer.js
"use strict";
var __importDefault = (this && this.__importDefault) || function (mod) {
return (mod && mod.__esModule) ? mod : { "default": mod };
};
Object.defineProperty(exports, "__esModule", { value: true });
exports.ensureVisionModelReady = ensureVisionModelReady;
exports.analyzeLmStudioVisionBatch = analyzeLmStudioVisionBatch;
exports.detectLmStudioVisionBatch = detectLmStudioVisionBatch;
const fs_1 = __importDefault(require("fs"));
const path_1 = __importDefault(require("path"));
const core_bundle_mjs_1 = require("../core-bundle.mjs");
const LlamaServerRouterManager_js_1 = require("./LlamaServerRouterManager.js");
const JSON_FENCE_RE = /```(?:json)?\s*([\s\S]*?)```/i;
const ITEM_RE = /\{\s*"bbox_2d":\s*\[(\d+),\s*(\d+),\s*(\d+),\s*(\d+)\],\s*"label":\s*"([^"]+)"\s*\}/gi;
const JSON_FORMAT = " Output JSON only — a JSON array where each element has" +
" 'bbox_2d' ([x1, y1, x2, y2] as integers normalized 0–1000) and 'label' (a string)." +
" No prose, no markdown, no explanation.";
const LABEL_FORMAT_RULE = "\n\nLABEL FORMAT RULE (mandatory):" +
"\n- Labels must be concise and specific: 2–4 words maximum." +
"\n- No commas or punctuation inside a label (no ',', '.', ';', ':', '/') — downstream tools split labels on commas." +
"\n- Examples: 'plugin list', 'plugin name', 'human face', 'left hand', 'red car', 'fluffy owl toy'";
function normalizeLmApiRoot(baseUrl) {
return String(baseUrl || "")
.trim()
.replace(/\/(api\/v1|v1)\/?$/i, "")
.replace(/\/+$/, "");
}
function authHeaders(apiKey, contentType = false) {
const headers = {};
if (contentType)
headers["Content-Type"] = "application/json";
if (apiKey?.trim())
headers.Authorization = `Bearer ${apiKey.trim()}`;
return headers;
}
function logVisionRequestMetadata(metadata) {
const line = `[LmStudioVisionAnalyzer] /v1/chat/completions request ${JSON.stringify(metadata)}`;
console.info(line);
try {
const logsDir = (0, core_bundle_mjs_1.getLogsDir)();
if (!fs_1.default.existsSync(logsDir))
fs_1.default.mkdirSync(logsDir, { recursive: true });
fs_1.default.appendFileSync(path_1.default.join(logsDir, (0, core_bundle_mjs_1.getPluginLogFilename)()), `${new Date().toISOString()} - ${line}\n`, "utf8");
}
catch { }
}
function hasLoadedInstances(modelInfo) {
return Array.isArray(modelInfo?.loaded_instances) && modelInfo.loaded_instances.length > 0;
}
async function getBionicVisionModelState(baseUrl, apiKey, modelKey) {
const apiRoot = normalizeLmApiRoot(baseUrl);
if (!apiRoot)
return { available: false, loaded: false };
const normalizedModelKey = modelKey.trim().toLowerCase();
const controller = new AbortController();
const timeout = setTimeout(() => controller.abort(), 5000);
try {
const response = await fetch(`${apiRoot}/api/v1/models`, {
headers: authHeaders(apiKey),
signal: controller.signal,
});
if (!response.ok)
return { available: false, loaded: false };
const data = await response.json();
const models = Array.isArray(data)
? data
: Array.isArray(data?.models)
? data.models
: Array.isArray(data?.data)
? data.data
: [];
const modelInfo = models.find((entry) => {
const key = String(entry?.key || entry?.id || "").trim().toLowerCase();
return key === normalizedModelKey;
});
if (!modelInfo)
return { available: false, loaded: false };
return {
available: true,
loaded: hasLoadedInstances(modelInfo),
modelKey: String(modelInfo?.key || modelInfo?.id || "").trim() || undefined,
};
}
catch {
return { available: false, loaded: false };
}
finally {
clearTimeout(timeout);
}
}
async function loadBionicVisionInstance(baseUrl, apiKey, modelKey) {
const apiRoot = normalizeLmApiRoot(baseUrl);
if (!apiRoot) {
return { ok: false, error: "Vision API base URL is empty." };
}
const controller = new AbortController();
const timeout = setTimeout(() => controller.abort(), 600_000);
try {
const response = await fetch(`${apiRoot}/api/v1/models/load`, {
method: "POST",
headers: authHeaders(apiKey, true),
body: JSON.stringify({
model: modelKey,
echo_load_config: true,
}),
signal: controller.signal,
});
const text = await response.text().catch(() => "");
let data = null;
if (text.trim()) {
try {
data = JSON.parse(text);
}
catch {
data = { raw: text };
}
}
const apiError = data?.error?.message || data?.error || data?.message;
if (!response.ok || apiError) {
const detail = apiError || text || `${response.status} ${response.statusText}`;
return {
ok: false,
error: `Vision API could not load '${modelKey}' via /api/v1/models/load. This can happen when there are not enough system resources available. Error: ${detail}`,
};
}
return { ok: true };
}
catch (error) {
const detail = error?.name === "AbortError"
? "request timed out after 600000 ms"
: error?.message || String(error);
return {
ok: false,
error: `Vision API could not load '${modelKey}' via /api/v1/models/load. This can happen when there are not enough system resources available. Error: ${detail}`,
};
}
finally {
clearTimeout(timeout);
}
}
/**
* force_cancel_active: true bypasses Unsloth Studio's refusal to load while another model still
* has an active generation in flight ("active_generations" error). alongside: true requests
* keeping other models loaded instead of evicting them -- confirmed NOT implemented in the
* installed 0.1.900-beta build (live-verified 2026-10-01: real curl + /v1/models + OS process
* list all show the previous model's llama-server killed regardless of this flag; the field is
* entirely absent from that build's LoadRequest schema, silently dropped). unslothai/unsloth#11591
* ("Studio: serve multiple models at once") adds real support for exactly this flag at load time
* ("send \"alongside\": true when loading a model") -- kept here deliberately, forward-compatible,
* for once that PR ships. Do NOT also send it in the chat request: that PR never puts it there.
*/
async function loadUnslothVisionInstance(baseUrl, apiKey, modelKey) {
const apiRoot = normalizeLmApiRoot(baseUrl);
if (!apiRoot) {
return { ok: false, error: "Vision API base URL is empty." };
}
const controller = new AbortController();
const timeout = setTimeout(() => controller.abort(), 600_000);
try {
const response = await fetch(`${apiRoot}/api/inference/load`, {
method: "POST",
headers: authHeaders(apiKey, true),
body: JSON.stringify({
model_path: modelKey,
gguf_variant: "Q4_K_M",
max_seq_length: 8192,
alongside: true,
force_cancel_active: true,
}),
signal: controller.signal,
});
const text = await response.text().catch(() => "");
let data = null;
if (text.trim()) {
try {
data = JSON.parse(text);
}
catch {
data = { raw: text };
}
}
const apiError = data?.error?.message || data?.error || data?.message;
if (!response.ok || apiError) {
const detail = apiError || text || `${response.status} ${response.statusText}`;
return {
ok: false,
error: `Vision API could not load '${modelKey}' via /api/inference/load. This can happen when there are not enough system resources available. Error: ${detail}`,
};
}
return { ok: true };
}
catch (error) {
const detail = error?.name === "AbortError"
? "request timed out after 600000 ms"
: error?.message || String(error);
return {
ok: false,
error: `Vision API could not load '${modelKey}' via /api/inference/load. This can happen when there are not enough system resources available. Error: ${detail}`,
};
}
finally {
clearTimeout(timeout);
}
}
/** Strand 1/3: LM Studio's own endpoints (also the default for `generic` server adapters that happen to be LM Studio). */
async function ensureBionicVisionModelReady(config) {
const modelKey = String(config.modelKey || "").trim();
if (!modelKey) {
return { ok: false, error: "Vision API mode is active, but Qwen3-VL model key is empty." };
}
const initialState = await getBionicVisionModelState(config.baseUrl, config.apiKey, modelKey);
if (!initialState.available) {
return { ok: false, error: `Vision API does not list '${modelKey}' as an available model via /api/v1/models.` };
}
if (initialState.loaded) {
return { ok: true, loaded: false };
}
try {
config.status?.(`Loading ${modelKey}...`);
}
catch { }
const loadResult = await loadBionicVisionInstance(config.baseUrl, config.apiKey, modelKey);
if (!loadResult.ok)
return loadResult;
const loadedState = await getBionicVisionModelState(config.baseUrl, config.apiKey, modelKey);
if (!loadedState.loaded) {
return {
ok: false,
error: `Vision API loaded '${modelKey}' via /api/v1/models/load, but /api/v1/models did not report it as loaded.`,
};
}
if (loadedState.modelKey?.trim().toLowerCase() !== modelKey.toLowerCase()) {
return {
ok: false,
error: `Vision API loaded a model, but /api/v1/models reports '${loadedState.modelKey || "unknown model"}' instead of '${modelKey}'.`,
};
}
return { ok: true, loaded: true };
}
/** Fast per-call status check (live-verified much faster than /v1/models on Unsloth Studio) -- only ever describes the single most-recently-active model, never a full multi-model registry. */
async function getUnslothVisionModelState(baseUrl, apiKey) {
const apiRoot = normalizeLmApiRoot(baseUrl);
if (!apiRoot)
return { loaded: false };
const controller = new AbortController();
const timeout = setTimeout(() => controller.abort(), 5000);
try {
const response = await fetch(`${apiRoot}/api/inference/status`, {
headers: authHeaders(apiKey),
signal: controller.signal,
});
if (!response.ok)
return { loaded: false };
const data = await response.json();
const activeModel = typeof data?.active_model === "string" ? data.active_model : undefined;
const loadedList = Array.isArray(data?.loaded) ? data.loaded : [];
return { loaded: loadedList.length > 0, activeModel };
}
catch {
return { loaded: false };
}
finally {
clearTimeout(timeout);
}
}
/**
* Strand 2/3: Unsloth Studio. Checks loaded state via GET /api/inference/status, NOT /v1/models --
* live-verified /v1/models is extremely slow on this server, /api/inference/status is fast.
* /api/inference/status only ever reports the single most-recently-active model (never a full
* registry), so this never checks availability up front -- a load failure doubles as the "not
* available" signal instead, same as bionic/generic do via their own availability check. Does NOT
* check whether any other ("agent") model stays loaded afterward -- confirmed impossible on this
* server (see unslothai/unsloth#11591 note on loadUnslothVisionInstance()).
*/
async function ensureUnslothVisionModelReady(config) {
const modelKey = String(config.modelKey || "").trim();
if (!modelKey) {
return { ok: false, error: "Vision API mode is active, but Qwen3-VL model key is empty." };
}
const targetBasename = visionModelBasename(modelKey);
const initialState = await getUnslothVisionModelState(config.baseUrl, config.apiKey);
if (initialState.loaded && initialState.activeModel && visionModelBasename(initialState.activeModel) === targetBasename) {
return { ok: true, loaded: false };
}
try {
config.status?.(`Loading ${modelKey}...`);
}
catch { }
const loadResult = await loadUnslothVisionInstance(config.baseUrl, config.apiKey, modelKey);
if (!loadResult.ok)
return loadResult;
const loadedState = await getUnslothVisionModelState(config.baseUrl, config.apiKey);
if (!loadedState.loaded || !loadedState.activeModel || visionModelBasename(loadedState.activeModel) !== targetBasename) {
return {
ok: false,
error: `Vision API loaded '${modelKey}' via /api/inference/load, but /api/inference/status does not report it as the active model.`,
};
}
return { ok: true, loaded: true };
}
/** Smallest common denominator across /v1/models id formats: only the final path segment, case-insensitively, since a server is free to report its own id shape for the same model. */
function visionModelBasename(key) {
const trimmed = key.trim();
const idx = trimmed.lastIndexOf("/");
return (idx >= 0 ? trimmed.slice(idx + 1) : trimmed).toLowerCase();
}
/** Only llama-server router mode's ids carry a trailing ":quant" tag (e.g. ":Q4_K_M") -- strip it before comparing against a bare hf-repo-style QWEN3_VL_MODEL value. */
function stripLlamaServerQuantTag(id) {
return id.replace(/:[^/:]+$/, "");
}
/**
* Resolves the sole *mmproj*.gguf file in the same directory as an absolute-path GGUF model file --
* required for Qwen3-VL's vision support when VISION_API=llama-server is pointed at a local file
* instead of an hf-repo id (see ensureLlamaServerVisionModelReady).
*/
function resolveLlamaServerMmprojPath(modelFilePath) {
const dir = path_1.default.dirname(modelFilePath);
let entries;
try {
entries = fs_1.default.readdirSync(dir);
}
catch {
return { ok: false, error: `Vision API (llama-server) could not read the model's directory: ${dir}` };
}
const candidates = entries.filter((e) => e.toLowerCase().endsWith(".gguf") && e.toLowerCase().includes("mmproj"));
if (candidates.length === 0) {
return { ok: false, error: `Vision API (llama-server) found no mmproj *.gguf file alongside '${modelFilePath}' in ${dir}.` };
}
if (candidates.length > 1) {
return { ok: false, error: `Vision API (llama-server) found multiple candidate mmproj files in ${dir}: ${candidates.join(", ")}. Keep exactly one.` };
}
return { ok: true, mmprojPath: path_1.default.join(dir, candidates[0]) };
}
/** Shared HTTP utility (not preset-branching logic). `loaded` is a real per-model flag on this endpoint -- LIVE-VERIFIED against both LM Studio-shaped and Unsloth Studio-shaped /v1/models responses. */
async function listOpenAiModels(baseUrl, apiKey) {
const apiRoot = normalizeLmApiRoot(baseUrl);
if (!apiRoot)
return null;
const controller = new AbortController();
const timeout = setTimeout(() => controller.abort(), 5000);
try {
const response = await fetch(`${apiRoot}/v1/models`, {
headers: authHeaders(apiKey),
signal: controller.signal,
});
if (!response.ok)
return null;
const data = await response.json();
const models = Array.isArray(data?.data) ? data.data : Array.isArray(data) ? data : [];
return models
.map((entry) => ({ id: String(entry?.id ?? entry?.key ?? "").trim(), loaded: Boolean(entry?.loaded) }))
.filter((m) => m.id);
}
catch {
return null;
}
finally {
clearTimeout(timeout);
}
}
/** Strand 3/3: plain OpenAI /v1/models listing, no explicit load step, since an available model is trusted to be usable on demand. */
async function ensureGenericVisionModelReady(config) {
const modelKey = String(config.modelKey || "").trim();
if (!modelKey) {
return { ok: false, error: "Vision API mode is active, but Qwen3-VL model key is empty." };
}
const models = await listOpenAiModels(config.baseUrl, config.apiKey);
if (models === null) {
return { ok: false, error: "Vision API could not be reached via /v1/models." };
}
const targetBasename = visionModelBasename(modelKey);
const isAvailable = models.some((m) => visionModelBasename(m.id) === targetBasename);
if (!isAvailable) {
return { ok: false, error: `Vision API does not list '${modelKey}' as an available model via /v1/models.` };
}
return { ok: true, loaded: false };
}
/**
* Strand 4/4: a standalone llama.cpp router-mode server we spawn and manage ourselves, entirely
* independent of any agent-model process (LM Studio Bionic, Unsloth Studio) -- live-verified
* 2026-10-01 that loading/using a vision model here never evicts an agent model loaded elsewhere.
* No explicit load call: router mode auto-loads the requested model on the first
* /v1/chat/completions request, same lazy "trust it's usable on demand" pattern as the generic
* adapter, after ensureLlamaServerRouterRunning() has made sure the router process itself exists.
* Listens on its own dedicated llamaServer.port -- NEVER derived from config.baseUrl, which stays
* the real Vision API's own fixed address (e.g. Unsloth Studio's 8888) and must not be hijacked as
* our router's bind address (would either collide with that real service or silently adopt it).
*/
async function ensureLlamaServerVisionModelReady(config) {
const modelKey = String(config.modelKey || "").trim();
if (!modelKey) {
return { ok: false, error: "Vision API mode is active, but Qwen3-VL model key is empty." };
}
const llamaServer = config.llamaServer;
if (!llamaServer) {
return { ok: false, error: "Vision API mode is 'llama-server', but no llamaServer config (port/binaryPath/ctxSize/modelsMax/idleTtlMinutes) was supplied." };
}
const host = "127.0.0.1";
const routerBaseUrl = `http://${host}:${llamaServer.port}/v1`;
// An absolute path to a .gguf file (as opposed to an hf-repo-style id like
// "unsloth/Qwen3-VL-8B-Instruct-GGUF") bypasses HF-cache auto-discovery entirely -- registered
// as a named custom preset section instead (see ensureLlamaServerRouterRunning's localModel).
let localModel;
if (path_1.default.isAbsolute(modelKey) && modelKey.toLowerCase().endsWith(".gguf")) {
if (!fs_1.default.existsSync(modelKey)) {
return { ok: false, error: `Vision API (llama-server) model file does not exist: ${modelKey}` };
}
const mmprojResult = resolveLlamaServerMmprojPath(modelKey);
if (!mmprojResult.ok) {
return { ok: false, error: mmprojResult.error };
}
localModel = { name: path_1.default.basename(modelKey, ".gguf"), modelPath: modelKey, mmprojPath: mmprojResult.mmprojPath };
}
try {
await (0, LlamaServerRouterManager_js_1.ensureLlamaServerRouterRunning)({
host,
port: llamaServer.port,
binaryPath: llamaServer.binaryPath,
ctxSize: llamaServer.ctxSize,
modelsMax: llamaServer.modelsMax,
idleTtlMinutes: llamaServer.idleTtlMinutes,
localModel,
});
}
catch (error) {
return { ok: false, error: error?.message || String(error) };
}
const models = await listOpenAiModels(routerBaseUrl, config.apiKey);
if (models === null) {
return { ok: false, error: `Vision API (llama-server router) could not be reached via ${routerBaseUrl}/models.` };
}
if (localModel) {
const matchedLocal = models.find((m) => m.id === localModel.name);
if (!matchedLocal) {
return { ok: false, error: `Vision API (llama-server router) does not list '${localModel.name}' as an available model via ${routerBaseUrl}/models.` };
}
return { ok: true, loaded: false, resolvedModelKey: matchedLocal.id, resolvedBaseUrl: routerBaseUrl };
}
// Router mode's own /v1/models ids are "hf-repo:quant" (e.g. "unsloth/Qwen3-VL-8B-Instruct-GGUF:Q4_K_M"),
// never just the bare hf-repo -- strip the trailing ":quant" tag on BOTH sides before the usual
// basename compare, scoped to this adapter only (other adapters' servers never emit this shape).
const targetBasename = visionModelBasename(stripLlamaServerQuantTag(modelKey));
const matched = models.find((m) => visionModelBasename(stripLlamaServerQuantTag(m.id)) === targetBasename);
if (!matched) {
return { ok: false, error: `Vision API (llama-server router) does not list '${modelKey}' as an available model via ${routerBaseUrl}/models.` };
}
// The real chat request must use the router's own exact tagged id and its own address, not the
// caller's originally-configured modelKey/baseUrl (see the ":quant" tag note above).
return { ok: true, loaded: false, resolvedModelKey: matched.id, resolvedBaseUrl: routerBaseUrl };
}
const VISION_API_ADAPTERS = {
bionic: ensureBionicVisionModelReady,
unsloth: ensureUnslothVisionModelReady,
generic: ensureGenericVisionModelReady,
"llama-server": ensureLlamaServerVisionModelReady,
};
/** Single dispatch point between the 4 independent adapters, add a new VISION_API value by adding one more entry here, never by branching inside an adapter. */
async function ensureVisionModelReady(visionApi, config) {
return VISION_API_ADAPTERS[visionApi](config);
}
function mimeFromPath(filePath) {
const ext = path_1.default.extname(filePath).toLowerCase();
if (ext === ".jpg" || ext === ".jpeg")
return "image/jpeg";
if (ext === ".webp")
return "image/webp";
if (ext === ".gif")
return "image/gif";
return "image/png";
}
function readUInt24LE(buffer, offset) {
return buffer[offset] | (buffer[offset + 1] << 8) | (buffer[offset + 2] << 16);
}
function readPngDimensions(buffer) {
if (buffer.length < 24)
return null;
if (buffer.toString("ascii", 1, 4) !== "PNG")
return null;
return {
width: buffer.readUInt32BE(16),
height: buffer.readUInt32BE(20),
};
}
function readGifDimensions(buffer) {
if (buffer.length < 10)
return null;
const signature = buffer.toString("ascii", 0, 6);
if (signature !== "GIF87a" && signature !== "GIF89a")
return null;
return {
width: buffer.readUInt16LE(6),
height: buffer.readUInt16LE(8),
};
}
function readWebpDimensions(buffer) {
if (buffer.length < 30)
return null;
if (buffer.toString("ascii", 0, 4) !== "RIFF" || buffer.toString("ascii", 8, 12) !== "WEBP") {
return null;
}
const chunkType = buffer.toString("ascii", 12, 16);
if (chunkType === "VP8X" && buffer.length >= 30) {
return {
width: readUInt24LE(buffer, 24) + 1,
height: readUInt24LE(buffer, 27) + 1,
};
}
if (chunkType === "VP8L" && buffer.length >= 25 && buffer[20] === 0x2f) {
const bits = buffer.readUInt32LE(21);
return {
width: (bits & 0x3fff) + 1,
height: ((bits >> 14) & 0x3fff) + 1,
};
}
if (chunkType === "VP8 " && buffer.length >= 30) {
return {
width: buffer.readUInt16LE(26) & 0x3fff,
height: buffer.readUInt16LE(28) & 0x3fff,
};
}
return null;
}
function readJpegDimensions(buffer) {
if (buffer.length < 4 || buffer[0] !== 0xff || buffer[1] !== 0xd8)
return null;
let offset = 2;
while (offset + 9 < buffer.length) {
if (buffer[offset] !== 0xff) {
offset += 1;
continue;
}
while (offset < buffer.length && buffer[offset] === 0xff)
offset += 1;
const marker = buffer[offset];
offset += 1;
if (marker === 0xd9 || marker === 0xda)
break;
if (offset + 2 > buffer.length)
break;
const segmentLength = buffer.readUInt16BE(offset);
if (segmentLength < 2 || offset + segmentLength > buffer.length)
break;
const isStartOfFrame = (marker >= 0xc0 && marker <= 0xc3) ||
(marker >= 0xc5 && marker <= 0xc7) ||
(marker >= 0xc9 && marker <= 0xcb) ||
(marker >= 0xcd && marker <= 0xcf);
if (isStartOfFrame && segmentLength >= 7) {
return {
height: buffer.readUInt16BE(offset + 3),
width: buffer.readUInt16BE(offset + 5),
};
}
offset += segmentLength;
}
return null;
}
async function readImageDimensions(filePath) {
const buffer = await fs_1.default.promises.readFile(filePath);
const dimensions = readPngDimensions(buffer) ||
readJpegDimensions(buffer) ||
readWebpDimensions(buffer) ||
readGifDimensions(buffer);
if (!dimensions || dimensions.width <= 0 || dimensions.height <= 0) {
throw new Error(`Could not determine image dimensions for ${filePath}`);
}
return dimensions;
}
function extractMessageText(data) {
const pieces = [];
// Standard OpenAI /v1/chat/completions shape.
const choices = Array.isArray(data?.choices) ? data.choices : [];
for (const choice of choices) {
const content = choice?.message?.content;
if (typeof content === "string") {
pieces.push(content);
}
else if (Array.isArray(content)) {
for (const part of content) {
if (typeof part === "string")
pieces.push(part);
else if (typeof part?.text === "string")
pieces.push(part.text);
}
}
}
// LM Studio's proprietary /api/v1/chat "responses"-style shape.
const output = Array.isArray(data?.output) ? data.output : [];
for (const item of output) {
if (item?.type !== "message")
continue;
const content = item?.content;
if (typeof content === "string") {
pieces.push(content);
}
else if (Array.isArray(content)) {
for (const part of content) {
if (typeof part === "string") {
pieces.push(part);
}
else if (typeof part?.text === "string") {
pieces.push(part.text);
}
else if (typeof part?.content === "string") {
pieces.push(part.content);
}
}
}
}
if (pieces.length === 0 && typeof data?.text === "string") {
pieces.push(data.text);
}
if (pieces.length === 0 && typeof data?.content === "string") {
pieces.push(data.content);
}
return pieces.join("\n").trim();
}
function buildDetectPrompt(task, odPrompt) {
const label = String(task || "").trim();
if (label) {
return `Detect all instances of '${label}' in the image.` + LABEL_FORMAT_RULE + JSON_FORMAT;
}
const instruction = String(odPrompt || "").trim();
if (!instruction) {
throw new Error("No OD prompt available: odPrompt not set and DETECT_OD_PROMPT env var not set");
}
return instruction + LABEL_FORMAT_RULE + JSON_FORMAT;
}
function bboxToCrop(bbox, width, height) {
const [x1, y1, x2, y2] = bbox;
return {
cropLeft: (x1 / width) * 100,
cropRight: ((width - x2) / width) * 100,
cropTop: (y1 / height) * 100,
cropBottom: ((height - y2) / height) * 100,
};
}
function parseQwen3VlDetectionOutput(text, width, height) {
const objects = [];
const seen = new Set();
const fenceMatch = JSON_FENCE_RE.exec(text);
const jsonText = fenceMatch ? fenceMatch[1].trim() : text.trim();
let items = null;
try {
const parsed = JSON.parse(jsonText);
items = Array.isArray(parsed) ? parsed : [parsed];
}
catch {
const recovered = [];
ITEM_RE.lastIndex = 0;
for (const match of text.matchAll(ITEM_RE)) {
recovered.push({
bbox_2d: [Number(match[1]), Number(match[2]), Number(match[3]), Number(match[4])],
label: match[5],
});
}
items = recovered.length > 0 ? recovered : [];
}
for (const item of items) {
if (!item || typeof item !== "object")
continue;
const bbox = item.bbox_2d;
const label = String(item.label || "");
if (!Array.isArray(bbox) || bbox.length !== 4)
continue;
const [nx1, ny1, nx2, ny2] = bbox.map((value) => Number(value));
if (![nx1, ny1, nx2, ny2].every((value) => Number.isFinite(value) && value >= 0 && value <= 1000)) {
continue;
}
if (nx2 <= nx1 || ny2 <= ny1)
continue;
if (nx1 < 10 && ny1 < 10 && nx2 > 990 && ny2 > 990)
continue;
const dedupKey = `${Math.round(nx1)}:${Math.round(ny1)}:${Math.round(nx2)}:${Math.round(ny2)}:${label}`;
if (seen.has(dedupKey))
continue;
seen.add(dedupKey);
const pixelBbox = [
(nx1 / 1000) * width,
(ny1 / 1000) * height,
(nx2 / 1000) * width,
(ny2 / 1000) * height,
];
objects.push({
label,
bbox: pixelBbox,
...bboxToCrop(pixelBbox, width, height),
});
}
return objects;
}
async function chatOnce(item, prompt, config) {
const apiRoot = normalizeLmApiRoot(config.baseUrl);
if (!apiRoot) {
throw new Error("Vision API base URL is empty");
}
const endpoint = `${apiRoot}/v1/chat/completions`;
const timeoutMs = config.timeoutMs ?? 180_000;
const model = config.model || "vision-capability-priming";
const buf = await fs_1.default.promises.readFile(item.filePath);
const dataUrl = `data:${mimeFromPath(item.filePath)};base64,${buf.toString("base64")}`;
const payload = {
model,
seed: 175308301,
messages: [
{
role: "user",
content: [
{ type: "text", text: prompt },
{ type: "image_url", image_url: { url: dataUrl } },
],
},
],
};
if (typeof config.maxTokens === "number" && Number.isFinite(config.maxTokens) && config.maxTokens > 0) {
payload.max_tokens = Math.floor(config.maxTokens);
}
if (typeof config.temperature === "number" && Number.isFinite(config.temperature)) {
payload.temperature = config.temperature;
}
logVisionRequestMetadata({
configuredBaseUrl: config.baseUrl,
apiRoot,
endpoint,
model,
store: payload.store ?? null,
max_output_tokens: payload.max_output_tokens ?? payload.max_tokens ?? null,
temperature: payload.temperature ?? null,
promptChars: prompt.length,
imageBytes: buf.byteLength,
payloadKeys: Object.keys(payload),
});
const headers = authHeaders(config.apiKey, true);
const controller = new AbortController();
const timeout = setTimeout(() => controller.abort(), timeoutMs);
const startedAt = Date.now();
let data;
try {
const resp = await fetch(endpoint, {
method: "POST",
headers,
body: JSON.stringify(payload),
signal: controller.signal,
});
clearTimeout(timeout);
if (!resp.ok) {
const detail = await resp.text().catch(() => "(no body)");
throw new Error(`Vision API ${resp.status}: ${detail}`);
}
data = await resp.json();
}
catch (error) {
clearTimeout(timeout);
if (error?.name === "AbortError") {
throw new Error(`Vision API timed out after ${timeoutMs}ms`);
}
throw new Error(`Vision API failed: ${error?.message || String(error)}`);
}
return {
text: extractMessageText(data),
elapsedMs: Date.now() - startedAt,
bytes: buf.byteLength,
modelInstanceId: typeof data?.model_instance_id === "string" ? data.model_instance_id : "",
};
}
async function analyzeLmStudioVisionBatch(items, config) {
if (!items.length) {
return {
results: [],
totalInferenceTimeMs: 0,
backend: "vision-api",
};
}
const results = [];
let totalInferenceTimeMs = 0;
for (const item of items) {
console.info(`[VisionAnalyzer] /v1/chat/completions start mode=analyze id=${item.id} timeoutMs=${config.timeoutMs ?? 180_000}`);
const response = await chatOnce(item, config.prompt || "Describe the image.", config);
console.info(`[VisionAnalyzer] /v1/chat/completions ok mode=analyze id=${item.id} bytes=${response.bytes} elapsedMs=${response.elapsedMs} modelInstance=${response.modelInstanceId || "?"}`);
results.push({
id: item.id,
text: response.text,
inferenceTimeMs: response.elapsedMs,
});
totalInferenceTimeMs += response.elapsedMs;
}
return {
results,
totalInferenceTimeMs,
backend: "vision-api",
};
}
async function detectLmStudioVisionBatch(items, config) {
if (!items.length) {
return {
results: [],
totalInferenceTimeMs: 0,
backend: "vision-api",
};
}
const prompt = buildDetectPrompt(config.task, config.odPrompt);
const results = [];
let totalInferenceTimeMs = 0;
for (const item of items) {
const { width, height } = await readImageDimensions(item.filePath);
console.info(`[VisionAnalyzer] /v1/chat/completions start mode=detect id=${item.id} timeoutMs=${config.timeoutMs ?? 120_000}`);
const response = await chatOnce(item, prompt, {
baseUrl: config.baseUrl,
apiKey: config.apiKey,
model: config.model || "vision-capability-priming",
maxTokens: config.maxTokens,
temperature: config.temperature,
timeoutMs: config.timeoutMs ?? 120_000,
});
const objects = parseQwen3VlDetectionOutput(response.text, width, height);
console.info(`[VisionAnalyzer] /v1/chat/completions ok mode=detect id=${item.id} objects=${objects.length} bytes=${response.bytes} elapsedMs=${response.elapsedMs} modelInstance=${response.modelInstanceId || "?"}`);
results.push({
id: item.id,
objects,
imageWidth: width,
imageHeight: height,
inferenceTimeMs: response.elapsedMs,
});
totalInferenceTimeMs += response.elapsedMs;
}
return {
results,
totalInferenceTimeMs,
backend: "vision-api",
};
}
"use strict";
var __importDefault = (this && this.__importDefault) || function (mod) {
return (mod && mod.__esModule) ? mod : { "default": mod };
};
Object.defineProperty(exports, "__esModule", { value: true });
exports.ensureVisionModelReady = ensureVisionModelReady;
exports.analyzeLmStudioVisionBatch = analyzeLmStudioVisionBatch;
exports.detectLmStudioVisionBatch = detectLmStudioVisionBatch;
const fs_1 = __importDefault(require("fs"));
const path_1 = __importDefault(require("path"));
const core_bundle_mjs_1 = require("../core-bundle.mjs");
const LlamaServerRouterManager_js_1 = require("./LlamaServerRouterManager.js");
const JSON_FENCE_RE = /```(?:json)?\s*([\s\S]*?)```/i;
const ITEM_RE = /\{\s*"bbox_2d":\s*\[(\d+),\s*(\d+),\s*(\d+),\s*(\d+)\],\s*"label":\s*"([^"]+)"\s*\}/gi;
const JSON_FORMAT = " Output JSON only — a JSON array where each element has" +
" 'bbox_2d' ([x1, y1, x2, y2] as integers normalized 0–1000) and 'label' (a string)." +
" No prose, no markdown, no explanation.";
const LABEL_FORMAT_RULE = "\n\nLABEL FORMAT RULE (mandatory):" +
"\n- Labels must be concise and specific: 2–4 words maximum." +
"\n- No commas or punctuation inside a label (no ',', '.', ';', ':', '/') — downstream tools split labels on commas." +
"\n- Examples: 'plugin list', 'plugin name', 'human face', 'left hand', 'red car', 'fluffy owl toy'";
function normalizeLmApiRoot(baseUrl) {
return String(baseUrl || "")
.trim()
.replace(/\/(api\/v1|v1)\/?$/i, "")
.replace(/\/+$/, "");
}
function authHeaders(apiKey, contentType = false) {
const headers = {};
if (contentType)
headers["Content-Type"] = "application/json";
if (apiKey?.trim())
headers.Authorization = `Bearer ${apiKey.trim()}`;
return headers;
}
function logVisionRequestMetadata(metadata) {
const line = `[LmStudioVisionAnalyzer] /v1/chat/completions request ${JSON.stringify(metadata)}`;
console.info(line);
try {
const logsDir = (0, core_bundle_mjs_1.getLogsDir)();
if (!fs_1.default.existsSync(logsDir))
fs_1.default.mkdirSync(logsDir, { recursive: true });
fs_1.default.appendFileSync(path_1.default.join(logsDir, (0, core_bundle_mjs_1.getPluginLogFilename)()), `${new Date().toISOString()} - ${line}\n`, "utf8");
}
catch { }
}
function hasLoadedInstances(modelInfo) {
return Array.isArray(modelInfo?.loaded_instances) && modelInfo.loaded_instances.length > 0;
}
async function getBionicVisionModelState(baseUrl, apiKey, modelKey) {
const apiRoot = normalizeLmApiRoot(baseUrl);
if (!apiRoot)
return { available: false, loaded: false };
const normalizedModelKey = modelKey.trim().toLowerCase();
const controller = new AbortController();
const timeout = setTimeout(() => controller.abort(), 5000);
try {
const response = await fetch(`${apiRoot}/api/v1/models`, {
headers: authHeaders(apiKey),
signal: controller.signal,
});
if (!response.ok)
return { available: false, loaded: false };
const data = await response.json();
const models = Array.isArray(data)
? data
: Array.isArray(data?.models)
? data.models
: Array.isArray(data?.data)
? data.data
: [];
const modelInfo = models.find((entry) => {
const key = String(entry?.key || entry?.id || "").trim().toLowerCase();
return key === normalizedModelKey;
});
if (!modelInfo)
return { available: false, loaded: false };
return {
available: true,
loaded: hasLoadedInstances(modelInfo),
modelKey: String(modelInfo?.key || modelInfo?.id || "").trim() || undefined,
};
}
catch {
return { available: false, loaded: false };
}
finally {
clearTimeout(timeout);
}
}
async function loadBionicVisionInstance(baseUrl, apiKey, modelKey) {
const apiRoot = normalizeLmApiRoot(baseUrl);
if (!apiRoot) {
return { ok: false, error: "Vision API base URL is empty." };
}
const controller = new AbortController();
const timeout = setTimeout(() => controller.abort(), 600_000);
try {
const response = await fetch(`${apiRoot}/api/v1/models/load`, {
method: "POST",
headers: authHeaders(apiKey, true),
body: JSON.stringify({
model: modelKey,
echo_load_config: true,
}),
signal: controller.signal,
});
const text = await response.text().catch(() => "");
let data = null;
if (text.trim()) {
try {
data = JSON.parse(text);
}
catch {
data = { raw: text };
}
}
const apiError = data?.error?.message || data?.error || data?.message;
if (!response.ok || apiError) {
const detail = apiError || text || `${response.status} ${response.statusText}`;
return {
ok: false,
error: `Vision API could not load '${modelKey}' via /api/v1/models/load. This can happen when there are not enough system resources available. Error: ${detail}`,
};
}
return { ok: true };
}
catch (error) {
const detail = error?.name === "AbortError"
? "request timed out after 600000 ms"
: error?.message || String(error);
return {
ok: false,
error: `Vision API could not load '${modelKey}' via /api/v1/models/load. This can happen when there are not enough system resources available. Error: ${detail}`,
};
}
finally {
clearTimeout(timeout);
}
}
/**
* force_cancel_active: true bypasses Unsloth Studio's refusal to load while another model still
* has an active generation in flight ("active_generations" error). alongside: true requests
* keeping other models loaded instead of evicting them -- confirmed NOT implemented in the
* installed 0.1.900-beta build (live-verified 2026-10-01: real curl + /v1/models + OS process
* list all show the previous model's llama-server killed regardless of this flag; the field is
* entirely absent from that build's LoadRequest schema, silently dropped). unslothai/unsloth#11591
* ("Studio: serve multiple models at once") adds real support for exactly this flag at load time
* ("send \"alongside\": true when loading a model") -- kept here deliberately, forward-compatible,
* for once that PR ships. Do NOT also send it in the chat request: that PR never puts it there.
*/
async function loadUnslothVisionInstance(baseUrl, apiKey, modelKey) {
const apiRoot = normalizeLmApiRoot(baseUrl);
if (!apiRoot) {
return { ok: false, error: "Vision API base URL is empty." };
}
const controller = new AbortController();
const timeout = setTimeout(() => controller.abort(), 600_000);
try {
const response = await fetch(`${apiRoot}/api/inference/load`, {
method: "POST",
headers: authHeaders(apiKey, true),
body: JSON.stringify({
model_path: modelKey,
gguf_variant: "Q4_K_M",
max_seq_length: 8192,
alongside: true,
force_cancel_active: true,
}),
signal: controller.signal,
});
const text = await response.text().catch(() => "");
let data = null;
if (text.trim()) {
try {
data = JSON.parse(text);
}
catch {
data = { raw: text };
}
}
const apiError = data?.error?.message || data?.error || data?.message;
if (!response.ok || apiError) {
const detail = apiError || text || `${response.status} ${response.statusText}`;
return {
ok: false,
error: `Vision API could not load '${modelKey}' via /api/inference/load. This can happen when there are not enough system resources available. Error: ${detail}`,
};
}
return { ok: true };
}
catch (error) {
const detail = error?.name === "AbortError"
? "request timed out after 600000 ms"
: error?.message || String(error);
return {
ok: false,
error: `Vision API could not load '${modelKey}' via /api/inference/load. This can happen when there are not enough system resources available. Error: ${detail}`,
};
}
finally {
clearTimeout(timeout);
}
}
/** Strand 1/3: LM Studio's own endpoints (also the default for `generic` server adapters that happen to be LM Studio). */
async function ensureBionicVisionModelReady(config) {
const modelKey = String(config.modelKey || "").trim();
if (!modelKey) {
return { ok: false, error: "Vision API mode is active, but Qwen3-VL model key is empty." };
}
const initialState = await getBionicVisionModelState(config.baseUrl, config.apiKey, modelKey);
if (!initialState.available) {
return { ok: false, error: `Vision API does not list '${modelKey}' as an available model via /api/v1/models.` };
}
if (initialState.loaded) {
return { ok: true, loaded: false };
}
try {
config.status?.(`Loading ${modelKey}...`);
}
catch { }
const loadResult = await loadBionicVisionInstance(config.baseUrl, config.apiKey, modelKey);
if (!loadResult.ok)
return loadResult;
const loadedState = await getBionicVisionModelState(config.baseUrl, config.apiKey, modelKey);
if (!loadedState.loaded) {
return {
ok: false,
error: `Vision API loaded '${modelKey}' via /api/v1/models/load, but /api/v1/models did not report it as loaded.`,
};
}
if (loadedState.modelKey?.trim().toLowerCase() !== modelKey.toLowerCase()) {
return {
ok: false,
error: `Vision API loaded a model, but /api/v1/models reports '${loadedState.modelKey || "unknown model"}' instead of '${modelKey}'.`,
};
}
return { ok: true, loaded: true };
}
/** Fast per-call status check (live-verified much faster than /v1/models on Unsloth Studio) -- only ever describes the single most-recently-active model, never a full multi-model registry. */
async function getUnslothVisionModelState(baseUrl, apiKey) {
const apiRoot = normalizeLmApiRoot(baseUrl);
if (!apiRoot)
return { loaded: false };
const controller = new AbortController();
const timeout = setTimeout(() => controller.abort(), 5000);
try {
const response = await fetch(`${apiRoot}/api/inference/status`, {
headers: authHeaders(apiKey),
signal: controller.signal,
});
if (!response.ok)
return { loaded: false };
const data = await response.json();
const activeModel = typeof data?.active_model === "string" ? data.active_model : undefined;
const loadedList = Array.isArray(data?.loaded) ? data.loaded : [];
return { loaded: loadedList.length > 0, activeModel };
}
catch {
return { loaded: false };
}
finally {
clearTimeout(timeout);
}
}
/**
* Strand 2/3: Unsloth Studio. Checks loaded state via GET /api/inference/status, NOT /v1/models --
* live-verified /v1/models is extremely slow on this server, /api/inference/status is fast.
* /api/inference/status only ever reports the single most-recently-active model (never a full
* registry), so this never checks availability up front -- a load failure doubles as the "not
* available" signal instead, same as bionic/generic do via their own availability check. Does NOT
* check whether any other ("agent") model stays loaded afterward -- confirmed impossible on this
* server (see unslothai/unsloth#11591 note on loadUnslothVisionInstance()).
*/
async function ensureUnslothVisionModelReady(config) {
const modelKey = String(config.modelKey || "").trim();
if (!modelKey) {
return { ok: false, error: "Vision API mode is active, but Qwen3-VL model key is empty." };
}
const targetBasename = visionModelBasename(modelKey);
const initialState = await getUnslothVisionModelState(config.baseUrl, config.apiKey);
if (initialState.loaded && initialState.activeModel && visionModelBasename(initialState.activeModel) === targetBasename) {
return { ok: true, loaded: false };
}
try {
config.status?.(`Loading ${modelKey}...`);
}
catch { }
const loadResult = await loadUnslothVisionInstance(config.baseUrl, config.apiKey, modelKey);
if (!loadResult.ok)
return loadResult;
const loadedState = await getUnslothVisionModelState(config.baseUrl, config.apiKey);
if (!loadedState.loaded || !loadedState.activeModel || visionModelBasename(loadedState.activeModel) !== targetBasename) {
return {
ok: false,
error: `Vision API loaded '${modelKey}' via /api/inference/load, but /api/inference/status does not report it as the active model.`,
};
}
return { ok: true, loaded: true };
}
/** Smallest common denominator across /v1/models id formats: only the final path segment, case-insensitively, since a server is free to report its own id shape for the same model. */
function visionModelBasename(key) {
const trimmed = key.trim();
const idx = trimmed.lastIndexOf("/");
return (idx >= 0 ? trimmed.slice(idx + 1) : trimmed).toLowerCase();
}
/** Only llama-server router mode's ids carry a trailing ":quant" tag (e.g. ":Q4_K_M") -- strip it before comparing against a bare hf-repo-style QWEN3_VL_MODEL value. */
function stripLlamaServerQuantTag(id) {
return id.replace(/:[^/:]+$/, "");
}
/**
* Resolves the sole *mmproj*.gguf file in the same directory as an absolute-path GGUF model file --
* required for Qwen3-VL's vision support when VISION_API=llama-server is pointed at a local file
* instead of an hf-repo id (see ensureLlamaServerVisionModelReady).
*/
function resolveLlamaServerMmprojPath(modelFilePath) {
const dir = path_1.default.dirname(modelFilePath);
let entries;
try {
entries = fs_1.default.readdirSync(dir);
}
catch {
return { ok: false, error: `Vision API (llama-server) could not read the model's directory: ${dir}` };
}
const candidates = entries.filter((e) => e.toLowerCase().endsWith(".gguf") && e.toLowerCase().includes("mmproj"));
if (candidates.length === 0) {
return { ok: false, error: `Vision API (llama-server) found no mmproj *.gguf file alongside '${modelFilePath}' in ${dir}.` };
}
if (candidates.length > 1) {
return { ok: false, error: `Vision API (llama-server) found multiple candidate mmproj files in ${dir}: ${candidates.join(", ")}. Keep exactly one.` };
}
return { ok: true, mmprojPath: path_1.default.join(dir, candidates[0]) };
}
/** Shared HTTP utility (not preset-branching logic). `loaded` is a real per-model flag on this endpoint -- LIVE-VERIFIED against both LM Studio-shaped and Unsloth Studio-shaped /v1/models responses. */
async function listOpenAiModels(baseUrl, apiKey) {
const apiRoot = normalizeLmApiRoot(baseUrl);
if (!apiRoot)
return null;
const controller = new AbortController();
const timeout = setTimeout(() => controller.abort(), 5000);
try {
const response = await fetch(`${apiRoot}/v1/models`, {
headers: authHeaders(apiKey),
signal: controller.signal,
});
if (!response.ok)
return null;
const data = await response.json();
const models = Array.isArray(data?.data) ? data.data : Array.isArray(data) ? data : [];
return models
.map((entry) => ({ id: String(entry?.id ?? entry?.key ?? "").trim(), loaded: Boolean(entry?.loaded) }))
.filter((m) => m.id);
}
catch {
return null;
}
finally {
clearTimeout(timeout);
}
}
/** Strand 3/3: plain OpenAI /v1/models listing, no explicit load step, since an available model is trusted to be usable on demand. */
async function ensureGenericVisionModelReady(config) {
const modelKey = String(config.modelKey || "").trim();
if (!modelKey) {
return { ok: false, error: "Vision API mode is active, but Qwen3-VL model key is empty." };
}
const models = await listOpenAiModels(config.baseUrl, config.apiKey);
if (models === null) {
return { ok: false, error: "Vision API could not be reached via /v1/models." };
}
const targetBasename = visionModelBasename(modelKey);
const isAvailable = models.some((m) => visionModelBasename(m.id) === targetBasename);
if (!isAvailable) {
return { ok: false, error: `Vision API does not list '${modelKey}' as an available model via /v1/models.` };
}
return { ok: true, loaded: false };
}
/**
* Strand 4/4: a standalone llama.cpp router-mode server we spawn and manage ourselves, entirely
* independent of any agent-model process (LM Studio Bionic, Unsloth Studio) -- live-verified
* 2026-10-01 that loading/using a vision model here never evicts an agent model loaded elsewhere.
* No explicit load call: router mode auto-loads the requested model on the first
* /v1/chat/completions request, same lazy "trust it's usable on demand" pattern as the generic
* adapter, after ensureLlamaServerRouterRunning() has made sure the router process itself exists.
* Listens on its own dedicated llamaServer.port -- NEVER derived from config.baseUrl, which stays
* the real Vision API's own fixed address (e.g. Unsloth Studio's 8888) and must not be hijacked as
* our router's bind address (would either collide with that real service or silently adopt it).
*/
async function ensureLlamaServerVisionModelReady(config) {
const modelKey = String(config.modelKey || "").trim();
if (!modelKey) {
return { ok: false, error: "Vision API mode is active, but Qwen3-VL model key is empty." };
}
const llamaServer = config.llamaServer;
if (!llamaServer) {
return { ok: false, error: "Vision API mode is 'llama-server', but no llamaServer config (port/binaryPath/ctxSize/modelsMax/idleTtlMinutes) was supplied." };
}
const host = "127.0.0.1";
const routerBaseUrl = `http://${host}:${llamaServer.port}/v1`;
// An absolute path to a .gguf file (as opposed to an hf-repo-style id like
// "unsloth/Qwen3-VL-8B-Instruct-GGUF") bypasses HF-cache auto-discovery entirely -- registered
// as a named custom preset section instead (see ensureLlamaServerRouterRunning's localModel).
let localModel;
if (path_1.default.isAbsolute(modelKey) && modelKey.toLowerCase().endsWith(".gguf")) {
if (!fs_1.default.existsSync(modelKey)) {
return { ok: false, error: `Vision API (llama-server) model file does not exist: ${modelKey}` };
}
const mmprojResult = resolveLlamaServerMmprojPath(modelKey);
if (!mmprojResult.ok) {
return { ok: false, error: mmprojResult.error };
}
localModel = { name: path_1.default.basename(modelKey, ".gguf"), modelPath: modelKey, mmprojPath: mmprojResult.mmprojPath };
}
try {
await (0, LlamaServerRouterManager_js_1.ensureLlamaServerRouterRunning)({
host,
port: llamaServer.port,
binaryPath: llamaServer.binaryPath,
ctxSize: llamaServer.ctxSize,
modelsMax: llamaServer.modelsMax,
idleTtlMinutes: llamaServer.idleTtlMinutes,
localModel,
});
}
catch (error) {
return { ok: false, error: error?.message || String(error) };
}
const models = await listOpenAiModels(routerBaseUrl, config.apiKey);
if (models === null) {
return { ok: false, error: `Vision API (llama-server router) could not be reached via ${routerBaseUrl}/models.` };
}
if (localModel) {
const matchedLocal = models.find((m) => m.id === localModel.name);
if (!matchedLocal) {
return { ok: false, error: `Vision API (llama-server router) does not list '${localModel.name}' as an available model via ${routerBaseUrl}/models.` };
}
return { ok: true, loaded: false, resolvedModelKey: matchedLocal.id, resolvedBaseUrl: routerBaseUrl };
}
// Router mode's own /v1/models ids are "hf-repo:quant" (e.g. "unsloth/Qwen3-VL-8B-Instruct-GGUF:Q4_K_M"),
// never just the bare hf-repo -- strip the trailing ":quant" tag on BOTH sides before the usual
// basename compare, scoped to this adapter only (other adapters' servers never emit this shape).
const targetBasename = visionModelBasename(stripLlamaServerQuantTag(modelKey));
const matched = models.find((m) => visionModelBasename(stripLlamaServerQuantTag(m.id)) === targetBasename);
if (!matched) {
return { ok: false, error: `Vision API (llama-server router) does not list '${modelKey}' as an available model via ${routerBaseUrl}/models.` };
}
// The real chat request must use the router's own exact tagged id and its own address, not the
// caller's originally-configured modelKey/baseUrl (see the ":quant" tag note above).
return { ok: true, loaded: false, resolvedModelKey: matched.id, resolvedBaseUrl: routerBaseUrl };
}
const VISION_API_ADAPTERS = {
bionic: ensureBionicVisionModelReady,
unsloth: ensureUnslothVisionModelReady,
generic: ensureGenericVisionModelReady,
"llama-server": ensureLlamaServerVisionModelReady,
};
/** Single dispatch point between the 4 independent adapters, add a new VISION_API value by adding one more entry here, never by branching inside an adapter. */
async function ensureVisionModelReady(visionApi, config) {
return VISION_API_ADAPTERS[visionApi](config);
}
function mimeFromPath(filePath) {
const ext = path_1.default.extname(filePath).toLowerCase();
if (ext === ".jpg" || ext === ".jpeg")
return "image/jpeg";
if (ext === ".webp")
return "image/webp";
if (ext === ".gif")
return "image/gif";
return "image/png";
}
function readUInt24LE(buffer, offset) {
return buffer[offset] | (buffer[offset + 1] << 8) | (buffer[offset + 2] << 16);
}
function readPngDimensions(buffer) {
if (buffer.length < 24)
return null;
if (buffer.toString("ascii", 1, 4) !== "PNG")
return null;
return {
width: buffer.readUInt32BE(16),
height: buffer.readUInt32BE(20),
};
}
function readGifDimensions(buffer) {
if (buffer.length < 10)
return null;
const signature = buffer.toString("ascii", 0, 6);
if (signature !== "GIF87a" && signature !== "GIF89a")
return null;
return {
width: buffer.readUInt16LE(6),
height: buffer.readUInt16LE(8),
};
}
function readWebpDimensions(buffer) {
if (buffer.length < 30)
return null;
if (buffer.toString("ascii", 0, 4) !== "RIFF" || buffer.toString("ascii", 8, 12) !== "WEBP") {
return null;
}
const chunkType = buffer.toString("ascii", 12, 16);
if (chunkType === "VP8X" && buffer.length >= 30) {
return {
width: readUInt24LE(buffer, 24) + 1,
height: readUInt24LE(buffer, 27) + 1,
};
}
if (chunkType === "VP8L" && buffer.length >= 25 && buffer[20] === 0x2f) {
const bits = buffer.readUInt32LE(21);
return {
width: (bits & 0x3fff) + 1,
height: ((bits >> 14) & 0x3fff) + 1,
};
}
if (chunkType === "VP8 " && buffer.length >= 30) {
return {
width: buffer.readUInt16LE(26) & 0x3fff,
height: buffer.readUInt16LE(28) & 0x3fff,
};
}
return null;
}
function readJpegDimensions(buffer) {
if (buffer.length < 4 || buffer[0] !== 0xff || buffer[1] !== 0xd8)
return null;
let offset = 2;
while (offset + 9 < buffer.length) {
if (buffer[offset] !== 0xff) {
offset += 1;
continue;
}
while (offset < buffer.length && buffer[offset] === 0xff)
offset += 1;
const marker = buffer[offset];
offset += 1;
if (marker === 0xd9 || marker === 0xda)
break;
if (offset + 2 > buffer.length)
break;
const segmentLength = buffer.readUInt16BE(offset);
if (segmentLength < 2 || offset + segmentLength > buffer.length)
break;
const isStartOfFrame = (marker >= 0xc0 && marker <= 0xc3) ||
(marker >= 0xc5 && marker <= 0xc7) ||
(marker >= 0xc9 && marker <= 0xcb) ||
(marker >= 0xcd && marker <= 0xcf);
if (isStartOfFrame && segmentLength >= 7) {
return {
height: buffer.readUInt16BE(offset + 3),
width: buffer.readUInt16BE(offset + 5),
};
}
offset += segmentLength;
}
return null;
}
async function readImageDimensions(filePath) {
const buffer = await fs_1.default.promises.readFile(filePath);
const dimensions = readPngDimensions(buffer) ||
readJpegDimensions(buffer) ||
readWebpDimensions(buffer) ||
readGifDimensions(buffer);
if (!dimensions || dimensions.width <= 0 || dimensions.height <= 0) {
throw new Error(`Could not determine image dimensions for ${filePath}`);
}
return dimensions;
}
function extractMessageText(data) {
const pieces = [];
// Standard OpenAI /v1/chat/completions shape.
const choices = Array.isArray(data?.choices) ? data.choices : [];
for (const choice of choices) {
const content = choice?.message?.content;
if (typeof content === "string") {
pieces.push(content);
}
else if (Array.isArray(content)) {
for (const part of content) {
if (typeof part === "string")
pieces.push(part);
else if (typeof part?.text === "string")
pieces.push(part.text);
}
}
}
// LM Studio's proprietary /api/v1/chat "responses"-style shape.
const output = Array.isArray(data?.output) ? data.output : [];
for (const item of output) {
if (item?.type !== "message")
continue;
const content = item?.content;
if (typeof content === "string") {
pieces.push(content);
}
else if (Array.isArray(content)) {
for (const part of content) {
if (typeof part === "string") {
pieces.push(part);
}
else if (typeof part?.text === "string") {
pieces.push(part.text);
}
else if (typeof part?.content === "string") {
pieces.push(part.content);
}
}
}
}
if (pieces.length === 0 && typeof data?.text === "string") {
pieces.push(data.text);
}
if (pieces.length === 0 && typeof data?.content === "string") {
pieces.push(data.content);
}
return pieces.join("\n").trim();
}
function buildDetectPrompt(task, odPrompt) {
const label = String(task || "").trim();
if (label) {
return `Detect all instances of '${label}' in the image.` + LABEL_FORMAT_RULE + JSON_FORMAT;
}
const instruction = String(odPrompt || "").trim();
if (!instruction) {
throw new Error("No OD prompt available: odPrompt not set and DETECT_OD_PROMPT env var not set");
}
return instruction + LABEL_FORMAT_RULE + JSON_FORMAT;
}
function bboxToCrop(bbox, width, height) {
const [x1, y1, x2, y2] = bbox;
return {
cropLeft: (x1 / width) * 100,
cropRight: ((width - x2) / width) * 100,
cropTop: (y1 / height) * 100,
cropBottom: ((height - y2) / height) * 100,
};
}
function parseQwen3VlDetectionOutput(text, width, height) {
const objects = [];
const seen = new Set();
const fenceMatch = JSON_FENCE_RE.exec(text);
const jsonText = fenceMatch ? fenceMatch[1].trim() : text.trim();
let items = null;
try {
const parsed = JSON.parse(jsonText);
items = Array.isArray(parsed) ? parsed : [parsed];
}
catch {
const recovered = [];
ITEM_RE.lastIndex = 0;
for (const match of text.matchAll(ITEM_RE)) {
recovered.push({
bbox_2d: [Number(match[1]), Number(match[2]), Number(match[3]), Number(match[4])],
label: match[5],
});
}
items = recovered.length > 0 ? recovered : [];
}
for (const item of items) {
if (!item || typeof item !== "object")
continue;
const bbox = item.bbox_2d;
const label = String(item.label || "");
if (!Array.isArray(bbox) || bbox.length !== 4)
continue;
const [nx1, ny1, nx2, ny2] = bbox.map((value) => Number(value));
if (![nx1, ny1, nx2, ny2].every((value) => Number.isFinite(value) && value >= 0 && value <= 1000)) {
continue;
}
if (nx2 <= nx1 || ny2 <= ny1)
continue;
if (nx1 < 10 && ny1 < 10 && nx2 > 990 && ny2 > 990)
continue;
const dedupKey = `${Math.round(nx1)}:${Math.round(ny1)}:${Math.round(nx2)}:${Math.round(ny2)}:${label}`;
if (seen.has(dedupKey))
continue;
seen.add(dedupKey);
const pixelBbox = [
(nx1 / 1000) * width,
(ny1 / 1000) * height,
(nx2 / 1000) * width,
(ny2 / 1000) * height,
];
objects.push({
label,
bbox: pixelBbox,
...bboxToCrop(pixelBbox, width, height),
});
}
return objects;
}
async function chatOnce(item, prompt, config) {
const apiRoot = normalizeLmApiRoot(config.baseUrl);
if (!apiRoot) {
throw new Error("Vision API base URL is empty");
}
const endpoint = `${apiRoot}/v1/chat/completions`;
const timeoutMs = config.timeoutMs ?? 180_000;
const model = config.model || "vision-capability-priming";
const buf = await fs_1.default.promises.readFile(item.filePath);
const dataUrl = `data:${mimeFromPath(item.filePath)};base64,${buf.toString("base64")}`;
const payload = {
model,
seed: 175308301,
messages: [
{
role: "user",
content: [
{ type: "text", text: prompt },
{ type: "image_url", image_url: { url: dataUrl } },
],
},
],
};
if (typeof config.maxTokens === "number" && Number.isFinite(config.maxTokens) && config.maxTokens > 0) {
payload.max_tokens = Math.floor(config.maxTokens);
}
if (typeof config.temperature === "number" && Number.isFinite(config.temperature)) {
payload.temperature = config.temperature;
}
logVisionRequestMetadata({
configuredBaseUrl: config.baseUrl,
apiRoot,
endpoint,
model,
store: payload.store ?? null,
max_output_tokens: payload.max_output_tokens ?? payload.max_tokens ?? null,
temperature: payload.temperature ?? null,
promptChars: prompt.length,
imageBytes: buf.byteLength,
payloadKeys: Object.keys(payload),
});
const headers = authHeaders(config.apiKey, true);
const controller = new AbortController();
const timeout = setTimeout(() => controller.abort(), timeoutMs);
const startedAt = Date.now();
let data;
try {
const resp = await fetch(endpoint, {
method: "POST",
headers,
body: JSON.stringify(payload),
signal: controller.signal,
});
clearTimeout(timeout);
if (!resp.ok) {
const detail = await resp.text().catch(() => "(no body)");
throw new Error(`Vision API ${resp.status}: ${detail}`);
}
data = await resp.json();
}
catch (error) {
clearTimeout(timeout);
if (error?.name === "AbortError") {
throw new Error(`Vision API timed out after ${timeoutMs}ms`);
}
throw new Error(`Vision API failed: ${error?.message || String(error)}`);
}
return {
text: extractMessageText(data),
elapsedMs: Date.now() - startedAt,
bytes: buf.byteLength,
modelInstanceId: typeof data?.model_instance_id === "string" ? data.model_instance_id : "",
};
}
async function analyzeLmStudioVisionBatch(items, config) {
if (!items.length) {
return {
results: [],
totalInferenceTimeMs: 0,
backend: "vision-api",
};
}
const results = [];
let totalInferenceTimeMs = 0;
for (const item of items) {
console.info(`[VisionAnalyzer] /v1/chat/completions start mode=analyze id=${item.id} timeoutMs=${config.timeoutMs ?? 180_000}`);
const response = await chatOnce(item, config.prompt || "Describe the image.", config);
console.info(`[VisionAnalyzer] /v1/chat/completions ok mode=analyze id=${item.id} bytes=${response.bytes} elapsedMs=${response.elapsedMs} modelInstance=${response.modelInstanceId || "?"}`);
results.push({
id: item.id,
text: response.text,
inferenceTimeMs: response.elapsedMs,
});
totalInferenceTimeMs += response.elapsedMs;
}
return {
results,
totalInferenceTimeMs,
backend: "vision-api",
};
}
async function detectLmStudioVisionBatch(items, config) {
if (!items.length) {
return {
results: [],
totalInferenceTimeMs: 0,
backend: "vision-api",
};
}
const prompt = buildDetectPrompt(config.task, config.odPrompt);
const results = [];
let totalInferenceTimeMs = 0;
for (const item of items) {
const { width, height } = await readImageDimensions(item.filePath);
console.info(`[VisionAnalyzer] /v1/chat/completions start mode=detect id=${item.id} timeoutMs=${config.timeoutMs ?? 120_000}`);
const response = await chatOnce(item, prompt, {
baseUrl: config.baseUrl,
apiKey: config.apiKey,
model: config.model || "vision-capability-priming",
maxTokens: config.maxTokens,
temperature: config.temperature,
timeoutMs: config.timeoutMs ?? 120_000,
});
const objects = parseQwen3VlDetectionOutput(response.text, width, height);
console.info(`[VisionAnalyzer] /v1/chat/completions ok mode=detect id=${item.id} objects=${objects.length} bytes=${response.bytes} elapsedMs=${response.elapsedMs} modelInstance=${response.modelInstanceId || "?"}`);
results.push({
id: item.id,
objects,
imageWidth: width,
imageHeight: height,
inferenceTimeMs: response.elapsedMs,
});
totalInferenceTimeMs += response.elapsedMs;
}
return {
results,
totalInferenceTimeMs,
backend: "vision-api",
};
}