src / core / tools.ts
src / core / tools.ts
import path from "path";
import fs from "fs";
import { createHash } from "crypto";
import { pathToFileURL } from "url";
import { z } from "zod";
import type { ToolsProviderController } from "@lmstudio/sdk";
import {
formatToolMetaBlock,
readCameraImageMetadata,
syncAttachmentsToState,
getActiveChatContext,
resolveActiveLMStudioChatId,
getLMStudioWorkingDir,
readState,
writeStateAtomic,
generatePreviewFromBuffer,
appendImages,
getHealthyServerBaseUrl,
toHttpOriginalUrl,
toHttpPreviewUrl,
buildAuditLogger,
getSelfPluginIdentifier,
VARIANT_FULL_CONFIG,
drawthingsLimits,
getSize,
encodeJpegPreviewFromBuffer,
} from "../core-bundle.mjs";
import { type ChatMediaState } from "../state.js";
import { readPngGenerationMeta, formatGenerationMeta, type PngGenerationMeta } from "../helpers/readPngMetadata.js";
import { reportToolStatus, reportToolStep, type ToolStatusContext } from "../helpers/toolProgress.js";
import { defaultPluginSettings, globalConfigSchematics } from "../config.js";
import { drawBboxesOnImage } from "../helpers/drawBboxesOnImage.js";
import { injectXmpIntoBuffer, type PngAnalysisMetadata } from "../helpers/pngMetadata.js";
import {
analyzeLmStudioVisionBatch,
ensureVisionModelReady,
detectLmStudioVisionBatch,
type LmStudioVisionAnalyzerConfig,
type VisionAnalysisItem,
type VisionDetectionAnalyzerConfig,
type VisionDetectionBatchResult,
type VisionApiName,
} from "../services/VisionAnalyzer.js";
/** The LM-Studio-plugin entrypoint has no preset concept and only ever talks to LM Studio itself, defaults to "bionic" when VISION_API isn't set (MCP's applyMcpConfig() is the only writer of this env var). */
function resolveVisionApiFromEnv(): VisionApiName {
const raw = (process.env.LMSTUDIO_VISION_API || "").trim().toLowerCase();
return raw === "generic" || raw === "unsloth" || raw === "llama-server" ? raw : "bionic";
}
// ─────────────────────────────────────────────────────────────────────────────
// Shared helpers (previously duplicated across src/tools/*.ts, verified identical)
// ─────────────────────────────────────────────────────────────────────────────
export function parsePrefixedNotation(
s: string
): { pool: "attachment" | "image" | "variant" | "picture"; index: number } | null {
const t = String(s || "").trim().toLowerCase();
const m = t.match(/^([avip])(\d+)$/);
if (!m) return null;
const idx = Math.max(1, parseInt(m[2], 10));
const pool =
m[1] === "a" ? "attachment" : m[1] === "v" ? "variant" : m[1] === "i" ? "image" : "picture";
return { pool, index: idx };
}
// A target's `id` is normally a short notation (e.g. "i1") but the MCP adapter may also pass a
// resolved absolute path (see made-for-bionic-core's resolveMcpSourcePathTarget, called from
// mcp/index.ts before this handler runs) — never embed it in a filename unsanitized.
function safeIdForFilename(id: string): string {
return path.basename(id).replace(/[^a-zA-Z0-9._-]/g, "_").slice(0, 80);
}
// A target that isn't aN/vN/iN/pN notation is a scratchpad basename or an absolute path (see
// generate-image's core/tools.ts for the same pattern) — deliberately local/non-shared. The MCP
// layer (src/mcp/index.ts) already resolves and containment-checks these against the bound
// scratchpad before this handler ever runs (made-for-bionic-core's resolveMcpSourceToken()) — by
// the time a target reaches here it is either aN/vN/iN/pN notation or an already-validated
// absolute path, so this is a plain existence check, not a security boundary.
async function resolveTargetPathToken(raw: string, baseDir: string | undefined): Promise<string | null> {
const trimmed = raw.trim();
const candidate = path.isAbsolute(trimmed)
? trimmed
: baseDir && trimmed === path.basename(trimmed)
? path.join(baseDir, trimmed)
: null;
if (!candidate) return null;
const exists = await fs.promises.stat(candidate).then((s) => s.isFile()).catch(() => false);
return exists ? candidate : null;
}
// Normalizes a path-specified source buffer for the vision model, mirroring find-images'
// previewPayloadBytes policy: images already within drawthingsLimits.previewMaxSum (w + h) are
// returned untouched (no resize, no format change — a fitting PNG stays a PNG); only oversize
// originals are re-encoded to a JPEG preview capped at previewMaxSum.
async function normalizeVisionBuffer(buf: Buffer): Promise<Buffer> {
const maxSum = drawthingsLimits.previewMaxSum;
const { width, height } = await getSize(buf);
if (width <= 0 || height <= 0 || width + height <= maxSum) return buf;
const { data } = await encodeJpegPreviewFromBuffer(buf, {
maxDim: maxSum,
quality: drawthingsLimits.previewQuality,
mode: "sum",
maxSum,
});
return data;
}
function isoStampCompact(): string {
const d = new Date();
const year = d.getUTCFullYear();
const month = String(d.getUTCMonth() + 1).padStart(2, "0");
const day = String(d.getUTCDate()).padStart(2, "0");
const hours = String(d.getUTCHours()).padStart(2, "0");
const minutes = String(d.getUTCMinutes()).padStart(2, "0");
const seconds = String(d.getUTCSeconds()).padStart(2, "0");
const millis = String(d.getUTCMilliseconds()).padStart(3, "0");
return `${year}${month}${day}T${hours}${minutes}${seconds}${millis}Z`;
}
function getGlobalConfig(ctl: ToolsProviderController): any | null {
const ctlAny = ctl as any;
const getter = ctlAny.getGlobalPluginConfig || ctlAny.getGlobalConfig;
if (!getter) return null;
try {
return getter.call(ctl, globalConfigSchematics);
} catch {
return null;
}
}
function getGlobalString(gcfg: any | null, key: string, fallback: string): string {
try {
const value = gcfg?.get(key);
return typeof value === "string" ? value : fallback;
} catch {
return fallback;
}
}
function getGlobalNumber(gcfg: any | null, key: string, fallback: number): number {
try {
const value = gcfg?.get(key);
return typeof value === "number" && Number.isFinite(value) ? value : fallback;
} catch {
return fallback;
}
}
function getGlobalBoolean(gcfg: any | null, key: string, fallback: boolean): boolean {
try {
const value = gcfg?.get(key);
return typeof value === "boolean" ? value : fallback;
} catch {
return fallback;
}
}
/** Shared by all 3 ensureVisionModelReady() call sites -- only consumed when VISION_API="llama-server" (see ensureLlamaServerVisionModelReady). GUI setting > MCP env var > defaultPluginSettings, matching every other config field in this file. */
function resolveLlamaServerConfig(globalConfig: any | null): { port: number; binaryPath: string; ctxSize: number; modelsMax: number; idleTtlMinutes: number } {
const envPort = Number.parseInt(process.env.LMSTUDIO_LLAMA_SERVER_PORT || "", 10);
const envCtxSize = Number.parseInt(process.env.LMSTUDIO_LLAMA_SERVER_CTX_SIZE || "", 10);
const envModelsMax = Number.parseInt(process.env.LMSTUDIO_LLAMA_SERVER_MODELS_MAX || "", 10);
const envIdleTtlMinutes = Number.parseInt(process.env.LMSTUDIO_LLAMA_SERVER_IDLE_TTL_MINUTES || "", 10);
return {
port: Math.floor(getGlobalNumber(
globalConfig,
"llamaServerPort",
Number.isFinite(envPort) && envPort > 0 ? envPort : defaultPluginSettings.llamaServerPort
)),
binaryPath: getGlobalString(globalConfig, "llamaServerBinary", process.env.LMSTUDIO_LLAMA_SERVER_BINARY || defaultPluginSettings.llamaServerBinary),
ctxSize: Math.floor(getGlobalNumber(
globalConfig,
"llamaServerCtxSize",
Number.isFinite(envCtxSize) && envCtxSize > 0 ? envCtxSize : defaultPluginSettings.llamaServerCtxSize
)),
modelsMax: Math.floor(getGlobalNumber(
globalConfig,
"llamaServerModelsMax",
Number.isFinite(envModelsMax) && envModelsMax > 0 ? envModelsMax : defaultPluginSettings.llamaServerModelsMax
)),
idleTtlMinutes: Math.floor(getGlobalNumber(
globalConfig,
"llamaServerIdleTtlMinutes",
Number.isFinite(envIdleTtlMinutes) && envIdleTtlMinutes >= 0 ? envIdleTtlMinutes : defaultPluginSettings.llamaServerIdleTtlMinutes
)),
};
}
/**
* Flexible targets schema: accepts array OR comma/space-separated string.
* Reduces parse errors when models output "a1, v2" instead of ["a1", "v2"].
*/
export const FlexibleTargetsList = z
.union([
z.string().transform((s) => (s.match(/[aivp]\d+/gi) ?? []).map((x) => x.toLowerCase())),
z.array(z.string()),
])
.refine((arr) => arr.length >= 1, "targets must contain at least one notation")
.refine((arr) => arr.length <= 16, "targets must contain at most 16 notations");
// ─────────────────────────────────────────────────────────────────────────────
// analyse_image
// ─────────────────────────────────────────────────────────────────────────────
export const AnalyseImageParamsShape = {
targets: FlexibleTargetsList.describe(
"One or more image notations to process. Each notation is a letter followed by a number: " +
"a=attachment (a1, a2, …), i=generated image (i1, i2, …), v=variant (v1, v2, …), p=picture (p1, p2, …). " +
"Also accepts a filename in the scratchpad folder or an absolute path. Pass as a JSON array, e.g. [\"a1\", \"i3\"]."
),
prompt: z.string().optional().describe("Optional prompt for the vision model. Empty = model default."),
} satisfies Record<string, z.ZodTypeAny>;
type ParsedTargets = { a: number[]; v: number[]; i: number[]; p: number[] };
function parseTargets(targets: string[]): {
parsed: ParsedTargets;
invalid: string[];
} {
const parsed: ParsedTargets = { a: [], v: [], i: [], p: [] };
const invalid: string[] = [];
for (const raw of targets) {
const s = typeof raw === "string" ? raw.trim() : "";
const m = /^([avip])(\d+)$/i.exec(s);
if (!m) {
invalid.push(String(raw));
continue;
}
const kind = m[1].toLowerCase() as "a" | "v" | "i" | "p";
const n = parseInt(m[2], 10);
if (!Number.isFinite(n) || n <= 0) {
invalid.push(String(raw));
continue;
}
parsed[kind].push(n);
}
// Unique + sort for stable behavior
(Object.keys(parsed) as Array<keyof ParsedTargets>).forEach((k) => {
parsed[k] = Array.from(new Set(parsed[k])).sort((a, b) => a - b);
});
return { parsed, invalid };
}
function getAvailable(state: ChatMediaState) {
const availableA = (state.attachments || [])
.map((x: any) => (typeof x?.a === "number" ? x.a : undefined))
.filter((x: any) => typeof x === "number" && x > 0)
.sort((a: number, b: number) => a - b);
const availableV = (state.variants || [])
.map((x: any) => (typeof x?.v === "number" ? x.v : undefined))
.filter((x: any) => typeof x === "number" && x > 0)
.sort((a: number, b: number) => a - b);
const availableI = (state.images || [])
.map((x: any) => (typeof x?.i === "number" ? x.i : undefined))
.filter((x: any) => typeof x === "number" && x > 0)
.sort((a: number, b: number) => a - b);
const availableP = (state.pictures || [])
.map((x: any) => (typeof x?.p === "number" ? x.p : undefined))
.filter((x: any) => typeof x === "number" && x > 0)
.sort((a: number, b: number) => a - b);
return { availableA, availableV, availableI, availableP };
}
async function ensurePreviewExists(chatWd: string, previewRel: string): Promise<void> {
const pAbs = path.join(chatWd, previewRel);
await fs.promises.access(pAbs, fs.constants.F_OK);
}
function classifyVisionError(errMsg: string): string {
if (/\b503\b/.test(errMsg)) {
return "The Vision API is reachable, but the configured vision model is not available for inference. Check the configured vision model key and loaded model state.";
}
if (/aborted|aborterror|timed out|timeout/i.test(errMsg)) {
return "The Vision API request timed out before the model returned.";
}
if (/ECONNREFUSED|ENOTFOUND|ECONNRESET|network socket|fetch failed/i.test(errMsg)) {
return "The Vision API is not reachable. Check the configured vision/embedding API base URL and try again.";
}
return /Vision API/i.test(errMsg) ? errMsg : `Vision API error: ${errMsg}`;
}
function analyzeTimeoutMs(itemCount: number): number {
return Math.min(600_000, Math.max(180_000, itemCount * 60_000));
}
/** One analysed item's structured content — same data as its plain-text block in the returned string, for callers (the MCP HTML report) that want per-item rendering instead of re-parsing the text. */
export type AnalyseImageItemReport = {
id: string;
filePath?: string;
displayName?: string;
visual?: string;
/** Present when origPath is a PNG with an embedded generation-metadata chunk — lets callers (the MCP HTML report) render fields individually instead of parsing formatGenerationMeta()'s text. */
meta?: PngGenerationMeta;
/** Human-readable fallback note when `meta` isn't available (EXIF JSON, "no embedded metadata", etc.) — same text as the corresponding line in the plain-text result. */
metaNote?: string;
};
export async function handleAnalyseImage(
args: any,
ctl: ToolsProviderController,
statusCtx: ToolStatusContext = {},
requestId?: string,
onItem?: (item: AnalyseImageItemReport) => void
): Promise<string> {
const visionTmpPaths: string[] = [];
const visionTmpIds = new Set<string>();
try {
// Tolerant mode: drop invalid items instead of failing
const strict = false;
let targets: string[];
if (Array.isArray(args?.targets)) {
targets = args.targets;
} else if (typeof args?.targets === "string" && args.targets.trim()) {
const trimmed = args.targets.trim();
if (trimmed.startsWith("[")) {
try {
const parsed = JSON.parse(trimmed);
targets = Array.isArray(parsed) ? parsed.map((s: any) => String(s).trim()).filter(Boolean) : [];
} catch {
targets = trimmed.split(/[\s,]+/).map((s: string) => s.trim()).filter(Boolean);
}
} else {
// A bare notation or comma/space-separated list (e.g. "i1" or "i1, i2") — the same
// fallback detect_object/annotate_image's own string parsing already has. The MCP path
// never re-runs this handler's args through FlexibleTargetsList's zod transform (only the
// LM-Studio plugin path does, via @lmstudio/sdk's tool() wrapper), so this handler's own
// string parsing must accept the same shapes on its own.
targets = trimmed.split(/[\s,]+/).map((s: string) => s.trim()).filter(Boolean);
}
} else {
targets = [];
}
const prompt = typeof args?.prompt === "string" ? args.prompt : "";
const globalConfig = getGlobalConfig(ctl);
const configuredVisionPrompt = getGlobalString(
globalConfig,
"visionPrompt",
process.env.VISION_PROMPT || defaultPluginSettings.visionPrompt
);
const effectivePrompt = (prompt || configuredVisionPrompt || "").trim();
// Parse explicit notations (aN/vN/iN/pN)
const { parsed, invalid: invalidRaw } = parseTargets(targets);
// LM-Studio plugin path: ctl.getWorkingDirectory() (LM-Studio SDK method, only present when
// ctl is a real ToolsProviderController). MCP path: ctl is `{}` and has no such method — fall
// back to the same chat-context-bridge resolution detect_object/annotate_image already use
// (bindChatContextToScratchpad() sets this before the MCP adapter calls this handler).
// currentLmChatId is resolved unconditionally (also used for this call's audit log entry).
let currentLmChatId: string | null = null;
let boundWorkingDir: string | null = null;
try {
const chatCtx = await getActiveChatContext();
if ((chatCtx as any)?.chatId) currentLmChatId = (chatCtx as any).chatId;
if ((chatCtx as any)?.workingDir) boundWorkingDir = (chatCtx as any).workingDir;
} catch {}
if (!currentLmChatId) {
try {
const resolved = await resolveActiveLMStudioChatId();
if ((resolved as any)?.ok) currentLmChatId = (resolved as any).chatId;
} catch {}
}
let workingDir: string | undefined;
try {
workingDir = (ctl as any)?.getWorkingDirectory?.();
} catch {
workingDir = undefined;
}
if (typeof workingDir !== "string" || !workingDir.trim()) {
workingDir = boundWorkingDir || (currentLmChatId ? getLMStudioWorkingDir(currentLmChatId) : undefined);
}
if (typeof workingDir !== "string" || !workingDir.trim()) {
return "analyse_image failed: working directory not available.";
}
const chatWd = workingDir;
// Sync attachments from conversation.json → chat_media_state.json.
// maxPreviewAttachments=MAX: importAttachmentBatch generates previews for all attachments
// and writes them to state atomically — the canonical path via draw-things-chat-core.
try {
await syncAttachmentsToState(chatWd, false, Number.MAX_SAFE_INTEGER);
} catch (syncErr: any) {
console.warn("[analyse_image] attachment sync failed (non-fatal):", syncErr?.message ?? syncErr);
}
const state = await readState(chatWd);
const { availableA, availableV, availableI, availableP } = getAvailable(state);
// If any invalid notations and strict=true, reject
if (invalidRaw.length > 0 && strict) {
return `analyse_image failed: invalid targets: ${invalidRaw
.map((s) => JSON.stringify(s))
.join(", ")}`;
}
// Collect analysis items (id + preview file path)
const analysisItems: VisionAnalysisItem[] = [];
// Map from notation id → absolute path of original file (for XMP reading)
const originalFilePaths = new Map<string, string>();
// Map from notation id → human-readable filename for the result
const displayNames = new Map<string, string>();
const missingNotations = new Set<string>();
const missingDetails: string[] = [];
// Helper: resolve & validate preview for a notation
const addItem = async (
notation: string,
rec: any,
previewField: string,
originalAbsPath?: string,
displayName?: string
): Promise<boolean> => {
if (!rec) {
missingNotations.add(notation);
missingDetails.push(notation);
return false;
}
const previewRel =
typeof rec[previewField] === "string" ? String(rec[previewField]) : "";
if (!previewRel.trim()) {
missingNotations.add(notation);
missingDetails.push(`${notation} (missing preview)`);
return false;
}
try {
await ensurePreviewExists(chatWd, previewRel);
analysisItems.push({
id: notation,
filePath: path.join(chatWd, previewRel),
});
if (originalAbsPath) {
originalFilePaths.set(notation, originalAbsPath);
}
const dn = displayName || (originalAbsPath ? path.basename(originalAbsPath) : undefined);
if (dn) {
displayNames.set(notation, dn);
}
return true;
} catch {
missingNotations.add(notation);
missingDetails.push(`${notation} (preview file missing)`);
return false;
}
};
// Process attachments (aN)
// Preview generation + state write already handled by syncAttachmentsToState above.
for (const n of parsed.a) {
const rec = (state.attachments || []).find((x: any) => x?.a === n);
const origAbs: string | undefined =
(rec?.originAbs as string | undefined) ?? (rec?.filename ? path.join(chatWd, rec.filename as string) : undefined);
// originalName holds the real user-visible filename (e.g. "Katze.png")
const origName: string | undefined =
typeof rec?.originalName === "string" && rec.originalName ? rec.originalName as string : undefined;
await addItem(`a${n}`, rec, "preview", origAbs, origName);
}
// Process variants (vN)
for (const n of parsed.v) {
const rec = (state.variants || []).find((x: any) => x?.v === n);
const origAbs: string | undefined = rec?.filename ? path.join(chatWd, rec.filename as string) : undefined;
await addItem(`v${n}`, rec, "preview", origAbs);
}
// Process images (iN)
for (const n of parsed.i) {
const rec = (state.images || []).find((x: any) => x?.i === n);
const origAbs: string | undefined = rec?.filename ? path.join(chatWd, rec.filename as string) : undefined;
await addItem(`i${n}`, rec, "preview", origAbs);
}
// Process pictures (pN)
for (const n of parsed.p) {
const rec = (state.pictures || []).find((x: any) => x?.p === n);
const origAbs: string | undefined = rec?.filename ? path.join(chatWd, rec.filename as string) : undefined;
await addItem(`p${n}`, rec, "preview", origAbs);
}
// Entries that didn't match aN/vN/iN/pN — a scratchpad filename or an absolute path is
// also accepted as a target, same as detect_object/annotate_image/every other Bionic MCP tool.
for (const raw of invalidRaw) {
const pathToken = await resolveTargetPathToken(raw, chatWd);
if (pathToken) {
let visionPath = pathToken;
if (effectivePrompt) {
const buf = await fs.promises.readFile(pathToken);
const norm = await normalizeVisionBuffer(buf);
if (norm !== buf) {
const tmpPath = path.join(chatWd, `_tmp_analyse_src_${safeIdForFilename(raw)}_${Date.now()}.png`);
await fs.promises.writeFile(tmpPath, norm);
visionTmpPaths.push(tmpPath);
visionTmpIds.add(raw);
visionPath = tmpPath;
}
}
analysisItems.push({ id: raw, filePath: visionPath });
originalFilePaths.set(raw, pathToken);
displayNames.set(raw, path.basename(pathToken));
}
}
// If strict, fail on any missing items
if (missingDetails.length > 0 && strict) {
const hint =
`Available: ` +
`a=[${availableA.map((x) => `a${x}`).join(", ") || "(none)"}] ` +
`v=[${availableV.map((x) => `v${x}`).join(", ") || "(none)"}] ` +
`i=[${availableI.map((x) => `i${x}`).join(", ") || "(none)"}] ` +
`p=[${availableP.map((x) => `p${x}`).join(", ") || "(none)"}]`;
return `analyse_image failed: unknown/invalid targets: ${missingDetails.join(", ")}. ${hint}`;
}
// If no valid items after filtering, return early
if (analysisItems.length === 0) {
const hint =
`Available: ` +
`a=[${availableA.map((x) => `a${x}`).join(", ") || "(none)"}] ` +
`v=[${availableV.map((x) => `v${x}`).join(", ") || "(none)"}] ` +
`i=[${availableI.map((x) => `i${x}`).join(", ") || "(none)"}] ` +
`p=[${availableP.map((x) => `p${x}`).join(", ") || "(none)"}]`;
return `analyse_image: no valid targets found. ${hint}`;
}
let visionError: string | null = null;
const visionResults = new Map<string, string>();
let totalInferenceTimeMs: number | null = null;
if (effectivePrompt) {
const envServerMaxTokens = Number.parseInt(process.env.SERVER_MAX_TOKENS || "", 10);
const envServerTemperature = Number.parseFloat(process.env.SERVER_TEMPERATURE || "");
const configuredMaxTokens = Math.floor(getGlobalNumber(
globalConfig,
"serverMaxTokens",
Number.isFinite(envServerMaxTokens) && envServerMaxTokens > 0 ? envServerMaxTokens : defaultPluginSettings.serverMaxTokens
));
const configuredTemperature = getGlobalNumber(
globalConfig,
"serverTemperature",
Number.isFinite(envServerTemperature) ? envServerTemperature : defaultPluginSettings.serverTemperature
);
const lmStudioConfig: LmStudioVisionAnalyzerConfig = {
baseUrl: getGlobalString(globalConfig, "embeddingBaseUrl", process.env.LMSTUDIO_VISION_API_BASE_URL || defaultPluginSettings.embeddingBaseUrl),
apiKey: getGlobalString(globalConfig, "embeddingApiKey", process.env.LMSTUDIO_VISION_API_KEY || defaultPluginSettings.embeddingApiKey),
model: getGlobalString(globalConfig, "qwen3VlModelPath", process.env.LMSTUDIO_VISION_MODEL_KEY || defaultPluginSettings.qwen3VlModelPath),
prompt: effectivePrompt,
maxTokens: configuredMaxTokens,
temperature: configuredTemperature,
timeoutMs: analyzeTimeoutMs(1),
};
try {
const totalSteps = analysisItems.length + 2;
reportToolStatus(statusCtx, `Analyzing ${analysisItems.length} image${analysisItems.length === 1 ? "" : "s"}...`);
reportToolStep(statusCtx, 1, totalSteps, `Preparing ${analysisItems.length} image${analysisItems.length === 1 ? "" : "s"} for visual analysis...`);
const visionApi = resolveVisionApiFromEnv();
const ready = await ensureVisionModelReady(visionApi, {
baseUrl: lmStudioConfig.baseUrl,
apiKey: lmStudioConfig.apiKey,
modelKey: lmStudioConfig.model || "",
status: (message) => { try { statusCtx.status?.(message); } catch {} },
llamaServer: visionApi === "llama-server" ? resolveLlamaServerConfig(globalConfig) : undefined,
});
if (!ready.ok) {
throw new Error(ready.error);
}
// llama-server strand: router's own exact tagged model id/address, not the configured ones.
if (ready.resolvedBaseUrl) lmStudioConfig.baseUrl = ready.resolvedBaseUrl;
if (ready.resolvedModelKey) lmStudioConfig.model = ready.resolvedModelKey;
let totalMs = 0;
for (let idx = 0; idx < analysisItems.length; idx++) {
const item = analysisItems[idx];
reportToolStep(statusCtx, idx + 2, totalSteps, `Analyzing ${item.id} (${idx + 1}/${analysisItems.length})...`);
const batchResult = await analyzeLmStudioVisionBatch([item], lmStudioConfig);
for (const r of batchResult.results) {
visionResults.set(r.id, r.text.trim() || "(no description)");
}
totalMs += batchResult.totalInferenceTimeMs;
}
totalInferenceTimeMs = totalMs;
reportToolStep(statusCtx, totalSteps, totalSteps, "Formatting analysis results...");
} catch (e) {
visionError = classifyVisionError((e as Error).message || String(e));
}
}
const includeGenMeta = process.env.INCLUDE_GENERATION_METADATA !== "false";
// Build structured result
const lines: string[] = [];
lines.push(`Analysis results (${analysisItems.length} image${analysisItems.length !== 1 ? "s" : ""}):`)
lines.push("");
if (visionError) {
lines.push(`Note: Visual analysis unavailable — ${visionError}`);
lines.push("");
}
for (const item of analysisItems) {
const { id } = item;
const displayName = displayNames.get(id);
const origPath = originalFilePaths.get(id);
const header = displayName ? `${id} — ${displayName}` : id;
lines.push(`- ${header}`);
let visual: string | undefined;
if (effectivePrompt) {
visual = visionError ? "(not available)" : (visionResults.get(id) ?? "(no description)");
lines.push(` VISUAL: ${visual}`);
}
let metaText: string | undefined;
let meta: PngGenerationMeta | undefined;
let metaNote: string | undefined;
if (includeGenMeta) {
if (origPath && origPath.toLowerCase().endsWith(".png")) {
meta = readPngGenerationMeta(origPath) ?? undefined;
if (meta) {
metaText = formatGenerationMeta(meta);
} else {
metaText = ` (No embedded generation metadata)`;
metaNote = "(No embedded generation metadata)";
}
} else if (origPath && /\.jpe?g$/i.test(origPath)) {
const metadata = await readCameraImageMetadata(origPath);
metaText = Object.keys(metadata).length > 0 ? ` EXIF JSON: ${JSON.stringify(metadata)}` : ` (No embedded EXIF metadata)`;
metaNote = Object.keys(metadata).length > 0 ? `EXIF JSON: ${JSON.stringify(metadata)}` : "(No embedded EXIF metadata)";
} else if (origPath) {
metaText = ` (No embedded generation metadata — not a PNG file)`;
metaNote = "(No embedded generation metadata — not a PNG file)";
}
if (metaText) lines.push(metaText);
}
lines.push("");
onItem?.({ id, filePath: visionTmpIds.has(id) ? (origPath ?? item.filePath) : item.filePath, displayName, visual, meta, metaNote });
}
if (totalInferenceTimeMs !== null) {
lines.push(`Total inference time: ${Math.round(totalInferenceTimeMs)}ms`);
}
// Status update: done
try {
const statusSuffix = visionError ? " (vision unavailable)" : " successfully";
statusCtx.status?.(`Analyzed ${analysisItems.length} image${analysisItems.length !== 1 ? "s" : ""}${statusSuffix}`);
} catch {
// best-effort
}
// Audit log — requestId is reused as-is by the MCP adapter for its HTML preview filename.
try {
const audit = buildAuditLogger({ backend: "analyse_image", mode: "analyse_image" as any, requestId });
if (currentLmChatId) audit.setChatId(currentLmChatId);
audit.setUserRequest({ targets, prompt: effectivePrompt } as any);
audit.setOutput({ items: analysisItems.map((item) => item.id) } as any);
await audit.write();
} catch (e) {
console.warn("[analyse_image] audit logging failed:", String(e));
}
return lines.join("\n");
} catch (e) {
return `analyse_image failed: ${String((e as any)?.message || e)}`;
} finally {
for (const tp of visionTmpPaths) await fs.promises.unlink(tp).catch(() => {});
}
}
// ─────────────────────────────────────────────────────────────────────────────
// detect_object
// ─────────────────────────────────────────────────────────────────────────────
export const DetectObjectParamsShape = {
targets: FlexibleTargetsList.optional().describe(
"One or more image notations to process. Each notation is a letter followed by a number: " +
"a=attachment (a1, a2, …), i=generated image (i1, i2, …), v=variant (v1, v2, …), p=picture (p1, p2, …). " +
"Also accepts a filename in the scratchpad folder or an absolute path. " +
"Pass as a JSON array, e.g. [\"a1\", \"i3\"]. " +
"Omit when there is exactly one image — it will be selected automatically."
),
task: z
.string()
.optional()
.default("")
.describe(
"What to detect. Omit for full-image general object detection. " +
"Use natural language to target specific subjects (e.g. 'all faces and hands', 'the dog', 'cars and bicycles')."
),
} satisfies Record<string, z.ZodTypeAny>;
export async function handleDetectObject(
args: any,
ctl: ToolsProviderController,
statusCtx: ToolStatusContext = {},
requestId?: string
): Promise<any> {
try {
// Parse targets: accept array or comma/space-separated string (analogous to analyse_image)
let rawTargets: string[] = [];
if (Array.isArray(args?.targets)) {
rawTargets = args.targets.map((s: any) => String(s).trim()).filter(Boolean);
} else if (typeof args?.targets === "string" && args.targets.trim()) {
const trimmed = args.targets.trim();
if (trimmed.startsWith("[")) {
try {
const parsed = JSON.parse(trimmed);
if (Array.isArray(parsed)) {
rawTargets = parsed.map((s: any) => String(s).trim()).filter(Boolean);
} else {
rawTargets = trimmed.split(/[\s,]+/).map((s: string) => s.trim()).filter(Boolean);
}
} catch {
rawTargets = trimmed.split(/[\s,]+/).map((s: string) => s.trim()).filter(Boolean);
}
} else {
rawTargets = trimmed.split(/[\s,]+/).map((s: string) => s.trim()).filter(Boolean);
}
}
const task =
typeof args?.task === "string" && args.task.trim() ? args.task.trim() : "";
console.log("[detect_object] invoked", { targets: rawTargets, task });
let currentLmChatId: string | null = null;
let currentLmWorkingDir: string | null = null;
try {
const chatCtx = await getActiveChatContext();
if ((chatCtx as any)?.chatId) currentLmChatId = (chatCtx as any).chatId;
if ((chatCtx as any)?.workingDir) currentLmWorkingDir = (chatCtx as any).workingDir;
} catch {}
if (!currentLmChatId) {
try {
const resolved = await resolveActiveLMStudioChatId();
if ((resolved as any)?.ok) currentLmChatId = (resolved as any).chatId;
} catch {}
}
const primaryOutDir: string | undefined =
currentLmWorkingDir ||
(currentLmChatId ? getLMStudioWorkingDir(currentLmChatId) : undefined);
if (!primaryOutDir) {
console.error("[detect_object] could not resolve working directory");
return {
content: [
{
type: "text",
text: "detect_object failed: could not resolve LM Studio chat working directory.",
},
],
isError: true as const,
};
}
console.log("[detect_object] primaryOutDir:", primaryOutDir);
await fs.promises.mkdir(primaryOutDir, { recursive: true }).catch(() => {});
// Sync attachments so state is up to date
console.log("[detect_object] syncing attachments...");
try {
await syncAttachmentsToState(primaryOutDir, false, Number.MAX_SAFE_INTEGER);
} catch (e) {
console.warn("[detect_object] attachment sync failed (non-fatal):", (e as any)?.message ?? e);
}
console.log("[detect_object] attachment sync done");
console.log("[detect_object] reading state...");
const st = await readState(primaryOutDir);
const attachments: any[] = Array.isArray(st?.attachments) ? st.attachments : [];
const pictures: any[] = Array.isArray(st?.pictures) ? st.pictures : [];
const imageRecords: any[] = Array.isArray(st?.images) ? st.images : [];
const images: Array<{ i: number; path: string }> = imageRecords
.filter((r: any) => r && typeof r.filename === "string")
.sort((a: any, b: any) => (a.i || 0) - (b.i || 0))
.map((r: any) => ({ i: r.i || 1, path: path.join(primaryOutDir, r.filename) }));
const variantRecords: any[] = Array.isArray(st?.variants) ? st.variants : [];
const variants: Array<{ v: number; path: string }> = variantRecords
.filter((v: any) => v && typeof v.filename === "string")
.map((v: any) => ({ v: v.v || 1, path: path.join(primaryOutDir, v.filename) }));
console.log("[detect_object] state:", {
attachments: attachments.length,
images: images.length,
variants: variants.length,
pictures: pictures.length,
});
// Resolve source buffers for each target (or auto-select if none given)
type SourceEntry = { id: string; buf: Buffer; origBuf: Buffer };
const sourceEntries: SourceEntry[] = [];
async function resolveOneBuf(rawCanvas: string): Promise<Buffer> {
const pref = parsePrefixedNotation(rawCanvas);
if (!pref) {
const pathToken = await resolveTargetPathToken(rawCanvas, primaryOutDir);
if (!pathToken) throw new Error(`Invalid canvas notation: ${rawCanvas}`);
return fs.promises.readFile(pathToken);
}
if (pref.pool === "attachment") {
const rec = attachments.find((a: any) => a?.a === pref.index);
const previewRel = rec && typeof rec.preview === "string" ? rec.preview : "";
if (!previewRel) throw new Error(`Preview for attachment a${pref.index} not found.`);
return fs.promises.readFile(path.join(primaryOutDir!, previewRel));
} else if (pref.pool === "image") {
const rec = imageRecords.find((r: any) => r?.i === pref.index);
const previewRel = rec && typeof rec.preview === "string" ? rec.preview : "";
if (!previewRel) throw new Error(`Preview for image i${pref.index} not found.`);
return fs.promises.readFile(path.join(primaryOutDir!, previewRel));
} else if (pref.pool === "variant") {
const rec = variantRecords.find((v: any) => v?.v === pref.index);
const previewRel = rec && typeof rec.preview === "string" ? rec.preview : "";
if (!previewRel) throw new Error(`Preview for variant v${pref.index} not found.`);
return fs.promises.readFile(path.join(primaryOutDir!, previewRel));
} else {
const rec = pictures.find((p: any) => p?.p === pref.index);
const previewRel = rec && typeof rec.preview === "string" ? rec.preview : "";
if (!previewRel) throw new Error(`Preview for picture p${pref.index} not found.`);
return fs.promises.readFile(path.join(primaryOutDir!, previewRel));
}
}
// Resolves the original (full-resolution) file for a canvas notation.
// Falls back to previewFallback when the original is unavailable.
async function resolveOriginalBuf(rawCanvas: string, previewFallback: Buffer): Promise<Buffer> {
try {
const pref = parsePrefixedNotation(rawCanvas);
if (!pref) {
const pathToken = await resolveTargetPathToken(rawCanvas, primaryOutDir);
return pathToken ? await fs.promises.readFile(pathToken) : previewFallback;
}
if (pref.pool === "attachment") {
const rec = attachments.find((a: any) => a?.a === pref.index);
const originAbs = rec && typeof rec.originAbs === "string" ? rec.originAbs : "";
if (!originAbs) return previewFallback;
return await fs.promises.readFile(originAbs);
}
let rec: any;
if (pref.pool === "image") rec = imageRecords.find((r: any) => r?.i === pref.index);
else if (pref.pool === "variant") rec = variantRecords.find((v: any) => v?.v === pref.index);
else rec = pictures.find((p: any) => p?.p === pref.index);
const filename = rec && typeof rec.filename === "string" ? rec.filename : "";
if (!filename) return previewFallback;
return await fs.promises.readFile(path.join(primaryOutDir!, filename));
} catch {
return previewFallback;
}
}
try {
if (rawTargets.length > 0) {
for (const t of rawTargets) {
const buf = await resolveOneBuf(t);
const origBuf = await resolveOriginalBuf(t, buf);
sourceEntries.push({ id: t, buf, origBuf });
}
} else {
// Auto-select single source
const total = attachments.length + variantRecords.length + imageRecords.length + pictures.length;
if (total === 0) throw new Error("No source image available.");
if (total > 1) throw new Error("Ambiguous source — specify targets explicitly.");
let buf: Buffer;
let id: string;
if (attachments.length === 1) {
const rec = attachments[0];
const previewRel = rec && typeof rec.preview === "string" ? rec.preview : "";
if (!previewRel) throw new Error("Preview for attachment not found.");
buf = await fs.promises.readFile(path.join(primaryOutDir!, previewRel));
id = `a${typeof rec.a === "number" ? rec.a : 1}`;
} else if (variantRecords.length === 1) {
const rec = variantRecords[0];
const previewRel = rec && typeof rec.preview === "string" ? rec.preview : "";
if (!previewRel) throw new Error("Preview for variant not found.");
buf = await fs.promises.readFile(path.join(primaryOutDir!, previewRel));
id = `v${typeof rec.v === "number" ? rec.v : 1}`;
} else if (imageRecords.length === 1) {
const rec = imageRecords[0];
const previewRel = rec && typeof rec.preview === "string" ? rec.preview : "";
if (!previewRel) throw new Error("Preview for image not found.");
buf = await fs.promises.readFile(path.join(primaryOutDir!, previewRel));
id = `i${typeof rec.i === "number" ? rec.i : 1}`;
} else {
const rec = pictures[0];
const previewRel = rec && typeof rec.preview === "string" ? rec.preview : "";
if (!previewRel) throw new Error("Preview for picture not found.");
buf = await fs.promises.readFile(path.join(primaryOutDir!, previewRel));
id = `p${rec.p ?? 1}`;
}
const origBuf = await resolveOriginalBuf(id, buf);
sourceEntries.push({ id, buf, origBuf });
}
} catch (e) {
return {
content: [{ type: "text", text: String((e as any)?.message || e) }],
isError: true as const,
};
}
console.log("[detect_object] sources resolved:", sourceEntries.map((s) => s.id));
const globalConfig = getGlobalConfig(ctl);
const envEmbedPngMetadata = process.env.EMBED_PNG_METADATA;
const embedPngMetadata = getGlobalBoolean(
globalConfig,
"embedPngMetadata",
envEmbedPngMetadata === undefined ? defaultPluginSettings.embedPngMetadata : envEmbedPngMetadata !== "false"
);
let visionBaseUrl = getGlobalString(globalConfig, "embeddingBaseUrl", process.env.LMSTUDIO_VISION_API_BASE_URL || defaultPluginSettings.embeddingBaseUrl);
const visionApiKey = getGlobalString(globalConfig, "embeddingApiKey", process.env.LMSTUDIO_VISION_API_KEY || defaultPluginSettings.embeddingApiKey);
let visionModelKey = getGlobalString(globalConfig, "qwen3VlModelPath", process.env.LMSTUDIO_VISION_MODEL_KEY || defaultPluginSettings.qwen3VlModelPath);
const envDetectMaxTokens = Number.parseInt(process.env.DETECT_MAX_TOKENS || "", 10);
const envDetectTemperature = Number.parseFloat(process.env.DETECT_TEMPERATURE || "");
const configuredDetectMaxTokens = Math.floor(getGlobalNumber(
globalConfig,
"detectMaxTokens",
Number.isFinite(envDetectMaxTokens) && envDetectMaxTokens > 0 ? envDetectMaxTokens : defaultPluginSettings.detectMaxTokens
));
const configuredDetectTemperature = getGlobalNumber(
globalConfig,
"detectTemperature",
Number.isFinite(envDetectTemperature) ? envDetectTemperature : defaultPluginSettings.detectTemperature
);
const detectionConfig: VisionDetectionAnalyzerConfig = {
task,
odPrompt: getGlobalString(globalConfig, "qwen3VlOdPrompt", process.env.DETECT_OD_PROMPT || defaultPluginSettings.qwen3VlOdPrompt) || undefined,
maxTokens: configuredDetectMaxTokens,
temperature: configuredDetectTemperature,
timeoutMs: 120_000,
};
const tmpPaths: string[] = [];
const detectionItems: VisionAnalysisItem[] = [];
for (const entry of sourceEntries) {
const visionBuf = await normalizeVisionBuffer(entry.buf);
const tmpPath = path.join(primaryOutDir, `_tmp_detect_src_${safeIdForFilename(entry.id)}_${Date.now()}.png`);
await fs.promises.writeFile(tmpPath, visionBuf);
tmpPaths.push(tmpPath);
detectionItems.push({ id: entry.id, filePath: tmpPath });
}
console.log("[detect_object] calling detection API for", detectionItems.length, "items");
const progressTotalSteps = sourceEntries.length * 2 + 4;
let batchResult: VisionDetectionBatchResult = {
results: [],
totalInferenceTimeMs: 0,
backend: "vision-api",
};
try {
reportToolStatus(statusCtx, `Detecting objects in ${sourceEntries.length} image${sourceEntries.length === 1 ? "" : "s"}...`);
reportToolStep(statusCtx, 1, progressTotalSteps, `Preparing ${sourceEntries.length} image${sourceEntries.length === 1 ? "" : "s"} for object detection...`);
const visionApi = resolveVisionApiFromEnv();
const ready = await ensureVisionModelReady(visionApi, {
baseUrl: visionBaseUrl,
apiKey: visionApiKey,
modelKey: visionModelKey,
status: (message) => { try { statusCtx.status?.(message); } catch {} },
llamaServer: visionApi === "llama-server" ? resolveLlamaServerConfig(globalConfig) : undefined,
});
if (!ready.ok) {
throw new Error(ready.error);
}
// llama-server strand: router's own exact tagged model id/address, not the configured ones.
if (ready.resolvedBaseUrl) visionBaseUrl = ready.resolvedBaseUrl;
if (ready.resolvedModelKey) visionModelKey = ready.resolvedModelKey;
for (let idx = 0; idx < detectionItems.length; idx++) {
const item = detectionItems[idx];
reportToolStep(statusCtx, idx + 2, progressTotalSteps, `Detecting objects in ${item.id} (${idx + 1}/${detectionItems.length})...`);
const singleResult = await detectLmStudioVisionBatch([item], {
...detectionConfig,
baseUrl: visionBaseUrl,
apiKey: visionApiKey,
model: visionModelKey,
});
batchResult.results.push(...singleResult.results);
batchResult.totalInferenceTimeMs += singleResult.totalInferenceTimeMs;
batchResult.backend = singleResult.backend;
}
console.log("[detect_object] detection API returned:", {
results: batchResult.results.length,
totalMs: batchResult.totalInferenceTimeMs,
});
try {
const totalObjects = batchResult.results.reduce((s, r) => s + (r.objects?.length ?? 0), 0);
const ms = Math.round(batchResult.totalInferenceTimeMs);
reportToolStep(statusCtx, sourceEntries.length + 2, progressTotalSteps, `${totalObjects} object${totalObjects === 1 ? "" : "s"} found across ${batchResult.results.length} image${batchResult.results.length === 1 ? "" : "s"} (${ms}ms); drawing bounding boxes...`);
} catch {}
} finally {
for (const tp of tmpPaths) await fs.promises.unlink(tp).catch(() => {});
}
if (!batchResult.results.length) {
return {
content: [{ type: "text", text: "detect_object: no results returned from detection API." }],
isError: true as const,
};
}
// Per-image: draw bboxes, save, generate preview, build state records
const variantPreviewSpec = (VARIANT_FULL_CONFIG as any).preview;
const stamp = isoStampCompact();
let nextI = Math.max(1, st.counters?.nextImageI ?? 1);
const imageRecordsForState: any[] = [];
const resultEntries: Array<{
id: string;
i: number;
detResult: typeof batchResult.results[0];
savedPath: string;
savedFileUrl: string;
savedSize: number;
preview: any | null;
httpOriginal: string;
httpPreview: string;
}> = [];
const httpBase = await getHealthyServerBaseUrl();
for (let idx = 0; idx < batchResult.results.length; idx++) {
const detResult = batchResult.results[idx];
const sourceId = sourceEntries[idx]?.id ?? `canvas${idx + 1}`;
const origBuf = sourceEntries[idx].origBuf;
const currentI = nextI++;
reportToolStep(statusCtx, sourceEntries.length + 3 + idx, progressTotalSteps, `Drawing boxes for ${sourceId} (${idx + 1}/${batchResult.results.length})...`);
const bboxes = detResult.objects.map((o) => o.bbox as [number, number, number, number]);
console.log(`[detect_object] drawing ${bboxes.length} bboxes for ${sourceId}...`);
// Draw on the original-resolution file; sourceDims carries the preview space in which
// the detection server reported its bbox coordinates so they get scaled up correctly.
const annotatedBuf = await drawBboxesOnImage(origBuf, bboxes, {
sourceDims: { width: detResult.imageWidth, height: detResult.imageHeight },
palette: true,
});
const analysis: PngAnalysisMetadata = {
schema: "ceveyne.image-analysis/v1",
tool: "detect_object",
sourceNotation: sourceId,
inference: {
model: visionModelKey,
...(task ? { query: task } : {}),
detectorPromptSha256: createHash("sha256").update(detectionConfig.odPrompt ?? "").digest("hex"),
maxTokens: configuredDetectMaxTokens,
temperature: configuredDetectTemperature,
},
render: { palette: true },
detections: detResult.objects.map((object) => ({
label: object.label,
bbox: { x1: object.bbox[0], y1: object.bbox[1], x2: object.bbox[2], y2: object.bbox[3] },
})),
};
const savedBuffer = embedPngMetadata
? injectXmpIntoBuffer(annotatedBuf, {
...(task ? { prompt: task } : {}),
model: visionModelKey,
mode: "object_detection",
generatedBy: `${getSelfPluginIdentifier()}/detect_object`,
creatorTool: `${getSelfPluginIdentifier()}/detect_object`,
analysis,
})
: annotatedBuf;
const baseName = `image-${stamp}-i${currentI}`;
const savedPath = path.join(primaryOutDir, `${baseName}.png`);
await fs.promises.writeFile(savedPath, savedBuffer);
const savedFileUrl = pathToFileURL(savedPath).toString();
const savedSize = savedBuffer.length;
console.log(`[detect_object] annotated image written: ${savedPath} (${savedSize} bytes)`);
let preview: any = null;
try {
const p = await generatePreviewFromBuffer(savedBuffer, primaryOutDir, `${baseName}.png`, variantPreviewSpec);
preview = {
ok: true as const,
filePath: p.previewAbs,
fileName: p.previewFilename,
fileUrl: pathToFileURL(p.previewAbs).toString(),
size_bytes: p.data.length,
width: p.width,
height: p.height,
mimeType: "image/jpeg" as const,
dataBase64: p.data.toString("base64"),
};
} catch (e) {
console.warn(`[detect_object] preview generation failed for ${sourceId}:`, String(e));
}
const httpOriginal = httpBase
? toHttpOriginalUrl(`${baseName}.png`, httpBase, currentLmChatId || undefined)
: "";
const httpPreview = (() => {
if (!httpBase || !currentLmChatId || !preview?.fileName) return "";
return toHttpPreviewUrl(preview.fileName, httpBase, currentLmChatId);
})();
imageRecordsForState.push({
filename: `${baseName}.png`,
preview: preview ? `preview-${baseName}.jpg` : undefined,
i: currentI,
sourceTool: `${getSelfPluginIdentifier()}/detect_object`,
detectSource: sourceId,
task,
imageWidth: detResult.imageWidth,
imageHeight: detResult.imageHeight,
analysisMetadata: analysis,
detections: detResult.objects.map((o) => ({
label: o.label,
bbox: { x1: o.bbox[0], y1: o.bbox[1], x2: o.bbox[2], y2: o.bbox[3] },
crop: {
cropLeft: o.cropLeft,
cropRight: o.cropRight,
cropTop: o.cropTop,
cropBottom: o.cropBottom,
},
})),
});
resultEntries.push({ id: sourceId, i: currentI, detResult, savedPath, savedFileUrl, savedSize, preview, httpOriginal, httpPreview });
}
// Update state once for all images
console.log("[detect_object] updating state...");
reportToolStep(statusCtx, progressTotalSteps - 1, progressTotalSteps, "Updating image state and audit log...");
try {
const stateForUpdate = await readState(primaryOutDir);
const appendResult = appendImages(stateForUpdate, imageRecordsForState);
if (appendResult.changed) {
await writeStateAtomic(primaryOutDir, stateForUpdate);
console.log("[detect_object] state written, nextImageI:", stateForUpdate.counters?.nextImageI);
}
} catch (e) {
console.warn("[detect_object] state update failed:", String(e));
}
// Audit log — effectiveRequestId is reused below in each summary entry, so a caller
// (the MCP adapter) can always name its HTML report after this exact audit requestId,
// even for a batch call whose top-level result is an array (see summaries below).
const effectiveRequestId = requestId ?? `${Date.now()}-${Math.random().toString(36).slice(2, 8)}`;
try {
const audit = buildAuditLogger({ backend: "detect_object", mode: "detect_object" as any, requestId: effectiveRequestId });
if (currentLmChatId) audit.setChatId(currentLmChatId);
audit.setUserRequest({ targets: rawTargets, task } as any);
audit.setOutput({
images: resultEntries.map((r) => ({
id: r.id,
i: r.i,
detections: r.detResult.objects.length,
path: r.savedPath,
url: r.savedFileUrl,
bytes: r.savedSize,
...(r.httpOriginal ? { http_url: r.httpOriginal } : {}),
...(r.preview ? { preview_path: r.preview.filePath, preview_url: r.preview.fileUrl } : {}),
...(r.httpPreview ? { http_preview_url: r.httpPreview } : {}),
})),
} as any);
await audit.write();
} catch (e) {
console.warn("[detect_object] audit logging failed:", String(e));
}
// Assemble tool result
const envPreviewRaw = process.env["PREVIEW_IN_CHAT"];
const previewInChat =
envPreviewRaw === undefined
? true
: envPreviewRaw === "1" || envPreviewRaw.toLowerCase() === "true";
const summaries = resultEntries.map((r) => ({
tool: "detect_object",
requestId: effectiveRequestId,
source: r.id,
i: r.i,
imageWidth: r.detResult.imageWidth,
imageHeight: r.detResult.imageHeight,
inferenceTimeMs: r.detResult.inferenceTimeMs,
detections: r.detResult.objects.map((o) => ({
label: o.label,
bbox: { x1: o.bbox[0], y1: o.bbox[1], x2: o.bbox[2], y2: o.bbox[3] },
crop: {
left: { pct: o.cropLeft, px: Math.round((o.cropLeft / 100) * r.detResult.imageWidth) },
right: { pct: o.cropRight, px: Math.round((o.cropRight / 100) * r.detResult.imageWidth) },
top: { pct: o.cropTop, px: Math.round((o.cropTop / 100) * r.detResult.imageHeight) },
bottom: { pct: o.cropBottom, px: Math.round((o.cropBottom / 100) * r.detResult.imageHeight) },
},
crop_tool_hint: "Pass crop.left.pct as cropLeft, crop.right.pct as cropRight, crop.top.pct as cropTop, crop.bottom.pct as cropBottom to the crop tool.",
})),
}));
const reviewHint = "Carefully examine the preview to make absolutely sure that the object detection matches your intent.";
const content: any[] = [];
reportToolStep(statusCtx, progressTotalSteps, progressTotalSteps, "Assembling detection result...");
for (const r of resultEntries) {
const fallbackPreviewUrl = r.preview?.fileUrl || r.savedFileUrl;
const previewLine = `Preview i${r.i}: ${r.httpPreview ? r.httpPreview : fallbackPreviewUrl}`;
const originalLine = `Original i${r.i}: ${r.httpOriginal ? r.httpOriginal : r.savedFileUrl}`;
if (previewInChat && r.preview) {
const fname = String(r.preview.fileName || "");
content.push({
type: "image",
fileName: fname,
mimeType: r.preview.mimeType,
markdown: ``,
$hint: "This is an image file. Present the image to the user by using the markdown above.",
} as any);
}
// TODO: Restore when LM Studio renders file/HTTP links again
// content.push({ type: "text", text: previewLine });
// content.push({ type: "text", text: originalLine });
}
const totalMs = Math.round(batchResult.totalInferenceTimeMs);
if (totalMs > 0) {
content.push({ type: "text", text: `Total inference time: ${totalMs}ms` });
}
content.push({
type: "text",
text: JSON.stringify(summaries.length === 1 ? summaries[0] : summaries),
...(previewInChat ? {} : { $hint: reviewHint }),
});
return { content };
} catch (error) {
return {
content: [
{
type: "text",
text: `detect_object failed: ${(error as Error).message || String(error)}`,
},
],
isError: true as const,
};
}
}
// ─────────────────────────────────────────────────────────────────────────────
// annotate_image
// ─────────────────────────────────────────────────────────────────────────────
/**
* Expand (positive) or shrink (negative) a bbox [x1, y1, x2, y2].
* Number → % of box diagonal. String → value + optional 'px' or '%' suffix.
*/
function applyFrameAdjust(
bbox: [number, number, number, number],
frameAdjust: number | string,
imgW: number,
imgH: number,
): [number, number, number, number] {
const [x1, y1, x2, y2] = bbox;
const diag = Math.hypot(x2 - x1, y2 - y1);
let d_px: number;
if (typeof frameAdjust === "string") {
const m = String(frameAdjust).trim().match(/^([+-]?\d+(?:\.\d+)?)\s*(%|px)?$/i);
if (!m) return bbox;
const val = parseFloat(m[1]);
d_px = m[2]?.toLowerCase() === "px" ? val : (val / 100) * diag;
} else {
d_px = (frameAdjust / 100) * diag;
}
return [
Math.max(0, Math.round(x1 - d_px)),
Math.max(0, Math.round(y1 - d_px)),
Math.min(imgW - 1, Math.round(x2 + d_px)),
Math.min(imgH - 1, Math.round(y2 + d_px)),
];
}
export const AnnotateImageParamsShape = {
targets: FlexibleTargetsList.optional().describe(
"One or more image notations to process. Each notation is a letter followed by a number: " +
"a=attachment (a1, a2, …), i=generated image (i1, i2, …), v=variant (v1, v2, …), p=picture (p1, p2, …). " +
"Also accepts a filename in the scratchpad folder or an absolute path. " +
"Pass via the targets field, e.g. annotate_image({\"targets\":[\"a1\", \"i3\"]}). " +
"Omit when there is exactly one image — it will be selected automatically."
),
task: z
.string()
.optional()
.default("")
.describe(
"What to detect. Omit for full-image general object detection. " +
"Use natural language to target specific subjects (e.g. 'all faces and hands', 'the dog', 'cars and bicycles'). " +
"Not used on correction calls — detections are loaded from state."
),
color: z
.string()
.optional()
.describe("Box color for all detected objects. CSS color name or hex, e.g. 'pink', 'red', '#FF0000'. Default: magenta."),
lineWeight: z
.coerce.number().int().min(1).max(50)
.optional()
.describe("Line thickness in pixels for all bounding boxes. Default: 5."),
frameAdjust: z
.union([z.number(), z.string()])
.optional()
.describe(
"Expand (positive) or shrink (negative) bounding boxes before drawing. " +
"Number: percent of box diagonal (e.g. 5 = +5 %). " +
"String: value + optional 'px' or '%' suffix, e.g. '10px', '-5%'. " +
"Without detectLabel: applied to all boxes. With detectLabel: applied to the selected box only. Default: 5."
),
detectLabel: z
.preprocess(
(val) => {
if (typeof val === "string") {
const t = val.trim();
if (t.startsWith("[")) {
try {
const parsed = JSON.parse(t);
if (Array.isArray(parsed)) {
return parsed.map((el: unknown) => String(el).trim()).filter((s: string) => s.length > 0);
}
} catch {}
}
}
return val;
},
z.union([
z.string().transform((s) =>
s.split(/\s*,\s*/).map((x) => x.trim()).filter((x) => x.length > 0)
),
z.array(z.string().min(1)),
])
)
.optional()
.describe(
"On a correction call: label(s) to match (case-insensitive). " +
"Single string, comma-separated list ('left eye, right eye'), or JSON array. " +
"When set, ONLY the matching detection(s) are drawn — all others are omitted. " +
"Single label with no detectIndex: auto-expands to ALL detections for that label (Option A). " +
"canvas may be the original source (e.g. a1) or the previous annotate_image result (e.g. i3)."
),
detectIndex: z
.union([
z.string().transform((s) => (s.match(/\d+/g) ?? []).map(Number)),
z.coerce.number().int().min(0).transform((n) => [n]),
z.array(z.coerce.number().int().min(0)),
])
.optional()
.describe(
"Zero-based index or list of indices, parallel to detectLabel. " +
"indices[li] ?? indices[0] ?? 0 for missing entries. Default: 0. " +
"Single label + multiple indices draws that label at each specified occurrence (Option B). " +
"Bracket notation accepted: '[2, 4, 7]'."
),
x1: z.preprocess(
(val) => { if (typeof val === "string" && val.trim().startsWith("[")) { try { return JSON.parse(val); } catch {} } return val; },
z.union([z.number(), z.array(z.union([z.number(), z.null()]))])
).optional().describe(
"Manual left edge(s) in original image pixels. " +
"Scalar: applies to all selected detections. Array (parallel to detectLabel): null = keep stored value. " +
"E.g. x1=[null,50,null] moves only the second detection's left edge."
),
y1: z.preprocess(
(val) => { if (typeof val === "string" && val.trim().startsWith("[")) { try { return JSON.parse(val); } catch {} } return val; },
z.union([z.number(), z.array(z.union([z.number(), z.null()]))])
).optional().describe("Manual top edge(s) in original image pixels. Scalar or array (null = keep stored). See x1."),
x2: z.preprocess(
(val) => { if (typeof val === "string" && val.trim().startsWith("[")) { try { return JSON.parse(val); } catch {} } return val; },
z.union([z.number(), z.array(z.union([z.number(), z.null()]))])
).optional().describe("Manual right edge(s) in original image pixels. Scalar or array (null = keep stored). See x1."),
y2: z.preprocess(
(val) => { if (typeof val === "string" && val.trim().startsWith("[")) { try { return JSON.parse(val); } catch {} } return val; },
z.union([z.number(), z.array(z.union([z.number(), z.null()]))])
).optional().describe("Manual bottom edge(s) in original image pixels. Scalar or array (null = keep stored). See x1."),
} satisfies Record<string, z.ZodTypeAny>;
type StoredDetection = {
label: string;
bbox: { x1: number; y1: number; x2: number; y2: number };
crop?: { cropLeft: number; cropRight: number; cropTop: number; cropBottom: number };
};
type ResolvedEntry =
| { mode: "detect"; id: string; previewBuf: Buffer; origBuf: Buffer }
| {
mode: "redraw";
id: string; // original draw source (e.g. "a1")
origBuf: Buffer;
task: string;
detections: StoredDetection[];
imageWidth: number;
imageHeight: number;
analysisMetadata?: PngAnalysisMetadata;
};
/**
* Resolve a per-box coordinate override, analogous to resolveBoxOverride in mask.
* override: scalar (applies to all boxes) | array (parallel, null = keep stored) | undefined.
* Returns undefined when no override → caller uses storedValue.
*/
function resolveCoordOverride(
override: number | (number | null)[] | undefined,
index: number,
): number | undefined {
if (override === undefined) return undefined;
if (Array.isArray(override)) {
const entry = override[index];
if (entry === null || entry === undefined) return undefined;
return entry;
}
return override;
}
/**
* Option A helper: count occurrences of label in detections and return [0..N-1] if N > 1, else [0].
*/
function expandDetectIndices(detections: StoredDetection[], label: string): number[] {
const lower = label.toLowerCase();
let count = 0;
for (const d of detections) { if (d.label.toLowerCase() === lower) count++; }
return count > 1 ? Array.from({ length: count }, (_, i) => i) : [0];
}
export async function handleAnnotateImage(
args: any,
ctl: ToolsProviderController,
statusCtx: ToolStatusContext = {},
requestId?: string
): Promise<any> {
try {
// ── Parse args ─────────────────────────────────────────────────────
let rawTargets: string[] = [];
if (Array.isArray(args?.targets)) {
rawTargets = args.targets.map((s: any) => String(s).trim()).filter(Boolean);
} else if (typeof args?.targets === "string" && args.targets.trim()) {
const trimmed = args.targets.trim();
if (trimmed.startsWith("[")) {
try {
const parsed = JSON.parse(trimmed);
rawTargets = Array.isArray(parsed)
? parsed.map((s: any) => String(s).trim()).filter(Boolean)
: trimmed.split(/[\s,]+/).map((s: string) => s.trim()).filter(Boolean);
} catch {
rawTargets = trimmed.split(/[\s,]+/).map((s: string) => s.trim()).filter(Boolean);
}
} else {
rawTargets = trimmed.split(/[\s,]+/).map((s: string) => s.trim()).filter(Boolean);
}
}
const taskArg = typeof args?.task === "string" && args.task.trim() ? args.task.trim() : "";
const globalColor = typeof args?.color === "string" && args.color.trim() ? args.color.trim() : "magenta";
const globalLineWeight = typeof args?.lineWeight === "number"
? Math.max(1, Math.min(50, Math.round(args.lineWeight))) : 5;
const globalFrameAdjust: number | string = args?.frameAdjust !== undefined ? args.frameAdjust : 5;
// detectLabel: string → string[] via comma-split; array → as-is
let globalLabels: string[] | undefined;
{
const dlRaw = args?.detectLabel;
if (Array.isArray(dlRaw)) {
const arr = dlRaw.map((s: any) => String(s).trim()).filter(Boolean);
if (arr.length > 0) globalLabels = arr;
} else if (typeof dlRaw === "string" && dlRaw.trim()) {
globalLabels = dlRaw.split(/\s*,\s*/).map((x) => x.trim()).filter(Boolean);
}
}
// detectIndex: number → [n]; string → extract digit runs; array → as-is
let globalIndices: number[] = [];
{
const diRaw = args?.detectIndex;
if (Array.isArray(diRaw)) {
globalIndices = diRaw.map((n: any) => typeof n === "number" ? Math.floor(n) : parseInt(String(n), 10)).filter((n) => !isNaN(n) && n >= 0);
} else if (typeof diRaw === "string" && diRaw.trim()) {
globalIndices = (diRaw.match(/\d+/g) ?? []).map(Number);
} else if (typeof diRaw === "number" && diRaw >= 0) {
globalIndices = [Math.floor(diRaw)];
}
}
const rawX1 = args?.x1;
const rawY1 = args?.y1;
const rawX2 = args?.x2;
const rawY2 = args?.y2;
// Normalise: scalar number | number-or-null array | undefined
function normaliseCoord(raw: any): number | (number | null)[] | undefined {
if (raw === undefined || raw === null) return undefined;
if (typeof raw === "number") return raw;
if (Array.isArray(raw)) return raw.map((v: any) => (v === null || v === undefined) ? null : Number(v));
if (typeof raw === "string") {
const t = raw.trim();
if (t.startsWith("[")) { try { const p = JSON.parse(t); if (Array.isArray(p)) return p.map((v: any) => (v === null || v === undefined) ? null : Number(v)); } catch {} }
const n = Number(t); return isNaN(n) ? undefined : n;
}
return undefined;
}
const manualX1 = normaliseCoord(rawX1);
const manualY1 = normaliseCoord(rawY1);
const manualX2 = normaliseCoord(rawX2);
const manualY2 = normaliseCoord(rawY2);
const hasAnyManualCoord = manualX1 !== undefined || manualY1 !== undefined || manualX2 !== undefined || manualY2 !== undefined;
console.log("[annotate_image] invoked", { targets: rawTargets, task: taskArg, color: globalColor, lineWeight: globalLineWeight, frameAdjust: globalFrameAdjust, detectLabels: globalLabels, detectIndices: globalIndices, manualCoords: { x1: manualX1, y1: manualY1, x2: manualX2, y2: manualY2 } });
// ── Resolve working directory ──────────────────────────────────────
let currentLmChatId: string | null = null;
let currentLmWorkingDir: string | null = null;
try {
const chatCtx = await getActiveChatContext();
if ((chatCtx as any)?.chatId) currentLmChatId = (chatCtx as any).chatId;
if ((chatCtx as any)?.workingDir) currentLmWorkingDir = (chatCtx as any).workingDir;
} catch {}
if (!currentLmChatId) {
try {
const resolved = await resolveActiveLMStudioChatId();
if ((resolved as any)?.ok) currentLmChatId = (resolved as any).chatId;
} catch {}
}
const primaryOutDir: string | undefined =
currentLmWorkingDir ||
(currentLmChatId ? getLMStudioWorkingDir(currentLmChatId) : undefined);
if (!primaryOutDir) {
return {
content: [{ type: "text", text: "annotate_image failed: could not resolve LM Studio chat working directory." }],
isError: true as const,
};
}
await fs.promises.mkdir(primaryOutDir, { recursive: true }).catch(() => {});
// ── Sync attachments ──────────────────────────────────────────────
try {
await syncAttachmentsToState(primaryOutDir, false, Number.MAX_SAFE_INTEGER);
} catch (e) {
console.warn("[annotate_image] attachment sync failed (non-fatal):", (e as any)?.message ?? e);
}
// ── Read state ────────────────────────────────────────────────────
const st = await readState(primaryOutDir);
const attachments: any[] = Array.isArray(st?.attachments) ? st.attachments : [];
const pictures: any[] = Array.isArray(st?.pictures) ? st.pictures : [];
const imageRecords: any[] = Array.isArray(st?.images) ? st.images : [];
const variantRecords: any[] = Array.isArray(st?.variants) ? st.variants : [];
// ── Buffer resolution helpers ─────────────────────────────────────
async function resolvePreviewBuf(notation: string): Promise<Buffer> {
const pref = parsePrefixedNotation(notation);
if (!pref) {
const pathToken = await resolveTargetPathToken(notation, primaryOutDir);
if (!pathToken) throw new Error(`Invalid notation: ${notation}`);
return fs.promises.readFile(pathToken);
}
if (pref.pool === "attachment") {
const rec = attachments.find((a: any) => a?.a === pref.index);
const r = rec && typeof rec.preview === "string" ? rec.preview : "";
if (!r) throw new Error(`Preview for a${pref.index} not found.`);
return fs.promises.readFile(path.join(primaryOutDir!, r));
}
if (pref.pool === "image") {
const rec = imageRecords.find((r: any) => r?.i === pref.index);
const r = rec && typeof rec.preview === "string" ? rec.preview : "";
if (!r) throw new Error(`Preview for i${pref.index} not found.`);
return fs.promises.readFile(path.join(primaryOutDir!, r));
}
if (pref.pool === "variant") {
const rec = variantRecords.find((v: any) => v?.v === pref.index);
const r = rec && typeof rec.preview === "string" ? rec.preview : "";
if (!r) throw new Error(`Preview for v${pref.index} not found.`);
return fs.promises.readFile(path.join(primaryOutDir!, r));
}
const rec = pictures.find((p: any) => p?.p === pref.index);
const r = rec && typeof rec.preview === "string" ? rec.preview : "";
if (!r) throw new Error(`Preview for p${pref.index} not found.`);
return fs.promises.readFile(path.join(primaryOutDir!, r));
}
async function resolveOriginalBuf(notation: string, fallback: Buffer): Promise<Buffer> {
try {
const pref = parsePrefixedNotation(notation);
if (!pref) {
const pathToken = await resolveTargetPathToken(notation, primaryOutDir);
return pathToken ? await fs.promises.readFile(pathToken) : fallback;
}
if (pref.pool === "attachment") {
const rec = attachments.find((a: any) => a?.a === pref.index);
const abs = rec && typeof rec.originAbs === "string" ? rec.originAbs : "";
if (!abs) return fallback;
return await fs.promises.readFile(abs);
}
let rec: any;
if (pref.pool === "image") rec = imageRecords.find((r: any) => r?.i === pref.index);
else if (pref.pool === "variant") rec = variantRecords.find((v: any) => v?.v === pref.index);
else rec = pictures.find((p: any) => p?.p === pref.index);
const fn = rec && typeof rec.filename === "string" ? rec.filename : "";
if (!fn) return fallback;
return await fs.promises.readFile(path.join(primaryOutDir!, fn));
} catch {
return fallback;
}
}
// ── Resolve auto-selected source when no targets given ─────────────
let autoId: string | null = null;
if (rawTargets.length === 0) {
const total = attachments.length + variantRecords.length + imageRecords.length + pictures.length;
if (total === 0) {
return { content: [{ type: "text", text: "No source image available." }], isError: true as const };
}
if (total > 1) {
return { content: [{ type: "text", text: "Ambiguous source — specify targets explicitly." }], isError: true as const };
}
if (attachments.length === 1) autoId = `a${typeof attachments[0]?.a === "number" ? attachments[0].a : 1}`;
else if (variantRecords.length === 1) autoId = `v${typeof variantRecords[0]?.v === "number" ? variantRecords[0].v : 1}`;
else if (imageRecords.length === 1) autoId = `i${typeof imageRecords[0]?.i === "number" ? imageRecords[0].i : 1}`;
else autoId = `p${pictures[0]?.p ?? 1}`;
rawTargets = [autoId];
}
// ── Classify each target: detect or redraw ────────────────────────
// If task is explicitly provided, always run fresh inference (ignore any prior record).
const forceDetect = taskArg.length > 0;
const resolvedEntries: ResolvedEntry[] = [];
for (const rawId of rawTargets) {
let drawSourceId = rawId;
let stateRec: any = null;
// Case A: target is an iN detection/annotation result
const pref = parsePrefixedNotation(rawId);
if (pref?.pool === "image") {
const imgRec = imageRecords.find((r: any) => r?.i === pref.index);
if (
imgRec &&
Array.isArray(imgRec.detections) &&
imgRec.detections.length > 0 &&
typeof imgRec.imageWidth === "number"
) {
stateRec = imgRec;
drawSourceId = (typeof imgRec.detectSource === "string" && imgRec.detectSource)
? imgRec.detectSource
: rawId;
}
}
// Case B: target is an original (a1, p1, …) with a prior detection on record
if (!stateRec) {
const prior = [...imageRecords]
.reverse()
.find(
(r: any) =>
(r?.detectSource === rawId) &&
Array.isArray(r.detections) &&
r.detections.length > 0 &&
typeof r.imageWidth === "number",
);
if (prior) {
stateRec = prior;
drawSourceId = rawId;
}
}
try {
if (stateRec && !forceDetect) {
const preview = await resolvePreviewBuf(drawSourceId).catch(() => null);
const origBuf = preview
? await resolveOriginalBuf(drawSourceId, preview)
: Buffer.alloc(0);
resolvedEntries.push({
mode: "redraw",
id: drawSourceId,
origBuf,
task: typeof stateRec.task === "string" ? stateRec.task : taskArg,
detections: stateRec.detections as StoredDetection[],
imageWidth: stateRec.imageWidth as number,
imageHeight: stateRec.imageHeight as number,
analysisMetadata: stateRec.analysisMetadata as PngAnalysisMetadata | undefined,
});
} else {
const previewBuf = await resolvePreviewBuf(rawId);
const origBuf = await resolveOriginalBuf(rawId, previewBuf);
resolvedEntries.push({ mode: "detect", id: rawId, previewBuf, origBuf });
}
} catch (e) {
return {
content: [{ type: "text", text: String((e as any)?.message || e) }],
isError: true as const,
};
}
}
// ── Run Qwen3-VL for detect-mode entries ───────────────────────────
const detectEntries = resolvedEntries.filter((e): e is Extract<ResolvedEntry, { mode: "detect" }> => e.mode === "detect");
const progressTotalSteps = detectEntries.length > 0
? detectEntries.length + resolvedEntries.length + 4
: resolvedEntries.length + 3;
let batchResult: VisionDetectionBatchResult | null = null;
const globalConfig = getGlobalConfig(ctl);
const envEmbedPngMetadata = process.env.EMBED_PNG_METADATA;
const embedPngMetadata = getGlobalBoolean(
globalConfig,
"embedPngMetadata",
envEmbedPngMetadata === undefined ? defaultPluginSettings.embedPngMetadata : envEmbedPngMetadata !== "false"
);
let visionModelKey = "";
let detectionConfig: VisionDetectionAnalyzerConfig | null = null;
if (detectEntries.length > 0) {
let visionBaseUrl = getGlobalString(globalConfig, "embeddingBaseUrl", process.env.LMSTUDIO_VISION_API_BASE_URL || defaultPluginSettings.embeddingBaseUrl);
const visionApiKey = getGlobalString(globalConfig, "embeddingApiKey", process.env.LMSTUDIO_VISION_API_KEY || defaultPluginSettings.embeddingApiKey);
visionModelKey = getGlobalString(globalConfig, "qwen3VlModelPath", process.env.LMSTUDIO_VISION_MODEL_KEY || defaultPluginSettings.qwen3VlModelPath);
const envDetectMaxTokens = Number.parseInt(process.env.DETECT_MAX_TOKENS || "", 10);
const envDetectTemperature = Number.parseFloat(process.env.DETECT_TEMPERATURE || "");
const configuredDetectMaxTokens = Math.floor(getGlobalNumber(
globalConfig,
"detectMaxTokens",
Number.isFinite(envDetectMaxTokens) && envDetectMaxTokens > 0 ? envDetectMaxTokens : defaultPluginSettings.detectMaxTokens
));
const configuredDetectTemperature = getGlobalNumber(
globalConfig,
"detectTemperature",
Number.isFinite(envDetectTemperature) ? envDetectTemperature : defaultPluginSettings.detectTemperature
);
detectionConfig = {
task: taskArg,
odPrompt: getGlobalString(globalConfig, "qwen3VlOdPrompt", process.env.DETECT_OD_PROMPT || defaultPluginSettings.qwen3VlOdPrompt) || undefined,
maxTokens: configuredDetectMaxTokens,
temperature: configuredDetectTemperature,
timeoutMs: 120_000,
};
const tmpPaths: string[] = [];
const detectionItems: VisionAnalysisItem[] = [];
for (const entry of detectEntries) {
const visionBuf = await normalizeVisionBuffer(entry.previewBuf);
const tmpPath = path.join(primaryOutDir, `_tmp_annotate_src_${safeIdForFilename(entry.id)}_${Date.now()}.png`);
await fs.promises.writeFile(tmpPath, visionBuf);
tmpPaths.push(tmpPath);
detectionItems.push({ id: entry.id, filePath: tmpPath });
}
try {
reportToolStatus(statusCtx, `Detecting objects in ${detectEntries.length} image${detectEntries.length === 1 ? "" : "s"}...`);
reportToolStep(statusCtx, 1, progressTotalSteps, `Preparing ${detectEntries.length} image${detectEntries.length === 1 ? "" : "s"} for annotation detection...`);
const visionApi = resolveVisionApiFromEnv();
const ready = await ensureVisionModelReady(visionApi, {
baseUrl: visionBaseUrl,
apiKey: visionApiKey,
modelKey: visionModelKey,
status: (message) => { try { statusCtx.status?.(message); } catch {} },
llamaServer: visionApi === "llama-server" ? resolveLlamaServerConfig(globalConfig) : undefined,
});
if (!ready.ok) {
throw new Error(ready.error);
}
// llama-server strand: router's own exact tagged model id/address, not the configured ones.
if (ready.resolvedBaseUrl) visionBaseUrl = ready.resolvedBaseUrl;
if (ready.resolvedModelKey) visionModelKey = ready.resolvedModelKey;
batchResult = {
results: [],
totalInferenceTimeMs: 0,
backend: "vision-api",
};
for (let idx = 0; idx < detectionItems.length; idx++) {
const item = detectionItems[idx];
reportToolStep(statusCtx, idx + 2, progressTotalSteps, `Detecting objects in ${item.id} (${idx + 1}/${detectionItems.length})...`);
const singleResult = await detectLmStudioVisionBatch([item], {
...detectionConfig,
baseUrl: visionBaseUrl,
apiKey: visionApiKey,
model: visionModelKey,
});
batchResult.results.push(...singleResult.results);
batchResult.totalInferenceTimeMs += singleResult.totalInferenceTimeMs;
batchResult.backend = singleResult.backend;
}
try {
const totalObjects = batchResult.results.reduce((s, r) => s + (r.objects?.length ?? 0), 0);
const ms = Math.round(batchResult.totalInferenceTimeMs);
reportToolStep(statusCtx, detectEntries.length + 2, progressTotalSteps, `${totalObjects} object${totalObjects === 1 ? "" : "s"} found (${ms}ms); drawing boxes...`);
} catch {}
} finally {
for (const tp of tmpPaths) await fs.promises.unlink(tp).catch(() => {});
}
if (!batchResult || !batchResult.results.length) {
return {
content: [{ type: "text", text: "annotate_image: no results returned from detection API." }],
isError: true as const,
};
}
} else {
reportToolStatus(statusCtx, `Redrawing ${resolvedEntries.length} annotated image${resolvedEntries.length === 1 ? "" : "s"} from stored detections...`);
reportToolStep(statusCtx, 1, progressTotalSteps, `Redrawing ${resolvedEntries.length} annotated image${resolvedEntries.length === 1 ? "" : "s"} from stored detections...`);
}
// ── Per-entry: apply frameAdjust, draw, save ───────────────────────
const variantPreviewSpec = (VARIANT_FULL_CONFIG as any).preview;
const stamp = isoStampCompact();
let nextI = Math.max(1, st.counters?.nextImageI ?? 1);
const imageRecordsForState: any[] = [];
const resultEntries: Array<{
id: string;
i: number;
isRedraw: boolean;
task: string;
detObjects: StoredDetection[];
imageWidth: number;
imageHeight: number;
savedPath: string;
savedFileUrl: string;
savedSize: number;
preview: any | null;
httpOriginal: string;
httpPreview: string;
inferenceTimeMs: number;
}> = [];
const httpBase = await getHealthyServerBaseUrl();
let detectResultIdx = 0;
let resolvedIdx = 0;
const drawBaseStep = detectEntries.length > 0 ? detectEntries.length + 3 : 2;
for (const entry of resolvedEntries) {
reportToolStep(statusCtx, drawBaseStep + resolvedIdx, progressTotalSteps, `Drawing annotation for ${entry.id} (${resolvedIdx + 1}/${resolvedEntries.length})...`);
resolvedIdx++;
let rawBboxes: [number, number, number, number][];
let imgW: number;
let imgH: number;
let detObjects: StoredDetection[];
let isRedraw: boolean;
let entryTask: string;
let inferenceTimeMs = 0;
let bboxesAlreadyAdjusted = false;
if (entry.mode === "redraw") {
imgW = entry.imageWidth;
imgH = entry.imageHeight;
isRedraw = true;
entryTask = entry.task;
if (globalLabels !== undefined && globalLabels.length > 0) {
// detectLabel mode: resolve label/index pairs, draw ONLY the selected detections
let labels = [...globalLabels];
let indices = [...globalIndices];
// Option A: single label + no explicit index → auto-expand to all detections for that label
if (labels.length === 1 && indices.length === 0) {
const allIndices = expandDetectIndices(entry.detections, labels[0]);
if (allIndices.length > 1) {
labels = Array(allIndices.length).fill(labels[0]);
indices = allIndices;
}
}
// Option B: single label + multiple explicit indices → expand labels to match
if (labels.length === 1 && indices.length > 1) {
labels = Array(indices.length).fill(labels[0]);
}
// Pre-group detections by lowercase label once — O(1) lookup per box.
const detByLabel = new Map<string, StoredDetection[]>();
for (const d of entry.detections) {
const key = d.label.toLowerCase();
if (!detByLabel.has(key)) detByLabel.set(key, []);
detByLabel.get(key)!.push(d);
}
const resolvedBoxes: { det: StoredDetection; bbox: [number, number, number, number] }[] = [];
for (let li = 0; li < labels.length; li++) {
const label = labels[li];
const idx = indices[li] ?? indices[0] ?? 0;
const selectedDet = detByLabel.get(label.toLowerCase())?.[idx];
if (!selectedDet) {
const available = [...new Set(entry.detections.map((d) => d.label))].join(", ");
return {
content: [{ type: "text", text: `annotate_image: label '${label}' (index ${idx}) not found in stored detections. Available: ${available || "(none)"}` }],
isError: true as const,
};
}
// Coord overrides: resolveCoordOverride per axis, fall back to stored value.
// Applies to all selected boxes; scalar = same for all, array = parallel (null = keep stored).
const ox1 = hasAnyManualCoord ? resolveCoordOverride(manualX1 as any, li) : undefined;
const oy1 = hasAnyManualCoord ? resolveCoordOverride(manualY1 as any, li) : undefined;
const ox2 = hasAnyManualCoord ? resolveCoordOverride(manualX2 as any, li) : undefined;
const oy2 = hasAnyManualCoord ? resolveCoordOverride(manualY2 as any, li) : undefined;
const bbox: [number, number, number, number] = [
ox1 ?? selectedDet.bbox.x1,
oy1 ?? selectedDet.bbox.y1,
ox2 ?? selectedDet.bbox.x2,
oy2 ?? selectedDet.bbox.y2,
];
resolvedBoxes.push({ det: selectedDet, bbox });
}
rawBboxes = resolvedBoxes.map(({ bbox }) => applyFrameAdjust(bbox, globalFrameAdjust, imgW, imgH));
detObjects = resolvedBoxes.map(({ det, bbox }) => ({
...det,
bbox: { x1: bbox[0], y1: bbox[1], x2: bbox[2], y2: bbox[3] },
}));
bboxesAlreadyAdjusted = true;
} else if (globalIndices.length > 0) {
// No detectLabel but explicit detectIndex: select stored detections by position.
const selected: StoredDetection[] = [];
for (const idx of globalIndices) {
const det = entry.detections[idx];
if (!det) {
return {
content: [{ type: "text", text: `annotate_image: detectIndex ${idx} out of range (${entry.detections.length} stored detections).` }],
isError: true as const,
};
}
selected.push(det);
}
rawBboxes = selected.map((d) => [d.bbox.x1, d.bbox.y1, d.bbox.x2, d.bbox.y2] as [number, number, number, number]);
detObjects = selected;
bboxesAlreadyAdjusted = false;
} else {
// No detectLabel, no detectIndex: draw all stored boxes, apply frameAdjust to all
rawBboxes = entry.detections.map((d) => [d.bbox.x1, d.bbox.y1, d.bbox.x2, d.bbox.y2] as [number, number, number, number]);
detObjects = entry.detections;
}
} else {
const detResult = batchResult!.results[detectResultIdx++];
rawBboxes = detResult.objects.map((o) => o.bbox as [number, number, number, number]);
imgW = detResult.imageWidth;
imgH = detResult.imageHeight;
detObjects = detResult.objects.map((o) => ({
label: o.label,
bbox: { x1: o.bbox[0], y1: o.bbox[1], x2: o.bbox[2], y2: o.bbox[3] },
crop: { cropLeft: o.cropLeft, cropRight: o.cropRight, cropTop: o.cropTop, cropBottom: o.cropBottom },
}));
isRedraw = false;
entryTask = taskArg;
inferenceTimeMs = detResult.inferenceTimeMs ?? 0;
}
const adjustedBboxes = bboxesAlreadyAdjusted
? rawBboxes
: rawBboxes.map((bbox) => applyFrameAdjust(bbox, globalFrameAdjust, imgW, imgH));
const annotatedBuf = await drawBboxesOnImage(entry.origBuf, adjustedBboxes, {
sourceDims: { width: imgW, height: imgH },
palette: false,
color: globalColor,
lineWeight: globalLineWeight,
});
const inheritedInference = entry.mode === "redraw" ? entry.analysisMetadata?.inference : undefined;
const analysis: PngAnalysisMetadata = {
schema: "ceveyne.image-analysis/v1",
tool: "annotate_image",
sourceNotation: entry.id,
inference: entry.mode === "redraw"
? inheritedInference
? { ...inheritedInference, reused: true }
: undefined
: {
model: visionModelKey,
...(entryTask ? { query: entryTask } : {}),
detectorPromptSha256: createHash("sha256").update(detectionConfig?.odPrompt ?? "").digest("hex"),
maxTokens: detectionConfig?.maxTokens,
temperature: detectionConfig?.temperature,
},
render: { color: globalColor, lineWeight: globalLineWeight, frameAdjust: globalFrameAdjust },
detections: detObjects.map((detection, index) => ({
label: detection.label,
bbox: {
x1: adjustedBboxes[index][0],
y1: adjustedBboxes[index][1],
x2: adjustedBboxes[index][2],
y2: adjustedBboxes[index][3],
},
})),
};
const savedBuffer = embedPngMetadata
? injectXmpIntoBuffer(annotatedBuf, {
...(entryTask ? { prompt: entryTask } : {}),
...(analysis.inference?.model ? { model: analysis.inference.model } : {}),
mode: isRedraw ? "image_annotation_redraw" : "image_annotation",
generatedBy: `${getSelfPluginIdentifier()}/annotate_image`,
creatorTool: `${getSelfPluginIdentifier()}/annotate_image`,
analysis,
})
: annotatedBuf;
const currentI = nextI++;
const baseName = `image-${stamp}-i${currentI}`;
const savedPath = path.join(primaryOutDir, `${baseName}.png`);
await fs.promises.writeFile(savedPath, savedBuffer);
const savedFileUrl = pathToFileURL(savedPath).toString();
const savedSize = savedBuffer.length;
let preview: any = null;
try {
const p = await generatePreviewFromBuffer(savedBuffer, primaryOutDir, `${baseName}.png`, variantPreviewSpec);
preview = {
ok: true as const,
filePath: p.previewAbs,
fileName: p.previewFilename,
fileUrl: pathToFileURL(p.previewAbs).toString(),
size_bytes: p.data.length,
width: p.width,
height: p.height,
mimeType: "image/jpeg" as const,
dataBase64: p.data.toString("base64"),
};
} catch (e) {
console.warn(`[annotate_image] preview generation failed for ${entry.id}:`, String(e));
}
const httpOriginal = httpBase
? toHttpOriginalUrl(`${baseName}.png`, httpBase, currentLmChatId || undefined) : "";
const httpPreview = (() => {
if (!httpBase || !currentLmChatId || !preview?.fileName) return "";
return toHttpPreviewUrl(preview.fileName, httpBase, currentLmChatId);
})();
imageRecordsForState.push({
filename: `${baseName}.png`,
preview: preview ? `preview-${baseName}.jpg` : undefined,
i: currentI,
sourceTool: `${getSelfPluginIdentifier()}/annotate_image`,
detectSource: entry.id,
task: entryTask,
annotateColor: globalColor,
annotateLineWeight: globalLineWeight,
annotateFrameAdjust: globalFrameAdjust,
imageWidth: imgW,
imageHeight: imgH,
analysisMetadata: analysis,
detections: detObjects.map((d) => ({
label: d.label,
bbox: { x1: d.bbox.x1, y1: d.bbox.y1, x2: d.bbox.x2, y2: d.bbox.y2 },
crop: d.crop ?? {},
})),
});
resultEntries.push({
id: entry.id,
i: currentI,
isRedraw,
task: entryTask,
detObjects,
imageWidth: imgW,
imageHeight: imgH,
savedPath,
savedFileUrl,
savedSize,
preview,
httpOriginal,
httpPreview,
inferenceTimeMs,
});
}
// ── Update state ──────────────────────────────────────────────────
reportToolStep(statusCtx, progressTotalSteps - 1, progressTotalSteps, "Updating image state and audit log...");
try {
const stateForUpdate = await readState(primaryOutDir);
const appendResult = appendImages(stateForUpdate, imageRecordsForState);
if (appendResult.changed) {
await writeStateAtomic(primaryOutDir, stateForUpdate);
}
} catch (e) {
console.warn("[annotate_image] state update failed:", String(e));
}
// ── Audit log ─────────────────────────────────────────────────────
// effectiveRequestId is reused below in each summary entry, so a caller (the MCP adapter)
// can always name its HTML report after this exact audit requestId, even for a batch call
// whose top-level result is an array (see summaries below).
const effectiveRequestId = requestId ?? `${Date.now()}-${Math.random().toString(36).slice(2, 8)}`;
try {
const audit = buildAuditLogger({ backend: "annotate_image", mode: "annotate_image" as any, requestId: effectiveRequestId });
if (currentLmChatId) audit.setChatId(currentLmChatId);
audit.setUserRequest({ targets: rawTargets, task: taskArg, color: globalColor, lineWeight: globalLineWeight, frameAdjust: globalFrameAdjust } as any);
audit.setOutput({
images: resultEntries.map((r) => ({
id: r.id,
i: r.i,
redraw: r.isRedraw,
detections: r.detObjects.length,
path: r.savedPath,
url: r.savedFileUrl,
bytes: r.savedSize,
...(r.httpOriginal ? { http_url: r.httpOriginal } : {}),
...(r.preview ? { preview_path: r.preview.filePath, preview_url: r.preview.fileUrl } : {}),
...(r.httpPreview ? { http_preview_url: r.httpPreview } : {}),
})),
} as any);
await audit.write();
} catch {}
// ── Assemble result ───────────────────────────────────────────────
reportToolStep(statusCtx, progressTotalSteps, progressTotalSteps, "Assembling annotation result...");
const summaries = resultEntries.map((r) => ({
tool: "annotate_image",
requestId: effectiveRequestId,
source: r.id,
i: r.i,
redraw: r.isRedraw,
color: globalColor,
lineWeight: globalLineWeight,
frameAdjust: globalFrameAdjust,
...(r.inferenceTimeMs > 0 ? { inferenceTimeMs: r.inferenceTimeMs } : {}),
detections: r.detObjects.map((d) => ({
label: d.label,
bbox: { x1: d.bbox.x1, y1: d.bbox.y1, x2: d.bbox.x2, y2: d.bbox.y2 },
})),
}));
const envPreviewRaw = process.env["PREVIEW_IN_CHAT"];
const previewInChat =
envPreviewRaw === undefined
? true
: envPreviewRaw === "1" || envPreviewRaw.toLowerCase() === "true";
// Build target notations for hints (e.g. ["i9", "i10"])
const resultNotations = resultEntries.map((r) => `i${r.i}`);
const targetsJson = JSON.stringify(resultNotations);
const reviewHintFalse =
`Carefully examine the preview to make absolutely sure that the object detection matches your intent. Registered as ${resultNotations.join(", ")}. Use review_image({"targets":${targetsJson}}) to review, or annotate_image({"targets":${targetsJson}}) to apply corrections.`;
const reviewHintTrue =
`Carefully examine the preview to make absolutely sure that the object detection matches your intent. This is an image file. Present the image to the user by using the markdown above. Registered as ${resultNotations.join(", ")}. Use review_image({"targets":${targetsJson}}) to review, or annotate_image({"targets":${targetsJson}}) to apply corrections.`;
const content: any[] = [];
for (const r of resultEntries) {
const fallbackPreviewUrl = r.preview?.fileUrl || r.savedFileUrl;
if (previewInChat && r.preview) {
const fname = String(r.preview.fileName || "");
content.push({
type: "image",
fileName: fname,
mimeType: r.preview.mimeType,
markdown: ``,
$hint: reviewHintTrue,
} as any);
}
// TODO: Restore when LM Studio renders file/HTTP links again
// content.push({ type: "text", text: `Preview i${r.i}: ${r.httpPreview || fallbackPreviewUrl}` });
// content.push({ type: "text", text: `Original i${r.i}: ${r.httpOriginal || r.savedFileUrl}` });
}
if (batchResult && batchResult.totalInferenceTimeMs > 0) {
content.push({ type: "text", text: `Total inference time: ${Math.round(batchResult.totalInferenceTimeMs)}ms` });
}
content.push({
type: "text",
text: JSON.stringify(summaries.length === 1 ? summaries[0] : summaries),
...(previewInChat ? {} : { $hint: reviewHintFalse }),
});
return { content };
} catch (error) {
return {
content: [{ type: "text", text: `annotate_image failed: ${(error as Error).message || String(error)}` }],
isError: true as const,
};
}
}
import path from "path";
import fs from "fs";
import { createHash } from "crypto";
import { pathToFileURL } from "url";
import { z } from "zod";
import type { ToolsProviderController } from "@lmstudio/sdk";
import {
formatToolMetaBlock,
readCameraImageMetadata,
syncAttachmentsToState,
getActiveChatContext,
resolveActiveLMStudioChatId,
getLMStudioWorkingDir,
readState,
writeStateAtomic,
generatePreviewFromBuffer,
appendImages,
getHealthyServerBaseUrl,
toHttpOriginalUrl,
toHttpPreviewUrl,
buildAuditLogger,
getSelfPluginIdentifier,
VARIANT_FULL_CONFIG,
drawthingsLimits,
getSize,
encodeJpegPreviewFromBuffer,
} from "../core-bundle.mjs";
import { type ChatMediaState } from "../state.js";
import { readPngGenerationMeta, formatGenerationMeta, type PngGenerationMeta } from "../helpers/readPngMetadata.js";
import { reportToolStatus, reportToolStep, type ToolStatusContext } from "../helpers/toolProgress.js";
import { defaultPluginSettings, globalConfigSchematics } from "../config.js";
import { drawBboxesOnImage } from "../helpers/drawBboxesOnImage.js";
import { injectXmpIntoBuffer, type PngAnalysisMetadata } from "../helpers/pngMetadata.js";
import {
analyzeLmStudioVisionBatch,
ensureVisionModelReady,
detectLmStudioVisionBatch,
type LmStudioVisionAnalyzerConfig,
type VisionAnalysisItem,
type VisionDetectionAnalyzerConfig,
type VisionDetectionBatchResult,
type VisionApiName,
} from "../services/VisionAnalyzer.js";
/** The LM-Studio-plugin entrypoint has no preset concept and only ever talks to LM Studio itself, defaults to "bionic" when VISION_API isn't set (MCP's applyMcpConfig() is the only writer of this env var). */
function resolveVisionApiFromEnv(): VisionApiName {
const raw = (process.env.LMSTUDIO_VISION_API || "").trim().toLowerCase();
return raw === "generic" || raw === "unsloth" || raw === "llama-server" ? raw : "bionic";
}
// ─────────────────────────────────────────────────────────────────────────────
// Shared helpers (previously duplicated across src/tools/*.ts, verified identical)
// ─────────────────────────────────────────────────────────────────────────────
export function parsePrefixedNotation(
s: string
): { pool: "attachment" | "image" | "variant" | "picture"; index: number } | null {
const t = String(s || "").trim().toLowerCase();
const m = t.match(/^([avip])(\d+)$/);
if (!m) return null;
const idx = Math.max(1, parseInt(m[2], 10));
const pool =
m[1] === "a" ? "attachment" : m[1] === "v" ? "variant" : m[1] === "i" ? "image" : "picture";
return { pool, index: idx };
}
// A target's `id` is normally a short notation (e.g. "i1") but the MCP adapter may also pass a
// resolved absolute path (see made-for-bionic-core's resolveMcpSourcePathTarget, called from
// mcp/index.ts before this handler runs) — never embed it in a filename unsanitized.
function safeIdForFilename(id: string): string {
return path.basename(id).replace(/[^a-zA-Z0-9._-]/g, "_").slice(0, 80);
}
// A target that isn't aN/vN/iN/pN notation is a scratchpad basename or an absolute path (see
// generate-image's core/tools.ts for the same pattern) — deliberately local/non-shared. The MCP
// layer (src/mcp/index.ts) already resolves and containment-checks these against the bound
// scratchpad before this handler ever runs (made-for-bionic-core's resolveMcpSourceToken()) — by
// the time a target reaches here it is either aN/vN/iN/pN notation or an already-validated
// absolute path, so this is a plain existence check, not a security boundary.
async function resolveTargetPathToken(raw: string, baseDir: string | undefined): Promise<string | null> {
const trimmed = raw.trim();
const candidate = path.isAbsolute(trimmed)
? trimmed
: baseDir && trimmed === path.basename(trimmed)
? path.join(baseDir, trimmed)
: null;
if (!candidate) return null;
const exists = await fs.promises.stat(candidate).then((s) => s.isFile()).catch(() => false);
return exists ? candidate : null;
}
// Normalizes a path-specified source buffer for the vision model, mirroring find-images'
// previewPayloadBytes policy: images already within drawthingsLimits.previewMaxSum (w + h) are
// returned untouched (no resize, no format change — a fitting PNG stays a PNG); only oversize
// originals are re-encoded to a JPEG preview capped at previewMaxSum.
async function normalizeVisionBuffer(buf: Buffer): Promise<Buffer> {
const maxSum = drawthingsLimits.previewMaxSum;
const { width, height } = await getSize(buf);
if (width <= 0 || height <= 0 || width + height <= maxSum) return buf;
const { data } = await encodeJpegPreviewFromBuffer(buf, {
maxDim: maxSum,
quality: drawthingsLimits.previewQuality,
mode: "sum",
maxSum,
});
return data;
}
function isoStampCompact(): string {
const d = new Date();
const year = d.getUTCFullYear();
const month = String(d.getUTCMonth() + 1).padStart(2, "0");
const day = String(d.getUTCDate()).padStart(2, "0");
const hours = String(d.getUTCHours()).padStart(2, "0");
const minutes = String(d.getUTCMinutes()).padStart(2, "0");
const seconds = String(d.getUTCSeconds()).padStart(2, "0");
const millis = String(d.getUTCMilliseconds()).padStart(3, "0");
return `${year}${month}${day}T${hours}${minutes}${seconds}${millis}Z`;
}
function getGlobalConfig(ctl: ToolsProviderController): any | null {
const ctlAny = ctl as any;
const getter = ctlAny.getGlobalPluginConfig || ctlAny.getGlobalConfig;
if (!getter) return null;
try {
return getter.call(ctl, globalConfigSchematics);
} catch {
return null;
}
}
function getGlobalString(gcfg: any | null, key: string, fallback: string): string {
try {
const value = gcfg?.get(key);
return typeof value === "string" ? value : fallback;
} catch {
return fallback;
}
}
function getGlobalNumber(gcfg: any | null, key: string, fallback: number): number {
try {
const value = gcfg?.get(key);
return typeof value === "number" && Number.isFinite(value) ? value : fallback;
} catch {
return fallback;
}
}
function getGlobalBoolean(gcfg: any | null, key: string, fallback: boolean): boolean {
try {
const value = gcfg?.get(key);
return typeof value === "boolean" ? value : fallback;
} catch {
return fallback;
}
}
/** Shared by all 3 ensureVisionModelReady() call sites -- only consumed when VISION_API="llama-server" (see ensureLlamaServerVisionModelReady). GUI setting > MCP env var > defaultPluginSettings, matching every other config field in this file. */
function resolveLlamaServerConfig(globalConfig: any | null): { port: number; binaryPath: string; ctxSize: number; modelsMax: number; idleTtlMinutes: number } {
const envPort = Number.parseInt(process.env.LMSTUDIO_LLAMA_SERVER_PORT || "", 10);
const envCtxSize = Number.parseInt(process.env.LMSTUDIO_LLAMA_SERVER_CTX_SIZE || "", 10);
const envModelsMax = Number.parseInt(process.env.LMSTUDIO_LLAMA_SERVER_MODELS_MAX || "", 10);
const envIdleTtlMinutes = Number.parseInt(process.env.LMSTUDIO_LLAMA_SERVER_IDLE_TTL_MINUTES || "", 10);
return {
port: Math.floor(getGlobalNumber(
globalConfig,
"llamaServerPort",
Number.isFinite(envPort) && envPort > 0 ? envPort : defaultPluginSettings.llamaServerPort
)),
binaryPath: getGlobalString(globalConfig, "llamaServerBinary", process.env.LMSTUDIO_LLAMA_SERVER_BINARY || defaultPluginSettings.llamaServerBinary),
ctxSize: Math.floor(getGlobalNumber(
globalConfig,
"llamaServerCtxSize",
Number.isFinite(envCtxSize) && envCtxSize > 0 ? envCtxSize : defaultPluginSettings.llamaServerCtxSize
)),
modelsMax: Math.floor(getGlobalNumber(
globalConfig,
"llamaServerModelsMax",
Number.isFinite(envModelsMax) && envModelsMax > 0 ? envModelsMax : defaultPluginSettings.llamaServerModelsMax
)),
idleTtlMinutes: Math.floor(getGlobalNumber(
globalConfig,
"llamaServerIdleTtlMinutes",
Number.isFinite(envIdleTtlMinutes) && envIdleTtlMinutes >= 0 ? envIdleTtlMinutes : defaultPluginSettings.llamaServerIdleTtlMinutes
)),
};
}
/**
* Flexible targets schema: accepts array OR comma/space-separated string.
* Reduces parse errors when models output "a1, v2" instead of ["a1", "v2"].
*/
export const FlexibleTargetsList = z
.union([
z.string().transform((s) => (s.match(/[aivp]\d+/gi) ?? []).map((x) => x.toLowerCase())),
z.array(z.string()),
])
.refine((arr) => arr.length >= 1, "targets must contain at least one notation")
.refine((arr) => arr.length <= 16, "targets must contain at most 16 notations");
// ─────────────────────────────────────────────────────────────────────────────
// analyse_image
// ─────────────────────────────────────────────────────────────────────────────
export const AnalyseImageParamsShape = {
targets: FlexibleTargetsList.describe(
"One or more image notations to process. Each notation is a letter followed by a number: " +
"a=attachment (a1, a2, …), i=generated image (i1, i2, …), v=variant (v1, v2, …), p=picture (p1, p2, …). " +
"Also accepts a filename in the scratchpad folder or an absolute path. Pass as a JSON array, e.g. [\"a1\", \"i3\"]."
),
prompt: z.string().optional().describe("Optional prompt for the vision model. Empty = model default."),
} satisfies Record<string, z.ZodTypeAny>;
type ParsedTargets = { a: number[]; v: number[]; i: number[]; p: number[] };
function parseTargets(targets: string[]): {
parsed: ParsedTargets;
invalid: string[];
} {
const parsed: ParsedTargets = { a: [], v: [], i: [], p: [] };
const invalid: string[] = [];
for (const raw of targets) {
const s = typeof raw === "string" ? raw.trim() : "";
const m = /^([avip])(\d+)$/i.exec(s);
if (!m) {
invalid.push(String(raw));
continue;
}
const kind = m[1].toLowerCase() as "a" | "v" | "i" | "p";
const n = parseInt(m[2], 10);
if (!Number.isFinite(n) || n <= 0) {
invalid.push(String(raw));
continue;
}
parsed[kind].push(n);
}
// Unique + sort for stable behavior
(Object.keys(parsed) as Array<keyof ParsedTargets>).forEach((k) => {
parsed[k] = Array.from(new Set(parsed[k])).sort((a, b) => a - b);
});
return { parsed, invalid };
}
function getAvailable(state: ChatMediaState) {
const availableA = (state.attachments || [])
.map((x: any) => (typeof x?.a === "number" ? x.a : undefined))
.filter((x: any) => typeof x === "number" && x > 0)
.sort((a: number, b: number) => a - b);
const availableV = (state.variants || [])
.map((x: any) => (typeof x?.v === "number" ? x.v : undefined))
.filter((x: any) => typeof x === "number" && x > 0)
.sort((a: number, b: number) => a - b);
const availableI = (state.images || [])
.map((x: any) => (typeof x?.i === "number" ? x.i : undefined))
.filter((x: any) => typeof x === "number" && x > 0)
.sort((a: number, b: number) => a - b);
const availableP = (state.pictures || [])
.map((x: any) => (typeof x?.p === "number" ? x.p : undefined))
.filter((x: any) => typeof x === "number" && x > 0)
.sort((a: number, b: number) => a - b);
return { availableA, availableV, availableI, availableP };
}
async function ensurePreviewExists(chatWd: string, previewRel: string): Promise<void> {
const pAbs = path.join(chatWd, previewRel);
await fs.promises.access(pAbs, fs.constants.F_OK);
}
function classifyVisionError(errMsg: string): string {
if (/\b503\b/.test(errMsg)) {
return "The Vision API is reachable, but the configured vision model is not available for inference. Check the configured vision model key and loaded model state.";
}
if (/aborted|aborterror|timed out|timeout/i.test(errMsg)) {
return "The Vision API request timed out before the model returned.";
}
if (/ECONNREFUSED|ENOTFOUND|ECONNRESET|network socket|fetch failed/i.test(errMsg)) {
return "The Vision API is not reachable. Check the configured vision/embedding API base URL and try again.";
}
return /Vision API/i.test(errMsg) ? errMsg : `Vision API error: ${errMsg}`;
}
function analyzeTimeoutMs(itemCount: number): number {
return Math.min(600_000, Math.max(180_000, itemCount * 60_000));
}
/** One analysed item's structured content — same data as its plain-text block in the returned string, for callers (the MCP HTML report) that want per-item rendering instead of re-parsing the text. */
export type AnalyseImageItemReport = {
id: string;
filePath?: string;
displayName?: string;
visual?: string;
/** Present when origPath is a PNG with an embedded generation-metadata chunk — lets callers (the MCP HTML report) render fields individually instead of parsing formatGenerationMeta()'s text. */
meta?: PngGenerationMeta;
/** Human-readable fallback note when `meta` isn't available (EXIF JSON, "no embedded metadata", etc.) — same text as the corresponding line in the plain-text result. */
metaNote?: string;
};
export async function handleAnalyseImage(
args: any,
ctl: ToolsProviderController,
statusCtx: ToolStatusContext = {},
requestId?: string,
onItem?: (item: AnalyseImageItemReport) => void
): Promise<string> {
const visionTmpPaths: string[] = [];
const visionTmpIds = new Set<string>();
try {
// Tolerant mode: drop invalid items instead of failing
const strict = false;
let targets: string[];
if (Array.isArray(args?.targets)) {
targets = args.targets;
} else if (typeof args?.targets === "string" && args.targets.trim()) {
const trimmed = args.targets.trim();
if (trimmed.startsWith("[")) {
try {
const parsed = JSON.parse(trimmed);
targets = Array.isArray(parsed) ? parsed.map((s: any) => String(s).trim()).filter(Boolean) : [];
} catch {
targets = trimmed.split(/[\s,]+/).map((s: string) => s.trim()).filter(Boolean);
}
} else {
// A bare notation or comma/space-separated list (e.g. "i1" or "i1, i2") — the same
// fallback detect_object/annotate_image's own string parsing already has. The MCP path
// never re-runs this handler's args through FlexibleTargetsList's zod transform (only the
// LM-Studio plugin path does, via @lmstudio/sdk's tool() wrapper), so this handler's own
// string parsing must accept the same shapes on its own.
targets = trimmed.split(/[\s,]+/).map((s: string) => s.trim()).filter(Boolean);
}
} else {
targets = [];
}
const prompt = typeof args?.prompt === "string" ? args.prompt : "";
const globalConfig = getGlobalConfig(ctl);
const configuredVisionPrompt = getGlobalString(
globalConfig,
"visionPrompt",
process.env.VISION_PROMPT || defaultPluginSettings.visionPrompt
);
const effectivePrompt = (prompt || configuredVisionPrompt || "").trim();
// Parse explicit notations (aN/vN/iN/pN)
const { parsed, invalid: invalidRaw } = parseTargets(targets);
// LM-Studio plugin path: ctl.getWorkingDirectory() (LM-Studio SDK method, only present when
// ctl is a real ToolsProviderController). MCP path: ctl is `{}` and has no such method — fall
// back to the same chat-context-bridge resolution detect_object/annotate_image already use
// (bindChatContextToScratchpad() sets this before the MCP adapter calls this handler).
// currentLmChatId is resolved unconditionally (also used for this call's audit log entry).
let currentLmChatId: string | null = null;
let boundWorkingDir: string | null = null;
try {
const chatCtx = await getActiveChatContext();
if ((chatCtx as any)?.chatId) currentLmChatId = (chatCtx as any).chatId;
if ((chatCtx as any)?.workingDir) boundWorkingDir = (chatCtx as any).workingDir;
} catch {}
if (!currentLmChatId) {
try {
const resolved = await resolveActiveLMStudioChatId();
if ((resolved as any)?.ok) currentLmChatId = (resolved as any).chatId;
} catch {}
}
let workingDir: string | undefined;
try {
workingDir = (ctl as any)?.getWorkingDirectory?.();
} catch {
workingDir = undefined;
}
if (typeof workingDir !== "string" || !workingDir.trim()) {
workingDir = boundWorkingDir || (currentLmChatId ? getLMStudioWorkingDir(currentLmChatId) : undefined);
}
if (typeof workingDir !== "string" || !workingDir.trim()) {
return "analyse_image failed: working directory not available.";
}
const chatWd = workingDir;
// Sync attachments from conversation.json → chat_media_state.json.
// maxPreviewAttachments=MAX: importAttachmentBatch generates previews for all attachments
// and writes them to state atomically — the canonical path via draw-things-chat-core.
try {
await syncAttachmentsToState(chatWd, false, Number.MAX_SAFE_INTEGER);
} catch (syncErr: any) {
console.warn("[analyse_image] attachment sync failed (non-fatal):", syncErr?.message ?? syncErr);
}
const state = await readState(chatWd);
const { availableA, availableV, availableI, availableP } = getAvailable(state);
// If any invalid notations and strict=true, reject
if (invalidRaw.length > 0 && strict) {
return `analyse_image failed: invalid targets: ${invalidRaw
.map((s) => JSON.stringify(s))
.join(", ")}`;
}
// Collect analysis items (id + preview file path)
const analysisItems: VisionAnalysisItem[] = [];
// Map from notation id → absolute path of original file (for XMP reading)
const originalFilePaths = new Map<string, string>();
// Map from notation id → human-readable filename for the result
const displayNames = new Map<string, string>();
const missingNotations = new Set<string>();
const missingDetails: string[] = [];
// Helper: resolve & validate preview for a notation
const addItem = async (
notation: string,
rec: any,
previewField: string,
originalAbsPath?: string,
displayName?: string
): Promise<boolean> => {
if (!rec) {
missingNotations.add(notation);
missingDetails.push(notation);
return false;
}
const previewRel =
typeof rec[previewField] === "string" ? String(rec[previewField]) : "";
if (!previewRel.trim()) {
missingNotations.add(notation);
missingDetails.push(`${notation} (missing preview)`);
return false;
}
try {
await ensurePreviewExists(chatWd, previewRel);
analysisItems.push({
id: notation,
filePath: path.join(chatWd, previewRel),
});
if (originalAbsPath) {
originalFilePaths.set(notation, originalAbsPath);
}
const dn = displayName || (originalAbsPath ? path.basename(originalAbsPath) : undefined);
if (dn) {
displayNames.set(notation, dn);
}
return true;
} catch {
missingNotations.add(notation);
missingDetails.push(`${notation} (preview file missing)`);
return false;
}
};
// Process attachments (aN)
// Preview generation + state write already handled by syncAttachmentsToState above.
for (const n of parsed.a) {
const rec = (state.attachments || []).find((x: any) => x?.a === n);
const origAbs: string | undefined =
(rec?.originAbs as string | undefined) ?? (rec?.filename ? path.join(chatWd, rec.filename as string) : undefined);
// originalName holds the real user-visible filename (e.g. "Katze.png")
const origName: string | undefined =
typeof rec?.originalName === "string" && rec.originalName ? rec.originalName as string : undefined;
await addItem(`a${n}`, rec, "preview", origAbs, origName);
}
// Process variants (vN)
for (const n of parsed.v) {
const rec = (state.variants || []).find((x: any) => x?.v === n);
const origAbs: string | undefined = rec?.filename ? path.join(chatWd, rec.filename as string) : undefined;
await addItem(`v${n}`, rec, "preview", origAbs);
}
// Process images (iN)
for (const n of parsed.i) {
const rec = (state.images || []).find((x: any) => x?.i === n);
const origAbs: string | undefined = rec?.filename ? path.join(chatWd, rec.filename as string) : undefined;
await addItem(`i${n}`, rec, "preview", origAbs);
}
// Process pictures (pN)
for (const n of parsed.p) {
const rec = (state.pictures || []).find((x: any) => x?.p === n);
const origAbs: string | undefined = rec?.filename ? path.join(chatWd, rec.filename as string) : undefined;
await addItem(`p${n}`, rec, "preview", origAbs);
}
// Entries that didn't match aN/vN/iN/pN — a scratchpad filename or an absolute path is
// also accepted as a target, same as detect_object/annotate_image/every other Bionic MCP tool.
for (const raw of invalidRaw) {
const pathToken = await resolveTargetPathToken(raw, chatWd);
if (pathToken) {
let visionPath = pathToken;
if (effectivePrompt) {
const buf = await fs.promises.readFile(pathToken);
const norm = await normalizeVisionBuffer(buf);
if (norm !== buf) {
const tmpPath = path.join(chatWd, `_tmp_analyse_src_${safeIdForFilename(raw)}_${Date.now()}.png`);
await fs.promises.writeFile(tmpPath, norm);
visionTmpPaths.push(tmpPath);
visionTmpIds.add(raw);
visionPath = tmpPath;
}
}
analysisItems.push({ id: raw, filePath: visionPath });
originalFilePaths.set(raw, pathToken);
displayNames.set(raw, path.basename(pathToken));
}
}
// If strict, fail on any missing items
if (missingDetails.length > 0 && strict) {
const hint =
`Available: ` +
`a=[${availableA.map((x) => `a${x}`).join(", ") || "(none)"}] ` +
`v=[${availableV.map((x) => `v${x}`).join(", ") || "(none)"}] ` +
`i=[${availableI.map((x) => `i${x}`).join(", ") || "(none)"}] ` +
`p=[${availableP.map((x) => `p${x}`).join(", ") || "(none)"}]`;
return `analyse_image failed: unknown/invalid targets: ${missingDetails.join(", ")}. ${hint}`;
}
// If no valid items after filtering, return early
if (analysisItems.length === 0) {
const hint =
`Available: ` +
`a=[${availableA.map((x) => `a${x}`).join(", ") || "(none)"}] ` +
`v=[${availableV.map((x) => `v${x}`).join(", ") || "(none)"}] ` +
`i=[${availableI.map((x) => `i${x}`).join(", ") || "(none)"}] ` +
`p=[${availableP.map((x) => `p${x}`).join(", ") || "(none)"}]`;
return `analyse_image: no valid targets found. ${hint}`;
}
let visionError: string | null = null;
const visionResults = new Map<string, string>();
let totalInferenceTimeMs: number | null = null;
if (effectivePrompt) {
const envServerMaxTokens = Number.parseInt(process.env.SERVER_MAX_TOKENS || "", 10);
const envServerTemperature = Number.parseFloat(process.env.SERVER_TEMPERATURE || "");
const configuredMaxTokens = Math.floor(getGlobalNumber(
globalConfig,
"serverMaxTokens",
Number.isFinite(envServerMaxTokens) && envServerMaxTokens > 0 ? envServerMaxTokens : defaultPluginSettings.serverMaxTokens
));
const configuredTemperature = getGlobalNumber(
globalConfig,
"serverTemperature",
Number.isFinite(envServerTemperature) ? envServerTemperature : defaultPluginSettings.serverTemperature
);
const lmStudioConfig: LmStudioVisionAnalyzerConfig = {
baseUrl: getGlobalString(globalConfig, "embeddingBaseUrl", process.env.LMSTUDIO_VISION_API_BASE_URL || defaultPluginSettings.embeddingBaseUrl),
apiKey: getGlobalString(globalConfig, "embeddingApiKey", process.env.LMSTUDIO_VISION_API_KEY || defaultPluginSettings.embeddingApiKey),
model: getGlobalString(globalConfig, "qwen3VlModelPath", process.env.LMSTUDIO_VISION_MODEL_KEY || defaultPluginSettings.qwen3VlModelPath),
prompt: effectivePrompt,
maxTokens: configuredMaxTokens,
temperature: configuredTemperature,
timeoutMs: analyzeTimeoutMs(1),
};
try {
const totalSteps = analysisItems.length + 2;
reportToolStatus(statusCtx, `Analyzing ${analysisItems.length} image${analysisItems.length === 1 ? "" : "s"}...`);
reportToolStep(statusCtx, 1, totalSteps, `Preparing ${analysisItems.length} image${analysisItems.length === 1 ? "" : "s"} for visual analysis...`);
const visionApi = resolveVisionApiFromEnv();
const ready = await ensureVisionModelReady(visionApi, {
baseUrl: lmStudioConfig.baseUrl,
apiKey: lmStudioConfig.apiKey,
modelKey: lmStudioConfig.model || "",
status: (message) => { try { statusCtx.status?.(message); } catch {} },
llamaServer: visionApi === "llama-server" ? resolveLlamaServerConfig(globalConfig) : undefined,
});
if (!ready.ok) {
throw new Error(ready.error);
}
// llama-server strand: router's own exact tagged model id/address, not the configured ones.
if (ready.resolvedBaseUrl) lmStudioConfig.baseUrl = ready.resolvedBaseUrl;
if (ready.resolvedModelKey) lmStudioConfig.model = ready.resolvedModelKey;
let totalMs = 0;
for (let idx = 0; idx < analysisItems.length; idx++) {
const item = analysisItems[idx];
reportToolStep(statusCtx, idx + 2, totalSteps, `Analyzing ${item.id} (${idx + 1}/${analysisItems.length})...`);
const batchResult = await analyzeLmStudioVisionBatch([item], lmStudioConfig);
for (const r of batchResult.results) {
visionResults.set(r.id, r.text.trim() || "(no description)");
}
totalMs += batchResult.totalInferenceTimeMs;
}
totalInferenceTimeMs = totalMs;
reportToolStep(statusCtx, totalSteps, totalSteps, "Formatting analysis results...");
} catch (e) {
visionError = classifyVisionError((e as Error).message || String(e));
}
}
const includeGenMeta = process.env.INCLUDE_GENERATION_METADATA !== "false";
// Build structured result
const lines: string[] = [];
lines.push(`Analysis results (${analysisItems.length} image${analysisItems.length !== 1 ? "s" : ""}):`)
lines.push("");
if (visionError) {
lines.push(`Note: Visual analysis unavailable — ${visionError}`);
lines.push("");
}
for (const item of analysisItems) {
const { id } = item;
const displayName = displayNames.get(id);
const origPath = originalFilePaths.get(id);
const header = displayName ? `${id} — ${displayName}` : id;
lines.push(`- ${header}`);
let visual: string | undefined;
if (effectivePrompt) {
visual = visionError ? "(not available)" : (visionResults.get(id) ?? "(no description)");
lines.push(` VISUAL: ${visual}`);
}
let metaText: string | undefined;
let meta: PngGenerationMeta | undefined;
let metaNote: string | undefined;
if (includeGenMeta) {
if (origPath && origPath.toLowerCase().endsWith(".png")) {
meta = readPngGenerationMeta(origPath) ?? undefined;
if (meta) {
metaText = formatGenerationMeta(meta);
} else {
metaText = ` (No embedded generation metadata)`;
metaNote = "(No embedded generation metadata)";
}
} else if (origPath && /\.jpe?g$/i.test(origPath)) {
const metadata = await readCameraImageMetadata(origPath);
metaText = Object.keys(metadata).length > 0 ? ` EXIF JSON: ${JSON.stringify(metadata)}` : ` (No embedded EXIF metadata)`;
metaNote = Object.keys(metadata).length > 0 ? `EXIF JSON: ${JSON.stringify(metadata)}` : "(No embedded EXIF metadata)";
} else if (origPath) {
metaText = ` (No embedded generation metadata — not a PNG file)`;
metaNote = "(No embedded generation metadata — not a PNG file)";
}
if (metaText) lines.push(metaText);
}
lines.push("");
onItem?.({ id, filePath: visionTmpIds.has(id) ? (origPath ?? item.filePath) : item.filePath, displayName, visual, meta, metaNote });
}
if (totalInferenceTimeMs !== null) {
lines.push(`Total inference time: ${Math.round(totalInferenceTimeMs)}ms`);
}
// Status update: done
try {
const statusSuffix = visionError ? " (vision unavailable)" : " successfully";
statusCtx.status?.(`Analyzed ${analysisItems.length} image${analysisItems.length !== 1 ? "s" : ""}${statusSuffix}`);
} catch {
// best-effort
}
// Audit log — requestId is reused as-is by the MCP adapter for its HTML preview filename.
try {
const audit = buildAuditLogger({ backend: "analyse_image", mode: "analyse_image" as any, requestId });
if (currentLmChatId) audit.setChatId(currentLmChatId);
audit.setUserRequest({ targets, prompt: effectivePrompt } as any);
audit.setOutput({ items: analysisItems.map((item) => item.id) } as any);
await audit.write();
} catch (e) {
console.warn("[analyse_image] audit logging failed:", String(e));
}
return lines.join("\n");
} catch (e) {
return `analyse_image failed: ${String((e as any)?.message || e)}`;
} finally {
for (const tp of visionTmpPaths) await fs.promises.unlink(tp).catch(() => {});
}
}
// ─────────────────────────────────────────────────────────────────────────────
// detect_object
// ─────────────────────────────────────────────────────────────────────────────
export const DetectObjectParamsShape = {
targets: FlexibleTargetsList.optional().describe(
"One or more image notations to process. Each notation is a letter followed by a number: " +
"a=attachment (a1, a2, …), i=generated image (i1, i2, …), v=variant (v1, v2, …), p=picture (p1, p2, …). " +
"Also accepts a filename in the scratchpad folder or an absolute path. " +
"Pass as a JSON array, e.g. [\"a1\", \"i3\"]. " +
"Omit when there is exactly one image — it will be selected automatically."
),
task: z
.string()
.optional()
.default("")
.describe(
"What to detect. Omit for full-image general object detection. " +
"Use natural language to target specific subjects (e.g. 'all faces and hands', 'the dog', 'cars and bicycles')."
),
} satisfies Record<string, z.ZodTypeAny>;
export async function handleDetectObject(
args: any,
ctl: ToolsProviderController,
statusCtx: ToolStatusContext = {},
requestId?: string
): Promise<any> {
try {
// Parse targets: accept array or comma/space-separated string (analogous to analyse_image)
let rawTargets: string[] = [];
if (Array.isArray(args?.targets)) {
rawTargets = args.targets.map((s: any) => String(s).trim()).filter(Boolean);
} else if (typeof args?.targets === "string" && args.targets.trim()) {
const trimmed = args.targets.trim();
if (trimmed.startsWith("[")) {
try {
const parsed = JSON.parse(trimmed);
if (Array.isArray(parsed)) {
rawTargets = parsed.map((s: any) => String(s).trim()).filter(Boolean);
} else {
rawTargets = trimmed.split(/[\s,]+/).map((s: string) => s.trim()).filter(Boolean);
}
} catch {
rawTargets = trimmed.split(/[\s,]+/).map((s: string) => s.trim()).filter(Boolean);
}
} else {
rawTargets = trimmed.split(/[\s,]+/).map((s: string) => s.trim()).filter(Boolean);
}
}
const task =
typeof args?.task === "string" && args.task.trim() ? args.task.trim() : "";
console.log("[detect_object] invoked", { targets: rawTargets, task });
let currentLmChatId: string | null = null;
let currentLmWorkingDir: string | null = null;
try {
const chatCtx = await getActiveChatContext();
if ((chatCtx as any)?.chatId) currentLmChatId = (chatCtx as any).chatId;
if ((chatCtx as any)?.workingDir) currentLmWorkingDir = (chatCtx as any).workingDir;
} catch {}
if (!currentLmChatId) {
try {
const resolved = await resolveActiveLMStudioChatId();
if ((resolved as any)?.ok) currentLmChatId = (resolved as any).chatId;
} catch {}
}
const primaryOutDir: string | undefined =
currentLmWorkingDir ||
(currentLmChatId ? getLMStudioWorkingDir(currentLmChatId) : undefined);
if (!primaryOutDir) {
console.error("[detect_object] could not resolve working directory");
return {
content: [
{
type: "text",
text: "detect_object failed: could not resolve LM Studio chat working directory.",
},
],
isError: true as const,
};
}
console.log("[detect_object] primaryOutDir:", primaryOutDir);
await fs.promises.mkdir(primaryOutDir, { recursive: true }).catch(() => {});
// Sync attachments so state is up to date
console.log("[detect_object] syncing attachments...");
try {
await syncAttachmentsToState(primaryOutDir, false, Number.MAX_SAFE_INTEGER);
} catch (e) {
console.warn("[detect_object] attachment sync failed (non-fatal):", (e as any)?.message ?? e);
}
console.log("[detect_object] attachment sync done");
console.log("[detect_object] reading state...");
const st = await readState(primaryOutDir);
const attachments: any[] = Array.isArray(st?.attachments) ? st.attachments : [];
const pictures: any[] = Array.isArray(st?.pictures) ? st.pictures : [];
const imageRecords: any[] = Array.isArray(st?.images) ? st.images : [];
const images: Array<{ i: number; path: string }> = imageRecords
.filter((r: any) => r && typeof r.filename === "string")
.sort((a: any, b: any) => (a.i || 0) - (b.i || 0))
.map((r: any) => ({ i: r.i || 1, path: path.join(primaryOutDir, r.filename) }));
const variantRecords: any[] = Array.isArray(st?.variants) ? st.variants : [];
const variants: Array<{ v: number; path: string }> = variantRecords
.filter((v: any) => v && typeof v.filename === "string")
.map((v: any) => ({ v: v.v || 1, path: path.join(primaryOutDir, v.filename) }));
console.log("[detect_object] state:", {
attachments: attachments.length,
images: images.length,
variants: variants.length,
pictures: pictures.length,
});
// Resolve source buffers for each target (or auto-select if none given)
type SourceEntry = { id: string; buf: Buffer; origBuf: Buffer };
const sourceEntries: SourceEntry[] = [];
async function resolveOneBuf(rawCanvas: string): Promise<Buffer> {
const pref = parsePrefixedNotation(rawCanvas);
if (!pref) {
const pathToken = await resolveTargetPathToken(rawCanvas, primaryOutDir);
if (!pathToken) throw new Error(`Invalid canvas notation: ${rawCanvas}`);
return fs.promises.readFile(pathToken);
}
if (pref.pool === "attachment") {
const rec = attachments.find((a: any) => a?.a === pref.index);
const previewRel = rec && typeof rec.preview === "string" ? rec.preview : "";
if (!previewRel) throw new Error(`Preview for attachment a${pref.index} not found.`);
return fs.promises.readFile(path.join(primaryOutDir!, previewRel));
} else if (pref.pool === "image") {
const rec = imageRecords.find((r: any) => r?.i === pref.index);
const previewRel = rec && typeof rec.preview === "string" ? rec.preview : "";
if (!previewRel) throw new Error(`Preview for image i${pref.index} not found.`);
return fs.promises.readFile(path.join(primaryOutDir!, previewRel));
} else if (pref.pool === "variant") {
const rec = variantRecords.find((v: any) => v?.v === pref.index);
const previewRel = rec && typeof rec.preview === "string" ? rec.preview : "";
if (!previewRel) throw new Error(`Preview for variant v${pref.index} not found.`);
return fs.promises.readFile(path.join(primaryOutDir!, previewRel));
} else {
const rec = pictures.find((p: any) => p?.p === pref.index);
const previewRel = rec && typeof rec.preview === "string" ? rec.preview : "";
if (!previewRel) throw new Error(`Preview for picture p${pref.index} not found.`);
return fs.promises.readFile(path.join(primaryOutDir!, previewRel));
}
}
// Resolves the original (full-resolution) file for a canvas notation.
// Falls back to previewFallback when the original is unavailable.
async function resolveOriginalBuf(rawCanvas: string, previewFallback: Buffer): Promise<Buffer> {
try {
const pref = parsePrefixedNotation(rawCanvas);
if (!pref) {
const pathToken = await resolveTargetPathToken(rawCanvas, primaryOutDir);
return pathToken ? await fs.promises.readFile(pathToken) : previewFallback;
}
if (pref.pool === "attachment") {
const rec = attachments.find((a: any) => a?.a === pref.index);
const originAbs = rec && typeof rec.originAbs === "string" ? rec.originAbs : "";
if (!originAbs) return previewFallback;
return await fs.promises.readFile(originAbs);
}
let rec: any;
if (pref.pool === "image") rec = imageRecords.find((r: any) => r?.i === pref.index);
else if (pref.pool === "variant") rec = variantRecords.find((v: any) => v?.v === pref.index);
else rec = pictures.find((p: any) => p?.p === pref.index);
const filename = rec && typeof rec.filename === "string" ? rec.filename : "";
if (!filename) return previewFallback;
return await fs.promises.readFile(path.join(primaryOutDir!, filename));
} catch {
return previewFallback;
}
}
try {
if (rawTargets.length > 0) {
for (const t of rawTargets) {
const buf = await resolveOneBuf(t);
const origBuf = await resolveOriginalBuf(t, buf);
sourceEntries.push({ id: t, buf, origBuf });
}
} else {
// Auto-select single source
const total = attachments.length + variantRecords.length + imageRecords.length + pictures.length;
if (total === 0) throw new Error("No source image available.");
if (total > 1) throw new Error("Ambiguous source — specify targets explicitly.");
let buf: Buffer;
let id: string;
if (attachments.length === 1) {
const rec = attachments[0];
const previewRel = rec && typeof rec.preview === "string" ? rec.preview : "";
if (!previewRel) throw new Error("Preview for attachment not found.");
buf = await fs.promises.readFile(path.join(primaryOutDir!, previewRel));
id = `a${typeof rec.a === "number" ? rec.a : 1}`;
} else if (variantRecords.length === 1) {
const rec = variantRecords[0];
const previewRel = rec && typeof rec.preview === "string" ? rec.preview : "";
if (!previewRel) throw new Error("Preview for variant not found.");
buf = await fs.promises.readFile(path.join(primaryOutDir!, previewRel));
id = `v${typeof rec.v === "number" ? rec.v : 1}`;
} else if (imageRecords.length === 1) {
const rec = imageRecords[0];
const previewRel = rec && typeof rec.preview === "string" ? rec.preview : "";
if (!previewRel) throw new Error("Preview for image not found.");
buf = await fs.promises.readFile(path.join(primaryOutDir!, previewRel));
id = `i${typeof rec.i === "number" ? rec.i : 1}`;
} else {
const rec = pictures[0];
const previewRel = rec && typeof rec.preview === "string" ? rec.preview : "";
if (!previewRel) throw new Error("Preview for picture not found.");
buf = await fs.promises.readFile(path.join(primaryOutDir!, previewRel));
id = `p${rec.p ?? 1}`;
}
const origBuf = await resolveOriginalBuf(id, buf);
sourceEntries.push({ id, buf, origBuf });
}
} catch (e) {
return {
content: [{ type: "text", text: String((e as any)?.message || e) }],
isError: true as const,
};
}
console.log("[detect_object] sources resolved:", sourceEntries.map((s) => s.id));
const globalConfig = getGlobalConfig(ctl);
const envEmbedPngMetadata = process.env.EMBED_PNG_METADATA;
const embedPngMetadata = getGlobalBoolean(
globalConfig,
"embedPngMetadata",
envEmbedPngMetadata === undefined ? defaultPluginSettings.embedPngMetadata : envEmbedPngMetadata !== "false"
);
let visionBaseUrl = getGlobalString(globalConfig, "embeddingBaseUrl", process.env.LMSTUDIO_VISION_API_BASE_URL || defaultPluginSettings.embeddingBaseUrl);
const visionApiKey = getGlobalString(globalConfig, "embeddingApiKey", process.env.LMSTUDIO_VISION_API_KEY || defaultPluginSettings.embeddingApiKey);
let visionModelKey = getGlobalString(globalConfig, "qwen3VlModelPath", process.env.LMSTUDIO_VISION_MODEL_KEY || defaultPluginSettings.qwen3VlModelPath);
const envDetectMaxTokens = Number.parseInt(process.env.DETECT_MAX_TOKENS || "", 10);
const envDetectTemperature = Number.parseFloat(process.env.DETECT_TEMPERATURE || "");
const configuredDetectMaxTokens = Math.floor(getGlobalNumber(
globalConfig,
"detectMaxTokens",
Number.isFinite(envDetectMaxTokens) && envDetectMaxTokens > 0 ? envDetectMaxTokens : defaultPluginSettings.detectMaxTokens
));
const configuredDetectTemperature = getGlobalNumber(
globalConfig,
"detectTemperature",
Number.isFinite(envDetectTemperature) ? envDetectTemperature : defaultPluginSettings.detectTemperature
);
const detectionConfig: VisionDetectionAnalyzerConfig = {
task,
odPrompt: getGlobalString(globalConfig, "qwen3VlOdPrompt", process.env.DETECT_OD_PROMPT || defaultPluginSettings.qwen3VlOdPrompt) || undefined,
maxTokens: configuredDetectMaxTokens,
temperature: configuredDetectTemperature,
timeoutMs: 120_000,
};
const tmpPaths: string[] = [];
const detectionItems: VisionAnalysisItem[] = [];
for (const entry of sourceEntries) {
const visionBuf = await normalizeVisionBuffer(entry.buf);
const tmpPath = path.join(primaryOutDir, `_tmp_detect_src_${safeIdForFilename(entry.id)}_${Date.now()}.png`);
await fs.promises.writeFile(tmpPath, visionBuf);
tmpPaths.push(tmpPath);
detectionItems.push({ id: entry.id, filePath: tmpPath });
}
console.log("[detect_object] calling detection API for", detectionItems.length, "items");
const progressTotalSteps = sourceEntries.length * 2 + 4;
let batchResult: VisionDetectionBatchResult = {
results: [],
totalInferenceTimeMs: 0,
backend: "vision-api",
};
try {
reportToolStatus(statusCtx, `Detecting objects in ${sourceEntries.length} image${sourceEntries.length === 1 ? "" : "s"}...`);
reportToolStep(statusCtx, 1, progressTotalSteps, `Preparing ${sourceEntries.length} image${sourceEntries.length === 1 ? "" : "s"} for object detection...`);
const visionApi = resolveVisionApiFromEnv();
const ready = await ensureVisionModelReady(visionApi, {
baseUrl: visionBaseUrl,
apiKey: visionApiKey,
modelKey: visionModelKey,
status: (message) => { try { statusCtx.status?.(message); } catch {} },
llamaServer: visionApi === "llama-server" ? resolveLlamaServerConfig(globalConfig) : undefined,
});
if (!ready.ok) {
throw new Error(ready.error);
}
// llama-server strand: router's own exact tagged model id/address, not the configured ones.
if (ready.resolvedBaseUrl) visionBaseUrl = ready.resolvedBaseUrl;
if (ready.resolvedModelKey) visionModelKey = ready.resolvedModelKey;
for (let idx = 0; idx < detectionItems.length; idx++) {
const item = detectionItems[idx];
reportToolStep(statusCtx, idx + 2, progressTotalSteps, `Detecting objects in ${item.id} (${idx + 1}/${detectionItems.length})...`);
const singleResult = await detectLmStudioVisionBatch([item], {
...detectionConfig,
baseUrl: visionBaseUrl,
apiKey: visionApiKey,
model: visionModelKey,
});
batchResult.results.push(...singleResult.results);
batchResult.totalInferenceTimeMs += singleResult.totalInferenceTimeMs;
batchResult.backend = singleResult.backend;
}
console.log("[detect_object] detection API returned:", {
results: batchResult.results.length,
totalMs: batchResult.totalInferenceTimeMs,
});
try {
const totalObjects = batchResult.results.reduce((s, r) => s + (r.objects?.length ?? 0), 0);
const ms = Math.round(batchResult.totalInferenceTimeMs);
reportToolStep(statusCtx, sourceEntries.length + 2, progressTotalSteps, `${totalObjects} object${totalObjects === 1 ? "" : "s"} found across ${batchResult.results.length} image${batchResult.results.length === 1 ? "" : "s"} (${ms}ms); drawing bounding boxes...`);
} catch {}
} finally {
for (const tp of tmpPaths) await fs.promises.unlink(tp).catch(() => {});
}
if (!batchResult.results.length) {
return {
content: [{ type: "text", text: "detect_object: no results returned from detection API." }],
isError: true as const,
};
}
// Per-image: draw bboxes, save, generate preview, build state records
const variantPreviewSpec = (VARIANT_FULL_CONFIG as any).preview;
const stamp = isoStampCompact();
let nextI = Math.max(1, st.counters?.nextImageI ?? 1);
const imageRecordsForState: any[] = [];
const resultEntries: Array<{
id: string;
i: number;
detResult: typeof batchResult.results[0];
savedPath: string;
savedFileUrl: string;
savedSize: number;
preview: any | null;
httpOriginal: string;
httpPreview: string;
}> = [];
const httpBase = await getHealthyServerBaseUrl();
for (let idx = 0; idx < batchResult.results.length; idx++) {
const detResult = batchResult.results[idx];
const sourceId = sourceEntries[idx]?.id ?? `canvas${idx + 1}`;
const origBuf = sourceEntries[idx].origBuf;
const currentI = nextI++;
reportToolStep(statusCtx, sourceEntries.length + 3 + idx, progressTotalSteps, `Drawing boxes for ${sourceId} (${idx + 1}/${batchResult.results.length})...`);
const bboxes = detResult.objects.map((o) => o.bbox as [number, number, number, number]);
console.log(`[detect_object] drawing ${bboxes.length} bboxes for ${sourceId}...`);
// Draw on the original-resolution file; sourceDims carries the preview space in which
// the detection server reported its bbox coordinates so they get scaled up correctly.
const annotatedBuf = await drawBboxesOnImage(origBuf, bboxes, {
sourceDims: { width: detResult.imageWidth, height: detResult.imageHeight },
palette: true,
});
const analysis: PngAnalysisMetadata = {
schema: "ceveyne.image-analysis/v1",
tool: "detect_object",
sourceNotation: sourceId,
inference: {
model: visionModelKey,
...(task ? { query: task } : {}),
detectorPromptSha256: createHash("sha256").update(detectionConfig.odPrompt ?? "").digest("hex"),
maxTokens: configuredDetectMaxTokens,
temperature: configuredDetectTemperature,
},
render: { palette: true },
detections: detResult.objects.map((object) => ({
label: object.label,
bbox: { x1: object.bbox[0], y1: object.bbox[1], x2: object.bbox[2], y2: object.bbox[3] },
})),
};
const savedBuffer = embedPngMetadata
? injectXmpIntoBuffer(annotatedBuf, {
...(task ? { prompt: task } : {}),
model: visionModelKey,
mode: "object_detection",
generatedBy: `${getSelfPluginIdentifier()}/detect_object`,
creatorTool: `${getSelfPluginIdentifier()}/detect_object`,
analysis,
})
: annotatedBuf;
const baseName = `image-${stamp}-i${currentI}`;
const savedPath = path.join(primaryOutDir, `${baseName}.png`);
await fs.promises.writeFile(savedPath, savedBuffer);
const savedFileUrl = pathToFileURL(savedPath).toString();
const savedSize = savedBuffer.length;
console.log(`[detect_object] annotated image written: ${savedPath} (${savedSize} bytes)`);
let preview: any = null;
try {
const p = await generatePreviewFromBuffer(savedBuffer, primaryOutDir, `${baseName}.png`, variantPreviewSpec);
preview = {
ok: true as const,
filePath: p.previewAbs,
fileName: p.previewFilename,
fileUrl: pathToFileURL(p.previewAbs).toString(),
size_bytes: p.data.length,
width: p.width,
height: p.height,
mimeType: "image/jpeg" as const,
dataBase64: p.data.toString("base64"),
};
} catch (e) {
console.warn(`[detect_object] preview generation failed for ${sourceId}:`, String(e));
}
const httpOriginal = httpBase
? toHttpOriginalUrl(`${baseName}.png`, httpBase, currentLmChatId || undefined)
: "";
const httpPreview = (() => {
if (!httpBase || !currentLmChatId || !preview?.fileName) return "";
return toHttpPreviewUrl(preview.fileName, httpBase, currentLmChatId);
})();
imageRecordsForState.push({
filename: `${baseName}.png`,
preview: preview ? `preview-${baseName}.jpg` : undefined,
i: currentI,
sourceTool: `${getSelfPluginIdentifier()}/detect_object`,
detectSource: sourceId,
task,
imageWidth: detResult.imageWidth,
imageHeight: detResult.imageHeight,
analysisMetadata: analysis,
detections: detResult.objects.map((o) => ({
label: o.label,
bbox: { x1: o.bbox[0], y1: o.bbox[1], x2: o.bbox[2], y2: o.bbox[3] },
crop: {
cropLeft: o.cropLeft,
cropRight: o.cropRight,
cropTop: o.cropTop,
cropBottom: o.cropBottom,
},
})),
});
resultEntries.push({ id: sourceId, i: currentI, detResult, savedPath, savedFileUrl, savedSize, preview, httpOriginal, httpPreview });
}
// Update state once for all images
console.log("[detect_object] updating state...");
reportToolStep(statusCtx, progressTotalSteps - 1, progressTotalSteps, "Updating image state and audit log...");
try {
const stateForUpdate = await readState(primaryOutDir);
const appendResult = appendImages(stateForUpdate, imageRecordsForState);
if (appendResult.changed) {
await writeStateAtomic(primaryOutDir, stateForUpdate);
console.log("[detect_object] state written, nextImageI:", stateForUpdate.counters?.nextImageI);
}
} catch (e) {
console.warn("[detect_object] state update failed:", String(e));
}
// Audit log — effectiveRequestId is reused below in each summary entry, so a caller
// (the MCP adapter) can always name its HTML report after this exact audit requestId,
// even for a batch call whose top-level result is an array (see summaries below).
const effectiveRequestId = requestId ?? `${Date.now()}-${Math.random().toString(36).slice(2, 8)}`;
try {
const audit = buildAuditLogger({ backend: "detect_object", mode: "detect_object" as any, requestId: effectiveRequestId });
if (currentLmChatId) audit.setChatId(currentLmChatId);
audit.setUserRequest({ targets: rawTargets, task } as any);
audit.setOutput({
images: resultEntries.map((r) => ({
id: r.id,
i: r.i,
detections: r.detResult.objects.length,
path: r.savedPath,
url: r.savedFileUrl,
bytes: r.savedSize,
...(r.httpOriginal ? { http_url: r.httpOriginal } : {}),
...(r.preview ? { preview_path: r.preview.filePath, preview_url: r.preview.fileUrl } : {}),
...(r.httpPreview ? { http_preview_url: r.httpPreview } : {}),
})),
} as any);
await audit.write();
} catch (e) {
console.warn("[detect_object] audit logging failed:", String(e));
}
// Assemble tool result
const envPreviewRaw = process.env["PREVIEW_IN_CHAT"];
const previewInChat =
envPreviewRaw === undefined
? true
: envPreviewRaw === "1" || envPreviewRaw.toLowerCase() === "true";
const summaries = resultEntries.map((r) => ({
tool: "detect_object",
requestId: effectiveRequestId,
source: r.id,
i: r.i,
imageWidth: r.detResult.imageWidth,
imageHeight: r.detResult.imageHeight,
inferenceTimeMs: r.detResult.inferenceTimeMs,
detections: r.detResult.objects.map((o) => ({
label: o.label,
bbox: { x1: o.bbox[0], y1: o.bbox[1], x2: o.bbox[2], y2: o.bbox[3] },
crop: {
left: { pct: o.cropLeft, px: Math.round((o.cropLeft / 100) * r.detResult.imageWidth) },
right: { pct: o.cropRight, px: Math.round((o.cropRight / 100) * r.detResult.imageWidth) },
top: { pct: o.cropTop, px: Math.round((o.cropTop / 100) * r.detResult.imageHeight) },
bottom: { pct: o.cropBottom, px: Math.round((o.cropBottom / 100) * r.detResult.imageHeight) },
},
crop_tool_hint: "Pass crop.left.pct as cropLeft, crop.right.pct as cropRight, crop.top.pct as cropTop, crop.bottom.pct as cropBottom to the crop tool.",
})),
}));
const reviewHint = "Carefully examine the preview to make absolutely sure that the object detection matches your intent.";
const content: any[] = [];
reportToolStep(statusCtx, progressTotalSteps, progressTotalSteps, "Assembling detection result...");
for (const r of resultEntries) {
const fallbackPreviewUrl = r.preview?.fileUrl || r.savedFileUrl;
const previewLine = `Preview i${r.i}: ${r.httpPreview ? r.httpPreview : fallbackPreviewUrl}`;
const originalLine = `Original i${r.i}: ${r.httpOriginal ? r.httpOriginal : r.savedFileUrl}`;
if (previewInChat && r.preview) {
const fname = String(r.preview.fileName || "");
content.push({
type: "image",
fileName: fname,
mimeType: r.preview.mimeType,
markdown: ``,
$hint: "This is an image file. Present the image to the user by using the markdown above.",
} as any);
}
// TODO: Restore when LM Studio renders file/HTTP links again
// content.push({ type: "text", text: previewLine });
// content.push({ type: "text", text: originalLine });
}
const totalMs = Math.round(batchResult.totalInferenceTimeMs);
if (totalMs > 0) {
content.push({ type: "text", text: `Total inference time: ${totalMs}ms` });
}
content.push({
type: "text",
text: JSON.stringify(summaries.length === 1 ? summaries[0] : summaries),
...(previewInChat ? {} : { $hint: reviewHint }),
});
return { content };
} catch (error) {
return {
content: [
{
type: "text",
text: `detect_object failed: ${(error as Error).message || String(error)}`,
},
],
isError: true as const,
};
}
}
// ─────────────────────────────────────────────────────────────────────────────
// annotate_image
// ─────────────────────────────────────────────────────────────────────────────
/**
* Expand (positive) or shrink (negative) a bbox [x1, y1, x2, y2].
* Number → % of box diagonal. String → value + optional 'px' or '%' suffix.
*/
function applyFrameAdjust(
bbox: [number, number, number, number],
frameAdjust: number | string,
imgW: number,
imgH: number,
): [number, number, number, number] {
const [x1, y1, x2, y2] = bbox;
const diag = Math.hypot(x2 - x1, y2 - y1);
let d_px: number;
if (typeof frameAdjust === "string") {
const m = String(frameAdjust).trim().match(/^([+-]?\d+(?:\.\d+)?)\s*(%|px)?$/i);
if (!m) return bbox;
const val = parseFloat(m[1]);
d_px = m[2]?.toLowerCase() === "px" ? val : (val / 100) * diag;
} else {
d_px = (frameAdjust / 100) * diag;
}
return [
Math.max(0, Math.round(x1 - d_px)),
Math.max(0, Math.round(y1 - d_px)),
Math.min(imgW - 1, Math.round(x2 + d_px)),
Math.min(imgH - 1, Math.round(y2 + d_px)),
];
}
export const AnnotateImageParamsShape = {
targets: FlexibleTargetsList.optional().describe(
"One or more image notations to process. Each notation is a letter followed by a number: " +
"a=attachment (a1, a2, …), i=generated image (i1, i2, …), v=variant (v1, v2, …), p=picture (p1, p2, …). " +
"Also accepts a filename in the scratchpad folder or an absolute path. " +
"Pass via the targets field, e.g. annotate_image({\"targets\":[\"a1\", \"i3\"]}). " +
"Omit when there is exactly one image — it will be selected automatically."
),
task: z
.string()
.optional()
.default("")
.describe(
"What to detect. Omit for full-image general object detection. " +
"Use natural language to target specific subjects (e.g. 'all faces and hands', 'the dog', 'cars and bicycles'). " +
"Not used on correction calls — detections are loaded from state."
),
color: z
.string()
.optional()
.describe("Box color for all detected objects. CSS color name or hex, e.g. 'pink', 'red', '#FF0000'. Default: magenta."),
lineWeight: z
.coerce.number().int().min(1).max(50)
.optional()
.describe("Line thickness in pixels for all bounding boxes. Default: 5."),
frameAdjust: z
.union([z.number(), z.string()])
.optional()
.describe(
"Expand (positive) or shrink (negative) bounding boxes before drawing. " +
"Number: percent of box diagonal (e.g. 5 = +5 %). " +
"String: value + optional 'px' or '%' suffix, e.g. '10px', '-5%'. " +
"Without detectLabel: applied to all boxes. With detectLabel: applied to the selected box only. Default: 5."
),
detectLabel: z
.preprocess(
(val) => {
if (typeof val === "string") {
const t = val.trim();
if (t.startsWith("[")) {
try {
const parsed = JSON.parse(t);
if (Array.isArray(parsed)) {
return parsed.map((el: unknown) => String(el).trim()).filter((s: string) => s.length > 0);
}
} catch {}
}
}
return val;
},
z.union([
z.string().transform((s) =>
s.split(/\s*,\s*/).map((x) => x.trim()).filter((x) => x.length > 0)
),
z.array(z.string().min(1)),
])
)
.optional()
.describe(
"On a correction call: label(s) to match (case-insensitive). " +
"Single string, comma-separated list ('left eye, right eye'), or JSON array. " +
"When set, ONLY the matching detection(s) are drawn — all others are omitted. " +
"Single label with no detectIndex: auto-expands to ALL detections for that label (Option A). " +
"canvas may be the original source (e.g. a1) or the previous annotate_image result (e.g. i3)."
),
detectIndex: z
.union([
z.string().transform((s) => (s.match(/\d+/g) ?? []).map(Number)),
z.coerce.number().int().min(0).transform((n) => [n]),
z.array(z.coerce.number().int().min(0)),
])
.optional()
.describe(
"Zero-based index or list of indices, parallel to detectLabel. " +
"indices[li] ?? indices[0] ?? 0 for missing entries. Default: 0. " +
"Single label + multiple indices draws that label at each specified occurrence (Option B). " +
"Bracket notation accepted: '[2, 4, 7]'."
),
x1: z.preprocess(
(val) => { if (typeof val === "string" && val.trim().startsWith("[")) { try { return JSON.parse(val); } catch {} } return val; },
z.union([z.number(), z.array(z.union([z.number(), z.null()]))])
).optional().describe(
"Manual left edge(s) in original image pixels. " +
"Scalar: applies to all selected detections. Array (parallel to detectLabel): null = keep stored value. " +
"E.g. x1=[null,50,null] moves only the second detection's left edge."
),
y1: z.preprocess(
(val) => { if (typeof val === "string" && val.trim().startsWith("[")) { try { return JSON.parse(val); } catch {} } return val; },
z.union([z.number(), z.array(z.union([z.number(), z.null()]))])
).optional().describe("Manual top edge(s) in original image pixels. Scalar or array (null = keep stored). See x1."),
x2: z.preprocess(
(val) => { if (typeof val === "string" && val.trim().startsWith("[")) { try { return JSON.parse(val); } catch {} } return val; },
z.union([z.number(), z.array(z.union([z.number(), z.null()]))])
).optional().describe("Manual right edge(s) in original image pixels. Scalar or array (null = keep stored). See x1."),
y2: z.preprocess(
(val) => { if (typeof val === "string" && val.trim().startsWith("[")) { try { return JSON.parse(val); } catch {} } return val; },
z.union([z.number(), z.array(z.union([z.number(), z.null()]))])
).optional().describe("Manual bottom edge(s) in original image pixels. Scalar or array (null = keep stored). See x1."),
} satisfies Record<string, z.ZodTypeAny>;
type StoredDetection = {
label: string;
bbox: { x1: number; y1: number; x2: number; y2: number };
crop?: { cropLeft: number; cropRight: number; cropTop: number; cropBottom: number };
};
type ResolvedEntry =
| { mode: "detect"; id: string; previewBuf: Buffer; origBuf: Buffer }
| {
mode: "redraw";
id: string; // original draw source (e.g. "a1")
origBuf: Buffer;
task: string;
detections: StoredDetection[];
imageWidth: number;
imageHeight: number;
analysisMetadata?: PngAnalysisMetadata;
};
/**
* Resolve a per-box coordinate override, analogous to resolveBoxOverride in mask.
* override: scalar (applies to all boxes) | array (parallel, null = keep stored) | undefined.
* Returns undefined when no override → caller uses storedValue.
*/
function resolveCoordOverride(
override: number | (number | null)[] | undefined,
index: number,
): number | undefined {
if (override === undefined) return undefined;
if (Array.isArray(override)) {
const entry = override[index];
if (entry === null || entry === undefined) return undefined;
return entry;
}
return override;
}
/**
* Option A helper: count occurrences of label in detections and return [0..N-1] if N > 1, else [0].
*/
function expandDetectIndices(detections: StoredDetection[], label: string): number[] {
const lower = label.toLowerCase();
let count = 0;
for (const d of detections) { if (d.label.toLowerCase() === lower) count++; }
return count > 1 ? Array.from({ length: count }, (_, i) => i) : [0];
}
export async function handleAnnotateImage(
args: any,
ctl: ToolsProviderController,
statusCtx: ToolStatusContext = {},
requestId?: string
): Promise<any> {
try {
// ── Parse args ─────────────────────────────────────────────────────
let rawTargets: string[] = [];
if (Array.isArray(args?.targets)) {
rawTargets = args.targets.map((s: any) => String(s).trim()).filter(Boolean);
} else if (typeof args?.targets === "string" && args.targets.trim()) {
const trimmed = args.targets.trim();
if (trimmed.startsWith("[")) {
try {
const parsed = JSON.parse(trimmed);
rawTargets = Array.isArray(parsed)
? parsed.map((s: any) => String(s).trim()).filter(Boolean)
: trimmed.split(/[\s,]+/).map((s: string) => s.trim()).filter(Boolean);
} catch {
rawTargets = trimmed.split(/[\s,]+/).map((s: string) => s.trim()).filter(Boolean);
}
} else {
rawTargets = trimmed.split(/[\s,]+/).map((s: string) => s.trim()).filter(Boolean);
}
}
const taskArg = typeof args?.task === "string" && args.task.trim() ? args.task.trim() : "";
const globalColor = typeof args?.color === "string" && args.color.trim() ? args.color.trim() : "magenta";
const globalLineWeight = typeof args?.lineWeight === "number"
? Math.max(1, Math.min(50, Math.round(args.lineWeight))) : 5;
const globalFrameAdjust: number | string = args?.frameAdjust !== undefined ? args.frameAdjust : 5;
// detectLabel: string → string[] via comma-split; array → as-is
let globalLabels: string[] | undefined;
{
const dlRaw = args?.detectLabel;
if (Array.isArray(dlRaw)) {
const arr = dlRaw.map((s: any) => String(s).trim()).filter(Boolean);
if (arr.length > 0) globalLabels = arr;
} else if (typeof dlRaw === "string" && dlRaw.trim()) {
globalLabels = dlRaw.split(/\s*,\s*/).map((x) => x.trim()).filter(Boolean);
}
}
// detectIndex: number → [n]; string → extract digit runs; array → as-is
let globalIndices: number[] = [];
{
const diRaw = args?.detectIndex;
if (Array.isArray(diRaw)) {
globalIndices = diRaw.map((n: any) => typeof n === "number" ? Math.floor(n) : parseInt(String(n), 10)).filter((n) => !isNaN(n) && n >= 0);
} else if (typeof diRaw === "string" && diRaw.trim()) {
globalIndices = (diRaw.match(/\d+/g) ?? []).map(Number);
} else if (typeof diRaw === "number" && diRaw >= 0) {
globalIndices = [Math.floor(diRaw)];
}
}
const rawX1 = args?.x1;
const rawY1 = args?.y1;
const rawX2 = args?.x2;
const rawY2 = args?.y2;
// Normalise: scalar number | number-or-null array | undefined
function normaliseCoord(raw: any): number | (number | null)[] | undefined {
if (raw === undefined || raw === null) return undefined;
if (typeof raw === "number") return raw;
if (Array.isArray(raw)) return raw.map((v: any) => (v === null || v === undefined) ? null : Number(v));
if (typeof raw === "string") {
const t = raw.trim();
if (t.startsWith("[")) { try { const p = JSON.parse(t); if (Array.isArray(p)) return p.map((v: any) => (v === null || v === undefined) ? null : Number(v)); } catch {} }
const n = Number(t); return isNaN(n) ? undefined : n;
}
return undefined;
}
const manualX1 = normaliseCoord(rawX1);
const manualY1 = normaliseCoord(rawY1);
const manualX2 = normaliseCoord(rawX2);
const manualY2 = normaliseCoord(rawY2);
const hasAnyManualCoord = manualX1 !== undefined || manualY1 !== undefined || manualX2 !== undefined || manualY2 !== undefined;
console.log("[annotate_image] invoked", { targets: rawTargets, task: taskArg, color: globalColor, lineWeight: globalLineWeight, frameAdjust: globalFrameAdjust, detectLabels: globalLabels, detectIndices: globalIndices, manualCoords: { x1: manualX1, y1: manualY1, x2: manualX2, y2: manualY2 } });
// ── Resolve working directory ──────────────────────────────────────
let currentLmChatId: string | null = null;
let currentLmWorkingDir: string | null = null;
try {
const chatCtx = await getActiveChatContext();
if ((chatCtx as any)?.chatId) currentLmChatId = (chatCtx as any).chatId;
if ((chatCtx as any)?.workingDir) currentLmWorkingDir = (chatCtx as any).workingDir;
} catch {}
if (!currentLmChatId) {
try {
const resolved = await resolveActiveLMStudioChatId();
if ((resolved as any)?.ok) currentLmChatId = (resolved as any).chatId;
} catch {}
}
const primaryOutDir: string | undefined =
currentLmWorkingDir ||
(currentLmChatId ? getLMStudioWorkingDir(currentLmChatId) : undefined);
if (!primaryOutDir) {
return {
content: [{ type: "text", text: "annotate_image failed: could not resolve LM Studio chat working directory." }],
isError: true as const,
};
}
await fs.promises.mkdir(primaryOutDir, { recursive: true }).catch(() => {});
// ── Sync attachments ──────────────────────────────────────────────
try {
await syncAttachmentsToState(primaryOutDir, false, Number.MAX_SAFE_INTEGER);
} catch (e) {
console.warn("[annotate_image] attachment sync failed (non-fatal):", (e as any)?.message ?? e);
}
// ── Read state ────────────────────────────────────────────────────
const st = await readState(primaryOutDir);
const attachments: any[] = Array.isArray(st?.attachments) ? st.attachments : [];
const pictures: any[] = Array.isArray(st?.pictures) ? st.pictures : [];
const imageRecords: any[] = Array.isArray(st?.images) ? st.images : [];
const variantRecords: any[] = Array.isArray(st?.variants) ? st.variants : [];
// ── Buffer resolution helpers ─────────────────────────────────────
async function resolvePreviewBuf(notation: string): Promise<Buffer> {
const pref = parsePrefixedNotation(notation);
if (!pref) {
const pathToken = await resolveTargetPathToken(notation, primaryOutDir);
if (!pathToken) throw new Error(`Invalid notation: ${notation}`);
return fs.promises.readFile(pathToken);
}
if (pref.pool === "attachment") {
const rec = attachments.find((a: any) => a?.a === pref.index);
const r = rec && typeof rec.preview === "string" ? rec.preview : "";
if (!r) throw new Error(`Preview for a${pref.index} not found.`);
return fs.promises.readFile(path.join(primaryOutDir!, r));
}
if (pref.pool === "image") {
const rec = imageRecords.find((r: any) => r?.i === pref.index);
const r = rec && typeof rec.preview === "string" ? rec.preview : "";
if (!r) throw new Error(`Preview for i${pref.index} not found.`);
return fs.promises.readFile(path.join(primaryOutDir!, r));
}
if (pref.pool === "variant") {
const rec = variantRecords.find((v: any) => v?.v === pref.index);
const r = rec && typeof rec.preview === "string" ? rec.preview : "";
if (!r) throw new Error(`Preview for v${pref.index} not found.`);
return fs.promises.readFile(path.join(primaryOutDir!, r));
}
const rec = pictures.find((p: any) => p?.p === pref.index);
const r = rec && typeof rec.preview === "string" ? rec.preview : "";
if (!r) throw new Error(`Preview for p${pref.index} not found.`);
return fs.promises.readFile(path.join(primaryOutDir!, r));
}
async function resolveOriginalBuf(notation: string, fallback: Buffer): Promise<Buffer> {
try {
const pref = parsePrefixedNotation(notation);
if (!pref) {
const pathToken = await resolveTargetPathToken(notation, primaryOutDir);
return pathToken ? await fs.promises.readFile(pathToken) : fallback;
}
if (pref.pool === "attachment") {
const rec = attachments.find((a: any) => a?.a === pref.index);
const abs = rec && typeof rec.originAbs === "string" ? rec.originAbs : "";
if (!abs) return fallback;
return await fs.promises.readFile(abs);
}
let rec: any;
if (pref.pool === "image") rec = imageRecords.find((r: any) => r?.i === pref.index);
else if (pref.pool === "variant") rec = variantRecords.find((v: any) => v?.v === pref.index);
else rec = pictures.find((p: any) => p?.p === pref.index);
const fn = rec && typeof rec.filename === "string" ? rec.filename : "";
if (!fn) return fallback;
return await fs.promises.readFile(path.join(primaryOutDir!, fn));
} catch {
return fallback;
}
}
// ── Resolve auto-selected source when no targets given ─────────────
let autoId: string | null = null;
if (rawTargets.length === 0) {
const total = attachments.length + variantRecords.length + imageRecords.length + pictures.length;
if (total === 0) {
return { content: [{ type: "text", text: "No source image available." }], isError: true as const };
}
if (total > 1) {
return { content: [{ type: "text", text: "Ambiguous source — specify targets explicitly." }], isError: true as const };
}
if (attachments.length === 1) autoId = `a${typeof attachments[0]?.a === "number" ? attachments[0].a : 1}`;
else if (variantRecords.length === 1) autoId = `v${typeof variantRecords[0]?.v === "number" ? variantRecords[0].v : 1}`;
else if (imageRecords.length === 1) autoId = `i${typeof imageRecords[0]?.i === "number" ? imageRecords[0].i : 1}`;
else autoId = `p${pictures[0]?.p ?? 1}`;
rawTargets = [autoId];
}
// ── Classify each target: detect or redraw ────────────────────────
// If task is explicitly provided, always run fresh inference (ignore any prior record).
const forceDetect = taskArg.length > 0;
const resolvedEntries: ResolvedEntry[] = [];
for (const rawId of rawTargets) {
let drawSourceId = rawId;
let stateRec: any = null;
// Case A: target is an iN detection/annotation result
const pref = parsePrefixedNotation(rawId);
if (pref?.pool === "image") {
const imgRec = imageRecords.find((r: any) => r?.i === pref.index);
if (
imgRec &&
Array.isArray(imgRec.detections) &&
imgRec.detections.length > 0 &&
typeof imgRec.imageWidth === "number"
) {
stateRec = imgRec;
drawSourceId = (typeof imgRec.detectSource === "string" && imgRec.detectSource)
? imgRec.detectSource
: rawId;
}
}
// Case B: target is an original (a1, p1, …) with a prior detection on record
if (!stateRec) {
const prior = [...imageRecords]
.reverse()
.find(
(r: any) =>
(r?.detectSource === rawId) &&
Array.isArray(r.detections) &&
r.detections.length > 0 &&
typeof r.imageWidth === "number",
);
if (prior) {
stateRec = prior;
drawSourceId = rawId;
}
}
try {
if (stateRec && !forceDetect) {
const preview = await resolvePreviewBuf(drawSourceId).catch(() => null);
const origBuf = preview
? await resolveOriginalBuf(drawSourceId, preview)
: Buffer.alloc(0);
resolvedEntries.push({
mode: "redraw",
id: drawSourceId,
origBuf,
task: typeof stateRec.task === "string" ? stateRec.task : taskArg,
detections: stateRec.detections as StoredDetection[],
imageWidth: stateRec.imageWidth as number,
imageHeight: stateRec.imageHeight as number,
analysisMetadata: stateRec.analysisMetadata as PngAnalysisMetadata | undefined,
});
} else {
const previewBuf = await resolvePreviewBuf(rawId);
const origBuf = await resolveOriginalBuf(rawId, previewBuf);
resolvedEntries.push({ mode: "detect", id: rawId, previewBuf, origBuf });
}
} catch (e) {
return {
content: [{ type: "text", text: String((e as any)?.message || e) }],
isError: true as const,
};
}
}
// ── Run Qwen3-VL for detect-mode entries ───────────────────────────
const detectEntries = resolvedEntries.filter((e): e is Extract<ResolvedEntry, { mode: "detect" }> => e.mode === "detect");
const progressTotalSteps = detectEntries.length > 0
? detectEntries.length + resolvedEntries.length + 4
: resolvedEntries.length + 3;
let batchResult: VisionDetectionBatchResult | null = null;
const globalConfig = getGlobalConfig(ctl);
const envEmbedPngMetadata = process.env.EMBED_PNG_METADATA;
const embedPngMetadata = getGlobalBoolean(
globalConfig,
"embedPngMetadata",
envEmbedPngMetadata === undefined ? defaultPluginSettings.embedPngMetadata : envEmbedPngMetadata !== "false"
);
let visionModelKey = "";
let detectionConfig: VisionDetectionAnalyzerConfig | null = null;
if (detectEntries.length > 0) {
let visionBaseUrl = getGlobalString(globalConfig, "embeddingBaseUrl", process.env.LMSTUDIO_VISION_API_BASE_URL || defaultPluginSettings.embeddingBaseUrl);
const visionApiKey = getGlobalString(globalConfig, "embeddingApiKey", process.env.LMSTUDIO_VISION_API_KEY || defaultPluginSettings.embeddingApiKey);
visionModelKey = getGlobalString(globalConfig, "qwen3VlModelPath", process.env.LMSTUDIO_VISION_MODEL_KEY || defaultPluginSettings.qwen3VlModelPath);
const envDetectMaxTokens = Number.parseInt(process.env.DETECT_MAX_TOKENS || "", 10);
const envDetectTemperature = Number.parseFloat(process.env.DETECT_TEMPERATURE || "");
const configuredDetectMaxTokens = Math.floor(getGlobalNumber(
globalConfig,
"detectMaxTokens",
Number.isFinite(envDetectMaxTokens) && envDetectMaxTokens > 0 ? envDetectMaxTokens : defaultPluginSettings.detectMaxTokens
));
const configuredDetectTemperature = getGlobalNumber(
globalConfig,
"detectTemperature",
Number.isFinite(envDetectTemperature) ? envDetectTemperature : defaultPluginSettings.detectTemperature
);
detectionConfig = {
task: taskArg,
odPrompt: getGlobalString(globalConfig, "qwen3VlOdPrompt", process.env.DETECT_OD_PROMPT || defaultPluginSettings.qwen3VlOdPrompt) || undefined,
maxTokens: configuredDetectMaxTokens,
temperature: configuredDetectTemperature,
timeoutMs: 120_000,
};
const tmpPaths: string[] = [];
const detectionItems: VisionAnalysisItem[] = [];
for (const entry of detectEntries) {
const visionBuf = await normalizeVisionBuffer(entry.previewBuf);
const tmpPath = path.join(primaryOutDir, `_tmp_annotate_src_${safeIdForFilename(entry.id)}_${Date.now()}.png`);
await fs.promises.writeFile(tmpPath, visionBuf);
tmpPaths.push(tmpPath);
detectionItems.push({ id: entry.id, filePath: tmpPath });
}
try {
reportToolStatus(statusCtx, `Detecting objects in ${detectEntries.length} image${detectEntries.length === 1 ? "" : "s"}...`);
reportToolStep(statusCtx, 1, progressTotalSteps, `Preparing ${detectEntries.length} image${detectEntries.length === 1 ? "" : "s"} for annotation detection...`);
const visionApi = resolveVisionApiFromEnv();
const ready = await ensureVisionModelReady(visionApi, {
baseUrl: visionBaseUrl,
apiKey: visionApiKey,
modelKey: visionModelKey,
status: (message) => { try { statusCtx.status?.(message); } catch {} },
llamaServer: visionApi === "llama-server" ? resolveLlamaServerConfig(globalConfig) : undefined,
});
if (!ready.ok) {
throw new Error(ready.error);
}
// llama-server strand: router's own exact tagged model id/address, not the configured ones.
if (ready.resolvedBaseUrl) visionBaseUrl = ready.resolvedBaseUrl;
if (ready.resolvedModelKey) visionModelKey = ready.resolvedModelKey;
batchResult = {
results: [],
totalInferenceTimeMs: 0,
backend: "vision-api",
};
for (let idx = 0; idx < detectionItems.length; idx++) {
const item = detectionItems[idx];
reportToolStep(statusCtx, idx + 2, progressTotalSteps, `Detecting objects in ${item.id} (${idx + 1}/${detectionItems.length})...`);
const singleResult = await detectLmStudioVisionBatch([item], {
...detectionConfig,
baseUrl: visionBaseUrl,
apiKey: visionApiKey,
model: visionModelKey,
});
batchResult.results.push(...singleResult.results);
batchResult.totalInferenceTimeMs += singleResult.totalInferenceTimeMs;
batchResult.backend = singleResult.backend;
}
try {
const totalObjects = batchResult.results.reduce((s, r) => s + (r.objects?.length ?? 0), 0);
const ms = Math.round(batchResult.totalInferenceTimeMs);
reportToolStep(statusCtx, detectEntries.length + 2, progressTotalSteps, `${totalObjects} object${totalObjects === 1 ? "" : "s"} found (${ms}ms); drawing boxes...`);
} catch {}
} finally {
for (const tp of tmpPaths) await fs.promises.unlink(tp).catch(() => {});
}
if (!batchResult || !batchResult.results.length) {
return {
content: [{ type: "text", text: "annotate_image: no results returned from detection API." }],
isError: true as const,
};
}
} else {
reportToolStatus(statusCtx, `Redrawing ${resolvedEntries.length} annotated image${resolvedEntries.length === 1 ? "" : "s"} from stored detections...`);
reportToolStep(statusCtx, 1, progressTotalSteps, `Redrawing ${resolvedEntries.length} annotated image${resolvedEntries.length === 1 ? "" : "s"} from stored detections...`);
}
// ── Per-entry: apply frameAdjust, draw, save ───────────────────────
const variantPreviewSpec = (VARIANT_FULL_CONFIG as any).preview;
const stamp = isoStampCompact();
let nextI = Math.max(1, st.counters?.nextImageI ?? 1);
const imageRecordsForState: any[] = [];
const resultEntries: Array<{
id: string;
i: number;
isRedraw: boolean;
task: string;
detObjects: StoredDetection[];
imageWidth: number;
imageHeight: number;
savedPath: string;
savedFileUrl: string;
savedSize: number;
preview: any | null;
httpOriginal: string;
httpPreview: string;
inferenceTimeMs: number;
}> = [];
const httpBase = await getHealthyServerBaseUrl();
let detectResultIdx = 0;
let resolvedIdx = 0;
const drawBaseStep = detectEntries.length > 0 ? detectEntries.length + 3 : 2;
for (const entry of resolvedEntries) {
reportToolStep(statusCtx, drawBaseStep + resolvedIdx, progressTotalSteps, `Drawing annotation for ${entry.id} (${resolvedIdx + 1}/${resolvedEntries.length})...`);
resolvedIdx++;
let rawBboxes: [number, number, number, number][];
let imgW: number;
let imgH: number;
let detObjects: StoredDetection[];
let isRedraw: boolean;
let entryTask: string;
let inferenceTimeMs = 0;
let bboxesAlreadyAdjusted = false;
if (entry.mode === "redraw") {
imgW = entry.imageWidth;
imgH = entry.imageHeight;
isRedraw = true;
entryTask = entry.task;
if (globalLabels !== undefined && globalLabels.length > 0) {
// detectLabel mode: resolve label/index pairs, draw ONLY the selected detections
let labels = [...globalLabels];
let indices = [...globalIndices];
// Option A: single label + no explicit index → auto-expand to all detections for that label
if (labels.length === 1 && indices.length === 0) {
const allIndices = expandDetectIndices(entry.detections, labels[0]);
if (allIndices.length > 1) {
labels = Array(allIndices.length).fill(labels[0]);
indices = allIndices;
}
}
// Option B: single label + multiple explicit indices → expand labels to match
if (labels.length === 1 && indices.length > 1) {
labels = Array(indices.length).fill(labels[0]);
}
// Pre-group detections by lowercase label once — O(1) lookup per box.
const detByLabel = new Map<string, StoredDetection[]>();
for (const d of entry.detections) {
const key = d.label.toLowerCase();
if (!detByLabel.has(key)) detByLabel.set(key, []);
detByLabel.get(key)!.push(d);
}
const resolvedBoxes: { det: StoredDetection; bbox: [number, number, number, number] }[] = [];
for (let li = 0; li < labels.length; li++) {
const label = labels[li];
const idx = indices[li] ?? indices[0] ?? 0;
const selectedDet = detByLabel.get(label.toLowerCase())?.[idx];
if (!selectedDet) {
const available = [...new Set(entry.detections.map((d) => d.label))].join(", ");
return {
content: [{ type: "text", text: `annotate_image: label '${label}' (index ${idx}) not found in stored detections. Available: ${available || "(none)"}` }],
isError: true as const,
};
}
// Coord overrides: resolveCoordOverride per axis, fall back to stored value.
// Applies to all selected boxes; scalar = same for all, array = parallel (null = keep stored).
const ox1 = hasAnyManualCoord ? resolveCoordOverride(manualX1 as any, li) : undefined;
const oy1 = hasAnyManualCoord ? resolveCoordOverride(manualY1 as any, li) : undefined;
const ox2 = hasAnyManualCoord ? resolveCoordOverride(manualX2 as any, li) : undefined;
const oy2 = hasAnyManualCoord ? resolveCoordOverride(manualY2 as any, li) : undefined;
const bbox: [number, number, number, number] = [
ox1 ?? selectedDet.bbox.x1,
oy1 ?? selectedDet.bbox.y1,
ox2 ?? selectedDet.bbox.x2,
oy2 ?? selectedDet.bbox.y2,
];
resolvedBoxes.push({ det: selectedDet, bbox });
}
rawBboxes = resolvedBoxes.map(({ bbox }) => applyFrameAdjust(bbox, globalFrameAdjust, imgW, imgH));
detObjects = resolvedBoxes.map(({ det, bbox }) => ({
...det,
bbox: { x1: bbox[0], y1: bbox[1], x2: bbox[2], y2: bbox[3] },
}));
bboxesAlreadyAdjusted = true;
} else if (globalIndices.length > 0) {
// No detectLabel but explicit detectIndex: select stored detections by position.
const selected: StoredDetection[] = [];
for (const idx of globalIndices) {
const det = entry.detections[idx];
if (!det) {
return {
content: [{ type: "text", text: `annotate_image: detectIndex ${idx} out of range (${entry.detections.length} stored detections).` }],
isError: true as const,
};
}
selected.push(det);
}
rawBboxes = selected.map((d) => [d.bbox.x1, d.bbox.y1, d.bbox.x2, d.bbox.y2] as [number, number, number, number]);
detObjects = selected;
bboxesAlreadyAdjusted = false;
} else {
// No detectLabel, no detectIndex: draw all stored boxes, apply frameAdjust to all
rawBboxes = entry.detections.map((d) => [d.bbox.x1, d.bbox.y1, d.bbox.x2, d.bbox.y2] as [number, number, number, number]);
detObjects = entry.detections;
}
} else {
const detResult = batchResult!.results[detectResultIdx++];
rawBboxes = detResult.objects.map((o) => o.bbox as [number, number, number, number]);
imgW = detResult.imageWidth;
imgH = detResult.imageHeight;
detObjects = detResult.objects.map((o) => ({
label: o.label,
bbox: { x1: o.bbox[0], y1: o.bbox[1], x2: o.bbox[2], y2: o.bbox[3] },
crop: { cropLeft: o.cropLeft, cropRight: o.cropRight, cropTop: o.cropTop, cropBottom: o.cropBottom },
}));
isRedraw = false;
entryTask = taskArg;
inferenceTimeMs = detResult.inferenceTimeMs ?? 0;
}
const adjustedBboxes = bboxesAlreadyAdjusted
? rawBboxes
: rawBboxes.map((bbox) => applyFrameAdjust(bbox, globalFrameAdjust, imgW, imgH));
const annotatedBuf = await drawBboxesOnImage(entry.origBuf, adjustedBboxes, {
sourceDims: { width: imgW, height: imgH },
palette: false,
color: globalColor,
lineWeight: globalLineWeight,
});
const inheritedInference = entry.mode === "redraw" ? entry.analysisMetadata?.inference : undefined;
const analysis: PngAnalysisMetadata = {
schema: "ceveyne.image-analysis/v1",
tool: "annotate_image",
sourceNotation: entry.id,
inference: entry.mode === "redraw"
? inheritedInference
? { ...inheritedInference, reused: true }
: undefined
: {
model: visionModelKey,
...(entryTask ? { query: entryTask } : {}),
detectorPromptSha256: createHash("sha256").update(detectionConfig?.odPrompt ?? "").digest("hex"),
maxTokens: detectionConfig?.maxTokens,
temperature: detectionConfig?.temperature,
},
render: { color: globalColor, lineWeight: globalLineWeight, frameAdjust: globalFrameAdjust },
detections: detObjects.map((detection, index) => ({
label: detection.label,
bbox: {
x1: adjustedBboxes[index][0],
y1: adjustedBboxes[index][1],
x2: adjustedBboxes[index][2],
y2: adjustedBboxes[index][3],
},
})),
};
const savedBuffer = embedPngMetadata
? injectXmpIntoBuffer(annotatedBuf, {
...(entryTask ? { prompt: entryTask } : {}),
...(analysis.inference?.model ? { model: analysis.inference.model } : {}),
mode: isRedraw ? "image_annotation_redraw" : "image_annotation",
generatedBy: `${getSelfPluginIdentifier()}/annotate_image`,
creatorTool: `${getSelfPluginIdentifier()}/annotate_image`,
analysis,
})
: annotatedBuf;
const currentI = nextI++;
const baseName = `image-${stamp}-i${currentI}`;
const savedPath = path.join(primaryOutDir, `${baseName}.png`);
await fs.promises.writeFile(savedPath, savedBuffer);
const savedFileUrl = pathToFileURL(savedPath).toString();
const savedSize = savedBuffer.length;
let preview: any = null;
try {
const p = await generatePreviewFromBuffer(savedBuffer, primaryOutDir, `${baseName}.png`, variantPreviewSpec);
preview = {
ok: true as const,
filePath: p.previewAbs,
fileName: p.previewFilename,
fileUrl: pathToFileURL(p.previewAbs).toString(),
size_bytes: p.data.length,
width: p.width,
height: p.height,
mimeType: "image/jpeg" as const,
dataBase64: p.data.toString("base64"),
};
} catch (e) {
console.warn(`[annotate_image] preview generation failed for ${entry.id}:`, String(e));
}
const httpOriginal = httpBase
? toHttpOriginalUrl(`${baseName}.png`, httpBase, currentLmChatId || undefined) : "";
const httpPreview = (() => {
if (!httpBase || !currentLmChatId || !preview?.fileName) return "";
return toHttpPreviewUrl(preview.fileName, httpBase, currentLmChatId);
})();
imageRecordsForState.push({
filename: `${baseName}.png`,
preview: preview ? `preview-${baseName}.jpg` : undefined,
i: currentI,
sourceTool: `${getSelfPluginIdentifier()}/annotate_image`,
detectSource: entry.id,
task: entryTask,
annotateColor: globalColor,
annotateLineWeight: globalLineWeight,
annotateFrameAdjust: globalFrameAdjust,
imageWidth: imgW,
imageHeight: imgH,
analysisMetadata: analysis,
detections: detObjects.map((d) => ({
label: d.label,
bbox: { x1: d.bbox.x1, y1: d.bbox.y1, x2: d.bbox.x2, y2: d.bbox.y2 },
crop: d.crop ?? {},
})),
});
resultEntries.push({
id: entry.id,
i: currentI,
isRedraw,
task: entryTask,
detObjects,
imageWidth: imgW,
imageHeight: imgH,
savedPath,
savedFileUrl,
savedSize,
preview,
httpOriginal,
httpPreview,
inferenceTimeMs,
});
}
// ── Update state ──────────────────────────────────────────────────
reportToolStep(statusCtx, progressTotalSteps - 1, progressTotalSteps, "Updating image state and audit log...");
try {
const stateForUpdate = await readState(primaryOutDir);
const appendResult = appendImages(stateForUpdate, imageRecordsForState);
if (appendResult.changed) {
await writeStateAtomic(primaryOutDir, stateForUpdate);
}
} catch (e) {
console.warn("[annotate_image] state update failed:", String(e));
}
// ── Audit log ─────────────────────────────────────────────────────
// effectiveRequestId is reused below in each summary entry, so a caller (the MCP adapter)
// can always name its HTML report after this exact audit requestId, even for a batch call
// whose top-level result is an array (see summaries below).
const effectiveRequestId = requestId ?? `${Date.now()}-${Math.random().toString(36).slice(2, 8)}`;
try {
const audit = buildAuditLogger({ backend: "annotate_image", mode: "annotate_image" as any, requestId: effectiveRequestId });
if (currentLmChatId) audit.setChatId(currentLmChatId);
audit.setUserRequest({ targets: rawTargets, task: taskArg, color: globalColor, lineWeight: globalLineWeight, frameAdjust: globalFrameAdjust } as any);
audit.setOutput({
images: resultEntries.map((r) => ({
id: r.id,
i: r.i,
redraw: r.isRedraw,
detections: r.detObjects.length,
path: r.savedPath,
url: r.savedFileUrl,
bytes: r.savedSize,
...(r.httpOriginal ? { http_url: r.httpOriginal } : {}),
...(r.preview ? { preview_path: r.preview.filePath, preview_url: r.preview.fileUrl } : {}),
...(r.httpPreview ? { http_preview_url: r.httpPreview } : {}),
})),
} as any);
await audit.write();
} catch {}
// ── Assemble result ───────────────────────────────────────────────
reportToolStep(statusCtx, progressTotalSteps, progressTotalSteps, "Assembling annotation result...");
const summaries = resultEntries.map((r) => ({
tool: "annotate_image",
requestId: effectiveRequestId,
source: r.id,
i: r.i,
redraw: r.isRedraw,
color: globalColor,
lineWeight: globalLineWeight,
frameAdjust: globalFrameAdjust,
...(r.inferenceTimeMs > 0 ? { inferenceTimeMs: r.inferenceTimeMs } : {}),
detections: r.detObjects.map((d) => ({
label: d.label,
bbox: { x1: d.bbox.x1, y1: d.bbox.y1, x2: d.bbox.x2, y2: d.bbox.y2 },
})),
}));
const envPreviewRaw = process.env["PREVIEW_IN_CHAT"];
const previewInChat =
envPreviewRaw === undefined
? true
: envPreviewRaw === "1" || envPreviewRaw.toLowerCase() === "true";
// Build target notations for hints (e.g. ["i9", "i10"])
const resultNotations = resultEntries.map((r) => `i${r.i}`);
const targetsJson = JSON.stringify(resultNotations);
const reviewHintFalse =
`Carefully examine the preview to make absolutely sure that the object detection matches your intent. Registered as ${resultNotations.join(", ")}. Use review_image({"targets":${targetsJson}}) to review, or annotate_image({"targets":${targetsJson}}) to apply corrections.`;
const reviewHintTrue =
`Carefully examine the preview to make absolutely sure that the object detection matches your intent. This is an image file. Present the image to the user by using the markdown above. Registered as ${resultNotations.join(", ")}. Use review_image({"targets":${targetsJson}}) to review, or annotate_image({"targets":${targetsJson}}) to apply corrections.`;
const content: any[] = [];
for (const r of resultEntries) {
const fallbackPreviewUrl = r.preview?.fileUrl || r.savedFileUrl;
if (previewInChat && r.preview) {
const fname = String(r.preview.fileName || "");
content.push({
type: "image",
fileName: fname,
mimeType: r.preview.mimeType,
markdown: ``,
$hint: reviewHintTrue,
} as any);
}
// TODO: Restore when LM Studio renders file/HTTP links again
// content.push({ type: "text", text: `Preview i${r.i}: ${r.httpPreview || fallbackPreviewUrl}` });
// content.push({ type: "text", text: `Original i${r.i}: ${r.httpOriginal || r.savedFileUrl}` });
}
if (batchResult && batchResult.totalInferenceTimeMs > 0) {
content.push({ type: "text", text: `Total inference time: ${Math.round(batchResult.totalInferenceTimeMs)}ms` });
}
content.push({
type: "text",
text: JSON.stringify(summaries.length === 1 ? summaries[0] : summaries),
...(previewInChat ? {} : { $hint: reviewHintFalse }),
});
return { content };
} catch (error) {
return {
content: [{ type: "text", text: `annotate_image failed: ${(error as Error).message || String(error)}` }],
isError: true as const,
};
}
}