dist-mcp / mcp / index.js
dist-mcp / mcp / index.js
import crypto from "node:crypto";
import fs from "node:fs/promises";
import path from "node:path";
import { fromJsonSchema } from "@modelcontextprotocol/server";
import { zodToJsonSchema } from "zod-to-json-schema";
import { z } from "zod";
import { bridgeToolErrorResult, escapeHtml, renderReportShell, startStdioMcpServer, looksLikeSourceNotation, normalizeSourceFieldToArray, resolveMcpSourceToken } from "./core-bundle.mjs";
import { AnalyseImageParamsShape, AnnotateImageParamsShape, DetectObjectParamsShape } from "../core/tools.js";
import { applyMcpConfig, getRawEnvSnapshot, readMcpConfig } from "./config.js";
import { bindChatContextToScratchpad } from "./chatContextBridge.js";
import { assertNoAttachmentNotation, collectAttachmentTokens } from "./sourceNotation.js";
import { logger } from "./mcpLogger.js";
import { extractSummary, extractSummaryText, materializeNewImages, snapshotImageIndices, writeHtmlReport } from "./resultMaterializer.js";
import { resolveEffectiveScratchpadPath } from "./scratchpadResolution.js";
import { buildGenericToolResult, buildMultiResultText, buildSingleResultText, buildUnslothToolResult } from "./toolResults.js";
/**
* Forwards analyse_image/detect_object/annotate_image's already-formatted status messages (see
* reportToolStatus/reportToolStep in core/tools.ts) to MCP's notifications/progress, tied to the
* current tool call via ctx.mcpReq.notify (same mechanism as generate-image/process-image's own
* createProgressForwarder). Only sends anything when the client actually opted in by sending a
* progressToken with the request (per MCP spec) -- returns undefined otherwise, so no progress
* machinery runs for clients that never asked for it. This is NOT merely cosmetic: per the MCP
* spec, a progress notification resets the client's wait timer for that request -- Unsloth
* Desktop's MCP setup has no exposed request-timeout setting at all, so this is what actually
* keeps a slow vision-model-load + multi-image call from timing out there.
*/
function createProgressForwarder(ctx, tool) {
const mcpReq = ctx?.mcpReq;
const progressToken = mcpReq?._meta?.progressToken;
if (progressToken === undefined || typeof mcpReq?.notify !== "function")
return undefined;
let progress = 0;
return (message) => {
progress += 1;
mcpReq.notify({
method: "notifications/progress",
params: { progressToken, progress, message },
}).catch((error) => {
logger.logError("progress-notify-failed", error, { tool }).catch(() => { });
});
};
}
/**
* zod-to-json-schema collapses a union of only-primitive types (e.g. annotate_image's
* frameAdjust: z.union([z.number(), z.string()]), or the nested number|null union inside its
* x1/y1/x2/y2 array-item schemas) into a single node with `type: ["number", "string"]` instead of
* `anyOf` branches. Several MCP clients only accept `type` as a single string and either reject
* the tool or silently drop the constraint on such a node. Recursively rewrites every such node
* into `{ anyOf: [{ type: "number" }, { type: "string" }], ...rest }`, walking into
* properties/items/anyOf/allOf/oneOf/additionalProperties so nested cases are fixed too
* (see /memories/repo/process-image-mcp-implementation.md).
*/
function splitTypeArrayUnions(node) {
if (Array.isArray(node))
return node.map(splitTypeArrayUnions);
if (node === null || typeof node !== "object")
return node;
const { type, properties, items, anyOf, allOf, oneOf, additionalProperties, ...rest } = node;
const walked = { ...rest };
if (properties && typeof properties === "object") {
walked.properties = Object.fromEntries(Object.entries(properties).map(([key, value]) => [key, splitTypeArrayUnions(value)]));
}
if (items !== undefined)
walked.items = splitTypeArrayUnions(items);
if (additionalProperties !== undefined)
walked.additionalProperties = splitTypeArrayUnions(additionalProperties);
if (anyOf !== undefined)
walked.anyOf = anyOf.map(splitTypeArrayUnions);
if (allOf !== undefined)
walked.allOf = allOf.map(splitTypeArrayUnions);
if (oneOf !== undefined)
walked.oneOf = oneOf.map(splitTypeArrayUnions);
if (Array.isArray(type)) {
walked.anyOf = [...(Array.isArray(walked.anyOf) ? walked.anyOf : []), ...type.map((t) => ({ type: t }))];
}
else if (type !== undefined) {
walked.type = type;
}
return walked;
}
/**
* aN (LM-Studio attachment notation) doesn't exist over MCP (see sourceNotation.ts) — but the
* shared Zod `.describe()` text in core/tools.ts still advertises it, since that same text is
* also the LM-Studio plugin's own tool description (where aN IS valid). Rewriting core/tools.ts
* would break the plugin's description, so this strips the aN mention only from the JSON schema
* text actually sent to MCP clients, leaving core/tools.ts untouched.
*/
function stripAttachmentNotationMentions(node) {
if (Array.isArray(node))
return node.map(stripAttachmentNotationMentions);
if (node === null || typeof node !== "object")
return node;
const out = {};
for (const [key, value] of Object.entries(node)) {
out[key] =
key === "description" && typeof value === "string"
? value.replace(/a=attachment \(a1, a2, …\), /g, "")
: stripAttachmentNotationMentions(value);
}
return out;
}
const ATTACHMENT_INVENTORY_HINT = " Pass the literal \"aN\" (or any unregistered number) to get back a list of all attachments currently available in this thread instead of running the tool.";
/**
* unsloth is the one preset where aN genuinely resolves (see sourceNotation.ts) — instead of
* stripping the aN mention, append the inventory-listing hint right after it so the agent knows
* how to discover valid numbers.
*/
function annotateAttachmentNotationForUnsloth(node) {
if (Array.isArray(node))
return node.map(annotateAttachmentNotationForUnsloth);
if (node === null || typeof node !== "object")
return node;
const out = {};
for (const [key, value] of Object.entries(node)) {
out[key] = key === "description" && typeof value === "string" && value.includes("a=attachment") ? `${value}${ATTACHMENT_INVENTORY_HINT}` : annotateAttachmentNotationForUnsloth(value);
}
return out;
}
/**
* Adds the MCP-only scratchpadFolder field on top of the plugin's real Zod shape — no field is
* hand-duplicated. scratchpadFolder's own requiredness comes from the resolved preset
* (made-for-bionic-core) — see presets/index.ts's doc comment for why this must not be
* re-derived locally. Required for all three tools (including analyse_image): `targets` is
* always resolved against chat_media_state.json in the bound scratchpad.
*/
function toMcpInputSchema(shape, requireScratchpadFolder, preset) {
const { $schema: _$schema, ...rawSchema } = zodToJsonSchema(z.object(shape).strict());
const schema = (preset.name === "unsloth" ? annotateAttachmentNotationForUnsloth(splitTypeArrayUnions(rawSchema)) : stripAttachmentNotationMentions(splitTypeArrayUnions(rawSchema)));
const visibleProperties = (schema.properties ?? {});
const baseRequired = schema.required ?? [];
return {
...schema,
properties: {
scratchpadFolder: {
type: "string",
description: scratchpadFolderGuidance(preset, requireScratchpadFolder),
},
...visibleProperties,
},
required: requireScratchpadFolder ? ["scratchpadFolder", ...baseRequired] : baseRequired,
};
}
const SOURCE_NOTATION_GUIDANCE = "Sources: pass one or more notations in the `targets` array field, e.g. [\"i3\", \"v2\"] — i=generated image, v=variant, p=picture. A comma/space-separated string is also accepted, as well as a scratchpad filename or absolute path.";
function sourceNotationGuidance(preset) {
if (preset.name !== "unsloth")
return SOURCE_NOTATION_GUIDANCE;
return `${SOURCE_NOTATION_GUIDANCE} Also accepts aN (attachment, e.g. a1).${ATTACHMENT_INVENTORY_HINT}`;
}
/**
* unsloth has no get_scratchpad_folder tool (see made-for-preset-rollout-plan.md) — its agent
* obtains the same value via \`cat .unsloth_sandbox\`. Reusing bionic's "get_scratchpad_folder"
* wording there would be actively wrong, not just imprecise.
*/
function scratchpadFolderGuidance(preset, requireScratchpadFolder) {
if (!requireScratchpadFolder)
return "Optional scratchpad session folder name, one level under CHAT_WORKING_DIRECTORIES. Omit to write directly into CHAT_WORKING_DIRECTORIES.";
if (preset.name === "unsloth")
return "Required scratchpad session folder. Obtain this value by running `cat .unsloth_sandbox` and passing its output unchanged.";
return "Required scratchpad session folder name, one level under CHAT_WORKING_DIRECTORIES. Obtain this value with get_scratchpad_folder and pass it back unchanged.";
}
/** Mirrors the actual gate in main()'s analyse_image handler: `if (preset.writeHtmlReportForMultipleResults)`. */
function analyseImageHtmlPreviewGuidance(preset) {
if (preset.writeHtmlReportForMultipleResults)
return "It always also writes a plain HTML preview of its result into the scratchpad — opening it via open_url_in_app_browser is optional, the machine-readable text response below is the primary channel.";
return "No HTML preview file is written — the machine-readable text response below is the only output.";
}
function analyseImageDescription(preset, requireScratchpadFolder) {
return `Inspect existing media items (images, pictures, variants, attachments) for generation metadata and, optionally, visual content.
PRIMARY use — generation metadata:
Call this tool whenever the user asks how an image was generated, what settings were used, or wants to reuse generation parameters (prompt, model, sampler, seed, steps, guidance scale, LoRA, source images, …). Those parameters are embedded in the PNG file and are returned automatically — no vision prompt needed.
SECONDARY use — visual description (on demand only):
Only request a visual description when the user explicitly asks you to describe or analyze the image content. Pass a prompt in the 'prompt' parameter. Without a prompt, no vision model is invoked and no description is returned.
This tool never writes a new file. ${analyseImageHtmlPreviewGuidance(preset)}
${scratchpadFolderGuidance(preset, requireScratchpadFolder)}
${sourceNotationGuidance(preset)}`;
}
function detectObjectDescription(preset, requireScratchpadFolder) {
return `Detect objects in one or more images and draw colored bounding boxes on each result.
For each source image, returns a new annotated image (saved as iN) with bounding boxes for each detected object, plus a JSON summary with labels, coordinates, and crop percentages. Uses Qwen3-VL for detection.
${scratchpadFolderGuidance(preset, requireScratchpadFolder)}
${sourceNotationGuidance(preset)}`;
}
function annotateImageDescription(preset, requireScratchpadFolder) {
return `Highlights specific areas or elements on images. If well described, these elements are precisely framed with bounding boxes in the chosen color.
Detections are saved to state for later correction, refinement, or re-drawing with adjusted labels and edge positions. Omit task on a correction call to redraw from stored detections without new inference; pass task to force fresh inference.
${scratchpadFolderGuidance(preset, requireScratchpadFolder)}
${sourceNotationGuidance(preset)}`;
}
// The MCP SDK's CallToolResult content-block union isn't re-exported for reuse here,
// so the handler result stays structurally typed (any) rather than hand-duplicating it.
function toToolResult(result) {
const record = result;
if (record && Array.isArray(record.content)) {
return record.isError === true ? { content: record.content, isError: true } : { content: record.content };
}
return { content: [{ type: "text", text: typeof result === "string" ? result : JSON.stringify(result) }] };
}
function targetsAsString(renderArgs) {
const { targets } = renderArgs;
if (Array.isArray(targets))
return targets.join(", ");
if (typeof targets === "string")
return targets;
return undefined;
}
/**
* Only the unsloth preset defines beforeToolCall (see made-for-bionic-core's presets/unsloth.ts)
* — it re-syncs newly discovered attachments into chat_media_state.json and validates any aN
* token this call referenced. `threadId` is the raw scratchpadFolder value the client passed:
* for unsloth that IS the chat_threads.id (see `cat .unsloth_sandbox`), not a resolved path.
*/
async function runBeforeToolCallHook(config, toolArgs, scratchpadPath, renderArgs) {
if (!config.preset.beforeToolCall)
return { blocked: false };
return config.preset.beforeToolCall({
scratchpadPath,
threadId: toolArgs.scratchpadFolder ?? "",
clientDbPath: config.clientDbLocation,
attachmentTokens: collectAttachmentTokens(renderArgs),
});
}
/**
* Resolves each basename/absolute-path/preview-filename token in `targets` (the 3 non-aN/vN/iN/pN
* MCP input shapes — see made-for-bionic-core's resolveMcpSourceToken()) into a containment-checked
* absolute path to the ORIGINAL file, BEFORE core/tools.ts's handler ever sees it. aN/vN/iN/pN
* tokens are left untouched for the handler's own notation parser. Mutates `renderArgs` in place.
*/
async function preResolveTargetsField(renderArgs, field, scratchpadPath) {
const raw = renderArgs[field];
if (raw === undefined)
return;
const wasArray = Array.isArray(raw);
const tokens = normalizeSourceFieldToArray(raw);
if (tokens.length === 0)
return;
const resolved = [];
for (const token of tokens) {
if (looksLikeSourceNotation(token)) {
resolved.push(token);
continue;
}
const abs = await resolveMcpSourceToken(token, scratchpadPath);
resolved.push(abs ?? token);
}
renderArgs[field] = wasArray || resolved.length > 1 ? resolved : resolved[0];
}
/**
* Runs detect_object/annotate_image, then diffs chat_media_state.json to find the iN entries it
* just appended (see resultMaterializer.ts) and turns them into a tool result, per the resolved
* preset (made-for-bionic-core's presets/{bionic,generic}.ts) and this tool's fixed tier
* "process" (see planning/analyse-image-mcp-plan.md, "Materialisierungs-Stufe").
*/
async function runAndMaterialize(tool, tier, scratchpadPath, scratchpadFolder, preset, renderArgs, requestId, render) {
const before = await snapshotImageIndices(scratchpadPath);
const result = await render();
const record = result;
if (record?.isError === true)
return toToolResult(result);
const materialized = await materializeNewImages(scratchpadPath, before);
if (materialized.length === 0)
return toToolResult(result);
if (preset.includeBase64Preview) {
const summaryText = extractSummaryText(result);
await logger.logEvent("tool-materialized", { tool, scratchpadPath, notations: materialized.map((m) => m.notation).join(",") });
if (preset.name === "unsloth") {
return buildUnslothToolResult(tool, tier, scratchpadPath, scratchpadFolder, materialized, summaryText);
}
return buildGenericToolResult(tool, tier, scratchpadPath, scratchpadFolder, materialized, summaryText);
}
if (materialized.length === 1 || !preset.writeHtmlReportForMultipleResults) {
const summary = extractSummary(result);
await logger.logEvent("tool-materialized", { tool, scratchpadPath, notations: materialized[0].notation });
return { content: [{ type: "text", text: buildSingleResultText(tool, tier, materialized[0], summary) }] };
}
const summary = extractSummary(result);
// requestId was generated by the caller BEFORE render() ran and passed into the handler, which
// used the exact same value for this call's audit-log entry (see core/tools.ts) — reused as-is
// here so the HTML report filename always matches, including for a batch (2+ target) call whose
// own per-target summaries serialize as a top-level array (extractSummary() can't recover a
// requestId from that shape, so it must be threaded through explicitly, not re-derived).
const reportPath = await writeHtmlReport(scratchpadPath, requestId, {
tool,
task: typeof renderArgs.task === "string" ? renderArgs.task : undefined,
source: targetsAsString(renderArgs),
summary,
}, materialized);
await logger.logEvent("tool-materialized", { tool, scratchpadPath, notations: materialized.map((m) => m.notation).join(","), reportPath });
return { content: [{ type: "text", text: buildMultiResultText(tool, tier, reportPath, materialized) }] };
}
/** Same compact "<strong>Label:</strong> value • ..." single-line style as resultMaterializer.ts's metadataCell(). */
function formatMetaBullets(meta) {
const parts = [];
const add = (label, value) => {
if (value === undefined || value === null || value === "")
return;
parts.push(`<strong>${escapeHtml(label)}:</strong> ${escapeHtml(String(value))}`);
};
add("Prompt", meta.prompt);
add("Negative Prompt", meta.negativePrompt);
add("Model", meta.model);
const techParts = [];
if (meta.sampler)
techParts.push(`Sampler: ${meta.sampler}`);
if (typeof meta.steps === "number")
techParts.push(`Steps: ${meta.steps}`);
if (typeof meta.guidanceScale === "number")
techParts.push(`Guidance Scale: ${meta.guidanceScale}`);
if (typeof meta.seed === "number")
techParts.push(`Seed: ${meta.seed}`);
if (techParts.length > 0)
parts.push(escapeHtml(techParts.join(" ")));
add("Size", meta.size);
if (typeof meta.strength === "number" && meta.strength !== 1.0)
add("Strength", meta.strength);
if (meta.loras && meta.loras.length > 0) {
add("LoRA", meta.loras.map((l) => (l.weight != null ? `${l.file} (${l.weight})` : l.file)).join(", "));
}
if (meta.sources && meta.sources.length > 0)
add("Source(s)", meta.sources.join(", "));
return parts.join(" • ");
}
/** analyse_image never materializes a file — it always additionally writes an HTML preview of its result, one row per analysed item with a source-image thumbnail (mirrors detect_object/annotate_image's report style, see resultMaterializer.ts). */
async function writeAnalysisHtmlReport(scratchpadPath, requestId, items) {
const rows = items
.map((item) => {
// item.id is either a short notation (e.g. "i14") or the raw path/basename the caller
// passed for a non-notation target — in the latter case it's identical to (or a superset
// of) displayName, so showing both would just duplicate the same filename twice.
const idIsPathLike = item.displayName !== undefined && path.basename(item.id) === item.displayName;
const label = idIsPathLike
? `<span class="rank">${escapeHtml(item.displayName)}</span>`
: `<span class="rank">${escapeHtml(item.id)}</span>${item.displayName ? ` • ${escapeHtml(item.displayName)}` : ""}`;
const visualHtml = item.visual ? `<p style="margin:8px 0 0">${escapeHtml(item.visual)}</p>` : "";
const metaLine = item.meta ? formatMetaBullets(item.meta) : (item.metaNote ? escapeHtml(item.metaNote) : "");
const metaHtml = metaLine ? `<p style="margin:8px 0 0;font-size:11px;opacity:.85">${metaLine}</p>` : "";
const preview = item.filePath ? `<img src="${escapeHtml(item.filePath)}" alt="Preview ${escapeHtml(item.id)}">` : "";
return `<tr><td>${preview}</td><td>${label}${visualHtml}${metaHtml}</td></tr>`;
})
.join("\n");
const html = renderReportShell(`analyse_image ${requestId}`, "<tr><th>Preview</th><th>Analysis</th></tr>", rows);
const reportPath = path.join(scratchpadPath, `${requestId}.html`);
await fs.writeFile(reportPath, html, "utf8");
return reportPath;
}
async function main() {
// Logged BEFORE readMcpConfig() so a full, unfiltered record of what Bionic actually passed
// in always reaches analyse-image-mcp.log — every var, whether explicitly set, defaulted, or
// unset/empty.
await logger.logEvent("env snapshot (raw)", getRawEnvSnapshot());
const config = readMcpConfig();
applyMcpConfig(config);
const preset = config.preset;
await logger.logEvent("MCP server starting", {
preset: preset.name,
presetIncludeBase64Preview: preset.includeBase64Preview,
presetWriteHtmlReportForMultipleResults: preset.writeHtmlReportForMultipleResults,
presetScratchpadFolderRequired: preset.scratchpadFolderRequired,
visionApiBaseUrl: config.visionApiBaseUrl,
qwen3VlModel: config.qwen3VlModel,
embedPngMetadata: config.embedPngMetadata,
includeGenerationMetadata: config.includeGenerationMetadata,
chatWorkingDirectories: config.chatWorkingDirectories,
nodeVersion: process.version,
});
startStdioMcpServer({
name: "analyse-image-mcp",
version: "0.1.0",
buildServer: (server) => {
server.registerTool("analyse_image", { description: analyseImageDescription(preset, preset.scratchpadFolderRequired), inputSchema: fromJsonSchema(toMcpInputSchema(AnalyseImageParamsShape, preset.scratchpadFolderRequired, preset)) }, async (args, ctx) => {
const toolArgs = args;
try {
const scratchpadPath = await resolveEffectiveScratchpadPath(config.chatWorkingDirectories, toolArgs.scratchpadFolder, preset.scratchpadFolderRequired, preset);
await logger.logEvent("tool-request", { tool: "analyse_image", scratchpadPath });
const { scratchpadFolder: _scratchpadFolder, ...renderArgsRaw } = toolArgs;
assertNoAttachmentNotation(renderArgsRaw, preset);
const renderArgs = renderArgsRaw;
const hookResult = await runBeforeToolCallHook(config, toolArgs, scratchpadPath, renderArgs);
if (hookResult.blocked)
return { content: [{ type: "text", text: hookResult.message }] };
await preResolveTargetsField(renderArgs, "targets", scratchpadPath);
bindChatContextToScratchpad(scratchpadPath);
// Generated once and reused both as this call's audit-log requestId (core/tools.ts)
// and its HTML preview's filename — matches generate_image's convention.
const requestId = crypto.randomUUID().slice(0, 8);
const onProgress = createProgressForwarder(ctx, "analyse_image");
const items = [];
const analysisText = await (await import("../core/tools.js")).handleAnalyseImage(renderArgs, {}, {
status: (message) => {
onProgress?.(message);
logger.logEvent("tool-status", { tool: "analyse_image", message }).catch(() => { });
},
}, requestId, (item) => items.push(item));
const content = [{ type: "text", text: analysisText }];
if (preset.writeHtmlReportForMultipleResults) {
const reportPath = await writeAnalysisHtmlReport(scratchpadPath, requestId, items);
await logger.logEvent("tool-materialized", { tool: "analyse_image", scratchpadPath, reportPath });
content.push({ type: "text", text: `HTML preview (optional to open via open_url_in_app_browser): ${new URL(`file://${reportPath}`).href}` });
}
return { content };
}
catch (error) {
return await bridgeToolErrorResult("analyse_image", error, logger, preset);
}
});
function registerImageTool(name, shape, description, handlerImport) {
server.registerTool(name, { description: description(preset, preset.scratchpadFolderRequired), inputSchema: fromJsonSchema(toMcpInputSchema(shape, preset.scratchpadFolderRequired, preset)) }, async (args, ctx) => {
const toolArgs = args;
try {
const scratchpadPath = await resolveEffectiveScratchpadPath(config.chatWorkingDirectories, toolArgs.scratchpadFolder, preset.scratchpadFolderRequired, preset);
await logger.logEvent("tool-request", { tool: name, scratchpadPath });
const { scratchpadFolder: _scratchpadFolder, ...renderArgsRaw } = toolArgs;
assertNoAttachmentNotation(renderArgsRaw, preset);
const renderArgs = renderArgsRaw;
const hookResult = await runBeforeToolCallHook(config, toolArgs, scratchpadPath, renderArgs);
if (hookResult.blocked)
return { content: [{ type: "text", text: hookResult.message }] };
await preResolveTargetsField(renderArgs, "targets", scratchpadPath);
bindChatContextToScratchpad(scratchpadPath);
// Generated once and reused both as this call's audit-log requestId (core/tools.ts)
// and its HTML report's filename (see runAndMaterialize) — matches analyse_image.
const requestId = crypto.randomUUID().slice(0, 8);
const onProgress = createProgressForwarder(ctx, name);
return await runAndMaterialize(name, "process", scratchpadPath, toolArgs.scratchpadFolder, preset, renderArgs, requestId, async () => {
const handle = await handlerImport();
return handle(renderArgs, {}, {
status: (message) => {
onProgress?.(message);
logger.logEvent("tool-status", { tool: name, message }).catch(() => { });
},
}, requestId);
});
}
catch (error) {
return await bridgeToolErrorResult(name, error, logger, preset);
}
});
}
registerImageTool("detect_object", DetectObjectParamsShape, detectObjectDescription, async () => (await import("../core/tools.js")).handleDetectObject);
registerImageTool("annotate_image", AnnotateImageParamsShape, annotateImageDescription, async () => (await import("../core/tools.js")).handleAnnotateImage);
},
});
}
main();
import crypto from "node:crypto";
import fs from "node:fs/promises";
import path from "node:path";
import { fromJsonSchema } from "@modelcontextprotocol/server";
import { zodToJsonSchema } from "zod-to-json-schema";
import { z } from "zod";
import { bridgeToolErrorResult, escapeHtml, renderReportShell, startStdioMcpServer, looksLikeSourceNotation, normalizeSourceFieldToArray, resolveMcpSourceToken } from "./core-bundle.mjs";
import { AnalyseImageParamsShape, AnnotateImageParamsShape, DetectObjectParamsShape } from "../core/tools.js";
import { applyMcpConfig, getRawEnvSnapshot, readMcpConfig } from "./config.js";
import { bindChatContextToScratchpad } from "./chatContextBridge.js";
import { assertNoAttachmentNotation, collectAttachmentTokens } from "./sourceNotation.js";
import { logger } from "./mcpLogger.js";
import { extractSummary, extractSummaryText, materializeNewImages, snapshotImageIndices, writeHtmlReport } from "./resultMaterializer.js";
import { resolveEffectiveScratchpadPath } from "./scratchpadResolution.js";
import { buildGenericToolResult, buildMultiResultText, buildSingleResultText, buildUnslothToolResult } from "./toolResults.js";
/**
* Forwards analyse_image/detect_object/annotate_image's already-formatted status messages (see
* reportToolStatus/reportToolStep in core/tools.ts) to MCP's notifications/progress, tied to the
* current tool call via ctx.mcpReq.notify (same mechanism as generate-image/process-image's own
* createProgressForwarder). Only sends anything when the client actually opted in by sending a
* progressToken with the request (per MCP spec) -- returns undefined otherwise, so no progress
* machinery runs for clients that never asked for it. This is NOT merely cosmetic: per the MCP
* spec, a progress notification resets the client's wait timer for that request -- Unsloth
* Desktop's MCP setup has no exposed request-timeout setting at all, so this is what actually
* keeps a slow vision-model-load + multi-image call from timing out there.
*/
function createProgressForwarder(ctx, tool) {
const mcpReq = ctx?.mcpReq;
const progressToken = mcpReq?._meta?.progressToken;
if (progressToken === undefined || typeof mcpReq?.notify !== "function")
return undefined;
let progress = 0;
return (message) => {
progress += 1;
mcpReq.notify({
method: "notifications/progress",
params: { progressToken, progress, message },
}).catch((error) => {
logger.logError("progress-notify-failed", error, { tool }).catch(() => { });
});
};
}
/**
* zod-to-json-schema collapses a union of only-primitive types (e.g. annotate_image's
* frameAdjust: z.union([z.number(), z.string()]), or the nested number|null union inside its
* x1/y1/x2/y2 array-item schemas) into a single node with `type: ["number", "string"]` instead of
* `anyOf` branches. Several MCP clients only accept `type` as a single string and either reject
* the tool or silently drop the constraint on such a node. Recursively rewrites every such node
* into `{ anyOf: [{ type: "number" }, { type: "string" }], ...rest }`, walking into
* properties/items/anyOf/allOf/oneOf/additionalProperties so nested cases are fixed too
* (see /memories/repo/process-image-mcp-implementation.md).
*/
function splitTypeArrayUnions(node) {
if (Array.isArray(node))
return node.map(splitTypeArrayUnions);
if (node === null || typeof node !== "object")
return node;
const { type, properties, items, anyOf, allOf, oneOf, additionalProperties, ...rest } = node;
const walked = { ...rest };
if (properties && typeof properties === "object") {
walked.properties = Object.fromEntries(Object.entries(properties).map(([key, value]) => [key, splitTypeArrayUnions(value)]));
}
if (items !== undefined)
walked.items = splitTypeArrayUnions(items);
if (additionalProperties !== undefined)
walked.additionalProperties = splitTypeArrayUnions(additionalProperties);
if (anyOf !== undefined)
walked.anyOf = anyOf.map(splitTypeArrayUnions);
if (allOf !== undefined)
walked.allOf = allOf.map(splitTypeArrayUnions);
if (oneOf !== undefined)
walked.oneOf = oneOf.map(splitTypeArrayUnions);
if (Array.isArray(type)) {
walked.anyOf = [...(Array.isArray(walked.anyOf) ? walked.anyOf : []), ...type.map((t) => ({ type: t }))];
}
else if (type !== undefined) {
walked.type = type;
}
return walked;
}
/**
* aN (LM-Studio attachment notation) doesn't exist over MCP (see sourceNotation.ts) — but the
* shared Zod `.describe()` text in core/tools.ts still advertises it, since that same text is
* also the LM-Studio plugin's own tool description (where aN IS valid). Rewriting core/tools.ts
* would break the plugin's description, so this strips the aN mention only from the JSON schema
* text actually sent to MCP clients, leaving core/tools.ts untouched.
*/
function stripAttachmentNotationMentions(node) {
if (Array.isArray(node))
return node.map(stripAttachmentNotationMentions);
if (node === null || typeof node !== "object")
return node;
const out = {};
for (const [key, value] of Object.entries(node)) {
out[key] =
key === "description" && typeof value === "string"
? value.replace(/a=attachment \(a1, a2, …\), /g, "")
: stripAttachmentNotationMentions(value);
}
return out;
}
const ATTACHMENT_INVENTORY_HINT = " Pass the literal \"aN\" (or any unregistered number) to get back a list of all attachments currently available in this thread instead of running the tool.";
/**
* unsloth is the one preset where aN genuinely resolves (see sourceNotation.ts) — instead of
* stripping the aN mention, append the inventory-listing hint right after it so the agent knows
* how to discover valid numbers.
*/
function annotateAttachmentNotationForUnsloth(node) {
if (Array.isArray(node))
return node.map(annotateAttachmentNotationForUnsloth);
if (node === null || typeof node !== "object")
return node;
const out = {};
for (const [key, value] of Object.entries(node)) {
out[key] = key === "description" && typeof value === "string" && value.includes("a=attachment") ? `${value}${ATTACHMENT_INVENTORY_HINT}` : annotateAttachmentNotationForUnsloth(value);
}
return out;
}
/**
* Adds the MCP-only scratchpadFolder field on top of the plugin's real Zod shape — no field is
* hand-duplicated. scratchpadFolder's own requiredness comes from the resolved preset
* (made-for-bionic-core) — see presets/index.ts's doc comment for why this must not be
* re-derived locally. Required for all three tools (including analyse_image): `targets` is
* always resolved against chat_media_state.json in the bound scratchpad.
*/
function toMcpInputSchema(shape, requireScratchpadFolder, preset) {
const { $schema: _$schema, ...rawSchema } = zodToJsonSchema(z.object(shape).strict());
const schema = (preset.name === "unsloth" ? annotateAttachmentNotationForUnsloth(splitTypeArrayUnions(rawSchema)) : stripAttachmentNotationMentions(splitTypeArrayUnions(rawSchema)));
const visibleProperties = (schema.properties ?? {});
const baseRequired = schema.required ?? [];
return {
...schema,
properties: {
scratchpadFolder: {
type: "string",
description: scratchpadFolderGuidance(preset, requireScratchpadFolder),
},
...visibleProperties,
},
required: requireScratchpadFolder ? ["scratchpadFolder", ...baseRequired] : baseRequired,
};
}
const SOURCE_NOTATION_GUIDANCE = "Sources: pass one or more notations in the `targets` array field, e.g. [\"i3\", \"v2\"] — i=generated image, v=variant, p=picture. A comma/space-separated string is also accepted, as well as a scratchpad filename or absolute path.";
function sourceNotationGuidance(preset) {
if (preset.name !== "unsloth")
return SOURCE_NOTATION_GUIDANCE;
return `${SOURCE_NOTATION_GUIDANCE} Also accepts aN (attachment, e.g. a1).${ATTACHMENT_INVENTORY_HINT}`;
}
/**
* unsloth has no get_scratchpad_folder tool (see made-for-preset-rollout-plan.md) — its agent
* obtains the same value via \`cat .unsloth_sandbox\`. Reusing bionic's "get_scratchpad_folder"
* wording there would be actively wrong, not just imprecise.
*/
function scratchpadFolderGuidance(preset, requireScratchpadFolder) {
if (!requireScratchpadFolder)
return "Optional scratchpad session folder name, one level under CHAT_WORKING_DIRECTORIES. Omit to write directly into CHAT_WORKING_DIRECTORIES.";
if (preset.name === "unsloth")
return "Required scratchpad session folder. Obtain this value by running `cat .unsloth_sandbox` and passing its output unchanged.";
return "Required scratchpad session folder name, one level under CHAT_WORKING_DIRECTORIES. Obtain this value with get_scratchpad_folder and pass it back unchanged.";
}
/** Mirrors the actual gate in main()'s analyse_image handler: `if (preset.writeHtmlReportForMultipleResults)`. */
function analyseImageHtmlPreviewGuidance(preset) {
if (preset.writeHtmlReportForMultipleResults)
return "It always also writes a plain HTML preview of its result into the scratchpad — opening it via open_url_in_app_browser is optional, the machine-readable text response below is the primary channel.";
return "No HTML preview file is written — the machine-readable text response below is the only output.";
}
function analyseImageDescription(preset, requireScratchpadFolder) {
return `Inspect existing media items (images, pictures, variants, attachments) for generation metadata and, optionally, visual content.
PRIMARY use — generation metadata:
Call this tool whenever the user asks how an image was generated, what settings were used, or wants to reuse generation parameters (prompt, model, sampler, seed, steps, guidance scale, LoRA, source images, …). Those parameters are embedded in the PNG file and are returned automatically — no vision prompt needed.
SECONDARY use — visual description (on demand only):
Only request a visual description when the user explicitly asks you to describe or analyze the image content. Pass a prompt in the 'prompt' parameter. Without a prompt, no vision model is invoked and no description is returned.
This tool never writes a new file. ${analyseImageHtmlPreviewGuidance(preset)}
${scratchpadFolderGuidance(preset, requireScratchpadFolder)}
${sourceNotationGuidance(preset)}`;
}
function detectObjectDescription(preset, requireScratchpadFolder) {
return `Detect objects in one or more images and draw colored bounding boxes on each result.
For each source image, returns a new annotated image (saved as iN) with bounding boxes for each detected object, plus a JSON summary with labels, coordinates, and crop percentages. Uses Qwen3-VL for detection.
${scratchpadFolderGuidance(preset, requireScratchpadFolder)}
${sourceNotationGuidance(preset)}`;
}
function annotateImageDescription(preset, requireScratchpadFolder) {
return `Highlights specific areas or elements on images. If well described, these elements are precisely framed with bounding boxes in the chosen color.
Detections are saved to state for later correction, refinement, or re-drawing with adjusted labels and edge positions. Omit task on a correction call to redraw from stored detections without new inference; pass task to force fresh inference.
${scratchpadFolderGuidance(preset, requireScratchpadFolder)}
${sourceNotationGuidance(preset)}`;
}
// The MCP SDK's CallToolResult content-block union isn't re-exported for reuse here,
// so the handler result stays structurally typed (any) rather than hand-duplicating it.
function toToolResult(result) {
const record = result;
if (record && Array.isArray(record.content)) {
return record.isError === true ? { content: record.content, isError: true } : { content: record.content };
}
return { content: [{ type: "text", text: typeof result === "string" ? result : JSON.stringify(result) }] };
}
function targetsAsString(renderArgs) {
const { targets } = renderArgs;
if (Array.isArray(targets))
return targets.join(", ");
if (typeof targets === "string")
return targets;
return undefined;
}
/**
* Only the unsloth preset defines beforeToolCall (see made-for-bionic-core's presets/unsloth.ts)
* — it re-syncs newly discovered attachments into chat_media_state.json and validates any aN
* token this call referenced. `threadId` is the raw scratchpadFolder value the client passed:
* for unsloth that IS the chat_threads.id (see `cat .unsloth_sandbox`), not a resolved path.
*/
async function runBeforeToolCallHook(config, toolArgs, scratchpadPath, renderArgs) {
if (!config.preset.beforeToolCall)
return { blocked: false };
return config.preset.beforeToolCall({
scratchpadPath,
threadId: toolArgs.scratchpadFolder ?? "",
clientDbPath: config.clientDbLocation,
attachmentTokens: collectAttachmentTokens(renderArgs),
});
}
/**
* Resolves each basename/absolute-path/preview-filename token in `targets` (the 3 non-aN/vN/iN/pN
* MCP input shapes — see made-for-bionic-core's resolveMcpSourceToken()) into a containment-checked
* absolute path to the ORIGINAL file, BEFORE core/tools.ts's handler ever sees it. aN/vN/iN/pN
* tokens are left untouched for the handler's own notation parser. Mutates `renderArgs` in place.
*/
async function preResolveTargetsField(renderArgs, field, scratchpadPath) {
const raw = renderArgs[field];
if (raw === undefined)
return;
const wasArray = Array.isArray(raw);
const tokens = normalizeSourceFieldToArray(raw);
if (tokens.length === 0)
return;
const resolved = [];
for (const token of tokens) {
if (looksLikeSourceNotation(token)) {
resolved.push(token);
continue;
}
const abs = await resolveMcpSourceToken(token, scratchpadPath);
resolved.push(abs ?? token);
}
renderArgs[field] = wasArray || resolved.length > 1 ? resolved : resolved[0];
}
/**
* Runs detect_object/annotate_image, then diffs chat_media_state.json to find the iN entries it
* just appended (see resultMaterializer.ts) and turns them into a tool result, per the resolved
* preset (made-for-bionic-core's presets/{bionic,generic}.ts) and this tool's fixed tier
* "process" (see planning/analyse-image-mcp-plan.md, "Materialisierungs-Stufe").
*/
async function runAndMaterialize(tool, tier, scratchpadPath, scratchpadFolder, preset, renderArgs, requestId, render) {
const before = await snapshotImageIndices(scratchpadPath);
const result = await render();
const record = result;
if (record?.isError === true)
return toToolResult(result);
const materialized = await materializeNewImages(scratchpadPath, before);
if (materialized.length === 0)
return toToolResult(result);
if (preset.includeBase64Preview) {
const summaryText = extractSummaryText(result);
await logger.logEvent("tool-materialized", { tool, scratchpadPath, notations: materialized.map((m) => m.notation).join(",") });
if (preset.name === "unsloth") {
return buildUnslothToolResult(tool, tier, scratchpadPath, scratchpadFolder, materialized, summaryText);
}
return buildGenericToolResult(tool, tier, scratchpadPath, scratchpadFolder, materialized, summaryText);
}
if (materialized.length === 1 || !preset.writeHtmlReportForMultipleResults) {
const summary = extractSummary(result);
await logger.logEvent("tool-materialized", { tool, scratchpadPath, notations: materialized[0].notation });
return { content: [{ type: "text", text: buildSingleResultText(tool, tier, materialized[0], summary) }] };
}
const summary = extractSummary(result);
// requestId was generated by the caller BEFORE render() ran and passed into the handler, which
// used the exact same value for this call's audit-log entry (see core/tools.ts) — reused as-is
// here so the HTML report filename always matches, including for a batch (2+ target) call whose
// own per-target summaries serialize as a top-level array (extractSummary() can't recover a
// requestId from that shape, so it must be threaded through explicitly, not re-derived).
const reportPath = await writeHtmlReport(scratchpadPath, requestId, {
tool,
task: typeof renderArgs.task === "string" ? renderArgs.task : undefined,
source: targetsAsString(renderArgs),
summary,
}, materialized);
await logger.logEvent("tool-materialized", { tool, scratchpadPath, notations: materialized.map((m) => m.notation).join(","), reportPath });
return { content: [{ type: "text", text: buildMultiResultText(tool, tier, reportPath, materialized) }] };
}
/** Same compact "<strong>Label:</strong> value • ..." single-line style as resultMaterializer.ts's metadataCell(). */
function formatMetaBullets(meta) {
const parts = [];
const add = (label, value) => {
if (value === undefined || value === null || value === "")
return;
parts.push(`<strong>${escapeHtml(label)}:</strong> ${escapeHtml(String(value))}`);
};
add("Prompt", meta.prompt);
add("Negative Prompt", meta.negativePrompt);
add("Model", meta.model);
const techParts = [];
if (meta.sampler)
techParts.push(`Sampler: ${meta.sampler}`);
if (typeof meta.steps === "number")
techParts.push(`Steps: ${meta.steps}`);
if (typeof meta.guidanceScale === "number")
techParts.push(`Guidance Scale: ${meta.guidanceScale}`);
if (typeof meta.seed === "number")
techParts.push(`Seed: ${meta.seed}`);
if (techParts.length > 0)
parts.push(escapeHtml(techParts.join(" ")));
add("Size", meta.size);
if (typeof meta.strength === "number" && meta.strength !== 1.0)
add("Strength", meta.strength);
if (meta.loras && meta.loras.length > 0) {
add("LoRA", meta.loras.map((l) => (l.weight != null ? `${l.file} (${l.weight})` : l.file)).join(", "));
}
if (meta.sources && meta.sources.length > 0)
add("Source(s)", meta.sources.join(", "));
return parts.join(" • ");
}
/** analyse_image never materializes a file — it always additionally writes an HTML preview of its result, one row per analysed item with a source-image thumbnail (mirrors detect_object/annotate_image's report style, see resultMaterializer.ts). */
async function writeAnalysisHtmlReport(scratchpadPath, requestId, items) {
const rows = items
.map((item) => {
// item.id is either a short notation (e.g. "i14") or the raw path/basename the caller
// passed for a non-notation target — in the latter case it's identical to (or a superset
// of) displayName, so showing both would just duplicate the same filename twice.
const idIsPathLike = item.displayName !== undefined && path.basename(item.id) === item.displayName;
const label = idIsPathLike
? `<span class="rank">${escapeHtml(item.displayName)}</span>`
: `<span class="rank">${escapeHtml(item.id)}</span>${item.displayName ? ` • ${escapeHtml(item.displayName)}` : ""}`;
const visualHtml = item.visual ? `<p style="margin:8px 0 0">${escapeHtml(item.visual)}</p>` : "";
const metaLine = item.meta ? formatMetaBullets(item.meta) : (item.metaNote ? escapeHtml(item.metaNote) : "");
const metaHtml = metaLine ? `<p style="margin:8px 0 0;font-size:11px;opacity:.85">${metaLine}</p>` : "";
const preview = item.filePath ? `<img src="${escapeHtml(item.filePath)}" alt="Preview ${escapeHtml(item.id)}">` : "";
return `<tr><td>${preview}</td><td>${label}${visualHtml}${metaHtml}</td></tr>`;
})
.join("\n");
const html = renderReportShell(`analyse_image ${requestId}`, "<tr><th>Preview</th><th>Analysis</th></tr>", rows);
const reportPath = path.join(scratchpadPath, `${requestId}.html`);
await fs.writeFile(reportPath, html, "utf8");
return reportPath;
}
async function main() {
// Logged BEFORE readMcpConfig() so a full, unfiltered record of what Bionic actually passed
// in always reaches analyse-image-mcp.log — every var, whether explicitly set, defaulted, or
// unset/empty.
await logger.logEvent("env snapshot (raw)", getRawEnvSnapshot());
const config = readMcpConfig();
applyMcpConfig(config);
const preset = config.preset;
await logger.logEvent("MCP server starting", {
preset: preset.name,
presetIncludeBase64Preview: preset.includeBase64Preview,
presetWriteHtmlReportForMultipleResults: preset.writeHtmlReportForMultipleResults,
presetScratchpadFolderRequired: preset.scratchpadFolderRequired,
visionApiBaseUrl: config.visionApiBaseUrl,
qwen3VlModel: config.qwen3VlModel,
embedPngMetadata: config.embedPngMetadata,
includeGenerationMetadata: config.includeGenerationMetadata,
chatWorkingDirectories: config.chatWorkingDirectories,
nodeVersion: process.version,
});
startStdioMcpServer({
name: "analyse-image-mcp",
version: "0.1.0",
buildServer: (server) => {
server.registerTool("analyse_image", { description: analyseImageDescription(preset, preset.scratchpadFolderRequired), inputSchema: fromJsonSchema(toMcpInputSchema(AnalyseImageParamsShape, preset.scratchpadFolderRequired, preset)) }, async (args, ctx) => {
const toolArgs = args;
try {
const scratchpadPath = await resolveEffectiveScratchpadPath(config.chatWorkingDirectories, toolArgs.scratchpadFolder, preset.scratchpadFolderRequired, preset);
await logger.logEvent("tool-request", { tool: "analyse_image", scratchpadPath });
const { scratchpadFolder: _scratchpadFolder, ...renderArgsRaw } = toolArgs;
assertNoAttachmentNotation(renderArgsRaw, preset);
const renderArgs = renderArgsRaw;
const hookResult = await runBeforeToolCallHook(config, toolArgs, scratchpadPath, renderArgs);
if (hookResult.blocked)
return { content: [{ type: "text", text: hookResult.message }] };
await preResolveTargetsField(renderArgs, "targets", scratchpadPath);
bindChatContextToScratchpad(scratchpadPath);
// Generated once and reused both as this call's audit-log requestId (core/tools.ts)
// and its HTML preview's filename — matches generate_image's convention.
const requestId = crypto.randomUUID().slice(0, 8);
const onProgress = createProgressForwarder(ctx, "analyse_image");
const items = [];
const analysisText = await (await import("../core/tools.js")).handleAnalyseImage(renderArgs, {}, {
status: (message) => {
onProgress?.(message);
logger.logEvent("tool-status", { tool: "analyse_image", message }).catch(() => { });
},
}, requestId, (item) => items.push(item));
const content = [{ type: "text", text: analysisText }];
if (preset.writeHtmlReportForMultipleResults) {
const reportPath = await writeAnalysisHtmlReport(scratchpadPath, requestId, items);
await logger.logEvent("tool-materialized", { tool: "analyse_image", scratchpadPath, reportPath });
content.push({ type: "text", text: `HTML preview (optional to open via open_url_in_app_browser): ${new URL(`file://${reportPath}`).href}` });
}
return { content };
}
catch (error) {
return await bridgeToolErrorResult("analyse_image", error, logger, preset);
}
});
function registerImageTool(name, shape, description, handlerImport) {
server.registerTool(name, { description: description(preset, preset.scratchpadFolderRequired), inputSchema: fromJsonSchema(toMcpInputSchema(shape, preset.scratchpadFolderRequired, preset)) }, async (args, ctx) => {
const toolArgs = args;
try {
const scratchpadPath = await resolveEffectiveScratchpadPath(config.chatWorkingDirectories, toolArgs.scratchpadFolder, preset.scratchpadFolderRequired, preset);
await logger.logEvent("tool-request", { tool: name, scratchpadPath });
const { scratchpadFolder: _scratchpadFolder, ...renderArgsRaw } = toolArgs;
assertNoAttachmentNotation(renderArgsRaw, preset);
const renderArgs = renderArgsRaw;
const hookResult = await runBeforeToolCallHook(config, toolArgs, scratchpadPath, renderArgs);
if (hookResult.blocked)
return { content: [{ type: "text", text: hookResult.message }] };
await preResolveTargetsField(renderArgs, "targets", scratchpadPath);
bindChatContextToScratchpad(scratchpadPath);
// Generated once and reused both as this call's audit-log requestId (core/tools.ts)
// and its HTML report's filename (see runAndMaterialize) — matches analyse_image.
const requestId = crypto.randomUUID().slice(0, 8);
const onProgress = createProgressForwarder(ctx, name);
return await runAndMaterialize(name, "process", scratchpadPath, toolArgs.scratchpadFolder, preset, renderArgs, requestId, async () => {
const handle = await handlerImport();
return handle(renderArgs, {}, {
status: (message) => {
onProgress?.(message);
logger.logEvent("tool-status", { tool: name, message }).catch(() => { });
},
}, requestId);
});
}
catch (error) {
return await bridgeToolErrorResult(name, error, logger, preset);
}
});
}
registerImageTool("detect_object", DetectObjectParamsShape, detectObjectDescription, async () => (await import("../core/tools.js")).handleDetectObject);
registerImageTool("annotate_image", AnnotateImageParamsShape, annotateImageDescription, async () => (await import("../core/tools.js")).handleAnnotateImage);
},
});
}
main();