src / core / godotDocs.ts
/**
* Godot documentation lookup tools.
*
* The Godot docs site (docs.godotengine.org) renders its search results with
* client-side JavaScript only, so a plain HTTP GET of `search.html` returns an
* empty result list. Instead we query the underlying Sphinx search index
* (`searchindex.js`, served per-doc-version) directly on the server side.
*
* The index exposes an inverted term map: `terms[<term>]` is an array of
* document indices into `docnames`/`titles`. We turn that into
* {title, url, snippet} results without needing a browser.
*
* These helpers do NOT require the Godot binary, so they are safe to expose to
* models even when no project is open.
*/
import axios from "axios";
export interface DocSearchResult {
title: string;
url: string;
snippet: string;
}
interface DocIndex {
terms: Record<string, number[]>;
docnames: string[];
titles: Record<string, string>;
}
const DEFAULT_TIMEOUT_MS = 20000;
const MAX_OUTPUT_CHARS = 20000;
const CACHE_TTL_MS = 24 * 60 * 60 * 1000; // refresh stale index once a day
const USER_AGENT =
"Mozilla/5.0 (godot-lmstudio-plugin; +https://github.com/Coding-Solo/godot-mcp)";
// In-memory cache of parsed search indices, keyed by doc version.
const indexCache = new Map<string, { index: DocIndex; fetchedAt: number }>();
function safeVersion(version: string): string {
return version && version !== "stable" ? version : "stable";
}
function isOkStatus(status: number | string): boolean {
return /^\d{3}$/.test(String(status)) && Number(status) >= 200 && Number(status) < 400;
}
async function fetchText(
url: string,
opts: { timeoutMs?: number; signal?: AbortSignal }
): Promise<string> {
let response: { data: unknown; status: number };
try {
response = await axios.get(url, {
timeout: opts.timeoutMs ?? DEFAULT_TIMEOUT_MS,
signal: opts.signal,
responseType: "text",
validateStatus: (status) => isOkStatus(status),
headers: { "User-Agent": USER_AGENT },
});
} catch (error) {
const reason = error instanceof Error ? error.message : "Unknown network error";
throw new Error(`Network request to ${url} failed: ${reason}`);
}
if (!isOkStatus(response.status)) {
throw new Error(`Request to ${url} returned HTTP ${response.status}.`);
}
const data = response.data;
return typeof data === "string" ? data : String(data ?? "");
}
/**
* Load (and cache) the parsed search index for a documentation version.
*/
async function loadDocIndex(
version: string,
opts: { timeoutMs?: number; signal?: AbortSignal } = {}
): Promise<DocIndex> {
const key = safeVersion(version);
const cached = indexCache.get(key);
if (cached && Date.now() - cached.fetchedAt < CACHE_TTL_MS) {
return cached.index;
}
const url = `https://docs.godotengine.org/en/${key}/searchindex.js`;
const raw = await fetchText(url, opts);
// The file is exactly: Search.setIndex({ ...json... });
const jsonBody = raw
.replace(/^Search\.setIndex\(/, "")
.replace(/\)\s*;?\s*$/, "");
let parsed: any;
try {
parsed = JSON.parse(jsonBody);
} catch (error) {
throw new Error(`Failed to parse Godot search index: ${error instanceof Error ? error.message : "Unknown error"}`);
}
const index: DocIndex = {
terms: parsed.terms ?? {},
docnames: parsed.docnames ?? [],
titles: parsed.titles ?? {},
};
indexCache.set(key, { index, fetchedAt: Date.now() });
return index;
}
function buildDocUrl(version: string, docname: string): string {
const clean = docname.replace(/\.html?$/i, "");
return `https://docs.godotengine.org/en/${safeVersion(version)}/${clean}.html`;
}
/**
* Normalize a raw search-index hit list. Different Sphinx versions store hits
* either as flat `[docIdx, ...]` or as `[docIdx, wordPos]` pairs; sometimes a
* single value is stored as a scalar. This unifies all of those into an array
* of numeric document indices.
*/
function collectDocIndexes(raw: unknown): number[] {
if (!Array.isArray(raw)) return [];
const out: number[] = [];
for (const h of raw) {
if (Array.isArray(h)) {
const d = h[0];
if (typeof d === "number") out.push(d);
} else if (typeof h === "number") {
out.push(h);
}
}
return out;
}
/**
* Search the Godot documentation and return up to `limit` candidates.
*
* Matching strategy (mirrors how a model would want it):
* - Exact term matches are always preferred.
* - Multi-word queries prefer documents containing ALL terms; if none match,
* we fall back to documents matching ANY term (ranked by frequency).
* - If there is no exact hit at all, we relax to prefix / substring matches.
*/
export async function searchDocs(
query: string,
version: string,
limit: number,
opts: { timeoutMs?: number; signal?: AbortSignal } = {}
): Promise<DocSearchResult[]> {
const index = await loadDocIndex(version, opts);
const queryTerms = (query ?? "")
.toLowerCase()
.split(/\s+/)
.map((t) => t.trim())
.filter(Boolean);
if (queryTerms.length === 0) return [];
// docIndex -> { score (terms matched), terms, positions }
const scores = new Map<
number,
{ score: number; terms: Set<string>; positions: number[] }
>();
const bump = (docIdx: number, term: string, position: number): void => {
const e = scores.get(docIdx);
if (!e) {
scores.set(docIdx, { score: 1, terms: new Set([term]), positions: [position] });
} else {
e.score += 1;
e.terms.add(term);
e.positions.push(position);
}
};
// 1) Exact matches.
let exactHits = 0;
for (const term of queryTerms) {
const hits = collectDocIndexes(index.terms[term]);
if (!hits.length) continue;
exactHits += hits.length;
hits.forEach((docIdx, position) => bump(docIdx, term, position));
}
// 2) Relaxation fallback when nothing matched exactly.
if (exactHits === 0) {
for (const [term, raw] of Object.entries(index.terms)) {
const hits = collectDocIndexes(raw);
const hit = queryTerms.some(
(q) => term === q || term.startsWith(q) || term.includes(q)
);
if (!hit || !hits.length) continue;
hits.forEach((docIdx) => bump(docIdx, term, 0));
}
}
// 3) Ranking: prefer AND matches, then by score.
const multiWord = queryTerms.length > 1;
const ranked = [...scores.entries()]
.map(([docIdx, info]) => ({ docIdx, ...info }))
.sort((a, b) => {
if (multiWord) {
// Prefer docs matching more distinct query terms.
const aMatched = queryTerms.filter((t) => a.terms.has(t)).length;
const bMatched = queryTerms.filter((t) => b.terms.has(t)).length;
if (aMatched !== bMatched) return bMatched - aMatched;
}
if (b.score !== a.score) return b.score - a.score;
// Tiebreak: fewer references is usually more focused/relevant.
return (a.positions.length ?? 0) - (b.positions.length ?? 0);
});
const top = ranked.slice(0, Math.max(1, Math.trunc(limit)));
return top.map(({ docIdx }) => {
const title =
index.titles[String(docIdx)] ||
index.docnames[docIdx] ||
"(untitled)";
const url = buildDocUrl(version, index.docnames[docIdx] ?? "");
return { title, url, snippet: "" };
});
}
/**
* Strip scripts, styles and navigation chrome from an HTML document.
*/
function stripChrome(html: string): string {
let out = html;
out = out.replace(/<script[\s\S]*?<\/script>/gi, " ");
out = out.replace(/<style[\s\S]*?<\/style>/gi, " ");
out = out.replace(/<noscript[\s\S]*?<\/noscript>/gi, " ");
out = out.replace(
/<(nav|header|footer|aside|form|button)\b[^>]*>[\s\S]*?<\/\1>/gi,
" "
);
out = out.replace(
/<div\b[^>]*\b(?:role\s*=\s*"navigation"|side-links|rst-aside|sidebar|prev-next|footer)\b[^>]*>[\s\S]*?<\/div>/gi,
" "
);
return out;
}
function collapseWhitespace(text: string): string {
return text.replace(/\s+/g, " ").trim();
}
/**
* Extract readable text from an HTML document body.
*/
function extractReadableText(html: string): string {
const chromeRemoved = stripChrome(html);
return collapseWhitespace(chromeRemoved.replace(/<[^>]+>/g, " "));
}
/**
* Fetch and reduce a documentation page into readable text.
*/
export async function fetchDocPage(
url: string,
opts: { maxChars?: number; timeoutMs?: number; signal?: AbortSignal } = {}
): Promise<string> {
const maxChars = opts.maxChars ?? MAX_OUTPUT_CHARS;
const text = await fetchText(url, opts);
const readable = extractReadableText(text);
if (readable.length > maxChars) {
return `${readable.slice(0, maxChars)}\n\n... [truncated]`;
}
return readable;
}
src / core / godotDocs.ts
/**
* Godot documentation lookup tools.
*
* The Godot docs site (docs.godotengine.org) renders its search results with
* client-side JavaScript only, so a plain HTTP GET of `search.html` returns an
* empty result list. Instead we query the underlying Sphinx search index
* (`searchindex.js`, served per-doc-version) directly on the server side.
*
* The index exposes an inverted term map: `terms[<term>]` is an array of
* document indices into `docnames`/`titles`. We turn that into
* {title, url, snippet} results without needing a browser.
*
* These helpers do NOT require the Godot binary, so they are safe to expose to
* models even when no project is open.
*/
import axios from "axios";
export interface DocSearchResult {
title: string;
url: string;
snippet: string;
}
interface DocIndex {
terms: Record<string, number[]>;
docnames: string[];
titles: Record<string, string>;
}
const DEFAULT_TIMEOUT_MS = 20000;
const MAX_OUTPUT_CHARS = 20000;
const CACHE_TTL_MS = 24 * 60 * 60 * 1000; // refresh stale index once a day
const USER_AGENT =
"Mozilla/5.0 (godot-lmstudio-plugin; +https://github.com/Coding-Solo/godot-mcp)";
// In-memory cache of parsed search indices, keyed by doc version.
const indexCache = new Map<string, { index: DocIndex; fetchedAt: number }>();
function safeVersion(version: string): string {
return version && version !== "stable" ? version : "stable";
}
function isOkStatus(status: number | string): boolean {
return /^\d{3}$/.test(String(status)) && Number(status) >= 200 && Number(status) < 400;
}
async function fetchText(
url: string,
opts: { timeoutMs?: number; signal?: AbortSignal }
): Promise<string> {
let response: { data: unknown; status: number };
try {
response = await axios.get(url, {
timeout: opts.timeoutMs ?? DEFAULT_TIMEOUT_MS,
signal: opts.signal,
responseType: "text",
validateStatus: (status) => isOkStatus(status),
headers: { "User-Agent": USER_AGENT },
});
} catch (error) {
const reason = error instanceof Error ? error.message : "Unknown network error";
throw new Error(`Network request to ${url} failed: ${reason}`);
}
if (!isOkStatus(response.status)) {
throw new Error(`Request to ${url} returned HTTP ${response.status}.`);
}
const data = response.data;
return typeof data === "string" ? data : String(data ?? "");
}
/**
* Load (and cache) the parsed search index for a documentation version.
*/
async function loadDocIndex(
version: string,
opts: { timeoutMs?: number; signal?: AbortSignal } = {}
): Promise<DocIndex> {
const key = safeVersion(version);
const cached = indexCache.get(key);
if (cached && Date.now() - cached.fetchedAt < CACHE_TTL_MS) {
return cached.index;
}
const url = `https://docs.godotengine.org/en/${key}/searchindex.js`;
const raw = await fetchText(url, opts);
// The file is exactly: Search.setIndex({ ...json... });
const jsonBody = raw
.replace(/^Search\.setIndex\(/, "")
.replace(/\)\s*;?\s*$/, "");
let parsed: any;
try {
parsed = JSON.parse(jsonBody);
} catch (error) {
throw new Error(`Failed to parse Godot search index: ${error instanceof Error ? error.message : "Unknown error"}`);
}
const index: DocIndex = {
terms: parsed.terms ?? {},
docnames: parsed.docnames ?? [],
titles: parsed.titles ?? {},
};
indexCache.set(key, { index, fetchedAt: Date.now() });
return index;
}
function buildDocUrl(version: string, docname: string): string {
const clean = docname.replace(/\.html?$/i, "");
return `https://docs.godotengine.org/en/${safeVersion(version)}/${clean}.html`;
}
/**
* Normalize a raw search-index hit list. Different Sphinx versions store hits
* either as flat `[docIdx, ...]` or as `[docIdx, wordPos]` pairs; sometimes a
* single value is stored as a scalar. This unifies all of those into an array
* of numeric document indices.
*/
function collectDocIndexes(raw: unknown): number[] {
if (!Array.isArray(raw)) return [];
const out: number[] = [];
for (const h of raw) {
if (Array.isArray(h)) {
const d = h[0];
if (typeof d === "number") out.push(d);
} else if (typeof h === "number") {
out.push(h);
}
}
return out;
}
/**
* Search the Godot documentation and return up to `limit` candidates.
*
* Matching strategy (mirrors how a model would want it):
* - Exact term matches are always preferred.
* - Multi-word queries prefer documents containing ALL terms; if none match,
* we fall back to documents matching ANY term (ranked by frequency).
* - If there is no exact hit at all, we relax to prefix / substring matches.
*/
export async function searchDocs(
query: string,
version: string,
limit: number,
opts: { timeoutMs?: number; signal?: AbortSignal } = {}
): Promise<DocSearchResult[]> {
const index = await loadDocIndex(version, opts);
const queryTerms = (query ?? "")
.toLowerCase()
.split(/\s+/)
.map((t) => t.trim())
.filter(Boolean);
if (queryTerms.length === 0) return [];
// docIndex -> { score (terms matched), terms, positions }
const scores = new Map<
number,
{ score: number; terms: Set<string>; positions: number[] }
>();
const bump = (docIdx: number, term: string, position: number): void => {
const e = scores.get(docIdx);
if (!e) {
scores.set(docIdx, { score: 1, terms: new Set([term]), positions: [position] });
} else {
e.score += 1;
e.terms.add(term);
e.positions.push(position);
}
};
// 1) Exact matches.
let exactHits = 0;
for (const term of queryTerms) {
const hits = collectDocIndexes(index.terms[term]);
if (!hits.length) continue;
exactHits += hits.length;
hits.forEach((docIdx, position) => bump(docIdx, term, position));
}
// 2) Relaxation fallback when nothing matched exactly.
if (exactHits === 0) {
for (const [term, raw] of Object.entries(index.terms)) {
const hits = collectDocIndexes(raw);
const hit = queryTerms.some(
(q) => term === q || term.startsWith(q) || term.includes(q)
);
if (!hit || !hits.length) continue;
hits.forEach((docIdx) => bump(docIdx, term, 0));
}
}
// 3) Ranking: prefer AND matches, then by score.
const multiWord = queryTerms.length > 1;
const ranked = [...scores.entries()]
.map(([docIdx, info]) => ({ docIdx, ...info }))
.sort((a, b) => {
if (multiWord) {
// Prefer docs matching more distinct query terms.
const aMatched = queryTerms.filter((t) => a.terms.has(t)).length;
const bMatched = queryTerms.filter((t) => b.terms.has(t)).length;
if (aMatched !== bMatched) return bMatched - aMatched;
}
if (b.score !== a.score) return b.score - a.score;
// Tiebreak: fewer references is usually more focused/relevant.
return (a.positions.length ?? 0) - (b.positions.length ?? 0);
});
const top = ranked.slice(0, Math.max(1, Math.trunc(limit)));
return top.map(({ docIdx }) => {
const title =
index.titles[String(docIdx)] ||
index.docnames[docIdx] ||
"(untitled)";
const url = buildDocUrl(version, index.docnames[docIdx] ?? "");
return { title, url, snippet: "" };
});
}
/**
* Strip scripts, styles and navigation chrome from an HTML document.
*/
function stripChrome(html: string): string {
let out = html;
out = out.replace(/<script[\s\S]*?<\/script>/gi, " ");
out = out.replace(/<style[\s\S]*?<\/style>/gi, " ");
out = out.replace(/<noscript[\s\S]*?<\/noscript>/gi, " ");
out = out.replace(
/<(nav|header|footer|aside|form|button)\b[^>]*>[\s\S]*?<\/\1>/gi,
" "
);
out = out.replace(
/<div\b[^>]*\b(?:role\s*=\s*"navigation"|side-links|rst-aside|sidebar|prev-next|footer)\b[^>]*>[\s\S]*?<\/div>/gi,
" "
);
return out;
}
function collapseWhitespace(text: string): string {
return text.replace(/\s+/g, " ").trim();
}
/**
* Extract readable text from an HTML document body.
*/
function extractReadableText(html: string): string {
const chromeRemoved = stripChrome(html);
return collapseWhitespace(chromeRemoved.replace(/<[^>]+>/g, " "));
}
/**
* Fetch and reduce a documentation page into readable text.
*/
export async function fetchDocPage(
url: string,
opts: { maxChars?: number; timeoutMs?: number; signal?: AbortSignal } = {}
): Promise<string> {
const maxChars = opts.maxChars ?? MAX_OUTPUT_CHARS;
const text = await fetchText(url, opts);
const readable = extractReadableText(text);
if (readable.length > maxChars) {
return `${readable.slice(0, maxChars)}\n\n... [truncated]`;
}
return readable;
}