src / extract.ts
import { JSDOM } from "jsdom";
import { Readability } from "@mozilla/readability";
import TurndownService from "turndown";
const turndownService = new TurndownService({ headingStyle: "atx" });
export interface ExtractedPage {
title: string;
url: string;
source: string;
published_date: string | null;
content: string;
}
function findPublishedDate(document: Document): string | null {
const metaSelectors = [
'meta[property="article:published_time"]',
'meta[name="article:published_time"]',
'meta[property="og:article:published_time"]',
'meta[name="date"]',
'meta[name="publish-date"]',
'meta[itemprop="datePublished"]',
];
for (const selector of metaSelectors) {
const el = document.querySelector(selector);
const content = el?.getAttribute("content");
if (content) {
return content;
}
}
const timeEl = document.querySelector("time[datetime]");
return timeEl?.getAttribute("datetime") ?? null;
}
export async function fetchAndExtract(
url: string,
timeoutMs: number,
): Promise<ExtractedPage> {
const controller = new AbortController();
const timer = setTimeout(() => controller.abort(), timeoutMs);
let response: Response;
try {
response = await fetch(url, {
signal: controller.signal,
headers: { "User-Agent": "Mozilla/5.0 (compatible; LMStudioWebSearchPlugin/1.0)" },
});
} finally {
clearTimeout(timer);
}
if (!response.ok) {
throw new Error(`Failed to fetch page: HTTP ${response.status}`);
}
const contentType = response.headers.get("content-type") ?? "";
if (!contentType.includes("text/html") && !contentType.includes("application/xhtml")) {
throw new Error(`Unsupported content type for extraction: ${contentType || "unknown"}`);
}
const html = await response.text();
const dom = new JSDOM(html, { url });
const publishedDate = findPublishedDate(dom.window.document);
const reader = new Readability(dom.window.document);
const article = reader.parse();
if (!article || !article.content) {
throw new Error("Could not extract readable content from this page.");
}
const markdown = turndownService.turndown(article.content);
return {
title: article.title || dom.window.document.title || url,
url,
source: new URL(url).hostname,
published_date: publishedDate,
content: markdown.trim(),
};
}
src / extract.ts
import { JSDOM } from "jsdom";
import { Readability } from "@mozilla/readability";
import TurndownService from "turndown";
const turndownService = new TurndownService({ headingStyle: "atx" });
export interface ExtractedPage {
title: string;
url: string;
source: string;
published_date: string | null;
content: string;
}
function findPublishedDate(document: Document): string | null {
const metaSelectors = [
'meta[property="article:published_time"]',
'meta[name="article:published_time"]',
'meta[property="og:article:published_time"]',
'meta[name="date"]',
'meta[name="publish-date"]',
'meta[itemprop="datePublished"]',
];
for (const selector of metaSelectors) {
const el = document.querySelector(selector);
const content = el?.getAttribute("content");
if (content) {
return content;
}
}
const timeEl = document.querySelector("time[datetime]");
return timeEl?.getAttribute("datetime") ?? null;
}
export async function fetchAndExtract(
url: string,
timeoutMs: number,
): Promise<ExtractedPage> {
const controller = new AbortController();
const timer = setTimeout(() => controller.abort(), timeoutMs);
let response: Response;
try {
response = await fetch(url, {
signal: controller.signal,
headers: { "User-Agent": "Mozilla/5.0 (compatible; LMStudioWebSearchPlugin/1.0)" },
});
} finally {
clearTimeout(timer);
}
if (!response.ok) {
throw new Error(`Failed to fetch page: HTTP ${response.status}`);
}
const contentType = response.headers.get("content-type") ?? "";
if (!contentType.includes("text/html") && !contentType.includes("application/xhtml")) {
throw new Error(`Unsupported content type for extraction: ${contentType || "unknown"}`);
}
const html = await response.text();
const dom = new JSDOM(html, { url });
const publishedDate = findPublishedDate(dom.window.document);
const reader = new Readability(dom.window.document);
const article = reader.parse();
if (!article || !article.content) {
throw new Error("Could not extract readable content from this page.");
}
const markdown = turndownService.turndown(article.content);
return {
title: article.title || dom.window.document.title || url,
url,
source: new URL(url).hostname,
published_date: publishedDate,
content: markdown.trim(),
};
}