first
This commit is contained in:
+179
@@ -0,0 +1,179 @@
|
||||
import {
|
||||
CdpConnection,
|
||||
findChromeExecutable as findChromeExecutableBase,
|
||||
findExistingChromeDebugPort,
|
||||
getFreePort,
|
||||
killChrome,
|
||||
launchChrome as launchChromeBase,
|
||||
sleep,
|
||||
waitForChromeDebugPort,
|
||||
type PlatformCandidates,
|
||||
} from 'baoyu-chrome-cdp';
|
||||
|
||||
import { resolveUrlToMarkdownChromeProfileDir } from './paths.js';
|
||||
import { NETWORK_IDLE_TIMEOUT_MS } from './constants.js';
|
||||
|
||||
const CHROME_CANDIDATES_FULL: PlatformCandidates = {
|
||||
darwin: [
|
||||
'/Applications/Google Chrome.app/Contents/MacOS/Google Chrome',
|
||||
'/Applications/Google Chrome Canary.app/Contents/MacOS/Google Chrome Canary',
|
||||
'/Applications/Google Chrome Beta.app/Contents/MacOS/Google Chrome Beta',
|
||||
'/Applications/Chromium.app/Contents/MacOS/Chromium',
|
||||
'/Applications/Microsoft Edge.app/Contents/MacOS/Microsoft Edge',
|
||||
],
|
||||
win32: [
|
||||
'C:\\Program Files\\Google\\Chrome\\Application\\chrome.exe',
|
||||
'C:\\Program Files (x86)\\Google\\Chrome\\Application\\chrome.exe',
|
||||
'C:\\Program Files\\Microsoft\\Edge\\Application\\msedge.exe',
|
||||
'C:\\Program Files (x86)\\Microsoft\\Edge\\Application\\msedge.exe',
|
||||
],
|
||||
default: [
|
||||
'/usr/bin/google-chrome',
|
||||
'/usr/bin/google-chrome-stable',
|
||||
'/usr/bin/chromium',
|
||||
'/usr/bin/chromium-browser',
|
||||
'/snap/bin/chromium',
|
||||
'/usr/bin/microsoft-edge',
|
||||
],
|
||||
};
|
||||
|
||||
export { CdpConnection, getFreePort, killChrome, sleep, waitForChromeDebugPort };
|
||||
|
||||
export async function findExistingChromePort(): Promise<number | null> {
|
||||
return await findExistingChromeDebugPort({
|
||||
profileDir: resolveUrlToMarkdownChromeProfileDir(),
|
||||
});
|
||||
}
|
||||
|
||||
export function findChromeExecutable(): string | null {
|
||||
return findChromeExecutableBase({
|
||||
candidates: CHROME_CANDIDATES_FULL,
|
||||
envNames: ['URL_CHROME_PATH'],
|
||||
}) ?? null;
|
||||
}
|
||||
|
||||
export async function launchChrome(url: string, port: number, headless = false) {
|
||||
const chromePath = findChromeExecutable();
|
||||
if (!chromePath) throw new Error('Chrome executable not found. Install Chrome or set URL_CHROME_PATH env.');
|
||||
|
||||
return await launchChromeBase({
|
||||
chromePath,
|
||||
profileDir: resolveUrlToMarkdownChromeProfileDir(),
|
||||
port,
|
||||
url,
|
||||
headless,
|
||||
extraArgs: ['--disable-popup-blocking'],
|
||||
});
|
||||
}
|
||||
|
||||
export async function waitForNetworkIdle(
|
||||
cdp: CdpConnection,
|
||||
sessionId: string,
|
||||
timeoutMs: number = NETWORK_IDLE_TIMEOUT_MS,
|
||||
): Promise<void> {
|
||||
return new Promise((resolve) => {
|
||||
let timer: ReturnType<typeof setTimeout> | null = null;
|
||||
let pending = 0;
|
||||
const cleanup = () => {
|
||||
if (timer) clearTimeout(timer);
|
||||
cdp.off('Network.requestWillBeSent', onRequest);
|
||||
cdp.off('Network.loadingFinished', onFinish);
|
||||
cdp.off('Network.loadingFailed', onFinish);
|
||||
};
|
||||
const done = () => { cleanup(); resolve(); };
|
||||
const resetTimer = () => {
|
||||
if (timer) clearTimeout(timer);
|
||||
timer = setTimeout(done, timeoutMs);
|
||||
};
|
||||
const onRequest = () => { pending++; resetTimer(); };
|
||||
const onFinish = () => { pending = Math.max(0, pending - 1); if (pending <= 2) resetTimer(); };
|
||||
cdp.on('Network.requestWillBeSent', onRequest);
|
||||
cdp.on('Network.loadingFinished', onFinish);
|
||||
cdp.on('Network.loadingFailed', onFinish);
|
||||
resetTimer();
|
||||
});
|
||||
}
|
||||
|
||||
export async function waitForPageLoad(
|
||||
cdp: CdpConnection,
|
||||
sessionId: string,
|
||||
timeoutMs: number = 30_000,
|
||||
): Promise<void> {
|
||||
void sessionId;
|
||||
return new Promise((resolve) => {
|
||||
const timer = setTimeout(() => {
|
||||
cdp.off('Page.loadEventFired', handler);
|
||||
resolve();
|
||||
}, timeoutMs);
|
||||
const handler = () => {
|
||||
clearTimeout(timer);
|
||||
cdp.off('Page.loadEventFired', handler);
|
||||
resolve();
|
||||
};
|
||||
cdp.on('Page.loadEventFired', handler);
|
||||
});
|
||||
}
|
||||
|
||||
export async function createTargetAndAttach(
|
||||
cdp: CdpConnection,
|
||||
url: string,
|
||||
): Promise<{ targetId: string; sessionId: string }> {
|
||||
const { targetId } = await cdp.send<{ targetId: string }>('Target.createTarget', { url });
|
||||
const { sessionId } = await cdp.send<{ sessionId: string }>('Target.attachToTarget', { targetId, flatten: true });
|
||||
await cdp.send('Network.enable', {}, { sessionId });
|
||||
await cdp.send('Page.enable', {}, { sessionId });
|
||||
return { targetId, sessionId };
|
||||
}
|
||||
|
||||
export async function navigateAndWait(
|
||||
cdp: CdpConnection,
|
||||
sessionId: string,
|
||||
url: string,
|
||||
timeoutMs: number,
|
||||
): Promise<void> {
|
||||
const loadPromise = new Promise<void>((resolve, reject) => {
|
||||
const timer = setTimeout(() => reject(new Error('Page load timeout')), timeoutMs);
|
||||
const handler = (params: unknown) => {
|
||||
const event = params as { name?: string };
|
||||
if (event.name === 'load' || event.name === 'DOMContentLoaded') {
|
||||
clearTimeout(timer);
|
||||
cdp.off('Page.lifecycleEvent', handler);
|
||||
resolve();
|
||||
}
|
||||
};
|
||||
cdp.on('Page.lifecycleEvent', handler);
|
||||
});
|
||||
await cdp.send('Page.navigate', { url }, { sessionId });
|
||||
await loadPromise;
|
||||
}
|
||||
|
||||
export async function evaluateScript<T>(
|
||||
cdp: CdpConnection,
|
||||
sessionId: string,
|
||||
expression: string,
|
||||
timeoutMs: number = 30_000,
|
||||
): Promise<T> {
|
||||
const result = await cdp.send<{ result: { value?: T } }>(
|
||||
'Runtime.evaluate',
|
||||
{ expression, returnByValue: true, awaitPromise: true },
|
||||
{ sessionId, timeoutMs },
|
||||
);
|
||||
return result.result.value as T;
|
||||
}
|
||||
|
||||
export async function autoScroll(
|
||||
cdp: CdpConnection,
|
||||
sessionId: string,
|
||||
steps: number = 8,
|
||||
waitMs: number = 600,
|
||||
): Promise<void> {
|
||||
let lastHeight = await evaluateScript<number>(cdp, sessionId, 'document.body.scrollHeight');
|
||||
for (let i = 0; i < steps; i++) {
|
||||
await evaluateScript<void>(cdp, sessionId, 'window.scrollTo(0, document.body.scrollHeight)');
|
||||
await sleep(waitMs);
|
||||
const newHeight = await evaluateScript<number>(cdp, sessionId, 'document.body.scrollHeight');
|
||||
if (newHeight === lastHeight) break;
|
||||
lastHeight = newHeight;
|
||||
}
|
||||
await evaluateScript<void>(cdp, sessionId, 'window.scrollTo(0, 0)');
|
||||
}
|
||||
+13
@@ -0,0 +1,13 @@
|
||||
import { resolveUrlToMarkdownChromeProfileDir } from "./paths.js";
|
||||
|
||||
export const DEFAULT_USER_AGENT =
|
||||
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/130.0.0.0 Safari/537.36";
|
||||
|
||||
export const USER_DATA_DIR = resolveUrlToMarkdownChromeProfileDir();
|
||||
|
||||
export const DEFAULT_TIMEOUT_MS = 30_000;
|
||||
export const CDP_CONNECT_TIMEOUT_MS = 15_000;
|
||||
export const NETWORK_IDLE_TIMEOUT_MS = 1_500;
|
||||
export const POST_LOAD_DELAY_MS = 800;
|
||||
export const SCROLL_STEP_WAIT_MS = 600;
|
||||
export const SCROLL_MAX_STEPS = 8;
|
||||
@@ -0,0 +1,58 @@
|
||||
import { JSDOM, VirtualConsole } from "jsdom";
|
||||
import { Defuddle } from "defuddle/node";
|
||||
|
||||
import {
|
||||
type ConversionResult,
|
||||
type PageMetadata,
|
||||
isMarkdownUsable,
|
||||
normalizeMarkdown,
|
||||
pickString,
|
||||
} from "./markdown-conversion-shared.js";
|
||||
|
||||
export async function tryDefuddleConversion(
|
||||
html: string,
|
||||
url: string,
|
||||
baseMetadata: PageMetadata
|
||||
): Promise<{ ok: true; result: ConversionResult } | { ok: false; reason: string }> {
|
||||
try {
|
||||
const virtualConsole = new VirtualConsole();
|
||||
virtualConsole.on("jsdomError", (error: Error & { type?: string }) => {
|
||||
if (error.type === "css parsing" || /Could not parse CSS stylesheet/i.test(error.message)) {
|
||||
return;
|
||||
}
|
||||
console.warn(`[url-to-markdown] jsdom: ${error.message}`);
|
||||
});
|
||||
|
||||
const dom = new JSDOM(html, { url, virtualConsole });
|
||||
const result = await Defuddle(dom, url, { markdown: true });
|
||||
const markdown = normalizeMarkdown(result.content || "");
|
||||
|
||||
if (!isMarkdownUsable(markdown, html)) {
|
||||
return { ok: false, reason: "Defuddle returned empty or incomplete markdown" };
|
||||
}
|
||||
|
||||
return {
|
||||
ok: true,
|
||||
result: {
|
||||
metadata: {
|
||||
...baseMetadata,
|
||||
title: pickString(result.title, baseMetadata.title) ?? "",
|
||||
description: pickString(result.description, baseMetadata.description) ?? undefined,
|
||||
author: pickString(result.author, baseMetadata.author) ?? undefined,
|
||||
published: pickString(result.published, baseMetadata.published) ?? undefined,
|
||||
coverImage: pickString(result.image, baseMetadata.coverImage) ?? undefined,
|
||||
language: pickString(result.language, baseMetadata.language) ?? undefined,
|
||||
},
|
||||
markdown,
|
||||
rawHtml: html,
|
||||
conversionMethod: "defuddle",
|
||||
variables: result.variables,
|
||||
},
|
||||
};
|
||||
} catch (error) {
|
||||
return {
|
||||
ok: false,
|
||||
reason: error instanceof Error ? error.message : String(error),
|
||||
};
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,135 @@
|
||||
import {
|
||||
createMarkdownDocument,
|
||||
extractMetadataFromHtml,
|
||||
formatMetadataYaml,
|
||||
type ConversionResult,
|
||||
type PageMetadata,
|
||||
isYouTubeUrl,
|
||||
} from "./markdown-conversion-shared.js";
|
||||
import { tryDefuddleConversion } from "./defuddle-converter.js";
|
||||
import {
|
||||
convertWithLegacyExtractor,
|
||||
scoreMarkdownQuality,
|
||||
shouldCompareWithLegacy,
|
||||
} from "./legacy-converter.js";
|
||||
|
||||
export type { ConversionResult, PageMetadata };
|
||||
export { createMarkdownDocument, formatMetadataYaml };
|
||||
|
||||
export const absolutizeUrlsScript = String.raw`
|
||||
(function() {
|
||||
const baseUrl = document.baseURI || location.href;
|
||||
const htmlClone = document.documentElement.cloneNode(true);
|
||||
|
||||
function materializeShadowDom(sourceRoot, cloneRoot) {
|
||||
const sourceElements = Array.from(sourceRoot.querySelectorAll("*"));
|
||||
const cloneElements = Array.from(cloneRoot.querySelectorAll("*"));
|
||||
|
||||
for (let i = sourceElements.length - 1; i >= 0; i--) {
|
||||
const sourceEl = sourceElements[i];
|
||||
const cloneEl = cloneElements[i];
|
||||
const shadowRoot = sourceEl && sourceEl.shadowRoot;
|
||||
if (!shadowRoot || !cloneEl || !shadowRoot.innerHTML) continue;
|
||||
|
||||
if (cloneEl.tagName && cloneEl.tagName.includes("-")) {
|
||||
const wrapper = document.createElement("div");
|
||||
wrapper.setAttribute("data-shadow-host", cloneEl.tagName.toLowerCase());
|
||||
wrapper.innerHTML = shadowRoot.innerHTML;
|
||||
cloneEl.replaceWith(wrapper);
|
||||
} else {
|
||||
cloneEl.innerHTML = shadowRoot.innerHTML;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
function toAbsolute(url) {
|
||||
if (!url) return url;
|
||||
try { return new URL(url, baseUrl).href; } catch { return url; }
|
||||
}
|
||||
|
||||
function absAttr(root, sel, attr) {
|
||||
root.querySelectorAll(sel).forEach(el => {
|
||||
const v = el.getAttribute(attr);
|
||||
if (v) {
|
||||
const a = toAbsolute(v);
|
||||
if (a) el.setAttribute(attr, a);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
function absSrcset(root, sel) {
|
||||
root.querySelectorAll(sel).forEach(el => {
|
||||
const s = el.getAttribute("srcset");
|
||||
if (!s) return;
|
||||
el.setAttribute("srcset", s.split(",").map(p => {
|
||||
const t = p.trim();
|
||||
if (!t) return "";
|
||||
const [url, ...d] = t.split(/\s+/);
|
||||
return d.length ? toAbsolute(url) + " " + d.join(" ") : toAbsolute(url);
|
||||
}).filter(Boolean).join(", "));
|
||||
});
|
||||
}
|
||||
|
||||
materializeShadowDom(document.documentElement, htmlClone);
|
||||
|
||||
htmlClone.querySelectorAll("img[data-src], video[data-src], audio[data-src], source[data-src]").forEach(el => {
|
||||
const ds = el.getAttribute("data-src");
|
||||
if (ds && (!el.getAttribute("src") || el.getAttribute("src") === "" || el.getAttribute("src")?.startsWith("data:"))) {
|
||||
el.setAttribute("src", ds);
|
||||
}
|
||||
});
|
||||
|
||||
absAttr(htmlClone, "a[href]", "href");
|
||||
absAttr(htmlClone, "img[src], video[src], audio[src], source[src], iframe[src]", "src");
|
||||
absAttr(htmlClone, "video[poster]", "poster");
|
||||
absSrcset(htmlClone, "img[srcset], source[srcset]");
|
||||
|
||||
return { html: "<!doctype html>\n" + htmlClone.outerHTML };
|
||||
})()
|
||||
`;
|
||||
|
||||
function shouldPreferDefuddle(result: ConversionResult): boolean {
|
||||
if (isYouTubeUrl(result.metadata.url)) {
|
||||
return true;
|
||||
}
|
||||
|
||||
const transcript = result.variables?.transcript?.trim();
|
||||
if (transcript) {
|
||||
return true;
|
||||
}
|
||||
|
||||
return /^##?\s+transcript\b/im.test(result.markdown);
|
||||
}
|
||||
|
||||
export async function extractContent(html: string, url: string): Promise<ConversionResult> {
|
||||
const capturedAt = new Date().toISOString();
|
||||
const baseMetadata = extractMetadataFromHtml(html, url, capturedAt);
|
||||
|
||||
const defuddleResult = await tryDefuddleConversion(html, url, baseMetadata);
|
||||
if (defuddleResult.ok) {
|
||||
if (shouldPreferDefuddle(defuddleResult.result)) {
|
||||
return defuddleResult.result;
|
||||
}
|
||||
|
||||
if (shouldCompareWithLegacy(defuddleResult.result.markdown)) {
|
||||
const legacyResult = convertWithLegacyExtractor(html, baseMetadata);
|
||||
const legacyScore = scoreMarkdownQuality(legacyResult.markdown);
|
||||
const defuddleScore = scoreMarkdownQuality(defuddleResult.result.markdown);
|
||||
|
||||
if (legacyScore > defuddleScore + 120) {
|
||||
return {
|
||||
...legacyResult,
|
||||
fallbackReason: "Legacy extractor produced higher-quality markdown than Defuddle",
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
return defuddleResult.result;
|
||||
}
|
||||
|
||||
const fallbackResult = convertWithLegacyExtractor(html, baseMetadata);
|
||||
return {
|
||||
...fallbackResult,
|
||||
fallbackReason: defuddleResult.reason,
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,629 @@
|
||||
import { Readability } from "@mozilla/readability";
|
||||
import TurndownService from "turndown";
|
||||
import { gfm } from "turndown-plugin-gfm";
|
||||
|
||||
import {
|
||||
type AnyRecord,
|
||||
type ConversionResult,
|
||||
type PageMetadata,
|
||||
GOOD_CONTENT_LENGTH,
|
||||
MIN_CONTENT_LENGTH,
|
||||
extractPublishedTime,
|
||||
extractTextFromHtml,
|
||||
extractTitle,
|
||||
normalizeMarkdown,
|
||||
parseDocument,
|
||||
pickString,
|
||||
sanitizeHtml,
|
||||
} from "./markdown-conversion-shared.js";
|
||||
|
||||
interface ExtractionCandidate {
|
||||
title: string | null;
|
||||
byline: string | null;
|
||||
excerpt: string | null;
|
||||
published: string | null;
|
||||
html: string | null;
|
||||
textContent: string;
|
||||
method: string;
|
||||
}
|
||||
|
||||
const CONTENT_SELECTORS = [
|
||||
"article",
|
||||
"main article",
|
||||
"[role='main'] article",
|
||||
"[itemprop='articleBody']",
|
||||
".article-content",
|
||||
".article-body",
|
||||
".post-content",
|
||||
".entry-content",
|
||||
".story-body",
|
||||
"main",
|
||||
"[role='main']",
|
||||
"#content",
|
||||
".content",
|
||||
];
|
||||
|
||||
const REMOVE_SELECTORS = [
|
||||
"script",
|
||||
"style",
|
||||
"noscript",
|
||||
"template",
|
||||
"iframe",
|
||||
"svg",
|
||||
"path",
|
||||
"nav",
|
||||
"aside",
|
||||
"footer",
|
||||
"header",
|
||||
"form",
|
||||
".advertisement",
|
||||
".ads",
|
||||
".social-share",
|
||||
".related-articles",
|
||||
".comments",
|
||||
".newsletter",
|
||||
".cookie-banner",
|
||||
".cookie-consent",
|
||||
"[role='navigation']",
|
||||
"[aria-label*='cookie' i]",
|
||||
];
|
||||
|
||||
const NEXT_DATA_CONTENT_PATHS = [
|
||||
"props.pageProps.content.body",
|
||||
"props.pageProps.article.body",
|
||||
"props.pageProps.article.content",
|
||||
"props.pageProps.post.body",
|
||||
"props.pageProps.post.content",
|
||||
"props.pageProps.data.body",
|
||||
"props.pageProps.story.body.content",
|
||||
];
|
||||
|
||||
const LOW_QUALITY_MARKERS = [
|
||||
/Join The Conversation/i,
|
||||
/One Community\. Many Voices/i,
|
||||
/Read our community guidelines/i,
|
||||
/Create a free account to share your thoughts/i,
|
||||
/Become a Forbes Member/i,
|
||||
/Subscribe to trusted journalism/i,
|
||||
/\bComments\b/i,
|
||||
];
|
||||
|
||||
function generateExcerpt(excerpt: string | null, textContent: string | null): string | null {
|
||||
if (excerpt) return excerpt;
|
||||
if (!textContent) return null;
|
||||
const trimmed = textContent.trim();
|
||||
if (!trimmed) return null;
|
||||
return trimmed.length > 200 ? `${trimmed.slice(0, 200)}...` : trimmed;
|
||||
}
|
||||
|
||||
function parseJsonLdItem(item: AnyRecord): ExtractionCandidate | null {
|
||||
const type = Array.isArray(item["@type"]) ? item["@type"][0] : item["@type"];
|
||||
if (typeof type !== "string" || !["Article", "NewsArticle", "BlogPosting", "WebPage", "ReportageNewsArticle"].includes(type)) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const rawContent =
|
||||
(typeof item.articleBody === "string" && item.articleBody) ||
|
||||
(typeof item.text === "string" && item.text) ||
|
||||
(typeof item.description === "string" && item.description) ||
|
||||
null;
|
||||
|
||||
if (!rawContent) return null;
|
||||
|
||||
const content = rawContent.trim();
|
||||
const htmlLike = /<\/?[a-z][\s\S]*>/i.test(content);
|
||||
const textContent = htmlLike ? extractTextFromHtml(content) : content;
|
||||
|
||||
if (textContent.length < MIN_CONTENT_LENGTH) return null;
|
||||
|
||||
return {
|
||||
title: pickString(item.headline, item.name),
|
||||
byline: extractAuthorFromJsonLd(item.author),
|
||||
excerpt: pickString(item.description),
|
||||
published: pickString(item.datePublished, item.dateCreated),
|
||||
html: htmlLike ? content : null,
|
||||
textContent,
|
||||
method: "json-ld",
|
||||
};
|
||||
}
|
||||
|
||||
function extractAuthorFromJsonLd(authorData: unknown): string | null {
|
||||
if (typeof authorData === "string") return authorData;
|
||||
if (!authorData || typeof authorData !== "object") return null;
|
||||
|
||||
if (Array.isArray(authorData)) {
|
||||
const names = authorData
|
||||
.map((author) => extractAuthorFromJsonLd(author))
|
||||
.filter((name): name is string => Boolean(name));
|
||||
return names.length > 0 ? names.join(", ") : null;
|
||||
}
|
||||
|
||||
const author = authorData as AnyRecord;
|
||||
return typeof author.name === "string" ? author.name : null;
|
||||
}
|
||||
|
||||
function flattenJsonLdItems(data: unknown): AnyRecord[] {
|
||||
if (!data || typeof data !== "object") return [];
|
||||
if (Array.isArray(data)) return data.flatMap(flattenJsonLdItems);
|
||||
|
||||
const item = data as AnyRecord;
|
||||
if (Array.isArray(item["@graph"])) {
|
||||
return (item["@graph"] as unknown[]).flatMap(flattenJsonLdItems);
|
||||
}
|
||||
|
||||
return [item];
|
||||
}
|
||||
|
||||
function tryJsonLdExtraction(document: Document): ExtractionCandidate | null {
|
||||
const scripts = document.querySelectorAll("script[type='application/ld+json']");
|
||||
|
||||
for (const script of scripts) {
|
||||
try {
|
||||
const data = JSON.parse(script.textContent ?? "");
|
||||
for (const item of flattenJsonLdItems(data)) {
|
||||
const extracted = parseJsonLdItem(item);
|
||||
if (extracted) return extracted;
|
||||
}
|
||||
} catch {
|
||||
// Ignore malformed blocks.
|
||||
}
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
function getByPath(value: unknown, path: string): unknown {
|
||||
let current = value;
|
||||
for (const part of path.split(".")) {
|
||||
if (!current || typeof current !== "object") return undefined;
|
||||
current = (current as AnyRecord)[part];
|
||||
}
|
||||
return current;
|
||||
}
|
||||
|
||||
function isContentBlockArray(value: unknown): value is AnyRecord[] {
|
||||
if (!Array.isArray(value) || value.length === 0) return false;
|
||||
return value.slice(0, 5).some((item) => {
|
||||
if (!item || typeof item !== "object") return false;
|
||||
const obj = item as AnyRecord;
|
||||
return "type" in obj || "text" in obj || "textHtml" in obj || "content" in obj;
|
||||
});
|
||||
}
|
||||
|
||||
function extractTextFromContentBlocks(blocks: AnyRecord[]): string {
|
||||
const parts: string[] = [];
|
||||
|
||||
function pushParagraph(text: string): void {
|
||||
const trimmed = text.trim();
|
||||
if (!trimmed) return;
|
||||
parts.push(trimmed, "\n\n");
|
||||
}
|
||||
|
||||
function walk(node: unknown): void {
|
||||
if (!node || typeof node !== "object") return;
|
||||
const block = node as AnyRecord;
|
||||
|
||||
if (typeof block.text === "string") {
|
||||
pushParagraph(block.text);
|
||||
return;
|
||||
}
|
||||
|
||||
if (typeof block.textHtml === "string") {
|
||||
pushParagraph(extractTextFromHtml(block.textHtml));
|
||||
return;
|
||||
}
|
||||
|
||||
if (Array.isArray(block.items)) {
|
||||
for (const item of block.items) {
|
||||
if (item && typeof item === "object") {
|
||||
const text = pickString((item as AnyRecord).text);
|
||||
if (text) parts.push(`- ${text}\n`);
|
||||
}
|
||||
}
|
||||
parts.push("\n");
|
||||
}
|
||||
|
||||
if (Array.isArray(block.components)) {
|
||||
for (const component of block.components) {
|
||||
walk(component);
|
||||
}
|
||||
}
|
||||
|
||||
if (Array.isArray(block.content)) {
|
||||
for (const child of block.content) {
|
||||
walk(child);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (const block of blocks) {
|
||||
walk(block);
|
||||
}
|
||||
|
||||
return parts.join("").replace(/\n{3,}/g, "\n\n").trim();
|
||||
}
|
||||
|
||||
function tryStringBodyExtraction(
|
||||
content: string,
|
||||
meta: AnyRecord,
|
||||
document: Document,
|
||||
method: string
|
||||
): ExtractionCandidate | null {
|
||||
if (!content || content.length < MIN_CONTENT_LENGTH) return null;
|
||||
|
||||
const isHtml = /<\/?[a-z][\s\S]*>/i.test(content);
|
||||
const html = isHtml ? sanitizeHtml(content) : null;
|
||||
const textContent = isHtml ? extractTextFromHtml(html) : content.trim();
|
||||
|
||||
if (textContent.length < MIN_CONTENT_LENGTH) return null;
|
||||
|
||||
return {
|
||||
title: pickString(meta.headline, meta.title, extractTitle(document)),
|
||||
byline: pickString(meta.byline, meta.author),
|
||||
excerpt: pickString(meta.description, meta.excerpt, generateExcerpt(null, textContent)),
|
||||
published: pickString(meta.datePublished, meta.publishedAt, extractPublishedTime(document)),
|
||||
html,
|
||||
textContent,
|
||||
method,
|
||||
};
|
||||
}
|
||||
|
||||
function tryNextDataExtraction(document: Document): ExtractionCandidate | null {
|
||||
try {
|
||||
const script = document.querySelector("script#__NEXT_DATA__");
|
||||
if (!script?.textContent) return null;
|
||||
|
||||
const data = JSON.parse(script.textContent) as AnyRecord;
|
||||
const pageProps = (getByPath(data, "props.pageProps") ?? {}) as AnyRecord;
|
||||
|
||||
for (const path of NEXT_DATA_CONTENT_PATHS) {
|
||||
const value = getByPath(data, path);
|
||||
|
||||
if (typeof value === "string") {
|
||||
const parentPath = path.split(".").slice(0, -1).join(".");
|
||||
const parent = (getByPath(data, parentPath) ?? {}) as AnyRecord;
|
||||
const meta = {
|
||||
...pageProps,
|
||||
...parent,
|
||||
title: parent.title ?? (pageProps.title as string | undefined),
|
||||
};
|
||||
|
||||
const candidate = tryStringBodyExtraction(value, meta, document, "next-data");
|
||||
if (candidate) return candidate;
|
||||
}
|
||||
|
||||
if (isContentBlockArray(value)) {
|
||||
const textContent = extractTextFromContentBlocks(value);
|
||||
if (textContent.length < MIN_CONTENT_LENGTH) continue;
|
||||
|
||||
return {
|
||||
title: pickString(
|
||||
getByPath(data, "props.pageProps.content.headline"),
|
||||
getByPath(data, "props.pageProps.article.headline"),
|
||||
getByPath(data, "props.pageProps.article.title"),
|
||||
getByPath(data, "props.pageProps.post.title"),
|
||||
pageProps.title,
|
||||
extractTitle(document)
|
||||
),
|
||||
byline: pickString(
|
||||
getByPath(data, "props.pageProps.author.name"),
|
||||
getByPath(data, "props.pageProps.article.author.name")
|
||||
),
|
||||
excerpt: pickString(
|
||||
getByPath(data, "props.pageProps.content.description"),
|
||||
getByPath(data, "props.pageProps.article.description"),
|
||||
pageProps.description,
|
||||
generateExcerpt(null, textContent)
|
||||
),
|
||||
published: pickString(
|
||||
getByPath(data, "props.pageProps.content.datePublished"),
|
||||
getByPath(data, "props.pageProps.article.datePublished"),
|
||||
getByPath(data, "props.pageProps.publishedAt"),
|
||||
extractPublishedTime(document)
|
||||
),
|
||||
html: null,
|
||||
textContent,
|
||||
method: "next-data",
|
||||
};
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
function buildReadabilityCandidate(
|
||||
article: ReturnType<Readability["parse"]>,
|
||||
document: Document,
|
||||
method: string
|
||||
): ExtractionCandidate | null {
|
||||
const textContent = article?.textContent?.trim() ?? "";
|
||||
if (textContent.length < MIN_CONTENT_LENGTH) return null;
|
||||
|
||||
return {
|
||||
title: pickString(article?.title, extractTitle(document)),
|
||||
byline: pickString((article as { byline?: string } | null)?.byline),
|
||||
excerpt: pickString(article?.excerpt, generateExcerpt(null, textContent)),
|
||||
published: pickString((article as { publishedTime?: string } | null)?.publishedTime, extractPublishedTime(document)),
|
||||
html: article?.content ? sanitizeHtml(article.content) : null,
|
||||
textContent,
|
||||
method,
|
||||
};
|
||||
}
|
||||
|
||||
function tryReadability(document: Document): ExtractionCandidate | null {
|
||||
try {
|
||||
const strictClone = document.cloneNode(true) as Document;
|
||||
const strictResult = buildReadabilityCandidate(
|
||||
new Readability(strictClone).parse(),
|
||||
document,
|
||||
"readability"
|
||||
);
|
||||
if (strictResult) return strictResult;
|
||||
|
||||
const relaxedClone = document.cloneNode(true) as Document;
|
||||
return buildReadabilityCandidate(
|
||||
new Readability(relaxedClone, { charThreshold: 120 }).parse(),
|
||||
document,
|
||||
"readability-relaxed"
|
||||
);
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
function trySelectorExtraction(document: Document): ExtractionCandidate | null {
|
||||
for (const selector of CONTENT_SELECTORS) {
|
||||
const element = document.querySelector(selector);
|
||||
if (!element) continue;
|
||||
|
||||
const clone = element.cloneNode(true) as Element;
|
||||
for (const removeSelector of REMOVE_SELECTORS) {
|
||||
for (const node of clone.querySelectorAll(removeSelector)) {
|
||||
node.remove();
|
||||
}
|
||||
}
|
||||
|
||||
const html = sanitizeHtml(clone.innerHTML);
|
||||
const textContent = extractTextFromHtml(html);
|
||||
if (textContent.length < MIN_CONTENT_LENGTH) continue;
|
||||
|
||||
return {
|
||||
title: extractTitle(document),
|
||||
byline: null,
|
||||
excerpt: generateExcerpt(null, textContent),
|
||||
published: extractPublishedTime(document),
|
||||
html,
|
||||
textContent,
|
||||
method: `selector:${selector}`,
|
||||
};
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
function tryBodyExtraction(document: Document): ExtractionCandidate | null {
|
||||
const body = document.body;
|
||||
if (!body) return null;
|
||||
|
||||
const clone = body.cloneNode(true) as Element;
|
||||
for (const removeSelector of REMOVE_SELECTORS) {
|
||||
for (const node of clone.querySelectorAll(removeSelector)) {
|
||||
node.remove();
|
||||
}
|
||||
}
|
||||
|
||||
const html = sanitizeHtml(clone.innerHTML);
|
||||
const textContent = extractTextFromHtml(html);
|
||||
if (!textContent) return null;
|
||||
|
||||
return {
|
||||
title: extractTitle(document),
|
||||
byline: null,
|
||||
excerpt: generateExcerpt(null, textContent),
|
||||
published: extractPublishedTime(document),
|
||||
html,
|
||||
textContent,
|
||||
method: "body-fallback",
|
||||
};
|
||||
}
|
||||
|
||||
function pickBestCandidate(candidates: ExtractionCandidate[]): ExtractionCandidate | null {
|
||||
if (candidates.length === 0) return null;
|
||||
|
||||
const methodOrder = [
|
||||
"readability",
|
||||
"readability-relaxed",
|
||||
"next-data",
|
||||
"json-ld",
|
||||
"selector:",
|
||||
"body-fallback",
|
||||
];
|
||||
|
||||
function methodRank(method: string): number {
|
||||
const idx = methodOrder.findIndex((entry) =>
|
||||
entry.endsWith(":") ? method.startsWith(entry) : method === entry
|
||||
);
|
||||
return idx === -1 ? methodOrder.length : idx;
|
||||
}
|
||||
|
||||
const ranked = [...candidates].sort((a, b) => {
|
||||
const rankA = methodRank(a.method);
|
||||
const rankB = methodRank(b.method);
|
||||
if (rankA !== rankB) return rankA - rankB;
|
||||
return (b.textContent.length ?? 0) - (a.textContent.length ?? 0);
|
||||
});
|
||||
|
||||
for (const candidate of ranked) {
|
||||
if (candidate.textContent.length >= GOOD_CONTENT_LENGTH) {
|
||||
return candidate;
|
||||
}
|
||||
}
|
||||
|
||||
for (const candidate of ranked) {
|
||||
if (candidate.textContent.length >= MIN_CONTENT_LENGTH) {
|
||||
return candidate;
|
||||
}
|
||||
}
|
||||
|
||||
return ranked[0];
|
||||
}
|
||||
|
||||
function extractFromHtml(html: string): ExtractionCandidate | null {
|
||||
const document = parseDocument(html);
|
||||
|
||||
const readabilityCandidate = tryReadability(document);
|
||||
const nextDataCandidate = tryNextDataExtraction(document);
|
||||
const jsonLdCandidate = tryJsonLdExtraction(document);
|
||||
const selectorCandidate = trySelectorExtraction(document);
|
||||
const bodyCandidate = tryBodyExtraction(document);
|
||||
|
||||
const candidates = [
|
||||
readabilityCandidate,
|
||||
nextDataCandidate,
|
||||
jsonLdCandidate,
|
||||
selectorCandidate,
|
||||
bodyCandidate,
|
||||
].filter((candidate): candidate is ExtractionCandidate => Boolean(candidate));
|
||||
|
||||
const winner = pickBestCandidate(candidates);
|
||||
if (!winner) return null;
|
||||
|
||||
return {
|
||||
...winner,
|
||||
title: winner.title ?? extractTitle(document),
|
||||
published: winner.published ?? extractPublishedTime(document),
|
||||
excerpt: winner.excerpt ?? generateExcerpt(null, winner.textContent),
|
||||
};
|
||||
}
|
||||
|
||||
const turndown = new TurndownService({
|
||||
headingStyle: "atx",
|
||||
hr: "---",
|
||||
bulletListMarker: "-",
|
||||
codeBlockStyle: "fenced",
|
||||
emDelimiter: "*",
|
||||
strongDelimiter: "**",
|
||||
linkStyle: "inlined",
|
||||
});
|
||||
|
||||
turndown.use(gfm);
|
||||
turndown.remove(["script", "style", "iframe", "noscript", "template", "svg", "path"]);
|
||||
|
||||
turndown.addRule("collapseFigure", {
|
||||
filter: "figure",
|
||||
replacement(content) {
|
||||
return `\n\n${content.trim()}\n\n`;
|
||||
},
|
||||
});
|
||||
|
||||
turndown.addRule("dropInvisibleAnchors", {
|
||||
filter(node) {
|
||||
return node.nodeName === "A" && !(node as Element).textContent?.trim();
|
||||
},
|
||||
replacement() {
|
||||
return "";
|
||||
},
|
||||
});
|
||||
|
||||
function convertHtmlToMarkdown(html: string): string {
|
||||
if (!html || !html.trim()) return "";
|
||||
|
||||
try {
|
||||
const sanitized = sanitizeHtml(html);
|
||||
return turndown.turndown(sanitized);
|
||||
} catch {
|
||||
return "";
|
||||
}
|
||||
}
|
||||
|
||||
function fallbackPlainText(html: string): string {
|
||||
const document = parseDocument(html);
|
||||
for (const selector of ["script", "style", "noscript", "template", "iframe", "svg", "path"]) {
|
||||
for (const el of document.querySelectorAll(selector)) {
|
||||
el.remove();
|
||||
}
|
||||
}
|
||||
const text = document.body?.textContent ?? document.documentElement?.textContent ?? "";
|
||||
return normalizeMarkdown(text.replace(/\s+/g, " "));
|
||||
}
|
||||
|
||||
function countBylines(markdown: string): number {
|
||||
return (markdown.match(/(^|\n)By\s+/g) || []).length;
|
||||
}
|
||||
|
||||
function countUsefulParagraphs(markdown: string): number {
|
||||
const paragraphs = normalizeMarkdown(markdown).split(/\n{2,}/);
|
||||
let count = 0;
|
||||
|
||||
for (const paragraph of paragraphs) {
|
||||
const trimmed = paragraph.trim();
|
||||
if (!trimmed) continue;
|
||||
if (/^!?\[[^\]]*\]\([^)]+\)$/.test(trimmed)) continue;
|
||||
if (/^#{1,6}\s+/.test(trimmed)) continue;
|
||||
if ((trimmed.match(/\b[\p{L}\p{N}']+\b/gu) || []).length < 8) continue;
|
||||
count++;
|
||||
}
|
||||
|
||||
return count;
|
||||
}
|
||||
|
||||
function countMarkerHits(markdown: string, markers: RegExp[]): number {
|
||||
let hits = 0;
|
||||
for (const marker of markers) {
|
||||
if (marker.test(markdown)) hits++;
|
||||
}
|
||||
return hits;
|
||||
}
|
||||
|
||||
export function scoreMarkdownQuality(markdown: string): number {
|
||||
const normalized = normalizeMarkdown(markdown);
|
||||
const wordCount = (normalized.match(/\b[\p{L}\p{N}']+\b/gu) || []).length;
|
||||
const usefulParagraphs = countUsefulParagraphs(normalized);
|
||||
const headingCount = (normalized.match(/^#{1,6}\s+/gm) || []).length;
|
||||
const markerHits = countMarkerHits(normalized, LOW_QUALITY_MARKERS);
|
||||
const bylineCount = countBylines(normalized);
|
||||
const staffCount = (normalized.match(/\bForbes Staff\b/gi) || []).length;
|
||||
|
||||
return (
|
||||
Math.min(wordCount, 4000) +
|
||||
usefulParagraphs * 40 +
|
||||
headingCount * 10 -
|
||||
markerHits * 180 -
|
||||
Math.max(0, bylineCount - 1) * 120 -
|
||||
Math.max(0, staffCount - 1) * 80
|
||||
);
|
||||
}
|
||||
|
||||
export function shouldCompareWithLegacy(markdown: string): boolean {
|
||||
const normalized = normalizeMarkdown(markdown);
|
||||
return (
|
||||
countMarkerHits(normalized, LOW_QUALITY_MARKERS) > 0 ||
|
||||
countBylines(normalized) > 1 ||
|
||||
countUsefulParagraphs(normalized) < 6
|
||||
);
|
||||
}
|
||||
|
||||
export function convertWithLegacyExtractor(html: string, baseMetadata: PageMetadata): ConversionResult {
|
||||
const extracted = extractFromHtml(html);
|
||||
|
||||
let markdown = extracted?.html ? convertHtmlToMarkdown(extracted.html) : "";
|
||||
if (!markdown.trim()) {
|
||||
markdown = extracted?.textContent?.trim() || fallbackPlainText(html);
|
||||
}
|
||||
|
||||
return {
|
||||
metadata: {
|
||||
...baseMetadata,
|
||||
title: pickString(extracted?.title, baseMetadata.title) ?? "",
|
||||
description: pickString(extracted?.excerpt, baseMetadata.description) ?? undefined,
|
||||
author: pickString(extracted?.byline, baseMetadata.author) ?? undefined,
|
||||
published: pickString(extracted?.published, baseMetadata.published) ?? undefined,
|
||||
},
|
||||
markdown: normalizeMarkdown(markdown),
|
||||
rawHtml: html,
|
||||
conversionMethod: extracted ? `legacy:${extracted.method}` : "legacy:plain-text",
|
||||
};
|
||||
}
|
||||
+314
@@ -0,0 +1,314 @@
|
||||
import { createInterface } from "node:readline";
|
||||
import { writeFile, mkdir, access } from "node:fs/promises";
|
||||
import path from "node:path";
|
||||
import process from "node:process";
|
||||
|
||||
import { CdpConnection, getFreePort, findExistingChromePort, launchChrome, waitForChromeDebugPort, waitForNetworkIdle, waitForPageLoad, autoScroll, evaluateScript, killChrome } from "./cdp.js";
|
||||
import { absolutizeUrlsScript, extractContent, createMarkdownDocument, type ConversionResult } from "./html-to-markdown.js";
|
||||
import { localizeMarkdownMedia, countRemoteMedia } from "./media-localizer.js";
|
||||
import { resolveUrlToMarkdownDataDir } from "./paths.js";
|
||||
import { DEFAULT_TIMEOUT_MS, CDP_CONNECT_TIMEOUT_MS, NETWORK_IDLE_TIMEOUT_MS, POST_LOAD_DELAY_MS, SCROLL_STEP_WAIT_MS, SCROLL_MAX_STEPS } from "./constants.js";
|
||||
|
||||
function sleep(ms: number): Promise<void> {
|
||||
return new Promise((resolve) => setTimeout(resolve, ms));
|
||||
}
|
||||
|
||||
async function fileExists(filePath: string): Promise<boolean> {
|
||||
try {
|
||||
await access(filePath);
|
||||
return true;
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
interface Args {
|
||||
url: string;
|
||||
output?: string;
|
||||
outputDir?: string;
|
||||
wait: boolean;
|
||||
timeout: number;
|
||||
downloadMedia: boolean;
|
||||
}
|
||||
|
||||
function parseArgs(argv: string[]): Args {
|
||||
const args: Args = { url: "", wait: false, timeout: DEFAULT_TIMEOUT_MS, downloadMedia: false };
|
||||
for (let i = 2; i < argv.length; i++) {
|
||||
const arg = argv[i];
|
||||
if (arg === "--wait" || arg === "-w") {
|
||||
args.wait = true;
|
||||
} else if (arg === "-o" || arg === "--output") {
|
||||
args.output = argv[++i];
|
||||
} else if (arg === "--timeout" || arg === "-t") {
|
||||
args.timeout = parseInt(argv[++i], 10) || DEFAULT_TIMEOUT_MS;
|
||||
} else if (arg === "--output-dir") {
|
||||
args.outputDir = argv[++i];
|
||||
} else if (arg === "--download-media") {
|
||||
args.downloadMedia = true;
|
||||
} else if (!arg.startsWith("-") && !args.url) {
|
||||
args.url = arg;
|
||||
}
|
||||
}
|
||||
return args;
|
||||
}
|
||||
|
||||
function generateSlug(title: string, url: string): string {
|
||||
const text = title || new URL(url).pathname.replace(/\//g, "-");
|
||||
return text
|
||||
.toLowerCase()
|
||||
.replace(/[^\w\s-]/g, "")
|
||||
.replace(/\s+/g, "-")
|
||||
.replace(/-+/g, "-")
|
||||
.replace(/^-|-$/g, "")
|
||||
.slice(0, 50) || "page";
|
||||
}
|
||||
|
||||
function formatTimestamp(): string {
|
||||
const now = new Date();
|
||||
const pad = (n: number) => n.toString().padStart(2, "0");
|
||||
return `${now.getFullYear()}${pad(now.getMonth() + 1)}${pad(now.getDate())}-${pad(now.getHours())}${pad(now.getMinutes())}${pad(now.getSeconds())}`;
|
||||
}
|
||||
|
||||
function deriveHtmlSnapshotPath(markdownPath: string): string {
|
||||
const parsed = path.parse(markdownPath);
|
||||
const basename = parsed.ext ? parsed.name : parsed.base;
|
||||
return path.join(parsed.dir, `${basename}-captured.html`);
|
||||
}
|
||||
|
||||
function extractTitleFromMarkdownDocument(document: string): string {
|
||||
const normalized = document.replace(/\r\n/g, "\n");
|
||||
const frontmatterMatch = normalized.match(/^---\n([\s\S]*?)\n---\n?/);
|
||||
if (frontmatterMatch) {
|
||||
const titleLine = frontmatterMatch[1]
|
||||
.split("\n")
|
||||
.find((line) => /^title:\s*/i.test(line));
|
||||
|
||||
if (titleLine) {
|
||||
const rawValue = titleLine.replace(/^title:\s*/i, "").trim();
|
||||
const unquoted = rawValue
|
||||
.replace(/^"(.*)"$/, "$1")
|
||||
.replace(/^'(.*)'$/, "$1")
|
||||
.replace(/\\"/g, '"');
|
||||
if (unquoted) return unquoted;
|
||||
}
|
||||
}
|
||||
|
||||
const headingMatch = normalized.match(/^#\s+(.+)$/m);
|
||||
return headingMatch?.[1]?.trim() ?? "";
|
||||
}
|
||||
|
||||
function buildDefuddleApiUrl(targetUrl: string): string {
|
||||
return `https://defuddle.md/${encodeURIComponent(targetUrl)}`;
|
||||
}
|
||||
|
||||
async function fetchDefuddleApiMarkdown(targetUrl: string): Promise<{ markdown: string; title: string }> {
|
||||
const apiUrl = buildDefuddleApiUrl(targetUrl);
|
||||
const response = await fetch(apiUrl, {
|
||||
headers: {
|
||||
accept: "text/markdown,text/plain;q=0.9,*/*;q=0.1",
|
||||
},
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
throw new Error(`defuddle.md returned ${response.status} ${response.statusText}`);
|
||||
}
|
||||
|
||||
const markdown = (await response.text()).replace(/\r\n/g, "\n").trim();
|
||||
if (!markdown) {
|
||||
throw new Error("defuddle.md returned empty markdown");
|
||||
}
|
||||
|
||||
return {
|
||||
markdown,
|
||||
title: extractTitleFromMarkdownDocument(markdown),
|
||||
};
|
||||
}
|
||||
|
||||
async function generateOutputPath(url: string, title: string, outputDir?: string): Promise<string> {
|
||||
const domain = new URL(url).hostname.replace(/^www\./, "");
|
||||
const slug = generateSlug(title, url);
|
||||
const dataDir = outputDir ? path.resolve(outputDir) : resolveUrlToMarkdownDataDir();
|
||||
const basePath = path.join(dataDir, domain, `${slug}.md`);
|
||||
|
||||
if (!(await fileExists(basePath))) {
|
||||
return basePath;
|
||||
}
|
||||
|
||||
const timestampSlug = `${slug}-${formatTimestamp()}`;
|
||||
return path.join(dataDir, domain, `${timestampSlug}.md`);
|
||||
}
|
||||
|
||||
async function waitForUserSignal(): Promise<void> {
|
||||
console.log("Page opened. Press Enter when ready to capture...");
|
||||
const rl = createInterface({ input: process.stdin, output: process.stdout });
|
||||
await new Promise<void>((resolve) => {
|
||||
rl.once("line", () => { rl.close(); resolve(); });
|
||||
});
|
||||
}
|
||||
|
||||
async function captureUrl(args: Args): Promise<ConversionResult> {
|
||||
const existingPort = await findExistingChromePort();
|
||||
const reusing = existingPort !== null;
|
||||
const port = existingPort ?? await getFreePort();
|
||||
const chrome = reusing ? null : await launchChrome(args.url, port, false);
|
||||
|
||||
if (reusing) console.log(`Reusing existing Chrome on port ${port}`);
|
||||
|
||||
let cdp: CdpConnection | null = null;
|
||||
let targetId: string | null = null;
|
||||
try {
|
||||
const wsUrl = await waitForChromeDebugPort(port, 30_000);
|
||||
cdp = await CdpConnection.connect(wsUrl, CDP_CONNECT_TIMEOUT_MS);
|
||||
|
||||
let sessionId: string;
|
||||
if (reusing) {
|
||||
const created = await cdp.send<{ targetId: string }>("Target.createTarget", { url: args.url });
|
||||
targetId = created.targetId;
|
||||
const attached = await cdp.send<{ sessionId: string }>("Target.attachToTarget", { targetId, flatten: true });
|
||||
sessionId = attached.sessionId;
|
||||
await cdp.send("Network.enable", {}, { sessionId });
|
||||
await cdp.send("Page.enable", {}, { sessionId });
|
||||
} else {
|
||||
const targets = await cdp.send<{ targetInfos: Array<{ targetId: string; type: string; url: string }> }>("Target.getTargets");
|
||||
const pageTarget = targets.targetInfos.find(t => t.type === "page" && t.url.startsWith("http"));
|
||||
if (!pageTarget) throw new Error("No page target found");
|
||||
targetId = pageTarget.targetId;
|
||||
const attached = await cdp.send<{ sessionId: string }>("Target.attachToTarget", { targetId, flatten: true });
|
||||
sessionId = attached.sessionId;
|
||||
await cdp.send("Network.enable", {}, { sessionId });
|
||||
await cdp.send("Page.enable", {}, { sessionId });
|
||||
}
|
||||
|
||||
if (args.wait) {
|
||||
await waitForUserSignal();
|
||||
} else {
|
||||
console.log("Waiting for page to load...");
|
||||
await Promise.race([
|
||||
waitForPageLoad(cdp, sessionId, 15_000),
|
||||
sleep(8_000)
|
||||
]);
|
||||
await waitForNetworkIdle(cdp, sessionId, NETWORK_IDLE_TIMEOUT_MS);
|
||||
await sleep(POST_LOAD_DELAY_MS);
|
||||
console.log("Scrolling to trigger lazy load...");
|
||||
await autoScroll(cdp, sessionId, SCROLL_MAX_STEPS, SCROLL_STEP_WAIT_MS);
|
||||
await sleep(POST_LOAD_DELAY_MS);
|
||||
}
|
||||
|
||||
console.log("Capturing page content...");
|
||||
const { html } = await evaluateScript<{ html: string }>(
|
||||
cdp, sessionId, absolutizeUrlsScript, args.timeout
|
||||
);
|
||||
|
||||
return await extractContent(html, args.url);
|
||||
} finally {
|
||||
if (reusing) {
|
||||
if (cdp && targetId) {
|
||||
try { await cdp.send("Target.closeTarget", { targetId }, { timeoutMs: 5_000 }); } catch {}
|
||||
}
|
||||
if (cdp) cdp.close();
|
||||
} else {
|
||||
if (cdp) {
|
||||
try { await cdp.send("Browser.close", {}, { timeoutMs: 5_000 }); } catch {}
|
||||
cdp.close();
|
||||
}
|
||||
if (chrome) killChrome(chrome);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async function main(): Promise<void> {
|
||||
const args = parseArgs(process.argv);
|
||||
if (!args.url) {
|
||||
console.error("Usage: bun main.ts <url> [-o output.md] [--output-dir dir] [--wait] [--timeout ms] [--download-media]");
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
try {
|
||||
new URL(args.url);
|
||||
} catch {
|
||||
console.error(`Invalid URL: ${args.url}`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
if (args.output) {
|
||||
const stat = await import("node:fs").then(fs => fs.statSync(args.output!, { throwIfNoEntry: false }));
|
||||
if (stat?.isDirectory()) {
|
||||
console.error(`Error: -o path is a directory, not a file: ${args.output}`);
|
||||
process.exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
console.log(`Fetching: ${args.url}`);
|
||||
console.log(`Mode: ${args.wait ? "wait" : "auto"}`);
|
||||
|
||||
let outputPath: string;
|
||||
let htmlSnapshotPath: string | null = null;
|
||||
let document: string;
|
||||
let conversionMethod: string;
|
||||
let fallbackReason: string | undefined;
|
||||
|
||||
try {
|
||||
const result = await captureUrl(args);
|
||||
outputPath = args.output || await generateOutputPath(args.url, result.metadata.title, args.outputDir);
|
||||
const outputDir = path.dirname(outputPath);
|
||||
htmlSnapshotPath = deriveHtmlSnapshotPath(outputPath);
|
||||
await mkdir(outputDir, { recursive: true });
|
||||
await writeFile(htmlSnapshotPath, result.rawHtml, "utf-8");
|
||||
|
||||
document = createMarkdownDocument(result);
|
||||
conversionMethod = result.conversionMethod;
|
||||
fallbackReason = result.fallbackReason;
|
||||
} catch (error) {
|
||||
const primaryError = error instanceof Error ? error.message : String(error);
|
||||
console.warn(`Primary capture failed: ${primaryError}`);
|
||||
console.warn("Trying defuddle.md API fallback...");
|
||||
|
||||
try {
|
||||
const remoteResult = await fetchDefuddleApiMarkdown(args.url);
|
||||
outputPath = args.output || await generateOutputPath(args.url, remoteResult.title, args.outputDir);
|
||||
await mkdir(path.dirname(outputPath), { recursive: true });
|
||||
|
||||
document = remoteResult.markdown;
|
||||
conversionMethod = "defuddle-api";
|
||||
fallbackReason = `Local browser capture failed: ${primaryError}`;
|
||||
} catch (remoteError) {
|
||||
const remoteMessage = remoteError instanceof Error ? remoteError.message : String(remoteError);
|
||||
throw new Error(`Local browser capture failed (${primaryError}); defuddle.md fallback failed (${remoteMessage})`);
|
||||
}
|
||||
}
|
||||
|
||||
if (args.downloadMedia) {
|
||||
const mediaResult = await localizeMarkdownMedia(document, {
|
||||
markdownPath: outputPath,
|
||||
log: console.log,
|
||||
});
|
||||
document = mediaResult.markdown;
|
||||
if (mediaResult.downloadedImages > 0 || mediaResult.downloadedVideos > 0) {
|
||||
console.log(`Downloaded: ${mediaResult.downloadedImages} images, ${mediaResult.downloadedVideos} videos`);
|
||||
}
|
||||
} else {
|
||||
const { images, videos } = countRemoteMedia(document);
|
||||
if (images > 0 || videos > 0) {
|
||||
console.log(`Remote media found: ${images} images, ${videos} videos`);
|
||||
}
|
||||
}
|
||||
|
||||
await writeFile(outputPath, document, "utf-8");
|
||||
|
||||
console.log(`Saved: ${outputPath}`);
|
||||
if (htmlSnapshotPath) {
|
||||
console.log(`Saved HTML: ${htmlSnapshotPath}`);
|
||||
} else {
|
||||
console.log("Saved HTML: unavailable (defuddle.md fallback)");
|
||||
}
|
||||
console.log(`Title: ${extractTitleFromMarkdownDocument(document) || "(no title)"}`);
|
||||
console.log(`Converter: ${conversionMethod}`);
|
||||
if (fallbackReason) {
|
||||
console.warn(`Fallback used: ${fallbackReason}`);
|
||||
}
|
||||
}
|
||||
|
||||
main().catch((err) => {
|
||||
console.error("Error:", err instanceof Error ? err.message : String(err));
|
||||
process.exit(1);
|
||||
});
|
||||
@@ -0,0 +1,305 @@
|
||||
import { parseHTML } from "linkedom";
|
||||
|
||||
export interface PageMetadata {
|
||||
url: string;
|
||||
title: string;
|
||||
description?: string;
|
||||
author?: string;
|
||||
published?: string;
|
||||
coverImage?: string;
|
||||
language?: string;
|
||||
captured_at: string;
|
||||
}
|
||||
|
||||
export interface ConversionResult {
|
||||
metadata: PageMetadata;
|
||||
markdown: string;
|
||||
rawHtml: string;
|
||||
conversionMethod: string;
|
||||
fallbackReason?: string;
|
||||
variables?: Record<string, string>;
|
||||
}
|
||||
|
||||
export type AnyRecord = Record<string, unknown>;
|
||||
|
||||
export const MIN_CONTENT_LENGTH = 120;
|
||||
export const GOOD_CONTENT_LENGTH = 900;
|
||||
|
||||
const PUBLISHED_TIME_SELECTORS = [
|
||||
"meta[property='article:published_time']",
|
||||
"meta[name='pubdate']",
|
||||
"meta[name='publishdate']",
|
||||
"meta[name='date']",
|
||||
"time[datetime]",
|
||||
];
|
||||
|
||||
const ARTICLE_TYPES = new Set([
|
||||
"Article",
|
||||
"NewsArticle",
|
||||
"BlogPosting",
|
||||
"WebPage",
|
||||
"ReportageNewsArticle",
|
||||
]);
|
||||
|
||||
export function pickString(...values: unknown[]): string | null {
|
||||
for (const value of values) {
|
||||
if (typeof value === "string") {
|
||||
const trimmed = value.trim();
|
||||
if (trimmed) return trimmed;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
export function normalizeMarkdown(markdown: string): string {
|
||||
return markdown
|
||||
.replace(/\r\n/g, "\n")
|
||||
.replace(/[ \t]+\n/g, "\n")
|
||||
.replace(/\n{3,}/g, "\n\n")
|
||||
.trim();
|
||||
}
|
||||
|
||||
export function parseDocument(html: string): Document {
|
||||
const normalized = /<\s*html[\s>]/i.test(html)
|
||||
? html
|
||||
: `<!doctype html><html><body>${html}</body></html>`;
|
||||
return parseHTML(normalized).document as unknown as Document;
|
||||
}
|
||||
|
||||
export function sanitizeHtml(html: string): string {
|
||||
const { document } = parseHTML(`<div id="__root">${html}</div>`);
|
||||
const root = document.querySelector("#__root");
|
||||
if (!root) return html;
|
||||
|
||||
for (const selector of ["script", "style", "iframe", "noscript", "template", "svg", "path"]) {
|
||||
for (const el of root.querySelectorAll(selector)) {
|
||||
el.remove();
|
||||
}
|
||||
}
|
||||
|
||||
return root.innerHTML;
|
||||
}
|
||||
|
||||
export function extractTextFromHtml(html: string): string {
|
||||
const { document } = parseHTML(`<!doctype html><html><body>${html}</body></html>`);
|
||||
for (const selector of ["script", "style", "noscript", "template", "iframe", "svg", "path"]) {
|
||||
for (const el of document.querySelectorAll(selector)) {
|
||||
el.remove();
|
||||
}
|
||||
}
|
||||
return document.body?.textContent?.replace(/\s+/g, " ").trim() ?? "";
|
||||
}
|
||||
|
||||
export function getMetaContent(document: Document, names: string[]): string | null {
|
||||
for (const name of names) {
|
||||
const element =
|
||||
document.querySelector(`meta[name="${name}"]`) ??
|
||||
document.querySelector(`meta[property="${name}"]`);
|
||||
const content = element?.getAttribute("content");
|
||||
if (content && content.trim()) return content.trim();
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function normalizeLanguageTag(value: string | null): string | null {
|
||||
if (!value) return null;
|
||||
|
||||
const trimmed = value.trim();
|
||||
if (!trimmed) return null;
|
||||
|
||||
const primary = trimmed.split(/[,\s;]/, 1)[0]?.trim();
|
||||
if (!primary) return null;
|
||||
|
||||
return primary.replace(/_/g, "-");
|
||||
}
|
||||
|
||||
function flattenJsonLdItems(data: unknown): AnyRecord[] {
|
||||
if (!data || typeof data !== "object") return [];
|
||||
if (Array.isArray(data)) return data.flatMap(flattenJsonLdItems);
|
||||
|
||||
const item = data as AnyRecord;
|
||||
if (Array.isArray(item["@graph"])) {
|
||||
return (item["@graph"] as unknown[]).flatMap(flattenJsonLdItems);
|
||||
}
|
||||
|
||||
return [item];
|
||||
}
|
||||
|
||||
function parseJsonLdScripts(document: Document): AnyRecord[] {
|
||||
const results: AnyRecord[] = [];
|
||||
const scripts = document.querySelectorAll("script[type='application/ld+json']");
|
||||
|
||||
for (const script of scripts) {
|
||||
try {
|
||||
const data = JSON.parse(script.textContent ?? "");
|
||||
results.push(...flattenJsonLdItems(data));
|
||||
} catch {
|
||||
// Ignore malformed blocks.
|
||||
}
|
||||
}
|
||||
|
||||
return results;
|
||||
}
|
||||
|
||||
function isArticleType(item: AnyRecord): boolean {
|
||||
const value = Array.isArray(item["@type"]) ? item["@type"][0] : item["@type"];
|
||||
return typeof value === "string" && ARTICLE_TYPES.has(value);
|
||||
}
|
||||
|
||||
function extractAuthorFromJsonLd(authorData: unknown): string | null {
|
||||
if (typeof authorData === "string") return authorData;
|
||||
if (!authorData || typeof authorData !== "object") return null;
|
||||
|
||||
if (Array.isArray(authorData)) {
|
||||
const names = authorData
|
||||
.map((author) => extractAuthorFromJsonLd(author))
|
||||
.filter((name): name is string => Boolean(name));
|
||||
return names.length > 0 ? names.join(", ") : null;
|
||||
}
|
||||
|
||||
const author = authorData as AnyRecord;
|
||||
return typeof author.name === "string" ? author.name : null;
|
||||
}
|
||||
|
||||
function extractPrimaryJsonLdMeta(document: Document): Partial<PageMetadata> {
|
||||
for (const item of parseJsonLdScripts(document)) {
|
||||
if (!isArticleType(item)) continue;
|
||||
|
||||
return {
|
||||
title: pickString(item.headline, item.name) ?? undefined,
|
||||
description: pickString(item.description) ?? undefined,
|
||||
author: extractAuthorFromJsonLd(item.author) ?? undefined,
|
||||
published: pickString(item.datePublished, item.dateCreated) ?? undefined,
|
||||
coverImage:
|
||||
pickString(
|
||||
item.image,
|
||||
(item.image as AnyRecord | undefined)?.url,
|
||||
(Array.isArray(item.image) ? item.image[0] : undefined) as unknown
|
||||
) ?? undefined,
|
||||
};
|
||||
}
|
||||
|
||||
return {};
|
||||
}
|
||||
|
||||
export function extractPublishedTime(document: Document): string | null {
|
||||
for (const selector of PUBLISHED_TIME_SELECTORS) {
|
||||
const el = document.querySelector(selector);
|
||||
if (!el) continue;
|
||||
const value = el.getAttribute("content") ?? el.getAttribute("datetime");
|
||||
if (value && value.trim()) return value.trim();
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
export function extractTitle(document: Document): string | null {
|
||||
const ogTitle = document.querySelector("meta[property='og:title']")?.getAttribute("content");
|
||||
if (ogTitle && ogTitle.trim()) return ogTitle.trim();
|
||||
|
||||
const twitterTitle = document.querySelector("meta[name='twitter:title']")?.getAttribute("content");
|
||||
if (twitterTitle && twitterTitle.trim()) return twitterTitle.trim();
|
||||
|
||||
const title = document.querySelector("title")?.textContent?.trim();
|
||||
if (title) {
|
||||
const cleaned = title.split(/\s*[-|–—]\s*/)[0]?.trim();
|
||||
if (cleaned) return cleaned;
|
||||
}
|
||||
|
||||
const h1 = document.querySelector("h1")?.textContent?.trim();
|
||||
return h1 || null;
|
||||
}
|
||||
|
||||
export function extractMetadataFromHtml(html: string, url: string, capturedAt: string): PageMetadata {
|
||||
const document = parseDocument(html);
|
||||
const jsonLd = extractPrimaryJsonLdMeta(document);
|
||||
const timeEl = document.querySelector("time[datetime]");
|
||||
const htmlLang = normalizeLanguageTag(document.documentElement?.getAttribute("lang"));
|
||||
const metaLanguage = normalizeLanguageTag(
|
||||
pickString(
|
||||
getMetaContent(document, ["language", "content-language", "og:locale"]),
|
||||
document.querySelector("meta[http-equiv='content-language']")?.getAttribute("content")
|
||||
)
|
||||
);
|
||||
|
||||
return {
|
||||
url,
|
||||
title:
|
||||
pickString(
|
||||
getMetaContent(document, ["og:title", "twitter:title"]),
|
||||
jsonLd.title,
|
||||
document.querySelector("h1")?.textContent,
|
||||
document.title
|
||||
) ?? "",
|
||||
description:
|
||||
pickString(
|
||||
getMetaContent(document, ["description", "og:description", "twitter:description"]),
|
||||
jsonLd.description
|
||||
) ?? undefined,
|
||||
author:
|
||||
pickString(
|
||||
getMetaContent(document, ["author", "article:author", "twitter:creator"]),
|
||||
jsonLd.author
|
||||
) ?? undefined,
|
||||
published:
|
||||
pickString(
|
||||
timeEl?.getAttribute("datetime"),
|
||||
getMetaContent(document, ["article:published_time", "datePublished", "publishdate", "date"]),
|
||||
jsonLd.published,
|
||||
extractPublishedTime(document)
|
||||
) ?? undefined,
|
||||
coverImage:
|
||||
pickString(
|
||||
getMetaContent(document, ["og:image", "twitter:image", "twitter:image:src"]),
|
||||
jsonLd.coverImage
|
||||
) ?? undefined,
|
||||
language: pickString(htmlLang, metaLanguage) ?? undefined,
|
||||
captured_at: capturedAt,
|
||||
};
|
||||
}
|
||||
|
||||
export function isMarkdownUsable(markdown: string, html: string): boolean {
|
||||
const normalized = normalizeMarkdown(markdown);
|
||||
if (!normalized) return false;
|
||||
|
||||
const htmlTextLength = extractTextFromHtml(html).length;
|
||||
if (htmlTextLength < MIN_CONTENT_LENGTH) return true;
|
||||
|
||||
if (normalized.length >= 80) return true;
|
||||
return normalized.length >= Math.min(200, Math.floor(htmlTextLength * 0.2));
|
||||
}
|
||||
|
||||
export function isYouTubeUrl(url: string): boolean {
|
||||
try {
|
||||
const hostname = new URL(url).hostname.toLowerCase();
|
||||
return hostname === "youtu.be" || hostname.endsWith(".youtube.com") || hostname === "youtube.com";
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
function escapeYamlValue(value: string): string {
|
||||
return value.replace(/\\/g, "\\\\").replace(/"/g, '\\"').replace(/\r?\n/g, "\\n");
|
||||
}
|
||||
|
||||
export function formatMetadataYaml(meta: PageMetadata): string {
|
||||
const lines = ["---"];
|
||||
lines.push(`url: ${meta.url}`);
|
||||
lines.push(`title: "${escapeYamlValue(meta.title)}"`);
|
||||
if (meta.description) lines.push(`description: "${escapeYamlValue(meta.description)}"`);
|
||||
if (meta.author) lines.push(`author: "${escapeYamlValue(meta.author)}"`);
|
||||
if (meta.published) lines.push(`published: "${escapeYamlValue(meta.published)}"`);
|
||||
if (meta.coverImage) lines.push(`coverImage: "${escapeYamlValue(meta.coverImage)}"`);
|
||||
if (meta.language) lines.push(`language: "${escapeYamlValue(meta.language)}"`);
|
||||
lines.push(`captured_at: "${escapeYamlValue(meta.captured_at)}"`);
|
||||
lines.push("---");
|
||||
return lines.join("\n");
|
||||
}
|
||||
|
||||
export function createMarkdownDocument(result: ConversionResult): string {
|
||||
const yaml = formatMetadataYaml(result.metadata);
|
||||
const escapedTitle = result.metadata.title.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
||||
const titleRegex = new RegExp(`^#\\s+${escapedTitle}\\s*(\\n|$)`, "i");
|
||||
const hasTitle = titleRegex.test(result.markdown.trimStart());
|
||||
const title = result.metadata.title && !hasTitle ? `\n\n# ${result.metadata.title}\n\n` : "\n\n";
|
||||
return yaml + title + result.markdown;
|
||||
}
|
||||
@@ -0,0 +1,317 @@
|
||||
import path from "node:path";
|
||||
import { mkdir, writeFile } from "node:fs/promises";
|
||||
|
||||
type MediaKind = "image" | "video";
|
||||
type MediaHint = "image" | "unknown";
|
||||
|
||||
type MarkdownLinkCandidate = {
|
||||
url: string;
|
||||
hint: MediaHint;
|
||||
};
|
||||
|
||||
export type LocalizeMarkdownMediaOptions = {
|
||||
markdownPath: string;
|
||||
log?: (message: string) => void;
|
||||
};
|
||||
|
||||
export type LocalizeMarkdownMediaResult = {
|
||||
markdown: string;
|
||||
downloadedImages: number;
|
||||
downloadedVideos: number;
|
||||
imageDir: string | null;
|
||||
videoDir: string | null;
|
||||
};
|
||||
|
||||
const MARKDOWN_LINK_RE = /(!?\[[^\]\n]*\])\((<)?(https?:\/\/[^)\s>]+)(>)?\)/g;
|
||||
const FRONTMATTER_COVER_RE = /^(coverImage:\s*")(https?:\/\/[^"]+)(")/m;
|
||||
|
||||
const IMAGE_EXTENSIONS = new Set([
|
||||
"jpg",
|
||||
"jpeg",
|
||||
"png",
|
||||
"webp",
|
||||
"gif",
|
||||
"bmp",
|
||||
"avif",
|
||||
"heic",
|
||||
"heif",
|
||||
"svg",
|
||||
]);
|
||||
|
||||
const VIDEO_EXTENSIONS = new Set(["mp4", "m4v", "mov", "webm", "mkv"]);
|
||||
|
||||
const MIME_EXTENSION_MAP: Record<string, string> = {
|
||||
"image/jpeg": "jpg",
|
||||
"image/jpg": "jpg",
|
||||
"image/png": "png",
|
||||
"image/webp": "webp",
|
||||
"image/gif": "gif",
|
||||
"image/bmp": "bmp",
|
||||
"image/avif": "avif",
|
||||
"image/heic": "heic",
|
||||
"image/heif": "heif",
|
||||
"image/svg+xml": "svg",
|
||||
"video/mp4": "mp4",
|
||||
"video/webm": "webm",
|
||||
"video/quicktime": "mov",
|
||||
"video/x-m4v": "m4v",
|
||||
};
|
||||
|
||||
const DOWNLOAD_USER_AGENT =
|
||||
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/130.0.0.0 Safari/537.36";
|
||||
|
||||
function normalizeContentType(raw: string | null): string {
|
||||
return raw?.split(";")[0]?.trim().toLowerCase() ?? "";
|
||||
}
|
||||
|
||||
function normalizeExtension(raw: string | undefined | null): string | undefined {
|
||||
if (!raw) return undefined;
|
||||
const trimmed = raw.replace(/^\./, "").trim().toLowerCase();
|
||||
if (!trimmed) return undefined;
|
||||
if (trimmed === "jpeg") return "jpg";
|
||||
if (trimmed === "jpg") return "jpg";
|
||||
return trimmed;
|
||||
}
|
||||
|
||||
function resolveExtensionFromUrl(rawUrl: string): string | undefined {
|
||||
try {
|
||||
const parsed = new URL(rawUrl);
|
||||
const extFromPath = normalizeExtension(path.posix.extname(parsed.pathname));
|
||||
if (extFromPath) return extFromPath;
|
||||
const extFromFormat = normalizeExtension(parsed.searchParams.get("format"));
|
||||
if (extFromFormat) return extFromFormat;
|
||||
} catch {
|
||||
return undefined;
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function resolveKindFromContentType(contentType: string): MediaKind | undefined {
|
||||
if (!contentType) return undefined;
|
||||
if (contentType.startsWith("image/")) return "image";
|
||||
if (contentType.startsWith("video/")) return "video";
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function resolveKindFromExtension(ext: string | undefined): MediaKind | undefined {
|
||||
if (!ext) return undefined;
|
||||
if (IMAGE_EXTENSIONS.has(ext)) return "image";
|
||||
if (VIDEO_EXTENSIONS.has(ext)) return "video";
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function resolveMediaKind(
|
||||
rawUrl: string,
|
||||
contentType: string,
|
||||
extension: string | undefined,
|
||||
hint: MediaHint
|
||||
): MediaKind | undefined {
|
||||
const kindFromType = resolveKindFromContentType(contentType);
|
||||
if (kindFromType) return kindFromType;
|
||||
|
||||
const kindFromExtension = resolveKindFromExtension(extension);
|
||||
if (kindFromExtension) return kindFromExtension;
|
||||
|
||||
if (contentType && contentType !== "application/octet-stream") {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
return hint === "image" ? "image" : undefined;
|
||||
}
|
||||
|
||||
function resolveOutputExtension(
|
||||
contentType: string,
|
||||
extension: string | undefined,
|
||||
kind: MediaKind
|
||||
): string {
|
||||
const extFromMime = normalizeExtension(MIME_EXTENSION_MAP[contentType]);
|
||||
if (extFromMime) return extFromMime;
|
||||
|
||||
const normalizedExt = normalizeExtension(extension);
|
||||
if (normalizedExt) return normalizedExt;
|
||||
|
||||
return kind === "video" ? "mp4" : "jpg";
|
||||
}
|
||||
|
||||
function safeDecodeURIComponent(value: string): string {
|
||||
try {
|
||||
return decodeURIComponent(value);
|
||||
} catch {
|
||||
return value;
|
||||
}
|
||||
}
|
||||
|
||||
function sanitizeFileSegment(input: string): string {
|
||||
return input
|
||||
.replace(/[^a-zA-Z0-9_-]+/g, "-")
|
||||
.replace(/-+/g, "-")
|
||||
.replace(/^[-_]+|[-_]+$/g, "")
|
||||
.slice(0, 48);
|
||||
}
|
||||
|
||||
function resolveFileStem(rawUrl: string, extension: string): string {
|
||||
try {
|
||||
const parsed = new URL(rawUrl);
|
||||
const base = path.posix.basename(parsed.pathname);
|
||||
if (!base) return "";
|
||||
const decodedBase = safeDecodeURIComponent(base);
|
||||
const normalizedExt = normalizeExtension(extension);
|
||||
const stripExt = normalizedExt ? new RegExp(`\\.${normalizedExt}$`, "i") : null;
|
||||
const rawStem = stripExt ? decodedBase.replace(stripExt, "") : decodedBase;
|
||||
return sanitizeFileSegment(rawStem);
|
||||
} catch {
|
||||
return "";
|
||||
}
|
||||
}
|
||||
|
||||
function buildFileName(kind: MediaKind, index: number, sourceUrl: string, extension: string): string {
|
||||
const stem = resolveFileStem(sourceUrl, extension);
|
||||
const prefix = kind === "image" ? "img" : "video";
|
||||
const serial = String(index).padStart(3, "0");
|
||||
const suffix = stem ? `-${stem}` : "";
|
||||
return `${prefix}-${serial}${suffix}.${extension}`;
|
||||
}
|
||||
|
||||
function collectMarkdownLinkCandidates(markdown: string): MarkdownLinkCandidate[] {
|
||||
const candidates: MarkdownLinkCandidate[] = [];
|
||||
const seen = new Set<string>();
|
||||
|
||||
const fmMatch = markdown.match(/^---\n([\s\S]*?)\n---/);
|
||||
if (fmMatch) {
|
||||
const coverMatch = fmMatch[1]?.match(FRONTMATTER_COVER_RE);
|
||||
if (coverMatch?.[2] && !seen.has(coverMatch[2])) {
|
||||
seen.add(coverMatch[2]);
|
||||
candidates.push({ url: coverMatch[2], hint: "image" });
|
||||
}
|
||||
}
|
||||
|
||||
MARKDOWN_LINK_RE.lastIndex = 0;
|
||||
let match: RegExpExecArray | null;
|
||||
while ((match = MARKDOWN_LINK_RE.exec(markdown))) {
|
||||
const label = match[1] ?? "";
|
||||
const rawUrl = match[3] ?? "";
|
||||
if (!rawUrl || seen.has(rawUrl)) continue;
|
||||
seen.add(rawUrl);
|
||||
candidates.push({
|
||||
url: rawUrl,
|
||||
hint: label.startsWith("![") ? "image" : "unknown",
|
||||
});
|
||||
}
|
||||
|
||||
return candidates;
|
||||
}
|
||||
|
||||
function rewriteMarkdownMediaLinks(markdown: string, replacements: Map<string, string>): string {
|
||||
if (replacements.size === 0) return markdown;
|
||||
MARKDOWN_LINK_RE.lastIndex = 0;
|
||||
|
||||
let result = markdown.replace(MARKDOWN_LINK_RE, (full, label, _openAngle, rawUrl) => {
|
||||
const localPath = replacements.get(rawUrl);
|
||||
if (!localPath) return full;
|
||||
return `${label}(${localPath})`;
|
||||
});
|
||||
|
||||
result = result.replace(FRONTMATTER_COVER_RE, (full, prefix, rawUrl, suffix) => {
|
||||
const localPath = replacements.get(rawUrl);
|
||||
if (!localPath) return full;
|
||||
return `${prefix}${localPath}${suffix}`;
|
||||
});
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
export async function localizeMarkdownMedia(
|
||||
markdown: string,
|
||||
options: LocalizeMarkdownMediaOptions
|
||||
): Promise<LocalizeMarkdownMediaResult> {
|
||||
const log = options.log ?? (() => {});
|
||||
const markdownDir = path.dirname(options.markdownPath);
|
||||
const candidates = collectMarkdownLinkCandidates(markdown);
|
||||
|
||||
if (candidates.length === 0) {
|
||||
return {
|
||||
markdown,
|
||||
downloadedImages: 0,
|
||||
downloadedVideos: 0,
|
||||
imageDir: null,
|
||||
videoDir: null,
|
||||
};
|
||||
}
|
||||
|
||||
const replacements = new Map<string, string>();
|
||||
let downloadedImages = 0;
|
||||
let downloadedVideos = 0;
|
||||
|
||||
for (const candidate of candidates) {
|
||||
try {
|
||||
const response = await fetch(candidate.url, {
|
||||
method: "GET",
|
||||
redirect: "follow",
|
||||
headers: {
|
||||
"user-agent": DOWNLOAD_USER_AGENT,
|
||||
},
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
log(`[url-to-markdown] Skip media (${response.status}): ${candidate.url}`);
|
||||
continue;
|
||||
}
|
||||
|
||||
const sourceUrl = response.url || candidate.url;
|
||||
const contentType = normalizeContentType(response.headers.get("content-type"));
|
||||
const extension = resolveExtensionFromUrl(sourceUrl) ?? resolveExtensionFromUrl(candidate.url);
|
||||
const kind = resolveMediaKind(sourceUrl, contentType, extension, candidate.hint);
|
||||
if (!kind) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const outputExtension = resolveOutputExtension(contentType, extension, kind);
|
||||
const nextIndex = kind === "image" ? downloadedImages + 1 : downloadedVideos + 1;
|
||||
const dirName = kind === "image" ? "imgs" : "videos";
|
||||
const targetDir = path.join(markdownDir, dirName);
|
||||
await mkdir(targetDir, { recursive: true });
|
||||
|
||||
const fileName = buildFileName(kind, nextIndex, sourceUrl, outputExtension);
|
||||
const absolutePath = path.join(targetDir, fileName);
|
||||
const relativePath = path.posix.join(dirName, fileName);
|
||||
const bytes = Buffer.from(await response.arrayBuffer());
|
||||
await writeFile(absolutePath, bytes);
|
||||
replacements.set(candidate.url, relativePath);
|
||||
|
||||
if (kind === "image") {
|
||||
downloadedImages = nextIndex;
|
||||
} else {
|
||||
downloadedVideos = nextIndex;
|
||||
}
|
||||
} catch (error) {
|
||||
const message = error instanceof Error ? error.message : String(error ?? "");
|
||||
log(`[url-to-markdown] Failed to download media ${candidate.url}: ${message}`);
|
||||
}
|
||||
}
|
||||
|
||||
return {
|
||||
markdown: rewriteMarkdownMediaLinks(markdown, replacements),
|
||||
downloadedImages,
|
||||
downloadedVideos,
|
||||
imageDir: downloadedImages > 0 ? path.join(markdownDir, "imgs") : null,
|
||||
videoDir: downloadedVideos > 0 ? path.join(markdownDir, "videos") : null,
|
||||
};
|
||||
}
|
||||
|
||||
export function countRemoteMedia(markdown: string): { images: number; videos: number; hasCoverImage: boolean } {
|
||||
const fmMatch = markdown.match(/^---\n([\s\S]*?)\n---/);
|
||||
const hasCoverImage = !!(fmMatch?.[1]?.match(FRONTMATTER_COVER_RE)?.[2]);
|
||||
const candidates = collectMarkdownLinkCandidates(markdown);
|
||||
let images = 0;
|
||||
let videos = 0;
|
||||
for (const c of candidates) {
|
||||
const ext = resolveExtensionFromUrl(c.url);
|
||||
const kind = resolveKindFromExtension(ext);
|
||||
if (kind === "video") {
|
||||
videos++;
|
||||
} else if (kind === "image" || c.hint === "image") {
|
||||
images++;
|
||||
}
|
||||
}
|
||||
return { images, videos, hasCoverImage };
|
||||
}
|
||||
+14
@@ -0,0 +1,14 @@
|
||||
{
|
||||
"name": "baoyu-url-to-markdown-scripts",
|
||||
"private": true,
|
||||
"type": "module",
|
||||
"dependencies": {
|
||||
"@mozilla/readability": "^0.6.0",
|
||||
"baoyu-chrome-cdp": "file:./vendor/baoyu-chrome-cdp",
|
||||
"defuddle": "^0.12.0",
|
||||
"jsdom": "^24.1.3",
|
||||
"linkedom": "^0.18.12",
|
||||
"turndown": "^7.2.2",
|
||||
"turndown-plugin-gfm": "^1.0.2"
|
||||
}
|
||||
}
|
||||
+29
@@ -0,0 +1,29 @@
|
||||
import os from "node:os";
|
||||
import path from "node:path";
|
||||
import process from "node:process";
|
||||
|
||||
const APP_DATA_DIR = "baoyu-skills";
|
||||
const URL_TO_MARKDOWN_DATA_DIR = "url-to-markdown";
|
||||
const PROFILE_DIR_NAME = "chrome-profile";
|
||||
|
||||
export function resolveUserDataRoot(): string {
|
||||
if (process.platform === "win32") {
|
||||
return process.env.APPDATA ?? path.join(os.homedir(), "AppData", "Roaming");
|
||||
}
|
||||
if (process.platform === "darwin") {
|
||||
return path.join(os.homedir(), "Library", "Application Support");
|
||||
}
|
||||
return process.env.XDG_DATA_HOME ?? path.join(os.homedir(), ".local", "share");
|
||||
}
|
||||
|
||||
export function resolveUrlToMarkdownDataDir(): string {
|
||||
const override = process.env.URL_DATA_DIR?.trim();
|
||||
if (override) return path.resolve(override);
|
||||
return path.join(process.cwd(), URL_TO_MARKDOWN_DATA_DIR);
|
||||
}
|
||||
|
||||
export function resolveUrlToMarkdownChromeProfileDir(): string {
|
||||
const override = process.env.BAOYU_CHROME_PROFILE_DIR?.trim() || process.env.URL_CHROME_PROFILE_DIR?.trim();
|
||||
if (override) return path.resolve(override);
|
||||
return path.join(resolveUserDataRoot(), APP_DATA_DIR, PROFILE_DIR_NAME);
|
||||
}
|
||||
@@ -0,0 +1,9 @@
|
||||
{
|
||||
"name": "baoyu-chrome-cdp",
|
||||
"private": true,
|
||||
"version": "0.1.0",
|
||||
"type": "module",
|
||||
"exports": {
|
||||
".": "./src/index.ts"
|
||||
}
|
||||
}
|
||||
+307
@@ -0,0 +1,307 @@
|
||||
import assert from "node:assert/strict";
|
||||
import { spawn, type ChildProcess } from "node:child_process";
|
||||
import fs from "node:fs/promises";
|
||||
import http from "node:http";
|
||||
import os from "node:os";
|
||||
import path from "node:path";
|
||||
import process from "node:process";
|
||||
import test, { type TestContext } from "node:test";
|
||||
|
||||
import {
|
||||
discoverRunningChromeDebugPort,
|
||||
findChromeExecutable,
|
||||
findExistingChromeDebugPort,
|
||||
getFreePort,
|
||||
openPageSession,
|
||||
resolveSharedChromeProfileDir,
|
||||
waitForChromeDebugPort,
|
||||
} from "./index.ts";
|
||||
|
||||
function useEnv(
|
||||
t: TestContext,
|
||||
values: Record<string, string | null>,
|
||||
): void {
|
||||
const previous = new Map<string, string | undefined>();
|
||||
for (const [key, value] of Object.entries(values)) {
|
||||
previous.set(key, process.env[key]);
|
||||
if (value == null) {
|
||||
delete process.env[key];
|
||||
} else {
|
||||
process.env[key] = value;
|
||||
}
|
||||
}
|
||||
|
||||
t.after(() => {
|
||||
for (const [key, value] of previous.entries()) {
|
||||
if (value == null) {
|
||||
delete process.env[key];
|
||||
} else {
|
||||
process.env[key] = value;
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
async function makeTempDir(prefix: string): Promise<string> {
|
||||
return fs.mkdtemp(path.join(os.tmpdir(), prefix));
|
||||
}
|
||||
|
||||
async function startDebugServer(port: number): Promise<http.Server> {
|
||||
const server = http.createServer((req, res) => {
|
||||
if (req.url === "/json/version") {
|
||||
res.writeHead(200, { "Content-Type": "application/json" });
|
||||
res.end(JSON.stringify({
|
||||
webSocketDebuggerUrl: `ws://127.0.0.1:${port}/devtools/browser/demo`,
|
||||
}));
|
||||
return;
|
||||
}
|
||||
|
||||
res.writeHead(404);
|
||||
res.end();
|
||||
});
|
||||
|
||||
await new Promise<void>((resolve, reject) => {
|
||||
server.once("error", reject);
|
||||
server.listen(port, "127.0.0.1", () => resolve());
|
||||
});
|
||||
|
||||
return server;
|
||||
}
|
||||
|
||||
async function closeServer(server: http.Server): Promise<void> {
|
||||
await new Promise<void>((resolve, reject) => {
|
||||
server.close((error) => {
|
||||
if (error) reject(error);
|
||||
else resolve();
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
function shellPathForPlatform(): string | null {
|
||||
if (process.platform === "win32") return null;
|
||||
return "/bin/bash";
|
||||
}
|
||||
|
||||
async function startFakeChromiumProcess(port: number): Promise<ChildProcess | null> {
|
||||
const shell = shellPathForPlatform();
|
||||
if (!shell) return null;
|
||||
|
||||
const child = spawn(
|
||||
shell,
|
||||
[
|
||||
"-lc",
|
||||
`exec -a chromium-mock ${JSON.stringify(process.execPath)} -e 'setInterval(() => {}, 1000)' -- --remote-debugging-port=${port}`,
|
||||
],
|
||||
{ stdio: "ignore" },
|
||||
);
|
||||
|
||||
await new Promise((resolve) => setTimeout(resolve, 250));
|
||||
return child;
|
||||
}
|
||||
|
||||
async function stopProcess(child: ChildProcess | null): Promise<void> {
|
||||
if (!child) return;
|
||||
if (child.exitCode !== null || child.signalCode !== null) return;
|
||||
|
||||
child.kill("SIGTERM");
|
||||
await new Promise((resolve) => setTimeout(resolve, 100));
|
||||
if (child.exitCode === null && child.signalCode === null) child.kill("SIGKILL");
|
||||
if (child.exitCode !== null || child.signalCode !== null) return;
|
||||
await new Promise((resolve) => child.once("exit", resolve));
|
||||
}
|
||||
|
||||
test("getFreePort honors a fixed environment override and otherwise allocates a TCP port", async (t) => {
|
||||
useEnv(t, { TEST_FIXED_PORT: "45678" });
|
||||
assert.equal(await getFreePort("TEST_FIXED_PORT"), 45678);
|
||||
|
||||
const dynamicPort = await getFreePort();
|
||||
assert.ok(Number.isInteger(dynamicPort));
|
||||
assert.ok(dynamicPort > 0);
|
||||
});
|
||||
|
||||
test("findChromeExecutable prefers env overrides and falls back to candidate paths", async (t) => {
|
||||
const root = await makeTempDir("baoyu-chrome-bin-");
|
||||
t.after(() => fs.rm(root, { recursive: true, force: true }));
|
||||
|
||||
const envChrome = path.join(root, "env-chrome");
|
||||
const fallbackChrome = path.join(root, "fallback-chrome");
|
||||
await fs.writeFile(envChrome, "");
|
||||
await fs.writeFile(fallbackChrome, "");
|
||||
|
||||
useEnv(t, { BAOYU_CHROME_PATH: envChrome });
|
||||
assert.equal(
|
||||
findChromeExecutable({
|
||||
envNames: ["BAOYU_CHROME_PATH"],
|
||||
candidates: { default: [fallbackChrome] },
|
||||
}),
|
||||
envChrome,
|
||||
);
|
||||
|
||||
useEnv(t, { BAOYU_CHROME_PATH: null });
|
||||
assert.equal(
|
||||
findChromeExecutable({
|
||||
envNames: ["BAOYU_CHROME_PATH"],
|
||||
candidates: { default: [fallbackChrome] },
|
||||
}),
|
||||
fallbackChrome,
|
||||
);
|
||||
});
|
||||
|
||||
test("resolveSharedChromeProfileDir supports env overrides, WSL paths, and default suffixes", (t) => {
|
||||
useEnv(t, { BAOYU_SHARED_PROFILE: "/tmp/custom-profile" });
|
||||
assert.equal(
|
||||
resolveSharedChromeProfileDir({
|
||||
envNames: ["BAOYU_SHARED_PROFILE"],
|
||||
appDataDirName: "demo-app",
|
||||
profileDirName: "demo-profile",
|
||||
}),
|
||||
path.resolve("/tmp/custom-profile"),
|
||||
);
|
||||
|
||||
useEnv(t, { BAOYU_SHARED_PROFILE: null });
|
||||
assert.equal(
|
||||
resolveSharedChromeProfileDir({
|
||||
wslWindowsHome: "/mnt/c/Users/demo",
|
||||
appDataDirName: "demo-app",
|
||||
profileDirName: "demo-profile",
|
||||
}),
|
||||
path.join("/mnt/c/Users/demo", ".local", "share", "demo-app", "demo-profile"),
|
||||
);
|
||||
|
||||
const fallback = resolveSharedChromeProfileDir({
|
||||
appDataDirName: "demo-app",
|
||||
profileDirName: "demo-profile",
|
||||
});
|
||||
assert.match(fallback, /demo-app[\\/]demo-profile$/);
|
||||
});
|
||||
|
||||
test("findExistingChromeDebugPort reads DevToolsActivePort and validates it against a live endpoint", async (t) => {
|
||||
const root = await makeTempDir("baoyu-cdp-profile-");
|
||||
t.after(() => fs.rm(root, { recursive: true, force: true }));
|
||||
|
||||
const port = await getFreePort();
|
||||
const server = await startDebugServer(port);
|
||||
t.after(() => closeServer(server));
|
||||
|
||||
await fs.writeFile(path.join(root, "DevToolsActivePort"), `${port}\n/devtools/browser/demo\n`);
|
||||
|
||||
const found = await findExistingChromeDebugPort({ profileDir: root, timeoutMs: 1000 });
|
||||
assert.equal(found, port);
|
||||
});
|
||||
|
||||
test("discoverRunningChromeDebugPort reads DevToolsActivePort from the provided user-data dir", async (t) => {
|
||||
const root = await makeTempDir("baoyu-cdp-user-data-");
|
||||
t.after(() => fs.rm(root, { recursive: true, force: true }));
|
||||
|
||||
const port = await getFreePort();
|
||||
const server = await startDebugServer(port);
|
||||
t.after(() => closeServer(server));
|
||||
|
||||
await fs.writeFile(path.join(root, "DevToolsActivePort"), `${port}\n/devtools/browser/demo\n`);
|
||||
|
||||
const found = await discoverRunningChromeDebugPort({
|
||||
userDataDirs: [root],
|
||||
timeoutMs: 1000,
|
||||
});
|
||||
assert.deepEqual(found, {
|
||||
port,
|
||||
wsUrl: `ws://127.0.0.1:${port}/devtools/browser/demo`,
|
||||
});
|
||||
});
|
||||
|
||||
test("discoverRunningChromeDebugPort ignores unrelated debugging processes", async (t) => {
|
||||
if (process.platform === "win32") {
|
||||
t.skip("Process discovery fallback is not used on Windows.");
|
||||
return;
|
||||
}
|
||||
|
||||
const root = await makeTempDir("baoyu-cdp-user-data-");
|
||||
t.after(() => fs.rm(root, { recursive: true, force: true }));
|
||||
|
||||
const port = await getFreePort();
|
||||
const server = await startDebugServer(port);
|
||||
t.after(() => closeServer(server));
|
||||
|
||||
const fakeChromium = await startFakeChromiumProcess(port);
|
||||
t.after(async () => { await stopProcess(fakeChromium); });
|
||||
|
||||
const found = await discoverRunningChromeDebugPort({
|
||||
userDataDirs: [root],
|
||||
timeoutMs: 1000,
|
||||
});
|
||||
assert.equal(found, null);
|
||||
});
|
||||
|
||||
test("openPageSession reports whether it created a new target", async () => {
|
||||
const calls: string[] = [];
|
||||
const cdpExisting = {
|
||||
send: async <T>(method: string): Promise<T> => {
|
||||
calls.push(method);
|
||||
if (method === "Target.getTargets") {
|
||||
return {
|
||||
targetInfos: [{ targetId: "existing-target", type: "page", url: "https://gemini.google.com/app" }],
|
||||
} as T;
|
||||
}
|
||||
if (method === "Target.attachToTarget") return { sessionId: "session-existing" } as T;
|
||||
throw new Error(`Unexpected method: ${method}`);
|
||||
},
|
||||
};
|
||||
|
||||
const existing = await openPageSession({
|
||||
cdp: cdpExisting as never,
|
||||
reusing: false,
|
||||
url: "https://gemini.google.com/app",
|
||||
matchTarget: (target) => target.url.includes("gemini.google.com"),
|
||||
activateTarget: false,
|
||||
});
|
||||
|
||||
assert.deepEqual(existing, {
|
||||
sessionId: "session-existing",
|
||||
targetId: "existing-target",
|
||||
createdTarget: false,
|
||||
});
|
||||
assert.deepEqual(calls, ["Target.getTargets", "Target.attachToTarget"]);
|
||||
|
||||
const createCalls: string[] = [];
|
||||
const cdpCreated = {
|
||||
send: async <T>(method: string): Promise<T> => {
|
||||
createCalls.push(method);
|
||||
if (method === "Target.getTargets") return { targetInfos: [] } as T;
|
||||
if (method === "Target.createTarget") return { targetId: "created-target" } as T;
|
||||
if (method === "Target.attachToTarget") return { sessionId: "session-created" } as T;
|
||||
throw new Error(`Unexpected method: ${method}`);
|
||||
},
|
||||
};
|
||||
|
||||
const created = await openPageSession({
|
||||
cdp: cdpCreated as never,
|
||||
reusing: false,
|
||||
url: "https://gemini.google.com/app",
|
||||
matchTarget: (target) => target.url.includes("gemini.google.com"),
|
||||
activateTarget: false,
|
||||
});
|
||||
|
||||
assert.deepEqual(created, {
|
||||
sessionId: "session-created",
|
||||
targetId: "created-target",
|
||||
createdTarget: true,
|
||||
});
|
||||
assert.deepEqual(createCalls, ["Target.getTargets", "Target.createTarget", "Target.attachToTarget"]);
|
||||
});
|
||||
|
||||
test("waitForChromeDebugPort retries until the debug endpoint becomes available", async (t) => {
|
||||
const port = await getFreePort();
|
||||
|
||||
const serverPromise = (async () => {
|
||||
await new Promise((resolve) => setTimeout(resolve, 200));
|
||||
const server = await startDebugServer(port);
|
||||
t.after(() => closeServer(server));
|
||||
})();
|
||||
|
||||
const websocketUrl = await waitForChromeDebugPort(port, 4000, {
|
||||
includeLastError: true,
|
||||
});
|
||||
await serverPromise;
|
||||
|
||||
assert.equal(websocketUrl, `ws://127.0.0.1:${port}/devtools/browser/demo`);
|
||||
});
|
||||
@@ -0,0 +1,523 @@
|
||||
import { spawn, spawnSync, type ChildProcess } from "node:child_process";
|
||||
import fs from "node:fs";
|
||||
import net from "node:net";
|
||||
import os from "node:os";
|
||||
import path from "node:path";
|
||||
import process from "node:process";
|
||||
|
||||
export type PlatformCandidates = {
|
||||
darwin?: string[];
|
||||
win32?: string[];
|
||||
default: string[];
|
||||
};
|
||||
|
||||
type PendingRequest = {
|
||||
resolve: (value: unknown) => void;
|
||||
reject: (error: Error) => void;
|
||||
timer: ReturnType<typeof setTimeout> | null;
|
||||
};
|
||||
|
||||
type CdpSendOptions = {
|
||||
sessionId?: string;
|
||||
timeoutMs?: number;
|
||||
};
|
||||
|
||||
type FetchJsonOptions = {
|
||||
timeoutMs?: number;
|
||||
};
|
||||
|
||||
type FindChromeExecutableOptions = {
|
||||
candidates: PlatformCandidates;
|
||||
envNames?: string[];
|
||||
};
|
||||
|
||||
type ResolveSharedChromeProfileDirOptions = {
|
||||
envNames?: string[];
|
||||
appDataDirName?: string;
|
||||
profileDirName?: string;
|
||||
wslWindowsHome?: string | null;
|
||||
};
|
||||
|
||||
type FindExistingChromeDebugPortOptions = {
|
||||
profileDir: string;
|
||||
timeoutMs?: number;
|
||||
};
|
||||
|
||||
export type ChromeChannel = "stable" | "beta" | "canary" | "dev";
|
||||
|
||||
export type DiscoveredChrome = {
|
||||
port: number;
|
||||
wsUrl: string;
|
||||
};
|
||||
|
||||
type DiscoverRunningChromeOptions = {
|
||||
channels?: ChromeChannel[];
|
||||
userDataDirs?: string[];
|
||||
timeoutMs?: number;
|
||||
};
|
||||
|
||||
type LaunchChromeOptions = {
|
||||
chromePath: string;
|
||||
profileDir: string;
|
||||
port: number;
|
||||
url?: string;
|
||||
headless?: boolean;
|
||||
extraArgs?: string[];
|
||||
};
|
||||
|
||||
type ChromeTargetInfo = {
|
||||
targetId: string;
|
||||
url: string;
|
||||
type: string;
|
||||
};
|
||||
|
||||
type OpenPageSessionOptions = {
|
||||
cdp: CdpConnection;
|
||||
reusing: boolean;
|
||||
url: string;
|
||||
matchTarget: (target: ChromeTargetInfo) => boolean;
|
||||
enablePage?: boolean;
|
||||
enableRuntime?: boolean;
|
||||
enableDom?: boolean;
|
||||
enableNetwork?: boolean;
|
||||
activateTarget?: boolean;
|
||||
};
|
||||
|
||||
export type PageSession = {
|
||||
sessionId: string;
|
||||
targetId: string;
|
||||
createdTarget: boolean;
|
||||
};
|
||||
|
||||
export function sleep(ms: number): Promise<void> {
|
||||
return new Promise((resolve) => setTimeout(resolve, ms));
|
||||
}
|
||||
|
||||
export async function getFreePort(fixedEnvName?: string): Promise<number> {
|
||||
const fixed = fixedEnvName ? Number.parseInt(process.env[fixedEnvName] ?? "", 10) : NaN;
|
||||
if (Number.isInteger(fixed) && fixed > 0) return fixed;
|
||||
|
||||
return await new Promise((resolve, reject) => {
|
||||
const server = net.createServer();
|
||||
server.unref();
|
||||
server.on("error", reject);
|
||||
server.listen(0, "127.0.0.1", () => {
|
||||
const address = server.address();
|
||||
if (!address || typeof address === "string") {
|
||||
server.close(() => reject(new Error("Unable to allocate a free TCP port.")));
|
||||
return;
|
||||
}
|
||||
const port = address.port;
|
||||
server.close((err) => {
|
||||
if (err) reject(err);
|
||||
else resolve(port);
|
||||
});
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
export function findChromeExecutable(options: FindChromeExecutableOptions): string | undefined {
|
||||
for (const envName of options.envNames ?? []) {
|
||||
const override = process.env[envName]?.trim();
|
||||
if (override && fs.existsSync(override)) return override;
|
||||
}
|
||||
|
||||
const candidates = process.platform === "darwin"
|
||||
? options.candidates.darwin ?? options.candidates.default
|
||||
: process.platform === "win32"
|
||||
? options.candidates.win32 ?? options.candidates.default
|
||||
: options.candidates.default;
|
||||
|
||||
for (const candidate of candidates) {
|
||||
if (fs.existsSync(candidate)) return candidate;
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
export function resolveSharedChromeProfileDir(options: ResolveSharedChromeProfileDirOptions = {}): string {
|
||||
for (const envName of options.envNames ?? []) {
|
||||
const override = process.env[envName]?.trim();
|
||||
if (override) return path.resolve(override);
|
||||
}
|
||||
|
||||
const appDataDirName = options.appDataDirName ?? "baoyu-skills";
|
||||
const profileDirName = options.profileDirName ?? "chrome-profile";
|
||||
|
||||
if (options.wslWindowsHome) {
|
||||
return path.join(options.wslWindowsHome, ".local", "share", appDataDirName, profileDirName);
|
||||
}
|
||||
|
||||
const base = process.platform === "darwin"
|
||||
? path.join(os.homedir(), "Library", "Application Support")
|
||||
: process.platform === "win32"
|
||||
? (process.env.APPDATA ?? path.join(os.homedir(), "AppData", "Roaming"))
|
||||
: (process.env.XDG_DATA_HOME ?? path.join(os.homedir(), ".local", "share"));
|
||||
return path.join(base, appDataDirName, profileDirName);
|
||||
}
|
||||
|
||||
async function fetchWithTimeout(url: string, timeoutMs?: number): Promise<Response> {
|
||||
if (!timeoutMs || timeoutMs <= 0) return await fetch(url, { redirect: "follow" });
|
||||
|
||||
const ctl = new AbortController();
|
||||
const timer = setTimeout(() => ctl.abort(), timeoutMs);
|
||||
try {
|
||||
return await fetch(url, { redirect: "follow", signal: ctl.signal });
|
||||
} finally {
|
||||
clearTimeout(timer);
|
||||
}
|
||||
}
|
||||
|
||||
async function fetchJson<T = unknown>(url: string, options: FetchJsonOptions = {}): Promise<T> {
|
||||
const response = await fetchWithTimeout(url, options.timeoutMs);
|
||||
if (!response.ok) {
|
||||
throw new Error(`Request failed: ${response.status} ${response.statusText}`);
|
||||
}
|
||||
return await response.json() as T;
|
||||
}
|
||||
|
||||
async function isDebugPortReady(port: number, timeoutMs = 3_000): Promise<boolean> {
|
||||
try {
|
||||
const version = await fetchJson<{ webSocketDebuggerUrl?: string }>(
|
||||
`http://127.0.0.1:${port}/json/version`,
|
||||
{ timeoutMs }
|
||||
);
|
||||
return !!version.webSocketDebuggerUrl;
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
function isPortListening(port: number, timeoutMs = 3_000): Promise<boolean> {
|
||||
return new Promise((resolve) => {
|
||||
const socket = new net.Socket();
|
||||
const timer = setTimeout(() => { socket.destroy(); resolve(false); }, timeoutMs);
|
||||
socket.once("connect", () => { clearTimeout(timer); socket.destroy(); resolve(true); });
|
||||
socket.once("error", () => { clearTimeout(timer); resolve(false); });
|
||||
socket.connect(port, "127.0.0.1");
|
||||
});
|
||||
}
|
||||
|
||||
function parseDevToolsActivePort(filePath: string): { port: number; wsPath: string } | null {
|
||||
try {
|
||||
const content = fs.readFileSync(filePath, "utf-8");
|
||||
const lines = content.split(/\r?\n/);
|
||||
const port = Number.parseInt(lines[0]?.trim() ?? "", 10);
|
||||
const wsPath = lines[1]?.trim();
|
||||
if (port > 0 && wsPath) return { port, wsPath };
|
||||
} catch {}
|
||||
return null;
|
||||
}
|
||||
|
||||
export async function findExistingChromeDebugPort(options: FindExistingChromeDebugPortOptions): Promise<number | null> {
|
||||
const timeoutMs = options.timeoutMs ?? 3_000;
|
||||
const parsed = parseDevToolsActivePort(path.join(options.profileDir, "DevToolsActivePort"));
|
||||
|
||||
if (parsed && parsed.port > 0 && await isDebugPortReady(parsed.port, timeoutMs)) return parsed.port;
|
||||
|
||||
if (process.platform === "win32") return null;
|
||||
|
||||
try {
|
||||
const result = spawnSync("ps", ["aux"], { encoding: "utf-8", timeout: 5_000 });
|
||||
if (result.status !== 0 || !result.stdout) return null;
|
||||
|
||||
const lines = result.stdout
|
||||
.split("\n")
|
||||
.filter((line) => line.includes(options.profileDir) && line.includes("--remote-debugging-port="));
|
||||
|
||||
for (const line of lines) {
|
||||
const portMatch = line.match(/--remote-debugging-port=(\d+)/);
|
||||
const port = Number.parseInt(portMatch?.[1] ?? "", 10);
|
||||
if (port > 0 && await isDebugPortReady(port, timeoutMs)) return port;
|
||||
}
|
||||
} catch {}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
export function getDefaultChromeUserDataDirs(channels: ChromeChannel[] = ["stable"]): string[] {
|
||||
const home = os.homedir();
|
||||
const dirs: string[] = [];
|
||||
|
||||
const channelDirs: Record<string, { darwin: string; linux: string; win32: string }> = {
|
||||
stable: {
|
||||
darwin: path.join(home, "Library", "Application Support", "Google", "Chrome"),
|
||||
linux: path.join(home, ".config", "google-chrome"),
|
||||
win32: path.join(process.env.LOCALAPPDATA ?? path.join(home, "AppData", "Local"), "Google", "Chrome", "User Data"),
|
||||
},
|
||||
beta: {
|
||||
darwin: path.join(home, "Library", "Application Support", "Google", "Chrome Beta"),
|
||||
linux: path.join(home, ".config", "google-chrome-beta"),
|
||||
win32: path.join(process.env.LOCALAPPDATA ?? path.join(home, "AppData", "Local"), "Google", "Chrome Beta", "User Data"),
|
||||
},
|
||||
canary: {
|
||||
darwin: path.join(home, "Library", "Application Support", "Google", "Chrome Canary"),
|
||||
linux: path.join(home, ".config", "google-chrome-canary"),
|
||||
win32: path.join(process.env.LOCALAPPDATA ?? path.join(home, "AppData", "Local"), "Google", "Chrome SxS", "User Data"),
|
||||
},
|
||||
dev: {
|
||||
darwin: path.join(home, "Library", "Application Support", "Google", "Chrome Dev"),
|
||||
linux: path.join(home, ".config", "google-chrome-dev"),
|
||||
win32: path.join(process.env.LOCALAPPDATA ?? path.join(home, "AppData", "Local"), "Google", "Chrome Dev", "User Data"),
|
||||
},
|
||||
};
|
||||
|
||||
const platform = process.platform === "darwin" ? "darwin" : process.platform === "win32" ? "win32" : "linux";
|
||||
|
||||
for (const ch of channels) {
|
||||
const entry = channelDirs[ch];
|
||||
if (entry) dirs.push(entry[platform]);
|
||||
}
|
||||
|
||||
return dirs;
|
||||
}
|
||||
|
||||
// Best-effort reuse of an already-running local CDP session discovered from
|
||||
// known Chrome user-data dirs. This is distinct from Chrome DevTools MCP's
|
||||
// prompt-based --autoConnect flow.
|
||||
export async function discoverRunningChromeDebugPort(options: DiscoverRunningChromeOptions = {}): Promise<DiscoveredChrome | null> {
|
||||
const channels = options.channels ?? ["stable", "beta", "canary", "dev"];
|
||||
const timeoutMs = options.timeoutMs ?? 3_000;
|
||||
|
||||
const userDataDirs = (options.userDataDirs ?? getDefaultChromeUserDataDirs(channels))
|
||||
.map((dir) => path.resolve(dir));
|
||||
for (const dir of userDataDirs) {
|
||||
const parsed = parseDevToolsActivePort(path.join(dir, "DevToolsActivePort"));
|
||||
if (!parsed) continue;
|
||||
if (await isPortListening(parsed.port, timeoutMs)) {
|
||||
return { port: parsed.port, wsUrl: `ws://127.0.0.1:${parsed.port}${parsed.wsPath}` };
|
||||
}
|
||||
}
|
||||
|
||||
if (process.platform !== "win32") {
|
||||
try {
|
||||
const result = spawnSync("ps", ["aux"], { encoding: "utf-8", timeout: 5_000 });
|
||||
if (result.status === 0 && result.stdout) {
|
||||
const lines = result.stdout
|
||||
.split("\n")
|
||||
.filter((line) =>
|
||||
line.includes("--remote-debugging-port=") &&
|
||||
userDataDirs.some((dir) => line.includes(dir))
|
||||
);
|
||||
|
||||
for (const line of lines) {
|
||||
const portMatch = line.match(/--remote-debugging-port=(\d+)/);
|
||||
const port = Number.parseInt(portMatch?.[1] ?? "", 10);
|
||||
if (port > 0 && await isDebugPortReady(port, timeoutMs)) {
|
||||
try {
|
||||
const version = await fetchJson<{ webSocketDebuggerUrl?: string }>(`http://127.0.0.1:${port}/json/version`, { timeoutMs });
|
||||
if (version.webSocketDebuggerUrl) return { port, wsUrl: version.webSocketDebuggerUrl };
|
||||
} catch {}
|
||||
}
|
||||
}
|
||||
}
|
||||
} catch {}
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
export async function waitForChromeDebugPort(
|
||||
port: number,
|
||||
timeoutMs: number,
|
||||
options?: { includeLastError?: boolean }
|
||||
): Promise<string> {
|
||||
const start = Date.now();
|
||||
let lastError: unknown = null;
|
||||
|
||||
while (Date.now() - start < timeoutMs) {
|
||||
try {
|
||||
const version = await fetchJson<{ webSocketDebuggerUrl?: string }>(
|
||||
`http://127.0.0.1:${port}/json/version`,
|
||||
{ timeoutMs: 5_000 }
|
||||
);
|
||||
if (version.webSocketDebuggerUrl) return version.webSocketDebuggerUrl;
|
||||
lastError = new Error("Missing webSocketDebuggerUrl");
|
||||
} catch (error) {
|
||||
lastError = error;
|
||||
}
|
||||
await sleep(200);
|
||||
}
|
||||
|
||||
if (options?.includeLastError && lastError) {
|
||||
throw new Error(
|
||||
`Chrome debug port not ready: ${lastError instanceof Error ? lastError.message : String(lastError)}`
|
||||
);
|
||||
}
|
||||
throw new Error("Chrome debug port not ready");
|
||||
}
|
||||
|
||||
export class CdpConnection {
|
||||
private ws: WebSocket;
|
||||
private nextId = 0;
|
||||
private pending = new Map<number, PendingRequest>();
|
||||
private eventHandlers = new Map<string, Set<(params: unknown) => void>>();
|
||||
private defaultTimeoutMs: number;
|
||||
|
||||
private constructor(ws: WebSocket, defaultTimeoutMs = 15_000) {
|
||||
this.ws = ws;
|
||||
this.defaultTimeoutMs = defaultTimeoutMs;
|
||||
|
||||
this.ws.addEventListener("message", (event) => {
|
||||
try {
|
||||
const data = typeof event.data === "string"
|
||||
? event.data
|
||||
: new TextDecoder().decode(event.data as ArrayBuffer);
|
||||
const msg = JSON.parse(data) as {
|
||||
id?: number;
|
||||
method?: string;
|
||||
params?: unknown;
|
||||
result?: unknown;
|
||||
error?: { message?: string };
|
||||
};
|
||||
|
||||
if (msg.method) {
|
||||
const handlers = this.eventHandlers.get(msg.method);
|
||||
if (handlers) {
|
||||
handlers.forEach((handler) => handler(msg.params));
|
||||
}
|
||||
}
|
||||
|
||||
if (msg.id) {
|
||||
const pending = this.pending.get(msg.id);
|
||||
if (pending) {
|
||||
this.pending.delete(msg.id);
|
||||
if (pending.timer) clearTimeout(pending.timer);
|
||||
if (msg.error?.message) pending.reject(new Error(msg.error.message));
|
||||
else pending.resolve(msg.result);
|
||||
}
|
||||
}
|
||||
} catch {}
|
||||
});
|
||||
|
||||
this.ws.addEventListener("close", () => {
|
||||
for (const [id, pending] of this.pending.entries()) {
|
||||
this.pending.delete(id);
|
||||
if (pending.timer) clearTimeout(pending.timer);
|
||||
pending.reject(new Error("CDP connection closed."));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
static async connect(
|
||||
url: string,
|
||||
timeoutMs: number,
|
||||
options?: { defaultTimeoutMs?: number }
|
||||
): Promise<CdpConnection> {
|
||||
const ws = new WebSocket(url);
|
||||
await new Promise<void>((resolve, reject) => {
|
||||
const timer = setTimeout(() => reject(new Error("CDP connection timeout.")), timeoutMs);
|
||||
ws.addEventListener("open", () => {
|
||||
clearTimeout(timer);
|
||||
resolve();
|
||||
});
|
||||
ws.addEventListener("error", () => {
|
||||
clearTimeout(timer);
|
||||
reject(new Error("CDP connection failed."));
|
||||
});
|
||||
});
|
||||
return new CdpConnection(ws, options?.defaultTimeoutMs ?? 15_000);
|
||||
}
|
||||
|
||||
on(method: string, handler: (params: unknown) => void): void {
|
||||
if (!this.eventHandlers.has(method)) {
|
||||
this.eventHandlers.set(method, new Set());
|
||||
}
|
||||
this.eventHandlers.get(method)?.add(handler);
|
||||
}
|
||||
|
||||
off(method: string, handler: (params: unknown) => void): void {
|
||||
this.eventHandlers.get(method)?.delete(handler);
|
||||
}
|
||||
|
||||
async send<T = unknown>(method: string, params?: Record<string, unknown>, options?: CdpSendOptions): Promise<T> {
|
||||
const id = ++this.nextId;
|
||||
const message: Record<string, unknown> = { id, method };
|
||||
if (params) message.params = params;
|
||||
if (options?.sessionId) message.sessionId = options.sessionId;
|
||||
|
||||
const timeoutMs = options?.timeoutMs ?? this.defaultTimeoutMs;
|
||||
const result = await new Promise<unknown>((resolve, reject) => {
|
||||
const timer = timeoutMs > 0
|
||||
? setTimeout(() => {
|
||||
this.pending.delete(id);
|
||||
reject(new Error(`CDP timeout: ${method}`));
|
||||
}, timeoutMs)
|
||||
: null;
|
||||
this.pending.set(id, { resolve, reject, timer });
|
||||
this.ws.send(JSON.stringify(message));
|
||||
});
|
||||
|
||||
return result as T;
|
||||
}
|
||||
|
||||
close(): void {
|
||||
try {
|
||||
this.ws.close();
|
||||
} catch {}
|
||||
}
|
||||
}
|
||||
|
||||
export async function launchChrome(options: LaunchChromeOptions): Promise<ChildProcess> {
|
||||
await fs.promises.mkdir(options.profileDir, { recursive: true });
|
||||
|
||||
const args = [
|
||||
`--remote-debugging-port=${options.port}`,
|
||||
`--user-data-dir=${options.profileDir}`,
|
||||
"--no-first-run",
|
||||
"--no-default-browser-check",
|
||||
...(options.extraArgs ?? []),
|
||||
];
|
||||
if (options.headless) args.push("--headless=new");
|
||||
if (options.url) args.push(options.url);
|
||||
|
||||
return spawn(options.chromePath, args, { stdio: "ignore" });
|
||||
}
|
||||
|
||||
export function killChrome(chrome: ChildProcess): void {
|
||||
try {
|
||||
chrome.kill("SIGTERM");
|
||||
} catch {}
|
||||
setTimeout(() => {
|
||||
if (!chrome.killed) {
|
||||
try {
|
||||
chrome.kill("SIGKILL");
|
||||
} catch {}
|
||||
}
|
||||
}, 2_000).unref?.();
|
||||
}
|
||||
|
||||
export async function openPageSession(options: OpenPageSessionOptions): Promise<PageSession> {
|
||||
let targetId: string;
|
||||
let createdTarget = false;
|
||||
|
||||
if (options.reusing) {
|
||||
const created = await options.cdp.send<{ targetId: string }>("Target.createTarget", { url: options.url });
|
||||
targetId = created.targetId;
|
||||
createdTarget = true;
|
||||
} else {
|
||||
const targets = await options.cdp.send<{ targetInfos: ChromeTargetInfo[] }>("Target.getTargets");
|
||||
const existing = targets.targetInfos.find(options.matchTarget);
|
||||
if (existing) {
|
||||
targetId = existing.targetId;
|
||||
} else {
|
||||
const created = await options.cdp.send<{ targetId: string }>("Target.createTarget", { url: options.url });
|
||||
targetId = created.targetId;
|
||||
createdTarget = true;
|
||||
}
|
||||
}
|
||||
|
||||
const { sessionId } = await options.cdp.send<{ sessionId: string }>(
|
||||
"Target.attachToTarget",
|
||||
{ targetId, flatten: true }
|
||||
);
|
||||
|
||||
if (options.activateTarget ?? true) {
|
||||
await options.cdp.send("Target.activateTarget", { targetId });
|
||||
}
|
||||
if (options.enablePage) await options.cdp.send("Page.enable", {}, { sessionId });
|
||||
if (options.enableRuntime) await options.cdp.send("Runtime.enable", {}, { sessionId });
|
||||
if (options.enableDom) await options.cdp.send("DOM.enable", {}, { sessionId });
|
||||
if (options.enableNetwork) await options.cdp.send("Network.enable", {}, { sessionId });
|
||||
|
||||
return { sessionId, targetId, createdTarget };
|
||||
}
|
||||
Reference in New Issue
Block a user