From be302e91d8cc88c7b2d138ee02c5cf1ce90d51d6 Mon Sep 17 00:00:00 2001 From: Taras Mankovski Date: Sat, 26 Sep 2026 08:45:41 -0400 Subject: [PATCH 1/2] =?UTF-8?q?=E2=9C=A8=20rewrite=20self-referencing=20ur?= =?UTF-8?q?ls=20outside=20html?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `--base` only reached html documents: everything else was streamed to disk byte for byte, so absolute urls in the text a site serves — `llms.txt`, the markdown twins of its pages, `feed.xml` — kept the crawl origin. A deployment had to tell the application its own public url a second way. Textual bodies (`text/*`, json, xml, javascript and any `+json`/`+xml` type) now have the crawl origin substituted for the base. There is no document to walk in text, so `rebaseText` is a substitution rather than a rewrite of known url-bearing attributes, but it maps the origin through the same `rebase` the html pass uses and so honors the base's path. Bodies that are not text are still streamed untouched, so nothing binary gets decoded and re-encoded. The html pass walked `link[href]`, `[src]` and `[content]`, which left `` behind; anchors are now rewritten too. They are not followed: the sitemap says what the site is made of, and downloading every link would crawl past it. Closes #13 --- README.md | 7 +++ downloader.ts | 43 ++++++++++++- rebase.ts | 20 ++++++ test/staticalize.test.ts | 129 +++++++++++++++++++++++++++++++++++++++ 4 files changed, 198 insertions(+), 1 deletion(-) diff --git a/README.md b/README.md index dcb685d..2f9c5c3 100644 --- a/README.md +++ b/README.md @@ -25,6 +25,13 @@ This will read `https://localhost:800/sitemap.xml` and download the entire website to the `dist/` directory in a format that can be served from a simple file server running at `frontside.com`. +Wherever your site refers to itself with an absolute url, staticalize replaces +the `--site` origin with `--base`. It does this in html documents — ``, +``, and any `src` or `content` attribute — and in textual bodies such +as `llms.txt`, markdown, feeds, json, css and javascript. Assets that are not +text are copied byte for byte. This means `--base` is the only place a +deployment has to say where it lives. + ### CLI ``` diff --git a/downloader.ts b/downloader.ts index b0c08d7..c14263e 100644 --- a/downloader.ts +++ b/downloader.ts @@ -11,7 +11,7 @@ import { fromHtml } from "hast-util-from-html"; import { toHtml } from "hast-util-to-html"; import { selectAll } from "hast-util-select"; import { useTaskBuffer } from "./task-buffer.ts"; -import { rebase } from "./rebase.ts"; +import { rebase, rebaseText } from "./rebase.ts"; import { createApi } from "@effectionx/context-api"; export interface Downloader extends Operation { @@ -74,6 +74,18 @@ export const DownloadApi = createApi("@staticalize/download", { } } + // anchors are rewritten but not followed: the sitemap says what the + // site is made of, and downloading every link would crawl past it + let anchors = selectAll("a[href]", html); + + for (let anchor of anchors) { + let href = anchor.properties.href as string; + + if (href.startsWith(host.origin)) { + anchor.properties.href = rebase(new URL(href), base).href; + } + } + let assets = selectAll("[src]", html); for (let element of assets) { @@ -103,6 +115,16 @@ export const DownloadApi = createApi("@staticalize/download", { ok: true, bytes: new TextEncoder().encode(output).byteLength, }; + } else if (isTextual(response.headers.get("Content-Type"))) { + let content = yield* until(response.text()); + let output = rebaseText(content, host, base); + let destdir = dirname(path); + yield* until(ensureDir(destdir)); + yield* until(Deno.writeTextFile(path, output)); + return { + ok: true, + bytes: new TextEncoder().encode(output).byteLength, + }; } else { let size = Number(response.headers.get("Content-Length") ?? 0); let destdir = dirname(path); @@ -120,6 +142,25 @@ export const DownloadApi = createApi("@staticalize/download", { }, }); +/** + * Is this a content type whose body is text we can rewrite urls in? + * + * Everything else is streamed to disk byte for byte, so a binary body is never + * decoded and re-encoded. + */ +function isTextual(contentType: string | null): boolean { + if (!contentType) { + return false; + } + let type = contentType.split(";")[0].trim().toLowerCase(); + return type.startsWith("text/") || + type.endsWith("+json") || + type.endsWith("+xml") || + ["application/json", "application/xml", "application/javascript"].includes( + type, + ); +} + function* fetchWithRetry( url: string, signal: AbortSignal, diff --git a/rebase.ts b/rebase.ts index b12fb20..61b988d 100644 --- a/rebase.ts +++ b/rebase.ts @@ -20,3 +20,23 @@ export function rebase(source: URL, base: URL): URL { url.pathname = `${base.pathname.replace(/\/$/, "")}${source.pathname}`; return url; } + +/** + * Replace self-referencing absolute urls in a text body with the public base. + * + * There is no document to walk in a text body, so unlike the html pass this is + * a substitution of the crawl origin rather than a rewrite of known + * url-bearing attributes. + * + * @param body text served by the crawled site + * @param host url of the site being crawled + * @param base public base url of the site, path included + * @returns the body with every url on the crawled site pointing at the base + */ +export function rebaseText(body: string, host: URL, base: URL): string { + // the crawl origin stands for the root of the site, so it maps onto the base + // the same way any other url does. the result carries a trailing slash, which + // the urls in the body bring themselves. + let prefix = `${rebase(new URL(host.origin), base)}`.replace(/\/$/, ""); + return body.replaceAll(host.origin, prefix); +} diff --git a/test/staticalize.test.ts b/test/staticalize.test.ts index fed628b..d9c846e 100644 --- a/test/staticalize.test.ts +++ b/test/staticalize.test.ts @@ -324,6 +324,135 @@ describe("staticalize", () => { "https://frontside.com/effection/about", ); }); + + it("replaces self-referencing urls in anchors", async () => { + app.get( + "/", + (c) => c.html(`about`), + ) + .get("/about", (c) => c.html("

About

")) + .get(...sitemap(["/", "/about"])); + + await staticalize({ + base: new URL("https://fs.com"), + host, + dir: "test/dist", + }); + + await expect(content("test/dist/index.html")).resolves.toContain( + `about`, + ); + }); + + it("does not download pages that are only linked from an anchor", async () => { + app.get("/", (c) => c.html(`unlisted`)) + .get("/unlisted", (c) => c.html("

Unlisted

")) + .get(...sitemap(["/"])); + + await staticalize({ + base: new URL("https://fs.com"), + host, + dir: "test/dist", + }); + + await expect(exists("test/dist/unlisted/index.html")).resolves.toEqual( + false, + ); + }); + + it("replaces self-referencing urls in text bodies", async () => { + app.get("/llms.txt", (c) => + c.text(`[API]: ${host}api.md\n`, 200, { + "Content-Type": "text/plain; charset=utf-8", + })) + .get("/AGENTS.md", (c) => + c.text(`see ${host}api.md\n`, 200, { + "Content-Type": "text/markdown; charset=utf-8", + })) + .get("/feed.xml", (c) => + c.text(`${host}about`, 200, { + "Content-Type": "application/rss+xml", + })) + .get("/data.json", (c) => + c.text(`{"self":"${host}data.json"}`, 200, { + "Content-Type": "application/json", + })) + .get( + "/styles.css", + (c) => + c.text(`body { background: url(${host}bg.png); }`, 200, { + "Content-Type": "text/css", + }), + ) + .get( + ...sitemap([ + "/llms.txt", + "/AGENTS.md", + "/feed.xml", + "/data.json", + "/styles.css", + ]), + ); + + await staticalize({ + base: new URL("https://fs.com"), + host, + dir: "test/dist", + }); + + await expect(content("test/dist/llms.txt")).resolves.toEqual( + "[API]: https://fs.com/api.md\n", + ); + await expect(content("test/dist/AGENTS.md")).resolves.toEqual( + "see https://fs.com/api.md\n", + ); + await expect(content("test/dist/feed.xml")).resolves.toEqual( + "https://fs.com/about", + ); + await expect(content("test/dist/data.json")).resolves.toEqual( + `{"self":"https://fs.com/data.json"}`, + ); + await expect(content("test/dist/styles.css")).resolves.toEqual( + "body { background: url(https://fs.com/bg.png); }", + ); + }); + + it("carries the path of the base url into text bodies", async () => { + app.get("/llms.txt", (c) => + c.text(`[API]: ${host}api.md\n`, 200, { + "Content-Type": "text/plain; charset=utf-8", + })) + .get(...sitemap(["/llms.txt"])); + + await staticalize({ + base: new URL("https://frontside.com/effection"), + host, + dir: "test/dist", + }); + + await expect(content("test/dist/llms.txt")).resolves.toEqual( + "[API]: https://frontside.com/effection/api.md\n", + ); + }); + + it("streams bodies that are not text byte for byte", async () => { + // a png that would be corrupted by a decode/encode round trip + let png = new Uint8Array([0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a]); + + app.get("/logo.png", (c) => + c.body(png.buffer, 200, { + "Content-Type": "image/png", + })) + .get(...sitemap(["/logo.png"])); + + await staticalize({ + base: new URL("https://fs.com"), + host, + dir: "test/dist", + }); + + await expect(Deno.readFile("test/dist/logo.png")).resolves.toEqual(png); + }); }); async function content(path: string): Promise { From 27afcfae540a3a953e38e75656e408cdc649906d Mon Sep 17 00:00:00 2001 From: Taras Mankovski Date: Sat, 26 Sep 2026 11:17:32 -0400 Subject: [PATCH 2/2] =?UTF-8?q?=E2=99=BB=EF=B8=8F=20stream=20the=20text=20?= =?UTF-8?q?rebase=20instead=20of=20slurping=20it?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Reading each text body into a string to scan it undid, for text, the one good property the byte-for-byte branch had: memory that does not depend on the size of the document. With the default concurrency of 75 that is 75 whole documents in flight, and css and js bundles are not small. `RebaseTextStream` is a TransformStream that does the substitution as the body arrives, after the shape of `TextLineStream` in effectionx's jsonl-store: the crawl origin is a fixed string, so it holds back the longest partial match at the end of each chunk and lets `flush` emit whatever is left. Nothing else is buffered. The text branch now writes the same way the byte-for-byte branch does, and inherits the same behavior on failure: a body that dies midway can leave a truncated file where slurping left none. Verified on a 12mb document: byte-identical output to the slurping version (same sha256) at half the peak rss, 190mb down to 89mb. --- downloader.ts | 28 +++++++---- rebase.ts | 46 ++++++++++++------ test/rebase.test.ts | 110 ++++++++++++++++++++++++++++++++++++++++++++ 3 files changed, 163 insertions(+), 21 deletions(-) create mode 100644 test/rebase.test.ts diff --git a/downloader.ts b/downloader.ts index c14263e..4cfa3ba 100644 --- a/downloader.ts +++ b/downloader.ts @@ -11,7 +11,7 @@ import { fromHtml } from "hast-util-from-html"; import { toHtml } from "hast-util-to-html"; import { selectAll } from "hast-util-select"; import { useTaskBuffer } from "./task-buffer.ts"; -import { rebase, rebaseText } from "./rebase.ts"; +import { rebase, RebaseTextStream } from "./rebase.ts"; import { createApi } from "@effectionx/context-api"; export interface Downloader extends Operation { @@ -116,15 +116,27 @@ export const DownloadApi = createApi("@staticalize/download", { bytes: new TextEncoder().encode(output).byteLength, }; } else if (isTextual(response.headers.get("Content-Type"))) { - let content = yield* until(response.text()); - let output = rebaseText(content, host, base); + let bytes = 0; + let counted = new TransformStream({ + transform(chunk, controller) { + bytes += chunk.byteLength; + controller.enqueue(chunk); + }, + }); + let destdir = dirname(path); yield* until(ensureDir(destdir)); - yield* until(Deno.writeTextFile(path, output)); - return { - ok: true, - bytes: new TextEncoder().encode(output).byteLength, - }; + yield* until( + Deno.writeFile( + path, + response.body! + .pipeThrough(new TextDecoderStream()) + .pipeThrough(new RebaseTextStream(host, base)) + .pipeThrough(new TextEncoderStream()) + .pipeThrough(counted), + ), + ); + return { ok: true, bytes }; } else { let size = Number(response.headers.get("Content-Length") ?? 0); let destdir = dirname(path); diff --git a/rebase.ts b/rebase.ts index 61b988d..e7cc5e5 100644 --- a/rebase.ts +++ b/rebase.ts @@ -22,21 +22,41 @@ export function rebase(source: URL, base: URL): URL { } /** - * Replace self-referencing absolute urls in a text body with the public base. + * A TransformStream that replaces self-referencing urls with the public base. * * There is no document to walk in a text body, so unlike the html pass this is * a substitution of the crawl origin rather than a rewrite of known - * url-bearing attributes. - * - * @param body text served by the crawled site - * @param host url of the site being crawled - * @param base public base url of the site, path included - * @returns the body with every url on the crawled site pointing at the base + * url-bearing attributes. It runs over the response as it arrives, so the + * memory it holds is a chunk and change rather than the whole document. */ -export function rebaseText(body: string, host: URL, base: URL): string { - // the crawl origin stands for the root of the site, so it maps onto the base - // the same way any other url does. the result carries a trailing slash, which - // the urls in the body bring themselves. - let prefix = `${rebase(new URL(host.origin), base)}`.replace(/\/$/, ""); - return body.replaceAll(host.origin, prefix); +export class RebaseTextStream extends TransformStream { + #carry = ""; + + constructor(host: URL, base: URL) { + let needle = host.origin; + // the crawl origin stands for the root of the site, so it maps onto the base + // the same way any other url does. the result carries a trailing slash, which + // the urls in the body bring themselves. + let prefix = `${rebase(new URL(needle), base)}`.replace(/\/$/, ""); + // a match can straddle a chunk boundary, so hold back the longest partial + // one and let the next chunk complete it + let keep = needle.length - 1; + + super({ + transform: (chunk, controller) => { + let text = (this.#carry + chunk).replaceAll(needle, prefix); + if (text.length > keep) { + controller.enqueue(text.slice(0, text.length - keep)); + this.#carry = text.slice(text.length - keep); + } else { + this.#carry = text; + } + }, + flush: (controller) => { + if (this.#carry.length > 0) { + controller.enqueue(this.#carry); + } + }, + }); + } } diff --git a/test/rebase.test.ts b/test/rebase.test.ts new file mode 100644 index 0000000..49513ba --- /dev/null +++ b/test/rebase.test.ts @@ -0,0 +1,110 @@ +import { expect } from "@std/expect"; +import { describe, it } from "@std/testing/bdd"; + +import { rebase, RebaseTextStream } from "../rebase.ts"; + +let host = new URL("http://localhost:8321"); + +async function rebased( + chunks: string[], + base: URL, + from: URL = host, +): Promise { + let stream = ReadableStream.from(chunks) + .pipeThrough(new RebaseTextStream(from, base)); + let out = ""; + for await (let chunk of stream) { + out += chunk; + } + return out; +} + +function cut(text: string, size: number): string[] { + let chunks: string[] = []; + for (let i = 0; i < text.length; i += size) { + chunks.push(text.slice(i, i + size)); + } + return chunks; +} + +describe("rebase", () => { + it("takes the protocol, host and port of the base", () => { + expect(rebase(new URL(`${host}about`), new URL("https://fs.com")).href) + .toEqual("https://fs.com/about"); + }); + + it("prefixes the path of the base", () => { + expect( + rebase(new URL(`${host}about`), new URL("https://fs.com/effection")).href, + ).toEqual("https://fs.com/effection/about"); + }); + + it("treats a trailing slash on the base as no path at all", () => { + expect(rebase(new URL(`${host}about`), new URL("https://fs.com/")).href) + .toEqual("https://fs.com/about"); + }); + + it("keeps the query and fragment of the source", () => { + expect( + rebase(new URL(`${host}search?q=hi#top`), new URL("https://fs.com")).href, + ).toEqual("https://fs.com/search?q=hi#top"); + }); +}); + +describe("RebaseTextStream", () => { + it("replaces every occurrence in a body", async () => { + await expect( + rebased( + [`[API]: ${host}api.md and ${host}b.md`], + new URL( + "https://fs.com", + ), + ), + ).resolves.toEqual("[API]: https://fs.com/api.md and https://fs.com/b.md"); + }); + + it("leaves urls on other hosts alone", async () => { + await expect( + rebased( + ["see https://google.com/a and http://localhost:9999/b"], + new URL( + "https://fs.com", + ), + ), + ).resolves.toEqual("see https://google.com/a and http://localhost:9999/b"); + }); + + it("carries the path of the base", async () => { + await expect( + rebased([`${host}api.md`], new URL("https://fs.com/effection")), + ).resolves.toEqual("https://fs.com/effection/api.md"); + }); + + it("matches urls split across chunk boundaries", async () => { + // the interesting cases are a match at the very start, at the very end, + // back to back with another, and bare with no path of its own + let body = + `${host}a.md then ${host}b.md${host}c.md and a bare ${host} to close`; + let want = body.replaceAll(host.origin, "https://fs.com"); + + for (let size of [1, 2, 3, 7, host.origin.length - 1, host.origin.length]) { + await expect(rebased(cut(body, size), new URL("https://fs.com"))) + .resolves.toEqual(want); + } + }); + + it("emits a body that ends mid-match as it stands", async () => { + // a truncated url is not a url, so it survives rather than disappearing + let partial = host.origin.slice(0, -3); + await expect(rebased( + cut(`ends with ${partial}`, 4), + new URL( + "https://fs.com", + ), + )).resolves.toEqual(`ends with ${partial}`); + }); + + it("passes an empty body through", async () => { + await expect(rebased([], new URL("https://fs.com"))).resolves.toEqual(""); + }); +});