diff --git a/README.md b/README.md index dcb685d..2f9c5c3 100644 --- a/README.md +++ b/README.md @@ -25,6 +25,13 @@ This will read `https://localhost:800/sitemap.xml` and download the entire website to the `dist/` directory in a format that can be served from a simple file server running at `frontside.com`. +Wherever your site refers to itself with an absolute url, staticalize replaces +the `--site` origin with `--base`. It does this in html documents — ``, +``, and any `src` or `content` attribute — and in textual bodies such +as `llms.txt`, markdown, feeds, json, css and javascript. Assets that are not +text are copied byte for byte. This means `--base` is the only place a +deployment has to say where it lives. + ### CLI ``` diff --git a/downloader.ts b/downloader.ts index b0c08d7..4cfa3ba 100644 --- a/downloader.ts +++ b/downloader.ts @@ -11,7 +11,7 @@ import { fromHtml } from "hast-util-from-html"; import { toHtml } from "hast-util-to-html"; import { selectAll } from "hast-util-select"; import { useTaskBuffer } from "./task-buffer.ts"; -import { rebase } from "./rebase.ts"; +import { rebase, RebaseTextStream } from "./rebase.ts"; import { createApi } from "@effectionx/context-api"; export interface Downloader extends Operation { @@ -74,6 +74,18 @@ export const DownloadApi = createApi("@staticalize/download", { } } + // anchors are rewritten but not followed: the sitemap says what the + // site is made of, and downloading every link would crawl past it + let anchors = selectAll("a[href]", html); + + for (let anchor of anchors) { + let href = anchor.properties.href as string; + + if (href.startsWith(host.origin)) { + anchor.properties.href = rebase(new URL(href), base).href; + } + } + let assets = selectAll("[src]", html); for (let element of assets) { @@ -103,6 +115,28 @@ export const DownloadApi = createApi("@staticalize/download", { ok: true, bytes: new TextEncoder().encode(output).byteLength, }; + } else if (isTextual(response.headers.get("Content-Type"))) { + let bytes = 0; + let counted = new TransformStream({ + transform(chunk, controller) { + bytes += chunk.byteLength; + controller.enqueue(chunk); + }, + }); + + let destdir = dirname(path); + yield* until(ensureDir(destdir)); + yield* until( + Deno.writeFile( + path, + response.body! + .pipeThrough(new TextDecoderStream()) + .pipeThrough(new RebaseTextStream(host, base)) + .pipeThrough(new TextEncoderStream()) + .pipeThrough(counted), + ), + ); + return { ok: true, bytes }; } else { let size = Number(response.headers.get("Content-Length") ?? 0); let destdir = dirname(path); @@ -120,6 +154,25 @@ export const DownloadApi = createApi("@staticalize/download", { }, }); +/** + * Is this a content type whose body is text we can rewrite urls in? + * + * Everything else is streamed to disk byte for byte, so a binary body is never + * decoded and re-encoded. + */ +function isTextual(contentType: string | null): boolean { + if (!contentType) { + return false; + } + let type = contentType.split(";")[0].trim().toLowerCase(); + return type.startsWith("text/") || + type.endsWith("+json") || + type.endsWith("+xml") || + ["application/json", "application/xml", "application/javascript"].includes( + type, + ); +} + function* fetchWithRetry( url: string, signal: AbortSignal, diff --git a/rebase.ts b/rebase.ts index b12fb20..e7cc5e5 100644 --- a/rebase.ts +++ b/rebase.ts @@ -20,3 +20,43 @@ export function rebase(source: URL, base: URL): URL { url.pathname = `${base.pathname.replace(/\/$/, "")}${source.pathname}`; return url; } + +/** + * A TransformStream that replaces self-referencing urls with the public base. + * + * There is no document to walk in a text body, so unlike the html pass this is + * a substitution of the crawl origin rather than a rewrite of known + * url-bearing attributes. It runs over the response as it arrives, so the + * memory it holds is a chunk and change rather than the whole document. + */ +export class RebaseTextStream extends TransformStream { + #carry = ""; + + constructor(host: URL, base: URL) { + let needle = host.origin; + // the crawl origin stands for the root of the site, so it maps onto the base + // the same way any other url does. the result carries a trailing slash, which + // the urls in the body bring themselves. + let prefix = `${rebase(new URL(needle), base)}`.replace(/\/$/, ""); + // a match can straddle a chunk boundary, so hold back the longest partial + // one and let the next chunk complete it + let keep = needle.length - 1; + + super({ + transform: (chunk, controller) => { + let text = (this.#carry + chunk).replaceAll(needle, prefix); + if (text.length > keep) { + controller.enqueue(text.slice(0, text.length - keep)); + this.#carry = text.slice(text.length - keep); + } else { + this.#carry = text; + } + }, + flush: (controller) => { + if (this.#carry.length > 0) { + controller.enqueue(this.#carry); + } + }, + }); + } +} diff --git a/test/rebase.test.ts b/test/rebase.test.ts new file mode 100644 index 0000000..49513ba --- /dev/null +++ b/test/rebase.test.ts @@ -0,0 +1,110 @@ +import { expect } from "@std/expect"; +import { describe, it } from "@std/testing/bdd"; + +import { rebase, RebaseTextStream } from "../rebase.ts"; + +let host = new URL("http://localhost:8321"); + +async function rebased( + chunks: string[], + base: URL, + from: URL = host, +): Promise { + let stream = ReadableStream.from(chunks) + .pipeThrough(new RebaseTextStream(from, base)); + let out = ""; + for await (let chunk of stream) { + out += chunk; + } + return out; +} + +function cut(text: string, size: number): string[] { + let chunks: string[] = []; + for (let i = 0; i < text.length; i += size) { + chunks.push(text.slice(i, i + size)); + } + return chunks; +} + +describe("rebase", () => { + it("takes the protocol, host and port of the base", () => { + expect(rebase(new URL(`${host}about`), new URL("https://fs.com")).href) + .toEqual("https://fs.com/about"); + }); + + it("prefixes the path of the base", () => { + expect( + rebase(new URL(`${host}about`), new URL("https://fs.com/effection")).href, + ).toEqual("https://fs.com/effection/about"); + }); + + it("treats a trailing slash on the base as no path at all", () => { + expect(rebase(new URL(`${host}about`), new URL("https://fs.com/")).href) + .toEqual("https://fs.com/about"); + }); + + it("keeps the query and fragment of the source", () => { + expect( + rebase(new URL(`${host}search?q=hi#top`), new URL("https://fs.com")).href, + ).toEqual("https://fs.com/search?q=hi#top"); + }); +}); + +describe("RebaseTextStream", () => { + it("replaces every occurrence in a body", async () => { + await expect( + rebased( + [`[API]: ${host}api.md and ${host}b.md`], + new URL( + "https://fs.com", + ), + ), + ).resolves.toEqual("[API]: https://fs.com/api.md and https://fs.com/b.md"); + }); + + it("leaves urls on other hosts alone", async () => { + await expect( + rebased( + ["see https://google.com/a and http://localhost:9999/b"], + new URL( + "https://fs.com", + ), + ), + ).resolves.toEqual("see https://google.com/a and http://localhost:9999/b"); + }); + + it("carries the path of the base", async () => { + await expect( + rebased([`${host}api.md`], new URL("https://fs.com/effection")), + ).resolves.toEqual("https://fs.com/effection/api.md"); + }); + + it("matches urls split across chunk boundaries", async () => { + // the interesting cases are a match at the very start, at the very end, + // back to back with another, and bare with no path of its own + let body = + `${host}a.md then ${host}b.md${host}c.md and a bare ${host} to close`; + let want = body.replaceAll(host.origin, "https://fs.com"); + + for (let size of [1, 2, 3, 7, host.origin.length - 1, host.origin.length]) { + await expect(rebased(cut(body, size), new URL("https://fs.com"))) + .resolves.toEqual(want); + } + }); + + it("emits a body that ends mid-match as it stands", async () => { + // a truncated url is not a url, so it survives rather than disappearing + let partial = host.origin.slice(0, -3); + await expect(rebased( + cut(`ends with ${partial}`, 4), + new URL( + "https://fs.com", + ), + )).resolves.toEqual(`ends with ${partial}`); + }); + + it("passes an empty body through", async () => { + await expect(rebased([], new URL("https://fs.com"))).resolves.toEqual(""); + }); +}); diff --git a/test/staticalize.test.ts b/test/staticalize.test.ts index fed628b..d9c846e 100644 --- a/test/staticalize.test.ts +++ b/test/staticalize.test.ts @@ -324,6 +324,135 @@ describe("staticalize", () => { "https://frontside.com/effection/about", ); }); + + it("replaces self-referencing urls in anchors", async () => { + app.get( + "/", + (c) => c.html(`about`), + ) + .get("/about", (c) => c.html("

About

")) + .get(...sitemap(["/", "/about"])); + + await staticalize({ + base: new URL("https://fs.com"), + host, + dir: "test/dist", + }); + + await expect(content("test/dist/index.html")).resolves.toContain( + `about`, + ); + }); + + it("does not download pages that are only linked from an anchor", async () => { + app.get("/", (c) => c.html(`unlisted`)) + .get("/unlisted", (c) => c.html("

Unlisted

")) + .get(...sitemap(["/"])); + + await staticalize({ + base: new URL("https://fs.com"), + host, + dir: "test/dist", + }); + + await expect(exists("test/dist/unlisted/index.html")).resolves.toEqual( + false, + ); + }); + + it("replaces self-referencing urls in text bodies", async () => { + app.get("/llms.txt", (c) => + c.text(`[API]: ${host}api.md\n`, 200, { + "Content-Type": "text/plain; charset=utf-8", + })) + .get("/AGENTS.md", (c) => + c.text(`see ${host}api.md\n`, 200, { + "Content-Type": "text/markdown; charset=utf-8", + })) + .get("/feed.xml", (c) => + c.text(`${host}about`, 200, { + "Content-Type": "application/rss+xml", + })) + .get("/data.json", (c) => + c.text(`{"self":"${host}data.json"}`, 200, { + "Content-Type": "application/json", + })) + .get( + "/styles.css", + (c) => + c.text(`body { background: url(${host}bg.png); }`, 200, { + "Content-Type": "text/css", + }), + ) + .get( + ...sitemap([ + "/llms.txt", + "/AGENTS.md", + "/feed.xml", + "/data.json", + "/styles.css", + ]), + ); + + await staticalize({ + base: new URL("https://fs.com"), + host, + dir: "test/dist", + }); + + await expect(content("test/dist/llms.txt")).resolves.toEqual( + "[API]: https://fs.com/api.md\n", + ); + await expect(content("test/dist/AGENTS.md")).resolves.toEqual( + "see https://fs.com/api.md\n", + ); + await expect(content("test/dist/feed.xml")).resolves.toEqual( + "https://fs.com/about", + ); + await expect(content("test/dist/data.json")).resolves.toEqual( + `{"self":"https://fs.com/data.json"}`, + ); + await expect(content("test/dist/styles.css")).resolves.toEqual( + "body { background: url(https://fs.com/bg.png); }", + ); + }); + + it("carries the path of the base url into text bodies", async () => { + app.get("/llms.txt", (c) => + c.text(`[API]: ${host}api.md\n`, 200, { + "Content-Type": "text/plain; charset=utf-8", + })) + .get(...sitemap(["/llms.txt"])); + + await staticalize({ + base: new URL("https://frontside.com/effection"), + host, + dir: "test/dist", + }); + + await expect(content("test/dist/llms.txt")).resolves.toEqual( + "[API]: https://frontside.com/effection/api.md\n", + ); + }); + + it("streams bodies that are not text byte for byte", async () => { + // a png that would be corrupted by a decode/encode round trip + let png = new Uint8Array([0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a]); + + app.get("/logo.png", (c) => + c.body(png.buffer, 200, { + "Content-Type": "image/png", + })) + .get(...sitemap(["/logo.png"])); + + await staticalize({ + base: new URL("https://fs.com"), + host, + dir: "test/dist", + }); + + await expect(Deno.readFile("test/dist/logo.png")).resolves.toEqual(png); + }); }); async function content(path: string): Promise {