Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 7 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -25,6 +25,13 @@ This will read `https://localhost:800/sitemap.xml` and download the entire
website to the `dist/` directory in a format that can be served from a simple
file server running at `frontside.com`.

Wherever your site refers to itself with an absolute url, staticalize replaces
the `--site` origin with `--base`. It does this in html documents — `<a href>`,
`<link href>`, and any `src` or `content` attribute — and in textual bodies such
as `llms.txt`, markdown, feeds, json, css and javascript. Assets that are not
text are copied byte for byte. This means `--base` is the only place a
deployment has to say where it lives.

### CLI

```
Expand Down
55 changes: 54 additions & 1 deletion downloader.ts
Original file line number Diff line number Diff line change
Expand Up @@ -11,7 +11,7 @@ import { fromHtml } from "hast-util-from-html";
import { toHtml } from "hast-util-to-html";
import { selectAll } from "hast-util-select";
import { useTaskBuffer } from "./task-buffer.ts";
import { rebase } from "./rebase.ts";
import { rebase, RebaseTextStream } from "./rebase.ts";
import { createApi } from "@effectionx/context-api";

export interface Downloader extends Operation<void> {
Expand Down Expand Up @@ -74,6 +74,18 @@ export const DownloadApi = createApi("@staticalize/download", {
}
}

// anchors are rewritten but not followed: the sitemap says what the
// site is made of, and downloading every link would crawl past it
let anchors = selectAll("a[href]", html);

for (let anchor of anchors) {
let href = anchor.properties.href as string;

if (href.startsWith(host.origin)) {
anchor.properties.href = rebase(new URL(href), base).href;
}
}

let assets = selectAll("[src]", html);

for (let element of assets) {
Expand Down Expand Up @@ -103,6 +115,28 @@ export const DownloadApi = createApi("@staticalize/download", {
ok: true,
bytes: new TextEncoder().encode(output).byteLength,
};
} else if (isTextual(response.headers.get("Content-Type"))) {
let bytes = 0;
let counted = new TransformStream<Uint8Array, Uint8Array>({
transform(chunk, controller) {
bytes += chunk.byteLength;
controller.enqueue(chunk);
},
});

let destdir = dirname(path);
yield* until(ensureDir(destdir));
yield* until(
Deno.writeFile(
path,
response.body!
.pipeThrough(new TextDecoderStream())
.pipeThrough(new RebaseTextStream(host, base))
.pipeThrough(new TextEncoderStream())
.pipeThrough(counted),
),
);
return { ok: true, bytes };
} else {
let size = Number(response.headers.get("Content-Length") ?? 0);
let destdir = dirname(path);
Expand All @@ -120,6 +154,25 @@ export const DownloadApi = createApi("@staticalize/download", {
},
});

/**
* Is this a content type whose body is text we can rewrite urls in?
*
* Everything else is streamed to disk byte for byte, so a binary body is never
* decoded and re-encoded.
*/
function isTextual(contentType: string | null): boolean {
if (!contentType) {
return false;
}
let type = contentType.split(";")[0].trim().toLowerCase();
return type.startsWith("text/") ||
type.endsWith("+json") ||
type.endsWith("+xml") ||
["application/json", "application/xml", "application/javascript"].includes(
type,
);
}

function* fetchWithRetry(
url: string,
signal: AbortSignal,
Expand Down
40 changes: 40 additions & 0 deletions rebase.ts
Original file line number Diff line number Diff line change
Expand Up @@ -20,3 +20,43 @@ export function rebase(source: URL, base: URL): URL {
url.pathname = `${base.pathname.replace(/\/$/, "")}${source.pathname}`;
return url;
}

/**
* A TransformStream that replaces self-referencing urls with the public base.
*
* There is no document to walk in a text body, so unlike the html pass this is
* a substitution of the crawl origin rather than a rewrite of known
* url-bearing attributes. It runs over the response as it arrives, so the
* memory it holds is a chunk and change rather than the whole document.
*/
export class RebaseTextStream extends TransformStream<string, string> {
#carry = "";

constructor(host: URL, base: URL) {
let needle = host.origin;
// the crawl origin stands for the root of the site, so it maps onto the base
// the same way any other url does. the result carries a trailing slash, which
// the urls in the body bring themselves.
let prefix = `${rebase(new URL(needle), base)}`.replace(/\/$/, "");
// a match can straddle a chunk boundary, so hold back the longest partial
// one and let the next chunk complete it
let keep = needle.length - 1;

super({
transform: (chunk, controller) => {
let text = (this.#carry + chunk).replaceAll(needle, prefix);
if (text.length > keep) {
controller.enqueue(text.slice(0, text.length - keep));
this.#carry = text.slice(text.length - keep);
} else {
this.#carry = text;
}
},
flush: (controller) => {
if (this.#carry.length > 0) {
controller.enqueue(this.#carry);
}
},
});
}
}
110 changes: 110 additions & 0 deletions test/rebase.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,110 @@
import { expect } from "@std/expect";
import { describe, it } from "@std/testing/bdd";

import { rebase, RebaseTextStream } from "../rebase.ts";

let host = new URL("http://localhost:8321");

async function rebased(
chunks: string[],
base: URL,
from: URL = host,
): Promise<string> {
let stream = ReadableStream.from(chunks)
.pipeThrough(new RebaseTextStream(from, base));
let out = "";
for await (let chunk of stream) {
out += chunk;
}
return out;
}

function cut(text: string, size: number): string[] {
let chunks: string[] = [];
for (let i = 0; i < text.length; i += size) {
chunks.push(text.slice(i, i + size));
}
return chunks;
}

describe("rebase", () => {
it("takes the protocol, host and port of the base", () => {
expect(rebase(new URL(`${host}about`), new URL("https://fs.com")).href)
.toEqual("https://fs.com/about");
});

it("prefixes the path of the base", () => {
expect(
rebase(new URL(`${host}about`), new URL("https://fs.com/effection")).href,
).toEqual("https://fs.com/effection/about");
});

it("treats a trailing slash on the base as no path at all", () => {
expect(rebase(new URL(`${host}about`), new URL("https://fs.com/")).href)
.toEqual("https://fs.com/about");
});

it("keeps the query and fragment of the source", () => {
expect(
rebase(new URL(`${host}search?q=hi#top`), new URL("https://fs.com")).href,
).toEqual("https://fs.com/search?q=hi#top");
});
});

describe("RebaseTextStream", () => {
it("replaces every occurrence in a body", async () => {
await expect(
rebased(
[`[API]: ${host}api.md and ${host}b.md`],
new URL(
"https://fs.com",
),
),
).resolves.toEqual("[API]: https://fs.com/api.md and https://fs.com/b.md");
});

it("leaves urls on other hosts alone", async () => {
await expect(
rebased(
["see https://google.com/a and http://localhost:9999/b"],
new URL(
"https://fs.com",
),
),
).resolves.toEqual("see https://google.com/a and http://localhost:9999/b");
});

it("carries the path of the base", async () => {
await expect(
rebased([`${host}api.md`], new URL("https://fs.com/effection")),
).resolves.toEqual("https://fs.com/effection/api.md");
});

it("matches urls split across chunk boundaries", async () => {
// the interesting cases are a match at the very start, at the very end,
// back to back with another, and bare with no path of its own
let body =
`${host}a.md then ${host}b.md${host}c.md and a bare ${host} to close`;
let want = body.replaceAll(host.origin, "https://fs.com");

for (let size of [1, 2, 3, 7, host.origin.length - 1, host.origin.length]) {
await expect(rebased(cut(body, size), new URL("https://fs.com")))
.resolves.toEqual(want);
}
});

it("emits a body that ends mid-match as it stands", async () => {
// a truncated url is not a url, so it survives rather than disappearing
let partial = host.origin.slice(0, -3);
await expect(rebased(
cut(`ends with ${partial}`, 4),
new URL(
"https://fs.com",
),
)).resolves.toEqual(`ends with ${partial}`);
});

it("passes an empty body through", async () => {
await expect(rebased([], new URL("https://fs.com"))).resolves.toEqual("");
});
});
129 changes: 129 additions & 0 deletions test/staticalize.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -324,6 +324,135 @@ describe("staticalize", () => {
"<loc>https://frontside.com/effection/about</loc>",
);
});

it("replaces self-referencing urls in anchors", async () => {
app.get(
"/",
(c) => c.html(`<a href="${host}about">about</a>`),
)
.get("/about", (c) => c.html("<h1>About</h1>"))
.get(...sitemap(["/", "/about"]));

await staticalize({
base: new URL("https://fs.com"),
host,
dir: "test/dist",
});

await expect(content("test/dist/index.html")).resolves.toContain(
`<a href="https://fs.com/about">about</a>`,
);
});

it("does not download pages that are only linked from an anchor", async () => {
app.get("/", (c) => c.html(`<a href="${host}unlisted">unlisted</a>`))
.get("/unlisted", (c) => c.html("<h1>Unlisted</h1>"))
.get(...sitemap(["/"]));

await staticalize({
base: new URL("https://fs.com"),
host,
dir: "test/dist",
});

await expect(exists("test/dist/unlisted/index.html")).resolves.toEqual(
false,
);
});

it("replaces self-referencing urls in text bodies", async () => {
app.get("/llms.txt", (c) =>
c.text(`[API]: ${host}api.md\n`, 200, {
"Content-Type": "text/plain; charset=utf-8",
}))
.get("/AGENTS.md", (c) =>
c.text(`see ${host}api.md\n`, 200, {
"Content-Type": "text/markdown; charset=utf-8",
}))
.get("/feed.xml", (c) =>
c.text(`<link>${host}about</link>`, 200, {
"Content-Type": "application/rss+xml",
}))
.get("/data.json", (c) =>
c.text(`{"self":"${host}data.json"}`, 200, {
"Content-Type": "application/json",
}))
.get(
"/styles.css",
(c) =>
c.text(`body { background: url(${host}bg.png); }`, 200, {
"Content-Type": "text/css",
}),
)
.get(
...sitemap([
"/llms.txt",
"/AGENTS.md",
"/feed.xml",
"/data.json",
"/styles.css",
]),
);

await staticalize({
base: new URL("https://fs.com"),
host,
dir: "test/dist",
});

await expect(content("test/dist/llms.txt")).resolves.toEqual(
"[API]: https://fs.com/api.md\n",
);
await expect(content("test/dist/AGENTS.md")).resolves.toEqual(
"see https://fs.com/api.md\n",
);
await expect(content("test/dist/feed.xml")).resolves.toEqual(
"<link>https://fs.com/about</link>",
);
await expect(content("test/dist/data.json")).resolves.toEqual(
`{"self":"https://fs.com/data.json"}`,
);
await expect(content("test/dist/styles.css")).resolves.toEqual(
"body { background: url(https://fs.com/bg.png); }",
);
});

it("carries the path of the base url into text bodies", async () => {
app.get("/llms.txt", (c) =>
c.text(`[API]: ${host}api.md\n`, 200, {
"Content-Type": "text/plain; charset=utf-8",
}))
.get(...sitemap(["/llms.txt"]));

await staticalize({
base: new URL("https://frontside.com/effection"),
host,
dir: "test/dist",
});

await expect(content("test/dist/llms.txt")).resolves.toEqual(
"[API]: https://frontside.com/effection/api.md\n",
);
});

it("streams bodies that are not text byte for byte", async () => {
// a png that would be corrupted by a decode/encode round trip
let png = new Uint8Array([0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a]);

app.get("/logo.png", (c) =>
c.body(png.buffer, 200, {
"Content-Type": "image/png",
}))
.get(...sitemap(["/logo.png"]));

await staticalize({
base: new URL("https://fs.com"),
host,
dir: "test/dist",
});

await expect(Deno.readFile("test/dist/logo.png")).resolves.toEqual(png);
});
});

async function content(path: string): Promise<string> {
Expand Down
Loading