Repository navigation
Expand file tree
/
Copy pathDockerfile
More file actions
259 lines (251 loc) · 18.3 KB
/
Copy pathDockerfile
File metadata and controls
259 lines (251 loc) · 18.3 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
# Runtime-only. The binaries are compiled OUTSIDE docker and copied in from `target/${PROFILE}/`:
#
# cargo build --release --locked && docker build --target server . # or agent, gateway
#
# They used to be compiled inside (cargo-chef, `type=gha` cache in `mode=max`). That design cached
# the dependency cook but never the workspace crates — those lived in the layer that also held the
# source, so every commit rebuilt them from scratch: 342 s of `cargo build --release` on EVERY run
# plus 114 s exporting the layer cache, before a single image was pushed. CI now compiles once on
# the runner under Swatinem/rust-cache (incremental across commits) and this file is only the
# apt-get and the COPY. See .github/workflows/image.yml.
#
# The compile MUST link against a glibc no newer than bookworm's 2.36, because these stages are
# bookworm-slim; CI builds inside `rust:1-bookworm` and checks the binary's GLIBC_ requirement.
# Building on a newer host (ubuntu 24.04, glibc 2.39) yields an image that dies at exec.
#
# `PROFILE` names the target dir the binaries come from: `release` (default) or `dev-image` for the
# deploy/k3s/dev-push.sh loop. Only the five kloudlite binaries make it into the context — see
# .dockerignore — so a fat `target/` costs nothing to send.
FROM debian:bookworm-slim@sha256:abd67ffcfa541b485a3dff59865ab629aa048a6c613e639d36e7456b0b229241 AS server
# openssh-client: the server shells out to ssh-keygen to generate its host key on first start.
# git: the merge worker performs merges by running it (see crates/pulls/src/merge_worker.rs) —
# bookworm ships 2.39, past the 2.38 that `merge-tree --write-tree` needs. One image serves all
# three processes, so git also lands on the srv and api pods, where nothing runs it; a few MB of
# unused binary is cheaper than a second image to keep in step with this one.
# curl: the srv preStop hook POSTs /peer/v1/drain to its own peer port to hand ownership over
# before the pod goes (deploy/kloudlite.yaml). Nothing in the processes shells out to it.
RUN apt-get update && apt-get install -y --no-install-recommends ca-certificates openssh-client git curl \
&& rm -rf /var/lib/apt/lists/*
# All three binaries. One image, three processes: the git server, the api server
# and the merge worker are built from the same source and deployed separately, so
# a shared image keeps them on the same commit without coupling their lifecycles.
# Each Deployment picks its process with `command`, so a binary missing here is a
# CrashLoopBackOff there, not a build error.
ARG PROFILE=release
COPY target/${PROFILE}/kloudlite /usr/local/bin/kloudlite
COPY target/${PROFILE}/kloudlite-api /usr/local/bin/kloudlite-api
COPY target/${PROFILE}/kloudlite-worker /usr/local/bin/kloudlite-worker
# Not root. Nothing here needs a capability: the listeners bind 8080/2222/8081/8082, the host
# key lives in a mounted Secret in the cluster, and the pack cache is a directory. The two
# directories the binaries write are created and owned here so a plain `docker run` (no
# mounts) works; in the cluster both are mounts and `fsGroup` on the pod makes them writable.
# uid 1001 matches web/Dockerfile so one securityContext convention serves both images.
RUN useradd --system --uid 1001 --user-group --no-create-home --shell /usr/sbin/nologin kloudlite \
&& mkdir -p /var/cache/kloudlite /var/lib/kloudlite \
&& chown kloudlite:kloudlite /var/cache/kloudlite /var/lib/kloudlite
ENV KLOUDLITE_CACHE_DIR=/var/cache/kloudlite KLOUDLITE_HOST_KEY=/var/lib/kloudlite/host_key
USER kloudlite
EXPOSE 8080 2222
ENTRYPOINT ["kloudlite"]
CMD ["serve"]
# The node controller. A separate IMAGE, not a fourth binary in the server one: this runs as root
# with btrfs-progs and the host pool mounted, and shipping root's toolchain to the three processes
# that must never have it is exactly what the split prevents.
FROM debian:bookworm-slim@sha256:abd67ffcfa541b485a3dff59865ab629aa048a6c613e639d36e7456b0b229241 AS agent
# btrfs-progs: every storage operation shells out to it.
# util-linux: losetup/mount for the block-layer restore path.
# ca-certificates: the registry client and Azure blob store speak TLS.
# git: a workspace can be seeded from a platform repository, which the controller clones into the
# fresh subvolume (`VolumeSource::GitRepo`).
# openssh-client: ssh-keygen makes each workspace's SSH host key.
# nfs-common: `mount -t nfs` is not built into mount(8) — it execs /sbin/mount.nfs, which ships
# here. Without it the agent's shared-home mount fails with "bad option; ... you might need a
# /sbin/mount.<type> helper program" and, because that mount is fail-closed, the agent refuses to
# start at all rather than serving anyone an empty home.
# netbase: /etc/protocols and /etc/services. `--no-install-recommends` leaves them out, and
# without /etc/protocols mount.nfs cannot resolve `proto=tcp` ("Failed to find 'tcp' protocol"),
# silently abandons the v3 negotiation it was told to use, and asks the server for v4 instead —
# which ZeroFS rejects as "Invalid NFS Version number 4 != 3" and the client reports, uselessly,
# as "Protocol not supported". Two lost deploys came from that chain; keep this package.
RUN apt-get update && apt-get install -y --no-install-recommends \
btrfs-progs util-linux ca-certificates git openssh-client nfs-common netbase \
&& rm -rf /var/lib/apt/lists/*
ARG PROFILE=release
COPY target/${PROFILE}/kloudlite-agent /usr/local/bin/kloudlite-agent
# Root, deliberately and unlike the server image: btrfs subvolume operations on the host pool are
# not something a capability set can be narrowed to.
ENTRYPOINT ["kloudlite-agent"]
# The SSH gateway. Its own image rather than a fourth binary in the server one: this pod runs with
# NET_BIND_SERVICE to hold hostPort 443 on a pool node, and that capability has no business on the
# git server's pods.
FROM debian:bookworm-slim@sha256:abd67ffcfa541b485a3dff59865ab629aa048a6c613e639d36e7456b0b229241 AS gateway
# ca-certificates only: the gateway talks to the kube API server over TLS and to nothing else.
# libcap2-bin is build-time only, for the setcap below.
RUN apt-get update && apt-get install -y --no-install-recommends ca-certificates libcap2-bin \
&& rm -rf /var/lib/apt/lists/*
ARG PROFILE=release
COPY target/${PROFILE}/kloudlite-gateway /usr/local/bin/kloudlite-gateway
# The FILE capability is what actually lets uid 1001 bind 443, and it is not optional.
# `capabilities.add: [NET_BIND_SERVICE]` in the pod spec sets the container's BOUNDING and
# permitted sets — but execve empties the permitted set of a non-root process unless the binary
# itself carries the capability, and Kubernetes has no way to request ambient capabilities. So the
# pod grant alone yields EACCES on 443; the pod grant plus this file capability is what works, and
# neither half is sufficient on its own (the bounding set must still permit it).
RUN setcap cap_net_bind_service=+ep /usr/local/bin/kloudlite-gateway \
&& apt-get purge -y libcap2-bin && apt-get autoremove -y
# uid 1001 as in the server image: the binary writes no files and needs no other privilege.
RUN useradd --system --uid 1001 --user-group --no-create-home --shell /usr/sbin/nologin kloudlite
USER kloudlite
EXPOSE 443 8080
ENTRYPOINT ["kloudlite-gateway"]
# The cluster controller. No capability, no hostPath, no secret: the API server is its only
# dependency, and its one listener is the health route on 8080 — so none of the gateway's setcap
# dance applies here.
FROM debian:bookworm-slim@sha256:abd67ffcfa541b485a3dff59865ab629aa048a6c613e639d36e7456b0b229241 AS controller
# ca-certificates only: the controller talks TLS to the kube API server and to nothing else.
RUN apt-get update && apt-get install -y --no-install-recommends ca-certificates \
&& rm -rf /var/lib/apt/lists/*
ARG PROFILE=release
COPY target/${PROFILE}/kloudlite-controller /usr/local/bin/kloudlite-controller
RUN useradd --system --uid 1001 --user-group --no-create-home --shell /usr/sbin/nologin kloudlite
USER kloudlite
EXPOSE 8080
ENTRYPOINT ["kloudlite-controller"]
# The build gate. Its own image for the same reason the gateway has one: a different pod, a
# different ServiceAccount, and no reason for the git server's pods to carry either binary.
FROM debian:bookworm-slim@sha256:abd67ffcfa541b485a3dff59865ab629aa048a6c613e639d36e7456b0b229241 AS builder-gate
# ca-certificates only: the gate talks TLS to the kube API server and to the api tier's public
# URL, and plain TCP to buildkit.
RUN apt-get update && apt-get install -y --no-install-recommends ca-certificates \
&& rm -rf /var/lib/apt/lists/*
ARG PROFILE=release
COPY target/${PROFILE}/kloudlite-builder-gate /usr/local/bin/kloudlite-builder-gate
# No file capability, unlike the gateway: 1234 and 8080 are both unprivileged.
RUN useradd --system --uid 1001 --user-group --no-create-home --shell /usr/sbin/nologin kloudlite
USER kloudlite
EXPOSE 1234 8080
ENTRYPOINT ["kloudlite-builder-gate"]
# The intercept proxy. `scratch`, not bookworm-slim like its neighbours: the binary is a static
# musl build that opens no file, resolves DNS through the kernel's own getaddrinfo-free path in
# std's resolver, and talks to nothing but two TCP sockets — so there is no libc, no CA bundle and
# no shell for it to need. `USER` is not set here: the pod spec runs it as uid 1000 with a
# read-only root, and a scratch image has no /etc/passwd to name an account in.
FROM scratch AS intercept-proxy
ARG PROFILE=release
COPY target/x86_64-unknown-linux-musl/${PROFILE}/kloudlite-intercept-proxy /kloudlite-intercept-proxy
ENTRYPOINT ["/kloudlite-intercept-proxy"]
# The default workspace image: what `ws-{id}` runs when a workspace names no image of its own.
# debian:bookworm-slim, the same pinned base as every other stage here, and GLIBC is the whole
# point (2026-09-17): it used to be alpine, and every tool a person actually uses comes from the
# Nix profile, which is glibc-linked. npm's native-binding loaders (rolldown, rollup, esbuild's
# optional peers) detect the libc by running `/usr/bin/ldd` and read "musl" from a musl base, then
# install and dlopen the musl binding — which a glibc `node` cannot load: "Cannot find native
# binding". A musl userland under a glibc toolchain is a lie the whole npm ecosystem believes.
# Stock bookworm plus exactly what the platform itself needs and cannot get from Nix:
# - libstdc++6/libgcc-s1: VS Code Remote-SSH's server dlopens both, and bookworm-slim carries
# neither by default; without them every connect downloads the server and dies relocating.
# Nix cannot supply them to a foreign binary.
# - ca-certificates: the alpine base bundled them; a debian slim does not, and npm, `kl` and
# the credential helper all speak TLS.
# - node + npm from the official node image's `/usr/local`, not bookworm's `nodejs` (18):
# graft below declares `engines.node >= 20`. Same bookworm userland, built against it.
# - the docker CLI and its buildx plugin, as static binaries from their own upstream images —
# debian has no `docker-cli` package, only `docker.io`, which drags in the daemon.
# - the `kl` account to log in as, and sshd's `sshd` account and chroot dir, which the alpine
# base shipped and debian only gets with `openssh-server` — a package whose only useful part
# here would be those two, since sshd itself comes from the Nix profile at run time.
# `useradd -p '*'` writes "no password", NOT the `!` that sshd reads as "locked" and refuses
# even a valid key for. The login shell is the Nix profile's zsh, mounted at run time —
# useradd does not check that the path exists yet.
# - the greeting.
# Everything a person actually uses (git, zsh, fish, starship, …) comes from the Nix profile the
# agent builds per workspace and mounts read-only, so this image stays stock apart from the above.
# Runtime steps that depend on mounts (chown of the volume, seeding rc files, exec sshd) live in
# `k8s::prelude`, not here.
# Pinned by digest like every other stage (2026-09-12 review #69): a tag is a pointer Docker Hub
# can move, and this is the image a person's whole working day runs inside.
FROM debian:bookworm-slim@sha256:abd67ffcfa541b485a3dff59865ab629aa048a6c613e639d36e7456b0b229241 AS workspace
ARG PROFILE=release
COPY --from=node:22-bookworm-slim@sha256:83f487e0a63425e5b4d146fb5e5be574bcbe1b7b843d3ebafdd95eaf7767a7e5 /usr/local/ /usr/local/
COPY --from=docker:28-cli@sha256:625d9431a9f54c5a2bc90f24f0e1c3d55b1349fd857dd85035f98c2c9acbdd4d /usr/local/bin/docker /usr/bin/docker
COPY --from=docker/buildx-bin:0.20.1@sha256:ead27bfcde6308a757b4a5a4a931937363c1fa0091f7e2994b9114521853cf69 /buildx /usr/libexec/docker/cli-plugins/docker-buildx
RUN apt-get update && apt-get install -y --no-install-recommends ca-certificates libstdc++6 libgcc-s1 \
&& rm -rf /var/lib/apt/lists/* \
&& mkdir -p /var/empty \
&& groupadd -g 1000 kl \
&& useradd -u 1000 -g 1000 -m -d /home/kl -s /nix/profile/current/bin/zsh -p '*' kl \
&& useradd -r -d /var/empty -s /usr/sbin/nologin -p '*' sshd \
&& printf '%s\n' 'Kloudlite workspace — you are kl (no root, no sudo).' > /etc/motd
# The docker CLI's credential-helper protocol resolves `credHelpers.<host>: kl` to
# `docker-credential-kl`, which reads the registry token `login_env`/`user_key_secret` keep fresh
# on disk — no `docker login`, nothing long-lived, rotation is just the next Secret projection.
COPY deploy/workspace-image/docker-credential-kl /usr/local/bin/docker-credential-kl
RUN chmod 0755 /usr/local/bin/docker-credential-kl
# `kl` is the workspace CLI (build, push). Still the musl build, unchanged by the move to glibc:
# a static musl binary runs anywhere, and it is the same artifact the bench image carries.
COPY target/x86_64-unknown-linux-musl/${PROFILE}/kl /usr/local/bin/kl
RUN chmod 0755 /usr/local/bin/kl
# `/etc/profile.d`, not the seeded rc files under `k8s::prelude`: those are copied into the
# person's home once and their own edits then survive forever, which is wrong for a config that
# must track `BUILDKIT_HOST`/`KL_REGISTRY_HOST` on every login.
COPY deploy/workspace-image/kl-build.sh /etc/profile.d/kl-build.sh
# Derived state the platform places inside a workspace directory, ignored by git GLOBALLY —
# git's default `core.excludesFile`. `prelude` appends this block to the person's own file once;
# a per-repository `.gitignore` line would be a diff they did not ask for, in every repository.
COPY deploy/workspace-image/gitignore-global /etc/kloudlite/gitignore-global
# graft, for `kl ide serve`: the graph of every symbol and call edge an outside agent asks about,
# proxied from `graft mcp` and kept fresh by the server. Pinned; the image is the version.
# tree-sitter's grammars are native modules: node-gyp needs python3, make and g++ for the install
# and nothing after it, so the toolchain is added and removed in the one layer.
RUN apt-get update && apt-get install -y --no-install-recommends python3 make g++ \
&& npm install -g @nanonets/graft@0.18.0 \
&& npm cache clean --force \
&& apt-get purge -y python3 make g++ && apt-get autoremove -y \
&& rm -rf /var/lib/apt/lists/*
ENV DO_NOT_TRACK=1
# No `USER kl`, unlike every other stage here, and that is the design rather than an omission
# (2026-09-12 review #69): this image's entrypoint is `k8s::prelude`, which chowns the mounted
# volume and the home, seeds rc files and writes the SSH host key before it execs sshd — all of
# which need root. sshd is what drops to `kl`, for the only session a person ever gets, and the
# pod is confined by gVisor plus the workspace admission policy rather than by this line.
# The SLO probe. Its own image because it is the only one that carries a toolbox — git, ssh,
# crane, kubectl, dig, openssl — and shipping that to the three server processes would hand a
# compromised request handler everything it needs to talk to the cluster.
FROM debian:bookworm-slim@sha256:abd67ffcfa541b485a3dff59865ab629aa048a6c613e639d36e7456b0b229241 AS slo
# git + openssh-client: stage 2 pushes and clones over both transports, with a real client, because
# a probe that used our own library would pass on a bug only a real client trips.
# curl is the build-time tool fetch below; the edge stage dials the origin with reqwest.
# openssl + bind9-dnsutils: `edge.cert` reads the served certificate and `edge.dns` resolves the
# hostnames without trusting the pod's resolver cache.
# bash: the edge stage opens `/dev/tcp/host/port` to prove a listener answers, which is a bash
# builtin — debian's default /bin/sh is dash and has no such thing. Explicit rather than relying
# on bookworm-slim shipping it, because a base-image slim-down would break the edge stage silently.
RUN apt-get update && apt-get install -y --no-install-recommends \
bash ca-certificates git openssh-client curl openssl bind9-dnsutils \
&& rm -rf /var/lib/apt/lists/*
# crane and kubectl are pinned by CONTENT, not by tag: both are fetched from a release URL a
# third party controls, and a moved tag would silently change what the probe runs. A checksum
# mismatch fails the build, which is the only place it can be caught.
ARG CRANE_VERSION=v0.20.3
ARG CRANE_SHA256=36c67a932f489b3f2724b64af90b599a8ef2aa7b004872597373c0ad694dc059
ARG KUBECTL_VERSION=v1.31.5
ARG KUBECTL_SHA256=fbecbfd375b3686002c2e81d51c390172f5ffba3d6b47920d55342cb03f557af
RUN set -eux; \
curl -fsSL -o /tmp/crane.tgz "https://github.com/google/go-containerregistry/releases/download/${CRANE_VERSION}/go-containerregistry_Linux_x86_64.tar.gz"; \
echo "${CRANE_SHA256} /tmp/crane.tgz" | sha256sum -c -; \
tar -xzf /tmp/crane.tgz -C /usr/local/bin crane; \
curl -fsSL -o /usr/local/bin/kubectl "https://dl.k8s.io/release/${KUBECTL_VERSION}/bin/linux/amd64/kubectl"; \
echo "${KUBECTL_SHA256} /usr/local/bin/kubectl" | sha256sum -c -; \
chmod +x /usr/local/bin/crane /usr/local/bin/kubectl; \
rm -f /tmp/crane.tgz
ARG PROFILE=release
COPY target/${PROFILE}/kloudlite-slo /usr/local/bin/kloudlite-slo
# `kl-connect` is the laptop CLI, built by the same `cargo build`: stage 1's `id.cli.flow` and stage 5's
# tunnel checks exercise the CLI a person actually runs, not a reimplementation of it.
COPY target/${PROFILE}/kl-connect /usr/local/bin/kl-connect
# uid 1001 as everywhere else. No home directory: the pod's root filesystem is read-only and
# everything git, ssh and crane write goes under the /tmp emptyDir (HOME is set to it in the
# CronJob), so a home baked in here would only be a read-only trap.
RUN useradd --system --uid 1001 --user-group --no-create-home --shell /usr/sbin/nologin kloudlite
USER kloudlite
ENTRYPOINT ["kloudlite-slo"]