summaryrefslogtreecommitdiff
path: root/workers/apex/src
diff options
context:
space:
mode:
authorYuriy Andamasov <yuriy@vyos.io>2026-09-25 17:11:19 +0200
committerGitHub <noreply@github.com>2026-09-25 16:11:19 +0100
commita783e56b774795c1f4bd8a6088dfcce96a307c5f (patch)
treedcd0c88ff02969351840c8222ea97c72d9258e18 /workers/apex/src
parent4c8d1d1a0f23e2106a97f4b672607121ff03a091 (diff)
downloadvyos-documentation-a783e56b774795c1f4bd8a6088dfcce96a307c5f.tar.gz
vyos-documentation-a783e56b774795c1f4bd8a6088dfcce96a307c5f.zip
ci: port the Cloudflare Workers docs pipeline to circinus (slug 1.5) (#2208)
* ci: port Cloudflare Workers docs pipeline files to circinus (verbatim) Copies the branch-agnostic half of the docs.vyos.io Cloudflare Workers pipeline from `rolling` at 8cb568bf, byte-identical: - .github/workflows/docs-build.yml - workers/ (entire tree) - scripts/docs_gates/ - docker/im-convert.sh - docs/_static/js/version-picker.js, js/pagefind-wrapper.js, css/version-picker.css (new files, no circinus counterpart) - docs/_templates/breadcrumbs.html, searchbox.html (new files) docs-build.yml already triggers on push to [rolling, circinus, sagitta] and resolves `circinus` -> worker vyos-docs-v15-en / slug 1.5 from workers/matrix.json; the files simply did not exist on this branch, so slug 1.5 still serves the bootstrap placeholder. workers/versions.json + workers/matrix.json are deliberately identical across all three branches and must be kept in sync. Advances: IS-572 * ci: wire circinus docs build into the Cloudflare Workers pipeline Hand-merges the CF-specific hunks onto circinus's own docker/Dockerfile and docs/conf.py rather than clobbering them with rolling's versions — circinus keeps its own content-driven history in both files. docker/Dockerfile: - imagemagick + librsvg2-bin (sphinx.ext.imgconverter backend) and poppler-utils (pdfinfo, used by docs-build.yml's PDF page-count completeness check), installed --no-install-recommends - install docker/im-convert.sh as /usr/local/bin/im-convert docs/conf.py: - enable sphinx.ext.imgconverter + image_converter = 'im-convert' so the LaTeX/PDF builder stops silently dropping .webp/.svg images - register js/version-picker.js + css/version-picker.css unconditionally (degrades silently on ReadTheDocs) - _vyos_cf_build gate off the raw DOCS_VERSION_SLUG env var; only CF builds load js/pagefind-wrapper.js, and html_context['vyos_cf_build'] lets _templates/searchbox.html fall back to the stock Sphinx searchbox via the "!" bang-include on RTD The RTD path stays the default in both files: circinus continues building on ReadTheDocs until RTD sunset, and every CF feature activates only when DOCS_VERSION_SLUG is present. .readthedocs.yml is untouched. circinus keeps its own version/release/html_title/source_suffix and its hardcoded html_baseurl (its CF slug is also `1.5`, so rolling's DOCS_VERSION_SLUG/READTHEDOCS_VERSION resolution block is a no-op here and was deliberately not ported). Advances: IS-572 * ci: IS-572: re-sync ported Cloudflare Workers pipeline files with rolling The category-1 files in this port are byte-identical copies from `rolling`. `rolling` has since moved: [vyos-documentation#2209](https://github.com/vyos/vyos-documentation/pull/2209) merged as `3a1c6c30`, thirteen rounds of hardening on exactly these files. Re-take all 14 category-1 paths from `origin/rolling` via `git checkout origin/rolling -- <paths>`, so byte-identity holds by construction rather than by hand-editing: .github/workflows/docs-build.yml scripts/docs_gates/{gates,parity,smoke,test_gates,test_parity,test_smoke}.py workers/.gitignore workers/apex/src/{index,special,uagate}.ts workers/apex/test/{router,uagate}.test.ts workers/apex/ua-policy.json Thirteen of the fourteen carry [vyos-documentation#2209](https://github.com/vyos/vyos-documentation/pull/2209) exactly — the pre-change tree was byte-identical to `3a1c6c30^` for those paths. `workers/.gitignore` additionally picks up the one-line `test-results/` entry from [vyos-documentation#2212](https://github.com/vyos/vyos-documentation/pull/2212); inert on circinus, since only the deliberately-unported `apex-deploy.yml` writes that directory. Deliberate exclusions are unchanged: `docs-canary-qa.yml` (cron runs on the default branch only, so it is not ported even though [vyos-documentation#2209](https://github.com/vyos/vyos-documentation/pull/2209) touched it on `rolling`), `apex-deploy.yml`, and the `docs-preview-*` workflows. `docs/conf.py` stays hand-merged and circinus-specific, with its ReadTheDocs fallback intact. 🤖 Generated by [robots](https://vyos.io)
Diffstat (limited to 'workers/apex/src')
-rw-r--r--workers/apex/src/dispatch.ts15
-rw-r--r--workers/apex/src/index.ts561
-rw-r--r--workers/apex/src/manifest.ts55
-rw-r--r--workers/apex/src/redirects.ts38
-rw-r--r--workers/apex/src/special.ts60
-rw-r--r--workers/apex/src/uagate.ts48
6 files changed, 777 insertions, 0 deletions
diff --git a/workers/apex/src/dispatch.ts b/workers/apex/src/dispatch.ts
new file mode 100644
index 00000000..9ad1f15d
--- /dev/null
+++ b/workers/apex/src/dispatch.ts
@@ -0,0 +1,15 @@
+export function resolveVersion(
+ pathname: string,
+ dispatch: Map<string, string>,
+): { slug: string; binding: string } | null {
+ const m = pathname.match(/^\/en\/([^/]+)\//);
+ if (!m) return null;
+ const binding = dispatch.get(m[1]);
+ return binding ? { slug: m[1], binding } : null;
+}
+
+export function bindingGuard(env: Record<string, unknown>, binding: string): Fetcher | null {
+ const b = env[binding];
+ if (b && typeof (b as Fetcher).fetch === "function") return b as Fetcher;
+ return null;
+}
diff --git a/workers/apex/src/index.ts b/workers/apex/src/index.ts
new file mode 100644
index 00000000..f97b4bc5
--- /dev/null
+++ b/workers/apex/src/index.ts
@@ -0,0 +1,561 @@
+import { loadManifest, buildDispatch } from "./manifest";
+import { resolveVersion, bindingGuard } from "./dispatch";
+import { redirectFor } from "./redirects";
+import { specialPathFor } from "./special";
+import { uaVerdict } from "./uagate";
+import policy from "../ua-policy.json";
+
+export interface ApexEnv extends Record<string, unknown> {
+ ASSETS: Fetcher;
+ APEX_BUILD_SHA: string;
+ DOCS_ENV: "production" | "canary";
+ DOCS_KB?: Fetcher;
+ // §5 apex PDF fallback — R2 bucket holding oversized legacy PDFs excluded from the
+ // content Worker's own asset tree (currently just the 1.3 PDF). Optional so the
+ // binding-guard path (503, not a crash) exercises on envs that omit it.
+ DOCS_PDFS?: R2Bucket;
+}
+
+const manifest = loadManifest();
+const dispatch = buildDispatch(manifest);
+
+// Security headers only — safe on content pass-through (never touches
+// Cache-Control or X-Docs-Build, which the content Worker owns).
+function securityHeaders(resp: Response): Response {
+ const out = new Response(resp.body, resp);
+ out.headers.set("X-Content-Type-Options", "nosniff");
+ out.headers.set("Referrer-Policy", "strict-origin-when-cross-origin");
+ out.headers.set("Content-Security-Policy-Report-Only", "default-src 'self'; img-src 'self' data:; style-src 'self' 'unsafe-inline'; script-src 'self'");
+ return out;
+}
+
+const DEFAULT_CACHE_CLASS = "public, max-age=0, s-maxage=300, must-revalidate";
+
+function apexHeaders(resp: Response, env: ApexEnv, cacheClass: string = DEFAULT_CACHE_CLASS): Response {
+ const out = securityHeaders(resp);
+ out.headers.set("X-Apex-Build", env.APEX_BUILD_SHA);
+ // §3.3 cache contract applies to apex-owned responses too. Error responses (4xx/5xx)
+ // must never carry the s-maxage cache class — the cache key excludes User-Agent, so a
+ // cached UA-gate 403 or themed 404/503 would poison the edge for every visitor for the
+ // full s-maxage window. Mirrors the branch worker's withDocsHeaders() precedence.
+ // `cacheClass` lets a specific caller (e.g. the §5 PDF R2 fallback) apply a
+ // differently-classed success cache-control; canary/error still always win.
+ out.headers.set(
+ "Cache-Control",
+ env.DOCS_ENV === "canary" || out.status >= 400 ? "no-store" : cacheClass,
+ );
+ return out;
+}
+
+// R2's `R2Range` is a three-shape union — `{offset, length?}`, `{length}` (offset implicitly 0)
+// and `{suffix}` (the trailing N bytes) — so `"offset" in range` is NOT a safe way to read it:
+// the two offset-less shapes would fall through to the full-object 200 branch and be served
+// with a Content-Length claiming the whole object while the body held only a slice. workerd is
+// observed to normalize every shape to `{offset, length}` before it reaches us, but the type
+// admits the others, so resolve all three to concrete byte bounds, clamped to the object size.
+// Exported for direct unit testing.
+export function resolveRange(
+ range: { offset?: number; length?: number; suffix?: number },
+ size: number,
+): { start: number; length: number } {
+ if (typeof range.suffix === "number") {
+ const suffix = Math.min(Math.max(range.suffix, 0), size); // a suffix past the start is the whole object
+ return { start: size - suffix, length: suffix };
+ }
+ const start = Math.min(Math.max(range.offset ?? 0, 0), size);
+ // The trailing Math.max(_, 0) keeps the documented "clamped to the object size" contract
+ // total: without it a negative `range.length` would pass straight through Math.min and
+ // yield a negative length (and so a negative Content-Length). A real R2 binding cannot
+ // produce that — see classifyRangeHeader's note on observed R2 behaviour — but this
+ // function is exported and unit-tested as a standalone utility over the R2Range union, so
+ // it should not have a documented invariant its own signature can violate. Deliberately
+ // NOT guarding non-finite inputs: NaN bounds are unreachable from the binding and the
+ // guard would be untestable-in-anger dead weight.
+ const length = Math.max(Math.min(range.length ?? size - start, size - start), 0);
+ return { start, length };
+}
+
+// A single `bytes=` range-spec. Anything with a comma is a multi-range and deliberately
+// fails to match. The whitespace class is `\s`, which is DELIBERATELY wider than the ` `
+// (ASCII space) that R2's own parser accepts — see classifyRangeHeader's contract note on
+// why the two grammars are allowed to disagree.
+const SINGLE_BYTE_RANGE = /^\s*bytes\s*=\s*(\d*)\s*-\s*(\d*)\s*$/i;
+
+/**
+ * Numeric comparison of two non-empty digit strings, without going through Number().
+ *
+ * A range-spec's positions are unbounded digit strings, and Number() silently rounds
+ * anything above 2^53: `Number("9007199254740993") === Number("9007199254740992")`, which
+ * collapsed `bytes=9007199254740993-9007199254740992` — an invalid spec (last < first)
+ * that §14.1.2 says to IGNORE, so 200 — into an apparently-valid one that then read as
+ * unsatisfiable and answered 416. Comparing normalised digit strings by length and then
+ * lexically is exact at every magnitude.
+ */
+function cmpDigits(a: string, b: string): number {
+ const x = a.replace(/^0+(?=\d)/, "");
+ const y = b.replace(/^0+(?=\d)/, "");
+ if (x.length !== y.length) return x.length - y.length;
+ return x < y ? -1 : x > y ? 1 : 0;
+}
+
+export type RangeIntent =
+ | { kind: "ignored" }
+ | { kind: "unsatisfiable" }
+ | { kind: "single"; start: number; length: number };
+
+/**
+ * What the client's Range header ASKS FOR, judged against the representation length.
+ *
+ * This exists because R2 does not tell us. Probed against a real R2 binding under
+ * vitest-pool-workers, `get(key, {range: <Headers>})` signals "I ignored your Range" by
+ * returning the WHOLE object with `range = {offset: 0, length: size}` — the byte-for-byte
+ * same shape it returns for a legitimately-satisfied whole-object range like `bytes=0-`.
+ * It does this for every unsatisfiable spec (`bytes=10-` / `bytes=99-` / `bytes=-0` on a
+ * 10-byte object), every malformed one (`bytes=abc`, `bytes=-`, `bytes=5-2`), multi-ranges
+ * (`bytes=0-1,4-5`) and unknown units (`items=0-5`). It does NOT throw for any of them and
+ * it never returns a zero/negative length except for a genuinely zero-length object.
+ * (The object-literal form `get(key, {range: {offset: 99}})` DOES throw
+ * "The requested range is not satisfiable (10039)" — but this Worker passes Headers, so
+ * that path is unreachable here.)
+ *
+ * Trusting `obj.range` alone therefore answered `Range: bytes=99-` with
+ * `206 + Content-Range: bytes 0-9/10` and the FULL body — a 206 that does not correspond to
+ * the request (RFC 9110 §15.3.7). That is actively dangerous for the resuming downloader
+ * this range forwarding exists to serve: a client resuming at byte 99 would append bytes
+ * 0-9 to its partial file and silently corrupt it. Re-deriving intent from the client's own
+ * header is the only way to separate the three cases.
+ *
+ * Satisfiability follows RFC 9110 §14.1.2 verbatim: an int-range is satisfiable iff
+ * first-pos < length; a suffix-range iff suffix-length is non-zero (so on a zero-length
+ * representation, a non-zero suffix-range is the ONLY satisfiable form). An invalid spec
+ * (last-pos < first-pos) MUST be ignored rather than rejected, hence "ignored", not
+ * "unsatisfiable".
+ *
+ * The `single` verdict carries the CONCRETE byte bounds the client asked for, clamped the
+ * way §14.1.2 clamps them. That is what makes this classifier safe to disagree with R2's
+ * parser. The two grammars are not identical and cannot be kept identical: R2's accepts
+ * only ASCII space around the tokens (miniflare's `/^ *(\d+)? *- *(\d+)? *$/`) while this
+ * one accepts `\s`, so `Range: bytes=2<TAB>-<TAB>4` parses here and is ignored there. When
+ * a caller compares these bounds against the bytes R2 actually handed back, any such
+ * divergence — this one, or the next one a parser change introduces — degrades to a plain
+ * 200 instead of a 206 whose Content-Range describes a body the client did not ask for.
+ * Chasing byte-for-byte grammar parity would put the guarantee back in the hands of two
+ * regexes staying in sync, which is the coupling that produced the bug.
+ */
+export function classifyRangeHeader(header: string, size: number): RangeIntent {
+ const m = SINGLE_BYTE_RANGE.exec(header);
+ if (!m) return { kind: "ignored" }; // multi-range, unknown unit, or unparseable
+ const [, firstRaw, lastRaw] = m;
+ if (firstRaw === "") {
+ if (lastRaw === "") return { kind: "ignored" }; // bare "bytes=-" is malformed
+ // §14.1.2: suffix-length 0 is unsatisfiable; a suffix past the start is the whole object.
+ if (cmpDigits(lastRaw, "0") <= 0) return { kind: "unsatisfiable" };
+ const suffix = Math.min(Number(lastRaw), size);
+ return { kind: "single", start: size - suffix, length: suffix };
+ }
+ // §14.1.2: an invalid spec (last-pos < first-pos) is ignored, not rejected.
+ if (lastRaw !== "" && cmpDigits(lastRaw, firstRaw) < 0) return { kind: "ignored" };
+ if (cmpDigits(firstRaw, String(size)) >= 0) return { kind: "unsatisfiable" };
+ const first = Number(firstRaw); // < size, so within safe-integer range
+ const last = lastRaw === "" ? size - 1 : Math.min(Number(lastRaw), size - 1);
+ return { kind: "single", start: first, length: last - first + 1 };
+}
+
+/** A quoted entity-tag list (`"a", W/"b"`) or `*`, compared per RFC 9110 §8.8.3.2. */
+function etagListMatches(list: string, etag: string, compare: "strong" | "weak"): boolean {
+ const items = list.split(",").map((s) => s.trim()).filter((s) => s !== "");
+ if (items.includes("*")) return true; // "*" matches iff a representation exists — one does
+ const weaken = (t: string) => t.replace(/^W\//, "");
+ // Strong comparison: neither side may be weak (§8.8.3.2).
+ if (compare === "strong" && etag.startsWith("W/")) return false;
+ return items.some((t) =>
+ compare === "strong" ? t === etag : weaken(t) === weaken(etag),
+ );
+}
+
+/**
+ * `uploaded <= <HTTP-date>`, at seconds granularity, or null when the date is unparseable
+ * (§13.1.3: an invalid date MUST be ignored, which callers map to "precondition passes").
+ * Seconds granularity matches R2's own comparison, which the Headers form of `onlyIf`
+ * selects — evaluating at millisecond precision here would disagree with the binding that
+ * produced the failure we are trying to name.
+ */
+function uploadedAtOrBefore(uploaded: Date | undefined, httpDate: string): boolean | null {
+ const at = Date.parse(httpDate);
+ if (Number.isNaN(at) || !uploaded) return null;
+ return Math.floor(uploaded.getTime() / 1000) <= Math.floor(at / 1000);
+}
+
+/**
+ * The conditional headers R2 is allowed to see, filtered to those RFC 9110 §13.2.2 says
+ * actually apply to THIS request.
+ *
+ * R2 ANDs together every validator it is handed; §13.2.2 instead defines a precedence in
+ * which a lower-ranked validator is not evaluated at all. Forwarding `request.headers`
+ * wholesale therefore let R2 fail a request on a validator the RFC says to ignore — most
+ * visibly `If-Modified-Since` on a non-GET/HEAD method, which §13.2.2 step 4 does not
+ * evaluate, but which R2 evaluated anyway and answered with a body-less object that this
+ * Worker could only turn into a 412. Filtering at the source means a body-less result now
+ * always corresponds to a precondition that genuinely applies.
+ *
+ * The METHOD decides this before any header does. §13.2.1: "a server MUST ignore the
+ * conditional request header fields defined by this specification when received with a
+ * request method that does not involve the selection or modification of a selected
+ * representation, such as CONNECT, OPTIONS, or TRACE." Filtering by validator applicability
+ * alone still handed those methods' conditionals to R2, so an OPTIONS carrying a stale
+ * `If-Match` a client had left lying around was refused 412 where the same request without
+ * it succeeded. Returning an empty set here is what "ignore" means at this layer: R2 is
+ * given nothing to evaluate, so it cannot answer body-less, so preconditionStatus() — which
+ * is only ever reached from a body-less result — is unreachable for these methods too.
+ */
+function applicablePreconditions(h: Headers, method: string): Headers {
+ const out = new Headers();
+ if (method === "OPTIONS" || method === "TRACE" || method === "CONNECT") return out;
+ const isGetOrHead = method === "GET" || method === "HEAD";
+ const ifMatch = h.get("if-match");
+ const ifNoneMatch = h.get("if-none-match");
+ if (ifMatch !== null) out.set("if-match", ifMatch);
+ else {
+ const ius = h.get("if-unmodified-since"); // §13.2.2 step 2: only when If-Match is absent
+ if (ius !== null) out.set("if-unmodified-since", ius);
+ }
+ if (ifNoneMatch !== null) out.set("if-none-match", ifNoneMatch);
+ else if (isGetOrHead) {
+ const ims = h.get("if-modified-since"); // step 4: only when If-None-Match is absent, GET/HEAD only
+ if (ims !== null) out.set("if-modified-since", ims);
+ }
+ return out;
+}
+
+/**
+ * Which status a FAILED `onlyIf` owes the client, decided by re-evaluating the request's
+ * conditionals against the object's own validators in RFC 9110 §13.2.2 order.
+ *
+ * R2 reports THAT a precondition failed and never WHICH one. Inferring from header
+ * PRESENCE cannot be right in both directions, which is how `If-Match: "x"` +
+ * `If-None-Match: "x"` on a matching object — If-Match satisfied, If-None-Match failed,
+ * so §13.1.2 owes a 304 — came back 412 purely because an If-Match header was present.
+ * Evaluating the validators removes the guess: presence selects which check runs, the
+ * comparison decides the answer.
+ */
+export function preconditionStatus(
+ h: Headers, isGetOrHead: boolean, etag: string, uploaded: Date | undefined,
+): 304 | 412 {
+ const ifMatch = h.get("if-match");
+ if (ifMatch !== null) {
+ if (!etagListMatches(ifMatch, etag, "strong")) return 412; // §13.2.2 step 1
+ } else {
+ const ius = h.get("if-unmodified-since"); // step 2
+ if (ius !== null && uploadedAtOrBefore(uploaded, ius) === false) return 412;
+ }
+ const ifNoneMatch = h.get("if-none-match");
+ if (ifNoneMatch !== null) {
+ // §13.1.2: a failed If-None-Match is 304 for GET/HEAD and 412 for every other method.
+ if (etagListMatches(ifNoneMatch, etag, "weak")) return isGetOrHead ? 304 : 412;
+ } else if (isGetOrHead) {
+ const ims = h.get("if-modified-since"); // step 4
+ if (ims !== null && uploadedAtOrBefore(uploaded, ims) === true) return 304;
+ }
+ // R2 refused for a reason this evaluation could not reproduce (a validator comparison
+ // that differs at the margins, say). 412 is the safe answer: a 304 would assert a cache
+ // validity we have not established.
+ return 412;
+}
+
+/**
+ * RFC 9110 §13.1.5 If-Range: does the client's validator still describe this object?
+ *
+ * R2 cannot answer this — its `R2Conditional` carries only etagMatches /
+ * etagDoesNotMatch / uploadedBefore / uploadedAfter, so an `If-Range` in the forwarded
+ * Headers is silently dropped and the range is applied unconditionally. For an object
+ * whose ETag has moved on, that answered `Range: bytes=100-` + `If-Range: "old"` with
+ * bytes 100+ of the NEW representation under a 206 — a resuming downloader then appends
+ * the new tail to its old prefix and silently corrupts the file, which is the precise
+ * failure this range forwarding exists to avoid.
+ *
+ * §13.1.5 requires a STRONG validator, so a weak entity-tag never matches. The date form
+ * likewise never matches here: it must compare against Last-Modified, and this Worker
+ * does not emit one, so no client can hold a date validator for this resource that we
+ * could honour — treating it as a mismatch (serve the complete representation) is both
+ * correct and the safe direction.
+ */
+function ifRangeMatches(value: string, etag: string): boolean {
+ const v = value.trim();
+ if (!v.startsWith('"')) return false; // weak tag or HTTP-date → not a strong match
+ return etagListMatches(v, etag, "strong");
+}
+
+/**
+ * Abandon a body stream this Worker has decided not to send.
+ *
+ * R2 hands back a body on paths whose response carries none. An unsatisfiable Range gets
+ * the COMPLETE object (29.2 MiB for the 1.3 PDF) and is answered 416 with a null body; a
+ * stale `If-Range` gets the sliced range and is answered from a re-read. Dropping the
+ * reference leaves the stream open until GC collects it, holding the connection; cancelling
+ * releases it now and aborts the transfer rather than draining it. Failures are swallowed —
+ * this is cleanup on a path whose response is already decided, and a stream that is already
+ * closed or errored is exactly the state we wanted.
+ */
+async function discardBody(body: ReadableStream | null | undefined): Promise<void> {
+ try {
+ await body?.cancel();
+ } catch {
+ /* already closed or errored — nothing left to release */
+ }
+}
+
+async function themed(env: ApexEnv, status: 404 | 503): Promise<Response> {
+ const page = await env.ASSETS.fetch(new Request(`https://apex.internal/${status}.html`));
+ return apexHeaders(new Response(page.body, { status, headers: { "content-type": "text/html; charset=utf-8" } }), env);
+}
+
+export default {
+ async fetch(request: Request, env: ApexEnv): Promise<Response> {
+ const url = new URL(request.url);
+
+ // 1. UA gate — production only (§3.2.1)
+ if (env.DOCS_ENV === "production") {
+ const v = uaVerdict(request.headers.get("user-agent") ?? "", policy);
+ if (v === "block") return apexHeaders(new Response("Forbidden", { status: 403 }), env);
+ if (v === "log") console.log(JSON.stringify({ event: "ua-log", ua: request.headers.get("user-agent"), path: url.pathname }));
+ }
+
+ // 2. Special paths (§3.2.2)
+ const special = await specialPathFor(request, manifest, env as never);
+ if (special) return apexHeaders(special, env);
+
+ // 2b. Legacy PDF R2 fallback (spec §5) — the 1.3 PDF (29.2 MiB) exceeds the 25 MiB
+ // static-asset cap and is excluded from the content Worker's own build. MUST run
+ // before version dispatch (step 6): the legacy Worker's asset tree lacks this file,
+ // so unconditional dispatch would 404 on the exact path the PDF 301 (§3.2.4) targets.
+ const pdfVersion = manifest.versions.find((v) => v.pdf_r2_key && v.pdf === url.pathname);
+ if (pdfVersion) {
+ const bucket = env.DOCS_PDFS;
+ if (!bucket) {
+ console.log(JSON.stringify({ event: "binding-missing", binding: "DOCS_PDFS" }));
+ return themed(env, 503);
+ }
+ const method = request.method.toUpperCase();
+ const isGetOrHead = method === "GET" || method === "HEAD";
+ // §14.2: "GET is the only method for which range handling is defined" — a Range on
+ // any other method MUST be ignored. Reading it as null here suppresses the whole
+ // partial-content path in one place: R2 is never asked to slice, and the 206/416
+ // branches below are unreachable. Gating only the R2 forward would leave the
+ // response side still seeing a Range header and answering a HEAD or a POST with a
+ // 416 or a Content-Range.
+ const rangeHeader = method === "GET" ? request.headers.get("range") : null;
+ const onlyIf = applicablePreconditions(request.headers, method);
+
+ let raw: R2ObjectBody | R2Object | null;
+ try {
+ // Forward Range + the APPLICABLE conditionals so a resumed download or a client
+ // with a fresh cached copy doesn't have to re-pull the full 29.2 MiB object.
+ raw = await bucket.get(pdfVersion.pdf_r2_key!, {
+ ...(rangeHeader !== null ? { range: request.headers } : {}),
+ onlyIf,
+ });
+ } catch (e) {
+ console.log(JSON.stringify({ event: "binding-error", binding: "DOCS_PDFS", error: String(e) }));
+ return themed(env, 503);
+ }
+ if (!raw) return themed(env, 404);
+
+ // Distinct (longer) cache class from apex's default control-response class — this
+ // is effectively content, just not content the legacy content Worker can serve.
+ // 304/206 are both <400 so apexHeaders() still applies this class, not "no-store".
+ const pdfCacheClass = "public, max-age=300, s-maxage=600, must-revalidate";
+ const pdfHeaders: Record<string, string> = {
+ etag: raw.httpEtag,
+ "accept-ranges": "bytes",
+ };
+
+ // A FAILED onlyIf precondition makes R2 hand back a body-less R2Object — just the
+ // validators, no content — and never says which validator failed. Because `onlyIf`
+ // was filtered to the conditionals §13.2.2 actually applies to this request, a
+ // body-less result here always means a precondition that genuinely applies failed;
+ // preconditionStatus() re-evaluates them against the object's own validators, in
+ // §13.2.2 order, to decide between 304 and 412.
+ if (!("body" in raw) || !raw.body) {
+ const status = preconditionStatus(
+ request.headers, isGetOrHead, raw.httpEtag, raw.uploaded);
+ return apexHeaders(new Response(null, { status, headers: pdfHeaders }), env, pdfCacheClass);
+ }
+ let obj = raw as R2ObjectBody;
+ pdfHeaders["content-type"] = "application/pdf";
+
+ // §13.1.5 If-Range, which R2 cannot evaluate (see ifRangeMatches). A failed validator
+ // means the client's partial copy is stale, so the Range is ignored ENTIRELY and the
+ // complete representation is served — including for a spec that would otherwise be
+ // unsatisfiable, since the 416 branch below must not fire on a range we have decided
+ // not to honour. R2 has already applied the range at this point, so the whole object
+ // has to be re-read; that costs one extra R2 read on the rare stale-resume path and
+ // nothing at all on the common one.
+ //
+ // The re-read carries the SAME `onlyIf`. §13.2.1 requires preconditions to hold for the
+ // representation ultimately selected, and the two reads need not see one object: a bare
+ // re-get answered `Range` + stale `If-Range` + `If-Match: "A"` with a 200 carrying
+ // object B, whose ETag the client had explicitly excluded, whenever the key was
+ // rewritten in between. That rewrite is not hypothetical — the legacy snapshot repo's
+ // deploy workflow re-uploads this exact key on its `force_pdf_refresh` input. Re-sending
+ // the conditionals ties the verdict to the bytes actually served, because R2 evaluates
+ // `onlyIf` against the very object it returns; the body-less outcome that produces is
+ // not a gap in this path but the correct answer, resolved by preconditionStatus()
+ // exactly as on the first read. When no conditionals were sent, `onlyIf` is empty and a
+ // body-less result cannot occur, so the common path is untouched.
+ //
+ // A head() before the get() would also expose the validators, and would avoid opening
+ // this slice stream at all — but it would put a second round-trip on the path where
+ // If-Range MATCHES, which is the normal resumed download, in exchange for tidying the
+ // rare one where it does not. It would also widen the window this shape keeps narrow:
+ // the object whose validators decide the verdict is the one R2 returns from the same
+ // call. So the get-then-re-read stays, and the slice we are abandoning is cancelled
+ // rather than left to GC.
+ const ifRange = rangeHeader !== null ? request.headers.get("if-range") : null;
+ let rangeApplies = rangeHeader !== null;
+ if (ifRange !== null && !ifRangeMatches(ifRange, obj.httpEtag)) {
+ rangeApplies = false;
+ await discardBody(obj.body);
+ let full: R2ObjectBody | R2Object | null;
+ try {
+ full = await bucket.get(pdfVersion.pdf_r2_key!, { onlyIf });
+ } catch (e) {
+ console.log(JSON.stringify({ event: "binding-error", binding: "DOCS_PDFS", error: String(e) }));
+ return themed(env, 503);
+ }
+ if (!full) return themed(env, 404); // deleted between the two reads
+ if (!("body" in full) || !full.body) {
+ // The key was rewritten between the two reads and the request's preconditions do
+ // not hold for the new representation. Same treatment as a first-read failure, on
+ // the new object's validators — never the old ones, which describe a representation
+ // this response is not about.
+ const status = preconditionStatus(
+ request.headers, isGetOrHead, full.httpEtag, full.uploaded);
+ return apexHeaders(new Response(null, {
+ status, headers: { etag: full.httpEtag, "accept-ranges": "bytes" },
+ }), env, pdfCacheClass);
+ }
+ obj = full as R2ObjectBody;
+ pdfHeaders.etag = obj.httpEtag;
+ }
+
+ // A satisfied Range request. R2 echoes the actually-served byte range on `obj.range` —
+ // but it does so for FULL gets too: against a real R2 binding under workerd, a get()
+ // whose forwarded Headers carry NO Range header still comes back with
+ // `range = {offset: 0, length: obj.size}`. Keying the 206 off `obj.range` alone therefore
+ // turned every plain GET of the 1.3 PDF into a 206 — which is exactly what the nightly
+ // canary sweep observed (`/en/1.3/vyos-documentation.pdf: status=206`) — and RFC 9110
+ // §15.3.7 only permits a 206 in answer to a request that actually carried a Range header.
+ // So: gate on the REQUEST first, then normalize whatever shape R2 handed back — and
+ // then CHECK that the two agree before promising a 206, because "R2 sliced it" and
+ // "R2 handed back everything" are the same shape on the wire.
+ //
+ // What R2 actually handed back, as concrete bounds. `obj.range` is absent only if a
+ // binding declines to report one, in which case the body is the complete object.
+ const actual = obj.range
+ ? resolveRange(obj.range, obj.size)
+ : { start: 0, length: obj.size };
+ const servedWhole = actual.start === 0 && actual.length === obj.size;
+
+ if (rangeApplies && rangeHeader !== null) {
+ const intent = classifyRangeHeader(rangeHeader, obj.size);
+ if (intent.kind === "unsatisfiable") {
+ // §14.2: "the server SHOULD send a 416"; §15.5.17: a 416 to a byte-range request
+ // SHOULD carry `Content-Range: bytes */<complete-length>`. Deliberately no
+ // content-type — there is no PDF payload on this response. 416 is >= 400 so
+ // apexHeaders() forces no-store, which is right: the verdict depends on the
+ // request's Range header and the cache key does not include it.
+ // R2 answers an unsatisfiable Range with the COMPLETE object, so the body being
+ // dropped here is the whole 29.2 MiB one — the largest abandoned stream on any
+ // path through this handler.
+ await discardBody(obj.body);
+ return apexHeaders(
+ new Response(null, {
+ status: 416,
+ headers: { etag: pdfHeaders.etag, "accept-ranges": "bytes",
+ "content-range": `bytes */${obj.size}` },
+ }),
+ env,
+ pdfCacheClass,
+ );
+ }
+ // A 206 is owed only when the bytes R2 selected are the bytes the client asked for.
+ // `length === 0` means a zero-length representation (the only way R2 yields it) —
+ // e.g. a non-zero suffix-range, which §14.1.2 calls satisfiable, against an empty
+ // object. No valid Content-Range exists for an empty selection (§14.4 forbids a
+ // last-pos below the first-pos), so a 206 is unrepresentable. Fall through to the
+ // 200: §15.5.17's own note records that servers are free to ignore Range and answer
+ // with the complete representation, which for an empty object is exactly this body.
+ if (intent.kind === "single" && intent.length > 0 &&
+ actual.start === intent.start && actual.length === intent.length) {
+ pdfHeaders["content-range"] =
+ `bytes ${intent.start}-${intent.start + intent.length - 1}/${obj.size}`;
+ pdfHeaders["content-length"] = String(intent.length);
+ return apexHeaders(new Response(obj.body, { status: 206, headers: pdfHeaders }), env, pdfCacheClass);
+ }
+ // Otherwise no 206 is owed: either the spec was one §14.1.2 says to ignore
+ // (multi-range / malformed / unknown unit), or R2 declined a spec that this parser
+ // accepted — the grammar divergence classifyRangeHeader documents. Both leave R2
+ // having returned the complete object, so fall through to the 200 below.
+ }
+
+ // Serve what R2 actually handed back, described truthfully. `servedWhole` is the
+ // normal case and the only one a 200 can describe; a partial body under a 200 would
+ // ship a Content-Length that contradicts it. A partial body reaching HERE — past the
+ // agreement check above — means R2 sliced to bounds the request did not ask for, and
+ // there is no honest success response left: a 200 would misstate the length, and a
+ // 206 would answer with a range the client never requested, which §14.4 does not
+ // permit (Content-Range on a 206 describes the selected range, and no range was
+ // selected). This used to ship that illegal 206.
+ //
+ // So: drop the slice and fail. Re-reading the object for a clean 200 is the other
+ // option Codex offered, but it means a second conditional read with its own
+ // object-rewritten-between-reads handling — a copy of the If-Range block above, or a
+ // refactor of that live and currently-correct path — bought for a branch that cannot
+ // execute under today's workerd (round 3 established empirically that the Headers
+ // form ignores multi-range and returns the whole object). The log line is the part
+ // that earns its keep: it is the alarm that R2's range semantics have moved, and it
+ // is what would justify writing that re-read for real.
+ if (!servedWhole) {
+ console.log(JSON.stringify({
+ event: "r2-range-divergence", path: url.pathname, size: obj.size,
+ served: `${actual.start}+${actual.length}`, requested: rangeHeader ?? null,
+ }));
+ await discardBody(obj.body);
+ return themed(env, 503);
+ }
+ pdfHeaders["content-length"] = String(obj.size);
+ return apexHeaders(new Response(obj.body, { status: 200, headers: pdfHeaders }), env, pdfCacheClass);
+ }
+
+ // 3+4. Trailing-slash + alias/codename/PDF 301s (§3.2.3-4)
+ const redir = redirectFor(url, manifest);
+ if (redir) return apexHeaders(redir, env);
+
+ // 5. /kb seam (§3.2.5)
+ if (url.pathname.startsWith("/kb/") || url.pathname === "/kb") {
+ if (env.DOCS_KB) return securityHeaders(await env.DOCS_KB.fetch(request));
+ return themed(env, 404);
+ }
+
+ // 6. Version dispatch (§3.2.6)
+ const hit = resolveVersion(url.pathname, dispatch);
+ if (hit) {
+ const fetcher = bindingGuard(env, hit.binding);
+ if (!fetcher) { // 7. runtime binding guard (§3.2.7)
+ console.log(JSON.stringify({ event: "binding-missing", binding: hit.binding }));
+ return themed(env, 503);
+ }
+ try {
+ const resp = await fetcher.fetch(request);
+ if (resp.status === 404) return themed(env, 404);
+ return securityHeaders(resp); // §3.3: security headers at apex; cache + X-Docs-Build stay content-owned
+ } catch (e) {
+ console.log(JSON.stringify({ event: "binding-error", binding: hit.binding, error: String(e) }));
+ return themed(env, 503);
+ }
+ }
+
+ // 7. Fallback
+ return themed(env, 404);
+ },
+} satisfies ExportedHandler<ApexEnv>;
diff --git a/workers/apex/src/manifest.ts b/workers/apex/src/manifest.ts
new file mode 100644
index 00000000..396f29ed
--- /dev/null
+++ b/workers/apex/src/manifest.ts
@@ -0,0 +1,55 @@
+import raw from "../../versions.json";
+
+export interface VersionEntry {
+ slug: string; label: string;
+ status: "dev" | "lts" | "eol";
+ binding: string; aliases: string[];
+ pdf: string | null;
+ // R2 object key for the apex PDF fallback (spec §5) — set only on versions whose PDF
+ // exceeds the 25 MiB static-asset cap and is therefore absent from the content
+ // Worker's own asset tree (currently just 1.3). Optional; most versions omit it.
+ pdf_r2_key?: string;
+}
+export interface Manifest {
+ schema_version: number;
+ default_lang: string; default_version: string;
+ languages: { code: string; label: string }[];
+ versions: VersionEntry[];
+}
+
+// Split out from loadManifest() so tests can validate synthetic manifests without
+// touching the real ../../versions.json import (which loadManifest() is hardwired to).
+export function validateManifest(m: Manifest): Manifest {
+ if (m.schema_version !== 2) throw new Error(`versions.json schema_version ${m.schema_version} != 2`);
+ const slugs = new Set<string>();
+ for (const v of m.versions) {
+ if (!/^DOCS_[A-Z0-9_]+$/.test(v.binding)) throw new Error(`bad binding for ${v.slug}`);
+ if (!["dev", "lts", "eol"].includes(v.status)) throw new Error(`bad status for ${v.slug}`);
+ if (slugs.has(v.slug)) throw new Error(`duplicate slug: ${v.slug}`);
+ // pdf_r2_key names an R2 fallback for the exact `pdf` URL — a null pdf has no URL
+ // for the fallback to ever be reached at, so the pairing is nonsensical.
+ if (v.pdf_r2_key && v.pdf === null) throw new Error(`pdf_r2_key set but pdf is null for ${v.slug}`);
+ slugs.add(v.slug);
+ }
+ // Separate pass: an alias must not collide with ANY canonical slug (not just an
+ // earlier one), so `slugs` must be fully populated before this check runs.
+ const aliases = new Set<string>();
+ for (const v of m.versions) {
+ for (const alias of v.aliases) {
+ if (slugs.has(alias)) throw new Error(`alias ${alias} (on ${v.slug}) collides with a canonical slug`);
+ if (aliases.has(alias)) throw new Error(`duplicate alias: ${alias}`);
+ aliases.add(alias);
+ }
+ }
+ if (!m.versions.some((v) => v.slug === m.default_version))
+ throw new Error(`default_version ${m.default_version} not in versions[]`);
+ return m;
+}
+
+export function loadManifest(): Manifest {
+ return validateManifest(raw as Manifest);
+}
+
+export function buildDispatch(m: Manifest): Map<string, string> {
+ return new Map(m.versions.map((v) => [v.slug, v.binding]));
+}
diff --git a/workers/apex/src/redirects.ts b/workers/apex/src/redirects.ts
new file mode 100644
index 00000000..835b50f7
--- /dev/null
+++ b/workers/apex/src/redirects.ts
@@ -0,0 +1,38 @@
+import type { Manifest } from "./manifest";
+
+function aliasMap(m: Manifest): Map<string, string> {
+ const map = new Map<string, string>();
+ for (const v of m.versions) for (const a of v.aliases) map.set(a, v.slug);
+ return map;
+}
+
+function r301(pathAndQuery: string): Response {
+ return new Response(null, { status: 301, headers: { Location: pathAndQuery } });
+}
+
+export function redirectFor(url: URL, m: Manifest): Response | null {
+ const { pathname, search } = url; // never touch url.hash — fragments don't reach the server
+
+ // RTD PDF URLs: /_/downloads/en/<ver>/pdf/* → /en/<slug>/vyos-documentation.pdf
+ const pdf = pathname.match(/^\/_\/downloads\/en\/([^/]+)\/pdf(?:\/|$)/);
+ if (pdf) {
+ const slug = aliasMap(m).get(pdf[1]) ?? pdf[1];
+ const entry = m.versions.find((v) => v.slug === slug);
+ // pdf: null means no PDF artifact exists for this version — don't 301 into a dead-end 404.
+ // The manifest's pdf value is the source of truth (not a hardcoded filename) so a future
+ // R2-fallback path for 1.3 (or any other version) can move the target without a code change.
+ if (entry && entry.pdf !== null) return r301(`${entry.pdf}${search}`);
+ }
+
+ // Alias / codename prefixes: /en/<alias>/* → /en/<slug>/*
+ const seg = pathname.match(/^\/en\/([^/]+)(\/.*)?$/);
+ if (seg) {
+ const [, first, rest = ""] = seg;
+ const target = aliasMap(m).get(first);
+ if (target) return r301(`/en/${target}${rest || "/"}${search}`);
+ // Trailing-slash normalization on bare version roots: /en/<slug> → /en/<slug>/
+ if (rest === "" && m.versions.some((v) => v.slug === first))
+ return r301(`/en/${first}/${search}`);
+ }
+ return null;
+}
diff --git a/workers/apex/src/special.ts b/workers/apex/src/special.ts
new file mode 100644
index 00000000..15e693b1
--- /dev/null
+++ b/workers/apex/src/special.ts
@@ -0,0 +1,60 @@
+import type { Manifest } from "./manifest";
+import { bindingGuard } from "./dispatch";
+
+// Static assets (404.html, 503.html, robots.txt, favicons, root) live in the apex
+// ASSETS binding; versions.json is served from the imported manifest so the body
+// always matches the dispatch map.
+export async function specialPathFor(
+ request: Request,
+ m: Manifest,
+ env: Record<string, unknown> & { ASSETS: Fetcher },
+): Promise<Response | null> {
+ const url = new URL(request.url);
+ const p = url.pathname;
+
+ if (p === "/") // default-version redirect (§3.2.2) — query preserved, like alias redirects
+ return new Response(null, { status: 301, headers: { Location: `/en/${m.default_version}/${url.search}` } });
+
+ if (p === "/versions.json")
+ return new Response(JSON.stringify(m), {
+ headers: { "content-type": "application/json; charset=utf-8" },
+ });
+
+ if (p === "/healthz")
+ return new Response(JSON.stringify({ status: "ok", versions: m.versions.length }), {
+ headers: { "content-type": "application/json" },
+ });
+
+ if (p === "/sitemap.xml") {
+ // Origin comes from the REQUEST, never a hard-coded production hostname: this same Worker
+ // also serves the canary origin (docs-next.vyos.io, DOCS_ENV=canary), and a canary sitemap
+ // index whose entries pointed at docs.vyos.io would send any checker that follows it
+ // straight to production — a candidate tree with broken or missing per-version sitemaps
+ // would then pass its own sitemap check by silently grading production instead of itself.
+ const entries = m.versions
+ .map((v) => `<sitemap><loc>${url.origin}/en/${v.slug}/sitemap.xml</loc></sitemap>`)
+ .join("");
+ return new Response(
+ `<?xml version="1.0" encoding="UTF-8"?><sitemapindex xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">${entries}</sitemapindex>`,
+ { headers: { "content-type": "application/xml" } },
+ );
+ }
+
+ if (p === "/llms.txt") { // direct body, not a redirect (§3.2.2)
+ const def = m.versions.find((v) => v.slug === m.default_version)!;
+ const b = bindingGuard(env, def.binding);
+ if (!b) // do NOT fall through — /llms.txt with a missing binding is a 503, not a 404
+ return new Response("service unavailable", { status: 503, headers: { "content-type": "text/plain" } });
+ // Forward the ORIGINAL request (method + conditional-GET headers), just retargeted
+ // to the versioned path — mirrors the robots.txt/favicon pass-through below.
+ const target = new URL(`/en/${def.slug}/llms.txt`, url);
+ return b.fetch(new Request(target, request));
+ }
+
+ if (p === "/robots.txt" || p === "/favicon.ico" || /^\/apple-touch-icon.*\.png$/.test(p))
+ // Pass the ORIGINAL request through (not a re-synthesized `new Request(url)`) so
+ // conditional-GET headers (If-None-Match / If-Modified-Since) reach ASSETS intact.
+ return env.ASSETS.fetch(request);
+
+ return null;
+}
diff --git a/workers/apex/src/uagate.ts b/workers/apex/src/uagate.ts
new file mode 100644
index 00000000..4a407e51
--- /dev/null
+++ b/workers/apex/src/uagate.ts
@@ -0,0 +1,48 @@
+export interface UaPolicy {
+ allow: string[];
+ log: string[];
+ block: string[];
+}
+
+export type UaVerdict = "allow" | "block" | "log";
+
+/**
+ * The LONGEST entry in `list` occurring in the (already-lowercased) UA, lowercased, or null.
+ * Longest rather than first-hit so the containment test in uaVerdict() compares against the
+ * most specific entry a multi-token UA matched, not an arbitrary earlier one.
+ */
+function bestMatch(lowerUa: string, list: string[]): string | null {
+ return list.reduce<string | null>((best, entry) => {
+ const needle = entry.toLowerCase();
+ if (!lowerUa.includes(needle)) return best;
+ return best === null || needle.length > best.length ? needle : best;
+ }, null);
+}
+
+export function uaVerdict(ua: string, policy: UaPolicy): UaVerdict {
+ const lowerUa = ua.toLowerCase();
+ // Explicit blocks take precedence — a request-controlled UA string that spoofs an
+ // allow-listed substring (e.g. "Googlebot EvilScraper") must not be able to bypass a
+ // block entry just by also matching the allow list.
+ if (bestMatch(lowerUa, policy.block) !== null) return "block";
+
+ // A log match WINS over any competing allow match, unconditionally. `log` is a telemetry
+ // verdict, not a denial (the request is served either way), so resolving a contest the
+ // wrong way is asymmetric: choosing `allow` loses the ua-log event permanently, while
+ // choosing `log` costs one log line. A UA presenting BOTH an allow token and a log token
+ // (e.g. "GPTBot/1.0 DuckDuckBot") is exactly the shape worth recording.
+ //
+ // There used to be a carve-out here: a matched allow entry that strictly CONTAINED the
+ // matched log entry won, so a policy could express a narrow allow exception inside a
+ // broader log entry (log "Foo", allow "Foo-Search"). It is gone, for two reasons. It was
+ // spoofable — containment was tested between the two matched ENTRIES, never against the
+ // UA's own token structure, so a caller writing "Bytespider/2.0 Bytespider-Search/1.0"
+ // matched both entries as independent tokens and bought itself `allow`, and the UA
+ // string is entirely request-controlled. And it bought nothing: no entry pair in
+ // ua-policy.json takes that branch. The pair the shipped policy does depend on runs the
+ // OTHER way — Apple ships "Applebot" (search, allow) and "Applebot-Extended" (AI
+ // training, log), where the log entry is the longer one, so there is no containment and
+ // log wins regardless. Losing the carve-out costs a future narrow-allow vendor variant
+ // nothing worse than being logged as well as served.
+ return bestMatch(lowerUa, policy.log) === null ? "allow" : "log"; // unknown UAs fail open
+}