From a783e56b774795c1f4bd8a6088dfcce96a307c5f Mon Sep 17 00:00:00 2001 From: Yuriy Andamasov Date: Fri, 25 Sep 2026 17:11:19 +0200 Subject: ci: port the Cloudflare Workers docs pipeline to circinus (slug 1.5) (#2208) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * ci: port Cloudflare Workers docs pipeline files to circinus (verbatim) Copies the branch-agnostic half of the docs.vyos.io Cloudflare Workers pipeline from `rolling` at 8cb568bf, byte-identical: - .github/workflows/docs-build.yml - workers/ (entire tree) - scripts/docs_gates/ - docker/im-convert.sh - docs/_static/js/version-picker.js, js/pagefind-wrapper.js, css/version-picker.css (new files, no circinus counterpart) - docs/_templates/breadcrumbs.html, searchbox.html (new files) docs-build.yml already triggers on push to [rolling, circinus, sagitta] and resolves `circinus` -> worker vyos-docs-v15-en / slug 1.5 from workers/matrix.json; the files simply did not exist on this branch, so slug 1.5 still serves the bootstrap placeholder. workers/versions.json + workers/matrix.json are deliberately identical across all three branches and must be kept in sync. Advances: IS-572 * ci: wire circinus docs build into the Cloudflare Workers pipeline Hand-merges the CF-specific hunks onto circinus's own docker/Dockerfile and docs/conf.py rather than clobbering them with rolling's versions โ€” circinus keeps its own content-driven history in both files. docker/Dockerfile: - imagemagick + librsvg2-bin (sphinx.ext.imgconverter backend) and poppler-utils (pdfinfo, used by docs-build.yml's PDF page-count completeness check), installed --no-install-recommends - install docker/im-convert.sh as /usr/local/bin/im-convert docs/conf.py: - enable sphinx.ext.imgconverter + image_converter = 'im-convert' so the LaTeX/PDF builder stops silently dropping .webp/.svg images - register js/version-picker.js + css/version-picker.css unconditionally (degrades silently on ReadTheDocs) - _vyos_cf_build gate off the raw DOCS_VERSION_SLUG env var; only CF builds load js/pagefind-wrapper.js, and html_context['vyos_cf_build'] lets _templates/searchbox.html fall back to the stock Sphinx searchbox via the "!" bang-include on RTD The RTD path stays the default in both files: circinus continues building on ReadTheDocs until RTD sunset, and every CF feature activates only when DOCS_VERSION_SLUG is present. .readthedocs.yml is untouched. circinus keeps its own version/release/html_title/source_suffix and its hardcoded html_baseurl (its CF slug is also `1.5`, so rolling's DOCS_VERSION_SLUG/READTHEDOCS_VERSION resolution block is a no-op here and was deliberately not ported). Advances: IS-572 * ci: IS-572: re-sync ported Cloudflare Workers pipeline files with rolling The category-1 files in this port are byte-identical copies from `rolling`. `rolling` has since moved: [vyos-documentation#2209](https://github.com/vyos/vyos-documentation/pull/2209) merged as `3a1c6c30`, thirteen rounds of hardening on exactly these files. Re-take all 14 category-1 paths from `origin/rolling` via `git checkout origin/rolling -- `, so byte-identity holds by construction rather than by hand-editing: .github/workflows/docs-build.yml scripts/docs_gates/{gates,parity,smoke,test_gates,test_parity,test_smoke}.py workers/.gitignore workers/apex/src/{index,special,uagate}.ts workers/apex/test/{router,uagate}.test.ts workers/apex/ua-policy.json Thirteen of the fourteen carry [vyos-documentation#2209](https://github.com/vyos/vyos-documentation/pull/2209) exactly โ€” the pre-change tree was byte-identical to `3a1c6c30^` for those paths. `workers/.gitignore` additionally picks up the one-line `test-results/` entry from [vyos-documentation#2212](https://github.com/vyos/vyos-documentation/pull/2212); inert on circinus, since only the deliberately-unported `apex-deploy.yml` writes that directory. Deliberate exclusions are unchanged: `docs-canary-qa.yml` (cron runs on the default branch only, so it is not ported even though [vyos-documentation#2209](https://github.com/vyos/vyos-documentation/pull/2209) touched it on `rolling`), `apex-deploy.yml`, and the `docs-preview-*` workflows. `docs/conf.py` stays hand-merged and circinus-specific, with its ReadTheDocs fallback intact. ๐Ÿค– Generated by [robots](https://vyos.io) --- scripts/docs_gates/smoke.py | 236 ++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 236 insertions(+) create mode 100644 scripts/docs_gates/smoke.py (limited to 'scripts/docs_gates/smoke.py') diff --git a/scripts/docs_gates/smoke.py b/scripts/docs_gates/smoke.py new file mode 100644 index 00000000..6c526d98 --- /dev/null +++ b/scripts/docs_gates/smoke.py @@ -0,0 +1,236 @@ +"""Scoped pre-traffic smoke + post-promote probe (spec ยง7.1 steps 2/5). + +Probes ONE version's pages plus apex special paths through a host, presenting a +CF Access service token. Header contract (ยง3.3): content probes assert +X-Docs-Build == --expect-sha; apex probes assert X-Apex-Build presence only. + +Phase-2 obligation (authorized addition, not in the original spec text): the +version's index.html probe also asserts the response body carries the +`#vyos-search` mount div (docs/_templates/searchbox.html), guarding the +Pagefind gate's silent-degrade failure mode โ€” a build that forgot to set +DOCS_VERSION_SLUG would otherwise ship stock RTD search without CI noticing. +""" +from __future__ import annotations + +import argparse +import dataclasses +import json +import os +import sys +import time +import urllib.request +from pathlib import Path + +APEX_PATHS = ["/versions.json", "/healthz", "/robots.txt", "/sitemap.xml"] +SEARCH_MOUNT_MARKER = 'id="vyos-search"' + +# Explicit UA so the gate never depends on a Cloudflare edge exemption for the default +# Python-urllib UA. The Browser Integrity Check blocked that UA until a skip rule was added; +# the gate must not silently rely on that rule surviving. +USER_AGENT = "vyos-docs-smoke/1.0 (+https://github.com/vyos/vyos-documentation)" + +# Round-based retry (module-level so tests can shrink them). A freshly-deployed worker version +# can lose a propagation race: for a few minutes a probe may be served by the PREVIOUS version +# (wrong status / stale X-Docs-Build). Rather than fail-fast, each round re-probes ONLY the +# still-failing probes โ€” this preserves the full per-probe failure enumeration (diagnostic +# value) while adding at most (MAX_ROUNDS - 1) inter-round sleeps. Envelope widened to 5 rounds +# x 30s after an observed propagation wave outlasted 3 rounds x 20s (a path still stale at +# round 3): 4 x 30s = 2 min now covers the observed 1-2+ min waves. The green path still costs +# zero extra time (no retries), and DEADLINE_SECONDS=480 still bounds the worst case. +MAX_ROUNDS = 5 +RETRY_SLEEP_SECONDS = 30 +DEADLINE_SECONDS = 480 +# Per-socket-op timeout (connect + read), capped down to the remaining deadline budget on each +# probe so a probe that starts late cannot overshoot DEADLINE_SECONDS. Pages are small, so this +# socket-op timeout also bounds body reads adequately โ€” no separate body-read deadline is needed. +PROBE_TIMEOUT_SECONDS = 30 + + +class _NoRedirect(urllib.request.HTTPRedirectHandler): + """Probes assert an EXACT status per-path (200 or 404) โ€” following a 3xx would + silently swap the probed status for whatever the redirect target returns, + masking an accidental redirect where a direct 200/404 was expected. Mirrors + parity.py's _NoRedirect/_OPENER pattern.""" + + def redirect_request(self, req, fp, code, msg, headers, newurl): # noqa: D401 + return None + + +_OPENER = urllib.request.build_opener(_NoRedirect) + + +@dataclasses.dataclass +class Probe: + path: str + expect_status: int + assert_docs_build: bool + assert_apex_build: bool + assert_search_mount: bool = False + + +def probe_plan(slug: str, pdf: str | None, critical: list[str]) -> list[Probe]: + # `critical` may itself list "index.html" (it does in critical-pages.txt); drop it so the + # index page is probed exactly once โ€” as plan[0], the sole search-mount probe below. + critical = [rel for rel in critical if rel != "index.html"] + plan = [Probe(f"/en/{slug}/{rel}", 200, True, False) for rel in ["index.html", *critical]] + plan.append(Probe(f"/en/{slug}/pagefind/pagefind.js", 200, True, False)) + if pdf: + # assert_docs_build=False: the PDF is the ONE content path that can legitimately be + # answered by the apex Worker instead of a branch content Worker. 1.3's PDF (29.2 MiB) + # exceeds the 25 MiB static-asset cap, so apex serves it straight from R2 (spec ยง5) and + # that response carries only etag / accept-ranges / content-type / content-length โ€” + # X-Docs-Build is a content-Worker header apex never sets on it. Asserting it made the + # probe structurally unpassable for 1.3 (observed nightly: "detail=status+docs-build"). + # The build SHA is still gated for this version: every HTML probe above asserts it. + plan.append(Probe(pdf, 200, False, False)) + plan.append(Probe(f"/en/{slug}/definitely-missing-page-xyz.html", 404, False, False)) + plan += [Probe(p, 200, False, True) for p in APEX_PATHS] + plan[0].assert_search_mount = True # plan[0] is always /en//index.html + return plan + + +def docs_build_ok(header_value: str | None, expect_sha: str) -> bool: + """SKIP sentinel (nightly sweep): header presence only; otherwise exact match.""" + if expect_sha == "SKIP": + return header_value is not None + return header_value == expect_sha + + +def search_mount_present(html: str) -> bool: + return SEARCH_MOUNT_MARKER in html + + +def _probe_once(host: str, probe: Probe, expect_sha: str, access_id: str, access_secret: str, + timeout: float = PROBE_TIMEOUT_SECONDS, + ) -> tuple[bool, int | None, str | None, str | None]: + """One probe attempt. Returns (ok, status, docs_build, detail). `detail` names the failed + check(s) โ€” "status" / "docs-build" / "apex-build" / "search-mount" joined by "+", or the + transport error text โ€” and is None when ok. `timeout` is the per-socket-op deadline (connect + + read); run() caps it to the remaining budget so a late probe can't overshoot the overall + deadline. ANY exception in the open OR body-read path (including a transport error DURING + HTTPError.read()) is contained and yields (False, None, None, ): a retryable + failure, never a traceback. The HTTPError response stream is always closed โ€” it owns a + socket, so a bare e.read() would leak it.""" + req = urllib.request.Request(f"https://{host}{probe.path}", method="GET") + req.add_header("CF-Access-Client-Id", access_id) + req.add_header("CF-Access-Client-Secret", access_secret) + req.add_header("User-Agent", USER_AGENT) + try: + try: + with _OPENER.open(req, timeout=timeout) as resp: + status, headers, body = resp.status, resp.headers, resp.read() + except urllib.error.HTTPError as e: # non-2xx still carries headers/body + with e: # HTTPError is file-like and owns the response socket โ€” always close it + status, headers, body = e.code, e.headers, e.read() + except Exception as e: # noqa: BLE001 โ€” open OR read failure โ†’ retryable probe result + return False, None, None, str(e) + reasons: list[str] = [] + if status != probe.expect_status: + reasons.append("status") + if probe.assert_docs_build and not docs_build_ok(headers.get("X-Docs-Build"), expect_sha): + reasons.append("docs-build") + if probe.assert_apex_build and not headers.get("X-Apex-Build"): + reasons.append("apex-build") + if probe.assert_search_mount and not search_mount_present( + body.decode("utf-8", errors="replace")): + reasons.append("search-mount") + detail = "+".join(reasons) if reasons else None + return not reasons, status, headers.get("X-Docs-Build"), detail + + +def run(host: str, slug: str, expect_sha: str, access_id: str, access_secret: str, + pdf: str | None, critical: list[str]) -> int: + """Probe the whole plan, then re-probe ONLY the still-failing probes each round (up to + MAX_ROUNDS, one RETRY_SLEEP_SECONDS between rounds). A probe passing in ANY round passes; + a single propagation blip served by the previous worker version cannot fail the gate. + DEADLINE_SECONDS bounds total wall-clock โ€” on breach, unresolved probes count as failed.""" + plan = probe_plan(slug, pdf, critical) + start = time.monotonic() + deadline = start + DEADLINE_SECONDS # absolute โ€” a hard upper bound on total wall-clock + pending = list(plan) # probes not yet passed + detail_by_path: dict[str, str] = {} # last failure detail per path, for logging + deadline_hit = False + + def _remaining() -> float: + return deadline - time.monotonic() + + for round_num in range(1, MAX_ROUNDS + 1): + if not pending: + break + still_failing: list[Probe] = [] + unprobed: list[Probe] = [] + for i, probe in enumerate(pending): + remaining = _remaining() # checked before each probe + if remaining < 1: # < 1s budget: don't start a probe that could overshoot + deadline_hit = True + unprobed = pending[i:] # not reached this round โ†’ still unresolved + break + ok, status, docs_build, detail = _probe_once( + host, probe, expect_sha, access_id, access_secret, + timeout=min(PROBE_TIMEOUT_SECONDS, max(1, remaining))) + if ok: + continue + still_failing.append(probe) + detail_by_path[probe.path] = ( + f"status={status} docs-build={docs_build} detail={detail}") + pending = still_failing + unprobed + if deadline_hit or not pending or round_num == MAX_ROUNDS: + break + remaining = _remaining() # checked before the inter-round sleep + if remaining <= 0: # no budget left โ†’ deadline path (skip the sleep) + deadline_hit = True + break + for probe in pending: + print(f"SMOKE-RETRY {probe.path}: round {round_num} " + f"{detail_by_path[probe.path]}", file=sys.stderr) + time.sleep(min(RETRY_SLEEP_SECONDS, remaining)) # never sleep past the deadline + + if deadline_hit: + print("SMOKE-DEADLINE: overall deadline reached โ€” remaining probes counted as failed", + file=sys.stderr) + for probe in pending: + print(f"SMOKE-FAIL {probe.path}: {detail_by_path.get(probe.path, 'unresolved')}", + file=sys.stderr) + failures = len(pending) + print(json.dumps({"failures": failures})) + return 1 if failures else 0 + + +def main() -> int: + ap = argparse.ArgumentParser() + ap.add_argument("--host", required=True) + ap.add_argument("--slug", required=True) + ap.add_argument("--expect-sha", required=True) + ap.add_argument("--pdf", default=None) + ap.add_argument("--critical-list", type=Path, + default=Path("scripts/docs_gates/critical-pages.txt")) + a = ap.parse_args() + # CF Access service-token credentials are read ONLY from the environment. They were also + # accepted as --access-id/--access-secret flags; that is removed rather than merely + # discouraged, because a value passed in argv publishes it in the process command line โ€” + # readable from the process table for the lifetime of the process, and captured verbatim + # by `set -x` shell traces, crash dumps and process-listing tooling. No call site used + # the flags (docs-build.yml and docs-canary-qa.yml both export the env vars), so there is + # nothing to migrate. Neither value nor its length is ever echoed. + access_id = os.environ.get("CF_ACCESS_CLIENT_ID", "") + access_secret = os.environ.get("CF_ACCESS_CLIENT_SECRET", "") + missing = [name for name, value in ( + ("CF_ACCESS_CLIENT_ID", access_id), + ("CF_ACCESS_CLIENT_SECRET", access_secret)) if not value] + if missing: + # The canary host is Access-gated, so an empty credential would turn every probe into + # an indistinguishable 403 โ€” fail loudly on the cause instead. Names only, no values. + print(f"missing CF Access credentials: {', '.join(missing)}", file=sys.stderr) + return 2 + # Strip BEFORE the comment test: an indented " # note" line is a comment, not a page that + # every deployable build must contain (it would fail the probe as a missing critical page). + # read_text() (rather than a bare open().read()) closes the handle deterministically, + # matching gates.py; the bare form leaked the descriptor until GC on any interpreter + # without CPython's refcounting. + lines = (line.strip() for line in a.critical_list.read_text().splitlines()) + critical = [line for line in lines if line and not line.startswith("#")] + return run(a.host, a.slug, a.expect_sha, access_id, access_secret, a.pdf, critical) + + +if __name__ == "__main__": + raise SystemExit(main()) -- cgit v1.2.3