summaryrefslogtreecommitdiff
path: root/scripts/docs_gates
diff options
context:
space:
mode:
Diffstat (limited to 'scripts/docs_gates')
-rw-r--r--scripts/docs_gates/__init__.py0
-rw-r--r--scripts/docs_gates/conftest.py55
-rw-r--r--scripts/docs_gates/critical-pages.txt9
-rw-r--r--scripts/docs_gates/gates.py96
-rw-r--r--scripts/docs_gates/parity.py253
-rw-r--r--scripts/docs_gates/smoke.py236
-rw-r--r--scripts/docs_gates/test_gates.py122
-rw-r--r--scripts/docs_gates/test_parity.py340
-rw-r--r--scripts/docs_gates/test_smoke.py483
9 files changed, 1594 insertions, 0 deletions
diff --git a/scripts/docs_gates/__init__.py b/scripts/docs_gates/__init__.py
new file mode 100644
index 00000000..e69de29b
--- /dev/null
+++ b/scripts/docs_gates/__init__.py
diff --git a/scripts/docs_gates/conftest.py b/scripts/docs_gates/conftest.py
new file mode 100644
index 00000000..38d6127e
--- /dev/null
+++ b/scripts/docs_gates/conftest.py
@@ -0,0 +1,55 @@
+"""Shared pytest fixtures for docs_gates tests.
+
+Provides a real local HTTP server (stdlib http.server, no TLS) used to exercise the
+_NoRedirect opener pattern (parity.py / smoke.py) end-to-end over an actual network
+round trip, rather than only unit-testing the handler class in isolation.
+"""
+from __future__ import annotations
+
+import threading
+from collections.abc import Iterator
+from http.server import BaseHTTPRequestHandler, HTTPServer
+
+import pytest
+
+REDIRECT_PATH = "/redirect-me"
+REDIRECT_LOCATION = "https://example.invalid/target"
+
+
+class _RedirectHandler(BaseHTTPRequestHandler):
+ """301+Location for REDIRECT_PATH; 200 for anything else."""
+
+ def do_GET(self) -> None: # noqa: N802 — stdlib handler method name
+ self._respond()
+
+ def do_HEAD(self) -> None: # noqa: N802
+ self._respond()
+
+ def _respond(self) -> None:
+ if self.path == REDIRECT_PATH:
+ self.send_response(301)
+ self.send_header("Location", REDIRECT_LOCATION)
+ self.end_headers()
+ else:
+ self.send_response(200)
+ self.send_header("Content-Type", "text/plain")
+ self.end_headers()
+ if self.command == "GET":
+ self.wfile.write(b"ok")
+
+ def log_message(self, format: str, *args: object) -> None: # noqa: A002 — quiet test output
+ pass
+
+
+@pytest.fixture
+def redirect_http_server() -> Iterator[str]:
+ """Starts the server on 127.0.0.1 (ephemeral port); yields 'host:port'."""
+ server = HTTPServer(("127.0.0.1", 0), _RedirectHandler)
+ thread = threading.Thread(target=server.serve_forever, daemon=True)
+ thread.start()
+ try:
+ yield f"127.0.0.1:{server.server_address[1]}"
+ finally:
+ server.shutdown()
+ server.server_close()
+ thread.join(timeout=5)
diff --git a/scripts/docs_gates/critical-pages.txt b/scripts/docs_gates/critical-pages.txt
new file mode 100644
index 00000000..1bd1a171
--- /dev/null
+++ b/scripts/docs_gates/critical-pages.txt
@@ -0,0 +1,9 @@
+# Paths relative to en/<slug>/ that must exist in every deployable build.
+# Verified 2026-07-10 against a real `sphinx-build -b html docs docs/_build/html-verify`
+# of the `rolling` tree (Python 3.12 venv; coverage.md excluded — pre-existing,
+# unrelated CfgcmdList/HTML5Translator crash, not touched by this task).
+index.html
+installation/index.html
+configuration/index.html
+cli.html
+search.html
diff --git a/scripts/docs_gates/gates.py b/scripts/docs_gates/gates.py
new file mode 100644
index 00000000..7b5cbf1d
--- /dev/null
+++ b/scripts/docs_gates/gates.py
@@ -0,0 +1,96 @@
+"""Deploy-blocking sanity gates (spec §7.1).
+
+Gates: file-count vs plan cap (<= 80% of 100k), per-file < 25 MiB, index.html +
+critical-page presence, page-count delta vs previous deploy, canonical URLs must
+start with the https://docs.vyos.io/en/<slug>/ prefix, declared PDF present,
+Pagefind non-empty.
+Exit 0 = deployable; exit 1 = blocked (one line per failed gate on stderr).
+"""
+from __future__ import annotations
+
+import argparse
+import json
+import re
+import sys
+from pathlib import Path
+
+FILE_CAP = 100_000
+CAP_FRACTION = 0.8
+MAX_FILE = 25 * 1024 * 1024
+COUNT_DELTA_MIN_RATIO = 0.5 # new build must have >= 50% of previous page count
+CANONICAL_RE = re.compile(r'<link\s+rel="canonical"\s+href="([^"]+)"')
+
+
+def _fail(msgs: list[str], msg: str) -> None:
+ msgs.append(msg)
+ print(f"GATE-FAIL: {msg}", file=sys.stderr)
+
+
+def run(artifact: Path, slug: str, versions: Path, previous_meta: Path | None,
+ critical: list[str]) -> int:
+ fails: list[str] = []
+ root = artifact / "en" / slug
+ manifest = json.loads(versions.read_text())
+ entry = next((v for v in manifest["versions"] if v["slug"] == slug), None)
+ if entry is None:
+ _fail(fails, f"slug {slug} not in versions.json")
+ return 1
+
+ files = [p for p in artifact.rglob("*") if p.is_file()]
+ if len(files) > FILE_CAP * CAP_FRACTION:
+ _fail(fails, f"file count {len(files)} > 80% of {FILE_CAP} cap")
+ for p in files:
+ if p.stat().st_size > MAX_FILE:
+ _fail(fails, f"{p.relative_to(artifact)} exceeds 25 MiB")
+
+ for rel in ["index.html", *critical]:
+ if not (root / rel).is_file():
+ _fail(fails, f"critical page missing: en/{slug}/{rel}")
+
+ pagefind = root / "pagefind"
+ if not pagefind.is_dir() or not any(pagefind.iterdir()):
+ _fail(fails, "Pagefind index missing or empty")
+
+ if entry.get("pdf"):
+ expected = artifact / entry["pdf"].lstrip("/")
+ if not expected.is_file():
+ _fail(fails, f"declared PDF missing: {entry['pdf']}")
+
+ pages = [p for p in root.rglob("*.html")]
+ if previous_meta is not None and previous_meta.is_file():
+ prev = json.loads(previous_meta.read_text())
+ if prev.get("page_count") and len(pages) < prev["page_count"] * COUNT_DELTA_MIN_RATIO:
+ _fail(fails, f"page count collapsed: {len(pages)} vs previous {prev['page_count']}")
+
+ want = f"https://docs.vyos.io/en/{slug}/"
+ for p in pages:
+ m = CANONICAL_RE.search(p.read_text(errors="ignore"))
+ if m is None:
+ _fail(fails, f"missing canonical link in en/{slug}/{p.relative_to(root)}")
+ break # one example is enough to block
+ if not m.group(1).startswith(want):
+ _fail(fails, f"bad canonical in en/{slug}/{p.relative_to(root)}: {m.group(1)}")
+ break # one example is enough to block
+
+ return 1 if fails else 0
+
+
+def main() -> int:
+ ap = argparse.ArgumentParser()
+ ap.add_argument("--artifact", type=Path, required=True)
+ ap.add_argument("--slug", required=True)
+ ap.add_argument("--versions", type=Path, required=True)
+ ap.add_argument("--previous-meta", type=Path, default=None)
+ ap.add_argument("--critical-list", type=Path,
+ default=Path("scripts/docs_gates/critical-pages.txt"))
+ a = ap.parse_args()
+ # Strip BEFORE the comment test: an indented " # note" line is a comment, not a page
+ # every deployable build must contain — the unstripped test turned it into a live entry,
+ # and a comment can never exist as a file, so it would block the deploy as a missing page.
+ lines = (line.strip() for line in a.critical_list.read_text().splitlines())
+ critical = [line for line in lines if line and not line.startswith("#")]
+ return run(a.artifact, a.slug, a.versions, a.previous_meta, critical)
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())
diff --git a/scripts/docs_gates/parity.py b/scripts/docs_gates/parity.py
new file mode 100644
index 00000000..c30536db
--- /dev/null
+++ b/scripts/docs_gates/parity.py
@@ -0,0 +1,253 @@
+"""URL-parity corpus + alias assertions (spec §11).
+
+Modes:
+ --sitemap-host X pull per-version sitemaps from X (RTD pre-cutover)
+ --probe-host Y probe every URL on Y (canary via Access, or production)
+Fails (exit 1) on any status mismatch (non-200 for corpus rows; wrong
+Location for alias rows).
+"""
+from __future__ import annotations
+
+import argparse
+import dataclasses
+import json
+import os
+import re
+import sys
+import urllib.parse
+import urllib.request
+from pathlib import Path
+
+SITEMAP_LOC = re.compile(r"<loc>([^<]+)</loc>")
+
+# Sitemap sweep covers only the versions THIS repo builds on CF (rolling/1.5/1.4).
+# 1.3/1.2 have NO RTD sitemaps (spec §15a.5) — their parity is the legacy snapshot
+# repo's crawl-inventory job. Their alias/PDF redirect rows below stay in scope.
+DEFAULT_SLUGS = "rolling,1.5,1.4"
+
+ALIASES = [("latest", "rolling"), ("stable", "1.5"), ("lts", "1.5"),
+ ("circinus", "1.5"), ("sagitta", "1.4"), ("equuleus", "1.3"), ("crux", "1.2")]
+
+
+def urls_from_sitemap(xml: str) -> list[str]:
+ out = []
+ for loc in SITEMAP_LOC.findall(xml):
+ path = re.sub(r"^https?://[^/]+", "", loc)
+ if path.startswith("/en/"):
+ out.append(path)
+ return out
+
+
+def alias_corpus() -> list[tuple[str, int, str]]:
+ rows = [(f"/en/{a}/", 301, f"/en/{s}/") for a, s in ALIASES]
+ # 1.2 excluded: never had an RTD PDF artifact (Phase-0 finding, spec §15a) — pdf: null
+ rows += [(f"/_/downloads/en/{s}/pdf/", 301, f"/en/{s}/vyos-documentation.pdf")
+ for s in ["rolling", "1.5", "1.4", "1.3"]]
+ rows.append(("/_/downloads/en/latest/pdf/", 301, "/en/rolling/vyos-documentation.pdf"))
+ return rows
+
+
+class _NoRedirect(urllib.request.HTTPRedirectHandler):
+ """The parity checker must SEE 301s, not follow them (alias assertions)."""
+
+ def redirect_request(self, req, fp, code, msg, headers, newurl): # noqa: D401
+ return None
+
+
+_OPENER = urllib.request.build_opener(_NoRedirect)
+
+# Overridable in tests (monkeypatched to "http") so fetch() can be exercised end-to-end
+# against a real local http.server instead of requiring TLS for a unit test.
+_SCHEME = "https"
+
+
+# The port each scheme already implies, so `host` and `host:443` are not two origins.
+_DEFAULT_PORTS = {"https": 443, "http": 80}
+
+
+def _authority(value: str) -> tuple[str, str | None, int | None]:
+ """The normalized ORIGIN of a full URL or of a bare `host[:port]` argument.
+
+ An origin is scheme + host + port, and all three are returned: a credential scoped to an
+ https host must not match a plaintext http URL. Dropping the scheme made
+ `Access("p.invalid", ...)` apply to `http://p.invalid/`, so the ONE choke point that
+ decides whether to attach the service token would have attached it to a cleartext
+ request. Nothing constructs such a URL today — every URL in this module is built from
+ _SCHEME, so a single run is single-scheme — but the scope of a credential should not
+ depend on that staying true.
+
+ Two spellings of one origin still have to compare equal, because the two sides of this
+ comparison come from different places: one is a URL this module built, the other is
+ whatever an operator typed after --probe-host. Comparing (hostname, port) verbatim made
+ `p.invalid` and `p.invalid:443` distinct, so spelling the default port out cost the
+ credential its own scope — post-cutover, where --sitemap-host and --probe-host name the
+ same Access-gated host, that silently 403'd every sitemap fetch and the sweep then
+ reported an empty corpus as a pass. Normalized here:
+
+ * case — `scheme` and `hostname` are already lowercased by urlsplit; kept explicit for
+ the reader.
+ * the root label's trailing dot — `p.invalid.` names the same host as `p.invalid`.
+ * the scheme's default port → None, so `:443` under https (or `:80` under http) is not
+ a separate authority. Folded against the origin's OWN scheme, so http `:80` and
+ https `:443` stay the distinct origins they are.
+ * a bare argument carries no scheme, so it is read under _SCHEME — the scheme every
+ URL in this module is built with.
+
+ Deliberately NOT normalized: IDN/punycode equivalence (`ünïcode.example` against its
+ `xn--` form). Both hosts here are ASCII literals passed by CI, idna encoding carries its
+ own failure modes, and the safe direction for a credential-scoping test is to leave a
+ Unicode spelling not matching its punycode one rather than to guess an equivalence.
+ """
+ parts = urllib.parse.urlsplit(value if "://" in value else f"//{value}")
+ scheme = (parts.scheme or _SCHEME).lower()
+ host = parts.hostname.lower() if parts.hostname else None
+ if host and host.endswith("."):
+ host = host[:-1]
+ port = parts.port
+ if port is not None and port == _DEFAULT_PORTS.get(scheme):
+ port = None
+ return scheme, host, port
+
+
+@dataclasses.dataclass(frozen=True)
+class Access:
+ """A CF Access service token BOUND TO THE ONE HOST it may be presented to.
+
+ The binding is the point. This run talks to two hosts that are not the same party:
+ --probe-host is our Access-gated canary, while --sitemap-host is (pre-cutover)
+ docs.vyos.io, still served by ReadTheDocs. Credentials modelled as a bare
+ (id, secret) tuple carry no notion of destination, so a single `if access:` test in
+ the request builder sent our service token to BOTH — handing it to a third party on
+ every nightly sitemap fetch. Pairing the secret with its host makes the destination
+ check part of the credential rather than a rule each call site has to remember.
+ """
+
+ host: str
+ client_id: str
+ # repr=False: the default dataclass repr renders every field, so a failed assertion, a
+ # debug print or any exception that interpolates an Access would put the service token
+ # verbatim into CI logs — which are durable and, for this repo, world-readable. The id
+ # stays: it names WHICH token without being the credential, and losing it would make a
+ # scoping failure much harder to read. Secret is fetched via the attribute, never shown.
+ client_secret: str = dataclasses.field(repr=False)
+
+ def applies_to(self, url: str) -> bool:
+ """True only for a URL whose ORIGIN is this credential's host (see _authority)."""
+ return _authority(url) == _authority(self.host)
+
+
+def build_request(url: str, access: Access | None,
+ method: str = "HEAD") -> urllib.request.Request:
+ """The ONE place that attaches CF Access credentials to a request.
+
+ Every outbound request in this module goes through here, and the attach decision is
+ made PER DESTINATION, never per run. Two failure modes meet at this function and only
+ a host-scoped single choke point closes both:
+
+ * Credential leak. The sitemap host and the probe host are different parties
+ pre-cutover; an unscoped `if access:` mailed our service token to ReadTheDocs
+ once a night. `Access.applies_to()` makes that structurally impossible.
+ * Split-brain. The sitemap fetch used to build its own bare Request, so pointing
+ --sitemap-host at the Access-gated canary 403'd every sitemap while the probe
+ requests worked. Post-cutover both flags name the same host, and because the
+ scoping test is on the URL rather than on which caller asked, that configuration
+ still gets credentialed sitemap fetches with no extra wiring.
+ """
+ req = urllib.request.Request(url, method=method)
+ if access is not None and access.applies_to(url):
+ req.add_header("CF-Access-Client-Id", access.client_id)
+ req.add_header("CF-Access-Client-Secret", access.client_secret)
+ return req
+
+
+def fetch(host: str, path: str, access: Access | None, method: str = "HEAD"):
+ req = build_request(f"{_SCHEME}://{host}{path}", access, method)
+ try:
+ with _OPENER.open(req, timeout=30) as r:
+ return r.status, r.headers.get("Location")
+ except urllib.error.HTTPError as e: # 3xx land here with the no-redirect handler
+ return e.code, e.headers.get("Location")
+ except Exception: # noqa: BLE001 — DNS blip/timeout fails THIS probe, not the run
+ return 0, None
+
+
+def main() -> int:
+ ap = argparse.ArgumentParser()
+ ap.add_argument("--sitemap-host", required=True)
+ ap.add_argument("--probe-host", required=True)
+ ap.add_argument("--slugs", default=DEFAULT_SLUGS)
+ ap.add_argument("--report", type=Path, default=Path("parity-report.json"))
+ a = ap.parse_args()
+ # CF Access service-token credentials are read ONLY from the environment. They were
+ # also accepted as --access-id/--access-secret flags; that is removed rather than
+ # merely discouraged, because a value passed in argv is readable from the process table
+ # for the lifetime of the process and is captured verbatim by `set -x` traces, crash
+ # dumps and CI process listings. No call site used the flags (both workflows export the
+ # env vars), so there is nothing to migrate and no ergonomic loss worth the exposure.
+ # Access itself stays OPTIONAL: the sitemap host may be a public origin needing no token.
+ access_id = os.environ.get("CF_ACCESS_CLIENT_ID", "")
+ access_secret = os.environ.get("CF_ACCESS_CLIENT_SECRET", "")
+ if bool(access_id) != bool(access_secret):
+ # Half a service token is never usable — every probe would 403 and the run would
+ # report a wholly misleading "parity broken". Names only, never the values.
+ print("CF Access needs BOTH an id and a secret, or neither "
+ "(CF_ACCESS_CLIENT_ID, CF_ACCESS_CLIENT_SECRET)", file=sys.stderr)
+ return 2
+ # Bound to the PROBE host, and to nothing else. --probe-host is the host we own and
+ # gate with Access; --sitemap-host is whatever currently publishes the truth sitemaps,
+ # which pre-cutover is ReadTheDocs. Should the two flags name the same host — the
+ # post-cutover configuration — build_request() credentials the sitemap fetch too,
+ # because the test is on the destination and not on the call site.
+ access = Access(a.probe_host, access_id, access_secret) if access_id else None
+ failures: list[dict] = []
+ checked = 0
+
+ for slug in a.slugs.split(","):
+ # ONE request per sitemap. This used to probe the status with fetch() and then fetch
+ # the whole document a second time — two full GETs of a multi-thousand-URL sitemap per
+ # slug — and the body fetch hard-coded "https://", so the _SCHEME override (the hook
+ # the tests use to drive this path against a local plain-HTTP server) was ignored.
+ # Two things the single-call rewrite must NOT lose:
+ # 1. The discarded pre-check asserted status == 200 exactly. _OPENER raises
+ # HTTPError for non-2xx (3xx included — it refuses to follow redirects), but it
+ # RETURNS normally for any other 2xx, so a sitemap answering 204/206 would yield
+ # an empty corpus and the gate would pass having probed nothing. The explicit
+ # status check below restores that strictness.
+ # 2. CF Access credentials WHEN — and only when — the sitemap host is the host the
+ # token belongs to. A bare Request here 403'd a --sitemap-host pointed at the
+ # Access-gated canary; an unconditionally credentialed one posted the token to
+ # ReadTheDocs. build_request() decides per destination and settles both.
+ try:
+ with _OPENER.open(build_request(
+ f"{_SCHEME}://{a.sitemap_host}/en/{slug}/sitemap.xml", access, "GET"),
+ timeout=30) as r:
+ if r.status != 200:
+ failures.append({"path": f"/en/{slug}/sitemap.xml",
+ "reason": f"sitemap status {r.status}"})
+ continue
+ urls = urls_from_sitemap(r.read().decode())
+ except Exception as e: # noqa: BLE001 — record per-slug, keep sweeping; report ALWAYS written
+ failures.append({"path": f"/en/{slug}/sitemap.xml",
+ "reason": f"sitemap fetch error: {e}"})
+ continue
+ for path in urls:
+ checked += 1
+ st, _ = fetch(a.probe_host, path, access)
+ if st != 200:
+ failures.append({"path": path, "reason": f"status {st}"})
+
+ for path, want_status, want_loc in alias_corpus():
+ checked += 1
+ st, loc = fetch(a.probe_host, path, access)
+ if st != want_status or (loc or "") != want_loc:
+ failures.append({"path": path, "reason": f"got {st} → {loc}, want {want_status} → {want_loc}"})
+
+ a.report.write_text(json.dumps({"checked": checked, "failures": failures}, indent=2))
+ for f in failures:
+ print(f"PARITY-FAIL {f['path']}: {f['reason']}", file=sys.stderr)
+ print(f"checked={checked} failures={len(failures)}")
+ return 1 if failures else 0
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())
diff --git a/scripts/docs_gates/smoke.py b/scripts/docs_gates/smoke.py
new file mode 100644
index 00000000..6c526d98
--- /dev/null
+++ b/scripts/docs_gates/smoke.py
@@ -0,0 +1,236 @@
+"""Scoped pre-traffic smoke + post-promote probe (spec §7.1 steps 2/5).
+
+Probes ONE version's pages plus apex special paths through a host, presenting a
+CF Access service token. Header contract (§3.3): content probes assert
+X-Docs-Build == --expect-sha; apex probes assert X-Apex-Build presence only.
+
+Phase-2 obligation (authorized addition, not in the original spec text): the
+version's index.html probe also asserts the response body carries the
+`#vyos-search` mount div (docs/_templates/searchbox.html), guarding the
+Pagefind gate's silent-degrade failure mode — a build that forgot to set
+DOCS_VERSION_SLUG would otherwise ship stock RTD search without CI noticing.
+"""
+from __future__ import annotations
+
+import argparse
+import dataclasses
+import json
+import os
+import sys
+import time
+import urllib.request
+from pathlib import Path
+
+APEX_PATHS = ["/versions.json", "/healthz", "/robots.txt", "/sitemap.xml"]
+SEARCH_MOUNT_MARKER = 'id="vyos-search"'
+
+# Explicit UA so the gate never depends on a Cloudflare edge exemption for the default
+# Python-urllib UA. The Browser Integrity Check blocked that UA until a skip rule was added;
+# the gate must not silently rely on that rule surviving.
+USER_AGENT = "vyos-docs-smoke/1.0 (+https://github.com/vyos/vyos-documentation)"
+
+# Round-based retry (module-level so tests can shrink them). A freshly-deployed worker version
+# can lose a propagation race: for a few minutes a probe may be served by the PREVIOUS version
+# (wrong status / stale X-Docs-Build). Rather than fail-fast, each round re-probes ONLY the
+# still-failing probes — this preserves the full per-probe failure enumeration (diagnostic
+# value) while adding at most (MAX_ROUNDS - 1) inter-round sleeps. Envelope widened to 5 rounds
+# x 30s after an observed propagation wave outlasted 3 rounds x 20s (a path still stale at
+# round 3): 4 x 30s = 2 min now covers the observed 1-2+ min waves. The green path still costs
+# zero extra time (no retries), and DEADLINE_SECONDS=480 still bounds the worst case.
+MAX_ROUNDS = 5
+RETRY_SLEEP_SECONDS = 30
+DEADLINE_SECONDS = 480
+# Per-socket-op timeout (connect + read), capped down to the remaining deadline budget on each
+# probe so a probe that starts late cannot overshoot DEADLINE_SECONDS. Pages are small, so this
+# socket-op timeout also bounds body reads adequately — no separate body-read deadline is needed.
+PROBE_TIMEOUT_SECONDS = 30
+
+
+class _NoRedirect(urllib.request.HTTPRedirectHandler):
+ """Probes assert an EXACT status per-path (200 or 404) — following a 3xx would
+ silently swap the probed status for whatever the redirect target returns,
+ masking an accidental redirect where a direct 200/404 was expected. Mirrors
+ parity.py's _NoRedirect/_OPENER pattern."""
+
+ def redirect_request(self, req, fp, code, msg, headers, newurl): # noqa: D401
+ return None
+
+
+_OPENER = urllib.request.build_opener(_NoRedirect)
+
+
+@dataclasses.dataclass
+class Probe:
+ path: str
+ expect_status: int
+ assert_docs_build: bool
+ assert_apex_build: bool
+ assert_search_mount: bool = False
+
+
+def probe_plan(slug: str, pdf: str | None, critical: list[str]) -> list[Probe]:
+ # `critical` may itself list "index.html" (it does in critical-pages.txt); drop it so the
+ # index page is probed exactly once — as plan[0], the sole search-mount probe below.
+ critical = [rel for rel in critical if rel != "index.html"]
+ plan = [Probe(f"/en/{slug}/{rel}", 200, True, False) for rel in ["index.html", *critical]]
+ plan.append(Probe(f"/en/{slug}/pagefind/pagefind.js", 200, True, False))
+ if pdf:
+ # assert_docs_build=False: the PDF is the ONE content path that can legitimately be
+ # answered by the apex Worker instead of a branch content Worker. 1.3's PDF (29.2 MiB)
+ # exceeds the 25 MiB static-asset cap, so apex serves it straight from R2 (spec §5) and
+ # that response carries only etag / accept-ranges / content-type / content-length —
+ # X-Docs-Build is a content-Worker header apex never sets on it. Asserting it made the
+ # probe structurally unpassable for 1.3 (observed nightly: "detail=status+docs-build").
+ # The build SHA is still gated for this version: every HTML probe above asserts it.
+ plan.append(Probe(pdf, 200, False, False))
+ plan.append(Probe(f"/en/{slug}/definitely-missing-page-xyz.html", 404, False, False))
+ plan += [Probe(p, 200, False, True) for p in APEX_PATHS]
+ plan[0].assert_search_mount = True # plan[0] is always /en/<slug>/index.html
+ return plan
+
+
+def docs_build_ok(header_value: str | None, expect_sha: str) -> bool:
+ """SKIP sentinel (nightly sweep): header presence only; otherwise exact match."""
+ if expect_sha == "SKIP":
+ return header_value is not None
+ return header_value == expect_sha
+
+
+def search_mount_present(html: str) -> bool:
+ return SEARCH_MOUNT_MARKER in html
+
+
+def _probe_once(host: str, probe: Probe, expect_sha: str, access_id: str, access_secret: str,
+ timeout: float = PROBE_TIMEOUT_SECONDS,
+ ) -> tuple[bool, int | None, str | None, str | None]:
+ """One probe attempt. Returns (ok, status, docs_build, detail). `detail` names the failed
+ check(s) — "status" / "docs-build" / "apex-build" / "search-mount" joined by "+", or the
+ transport error text — and is None when ok. `timeout` is the per-socket-op deadline (connect
+ + read); run() caps it to the remaining budget so a late probe can't overshoot the overall
+ deadline. ANY exception in the open OR body-read path (including a transport error DURING
+ HTTPError.read()) is contained and yields (False, None, None, <error text>): a retryable
+ failure, never a traceback. The HTTPError response stream is always closed — it owns a
+ socket, so a bare e.read() would leak it."""
+ req = urllib.request.Request(f"https://{host}{probe.path}", method="GET")
+ req.add_header("CF-Access-Client-Id", access_id)
+ req.add_header("CF-Access-Client-Secret", access_secret)
+ req.add_header("User-Agent", USER_AGENT)
+ try:
+ try:
+ with _OPENER.open(req, timeout=timeout) as resp:
+ status, headers, body = resp.status, resp.headers, resp.read()
+ except urllib.error.HTTPError as e: # non-2xx still carries headers/body
+ with e: # HTTPError is file-like and owns the response socket — always close it
+ status, headers, body = e.code, e.headers, e.read()
+ except Exception as e: # noqa: BLE001 — open OR read failure → retryable probe result
+ return False, None, None, str(e)
+ reasons: list[str] = []
+ if status != probe.expect_status:
+ reasons.append("status")
+ if probe.assert_docs_build and not docs_build_ok(headers.get("X-Docs-Build"), expect_sha):
+ reasons.append("docs-build")
+ if probe.assert_apex_build and not headers.get("X-Apex-Build"):
+ reasons.append("apex-build")
+ if probe.assert_search_mount and not search_mount_present(
+ body.decode("utf-8", errors="replace")):
+ reasons.append("search-mount")
+ detail = "+".join(reasons) if reasons else None
+ return not reasons, status, headers.get("X-Docs-Build"), detail
+
+
+def run(host: str, slug: str, expect_sha: str, access_id: str, access_secret: str,
+ pdf: str | None, critical: list[str]) -> int:
+ """Probe the whole plan, then re-probe ONLY the still-failing probes each round (up to
+ MAX_ROUNDS, one RETRY_SLEEP_SECONDS between rounds). A probe passing in ANY round passes;
+ a single propagation blip served by the previous worker version cannot fail the gate.
+ DEADLINE_SECONDS bounds total wall-clock — on breach, unresolved probes count as failed."""
+ plan = probe_plan(slug, pdf, critical)
+ start = time.monotonic()
+ deadline = start + DEADLINE_SECONDS # absolute — a hard upper bound on total wall-clock
+ pending = list(plan) # probes not yet passed
+ detail_by_path: dict[str, str] = {} # last failure detail per path, for logging
+ deadline_hit = False
+
+ def _remaining() -> float:
+ return deadline - time.monotonic()
+
+ for round_num in range(1, MAX_ROUNDS + 1):
+ if not pending:
+ break
+ still_failing: list[Probe] = []
+ unprobed: list[Probe] = []
+ for i, probe in enumerate(pending):
+ remaining = _remaining() # checked before each probe
+ if remaining < 1: # < 1s budget: don't start a probe that could overshoot
+ deadline_hit = True
+ unprobed = pending[i:] # not reached this round → still unresolved
+ break
+ ok, status, docs_build, detail = _probe_once(
+ host, probe, expect_sha, access_id, access_secret,
+ timeout=min(PROBE_TIMEOUT_SECONDS, max(1, remaining)))
+ if ok:
+ continue
+ still_failing.append(probe)
+ detail_by_path[probe.path] = (
+ f"status={status} docs-build={docs_build} detail={detail}")
+ pending = still_failing + unprobed
+ if deadline_hit or not pending or round_num == MAX_ROUNDS:
+ break
+ remaining = _remaining() # checked before the inter-round sleep
+ if remaining <= 0: # no budget left → deadline path (skip the sleep)
+ deadline_hit = True
+ break
+ for probe in pending:
+ print(f"SMOKE-RETRY {probe.path}: round {round_num} "
+ f"{detail_by_path[probe.path]}", file=sys.stderr)
+ time.sleep(min(RETRY_SLEEP_SECONDS, remaining)) # never sleep past the deadline
+
+ if deadline_hit:
+ print("SMOKE-DEADLINE: overall deadline reached — remaining probes counted as failed",
+ file=sys.stderr)
+ for probe in pending:
+ print(f"SMOKE-FAIL {probe.path}: {detail_by_path.get(probe.path, 'unresolved')}",
+ file=sys.stderr)
+ failures = len(pending)
+ print(json.dumps({"failures": failures}))
+ return 1 if failures else 0
+
+
+def main() -> int:
+ ap = argparse.ArgumentParser()
+ ap.add_argument("--host", required=True)
+ ap.add_argument("--slug", required=True)
+ ap.add_argument("--expect-sha", required=True)
+ ap.add_argument("--pdf", default=None)
+ ap.add_argument("--critical-list", type=Path,
+ default=Path("scripts/docs_gates/critical-pages.txt"))
+ a = ap.parse_args()
+ # CF Access service-token credentials are read ONLY from the environment. They were also
+ # accepted as --access-id/--access-secret flags; that is removed rather than merely
+ # discouraged, because a value passed in argv publishes it in the process command line —
+ # readable from the process table for the lifetime of the process, and captured verbatim
+ # by `set -x` shell traces, crash dumps and process-listing tooling. No call site used
+ # the flags (docs-build.yml and docs-canary-qa.yml both export the env vars), so there is
+ # nothing to migrate. Neither value nor its length is ever echoed.
+ access_id = os.environ.get("CF_ACCESS_CLIENT_ID", "")
+ access_secret = os.environ.get("CF_ACCESS_CLIENT_SECRET", "")
+ missing = [name for name, value in (
+ ("CF_ACCESS_CLIENT_ID", access_id),
+ ("CF_ACCESS_CLIENT_SECRET", access_secret)) if not value]
+ if missing:
+ # The canary host is Access-gated, so an empty credential would turn every probe into
+ # an indistinguishable 403 — fail loudly on the cause instead. Names only, no values.
+ print(f"missing CF Access credentials: {', '.join(missing)}", file=sys.stderr)
+ return 2
+ # Strip BEFORE the comment test: an indented " # note" line is a comment, not a page that
+ # every deployable build must contain (it would fail the probe as a missing critical page).
+ # read_text() (rather than a bare open().read()) closes the handle deterministically,
+ # matching gates.py; the bare form leaked the descriptor until GC on any interpreter
+ # without CPython's refcounting.
+ lines = (line.strip() for line in a.critical_list.read_text().splitlines())
+ critical = [line for line in lines if line and not line.startswith("#")]
+ return run(a.host, a.slug, a.expect_sha, access_id, access_secret, a.pdf, critical)
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())
diff --git a/scripts/docs_gates/test_gates.py b/scripts/docs_gates/test_gates.py
new file mode 100644
index 00000000..e65be3af
--- /dev/null
+++ b/scripts/docs_gates/test_gates.py
@@ -0,0 +1,122 @@
+import json
+import sys
+from pathlib import Path
+
+import pytest
+from scripts.docs_gates import gates
+
+
+@pytest.fixture()
+def artifact(tmp_path: Path) -> Path:
+ root = tmp_path / "en" / "rolling"
+ root.mkdir(parents=True)
+ for i in range(50):
+ (root / f"p{i}.html").write_text(
+ f'<html><head><link rel="canonical" href="https://docs.vyos.io/en/rolling/p{i}.html"/></head><body>x</body></html>'
+ )
+ (root / "index.html").write_text(
+ '<html><head><link rel="canonical" href="https://docs.vyos.io/en/rolling/index.html"/></head></html>'
+ )
+ (root / "vyos-documentation.pdf").write_bytes(b"%PDF-1.4 fake")
+ (root / "pagefind").mkdir()
+ (root / "pagefind" / "pagefind.js").write_text("// index")
+ (root / "installation").mkdir()
+ (root / "installation" / "index.html").write_text(
+ '<html><head><link rel="canonical" href="https://docs.vyos.io/en/rolling/installation/index.html"/></head></html>'
+ )
+ return tmp_path
+
+
+def versions_arg(tmp_path: Path) -> Path:
+ """Hermetic stand-in for workers/versions.json — writes a fixture-local
+ manifest with the minimal schema the gates consume (mirrors the real
+ file's shape for the rolling entry) so tests never depend on, or break
+ from, edits to the repo file."""
+ p = tmp_path / "versions.json"
+ p.write_text(json.dumps({
+ "schema_version": 2,
+ "default_lang": "en",
+ "default_version": "rolling",
+ "languages": [{"code": "en", "label": "English"}],
+ "versions": [
+ {"slug": "rolling", "label": "Rolling (development)", "status": "dev",
+ "binding": "DOCS_ROLLING", "aliases": ["latest"],
+ "pdf": "/en/rolling/vyos-documentation.pdf"},
+ ],
+ }))
+ return p
+
+
+@pytest.fixture()
+def versions(tmp_path: Path) -> Path:
+ return versions_arg(tmp_path)
+
+
+def test_pass_on_good_artifact(artifact: Path, versions: Path):
+ rc = gates.run(artifact=artifact, slug="rolling", versions=versions,
+ previous_meta=None, critical=["index.html", "installation/index.html"])
+ assert rc == 0
+
+
+def test_fail_on_missing_critical_page(artifact: Path, versions: Path):
+ (artifact / "en/rolling/installation/index.html").unlink()
+ rc = gates.run(artifact=artifact, slug="rolling", versions=versions,
+ previous_meta=None, critical=["index.html", "installation/index.html"])
+ assert rc == 1
+
+
+def test_fail_on_count_collapse(artifact: Path, versions: Path, tmp_path: Path):
+ meta = tmp_path / "meta.json"
+ meta.write_text(json.dumps({"sha": "old", "page_count": 5000})) # previous build 100x bigger
+ rc = gates.run(artifact=artifact, slug="rolling", versions=versions,
+ previous_meta=meta, critical=["index.html"])
+ assert rc == 1
+
+
+def test_fail_on_alias_canonical(artifact: Path, versions: Path):
+ (artifact / "en/rolling/bad.html").write_text(
+ '<html><head><link rel="canonical" href="https://docs.vyos.io/en/latest/bad.html"/></head></html>'
+ )
+ rc = gates.run(artifact=artifact, slug="rolling", versions=versions,
+ previous_meta=None, critical=["index.html"])
+ assert rc == 1
+
+
+def test_fail_on_missing_canonical(artifact: Path, versions: Path):
+ (artifact / "en/rolling/nocanon.html").write_text(
+ '<html><head></head><body>no canonical link at all</body></html>'
+ )
+ rc = gates.run(artifact=artifact, slug="rolling", versions=versions,
+ previous_meta=None, critical=["index.html"])
+ assert rc == 1
+
+
+def test_fail_on_oversize_file(artifact: Path, versions: Path):
+ big = artifact / "en/rolling/huge.bin"
+ big.write_bytes(b"\0" * (26 * 1024 * 1024)) # > 25 MiB
+ rc = gates.run(artifact=artifact, slug="rolling", versions=versions,
+ previous_meta=None, critical=["index.html"])
+ assert rc == 1
+
+
+def test_fail_when_declared_pdf_missing(artifact: Path, versions: Path):
+ (artifact / "en/rolling/vyos-documentation.pdf").unlink()
+ rc = gates.run(artifact=artifact, slug="rolling", versions=versions,
+ previous_meta=None, critical=["index.html"])
+ assert rc == 1
+
+
+def test_critical_list_strips_before_testing_for_comments(monkeypatch, tmp_path, artifact):
+ # An INDENTED comment used to survive the `line.startswith("#")` test (applied to the
+ # UNSTRIPPED line) and become a live critical-page entry. A comment can never exist as a
+ # file, so it would block every deploy with "critical page missing: en/rolling/ # ...".
+ crit = tmp_path / "critical.txt"
+ crit.write_text("# leading comment\n # indented comment\n\n index.html \n")
+ seen: list[str] = []
+ monkeypatch.setattr(gates, "run",
+ lambda art, slug, versions, prev, critical: seen.extend(critical) or 0)
+ monkeypatch.setattr(sys, "argv", [
+ "gates", "--artifact", str(artifact), "--slug", "rolling",
+ "--versions", str(versions_arg(tmp_path)), "--critical-list", str(crit)])
+ assert gates.main() == 0
+ assert seen == ["index.html"]
diff --git a/scripts/docs_gates/test_parity.py b/scripts/docs_gates/test_parity.py
new file mode 100644
index 00000000..053eb79f
--- /dev/null
+++ b/scripts/docs_gates/test_parity.py
@@ -0,0 +1,340 @@
+import json
+import sys
+import urllib.error
+import urllib.request
+
+import pytest
+
+from scripts.docs_gates import parity
+from scripts.docs_gates.conftest import REDIRECT_LOCATION, REDIRECT_PATH
+
+
+def test_sitemap_url_extraction():
+ xml = ('<?xml version="1.0"?><urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">'
+ "<url><loc>https://docs.vyos.io/en/1.5/a.html</loc></url>"
+ "<url><loc>https://docs.vyos.io/en/1.5/b/</loc></url></urlset>")
+ assert parity.urls_from_sitemap(xml) == ["/en/1.5/a.html", "/en/1.5/b/"]
+
+
+def test_alias_corpus_includes_pdf_and_alias_rows():
+ rows = parity.alias_corpus()
+ assert ("/en/latest/", 301, "/en/rolling/") in rows
+ assert ("/_/downloads/en/1.5/pdf/", 301, "/en/1.5/vyos-documentation.pdf") in rows
+
+
+def test_fetch_never_follows_redirects_unit():
+ # unit-level sanity check on the handler class in isolation (kept alongside the
+ # end-to-end test below, which is what actually proves the opener is wired up)
+ handler = parity._NoRedirect()
+ assert handler.redirect_request(None, None, 301, "Moved", {}, "https://x/") is None
+
+
+def test_fetch_never_follows_redirects(redirect_http_server, monkeypatch):
+ # End-to-end: real local HTTP server returns a 301, exercised through the PUBLIC
+ # fetch() entrypoint (not just the handler class) — proves _OPENER is actually
+ # wired into fetch() and surfaces (301, Location) instead of following it.
+ monkeypatch.setattr(parity, "_SCHEME", "http")
+ status, location = parity.fetch(redirect_http_server, REDIRECT_PATH, None, "GET")
+ assert status == 301
+ assert location == REDIRECT_LOCATION
+
+
+def test_default_slugs_scoped_to_cf_built_versions():
+ # 1.3/1.2 have NO RTD sitemaps (spec §15a.5); legacy parity belongs to the
+ # snapshot repo's crawl-inventory job, not this sweep
+ assert parity.DEFAULT_SLUGS == "rolling,1.5,1.4"
+
+
+def test_fetch_records_transport_error_as_status_zero(monkeypatch):
+ # a DNS blip / timeout must fail the single probe, not abort the whole run
+ class _Boom:
+ def open(self, req, timeout=None):
+ raise urllib.error.URLError("dns blip")
+
+ monkeypatch.setattr(parity, "_OPENER", _Boom())
+ assert parity.fetch("host.invalid", "/en/rolling/", None) == (0, None)
+
+
+def test_main_always_writes_report_on_transport_errors(tmp_path, monkeypatch):
+ # sitemap status probe says 200, but the body fetch raises mid-sweep:
+ # the run must record per-slug failures, keep going, and STILL write the report
+ report = tmp_path / "parity-report.json"
+ monkeypatch.setattr(parity, "fetch", lambda *a, **k: (200, None))
+
+ def _boom(*a, **k):
+ raise urllib.error.URLError("timed out")
+
+ monkeypatch.setattr(parity._OPENER, "open", _boom)
+ monkeypatch.setattr(sys, "argv", ["parity", "--sitemap-host", "sitemap.invalid",
+ "--probe-host", "probe.invalid",
+ "--report", str(report)])
+ rc = parity.main()
+ assert rc == 1
+ data = json.loads(report.read_text())
+ assert data["failures"] # report written despite transport errors
+ assert any("sitemap" in f["reason"] for f in data["failures"])
+
+
+# --- CF Access credentials come from the ENVIRONMENT ONLY. The --access-id/--access-secret
+# flags were REMOVED: an argv-passed secret is readable from the process table and captured
+# by `set -x` traces. Access stays OPTIONAL here — the sitemap host may be public — but HALF
+# a service token is never usable, so an id/secret mismatch is rejected outright. ---
+
+def _parity_argv(monkeypatch, tmp_path, *extra):
+ monkeypatch.setattr(sys, "argv", ["parity", "--sitemap-host", "s.invalid",
+ "--probe-host", "p.invalid", "--slugs", "rolling",
+ "--report", str(tmp_path / "r.json"), *extra])
+
+
+def test_access_credentials_default_from_environment(monkeypatch, tmp_path):
+ _parity_argv(monkeypatch, tmp_path)
+ monkeypatch.setenv("CF_ACCESS_CLIENT_ID", "env-id")
+ monkeypatch.setenv("CF_ACCESS_CLIENT_SECRET", "env-secret")
+ seen: list[parity.Access | None] = []
+
+ def _probe(host, path, access, method="HEAD"):
+ seen.append(access)
+ return 200, None
+
+ monkeypatch.setattr(parity, "fetch", _probe)
+ monkeypatch.setattr(parity._OPENER, "open",
+ lambda *a, **k: _sitemap_response("<urlset></urlset>"))
+ parity.main()
+ # scoped to --probe-host, which is the only host the token may ever be presented to
+ assert parity.Access("p.invalid", "env-id", "env-secret") in seen
+
+
+def test_half_a_service_token_is_rejected(monkeypatch, tmp_path, capsys):
+ _parity_argv(monkeypatch, tmp_path)
+ monkeypatch.setenv("CF_ACCESS_CLIENT_ID", "only-an-id")
+ monkeypatch.delenv("CF_ACCESS_CLIENT_SECRET", raising=False)
+ monkeypatch.setattr(parity, "fetch", lambda *a, **k: (_ for _ in ()).throw(
+ AssertionError("must not probe with half a token")))
+ assert parity.main() == 2
+ assert "CF_ACCESS_CLIENT_SECRET" in capsys.readouterr().err
+
+
+def test_secret_bearing_flags_are_rejected_not_silently_ignored(monkeypatch, tmp_path):
+ # The flags are GONE, not deprecated. argparse must reject them outright so an operator
+ # reaching for the old muscle-memory invocation gets an error instead of a run that
+ # silently ignores the credential they passed and then 403s on every probe.
+ for flag, value in (("--access-id", "an-id"), ("--access-secret", "a-secret")):
+ _parity_argv(monkeypatch, tmp_path, flag, value)
+ monkeypatch.setenv("CF_ACCESS_CLIENT_ID", "env-id")
+ monkeypatch.setenv("CF_ACCESS_CLIENT_SECRET", "env-secret")
+ with pytest.raises(SystemExit) as exc:
+ parity.main()
+ assert exc.value.code == 2
+
+
+# --- The sitemap used to be fetched TWICE per slug (a status probe via fetch(), then the
+# body via a second GET) and the body fetch hard-coded "https://", ignoring _SCHEME. ---
+
+class _CountingSitemap:
+ """Records the Request objects the opener is handed, so a test can assert both the URL
+ (once per slug, honouring _SCHEME) and the CF Access headers actually attached to it."""
+
+ def __init__(self, status: int = 200) -> None:
+ self.requests: list[urllib.request.Request] = []
+ self.status = status
+
+ @property
+ def urls(self) -> list[str]:
+ return [r.full_url for r in self.requests]
+
+ def __call__(self, req, *a, **k):
+ self.requests.append(req)
+ return _sitemap_response(
+ '<urlset><url><loc>http://h/en/rolling/a.html</loc></url></urlset>',
+ status=self.status)
+
+
+def _sitemap_response(body: str, status: int = 200):
+ class _R:
+ def __init__(self):
+ self.status = status
+
+ def read(self):
+ return body.encode()
+
+ def __enter__(self):
+ return self
+
+ def __exit__(self, *a):
+ return False
+ return _R()
+
+
+def test_sitemap_fetched_once_per_slug_and_honours_the_scheme_override(monkeypatch, tmp_path):
+ _parity_argv(monkeypatch, tmp_path)
+ monkeypatch.delenv("CF_ACCESS_CLIENT_ID", raising=False)
+ monkeypatch.delenv("CF_ACCESS_CLIENT_SECRET", raising=False)
+ monkeypatch.setattr(parity, "_SCHEME", "http")
+ counter = _CountingSitemap()
+ monkeypatch.setattr(parity._OPENER, "open", counter)
+ monkeypatch.setattr(parity, "fetch", lambda *a, **k: (200, None))
+ parity.main()
+ assert counter.urls == ["http://s.invalid/en/rolling/sitemap.xml"] # once, and NOT https
+
+
+_ACCESS_HEADERS = ("Cf-access-client-id", "Cf-access-client-secret") # urllib capitalises
+
+
+def test_sitemap_host_that_is_not_the_probe_host_gets_NO_access_headers(monkeypatch, tmp_path):
+ # THE credential-scoping assertion, and the inverse of what this test used to demand.
+ # Pre-cutover the two flags name different parties: --sitemap-host is docs.vyos.io,
+ # still served by ReadTheDocs, while --probe-host is our Access-gated canary. Crediting
+ # every outbound request "because the run holds a token" handed our CF Access service
+ # token to a host we do not control, once every night.
+ _parity_argv(monkeypatch, tmp_path)
+ monkeypatch.setenv("CF_ACCESS_CLIENT_ID", "env-id")
+ monkeypatch.setenv("CF_ACCESS_CLIENT_SECRET", "env-secret")
+ counter = _CountingSitemap()
+ monkeypatch.setattr(parity._OPENER, "open", counter)
+ monkeypatch.setattr(parity, "fetch", lambda *a, **k: (200, None))
+ parity.main()
+ assert len(counter.requests) == 1
+ req = counter.requests[0]
+ assert req.full_url.startswith("https://s.invalid/") # the third-party host
+ for header in _ACCESS_HEADERS:
+ assert req.get_header(header) is None
+
+
+def test_sitemap_host_equal_to_the_probe_host_IS_credentialed(monkeypatch, tmp_path):
+ # The other direction, and the reason the scoping lives inside build_request() rather
+ # than at each call site: post-cutover both flags name the same Access-gated host and
+ # that sitemap fetch must still carry the token. A bare Request here (the shape before
+ # round 2) 403'd every sitemap, and the sweep then reported an empty corpus as a pass.
+ monkeypatch.setattr(sys, "argv", ["parity", "--sitemap-host", "p.invalid",
+ "--probe-host", "p.invalid", "--slugs", "rolling",
+ "--report", str(tmp_path / "r.json")])
+ monkeypatch.setenv("CF_ACCESS_CLIENT_ID", "env-id")
+ monkeypatch.setenv("CF_ACCESS_CLIENT_SECRET", "env-secret")
+ counter = _CountingSitemap()
+ monkeypatch.setattr(parity._OPENER, "open", counter)
+ monkeypatch.setattr(parity, "fetch", lambda *a, **k: (200, None))
+ parity.main()
+ req = counter.requests[0]
+ assert req.get_header("Cf-access-client-id") == "env-id"
+ assert req.get_header("Cf-access-client-secret") == "env-secret"
+
+
+def test_build_request_attaches_the_token_to_its_own_host_and_to_nothing_else():
+ # build_request() in isolation: one Access object, many destinations.
+ access = parity.Access("p.invalid", "an-id", "a-secret")
+ own = parity.build_request("https://P.Invalid/en/rolling/", access) # case-insensitive
+ assert own.get_header("Cf-access-client-id") == "an-id"
+ assert own.get_header("Cf-access-client-secret") == "a-secret"
+ for other in ("https://s.invalid/en/rolling/", # a different host entirely
+ "https://p.invalid.evil.example/en/", # suffix-extended lookalike
+ "https://notp.invalid/en/", # prefix-extended lookalike
+ "https://p.invalid:8443/en/rolling/"): # same name, different authority
+ for header in _ACCESS_HEADERS:
+ assert parity.build_request(other, access).get_header(header) is None
+ for header in _ACCESS_HEADERS: # no token configured at all
+ assert parity.build_request("https://p.invalid/", None).get_header(header) is None
+
+
+# --- The scoping test compares ORIGINS, not spellings. `p.invalid` and `p.invalid:443` are
+# the same HTTPS origin, and so is the trailing-dot FQDN form; comparing (hostname, port)
+# verbatim made all three distinct. The case that matters is post-cutover, where BOTH flags
+# name the same host: write either one with an explicit `:443` and the sitemap fetch silently
+# lost its token and 403'd — reverting the keeper case two tests up. ---
+
+def test_equivalent_spellings_of_one_origin_all_get_the_token():
+ for host, url in (("p.invalid", "https://p.invalid:443/en/rolling/"), # default port explicit
+ ("p.invalid:443", "https://p.invalid/en/rolling/"), # ...and the reverse
+ ("p.invalid:443", "https://p.invalid:443/en/"), # explicit on both
+ ("p.invalid.", "https://p.invalid/en/rolling/"), # trailing-dot FQDN
+ ("p.invalid", "https://p.invalid./en/rolling/"), # ...and the reverse
+ ("P.INVALID.:443", "https://p.invalid/en/")): # every axis at once
+ req = parity.build_request(url, parity.Access(host, "an-id", "a-secret"))
+ assert req.get_header("Cf-access-client-id") == "an-id", f"{host} vs {url}"
+ assert req.get_header("Cf-access-client-secret") == "a-secret", f"{host} vs {url}"
+
+
+def test_normalization_does_not_widen_the_scope_to_a_different_origin():
+ # The inverse pin: normalizing the default port and the trailing dot must not smear the
+ # comparison into matching anything else. A non-default port stays a distinct origin in
+ # BOTH directions, and a trailing dot on a lookalike is still a lookalike.
+ for host, url in (("p.invalid", "https://s.invalid:443/en/"), # different host, :443
+ ("p.invalid:8443", "https://p.invalid/en/"), # non-default on the cred
+ ("p.invalid", "https://p.invalid:8443/en/"), # non-default on the URL
+ ("p.invalid.", "https://p.invalid.evil.example./en/")): # dotted lookalike
+ for header in _ACCESS_HEADERS:
+ req = parity.build_request(url, parity.Access(host, "an-id", "a-secret"))
+ assert req.get_header(header) is None, f"{host} vs {url}"
+
+
+def test_the_default_port_that_normalizes_is_the_one_for_the_scheme_in_use(monkeypatch):
+ # A bare `host[:port]` argument carries no scheme, so the default it is compared against
+ # is the scheme every URL in this module is built with (_SCHEME) — not a hard-coded 443.
+ # Under the http override the tests use, 80 is the default and 443 is a real distinct port.
+ monkeypatch.setattr(parity, "_SCHEME", "http")
+ token = parity.Access("p.invalid:80", "an-id", "a-secret")
+ assert parity.build_request("http://p.invalid/en/", token).get_header(
+ "Cf-access-client-id") == "an-id"
+ assert parity.build_request("http://p.invalid:443/en/", token).get_header(
+ "Cf-access-client-id") is None
+
+
+def test_probe_host_written_with_an_explicit_port_still_credentials_its_own_sitemap(
+ monkeypatch, tmp_path):
+ # The end-to-end shape of the bug: post-cutover both flags name the same host, but one
+ # of them spells the default port out. Before origin normalization the sitemap request
+ # went out bare, 403'd behind Access, and the sweep reported an empty corpus as a pass.
+ monkeypatch.setattr(sys, "argv", ["parity", "--sitemap-host", "p.invalid",
+ "--probe-host", "p.invalid:443", "--slugs", "rolling",
+ "--report", str(tmp_path / "r.json")])
+ monkeypatch.setenv("CF_ACCESS_CLIENT_ID", "env-id")
+ monkeypatch.setenv("CF_ACCESS_CLIENT_SECRET", "env-secret")
+ counter = _CountingSitemap()
+ monkeypatch.setattr(parity._OPENER, "open", counter)
+ monkeypatch.setattr(parity, "fetch", lambda *a, **k: (200, None))
+ parity.main()
+ req = counter.requests[0]
+ assert req.get_header("Cf-access-client-id") == "env-id"
+ assert req.get_header("Cf-access-client-secret") == "env-secret"
+
+
+def test_a_plaintext_http_url_never_gets_an_https_scoped_token():
+ # An ORIGIN is scheme + host + port. Comparing only (host, port) left the transport out
+ # of the credential's scope, so a token bound to an https host also applied to the
+ # cleartext http URL of the same name — the choke point would have attached the service
+ # token to a request that puts it on the wire in plaintext. `http://p.invalid:80/` is the
+ # sharp case: 80 folds to None under http, so the authority-only comparison matched the
+ # https-scoped ("p.invalid", None) exactly.
+ access = parity.Access("p.invalid", "an-id", "a-secret") # bare host → _SCHEME (https)
+ for url in ("http://p.invalid/en/rolling/", "http://p.invalid:80/en/rolling/"):
+ for header in _ACCESS_HEADERS:
+ assert parity.build_request(url, access).get_header(header) is None, url
+ # control, same test: its own scheme still gets the token
+ assert parity.build_request("https://p.invalid/en/rolling/", access).get_header(
+ "Cf-access-client-id") == "an-id"
+
+
+def test_the_service_token_is_not_rendered_by_repr():
+ # The default dataclass repr renders every field. A failed assertion, a debug print or an
+ # exception that interpolates an Access would then put the token into CI output, which is
+ # durable. str() delegates to __repr__, so it covers f-string interpolation too.
+ access = parity.Access("p.invalid", "an-id", "sekrit-must-not-be-rendered")
+ for rendered in (repr(access), str(access), f"{access}"):
+ assert "sekrit-must-not-be-rendered" not in rendered
+ assert access.client_secret == "sekrit-must-not-be-rendered" # still readable as a field
+ assert "an-id" in repr(access) # the id is NOT the credential; keep it for diagnosis
+
+
+def test_non_200_sitemap_is_a_failure_not_an_empty_corpus(monkeypatch, tmp_path):
+ # _OPENER only raises for non-2xx. A sitemap answering 204 (or any other 2xx) returned
+ # normally with an empty/irrelevant body, so the corpus came back empty and the parity
+ # gate PASSED having probed nothing at all — the exact silent-degrade the discarded
+ # exact-200 pre-check existed to prevent.
+ _parity_argv(monkeypatch, tmp_path)
+ monkeypatch.delenv("CF_ACCESS_CLIENT_ID", raising=False)
+ monkeypatch.delenv("CF_ACCESS_CLIENT_SECRET", raising=False)
+ report = tmp_path / "r.json"
+ monkeypatch.setattr(parity._OPENER, "open", _CountingSitemap(status=204))
+ monkeypatch.setattr(parity, "fetch", lambda *a, **k: (200, None))
+ assert parity.main() == 1
+ data = json.loads(report.read_text())
+ assert any(f["reason"] == "sitemap status 204" for f in data["failures"])
diff --git a/scripts/docs_gates/test_smoke.py b/scripts/docs_gates/test_smoke.py
new file mode 100644
index 00000000..790f953d
--- /dev/null
+++ b/scripts/docs_gates/test_smoke.py
@@ -0,0 +1,483 @@
+from __future__ import annotations
+
+import io
+import sys
+import urllib.error
+import urllib.request
+from email.message import Message
+from urllib.parse import urlsplit
+
+import pytest
+
+from scripts.docs_gates import smoke
+from scripts.docs_gates.conftest import REDIRECT_LOCATION, REDIRECT_PATH
+
+
+def test_probe_plan_scoped_to_slug():
+ plan = smoke.probe_plan("1.5", pdf="/en/1.5/vyos-documentation.pdf",
+ critical=["index.html", "cli/index.html"])
+ urls = [p.path for p in plan]
+ assert "/en/1.5/index.html" in urls
+ assert "/en/1.5/cli/index.html" in urls
+ assert "/en/1.5/vyos-documentation.pdf" in urls
+ assert "/en/1.5/definitely-missing-page-xyz.html" in urls # 404-status probe
+ assert "/versions.json" in urls and "/healthz" in urls # apex specials
+ assert not any(u.startswith("/en/rolling/") for u in urls) # scoped (§7.1.2)
+
+
+def test_assertions_follow_header_contract():
+ plan = smoke.probe_plan("1.5", pdf=None, critical=["index.html"])
+ content = next(p for p in plan if p.path == "/en/1.5/index.html")
+ assert content.expect_status == 200 and content.assert_docs_build is True
+ apex = next(p for p in plan if p.path == "/versions.json")
+ assert apex.assert_docs_build is False and apex.assert_apex_build is True
+ missing = next(p for p in plan if "definitely-missing" in p.path)
+ assert missing.expect_status == 404
+
+
+def test_skip_sentinel_relaxes_sha_to_presence_only():
+ # expect_sha == "SKIP" (nightly sweep, Task 3.5): header must be PRESENT but any value passes
+ assert smoke.docs_build_ok("anything", expect_sha="SKIP") is True
+ assert smoke.docs_build_ok(None, expect_sha="SKIP") is False
+ assert smoke.docs_build_ok("abc", expect_sha="abc") is True
+ assert smoke.docs_build_ok("abc", expect_sha="def") is False
+
+
+# --- Phase-2 obligation (authorized addition, not in the original brief): the
+# index.html probe must assert the CF-built HTML carries the `#vyos-search`
+# mount div, so CI catches a build that silently forgot to set
+# DOCS_VERSION_SLUG (which would ship stock RTD search instead of Pagefind). ---
+
+def test_probe_plan_asserts_search_mount_on_index_only():
+ plan = smoke.probe_plan("1.5", pdf=None, critical=["index.html", "cli/index.html"])
+ index = next(p for p in plan if p.path == "/en/1.5/index.html")
+ assert index.assert_search_mount is True
+ other_content = [p for p in plan if p.path != "/en/1.5/index.html" and p.assert_docs_build]
+ assert other_content and all(p.assert_search_mount is False for p in other_content)
+
+
+# --- CodeRabbit finding: smoke's probe requests must mirror parity.py's _NoRedirect
+# opener — an exact-status probe (200/404) that silently followed a 3xx would report
+# whatever the redirect target returns instead of the redirect itself. ---
+
+def test_opener_observes_redirect_directly_not_followed(redirect_http_server):
+ # End-to-end: a real local HTTP server returns a 301, opened through smoke.py's
+ # module-level _OPENER (the exact object `run()` uses) — proves it's actually
+ # wired to refuse the redirect, matching run()'s HTTPError-catch handling of 3xx.
+ req = urllib.request.Request(f"http://{redirect_http_server}{REDIRECT_PATH}", method="GET")
+ try:
+ smoke._OPENER.open(req, timeout=5)
+ raise AssertionError("expected HTTPError for a 301 with the no-redirect opener")
+ except urllib.error.HTTPError as e:
+ assert e.code == 301
+ assert e.headers.get("Location") == REDIRECT_LOCATION
+
+
+def test_search_mount_present():
+ assert smoke.search_mount_present('<div id="vyos-search" role="search"></div>') is True
+ assert smoke.search_mount_present('<html><body>no search here</body></html>') is False
+
+
+# --- Hardening: explicit UA + round-based retry with an overall deadline. Mock at the _OPENER
+# boundary (the object _probe_once() opens through, mirroring the redirect test above which
+# drives smoke._OPENER directly). run()-level round tests inject a small plan via probe_plan and
+# a path-keyed opener; sleeps are spied (or constants shrunk) so nothing wall-clocks. ---
+
+
+class _FakeResp:
+ """Stand-in for what _OPENER.open() yields: a context manager exposing .status /
+ .headers / .read()."""
+
+ def __init__(self, status: int, headers: dict[str, str], body: bytes = b""):
+ self.status = status
+ self.headers = headers
+ self._body = body
+
+ def read(self) -> bytes:
+ return self._body
+
+ def __enter__(self) -> "_FakeResp":
+ return self
+
+ def __exit__(self, *exc: object) -> bool:
+ return False
+
+
+class _FakeOpener:
+ """Yields queued responses in order; an Exception item is raised (models a transport error,
+ or a non-2xx delivered as HTTPError). Records each opened Request + the timeout it was
+ called with, for assertions."""
+
+ def __init__(self, responses: list[object]):
+ self._responses = list(responses)
+ self.calls: list[urllib.request.Request] = []
+ self.timeouts: list[float | None] = []
+
+ def open(self, req: urllib.request.Request, timeout: float | None = None) -> object:
+ self.calls.append(req)
+ self.timeouts.append(timeout)
+ item = self._responses.pop(0)
+ if isinstance(item, Exception):
+ raise item
+ return item
+
+
+class _RecordingBody(io.BytesIO):
+ """HTTPError.fp stand-in: records close() (so tests can assert the response stream is
+ closed) and can optionally raise on read (a transport error DURING body read). urllib's
+ addbase binds read through to fp and closes fp on close(), so overriding them here is what
+ e.read() / closing e actually hit."""
+
+ def __init__(self, data: bytes = b"", raise_on_read: bool = False):
+ super().__init__(data)
+ self.close_calls = 0
+ self._raise_on_read = raise_on_read
+
+ def read(self, *args: object) -> bytes: # noqa: D401
+ if self._raise_on_read:
+ raise OSError("reset during body read")
+ return super().read(*args)
+
+ def close(self) -> None:
+ self.close_calls += 1
+ super().close()
+
+
+def _http_error_read_boom(code: int) -> urllib.error.HTTPError:
+ return urllib.error.HTTPError(
+ "https://host.invalid/x", code, "msg", Message(), _RecordingBody(raise_on_read=True))
+
+
+class _PathOpener:
+ """Opener keyed by request path: each path maps to a queue consumed one item per probe of
+ that path (item = _FakeResp, or an Exception that is raised). Records probed paths in order,
+ so round scoping / retry counts are assertable across rounds that re-probe only failures."""
+
+ def __init__(self, by_path: dict[str, list[object]]):
+ self._by_path = {p: list(v) for p, v in by_path.items()}
+ self.calls: list[str] = []
+ self.timeouts: list[float | None] = []
+
+ def open(self, req: urllib.request.Request, timeout: float | None = None) -> object:
+ path = urlsplit(req.full_url).path
+ self.calls.append(path)
+ self.timeouts.append(timeout)
+ item = self._by_path[path].pop(0)
+ if isinstance(item, Exception):
+ raise item
+ return item
+
+
+def _plan(paths: list[str]) -> list[smoke.Probe]:
+ """Minimal status-only plan (no build/search assertions) for run() round tests."""
+ return [smoke.Probe(p, 200, assert_docs_build=False, assert_apex_build=False) for p in paths]
+
+
+def test_probe_sends_explicit_user_agent(monkeypatch):
+ opener = _FakeOpener([_FakeResp(200, {"X-Docs-Build": "sha1"})])
+ monkeypatch.setattr(smoke, "_OPENER", opener)
+ probe = smoke.Probe("/en/1.5/index.html", 200, assert_docs_build=True, assert_apex_build=False)
+ ok, *_ = smoke._probe_once("host.example", probe, "sha1", "cf-id", "cf-secret")
+ assert ok is True
+ req = opener.calls[0]
+ assert req.get_header("User-agent") == smoke.USER_AGENT # urllib capitalizes the key
+ assert req.get_header("Cf-access-client-id") == "cf-id" # CF-Access headers still sent
+
+
+def test_transport_error_recovers_in_second_round(monkeypatch):
+ monkeypatch.setattr(smoke, "probe_plan",
+ lambda slug, pdf, critical: _plan(["/en/rolling/index.html"]))
+ monkeypatch.setattr(smoke, "RETRY_SLEEP_SECONDS", 0)
+ opener = _PathOpener({
+ "/en/rolling/index.html": [OSError("dns hiccup"), _FakeResp(200, {})],
+ })
+ monkeypatch.setattr(smoke, "_OPENER", opener)
+ assert smoke.run("host", "rolling", "sha", "id", "sec", None, []) == 0
+ # round 1 transport error, round 2 success — the same path is re-probed
+ assert opener.calls == ["/en/rolling/index.html", "/en/rolling/index.html"]
+
+
+def test_httperror_read_crash_is_contained_as_failure(monkeypatch):
+ # agy-critical: a transport error DURING e.read() must not escape as a traceback.
+ monkeypatch.setattr(smoke, "_OPENER", _FakeOpener([_http_error_read_boom(502)]))
+ probe = smoke.Probe("/en/rolling/index.html", 200,
+ assert_docs_build=False, assert_apex_build=False)
+ ok, status, docs_build, detail = smoke._probe_once("host", probe, "sha", "id", "sec")
+ assert ok is False
+ assert status is None and docs_build is None
+ assert detail is not None and "reset during body read" in detail
+
+
+def test_sleeps_once_per_inter_round_gap_not_per_probe(monkeypatch):
+ monkeypatch.setattr(smoke, "probe_plan",
+ lambda slug, pdf, critical: _plan(["/a", "/b", "/c"]))
+ sleeps: list[float] = []
+ monkeypatch.setattr(smoke.time, "sleep", lambda s: sleeps.append(s))
+ # /a and /b fail round 1 then pass round 2; /c passes round 1
+ opener = _PathOpener({
+ "/a": [_FakeResp(500, {}), _FakeResp(200, {})],
+ "/b": [_FakeResp(500, {}), _FakeResp(200, {})],
+ "/c": [_FakeResp(200, {})],
+ })
+ monkeypatch.setattr(smoke, "_OPENER", opener)
+ assert smoke.run("host", "rolling", "sha", "id", "sec", None, []) == 0
+ assert sleeps == [smoke.RETRY_SLEEP_SECONDS] # ONE inter-round sleep despite 2 failing probes
+
+
+def test_second_round_reprobes_only_the_failed_path(monkeypatch):
+ monkeypatch.setattr(smoke, "probe_plan",
+ lambda slug, pdf, critical: _plan(["/a", "/b", "/c"]))
+ monkeypatch.setattr(smoke.time, "sleep", lambda s: None)
+ opener = _PathOpener({
+ "/a": [_FakeResp(200, {})], # passes round 1
+ "/b": [_FakeResp(500, {}), _FakeResp(200, {})], # fails r1, passes r2
+ "/c": [_FakeResp(200, {})], # passes round 1
+ })
+ monkeypatch.setattr(smoke, "_OPENER", opener)
+ assert smoke.run("host", "rolling", "sha", "id", "sec", None, []) == 0
+ assert opener.calls == ["/a", "/b", "/c", "/b"] # only the failed path is re-probed
+
+
+def test_run_reports_failure_count_and_nonzero_exit(monkeypatch, capsys):
+ monkeypatch.setattr(smoke, "probe_plan", lambda slug, pdf, critical: _plan(["/a", "/b"]))
+ monkeypatch.setattr(smoke, "MAX_ROUNDS", 1) # single round, no retries
+ monkeypatch.setattr(smoke, "_OPENER", _PathOpener({
+ "/a": [_FakeResp(200, {})],
+ "/b": [_FakeResp(500, {})],
+ }))
+ assert smoke.run("host", "rolling", "sha", "id", "sec", None, []) == 1
+ assert '{"failures": 1}' in capsys.readouterr().out
+
+
+def test_run_reports_clean_and_zero_exit(monkeypatch, capsys):
+ monkeypatch.setattr(smoke, "probe_plan", lambda slug, pdf, critical: _plan(["/a", "/b"]))
+ monkeypatch.setattr(smoke, "_OPENER", _PathOpener({
+ "/a": [_FakeResp(200, {})],
+ "/b": [_FakeResp(200, {})],
+ }))
+ assert smoke.run("host", "rolling", "sha", "id", "sec", None, []) == 0
+ assert '{"failures": 0}' in capsys.readouterr().out
+
+
+def test_deadline_counts_unresolved_as_failed(monkeypatch, capsys):
+ monkeypatch.setattr(smoke, "probe_plan", lambda slug, pdf, critical: _plan(["/a", "/b"]))
+ monkeypatch.setattr(smoke, "DEADLINE_SECONDS", 0) # trips before the first probe
+ opener = _PathOpener({"/a": [_FakeResp(200, {})], "/b": [_FakeResp(200, {})]})
+ monkeypatch.setattr(smoke, "_OPENER", opener)
+ assert smoke.run("host", "rolling", "sha", "id", "sec", None, []) == 1
+ captured = capsys.readouterr()
+ assert "SMOKE-DEADLINE" in captured.err
+ assert '{"failures": 2}' in captured.out
+ assert opener.calls == [] # deadline reached before any probe ran
+
+
+def test_probe_plan_dedups_index_html():
+ plan = smoke.probe_plan("rolling", None, ["index.html", "cli.html"])
+ index_probes = [p for p in plan if p.path == "/en/rolling/index.html"]
+ assert len(index_probes) == 1 # not duplicated by the critical list
+ assert index_probes[0].assert_search_mount is True # still the single search-mount probe
+ assert sum(1 for p in plan if p.path == "/en/rolling/cli.html") == 1 # cli.html preserved
+
+
+# --- CR round 2: close the HTTPError response stream (it owns a socket) + name the failed
+# assertion in retry/fail logs. ---
+
+def test_httperror_response_is_closed(monkeypatch):
+ # HTTPError owns the response socket; the expected-404 path must close it, not leak.
+ body = _RecordingBody(b"not found")
+ monkeypatch.setattr(smoke, "_OPENER", _FakeOpener([
+ urllib.error.HTTPError("https://host.invalid/x", 404, "msg", Message(), body)]))
+ probe = smoke.Probe("/en/rolling/missing.html", 404,
+ assert_docs_build=False, assert_apex_build=False)
+ ok, *_ = smoke._probe_once("host", probe, "sha", "id", "sec")
+ assert ok is True # 404 expected → passes
+ assert body.close_calls >= 1 # response stream closed (no socket leak / ResourceWarning)
+
+
+def test_httperror_response_is_closed_even_when_read_raises(monkeypatch):
+ body = _RecordingBody(raise_on_read=True)
+ monkeypatch.setattr(smoke, "_OPENER", _FakeOpener([
+ urllib.error.HTTPError("https://host.invalid/x", 502, "msg", Message(), body)]))
+ probe = smoke.Probe("/en/rolling/index.html", 200,
+ assert_docs_build=False, assert_apex_build=False)
+ ok, _, _, detail = smoke._probe_once("host", probe, "sha", "id", "sec")
+ assert ok is False
+ assert detail is not None and "reset during body read" in detail
+ assert body.close_calls >= 1 # close still attempted despite the read raising
+
+
+def test_apex_build_failure_is_named_in_logs(monkeypatch, capsys):
+ # 200 but no X-Apex-Build: without a named detail the log read "status=200 docs-build=None".
+ probe = smoke.Probe("/versions.json", 200, assert_docs_build=False, assert_apex_build=True)
+ monkeypatch.setattr(smoke, "probe_plan", lambda slug, pdf, critical: [probe])
+ monkeypatch.setattr(smoke, "MAX_ROUNDS", 1)
+ monkeypatch.setattr(smoke, "_OPENER", _PathOpener({"/versions.json": [_FakeResp(200, {})]}))
+ assert smoke.run("host", "rolling", "sha", "id", "sec", None, []) == 1
+ err = capsys.readouterr().err
+ assert "SMOKE-FAIL /versions.json:" in err
+ assert "detail=apex-build" in err # names the failed check
+ assert "status=200" in err # existing fields preserved
+
+
+def test_search_mount_failure_is_named_in_logs(monkeypatch, capsys):
+ # 200 + correct build, but the HTML lacks the #vyos-search mount div.
+ probe = smoke.Probe("/en/rolling/index.html", 200, assert_docs_build=True,
+ assert_apex_build=False, assert_search_mount=True)
+ monkeypatch.setattr(smoke, "probe_plan", lambda slug, pdf, critical: [probe])
+ monkeypatch.setattr(smoke, "MAX_ROUNDS", 1)
+ monkeypatch.setattr(smoke, "_OPENER", _PathOpener({
+ "/en/rolling/index.html": [_FakeResp(200, {"X-Docs-Build": "goodsha"},
+ b"<html>no search</html>")]}))
+ assert smoke.run("host", "rolling", "goodsha", "id", "sec", None, []) == 1
+ assert "detail=search-mount" in capsys.readouterr().err
+
+
+def test_probe_once_detail_joins_multiple_failed_checks(monkeypatch):
+ # wrong status AND missing X-Apex-Build → detail "status+apex-build"
+ monkeypatch.setattr(smoke, "_OPENER", _FakeOpener([_FakeResp(500, {})]))
+ probe = smoke.Probe("/versions.json", 200, assert_docs_build=False, assert_apex_build=True)
+ ok, _, _, detail = smoke._probe_once("host", probe, "sha", "id", "sec")
+ assert ok is False
+ assert detail == "status+apex-build"
+
+
+# --- Adversarial round: DEADLINE_SECONDS is a HARD bound — the per-probe socket timeout and the
+# inter-round sleep are both capped to the remaining budget so neither can overshoot it. ---
+
+def test_probe_timeout_capped_to_remaining_budget(monkeypatch):
+ monkeypatch.setattr(smoke, "probe_plan", lambda slug, pdf, critical: _plan(["/a"]))
+ monkeypatch.setattr(smoke, "DEADLINE_SECONDS", 5) # << PROBE_TIMEOUT_SECONDS (30)
+ opener = _PathOpener({"/a": [_FakeResp(200, {})]})
+ monkeypatch.setattr(smoke, "_OPENER", opener)
+ smoke.run("host", "rolling", "sha", "id", "sec", None, [])
+ assert opener.timeouts, "the probe recorded the timeout it was opened with"
+ t = opener.timeouts[0]
+ assert 1 <= t <= smoke.DEADLINE_SECONDS # capped DOWN to the ~5s remaining budget
+ assert t < smoke.PROBE_TIMEOUT_SECONDS # NOT the full 30s socket timeout
+
+
+def test_inter_round_sleep_capped_to_remaining_budget(monkeypatch):
+ monkeypatch.setattr(smoke, "probe_plan", lambda slug, pdf, critical: _plan(["/a"]))
+ monkeypatch.setattr(smoke, "DEADLINE_SECONDS", 10) # << RETRY_SLEEP_SECONDS (30)
+ sleeps: list[float] = []
+ monkeypatch.setattr(smoke.time, "sleep", lambda s: sleeps.append(s))
+ opener = _PathOpener({"/a": [_FakeResp(500, {}), _FakeResp(200, {})]}) # fail r1, pass r2
+ monkeypatch.setattr(smoke, "_OPENER", opener)
+ assert smoke.run("host", "rolling", "sha", "id", "sec", None, []) == 0
+ assert len(sleeps) == 1
+ assert 0 < sleeps[0] <= smoke.DEADLINE_SECONDS # capped to the ~10s remaining budget
+ assert sleeps[0] < smoke.RETRY_SLEEP_SECONDS # NOT the full 30s sleep
+
+
+# --- The PDF probe was structurally unpassable: it asserted X-Docs-Build, but 1.3's PDF is
+# served by the APEX Worker straight from R2 (spec §5, 29.2 MiB > the 25 MiB asset cap) and
+# that path sets only etag / accept-ranges / content-type / content-length. Observed nightly:
+# "/en/1.3/vyos-documentation.pdf: status=206 docs-build=None detail=status+docs-build". ---
+
+def test_pdf_probe_does_not_assert_docs_build():
+ plan = smoke.probe_plan("1.3", pdf="/en/1.3/vyos-documentation.pdf", critical=["index.html"])
+ pdf = next(p for p in plan if p.path.endswith(".pdf"))
+ assert pdf.assert_docs_build is False # apex's R2 path legitimately never sets it
+ assert pdf.assert_apex_build is False # nor does the content Worker set X-Apex-Build
+ # ...but the build SHA is still gated for this version, via the HTML probes:
+ assert next(p for p in plan if p.path.endswith("/index.html")).assert_docs_build is True
+
+
+def test_pdf_probe_still_demands_an_exact_200():
+ # The 206 seen alongside the docs-build failure was an apex defect (a 206 answered to a
+ # request carrying no Range header), fixed in workers/apex/src/index.ts — NOT something
+ # this gate should learn to tolerate.
+ plan = smoke.probe_plan("1.3", pdf="/en/1.3/vyos-documentation.pdf", critical=[])
+ assert next(p for p in plan if p.path.endswith(".pdf")).expect_status == 200
+
+
+# --- CF Access credentials: env by default (argv publishes secrets to the process table),
+# flags as a manual fallback, and an empty value is rejected rather than sent as a blank
+# header (every probe would then 403 and the report would blame the wrong thing). ---
+
+def _argv(monkeypatch, tmp_path, *extra):
+ crit = tmp_path / "critical.txt"
+ crit.write_text("index.html\n")
+ monkeypatch.setattr(sys, "argv", ["smoke", "--host", "h", "--slug", "rolling",
+ "--expect-sha", "SKIP",
+ "--critical-list", str(crit), *extra])
+ return crit
+
+
+def test_access_credentials_default_from_environment(monkeypatch, tmp_path):
+ _argv(monkeypatch, tmp_path)
+ monkeypatch.setenv("CF_ACCESS_CLIENT_ID", "env-id")
+ monkeypatch.setenv("CF_ACCESS_CLIENT_SECRET", "env-secret")
+ seen: dict[str, str] = {}
+ monkeypatch.setattr(smoke, "run",
+ lambda host, slug, sha, aid, asec, pdf, critical:
+ seen.update(id=aid, secret=asec) or 0)
+ assert smoke.main() == 0
+ assert seen == {"id": "env-id", "secret": "env-secret"}
+
+
+def test_secret_bearing_flags_are_rejected_not_silently_ignored(monkeypatch, tmp_path):
+ # --access-id/--access-secret are GONE, not deprecated: an argv-passed secret is readable
+ # from the process table and captured verbatim by `set -x` traces. argparse must reject
+ # them so the old muscle-memory invocation errors out instead of silently ignoring the
+ # credential the operator passed and then 403ing on every probe.
+ for flag, value in (("--access-id", "an-id"), ("--access-secret", "a-secret")):
+ _argv(monkeypatch, tmp_path, flag, value)
+ monkeypatch.setenv("CF_ACCESS_CLIENT_ID", "env-id")
+ monkeypatch.setenv("CF_ACCESS_CLIENT_SECRET", "env-secret")
+ monkeypatch.setattr(smoke, "run", lambda *a, **k: pytest.fail("must not probe"))
+ with pytest.raises(SystemExit) as exc:
+ smoke.main()
+ assert exc.value.code == 2
+
+
+def test_critical_list_is_read_through_a_path_without_resource_warnings(monkeypatch, tmp_path):
+ # `open(a.critical_list).read()` left the descriptor to be closed by GC; --critical-list
+ # is now `type=Path` and read via Path.read_text(), which closes deterministically.
+ # HONEST SCOPE: this is not a strict regression test for the close itself — CPython's
+ # refcounting also closes the bare-open form immediately, so no ResourceWarning fires
+ # either way and this test passes against the pre-fix source (verified). What it DOES
+ # pin is the argparse `type=Path` change (a str would have no .read_text()) plus
+ # warning-free reading on interpreters without refcounting, e.g. PyPy, where the
+ # bare-open form genuinely leaks until GC.
+ import warnings
+
+ crit = _argv(monkeypatch, tmp_path)
+ monkeypatch.setenv("CF_ACCESS_CLIENT_ID", "id")
+ monkeypatch.setenv("CF_ACCESS_CLIENT_SECRET", "sec")
+ seen: list[str] = []
+ monkeypatch.setattr(smoke, "run",
+ lambda host, slug, sha, aid, asec, pdf, critical:
+ seen.extend(critical) or 0)
+ with warnings.catch_warnings():
+ warnings.simplefilter("error", ResourceWarning)
+ assert smoke.main() == 0
+ assert seen == ["index.html"]
+ assert crit.exists()
+
+
+def test_missing_access_credentials_fail_loudly_without_probing(monkeypatch, tmp_path, capsys):
+ _argv(monkeypatch, tmp_path)
+ monkeypatch.delenv("CF_ACCESS_CLIENT_ID", raising=False)
+ monkeypatch.delenv("CF_ACCESS_CLIENT_SECRET", raising=False)
+ monkeypatch.setattr(smoke, "run", lambda *a, **k: pytest.fail("must not probe"))
+ assert smoke.main() == 2
+ err = capsys.readouterr().err
+ assert "CF_ACCESS_CLIENT_ID" in err and "CF_ACCESS_CLIENT_SECRET" in err
+
+
+def test_critical_list_strips_before_testing_for_comments(monkeypatch, tmp_path):
+ # An INDENTED comment used to survive the `line.startswith("#")` test (applied to the
+ # unstripped line) and become a live critical page — which can never exist as a file.
+ crit = tmp_path / "critical.txt"
+ crit.write_text("# leading comment\n # indented comment\n\n cli.html \n")
+ monkeypatch.setenv("CF_ACCESS_CLIENT_ID", "id")
+ monkeypatch.setenv("CF_ACCESS_CLIENT_SECRET", "sec")
+ monkeypatch.setattr(sys, "argv", ["smoke", "--host", "h", "--slug", "rolling",
+ "--expect-sha", "SKIP", "--critical-list", str(crit)])
+ seen: list[str] = []
+ monkeypatch.setattr(smoke, "run",
+ lambda host, slug, sha, aid, asec, pdf, critical:
+ seen.extend(critical) or 0)
+ assert smoke.main() == 0
+ assert seen == ["cli.html"]