From 4d605edb29a7ce028996ac4edec2be8ebf53606d Mon Sep 17 00:00:00 2001 From: Seongjae Date: Mon, 28 Sep 2026 22:41:34 +0900 Subject: [PATCH 1/6] ci(validate): select validation stages without the Rust toolchain `scripts/validate.py --stages LIST` runs only the named stages, always in the fixed order whitespace, source-contract, documentation, protocol-analysis, dependency-boundary, format, clippy, contracts. The Rust toolchain is probed only when a Cargo stage is selected, so the documentation and source-contract checks run without Rust. A stage left out is recorded as not_selected_by_plan, never as passed, and a run of only some stages does not claim the DG-0 milestone. The default run is unchanged: every stage except whitespace, the same commands in the same order. test_measure.py is its own stage (protocol-analysis) instead of part of documentation. The new whitespace stage is `git diff --check` of `--diff-base`..HEAD, or of the whole tree at HEAD when no base is given. The whole tree exempts only files whose bytes match the SHA-256 the CI policy pins. The workspace graph moves to module level (WORKSPACE_GRAPH) so the CI planner can derive the affected functional suites from it. --- scripts/test_ci_validate.py | 171 +++++++++++++++++++++++++++ scripts/validate.py | 229 +++++++++++++++++++++++++++--------- 2 files changed, 342 insertions(+), 58 deletions(-) create mode 100644 scripts/test_ci_validate.py diff --git a/scripts/test_ci_validate.py b/scripts/test_ci_validate.py new file mode 100644 index 0000000..09bbf86 --- /dev/null +++ b/scripts/test_ci_validate.py @@ -0,0 +1,171 @@ +"""Selectable validation stages: fixed order, refusals, runs without Rust and recorded omissions.""" +import hashlib +import json +import os +from pathlib import Path +import shutil +import subprocess +import sys +import tempfile +import unittest +import uuid + +import validate + + +def git(root, *args): + environment = dict(os.environ, GIT_AUTHOR_NAME="CI", GIT_AUTHOR_EMAIL="ci@example.invalid", + GIT_COMMITTER_NAME="CI", GIT_COMMITTER_EMAIL="ci@example.invalid") + command = ["git", "-c", "commit.gpgsign=false", "-c", "core.hooksPath=/dev/null", *args] + return subprocess.run(command, cwd=root, env=environment, check=True, capture_output=True, + text=True).stdout.strip() + + +class Repository: + """A throwaway repository for the whitespace stage.""" + + def __init__(self, root): + self.root = Path(root) + self.root.mkdir() + git(self.root, "init", "-q", "-b", "main") + + def write(self, path, data): + (self.root / path).parent.mkdir(parents=True, exist_ok=True) + (self.root / path).write_bytes(data) + + def commit(self, message): + git(self.root, "add", "-A") + git(self.root, "commit", "-q", "--allow-empty", "-m", message) + return git(self.root, "rev-parse", "HEAD") + + def policy(self, exemptions): + self.write(validate.POLICY, json.dumps({"whitespace": {"tree_exemptions": exemptions}}).encode()) + + +class StageSelection(unittest.TestCase): + def test_the_default_is_every_stage_but_whitespace_in_the_legacy_order(self): + self.assertEqual(validate.STAGES, ("whitespace",) + validate.DEFAULT_STAGES) + self.assertEqual(validate.DEFAULT_STAGES, ("source-contract", "documentation", "protocol-analysis", + "dependency-boundary", "format", "clippy", "contracts")) + + def test_a_selection_runs_in_the_fixed_order(self): + self.assertEqual(validate.parse_stages("documentation, whitespace"), ["whitespace", "documentation"]) + + def test_unknown_repeated_or_empty_selections_are_refused(self): + for text in ("docs", "documentation,documentation", "", " , "): + with self.subTest(text=text), self.assertRaises(ValueError): + validate.parse_stages(text) + + def test_only_the_cargo_stages_need_rust(self): + self.assertEqual(validate.RUST_STAGES, {"dependency-boundary", "format", "clippy", "contracts"}) + + +class RunWithoutRust(unittest.TestCase): + """Non-Rust stages run with no rustc or cargo reachable, and the report says what was left out.""" + + def setUp(self): + self.bin = tempfile.TemporaryDirectory() + self.addCleanup(self.bin.cleanup) + # Only git is reachable: the validator runs Python checks with its own interpreter. + os.symlink(shutil.which("git"), Path(self.bin.name) / "git") + self.output = validate.ROOT / "target" / "qualification" / ("test-ci-validate-" + uuid.uuid4().hex) + self.addCleanup(shutil.rmtree, self.output, ignore_errors=True) + + def validate(self, *args): + return subprocess.run([sys.executable, "-B", str(validate.ROOT / "scripts/validate.py"), *args, + "--output", str(self.output)], + cwd=validate.ROOT, env=dict(os.environ, PATH=self.bin.name), + capture_output=True, text=True) + + def test_selected_non_rust_stages_pass_without_a_toolchain(self): + completed = self.validate("--stages", "documentation,source-contract,whitespace", "--diff-base", "HEAD") + self.assertEqual(completed.returncode, 0, completed.stdout + completed.stderr) + report = json.loads((self.output / "report.json").read_text()) + self.assertEqual((report["status"], report["milestone"], report["execution_mode"]), + ("passed", None, "selected-stages")) + self.assertIsNone(report["toolchain_matches"]) + self.assertNotIn("rustc", report) + statuses = {stage["name"]: stage["status"] for stage in report["stages"]} + self.assertEqual([stage["name"] for stage in report["stages"]], list(validate.STAGES)) + self.assertEqual(statuses, { + "whitespace": "passed", "source-contract": "passed", "documentation": "passed", + "protocol-analysis": "not_selected_by_plan", "dependency-boundary": "not_selected_by_plan", + "format": "not_selected_by_plan", "clippy": "not_selected_by_plan", "contracts": "not_selected_by_plan"}) + whitespace = report["stages"][0] + self.assertEqual((whitespace["mode"], whitespace["base"]), ("diff", whitespace["head"])) + self.assertEqual(report["source"]["head"], whitespace["head"]) + + def test_a_rust_stage_without_a_toolchain_fails_before_any_stage_runs(self): + completed = self.validate("--stages", "source-contract,format") + self.assertEqual(completed.returncode, 1) + report = json.loads((self.output / "report.json").read_text()) + self.assertEqual(report["status"], "failed") + self.assertEqual({stage["name"]: stage["status"] for stage in report["stages"] + if stage["name"] in ("source-contract", "format")}, + {"source-contract": "not_run", "format": "not_run"}) + + def test_a_diff_base_needs_the_whitespace_stage(self): + completed = self.validate("--stages", "documentation", "--diff-base", "HEAD") + self.assertEqual(completed.returncode, 2) + self.assertIn("--diff-base applies only to the whitespace stage", completed.stderr) + self.assertFalse(self.output.exists()) + + +class Whitespace(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory() + self.addCleanup(self.temporary.cleanup) + self.repo = Repository(Path(self.temporary.name) / "repo") + self.output = Path(self.temporary.name) / "out" + self.output.mkdir() + + def check(self, base): + return validate.whitespace(self.output, base, root=self.repo.root) + + def test_a_diff_is_checked_for_its_own_lines_only(self): + self.repo.write("old.md", b"trailing \n") + base = self.repo.commit("base") + self.repo.write("clean.md", b"clean\n") + self.repo.commit("clean") + self.assertEqual(self.check(base)["status"], "passed") + self.repo.write("new.md", b"trailing \n") + entry = self.check(base) + self.assertEqual(entry["status"], "passed", "an uncommitted file is not part of HEAD") + self.repo.commit("dirty") + entry = self.check(base) + self.assertEqual((entry["status"], entry["mode"], entry["base"]), ("failed", "diff", base)) + log = (self.output / entry["log"]).read_text() + self.assertIn("new.md:1: trailing whitespace", log) + self.assertNotIn("old.md", log) + + def test_the_whole_tree_exempts_only_pinned_bytes(self): + license_text = b"License.\n\n" + self.repo.write("LICENSE", license_text) + self.repo.policy({"LICENSE": hashlib.sha256(license_text).hexdigest()}) + self.repo.commit("pinned") + entry = self.check(None) + self.assertEqual((entry["status"], entry["mode"], entry["base"]), ("passed", "tree", None)) + self.assertEqual((list(entry["exempt"]), entry["exemption_lost"]), (["LICENSE"], [])) + self.repo.write("LICENSE", b"Changed license.\n\n") + self.repo.commit("changed") + entry = self.check(None) + self.assertEqual((entry["status"], entry["exempt"], entry["exemption_lost"]), ("failed", {}, ["LICENSE"])) + self.assertIn("LICENSE:2: new blank line at EOF", (self.output / entry["log"]).read_text()) + + def test_the_whole_tree_still_checks_every_other_file(self): + self.repo.policy({}) + self.repo.write("guide.md", b"a\ttab and trailing \n") + self.repo.commit("dirty") + entry = self.check(None) + self.assertEqual(entry["status"], "failed") + self.assertIn("guide.md:1: trailing whitespace", (self.output / entry["log"]).read_text()) + + def test_an_unknown_base_is_an_error_not_a_pass(self): + self.repo.policy({}) + self.repo.commit("base") + with self.assertRaises(subprocess.CalledProcessError): + self.check("f" * 40) + + +if __name__ == "__main__": + unittest.main() diff --git a/scripts/validate.py b/scripts/validate.py index bc07ca4..748c4ca 100644 --- a/scripts/validate.py +++ b/scripts/validate.py @@ -1,5 +1,11 @@ #!/usr/bin/env python3 -"""DG-0 qualification only. Never infer runtime or OS qualification from unit tests.""" +"""DG-0 qualification only. Never infer runtime or OS qualification from unit tests. + +Without --stages every default stage runs, as before. --stages selects a subset, run in the +fixed stage order, so a CI plan can run its non-Rust checks without the Rust toolchain. A stage +left out is recorded as not_selected_by_plan, never as passed, and a run of only some stages +does not claim the DG-0 milestone. +""" import argparse import datetime import hashlib @@ -15,12 +21,62 @@ ROOT = Path(__file__).resolve().parents[1] PINNED_RUST = "1.95.0" +POLICY = "scripts/ci-policy.json" +# Every stage, in the order it runs. The whitespace check runs only when it is selected: a CI +# plan checks its diff, or the whole tree when it has no trusted base. +STAGES = ("whitespace", "source-contract", "documentation", "protocol-analysis", + "dependency-boundary", "format", "clippy", "contracts") +DEFAULT_STAGES = STAGES[1:] +# Only these stages need the pinned Rust toolchain, so only they probe it. The whitespace check and +# the three Cargo commands record a failure and let the next stage run; any other failed stage ends +# the run, and the selected stages after it stay not_run. +RUST_STAGES = frozenset({"dependency-boundary", "format", "clippy", "contracts"}) +# The explicit workspace graph: the workspace packages each package may depend on, in any +# dependency kind. The dependency-boundary stage enforces it; the CI planner derives which +# functional suites a changed crate can affect from it. +WORKSPACE_GRAPH = { + "devguard-contract": set(), + "devguard-core": {"devguard-contract"}, + "devguard-macos": {"devguard-contract", "devguard-core"}, + # The self edge only enables the daemon's fixtures in its own tests. + "devguard-daemon": {"devguard-contract", "devguard-core", "devguard-client", "devguard-macos", + "devguard-daemon"}, + "devguard-client": {"devguard-contract"}, + # The daemon edge is a test-only dependency for isolated fixture authorities. + "devguard-launch": {"devguard-contract", "devguard-client", "devguard-macos", "devguard-daemon"}, + # The CLI shares the daemon's canonical paths and configuration; it never + # depends on the native backend or core directly. + "devguard-cli": {"devguard-contract", "devguard-client", "devguard-daemon", "devguard-cargo"}, + # The Cargo adapter is a pure transformation over the contract types. + "devguard-cargo": {"devguard-contract"}, + # The qualification harness measures a service as an ordinary CLI owner; + # it is never packaged in a release. + "devguard-qualify": {"devguard-contract", "devguard-client", "devguard-daemon", "devguard-cli"}, +} def capture(*args): return subprocess.check_output(args, cwd=ROOT, text=True).strip() +def git(root, *args, stdin=None): + return subprocess.run(["git", *args], cwd=root, check=True, capture_output=True, text=True, + input=stdin).stdout.strip() + + +def parse_stages(text): + """The named stages in the fixed stage order; unknown, repeated or no names are refused.""" + names = [name.strip() for name in text.split(",") if name.strip()] + if not names: + raise ValueError("no stage selected") + unknown = sorted(set(names) - set(STAGES)) + if unknown: + raise ValueError("unknown stage: " + ", ".join(unknown)) + if len(set(names)) != len(names): + raise ValueError("a stage is selected twice") + return [stage for stage in STAGES if stage in names] + + def source_fingerprint(): names = subprocess.check_output( ["git", "ls-files", "-z", "--cached", "--others", "--exclude-standard"], cwd=ROOT @@ -54,6 +110,44 @@ def source_contract(): raise RuntimeError("Linux qualification must remain required") +def tree_exemptions(root, head): + """Files whose whitespace findings the policy accepts, only while their bytes are the pinned ones.""" + pinned = json.loads((root / POLICY).read_text())["whitespace"]["tree_exemptions"] + exempt, lost = {}, [] + for path, digest in sorted(pinned.items()): + blob = subprocess.run(["git", "cat-file", "blob", f"{head}:{path}"], cwd=root, capture_output=True) + if blob.returncode == 0 and hashlib.sha256(blob.stdout).hexdigest() == digest: + exempt[path] = digest + else: + lost.append(path) + return exempt, lost + + +def whitespace(output, base, root=ROOT): + """`git diff --check` of BASE..HEAD or, without a base, of the whole tree at HEAD.""" + head = git(root, "rev-parse", "--verify", "HEAD^{commit}") + entry = {"name": "whitespace", "status": "failed", "head": head, "log": "whitespace.log"} + if base is not None: + entry.update(mode="diff", base=git(root, "rev-parse", "--verify", base + "^{commit}")) + command = ["git", "diff", "--check", "--no-color", entry["base"], head] + else: + exempt, lost = tree_exemptions(root, head) + entry.update(mode="tree", base=None, exempt=exempt, exemption_lost=lost) + empty = git(root, "hash-object", "-t", "tree", "--stdin", stdin="") + command = ["git", "diff", "--check", "--no-color", empty, head, "--", "."] + command += [":(exclude,literal)" + path for path in exempt] + entry["command"] = command + with (output / entry["log"]).open("w") as log: + completed = subprocess.run(command, cwd=root, stdout=log, stderr=subprocess.STDOUT) + entry["status"] = "passed" if completed.returncode == 0 else "failed" + return entry + + +def unittest(pattern): + subprocess.run([sys.executable, "-B", "-m", "unittest", "discover", "-s", "scripts", "-p", pattern], + cwd=ROOT, check=True) + + def dependency_boundary(offline, environment, output): command = ["cargo", "metadata", "--locked", "--format-version", "1"] if offline: @@ -64,25 +158,7 @@ def dependency_boundary(offline, environment, output): if forbidden: raise RuntimeError("independent graph includes product dependencies: " + ", ".join(forbidden)) roots = {p["name"] for p in packages if p["id"] in metadata["workspace_members"]} - allowed = { - "devguard-contract": set(), - "devguard-core": {"devguard-contract"}, - "devguard-macos": {"devguard-contract", "devguard-core"}, - # The self edge only enables the daemon's fixtures in its own tests. - "devguard-daemon": {"devguard-contract", "devguard-core", "devguard-client", "devguard-macos", - "devguard-daemon"}, - "devguard-client": {"devguard-contract"}, - # The daemon edge is a test-only dependency for isolated fixture authorities. - "devguard-launch": {"devguard-contract", "devguard-client", "devguard-macos", "devguard-daemon"}, - # The CLI shares the daemon's canonical paths and configuration; it never - # depends on the native backend or core directly. - "devguard-cli": {"devguard-contract", "devguard-client", "devguard-daemon", "devguard-cargo"}, - # The Cargo adapter is a pure transformation over the contract types. - "devguard-cargo": {"devguard-contract"}, - # The qualification harness measures a service as an ordinary CLI owner; - # it is never packaged in a release. - "devguard-qualify": {"devguard-contract", "devguard-client", "devguard-daemon", "devguard-cli"}, - } + allowed = WORKSPACE_GRAPH if roots != set(allowed): raise RuntimeError("unexpected workspace graph; update explicit boundaries with new crates") for package in packages: @@ -120,67 +196,104 @@ def main(): parser.add_argument("--offline", action="store_true") parser.add_argument("--allow-toolchain-mismatch", action="store_true") parser.add_argument("--output", type=Path) + parser.add_argument("--stages", metavar="LIST", + help="comma-separated stages to run, always in the order " + ", ".join(STAGES) + + "; default: every stage except whitespace") + parser.add_argument("--diff-base", metavar="COMMIT", + help="check whitespace in COMMIT..HEAD; without it the whole tree at HEAD is checked") args = parser.parse_args() + try: + selected = parse_stages(args.stages) if args.stages is not None else list(DEFAULT_STAGES) + except ValueError as error: + parser.error(str(error)) + if args.diff_base is not None and "whitespace" not in selected: + parser.error("--diff-base applies only to the whitespace stage") run_id = datetime.datetime.now(datetime.timezone.utc).strftime("%Y%m%dT%H%M%SZ") + "-" + uuid.uuid4().hex[:8] output = args.output or ROOT / "target/qualification" / run_id output = output.resolve() if subprocess.run(["git", "check-ignore", "-q", str(output / "report.json")], cwd=ROOT).returncode: parser.error("qualification output must be Git-ignored inside this checkout") output.mkdir(parents=True, exist_ok=False) - report = {"schema": "devguard-qualification/v1", "milestone": "DG-0", "run_id": run_id, - "platform": platform.platform(), "status": "failed", "stages": [], - "execution_mode": "bootstrap-contract-validation", "self_governed": False, + # Only a run of every default stage is DG-0 qualification. + complete = set(DEFAULT_STAGES) <= set(selected) + omitted = "not selected by --stages" if args.stages is not None else "not a default stage" + stages = [{"name": name, "status": "not_run", "reason": "not reached"} if name in selected + else {"name": name, "status": "not_selected_by_plan", "reason": omitted} for name in STAGES] + report = {"schema": "devguard-qualification/v1", "milestone": "DG-0" if complete else None, "run_id": run_id, + "platform": platform.platform(), "status": "failed", "stages": stages, "selected_stages": selected, + "execution_mode": "bootstrap-contract-validation" if complete else "selected-stages", + "self_governed": False, "runtime_qualification": {"macos_launch": "not_run", "linux_cgroups": "not_run", "browser_slo": "not_run", "candidate_self_use": "not_run", "codespace_integration": "not_run"}} + by_name = {entry["name"]: entry for entry in stages} environment = os.environ.copy() environment["CARGO_BUILD_JOBS"] = "1" environment["RUST_TEST_THREADS"] = "1" failed = False + current = None try: before = source_fingerprint() report["source"] = before - report["rustc"] = capture("rustc", "-vV") - report["cargo"] = capture("cargo", "--version") - actual = report["rustc"].splitlines()[0].split()[1] report["pinned_toolchain"] = PINNED_RUST - report["toolchain_matches"] = actual == PINNED_RUST - if not report["toolchain_matches"] and not args.allow_toolchain_mismatch: - raise RuntimeError(f"expected Rust {PINNED_RUST}, observed {actual}; put the pinned toolchain first in PATH") - source_contract() - report["stages"].append({"name": "source-contract", "status": "passed"}) - subprocess.run([sys.executable, "scripts/check_docs.py"], cwd=ROOT, check=True) - for tests in ("test_check_docs.py", "test_measure.py"): - subprocess.run([sys.executable, "-B", "-m", "unittest", "discover", "-s", "scripts", - "-p", tests], cwd=ROOT, check=True) - report["stages"].append({"name": "documentation", "status": "passed"}) - dependency_boundary(args.offline, environment, output) - report["stages"].append({"name": "dependency-boundary", "status": "passed"}) + if RUST_STAGES & set(selected): + report["rustc"] = capture("rustc", "-vV") + report["cargo"] = capture("cargo", "--version") + actual = report["rustc"].splitlines()[0].split()[1] + report["toolchain_matches"] = actual == PINNED_RUST + if not report["toolchain_matches"] and not args.allow_toolchain_mismatch: + raise RuntimeError(f"expected Rust {PINNED_RUST}, observed {actual}; put the pinned toolchain first in PATH") + else: + report["toolchain_matches"] = None + report["toolchain"] = "not probed: no selected stage needs Rust" cargo_flags = ["--locked"] + (["--offline"] if args.offline else []) - commands = [ - ("format", ["cargo", "fmt", "--all", "--", "--check"]), - ("clippy", ["cargo", "clippy", *cargo_flags, "--workspace", "--all-targets", "--", "-D", "warnings"]), - ("contracts", ["cargo", "test", *cargo_flags, "--workspace"]), - ] - for name, command in commands: - started = time.monotonic() - entry = {"name": name, "status": "failed", "command": command, "log": name + ".log"} - report["stages"].append(entry) - with (output / entry["log"]).open("w") as log: - completed = subprocess.run(command, cwd=ROOT, env=environment, stdout=log, stderr=subprocess.STDOUT) - entry["seconds"] = round(time.monotonic() - started, 3) - entry["status"] = "passed" if completed.returncode == 0 else "failed" - if name == "contracts" and completed.returncode == 0: - text = (output / entry["log"]).read_text(errors="replace") - entry["tests_passed"] = sum(map(int, re.findall(r"test result: ok\. (\d+) passed", text))) - failed |= completed.returncode != 0 - print(name + ": " + entry["status"], flush=True) + commands = { + "format": ["cargo", "fmt", "--all", "--", "--check"], + "clippy": ["cargo", "clippy", *cargo_flags, "--workspace", "--all-targets", "--", "-D", "warnings"], + "contracts": ["cargo", "test", *cargo_flags, "--workspace"], + } + for name in selected: + current = by_name[name] + current.pop("reason") + current["status"] = "failed" + if name in commands: + started = time.monotonic() + current.update(command=commands[name], log=name + ".log") + with (output / current["log"]).open("w") as log: + completed = subprocess.run(commands[name], cwd=ROOT, env=environment, stdout=log, + stderr=subprocess.STDOUT) + current["seconds"] = round(time.monotonic() - started, 3) + current["status"] = "passed" if completed.returncode == 0 else "failed" + if name == "contracts" and completed.returncode == 0: + text = (output / current["log"]).read_text(errors="replace") + current["tests_passed"] = sum(map(int, re.findall(r"test result: ok\. (\d+) passed", text))) + elif name == "whitespace": + current.update(whitespace(output, args.diff_base)) + elif name == "source-contract": + source_contract() + current["status"] = "passed" + elif name == "documentation": + subprocess.run([sys.executable, "scripts/check_docs.py"], cwd=ROOT, check=True) + unittest("test_check_docs.py") + current["status"] = "passed" + elif name == "protocol-analysis": + unittest("test_measure.py") + current["status"] = "passed" + elif name == "dependency-boundary": + dependency_boundary(args.offline, environment, output) + current["status"] = "passed" + failed |= current["status"] != "passed" + print(name + ": " + current["status"], flush=True) + current = None if source_fingerprint() != before: raise RuntimeError("source inputs changed during qualification") - report["status"] = "failed" if failed else ("passed" if report["toolchain_matches"] else "incomplete") - except (OSError, ValueError, RuntimeError, subprocess.CalledProcessError) as error: + matches = report["toolchain_matches"] is not False + report["status"] = "failed" if failed else ("passed" if matches else "incomplete") + except (OSError, ValueError, KeyError, RuntimeError, subprocess.CalledProcessError) as error: failed = True report["error"] = str(error) + if current is not None: + current["error"] = str(error) finally: (output / "report.json").write_text(json.dumps(report, indent=2, ensure_ascii=False) + "\n") print(json.dumps({"status": report["status"], "report": str(output / "report.json")}, ensure_ascii=False)) From 73991ed4b76a52398ac25255582531abfbd51dc0 Mon Sep 17 00:00:00 2001 From: Seongjae Date: Mon, 28 Sep 2026 22:41:48 +0900 Subject: [PATCH 2/6] ci: classify changed paths and plan affected checks scripts/ci-policy.json classifies every changed path, first match in this order: - full: .github, scripts, Cargo manifests and locks, toolchain files, Git ignore/attribute files and .devguard.toml; - historical: docs/handoff records, which run whitespace and documentation checks only; - normative: design, contract, operations, planning, translation and ledger inputs, AGENTS.md, README.md, LICENSE and NOTICE, which run whitespace and documentation, plus source-contract for the approved design, its source record and milestones.json; - a workspace crate, which runs the complete validator on both platforms and the functional suites whose packages, or the binaries they build first (qualify.py PREBUILD), depend on it. A path in no class, a missing or untrusted base, an empty diff and any change to a planning input make the plan full. scripts/ci_plan.py binds the plan to GITHUB_SHA, the event, the policy digest, the run id and the attempt. pull_request plans the merge against its first parent and push plans main's before..after; schedule and workflow_dispatch are full; any other event, merge_group included, has no plan. PR #17's one-file diff (a docs/handoff record) is a regression case: it selects whitespace and documentation only, no Rust and no suite. Drift tests tie the policy to the workspace members, the suite map to the dependency closure recomputed from qualify.py, the native set to qualify.NATIVE, the source-contract inputs to what source_contract() reads, and every tracked path to a class. --- scripts/ci-policy.json | 165 +++++++++++++++++ scripts/ci_plan.py | 226 +++++++++++++++++++++++ scripts/test_ci_plan.py | 393 ++++++++++++++++++++++++++++++++++++++++ 3 files changed, 784 insertions(+) create mode 100644 scripts/ci-policy.json create mode 100644 scripts/ci_plan.py create mode 100644 scripts/test_ci_plan.py diff --git a/scripts/ci-policy.json b/scripts/ci-policy.json new file mode 100644 index 0000000..3bed19a --- /dev/null +++ b/scripts/ci-policy.json @@ -0,0 +1,165 @@ +{ + "schema": "devguard-ci-policy/v1", + "classes": { + "full": [ + ".github/**", + "scripts/**", + "**/Cargo.toml", + "**/Cargo.lock", + "**/rust-toolchain", + "**/rust-toolchain.toml", + "**/.cargo/**", + "**/.gitignore", + "**/.gitattributes", + ".gitmodules", + ".devguard.toml" + ], + "historical": [ + "docs/handoff/**" + ], + "normative": [ + "AGENTS.md", + "README.md", + "LICENSE", + "NOTICE", + "milestones.json", + "docs/design.ko.md", + "docs/design-source.json", + "docs/design.md", + "docs/design-revision-1.md", + "docs/contracts.md", + "docs/operations.md", + "docs/milestones.md", + "docs/translations.json", + "docs/planning/**", + "docs/ko/**" + ] + }, + "components": { + "contract": ["crates/contract/**"], + "core": ["crates/core/**"], + "macos": ["crates/macos/**"], + "daemon": ["crates/daemon/**"], + "client": ["crates/client/**"], + "launch": ["crates/launch/**"], + "cli": ["crates/cli/**"], + "cargo": ["crates/cargo/**"], + "qualify": ["crates/qualify/**"] + }, + "repository": { + "always": ["whitespace"], + "documents": ["documentation"], + "source_contract_inputs": ["docs/design-source.json", "docs/design.ko.md", "milestones.json"], + "full": ["whitespace", "source-contract", "documentation", "protocol-analysis"] + }, + "contracts": { + "platforms": ["macos-14", "ubuntu-24.04"] + }, + "suites": { + "dg1-authority": { + "platforms": ["macos-14", "ubuntu-24.04"], + "components": ["client", "contract", "core", "daemon", "macos"] + }, + "dg1-auth": { + "platforms": ["macos-14", "ubuntu-24.04"], + "components": ["client", "contract", "core", "daemon", "macos"] + }, + "dg1-probes": { + "platforms": ["macos-14"], + "components": ["client", "contract", "core", "daemon", "macos"] + }, + "dg1-scopes": { + "platforms": ["macos-14"], + "components": ["contract", "core", "macos"] + }, + "dg1-launch": { + "platforms": ["macos-14"], + "components": ["client", "contract", "core", "daemon", "launch", "macos"] + }, + "dg1-reconcile": { + "platforms": ["macos-14"], + "components": ["client", "contract", "core", "daemon", "launch", "macos"] + }, + "dg1-cli": { + "platforms": ["macos-14"], + "components": ["cargo", "cli", "client", "contract", "core", "daemon", "launch", "macos"] + }, + "dg1-cargo": { + "platforms": ["macos-14"], + "components": ["cargo", "cli", "client", "contract", "core", "daemon", "launch", "macos"] + }, + "dg1-bootstrap": { + "platforms": ["macos-14"], + "components": ["client", "contract", "core", "daemon", "macos"] + }, + "dg1-self-use": { + "platforms": ["macos-14"], + "components": ["cargo", "cli", "client", "contract", "core", "daemon", "launch", "macos"] + }, + "dg1-upgrade": { + "platforms": ["macos-14"], + "components": ["cargo", "cli", "client", "contract", "core", "daemon", "macos"] + }, + "dg1-macos": { + "platforms": ["macos-14"], + "components": ["cargo", "cli", "client", "contract", "core", "daemon", "launch", "macos", "qualify"] + } + }, + "allowances": { + "macos-14": { + "dg1-scopes": [ + { + "stage": "native-scopes", + "cases": ["scope-refusals"], + "path": "unclamped_root", + "reason": "the environment clamps every child of this harness, so an unclamped root cannot be produced", + "source": "crates/macos/tests/scopes.rs" + } + ], + "dg1-launch": [ + { + "stage": "native-launch", + "cases": ["claimed-then-refused"], + "path": "", + "reason": "the environment clamps every child of this harness, so an unclamped helper cannot be produced", + "source": "crates/launch/tests/launch.rs" + } + ], + "dg1-cargo": [ + { + "stage": "native-cargo", + "cases": ["cargo-test", "concurrent-consumers", "direct-build", "explicit-jobs", "inherited-jobserver", + "inherited-pipe-jobserver", "nested-cargo", "pipeline-cancelled", "pipeline-shared", + "stale-jobserver"], + "path": "", + "reason": "this host's work capacity cannot fit the Cargo jobs the case needs", + "source": "crates/cli/tests/cargo.rs" + } + ], + "dg1-bootstrap": [ + { + "stage": "native-install", + "cases": ["launchd-lifecycle"], + "path": "", + "reason": "this session has no launchd gui domain", + "source": "crates/daemon/tests/install.rs" + } + ], + "dg1-macos": [ + { + "stage": "native-fixture", + "cases": ["fixture-check"], + "path": "", + "reason": "the fixture needs Google Chrome on macOS", + "source": "scripts/test_fixture.py" + } + ] + } + }, + "whitespace": { + "tree_exemptions": { + "LICENSE": "5842a12e4cc14135dffb108b7e7d57de5aa4834c070c7e0928a290238fb76a71", + "docs/design.ko.md": "97b67a1f9518c1781156a4b3b26829b285f84f5c9a44da60f3c5dcf1bc768df8" + } + } +} diff --git a/scripts/ci_plan.py b/scripts/ci_plan.py new file mode 100644 index 0000000..ca09ce9 --- /dev/null +++ b/scripts/ci_plan.py @@ -0,0 +1,226 @@ +#!/usr/bin/env python3 +"""Plan the CI checks a change needs and bind the plan to the checked-out source and run. + +Each changed path is classified by scripts/ci-policy.json, first match in this order: +- full: CI, planning, validation and qualification scripts, Cargo, toolchain and Git inputs; +- historical: dated records, which run the whitespace and documentation checks only; +- normative: policy, contract and design inputs, which run the non-Rust checks they feed; +- a component crate, which runs the complete validator on both platforms and the functional + suites whose packages, or the binaries they start, depend on it. +A path in no class makes the plan full. So do a missing or untrusted base, an empty diff and any +change to a planning input, so a pull request cannot narrow its own checks. Scheduled and manual +runs are full. Any other event, merge_group included, has no plan: planning fails, and with it +the required gate. +""" +import argparse +import hashlib +import json +import os +from pathlib import Path +import re +import subprocess + +ROOT = Path(__file__).resolve().parents[1] +POLICY = "scripts/ci-policy.json" +SCHEMA = "devguard-ci-plan/v1" +# The inputs that decide what runs. A difference in any of them between base and head makes the +# plan full, whatever the path classes say. +PLANNING = (".github/workflows/ci.yml", POLICY, "scripts/ci_plan.py", "scripts/ci_run.py", + "scripts/check_ci_results.py", "scripts/validate.py", "scripts/qualify.py") +FULL_EVENTS = ("schedule", "workflow_dispatch") +EVENTS = ("pull_request", "push") + FULL_EVENTS +SHA = re.compile("[0-9a-f]{40}") +RUN = re.compile("[1-9][0-9]*") +GLOB = {"**/": "(?:.*/)?", "**": ".*", "*": "[^/]*", "?": "[^/]"} +# Reasons name at most this many unclassified paths; the plan artifact lists every changed path. +SHOWN = 10 + + +class PlanError(Exception): + pass + + +def git(root, *args): + return subprocess.run(["git", *args], cwd=root, check=True, capture_output=True).stdout + + +def read_policy(root=ROOT): + raw = (root / POLICY).read_bytes() + return json.loads(raw), hashlib.sha256(raw).hexdigest() + + +def matches(pattern, path): + """`*` and `?` stay inside one path segment; `**` spans segments.""" + parts = re.split(r"(\*\*/|\*\*|\*|\?)", pattern) + regex = "".join(GLOB.get(part, re.escape(part)) for part in parts) + return re.fullmatch(regex, path, flags=re.DOTALL) is not None + + +def classify(policy, path): + """The class of one path and what matched it (a pattern or a component), or (None, None).""" + for name in ("full", "historical", "normative"): + for pattern in policy["classes"][name]: + if matches(pattern, path): + return name, pattern + for component, patterns in policy["components"].items(): + if any(matches(pattern, path) for pattern in patterns): + return "component", component + return None, None + + +def paths_digest(paths): + data = "\0".join(sorted(paths)).encode("utf-8", "surrogateescape") + return hashlib.sha256(data).hexdigest() + + +def build_plan(policy, digest, event, source, base, paths, reasons, run=(None, None)): + """The plan for PATHS changed from BASE to SOURCE. Any reason makes it full.""" + reasons = set(reasons) + counts = {"full": 0, "historical": 0, "normative": 0, "component": 0, "unclassified": 0} + components, unclassified, source_inputs = set(), [], False + repository = policy["repository"] + for path in paths: + kind, detail = classify(policy, path) + counts[kind or "unclassified"] += 1 + if kind is None: + unclassified.append(path) + elif kind == "full": + reasons.add("full-path:" + detail) + elif kind == "component": + components.add(detail) + source_inputs |= path in repository["source_contract_inputs"] + reasons |= {"unclassified:" + path for path in unclassified[:SHOWN]} + if len(unclassified) > SHOWN: + reasons.add(f"unclassified:{len(unclassified) - SHOWN} more") + if base is not None and not paths: + reasons.add("empty-diff") + full = bool(reasons) + if full: + chosen = set(repository["full"]) + else: + chosen = set(repository["always"]) + if counts["historical"] or counts["normative"]: + chosen |= set(repository["documents"]) + if source_inputs: + chosen.add("source-contract") + suites = policy["suites"] + rust = full or bool(components) + selected = [name for name, spec in suites.items() if full or components & set(spec["components"])] + return { + "schema": SCHEMA, "event": event, "source_sha": source, "base_sha": base, "policy_sha256": digest, + "run_id": run[0], "run_attempt": run[1], + "profile": "full" if full else "affected", "reasons": sorted(reasons), + "paths": {"count": len(paths), "sha256": paths_digest(paths), **counts}, + "components": sorted(components), + "repository_stages": [stage for stage in repository["full"] if stage in chosen], + "whitespace": {"mode": "diff", "base": base} if base is not None else {"mode": "tree", "base": None}, + "rust": rust, + "suites": {platform: [name for name in selected if rust and platform in suites[name]["platforms"]] + for platform in policy["contracts"]["platforms"]}, + } + + +def blob(root, commit, path): + found = subprocess.run(["git", "rev-parse", "--verify", "-q", f"{commit}:{path}"], cwd=root, + capture_output=True, text=True) + return found.stdout.strip() if found.returncode == 0 else None + + +def planning_changed(root, base, head): + """Whether any planning input differs; one absent from both sides is unchanged.""" + return any(blob(root, base, path) != blob(root, head, path) for path in PLANNING) + + +def changed_paths(root, base, head): + out = git(root, "diff", "--name-only", "--no-renames", "-z", base, head) + return sorted(os.fsdecode(path) for path in out.split(b"\0") if path) + + +def plan_diff(root, policy, digest, event, base, source, run): + paths = changed_paths(root, base, source) + reasons = {"planning-changed"} if planning_changed(root, base, source) else set() + return build_plan(policy, digest, event, source, base, paths, reasons, run), paths + + +def prepare(root, env, event): + """The plan of this run and the paths it changed, from the GitHub environment and event payload.""" + name = env.get("GITHUB_EVENT_NAME") or "" + if name not in EVENTS: + raise PlanError(f"no plan is defined for the event {name or '(none)'}") + source = git(root, "rev-parse", "HEAD").decode().strip() + if not SHA.fullmatch(source) or source != env.get("GITHUB_SHA"): + raise PlanError("the checkout is not GITHUB_SHA") + run = (env.get("GITHUB_RUN_ID") or "", env.get("GITHUB_RUN_ATTEMPT") or "") + if not all(RUN.fullmatch(value) for value in run): + raise PlanError("the run id or attempt is missing") + policy, digest = read_policy(root) + if name in FULL_EVENTS: + return build_plan(policy, digest, name, source, None, [], {"event:" + name}, run), [] + if name == "pull_request": + parents = git(root, "rev-list", "--parents", "-n", "1", source).decode().split() + head = ((event.get("pull_request") or {}).get("head") or {}).get("sha") + if len(parents) != 3 or parents[2] != head: + return build_plan(policy, digest, name, source, None, [], {"pull-request-base-untrusted"}, run), [] + return plan_diff(root, policy, digest, name, parents[1], source, run) + if env.get("GITHUB_REF") != "refs/heads/main": + raise PlanError("push runs are planned for refs/heads/main only") + base = event.get("before") or "" + trusted = (SHA.fullmatch(base) and base != "0" * 40 and not event.get("forced") + and subprocess.run(["git", "merge-base", "--is-ancestor", base, source], cwd=root, + capture_output=True).returncode == 0) + if not trusted: + return build_plan(policy, digest, name, source, None, [], {"push-base-untrusted"}, run), [] + return plan_diff(root, policy, digest, name, base, source, run) + + +def outputs(plan): + """GITHUB_OUTPUT lines. The plan is one line of ASCII JSON.""" + return ["plan=" + json.dumps(plan, separators=(",", ":"), sort_keys=True), + "rust=" + str(plan["rust"]).lower(), "profile=" + plan["profile"]] + + +def summary(plan): + lines = ["### CI plan", "", f"Profile **{plan['profile']}** for `{plan['source_sha']}` ({plan['event']}" + + (f", base `{plan['base_sha']}`" if plan["base_sha"] else ", no base") + ").", ""] + if plan["reasons"]: + lines += ["Full because: " + ", ".join(f"`{reason}`" for reason in plan["reasons"]), ""] + lines += [f"- Changed paths: {plan['paths']['count']}", + "- Repository checks: " + ", ".join(plan["repository_stages"]), + "- Rust validation: " + ("complete validator on " + ", ".join(plan["suites"]) if plan["rust"] + else "not selected by plan")] + for platform, suites in plan["suites"].items(): + lines.append(f"- Functional suites on {platform}: " + (", ".join(suites) or "none")) + return "\n".join(lines) + "\n" + + +def main(argv=None): + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("--base", help="print the plan for the committed diff BASE..HEAD instead") + parser.add_argument("--head", default="HEAD") + args = parser.parse_args(argv) + try: + if args.base: + policy, digest = read_policy() + base, head = (git(ROOT, "rev-parse", "--verify", rev + "^{commit}").decode().strip() + for rev in (args.base, args.head)) + plan, _ = plan_diff(ROOT, policy, digest, "local", base, head, (None, None)) + print(json.dumps(plan, indent=2)) + return + plan, paths = prepare(ROOT, os.environ, json.loads(Path(os.environ["GITHUB_EVENT_PATH"]).read_text())) + out = ROOT / "target" / "ci" / "plan" + out.mkdir(parents=True, exist_ok=True) + (out / "plan.json").write_text(json.dumps(plan, indent=2, sort_keys=True) + "\n") + (out / "changed-paths.json").write_text(json.dumps(paths, indent=2) + "\n") + with open(os.environ["GITHUB_OUTPUT"], "a") as handle: + handle.write("\n".join(outputs(plan)) + "\n") + if os.environ.get("GITHUB_STEP_SUMMARY"): + with open(os.environ["GITHUB_STEP_SUMMARY"], "a") as handle: + handle.write(summary(plan)) + print(f"CI plan: {plan['profile']}; repository checks {', '.join(plan['repository_stages'])}; " + f"Rust {'selected' if plan['rust'] else 'not selected'}") + except (OSError, ValueError, KeyError, TypeError, subprocess.CalledProcessError, PlanError) as error: + raise SystemExit(f"CI planning failed ({error}); no reduced coverage is authorized") + + +if __name__ == "__main__": + main() diff --git a/scripts/test_ci_plan.py b/scripts/test_ci_plan.py new file mode 100644 index 0000000..ba26bc6 --- /dev/null +++ b/scripts/test_ci_plan.py @@ -0,0 +1,393 @@ +"""The CI plan: path classes, suite selection, event binding and the policy's agreement with the repository.""" +import hashlib +import json +import os +from pathlib import Path +import re +import subprocess +import tempfile +import unittest +from unittest import mock + +import ci_plan +import qualify +import validate + +ROOT = ci_plan.ROOT +POLICY, DIGEST = ci_plan.read_policy() +MACOS, UBUNTU = POLICY["contracts"]["platforms"] +ALL = list(POLICY["suites"]) +PORTABLE = [suite for suite in ALL if UBUNTU in POLICY["suites"][suite]["platforms"]] +REPOSITORY = POLICY["repository"]["full"] +# The one-file diff of PR #17 (base 9e21cc8f, head 70c814c9). +PR17 = "docs/handoff/2026-09-28-w3-decision-packet.md" + + +def plan(*paths, base="e" * 40, reasons=()): + return ci_plan.build_plan(POLICY, DIGEST, "local", "f" * 40, base, list(paths), set(reasons)) + + +def suites(*names): + return [suite for suite in ALL if suite in names] + + +def workspace(): + members = re.search(r"members = \[([^\]]*)\]", (ROOT / "Cargo.toml").read_text()).group(1) + return sorted(Path(member).name for member in re.findall(r'"([^"]+)"', members)) + + +def tracked(): + out = subprocess.run(["git", "ls-files", "-z"], cwd=ROOT, check=True, capture_output=True).stdout + return [os.fsdecode(path) for path in out.split(b"\0") if path] + + +class Selection(unittest.TestCase): + def test_pr17_documentation_record_selects_no_rust_and_no_suite(self): + record = plan(PR17) + self.assertEqual((record["profile"], record["reasons"], record["repository_stages"], record["rust"]), + ("affected", [], ["whitespace", "documentation"], False)) + self.assertEqual(record["suites"], {MACOS: [], UBUNTU: []}) + self.assertFalse(any("dg1-upgrade" in names for names in record["suites"].values())) + self.assertEqual(record["whitespace"], {"mode": "diff", "base": "e" * 40}) + + def test_historical_records_and_normative_documents_compile_nothing(self): + for paths in ([PR17, "docs/handoff/2026-09-27-session-close.md"], + ["docs/contracts.md", "docs/ko/contracts.md", "docs/translations.json"], + ["docs/planning/verification.md", "docs/ko/planning/verification.md"], + ["AGENTS.md"], ["README.md"], ["LICENSE", "NOTICE"], ["docs/design-revision-1.md"]): + with self.subTest(paths=paths): + result = plan(*paths) + self.assertEqual((result["profile"], result["rust"], result["repository_stages"]), + ("affected", False, ["whitespace", "documentation"])) + + def test_source_contract_inputs_add_only_the_source_contract_check(self): + for path in POLICY["repository"]["source_contract_inputs"]: + with self.subTest(path=path): + result = plan(path) + self.assertEqual((result["profile"], result["rust"], result["repository_stages"]), + ("affected", False, ["whitespace", "source-contract", "documentation"])) + + def test_a_component_runs_the_complete_validator_and_its_suites(self): + rows = { + "crates/launch/src/main.rs": suites("dg1-launch", "dg1-reconcile", "dg1-cli", "dg1-cargo", + "dg1-self-use", "dg1-macos"), + "crates/cargo/src/lib.rs": suites("dg1-cli", "dg1-cargo", "dg1-self-use", "dg1-upgrade", "dg1-macos"), + "crates/cli/tests/exec.rs": suites("dg1-cli", "dg1-cargo", "dg1-self-use", "dg1-upgrade", "dg1-macos"), + "crates/qualify/fixture/foreground.html": ["dg1-macos"], + "crates/macos/src/scope.rs": ALL, + "crates/core/src/journal.rs": ALL, + "crates/contract/src/lib.rs": ALL, + "crates/daemon/tests/upgrade.rs": [suite for suite in ALL if suite != "dg1-scopes"], + "crates/client/src/protocol.rs": [suite for suite in ALL if suite != "dg1-scopes"], + } + for path, expected in rows.items(): + with self.subTest(path=path): + result = plan(path) + self.assertEqual((result["profile"], result["rust"], result["repository_stages"]), + ("affected", True, ["whitespace"])) + self.assertEqual(result["suites"][MACOS], expected) + self.assertEqual(result["suites"][UBUNTU], [suite for suite in expected if suite in PORTABLE]) + + def test_documents_and_a_component_together_take_both(self): + result = plan("docs/contracts.md", "crates/cargo/src/lib.rs") + self.assertEqual((result["profile"], result["repository_stages"], result["components"]), + ("affected", ["whitespace", "documentation"], ["cargo"])) + self.assertEqual(result["suites"][MACOS], plan("crates/cargo/src/lib.rs")["suites"][MACOS]) + + def test_full_inputs_and_unknown_paths_run_everything(self): + for path in (".github/workflows/ci.yml", "scripts/ci-policy.json", "scripts/qualify.py", + "scripts/check_docs.py", "Cargo.toml", "Cargo.lock", "crates/cli/Cargo.toml", + "rust-toolchain.toml", ".cargo/config.toml", "crates/core/.cargo/config.toml", ".gitignore", + ".devguard.toml", "Makefile", "docs/new-guide.md", "crates/README.md", ".kiro/settings.json"): + with self.subTest(path=path): + result = plan(PR17, path) + self.assertEqual((result["profile"], result["rust"], result["repository_stages"]), + ("full", True, REPOSITORY)) + self.assertEqual(result["suites"], {MACOS: ALL, UBUNTU: PORTABLE}) + self.assertEqual(plan("Makefile")["reasons"], ["unclassified:Makefile"]) + self.assertEqual(plan("Cargo.lock")["reasons"], ["full-path:**/Cargo.lock"]) + + def test_an_empty_diff_and_given_reasons_are_full(self): + self.assertEqual(plan()["reasons"], ["empty-diff"]) + self.assertEqual(plan(PR17, reasons={"planning-changed"})["profile"], "full") + + def test_without_a_base_whitespace_covers_the_whole_tree(self): + result = plan(base=None, reasons={"event:schedule"}) + self.assertEqual(result["whitespace"], {"mode": "tree", "base": None}) + self.assertEqual(result["profile"], "full") + + def test_unclassified_reasons_are_capped_but_counted(self): + result = plan(*[f"unknown/{index:02}" for index in range(12)]) + self.assertEqual(len(result["reasons"]), ci_plan.SHOWN + 1) + self.assertIn("unclassified:2 more", result["reasons"]) + self.assertEqual(result["paths"]["unclassified"], 12) + + def test_globs_respect_segments(self): + self.assertTrue(ci_plan.matches("**/Cargo.toml", "Cargo.toml")) + self.assertTrue(ci_plan.matches("**/Cargo.toml", "crates/cli/Cargo.toml")) + self.assertTrue(ci_plan.matches("docs/ko/**", "docs/ko/planning/README.md")) + self.assertTrue(ci_plan.matches("docs/handoff/**", "docs/handoff/a\nb.md")) + self.assertFalse(ci_plan.matches("docs/handoff/**", "docs/handoff")) + self.assertFalse(ci_plan.matches("README.md", "crates/cli/README.md")) + self.assertFalse(ci_plan.matches("crates/*/src", "crates/a/b/src")) + self.assertFalse(ci_plan.matches("docs/design.md", "docs/designXmd")) + + def test_the_path_digest_covers_every_changed_path(self): + self.assertEqual(plan("b", "a")["paths"]["sha256"], hashlib.sha256(b"a\0b").hexdigest()) + self.assertNotEqual(plan("a")["paths"]["sha256"], plan("a", "b")["paths"]["sha256"]) + + +class Repo: + """A throwaway repository holding the real planning inputs.""" + + def __init__(self, root): + self.root = Path(root) + self.git("init", "-q", "-b", "main") + for path in ci_plan.PLANNING: + if (ROOT / path).is_file(): + self.write(path, (ROOT / path).read_text()) + self.write("docs/handoff/record.md", "record\n") + self.base = self.commit("base") + + def git(self, *args): + env = dict(os.environ, GIT_AUTHOR_NAME="CI", GIT_AUTHOR_EMAIL="ci@example.invalid", + GIT_COMMITTER_NAME="CI", GIT_COMMITTER_EMAIL="ci@example.invalid") + command = ["git", "-c", "commit.gpgsign=false", "-c", "core.hooksPath=/dev/null", *args] + return subprocess.run(command, cwd=self.root, env=env, check=True, capture_output=True).stdout.decode().strip() + + def write(self, path, text): + (self.root / path).parent.mkdir(parents=True, exist_ok=True) + (self.root / path).write_text(text) + + def commit(self, message): + self.git("add", "-A") + self.git("commit", "-q", "--allow-empty", "-m", message) + return self.git("rev-parse", "HEAD") + + def pull_request(self, path=None, text="changed\n"): + self.git("checkout", "-q", "-b", "topic", self.base) + if path: + self.write(path, text) + head = self.commit("head") + self.git("checkout", "-q", "main") + self.git("merge", "-q", "--no-ff", "-m", "merge", "topic") + return head, self.git("rev-parse", "HEAD") + + def prepare(self, event, name, sha=None, **env): + values = {"GITHUB_EVENT_NAME": name, "GITHUB_SHA": sha or self.git("rev-parse", "HEAD"), + "GITHUB_RUN_ID": "4242", "GITHUB_RUN_ATTEMPT": "1", "GITHUB_REF": "refs/heads/main", **env} + return ci_plan.prepare(self.root, {key: value for key, value in values.items() if value is not None}, event) + + +class Events(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory() + self.addCleanup(self.temporary.cleanup) + self.repo = Repo(self.temporary.name) + + def test_a_pull_request_plans_its_merge_against_the_base(self): + head, merge = self.repo.pull_request("crates/cargo/src/lib.rs") + result, paths = self.repo.prepare({"pull_request": {"head": {"sha": head}}}, "pull_request") + self.assertEqual((result["source_sha"], result["base_sha"], result["event"]), + (merge, self.repo.base, "pull_request")) + self.assertEqual((result["profile"], result["components"], paths), ("affected", ["cargo"], + ["crates/cargo/src/lib.rs"])) + self.assertEqual((result["run_id"], result["run_attempt"], result["policy_sha256"]), ("4242", "1", DIGEST)) + + def test_a_documentation_record_pull_request_selects_no_rust(self): + head, _ = self.repo.pull_request("docs/handoff/2026-09-28-record.md") + result, _ = self.repo.prepare({"pull_request": {"head": {"sha": head}}}, "pull_request") + self.assertEqual((result["profile"], result["rust"], result["repository_stages"]), + ("affected", False, ["whitespace", "documentation"])) + + def test_a_changed_planning_input_is_full(self): + head, _ = self.repo.pull_request(ci_plan.POLICY, (ROOT / ci_plan.POLICY).read_text() + "\n") + result, _ = self.repo.prepare({"pull_request": {"head": {"sha": head}}}, "pull_request") + self.assertIn("planning-changed", result["reasons"]) + self.assertEqual((result["profile"], result["suites"]), ("full", {MACOS: ALL, UBUNTU: PORTABLE})) + + def test_an_empty_merge_is_full(self): + head, _ = self.repo.pull_request() + result, _ = self.repo.prepare({"pull_request": {"head": {"sha": head}}}, "pull_request") + self.assertEqual(result["reasons"], ["empty-diff"]) + + def test_the_checkout_must_be_the_event_commit(self): + head, _ = self.repo.pull_request("docs/handoff/new.md") + with self.assertRaisesRegex(ci_plan.PlanError, "GITHUB_SHA"): + self.repo.prepare({"pull_request": {"head": {"sha": head}}}, "pull_request", sha=head) + + def test_a_merge_of_another_head_is_full_without_a_base(self): + self.repo.pull_request("docs/handoff/new.md") + result, paths = self.repo.prepare({"pull_request": {"head": {"sha": self.repo.base}}}, "pull_request") + self.assertEqual((result["reasons"], result["base_sha"], paths), (["pull-request-base-untrusted"], None, [])) + self.assertEqual(result["whitespace"], {"mode": "tree", "base": None}) + + def test_a_push_to_main_plans_before_to_after(self): + self.repo.write("docs/handoff/record.md", "more\n") + after = self.repo.commit("docs") + result, _ = self.repo.prepare({"before": self.repo.base}, "push") + self.assertEqual((result["source_sha"], result["base_sha"], result["profile"], result["rust"]), + (after, self.repo.base, "affected", False)) + + def test_a_push_without_a_trusted_base_is_full(self): + self.repo.write("docs/handoff/record.md", "more\n") + self.repo.commit("docs") + self.repo.git("checkout", "-q", "--orphan", "other") + unrelated = self.repo.commit("unrelated") + self.repo.git("checkout", "-q", "main") + for event in ({"before": "0" * 40}, {"before": "1" * 40}, {"before": unrelated}, {}, {"before": "HEAD~1"}, + {"before": self.repo.base, "forced": True}): + with self.subTest(event=event): + result, _ = self.repo.prepare(event, "push") + self.assertEqual((result["reasons"], result["base_sha"], result["profile"]), + (["push-base-untrusted"], None, "full")) + + def test_a_push_outside_main_has_no_plan(self): + with self.assertRaisesRegex(ci_plan.PlanError, "refs/heads/main"): + self.repo.prepare({"before": self.repo.base}, "push", GITHUB_REF="refs/heads/topic") + + def test_scheduled_and_manual_runs_are_full(self): + for name in ("schedule", "workflow_dispatch"): + with self.subTest(event=name): + result, paths = self.repo.prepare({}, name) + self.assertEqual((result["profile"], result["reasons"], result["base_sha"], paths), + ("full", ["event:" + name], None, [])) + self.assertEqual((result["repository_stages"], result["whitespace"]["mode"]), (REPOSITORY, "tree")) + + def test_other_events_have_no_plan(self): + for name in ("merge_group", "issue_comment", "pull_request_target", ""): + with self.subTest(event=name), self.assertRaisesRegex(ci_plan.PlanError, "no plan is defined"): + self.repo.prepare({}, name) + + def test_the_run_identity_is_required(self): + for values in ({"GITHUB_RUN_ID": None}, {"GITHUB_RUN_ATTEMPT": None}, {"GITHUB_RUN_ATTEMPT": "0"}): + with self.subTest(values=values), self.assertRaisesRegex(ci_plan.PlanError, "run id or attempt"): + self.repo.prepare({}, "schedule", **values) + + +class Outputs(unittest.TestCase): + def test_outputs_are_single_ascii_lines_that_round_trip(self): + result = plan(PR17, "docs/handoff/caf\u00e9\nnote.md", "unknown/\u2028x") + lines = ci_plan.outputs(result) + self.assertEqual([line.split("=", 1)[0] for line in lines], ["plan", "rust", "profile"]) + for line in lines: + self.assertNotIn("\n", line) + self.assertTrue(line.isascii()) + self.assertEqual(json.loads(lines[0].split("=", 1)[1]), result) + self.assertEqual(lines[1:], ["rust=true", "profile=full"]) + self.assertEqual(ci_plan.outputs(plan(PR17))[1:], ["rust=false", "profile=affected"]) + + def test_the_summary_names_what_was_left_out(self): + text = ci_plan.summary(plan(PR17)) + self.assertIn("Rust validation: not selected by plan", text) + self.assertIn(f"Functional suites on {MACOS}: none", text) + + +def closure(package): + seen, todo = set(), [package] + while todo: + name = todo.pop() + if name not in seen: + seen.add(name) + todo += validate.WORKSPACE_GRAPH[name] - {name} + return seen + + +def suite_packages(suite): + """Workspace packages a suite tests, and those whose binaries it builds first.""" + packages = set() + for _, selectors, *_ in qualify.SUITES[suite]: + if selectors[0] != "unittest": + packages.add(selectors[selectors.index("-p") + 1]) + for selectors in qualify.PREBUILD.get(suite, []): + packages.add(selectors[selectors.index("-p") + 1]) + return packages + + +class PolicyAgreesWithTheRepository(unittest.TestCase): + def test_components_are_the_workspace_members(self): + self.assertEqual(sorted(POLICY["components"]), workspace()) + self.assertEqual(set(validate.WORKSPACE_GRAPH), {"devguard-" + name for name in workspace()}) + for name, patterns in POLICY["components"].items(): + self.assertEqual(patterns, [f"crates/{name}/**"]) + + def test_the_suite_map_is_the_dependency_closure_of_each_suite(self): + self.assertEqual(set(ALL), set(qualify.SUITES)) + for suite in ALL: + with self.subTest(suite=suite): + packages = set().union(*(closure(package) for package in suite_packages(suite))) + expected = sorted(package.removeprefix("devguard-") for package in packages) + self.assertEqual(POLICY["suites"][suite]["components"], expected) + + def test_native_suites_run_only_on_macos(self): + for suite in ALL: + with self.subTest(suite=suite): + expected = [MACOS] if suite in qualify.NATIVE else [MACOS, UBUNTU] + self.assertEqual(POLICY["suites"][suite]["platforms"], expected) + + def test_source_contract_inputs_are_the_files_it_reads(self): + read = set() + originals = {name: getattr(Path, name) for name in ("read_text", "read_bytes")} + + def recording(name): + def method(self, *args, **kwargs): + read.add(self.resolve().relative_to(ROOT.resolve()).as_posix()) + return originals[name](self, *args, **kwargs) + return method + + with mock.patch.object(Path, "read_text", recording("read_text")), \ + mock.patch.object(Path, "read_bytes", recording("read_bytes")): + validate.source_contract() + self.assertEqual(read, set(POLICY["repository"]["source_contract_inputs"])) + + def test_every_tracked_path_is_classified(self): + unclassified = [path for path in tracked() if ci_plan.classify(POLICY, path) == (None, None)] + self.assertEqual(unclassified, []) + + def test_every_document_pattern_matches_a_tracked_file(self): + files = tracked() + for pattern in POLICY["classes"]["historical"] + POLICY["classes"]["normative"]: + with self.subTest(pattern=pattern): + self.assertTrue(any(ci_plan.matches(pattern, path) for path in files)) + + def test_validation_and_planning_inputs_are_full(self): + paths = ["Cargo.toml", "Cargo.lock", "rust-toolchain.toml", ".gitignore", ".devguard.toml", + *ci_plan.PLANNING, *(f"crates/{name}/Cargo.toml" for name in workspace())] + paths += [path for path in tracked() if path.startswith(("scripts/", ".github/"))] + for path in paths: + with self.subTest(path=path): + self.assertEqual(ci_plan.classify(POLICY, path)[0], "full") + + def test_repository_stages_are_non_rust_validator_stages_in_order(self): + self.assertEqual(REPOSITORY, [stage for stage in validate.STAGES if stage in REPOSITORY]) + self.assertFalse(set(REPOSITORY) & validate.RUST_STAGES) + for key in ("always", "documents"): + self.assertLessEqual(set(POLICY["repository"][key]), set(REPOSITORY)) + self.assertIn("whitespace", POLICY["repository"]["always"]) + + def test_suite_platforms_are_the_contract_platforms(self): + used = {platform for spec in POLICY["suites"].values() for platform in spec["platforms"]} + self.assertEqual(used, set(POLICY["contracts"]["platforms"])) + + def test_allowances_name_real_cases_and_quote_their_sources(self): + for platform, allowed in POLICY["allowances"].items(): + for suite, entries in allowed.items(): + self.assertIn(platform, POLICY["suites"][suite]["platforms"]) + stages = {name: expected[0] for name, _, *expected in qualify.SUITES[suite] if expected} + for entry in entries: + with self.subTest(platform=platform, suite=suite, stage=entry["stage"]): + self.assertEqual(set(entry), {"stage", "cases", "path", "reason", "source"}) + self.assertLessEqual(set(entry["cases"]), set(stages[entry["stage"]])) + self.assertIn(entry["reason"], (ROOT / entry["source"]).read_text()) + + def test_whitespace_exemptions_are_pinned_to_the_current_bytes(self): + exemptions = POLICY["whitespace"]["tree_exemptions"] + for path, digest in exemptions.items(): + with self.subTest(path=path): + self.assertEqual(hashlib.sha256((ROOT / path).read_bytes()).hexdigest(), digest) + design = json.loads((ROOT / "docs/design-source.json").read_text()) + self.assertEqual(exemptions[design["document"]], design["sha256"]) + + +if __name__ == "__main__": + unittest.main() From 02d2167c1ba3dd91a1e622a2ee0d5dd66700a57b Mon Sep 17 00:00:00 2001 From: Seongjae Date: Mon, 28 Sep 2026 22:42:01 +0900 Subject: [PATCH 3/6] ci: run planned jobs and gate on the current attempt's evidence scripts/ci_run.py runs one job's share of the plan after refusing any plan not made for this commit, event, policy, run and attempt, and records that binding beside the job's evidence: - repository: whitespace and the plan's other non-Rust validator stages; - bind contracts / contracts --os OS: the functional suites the plan selects for OS, in policy order, continuing after a failure and never starting an unplanned suite. The hosted-runner exceptions moved into the policy: a suite listed there runs with --allow-incomplete, and each case its report records as not run must match an allowance's suite, stage, case, path and reason exactly; - summary --os OS: the job summary, with today's statement of scope. scripts/check_ci_results.py derives the plan again and requires the plan job's to equal it, the jobs to have the planned results, and evidence only from artifacts of the current attempt whose bindings name this run and attempt. An earlier attempt's artifact never satisfies the current one, so re-running only the failed jobs cannot pass. --- scripts/check_ci_results.py | 167 ++++++++++++++++++++++ scripts/ci_run.py | 270 ++++++++++++++++++++++++++++++++++++ scripts/test_ci_results.py | 265 +++++++++++++++++++++++++++++++++++ scripts/test_ci_run.py | 269 +++++++++++++++++++++++++++++++++++ 4 files changed, 971 insertions(+) create mode 100644 scripts/check_ci_results.py create mode 100644 scripts/ci_run.py create mode 100644 scripts/test_ci_results.py create mode 100644 scripts/test_ci_run.py diff --git a/scripts/check_ci_results.py b/scripts/check_ci_results.py new file mode 100644 index 0000000..028c602 --- /dev/null +++ b/scripts/check_ci_results.py @@ -0,0 +1,167 @@ +#!/usr/bin/env python3 +"""The required gate: pass only when this run and attempt did exactly what its plan requires. + +The plan is derived again here from the same event and must equal the plan job's, so it is bound +to this commit, event, policy, run and attempt. Every job must have the planned result: the +repository job succeeds, and the contracts jobs succeed when the plan selects Rust and are skipped +otherwise. Only evidence uploaded by this run in this attempt counts. Each artifact name ends in +the current attempt and carries a binding to the run; an artifact of an earlier attempt is +ignored and never satisfies the current one. Re-running only the failed jobs therefore cannot +pass; re-run every job. A suite may be incomplete only for cases the policy allows on its platform. +""" +import argparse +import json +import os +from pathlib import Path +import re +import subprocess + +import ci_plan +import ci_run +import validate + +ROOT = ci_plan.ROOT +REQUIRED_JOBS = {"plan", "repository", "contracts"} + + +def read(path): + try: + return json.loads(path.read_text()) + except (OSError, ValueError): + return None + + +def artifact_names(artifacts): + return sorted(path.name for path in artifacts.iterdir() if path.is_dir()) if artifacts.is_dir() else [] + + +def check_repository(plan, directory): + """Problems with the repository job's evidence, and the status found for each stage.""" + problems = [] + if read(directory / "binding.json") != ci_run.binding(plan, "repository"): + problems.append("repository: its evidence is not bound to this run and attempt") + report = read(directory / "validate" / "report.json") + if not isinstance(report, dict): + return problems + ["repository: no validator report"], {} + if report.get("status") != "passed" or (report.get("source") or {}).get("head") != plan["source_sha"]: + problems.append("repository: the validator report does not show a pass at the planned source") + stages = {stage.get("name"): stage for stage in report.get("stages") or []} + statuses = {name: stage.get("status") for name, stage in stages.items()} + if list(stages) != list(validate.STAGES): + return problems + ["repository: the validator report does not list every stage"], statuses + for name, status in statuses.items(): + expected = "passed" if name in plan["repository_stages"] else "not_selected_by_plan" + if status != expected: + problems.append(f"repository: stage {name} is {status}, expected {expected}") + whitespace = stages["whitespace"] + if (whitespace.get("mode"), whitespace.get("base")) != (plan["whitespace"]["mode"], plan["whitespace"]["base"]): + problems.append("repository: the whitespace check did not cover the planned range") + return problems, statuses + + +def check_contracts(policy, plan, platform, directory): + """Problems with one contracts job's evidence, the validator's status and each planned suite's.""" + label = "contracts " + platform + problems = [] + if read(directory / "binding.json") != ci_run.binding(plan, "contracts", platform): + problems.append(f"{label}: its evidence is not bound to this run and attempt") + report = read(directory / "ci" / "report.json") + report = report if isinstance(report, dict) else {} + stages = {stage.get("name"): stage.get("status") for stage in report.get("stages") or []} + expected = {"whitespace": "not_selected_by_plan", **{name: "passed" for name in validate.DEFAULT_STAGES}} + if (report.get("status") != "passed" or report.get("milestone") != "DG-0" + or report.get("toolchain_matches") is not True or stages != expected + or (report.get("source") or {}).get("head") != plan["source_sha"]): + problems.append(f"{label}: the complete validator did not pass at the planned source") + planned = ci_run.planned_suites(policy, plan, platform) + leg = read(directory / "leg.json") + if not isinstance(leg, dict) or [entry.get("suite") for entry in leg.get("suites") or []] != planned: + problems.append(f"{label}: the runner did not run exactly the planned suites") + results = {} + for suite in planned: + results[suite], found = ci_run.evaluate_suite(policy, platform, suite, read(directory / suite / "report.json"), + plan["source_sha"]) + problems += [f"{label}: {suite}: {problem}" for problem in found] + extra = sorted(path.parent.name for path in directory.glob("dg1-*/report.json") if path.parent.name not in planned) + problems += [f"{label}: {suite} ran although the plan did not select it" for suite in extra] + return problems, report.get("status") or "no report", results + + +def problems(results, plan_text, env, event, artifacts, root=ROOT): + """Every way this attempt differs from its plan, and a table of what was found.""" + if not isinstance(results, dict) or set(results) != REQUIRED_JOBS: + return ["the gate does not see exactly the required jobs"], [] + outcome = {name: (results[name] or {}).get("result") for name in REQUIRED_JOBS} + if outcome["plan"] != "success": + return [f"plan: {outcome['plan']}; without a plan no check is satisfied"], [] + try: + plan = json.loads(plan_text or "") + except ValueError: + return ["the plan job's output is not a plan"], [] + attempt = env.get("GITHUB_RUN_ATTEMPT") + if not isinstance(plan, dict) or plan.get("run_attempt") != attempt: + return [f"the plan is not from this attempt {attempt}; re-run all jobs"], [] + try: + derived, _ = ci_plan.prepare(root, env, event) + except (ci_plan.PlanError, OSError, ValueError, KeyError, subprocess.CalledProcessError) as error: + return [f"the plan cannot be derived again: {error}"], [] + if plan != derived: + return ["the plan job's plan differs from the plan derived for this run"], [] + if plan["event"] in ci_plan.FULL_EVENTS and plan["profile"] != "full": + return ["a scheduled or manual run must be full"], [] + policy, _ = ci_plan.read_policy(root) + found, table = [], [("plan", f"{plan['profile']}; job {outcome['plan']}")] + for name, expected in (("repository", "success"), ("contracts", "success" if plan["rust"] else "skipped")): + table.append((f"job {name}", outcome[name])) + if outcome[name] != expected: + found.append(f"{name}: {outcome[name]}, expected {expected}") + names = artifact_names(artifacts) + current = {name for name in names if re.fullmatch(r".+-[1-9][0-9]*", name) and name.rsplit("-", 1)[1] == attempt} + wanted = {f"ci-plan-{attempt}", f"ci-repository-{attempt}"} + if plan["rust"]: + wanted |= {f"dg0-{platform}-{attempt}" for platform in policy["contracts"]["platforms"]} + found += [f"{name}: no artifact from this attempt" for name in sorted(wanted - current)] + found += [f"{name}: an artifact the plan does not produce" for name in sorted(current - wanted)] + if f"ci-plan-{attempt}" in current and read(artifacts / f"ci-plan-{attempt}" / "plan.json") != plan: + found.append("ci-plan: the uploaded plan is not the planned one") + if f"ci-repository-{attempt}" in current: + repository, statuses = check_repository(plan, artifacts / f"ci-repository-{attempt}") + found += repository + table += [(f"repository {stage}", statuses.get(stage, "no report")) for stage in validate.STAGES] + for platform in policy["contracts"]["platforms"]: + name = f"dg0-{platform}-{attempt}" + if not plan["rust"]: + table.append((f"contracts {platform}", "not_selected_by_plan")) + elif name in current: + contracts, validator, suites = check_contracts(policy, plan, platform, artifacts / name) + found += contracts + table.append((f"contracts {platform} validator", validator)) + table += [(f"contracts {platform} {suite}", suites.get(suite, "not_selected_by_plan")) + for suite, spec in policy["suites"].items() if platform in spec["platforms"]] + ignored = sorted(set(names) - current) + if ignored: + table.append(("ignored: other attempts", ", ".join(ignored))) + return found, table + + +def main(argv=None): + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("--artifacts", type=Path, required=True, help="the directory of this run's downloaded artifacts") + args = parser.parse_args(argv) + env = os.environ + try: + event = json.loads(Path(env["GITHUB_EVENT_PATH"]).read_text()) + found, table = problems(json.loads(env["CI_RESULTS"]), env.get("CI_PLAN"), env, event, args.artifacts) + except (OSError, ValueError, KeyError, TypeError, AttributeError, ci_run.RunError) as error: + found, table = [f"cannot evaluate the CI results: {error}"], [] + verdict = "Required checks: " + ("failed" if found else "passed") + print("\n".join([verdict] + ["- " + line for line in found])) + if env.get("GITHUB_STEP_SUMMARY"): + rows = ["| Check | Result |", "| --- | --- |"] + [f"| {check} | {result} |" for check, result in table] + with open(env["GITHUB_STEP_SUMMARY"], "a") as handle: + handle.write("\n".join([f"### {verdict}", ""] + ["- " + line for line in found] + [""] + rows) + "\n") + raise SystemExit(1 if found else 0) + + +if __name__ == "__main__": + main() diff --git a/scripts/ci_run.py b/scripts/ci_run.py new file mode 100644 index 0000000..79f6a27 --- /dev/null +++ b/scripts/ci_run.py @@ -0,0 +1,270 @@ +#!/usr/bin/env python3 +"""Run one CI job's share of the plan and bind its evidence to this commit, event, policy, run and attempt. + + ci_run.py repository whitespace and the plan's other non-Rust validator stages + ci_run.py bind contracts --os OS bind a contracts job before its validator runs + ci_run.py contracts --os OS the functional suites the plan selects for OS, in policy order + ci_run.py summary --os OS the contracts job summary, written even after a failure + +The plan comes from the CI_PLAN environment variable. A job refuses a plan made for another +commit, event, policy, run or attempt. Hosted-runner exceptions come only from the policy: a suite +the policy allows to be incomplete on OS runs with --allow-incomplete, and each case its report +records as not run must then match one of the suite's allowances exactly. +""" +import argparse +import hashlib +import json +import os +from pathlib import Path +import subprocess +import sys + +import ci_plan +import qualify +import validate + +ROOT = ci_plan.ROOT +BINDING_SCHEMA = "devguard-ci-binding/v1" +LEG_SCHEMA = "devguard-ci-leg/v1" +REPOSITORY_OUT = Path("target/ci/repository") +CONTRACTS_OUT = Path("target/qualification") +# The summary keeps the old workflow's statement of what the functional checks do not cover. +SCOPE = ("Contract regression, service/UDS, native host evidence, native scope, launch helper, reconciliation, " + "command-line owner, Cargo adapter, installation, parent lease, upgrade and repair, and SLO harness " + "functional checks only, as planned; native suites run on macOS. Real self-use under an installed " + "parent and the SLO protocol itself are not run.") + + +class RunError(Exception): + pass + + +def canonical(value): + return json.dumps(value, sort_keys=True, separators=(",", ":")) + + +def digest(value): + return hashlib.sha256(canonical(value).encode()).hexdigest() + + +def head(root): + return ci_plan.git(root, "rev-parse", "HEAD").decode().strip() + + +def load_plan(env, root=ROOT): + """The plan handed to this job, refused unless it was made for this very run.""" + try: + plan = json.loads(env.get("CI_PLAN") or "") + except ValueError: + raise RunError("CI_PLAN is not a plan") from None + _, policy_digest = ci_plan.read_policy(root) + expected = {"schema": ci_plan.SCHEMA, "source_sha": env.get("GITHUB_SHA"), "event": env.get("GITHUB_EVENT_NAME"), + "policy_sha256": policy_digest, "run_id": env.get("GITHUB_RUN_ID"), + "run_attempt": env.get("GITHUB_RUN_ATTEMPT")} + if not isinstance(plan, dict): + raise RunError("CI_PLAN is not a plan") + wrong = sorted(key for key, value in expected.items() if not value or plan.get(key) != value) + if wrong: + raise RunError("the plan is not bound to this run: " + ", ".join(wrong)) + if head(root) != plan["source_sha"]: + raise RunError("the checkout is not the planned source") + return plan + + +def binding(plan, job, platform=None): + return {"schema": BINDING_SCHEMA, "job": job, "platform": platform, + **{key: plan[key] for key in ("source_sha", "event", "policy_sha256", "run_id", "run_attempt")}, + "plan_sha256": digest(plan)} + + +def write_json(path, value): + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(value, indent=2) + "\n") + + +def planned_suites(policy, plan, platform): + if platform not in policy["contracts"]["platforms"]: + raise RunError("no contracts job is defined for " + str(platform)) + suites = plan["suites"][platform] + if (not isinstance(suites, list) or len(set(suites)) != len(suites) + or any(suite not in policy["suites"] or platform not in policy["suites"][suite]["platforms"] + for suite in suites)): + raise RunError("the plan names a suite the policy does not run on " + platform) + return [suite for suite in policy["suites"] if suite in suites] + + +def allowed(policy, platform, suite, case): + """The allowance that covers one not-run case exactly, or None.""" + for entry in policy["allowances"].get(platform, {}).get(suite, []): + for name in entry["cases"]: + if case == {"file": f"raw/{entry['stage']}/{name}.json", "path": entry["path"], "reason": entry["reason"]}: + return entry + return None + + +def evaluate_suite(policy, platform, suite, report, source): + """passed, incomplete-allowed or failed, and why a report is not accepted.""" + if not isinstance(report, dict): + return "failed", ["no report"] + problems = [] + if report.get("suite") != suite or report.get("schema") != "devguard-functional-qualification/v1": + problems.append("the report is not for this suite") + if (report.get("source") or {}).get("head") != source: + problems.append("the report is not for the planned source") + stages = report.get("stages") or [] + if [stage.get("name") for stage in stages] != [name for name, *_ in qualify.SUITES[suite]]: + problems.append("the report does not show every stage of the suite") + cases = [case for stage in stages for case in stage.get("not_run_cases") or []] + status = report.get("status") + if status == "passed": + if cases or any(stage.get("status") != "passed" for stage in stages): + problems.append("a passed report records a case as not run") + elif status == "incomplete": + if not cases: + problems.append("an incomplete report names no case") + for case in cases: + if allowed(policy, platform, suite, case) is None: + problems.append(f"not allowed to be not run on {platform}: {canonical(case)}") + if any(stage.get("status") not in ("passed", "incomplete") for stage in stages): + problems.append("a stage did not pass") + else: + problems.append(f"status {status}") + if problems: + return "failed", problems + return ("passed" if status == "passed" else "incomplete-allowed"), [] + + +def run_repository(env, root=ROOT, runner=subprocess.run): + plan = load_plan(env, root) + stages = plan["repository_stages"] + if (not isinstance(stages, list) or "whitespace" not in stages + or not set(stages) <= set(validate.STAGES) - validate.RUST_STAGES): + raise RunError("the plan's repository checks are not non-Rust validator stages") + out = root / REPOSITORY_OUT + out.mkdir(parents=True, exist_ok=False) + write_json(out / "binding.json", binding(plan, "repository")) + command = [sys.executable, "-B", "scripts/validate.py", "--stages", ",".join(stages), + "--output", str(REPOSITORY_OUT / "validate")] + if plan["whitespace"]["mode"] == "diff": + command += ["--diff-base", plan["whitespace"]["base"]] + completed = runner(command, cwd=root) + write_json(out / "leg.json", {"schema": LEG_SCHEMA, "job": "repository", "command": command, + "exit_code": completed.returncode, + "status": "passed" if completed.returncode == 0 else "failed"}) + return 0 if completed.returncode == 0 else 1 + + +def bind_contracts(env, platform, root=ROOT): + plan = load_plan(env, root) + policy, _ = ci_plan.read_policy(root) + if plan.get("rust") is not True: + raise RunError("the plan does not select the contracts jobs") + planned_suites(policy, plan, platform) + out = root / CONTRACTS_OUT + out.mkdir(parents=True, exist_ok=True) + if (out / "binding.json").exists(): + raise RunError("this job is already bound") + write_json(out / "binding.json", binding(plan, "contracts", platform)) + return 0 + + +def run_contracts(env, platform, root=ROOT, runner=subprocess.run): + plan = load_plan(env, root) + policy, _ = ci_plan.read_policy(root) + out = root / CONTRACTS_OUT + try: + bound = json.loads((out / "binding.json").read_text()) + except (OSError, ValueError): + bound = None + if bound != binding(plan, "contracts", platform): + raise RunError("this job was not bound to the plan before its validator ran") + results = [] + for suite in planned_suites(policy, plan, platform): + command = [sys.executable, "scripts/qualify.py", suite] + if policy["allowances"].get(platform, {}).get(suite): + command.append("--allow-incomplete") + command += ["--output", str(CONTRACTS_OUT / suite)] + print(f"::group::{suite}", flush=True) + completed = runner(command, cwd=root) + print("::endgroup::", flush=True) + try: + report = json.loads((out / suite / "report.json").read_text()) + except (OSError, ValueError): + report = None + status, problems = evaluate_suite(policy, platform, suite, report, plan["source_sha"]) + if completed.returncode != 0: + status, problems = "failed", problems + [f"exit code {completed.returncode}"] + results.append({"suite": suite, "command": command, "exit_code": completed.returncode, + "status": status, "problems": problems}) + print(f"{suite}: {status}" + "".join("\n - " + problem for problem in problems), flush=True) + failed = any(result["status"] == "failed" for result in results) + write_json(out / "leg.json", {"schema": LEG_SCHEMA, "job": "contracts", "platform": platform, + "suites": results, "status": "failed" if failed else "passed"}) + return 1 if failed else 0 + + +def status_of(path): + try: + return json.loads(path.read_text()).get("status") or "unknown" + except (OSError, ValueError, AttributeError): + return None + + +def summarize(env, platform, root=ROOT): + """The contracts job summary; it describes what exists and never fails the job.""" + policy, _ = ci_plan.read_policy(root) + try: + planned = json.loads(env.get("CI_PLAN") or "{}").get("suites", {}).get(platform, []) + except (ValueError, AttributeError): + planned = [] + out = root / CONTRACTS_OUT + lines = [f"### Contracts on {platform}", "", "| Check | Result |", "| --- | --- |", + f"| validator (ci) | {status_of(out / 'ci' / 'report.json') or 'not_run'} |"] + for suite, spec in policy["suites"].items(): + found = status_of(out / suite / "report.json") + if found: + result = found + elif suite in planned: + result = "not_run (not reached)" + elif platform in spec["platforms"]: + result = "not_selected_by_plan" + else: + result = "not_run (native macOS suite)" + lines.append(f"| {suite} | {result} |") + text = "\n".join(lines + ["", SCOPE, ""]) + if env.get("GITHUB_STEP_SUMMARY"): + with open(env["GITHUB_STEP_SUMMARY"], "a") as handle: + handle.write(text) + else: + print(text) + return 0 + + +def main(argv=None): + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("command", choices=["repository", "bind", "contracts", "summary"]) + parser.add_argument("job", nargs="?", choices=["contracts"]) + parser.add_argument("--os", dest="platform") + args = parser.parse_args(argv) + if args.command in ("bind", "contracts", "summary") and not args.platform: + parser.error(args.command + " needs --os") + if args.command == "bind" and args.job != "contracts": + parser.error("bind needs the job: contracts") + try: + if args.command == "repository": + return run_repository(os.environ) + if args.command == "bind": + return bind_contracts(os.environ, args.platform) + if args.command == "contracts": + return run_contracts(os.environ, args.platform) + return summarize(os.environ, args.platform) + except (RunError, OSError, ValueError, KeyError, TypeError, subprocess.CalledProcessError) as error: + if args.command == "summary": + print(f"no summary: {error}") + return 0 + print(f"CI job refused: {error}", file=sys.stderr) + return 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/test_ci_results.py b/scripts/test_ci_results.py new file mode 100644 index 0000000..f959270 --- /dev/null +++ b/scripts/test_ci_results.py @@ -0,0 +1,265 @@ +"""The required gate: plan re-derivation, job results, current-attempt evidence and policy-listed exceptions.""" +import json +import os +from pathlib import Path +import shutil +import subprocess +import tempfile +import unittest + +import check_ci_results as gate +import ci_plan +import ci_run +import qualify +import validate + +POLICY, DIGEST = ci_plan.read_policy() +MACOS, UBUNTU = POLICY["contracts"]["platforms"] +ALLOWED_CARGO = POLICY["allowances"][MACOS]["dg1-cargo"][0] + + +def git(root, *args): + env = dict(os.environ, GIT_AUTHOR_NAME="CI", GIT_AUTHOR_EMAIL="ci@example.invalid", + GIT_COMMITTER_NAME="CI", GIT_COMMITTER_EMAIL="ci@example.invalid") + command = ["git", "-c", "commit.gpgsign=false", "-c", "core.hooksPath=/dev/null", *args] + return subprocess.run(command, cwd=root, env=env, check=True, capture_output=True, text=True).stdout.strip() + + +def validator(source, selected, whitespace=None): + """A validator report shaped like scripts/validate.py's.""" + stages = [] + for name in validate.STAGES: + stage = {"name": name, "status": "passed" if name in selected else "not_selected_by_plan"} + if name == "whitespace" and name in selected: + stage.update(whitespace) + stages.append(stage) + complete = set(validate.DEFAULT_STAGES) <= set(selected) + return {"schema": "devguard-qualification/v1", "status": "passed", "milestone": "DG-0" if complete else None, + "toolchain_matches": True if complete else None, "source": {"head": source}, "stages": stages} + + +def suite_report(suite, source, status="passed", cases=()): + stages = [] + for name, _, *expected in qualify.SUITES[suite]: + stage = {"name": name, "status": "passed"} + if expected: + stage["not_run_cases"] = [case for case in cases if case["file"].startswith(f"raw/{name}/")] + stage["status"] = "incomplete" if stage["not_run_cases"] else "passed" + stages.append(stage) + return {"schema": "devguard-functional-qualification/v1", "suite": suite, "status": status, + "source": {"head": source}, "stages": stages} + + +def write(path, value): + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(value)) + + +class Run(unittest.TestCase): + """One workflow run: a repository, its event, the plan, job results and uploaded artifacts.""" + + def setUp(self): + self.temporary = tempfile.TemporaryDirectory() + self.addCleanup(self.temporary.cleanup) + self.root = Path(self.temporary.name) / "repo" + self.root.mkdir() + git(self.root, "init", "-q", "-b", "main") + for path in ci_plan.PLANNING: + if (ci_plan.ROOT / path).is_file(): + (self.root / path).parent.mkdir(parents=True, exist_ok=True) + shutil.copy2(ci_plan.ROOT / path, self.root / path) + (self.root / "docs/handoff").mkdir(parents=True) + (self.root / "docs/handoff/record.md").write_text("record\n") + git(self.root, "add", "-A") + git(self.root, "commit", "-q", "-m", "base") + self.base = git(self.root, "rev-parse", "HEAD") + self.artifacts = Path(self.temporary.name) / "artifacts" + + def pull_request(self, path, attempt="2"): + git(self.root, "checkout", "-q", "-b", "topic") + (self.root / path).parent.mkdir(parents=True, exist_ok=True) + (self.root / path).write_text("changed\n") + git(self.root, "add", "-A") + git(self.root, "commit", "-q", "-m", "head") + head = git(self.root, "rev-parse", "HEAD") + git(self.root, "checkout", "-q", "main") + git(self.root, "merge", "-q", "--no-ff", "-m", "merge", "topic") + self.use("pull_request", {"pull_request": {"head": {"sha": head}}}, attempt) + + def use(self, event_name, event, attempt="2"): + self.source = git(self.root, "rev-parse", "HEAD") + self.event = event + self.env = {"GITHUB_EVENT_NAME": event_name, "GITHUB_SHA": self.source, "GITHUB_RUN_ID": "77", + "GITHUB_RUN_ATTEMPT": attempt, "GITHUB_REF": "refs/heads/main"} + self.plan, _ = ci_plan.prepare(self.root, self.env, event) + self.results = {"plan": {"result": "success"}, "repository": {"result": "success"}, + "contracts": {"result": "success" if self.plan["rust"] else "skipped"}} + self.upload(attempt) + + def upload(self, attempt, plan=None): + plan = plan or self.plan + write(self.artifacts / f"ci-plan-{attempt}" / "plan.json", plan) + repository = self.artifacts / f"ci-repository-{attempt}" + write(repository / "binding.json", ci_run.binding(plan, "repository")) + write(repository / "validate" / "report.json", + validator(self.source, plan["repository_stages"], plan["whitespace"])) + if not plan["rust"]: + return + for platform in POLICY["contracts"]["platforms"]: + directory = self.artifacts / f"dg0-{platform}-{attempt}" + write(directory / "binding.json", ci_run.binding(plan, "contracts", platform)) + write(directory / "ci" / "report.json", validator(self.source, validate.DEFAULT_STAGES)) + suites = ci_run.planned_suites(POLICY, plan, platform) + write(directory / "leg.json", {"suites": [{"suite": suite} for suite in suites]}) + for suite in suites: + cases = [] + if suite == "dg1-cargo": + cases = [{"file": f"raw/native-cargo/{name}.json", "path": "", "reason": ALLOWED_CARGO["reason"]} + for name in ALLOWED_CARGO["cases"]] + write(directory / suite / "report.json", + suite_report(suite, self.source, "incomplete" if cases else "passed", cases)) + + def problems(self, plan=None, results=None, env=None, event=None): + plan_text = json.dumps(plan if plan is not None else self.plan) + return gate.problems(results or self.results, plan_text, env or self.env, event or self.event, + self.artifacts, self.root)[0] + + +class Passing(Run): + def test_a_documentation_record_run_passes_without_contracts(self): + self.pull_request("docs/handoff/2026-09-28-w3-decision-packet.md") + self.assertEqual((self.plan["rust"], self.results["contracts"]["result"]), (False, "skipped")) + found, table = gate.problems(self.results, json.dumps(self.plan), self.env, self.event, self.artifacts, self.root) + self.assertEqual(found, []) + self.assertIn(("repository contracts", "not_selected_by_plan"), table) + self.assertIn((f"contracts {MACOS}", "not_selected_by_plan"), table) + + def test_a_component_run_passes_with_only_allowed_incomplete_suites(self): + self.pull_request("crates/cargo/src/lib.rs") + found, table = gate.problems(self.results, json.dumps(self.plan), self.env, self.event, self.artifacts, self.root) + self.assertEqual(found, []) + self.assertIn((f"contracts {MACOS} dg1-cargo", "incomplete-allowed"), table) + self.assertIn((f"contracts {MACOS} dg1-scopes", "not_selected_by_plan"), table) + + def test_a_scheduled_run_is_full_and_passes(self): + self.use("schedule", {}) + self.assertEqual((self.plan["profile"], self.plan["suites"][MACOS]), ("full", list(POLICY["suites"]))) + self.assertEqual(self.problems(), []) + + +class Jobs(Run): + def test_every_job_must_have_its_planned_result(self): + self.pull_request("crates/cargo/src/lib.rs") + for job, result in (("repository", "failure"), ("contracts", "skipped"), ("contracts", "cancelled"), + ("contracts", "failure"), ("repository", "skipped")): + with self.subTest(job=job, result=result): + self.assertIn(f"{job}: {result}, expected success", + self.problems(results=dict(self.results, **{job: {"result": result}}))) + + def test_contracts_must_be_skipped_when_rust_is_not_planned(self): + self.pull_request("docs/handoff/new.md") + self.assertEqual(self.problems(results=dict(self.results, contracts={"result": "success"})), + ["contracts: success, expected skipped"]) + + def test_a_failed_plan_or_a_changed_job_set_fails(self): + self.pull_request("docs/handoff/new.md") + self.assertIn("without a plan", self.problems(results=dict(self.results, plan={"result": "failure"}))[0]) + for results in ({"plan": {"result": "success"}}, dict(self.results, extra={"result": "success"})): + self.assertEqual(self.problems(results=results), ["the gate does not see exactly the required jobs"]) + + +class PlanBinding(Run): + def test_the_plan_must_be_the_one_derived_for_this_run(self): + self.pull_request("crates/cargo/src/lib.rs") + narrowed = json.loads(json.dumps(self.plan)) + narrowed["suites"][MACOS] = narrowed["suites"][MACOS][:1] + for plan in (narrowed, dict(self.plan, profile="affected", rust=False), dict(self.plan, policy_sha256="0" * 64), + dict(self.plan, source_sha="c" * 40), dict(self.plan, run_id="78")): + with self.subTest(plan={key: plan[key] for key in ("profile", "policy_sha256", "run_id")}): + self.assertEqual(self.problems(plan=plan), + ["the plan job's plan differs from the plan derived for this run"]) + + def test_a_plan_from_an_earlier_attempt_never_satisfies_this_one(self): + self.pull_request("docs/handoff/new.md") + earlier = dict(self.plan, run_attempt="1") + self.assertEqual(self.problems(plan=earlier), ["the plan is not from this attempt 2; re-run all jobs"]) + + def test_an_event_without_a_plan_fails(self): + self.pull_request("docs/handoff/new.md") + found = self.problems(env=dict(self.env, GITHUB_EVENT_NAME="merge_group")) + self.assertEqual(len(found), 1) + self.assertIn("cannot be derived again: no plan is defined for the event merge_group", found[0]) + + +class Evidence(Run): + def test_evidence_from_an_earlier_attempt_is_ignored(self): + self.pull_request("crates/cargo/src/lib.rs") + for name in list(gate.artifact_names(self.artifacts)): + (self.artifacts / name).rename(self.artifacts / (name.rsplit("-", 1)[0] + "-1")) + found = self.problems() + self.assertIn("ci-repository-2: no artifact from this attempt", found) + self.assertIn(f"dg0-{MACOS}-2: no artifact from this attempt", found) + self.upload("2") + found, table = gate.problems(self.results, json.dumps(self.plan), self.env, self.event, self.artifacts, self.root) + self.assertEqual(found, []) + self.assertEqual(table[-1][0], "ignored: other attempts") + + def test_a_current_name_with_an_earlier_binding_fails(self): + self.pull_request("crates/cargo/src/lib.rs") + write(self.artifacts / f"dg0-{MACOS}-2" / "binding.json", + ci_run.binding(dict(self.plan, run_attempt="1"), "contracts", MACOS)) + write(self.artifacts / "ci-repository-2" / "binding.json", ci_run.binding(self.plan, "contracts", MACOS)) + found = self.problems() + self.assertIn(f"contracts {MACOS}: its evidence is not bound to this run and attempt", found) + self.assertIn("repository: its evidence is not bound to this run and attempt", found) + + def test_evidence_the_plan_does_not_produce_fails(self): + self.pull_request("docs/handoff/new.md") + write(self.artifacts / f"dg0-{MACOS}-2" / "binding.json", {}) + self.assertEqual(self.problems(), [f"dg0-{MACOS}-2: an artifact the plan does not produce"]) + + def test_the_uploaded_plan_must_be_the_planned_one(self): + self.pull_request("docs/handoff/new.md") + write(self.artifacts / "ci-plan-2" / "plan.json", dict(self.plan, profile="full")) + self.assertEqual(self.problems(), ["ci-plan: the uploaded plan is not the planned one"]) + + def test_repository_stages_must_be_exactly_the_planned_ones(self): + self.pull_request("docs/handoff/new.md") + path = self.artifacts / "ci-repository-2" / "validate" / "report.json" + cases = { + "an unplanned stage shown as passed": validator(self.source, ["whitespace", "documentation", "contracts"], + self.plan["whitespace"]), + "a planned stage left out": validator(self.source, ["whitespace"], self.plan["whitespace"]), + "another range": validator(self.source, self.plan["repository_stages"], {"mode": "tree", "base": None}), + "another source": validator("c" * 40, self.plan["repository_stages"], self.plan["whitespace"]), + } + for label, report in cases.items(): + with self.subTest(case=label): + write(path, report) + self.assertTrue(self.problems()) + path.unlink() + self.assertIn("repository: no validator report", self.problems()) + + def test_the_contracts_jobs_need_the_complete_validator_and_exactly_the_planned_suites(self): + self.pull_request("crates/cargo/src/lib.rs") + directory = self.artifacts / f"dg0-{MACOS}-2" + partial = validator(self.source, validate.DEFAULT_STAGES) + partial["stages"][-1]["status"] = "failed" + for label, change in ( + ("a failed stage", lambda: write(directory / "ci" / "report.json", partial)), + ("an unplanned suite", lambda: write(directory / "dg1-scopes" / "report.json", + suite_report("dg1-scopes", self.source))), + ("a missing suite", lambda: shutil.rmtree(directory / "dg1-upgrade")), + ("a disallowed case", lambda: write(directory / "dg1-cli" / "report.json", suite_report( + "dg1-cli", self.source, "incomplete", + [{"file": "raw/native-exec/doctor.json", "path": "", "reason": ALLOWED_CARGO["reason"]}]))), + ("another runner record", lambda: write(directory / "leg.json", {"suites": []}))): + with self.subTest(case=label): + shutil.rmtree(self.artifacts) + self.upload("2") + change() + self.assertTrue(any(line.startswith(f"contracts {MACOS}") for line in self.problems()), label) + + +if __name__ == "__main__": + unittest.main() diff --git a/scripts/test_ci_run.py b/scripts/test_ci_run.py new file mode 100644 index 0000000..de957b9 --- /dev/null +++ b/scripts/test_ci_run.py @@ -0,0 +1,269 @@ +"""The CI job runner: run binding, planned stages and suites only, and policy-listed hosted exceptions.""" +import contextlib +import io +import json +import os +from pathlib import Path +import shutil +import subprocess +import sys +import tempfile +import unittest + +import ci_plan +import ci_run +import qualify + +POLICY, DIGEST = ci_plan.read_policy() +MACOS, UBUNTU = POLICY["contracts"]["platforms"] +CARGO = ["dg1-cli", "dg1-cargo", "dg1-self-use", "dg1-upgrade", "dg1-macos"] +CAPACITY = "this host's work capacity cannot fit the Cargo jobs the case needs" + + +def git(root, *args): + env = dict(os.environ, GIT_AUTHOR_NAME="CI", GIT_AUTHOR_EMAIL="ci@example.invalid", + GIT_COMMITTER_NAME="CI", GIT_COMMITTER_EMAIL="ci@example.invalid") + command = ["git", "-c", "commit.gpgsign=false", "-c", "core.hooksPath=/dev/null", *args] + return subprocess.run(command, cwd=root, env=env, check=True, capture_output=True, text=True).stdout.strip() + + +def cargo_case(case, reason=CAPACITY): + return {"file": f"raw/native-cargo/{case}.json", "path": "", "reason": reason} + + +def report(suite, source, status="passed", cases=()): + """A functional report shaped like scripts/qualify.py's.""" + stages = [] + for name, _, *expected in qualify.SUITES[suite]: + stage = {"name": name, "status": "passed"} + if expected: + stage["not_run_cases"] = [case for case in cases if case["file"].startswith(f"raw/{name}/")] + if stage["not_run_cases"]: + stage["status"] = "incomplete" + stages.append(stage) + return {"schema": "devguard-functional-qualification/v1", "suite": suite, "status": status, + "source": {"head": source}, "stages": stages} + + +class Runner: + """Stands in for subprocess.run: records commands and writes the report each suite would.""" + + def __init__(self, root, source, reports=None, codes=None): + self.root, self.source, self.commands = root, source, [] + self.reports, self.codes = reports or {}, codes or {} + + def __call__(self, command, cwd): + self.commands.append(command) + if command[1] == "scripts/qualify.py": + suite = command[2] + value = self.reports.get(suite, report(suite, self.source)) + if value is not None: + path = Path(cwd) / command[command.index("--output") + 1] / "report.json" + path.parent.mkdir(parents=True) + path.write_text(json.dumps(value)) + return subprocess.CompletedProcess(command, self.codes.get(command[2] if len(command) > 2 else None, 0)) + + +class Job(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory() + self.addCleanup(self.temporary.cleanup) + self.root = Path(self.temporary.name) + git(self.root, "init", "-q", "-b", "main") + (self.root / "scripts").mkdir() + shutil.copy2(ci_plan.ROOT / ci_plan.POLICY, self.root / ci_plan.POLICY) + git(self.root, "add", "-A") + git(self.root, "commit", "-q", "-m", "source") + self.source = git(self.root, "rev-parse", "HEAD") + self.base = "b" * 40 + + def plan(self, *paths, base=True, reasons=()): + return ci_plan.build_plan(POLICY, DIGEST, "pull_request", self.source, self.base if base else None, + list(paths), set(reasons), ("77", "2")) + + def env(self, plan, **changes): + values = {"CI_PLAN": json.dumps(plan), "GITHUB_SHA": self.source, "GITHUB_EVENT_NAME": "pull_request", + "GITHUB_RUN_ID": "77", "GITHUB_RUN_ATTEMPT": "2", **changes} + return {key: value for key, value in values.items() if value is not None} + + +class Binding(Job): + def test_only_a_plan_for_this_commit_event_policy_run_and_attempt_is_accepted(self): + plan = self.plan("docs/handoff/x.md") + self.assertEqual(ci_run.load_plan(self.env(plan), self.root), plan) + for changes in ({"GITHUB_SHA": "c" * 40}, {"GITHUB_EVENT_NAME": "push"}, {"GITHUB_RUN_ID": "78"}, + {"GITHUB_RUN_ATTEMPT": "1"}, {"GITHUB_RUN_ATTEMPT": None}, {"CI_PLAN": ""}, + {"CI_PLAN": "[]"}, {"CI_PLAN": json.dumps(dict(plan, policy_sha256="0" * 64))}, + {"CI_PLAN": json.dumps(dict(plan, schema="other"))}): + with self.subTest(changes=changes), self.assertRaises(ci_run.RunError): + ci_run.load_plan(self.env(plan, **changes), self.root) + + def test_the_checkout_must_be_the_planned_source(self): + plan = dict(self.plan("docs/handoff/x.md"), source_sha="c" * 40) + with self.assertRaisesRegex(ci_run.RunError, "checkout"): + ci_run.load_plan(self.env(plan, GITHUB_SHA="c" * 40), self.root) + + def test_the_binding_names_the_run_and_the_plan(self): + plan = self.plan("crates/cargo/src/lib.rs") + self.assertEqual(ci_run.binding(plan, "contracts", MACOS), { + "schema": ci_run.BINDING_SCHEMA, "job": "contracts", "platform": MACOS, "source_sha": self.source, + "event": "pull_request", "policy_sha256": DIGEST, "run_id": "77", "run_attempt": "2", + "plan_sha256": ci_run.digest(plan)}) + + +class Repository(Job): + def test_the_planned_stages_run_against_the_planned_base(self): + runner = Runner(self.root, self.source) + self.assertEqual(ci_run.run_repository(self.env(self.plan("docs/handoff/x.md")), self.root, runner), 0) + self.assertEqual(runner.commands, [[sys.executable, "-B", "scripts/validate.py", "--stages", + "whitespace,documentation", "--output", "target/ci/repository/validate", + "--diff-base", self.base]]) + out = self.root / ci_run.REPOSITORY_OUT + self.assertEqual(json.loads((out / "binding.json").read_text())["job"], "repository") + self.assertEqual(json.loads((out / "leg.json").read_text())["status"], "passed") + + def test_without_a_base_the_whole_tree_is_checked(self): + runner = Runner(self.root, self.source) + plan = self.plan(base=False, reasons={"event:schedule"}) + ci_run.run_repository(self.env(plan), self.root, runner) + self.assertNotIn("--diff-base", runner.commands[0]) + self.assertIn("whitespace,source-contract,documentation,protocol-analysis", runner.commands[0]) + + def test_a_failed_check_fails_the_job(self): + runner = Runner(self.root, self.source, codes={"scripts/validate.py": 1}) + self.assertEqual(ci_run.run_repository(self.env(self.plan("docs/handoff/x.md")), self.root, runner), 1) + leg = json.loads((self.root / ci_run.REPOSITORY_OUT / "leg.json").read_text()) + self.assertEqual((leg["status"], leg["exit_code"]), ("failed", 1)) + + def test_a_rust_stage_is_never_a_repository_check(self): + plan = self.plan("docs/handoff/x.md") + plan["repository_stages"] = ["whitespace", "contracts"] + with self.assertRaises(ci_run.RunError): + ci_run.run_repository(self.env(plan), self.root, Runner(self.root, self.source)) + + +class Contracts(Job): + def run_suites(self, plan, platform=MACOS, **runner): + env = self.env(plan) + ci_run.bind_contracts(env, platform, self.root) + stub = Runner(self.root, self.source, **runner) + with contextlib.redirect_stdout(io.StringIO()): + code = ci_run.run_contracts(env, platform, self.root, stub) + leg = json.loads((self.root / ci_run.CONTRACTS_OUT / "leg.json").read_text()) + return code, stub, leg + + def test_planned_suites_run_in_order_with_allow_incomplete_only_where_the_policy_allows(self): + cases = [cargo_case(name) for name in POLICY["allowances"][MACOS]["dg1-cargo"][0]["cases"]] + code, stub, leg = self.run_suites(self.plan("crates/cargo/src/lib.rs"), reports={ + "dg1-cargo": report("dg1-cargo", self.source, "incomplete", cases)}) + self.assertEqual(code, 0) + self.assertEqual([command[2] for command in stub.commands], CARGO) + flagged = {command[2] for command in stub.commands if "--allow-incomplete" in command} + self.assertEqual(flagged, {"dg1-cargo", "dg1-macos"}) + self.assertEqual({result["suite"]: result["status"] for result in leg["suites"]}, + {**{suite: "passed" for suite in CARGO}, "dg1-cargo": "incomplete-allowed"}) + self.assertEqual(leg["status"], "passed") + + def test_a_case_not_run_for_an_unlisted_reason_fails_but_the_rest_still_run(self): + code, stub, leg = self.run_suites(self.plan("crates/cargo/src/lib.rs"), reports={ + "dg1-cargo": report("dg1-cargo", self.source, "incomplete", [cargo_case("direct-build", "no reason")])}) + self.assertEqual(code, 1) + self.assertEqual([command[2] for command in stub.commands], CARGO) + failed = [result for result in leg["suites"] if result["status"] == "failed"] + self.assertEqual([result["suite"] for result in failed], ["dg1-cargo"]) + self.assertIn("not allowed to be not run", failed[0]["problems"][0]) + + def test_an_allowance_covers_only_its_own_suite_and_case(self): + stray = {"file": "raw/native-upgrade/drain-timeout.json", "path": "", "reason": CAPACITY} + code, _, leg = self.run_suites(self.plan("crates/cargo/src/lib.rs"), reports={ + "dg1-upgrade": report("dg1-upgrade", self.source, "incomplete", [stray])}, codes={"dg1-upgrade": 2}) + self.assertEqual(code, 1) + self.assertEqual({result["suite"] for result in leg["suites"] if result["status"] == "failed"}, + {"dg1-upgrade"}) + + def test_a_failed_or_missing_report_fails_and_later_suites_still_run(self): + code, stub, leg = self.run_suites(self.plan("crates/cargo/src/lib.rs"), + reports={"dg1-cli": None}, codes={"dg1-cli": 1}) + self.assertEqual(code, 1) + self.assertEqual([command[2] for command in stub.commands], CARGO) + self.assertEqual([result["suite"] for result in leg["suites"] if result["status"] == "failed"], ["dg1-cli"]) + + def test_only_suites_planned_for_this_platform_start(self): + code, stub, leg = self.run_suites(self.plan("crates/cargo/src/lib.rs"), platform=UBUNTU) + self.assertEqual((code, stub.commands, leg["suites"]), (0, [], [])) + _, stub, _ = self.run_suites_fresh(self.plan("crates/core/src/lib.rs"), UBUNTU) + self.assertEqual([command[2] for command in stub.commands], ["dg1-authority", "dg1-auth"]) + + def run_suites_fresh(self, plan, platform): + shutil.rmtree(self.root / ci_run.CONTRACTS_OUT) + return self.run_suites(plan, platform) + + def test_suites_need_the_binding_written_before_the_validator(self): + plan = self.plan("crates/cargo/src/lib.rs") + with self.assertRaisesRegex(ci_run.RunError, "not bound"): + ci_run.run_contracts(self.env(plan), MACOS, self.root, Runner(self.root, self.source)) + ci_run.bind_contracts(self.env(plan), UBUNTU, self.root) + with self.assertRaisesRegex(ci_run.RunError, "not bound"): + ci_run.run_contracts(self.env(plan), MACOS, self.root, Runner(self.root, self.source)) + + def test_a_plan_without_rust_or_an_unknown_platform_binds_nothing(self): + with self.assertRaisesRegex(ci_run.RunError, "does not select"): + ci_run.bind_contracts(self.env(self.plan("docs/handoff/x.md")), MACOS, self.root) + with self.assertRaises(ci_run.RunError): + ci_run.bind_contracts(self.env(self.plan("crates/cargo/src/lib.rs")), "windows-2022", self.root) + self.assertFalse((self.root / ci_run.CONTRACTS_OUT / "binding.json").exists()) + + def test_a_plan_that_names_an_unknown_or_misplaced_suite_is_refused(self): + for suites in (["dg1-other"], ["dg1-scopes"], ["dg1-auth", "dg1-auth"]): + plan = self.plan("crates/core/src/lib.rs") + plan["suites"][UBUNTU] = suites + with self.subTest(suites=suites), self.assertRaises(ci_run.RunError): + ci_run.planned_suites(POLICY, plan, UBUNTU) + + +class Reports(Job): + def check(self, value, suite="dg1-upgrade", platform=MACOS): + return ci_run.evaluate_suite(POLICY, platform, suite, value, self.source) + + def test_only_a_complete_matching_report_passes(self): + self.assertEqual(self.check(report("dg1-upgrade", self.source)), ("passed", [])) + partial = report("dg1-upgrade", self.source) + partial["stages"] = partial["stages"][:-1] + stray = report("dg1-upgrade", self.source) + stray["stages"][-1]["not_run_cases"] = [cargo_case("direct-build")] + for value in (None, report("dg1-cargo", self.source), report("dg1-upgrade", "c" * 40), partial, stray, + report("dg1-upgrade", self.source, "failed"), report("dg1-upgrade", self.source, "not_run"), + report("dg1-upgrade", self.source, "incomplete")): + with self.subTest(value=value): + self.assertEqual(self.check(value)[0], "failed") + + def test_allowances_are_per_platform(self): + value = report("dg1-scopes", self.source, "incomplete", [{ + "file": "raw/native-scopes/scope-refusals.json", "path": "unclamped_root", + "reason": POLICY["allowances"][MACOS]["dg1-scopes"][0]["reason"]}]) + self.assertEqual(self.check(value, "dg1-scopes"), ("incomplete-allowed", [])) + self.assertEqual(self.check(value, "dg1-scopes", UBUNTU)[0], "failed") + value["stages"][-1]["not_run_cases"][0]["path"] = "" + self.assertEqual(self.check(value, "dg1-scopes")[0], "failed") + + +class Summary(Job): + def test_the_summary_says_what_ran_and_what_the_plan_left_out(self): + plan = self.plan("crates/cargo/src/lib.rs") + out = self.root / ci_run.CONTRACTS_OUT + ci_run.write_json(out / "ci" / "report.json", {"status": "passed"}) + ci_run.write_json(out / "dg1-cli" / "report.json", {"status": "passed"}) + summary = self.root / "summary.md" + ci_run.summarize(dict(self.env(plan), GITHUB_STEP_SUMMARY=str(summary)), MACOS, self.root) + text = summary.read_text() + self.assertIn("| validator (ci) | passed |", text) + self.assertIn("| dg1-cli | passed |", text) + self.assertIn("| dg1-cargo | not_run (not reached) |", text) + self.assertIn("| dg1-scopes | not_selected_by_plan |", text) + self.assertIn(ci_run.SCOPE, text) + ci_run.summarize(dict(self.env(plan), GITHUB_STEP_SUMMARY=str(summary)), UBUNTU, self.root) + self.assertIn("| dg1-scopes | not_run (native macOS suite) |", summary.read_text()) + + +if __name__ == "__main__": + unittest.main() From 673914283b0243a1459393df0f37e988eda73832 Mon Sep 17 00:00:00 2001 From: Seongjae Date: Mon, 28 Sep 2026 22:42:12 +0900 Subject: [PATCH 4/6] ci: drive the workflow from the plan with a required gate The workflow keeps its name, the contracts matrix (contracts (macos-14), contracts (ubuntu-24.04)), the toolchain step, the exact validator step and the dg0-- artifacts, and adds: - Plan: runs the CI policy tests, then scripts/ci_plan.py; - Repository checks: whitespace (git diff --check, now a real CI check) and the planned non-Rust stages, without Rust; - contracts: only when the plan selects Rust; the suites and the hosted-runner exceptions come from the policy; - Required checks: always runs and passes only when this run and attempt did what the plan requires. Pull requests and pushes to main are planned; branch pushes no longer duplicate pull-request runs. A daily scheduled run and every manual run are full: the scheduled run is the compensating control that keeps qualifying the whole repository while pull requests and main pushes run only affected checks. A newer push to a pull request cancels its older run. No expression is expanded inside a script, and every checkout drops its credentials. --- .github/workflows/ci.yml | 156 +++++++++++++++++++++++------------- scripts/test_ci_workflow.py | 125 +++++++++++++++++++++++++++++ 2 files changed, 226 insertions(+), 55 deletions(-) create mode 100644 scripts/test_ci_workflow.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 1a41a43..9292391 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -1,15 +1,79 @@ name: Contracts and service boundary +# Pull requests and pushes to main run the checks their changed paths can affect, as +# scripts/ci-policy.json classifies them. Unknown paths, CI, planning, qualification, Cargo and +# toolchain inputs, and a missing or untrusted base run everything. The daily scheduled run and +# every manual run are always full: they are the compensating control that keeps qualifying the +# whole repository on both platforms, every native suite included, now that pull requests and +# main pushes run only affected checks. "Required checks" passes only when this run and attempt +# did exactly what its plan requires. on: - push: pull_request: + push: + branches: [main] + schedule: + - cron: '17 18 * * *' workflow_dispatch: permissions: contents: read +concurrency: + # A newer push to a pull request replaces its running checks, and the replaced run's gate + # fails; main, scheduled and manual runs always complete. + group: ci-${{ github.event_name == 'pull_request' && format('pr-{0}', github.event.pull_request.number) || format('run-{0}', github.run_id) }} + cancel-in-progress: ${{ github.event_name == 'pull_request' }} + jobs: + plan: + name: Plan + runs-on: ubuntu-24.04 + timeout-minutes: 10 + outputs: + plan: ${{ steps.plan.outputs.plan }} + rust: ${{ steps.plan.outputs.rust }} + steps: + - uses: actions/checkout@v5 + with: + fetch-depth: 0 + persist-credentials: false + - name: Verify the CI policy + run: python3 -B -m unittest discover -s scripts -p 'test_ci_*.py' + - id: plan + name: Plan affected checks + run: python3 -B scripts/ci_plan.py + - name: Preserve the plan + uses: actions/upload-artifact@v6 + with: + name: ci-plan-${{ github.run_attempt }} + path: target/ci/plan/ + if-no-files-found: error + + repository: + name: Repository checks + needs: plan + runs-on: ubuntu-24.04 + timeout-minutes: 15 + env: + CI_PLAN: ${{ needs.plan.outputs.plan }} + steps: + - uses: actions/checkout@v5 + with: + fetch-depth: 0 + persist-credentials: false + - name: Whitespace and planned non-Rust validation + run: python3 -B scripts/ci_run.py repository + - name: Preserve repository evidence + if: always() + uses: actions/upload-artifact@v6 + with: + name: ci-repository-${{ github.run_attempt }} + path: target/ci/repository/ + if-no-files-found: error + contracts: + needs: plan + if: ${{ needs.plan.outputs.rust == 'true' }} strategy: fail-fast: false matrix: @@ -18,57 +82,22 @@ jobs: env: CARGO_BUILD_JOBS: "1" RUST_TEST_THREADS: "1" + CI_PLAN: ${{ needs.plan.outputs.plan }} + CI_OS: ${{ matrix.os }} steps: - uses: actions/checkout@v5 with: persist-credentials: false + - name: Bind this job to the plan + run: python3 -B scripts/ci_run.py bind contracts --os "$CI_OS" - name: Select the qualification toolchain run: rustup toolchain install 1.95.0 --profile minimal --component clippy --component rustfmt - name: Contract regression and workspace validation run: python3 scripts/validate.py --output target/qualification/ci - - name: Canonical authority functional checks - run: python3 scripts/qualify.py dg1-authority --output target/qualification/dg1-authority - - name: Native local authentication functional checks - run: python3 scripts/qualify.py dg1-auth --output target/qualification/dg1-auth - - name: Native host evidence functional checks - if: runner.os == 'macOS' - run: python3 scripts/qualify.py dg1-probes --output target/qualification/dg1-probes - - name: Native scope and policy readback functional checks - if: runner.os == 'macOS' - # Hosted runners clamp every process, so the unclamped-root case can only - # be recorded as not run there; the summary still shows incomplete. - run: python3 scripts/qualify.py dg1-scopes --allow-incomplete --output target/qualification/dg1-scopes - - name: Native launch helper functional checks - if: runner.os == 'macOS' - # The same clamp leaves the unclamped-helper refusal case not run there. - run: python3 scripts/qualify.py dg1-launch --allow-incomplete --output target/qualification/dg1-launch - - name: Native reconciliation functional checks - if: runner.os == 'macOS' - run: python3 scripts/qualify.py dg1-reconcile --output target/qualification/dg1-reconcile - - name: Native command-line owner functional checks - if: runner.os == 'macOS' - run: python3 scripts/qualify.py dg1-cli --output target/qualification/dg1-cli - - name: Native Cargo adapter functional checks - if: runner.os == 'macOS' - # Hosted runners' work capacity cannot fit one Cargo job (1 CPU and - # 2 GiB), so real-build cases are recorded not run there; refusals still run. - run: python3 scripts/qualify.py dg1-cargo --allow-incomplete --output target/qualification/dg1-cargo - - name: Native installation functional checks - if: runner.os == 'macOS' - # Transient launchd jobs need the user's gui domain; a runner without - # one records the launchd case not run, and the summary shows incomplete. - run: python3 scripts/qualify.py dg1-bootstrap --allow-incomplete --output target/qualification/dg1-bootstrap - - name: Native parent lease and candidate functional checks - if: runner.os == 'macOS' - run: python3 scripts/qualify.py dg1-self-use --output target/qualification/dg1-self-use - - name: Native upgrade and repair functional checks - if: runner.os == 'macOS' - run: python3 scripts/qualify.py dg1-upgrade --output target/qualification/dg1-upgrade - - name: Native SLO harness functional checks - if: runner.os == 'macOS' - # The fixture case needs Google Chrome; a runner without it records the - # case not run, and the summary shows incomplete. No SLO is measured here. - run: python3 scripts/qualify.py dg1-macos --allow-incomplete --output target/qualification/dg1-macos + - name: Planned functional checks + # The suites, their order and the hosted-runner exceptions for incomplete suites come + # from scripts/ci-policy.json ("suites" and "allowances"). + run: python3 -B scripts/ci_run.py contracts --os "$CI_OS" - name: Preserve qualification evidence if: always() uses: actions/upload-artifact@v6 @@ -78,14 +107,31 @@ jobs: if-no-files-found: error - name: Summarize qualification if: always() - run: | - python3 - <<'PY' - import json, os - from pathlib import Path - with open(os.environ['GITHUB_STEP_SUMMARY'], 'a') as out: - for name in ['ci', 'dg1-authority', 'dg1-auth', 'dg1-probes', 'dg1-scopes', 'dg1-launch', 'dg1-reconcile', 'dg1-cli', 'dg1-cargo', 'dg1-bootstrap', 'dg1-self-use', 'dg1-upgrade', 'dg1-macos']: - path = Path('target/qualification') / name / 'report.json' - status = json.loads(path.read_text())['status'] if path.exists() else 'not_run' - out.write(name + ': ' + status + '\n\n') - out.write('Contract regression, service/UDS, native host evidence, native scope, launch helper, reconciliation, command-line owner, Cargo adapter, installation, parent lease, upgrade and repair, and SLO harness functional checks only; native suites run on macOS. Real self-use under an installed parent and the SLO protocol itself are not run.\n') - PY + run: python3 -B scripts/ci_run.py summary --os "$CI_OS" + + required: + name: Required checks + # Never skipped: it proves that the planned jobs passed with evidence from this attempt and + # that the others were skipped. + if: ${{ always() }} + needs: [plan, repository, contracts] + runs-on: ubuntu-24.04 + timeout-minutes: 10 + steps: + - uses: actions/checkout@v5 + with: + fetch-depth: 0 + persist-credentials: false + - name: Download this run's evidence + # Every artifact of this run, each in a directory named after it (a lone artifact is + # unpacked in place and so matches no expected name). A missing artifact fails the + # check below, not this download. + continue-on-error: true + uses: actions/download-artifact@v7 + with: + path: target/ci-artifacts + - name: Require the planned checks + env: + CI_RESULTS: ${{ toJSON(needs) }} + CI_PLAN: ${{ needs.plan.outputs.plan }} + run: python3 -B scripts/check_ci_results.py --artifacts target/ci-artifacts diff --git a/scripts/test_ci_workflow.py b/scripts/test_ci_workflow.py new file mode 100644 index 0000000..f605563 --- /dev/null +++ b/scripts/test_ci_workflow.py @@ -0,0 +1,125 @@ +"""The workflow consumes the plan: job wiring, the unskippable gate, triggers and no inline policy.""" +import re +import unittest + +import check_ci_results +import ci_plan + +WORKFLOW = (ci_plan.ROOT / ".github/workflows/ci.yml").read_text() +POLICY, _ = ci_plan.read_policy() + + +def jobs(text): + body = text.split("\njobs:\n", 1)[1] + parts = re.split(r"^ ([a-z0-9-]+):\n", body, flags=re.M) + return dict(zip(parts[1::2], parts[2::2])) + + +def steps(job): + return re.split(r"^ - ", job, flags=re.M)[1:] + + +class Workflow(unittest.TestCase): + def setUp(self): + self.jobs = jobs(WORKFLOW) + + def test_the_jobs_are_the_gate_requirements_and_the_gate(self): + self.assertEqual(set(self.jobs), check_ci_results.REQUIRED_JOBS | {"required"}) + + def test_the_gate_always_runs_needs_every_job_and_has_a_stable_name(self): + gate = self.jobs["required"] + self.assertIn(" name: Required checks\n", gate) + self.assertIn(" if: ${{ always() }}\n", gate) + needs = re.search(r"^ needs: \[([^\]]*)\]$", gate, re.M).group(1) + self.assertEqual({name.strip() for name in needs.split(",")}, check_ci_results.REQUIRED_JOBS) + self.assertIn("CI_RESULTS: ${{ toJSON(needs) }}", gate) + self.assertIn("CI_PLAN: ${{ needs.plan.outputs.plan }}", gate) + self.assertIn("run: python3 -B scripts/check_ci_results.py --artifacts target/ci-artifacts", gate) + download = next(step for step in steps(gate) if "download-artifact" in step) + self.assertIn("uses: actions/download-artifact@v7", download) + self.assertIn("path: target/ci-artifacts", download) + # Every artifact of the run, each in its own directory; the gate picks this attempt's. + for key in ("name:", "pattern:", "merge-multiple:", "run-id:", "github-token:"): + self.assertNotIn(key, download.split("with:", 1)[1]) + self.assertIn("fetch-depth: 0", gate) + + def test_the_plan_job_verifies_the_policy_and_exports_the_plan(self): + plan = self.jobs["plan"] + self.assertIn("fetch-depth: 0", plan) + self.assertIn("run: python3 -B -m unittest discover -s scripts -p 'test_ci_*.py'", plan) + self.assertIn("run: python3 -B scripts/ci_plan.py", plan) + self.assertLess(plan.index("test_ci_*.py"), plan.index("scripts/ci_plan.py")) + for output in ("plan", "rust"): + self.assertIn(f" {output}: ${{{{ steps.plan.outputs.{output} }}}}\n", plan) + self.assertIn("name: ci-plan-${{ github.run_attempt }}", plan) + + def test_repository_checks_always_follow_the_plan(self): + repository = self.jobs["repository"] + self.assertIn(" needs: plan\n", repository) + self.assertNotIn("\n if:", repository) + self.assertIn("fetch-depth: 0", repository) + self.assertIn("CI_PLAN: ${{ needs.plan.outputs.plan }}", repository) + self.assertIn("run: python3 -B scripts/ci_run.py repository", repository) + self.assertIn("name: ci-repository-${{ github.run_attempt }}", repository) + + def test_contracts_keep_the_legacy_validation_and_run_only_planned_suites(self): + contracts = self.jobs["contracts"] + self.assertIn(" needs: plan\n", contracts) + self.assertIn(" if: ${{ needs.plan.outputs.rust == 'true' }}\n", contracts) + platforms = re.search(r"^ os: \[([^\]]*)\]$", contracts, re.M).group(1).split(", ") + self.assertEqual(platforms, POLICY["contracts"]["platforms"]) + for line in ('CARGO_BUILD_JOBS: "1"', 'RUST_TEST_THREADS: "1"', "CI_OS: ${{ matrix.os }}", + "CI_PLAN: ${{ needs.plan.outputs.plan }}", + "run: rustup toolchain install 1.95.0 --profile minimal --component clippy --component rustfmt", + "run: python3 scripts/validate.py --output target/qualification/ci", + "name: dg0-${{ matrix.os }}-${{ github.run_attempt }}"): + self.assertIn(line, contracts) + order = [contracts.index(text) for text in ( + 'run: python3 -B scripts/ci_run.py bind contracts --os "$CI_OS"', "rustup toolchain install", + "scripts/validate.py --output", 'run: python3 -B scripts/ci_run.py contracts --os "$CI_OS"', + "actions/upload-artifact@v6", 'run: python3 -B scripts/ci_run.py summary --os "$CI_OS"')] + self.assertEqual(order, sorted(order)) + self.assertNotIn("\n name:", contracts, "the check names stay contracts ()") + + def test_evidence_is_always_preserved_and_required(self): + for name in ("repository", "contracts"): + upload = next(step for step in steps(self.jobs[name]) if "upload-artifact" in step) + self.assertIn("if: always()", upload) + self.assertIn("if-no-files-found: error", upload) + summary = next(step for step in steps(self.jobs["contracts"]) if "ci_run.py summary" in step) + self.assertIn("if: always()", summary) + + def test_no_policy_lives_in_the_workflow(self): + self.assertNotIn("--allow-incomplete", WORKFLOW) + self.assertIsNone(re.search(r"dg1-[a-z]", WORKFLOW)) + self.assertNotIn("runner.os", WORKFLOW) + self.assertNotIn("paths:", WORKFLOW) + self.assertNotIn("paths-ignore:", WORKFLOW) + + def test_no_expression_is_expanded_inside_a_script(self): + self.assertNotRegex(WORKFLOW, r"run: *[|>]") + for line in WORKFLOW.splitlines(): + if line.strip().startswith("run:"): + self.assertNotIn("${{", line) + + def test_every_checkout_drops_its_credentials(self): + checkouts = [step for job in self.jobs.values() for step in steps(job) if "actions/checkout@" in step] + self.assertEqual(len(checkouts), len(self.jobs)) + for step in checkouts: + self.assertIn("uses: actions/checkout@v5", step) + self.assertIn("persist-credentials: false", step) + + def test_triggers_permissions_and_concurrency(self): + header = WORKFLOW.split("\njobs:\n", 1)[0] + self.assertIn("on:\n pull_request:\n push:\n branches: [main]\n schedule:\n - cron: '17 18 * * *'\n" + " workflow_dispatch:\n", header) + self.assertNotIn("merge_group", header) + self.assertIn("compensating control", header) + self.assertIn("\npermissions:\n contents: read\n", header) + self.assertIn("format('pr-{0}', github.event.pull_request.number) || format('run-{0}', github.run_id)", header) + self.assertIn("cancel-in-progress: ${{ github.event_name == 'pull_request' }}", header) + self.assertEqual(set(ci_plan.EVENTS), {"pull_request", "push", "schedule", "workflow_dispatch"}) + + +if __name__ == "__main__": + unittest.main() From de87e2bd451f74aa61f0fa6044b12866309e7a69 Mon Sep 17 00:00:00 2001 From: Seongjae Date: Mon, 28 Sep 2026 22:42:23 +0900 Subject: [PATCH 5/6] docs(planning): describe affected-check CI in the verification plan The verification plan's commands, status list, suite paragraph and evidence rules now describe the affected-check model: path classes, full-coverage triggers, the daily scheduled full run as its compensating control, not_selected_by_plan, policy-listed hosted exceptions and the current-attempt gate. The CS-RG sections are unchanged. The sentence that CodeSpace's existing CI checks still run for documentation PRs is out of date since CodeSpace #70; it now says what CodeSpace main (ab0341b) runs for a documentation-only change, as its workflow, policy and docs-only run 36336288225 show. The Korean counterpart was reviewed and only the planning-verification pair's hashes were recorded. --- docs/ko/planning/verification.md | 34 ++++++++++++++++++++++++++------ docs/planning/verification.md | 32 +++++++++++++++++++++++++----- docs/translations.json | 4 ++-- 3 files changed, 57 insertions(+), 13 deletions(-) diff --git a/docs/ko/planning/verification.md b/docs/ko/planning/verification.md index bd5f0fa..1588d40 100644 --- a/docs/ko/planning/verification.md +++ b/docs/ko/planning/verification.md @@ -19,7 +19,7 @@ | V-LINUX | 실제 Linux scope와 제품 조합 | DGL-C05/C06 controller·ancestor·권한·자손·SLO | 미구현; fake cgroup과 분리 | | V-CACHE / V-ADAPTER | cache/tool/executor | DGC-C06, DGA-C02/C04/C06/C08의 해당 조합 | 미구현 | -상태는 `passed`, `failed`, `not_run`, `inconclusive`를 구분한다. 기존 도구가 `incomplete`를 출력하면 그 값을 보존하고 어떤 필수 범위가 미완료인지 덧붙인다. 실행하지 않은 시험과 환경 부적합을 통과로 바꾸지 않는다. 새 suite는 발견/실행 case 수가 0일 때 실패해야 한다. +상태는 `passed`, `failed`, `not_run`, `inconclusive`를 구분한다. CI 계획이 제외한 단계나 suite는 `passed`가 아니라 `not_selected_by_plan`으로 기록한다. 기존 도구가 `incomplete`를 출력하면 그 값을 보존하고 어떤 필수 범위가 미완료인지 덧붙인다. 실행하지 않은 시험과 환경 부적합을 통과로 바꾸지 않는다. 새 suite는 발견/실행 case 수가 0일 때 실패해야 한다. ## 현재 재현 가능한 명령 @@ -41,9 +41,12 @@ python3 scripts/qualify.py dg1-self-use --offline python3 scripts/qualify.py dg1-upgrade --offline python3 scripts/qualify.py dg1-macos --offline git diff --check +python3 scripts/validate.py --stages whitespace,source-contract,documentation --diff-base origin/main +python3 scripts/ci_plan.py --base origin/main --head HEAD +python3 -B -m unittest discover -s scripts -p 'test_ci_*.py' ``` -locked dependency가 없으면 offline을 제거한다. `--output`은 존재하지 않는 ignored repo 내부 경로만 사용한다. 정상 PATH가 다른 Rust면 설치된 1.95.0 toolchain을 먼저 둔다. mismatch 허용 결과는 supplemental이며 지정 toolchain qualification이 아니다. 이 검사는 OS/foreground/self-use를 `not_run`으로 남긴다. +locked dependency가 없으면 offline을 제거한다. `--output`은 존재하지 않는 ignored repo 내부 경로만 사용한다. 정상 PATH가 다른 Rust면 설치된 1.95.0 toolchain을 먼저 둔다. mismatch 허용 결과는 supplemental이며 지정 toolchain qualification이 아니다. 이 검사는 OS/foreground/self-use를 `not_run`으로 남긴다. `--stages`는 지정한 validator 단계만 고정된 순서로 실행하며, Cargo 단계를 지정했을 때만 Rust toolchain을 확인한다. 일부 단계만 실행한 결과는 DG-0을 주장하지 않는다. `--diff-base`는 `whitespace` 단계(`git diff --check`)를 commit된 변경으로 한정한다. `ci_plan.py --base`는 그 diff에 대해 CI가 만들 계획을 출력한다. CodeSpace root, Node **24.21.0**, npm **11.19.0**, Python 3.11 이상(CI 3.14): @@ -57,7 +60,7 @@ python3 -B docs-site/scripts/site.py check `DOCS_PYTHON`으로 Python을 명시할 수 있다. 번역 pair를 실제 대조한 뒤에만 `python3 -B scripts/check_docs.py record --id devguard-integration`으로 해당 pair의 hash를 갱신한다. 전체 registry를 일괄 재기록하지 않는다. 사이트 source Markdown·언어 이동·긴 표·desktop/narrow·light/dark를 확인한다. 문서 PR 병합 후 별도 main CI와 CodeSpace 문서 배포까지 확인해야 준비 단계가 완료된다. -현재 upstream runtime gate는 다음과 같다. **이번 문서 변경에서는 기존 CI가 이를 실행하며, 아래 명령 존재를 새 DevGuard 기능 구현으로 해석하지 않는다.** +현재 upstream runtime gate는 다음과 같다. **아래 명령 존재를 새 DevGuard 기능 구현으로 해석하지 않는다.** ```sh python3 scripts/validate-upstream.py all @@ -65,7 +68,26 @@ python3 scripts/validate-upstream.py macos-core dependencies python3 scripts/validate-upstream.py linux-isolation ``` -첫 명령은 macos-core를 포함하지 않는다. macOS에서 Linux isolation skip/incomplete는 실제 Linux 검증을 대체하지 않는다. Linux controller와 sandbox 권한이 있는 환경의 결과를 별도 확인한다. target은 기존 `target/upstream-validation`, 보호 보고서는 `target/upstream-reports/local` 규칙을 유지한다. +첫 명령은 macos-core를 포함하지 않는다. macOS에서 Linux isolation skip/incomplete는 실제 Linux 검증을 대체하지 않는다. Linux controller와 sandbox 권한이 있는 환경의 결과를 별도 확인한다. target은 기존 `target/upstream-validation`, 보호 보고서는 `target/upstream-reports/local` 규칙을 유지한다. CodeSpace CI는 변경 경로로 Rust leg를 계획한다. 문서만 바꾼 변경은 Rust leg를 선택하지 않으며, policy scan·format과 pin 검사·결과 gate와 별도 문서 사이트 workflow는 그대로 실행한다. 예약·수동 실행은 전체를 실행한다. + +## CI 선택 + +DevGuard CI는 `scripts/ci-policy.json`이 변경 경로를 분류한 결과에 따라, 변경이 영향을 줄 수 있는 검사만 실행한다. `scripts/ci_plan.py`가 계획을 만들고 `scripts/ci_run.py`가 각 job의 몫을 실행하며, `scripts/check_ci_results.py`가 **Required checks** job에서 계획의 이행을 강제한다. + +| 변경 | Rust 없는 저장소 검사 | Rust 검증과 기능 suite | +| --- | --- | --- | +| 이력 기록(`docs/handoff/**`) | whitespace, documentation | 선택하지 않음 | +| 규범 입력: 설계 문서와 그 기록, 계약, 운영, 계획, 번역, milestone ledger, `AGENTS.md`, `README.md`, `LICENSE`, `NOTICE` | whitespace와 documentation. 승인 설계·`docs/design-source.json`·`milestones.json`이 바뀌면 source-contract도 실행 | 선택하지 않음 | +| workspace crate | whitespace | macOS와 Ubuntu의 전체 validator, 그리고 package나 먼저 빌드하는 binary가 그 crate에 의존하는 기능 suite | +| CI·계획·검증·qualification script, Cargo·toolchain·Git 입력, `.devguard.toml`, 어느 분류에도 속하지 않는 경로 | 모든 비 Rust 단계 | 전체 validator와 모든 suite | + +- whitespace 검사는 변경에 대한 `git diff --check`이며, 신뢰할 base가 없으면 tree 전체를 검사한다. tree 전체 검사는 policy가 SHA-256으로 고정한 파일만 예외로 둔다. +- pull request와 `main` push는 계획대로 실행한다. base가 없거나 신뢰할 수 없을 때, diff가 비었을 때, 계획 입력이 바뀌었을 때도 전체를 실행하므로 pull request가 자기 검사를 줄일 수 없다. 예약·수동 실행은 항상 전체다. `merge_group`을 포함한 그 밖의 event에는 계획이 없으며 gate가 실패한다. +- 매일 예약된 전체 실행은 이 모델을 위한 보완 통제다. pull request와 `main` push가 영향받는 검사만 실행하는 동안, 두 플랫폼에서 모든 native suite를 포함해 저장소 전체를 계속 qualification한다. 문서만 바꾼 병합은 `main`에서도 문서 검사만 실행한다. +- Rust 변경은 여전히 전체 validator를 실행하며, workspace clippy와 시험의 범위는 그대로다. 선택하는 것은 기능 suite뿐이다. +- 계획이 제외한 단계나 suite는 `not_selected_by_plan`으로 기록한다. hosted runner 예외, 즉 `--allow-incomplete`로 실행하는 suite는 정확한 case와 이유와 함께 policy에 나열한다. 그 밖의 case가 `not_run`으로 기록되면 실행은 실패한다. +- gate는 계획을 다시 도출하며, 이 commit·event·policy digest에 묶인 현재 run과 현재 attempt의 증거만 받는다. 실패한 job만 다시 실행하지 말고 모든 job을 다시 실행한다. +- 추적되는 경로가 어느 분류에도 속하지 않거나, 문서 pattern이 아무 파일과도 맞지 않거나, 고정한 예외 파일의 byte가 바뀌면 policy 시험이 실패하므로, 그런 변경은 policy와 함께 검토한다. ## 이번 문서의 정합성 검사 @@ -81,7 +103,7 @@ V-DOC-DG는 scripts/check_docs.py와 수동 의미 검토로 영어/한국어 ha ## 향후 suite와 장애 주입 계약 -scripts/qualify.py dg1-authority --offline은 C01 설정·저장소, scripts/qualify.py dg1-auth --offline은 C02 인증·transport, scripts/qualify.py dg1-probes --offline은 C03 native 증거, scripts/qualify.py dg1-scopes --offline은 C04 정책·scope 증거, scripts/qualify.py dg1-launch --offline은 C05 launch helper, scripts/qualify.py dg1-reconcile --offline은 C06 대조, scripts/qualify.py dg1-cli --offline은 C07 명령행 owner, scripts/qualify.py dg1-cargo --offline은 C08 Cargo adapter, scripts/qualify.py dg1-bootstrap --offline은 C09 설치, scripts/qualify.py dg1-self-use --offline은 C10 부모 lease와 후보 authority, scripts/qualify.py dg1-upgrade --offline은 C11 upgrade와 repair, scripts/qualify.py dg1-macos --offline은 C12 SLO harness 검증을 제공하며, 그 protocol은 scripts/measure.py macos이다. 그 밖의 DevGuard suite와 CodeSpace scripts/qualify-devguard.py는 해당 작업에서 제공할 예정 인터페이스다. 각 구현 PR이 실제 CLI·case inventory·nonzero case assertion·timeout·log 수집·격리 cleanup을 구현하고 제공 명령을 갱신해야 한다. macOS/Ubuntu CI는 전체 validator와 이식 가능한 두 기능 suite를 실행하고 각각의 report·log를 보존한다. Native suite는 macOS CI에서만 실행하고 그 밖의 환경에서는 `not_run`으로 기록한다. 증거를 제공할 수 없는 플랫폼에서 통과로 처리하지 않는다. 각 native 단계는 만들어야 할 raw receipt를 선언한다. 어떤 경우를 `not_run`으로 기록한 receipt가 있으면 suite는 `passed`가 아니라 `incomplete`가 된다. +scripts/qualify.py dg1-authority --offline은 C01 설정·저장소, scripts/qualify.py dg1-auth --offline은 C02 인증·transport, scripts/qualify.py dg1-probes --offline은 C03 native 증거, scripts/qualify.py dg1-scopes --offline은 C04 정책·scope 증거, scripts/qualify.py dg1-launch --offline은 C05 launch helper, scripts/qualify.py dg1-reconcile --offline은 C06 대조, scripts/qualify.py dg1-cli --offline은 C07 명령행 owner, scripts/qualify.py dg1-cargo --offline은 C08 Cargo adapter, scripts/qualify.py dg1-bootstrap --offline은 C09 설치, scripts/qualify.py dg1-self-use --offline은 C10 부모 lease와 후보 authority, scripts/qualify.py dg1-upgrade --offline은 C11 upgrade와 repair, scripts/qualify.py dg1-macos --offline은 C12 SLO harness 검증을 제공하며, 그 protocol은 scripts/measure.py macos이다. 그 밖의 DevGuard suite와 CodeSpace scripts/qualify-devguard.py는 해당 작업에서 제공할 예정 인터페이스다. 각 구현 PR이 실제 CLI·case inventory·nonzero case assertion·timeout·log 수집·격리 cleanup을 구현하고 제공 명령을 갱신해야 한다. CI는 계획이 선택한 validator와 기능 suite를 실행하고([CI 선택](#ci-선택) 참조) 각각의 report·log를 보존한다. Native suite는 macOS CI에서만 실행하고 그 밖의 환경에서는 `not_run`으로 기록한다. 증거를 제공할 수 없는 플랫폼에서 통과로 처리하지 않는다. 각 native 단계는 만들어야 할 raw receipt를 선언한다. 어떤 경우를 `not_run`으로 기록한 receipt가 있으면 suite는 `passed`가 아니라 `incomplete`가 된다. C03 증거는 다음 파일에 기록한다. - **Raw receipt** (보고서의 `raw/` 디렉터리): 단위를 포함한 boot ID·시계 읽기, 호스트 용량, zombie·reap·거부 관측을 포함한 반복 프로세스 정체성, 계산된 비율을 포함한 native 압력 읽기, 마지막 sample 이후 admission이 닫히기까지 측정한 시간, 주입한 실패부터 Critical까지 걸린 서비스 loop 시간과 멈춘 probe에 대한 서비스 loop의 동작. @@ -244,6 +266,6 @@ DG-1에서는 standalone CLI/daemon의 대응되는 조회·종료와 개발/for raw pressure/latency/jobs 전환, peak memory, 완료 시간·처리량, 거절 이유, 큐/버퍼 peak, attempt/slot/lease lifecycle, 장애 지점, 실제 종료/readback을 저장한다. report·로그·원시 파일 manifest/hash를 함께 보존하고 token/credential/사용자 payload를 정제한다. source와 evidence를 같은 의미로 취급하지 않는다. -qualification은 정확한 artifact·정책·환경 조합에 귀속한다. documentation-only head에서 계약 회귀가 통과했다고 기존 binary의 SLO를 새로 측정한 것처럼 표시하지 않는다. CI run URL·job·event·head·artifact 이름을 PR에 연결하고 PR head 검사와 merge/push-main 검사를 구분한다. 이번 승인 범위는 PR 검사·정상 병합·별도 main 검사와 증거 보존/정리까지다. +qualification은 정확한 artifact·정책·환경 조합에 귀속한다. documentation-only head에서 계약 회귀가 통과했다고 기존 binary의 SLO를 새로 측정한 것처럼 표시하지 않는다. CI run URL·job·event·head·artifact 이름과 CI 계획을 PR에 연결하고 PR head 검사와 merge/push-main 검사를 구분한다. 이번 승인 범위는 PR 검사·정상 병합·별도 main 검사와 증거 보존/정리까지다. C10 기능 부모와 실제 자기 적용 receipt를 보존하고 C12 측정 artifact/정책/환경만 승격한다. 현재 대상은 8논리CPU/16GiB macOS다. foreground visibility/focus는 전체 측정 구간에서 검증하며 무효이면 inconclusive다. 정리 전 raw/report/manifest/log를 worktree 밖 보호 경로로 복사하고 hash를 확인한다. diff --git a/docs/planning/verification.md b/docs/planning/verification.md index 5273b8f..451e02a 100644 --- a/docs/planning/verification.md +++ b/docs/planning/verification.md @@ -19,7 +19,7 @@ Preserve the approved [design SLOs](../design.md#verification-and-promotion). De | V-LINUX | Actual Linux scopes/product | DGL-C05/C06 controllers, ancestors, privileges, descendants and SLO | Future; fake cgroups do not qualify | | V-CACHE / V-ADAPTER | Cache/tools/executors | DGC-C06; DGA-C02/C04/C06/C08 supported combinations | Future | -Keep `passed`, `failed`, `not_run`, `inconclusive` distinct. Preserve an existing tool's `incomplete` status and identify missing requirements. Zero discovered/executed cases cannot pass a suite. +Keep `passed`, `failed`, `not_run`, `inconclusive` distinct. A stage or suite that a CI plan leaves out is `not_selected_by_plan`, never `passed`. Preserve an existing tool's `incomplete` status and identify missing requirements. Zero discovered/executed cases cannot pass a suite. ## Available commands @@ -41,9 +41,12 @@ python3 scripts/qualify.py dg1-self-use --offline python3 scripts/qualify.py dg1-upgrade --offline python3 scripts/qualify.py dg1-macos --offline git diff --check +python3 scripts/validate.py --stages whitespace,source-contract,documentation --diff-base origin/main +python3 scripts/ci_plan.py --base origin/main --head HEAD +python3 -B -m unittest discover -s scripts -p 'test_ci_*.py' ``` -Remove offline only when locked dependencies must be downloaded. Report output must be a new ignored path inside the checkout. Put the installed pinned toolchain first in PATH. Toolchain mismatch results are supplemental/incomplete, not qualification. Bootstrap uses one Cargo job and one test thread. The existing validator leaves unmeasured runtime scopes `not_run`. +Remove offline only when locked dependencies must be downloaded. Report output must be a new ignored path inside the checkout. Put the installed pinned toolchain first in PATH. Toolchain mismatch results are supplemental/incomplete, not qualification. Bootstrap uses one Cargo job and one test thread. The existing validator leaves unmeasured runtime scopes `not_run`. `--stages` runs only the named validator stages, in their fixed order, and probes the Rust toolchain only when a Cargo stage is named; a run of only some stages does not claim DG-0. `--diff-base` limits the `whitespace` stage (`git diff --check`) to the committed change. `ci_plan.py --base` prints the plan CI would make for that diff. CodeSpace requires Node **24.21.0**, npm **11.19.0**, Python 3.11+ (CI 3.14): @@ -65,7 +68,26 @@ python3 scripts/validate-upstream.py macos-core dependencies python3 scripts/validate-upstream.py linux-isolation ``` -`all` does not include macos-core. Linux skips on macOS do not substitute for actual Linux evidence. Preserve `target/upstream-validation` and protected `target/upstream-reports/local` behavior. Existing CI checks still run for documentation PRs. +`all` does not include macos-core. Linux skips on macOS do not substitute for actual Linux evidence. Preserve `target/upstream-validation` and protected `target/upstream-reports/local` behavior. CodeSpace CI plans its Rust legs from the changed paths: a documentation-only change selects none, while its policy scan, format and pin checks, its result gate and the separate documentation-site workflow still run. Its scheduled and manual runs are full. + +## CI selection + +DevGuard CI runs the checks a change can affect, as `scripts/ci-policy.json` classifies its changed paths. `scripts/ci_plan.py` makes the plan, `scripts/ci_run.py` runs each job's share, and `scripts/check_ci_results.py` enforces it in the job **Required checks**. + +| Change | Repository checks, without Rust | Rust validation and functional suites | +| --- | --- | --- | +| Historical records (`docs/handoff/**`) | whitespace, documentation | not selected | +| Normative inputs: the design documents and their record, contracts, operations, planning, translations, the milestone ledger, `AGENTS.md`, `README.md`, `LICENSE`, `NOTICE` | whitespace and documentation; source-contract as well when the approved design, `docs/design-source.json` or `milestones.json` changes | not selected | +| A workspace crate | whitespace | the complete validator on macOS and Ubuntu, and each functional suite whose packages, or the binaries it builds first, depend on the crate | +| CI, planning, validation and qualification scripts, Cargo, toolchain and Git inputs, `.devguard.toml`, or a path in no class | every non-Rust stage | the complete validator and every suite | + +- The whitespace check is `git diff --check` over the change, or over the whole tree when there is no trusted base. The whole tree exempts only the files the policy pins by SHA-256. +- Pull requests and pushes to `main` are planned. A plan is also full when its base is missing or untrusted, its diff is empty, or it changes a planning input, so a pull request cannot narrow its own checks. Scheduled and manual runs are always full. Any other event, `merge_group` included, has no plan and fails the gate. +- The daily scheduled full run is a compensating control for this model: it keeps qualifying the whole repository on both platforms, every native suite included, while pull requests and `main` pushes run only affected checks. A documentation-only merge runs only the documentation checks on `main`. +- A Rust change still runs the complete validator, whose workspace clippy and tests keep their full scope; only the functional suites are selected. +- A stage or suite the plan leaves out is recorded `not_selected_by_plan`. The hosted-runner exceptions, suites run with `--allow-incomplete`, are listed in the policy with their exact cases and reasons; any other case recorded as `not_run` fails the run. +- The gate derives the plan again and accepts only evidence of the current run and attempt, bound to its commit, event and policy digest. Re-run every job, not only the failed ones. +- The policy's tests fail when a tracked path is in no class, a document pattern matches nothing or a pinned exemption's bytes change, so such a change is reviewed together with the policy. ## Document consistency @@ -77,7 +99,7 @@ For the preparation PRs, preserve runtime/Cargo/journal state. Documentation che ## Planned suites and fault injection -`scripts/qualify.py dg1-authority --offline` is available for C01 configuration/storage, `scripts/qualify.py dg1-auth --offline` for C02 authentication/transport, `scripts/qualify.py dg1-probes --offline` for C03 native evidence, `scripts/qualify.py dg1-scopes --offline` for C04 policy and scope evidence, `scripts/qualify.py dg1-launch --offline` for the C05 launch helper, `scripts/qualify.py dg1-reconcile --offline` for C06 reconciliation, `scripts/qualify.py dg1-cli --offline` for the C07 command-line owner, `scripts/qualify.py dg1-cargo --offline` for the C08 Cargo adapters, `scripts/qualify.py dg1-bootstrap --offline` for C09 installation, `scripts/qualify.py dg1-self-use --offline` for C10 parent leases and candidate authorities, `scripts/qualify.py dg1-upgrade --offline` for C11 upgrade and repair, and `scripts/qualify.py dg1-macos --offline` for the C12 SLO harness, whose protocol is `scripts/measure.py macos`. Other DevGuard suites and CodeSpace `scripts/qualify-devguard.py ` remain planned interfaces until supplied by their work units. Each implementation PR supplies the actual interface, nonzero case inventory, timeouts, logs, isolation and cleanup, then updates its task command documentation. macOS/Ubuntu CI retains the full validator and both portable functional suites, preserving their separate reports and logs. Native suites run on macOS CI only and record `not_run` elsewhere; they never pass on a platform that cannot supply the evidence. Each native stage declares the raw receipts it must produce. A receipt that records a case as `not_run` makes the suite `incomplete`, not `passed`. +`scripts/qualify.py dg1-authority --offline` is available for C01 configuration/storage, `scripts/qualify.py dg1-auth --offline` for C02 authentication/transport, `scripts/qualify.py dg1-probes --offline` for C03 native evidence, `scripts/qualify.py dg1-scopes --offline` for C04 policy and scope evidence, `scripts/qualify.py dg1-launch --offline` for the C05 launch helper, `scripts/qualify.py dg1-reconcile --offline` for C06 reconciliation, `scripts/qualify.py dg1-cli --offline` for the C07 command-line owner, `scripts/qualify.py dg1-cargo --offline` for the C08 Cargo adapters, `scripts/qualify.py dg1-bootstrap --offline` for C09 installation, `scripts/qualify.py dg1-self-use --offline` for C10 parent leases and candidate authorities, `scripts/qualify.py dg1-upgrade --offline` for C11 upgrade and repair, and `scripts/qualify.py dg1-macos --offline` for the C12 SLO harness, whose protocol is `scripts/measure.py macos`. Other DevGuard suites and CodeSpace `scripts/qualify-devguard.py ` remain planned interfaces until supplied by their work units. Each implementation PR supplies the actual interface, nonzero case inventory, timeouts, logs, isolation and cleanup, then updates its task command documentation. CI runs the validator and the functional suites its plan selects (see [CI selection](#ci-selection)), preserving their separate reports and logs. Native suites run on macOS CI only and record `not_run` elsewhere; they never pass on a platform that cannot supply the evidence. Each native stage declares the raw receipts it must produce. A receipt that records a case as `not_run` makes the suite `incomplete`, not `passed`. C03 evidence is recorded in these files: - **Raw receipts** (in the report's `raw/` directory): boot ID and clock readings with units, host capacity, repeated process identities, including zombie, reaped and refused observations, native pressure readings with their derived rates, the measured time from the last sample to closed admission, and the service loop's time from an injected failure to Critical and its behavior with a stuck probe. @@ -240,4 +262,4 @@ Manifest: source heads and dirty fingerprints, actual daemon/helper hashes, clie Retain raw latency/pressure/jobs, peak memory, completion time/throughput, refusal reasons, queue/buffer peaks, attempt/slot/lease transitions, fault points and termination/readback evidence. Hash reports and raw files, redact credentials and payloads, and preserve them outside disposable worktrees before cleanup. A documentation-only commit does not remeasure an old binary. -Record CI URL/job/event/head/artifact and distinguish PR checks from merge/push-main checks. At C10 preserve a functionally tested parent and real self-use receipts. At C12 promote only the artifact/policy/environment actually measured. Implementation status and platform qualification remain separate; Linux and CodeSpace runtime stay unqualified by DG-1. +Record CI URL/job/event/head/artifact and the CI plan, and distinguish PR checks from merge/push-main checks. At C10 preserve a functionally tested parent and real self-use receipts. At C12 promote only the artifact/policy/environment actually measured. Implementation status and platform qualification remain separate; Linux and CodeSpace runtime stay unqualified by DG-1. diff --git a/docs/translations.json b/docs/translations.json index 52f81fa..dbf447e 100644 --- a/docs/translations.json +++ b/docs/translations.json @@ -105,8 +105,8 @@ "id": "planning-verification", "source": "docs/planning/verification.md", "translation": "docs/ko/planning/verification.md", - "reviewed_source_sha256": "bfd7970650868b0ee6dcb23718283e8744327137a3ac08cb9d245bdeec0c35bd", - "reviewed_translation_sha256": "ebd5c8ac5f0e69aeac8b3f3a3d784b0f5e0a5bf9e362b36a9ad19022b0960427" + "reviewed_source_sha256": "c8afea075d25a82255ead448fb18873fa31668362eab628c96de3502f6420c73", + "reviewed_translation_sha256": "528a502e836684d24b2273747e75d9bdb82b2d4fd87a5f2151da8fa2bac97a6e" }, { "id": "operations", From 094b0b7b46c7ebe6f1a37ccdaaa08edb5f7c9ca0 Mon Sep 17 00:00:00 2001 From: Seongjae Date: Tue, 29 Sep 2026 00:16:03 +0900 Subject: [PATCH 6/6] ci: validate exact leg and allowance evidence Require changed-paths.json, both leg records and every current-run artifact name to match the current attempt's plan. Earlier attempts remain ignored and can never satisfy it; unexpected unsuffixed artifacts now fail instead of being mislabeled as earlier evidence. Validate incomplete suites stage by stage: each allowed not-run case must appear in its policy-named incomplete stage, every other stage must pass, and duplicate or unexplained cases fail. --- scripts/check_ci_results.py | 44 +++++++++++++++++++++++++++++-------- scripts/ci_run.py | 26 +++++++++++++++------- scripts/test_ci_results.py | 41 +++++++++++++++++++++++++++++++--- scripts/test_ci_run.py | 26 ++++++++++++++++++++++ 4 files changed, 117 insertions(+), 20 deletions(-) diff --git a/scripts/check_ci_results.py b/scripts/check_ci_results.py index 028c602..8b25bc3 100644 --- a/scripts/check_ci_results.py +++ b/scripts/check_ci_results.py @@ -40,6 +40,10 @@ def check_repository(plan, directory): problems = [] if read(directory / "binding.json") != ci_run.binding(plan, "repository"): problems.append("repository: its evidence is not bound to this run and attempt") + leg = read(directory / "leg.json") + if (not isinstance(leg, dict) or leg.get("schema") != ci_run.LEG_SCHEMA or leg.get("job") != "repository" + or leg.get("status") != "passed" or leg.get("exit_code") != 0): + problems.append("repository: its leg record does not show the repository runner passing") report = read(directory / "validate" / "report.json") if not isinstance(report, dict): return problems + ["repository: no validator report"], {} @@ -75,13 +79,22 @@ def check_contracts(policy, plan, platform, directory): problems.append(f"{label}: the complete validator did not pass at the planned source") planned = ci_run.planned_suites(policy, plan, platform) leg = read(directory / "leg.json") - if not isinstance(leg, dict) or [entry.get("suite") for entry in leg.get("suites") or []] != planned: - problems.append(f"{label}: the runner did not run exactly the planned suites") + entries = leg.get("suites") if isinstance(leg, dict) else None results = {} for suite in planned: results[suite], found = ci_run.evaluate_suite(policy, platform, suite, read(directory / suite / "report.json"), plan["source_sha"]) problems += [f"{label}: {suite}: {problem}" for problem in found] + if (not isinstance(leg, dict) or leg.get("schema") != ci_run.LEG_SCHEMA or leg.get("job") != "contracts" + or leg.get("platform") != platform or leg.get("status") != "passed" + or not isinstance(entries, list) or [entry.get("suite") for entry in entries] != planned): + problems.append(f"{label}: the runner did not record exactly the planned suites passing") + else: + for entry in entries: + suite = entry["suite"] + if (entry.get("status") != results[suite] or entry.get("exit_code") != 0 + or entry.get("problems") != []): + problems.append(f"{label}: the leg record does not show {suite} passing as {results[suite]}") extra = sorted(path.parent.name for path in directory.glob("dg1-*/report.json") if path.parent.name not in planned) problems += [f"{label}: {suite} ran although the plan did not select it" for suite in extra] return problems, report.get("status") or "no report", results @@ -102,7 +115,7 @@ def problems(results, plan_text, env, event, artifacts, root=ROOT): if not isinstance(plan, dict) or plan.get("run_attempt") != attempt: return [f"the plan is not from this attempt {attempt}; re-run all jobs"], [] try: - derived, _ = ci_plan.prepare(root, env, event) + derived, changed_paths = ci_plan.prepare(root, env, event) except (ci_plan.PlanError, OSError, ValueError, KeyError, subprocess.CalledProcessError) as error: return [f"the plan cannot be derived again: {error}"], [] if plan != derived: @@ -116,14 +129,28 @@ def problems(results, plan_text, env, event, artifacts, root=ROOT): if outcome[name] != expected: found.append(f"{name}: {outcome[name]}, expected {expected}") names = artifact_names(artifacts) - current = {name for name in names if re.fullmatch(r".+-[1-9][0-9]*", name) and name.rsplit("-", 1)[1] == attempt} + prefixes = {"ci-plan", "ci-repository", *{"dg0-" + platform for platform in policy["contracts"]["platforms"]}} + current, earlier, unexpected = set(), set(), set() + for name in names: + matches = [prefix for prefix in prefixes if re.fullmatch(re.escape(prefix) + r"-[1-9][0-9]*", name)] + if len(matches) != 1: + unexpected.add(name) + elif name.rsplit("-", 1)[1] == attempt: + current.add(name) + else: + earlier.add(name) wanted = {f"ci-plan-{attempt}", f"ci-repository-{attempt}"} if plan["rust"]: wanted |= {f"dg0-{platform}-{attempt}" for platform in policy["contracts"]["platforms"]} found += [f"{name}: no artifact from this attempt" for name in sorted(wanted - current)] found += [f"{name}: an artifact the plan does not produce" for name in sorted(current - wanted)] - if f"ci-plan-{attempt}" in current and read(artifacts / f"ci-plan-{attempt}" / "plan.json") != plan: - found.append("ci-plan: the uploaded plan is not the planned one") + found += [f"{name}: an artifact name not bound to an attempt" for name in sorted(unexpected)] + plan_artifact = artifacts / f"ci-plan-{attempt}" + if f"ci-plan-{attempt}" in current: + if read(plan_artifact / "plan.json") != plan: + found.append("ci-plan: the uploaded plan is not the planned one") + if read(plan_artifact / "changed-paths.json") != changed_paths: + found.append("ci-plan: the uploaded changed paths are not the derived diff") if f"ci-repository-{attempt}" in current: repository, statuses = check_repository(plan, artifacts / f"ci-repository-{attempt}") found += repository @@ -138,9 +165,8 @@ def problems(results, plan_text, env, event, artifacts, root=ROOT): table.append((f"contracts {platform} validator", validator)) table += [(f"contracts {platform} {suite}", suites.get(suite, "not_selected_by_plan")) for suite, spec in policy["suites"].items() if platform in spec["platforms"]] - ignored = sorted(set(names) - current) - if ignored: - table.append(("ignored: other attempts", ", ".join(ignored))) + if earlier: + table.append(("ignored: earlier attempts", ", ".join(sorted(earlier)))) return found, table diff --git a/scripts/ci_run.py b/scripts/ci_run.py index 79f6a27..37129c0 100644 --- a/scripts/ci_run.py +++ b/scripts/ci_run.py @@ -114,19 +114,29 @@ def evaluate_suite(policy, platform, suite, report, source): stages = report.get("stages") or [] if [stage.get("name") for stage in stages] != [name for name, *_ in qualify.SUITES[suite]]: problems.append("the report does not show every stage of the suite") - cases = [case for stage in stages for case in stage.get("not_run_cases") or []] status = report.get("status") + seen, case_count = set(), 0 + for stage in stages: + name = stage.get("name") + cases = stage.get("not_run_cases") or [] + case_count += len(cases) + for case in cases: + encoded = canonical(case) + if encoded in seen: + problems.append(f"the report repeats a not-run case: {encoded}") + seen.add(encoded) + allowance = allowed(policy, platform, suite, case) + if allowance is None or allowance["stage"] != name: + problems.append(f"not allowed to be not run in {name} on {platform}: {encoded}") + expected = "incomplete" if cases else "passed" + if stage.get("status") != expected: + problems.append(f"stage {name} is {stage.get('status')}, expected {expected}") if status == "passed": - if cases or any(stage.get("status") != "passed" for stage in stages): + if case_count: problems.append("a passed report records a case as not run") elif status == "incomplete": - if not cases: + if not case_count: problems.append("an incomplete report names no case") - for case in cases: - if allowed(policy, platform, suite, case) is None: - problems.append(f"not allowed to be not run on {platform}: {canonical(case)}") - if any(stage.get("status") not in ("passed", "incomplete") for stage in stages): - problems.append("a stage did not pass") else: problems.append(f"status {status}") if problems: diff --git a/scripts/test_ci_results.py b/scripts/test_ci_results.py index f959270..873df7d 100644 --- a/scripts/test_ci_results.py +++ b/scripts/test_ci_results.py @@ -91,7 +91,7 @@ def use(self, event_name, event, attempt="2"): self.event = event self.env = {"GITHUB_EVENT_NAME": event_name, "GITHUB_SHA": self.source, "GITHUB_RUN_ID": "77", "GITHUB_RUN_ATTEMPT": attempt, "GITHUB_REF": "refs/heads/main"} - self.plan, _ = ci_plan.prepare(self.root, self.env, event) + self.plan, self.changed_paths = ci_plan.prepare(self.root, self.env, event) self.results = {"plan": {"result": "success"}, "repository": {"result": "success"}, "contracts": {"result": "success" if self.plan["rust"] else "skipped"}} self.upload(attempt) @@ -99,8 +99,11 @@ def use(self, event_name, event, attempt="2"): def upload(self, attempt, plan=None): plan = plan or self.plan write(self.artifacts / f"ci-plan-{attempt}" / "plan.json", plan) + write(self.artifacts / f"ci-plan-{attempt}" / "changed-paths.json", self.changed_paths) repository = self.artifacts / f"ci-repository-{attempt}" write(repository / "binding.json", ci_run.binding(plan, "repository")) + write(repository / "leg.json", {"schema": ci_run.LEG_SCHEMA, "job": "repository", + "status": "passed", "exit_code": 0}) write(repository / "validate" / "report.json", validator(self.source, plan["repository_stages"], plan["whitespace"])) if not plan["rust"]: @@ -110,14 +113,18 @@ def upload(self, attempt, plan=None): write(directory / "binding.json", ci_run.binding(plan, "contracts", platform)) write(directory / "ci" / "report.json", validator(self.source, validate.DEFAULT_STAGES)) suites = ci_run.planned_suites(POLICY, plan, platform) - write(directory / "leg.json", {"suites": [{"suite": suite} for suite in suites]}) + entries = [] for suite in suites: cases = [] if suite == "dg1-cargo": cases = [{"file": f"raw/native-cargo/{name}.json", "path": "", "reason": ALLOWED_CARGO["reason"]} for name in ALLOWED_CARGO["cases"]] + status = "incomplete-allowed" if cases else "passed" + entries.append({"suite": suite, "status": status, "exit_code": 0, "problems": []}) write(directory / suite / "report.json", suite_report(suite, self.source, "incomplete" if cases else "passed", cases)) + write(directory / "leg.json", {"schema": ci_run.LEG_SCHEMA, "job": "contracts", + "platform": platform, "status": "passed", "suites": entries}) def problems(self, plan=None, results=None, env=None, event=None): plan_text = json.dumps(plan if plan is not None else self.plan) @@ -202,7 +209,7 @@ def test_evidence_from_an_earlier_attempt_is_ignored(self): self.upload("2") found, table = gate.problems(self.results, json.dumps(self.plan), self.env, self.event, self.artifacts, self.root) self.assertEqual(found, []) - self.assertEqual(table[-1][0], "ignored: other attempts") + self.assertEqual(table[-1][0], "ignored: earlier attempts") def test_a_current_name_with_an_earlier_binding_fails(self): self.pull_request("crates/cargo/src/lib.rs") @@ -223,6 +230,34 @@ def test_the_uploaded_plan_must_be_the_planned_one(self): write(self.artifacts / "ci-plan-2" / "plan.json", dict(self.plan, profile="full")) self.assertEqual(self.problems(), ["ci-plan: the uploaded plan is not the planned one"]) + def test_the_uploaded_changed_paths_must_be_the_derived_diff(self): + self.pull_request("docs/handoff/new.md") + path = self.artifacts / "ci-plan-2" / "changed-paths.json" + path.unlink() + self.assertEqual(self.problems(), ["ci-plan: the uploaded changed paths are not the derived diff"]) + write(path, ["another/path"]) + self.assertEqual(self.problems(), ["ci-plan: the uploaded changed paths are not the derived diff"]) + + def test_an_artifact_without_a_recognized_attempt_binding_fails(self): + self.pull_request("docs/handoff/new.md") + write(self.artifacts / "diagnostic" / "record.json", {}) + self.assertEqual(self.problems(), ["diagnostic: an artifact name not bound to an attempt"]) + shutil.rmtree(self.artifacts / "diagnostic") + write(self.artifacts / "ci-plan-latest" / "record.json", {}) + self.assertEqual(self.problems(), ["ci-plan-latest: an artifact name not bound to an attempt"]) + + def test_each_leg_record_must_show_its_runner_passing(self): + self.pull_request("crates/cargo/src/lib.rs") + repository = self.artifacts / "ci-repository-2" / "leg.json" + repository.unlink() + self.assertIn("repository: its leg record does not show the repository runner passing", self.problems()) + self.upload("2") + contracts = self.artifacts / f"dg0-{MACOS}-2" / "leg.json" + leg = json.loads(contracts.read_text()) + leg["suites"][0]["exit_code"] = 1 + write(contracts, leg) + self.assertTrue(any("leg record does not show" in problem for problem in self.problems())) + def test_repository_stages_must_be_exactly_the_planned_ones(self): self.pull_request("docs/handoff/new.md") path = self.artifacts / "ci-repository-2" / "validate" / "report.json" diff --git a/scripts/test_ci_run.py b/scripts/test_ci_run.py index de957b9..a656517 100644 --- a/scripts/test_ci_run.py +++ b/scripts/test_ci_run.py @@ -246,6 +246,32 @@ def test_allowances_are_per_platform(self): value["stages"][-1]["not_run_cases"][0]["path"] = "" self.assertEqual(self.check(value, "dg1-scopes")[0], "failed") + def test_an_allowed_case_must_be_in_its_own_incomplete_stage(self): + case = cargo_case("direct-build") + value = report("dg1-cargo", self.source, "incomplete", [case]) + native = next(stage for stage in value["stages"] if stage["name"] == "native-cargo") + other = next(stage for stage in value["stages"] if stage["name"] == "cargo-plan") + native["not_run_cases"] = [] + native["status"] = "passed" + other["not_run_cases"] = [case] + other["status"] = "incomplete" + status, problems = self.check(value, "dg1-cargo") + self.assertEqual(status, "failed") + self.assertTrue(any("not allowed to be not run in cargo-plan" in problem for problem in problems)) + + def test_each_incomplete_stage_needs_a_case_and_cases_cannot_repeat(self): + unexplained = report("dg1-cargo", self.source, "incomplete", []) + unexplained["stages"][0]["status"] = "incomplete" + status, problems = self.check(unexplained, "dg1-cargo") + self.assertEqual(status, "failed") + self.assertTrue(any("expected passed" in problem for problem in problems)) + duplicated = report("dg1-cargo", self.source, "incomplete", [cargo_case("direct-build")]) + native = next(stage for stage in duplicated["stages"] if stage["name"] == "native-cargo") + native["not_run_cases"].append(dict(native["not_run_cases"][0])) + status, problems = self.check(duplicated, "dg1-cargo") + self.assertEqual(status, "failed") + self.assertTrue(any("repeats a not-run case" in problem for problem in problems)) + class Summary(Job): def test_the_summary_says_what_ran_and_what_the_plan_left_out(self):