diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 3aaf3515..a05deaae 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -25,6 +25,12 @@ jobs: - name: Install pinned toolchain run: rustup show + - name: Install Linux syscall observation tool + run: | + sudo apt-get update + sudo apt-get install --yes strace + strace --version + - name: Install independent vector tool uses: taiki-e/install-action@41049aa56687c35e0afa74eed4f09cec4f9afabf # v2.85.2 with: diff --git a/CHANGELOG.md b/CHANGELOG.md index 0f056a26..c23e7c27 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,8 +8,124 @@ after its public API and format compatibility policies are established. ## [Unreleased] +- Retention recovery execution errors report the exact failed boundary, original typed cause, known namespace effects and uncertain effect/durability; retries freshly observe the store. Observed stage identity remains binding across reopening, and cleanup preserves verified pool evidence rather than promising the removed pathname survives (#99). + +- Retention recovery now preserves incomplete stages and requires explicit disposition before any recovery mutation or publication retry; automatic incomplete-stage disposal is deferred by maintainer decision (#99). + +- Complete-stage cleanup now verifies the retained source as well as the pool target before unlink, detecting source substitution while retaining the opened evidence (#99). + +- Earlier retention prefix-validation entries describe historical diagnostic fixes. They do not claim that the current landing automatically discards incomplete stages or proves every prefix has a canonical completion (#99). +- Recovery preserves interrupted initial roots, manifests and heads as corrupt evidence as soon as any available predecessor byte is nonzero, while retaining possible successor prefixes (#99). + +- Recovery rejects partial anchors when even their greatest canonical completion cannot follow the preceding anchor, preserving the exact offending stage (#99). + +- Recovery preserves partial layout-length fields whose minimum possible completion exceeds the format ceiling, with an explicit prefix-bound diagnostic (#99). + +- Recovery preserves partial root anchors with contradictory embedded identity bytes or invalid complete layout-length fields, sharing the identity codecs' rules (#99). + +- Recovery preserves partial manifest entries whose namespace prefix cannot satisfy strict ordering or whose completed root-generation field is zero (#99). + +- Recovery validates complete anchors and manifest entries inside interrupted bodies, preserving malformed identities, zero generations and duplicate ordering violations before discard (#99). + +- Recovery verifies available root and manifest digest and checksum bytes and complete body-set digests before discarding interrupted stages, preserving provably corrupt records (#99). + +- Recovery admits each closure-limit field as bytes arrive, preserving impossible partial bounds and invalid complete fields with precise typed diagnostics (#99). + +- Recovery rejects every available contradictory byte of an interrupted registered-profile identity, version or digest and preserves the root stage (#99). + +- Recovery preserves interrupted roots with invalid complete realization-profile or closure-policy groups, returning the exact domain refusal before stage discard (#99). + +- Recovery preserves interrupted roots and manifests whose complete generation/predecessor fields contradict initial or successor history (#99). + +- Recovery preserves interrupted roots declaring empty or oversized namespaces, applying the domain length rule before payload arrival (#99). + +- Recovery preserves interrupted root and manifest stages that declare counts above the format ceiling, even when their declared lengths are self-consistent (#99). + +- Recovery rejects and preserves short root and manifest headers whose complete size fields contradict their declared record length (#99). + +- Recovery verifies available checksum bytes in interrupted head stages and preserves contradictory prefixes as corrupt evidence instead of discarding them (#99). + +- Recovery preserves short heads whose complete generation/predecessor fields contradict initial or successor history, sharing semantic admission with complete head decoding (#99). + +- Recovery preserves interrupted head stages whose complete manifest-length field violates canonical bounds or alignment, returning the precise corruption cause instead of discarding them (#99). + +- Clarify that retention recovery and fenced readers are implemented but still under correctness remediation and independent review; link the completed process-death reader evidence and name remaining recovery gaps (#99). + +- Filesystem recovery tests now require the exact checksum-corrupt root-stage refusal and unchanged retained evidence, with calibrated diagnostic and deletion checks (#99). + +- Recovery now reports `ManifestStageWithoutRootStage` when complete head and manifest stages lack their root stage, instead of incorrectly reporting a missing manifest; refusal preserves retained evidence (#99). + +- The retention process-death matrix now independently reads the recovered head generation and exact selected-root bytes before forward retry, including absence before publication (#99). + +- Successor recovery evidence now covers every ordered publication prefix and verifies the recovered generation and exact selected-root bytes through the fenced reader; the ledger distinguishes this from sampled mid-write truncations (#99). + +- Publication now identifies recovery-observation failures through `RecoveryObservationRefused`, preserving the original typed or OS cause in its error chain; explicit recovery and later planning/execution boundaries retain their existing behavior (#99). + +- Publication tests now require the exact missing-manifest recovery refusal for a retained complete head, rejecting unrelated recovery errors (#99). + +- The fenced reader corruption law now requires the precise checksum-mismatch cause and expected/observed checksum bytes instead of accepting any root error (#99). + +- Correct the retention recovery documentation's obsolete integration status and explicitly retain its open incomplete-stage pinning and post-removal failure findings (#99). + +- Recovery fixtures now use production's head-to-manifest binding validation, preventing independently valid but contradictory records from becoming observed retention state (#99). + +- Repository-task migration admission now reports root capability-clone failures as namespace failures and identity-probe failures as root-identity failures, preserving their I/O causes (#99). + +- Reader collection tests now reject same-generation digest changes and require preserved I/O causes at all three read boundaries, using returned payloads instead of script load counters (#99). + +- Retention model tests now require exact preparation and superseded-publication refusals, including their identifying coordinates; unrelated errors no longer satisfy a rejected transition, and the harness sequence-count assertion is removed (#99). + +- Derive the recovery manifest read bound from the codec's checked framing calculation and semantic entry limit, removing a separately maintained size literal (#99). + +- Keep the reader fence private to the crate; callers retain snapshots through `FilesystemRetentionSnapshot`, which owns the fence for its lifetime (#99). + +- Retention readers load catalog bytes through the originally pinned store directory, so replacement of its ambient path cannot redirect catalog collection into another directory. + ### Added +- Model-based retention evidence: every three-operation sequence over initial + publications of two namespaces, a successor, a byte-identical retry, and a + stale initial (125 sequences, each in a fresh migrated store) agrees with a + deterministic namespace-to-(generation, anchor-set) map and liveness after + every step, observed through the fenced reader view; a source contract + keeps clocks, paths, environment, and identity out of the retention core. +- `FilesystemRetentionSnapshot` is the version-two reader view: it admits the + root as version two, acquires a shared `ReaderFence` on `reader.lock`, + double-collects the catalog and retention heads around loading through + `collect_retention_view` (bounded by `ReaderAttemptLimit`, refusing an + exhausted limit or an absent catalog), binds the catalog snapshot, the + retention head, and its manifest, and verifies each selected root against + the manifest on demand while the fence is held. +- Storage-independent retention recovery planning: `assess_root_stage`, + `assess_manifest_stage`, and `assess_head_stage` classify each fixed stage + as absent, complete, truncated, or corrupt through the decoders' own + truncation laws; `plan_retention_recovery` turns that evidence, the observed + current state, and pool-entry observations into an ordered + `RetentionRecoveryPlan` (discard a pre-effect truncated stage, link and + protect complete orphans, finalize a complete head over linked stages, clean + up stages the published head already names) or a typed + `RetentionRecoveryRefusal`. `RetentionRecoveryStorage` names one blocking + capability per step and `execute_retention_recovery` runs a plan in order, + stopping at the first refused step with the completed prefix named in + `RetentionRecoveryError`. `FilesystemRetentionPublicationAuthority::recover` + observes the stages within their format bounds, reopens complete stages + bound to their identity, and executes the plan under the retained writer + lock, so a crash after the head stage is synchronized finalizes on restart + and a byte-identical retry is already committed. Laws drive every + publication prefix from 0 through 18 phases, truncate each stage mid-write, + and replay successor prefixes over a published generation; each recovers to + its documented state, recovery is idempotent, and the forward retry reports + the predicted outcome. Publication runs that recovery as its first step, so + an interrupted publication no longer waits for a human unless it left a + complete orphan; `RecoveryRefused` and `RecoveryStepRefused` carry + recovery's own errors through `RetentionCurrentStateRefusal`. The crash + matrix gains `KEEP-CRASH-036` through `052`: a child migrates a golden + bundle store, publishes retention generation one, and is killed before, + during, or after each of the seventeen phases; restart reopens the store, + runs recovery, and requires the documented steps, outcome, and forward + retry. `FilesystemVersionTwoAdmission::reopen_unchecked_for_repository_tasks` + and `FilesystemStoreMigrationAuthority::open_unchecked_for_repository_tasks` + give repository tools the same bypass version one already had. - Recovery receipts retain the exact observed namespace prefix and bound migration intent digest, including observations of completed migration. @@ -624,6 +740,38 @@ after its public API and format compatibility policies are established. ### Fixed +- Retention recovery capabilities and execution errors preserve typed exact-record refusals and original I/O causes through `RetentionStorageError`. Callers can inspect `RetentionRecoveryError::storage_error()` directly; shared forward-stage adapters retain the typed source when adapting to their existing I/O boundary (#99). + +- Reader-fence tests now replace the lock inode during acquisition, distinguish that case from nonempty lock bytes, and require the exact kernel contention errno; isolated mutations verify that the identity and exclusion assertions detect removed protections (#99). + +- Reader double collection preserves catalog length and the complete retention head, so changes to validated length or predecessor coordinates cannot be collapsed into an unchanged generation/digest pair. The public `RetentionViewCoordinates` catalog tuple now includes `CatalogLength`, and its retention field carries `RetentionHead` (#99). + +- Fenced retention readers preserve selected catalog and segment admission failures as `FilesystemRetentionSnapshotError::Catalog` with their exact restart source, while coordinate failures remain `View` errors; a moving head retries the speculative catalog result (#99). + +- Fenced retention readers compare the opened root's restart-stable device and inode with the jointly admitted migration records before acquiring a fence or loading a catalog. A foreign binding refuses at reader admission with the exact typed identity coordinate, expected value, and observed value (#99). + +- Retention recovery reauthenticates the current catalog and selected segments and replays staged-root closure laws before publication effects. Forward publication shares live verification after pure preflight and preserves prepared catalog binding. Recovery exposes an explicit catalog loading policy; the default aggregate retained-segment cap is one protocol-maximum segment (1 GiB), with larger selections refusing before mutation (#99). + +- Retention recovery refuses truncated manifest or head stages whose earlier complete stage or immutable pool link is missing, preserving impossible-prefix evidence instead of deleting it (#99). + +- Retention recovery requires an uncommitted head's staged root to satisfy the same generation and predecessor rules as root-only and manifest-only recovery. Already committed cleanup remains admissible (#99). + +- Retention recovery reopens and authenticates the manifest-selected predecessor before publication effects, refusing missing, corrupt, or substituted roots while preserving retained evidence. + +- Retention recovery refuses successor manifests that add, drop, or alter namespace entries unrelated to the staged root before linking the manifest or finalizing the head. + +- Retention recovery binds a complete staged head to its manifest's exact byte length and predecessor before committing or cleaning up. Canonical but inconsistent heads refuse with existing typed planning errors and preserve published and retained evidence (#99). + +- Retention recovery refuses byte-identical root or manifest pool substitutions before committing the head or removing stages. Pool observation binds exact bytes to the retained stage's device and inode and preserves the typed pool refusal (#99). + +- Retention recovery admits namespace names and pool kinds before stage observation or mutation. Direct and publication-triggered recovery preserve retained evidence when unknown entries or noncanonical pool names refuse (#99). + +- Recovery synchronizes the exact complete retention stage before creating its immutable root or manifest pool link or replacing the retention head. A synchronization error refuses before that publication; the retained on-disk stage remains recovery evidence. + +- Interrupted retention stages with complete zero root or liveness generations now refuse recovery with the existing typed generation error before any mutation (#99). Incomplete generation fields remain eligible for canonical-prefix recovery; complete-record decoding and durable encodings are unchanged. + +- Retention recovery refuses an interrupted root, manifest, or head stage when any available magic, version, fixed width, flag, or reserved byte contradicts the canonical format (#99). The typed `PrefixByteMismatch` names the actual byte and offset without inventing missing bytes; refusal preserves retained evidence. Canonical interrupted prefixes remain recoverable. Complete-record decoding and on-disk bytes are unchanged; the decode-error enums gain a diagnostic variant. + - Subprocess test-support readiness admits a queued valid signal after sender exit, preventing a false process-group cleanup failure without retrying the test gate. - The version-one catalog ledger names its executable ordering integration @@ -717,6 +865,11 @@ after its public API and format compatibility policies are established. Review corrections to the unreleased retention and migration work above; none of these shipped in a release. +- Retention and migration crash tasks share the migration repository-task + constructor after main integration instead of defining it twice (#19). +- Explicit retention recovery verifies the pinned retention and immutable-pool + directory identities before any recovery effect, preserving retained stage + evidence when a protocol directory was replaced (#19, PR #99). - Retention publication reopens the catalog pool entry this store's `HEAD` selects, bounded by the head's declared length, and requires it to decode to that generation and digest (`CatalogAbsent`, `CatalogRefused`, diff --git a/README.md b/README.md index 6c62ecce..5a470d1f 100644 --- a/README.md +++ b/README.md @@ -51,10 +51,10 @@ Keep is required to refuse all three, before mutating anything. generation-versioned catalogs, and a fixed-width `HEAD` are published through an ordered protocol whose every step is a named crash point. Platform admission is Linux ext4, non-casefolded, one writer. -- **Proven restart recovery for version 1.** The crash matrix kills real - writer processes at 105 before/during/after coordinates - (`KEEP-CRASH-001`–`035`) and verifies the store lands in exactly one - documented lawful state each time. +- **Process-death recovery evidence.** The crash matrix kills real writer processes + at 156 before/during/after coordinates (`KEEP-CRASH-001`–`052`) and + verifies the store lands in exactly one documented lawful state each time, + for version-1 publication and for version-2 retention publication. - **Version-2 retention and migration, forward path.** Explicit retention roots, deterministic closure verification, a one-way 21-phase migration, and a 17-phase retention publication — all with production filesystem @@ -77,18 +77,22 @@ Keep is required to refuse all three, before mutating anything. ## What it does not do yet -Version-2 retention publication writes from a clean start and refuses the -residue of an interrupted retention publication. Retention publication recovery -and the reader snapshot fence await integration from PR #99, so **an -interrupted retention publication waits for explicit recovery.** Migration -restart recovery is implemented and does not grant retention authority. A -version-1 store stays admitted until its owner migrates it; migrate only if -you accept that wait. +Version-2 retention recovery, fenced reader snapshots and model-based transitions are implemented in this branch but still require the correctness corrections and independent acceptance review tracked in PR #99. + +The retention process-death sequence checks the recovered head generation and exact selected-root bytes before retry; its [evidence receipt](docs/testing-evidence/retention-crash-reader-oracle.md) bounds that claim to the declared initial-publication crash coordinates. + +Incomplete retention stages are preserved and block publication pending explicit disposition; automatic disposal is deferred. The [landing ledger](docs/testing-evidence/retention-landing.md) tracks execution-failure reporting and final acceptance under the [approved recovery contract](docs/formats/segment-store-v2/retention-recovery.md). + +Complete orphans remain recovery-protected until explicit disposition lands with garbage collection (#21). + +Migration restart recovery is implemented and does not grant retention authority. + +A version-1 store stays admitted until its owner migrates it. | Gap | Tracked | | --- | --- | -| Retention publication restart recovery and broader migration corruption coverage | [PR #99](https://github.com/flyingrobots/keep/pull/99), [#111](https://github.com/flyingrobots/keep/issues/111) | -| Reader fence binding one consistent catalog + retention snapshot | [PR #99](https://github.com/flyingrobots/keep/pull/99) | +| Retention recovery correctness remediation and broader migration corruption coverage | [PR #99](https://github.com/flyingrobots/keep/pull/99), [#111](https://github.com/flyingrobots/keep/issues/111) | +| Fenced reader correctness remediation and independent acceptance | [PR #99](https://github.com/flyingrobots/keep/pull/99) | | Precise verification reports at explicit depths | [#20](https://github.com/flyingrobots/keep/issues/20) | | Garbage collection and identity-preserving compaction | [#21](https://github.com/flyingrobots/keep/issues/21) | | Bounded production ingestion through the durable store | [#82](https://github.com/flyingrobots/keep/issues/82) | @@ -194,6 +198,22 @@ cargo xtask golden-file-worldline-check cargo xtask conformance-check ``` +Select a crash campaign with `--sequence NAME` using these exact CLI names: + +| Name | Campaign | +| --- | --- | +| `segment` | Segment publication | +| `catalog` | Catalog publication | +| `head` | Publication-head replacement | +| `recovery-discard` | Explicit recovery evidence discard | +| `initialization` | Writer-locked initialization | +| `retention` | Retention root, manifest, and head publication | +| `migration` | Version-one to version-two migration | + +```bash +cargo xtask durability-crash-matrix --sequence retention +``` + ## Design boundary Keep owns physical content storage: exact byte identity; chunking and diff --git a/docs/adr/reader-fence-visibility.md b/docs/adr/reader-fence-visibility.md new file mode 100644 index 00000000..e9bda2ec --- /dev/null +++ b/docs/adr/reader-fence-visibility.md @@ -0,0 +1,15 @@ +# Reader fence visibility + +Status: accepted. + +Change kind: deliberate API surface reduction, with no runtime behavior change. + +`FilesystemRetentionSnapshot` owns the reader fence for the snapshot's lifetime; no public operation accepts, returns, or constructs the fence itself. + +The previously exported `ReaderFence` exposed an unusable implementation detail and invited an unnecessary compatibility obligation, so its declaration is now visible only within retention, its parent import is private, and its crate-root export is removed. + +This changes the unreleased source API without changing locking, snapshot lifetime, durable encoding, or recovery behavior. + +Static before/after evidence is the public declaration and two export sites at parent `429e3f7`; compiler validation and the existing reader-fence and snapshot runtime laws check that internal consumers continue to work. + +No new runtime assertion is added for visibility: source/API evidence establishes this change, while the existing behavioral laws retain their independent oracles and calibration evidence. diff --git a/docs/adr/retention-bounded-recovery-landing.md b/docs/adr/retention-bounded-recovery-landing.md new file mode 100644 index 00000000..04702d7d --- /dev/null +++ b/docs/adr/retention-bounded-recovery-landing.md @@ -0,0 +1,15 @@ +# Bounded retention recovery landing + +Status: accepted maintainer decision for #99. This decision supersedes earlier retention automatic-discard requirements and claims, without changing durable record formats or the migration recovery protocol. + +Incomplete retained stages are preserved. Planning refuses before recovery effects with a precise demonstrated-corruption cause or a typed disposition-required result. Available checks finding no contradiction do not prove that a complete canonical record exists. The accepted availability cost is blocked publication until explicit disposition is designed; users must not be instructed to delete evidence blindly. + +Automatic incomplete-stage disposition and stronger completion feasibility belong to a focused follow-up. Adding a quarantine namespace, journal or durable format to this landing is rejected. Existing useful partial validation remains diagnostic evidence, not disposal authorization. + +The supported namespace mutation model consists of cooperating writers under Keep authority. No writer-lock isolation is promised against arbitrary concurrent raw mutation. Keep retains no-follow, identity, exact-byte, namespace and corruption checks, and rejects observed substitution. Retaining a handle and checking metadata do not make pathname unlink or rename identity-conditional. + +Pre-effect refusal initiates no recovery mutation. Execution failure is not rollback: completed earlier steps and the failing capability's known or uncertain effects must be distinguished, with original typed causes and durability uncertainty retained. No later recovery steps execute after failure; a new attempt must observe again. An empty completed-step list says nothing about whether the failing capability changed the namespace. + +Complete-stage cleanup verifies its source and surviving pool evidence while retaining the opened stage. Its guarantee is preservation of verified surviving pool evidence under the supported mutation model, not preservation of a stage pathname that cleanup intentionally removes. Directory synchronization failure cannot undo an earlier unlink, link or rename. + +The finite review scope is the existing recovery capabilities, shared stage operations, executor and error boundaries. The closure ledger and stable-candidate validation live in [retention-landing.md](../testing-evidence/retention-landing.md). Independent review assesses this approved contract; unrelated improvements are follow-ups. Implementation stops when the ledger closes, exact-head independent approval is recorded and required checks are green; human merge approval remains required. diff --git a/docs/adr/retention-live-closure-admission.md b/docs/adr/retention-live-closure-admission.md new file mode 100644 index 00000000..3362025e --- /dev/null +++ b/docs/adr/retention-live-closure-admission.md @@ -0,0 +1,15 @@ +# Retention live closure admission + +Status: Accepted + +Pure preflight proves reconstructability against its supplied snapshot, which may predate corruption or belong to another store. Canonical retained root bytes and matching publication records cannot establish that live segment bytes still reconstruct their anchors. Recovery and forward authority therefore share capability-relative catalog loading and the existing pure closure verifier before mutation. + +The loader receives the retained root directory capability directly rather than reopening an ambient pathname. It authenticates the exact current head, catalog, and selected segments. Forward publication additionally compares the resulting proof's catalog coordinates with its prepared proof, so a fresh but different catalog cannot launder a stale preparation. Recovery has no persisted original catalog coordinate in the root format and instead verifies the anchors against the current authenticated catalog. No format, root digest, or closure transcript encoding changes. + +The existing snapshot loader materializes all selected segments. Its aggregate retained-segment policy is independent of the root's closure physical-byte counter, which counts selected records rather than complete segment allocation and cannot bound unrelated catalog members. The explicit recovery policy admits caller-selected bounds; the default and forward authority use the protocol maximum segment length as a conservative aggregate ceiling, 1 GiB. Selection above that default refuses before mutation and may be admitted for recovery with an explicit larger policy. Catalog encoding, segment record counts, and decoded indexes retain their separate existing bounds. This is not a total RSS limit or a measured optimal budget. + +Reusing the authenticated snapshot and pure closure verifier preserves one semantic oracle. A new on-demand record loader could reduce memory and unrelated reads but would need independent admission and parity evidence; implementing an unchecked partial catalog or trusting old preflight would weaken the contract. This change prioritizes verification and explicit bounded refusal. It adds segment I/O and reconstruction work to forward verification and complete-stage recovery; no throughput or latency improvement is claimed. Clean or incomplete-root recovery avoids this loading work. + +Catalog restart and closure failures remain concrete error sources in the observation or current-verification boundary. Invalid staged framing, history, conflicting pool identity, and selected predecessor admission retain their existing earlier refusal points. The new proof is required for committed-stage cleanup as well as new publication; cleaning up after losing reconstructability would erase useful evidence. + +Runtime regressions cover missing and corrupted catalog or segment artifacts at root-link, manifest-link, and head-finalization boundaries, absent anchor membership, a physical closure-limit violation, and corruption after forward preflight. Explicit loading-policy boundary tests cover a budget below and exactly equal to the selected segment bytes. These experiments do not establish physical power-loss durability, every catalog size or profile, atomic unlink under concurrent replacement, or per-test resource ceilings. diff --git a/docs/adr/retention-reader-catalog-errors.md b/docs/adr/retention-reader-catalog-errors.md new file mode 100644 index 00000000..09e03e17 --- /dev/null +++ b/docs/adr/retention-reader-catalog-errors.md @@ -0,0 +1,19 @@ +# Retention reader catalog refusal boundary + +Status: accepted. + +The public fenced reader must distinguish catalog restart admission failures from failures to collect consistent head coordinates. + +Previously the filesystem source wrapped `CatalogRestartError` in `io::Error`, the collector wrapped it again, and the reader returned `FilesystemRetentionSnapshotError::View`; the documented `Catalog` variant was unreachable. + +The filesystem source now carries the catalog admission result as its collected view value, retaining the concrete restart error without converting it into I/O. + +After consistent collection, the public loader maps that admission error directly into `FilesystemRetentionSnapshotError::Catalog`, while coordinate and retention I/O errors remain `View` failures. + +A catalog refusal observed between moving heads is discarded with that attempt, just like a successful speculative view; the existing bounded collector retries, and only a stable pair selects the admission result. + +This uses the existing view-value abstraction instead of adding a breaking public error parameter to the port or storing a separate mutable error sentinel on the filesystem source. + +No format, identity preimage, writer protocol, or public signature changes; catalog refusal paths perform the collector's second coordinate read before reporting an admission error, and successful paths retain their existing allocation and I/O behavior. + +The public error regressions cover absent selected catalogs, absent selected segments, corrupt selected catalog bytes, and a corrupt coordinate HEAD; a controlled real-filesystem schedule additionally restores a missing catalog and publishes retention between the load and second coordinate read. diff --git a/docs/adr/retention-reader-catalog-pinning.md b/docs/adr/retention-reader-catalog-pinning.md new file mode 100644 index 00000000..2b2cf85c --- /dev/null +++ b/docs/adr/retention-reader-catalog-pinning.md @@ -0,0 +1,15 @@ +# Retention reader catalog pinning + +Status: accepted. + +The retention reader must load catalog bytes from the same opened store directory that supplies its collected publication coordinates. + +Reopening the ambient path during collection can select a replacement directory while coordinate reads continue through the original directory capability, producing a refusal or a mixed view unrelated to the admitted root. + +The filesystem reader therefore uses the existing directory-based catalog restart loader with its pinned root capability and unchanged restart policy. + +This changes neither the durable format nor the public API and adds no catalog allocation or retry beyond the existing restart loader. + +The regression drives the production collection port with a real migrated store, renames that store after pinning, places an empty replacement at the ambient path, and requires the original catalog generation and digest. + +Full public-loader scheduling during admission, head-coordinate completeness, and catalog error classification remain separate obligations. diff --git a/docs/adr/retention-reader-complete-coordinates.md b/docs/adr/retention-reader-complete-coordinates.md new file mode 100644 index 00000000..e4b3fbd3 --- /dev/null +++ b/docs/adr/retention-reader-complete-coordinates.md @@ -0,0 +1,17 @@ +# Complete retention reader coordinates + +Status: accepted. + +Reader double collection must compare every validated semantic field of both selected heads, including catalog length and retention manifest length and predecessor. + +The filesystem reader previously projected both heads to generation and digest, so a correctly checksummed catalog head with a changed admitted length could appear unchanged and select a view that the later head would not admit. + +Catalog coordinates now retain validated generation, length, and digest; retention coordinates retain the existing validated `RetentionHead` value, whose equality includes generation, manifest length, digest, and predecessor. + +Invalid head checksums and other decoding failures refuse through the existing source-error boundary; they are not retried as successfully decoded coordinate changes. + +This intentionally changes the unreleased public `RetentionViewCoordinates` field types, requiring port implementations to supply the missing validated coordinates; it does not change durable bytes, identity preimages, retry limits, or write/recovery protocols. + +Retaining exact encoded head bytes was rejected because the storage-independent collection port exchanges validated semantic values rather than codecs or filesystem representations. + +Full coordinate equality still cannot observe an arbitrary external replacement and restoration entirely between its reads, or establish that an arbitrary port's loaded value actually belongs to those coordinates; these are separate limits of the existing collection contract. diff --git a/docs/adr/retention-reader-root-identity.md b/docs/adr/retention-reader-root-identity.md new file mode 100644 index 00000000..d3512c51 --- /dev/null +++ b/docs/adr/retention-reader-root-identity.md @@ -0,0 +1,9 @@ +# Retention reader root identity + +Status: Accepted + +Joint admission of `FORMAT`, `migration.intent`, and `migration.receipt` establishes that those records agree with each other. It does not establish that the directory receiving the read request is the directory the migration intent names. Reader admission must compare the opened root's physical identity before accepting the version-two read capability. + +The reader uses the same root identity probe and `require_root_identity` predicate as writer admission. Restart-stable device and inode coordinates must match; mount-instance identity is not compared, preserving the established remount rule. A mismatch remains a concrete `FilesystemPlatformAdmissionError::RootIdentityChanged` source inside the reader admission error, with the coordinate and both values intact. No record, identity preimage, or format bytes change. + +The regression transplants a complete mutually consistent migration-record set from one controlled migrated directory into another on the same filesystem. Reading the recipient must refuse with the donor inode as expected and the recipient inode as observed. Ordinary migrated-root snapshot laws establish the positive path. This experiment does not exercise an actual device move or remount, and does not establish catalog-path pinning, every read interleaving, or platform admission beyond the existing identity probe. diff --git a/docs/adr/retention-recovery-directory-binding.md b/docs/adr/retention-recovery-directory-binding.md new file mode 100644 index 00000000..84467b2a --- /dev/null +++ b/docs/adr/retention-recovery-directory-binding.md @@ -0,0 +1,32 @@ +# Bind explicit recovery to admitted protocol directories + +This decision owns directory binding at the retention recovery entry point. +It implements the existing exact-capability invariant from ADR-0009 and +addresses the directory-replacement finding on PR #99 for issue #19. + +Writer admission pins the store root, retention directory and both immutable +pools. Retaining those handles avoids following a replacement directory, but +does not prove that the live protocol names still identify the admitted +directories. Recovery through an unreachable old handle can otherwise delete +stage evidence while returning success for a different visible store. + +`FilesystemRetentionPublicationAuthority::recover` runs the existing +device/inode guard before observing stages or executing recovery effects. +Publication invokes this same entry point, so both paths apply one rule. A +replacement returns the existing `ProtocolDirectoryReplaced` refusal through +the recovery `Observe` boundary, preserving its source. No durable bytes, +identity encoding, synchronization order or successful recovery classification +changes. + +Keeping the guard only in publication was rejected because explicit recovery +is a public production path. Reopening replacement directories was rejected +because they did not supply the authority's admission proof. The check is +synchronous capability-relative I/O with bounded extra directory opens and no +record-body allocation. + +Three permanent laws rename and recreate retention, roots and manifests in +turn, then require the exact refusal and unchanged retained root-stage bytes. +The laws fail on head `c9277eadbcb22a5ba1330760a3b908178957d4e5` at the +intended refusal assertion. They use filesystem unit fixtures; they establish +post-admission binding rather than production platform admission or physical +power-loss durability. Other recovery findings remain independently open. diff --git a/docs/adr/retention-recovery-discard-prefix.md b/docs/adr/retention-recovery-discard-prefix.md new file mode 100644 index 00000000..2cc80c57 --- /dev/null +++ b/docs/adr/retention-recovery-discard-prefix.md @@ -0,0 +1,13 @@ +# Retention recovery discard prefix + +Historical decision: its automatic-discard policy is superseded by [bounded retention recovery landing](retention-bounded-recovery-landing.md). Current recovery preserves incomplete stages before effects; the older diagnostic and test evidence below does not authorize disposal. + +Status: Accepted + +The forward protocol begins a manifest write only after the complete root stage is linked, and begins a head write only after complete root and manifest stages are linked. Retained earlier stages are removed only after head replacement. An incomplete later stage with absent earlier evidence therefore cannot be treated as a routine interrupted write. + +Pure recovery planning admits the complete linked earlier prefix before scheduling an incomplete manifest or head for deletion. Missing, incomplete, or differently linked earlier evidence returns `TruncatedStageWithoutEarlierEvidence`, naming the incomplete stage and the earlier stage that failed admission. Subsequent existing planning still checks root history, manifest history, and cross-record relationships before executing any scheduled discard. + +The filesystem observer supplies exact record admission and retained-stage inode binding. This guard does not infer completeness or identity from pathname existence, and it does not change on-disk encoding. The refusal enum is nonexhaustive; the new typed variant makes an impossible prefix distinguishable from a later-effect conflict. + +Runtime regressions remove one earlier stage or link from an otherwise valid forward prefix and shorten the later stage to an admissible framing prefix. Refusal must preserve every retained byte. Legitimate incomplete prefixes remain discardable under the existing forward-prefix laws; concurrent replacement during unlink and closure verification remain separate obligations. diff --git a/docs/adr/retention-recovery-head-binding.md b/docs/adr/retention-recovery-head-binding.md new file mode 100644 index 00000000..007802e2 --- /dev/null +++ b/docs/adr/retention-recovery-head-binding.md @@ -0,0 +1,13 @@ +# Retention recovery head binding + +This decision owns the cross-record relationship of a complete staged retention head and its selected manifest in PR #99. Canonical framing and checksums admit each record independently; they do not prove that the records describe one transition. + +Previously, recovery compared head and staged manifest digest and generation but omitted the exact manifest byte length and predecessor. A checksummed head with another valid length or another predecessor could therefore replace `HEAD` and permit stage cleanup, leaving a committed head that ordinary current-state observation rejects. Test commit `cd9e948` records both real-filesystem failures on unfixed production `6852d41`. + +The pure recovery planner now compares all four coordinates before finalization. It converts the admitted manifest's encoded length with checked conversion, requires equality with the head's declared length, and returns `HeadStageNamesOtherManifest` on disagreement. It requires head and manifest predecessor equality and returns `HeadPredecessorMismatch` on disagreement. The existing comparison against published predecessor history remains a separate requirement. + +Accepting a shared digest as sufficient was rejected because other persisted head fields remain user-visible protocol claims. Deferring the check to post-publication observation was rejected because recovery would already have committed ambiguous evidence. Re-encoding or repairing the head was rejected under the Core Law. No durable bytes, logical identities, or public error variants change. + +The filesystem regressions require preservation of the absent or previous published head, the exact typed planning refusal, and all retained bytes. A deterministic public-planner sweep exercises every canonical manifest length derived from the bounded entry domain, refusing every mismatch and accepting the exact length. This finite sweep reports the first counterexample directly; it uses no ambient randomness or arbitrary case total as its oracle. + +These checks establish staged head-to-manifest binding. They do not establish complete successor entry-set preservation, predecessor root reopening, closure verification, or physical power-loss durability. Those remain independent review obligations. diff --git a/docs/adr/retention-recovery-head-root-history.md b/docs/adr/retention-recovery-head-root-history.md new file mode 100644 index 00000000..1a0bdc10 --- /dev/null +++ b/docs/adr/retention-recovery-head-root-history.md @@ -0,0 +1,11 @@ +# Retention recovery head root history + +Status: Accepted + +An uncommitted complete head must name a staged root satisfying the same root-history predicate used for root-only and manifest-only recovery. Matching canonical head and manifest records, authenticated selected predecessor bytes, and identical pool inodes do not establish that the candidate generation is the exact successor. + +Head planning applies `root_succeeds` before publication effects and returns the existing typed `RootNotSuccessor` refusal on disagreement. The predicate requires generation one without a predecessor for initial or newly inserted namespaces, and the checked successor generation with the selected predecessor digest for existing namespaces. + +Already committed recovery retains its cleanup exemption because its selected root is the candidate itself rather than the candidate's predecessor. Requiring that root to succeed itself would reject legitimate interrupted cleanup. + +Filesystem regressions construct canonical, mutually consistent root, manifest, and head stages at the pre-head publication boundary with skipped generations, noninitial first publication, and noninitial namespace insertion. They require exact typed refusal and preservation of every retained byte. A bounded candidate-generation sweep includes the unsigned maximum; the evidence receipt states its domain and remaining blind spots. These experiments demonstrate runtime publication behavior without claiming every generation, operating-system schedule, or power-loss outcome. diff --git a/docs/adr/retention-recovery-namespace-admission.md b/docs/adr/retention-recovery-namespace-admission.md new file mode 100644 index 00000000..41640bad --- /dev/null +++ b/docs/adr/retention-recovery-namespace-admission.md @@ -0,0 +1,13 @@ +# Retention recovery namespace admission + +This decision owns namespace admission before filesystem retention recovery effects in PR #99. It preserves the existing rule that unknown or noncanonical protocol state is ambiguity and must refuse without destroying evidence. + +Previously, publication invoked recovery before its forward namespace census. An unknown retention entry could therefore receive the correct publication refusal after recovery had deleted an otherwise canonical incomplete stage. Direct recovery omitted that census entirely and could return `Clean` for the same invalid namespace. Test commit `282c105` records both failures on unfixed production `a038b1c`. + +Recovery first revalidates the pinned protocol directories and invalidates any pending in-memory publication attempt. It then admits retention entry names and root and manifest pool names and kinds, before observing stage bytes, planning, reopening stage handles, or executing filesystem effects. The existing typed namespace refusal remains the source of its `Observe` error. Publication invokes this same recovery entry point and retains its strict forward census afterward. + +The forward and recovery censuses share one admission implementation. Forward publication permits only `HEAD`, `roots`, and `manifests`; recovery additionally permits `root.next`, `manifest.next`, and `head.next`. Accepting these stage names is not accepting their contents: the existing bounded, no-follow stage observation and canonical assessment still decide their kinds, bytes, and semantics. + +Using the forward-only name list for restart was rejected because lawful crash stages would refuse. Keeping admission after recovery was rejected because it destroys unadmitted evidence. Silently deleting unknown entries or treating pool filenames as content proof was rejected under the Core Law. No new durable encoding or public error variant is introduced. + +The real-filesystem laws exercise explicit and publication-triggered recovery, check the exact typed namespace source, and compare every retained file's bytes before and after refusal. This establishes names-and-kinds admission ordering; it does not prove complete pool-content validation, capacity enforcement in every recovery transition, or physical power-loss survival. Those remain independent review obligations. diff --git a/docs/adr/retention-recovery-observation-errors.md b/docs/adr/retention-recovery-observation-errors.md new file mode 100644 index 00000000..d853712f --- /dev/null +++ b/docs/adr/retention-recovery-observation-errors.md @@ -0,0 +1,13 @@ +# Publication preserves recovery observation failures + +Status: accepted. + +Publication performs restart recovery before admitting a forward attempt, so callers must distinguish a failure while observing recovery evidence from a later forward current-state refusal. + +`RetentionCurrentStateRefusal::RecoveryObservationRefused` wraps the original observation `io::Error`, exposes it through the standard error source chain, and is itself returned through publication's existing `CurrentVerification` boundary. + +The outer storage I/O kind is `InvalidData`, consistent with the other typed current-state refusals; the original I/O kind, raw OS code and any typed source remain in the nested observation error. This is an intentional diagnostic API change for the unreleased publication API, including failures that previously escaped as raw I/O errors. + +Explicit recovery continues to return `FilesystemRetentionRecoveryError::Observe`. Planning and execution failures retain their existing `RecoveryRefused` and `RecoveryStepRefused` wrappers during publication. + +Returning observation failures directly was rejected because it erased which recovery phase had failed. Stringifying them was rejected because callers need their original typed or operating-system cause. This changes diagnostic structure without changing effects, publication ordering, durable encodings or the observation refusal itself. diff --git a/docs/adr/retention-recovery-pool-identity.md b/docs/adr/retention-recovery-pool-identity.md new file mode 100644 index 00000000..9736694b --- /dev/null +++ b/docs/adr/retention-recovery-pool-identity.md @@ -0,0 +1,13 @@ +# Retention recovery pool identity + +This decision owns exact pool-link admission before filesystem retention recovery effects in PR #99. Immutable pool names identify canonical content, but a retained publication stage also witnesses the exact file object the publication protocol linked. + +Previously, pool observation compared only bytes. Replacing a root or manifest pool link with a distinct byte-identical file therefore yielded `Identical`. Recovery finalized `HEAD`, then stage removal refused the inode mismatch; the manifest substitution also allowed root-stage removal before refusal. Test commit `7d45af2` records both real-filesystem failures on unfixed production `aa71bc0`. + +Each complete stage observation carries the device and inode identity read from its opened handle. Pool observation passes that identity and the exact canonical bytes to the existing bounded, no-follow exact-record verifier. The verifier checks the opened pool handle and named entry before and after reading. Absence remains `Absent`; a kind, length, identity, or byte contradiction becomes `Different`; operational I/O failures retain their source. + +The pure planner continues to return the existing pool-specific `PoolEntryDiffers` refusal for `Different` before executing any recovery effects. Pool identity is a storage transition witness and never becomes stable public content identity. Forward stage publication and removal already enforce the same physical binding, so recovery now admits it before head publication rather than discovering its failure during cleanup. + +Adopting equal bytes on a different inode was rejected because it weakens the exact hard-link publication contract. Repairing the pool link or delaying validation until cleanup was rejected because ambiguous evidence must refuse before mutation. No format bytes, public error variants, or logical identities change; the public observation documentation now states the required stage-object binding. + +The regressions confirm equal bytes on distinct inodes, require `HEAD` to remain absent, require the exact pool-specific planning refusal, and compare every retained file's bytes. Existing successful recovery laws retain same-inode hard-link admission. This establishes observation-time substitution refusal; it does not establish safety against every later concurrent replacement, full closure or predecessor admission, or physical power loss. Those remain independent review obligations. diff --git a/docs/adr/retention-recovery-selected-root-admission.md b/docs/adr/retention-recovery-selected-root-admission.md new file mode 100644 index 00000000..3f35fb16 --- /dev/null +++ b/docs/adr/retention-recovery-selected-root-admission.md @@ -0,0 +1,13 @@ +# Retention recovery selected root admission + +This decision owns reopening the published root selection before filesystem retention recovery effects. A manifest entry supplies coordinates; it does not prove that the corresponding predecessor pool record remains available or authentic after a crash. + +Regression commit `3cd7bde` records real-filesystem RED on production `48187bd`, whose documentation-only successor is `8686931`. Missing, checksum-corrupt, and canonical substituted predecessor records each allowed recovery to link the successor root, link its manifest, or replace the head. Every tested transition failed the named retained-byte preservation assertion. + +After pure planning and before reopening executable stages, recovery now reuses forward publication's selected-root admission. An uncommitted successor reopens the manifest-selected predecessor through the pinned roots capability, bounds the read, decodes the record, and compares its generation and digest to that selection. An already-selected staged root instead uses committed-root admission to require its exact bytes; a newly inserted namespace has no published predecessor to reopen. + +Both direct restart recovery and publication's automatic recovery enter this guard. Missing predecessors retain `PredecessorRootAbsent`; corrupted or substituted predecessors retain `PredecessorRootChanged`, carried by the observation error. The guard performs no writes, links, cleanup, or directory creation. + +Trusting manifest coordinates without reopening their record was rejected because it permits publication from unavailable evidence. Deferring the read until cleanup was rejected because head replacement would already have occurred. Repairing or substituting a predecessor was rejected under the Core Law. Durable formats and identities remain unchanged. + +The fault matrix crosses missing, corrupted, and substituted records with root linking, manifest linking, and head replacement. The tests observe all retained file bytes and the exact typed refusal. These checks do not establish full staged closure verification, every predecessor-coordinate rule in the pure planner, safety against arbitrary later out-of-band replacement, or physical power-loss durability. diff --git a/docs/adr/retention-recovery-successor-entries.md b/docs/adr/retention-recovery-successor-entries.md new file mode 100644 index 00000000..ea4ce1de --- /dev/null +++ b/docs/adr/retention-recovery-successor-entries.md @@ -0,0 +1,13 @@ +# Retention recovery successor entries + +This decision owns preservation of unrelated namespace entries when recovery admits a staged successor manifest. A valid checksum, successor generation, predecessor digest, and matching staged root do not establish that the remaining entries preserve the published state. + +Test commit `f036229` records real-filesystem failures on unfixed production `514a521`: canonical staged manifests dropped or altered an unrelated namespace, and recovery linked the manifest or finalized the head. The retained-byte preservation assertion failed in each production path. + +Recovery now compares the canonically ordered entry streams after excluding the staged root's namespace. Every remaining namespace, root generation, and root digest must agree exactly. The comparison allocates nothing and rejects additions, omissions, and coordinate changes with `ManifestNotSuccessor` before publication effects. With no published state, no unrelated entries are admitted. + +Manifest-link planning and head-finalization planning enforce the same rule. Already-committed cleanup retains its existing handling: the observed current manifest is the staged manifest itself, and this comparison cannot reconstruct an unavailable predecessor. Reopening predecessor roots and verifying staged closure remain separate obligations. + +Checking only that the candidate root appears was rejected because that permits unrelated retention changes. Reconstructing a replacement manifest was rejected because recovery must refuse contradictory evidence rather than silently repair it. Durable encodings, logical identities, and public error variants remain unchanged. + +The filesystem regressions observe exact retained bytes and the typed refusal for both production paths. They do not establish exhaustive generated-input coverage, all historical-state admission, concurrency safety, or physical power-loss durability. diff --git a/docs/adr/retention-stage-prefix-refusal.md b/docs/adr/retention-stage-prefix-refusal.md new file mode 100644 index 00000000..e08396b6 --- /dev/null +++ b/docs/adr/retention-stage-prefix-refusal.md @@ -0,0 +1,23 @@ +# Refuse contradictory fixed bytes in interrupted retention stages + +This decision owns the fixed-field admission rule for incomplete `root.next`, `manifest.next`, and `head.next` records in PR #99. It implements the existing requirement that corruption refuses recovery without mutation. + +A length-first decoder correctly reports that a record is incomplete, but length alone cannot distinguish an interrupted canonical write from garbage. The former implementation admitted `invalid` as truncation, planned a discard, removed the file, and reported successful recovery. The public filesystem regression observes this failure on unfixed production code; the generated public-assessment laws also fail on unfixed commit `4513a776d1c901c659d20a75bcf436c96eedf43a`. + +After a decoder reports truncation, recovery compares every available byte of the format's fixed magic, version, header or record width, flags, anchor or entry width, and reserved fields against its canonical value. Contradiction produces the record-specific `PrefixByteMismatch` decode error and a `StageCorrupt` planning refusal. Its diagnostic contains the actual byte offset, expected byte, and observed byte; missing bytes are neither padded nor presented as observations. + +The public assessment, filesystem observation, and recovery-context reopening paths use the same stage assessment. Publication invokes explicit recovery before creating its stages, so forward publication and direct restart recovery use the same refusal rule. + +Accepting every short byte string was rejected because it destroys corrupt evidence. Rejecting every interrupted write was rejected because the protocol requires recovery of canonical pre-effect prefixes. Checking magic alone was rejected because a short record can already contradict another available fixed field. Completing an incomplete field with fabricated bytes was rejected because that would misrepresent the evidence. + +The comparison allocates no memory and examines only bounded fixed-field bytes. Complete-record decoding retains its existing admission and failure order. Durable bytes, identity, publication order, and sync behavior do not change. The public decode-error enums gain a variant, so consumers with exhaustive matches must update their matches. + +The filesystem regression requires the exact stage-specific refusal and equality of every retained file's bytes before and after refusal. Generated laws verify strict canonical prefixes and contradictory fixed bytes across interrupted lengths. Diagnostic, valid-prefix, and byte-preservation assertions were separately calibrated by mutations that changed production behavior and failed the named checks. The retention fuzz target now exercises stage assessment as well as complete decoding. + +This decision does not claim complete semantic admission of incomplete variable fields, validation of every recovery transition, or physical power-loss evidence. Those remain separate review obligations; the PR is not approved by this focused fix alone. + +## Complete generation fields + +A complete generation field in an incomplete record is independently decidable. Root generations and global liveness generations must be positive, so recovery uses the existing domain constructors as soon as all eight generation bytes are present. Zero produces the existing record-specific generation error and `StageCorrupt`; a shorter field remains unknown and is not padded or rejected as zero. + +Public laws sweep every later strict prefix after a complete zero generation, and the filesystem law requires corruption refusal and equality of every retained file's bytes. They fail on unfixed production `ce54ae50eb99ba6b44d45c43f97b80e64b1e55c4`. A separate production mutation deleting evidence before planning confirms the byte-preservation assertion independently of the refusal. This extends the fixed-field correction without claiming admission of other incomplete semantic fields. diff --git a/docs/adr/retention-storage-errors.md b/docs/adr/retention-storage-errors.md new file mode 100644 index 00000000..1b9a3ef8 --- /dev/null +++ b/docs/adr/retention-storage-errors.md @@ -0,0 +1,19 @@ +# Typed retention storage failures + +Status: accepted. + +Recovery callers must distinguish an operational failure from a disagreement between a retained record and its admitted evidence. + +The shared stage adapter previously merged distinct exact-record refusals into message strings, losing the variant before recovery execution could report it. + +`RetentionStorageError` now separates source-preserving I/O failures from `RetentionRecordRefusal`, a public semantic vocabulary that maps the filesystem exact-record refusals without importing private filesystem machinery into the recovery port. + +Every recovery capability returns this error and `RetentionRecoveryError` retains it with a typed accessor and its standard error source chain. + +Stage operations share the typed boundary across recovery and forward publication; forward publication keeps its existing I/O signature by preserving operational errors directly and boxing typed record refusals as the source of `InvalidData`. + +This is an intentional source-level change to the unreleased recovery-storage trait; downstream implementations must return the new error type, while durable encodings and operation ordering are unaffected. + +An I/O-only recovery port with downcasting at every consumer was rejected because it obscures the distinction the recovery API needs to expose directly. + +The existing messages for invalid recovery invocation state are retained as operational errors; this decision addresses exact-record refusal preservation and does not claim every internal protocol-state diagnostic is now a dedicated variant. diff --git a/docs/formats/segment-store-v2/README.md b/docs/formats/segment-store-v2/README.md index 71194b52..ddad4eef 100644 --- a/docs/formats/segment-store-v2/README.md +++ b/docs/formats/segment-store-v2/README.md @@ -6,14 +6,7 @@ catalog, and publication-head byte while adding explicit retention state, reader fences, migration evidence, and reserved GC and recovery-disposition namespaces. -ADR-0009 owns the cross-cutting retention and liveness decision. These pages -own its durable representation. The one-way migration, version-two reopen, and -forward retention publication, partial-prefix migration recovery, and the -68-case migration process-death matrix are implemented with executable -evidence. Retention recovery and reader fencing await integration from PR #99; -collection remains planned in #21. The [requirements ledger](requirements.md) -records exactly which requirements are proven. A version-1 store remains -admitted until its owner migrates it. +ADR-0009 owns the cross-cutting retention and liveness decision. These pages own its durable representation. The one-way migration, version-two reopen, forward retention publication, partial-prefix migration recovery, and the 68-case migration process-death matrix are implemented with executable evidence. Retention recovery and reader fencing are implemented in this branch; correctness remediation and independent acceptance remain tracked in PR #99. Collection remains planned in #21. The [requirements ledger](requirements.md) records requirements and their evidence status. A version-1 store remains admitted until its owner migrates it. ## Core laws @@ -45,6 +38,7 @@ The following pages form one protocol: manifest, and retention-head rules. - [Retention publication](retention-publication.md) owns closure admission and the generation transition. +- [Retention publication recovery](retention-recovery.md) owns interrupted stages, restart publication, and retention process-death boundaries. - [Closure verification](closure.md) owns deterministic traversal, exact resource accounting, authenticated reconstruction, and closure evidence. - [Closure corruption boundary](closure-corruption.md) owns the admitted-record @@ -87,23 +81,25 @@ one pinned catalog, preflight, preparation, and the 17-phase publication port; fresh writer-locked filesystem migration through all 21 phases, refusing a version-one store that still holds a retained stage; `FilesystemVersionTwoAdmission::reopen`, which jointly admits the marker, -intent, and receipt, binds the root's device, mount, and inode identity to the -intent, and pins the retention directories it admitted; and +intent, and receipt, binds the root's restart-stable device and inode identity +to the intent, and pins the retention directories it admitted; and `FilesystemRetentionPublicationAuthority`, which publishes initial and successor generations against the observed head, binds this store's catalog head and the catalog it selects, and refuses superseded candidates, retained stages, replaced protocol directories, and every namespace or capacity violation before mutation, each as a typed `RetentionCurrentStateRefusal`. -Not implemented: retention publication recovery and `KEEP-CRASH-036..052` -process-death evidence, the reader fence, model-based transition evidence, and -garbage collection. Partial-prefix migration recovery and the 68-case -`KEEP-CRASH-053..073` process-death matrix are implemented. Broader migration -restart corruption and compatibility coverage remain in #111 and #112. -PR #99 contains retention recovery and reader fencing awaiting integration; -issue #21 owns garbage collection. -Reopen compares only the restart-stable root coordinates, device and inode, -against the intent; see +Retention publication recovery, fenced reader snapshots and model-based transition evidence are implemented in this branch. + +Their bounded landing and independent acceptance remain tracked in PR #99 and the [landing ledger](../../testing-evidence/retention-landing.md). Incomplete retention stages are preserved pending explicit disposition; automatic disposal is deferred. Execution-failure reporting remains part of the [retention recovery contract](retention-recovery.md). + +The `KEEP-CRASH-036..052` initial-publication process-death sequence includes independent recovered-reader checks; its [evidence receipt](../../testing-evidence/retention-crash-reader-oracle.md) records the assertions, calibration, and scope. + +Partial-prefix migration recovery and the 68-case `KEEP-CRASH-053..073` +process-death matrix are implemented. Broader migration restart corruption +and compatibility coverage remain in #111 and #112; issue #21 owns garbage +collection. Reopen compares only the restart-stable root coordinates, device +and inode, against the intent; see [root identity across restart](recovery.md#root-identity-across-restart). A version-1 store remains admitted until its owner migrates it, and the diff --git a/docs/formats/segment-store-v2/recovery.md b/docs/formats/segment-store-v2/recovery.md index 646c0b4e..d339caae 100644 --- a/docs/formats/segment-store-v2/recovery.md +++ b/docs/formats/segment-store-v2/recovery.md @@ -88,8 +88,6 @@ Migration is a one-way explicit migration under exclusive writer authority. `migration.intent` is exactly 256 bytes: - - | Offset | Width | Field | Canonical value | | ---: | ---: | --- | --- | | 0 | 16 | magic | `KEEP:MIG:INT2\0\0\0` | @@ -108,8 +106,6 @@ Migration is a one-way explicit migration under exclusive writer authority. | 192 | 32 | new store identifier | deterministic derivation below | | 224 | 32 | checksum | BLAKE3-256 over bytes `0..224` | - - The checksum domain is `keep.store-migration-intent-checksum/v2\0`. The receipt's intent digest is BLAKE3-256 of `keep.store-migration-intent/v2\0` followed by all 256 intent bytes. @@ -181,8 +177,6 @@ necessary. `migration.receipt` is exactly 256 bytes: - - | Offset | Width | Field | Canonical value | | ---: | ---: | --- | --- | | 0 | 16 | magic | `KEEP:MIG:REC2\0\0\0` | @@ -198,8 +192,6 @@ necessary. | 216 | 8 | completed synchronization mask | every mandatory bit set | | 224 | 32 | checksum | BLAKE3-256 over bytes `0..224` | - - Its checksum domain is `keep.store-migration-receipt-checksum/v2\0`. Unknown synchronization bits, a missing mandatory bit, or any mismatch with the intent refuses. @@ -222,70 +214,6 @@ issue #112. ## Retention publication recovery -At restart, a fixed retention stage is classified from its exact framing and -transitive evidence: - -The forward protocol guarantees that `root.next` is durable before a new -namespace directory is created. A new digest-named directory is created -exclusively, verified as the exact regular directory rather than a link, and -followed by synchronization of `retention/roots` before the immutable root is -linked. An existing exact directory is idempotent; any wrong kind, substituted -namespace, or unexpected entry refuses. Directory existence alone never proves -a retained root. - - - -| Fixed stage | Complete evidence | Recovery | -| --- | --- | --- | -| `root.next` | canonical successor root, matching namespace and closure proof | finalize its immutable pool link and retain the stage | -| `manifest.next` | canonical successor manifest naming only admitted roots | finalize its immutable pool link and retain both stages | -| `head.next` | canonical successor head naming the staged manifest | finalize the head, synchronize it, then remove retained stages | - - - -A pre-effect incomplete stage may be removed only when every later-ordered -effect is absent and all earlier evidence admits exactly. Recovery pins that -regular file, removes it, synchronizes `retention`, and returns a typed discard -report. Any later effect, stale generation, mismatched digest, missing -transitive member, reappeared stage, conflicting pool entry, or other -corruption is a typed refusal. A complete valid orphan remains -recovery-protected until explicit disposition. - -This retention protocol requires pinning the incomplete regular file. The -separate [migration discard path](migration-crash.md#fixed-stage-law) -revalidates the current entry's regular kind and incomplete length before -removal without retaining an incomplete-stage handle. These are distinct -protocol boundaries; retention recovery awaits integration from PR #99. - -The retention crash points are: - -| Identifier | Boundary | -| --- | --- | -| `KEEP-CRASH-036` | root stage write | -| `KEEP-CRASH-037` | root stage synchronization | -| `KEEP-CRASH-038` | new namespace-directory creation or exact admission | -| `KEEP-CRASH-039` | namespace-pool synchronization after creation | -| `KEEP-CRASH-040` | immutable root link | -| `KEEP-CRASH-041` | root namespace-directory synchronization | -| `KEEP-CRASH-042` | manifest stage write | -| `KEEP-CRASH-043` | manifest stage synchronization | -| `KEEP-CRASH-044` | immutable manifest link | -| `KEEP-CRASH-045` | manifest pool synchronization | -| `KEEP-CRASH-046` | retention-head stage write | -| `KEEP-CRASH-047` | retention-head stage synchronization | -| `KEEP-CRASH-048` | retention-head atomic replacement | -| `KEEP-CRASH-049` | committed retention namespace synchronization | -| `KEEP-CRASH-050` | retained root-stage removal | -| `KEEP-CRASH-051` | retained manifest-stage removal | -| `KEEP-CRASH-052` | retention cleanup synchronization | - -`RetentionPublicationPhase::ALL` freezes this exact order as a typed public -vocabulary. Storage execution and process-death evidence remain unimplemented. - -Each point requires before, during, and after process-death evidence. Restart -must establish exact catalog visibility, retention head, namespace generation, -orphan classification, stage disposition, and recovery report. - -`GcRetirementIntent`, `GcRetirementReceipt`, and -`RecoveryDispositionReceipt` are owned by the [GC specification](gc.md). Until -issue #21 implements them, any such artifact is unsupported and refuses. +[Retention publication recovery](retention-recovery.md) owns fixed-stage classification, synchronization before publication, exact retained evidence, and the `KEEP-CRASH-036` through `052` process-death boundaries. + +`GcRetirementIntent`, `GcRetirementReceipt`, and `RecoveryDispositionReceipt` are owned by the [GC specification](gc.md). Until issue #21 implements them, any such artifact is unsupported and refuses. diff --git a/docs/formats/segment-store-v2/requirements.md b/docs/formats/segment-store-v2/requirements.md index 172bcb89..b58eaf19 100644 --- a/docs/formats/segment-store-v2/requirements.md +++ b/docs/formats/segment-store-v2/requirements.md @@ -15,10 +15,10 @@ case is not evidence. | `KEEP-RETENTION-004` | Retain and release compare expected and observed generations and publish exact successors only | unforgeable readiness and preflight proofs in `tests/retention_transition.rs` and `tests/retention_preflight.rs`; exact successor preparation and complete receipt evidence in `tests/retention_publication_preparation.rs` and `tests/retention_publication_execution.rs`; writer-locked initial filesystem publication in `filesystem_retention_storage_tests`; observed-head successor publication, exact predecessor binding, and absent-head refusal in `filesystem_retention_successor_tests`; the store's catalog head must name the closure's catalog generation and digest before any forward write in `filesystem_retention_catalog_tests`; a head whose predecessor disagrees with its manifest refuses in `filesystem_retention_current_tests`; a successor reopens and decodes the manifest-selected predecessor root and refuses an absent or changed one in `filesystem_retention_expectation_tests` | Implemented | | `KEEP-RETENTION-005` | Closure derivation is deterministic, bounded, cycle-safe, fail-closed, and verifies complete blob reconstruction | exact accounting, reconstruction, adversarial-catalog, and exhaustive model laws in `tests/retention_closure.rs`; corrupt members refuse through the inherited segment-record admission laws and seeded `segment_format` fuzz target routed by `closure-corruption.md` | Implemented | | `KEEP-RETENTION-006` | Publication follows the exact ordered durability protocol, including new namespace-directory admission and retention of fixed-stage evidence until head commit, and returns only after cleanup synchronization | typed vocabulary and blocking port in `tests/retention_publication_phase.rs` and `tests/retention_publication_storage.rs`; ordered execution, conditional namespace sync, and all 17 exact storage-fault boundaries in `tests/retention_publication_execution.rs`; production 17-phase forward filesystem execution, exclusive staging, byte-equal inode-substitution refusal, and retained-stage recovery refusal in `filesystem_retention_storage_tests`; orphan namespace directories count against the 4,096 ceiling and refuse a new namespace before any stage is written in `filesystem_retention_capacity_tests`; crash injection remains | In progress in #19 | -| `KEEP-RETENTION-007` | Restart resolves every fixed-stage crash prefix to one documented lawful state or typed ambiguity | recovery-required refusals before any mutation in `filesystem_retention_expectation_tests`: an absent head over populated pools, a non-initial head prepared against an absent head, an orphan directory for a namespace expected absent, and an absent directory for a namespace expected current; debug and release crash matrix remains; replaced protocol directories, an absent or changed head-selected catalog, an over-full census, zero-generation pool names, and a stage retained by a failed write refuse in `filesystem_retention_*_tests` | In progress in #19 | -| `KEEP-RETENTION-008` | Readers double-collect catalog and retention heads and bind one complete catalog, manifest, and root-generation view under a `ReaderFence` | immutable snapshot and concurrency tests | Planned in #19 | +| `KEEP-RETENTION-007` | Complete-stage recovery preserves canonical history; incomplete stages require disposition before mutation | Complete publication-prefix recovery and reader-state laws remain; incomplete direct/publication recovery and process-death laws now require typed refusal and preserved bytes. See [landing ledger](../../testing-evidence/retention-landing.md). | Bounded landing in #99; automatic incomplete-stage disposition explicitly deferred | +| `KEEP-RETENTION-008` | Readers double-collect catalog and retention heads and bind one complete catalog, manifest, and root-generation view under a `ReaderFence` | `ReaderFence` holds a shared kernel lock on a verified zero-length `reader.lock`; `collect_retention_view` accepts a view only when both head coordinates agree before and after loading and refuses an exhausted attempt limit (`retention_view_collector_tests`); `FilesystemRetentionSnapshot` binds the catalog snapshot, retention head, and manifest under the fence and verifies each selected root on demand while the fence is held, refusing a substituted root and a replaced fence, and two readers share the fence while an exclusive lock waits (`filesystem_retention_snapshot_tests`) | Implemented | | `KEEP-RETENTION-009` | Exact already-committed retry is idempotent only while its successor remains current | byte-identical planning in `tests/retention_transition.rs`; authority-revalidated zero-mutation retry receipt in `tests/retention_publication_execution.rs`; exact already-committed filesystem retry with a byte-identical retention witness in `filesystem_retention_storage_tests`; superseded-candidate filesystem refusal with zero mutation in `filesystem_retention_successor_tests`; committed retry reopens the head-selected manifest entry and root pool bytes, refusing absent, changed, or corrupt evidence in `filesystem_retention_current_tests`; every refusal is a typed `RetentionCurrentStateRefusal` source, with superseded, committed-root-absent, committed-root-changed, and head-absent-with-artifacts pinned by downcast | Implemented | -| `KEEP-RETENTION-010` | Model operation sequences agree with a deterministic namespace-to-anchor-set map and never admit caller identity, paths, clocks, or application policy | model-based and source-architecture tests | Planned in #19 | +| `KEEP-RETENTION-010` | Model operation sequences agree with a deterministic namespace-to-anchor-set map and never admit caller identity, paths, clocks, or application policy | every three-operation sequence over initial publications of two namespaces, a successor, a byte-identical retry, and a stale initial (125 sequences, each in a fresh migrated store) agrees with a deterministic namespace-to-(generation, anchor-set) map plus liveness after every step, observed through the fenced reader view, in `retention_model_tests`; `tests/retention_core_architecture_contract.rs` refuses any clock, path, environment, or identity token in the retention core | Implemented | @@ -60,13 +60,13 @@ case is not evidence. - A fresh forward writer is not proof that version 2 is restart-safe or production-admitted; partial-prefix recovery and crash evidence remain mandatory. This applies to retention publication exactly as it applies to - migration: the filesystem publication writer refuses every retained stage - instead of continuing it. A stage left behind by a failed write is recovery - evidence like any crash residue; it is never unlinked, and the next - publication refuses until recovery classifies it. + migration: the filesystem publication writer never continues a retained + stage; it recovers complete evidence first, preserving incomplete stages for disposition, + finalizing a complete head, and refusing a complete orphan until explicit + disposition. A stage left behind by a failed write is recovery evidence like + any crash residue and is classified the same way. - Publication binds this store's catalog `HEAD` to the verified closure and - reopens the head-selected catalog pool entry under authority, but it does - not re-read closure-member segments: every read authenticates them, and - their re-verification under authority belongs to retention recovery. + reopens the head-selected catalog and its segments under authority. Forward + publication and recovery both replay the bounded live closure before effects. - Benchmarks are required before performance-sensitive retention or migration optimization. diff --git a/docs/formats/segment-store-v2/retention-recovery.md b/docs/formats/segment-store-v2/retention-recovery.md new file mode 100644 index 00000000..c2070841 --- /dev/null +++ b/docs/formats/segment-store-v2/retention-recovery.md @@ -0,0 +1,95 @@ +# Retention Publication Recovery + +This page owns fixed retention stages, restart publication, and the retention process-death boundaries. + +Both publication-triggered and explicit recovery first verify that `retention`, `retention/roots`, and `retention/manifests` still name the directories writer admission pinned. A replaced directory refuses as `ProtocolDirectoryReplaced` before stage observation or mutation; explicit recovery preserves that refusal inside its `Observe` error boundary. + +Both entry points then admit every retention entry name and every root and manifest pool entry's canonical name and regular kind before observing stages or executing recovery. Recovery permits only the three fixed stage names in addition to the forward namespace; stage observation separately verifies their exact kinds, bounds, and bytes. An unknown retention entry, non-namespace root entry, or noncanonical pool entry refuses inside `Observe` with its existing typed namespace source and preserves every retained file's bytes. + +A root or manifest pool entry named by a complete retained stage must match that stage's device and inode identity as well as its exact bytes. Observation verifies the opened pool handle and named entry on both sides of the bounded read. Equal bytes on a substituted inode classify as `Different`, so planning returns `PoolEntryDiffers` for the exact pool before finalizing `HEAD` or removing stages. Inode coordinates are transition evidence, not public content identity. + +A complete head stage without a manifest stage refuses as `HeadStageWithoutManifestStage`. When both head and manifest stages are complete but the root stage is absent, recovery instead reports `ManifestStageWithoutRootStage`, preserving the retained evidence before any execution. + +A complete staged head must name the staged manifest's exact digest, generation, byte length, and predecessor. A length disagreement refuses as `HeadStageNamesOtherManifest`; a predecessor disagreement refuses as `HeadPredecessorMismatch`. Checksums and valid individual field values do not establish this relationship. Recovery compares these coordinates before head finalization or stage cleanup, preserving the previously published head and all retained bytes on refusal. + +A staged successor manifest may change only the staged root's namespace entry. Every unrelated namespace entry must preserve its exact namespace, root generation, and root digest; additions, omissions, and changes refuse as `ManifestNotSuccessor` before manifest linking or head finalization. Already-committed cleanup uses its existing current-state handling rather than reconstructing unavailable predecessor history. + +Before executing a recovery plan for a complete staged root, the filesystem authority reopens the published root selected for that namespace through its pinned roots capability. A successor requires the predecessor record to decode to the manifest's exact generation and digest; an already-selected staged root requires exact committed bytes. Missing, corrupt, or substituted selections refuse at observation before any recovery effects. A newly inserted namespace has no published predecessor. + +Every complete staged root also requires a fresh catalog snapshot loaded relative to the authority's pinned root capability. Recovery authenticates the current head, catalog, and all selected segments, then replays every staged anchor and its stored closure limits through the same pure verifier used by preflight. Missing members, corruption, coordinate disagreement, and closure-limit failures remain typed observation sources and refuse before root linking, manifest linking, head replacement, or cleanup. Forward publication performs the same live verification after checking the prepared catalog coordinates; the fresh proof must still name that prepared catalog. + +`recover_with_catalog_policy` accepts explicit segment-admission and aggregate retained-segment byte limits. It materializes the selected catalog and all selected segments; catalog encoding and decoded indexes have additional protocol bounds. `recover` and forward publication use an aggregate retained-segment cap equal to the protocol's maximum segment length, 1 GiB, independently of the staged root's closure counters. Larger selections refuse before mutation; a recovery caller may supply a larger explicit policy. This cap is a loading policy rather than an on-disk format limit or a total resident-memory claim. Clean recovery and recovery without a complete staged root do not load segment bodies. + +At restart, a fixed retention stage is classified from its exact framing and transitive evidence: + +The forward protocol guarantees that `root.next` is durable before a new namespace directory is created. A new digest-named directory is created exclusively, verified as the exact regular directory rather than a link, and followed by synchronization of `retention/roots` before the immutable root is linked. An existing exact directory is idempotent; any wrong kind, substituted namespace, or unexpected entry refuses. Directory existence alone never proves a retained root. + +| Fixed stage | Complete evidence | Recovery | +| --- | --- | --- | +| `root.next` | canonical successor root, matching namespace and closure proof | finalize its immutable pool link and retain the stage | +| `manifest.next` | canonical successor manifest naming only admitted roots | finalize its immutable pool link and retain both stages | +| `head.next` | canonical successor head naming the staged manifest | finalize the head, synchronize it, then remove retained stages | + +Available fixed-field bytes in an interrupted retention stage must match the canonical magic, version, header or record width, flags, anchor or entry width, and reserved fields. A contradiction is `StageCorrupt` with a `PrefixByteMismatch` source naming the exact offset, expected byte, and observed byte; recovery refuses before mutation and preserves every retained file. Absent bytes are not padded or reported as observed. Complete generation fields must also admit through the positive root or liveness generation constructor; zero is corruption even if later fields are absent. Incomplete generation fields remain undecided. These checks do not establish full completion feasibility; incomplete stages remain preserved. + +Recovery synchronizes each complete staged file and reverifies its exact bytes before creating its root or manifest pool link or replacing `HEAD`. The file synchronization precedes namespace creation for a recovered root. A failed synchronization stops that recovery step before publication and preserves the on-disk stage; directory synchronization alone does not establish durability of the file contents. + +Automatic disposition of incomplete retention stages is deferred for #99. If any stage cannot be admitted as a complete canonical record, planning returns either its precise known corruption refusal or `IncompleteStageRequiresDisposition`, before any recovery mutation. Complete earlier stages are not linked or cleaned up first. The typed incomplete result reports actual observed length and the decoder's required boundary; it does not certify that a valid completion exists. Publication-triggered recovery follows the same rule and remains blocked until an explicit disposition protocol is designed. Do not blindly delete retained stages. A complete valid orphan remains recovery-protected until explicit disposition. + +Once all bytes of a staged head's manifest-length field are available, recovery requires that value to satisfy the canonical manifest bounds and entry alignment even if the rest of the head is missing. An invalid value refuses as `StageCorrupt(Head)` with its typed `ManifestLength` cause, preserving the retained evidence. + +When a short head contains all generation and predecessor bytes, recovery applies the same semantic history rules as complete head decoding: generation one cannot name a predecessor, and a successor must name one. Contradictions refuse as `StageCorrupt(Head)` with the exact semantic cause before any stage removal. + +Once a short head contains the entire checksum preimage, recovery computes its checksum and checks every checksum byte already present. Any mismatch refuses as `StageCorrupt(Head)` with `PrefixByteMismatch` naming the offset and exact expected and observed bytes; the incomplete stage remains intact. + +A root or manifest prefix containing all framing-size fields must declare the exact length those fields imply. Recovery uses the same checked framing calculation as complete decoding and refuses a disagreement with `DeclaredLengthMismatch` inside the exact stage-corruption error before returning a disposition requirement. + +Available complete anchor and manifest-entry counts must satisfy their format ceilings during interrupted-stage assessment. Excessive counts refuse with `AnchorCountExceeded` or `EntryCountExceeded` inside the corresponding stage-corruption error, even when the declared record length matches the excessive count. + +A complete namespace-length field in an interrupted root must declare between one and 255 bytes. Recovery uses the domain namespace admission rule without allocating a namespace or waiting for its payload; empty or oversized declarations refuse with the exact namespace cause and preserve the stage. Namespace contents remain opaque and are not interpreted as text or paths. + +When an interrupted root or manifest contains its complete generation and predecessor fields, recovery applies the same domain history rules as complete-record construction. Initial records naming a predecessor and successors omitting one refuse with the precise semantic cause during interrupted-stage assessment; all retained evidence remains intact. + +For an initial root, manifest or head with an incomplete predecessor field, each available predecessor byte must be zero. A nonzero byte already excludes every valid initial record, so recovery refuses with the stage-specific `PrefixByteMismatch` and preserves the evidence. An incomplete successor predecessor remains possible even when every available byte is zero; a missing byte may still make the complete digest nonzero. Complete predecessor fields retain their existing semantic diagnostics. + +When an interrupted root contains the complete realization-profile group (through byte 88) or closure-policy group (through byte 116), recovery applies the domain profile and closure-limit validators during interrupted-stage assessment. Unsupported profile coordinates, definition-digest mismatches, zero limits, and excessive limits refuse as `StageCorrupt(Root)` with their exact typed cause and preserve retained evidence. For an incomplete profile group, every available identity, version and digest byte must match the sole registered profile; a contradiction refuses with `PrefixByteMismatch` naming only an observed byte. Each complete closure-limit field is admitted independently as soon as it arrives. For a partial big-endian limit, recovery checks whether any positive completion fits the domain ceiling; an impossible prefix refuses with `ClosureLimitPrefixAboveMaximum`, reporting the resource, minimum possible completion and maximum admitted value. Missing bytes are never reported as observed, and incomplete all-zero prefixes remain admissible because they can still become positive. + +For interrupted root and manifest records, recovery verifies each available trailer digest or checksum byte once its complete preimage is present. After the body has arrived, it also verifies the header's anchor-set or entry-set digest. Contradictions refuse before mutation with the exact available-byte coordinates or body-set digest cause; canonical prefixes remain interrupted writes. Complete-record decoding retains its existing checksum, digest and semantic validation order. + +Each complete anchor or manifest entry already present in an interrupted body is admitted by the same identity, generation and strict-order rules used by full decoding. Recovery scans only available complete entries without allocating the unavailable suffix; typed entry failures preserve the stage. For a partial manifest entry, recovery admits its root generation once complete and compares the greatest possible namespace completion against the preceding namespace; an impossible strict ordering refuses with the current entry index. Unknown namespace bytes are used only to compute that bound, never exposed as an observed identity. Partial root anchors now verify every available fixed BlobId and LayoutId byte against the embedded codecs' constants and admit a complete layout-length field with the existing codec validator. Refusals report absolute record-byte coordinates or the exact anchor index and layout-length cause. For an incomplete layout length, recovery computes its minimum completion and refuses `LayoutLengthPrefixAboveMaximum` when that lower bound exceeds the domain maximum; zero-leading prefixes remain possible and complete lengths retain their existing bounds and congruence checks. For a partial anchor with a predecessor, recovery constructs the greatest canonical completion consistent with the available bytes, caps and aligns the layout length through the domain rule, and applies the normal typed anchor ordering check. This completion is used only to prove possibility and is never returned as observed data. These local checks do not establish completion feasibility of every declared future entry; incomplete records still require disposition regardless of these checks. + +Incomplete manifests and heads require disposition even when all earlier evidence is present. Missing earlier evidence does not authorize deletion either. Known record-corruption diagnostics retain precedence over the incomplete result. Cross-record and history admission still precede execution for complete stages. + +Observation and planning refusals initiate no recovery mutation. Execution failure returns the failed step, the successfully completed earlier steps, the original typed storage cause, and the failing capability's `RetentionStorageProgress`. An empty completed-step list does not imply that the failing capability made no changes. Known effects carry either `Synchronized` or `Unconfirmed` directory durability; an uncertain attempted effect is reported separately. A failed synchronization never rolls back an earlier link, rename or unlink. An adapter that supplies no progress leaves effects unreported, not absent. + +Recovery stops immediately after an execution error and drops its execution context. Another attempt must observe and plan again. Complete-stage cleanup retains the opened source handle, verifies its observed identity and exact bytes as well as the immutable pool target before unlink, and checks source absence and surviving pool evidence afterwards. Successful unlink intentionally removes the original stage pathname; the preservation guarantee concerns verified pool evidence under the supported mutation model. A later verification or synchronization failure reports the completed removal with unconfirmed durability. + +The supported mutation model is cooperating writers under Keep authority in a managed namespace. The writer lock supplies no isolation guarantee against arbitrary concurrent out-of-band namespace mutation. Existing no-follow, identity, exact-byte, namespace and corruption checks remain required and observed substitutions refuse. Open handles and metadata checks do not make pathname unlink or rename conditional on inode identity. The separate [migration discard path](migration-crash.md#fixed-stage-law) has its own contract and is unchanged by this retention scope decision. + +Complete-stage recovery and incomplete-stage preservation are implemented in this branch. The [landing ledger](../../testing-evidence/retention-landing.md) tracks remaining execution-failure reporting and exact-head acceptance for [PR #99](https://github.com/flyingrobots/keep/pull/99). Automatic incomplete-stage disposal and its stronger prefix-completion contract are explicitly deferred; historical discard-success evidence is not a claim of current behavior. + +The retention crash points are: + +| Identifier | Boundary | +| --- | --- | +| `KEEP-CRASH-036` | root stage write | +| `KEEP-CRASH-037` | root stage synchronization | +| `KEEP-CRASH-038` | new namespace-directory creation or exact admission | +| `KEEP-CRASH-039` | namespace-pool synchronization after creation | +| `KEEP-CRASH-040` | immutable root link | +| `KEEP-CRASH-041` | root namespace-directory synchronization | +| `KEEP-CRASH-042` | manifest stage write | +| `KEEP-CRASH-043` | manifest stage synchronization | +| `KEEP-CRASH-044` | immutable manifest link | +| `KEEP-CRASH-045` | manifest pool synchronization | +| `KEEP-CRASH-046` | retention-head stage write | +| `KEEP-CRASH-047` | retention-head stage synchronization | +| `KEEP-CRASH-048` | retention-head atomic replacement | +| `KEEP-CRASH-049` | committed retention namespace synchronization | +| `KEEP-CRASH-050` | retained root-stage removal | +| `KEEP-CRASH-051` | retained manifest-stage removal | +| `KEEP-CRASH-052` | retention cleanup synchronization | + +`RetentionPublicationPhase::ALL` freezes this exact order as a typed public vocabulary. `FilesystemRetentionPublicationAuthority::recover` implements the classification above and its effects, and the crash matrix kills a real writer before, during, and after every point and requires restart to recover to the documented state. + +Each point requires before, during, and after process-death evidence. Restart must establish exact catalog visibility, retention head, namespace generation, orphan classification, stage disposition, and recovery report. diff --git a/docs/formats/segment-store-v2/retention.md b/docs/formats/segment-store-v2/retention.md index 6f7402f2..44a8668c 100644 --- a/docs/formats/segment-store-v2/retention.md +++ b/docs/formats/segment-store-v2/retention.md @@ -165,8 +165,9 @@ implements root, manifest, and head codecs with a typed verified anchor-set digest, expected-state transition planning, deterministic closure verification, a blocking publication storage capability port, and ordered storage-port orchestration. `FilesystemRetentionPublicationAuthority` publishes initial and -successor generations against its observed head and refuses superseded -candidates and retained stages; recovery, fencing, and collection remain absent. +successor generations against its observed head, recovers retained stages +first, and refuses superseded candidates and protected orphans; fencing and +collection remain absent. ## Global retention manifest diff --git a/docs/testing-evidence/migration-clone-failure.md b/docs/testing-evidence/migration-clone-failure.md new file mode 100644 index 00000000..12dd4a22 --- /dev/null +++ b/docs/testing-evidence/migration-clone-failure.md @@ -0,0 +1,21 @@ +# Migration capability-clone failure + +Change kind: bug fix. Owner: `@flyingrobots`. Oracle: a failed root capability duplication reports `FilesystemMigrationAuthorityError::Namespace` with the original operating-system cause; root identity observation has its separate `RootIdentity` boundary. + +At base `84f5861`, the repository-task migration constructor used the unchecked-admission helper and labeled either failure `Platform`, despite performing no platform admission. The original review referred to an older location and a `RootIdentity` wrapper; the current defect was verified at the moved repository-task constructor. + +RED commit `50d0d0b` adds a Linux child-process regression that creates a valid version-one store, reacquires its real writer lock, fills a child-owned descriptor table, then invokes the public repository-task migration constructor. The kernel returns `EMFILE` during root capability cloning; the unfixed code reports `Platform` and fails the named namespace assertion. + +The child runs with a 64-descriptor limit and a 20-second deadline through the Docker image's shell and timeout utility. The parent retains its descriptor limits. Descriptors are released before inspecting the returned error, and no test mutates global process limits in the shared test runner. + +The fix performs capability cloning and lenient identity observation separately, preserving each source at its proper boundary, and releases the temporary cloned root before opening inventory capabilities as the previous helper did. + +The physical clone failure is exercised directly; identity-probe failure mapping is verified by code inspection, not claimed as a separately injected kernel fault. This evidence does not broaden the repository-task API into production platform admission. + +A copied-source mutation retained the `Namespace` variant but reconstructed the I/O error from its kind, losing `EMFILE`; the same assertion failed, independently calibrating source preservation in addition to the original wrong-boundary RED. + +The migration suite passed in debug and release. After explicitly restoring the temporary root handle's original lifetime, the focused regression passed again in both modes, with both workspace Clippy feature configurations, formatting and source-structure checks. Markdown validation passed separately. + +The descriptor law now lives in the dedicated `migration_descriptor_exhaustion` integration binary, so spawning its child cannot inherit descriptors from concurrent library tests. A full workspace run caught the original source-layout violation; the guard was retained and the test moved. Setup uses the public repository initialization API to obtain writer authority over a fresh version-one namespace before exhausting descriptors. + +Replay uses `cargo test --test migration_descriptor_exhaustion --all-features`, adding `--release` for optimized execution, inside copied Linux Docker. The fault schedule is deterministic descriptor exhaustion after writer-lock acquisition. Other targets do not run this Linux-specific regression. diff --git a/docs/testing-evidence/reader-root-checksum-refusal.md b/docs/testing-evidence/reader-root-checksum-refusal.md new file mode 100644 index 00000000..adaad915 --- /dev/null +++ b/docs/testing-evidence/reader-root-checksum-refusal.md @@ -0,0 +1,17 @@ +# Reader root checksum refusal + +Change kind: test-oracle repair. Owner: `@flyingrobots`. Subject: the exact public reader diagnostic for a corrupted selected root. + +At parent `68df384`, the filesystem test flipped the final checksum byte but accepted any `FilesystemRetentionSnapshotError::Root`, so an unrelated root failure satisfied the test. + +The renamed law now requires `Root` containing `InvalidData` with `RetentionRootDecodeError::ChecksumMismatch`. Its expected checksum comes from the independently checked-in root fixture's final checksum bytes; its observed checksum comes from those bytes after the specified bit flip. It checks both fields without calling the production checksum helper to compute its expectation. + +The mutation setup now refuses an empty fixture instead of silently skipping the byte flip. The test still publishes through real filesystem authority and reads the selected root through the public fenced snapshot API. + +Copied-source calibration substitutes `RootDigestMismatch` for the checksum variant and, separately, swaps the expected and observed checksum fields. These exercise the named runtime diagnostic rather than the fixture or test harness. + +Both mutations passed the parent test and failed the revised exact-checksum assertion. Each revision used a fresh build target, and neither runtime RED result was a setup or compilation failure. + +The filesystem snapshot suite passes in copied Linux Docker in debug and release; both workspace Clippy feature configurations, formatting and source-structure checks pass. Replay uses `cargo test --lib root_checksum_damage --all-features`, adding `--release` for optimized execution. Markdown lint is separate static validation. + +This evidence covers a single selected-root checksum mutation and its diagnostic fields. It does not establish arbitrary corruption coverage, physical power-loss behavior, or kernel fault handling. diff --git a/docs/testing-evidence/retention-closure-limit-prefix.md b/docs/testing-evidence/retention-closure-limit-prefix.md new file mode 100644 index 00000000..a0cf8695 --- /dev/null +++ b/docs/testing-evidence/retention-closure-limit-prefix.md @@ -0,0 +1,21 @@ +# Interrupted closure-limit admission + +Change kind: bug fix. Owner: `@flyingrobots`. Subject: Keep runtime refusal and preservation of interrupted roots whose available closure limits cannot admit a legal completion (#99). + +Regression `25f4be2` was observed RED with unchanged admission behavior from `3f3b7ea`: a node-limit prefix containing one `0xff` byte was discarded with `Clean`. The RED commit adds the diagnostic variant and formatting/source support so the regression compiles, but does not add prefix admission. + +The medium filesystem law exercises every nonempty strict prefix of each limit with a leading `0xff`, plus each complete field containing zero or the specified ceiling plus one. It requires the exact corrupt stage, typed domain error or impossible-prefix coordinates, and unchanged retained paths and bytes. Its specified oracle is the format's positive bounded big-endian integer contract; test ceilings are independent protocol literals. + +The small public-assessment law deterministically generates admitted powers of two through each ceiling and the ceiling itself, then checks every nonempty prefix. This protects the existence of a legal completion, including all-zero partial prefixes. Canonical strict-prefix controls and complete-policy filesystem laws retain earlier promises. There is no random seed: replay inputs are field offset, generated value and prefix length; the checked-in tests and fixture are the permanent corpus, and ascending enumeration provides the first failing boundary. + +The domain now owns independent resource admission shared with complete `RetentionClosureLimits` construction. The boundary computes a partial big-endian field's smallest completion by zero-filling only for that calculation. It reports a new `ClosureLimitPrefixAboveMaximum` diagnostic when this minimum exceeds the domain ceiling; complete fields retain the existing `ClosureLimit` cause. No missing byte is presented as an observation, no candidate policy is synthesized, and no wire-format or public constructor behavior changes. The public decode-error enum gains the named variant, which exhaustive consumers must account for. + +Replay uses `cargo test --lib closure_prefix --all-features`, `cargo test --lib short_policy --all-features`, `cargo test --test retention_stage_prefix --all-features` and `cargo test --test retention_root_encoding --all-features`, with `--release` for optimized execution, in copied Docker source. The environment is Linux aarch64, Rust 1.96.0 and an owned ext4 sandbox; the process-death replay is `cargo xtask durability-crash-matrix --sequence retention`. + +The fault model is interrupted limit bytes and real filesystem recovery, not physical power loss or concurrent namespace replacement. Other partial header fields, body admission, integrity prefixes, incomplete-stage pinning and post-removal failure semantics remain open. Per-test resource ceilings and suite latency admission remain unapproved enforcement gaps. No performance claim is made. + +Deletion requires stronger public recovery evidence that subsumes both impossible-prefix refusal/preservation and valid-completion controls; no existing behavioral expectation was weakened. + +Four disposable product mutations use copied trees and fresh targets: subtracting one from the reported minimum fails the exact-cause assertion; reporting `Head` fails the stage assertion; deleting `root.next` on refusal fails retained-evidence equality; and rejecting zero-valued partial prefixes fails the generated valid-completion law. All reach the intended assertions rather than failing setup or compilation. + +The focused runtime laws, canonical-prefix controls and root-encoding contract suite pass in debug and release. The retention process-death matrix, both workspace Clippy configurations with `-D warnings`, formatting and source-structure checks pass. diff --git a/docs/testing-evidence/retention-complete-body-prefix.md b/docs/testing-evidence/retention-complete-body-prefix.md new file mode 100644 index 00000000..e95ff659 --- /dev/null +++ b/docs/testing-evidence/retention-complete-body-prefix.md @@ -0,0 +1,21 @@ +# Complete entries in interrupted retention bodies + +Change kind: bug fix. Owner: `@flyingrobots`. Subject: Keep runtime refusal and preservation when an interrupted root or manifest already contains invalid complete body entries (#99). + +Regression `10ebd6f` was observed RED on unfixed production `1c46c99`: a malformed blob identity and duplicate anchors in an interrupted root body were discarded with `Clean`. + +The medium filesystem laws also exercise malformed layout identity, manifest generation zero and duplicate manifest namespaces. Fixtures declare one further unavailable entry with self-consistent framing so body completion and its set-digest check cannot mask the semantic finding. Each law requires the precise corrupt stage, entry index and typed admission failure, then compares retained paths and bytes across real recovery. + +The specified oracle is canonical identity framing, positive root generations and strictly ordered unique anchors/namespaces. Expected magic bytes, indices and zero-generation errors are stated independently of production parsing; the checked-in conformance records supply otherwise valid context. Inputs are deterministic minimal malformed/duplicate entries, identified by stage and fault; no random seed or reducer is needed. + +Full decoding and interrupted-body admission share entry parsing and ordering logic. The prefix path walks only available complete chunks and retains only the preceding semantic coordinate; it neither allocates the declared unavailable suffix nor treats an incomplete entry as admitted. Existing integrity-before-semantics ordering at complete-record decoding is retained. + +Replay uses `cargo test --lib body_prefix --all-features` and `cargo test --test retention_stage_prefix --test retention_root_decoding --test retention_manifest_codec --all-features`, adding `--release` for optimized execution. Runs use copied Docker source, Rust 1.96.0, Linux aarch64 and an owned ext4 sandbox. The retention process-death replay is `cargo xtask durability-crash-matrix --sequence retention`. + +The failure model covers complete bad entries inside incomplete bodies, not every incomplete entry field, physical power loss or out-of-band replacement. Partial header and partial-entry constraints, incomplete-stage pinning and post-removal failures remain open. Per-test resource ceilings and suite latency admission remain unapproved enforcement gaps. There is no wire/API change or performance claim. + +Delete these laws only when stronger public recovery evidence subsumes malformed identity, generation and ordering refusals with the same evidence-preservation guarantee; no existing expectation was weakened. + +Disposable copied product mutations with fresh targets alter the reported entry index, name the wrong corrupt stage, and delete the root stage while returning the correct refusal. Both filesystem laws fail at the corresponding exact-cause, stage and retained-evidence assertions. + +Debug/release filesystem laws, canonical-prefix controls and complete root/manifest codec suites pass, as does the retention process-death matrix. An initial Clippy semicolon finding in the fixture was corrected; the focused runtime laws and both workspace Clippy configurations with `-D warnings`, formatting and source-structure checks then pass. The original static failure remains recorded. diff --git a/docs/testing-evidence/retention-corrupt-stage-refusal.md b/docs/testing-evidence/retention-corrupt-stage-refusal.md new file mode 100644 index 00000000..207d9128 --- /dev/null +++ b/docs/testing-evidence/retention-corrupt-stage-refusal.md @@ -0,0 +1,27 @@ +# Filesystem corruption refusal + +Change kind: test coverage repair. Owner: `@flyingrobots`. Subject: refusal and evidence preservation when a complete retained root stage has a damaged checksum (#99). + +At parent `c3fca1c`, short corrupt prefixes were covered through filesystem recovery, but the complete checksum-corrupt root stage had no corresponding filesystem recovery law. + +The new law writes the golden root as `root.next` with one checksum bit flipped, then invokes the public filesystem authority's recovery method. + +Its specified oracle requires `FilesystemRetentionRecoveryError::Plan` containing `StageCorrupt` for the root stage, with the exact `RetentionRootDecodeError::ChecksumMismatch` expected and observed checksum bytes taken from the original and damaged fixtures. + +The complete retained-file witness must remain byte-identical after refusal, including the damaged stage. + +No assertion reaches into observation or classification internals: the typed public refusal and retained bytes establish the behavior through the real filesystem observation, decoding, planning, and refusal path. + +Production already satisfies this behavior; the change adds missing runtime evidence rather than claiming a production bug fix or a failing parent reproduction. + +Disposable production mutations independently mislabel the corrupt root as a head, swap the checksum coordinates, and delete `root.next` on the planning-error path. + +The diagnostic assertion rejects both incorrect error mutations; the retained-byte assertion rejects deletion while the correct refusal is still returned. + +Each mutation runs in a separate copied Docker tree with a fresh build target. + +The filesystem recovery suite passes in copied Docker debug and release using `cargo test --lib filesystem_retention_recovery_tests --all-features`, adding `--release` for optimized execution. + +Both workspace Clippy configurations with `-D warnings`, formatting, and source-structure checks pass. + +The calibration replay filter is `checksum_corrupt_root_stage_is_refused_without_changing_evidence`; this law covers complete root-stage checksum damage, not every corruption class or mid-recovery filesystem fault. diff --git a/docs/testing-evidence/retention-crash-reader-oracle.md b/docs/testing-evidence/retention-crash-reader-oracle.md new file mode 100644 index 00000000..604b34e4 --- /dev/null +++ b/docs/testing-evidence/retention-crash-reader-oracle.md @@ -0,0 +1,29 @@ +# Reader-visible state after process death + +Change kind: test-oracle repair. Owner: `@flyingrobots`. Subject: retention generation and selected-root bytes observed after a real publication child is killed (#99). + +At parent `276f0e7`, the retention crash runner checked the recovery receipt and forward retry, without independently reading the recovered snapshot. + +The restart verifier now loads `FilesystemRetentionSnapshot` after recovery and before retry, with a bounded catalog read policy. + +The crash coordinate determines the expected state: before a complete head-stage write, no retention head or selected root may be visible; from completion of that write onward, recovery must expose generation one and exactly the golden publication input's root bytes. + +The expected state does not come from the recovery receipt; the root decoder supplies only the golden input's namespace for the public read. + +The reader snapshot is dropped before forward retry, so the new verification does not retain its fence across publication. + +Calibration substitutes `wrong` for the public reader's verified returned root bytes and runs `cargo xtask durability-crash-matrix --case KEEP-CRASH-046 after` in a disposable copied Docker tree. + +With that same production mutation, the parent verifier passes and the revised verifier fails at `verify recovered selected root bytes`, reporting expected and observed payloads. + +A separate production mutation makes `retention_head()` return `None`; the revised verifier fails at `verify recovered retention generation`, reporting `expected Some(1), observed None` at the same crash coordinate. + +Each calibration uses a fresh build target, and neither mutation enters the committed product code. + +The complete retention process-death sequence passes in copied Docker with `cargo xtask durability-crash-matrix --sequence retention` and `cargo run --quiet --locked --release --package xtask -- durability-crash-matrix --sequence retention`. + +Both workspace Clippy configurations with `-D warnings`, formatting, and the source-structure check pass. + +This is product runtime evidence delivered by repository tooling: the assertions read Keep's public outputs after killing and reaping a real child, rather than asserting harness case counts or implementation structure. + +The crash coordinate is the deterministic replay artifact; the sequence covers the runner's declared before/during/after positions for initial retention publication, not arbitrary schedules, successor publication, stacked faults, or power-loss behavior. diff --git a/docs/testing-evidence/retention-first-step-refusal.md b/docs/testing-evidence/retention-first-step-refusal.md new file mode 100644 index 00000000..60cbae9e --- /dev/null +++ b/docs/testing-evidence/retention-first-step-refusal.md @@ -0,0 +1,23 @@ +# First-step recovery refusal + +Change kind: test coverage repair. Owner: `@flyingrobots`. Subject: the public recovery executor's first-refusal contract (#99). + +At parent `9712be6`, the refusal law exercised a failure after successful finalization, leaving refusal of finalization itself uncovered. + +The added law passes a finalization plan to `execute_retention_recovery` and injects refusal at the public storage port's first capability. + +The returned error must name `FinalizeHead`, report an empty completed prefix, and leave later cleanup capabilities unexecuted. + +The recording adapter observes successful effects across the storage port; this is a test of the product executor's contract, not a test of fixture cardinality or an internal call sequence. + +Three disposable production mutations independently misname the refused step, include that refused step in the completed prefix, and invoke root-stage cleanup after refusal. + +Each mutation fails its corresponding named assertion with a fresh build target and the same unchanged adapter. + +The executor suite passes in copied Docker debug and release with `cargo test --lib recovery_execution_tests --all-features`, adding `--release` for optimized execution. + +Both workspace Clippy configurations with `-D warnings`, formatting, and source-structure checks pass. + +Production already satisfies the contract; no failing-parent bug reproduction is claimed. + +This small test establishes port-level execution behavior only; the fake does not model filesystem durability or partial effects within a failed storage capability, which remain the responsibility of filesystem and fault-injection evidence. diff --git a/docs/testing-evidence/retention-head-length-domain.md b/docs/testing-evidence/retention-head-length-domain.md new file mode 100644 index 00000000..3690d5c0 --- /dev/null +++ b/docs/testing-evidence/retention-head-length-domain.md @@ -0,0 +1,21 @@ +# Canonical head-length inputs + +Change kind: test-maintenance refactoring. Owner: `@flyingrobots`. Subject: generated inputs for recovery's exact head-to-manifest length contract (#99). + +At parent `5cabf82`, the finite length sweep repeated the manifest entry width as a literal `72`. + +The sweep now derives its minimum and positive entry stride from the public encoder's empty and single-entry manifest outputs, then generates lengths through the manifest's admitted entry-count bound. + +Only input construction changes: the specified oracle still requires recovery precisely when the head length equals the frozen manifest fixture's actual byte length, and the exact `HeadStageNamesOtherManifest` refusal otherwise. + +The encoder supplies inputs, not expected outcomes, so no private codec constant is exposed for testing and the planner cannot validate its own answer through the generator. + +This relies on the version-two protocol's fixed-width entry representation; it is not a generator for arbitrary future variable-width formats. + +Removing the production planner's head-length comparison in a disposable copied tree makes the revised law fail at the first mismatched length: head length 224 must refuse the 296-byte fixture. + +The unchanged matching-length assertion passes on production, while the mismatched-length assertion rejects that mutation; no harness count assertion is introduced. + +The focused sweep passes in Docker debug and release, with both workspace Clippy configurations, formatting, and source-structure checks. + +Replay uses `cargo test --lib recovery_binds_head_length --all-features`, adding `--release` for optimized execution; calibration uses a fresh build target. diff --git a/docs/testing-evidence/retention-head-without-manifest-refusal.md b/docs/testing-evidence/retention-head-without-manifest-refusal.md new file mode 100644 index 00000000..82b98710 --- /dev/null +++ b/docs/testing-evidence/retention-head-without-manifest-refusal.md @@ -0,0 +1,15 @@ +# Complete head without manifest refusal + +Change kind: test-oracle repair. Owner: `@flyingrobots`. Subject: the precise refusal exposed by publication when restart finds a complete head stage without its manifest stage. + +At parent `88eaa98`, the direct storage-attempt law and the public publication-executor law accepted any `RecoveryRefused` cause despite constructing the specific missing-manifest state. + +Both laws now require `RetentionRecoveryRefusal::HeadStageWithoutManifestStage` inside the existing recovery wrapper. Their real migrated-store setup and existing no-publication or retained-evidence assertions remain unchanged. + +Calibration changes the actual recovery planner to return `HeadStageNamesOtherManifest` at that branch. The observation distinguishes an absent manifest from an existing but disagreeing manifest; the wrong variant must not satisfy either public boundary's diagnostic contract. + +Both parent tests passed that wrong production variant, while both revised assertions failed at their specific missing-manifest checks. The final diagnostic-message additions received a further focused debug run and all-feature Clippy check. + +The attempt and publication-storage suites pass in copied Docker debug and release, with both workspace Clippy feature configurations, formatting and source-structure checks. Replay uses `cargo test --lib refused_verification_admits_no_later_phase --all-features` and `cargo test --lib retained_stage_refuses_publication_before_recovery --all-features`, with `--release` for optimized execution. + +The first calibration copy exhausted the disposable filesystem's inode capacity and was excluded as a setup failure. Completed calibration trees were preserved outside that filesystem, then a complete copy and fresh build targets were used for the runtime calibration. This change claims neither new crash-schedule coverage nor a production behavior change. diff --git a/docs/testing-evidence/retention-landing.md b/docs/testing-evidence/retention-landing.md new file mode 100644 index 00000000..53451409 --- /dev/null +++ b/docs/testing-evidence/retention-landing.md @@ -0,0 +1,93 @@ +# Retention recovery landing ledger + +Owner: `@flyingrobots`. Change kinds: approved behavior change for incomplete stages, bug fixes for source binding and execution-failure reporting, and documentation reconciliation. This is the single closure ledger and evidence receipt for the bounded #99 landing. + +## Baseline and approved contract + +Baseline inspection found clean local `51bae7f0bdabaeec0f9700358649bc2657b00aa1` and pushed `821c60d0ba48ae4963cdeb8719a2598aafcc0249`. Local `aa6999c` and `51bae7f` preserve the unfinished framing correction; they remain in history. Target main was `379b24141bbcc4b883bd169e746ed76c77f7699c`. + +Maintainer decision A supersedes automatic incomplete-stage disposal for this landing: incomplete stages require explicit disposition before any recovery mutation, while demonstrated corruption keeps its precise refusal. Finding no contradiction does not prove canonical completion. Automatic disposition and stronger completion feasibility are deferred to a focused follow-up; no new namespace, journal or wire format is introduced. + +Decision B supports cooperating writers under Keep authority in a managed namespace. Concurrent raw namespace mutation is outside writer-lock isolation. Existing no-follow, identity, byte, namespace and corruption checks remain required; neither a handle nor a metadata check makes pathname mutation conditional on inode identity. + +Decision C distinguishes pre-effect refusal from execution failure. Planning refusal initiates no recovery mutation. Execution errors preserve their cause and report the failed boundary, completed earlier steps, known effects and uncertain effects/durability of the failing capability. Failure is not rollback; retry requires fresh observation. + +## Review reconciliation + +The baseline inventory includes all 40 inline threads with their complete comments, 73 review records and 133 top-level comments, verified against paginated API results. These are discovery coordinates, not runtime assertions. Inline IDs below omit the common `PRRT_kwDOTdinZc6` prefix. The independent review at [5946292798](https://github.com/flyingrobots/keep/pull/99#issuecomment-5946292798), its primary reconciliation at [5946293023](https://github.com/flyingrobots/keep/pull/99#issuecomment-5946293023), review-body outside-diff findings and later activity/self-findings are included. Resolution flags alone do not close an obligation. + +| Obligation / sources | Invariant and baseline evidence | Disposition / remaining implementation | Concrete exit | +| --- | --- | --- | --- | +| Short-stage destruction: g09p5; independent review; partial-prefix activity comments | Existing corruption checks preserve many contradictory prefixes; planner still schedules automatic discard | IMPLEMENTED under decision A; incomplete stages refuse and lower filesystem discard capabilities are disabled; diagnostics retained | Direct/publication recovery preserves all stages, including maximum-namespace future-entry counterexample; no discard path executes | +| Incomplete pinning / earlier-prefix discard: oR9Xz, g09qR | Discard stores metadata without a retained handle | APPROVED DEFERRAL of disposal, not a completed pinning implementation; no automatic deletion under A/B | Normative contract and requirements mark disposal deferred; retained-stage refusal tests pass | +| Source-stage removal binding: maintainer inspection | `FilesystemRetentionStage::remove` verifies destination but not source immediately before unlink | IMPLEMENTED bug fix: retain handle, preserve observation identity across reopening, verify source and pool evidence | Replaced source after observation refuses before unlink with exact identity cause and unchanged evidence; RED on baseline | +| Execution partial effects: maintainer C; prior global recovery gap reports | `RetentionRecoveryError` records previous completed steps only; capabilities can fail after link/rename/unlink | IMPLEMENTED bug fix: finite capability inventory below, typed failing-capability status and fresh observation; stable validation passed | Pre-effect, post-effect, sync failure and restart laws pass with exact causes and honest effects | +| Namespace and directory admission: g09p8, g09qo; independent review | Recovery starts with pinned-directory and recovery-census admission | IMPLEMENTED, retain; final stable runtime suite verifies both direct and publication paths | Namespace/directory refusal laws pass unchanged | +| Pool inode binding: g09qD; independent review | Observation binds exact pool bytes and identity to stage | IMPLEMENTED, retain | Byte-equal pool substitution refuses before head finalization | +| Recovery file synchronization: g09qI; independent review | Recovered stages synchronize before publication | IMPLEMENTED, retain; process/syscall evidence is not power-loss proof | Existing syscall-order and restart tests pass | +| Head coordinates / root history / successor entries: g09qX, g09qc; global 5950156938 | Planner compares head coordinates, successor root generation and unrelated manifest entries | IMPLEMENTED, retain | Exact history/binding/entry-set refusals and legitimate committed cleanup pass | +| Predecessor and live closure: g09qi, g09qv; global 5951046189 | Selected roots reopen; forward and recovery paths share bounded live closure admission | IMPLEMENTED, retain | Missing/corrupt predecessor, catalog and segment laws pass | +| Typed planner/storage/observation failures: g0-Vj, g0-Vr, g0-VQ, g0-VL, g0-VH | Dedicated storage causes, missing-root diagnostic and filesystem corruption laws exist | IMPLEMENTED, extend only for C | Original error variants/sources and corruption preservation remain observable | +| Reader capability, binding, coordinates and errors: g2O5E/g2pTv, g2pT1, g2O5N/g2pT6, g2pT_, g2O4x | Reader uses pinned root, migration binding, catalog-length coordinates, typed catalog error; fence is crate-private | IMPLEMENTED; duplicate sources grouped | Stable reader suite and public API checks pass; no new isolation claim | +| Reader/model test oracles: g2O40, g2O42, g2O5B, g2O5Q, g2pUF | Exact checksum/errno/identity and collector/model refusal laws implemented | IMPLEMENTED, retain | Existing boundary tests pass | +| Fixture, codec bounds, constructor and migration error consistency: g0-U3, g0-VB, g0-VW, g2O4u, g2O5S, oTAeP | Shared admission/phase vocabulary, codec bounds and typed migration errors exist | IMPLEMENTED, retain | Stable workspace checks pass | +| Executor first failure: g0-Vc | Existing test covers empty prior completed-step prefix | IMPLEMENTED, clarify under C: empty prefix says nothing about failing capability effects | New partial-effect law distinguishes those facts and verifies stop-on-error | +| Recovery process-death state oracle and successor claims: g2O5Y, g0-Ut | Reader checks recovered generation/root; initial and successor publication-phase coverage expanded | IMPLEMENTED; update partial-write expectations under A | Complete-stage restart success retained; partial writes refuse without deletion | +| Merge integration: independent review; globals 5946442838, 5946468733, 5947325527 | Main merged without rewriting history; mount stability, migration crash coordinates and Clippy duplicate arms corrected | IMPLEMENTED; previous case-count assertion is not durability evidence | Stable candidate builds, merge invariants and required crash suites pass | +| Documentation and outside-diff comments: g0-Uw, g2O4s, oR9X3; review-body lib.rs/recovery.md/crash CLI | Status, count and boundary claims previously corrected; A/C now supersede discard promises | IMPLEMENTED reconciliation across normative docs, requirements, API, PR description and superseded historical receipts | No current claim of automatic disposition, rollback or raw-mutation isolation; historical evidence clearly scoped | +| Current-head acceptance | Earlier agy review requested changes; earlier CI applies only to earlier heads | OPEN: exact-head agy under A/B/C, complete checklist, required checks | Exact pushed SHA approved and green; human merge approval remains required | + +Independent-review duplicate findings map to the rows above. Its requested rebase and claimed 300-line hard limit were rejected in the original reconciliation: preserve history with normal merges; 300 lines is a review threshold, 500 the hard file limit. No repeated hardening pass is authorized after the landing exits pass. + +The separate catalog publisher admission issue #150 remains tracked outside this recovery landing; its public unchecked-constructor scope is not discharged by these tests. It is not a new recovery-disposition requirement. + +## Landing evidence + +The focused automatic-disposition follow-up is [#155](https://github.com/flyingrobots/keep/issues/155). It is not a prerequisite for this landing. Historical per-field receipts remain preserved as evidence of their named assertions at their recorded commits; they do not prove current automatic-discard safety, exhaustive prefix completion, or the approved landing complete. + +Regression commit `47013bf` was observed RED with production `51bae7f`: cleanup deleted a byte-identical substituted source; incomplete initial and maximum-namespace manifests were discarded. The incomplete outcomes are an authorized behavior change, while source identity is a reproduced bug. The filesystem suite now preserves incomplete stages in direct and publication-triggered recovery, retains known corruption diagnostics, and detects source substitution before cleanup. Existing complete-stage recovery laws and the retention process-death matrix pass in copied Docker execution. + +Prior success expectations deliberately changed: isolated truncated-root planning/recovery, truncated-head planning with earlier linked evidence, publication retry after an incomplete root, and each interrupted stage-write case. Missing-earlier-evidence scenarios now require incomplete disposition without asserting that the unavailable full record is corrupt. Complete publication-prefix success expectations remain unchanged. Direct calls to reserved filesystem discard capabilities also refuse without mutation. + +Decision C regression uses a real filesystem unlink followed by a deterministic injected `EIO` before directory synchronization. The baseline reports the failing step but lacks its removed-stage effect and uncertain durability; the regression fails with `missing failing-capability effects after unlink`. This is filesystem execution with an injected failure, not physical power-loss evidence. The effect-reporting implementation passes focused filesystem and port-level tests. Distinct-assertion calibration and stable validation are recorded below; exact-head independent acceptance remains an external gate. + +## Finite capability and effect inventory + +The inventory covers `filesystem_retention_recovery.rs`, its shared `filesystem_retention_stage.rs` operations, `recovery_execution.rs`, and their error types. All complete stages retain their opened source through the operation and bind reopening to observation identity. Source verification checks handle and pathname identity and exact bytes; it is not an inode-conditional namespace syscall. The supported concurrency contract is decision B. + +| Capability | Before first namespace effect | First effect and subsequent fallible boundaries | Failure report / retained evidence | +| --- | --- | --- | --- | +| Reserved incomplete-stage discard capabilities | Unconditional typed disposition refusal | None | No namespace effects; no incomplete stage is removed, even through a direct capability call | +| `link_root` | Context admission, source verification, file synchronization, source recheck | Namespace creation if absent; no-follow namespace open; roots-directory sync; hard link; source/pool verification; pool-directory sync | Known created namespace with observed sync status; known created pool link only after successful link; failed creation/link attempts remain uncertain; source retained | +| `link_manifest` | Context admission, source verification, file synchronization, source recheck | Hard link; source/pool verification; pool-directory sync | Successful new link is known with unconfirmed durability until directory sync; failed link attempt is uncertain; existing matching link is not reported as newly created; source retained | +| `finalize_head` | Context admission, source verification, file synchronization, source recheck | Rename onto `HEAD`; source-absence and head verification; retention-directory sync | Successful rename is known even if verification/sync fails; failed rename attempt is uncertain; opened source retained through operation, pathname intentionally moved | +| `remove_root_stage` | Context admission, no-follow namespace open, source and pool verification | Unlink; source-absence and pool verification; retention-directory sync | Successful removal is known even if verification/sync fails; failed unlink attempt is uncertain; verified pool evidence survives under decision B; stage pathname intentionally removed | +| `remove_manifest_stage` | Context admission, source and pool verification | Unlink; source-absence and pool verification; retention-directory sync | Same cleanup guarantee as root removal | + +A pre-effect guard reports its exact boundary with no known or uncertain namespace effects. Each failed namespace syscall reports an uncertain attempted effect conservatively; it is not promoted to a successful effect. Successful calls followed by failure report known effects separately from durability. `executed()` contains only preceding successful capabilities; `progress()` reports the failing capability. `None` means the adapter did not report effects. The executor stops on the first error, and filesystem recovery clears its context so retry starts with new observation. + +Source identity across reopening is a verified extension of the same source-binding obligation: regression `substitution_before_reopening_refuses_without_rebinding_evidence` is RED on `dd79a0423d35536e5ee1a1b5f6494e2e25e726b3`, with “reopening silently rebound observed stage identity”. The fix carries the original observed identity into the reopened stage. It does not promise protection from the final check-to-pathname-syscall race outside decision B. + +Filesystem failure coverage injects `EIO` immediately before directory synchronization, after real namespace creation, pool link, head rename or stage unlink. Tests assert the original errno, exact failed boundary, known effects and durability, retained path/byte evidence, stop-on-error, then drop writer authority, reopen and recover. This is actual filesystem execution plus deterministic injected failure, not a physical crash or power-loss model. A separate port-level test verifies uncertain-effect propagation and prevents later cleanup; it does not claim syscall evidence. Existing production process-death campaigns supply their separately scoped restart evidence. + +Stable validation exposed a static documentation guard requiring the retired phrase `pre-effect incomplete stage`. That assertion was removed because decision A replaced its automatic-discard requirement. Deletion criterion: the guarded policy was superseded; displaced runtime risk is covered by incomplete-stage preservation and precise-corruption laws. Other migration documentation guards remain unchanged. This is a static-policy correction, not runtime bug evidence. + +One bounded calibration pass on production `eb1017b175a225259595fd75b08a21aca0fae594` exercised distinct assertions: deletion despite a disposition refusal, corruption-priority loss, omitted known effect, false synchronized durability, wrong failing boundary, wrong errno, erased uncertain effect, and execution of later cleanup after failure. Each executable mutation failed its intended runtime assertion. An initial omitted-effect mutant failed compilation due to an unused import; it was corrected and rerun, and that compiler failure is not calibration evidence. Source-binding regressions already supply observed RED evidence for the omitted guards. No mutation score or field-by-field campaign is claimed. + +## Stable candidate validation and acceptance boundary + +Production and tests at `7f0b9e55eb1732769cbb35ca279b3fef77e3534b` passed copied-container validation on Linux aarch64 with pinned Rust 1.96.0, a private ext4 test filesystem, `strace` 6.13 and independent `b3sum` 1.8.5. The final ledger-only commit does not change those sources. `clean-validation.log` is the acceptance run, using a dedicated initially empty Cargo target. An earlier shared-target run accidentally reused the final mutation binary; its failure paths identify the mutant checkout and it is excluded from acceptance. Historical logs are preserved, not relabeled green. + +| Validation | Result and scope | +| --- | --- | +| Golden File Worldline and protocol conformance | Passed; independent vector/corpus verification | +| Full process-death crash matrix, debug and optimized | Passed; production subprocess restart evidence, not physical power-loss simulation | +| Source structure and `cargo fmt --all --check` | Passed; static tooling evidence | +| Workspace checks and Clippy `-D warnings`, all/minimal features | Passed with the locked pinned toolchain | +| Workspace tests, debug and release | Passed; includes retention, reader/model, migration integration and tool-contract suites; tooling tests are not runtime storage evidence | +| Doctests, documentation build and explicit MSRV check | Passed | +| Fuzz manifest formatting, target checks and Clippy | Passed; target build evidence, not runtime fuzzing | +| Markdown lint 0.23.2 | Passed in a separate copied Node container | +| Complete documentation/workflow integrity, runtime fuzz smoke, dependency policy/audits | Exact-head hosted gates required; the local documentation-refusal command stopped because pinned external documentation tools were unavailable in the ARM Rust container | + +The implementation exits in the review table are closed under decisions A/B/C by the source, focused RED/GREEN evidence, calibrated assertions and stable validation above. Automatic incomplete-stage disposition remains an approved deferral in #155, not an implemented requirement. The remaining acceptance obligations are exact-head independent approval and green hosted checks. Their final SHA, reviewer identity, complete checklist and check URLs are recorded in PR #99 discussion against the reviewed commit, avoiding a self-referential receipt commit that would invalidate that exact-head approval. A READY FOR MERGE report requires both records; human merge approval is still separate. diff --git a/docs/testing-evidence/retention-live-closure-admission.md b/docs/testing-evidence/retention-live-closure-admission.md new file mode 100644 index 00000000..c0dfe072 --- /dev/null +++ b/docs/testing-evidence/retention-live-closure-admission.md @@ -0,0 +1,19 @@ +# Retention live closure admission evidence + +Change kind: bug fix plus an additive explicit recovery loading-policy API. Subject: Keep runtime retention publication and recovery. Size: medium. Oracle: catalog-selected physical bytes and every root anchor must authenticate live under authority before retention publication or cleanup; refusal preserves retained bytes and exact typed sources. + +Commit `6d1ed71` records the staged-recovery regression laws on unfixed recovery. Missing or corrupted catalog and segment artifacts after root staging, manifest staging, or complete head staging caused the old implementation to link artifacts or commit the head and remove stages. The retained-byte assertions were observed RED at the intended runtime boundary. The first catalog-corruption fixture flipped the terminal digest while expecting a checksum diagnostic; the corrected fixture flips the checksum field at `length - 64` and was rerun RED against an archived `6d1ed71` source tree with the revised test file. Expectations were not weakened to absorb a decoder refusal. + +Additional semantic laws replace the retained stages with canonical, mutually consistent root, manifest, and head records naming an absent anchor layout or a physical closure budget of one byte. Those laws were observed RED on the unfixed `6d1ed71` implementation with only the added test module and declaration overlaid: recovery committed the head and removed stages. The fixed code refuses with exact `MissingMember` identity or `LimitExceeded` physical-counter evidence before effects. + +Forward publication regressions remove or corrupt a selected segment after pure preflight, then invoke the actual publication executor without retained stages. Before the live forward hook, the unchanged catalog-binding path from `6d1ed71` returned `Published` and changed retention bytes despite lost reconstructability. Both laws were observed RED before the forward hook was added. They now require a current-verification refusal with the exact catalog restart phase or segment seal checksum source and unchanged retained bytes. + +Loading-policy boundary laws derive selected segment length from their controlled filesystem fixture. A budget one byte below the observed selection refuses with exact maximum and observed values and leaves retained bytes unchanged. A budget exactly equal to the selection admits recovery and returns the protected-root outcome. These are runtime policy assertions, not tests of case enumeration or source spelling. + +Calibration runs use independent copied source directories and fresh Cargo target directories. Stringifying catalog errors fails the precise recovery and forward source assertions and the policy counter assertion while preserving bytes. Replacing closure errors with a wrong typed counter-overflow error fails both semantic closure oracles while preserving bytes. Ignoring caller policy fails the budget refusal's retained-byte assertion. Returning `Clean` after a valid root link fails the exact-size policy's protected-root outcome assertion. Expected mutation failures are evidence, not gating green runs. + +Final implementation validation ran in a fresh copied Git checkout based on main integration `379b241`, overlaid with the complete source inventory including legitimate nested `target` source modules and new modules. Full all-feature workspace debug and release suites, formatting, Clippy for all and minimal features with warnings denied, and source-structure admission passed on pinned Rust 1.96.0 in Linux aarch64 Docker isolation with the existing ext4 audit filesystem. A fixture-constant wrapping correction preceded the final formatting run; its setup refusal is not RED evidence. Raw RED, mutation, environment, and full validation logs are retained outside tracked source. + +The default retained-segment loading cap is the existing protocol maximum segment length, 1,073,741,824 bytes (1 GiB), verified from `segment_header::MAXIMUM_SEGMENT_LENGTH` and the default policy constructor. It is independent of the root's selected-record closure counters. Explicit caller-policy boundary experiments establish the budget mechanism on the fixture; they do not measure total RSS or execute a GiB-sized successful load. Full catalog and segment materialization adds I/O and replay work; no performance improvement or measured optimal default is claimed. + +Remaining blind spots include physical power loss, every corruption position or closure profile, all catalog sizes, concurrent external replacement after observation, atomic truncated-stage unlink, per-test resource ceilings, and the other open PR findings. Codec encodings, hashes, and dependencies are unchanged. Markdown checks establish documentation policy rather than Keep runtime behavior. diff --git a/docs/testing-evidence/retention-manifest-read-bound.md b/docs/testing-evidence/retention-manifest-read-bound.md new file mode 100644 index 00000000..81869df1 --- /dev/null +++ b/docs/testing-evidence/retention-manifest-read-bound.md @@ -0,0 +1,13 @@ +# Retention manifest read-bound ownership + +Change kind: refactoring. Subject: static ownership of the recovery read bound, with no intended runtime change. + +At parent `be71fa5`, recovery independently fixed the manifest bound at 295,136 bytes while the decoder computed canonical length from a 160-byte header, 72-byte entries and a 64-byte trailer, with the semantic limit at 4,096 entries. + +Recovery now calls the decoder's checked canonical-length calculation at that semantic limit: 160 + 72 × 4,096 + 64 = 295,136, preserving the existing bound exactly and retaining the existing one-extra-byte oversize classification. + +A checked calculation is used instead of an independently evaluated constant so conversion to the platform's `usize` and arithmetic overflow follow the decoder's existing typed error path; recovery preserves that source in its I/O boundary. + +This before/after evidence establishes ownership and numeric equivalence, not new runtime fault coverage; no assertion on a constant or source text is added to the product test suite. + +Existing manifest and recovery behavior tests provide regression checks; their expectations are unchanged. No durable format, parser acceptance rule, or publication order changes. diff --git a/docs/testing-evidence/retention-missing-root-refusal.md b/docs/testing-evidence/retention-missing-root-refusal.md new file mode 100644 index 00000000..8d090d95 --- /dev/null +++ b/docs/testing-evidence/retention-missing-root-refusal.md @@ -0,0 +1,19 @@ +# Missing root stage diagnostic + +Change kind: bug fix. Owner: `@flyingrobots`. Subject: exact filesystem recovery refusal when complete head and manifest stages remain without their root stage (#99). + +The regression in `5eb3428` runs against unfixed production code from `6661736`: it creates a real migrated store, drives publication through head-stage synchronization, removes only `root.next`, then calls public filesystem recovery. + +The expected `ManifestStageWithoutRootStage` assertion fails RED with the observed `Plan { source: HeadStageWithoutManifestStage }`, proving that recovery named an artifact that was present. + +The planner now distinguishes a missing manifest from a complete manifest with a missing root, using the existing typed refusal variants. + +The regression also compares the complete retained-file witness before and after refusal, so a precise diagnostic cannot hide deletion or modification of evidence. + +A disposable production mutation deletes `head.next` on the planning-error path while returning the corrected refusal; the preservation assertion fails at `missing-root refusal must preserve all retained bytes`. + +The recovery-filtered library suites pass in copied Docker debug and release with `cargo test --lib recovery --all-features`, adding `--release` for optimized execution. + +Both workspace Clippy configurations with `-D warnings`, formatting, and source-structure checks pass. + +This changes the diagnostic variant for one ambiguous state; it does not permit publication, change durable encoding, or weaken recovery's refusal-before-effects contract. diff --git a/docs/testing-evidence/retention-model-refusals.md b/docs/testing-evidence/retention-model-refusals.md new file mode 100644 index 00000000..00dee17f --- /dev/null +++ b/docs/testing-evidence/retention-model-refusals.md @@ -0,0 +1,21 @@ +# Exact retention model refusals + +Change kind: test-oracle bug fix. Owner: `@flyingrobots`. Subject: the runtime diagnostics returned by rejected retention operations, alongside the existing persisted-state model. + +At base `5242b05`, the model accepted any error for a refused operation, including unrelated preparation failures and operating-system errors; unchanged stored state could therefore conceal an incorrect diagnostic. + +The model now distinguishes repeated initial roots from stale publications: the former must return `ManifestSuccessorMismatch` with the selected namespace, root generation and digest, an initial candidate generation and no predecessor; the latter must return `CurrentVerification` containing `Superseded` with the current liveness generation and manifest digest. + +The model determines acceptance from its namespace map and liveness generation; diagnostic coordinates come from the admitted pre-operation manifest, never from the returned error. That coordinate oracle shares the production decoder and therefore does not independently establish codec correctness; independent canonical-format tests remain necessary. + +Calibration changed actual production refusal paths in isolated copied trees. Replacing the preparation mismatch with `ManifestEntryIndex` passed the old model and failed the revised mismatch assertion. Replacing the superseded cause with `PermissionDenied` passed the old model and failed the revised exact-superseding-manifest assertion. These observed runtime RED results establish the oracle repair; the unmutated product passes the tightened checks. + +An additional mutation replaces the publication error wrapper with `MissingPublicationArtifacts` to calibrate the current-verification boundary separately. + +The sequence-count assertion was deleted because it froze harness structure without asserting Keep behavior. No runtime risk moves to another count: the suite retains publication outcomes, exact refusals, manifest entries, liveness generation, root generation and anchor-set checks. + +These are medium filesystem tests with deterministic enumeration of the existing short operation histories. Replay uses `cargo test --lib retention_model_tests --all-features`, adding `--release` for optimized execution, inside the copied Docker runner. They do not claim arbitrary-length history exploration, automatic shrinking, independent decoder validation or newly enforced per-test resource ceilings. + +Debug and release model runs and both workspace Clippy feature configurations passed, with formatting and source-structure checks. Final metadata-only edits received a further focused debug run. Markdown lint validates this receipt separately from product behavior. + +The initial preparation mutation failed compilation because its replacement left unused arguments; that setup result was excluded and the corrected mutant was rerun. An additional wrapper-calibration copy initially exhausted the disposable filesystem's file capacity; completed artifacts were preserved outside it before replay. Raw setup failures, mutant runs and successful unmutated runs remain separate evidence. diff --git a/docs/testing-evidence/retention-observation-binding.md b/docs/testing-evidence/retention-observation-binding.md new file mode 100644 index 00000000..c32a8f4b --- /dev/null +++ b/docs/testing-evidence/retention-observation-binding.md @@ -0,0 +1,13 @@ +# Shared retention observation binding + +Change kind: refactoring with test-fixture admission repair. Owner: `@flyingrobots`. Subject: the validated relationship between an observed retention head and its selected manifest. + +At parent `d118f27`, production observation checked the selected manifest's digest, generation and predecessor after reading exactly the length named by the head, while `ObservedRetentionState::for_tests` decoded its supplied records separately and could construct contradictory state. + +Both paths now call the same private binding validator before constructing `ObservedRetentionState`. The validator requires the named length and the same digest, generation and predecessor checks; the production read retains its existing earlier exact-length check and error precedence. + +The repair is established by static before/after constructor evidence and the single shared validation path. No test of a test fixture or source-text assertion is added. The existing planner scenarios continue to enter through their narrow semantic boundary, and existing filesystem observation tests exercise the real authority boundary. + +Recovery planner and filesystem current-state laws pass unchanged in Docker debug and release, with both workspace Clippy feature configurations, formatting and source-structure checks. A copied-source calibration removes the shared predecessor check and runs the existing real-filesystem predecessor-disagreement law to verify that the production path still depends on that check. + +Replay uses `cargo test --lib recovery_planner_tests --all-features` and `cargo test --lib filesystem_retention_current_tests --all-features`, adding `--release` for optimized runs. This is fixture-integrity and existing runtime regression evidence, not a claim of a newly reproduced production bug, new power-loss coverage or exhaustive input equivalence. diff --git a/docs/testing-evidence/retention-partial-anchor-coordinate.md b/docs/testing-evidence/retention-partial-anchor-coordinate.md new file mode 100644 index 00000000..6aa744e4 --- /dev/null +++ b/docs/testing-evidence/retention-partial-anchor-coordinate.md @@ -0,0 +1,21 @@ +# Partial anchor coordinate admission + +Change kind: bug fix. Owner: `@flyingrobots`. Subject: Keep runtime preservation of short root anchors whose available identity framing or complete layout length already contradicts its embedded codec (#99). + +Regression `8b299da` was observed RED on unfixed production `067966c`: a bad first blob-identity byte and a complete zero layout length inside short anchors were discarded with `Clean`. + +The medium filesystem laws flip each available fixed byte of blob magic, identity version and hash algorithm, and layout magic, identity version and codec, stopping at that byte. They also supply zero, noncongruent and above-maximum complete layout lengths at the earliest available field boundary and near anchor completion. Both first and second anchors are exercised, requiring absolute offending-byte coordinates or exact anchor index and layout-length cause. Every refusal independently checks that all retained paths and bytes remain unchanged. + +The specified/derived oracle is the frozen canonical root's embedded identity bytes and the layout protocol's 176-byte minimum, 46,137,520-byte maximum and 44-byte entry stride. Test expectations use protocol literals rather than invoking production admission. Inputs are deterministic field/anchor positions and malformed lengths; the checked-in laws and fixture preserve the reproducer without a random seed. + +Production checks the final incomplete anchor without allocating an unavailable suffix. Fixed fields refer to the existing boundary codecs' constants, and complete layout lengths reuse their existing validator; only internal visibility expands. Full anchor admission retains its original typed parsing and ordering path. No wire format or public API changes. + +Replay uses `cargo test --lib partial_anchor --all-features`, `cargo test --lib body_prefix --all-features` and `cargo test --test retention_stage_prefix --test retention_root_decoding --all-features`, with `--release` for optimized execution. Runs use copied Docker source, Rust 1.96.0, Linux aarch64 and the owned ext4 sandbox. The retention process-death replay is `cargo xtask durability-crash-matrix --sequence retention`. + +This evidence covers fixed coordinate fields and complete layout lengths, not every partial numeric length, partial-anchor ordering, future-entry feasibility, partial header rule, concurrent replacement or power loss. Incomplete-stage pinning and post-removal failures remain open. Per-test resource ceilings and suite latency admission remain unapproved enforcement gaps. No performance claim is made. + +Delete these laws only when stronger public runtime evidence subsumes their exact coordinate/length refusals and retained-evidence guarantees; no existing expectation was weakened. + +Disposable copied product mutations with fresh targets alter reported byte offsets and anchor indices, name the wrong stage, and remove the root stage on refusal. Both filesystem laws fail at the corresponding exact-cause, stage and retained-evidence assertions. + +Focused and complete-anchor filesystem laws, canonical-prefix controls and complete root-decoding controls pass in debug and release; the retention process-death matrix passes. Initial Clippy findings for an option match and a tuple-shaped fixture return were corrected, followed by passing focused debug/release tests, both workspace Clippy configurations with `-D warnings`, formatting and source-structure checks. Original failed attempts remain recorded. diff --git a/docs/testing-evidence/retention-partial-anchor-order.md b/docs/testing-evidence/retention-partial-anchor-order.md new file mode 100644 index 00000000..959a18f8 --- /dev/null +++ b/docs/testing-evidence/retention-partial-anchor-order.md @@ -0,0 +1,21 @@ +# Partial anchor ordering + +Change kind: bug fix. Owner: `@flyingrobots`. Subject: Keep runtime refusal and preservation when no canonical completion of a partial anchor can follow its predecessor (#99). + +Regression `575e742` was observed RED on unfixed production `aec721f`: a partial blob length already below its predecessor was assessed as `Truncated` and discarded by filesystem recovery. Ordered-prefix controls passed on the unfixed revision. + +Medium filesystem laws cover decisive and near-complete prefixes at each ordering level: blob length, blob digest, layout length and layout digest. They also cover a predecessor at the greatest canonical anchor, where no successor exists. Refusal must identify the root stage and second anchor exactly and preserve every retained path and byte. Small public-assessment sweeps check every decisive malformed prefix and every partial prefix of the reversed, strictly ordered pairs. + +The specified oracle is lexicographic ordering of typed blob length/digest followed by typed layout length/digest. Test pairs differ at one chosen ordering field and use independent protocol limits; reversing each strict descending pair supplies a known valid completion. Inputs are deterministic field choices and prefix lengths, permanently retained in the test; there is no random seed or probabilistic reduction. + +The boundary builds a greatest candidate consistent with observed bytes, using the domain's greatest canonical layout length under a bound to respect both the ceiling and entry alignment. It confirms that canonicalization preserved every available byte, then invokes the existing typed anchor parser and ordering check. The candidate is only a feasibility witness and is never returned as observed state. The process uses a fixed-size buffer and does not allocate the unavailable suffix. No wire or public API change is introduced. + +Replay uses `cargo test --lib anchor_order_prefix --all-features`, `cargo test --lib partial_anchor --all-features` and `cargo test --test retention_stage_prefix --test retention_root_decoding --all-features`, adding `--release` for optimized execution. Runs use copied Docker source, Rust 1.96.0, Linux aarch64 and the owned ext4 sandbox. The retention process-death replay is `cargo xtask durability-crash-matrix --sequence retention`. + +This covers the current partial anchor, not whether all future entries declared by a header remain possible. Partial header constraints, remaining-entry feasibility, incomplete-stage pinning and post-removal failures remain open; physical power loss and out-of-band replacement are outside this evidence. Per-test resource ceilings and suite latency admission remain unapproved enforcement gaps. No performance claim is made. + +Delete these laws only when stronger public runtime evidence subsumes the exact refusal, preservation and valid-completion promises; no existing expectation was weakened. + +Disposable copied product mutations with fresh targets alter the offending anchor index, name the wrong stage, and delete the root on refusal. The runtime laws fail the corresponding assessment/diagnostic, stage and retained-evidence checks. Replacing unknown maxima with zero makes the positive ordered-prefix law fail, calibrating the absence of false refusals. + +The focused ordering and partial-anchor laws, canonical-prefix controls and complete root-decoding controls pass in debug and release. The retention process-death matrix, both workspace Clippy configurations with `-D warnings`, formatting and source-structure checks pass. diff --git a/docs/testing-evidence/retention-partial-head-checksum.md b/docs/testing-evidence/retention-partial-head-checksum.md new file mode 100644 index 00000000..779d44b9 --- /dev/null +++ b/docs/testing-evidence/retention-partial-head-checksum.md @@ -0,0 +1,21 @@ +# Interrupted head checksum admission + +Change kind: bug fix. Owner: `@flyingrobots`. Subject: preserving interrupted heads with already contradictory checksum bytes (#99). + +Regression `9decf08` is observed RED against unfixed production `c9c27b4`: a head ending after its first checksum byte, with that byte flipped, returns a successful `DiscardHeadStage` receipt. + +Recovery now computes the checksum once its entire preimage is available and compares every checksum byte present in the interrupted record. + +The filesystem regression flips the last available byte for each strict checksum prefix ending from 113 through 143 bytes, requiring the exact head corruption refusal, mismatch offset and byte values, and an unchanged retained-file witness. + +Expected checksum bytes come from the canonical publication input before corruption, while recovery computes its own checksum over the preimage. + +A disposable production mutation deletes the stage on the planning-error path; the preservation assertion fails at the first corrupted prefix. + +The short-head filesystem laws and the existing canonical-prefix integration laws pass in copied Docker debug and release; the retention process-death matrix also passes. + +Both workspace Clippy configurations with `-D warnings`, formatting, and source-structure checks pass. + +Replay uses `cargo test --lib short_head --all-features` and `cargo test --test retention_stage_prefix --all-features`, adding `--release` for optimized execution; calibration uses a fresh build target. + +This closes available head-checksum bytes only; other partially available semantic fields, root and manifest validation, and incomplete-stage pinning remain under audit. diff --git a/docs/testing-evidence/retention-partial-layout-length.md b/docs/testing-evidence/retention-partial-layout-length.md new file mode 100644 index 00000000..587b0092 --- /dev/null +++ b/docs/testing-evidence/retention-partial-layout-length.md @@ -0,0 +1,21 @@ +# Partial layout-length admission + +Change kind: bug fix. Owner: `@flyingrobots`. Subject: Keep runtime preservation of partial anchor layout lengths whose available bytes force every completion above the protocol maximum (#99). + +Regression `18a44e3` was observed RED with admission behavior unchanged from `6c70a28`: a one-byte layout-length prefix beginning `0xff` was discarded with `Clean`. The RED commit adds only the new diagnostic variant and its display/source handling alongside the tests, so the regression compiles without fixing admission. Its valid-prefix control already passed. + +The medium filesystem law covers every nonempty strict length-field prefix with an excessive leading byte, plus the nearest impossible seven-byte prefix above the ceiling, in both the first and second anchor. It requires the exact corrupt-root stage, anchor index, minimum possible completion and format maximum, followed by unchanged retained paths and bytes across real recovery. + +The small public-assessment control generates canonical layout lengths from zero entries and power-of-two entry counts through the format ceiling, then checks every strict length prefix. Each full canonical value witnesses a legal completion; the zero-leading prefixes must remain admissible. The independent specified oracle is the 176-byte fixed overhead, 44-byte entry stride and 46,137,520-byte maximum. Deterministic values, anchor indices and prefix lengths are replay coordinates; the checked-in laws preserve the corpus. + +Production zero-fills unavailable bytes only to compute a lower bound, compares it with the domain maximum and never admits that bound as an observed layout identity. Complete fields retain the existing codec's bounds and congruence checks. The new public `LayoutLengthPrefixAboveMaximum` decode-error variant distinguishes a minimum possible completion from an observed complete value; exhaustive consumers must handle it. No wire-format change is made. + +Replay uses `cargo test --lib partial_anchor --all-features` and `cargo test --test retention_stage_prefix --test retention_root_decoding --all-features`, with `--release` for optimized execution. Runs use copied Docker source, Rust 1.96.0, Linux aarch64 and the owned ext4 sandbox. The retention process-death replay is `cargo xtask durability-crash-matrix --sequence retention`. + +The failure model is interrupted length fields and real filesystem recovery, not partial-anchor ordering, feasibility of all future entries, other partial header constraints, concurrent replacement or physical power loss. Incomplete-stage pinning and post-removal failures remain open. Per-test resource ceilings and suite latency admission remain unapproved enforcement gaps. No performance claim is made. + +Delete the new laws only when stronger public runtime evidence subsumes their exact refusal, preservation and valid-completion promises; no existing expectation was weakened. + +Disposable copied product mutations with fresh targets change the reported minimum, name the wrong stage and delete the root on refusal; the new refusal law fails its exact-cause, stage and retained-evidence checks respectively. A mutation rejecting zero lower bounds fails the generated valid-prefix control, calibrating the absence of false refusals. + +Partial-anchor laws and canonical-prefix/root-decoding controls pass in debug and release, and the retention process-death matrix passes. Initial Clippy findings for the expanded diagnostic formatter and singleton iterator were corrected by sharing identity-message formatting and using `iter::once`; focused debug/release tests, both workspace Clippy configurations with `-D warnings`, formatting and source-structure checks then pass. Original failed attempts remain recorded. diff --git a/docs/testing-evidence/retention-partial-manifest-entry.md b/docs/testing-evidence/retention-partial-manifest-entry.md new file mode 100644 index 00000000..68204f04 --- /dev/null +++ b/docs/testing-evidence/retention-partial-manifest-entry.md @@ -0,0 +1,21 @@ +# Partial manifest entry admission + +Change kind: bug fix. Owner: `@flyingrobots`. Subject: Keep runtime preservation of partial manifest entries whose available bytes already violate generation or namespace-order contracts (#99). + +Regression `574619d` was observed RED on unfixed production `e52b4fd`: both an impossible namespace prefix and a zero root-generation field caused `DiscardManifestStage`, leaving only the root protected. The positive ordered-prefix control already passed on that revision. + +The medium filesystem laws cover descending namespace prefixes, fully available duplicate namespaces, and equal prefixes whose unknown suffix cannot exceed an all-maximum predecessor. A separate sweep places generation zero in every partial-entry length after that field is complete. Each invokes real recovery, requires the exact manifest-stage refusal, entry index and cause, and independently compares retained paths and bytes before and after refusal. + +The small public-assessment control generates every nonempty partial-entry length from two known ordered namespace pairs: one differs in its first byte, the other only in its last. Each full ordered value witnesses a possible completion of its prefix; unknown zero-leading generation bytes remain permitted. These are deterministic prefix sweeps, with field values and lengths as replay coordinates; the checked-in test is the permanent reproducer and no random seed is needed. + +The specified oracle is strict unsigned lexicographic namespace order and positive root generation. Production shares generation and ordering admission with complete entries. For an incomplete namespace, it computes the greatest possible completion by filling only unknown suffix bytes with `0xff`; refusal follows only if even that bound cannot be greater than the predecessor. The bound is never returned as an observed identity. + +Replay uses `cargo test --lib partial_entry --all-features`, `cargo test --lib body_prefix --all-features` and `cargo test --test retention_stage_prefix --test retention_manifest_codec --all-features`, adding `--release` for optimized execution. Runs use copied Docker source, Rust 1.96.0, Linux aarch64 and the owned ext4 sandbox. The retention process-death replay is `cargo xtask durability-crash-matrix --sequence retention`. + +This establishes constraints decidable from the current partial manifest entry, not feasibility of all declared future entries, partial anchor admission or every partial header constraint. Incomplete-stage pinning, post-removal failures and physical power-loss guarantees remain outside this evidence. Per-test resource ceilings and suite latency admission remain unapproved enforcement gaps. There is no wire/API change or performance claim. + +Delete the laws only when stronger public runtime evidence subsumes these refusal, exact diagnostic, preservation and valid-completion promises; no existing expectation was weakened. + +Disposable copied product mutations with fresh targets change the reported entry index, name the wrong stage, and delete the manifest on refusal; both refusal laws fail the corresponding diagnostic, stage and retained-evidence assertions. Filling unknown namespace bytes with zero instead of computing the greatest completion makes the positive law fail at a valid prefix, calibrating the absence of false refusals. + +Focused laws and complete-entry controls pass in debug and release, as do the canonical-prefix and manifest-codec suites. The retention process-death matrix passes. An initial Clippy request to express a boolean pattern match with `matches!` was corrected, followed by fresh focused debug/release runs and passing all-feature/no-default-feature workspace Clippy with `-D warnings`, formatting and source-structure checks; earlier failed attempts remain recorded. diff --git a/docs/testing-evidence/retention-partial-predecessor.md b/docs/testing-evidence/retention-partial-predecessor.md new file mode 100644 index 00000000..8205c251 --- /dev/null +++ b/docs/testing-evidence/retention-partial-predecessor.md @@ -0,0 +1,21 @@ +# Partial initial predecessor admission + +Change kind: bug fix. Owner: `@flyingrobots`. Subject: Keep runtime refusal and evidence preservation for impossible initial-record predecessor prefixes (#99). + +Regression `a0a8133` was observed RED against unfixed production `b941d58`: an initial root with a nonzero byte in its incomplete predecessor field returned clean recovery after `DiscardRootStage`. Successor-prefix controls passed on that same unfixed code. + +The medium filesystem law sweeps every nonempty strict predecessor prefix for roots, manifests and heads. A nonzero final available byte must produce the exact stage-specific corruption refusal, byte offset, expected zero and observed value, while preserving every retained path and byte. It invokes the real filesystem recovery authority after driving the required earlier publication phases. + +The small public-assessment law sweeps the same prefix endpoints for successor generations two and the maximum generation, with both zero and nonzero available predecessor bytes. Each remains an interrupted record with a possible nonzero completion. The specified oracle is the protocol's initial-versus-successor history rule, independently of the admission helper. Inputs and endpoints are deterministic and permanently represented in the test; no probabilistic seed or shrinking step is involved. + +Admission compares only available predecessor bytes when the complete generation is initial. It neither pads missing bytes into observed state nor changes complete-field semantic errors. Root and manifest history admission reuse the existing boundary module; head admission calls it only while the predecessor is incomplete. No wire-format or public API change is introduced. + +Replay uses `cargo test --lib history --all-features` and `cargo test --test retention_stage_prefix --all-features`, adding `--release` for optimized execution. Runs use copied Docker source, Rust 1.96.0, Linux aarch64 and the owned ext4 sandbox. The retention process-death replay is `cargo xtask durability-crash-matrix --sequence retention`. + +This evidence does not establish partial numeric-header admission, feasibility of every future declared entry, incomplete-stage pinning, post-removal failure handling, physical power-loss survival or out-of-band replacement safety. Per-test resource ceilings and suite latency admission remain unapproved enforcement gaps. No performance claim is made. + +Delete these laws only when stronger public runtime evidence subsumes the same initial refusal, exact diagnostics, evidence preservation and valid-successor completion promises. No existing expectation is weakened. + +Disposable copied product mutations with fresh build targets change the reported expected byte, name the wrong stage, delete the root on refusal and apply the initial-record zero rule to successors. Each fails its corresponding exact-cause, stage, retained-evidence or possible-successor assertion. The first diagnostic mutation command matched no formatted source and changed nothing; that run is excluded. The corrected mutation was inspected and failed the named assertion in a fresh target. + +The focused history laws and canonical-prefix controls pass in debug and release. The retention process-death matrix, both workspace Clippy configurations with `-D warnings`, formatting and source-structure checks pass. diff --git a/docs/testing-evidence/retention-partial-profile.md b/docs/testing-evidence/retention-partial-profile.md new file mode 100644 index 00000000..2dba133e --- /dev/null +++ b/docs/testing-evidence/retention-partial-profile.md @@ -0,0 +1,19 @@ +# Interrupted registered-profile prefixes + +Change kind: bug fix. Owner: `@flyingrobots`. Subject: Keep runtime refusal and evidence preservation when an interrupted root already contradicts its registered profile (#99). + +Regression `f4875fe` was observed RED on unfixed production `fc29870`: the first available identity byte contradicted the registered encoding, but real filesystem recovery discarded the root stage and returned `Clean`. + +The medium filesystem law enumerates every interrupted profile end position from 49 through 87, flips the last available profile byte, and requires `StageCorrupt(Root)` with exact `PrefixByteMismatch` offset, expected byte and observed byte. It independently compares retained paths and bytes before and after refusal. The oracle is the checked-in canonical root fixture and the format's closed single-profile registry; no unavailable byte is asserted as observed. + +The fixed-field comparison implementation is shared with existing stage-prefix admission. For complete profile groups the existing typed domain coordinate/digest admission remains in force. The deterministic canonical-prefix integration sweep protects every valid truncated fixture prefix; complete-policy filesystem laws protect the prior diagnostic contract. Registration of another valid profile must update this single-profile prefix rule and its conformance evidence together. + +Replay uses `cargo test --lib partial_profile --all-features`, `cargo test --lib short_policy --all-features` and `cargo test --test retention_stage_prefix --all-features`, adding `--release` for optimized execution. These run in copied Docker source with Rust 1.96.0 on Linux aarch64 and an owned ext4 sandbox. There is no random seed: the ordered end position and fixture identify each minimized one-byte counterexample. The permanent corpus is the checked-in law and fixture. + +Fault scope is deterministic interrupted profile bytes and real filesystem recovery, not physical power loss or concurrent out-of-band replacement. Closure fields, record bodies, root/manifest integrity prefixes, incomplete-stage pinning and post-removal failure semantics remain open. Per-test resource ceilings and suite latency admission remain unapproved enforcement gaps; container execution alone does not establish those controls. + +No public API or wire encoding changes. No performance claim is made. Delete the law only when stronger public recovery evidence subsumes these exact contradiction and preservation promises; no test expectation was weakened. + +Disposable copied product mutations with fresh build targets prove the load-bearing checks: a wrong byte offset fails `name the exact available profile contradiction`, a wrong stage fails `identify the corrupt profile stage`, and deletion on refusal fails `must preserve retained evidence`. The first coordinate mutation did not compile because of an unused variable and is excluded from calibration; its corrected build reaches and fails the intended assertion. + +The focused filesystem laws, complete-policy controls and canonical strict-prefix integration sweep pass in debug and release. The retention process-death matrix, both workspace Clippy configurations with `-D warnings`, formatting and source-structure checks pass. diff --git a/docs/testing-evidence/retention-partial-record-integrity.md b/docs/testing-evidence/retention-partial-record-integrity.md new file mode 100644 index 00000000..5918e37b --- /dev/null +++ b/docs/testing-evidence/retention-partial-record-integrity.md @@ -0,0 +1,21 @@ +# Interrupted record integrity + +Change kind: bug fix. Owner: `@flyingrobots`. Subject: Keep runtime preservation of interrupted root and manifest stages whose available integrity fields contradict their completed preimages (#99). + +Regression `838736c` was observed RED on unfixed production `fd3fbc1`: a root with a contradictory first trailer byte and a root with an incorrect anchor-set digest but complete body were both discarded with `Clean`. + +The medium filesystem laws cover root and manifest stages. One enumerates every strict trailer-prefix endpoint and flips the last available byte; the other changes the body-set digest and stops exactly at body completion. Both require the precise corrupt stage and available-byte or full body-set digest coordinates, then independently compare retained paths and bytes before and after real recovery. The specified/derived oracle is the checked-in canonical conformance bytes, with their preimages unchanged by trailer mutations and their bodies unchanged by header-set mutations. + +Production shares each full decoder's hash and set-verification implementations. The new boundary admits integrity only after framing validation and availability of the relevant preimage; it allocates no unavailable body and never invents missing digest bytes. Complete-record decode order remains unchanged. The existing `PrefixByteMismatch` diagnostic now explicitly includes computed integrity fields; no public enum variant or wire-format change is introduced here. + +Replay uses `cargo test --lib partial_integrity --all-features` and `cargo test --test retention_stage_prefix --all-features`, adding `--release` for optimized execution. Docker runs copied source with Rust 1.96.0 on Linux aarch64 and an owned ext4 sandbox. The canonical-prefix sweep supplies positive controls for all three record types. The process-death replay is `cargo xtask durability-crash-matrix --sequence retention`. + +Inputs are deterministically generated by stage and prefix endpoint, with a single-byte counterexample retained in the test law and fixture. No random seed or probabilistic shrinking is needed. The body-set cases stop at the first boundary where the completed body determines that digest. + +The failure model is interrupted records and real filesystem recovery, not physical power loss or out-of-band replacement. Partial size/history constraints and body semantic admission remain open, as do incomplete-stage pinning and post-removal failures. Per-test resource ceilings and suite latency admission remain unapproved enforcement gaps. No performance claim is made. + +Delete these laws only when stronger public recovery evidence subsumes every promised integrity refusal and preservation observation. No existing test expectation was relaxed. + +Disposable copied product mutations with fresh targets swap expected/observed integrity coordinates, misname the corrupt stage, and remove the root stage while returning the correct refusal. Both filesystem laws fail the corresponding diagnostic, stage and retained-evidence assertions. + +The new filesystem laws and canonical strict-prefix controls pass in debug and release. Complete-record controls also pass using `cargo test --test retention_root_decoding --test retention_manifest_codec --all-features`, including `--release`. An initial command named a nonexistent manifest test target and ran no tests; the corrected targets above supply the evidence. The retention process-death matrix, both workspace Clippy configurations with `-D warnings`, formatting and source-structure checks pass. diff --git a/docs/testing-evidence/retention-publication-phase-fixture.md b/docs/testing-evidence/retention-publication-phase-fixture.md new file mode 100644 index 00000000..f83904d9 --- /dev/null +++ b/docs/testing-evidence/retention-publication-phase-fixture.md @@ -0,0 +1,19 @@ +# Publication prefix fixture binding + +Change kind: refactoring. Owner: `@flyingrobots`. Subject: the test fixture that creates interrupted filesystem publication prefixes (#99). + +At parent `627782d`, the fixture maintained an independent positional array of storage closures beside the public `RetentionPublicationPhase::ALL` sequence. + +The fixture now verifies current state once when requested, then iterates the requested prefix of `ALL` and exhaustively dispatches each phase to its matching storage capability. + +The prefix bound derives from the public sequence plus current-state verification; it is an input-generation bound, not an asserted harness count. + +A zero prefix still performs no storage operation, and every later prefix retains the prior verification-before-publication order. + +Existing runtime assertions remain unchanged: the initial and successor prefix laws independently require their documented recovery results, and the successor law reads the recovered generation and exact root bytes through the public reader. + +The expected disposition table is not derived from `ALL`, so changing the generated phase order does not automatically bless changed recovery outcomes. + +Validation uses the filesystem retention library suites in copied Docker debug and release, including the complete ordered-prefix sweeps, with both workspace Clippy configurations, formatting, and source-structure checks. + +No test of the fixture itself or production behavior change is introduced; the finite prefix evidence does not claim arbitrary filesystem interleaving coverage. diff --git a/docs/testing-evidence/retention-reader-catalog-errors.md b/docs/testing-evidence/retention-reader-catalog-errors.md new file mode 100644 index 00000000..fd7df0c2 --- /dev/null +++ b/docs/testing-evidence/retention-reader-catalog-errors.md @@ -0,0 +1,23 @@ +# Retention reader catalog error evidence + +Change kind: bug fix. Subject: Keep runtime reader refusal. Contract: `FilesystemRetentionSnapshot::load`. Owner: `@flyingrobots`. Oracle: specified documented error boundaries and the exact catalog restart phase, decoder refusal, and underlying I/O kind. + +The public missing-catalog, missing-segment, and corrupt-catalog regressions were observed RED on unfixed head `57cd6df3b15023a9c21307e84f5514040fa4de9e` with the test overlay; regression commit `f5ecbed` preserves the reproduction. Their named assertions reported `View` wrappers where `Catalog` was required, rather than setup or compilation failures. + +The corrupt coordinate-HEAD control passed before the fix and requires a `View` refusal with its concrete `PublicationHeadDecodeError::ChecksumMismatch` source, preventing overbroad catalog classification. + +All new laws are medium because they operate on owned real filesystem stores. The pending-result schedule runs the production source under a shared reader fence, restores the selected catalog and publishes retention synchronously after the first load, then requires a collected view bound to the new retention head and the original catalog. Its temporary missing catalog is a controlled fault rather than production evidence of a legitimate writer losing published files. + +The pinning regression's setup was adapted to the private source's collected admission-result value; its original path-replacement operation and catalog generation/digest oracles are unchanged. + +Independent copied-source mutation builds used fresh Cargo target directories. Returning the catalog failure as I/O reproduced the catalog-boundary failures and failed the named moving-head assertion. Hiding the public error's catalog source failed its direct-source assertion. Stringifying a coordinate decoder refusal failed the head's exact typed-source assertion while retaining a `View` error. + +Mutating the returned catalog generation failed both pinning and moving-head generation checks. Flipping a returned digest byte failed both digest checks. Omitting the loaded retention state failed the exact newly published head assertion. The initial constant-digest mutation triggered a dead-field compile error and is excluded; the corrected mutation retained the original field read and failed at the runtime digest assertions. The scheduled-publication fixture's initial unused receipt compile error is likewise excluded. + +Complete workspace all-feature debug and release suites, default-feature and all-feature Clippy with warnings denied, formatting, and staged source-structure checking passed. The final strengthening of the moving-head assertion and error documentation then passed focused snapshot laws in debug and release, all-feature Clippy, formatting, and staged source checks. Markdown lint passed separately. Raw RED, calibration, and verification logs are retained outside tracked source. + +Replay uses `cargo test --lib filesystem_retention_snapshot --all-features` for the reader laws; add `--release` for optimized execution. Full checks use `cargo test --workspace --all-features`, the same command with `--release`, `cargo clippy --workspace --all-targets --all-features -- -D warnings`, its default-feature counterpart, `cargo xtask source-structure-check`, and `cargo fmt --check`, all from copied source inside Docker. + +The execution profile is copied Docker isolation on pinned Rust 1.96.0, Linux aarch64, and the existing ext4 audit filesystem. The owned scratch stores and synchronous publication schedule control observed inputs. No new network, process-death, power-loss, randomized-input, golden, performance, or fuzz claim is made because this change adds no parser or write protocol. + +Ordinary per-test resource ceilings remain the existing enforcement-profile gap, not an approved waiver. Complete head coordinates, full reader platform admission, distinct valid replacement catalogs, and unrelated retention recovery findings remain separate unresolved work. diff --git a/docs/testing-evidence/retention-reader-catalog-pinning.md b/docs/testing-evidence/retention-reader-catalog-pinning.md new file mode 100644 index 00000000..c31584e6 --- /dev/null +++ b/docs/testing-evidence/retention-reader-catalog-pinning.md @@ -0,0 +1,15 @@ +# Retention reader catalog pinning evidence + +Change kind: bug fix. + +The specified oracle is that replacing an ambient store path after directory pinning cannot change the catalog selected through that pinned directory. + +The medium filesystem regression `replacing_the_ambient_root_preserves_the_pinned_catalog` enters the actual `RetentionViewSource` collection contract with a real migrated store and drives path replacement synchronously, without sleeps or scheduler assumptions. + +The assertion that ambient replacement must not displace the pinned catalog was observed red on unfixed head `0531fca0aa92f7a43fc1a8e3da3660de5a119e0f` with the test overlay; regression commit `ceefca2` preserves that reproduction. Initial compilation mistakes in fixture and digest accessor names were corrected before this runtime red and are excluded from the evidence. + +The remaining assertions compare the returned catalog generation and digest with the original validated HEAD; subsequent [catalog-error remediation](retention-reader-catalog-errors.md) independently mutated both returned coordinates in copied-source builds and observed the corresponding named assertions fail. Distinct valid replacement catalogs remain unexercised here. + +The fixed implementation passed the complete workspace all-feature test suites in debug and release, Clippy with warnings denied both with default features and all features, formatting, staged source-structure checking, and Markdown lint, using copied Docker isolation. + +This evidence does not establish public admission scheduling, complete head-coordinate comparison, total resource ceilings, or absence of other reader defects. diff --git a/docs/testing-evidence/retention-reader-complete-coordinates.md b/docs/testing-evidence/retention-reader-complete-coordinates.md new file mode 100644 index 00000000..97c202d8 --- /dev/null +++ b/docs/testing-evidence/retention-reader-complete-coordinates.md @@ -0,0 +1,21 @@ +# Complete reader coordinate evidence + +Change kind: bug fix. Subject: Keep runtime reader selection. Owner: `@flyingrobots`. Oracle: specified equality of every validated head coordinate across reader collection. + +The medium filesystem regression `a_changed_catalog_length_refuses_the_loaded_reader_view` was observed RED on unfixed head `206ef7b47041926e730e906a786804a6da973c09` with its test overlay; regression commit `953f7fc` preserves the reproduction. The original code accepted the view and failed the named refusal assertion after a correctly checksummed HEAD changed only its admitted catalog length. + +The deterministic schedule runs the production filesystem source under a real shared reader fence, changes HEAD after loading, and requires `AttemptsExhausted { attempts: 1 }` under a single-attempt policy. The added length is one canonical catalog entry, 160 bytes, at the version-one head's catalog-length offset; the changed head is admitted before installation. The initial one-byte length perturbation violated the length congruence rule and is excluded as fixture setup failure. + +Small storage-independent laws assert the returned semantic view, never a harness call count. One generated domain covers every admitted manifest length above the minimum with the original generation, digest and predecessor fixed; another covers every nonzero first predecessor byte with all other bytes fixed. Both require the later stable view instead of the superseded view. + +The generation/digest pinning and moving-error regression setup was adapted to the catalog tuple's extra field without changing their observable expectations. The public coordinate value now requires complete retention heads rather than generation/digest pairs. + +Independent copied-source mutations used fresh Cargo target directories. Normalizing the filesystem source's catalog length to a constant failed the exact single-attempt refusal assertion. Comparing retention heads only by generation and digest failed both generated semantic laws, which returned `superseded` instead of `stable`; the first counterexamples were the next admitted manifest length and predecessor byte one. + +The generated domains are deterministic ascending enumerations, with the smallest changed value first and its coordinates printed on failure. They use no ambient randomness or schedule; replay is the same focused command and the smallest counterexamples remain represented by those first inputs. + +Tests run from copied source in Docker on pinned Rust 1.96.0, Linux aarch64 and the existing ext4 audit filesystem; ordinary per-test resource ceilings remain an explicitly unresolved enforcement gap. Replay uses `cargo test --lib coordinate --all-features`; add `--release` for optimized execution. + +Complete workspace all-feature debug and release suites, default-feature and all-feature Clippy with warnings denied, formatting, staged source-structure checking, and Markdown lint passed. Raw runtime RED, excluded fixture failure, mutation and complete validation logs are retained outside tracked source. Core source and regression hashes were compared with the copied Docker files before publication. + +The domains do not exhaust predecessor digest space, physical filesystem faults, platform admission, or arbitrary head replacement/restoration between reads. Generic port loaded-value binding and other audit findings remain separate work. No parser, fuzz, process-death, power-loss, performance or durability claim is added by this semantic coordinate change. diff --git a/docs/testing-evidence/retention-reader-fence-coverage.md b/docs/testing-evidence/retention-reader-fence-coverage.md new file mode 100644 index 00000000..72ebbbce --- /dev/null +++ b/docs/testing-evidence/retention-reader-fence-coverage.md @@ -0,0 +1,21 @@ +# Reader fence acquisition coverage + +Change kind: test correction with a behavior-preserving acquisition refactor. Subject: Keep runtime reader exclusion and fence identity. Owner: `@flyingrobots`. Oracle: the acquired shared lock must name the same empty regular-file inode as the directory entry, and an exclusive nonblocking collector must receive `Errno::WOULDBLOCK` while readers hold it. + +At base `e6edb4357d89819506729ecf3016fe33750199c4`, acquisition already verified identity both before and after flock, but the test named for replacement only wrote nonempty bytes to the existing inode. That case is renamed and now requires the precise public `Fence` refusal with `InvalidData`. + +The new medium filesystem law replaces the zero-length lock after the initial verified open and before flock, while the original inode remains pinned by its opened handle. Acquisition must refuse with `InvalidData` when the post-lock entry no longer names that handle. + +A private scheduling callback shares the actual acquisition implementation with the ordinary no-op production path. There are no global hooks, sleeps, arbitrary scheduler races, or simulated filesystem results. The law enters the narrow reader-fence contract that owns identity verification; the existing nonempty-file law separately exercises the public snapshot's fence-error mapping. This intentionally avoids adding a second injection interface to the outer snapshot loader solely to duplicate that mapping check. + +The existing shared-reader law now asserts the exact kernel contention errno rather than accepting every possible I/O error. Its final exclusive acquisition after dropping the reader fences remains the positive release control. + +Fresh-target copied-source calibration removed the post-flock verification: the prior nonempty-file law still passed, while the new replacement law failed at its named assertion because acquisition returned a fence. Replacing shared acquisition with an unlock failed the exact contention assertion when the collector incorrectly acquired its exclusive lock. Because the identity protection already existed, this is not a claim that unfixed production lacked the check: these mutations expose the prior coverage gap. + +Changing the identity/length refusal from `InvalidData` to `Other` failed both precise refusal assertions while preserving generic failure, demonstrating why a bare error predicate is insufficient. + +The unchanged acquisition protocol passed complete workspace all-feature debug and release suites, default-feature and all-feature Clippy with warnings denied, formatting, and staged source-structure checks in copied Docker isolation. Final metadata and diagnostic-message changes passed focused debug/release fence laws, all-feature Clippy, formatting and source checks. Raw calibration and validation logs remain outside tracked source. + +All affected laws are medium and use owned migrated stores on the copied Docker ext4 audit filesystem with pinned Rust 1.96.0 and Linux aarch64. Replay uses `cargo test --lib fence --all-features`; optimized replay adds `--release`. Ordinary per-test resource ceilings remain an unresolved enforcement gap. + +This schedule covers replacement during acquisition, not arbitrary replacement after the final verification, hostile deletion outside the cooperative locking protocol, process-death durability, or other platforms. The change does not establish collector integration, repair other reader admission gaps, or change a durable format. diff --git a/docs/testing-evidence/retention-reader-root-identity.md b/docs/testing-evidence/retention-reader-root-identity.md new file mode 100644 index 00000000..7adaed80 --- /dev/null +++ b/docs/testing-evidence/retention-reader-root-identity.md @@ -0,0 +1,13 @@ +# Retention reader root-identity evidence + +Change kind: bug fix. Subject: Keep runtime fenced snapshot admission. Size: medium. Oracle: mutually consistent migration records are admissible for reading only when their restart-stable device and inode binding names the opened root; identity diagnostics preserve the bound and observed values. + +Regression commit `5c32126`, based on `5d73016`, transplants the complete migration-record set from one controlled migrated directory into a second directory on the same filesystem, then invokes `FilesystemRetentionSnapshot::load`. The unfixed reader returned a snapshot and the regression was observed RED with `reader accepted migration records naming another physical root`. This was a runtime acceptance failure rather than setup or source-spelling evidence. + +The fixed reader probes the opened directory's identity and applies the writer's existing `require_root_identity` predicate before fence acquisition and catalog loading. The regression requires the exact `File` coordinate, donor inode as expected, and recipient inode as observed in the retained `FilesystemPlatformAdmissionError::RootIdentityChanged` source. Its numerical oracle comes from independent filesystem metadata observations, not the production comparison helper. + +Independent copied-source mutation builds use fresh Cargo target directories. Swapping expected and observed values in the shared refusal constructor fails the precise runtime assertion. Stringifying the reader's identity source also fails that assertion while preserving the admission refusal, demonstrating that a generic error is insufficient. The mutation failures name the required coordinate and both observed values. + +Full all-feature workspace debug and release suites, both Clippy feature configurations with warnings denied, formatting, and staged source-structure checks passed in copied Docker isolation on pinned Rust 1.96.0 and Linux aarch64 with the existing ext4 audit filesystem. A final import-only readability change passed focused snapshot laws and all-feature Clippy. Markdown checks passed separately. Raw RED, mutation, and validation artifacts remain outside tracked source. + +This experiment covers a foreign inode binding on one filesystem and preserves the ordinary migrated-reader positive path. It does not execute an actual device move or remount, establish the full production platform profile for readers, repair catalog-path re-resolution, complete collected head coordinates, or enforce per-test resource ceilings. Mount-instance identity remains excluded by the unchanged shared predicate. No durable bytes, hash preimages, or API signatures change. diff --git a/docs/testing-evidence/retention-recovery-discard-prefix.md b/docs/testing-evidence/retention-recovery-discard-prefix.md new file mode 100644 index 00000000..1b28d2b6 --- /dev/null +++ b/docs/testing-evidence/retention-recovery-discard-prefix.md @@ -0,0 +1,17 @@ +# Retention recovery discard-prefix evidence + +Historical evidence at the recorded revisions. The [bounded landing contract](retention-landing.md) supersedes automatic incomplete-stage disposal and changes the corresponding success expectations; this receipt does not establish current disposal safety. + +Change kind: bug fix with a more precise refusal diagnostic for an already-refused case. Subject: Keep runtime recovery. Size: medium. Oracle: the ordered forward publication protocol requires complete linked earlier stages before any manifest or head write begins, and impossible-prefix refusal preserves retained bytes. + +Regression commit `0e5de49` introduced the typed refusal vocabulary and runtime laws without adding the planner guard. Its first parallel run had shared scratch-directory collisions; that run is not evidence for the affected cases. Commit `382abd3` gives each fault case its own stage-and-fault label. A separate copied tree carrying those isolated laws and the unfixed planner was observed RED at the intended assertions. A setup run with an unused new guard module was also excluded from RED evidence. + +The missing root stage and link at a truncated manifest caused the old planner to delete the manifest, with the missing-link case also relinking the root. A truncated head with a missing root link or missing manifest stage or link caused deletion or relinking. Each fails the retained-byte assertion. A truncated head with a missing root stage already refused as `ManifestStageWithoutRootStage`; that case preserves bytes and fails the expected precise earlier-evidence diagnostic instead. The fixture derives canonical stage bytes through the real forward authority before shortening the later record and removing the named earlier evidence. + +The fix computes earlier-prefix admission before consuming evidence and applies its result after existing corruption and conflicting-pool checks, preserving their typed diagnostics. Fixed runtime laws and the full all-feature workspace passed in debug and release on pinned Rust 1.96.0 in copied Linux aarch64 Docker isolation. Formatting, both Clippy feature configurations with warnings denied, and source-structure checks including the new staged modules passed. Existing legitimate interrupted-publication prefix laws still pass. + +An older publication test constructed a truncated manifest without any earlier root, so its setup represented the newly refused impossible state. Its unchanged publication, stage-disposition, and exact head-byte assertions now use a canonical incomplete root, the only first stage that permits resumed publication without protected earlier artifacts. Valid manifest and head interrupted prefixes remain exercised by the forward-prefix recovery laws; their protected orphans are not silently discarded to resume publication. + +Independent mutation builds use fresh Cargo target directories to avoid stale incremental artifacts after archive overlays. A wrong truncated-stage coordinate fails every precise refusal assertion while preserving bytes. A wrong earlier-stage coordinate fails the missing-manifest cases while preserving bytes. Raw RED, mutation, and validation artifacts are retained outside tracked source. A reused-target mutation run that executed stale baseline artifacts is excluded from calibration evidence. + +These are deterministic missing-evidence experiments at the declared manifest and head write boundaries. They do not establish every incomplete byte length, every multi-namespace state, physical power-loss behavior, atomic unlink under concurrent replacement, staged-root closure verification, or per-test resource ceilings. Markdown policy checks are documentation evidence rather than product runtime evidence. diff --git a/docs/testing-evidence/retention-recovery-head-history.md b/docs/testing-evidence/retention-recovery-head-history.md new file mode 100644 index 00000000..15e6a466 --- /dev/null +++ b/docs/testing-evidence/retention-recovery-head-history.md @@ -0,0 +1,15 @@ +# Retention recovery head-history evidence + +Change kind: bug fix. Subject: Keep runtime retention recovery. Size: medium. Oracle: an existing namespace permits only the checked successor root generation with the selected predecessor digest; refusal preserves retained bytes. + +The regression in `filesystem_retention_recovery_history_tests::recovered_head_refuses_a_skipped_root_generation` was observed RED on commit `91762ea`, based on the ordinary integration of main in `47d7f07`. Canonical generation-three root bytes named the authenticated generation-one predecessor, with matching manifest and head coordinates and retained-stage pool hard links. Recovery returned `Committed`, replaced the head, and removed stages, failing the retained-byte assertion. The reproduction failed at the intended runtime assertion rather than setup. + +The fix applies the existing `root_succeeds` predicate during uncommitted head planning. The regression passes in debug and release. A separate copied-source mutation changed the new refusal to `ManifestNotSuccessor`; byte preservation still passed, but the exact typed `RootNotSuccessor` assertion failed with the observed wrong variant. The original RED run calibrates the retained-byte assertion; the mutation calibrates the typed refusal assertion. + +Validation ran in copied Docker source on pinned Rust 1.96.0, Linux aarch64, with the existing isolated ext4 audit filesystem. The retention suite and full all-feature workspace suites passed in debug and release. Formatting, Clippy for all and minimal features with warnings denied, source-structure checks including the new staged regression module, and Markdown lint passed. Raw RED, mutation, focused GREEN, and workspace logs are retained as review artifacts outside tracked source. + +Follow-up runtime laws cover a generation-two root at initial head publication and a generation-two root inserted into a previously absent namespace. A deterministic filesystem sweep covers candidate generations `3..=16` and `u64::MAX` against the selected generation-one predecessor. Each case builds canonical mutually consistent stages with matching pool inodes, invokes real recovery, and requires exact refusal with every retained byte unchanged. Ascending enumeration reports the minimal generated counterexample, generation three, and the module records the replay command. + +All added laws were observed RED with the exact planner from unfixed integration commit `47d7f07` overlaid into an independent copied source tree carrying the new tests. The runtime failures returned `Committed` and removed retained stages. A separate wrong-refusal mutation failed the exact typed assertions for the initial and inserted namespace laws and at the generation sweep's minimal counterexample. The fixed history laws passed in debug; the complete retention suite passed in release. Formatting, all-feature Clippy, and source checks including both new modules passed in Docker. + +This evidence covers the declared finite domain and named initial/insertion examples. It does not establish the entire generation domain, every selected predecessor generation, every filesystem schedule, physical power-loss behavior, or per-test resource-ceiling enforcement. Existing legitimate committed-cleanup and publication-prefix laws passed in the retention suite; green checks do not establish absence of every regression. diff --git a/docs/testing-evidence/retention-recovery-observation-errors.md b/docs/testing-evidence/retention-recovery-observation-errors.md new file mode 100644 index 00000000..af344675 --- /dev/null +++ b/docs/testing-evidence/retention-recovery-observation-errors.md @@ -0,0 +1,19 @@ +# Recovery observation error evidence + +Change kind: bug fix with an intentional diagnostic API change. Owner: `@flyingrobots`. Oracle: publication identifies failure during recovery observation while preserving its original typed or operating-system cause. + +On base `1c6bab2`, real filesystem regressions for an unknown retention entry and a symbolic-link root stage both failed because publication returned the observation cause directly inside `CurrentVerification`, omitting a recovery-observation boundary. RED commit `27f7259` records the new error vocabulary and regressions before production wiring. + +The fix preserves `RecoveryObservationRefused` inside the existing publication error, with the original I/O error as its standard `Error::source`. The tests require that boundary, then read the original cause through the standard source chain: the exact `UnknownRetentionEntry` variant or the kernel's no-follow `ELOOP` code. + +Existing current-state tests still assert their underlying specific refusals. Their shared extraction helper explicitly traverses the new observation layer; the new boundary tests inspect that layer directly, so the compatibility adaptation does not stand in for proof of the wrapper itself. + +Copied-source mutations stringify the original observation cause or hide it from `Error::source`, calibrating cause fidelity and source-chain access separately from the original missing-boundary RED. + +The first workspace run exhausted inode capacity during fixture setup; that run is not reported green. Completed audit copies were preserved on the separate build volume before revalidation, leaving the ext4 audit filesystem available for the capacity scenarios. + +The next workspace run exposed an earlier descriptor regression test spawning a child inside the library test binary, violating the existing source-layout guard. Commit `3fe8ed3` moves that law into its own integration binary and retains the guard; restoring the wrong clone-error mapping still makes the moved runtime test fail. The prior hosted Rust failure had the same source-layout cause. + +After those corrections, complete workspace all-feature debug and release suites passed. Both workspace Clippy feature configurations, formatting and source-structure checks passed. Both cause-erasure mutations failed the intended runtime checks; pinned Markdown lint passed separately. + +Replay uses `cargo test --lib filesystem_retention_observation_error_tests --all-features`, adding `--release` for optimized execution. The symlink law is Unix-specific; this run uses copied Linux Docker on the ext4 audit filesystem. These tests establish diagnostic behavior, not new crash or power-loss coverage. diff --git a/docs/testing-evidence/retention-short-count-bounds.md b/docs/testing-evidence/retention-short-count-bounds.md new file mode 100644 index 00000000..08b2a98f --- /dev/null +++ b/docs/testing-evidence/retention-short-count-bounds.md @@ -0,0 +1,23 @@ +# Interrupted stage count bounds + +Change kind: bug fix. Owner: `@flyingrobots`. Subject: preserving short root and manifest stages with excessive complete count fields (#99). + +Regression `323578c` is observed RED against unfixed production `85d7fe9`: recovery discards a 48-byte root header declaring 65,537 anchors with a self-consistent record length. + +Interrupted framing admission now calls the same count-admission functions as complete semantic header admission, rejecting excessive counts before calculating the rest of the interrupted record's framing. + +Full-record decoding retains its existing checksum-before-semantic-admission order. + +The filesystem law covers root and manifest counts immediately above their specified ceilings and at the largest unsigned count, while keeping the declared length consistent so a length mismatch cannot mask the count defect. + +Each case requires the exact stage, maximum and observed count, and byte-identical retained evidence after refusal. + +Three disposable production mutations misname the corrupt stage, replace the observed count with zero, and delete the root stage while returning the correct refusal; each fails its corresponding runtime assertion. + +Recovery suites, the focused count law, and canonical-prefix integration tests pass in copied Docker debug and release; the retention process-death matrix also passes. + +Clippy initially required the extracted count-admission functions to be `const`; after that declaration-only correction, both workspace Clippy configurations, formatting, and source-structure checks pass. + +Replay uses `cargo test --lib excessive_counts_in_short_headers --all-features`, adding `--release` for optimized execution; calibrations use separate copied trees and fresh targets. + +This closes complete count bounds only; other semantic fields, partially available fields, and incomplete-stage pinning remain in the broader audit. diff --git a/docs/testing-evidence/retention-short-head-history.md b/docs/testing-evidence/retention-short-head-history.md new file mode 100644 index 00000000..331dce72 --- /dev/null +++ b/docs/testing-evidence/retention-short-head-history.md @@ -0,0 +1,21 @@ +# Interrupted head history admission + +Change kind: bug fix. Owner: `@flyingrobots`. Subject: preserving short head stages whose complete semantic fields already contradict retention history (#99). + +Regression commit `de271d8` is observed RED against unfixed production `785275a`: a 104-byte generation-one head prefix with a nonzero predecessor is successfully discarded rather than refused. + +The filesystem law constructs that contradiction and the complementary successor-without-predecessor case, requiring `Plan/StageCorrupt(Head)/Semantic` with each exact typed cause and an unchanged retained-file witness. + +Full head decoding and interrupted-head admission now share semantic field admission; the complete decoder still checks exact length, fixed fields, and checksum first. + +Interrupted recovery calls that shared admission only once all semantic fields are available, and never publishes or exposes the resulting unchecksummed head as admitted state. + +The head-filtered library suites pass in copied Docker debug and release, and the retention process-death matrix passes with valid interrupted prefixes. + +A disposable production mutation deleting `head.next` on the planning-error path fails the retained-evidence assertion while still reporting the correct semantic refusal. + +Both workspace Clippy configurations with `-D warnings`, formatting, and source-structure checks pass. + +Replay uses `cargo test --lib contradictory_history_in_short_heads --all-features`, adding `--release` for optimized execution; the calibration uses a fresh build target. + +This closes complete semantic head fields only; partially available fields, root and manifest variable fields, and incomplete-stage pinning remain in the broader recovery audit. diff --git a/docs/testing-evidence/retention-short-head-length.md b/docs/testing-evidence/retention-short-head-length.md new file mode 100644 index 00000000..6be9adc1 --- /dev/null +++ b/docs/testing-evidence/retention-short-head-length.md @@ -0,0 +1,23 @@ +# Interrupted head manifest-length admission + +Change kind: bug fix. Owner: `@flyingrobots`. Subject: preservation of corrupt evidence when an interrupted head contains a complete invalid manifest-length field (#99). + +The regression in `f45a748` runs against unfixed production from `02f710b`, creating real linked root and manifest stages and a 40-byte head prefix whose complete length field is invalid. + +The zero-length case fails RED because recovery returns a successful receipt executing `DiscardHeadStage`, instead of the expected typed corruption refusal. + +Prefix admission now calls the existing semantic manifest-length constructor once all eight length bytes are available, returning `RetentionHeadDecodeError::ManifestLength` on refusal. + +The filesystem law covers zero, just below the minimum, a within-bounds misaligned length, just above the maximum, and the largest unsigned value; each must return `Plan/StageCorrupt(Head)` with exact error coordinates and preserve the retained-file witness. + +The expected bounds and alignment cases are specified directly from the version-two format rather than computed with the admission constructor under test. + +A disposable production mutation deletes `head.next` while returning the correct planning error; the preservation assertion fails at `invalid short head length 0 must preserve retained evidence`. + +Recovery suites and the focused short-head law pass in copied Docker debug and release; the retention process-death sequence also passes, retaining valid interrupted-write recovery behavior. + +Both workspace Clippy configurations with `-D warnings`, formatting, and source-structure checks pass. + +Replay uses `cargo test --lib invalid_manifest_lengths_in_short_heads --all-features`, adding `--release` for optimized execution; the process-death replay is `cargo xtask durability-crash-matrix --sequence retention`. + +This closes the complete head manifest-length field case only; other variable fields, partially available numeric fields, and incomplete-stage pinning remain under the broader open recovery audit. diff --git a/docs/testing-evidence/retention-short-namespace.md b/docs/testing-evidence/retention-short-namespace.md new file mode 100644 index 00000000..e24ded64 --- /dev/null +++ b/docs/testing-evidence/retention-short-namespace.md @@ -0,0 +1,23 @@ +# Interrupted root namespace size + +Change kind: bug fix. Owner: `@flyingrobots`. Subject: preservation of root prefixes with impossible complete namespace-length fields (#99). + +Regression `1d739fa` is observed RED on unfixed production `6dde780`: a 42-byte root prefix declaring an empty namespace is discarded successfully. + +Prefix admission now calls the same allocation-free domain length admission used by namespace construction as soon as the complete length field is available. + +The helper becomes crate-visible for boundary admission, without extending the public API or introducing an adapter dependency into the domain. + +The filesystem law covers zero, 256, and the maximum encoded unsigned length at prefixes ending immediately after the namespace length and after all framing fields; declared record lengths remain consistent with the namespace size. + +Each case requires the exact root corruption and namespace cause, and an unchanged retained-file witness. + +Disposable production mutations replace the namespace cause with an unrelated overflow error and delete the stage while returning the correct cause; each fails the corresponding runtime assertion. + +Namespace-filtered library suites and canonical-prefix integration tests pass in copied Docker debug and release; the retention process-death matrix also passes. + +Both workspace Clippy configurations with `-D warnings`, formatting, and source-structure checks pass. + +Replay uses `cargo test --lib namespace --all-features` and `cargo test --test retention_stage_prefix --all-features`, adding `--release` for optimized execution; calibrations use separate copied trees and fresh targets. + +Namespace bytes are opaque, so no text, path, or Unicode validation is added; other root semantic fields, partial fields, and incomplete-stage pinning remain under audit. diff --git a/docs/testing-evidence/retention-short-record-framing.md b/docs/testing-evidence/retention-short-record-framing.md new file mode 100644 index 00000000..08579c1a --- /dev/null +++ b/docs/testing-evidence/retention-short-record-framing.md @@ -0,0 +1,21 @@ +# Interrupted root and manifest framing + +Change kind: bug fix. Owner: `@flyingrobots`. Subject: preserving short stage headers whose declared size contradicts complete framing fields (#99). + +Regression `c817fa0` is observed RED on unfixed production `0868675`: a 48-byte root prefix declaring zero length is successfully discarded as a clean recovery. + +Once all size fields are available, prefix admission now reuses each decoder's checked canonical-size calculation and declared-length comparison. + +The filesystem law covers root and manifest stages with zero, actual-length-plus-one, and maximum unsigned declared lengths, requiring the exact stage and `DeclaredLengthMismatch` coordinates plus unchanged retained evidence. + +Expected sizes come from the original golden record bytes, independently of the framing functions used by admission. + +Three disposable production mutations mislabel the root corruption as a head, swap expected and observed size coordinates, and delete the root stage while returning the correct error; each fails its corresponding named runtime assertion. + +Recovery suites, the focused framing law, and canonical-prefix integration tests pass in copied Docker debug and release; the retention process-death matrix also passes. + +Both workspace Clippy configurations with `-D warnings`, formatting, and source-structure checks pass. + +Replay uses `cargo test --lib contradictory_lengths_in_short_stage_headers --all-features` and `cargo test --test retention_stage_prefix --all-features`, adding `--release` for optimized execution; calibration uses separate copied trees and fresh targets. + +This closes consistency of complete framing-size fields only; bounds and semantics of other root/manifest fields, partially available numeric fields, and incomplete-stage pinning remain under audit. diff --git a/docs/testing-evidence/retention-short-record-history.md b/docs/testing-evidence/retention-short-record-history.md new file mode 100644 index 00000000..d8c23ee2 --- /dev/null +++ b/docs/testing-evidence/retention-short-record-history.md @@ -0,0 +1,21 @@ +# Interrupted root and manifest history + +Change kind: bug fix. Owner: `@flyingrobots`. Subject: preservation of interrupted root and manifest records with contradictory complete history fields (#99). + +Regression `db7e398` is observed RED on unfixed production `f475211`: an interrupted generation-one root naming a predecessor is discarded as clean recovery. + +The filesystem law covers both root and manifest initial-with-predecessor and successor-without-predecessor contradictions, requiring exact semantic causes and unchanged retained-file witnesses. + +Domain predecessor admission is now crate-visible through each record type and shared by complete construction and prefix admission; it neither requires nor allocates unavailable body entries. + +Complete decoding retains its existing integrity checks before semantic construction. + +Three disposable product mutations replace the history cause with overflow, misname the corrupt stage, and delete the root stage while returning the correct error; each fails the corresponding runtime assertion. + +History-filtered library suites and canonical-prefix integration tests pass in copied Docker debug and release, and the retention process-death matrix passes. + +Both workspace Clippy configurations with `-D warnings`, formatting, and source-structure checks pass. + +Replay uses `cargo test --lib history --all-features` and `cargo test --test retention_stage_prefix --all-features`, adding `--release` for optimized execution; calibration uses separate copied trees and fresh targets. + +This closes complete root and manifest history fields only; root profile/limit fields, partial fields, record bodies and integrity prefixes, and incomplete-stage pinning remain under audit. diff --git a/docs/testing-evidence/retention-short-root-policy.md b/docs/testing-evidence/retention-short-root-policy.md new file mode 100644 index 00000000..6b846776 --- /dev/null +++ b/docs/testing-evidence/retention-short-root-policy.md @@ -0,0 +1,23 @@ +# Interrupted root policy admission + +Change kind: bug fix. Owner: `@flyingrobots`. Subject: Keep runtime preservation of invalid complete policy groups in interrupted retention roots (#99). + +Regression commit `452e229` was observed RED on unfixed production `79e0530`: recovery discarded roots with an unsupported profile identity or a zero node limit and returned `Clean`. + +An initial test attempt shared a sandbox name across concurrent laws and produced an `AlreadyExists` setup failure; that failure is excluded from the RED evidence. The committed regression isolates each field and prefix case, and both laws then fail at the intended runtime refusal check. + +The medium filesystem laws invoke real authority recovery, require the exact corrupt-root stage and profile or closure-limit cause, and compare retained paths and bytes before and after refusal. The specified oracle is the version-2 registered profile and positive bounded closure-resource contract; profile digest expectations use the registered public profile constant, so an independently corrupted registry definition is outside this oracle. + +The cases cover unsupported identity and version, a wrong profile digest, and zero and above-maximum values for nodes, depth, encoded bytes and physical bytes. They use deterministic literal byte mutations of the checked-in root fixture, with no random seed or reducer needed; the permanent reproducer is `filesystem_retention_short_policy_tests`. + +Production now reuses `RegisteredRetentionProfile::admit` when the complete group is available through byte 88 and `RetentionClosureLimits::new` through byte 116. It does not allocate unavailable body entries or change the integrity-before-semantics order of complete record decoding. + +Replay uses `cargo test --lib short_policy --all-features` and `cargo test --test retention_stage_prefix --all-features`, with `--release` for optimized execution, in copied Docker source on Linux aarch64 with Rust 1.96.0 and the owned ext4 sandbox. The retention process-death replay is `cargo xtask durability-crash-matrix --sequence retention`. + +These laws establish refusal and byte preservation for complete policy groups, not partial fields, body ordering, root/manifest integrity prefixes, stage pinning against concurrent replacement, or physical power loss. Per-test resource ceilings and suite latency admission remain enforcement-profile gaps, not approved waivers. No performance claim or format/API change is made. + +Delete these tests only when stronger public recovery coverage subsumes their malformed-policy cases and exact preservation promises; no existing test expectations were relaxed. + +Calibration used disposable copied product trees and fresh build targets: wrong profile/limit causes fail `report exact root policy violation`, naming `Head` for root corruption fails `identify the corrupt policy stage`, and deleting `root.next` while returning the correct refusal fails `invalid policy must preserve retained evidence`. Both filesystem laws fail each injected defect at the corresponding assertion. + +Debug and release regressions and canonical strict-prefix controls pass, as does the retention process-death matrix. Both workspace Clippy configurations pass with `-D warnings`, alongside formatting and source-structure checks. An initial Clippy failure for a needless by-value test parameter was corrected to a borrow and the focused debug/release tests and checks rerun; the original failure remains recorded. diff --git a/docs/testing-evidence/retention-storage-errors.md b/docs/testing-evidence/retention-storage-errors.md new file mode 100644 index 00000000..9e84250d --- /dev/null +++ b/docs/testing-evidence/retention-storage-errors.md @@ -0,0 +1,25 @@ +# Retention storage error evidence + +Change kind: bug fix. Subject: Keep recovery diagnostics and refusal preservation. Owner: `@flyingrobots`. Oracle: exact admitted record disagreement must remain a typed refusal, while missing-file failures preserve the operating system cause and pre-effect refusal must preserve retained evidence. + +Runtime RED was observed on base `6b35af4da66c42841587df3eda2893059ea2467b` with the regression overlay; commit `16d2dcf` contains the regression and public error vocabulary before the production wiring. Byte damage and byte-identical inode substitution failed their named typed-source assertions because recovery returned flattened strings. The missing-file I/O control passed. + +Medium filesystem laws drive a real publication prefix, reopen its stages through the production recovery observation/context code, inject damage after admission, then call the public recovery executor against the actual filesystem authority. They assert the refused root-link step, retained evidence, precise record cause, and original missing-file I/O kind and OS code. + +A small port-level law supplies each public record refusal at each recovery capability and checks the public executor's resulting typed storage error. It asserts returned software behavior and makes no claim that every supplied refusal was generated by a real filesystem fault. + +The companion forward-publication law was observed RED against `16d2dcf` with its own test overlay: the I/O error held only a message string. It now verifies the typed byte refusal through the existing public forward-storage capability and verifies retained evidence after refusal. + +An additional copied-source mutation overwrites the forward stage after detecting its byte refusal; the typed diagnostic still matches, but the retained-evidence assertion fails, calibrating that independent preservation claim. + +The filesystem cases directly exercise byte and identity disagreement plus missing-file I/O; length overflow, trailing-byte races, and post-removal reappearance are not physically induced by this change. Their exhaustive adapter mapping is reviewed separately from the executor propagation law. + +All eight capabilities propagate `RetentionStorageError`, and the shared stage adapter preserves the same typed refusal when forward publication must expose an I/O result. The truncated-stage absence check now retains its exact source rather than replacing every failure with a remained-visible message. + +Fresh-target copied-source mutations substitute the wrong record variants, reconstruct an I/O error from its kind alone, stringify the executor's cause, report the wrong failed step, and overwrite retained evidence after refusal. Each mutation fails its corresponding diagnostic or preservation assertion. These are product-path and executor-contract calibration results, not mutation-score targets. + +Complete workspace all-feature debug and release suites passed. Clippy initially rejected a nonconsumed fixture enum passed by value; after declaring that small scenario value Copy, focused debug/release laws, both Clippy feature configurations with warnings denied, formatting and staged source checks passed. The added forward law also passed focused debug/release execution and the same lint/source checks. Raw RED, original lint failure, calibration and final validation artifacts remain outside tracked source. + +Runs use copied Docker source, pinned Rust 1.96.0, Linux aarch64 and the existing ext4 audit filesystem. The fixed scenario names and synchronous post-admission damage define the replay schedule; no ambient random seed is used. + +Replay uses `cargo test --lib storage_error --all-features` and `cargo test --lib forward_stage_refusal --all-features`, with `--release` for optimized execution. Ordinary per-test resource ceilings, generic invocation-state diagnostics, recovery unlink races, and the wider audit queue remain unresolved. This change supplies no new power-loss, performance, or parser-fuzz claim. diff --git a/docs/testing-evidence/retention-successor-prefix-coverage.md b/docs/testing-evidence/retention-successor-prefix-coverage.md new file mode 100644 index 00000000..c57b9fd1 --- /dev/null +++ b/docs/testing-evidence/retention-successor-prefix-coverage.md @@ -0,0 +1,27 @@ +# Successor publication prefix recovery + +Change kind: test coverage and oracle repair. Owner: `@flyingrobots`. Subject: the runtime state visible after recovery of an interrupted successor publication (#99). + +At parent `d353252`, the successor law exercised only prefixes 2, 9, 13, and 15, while the requirement ledger suggested exhaustive prefix evidence. + +The revised law exercises every ordered phase prefix from 0 through 18 in a fresh migrated store and loads a public `FilesystemRetentionSnapshot` after recovery. + +The observable oracle requires generation 1 and the original root bytes before completion of the head-stage write at prefix 12, and generation 2 and the prepared successor's exact bytes thereafter. + +The expected root payload is the publication input, independent of recovery receipts; its decoder supplies only the namespace used to request the selected root. + +Calibration deletes the committed immutable root immediately after recovery removes its root stage, while preserving the successful recovery receipt: the parent successor test passes, but the revised test fails when the public reader reports the missing root. + +A separate calibration changes the public reader to return `wrong` after its normal verification: the revised test fails at `successor prefix 0: exact selected root bytes`. + +Both calibrations use disposable copied Docker trees and fresh build targets; the production code in the final change is unchanged. + +The expanded law passes in copied Docker debug and release with `cargo test --lib successor_prefixes_recover --all-features`, adding `--release` for optimized execution. + +Both workspace Clippy feature configurations with `-D warnings`, formatting, and `cargo xtask source-structure-check` pass. + +This evidence covers ordered operation boundaries, not arbitrary byte offsets, power loss, or every possible syscall interruption. + +The separate mid-write law samples one 100-byte interruption in each of the root, manifest, and head stage writes; the ledger now states that scope explicitly. + +Independent reader assertions after process death in the durability crash runner remain a separate open review finding; this in-process law does not close it. diff --git a/docs/testing-evidence/retention-view-collector-laws.md b/docs/testing-evidence/retention-view-collector-laws.md new file mode 100644 index 00000000..cb576f21 --- /dev/null +++ b/docs/testing-evidence/retention-view-collector-laws.md @@ -0,0 +1,19 @@ +# Reader collector digest and I/O evidence + +Change kind: test coverage and oracle repair. Subject: returned view contents and source-preserving errors at Keep's public `RetentionViewSource` contract. Owner: `@flyingrobots`. + +At parent `e0d0893`, collector tests covered generation changes but omitted same-generation digest changes and the initial-coordinate, view-load and final-coordinate I/O failures. + +The revised tests return distinct superseded and stable payloads and require the stable payload after a coordinate change. Separate deterministic sweeps change every nonzero first digest byte for catalog and retention heads while preserving generation and other coordinates; this is a bounded digest domain, not exhaustive hash-space coverage. + +Each failing read must return `RetentionViewError::Io` preserving the original raw error code. The synthetic code 123 is a controlled opaque source value, not a claim about an operating system's interpretation or an actual kernel fault. + +Independent copied-source mutations ignore catalog digests, ignore retention digests, reconstruct I/O errors from their kinds, or reclassify them as catalog absence. The revised tests fail at the intended returned-payload, original-cause or error-variant checks, respectively. These are assertions on the running collector's outputs through its public port; they do not claim filesystem integration coverage. + +The parent collector tests were then run against each same production mutation with fresh build targets and passed, demonstrating the original coverage gaps separately from the revised tests' observed RED results. + +The old load-count assertions are removed because they read script internals. The review suggestion to assert an empty script queue is intentionally declined for the same reason. Bounded refusal remains asserted through `AttemptsExhausted { attempts: 2 }`; absence remains asserted through `CatalogAbsent`; successful collection and retry remain asserted through returned payloads. + +Debug and release collector laws, both workspace Clippy feature configurations, formatting and source-structure checks passed in copied Docker. Initial Clippy rejected an unnecessary result type on the now-infallible absence test; that signature was corrected before the final checks. Markdown validation is separate static evidence. + +Replay uses `cargo test --lib retention_view_collector --all-features`, adding `--release` for optimized execution. The sweeps are deterministic, consult no ambient randomness and report their mismatched coordinates on failure. Per-test resource-ceiling enforcement and broader physical-filesystem race coverage remain outside this change. diff --git a/docs/testing-evidence/version-two-reopen-consolidation.md b/docs/testing-evidence/version-two-reopen-consolidation.md new file mode 100644 index 00000000..419628b2 --- /dev/null +++ b/docs/testing-evidence/version-two-reopen-consolidation.md @@ -0,0 +1,17 @@ +# One unchecked version-two reopen path + +Change kind: refactoring. Owner: `@flyingrobots`. Subject: duplicated repository-only version-two admission constructors (#99). + +At parent `3f2d7d8`, the test-only and repository-task constructors both opened an ambient directory, mapped the same platform error, and called `Self::admit` with identical code. + +The test-only duplicate is removed; its callers use `reopen_unchecked_for_repository_tasks` under `cfg(any(test, feature = "repository-tasks"))`. + +Normal library builds without the repository-task feature still expose only the production `reopen` entry point, which performs platform admission before the shared namespace, lock, record, and identity checks. + +The misplaced capability-release comment now documents `into_parts`, rather than the unchecked constructor. + +Existing test expectations are unchanged; only constructor calls move to the retained implementation, so no new assertion or fabricated bug reproduction is introduced. + +Validation runs the complete library suites with all features in debug and release, the version-two admission suite without default features, both workspace Clippy configurations with `-D warnings`, formatting, and source-structure checks in copied Docker. + +The generated model sequences and existing filesystem admission, recovery, and refusal laws continue exercising the same shared admission implementation through the consolidated entry point; these finite checks do not prove equivalence for every possible filesystem state. diff --git a/docs/testing/retention-stage-prefix-evidence.md b/docs/testing/retention-stage-prefix-evidence.md new file mode 100644 index 00000000..096718ac --- /dev/null +++ b/docs/testing/retention-stage-prefix-evidence.md @@ -0,0 +1,119 @@ +# Retention stage prefix refusal evidence + +This receipt owns the focused fixed-field recovery correction in PR #99, production fix `ffb5074`, and the policy-only integration of reviewed PR #152 in `813df95`. The change kind is a bug fix with an additional public error diagnostic; it does not certify the entire retention PR. + +## Runtime claim and oracle + +When an interrupted root, manifest, or head stage contains an available fixed byte that contradicts the version-two grammar, recovery must return the stage-specific `StageCorrupt` refusal and preserve every retained file's bytes. Canonical strict prefixes must remain classified as interrupted writes. The oracle is the format grammar and checked-in canonical conformance records, independently of the case generator. + +The public assessment laws are in [retention_stage_prefix.rs](../../tests/retention_stage_prefix.rs). The filesystem regression is `noncanonical_short_stages_refuse_recovery_without_changing_retained_bytes` in [filesystem_retention_recovery_prefix_tests.rs](../../src/adapters/retention/filesystem_retention_recovery_prefix_tests.rs). These assert production classification, typed diagnostics, and persistent bytes. They do not assert test-harness case totals. + +## Observed RED + +On unfixed production `ee0e585864d8b17ee9b2b2e52950c5f6f5fad4a5` with the filesystem regression copied into the isolated Docker checkout, `cargo test -p keep --lib noncanonical_short_stages_refuse_recovery_without_changing_retained_bytes -- --nocapture` exited 101 at the intended assertion: malformed `root.next` returned `Ok` with `DiscardRootStage` and `Clean`. Test commit `8b3557d` records that reproduction. + +On exact unfixed commit `4513a776d1c901c659d20a75bcf436c96eedf43a`, `cargo test -p keep --test retention_stage_prefix` exited 101. Each corrupt-prefix law failed at fixed byte zero and prefix length one; canonical-prefix laws passed. This was repeated in a dedicated build directory, independent of the fixed checkout's artifacts. + +## Calibration + +Changing the production prefix comparator to report the wrong expected byte made each `short_*_refusal_reports_only_observed_bytes` assertion fail. Changing production admission to reject canonical prefixes made each `canonical_*_prefixes_remain_recoverable_interrupted_writes` assertion fail at its named check. + +Deleting the retained root stage after observing it but before planning left the correct typed corruption refusal intact and made the exact-byte preservation assertion fail: `refusal must preserve every retained byte for root.next`. This demonstrates the preservation assertion independently of the refusal assertion. + +One early calibration selected zero matching tests after independent checkouts shared a build directory. That run is invalid and is excluded from this evidence. Parent, fixed, and mutated checkouts were re-executed with separate Cargo target directories; mutations affected only disposable Docker copies. + +## Observed GREEN + +All execution used copy-based Docker isolation on Linux with pinned Rust 1.96.0. `cargo test -p keep` passed, including the library, public integrations and doctests. `cargo test -p keep --release --test retention_stage_prefix` and `cargo test -p keep --release --lib retention` passed. Full release workspace validation is not claimed. + +Both workspace Clippy configurations passed with all targets and `-D warnings`, with and without all features. Rust formatting, fuzz formatting, the source structure check, whitespace validation, and Docker Markdown lint passed. The duplicate cross-sequence match arms encountered during integration were corrected separately in `5b9fc4d` without changing their typed refusal. + +The old successful interrupted-write fixture used literal garbage. It now uses a strict prefix of the actual canonical manifest and retains its existing publication/head-byte assertions. The new refusal regression separately protects malformed short bytes; this preserves the distinction between interrupted canonical writes and corruption. + +## Bounded fuzz exploration + +The existing retention format target now calls all three stage assessments as well as complete-record decoders. The focused Docker command was `cargo +nightly-2026-07-24 fuzz run retention_format --debug-assertions --sanitizer address -- -seed=99 -max_total_time=15 -timeout=5 -max_len=1048576 -rss_limit_mb=1024 -print_final_stats=1`, using cargo-fuzz 0.13.2 and the prepared canonical corpus. Seed 99 was fixed in the invocation outside the fuzz process; the resource arguments match [campaign.env](../../fuzz/campaign.env). The command completed successfully without a reported finding. + +An earlier exploratory run used different input/RSS limits and is not the policy smoke receipt. Fuzz completion is robustness evidence for the exercised target, not proof that every semantic corruption is refused or every trust-boundary parser is covered. + +## Remaining blind spots + +The focused fix checks fixed fields. Incomplete semantic and variable fields, sync-before-publication recovery ordering, current-generation and namespace admission, error boundaries, and stronger persistent-view oracles remain review obligations. The runtime fixtures bypass production platform admission deliberately to isolate post-admission storage behavior; they do not establish platform eligibility or physical power-loss durability. PR #99 remains gated on its complete review queue and current-head validation. + +## Complete generation field correction + +Test commit `9431fc0` demonstrated that the published unfixed production implementation at `ce54ae5` classified complete zero-generation fields as discardable truncation. The public root and manifest laws failed at prefix length 40, and the head law failed at prefix length 32; the filesystem law returned `DiscardRootStage` instead of the required typed corruption refusal. Production fix `b365f6b` admits complete fields through the existing checked generation constructors, without fabricating missing bytes or changing complete-record decoding precedence. + +The public generation and fixed-prefix laws, filesystem preservation law, release retention library suite, both workspace Clippy configurations, formatting, source structure check, and the same bounded seeded ASan retention campaign passed. A disposable production mutation deleting `root.next` before planning made the filesystem preservation assertion fail by name even though the typed refusal remained correct. This is focused recovery evidence, not a complete workspace or power-loss claim. + +## Tool verification and test curation + +Commit `69def44` replaces the crash CLI test's implementation-derived round trip and obsolete refusal of `retention` with independent documented accepted and rejected spellings. These assertions verify repository-tool runtime behavior, not Keep durability. Changing the production retention spelling to `retentions` made both the admission and refusal laws fail at their intended assertions. The first calibration attempt failed during compilation because a source-copy exclusion accidentally removed a legitimate directory named `target`; it is invalid evidence. The corrected complete source copy produced the two assertion failures in a separate build directory. + +The deletion criterion for `requirement_ledger_names_planned_and_executable_evidence` is an uninterpretable class-5 change detector under Testing Standards Rule 18: substring presence neither establishes the listed claims nor validates their evidence, while the frozen phrase `Planned in #19` rejects legitimate progress to implementation. The controlled full workspace run exposed that obsolete phrase assertion; changing it to a new status would preserve the same defective oracle. Removing it does not waive any ledger requirement. Runtime contract laws and requirement-by-requirement acceptance review retain responsibility for those claims; Markdown and link validation retain responsibility for document integrity. Neighboring documentation guards are not promoted to runtime evidence by this deletion. + +After that focused deletion, `cargo test --workspace --all-features` and Rust formatting passed in the controlled Docker checkout. The external BLAKE3 oracle was supplied through `/tools/bin` in the test environment. Earlier runs with a transport-generated AppleDouble file or a missing oracle executable were setup failures, not product regressions or retry-to-green evidence. Full release workspace validation and an independent current-head review remain outstanding. + +## Recovery file synchronization + +The OS-boundary regressions in test commit `97b7b96` were observed red on unfixed production `fb34603`: all three complete stages were published by the surviving restart process without a preceding successful `fsync` on the exact staged file. Linux strace 6.13 observed production syscalls while the existing crash command killed the writer immediately after writing each stage. The oracle distinguishes the killed writer from the surviving restart process, requires an actual publication witness, and checks the stage pathname rather than descriptor numbers or syscall counts. This tests a contractual interaction with the operating system, not internal choreography or physical power-loss survival. The Linux test environment requires strace and permission to trace its own children; CI installs and reports the system tool version. + +Adding stage synchronization before root namespace creation/link, manifest link, and head replacement made all three laws pass in debug and release. The retention library suite and all-feature workspace Clippy also passed. Rust formatting initially flagged the new test layout and was corrected; this is not a runtime failure. + +Hosted validation of `fb34603` exposed a stale-document-copy gap in the earlier local full-workspace claim: the Docker checkout did not contain the latest recovery document. The hosted documentation-size guard refused its 306 lines. The earlier pass therefore proves the executed Rust suite against that local document snapshot, not full current-tree validation. Current-tree verification must use a complete source copy and resolve the document ownership issue before claiming the broad gate is green. + +The retention restart section now lives in [retention-recovery.md](../formats/segment-store-v2/retention-recovery.md), linked from the namespace page and protocol index; its grammar and boundaries are preserved. The existing documentation guards read the new owner explicitly and remain documentation checks. Obsolete inline line-length overrides were removed from the touched pages so the accepted global line-length policy governs prose. A complete copy of the changed source and documents passed both `cargo test --workspace --all-features` and its release counterpart, including the OS-boundary regressions. All-feature Clippy, source structure, formatting, and Markdown passed. A preliminary complete-copy run failed Git's ownership check after archive extraction changed the owned sandbox directory's UID; correcting that sandbox ownership, without global Git exceptions or source changes, allowed the controlled run. Physical power loss, kernel I/O error injection, and the remaining PR review queue are still not certified by these results. + +## Namespace refusal before recovery effects + +Test commit `282c105` records RED on unfixed production `a038b1c`. Direct recovery returned `Clean` after discarding `root.next` despite an unknown retention entry; publication returned the expected typed unknown-entry refusal after the stage was deleted, failing the exact-byte witness. An initial parallel run reused a fixture name and encountered `AlreadyExists`; that setup failure is excluded. Each entry point and case now owns a distinct sandbox name. + +Recovery now shares forward namespace admission while additionally permitting the three fixed stage names. The census checks entry names and pool kinds before stage observation or execution; bounded stage observation still checks the admitted names' contents. The real-filesystem laws cover each fixed stage prefix with an unknown retention entry, a non-namespace root entry, or a malformed manifest pool name, and require the exact typed source plus unchanged retained bytes. A disposable production mutation changing the unknown-entry refusal to a different namespace refusal made both diagnostic assertions fail by name. A separate mutation changing the preserved I/O kind to `InvalidInput` made the kind assertions fail. Deleting the root stage before admission while retaining the correct typed refusal made both exact-byte preservation assertions fail by name. These are filesystem outcomes, not census-count assertions. + +The corrected code passed focused retention laws in debug and release, the existing syscall synchronization regressions, full all-feature debug and release workspace suites, both Clippy configurations, source structure, formatting, and Markdown. Initial compilation caught a discarded must-use census, and Clippy caught test-helper style violations; those were corrected and are not runtime RED evidence. The [namespace admission decision](../adr/retention-recovery-namespace-admission.md) records ordering, alternatives, and remaining obligations. This correction establishes names-and-kinds refusal before effects; complete pool-content admission, capacity limits across all recovery transitions, and independent final-head approval remain open. + +## Pool stage identity before recovery publication + +Test commit `7d45af2` records real-filesystem RED on unfixed production `aa71bc0`. A byte-equal root pool replacement let recovery execute `FinalizeHead` before refusing at `RemoveRootStage`. A manifest replacement let it finalize the head and remove the root stage before refusing at `RemoveManifestStage`. Both regressions failed the named requirement that `HEAD` remain absent. The fixture confirms identical bytes on distinct device/inode coordinates while retaining the original stage hard link; a failed fixture setup is not accepted as RED evidence. + +Pool observation now verifies each complete stage's exact bytes and retained identity with the existing bounded, no-follow exact-record verifier. Absence remains `Absent`; kind, length, identity, or byte disagreement becomes `Different`, which the pure planner refuses through the existing pool-specific `PoolEntryDiffers`. The public observation documentation states the stage-object binding; logical identity and durable encoding are unchanged. The [pool identity decision](../adr/retention-recovery-pool-identity.md) records why byte-equal adoption and cleanup-time admission were rejected. + +The root diagnostic assertion failed when a disposable planner mutation reported the manifest pool instead. Deleting the retained root stage before planning while leaving the correct refusal intact made both preservation assertions fail by name. The unfixed production runs already demonstrated that both head-absence assertions fail when recovery actually commits the head. Mutated and fixed checkouts used separate Cargo target directories. + +Focused retention laws and the full all-feature workspace suites passed in debug and release; both Clippy configurations, formatting, source structure, and Markdown passed. Clippy initially flagged the fixture's inline tuple comparison; naming the two identity tuples clarified the intended comparison without changing its oracle. Observation-time inode admission does not prove safety against every later concurrent substitution, full predecessor/closure admission, or physical power loss; the remaining review queue and final independent approval remain binding. + +## Complete staged head binding + +Test commit `cd9e948` records real-filesystem RED on unfixed production `6852d41`: individually canonical, checksummed staged heads with a wrong manifest length or another predecessor reached `FinalizeHead`, removed both retained stages, and returned `Committed`. The named persistent-head preservation assertion failed for both the initial and successor fixtures. Record construction uses the checked semantic constructor and canonical encoder, so malformed framing or checksum is not the trigger. + +The pure planner now binds the head's digest, generation, exact manifest length, and predecessor to the staged manifest before any effects. Checked conversion of the admitted manifest byte length preserves arithmetic safety. The existing `HeadStageNamesOtherManifest` and `HeadPredecessorMismatch` variants identify the two new refusals; the independent published-history check remains in place. + +Property commit `f72727e` sweeps the complete bounded canonical length domain derived from entry counts and the format's 72-byte entry width. On unfixed `6852d41` planner code, it failed at the minimal tested length: head length 224 admitted a 296-byte manifest. The exact-length case remains accepted. Changing the fixed planner's committed outcome to `Clean` made that positive assertion fail by name; changing its refusals to another variant made both filesystem diagnostic assertions fail. Deleting a root stage after observation while preserving the correct refusal made both retained-byte assertions fail. Fixed, unfixed, and mutated Docker checkouts use separate build directories. + +The [head binding decision](../adr/retention-recovery-head-binding.md) records the boundary and alternatives. These checks do not establish exact preservation of unrelated successor entries, predecessor pool reopening, closure verification, or physical power-loss durability; those remain independent review obligations. + +Focused retention laws and the full all-feature workspace suites passed in debug and release, including the canonical-length sweep. Both Clippy configurations, formatting, source structure, and Markdown passed. The prior published `6852d41` also passed every hosted CI gate; the new published fix still requires its own current-head validation and independent final review. + +## Unrelated successor entries + +Test commit `f036229` records real-filesystem RED on unfixed production `514a521`. Dropping or altering an unrelated namespace from an otherwise canonical successor manifest let manifest-stage recovery execute `LinkManifest` and head-stage recovery execute `FinalizeHead` with cleanup. All four retained-byte preservation checks failed. The fixture keeps valid successor coordinates and, for head-stage cases, a matching canonical head and same-inode manifest pool link, isolating entry-set admission from earlier binding checks. + +The planner now compares all unrelated entries as ordered semantic streams before either publication path. Focused retention tests passed in debug and release; both workspace Clippy configurations, source structure, and formatting passed in the owned Docker sandbox. These results establish the exercised filesystem counterexamples; they do not claim a new full-workspace run or exhaustive generated-input coverage. Independent diagnostic calibration, current-head hosted validation, predecessor reopening, closure admission, and final independent approval remain outstanding. + +The [successor entry decision](../adr/retention-recovery-successor-entries.md) records the allocation-free comparison, already-committed cleanup distinction, and remaining blind spots. + +A disposable Docker copy of `48187bd` changed the manifest refusal to `RootNotSuccessor` while preserving the no-effect refusal path. Every successor-entry diagnostic assertion failed by name with the substituted variant; the retained-byte assertions passed. This independently calibrates the typed oracle rather than accepting any error as evidence. The mutant and fixed tree used separate Cargo build directories. + +A complete current-tree copy of `48187bd` passed both all-feature workspace suites in debug and release, including the OS-boundary synchronization laws, and formatting. The focused verification above additionally covered both Clippy configurations and source structure; Markdown passed after the accompanying documentation changes. These executions do not close the generated-input coverage or remaining recovery-admission gaps. + +## Selected predecessor reopening + +Regression commit `3cd7bde` records real-filesystem RED on production `48187bd` with documentation-only successor `8686931`. Missing, checksum-corrupt, and canonical substituted predecessor roots each reached `LinkRoot`, `LinkManifest`, or `FinalizeHead` with cleanup, depending on the staged prefix. Every tested transition failed the named retained-byte preservation assertion. The injected fault occurs after valid publication preparation and staging, so the failure demonstrates restart admission rather than fixture construction. + +The filesystem authority now reuses forward predecessor admission after pure planning and before executable-stage reopening. When the current manifest already selects the staged root, it instead reuses committed-root admission. Initial namespaces have no published predecessor. The [selected-root decision](../adr/retention-recovery-selected-root-admission.md) records ordering, alternatives, exact observation refusals, and remaining blind spots. + +Focused retention laws passed in debug and release. Clippy initially required borrowing the recovery evidence instead of passing it by value; correcting that signature left the admission behavior unchanged, and both Clippy configurations, source structure, and formatting passed. This style failure is excluded from runtime RED evidence. + +A separate Docker mutant changed missing and changed predecessor refusals to `PredecessorMismatch` while retaining the early no-effect refusal. Every fault-matrix diagnostic assertion failed by name, while every retained-byte assertion passed. Fixed and mutated checkouts used independent Cargo build directories. The parent RED runs independently calibrate preservation against actual unwanted publication effects. + +The complete changed tree passed both all-feature workspace suites in debug and release, including syscall synchronization checks, and Markdown. Hosted run `36991417726` for the earlier published `8686931` passed documentation, fuzz, and dependency checks but failed release Rust validation: `cleanup_terminates_the_entire_child_process_group` returned `UnexpectedEof` because its child exited before the readiness observer accepted the queued signal. This is a separate subprocess-fixture synchronization defect under investigation, not a retention GREEN result or permission to retry the gate into green. Current-head validation and independent final approval remain required. diff --git a/fuzz/fuzz_targets/retention_format.rs b/fuzz/fuzz_targets/retention_format.rs index b35808cc..3bf13c52 100644 --- a/fuzz/fuzz_targets/retention_format.rs +++ b/fuzz/fuzz_targets/retention_format.rs @@ -17,18 +17,21 @@ fuzz_target!(|bytes: &[u8]| { }); fn root(input: &[u8]) { + let _ = keep::assess_root_stage(Some(input)); if let Ok(root) = AdmittedRetentionRoot::decode(input) { assert_eq!(root.encoded(), input); } } fn manifest(input: &[u8]) { + let _ = keep::assess_manifest_stage(Some(input)); if let Ok(manifest) = AdmittedRetentionManifest::decode(input) { assert_eq!(manifest.encoded(), input); } } fn head(input: &[u8]) { + let _ = keep::assess_head_stage(Some(input)); if let Ok(head) = ChecksummedRetentionHead::decode(input) { assert_eq!(head.encoded(), input); } diff --git a/src/adapters/blob_id_binary.rs b/src/adapters/blob_id_binary.rs index 0e1b4ad7..1cee98bf 100644 --- a/src/adapters/blob_id_binary.rs +++ b/src/adapters/blob_id_binary.rs @@ -3,9 +3,9 @@ use super::blob_id_binary_error::BlobIdBinaryParseError; use crate::blob::{BlobId, BlobLength}; -const BINARY_MAGIC: [u8; 16] = *b"KEEP:BLOB:ID\0\0\0\0"; -const IDENTITY_VERSION: u16 = 1; -const HASH_ALGORITHM: u8 = 1; +pub(super) const BINARY_MAGIC: [u8; 16] = *b"KEEP:BLOB:ID\0\0\0\0"; +pub(super) const IDENTITY_VERSION: u16 = 1; +pub(super) const HASH_ALGORITHM: u8 = 1; const BINARY_BYTES: usize = 59; impl BlobId { diff --git a/src/adapters/filesystem_exact_record.rs b/src/adapters/filesystem_exact_record.rs index bda29df7..44707fab 100644 --- a/src/adapters/filesystem_exact_record.rs +++ b/src/adapters/filesystem_exact_record.rs @@ -251,7 +251,8 @@ pub(super) fn read_bounded_optional( Ok(Some(bytes)) } -fn open_read(directory: &Dir, name: &str) -> io::Result { +/// Opens a record read-only without following links or blocking. +pub(super) fn open_read(directory: &Dir, name: &str) -> io::Result { let mut options = OpenOptions::new(); options.read(true).follow(FollowSymlinks::No).nonblock(true); directory.open_with(name, &options) diff --git a/src/adapters/filesystem_version_two_admission.rs b/src/adapters/filesystem_version_two_admission.rs index 8acf6a42..3ef32471 100644 --- a/src/adapters/filesystem_version_two_admission.rs +++ b/src/adapters/filesystem_version_two_admission.rs @@ -3,7 +3,7 @@ use std::path::Path; use cap_fs_ext::DirExt; -#[cfg(test)] +#[cfg(any(test, feature = "repository-tasks"))] use cap_std::ambient_authority; use cap_std::fs::Dir; @@ -54,8 +54,18 @@ impl FilesystemVersionTwoAdmission { Self::admit(root) } - #[cfg(test)] - pub(super) fn reopen_unchecked_for_tests( + /// Reopens a migrated root without platform admission for repository tasks. + /// + /// The crash matrix and other repository tools run on hosts outside the + /// admitted Linux profile; every namespace, record, and identity law still + /// applies. Production callers use [`Self::reopen`]. + /// + /// # Errors + /// + /// Returns [`FilesystemPlatformAdmissionError`] exactly as [`Self::reopen`] + /// does for every boundary after platform admission. + #[cfg(any(test, feature = "repository-tasks"))] + pub fn reopen_unchecked_for_repository_tasks( store_root: &Path, ) -> Result { let root = Dir::open_ambient_dir(store_root, ambient_authority()) diff --git a/src/adapters/layout_id_binary.rs b/src/adapters/layout_id_binary.rs index b72667af..5d792890 100644 --- a/src/adapters/layout_id_binary.rs +++ b/src/adapters/layout_id_binary.rs @@ -3,9 +3,9 @@ use super::layout_id_binary_error::LayoutIdBinaryParseError; use crate::{LayoutId, LayoutRecordLength}; -const BINARY_MAGIC: [u8; 16] = *b"KEEP:LAYOUT:ID\0\0"; -const IDENTITY_VERSION: u16 = 1; -const LAYOUT_CODEC: u16 = 1; +pub(super) const BINARY_MAGIC: [u8; 16] = *b"KEEP:LAYOUT:ID\0\0"; +pub(super) const IDENTITY_VERSION: u16 = 1; +pub(super) const LAYOUT_CODEC: u16 = 1; const BINARY_BYTES: usize = 60; impl LayoutId { @@ -108,7 +108,9 @@ const fn validate_codec(observed: u16) -> Result<(), LayoutIdBinaryParseError> { }) } -fn validate_plan_length(observed: u64) -> Result { +pub(super) fn validate_plan_length( + observed: u64, +) -> Result { if !(LayoutRecordLength::MINIMUM.get()..=LayoutRecordLength::MAXIMUM.get()).contains(&observed) { return Err(LayoutIdBinaryParseError::PlanLengthOutOfBounds { diff --git a/src/adapters/retention.rs b/src/adapters/retention.rs index 334f250f..54f5afc0 100644 --- a/src/adapters/retention.rs +++ b/src/adapters/retention.rs @@ -15,16 +15,23 @@ mod closure_profile_error; mod closure_verifier; #[cfg(test)] mod filesystem_recovery_admission_tests; +#[cfg(test)] +mod filesystem_retention_anchor_order_prefix_tests; mod filesystem_retention_attempt; #[cfg(test)] mod filesystem_retention_attempt_tests; mod filesystem_retention_authority; mod filesystem_retention_authority_error; #[cfg(test)] +mod filesystem_retention_body_prefix_tests; +#[cfg(test)] mod filesystem_retention_capacity_tests; mod filesystem_retention_catalog; #[cfg(test)] mod filesystem_retention_catalog_tests; +mod filesystem_retention_closure_admission; +#[cfg(test)] +mod filesystem_retention_closure_prefix_tests; mod filesystem_retention_current; #[cfg(test)] mod filesystem_retention_current_tests; @@ -32,11 +39,84 @@ mod filesystem_retention_current_tests; mod filesystem_retention_expectation_tests; #[cfg(all(test, target_os = "linux"))] mod filesystem_retention_fifo_tests; +#[cfg(test)] +mod filesystem_retention_framing_prefix_tests; +#[cfg(test)] +mod filesystem_retention_generation_refusal_tests; +#[cfg(test)] +mod filesystem_retention_incomplete_disposition_tests; mod filesystem_retention_namespace; #[cfg(test)] mod filesystem_retention_namespace_tests; +#[cfg(test)] +mod filesystem_retention_observation_error_tests; +#[cfg(test)] +mod filesystem_retention_partial_anchor_tests; +#[cfg(test)] +mod filesystem_retention_partial_entry_tests; +#[cfg(test)] +mod filesystem_retention_partial_history_tests; +#[cfg(test)] +mod filesystem_retention_partial_integrity_tests; +#[cfg(test)] +mod filesystem_retention_partial_profile_tests; mod filesystem_retention_pool_name; +#[cfg(test)] +mod filesystem_retention_publication_closure_tests; +mod filesystem_retention_recovery; +#[cfg(test)] +mod filesystem_retention_recovery_closure_law_tests; +#[cfg(test)] +mod filesystem_retention_recovery_closure_tests; +#[cfg(test)] +mod filesystem_retention_recovery_directory_tests; +#[cfg(test)] +mod filesystem_retention_recovery_discard_prefix_tests; +#[cfg(test)] +mod filesystem_retention_recovery_entry_set_tests; +mod filesystem_retention_recovery_error; +#[cfg(test)] +mod filesystem_retention_recovery_head_binding_tests; +#[cfg(test)] +mod filesystem_retention_recovery_history_domain_tests; +#[cfg(test)] +mod filesystem_retention_recovery_history_tests; +#[cfg(test)] +mod filesystem_retention_recovery_identity_tests; +#[cfg(test)] +mod filesystem_retention_recovery_namespace_tests; +mod filesystem_retention_recovery_observation; +mod filesystem_retention_recovery_policy; +#[cfg(test)] +mod filesystem_retention_recovery_policy_tests; +#[cfg(test)] +mod filesystem_retention_recovery_predecessor_tests; +#[cfg(test)] +mod filesystem_retention_recovery_prefix_tests; +mod filesystem_retention_recovery_roots; +#[cfg(test)] +mod filesystem_retention_recovery_tests; mod filesystem_retention_refusal; +#[cfg(test)] +mod filesystem_retention_short_count_tests; +#[cfg(test)] +mod filesystem_retention_short_framing_tests; +#[cfg(test)] +mod filesystem_retention_short_head_tests; +#[cfg(test)] +mod filesystem_retention_short_history_tests; +#[cfg(test)] +mod filesystem_retention_short_namespace_tests; +#[cfg(test)] +mod filesystem_retention_short_policy_tests; +mod filesystem_retention_snapshot; +mod filesystem_retention_snapshot_error; +#[cfg(test)] +mod filesystem_retention_snapshot_error_tests; +#[cfg(test)] +mod filesystem_retention_snapshot_identity_tests; +#[cfg(test)] +mod filesystem_retention_snapshot_tests; mod filesystem_retention_stage; mod filesystem_retention_storage; #[cfg(test)] @@ -73,6 +153,8 @@ mod publication_preparation_error; mod publication_receipt; mod publication_storage; mod root_anchor_decoder; +mod root_anchor_order_prefix; +mod root_anchor_prefix; mod root_decode_error; mod root_decode_error_display; mod root_decoder; @@ -91,6 +173,42 @@ mod transition_preflight_error; mod transition_readiness; mod verified_closure; +#[cfg(test)] +mod filesystem_retention_forward_error_tests; +mod reader_attempt_limit; +mod reader_fence; +mod recovery_evidence; +mod recovery_execution; +#[cfg(test)] +mod recovery_execution_tests; +#[cfg(test)] +mod recovery_head_length_tests; +mod recovery_manifest_entries; +mod recovery_plan; +mod recovery_planner; +#[cfg(test)] +mod recovery_planner_tests; +mod recovery_refusal; +mod recovery_stage_assessment; +mod recovery_storage; +#[cfg(test)] +mod retention_model_tests; +mod retention_record_refusal; +mod retention_storage_error; +#[cfg(test)] +mod retention_storage_error_law_tests; +mod retention_storage_progress; +mod retention_view_collector; +#[cfg(test)] +mod retention_view_collector_tests; +mod stage_closure_limit_admission; +mod stage_fixed_field_admission; +mod stage_framing_prefix; +mod stage_generation_admission; +mod stage_history_admission; +mod stage_prefix_admission; +mod stage_record_integrity; +mod stage_root_policy_admission; pub use admitted_manifest::AdmittedRetentionManifest; pub use admitted_root::AdmittedRetentionRoot; pub use canonical_head::CanonicalRetentionHead; @@ -104,7 +222,10 @@ pub use filesystem_retention_authority_error::{ FilesystemRetentionAuthorityError, RetentionAuthorityDirectory, }; pub use filesystem_retention_current::ObservedRetentionState; +pub use filesystem_retention_recovery_error::FilesystemRetentionRecoveryError; pub use filesystem_retention_refusal::RetentionCurrentStateRefusal; +pub use filesystem_retention_snapshot::FilesystemRetentionSnapshot; +pub use filesystem_retention_snapshot_error::FilesystemRetentionSnapshotError; pub use head_decode_error::RetentionHeadDecodeError; pub use manifest_decode_error::RetentionManifestDecodeError; pub use manifest_encode_error::RetentionManifestEncodeError; @@ -118,6 +239,32 @@ pub use publication_preparation::prepare_retention_publication; pub use publication_preparation_error::RetentionPublicationPreparationError; pub use publication_receipt::RetentionPublicationReceipt; pub use publication_storage::RetentionPublicationStorage; +pub use reader_attempt_limit::ReaderAttemptLimit; +use reader_fence::ReaderFence; +pub use recovery_evidence::{ + RetentionPoolEntryObservation, RetentionPoolObservations, RetentionRecoveryEvidence, + RetentionStageAssessments, +}; +pub use recovery_execution::{ + RetentionRecoveryError, RetentionRecoveryReceipt, execute_retention_recovery, +}; +pub use recovery_plan::{RetentionRecoveryOutcome, RetentionRecoveryPlan, RetentionRecoveryStep}; +pub use recovery_planner::plan_retention_recovery; +pub use recovery_refusal::{RetentionFixedStage, RetentionPool, RetentionRecoveryRefusal}; +pub use recovery_stage_assessment::{ + RetentionHeadStageAssessment, RetentionManifestStageAssessment, RetentionRootStageAssessment, + RetentionStageAssessment, assess_head_stage, assess_manifest_stage, assess_root_stage, +}; +pub use recovery_storage::RetentionRecoveryStorage; +pub use retention_record_refusal::RetentionRecordRefusal; +pub use retention_storage_error::RetentionStorageError; +pub use retention_storage_progress::{ + RetentionEffectDurability, RetentionKnownEffect, RetentionNamespaceEffect, + RetentionStorageBoundary, RetentionStorageProgress, +}; +pub use retention_view_collector::{ + RetentionViewCoordinates, RetentionViewError, RetentionViewSource, collect_retention_view, +}; pub use root_decode_error::RetentionRootDecodeError; pub use root_encode_error::RetentionRootEncodeError; pub use transition_disposition::RetentionTransitionDisposition; diff --git a/src/adapters/retention/filesystem_retention_anchor_order_prefix_tests.rs b/src/adapters/retention/filesystem_retention_anchor_order_prefix_tests.rs new file mode 100644 index 00000000..9ae0865e --- /dev/null +++ b/src/adapters/retention/filesystem_retention_anchor_order_prefix_tests.rs @@ -0,0 +1,182 @@ +//! This module owns runtime ordering laws for incomplete root anchors. + +use super::filesystem_retention_test_fixture::{ + ROOT_HEX, fixture, open_authority, retention_witness, +}; +use super::{ + FilesystemRetentionRecoveryError, RetentionFixedStage, RetentionRecoveryRefusal, + RetentionRootDecodeError, RetentionStageAssessment, assess_root_stage, +}; +use std::{error::Error, fs}; + +struct Case { + name: &'static str, + prior: Vec, + next: Vec, + earliest: usize, +} + +// Size: medium. Oracle: anchors strictly order blob length/digest, then layout length/digest. +// Delete only when stronger public recovery laws subsume these decisive partial-order boundaries. +#[test] +fn impossible_partial_anchor_order_preserves_evidence() -> Result<(), Box> { + for case in cases()? { + for length in [case.earliest, 118] { + let bytes = prefix(&case.prior, &case.next, length)?; + let (sandbox, mut authority) = + open_authority(&format!("anchor-order-{}-{length}", case.name))?; + fs::write(sandbox.path().join("retention/root.next"), bytes)?; + let before = retention_witness(sandbox.path())?; + match authority.recover() { + Err(FilesystemRetentionRecoveryError::Plan { + source: RetentionRecoveryRefusal::StageCorrupt { stage, source }, + }) => { + assert_eq!( + stage, + RetentionFixedStage::Root, + "identify the corrupt anchor-order stage" + ); + assert!( + matches!( + source.downcast_ref::(), + Some(RetentionRootDecodeError::NonCanonicalAnchorOrder { index: 1 }) + ), + "report the exact impossible anchor index: {source:?}" + ); + } + result => { + return Err(format!( + "{} prefix {length} must refuse impossible order: {result:?}", + case.name + ) + .into()); + } + } + assert_eq!( + retention_witness(sandbox.path())?, + before, + "impossible anchor order must preserve retained evidence" + ); + } + } + Ok(()) +} + +// Size: small. Oracle: every prefix of a strictly greater complete anchor has that completion. +// Delete only when stronger public assessment laws subsume all lexicographic levels and lengths. +#[test] +fn ordered_anchors_admit_every_partial_length() -> Result<(), Box> { + for case in cases()?.into_iter().filter(|case| case.name != "exhausted") { + for length in 1..119 { + let bytes = prefix(&case.next, &case.prior, length)?; + let assessment = assess_root_stage(Some(&bytes)); + assert!( + matches!(assessment, RetentionStageAssessment::Truncated { .. }), + "{} ordered prefix {length} must admit completion: {assessment:?}", + case.name + ); + } + } + Ok(()) +} + +// Size: small. Oracle: once the most significant differing field forces descent, no suffix repairs it. +// Delete only when stronger runtime laws subsume the full decisive-prefix sweep. +#[test] +fn every_decisive_anchor_order_prefix_is_corrupt() -> Result<(), Box> { + for case in cases()? { + for length in case.earliest..119 { + let bytes = prefix(&case.prior, &case.next, length)?; + let assessment = assess_root_stage(Some(&bytes)); + assert!( + matches!( + assessment, + RetentionStageAssessment::Corrupt( + RetentionRootDecodeError::NonCanonicalAnchorOrder { index: 1 } + ) + ), + "{} descending prefix {length} must refuse: {assessment:?}", + case.name + ); + } + } + Ok(()) +} + +fn cases() -> Result, Box> { + let root = fixture(ROOT_HEX)?; + let anchor = root.get(195..314).ok_or("missing anchor")?; + let mut cases = Vec::new(); + for (name, start, end, prior_bytes, next_bytes, earliest) in [ + ( + "blob-length", + 19, + 27, + 0x8000_0000_0000_0000_u64.to_be_bytes().to_vec(), + 0x7fff_ffff_ffff_ffff_u64.to_be_bytes().to_vec(), + 20, + ), + ("blob-digest", 27, 59, vec![0x80; 32], vec![0x7f; 32], 28), + ( + "layout-length", + 79, + 87, + 46_137_520_u64.to_be_bytes().to_vec(), + 176_u64.to_be_bytes().to_vec(), + 84, + ), + ("layout-digest", 87, 119, vec![0x80; 32], vec![0x7f; 32], 88), + ] { + let mut prior = anchor.to_vec(); + let mut next = anchor.to_vec(); + prior + .get_mut(start..end) + .ok_or("missing prior field")? + .copy_from_slice(&prior_bytes); + next.get_mut(start..end) + .ok_or("missing next field")? + .copy_from_slice(&next_bytes); + cases.push(Case { + name, + prior, + next, + earliest, + }); + } + let mut maximum = anchor.to_vec(); + maximum + .get_mut(19..59) + .ok_or("missing blob coordinate")? + .fill(255); + maximum + .get_mut(79..87) + .ok_or("missing layout length")? + .copy_from_slice(&46_137_520_u64.to_be_bytes()); + maximum + .get_mut(87..119) + .ok_or("missing layout digest")? + .fill(255); + cases.push(Case { + name: "exhausted", + prior: maximum.clone(), + next: maximum, + earliest: 1, + }); + Ok(cases) +} + +fn prefix(prior: &[u8], next: &[u8], length: usize) -> Result, Box> { + let root = fixture(ROOT_HEX)?; + let mut prefix = root.get(..195).ok_or("missing header")?.to_vec(); + prefix + .get_mut(24..32) + .ok_or("missing length")? + .copy_from_slice(&497_u64.to_be_bytes()); + prefix + .get_mut(44..48) + .ok_or("missing count")? + .copy_from_slice(&2_u32.to_be_bytes()); + prefix.extend_from_slice(prior); + prefix.extend_from_slice(next.get(..length).ok_or("missing anchor prefix")?); + Ok(prefix) +} diff --git a/src/adapters/retention/filesystem_retention_attempt_tests.rs b/src/adapters/retention/filesystem_retention_attempt_tests.rs index 01c5cec5..3fcdfc4b 100644 --- a/src/adapters/retention/filesystem_retention_attempt_tests.rs +++ b/src/adapters/retention/filesystem_retention_attempt_tests.rs @@ -13,6 +13,8 @@ use super::{ RetentionPublicationStorage, RetentionTransitionDisposition, }; +// Size: medium. Oracle: a complete head without its manifest refuses with HeadStageWithoutManifestStage. +// Delete when this recovery state is removed or stronger public-boundary coverage subsumes it. #[test] fn refused_verification_admits_no_later_phase() -> Result<(), Box> { let (sandbox, mut authority) = open_authority("filesystem-retention-attempt-refused")?; @@ -20,18 +22,23 @@ fn refused_verification_admits_no_later_phase() -> Result<(), Box> { let preparation = initial_preparation(&root_bytes)?; fs::write( sandbox.path().join("retention").join("head.next"), - b"retained", + fixture(super::filesystem_retention_test_fixture::HEAD_HEX)?, )?; let before = retention_witness(sandbox.path())?; let error = authority .verify_current(&preparation) .err() - .ok_or("retained head stage was admitted")?; - assert!(matches!( - refusal(&error), - Some(RetentionCurrentStateRefusal::RetainedStage) - )); + .ok_or("an ambiguous head stage was admitted")?; + assert!( + matches!( + refusal(&error), + Some(RetentionCurrentStateRefusal::RecoveryRefused { + source: super::RetentionRecoveryRefusal::HeadStageWithoutManifestStage + }) + ), + "head-only evidence must report its missing manifest: {error:?}" + ); let error = authority .write_root_stage(preparation.candidate()) .err() diff --git a/src/adapters/retention/filesystem_retention_authority.rs b/src/adapters/retention/filesystem_retention_authority.rs index ada4ed58..36d61efa 100644 --- a/src/adapters/retention/filesystem_retention_authority.rs +++ b/src/adapters/retention/filesystem_retention_authority.rs @@ -9,6 +9,7 @@ use super::filesystem_retention_authority_error::{ FilesystemRetentionAuthorityError as Error, RetentionAuthorityDirectory as Directory, }; use super::filesystem_retention_current::{self, ObservedRetentionState}; +use super::filesystem_retention_recovery::RetentionRecoveryContext; use crate::adapters::{FilesystemVersionTwoAdmission, FilesystemWriterLock}; /// Exclusive authority to publish retention transitions on one pinned root. @@ -32,6 +33,9 @@ pub struct FilesystemRetentionPublicationAuthority { pub(super) roots: Dir, pub(super) manifests: Dir, pub(super) attempt: Option, + pub(super) recovery: Option, + #[cfg(test)] + pub(super) recovery_sync_failure: Option, _lock: FilesystemWriterLock, } @@ -68,6 +72,9 @@ impl FilesystemRetentionPublicationAuthority { roots, manifests, attempt: None, + recovery: None, + #[cfg(test)] + recovery_sync_failure: None, _lock: lock, }) } diff --git a/src/adapters/retention/filesystem_retention_body_prefix_tests.rs b/src/adapters/retention/filesystem_retention_body_prefix_tests.rs new file mode 100644 index 00000000..5852c1b7 --- /dev/null +++ b/src/adapters/retention/filesystem_retention_body_prefix_tests.rs @@ -0,0 +1,171 @@ +//! This module owns preservation of invalid complete entries in interrupted retention bodies. + +use super::filesystem_retention_test_fixture::{ + MANIFEST_HEX, ROOT_HEX, drive_publication, fixture, initial_preparation, open_authority, + retention_witness, +}; +use super::{ + FilesystemRetentionRecoveryError, RetentionFixedStage as Stage, RetentionManifestDecodeError, + RetentionRecoveryRefusal, RetentionRootDecodeError, +}; +use crate::{BlobIdBinaryParseError, LayoutIdBinaryParseError, RootGenerationError}; +use std::{error::Error, fs}; + +#[derive(Clone, Copy, Debug)] +enum Fault { + Blob, + Layout, + Generation, + Order, +} + +// Size: medium. Oracle: complete identities and root generations obey their public admission rules. +// Delete only when stronger recovery coverage subsumes incomplete bodies containing invalid entries. +#[test] +fn invalid_complete_entries_in_short_bodies_preserve_evidence() -> Result<(), Box> { + for (stage, fault) in [ + (Stage::Root, Fault::Blob), + (Stage::Root, Fault::Layout), + (Stage::Manifest, Fault::Generation), + ] { + require_preservation(stage, fault)?; + } + Ok(()) +} + +// Size: medium. Oracle: root anchors and manifest namespaces are strictly ordered without duplicates. +// Delete only when stronger recovery coverage subsumes duplicate entries before body completion. +#[test] +fn duplicate_entries_in_short_bodies_preserve_evidence() -> Result<(), Box> { + for stage in [Stage::Root, Stage::Manifest] { + require_preservation(stage, Fault::Order)?; + } + Ok(()) +} + +fn malformed_prefix(stage: Stage, fault: Fault) -> Result, Box> { + let (corpus, width) = match stage { + Stage::Root => (ROOT_HEX, 119_usize), + _ => (MANIFEST_HEX, 72), + }; + let bytes = fixture(corpus)?; + let body_end = bytes.len().checked_sub(64).ok_or("missing trailer")?; + let body_start = body_end.checked_sub(width).ok_or("missing body")?; + let mut prefix = bytes.get(..body_end).ok_or("missing prefix")?.to_vec(); + let count = if matches!(fault, Fault::Order) { + 3_u32 + } else { + 2 + }; + if matches!(fault, Fault::Order) { + prefix.extend_from_slice(bytes.get(body_start..body_end).ok_or("missing entry")?); + } + let extra = usize::try_from(count.checked_sub(1).ok_or("count underflow")?)? + .checked_mul(width) + .ok_or("length overflow")?; + let total = u64::try_from(bytes.len().checked_add(extra).ok_or("length overflow")?)?; + prefix + .get_mut(24..32) + .ok_or("missing length")? + .copy_from_slice(&total.to_be_bytes()); + prefix + .get_mut(44..48) + .ok_or("missing count")? + .copy_from_slice(&count.to_be_bytes()); + match fault { + Fault::Blob => *prefix.get_mut(body_start).ok_or("missing blob magic")? = 0, + Fault::Layout => { + *prefix + .get_mut(body_start.checked_add(59).ok_or("offset overflow")?) + .ok_or("missing layout magic")? = 0; + } + Fault::Generation => { + let start = body_start.checked_add(32).ok_or("offset overflow")?; + let end = start.checked_add(8).ok_or("offset overflow")?; + prefix + .get_mut(start..end) + .ok_or("missing generation")? + .fill(0); + } + Fault::Order => {} + } + Ok(prefix) +} + +fn require_preservation(stage: Stage, fault: Fault) -> Result<(), Box> { + let (name, phases) = match stage { + Stage::Root => ("root.next", 1), + _ => ("manifest.next", 7), + }; + let (sandbox, mut authority) = open_authority(&format!("body-prefix-{name}-{fault:?}"))?; + let root = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root)?; + drive_publication(&mut authority, &preparation, phases)?; + fs::write( + sandbox.path().join("retention").join(name), + malformed_prefix(stage, fault)?, + )?; + let before = retention_witness(sandbox.path())?; + match authority.recover() { + Err(FilesystemRetentionRecoveryError::Plan { + source: + RetentionRecoveryRefusal::StageCorrupt { + stage: actual, + source, + }, + }) => { + assert_eq!(actual, stage, "identify the invalid body stage"); + require_cause(source.as_ref(), stage, fault); + } + result => { + return Err( + format!("{name} {fault:?} must refuse before body completion: {result:?}").into(), + ); + } + } + assert_eq!( + retention_witness(sandbox.path())?, + before, + "invalid body entry must preserve retained evidence" + ); + Ok(()) +} + +fn require_cause(source: &(dyn Error + 'static), stage: Stage, fault: Fault) { + let mut blob_magic = *b"KEEP:BLOB:ID\0\0\0\0"; + let mut layout_magic = *b"KEEP:LAYOUT:ID\0\0"; + if let Some(first) = blob_magic.first_mut() { + *first = 0; + } + if let Some(first) = layout_magic.first_mut() { + *first = 0; + } + let exact = match (stage, fault) { + (Stage::Root, Fault::Blob) => { + matches!(source.downcast_ref::(), Some(RetentionRootDecodeError::BlobId { index: 0, source: BlobIdBinaryParseError::InvalidMagic { observed } }) if *observed == blob_magic) + } + (Stage::Root, Fault::Layout) => { + matches!(source.downcast_ref::(), Some(RetentionRootDecodeError::LayoutId { index: 0, source: LayoutIdBinaryParseError::InvalidMagic { observed } }) if *observed == layout_magic) + } + (Stage::Root, Fault::Order) => matches!( + source.downcast_ref::(), + Some(RetentionRootDecodeError::NonCanonicalAnchorOrder { index: 1 }) + ), + (Stage::Manifest, Fault::Generation) => matches!( + source.downcast_ref::(), + Some(RetentionManifestDecodeError::RootGeneration { + index: 0, + source: RootGenerationError::Zero + }) + ), + (Stage::Manifest, Fault::Order) => matches!( + source.downcast_ref::(), + Some(RetentionManifestDecodeError::NonCanonicalEntryOrder { index: 1 }) + ), + _ => false, + }; + assert!( + exact, + "report exact body-entry violation for {stage:?} {fault:?}: {source:?}" + ); +} diff --git a/src/adapters/retention/filesystem_retention_catalog.rs b/src/adapters/retention/filesystem_retention_catalog.rs index 446f8534..a85c82b6 100644 --- a/src/adapters/retention/filesystem_retention_catalog.rs +++ b/src/adapters/retention/filesystem_retention_catalog.rs @@ -21,9 +21,10 @@ const HEAD_LENGTH: usize = crate::adapters::publication_head_decoder::ENCODED_LE /// lost a pool entry. Publication must not proceed unless this store's own /// `HEAD` names exactly that catalog generation and digest and the selected /// catalog pool entry reopens under this authority, bounded by the head's -/// declared length, and decodes to that generation and digest. Closure-member -/// segments are not re-read here: every read authenticates them, and their -/// re-verification under authority belongs to retention recovery. +/// declared length, and decodes to that generation and digest. Live closure +/// verification then reloads all selected segments with the explicit default +/// catalog byte policy and replays the candidate anchors under writer authority. +/// The fresh proof must remain bound to the prepared catalog coordinates. pub(super) fn require_current_catalog( root: &Dir, preparation: &RetentionPublicationPreparation<'_>, @@ -47,7 +48,23 @@ pub(super) fn require_current_catalog( } .into_io()); } - require_selected_catalog(root, head) + require_selected_catalog(root, head)?; + let policy = super::filesystem_retention_recovery_policy::default_catalog_policy()?; + let live = super::filesystem_retention_closure_admission::verify( + root, + preparation.candidate().root(), + policy, + )?; + if live.catalog_generation() != expected_generation + || live.catalog_digest() != closure.catalog_digest() + { + return Err(RetentionCurrentStateRefusal::CatalogDisagreed { + expected_generation, + observed_generation: Some(live.catalog_generation()), + } + .into_io()); + } + Ok(()) } /// Reopens the catalog pool entry `head` selects and requires it to be that catalog. diff --git a/src/adapters/retention/filesystem_retention_closure_admission.rs b/src/adapters/retention/filesystem_retention_closure_admission.rs new file mode 100644 index 00000000..d6e208e4 --- /dev/null +++ b/src/adapters/retention/filesystem_retention_closure_admission.rs @@ -0,0 +1,37 @@ +//! This module owns capability-relative live closure admission under retention authority. + +use std::io; + +use cap_std::fs::Dir; + +use super::{ + RetentionRecoveryEvidence, RetentionStageAssessment, VerifiedRetentionClosure, + verify_retention_closure, +}; +use crate::RetentionRoot; +use crate::adapters::{CatalogRestartPolicy, catalog_restart_loader}; + +pub(super) fn admit_recovery( + root: &Dir, + evidence: &RetentionRecoveryEvidence<'_, '_>, + policy: CatalogRestartPolicy, +) -> io::Result<()> { + let RetentionStageAssessment::Complete(candidate) = &evidence.stages().root else { + return Ok(()); + }; + verify(root, candidate.root(), policy).map(|_verified| ()) +} + +pub(super) fn verify( + root: &Dir, + candidate: &RetentionRoot, + policy: CatalogRestartPolicy, +) -> io::Result { + let loaded = catalog_restart_loader::load_from_directory(root, "HEAD", policy) + .map_err(|source| io::Error::new(io::ErrorKind::InvalidData, source))?; + let snapshot = loaded + .snapshot() + .map_err(|source| io::Error::new(io::ErrorKind::InvalidData, source))?; + verify_retention_closure(candidate, &snapshot) + .map_err(|source| io::Error::new(io::ErrorKind::InvalidData, source)) +} diff --git a/src/adapters/retention/filesystem_retention_closure_prefix_tests.rs b/src/adapters/retention/filesystem_retention_closure_prefix_tests.rs new file mode 100644 index 00000000..0d7e2ce7 --- /dev/null +++ b/src/adapters/retention/filesystem_retention_closure_prefix_tests.rs @@ -0,0 +1,128 @@ +//! This module owns refusal and preservation of impossible closure-limit prefixes. + +use super::filesystem_retention_test_fixture::{ + ROOT_HEX, fixture, open_authority, retention_witness, +}; +use super::{ + FilesystemRetentionRecoveryError, RetentionFixedStage, RetentionRecoveryRefusal, + RetentionRootDecodeError, +}; +use crate::{RetentionClosureLimit as Limit, RetentionClosureLimitError as LimitError}; +use std::{error::Error, fs}; + +// Size: medium. Oracle: v2 closure limits are positive bounded big-endian integers. +// Delete only when stronger recovery coverage subsumes early complete and partial limit refusals. +#[test] +fn impossible_closure_prefixes_preserve_evidence() -> Result<(), Box> { + for (offset, width, limit, maximum) in [ + (88_usize, 8_usize, Limit::Nodes, 1_048_576_u64), + (96, 2, Limit::Depth, 8), + (100, 8, Limit::EncodedBytes, 16_777_216), + (108, 8, Limit::PhysicalBytes, 1_073_741_824), + ] { + for available in 1..width { + let mut bytes = vec![0; width]; + // A leading 0xff forces every completion above each specified ceiling. + *bytes.first_mut().ok_or("empty limit")? = 255; + let minimum = if width == 2 { + 0xff00 + } else { + 0xff00_0000_0000_0000 + }; + let expected = RetentionRootDecodeError::ClosureLimitPrefixAboveMaximum { + limit, + minimum, + maximum, + }; + require_refusal( + offset, + bytes.get(..available).ok_or("missing prefix")?, + &expected, + )?; + } + for observed in [0, maximum.checked_add(1).ok_or("boundary overflow")?] { + let raw = observed.to_be_bytes(); + let start = raw.len().checked_sub(width).ok_or("invalid width")?; + let source = if observed == 0 { + LimitError::Zero { limit } + } else { + LimitError::AboveMaximum { + limit, + maximum, + observed, + } + }; + require_refusal( + offset, + raw.get(start..).ok_or("missing field")?, + &RetentionRootDecodeError::ClosureLimit { source }, + )?; + } + } + Ok(()) +} + +fn require_refusal( + offset: usize, + field: &[u8], + expected: &RetentionRootDecodeError, +) -> Result<(), Box> { + let end = offset.checked_add(field.len()).ok_or("offset overflow")?; + let root = fixture(ROOT_HEX)?; + let mut partial = root.get(..end).ok_or("missing fixture bytes")?.to_vec(); + partial + .get_mut(offset..end) + .ok_or("missing field")? + .copy_from_slice(field); + let (sandbox, mut authority) = open_authority(&format!("closure-prefix-{offset}-{end}"))?; + fs::write(sandbox.path().join("retention/root.next"), partial)?; + let before = retention_witness(sandbox.path())?; + + match authority.recover() { + Err(FilesystemRetentionRecoveryError::Plan { + source: RetentionRecoveryRefusal::StageCorrupt { stage, source }, + }) => { + assert_eq!( + stage, + RetentionFixedStage::Root, + "identify the invalid closure stage" + ); + require_cause(source.as_ref(), expected); + } + result => { + return Err(format!("closure prefix {offset}..{end} must refuse: {result:?}").into()); + } + } + assert_eq!( + retention_witness(sandbox.path())?, + before, + "closure prefix {offset}..{end} must preserve retained evidence" + ); + Ok(()) +} + +fn require_cause(source: &(dyn Error + 'static), expected: &RetentionRootDecodeError) { + let exact = match (source.downcast_ref::(), expected) { + ( + Some(RetentionRootDecodeError::ClosureLimitPrefixAboveMaximum { + limit, + minimum, + maximum, + }), + RetentionRootDecodeError::ClosureLimitPrefixAboveMaximum { + limit: expected_limit, + minimum: expected_minimum, + maximum: expected_maximum, + }, + ) => limit == expected_limit && minimum == expected_minimum && maximum == expected_maximum, + ( + Some(RetentionRootDecodeError::ClosureLimit { source }), + RetentionRootDecodeError::ClosureLimit { source: expected }, + ) => source == expected, + _ => false, + }; + assert!( + exact, + "report exact closure-prefix cause: expected {expected:?}, observed {source:?}" + ); +} diff --git a/src/adapters/retention/filesystem_retention_current.rs b/src/adapters/retention/filesystem_retention_current.rs index fd22ae2a..e12a79ce 100644 --- a/src/adapters/retention/filesystem_retention_current.rs +++ b/src/adapters/retention/filesystem_retention_current.rs @@ -57,6 +57,24 @@ impl ObservedRetentionState { } } +#[cfg(test)] +impl ObservedRetentionState { + /// Builds the observed state from exact head and manifest bytes. + pub(super) fn for_tests(head: &[u8], manifest: &[u8]) -> io::Result { + let decoded = ChecksummedRetentionHead::decode(head) + .map_err(|source| RetentionCurrentStateRefusal::HeadRefused { source }.into_io())?; + let admitted = AdmittedRetentionManifest::decode(manifest) + .map_err(|source| RetentionCurrentStateRefusal::ManifestRefused { source }.into_io())?; + require_binding(decoded.head(), &admitted)?; + Ok(Self { + head: Box::from(head), + manifest: Box::from(manifest), + decoded_head: *decoded.head(), + decoded_manifest: admitted.manifest().clone(), + }) + } +} + /// The verified relationship between one preparation and the observed state. #[derive(Clone, Copy)] pub(super) enum ObservedDisposition<'state> { @@ -97,14 +115,7 @@ pub(super) fn observe( .ok_or_else(|| RetentionCurrentStateRefusal::ManifestAbsent.into_io())?; let admitted = AdmittedRetentionManifest::decode(&manifest) .map_err(|source| RetentionCurrentStateRefusal::ManifestRefused { source }.into_io())?; - if admitted.digest() != selected.manifest_digest() - || admitted.manifest().generation() != selected.generation() - { - return Err(RetentionCurrentStateRefusal::ManifestDisagreed.into_io()); - } - if admitted.manifest().predecessor() != selected.predecessor() { - return Err(RetentionCurrentStateRefusal::HeadPredecessorDisagreed.into_io()); - } + require_binding(selected, &admitted)?; let decoded_head = *selected; let decoded_manifest = admitted.manifest().clone(); Ok(Some(ObservedRetentionState { @@ -292,3 +303,24 @@ pub(super) fn read_exact_optional( .into_io()), } } + +/// Keeps production observation and byte-built recovery fixtures under one contract. +fn require_binding( + selected: &RetentionHead, + admitted: &AdmittedRetentionManifest<'_>, +) -> io::Result<()> { + let length = usize::try_from(selected.manifest_length().get()) + .map_err(|_source| RetentionCurrentStateRefusal::RecordLengthOverflow.into_io())?; + if admitted.encoded().len() != length { + return Err(RetentionCurrentStateRefusal::RecordKindOrLength.into_io()); + } + if admitted.digest() != selected.manifest_digest() + || admitted.manifest().generation() != selected.generation() + { + return Err(RetentionCurrentStateRefusal::ManifestDisagreed.into_io()); + } + if admitted.manifest().predecessor() != selected.predecessor() { + return Err(RetentionCurrentStateRefusal::HeadPredecessorDisagreed.into_io()); + } + Ok(()) +} diff --git a/src/adapters/retention/filesystem_retention_fifo_tests.rs b/src/adapters/retention/filesystem_retention_fifo_tests.rs index 8bbbc2a7..2aa951ab 100644 --- a/src/adapters/retention/filesystem_retention_fifo_tests.rs +++ b/src/adapters/retention/filesystem_retention_fifo_tests.rs @@ -35,7 +35,7 @@ fn a_fifo_at_the_format_marker_refuses_instead_of_blocking() -> Result<(), Box Result<(), Box> { + let (sandbox, mut authority) = open_authority("forward-stage-typed-refusal")?; + let root = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root)?; + drive_publication(&mut authority, &preparation, 2)?; + let path = sandbox.path().join("retention/root.next"); + let mut bytes = fs::read(&path)?; + *bytes.last_mut().ok_or("empty root stage")? ^= 1; + fs::write(path, bytes)?; + let before = retention_witness(sandbox.path())?; + + let error = RetentionPublicationStorage::synchronize_root_stage(&mut authority) + .err() + .ok_or("changed forward stage was synchronized successfully")?; + + let storage = error + .get_ref() + .and_then(|source| source.downcast_ref::()) + .ok_or("forward failure omitted storage cause")?; + let RetentionStorageError::Operation { source, .. } = storage else { + return Err("forward stage failure omitted operation boundary".into()); + }; + assert!( + matches!( + source.as_ref(), + RetentionStorageError::Refused { + source: RetentionRecordRefusal::Bytes + } + ), + "forward I/O failure must retain the typed byte refusal: {error:?}" + ); + assert_eq!( + retention_witness(sandbox.path())?, + before, + "forward refusal must preserve retained evidence" + ); + Ok(()) +} diff --git a/src/adapters/retention/filesystem_retention_framing_prefix_tests.rs b/src/adapters/retention/filesystem_retention_framing_prefix_tests.rs new file mode 100644 index 00000000..990d1457 --- /dev/null +++ b/src/adapters/retention/filesystem_retention_framing_prefix_tests.rs @@ -0,0 +1,244 @@ +//! This module owns runtime refusal of incomplete, impossible retention size fields. + +use super::filesystem_retention_test_fixture::{ + HEAD_HEX, MANIFEST_HEX, ROOT_HEX, drive_publication, fixture, initial_preparation, + open_authority, retention_witness, +}; +use super::{ + FilesystemRetentionRecoveryError, RetentionFixedStage as Stage, RetentionHeadDecodeError, + RetentionManifestDecodeError, RetentionRecoveryRefusal, RetentionRootDecodeError, + RetentionStageAssessment, assess_head_stage, assess_manifest_stage, assess_root_stage, +}; +use std::{error::Error, fs}; + +type StagePrefix = (Stage, Vec); + +// Size: medium. Oracle: canonical framing has bounded counts, namespace lengths and entry widths. +// Delete only when stronger runtime laws subsume impossible-prefix refusal and evidence preservation. +#[test] +fn impossible_framing_prefixes_preserve_evidence() -> Result<(), Box> { + for (index, (stage, bytes)) in impossible_prefixes()?.into_iter().enumerate() { + let (name, phases) = match stage { + Stage::Root => ("root.next", 1), + Stage::Manifest => ("manifest.next", 7), + Stage::Head => ("head.next", 11), + }; + let (sandbox, mut authority) = open_authority(&format!("framing-prefix-{index}"))?; + let root = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root)?; + drive_publication(&mut authority, &preparation, phases)?; + fs::write(sandbox.path().join("retention").join(name), &bytes)?; + let before = retention_witness(sandbox.path())?; + match authority.recover() { + Err(FilesystemRetentionRecoveryError::Plan { + source: + RetentionRecoveryRefusal::StageCorrupt { + stage: actual, + source, + }, + }) => { + assert_eq!(actual, stage, "identify the impossible framing stage"); + require_cause(source.as_ref(), stage, bytes.len()); + } + result => { + return Err( + format!("{name} impossible prefix {index} must refuse: {result:?}").into(), + ); + } + } + assert_eq!( + retention_witness(sandbox.path())?, + before, + "framing refusal must preserve retained evidence" + ); + } + Ok(()) +} + +fn impossible_prefixes() -> Result, Box> { + let mut cases = Vec::new(); + // A nonzero high length byte exceeds every format ceiling at each partial endpoint. + for (stage, corpus, offset) in [ + (Stage::Root, ROOT_HEX, 24_usize), + (Stage::Manifest, MANIFEST_HEX, 24), + (Stage::Head, HEAD_HEX, 32), + ] { + for available in 1..8 { + let end = offset.checked_add(available).ok_or("prefix overflow")?; + let mut bytes = fixture(corpus)?; + *bytes.get_mut(offset).ok_or("length byte absent")? = 1; + bytes.truncate(end); + cases.push((stage, bytes)); + } + } + for (stage, corpus, declared) in [ + (Stage::Root, ROOT_HEX, 256_u64), + (Stage::Root, ROOT_HEX, 7_799_296), + (Stage::Manifest, MANIFEST_HEX, 223), + (Stage::Manifest, MANIFEST_HEX, 225), + (Stage::Manifest, MANIFEST_HEX, 295_137), + ] { + let mut bytes = fixture(corpus)?; + bytes + .get_mut(24..32) + .ok_or("length absent")? + .copy_from_slice(&declared.to_be_bytes()); + bytes.truncate(32); + cases.push((stage, bytes)); + } + let mut namespace = fixture(ROOT_HEX)?; + *namespace.get_mut(40).ok_or("namespace absent")? = 1; + namespace.truncate(41); + cases.push((Stage::Root, namespace)); + // The declared length requires one item; any available high count byte excludes it. + for (stage, corpus) in [(Stage::Root, ROOT_HEX), (Stage::Manifest, MANIFEST_HEX)] { + for end in 45_usize..48 { + let mut bytes = fixture(corpus)?; + let offset = end.checked_sub(1).ok_or("empty count")?; + *bytes.get_mut(offset).ok_or("count absent")? = 1; + bytes.truncate(end); + cases.push((stage, bytes)); + } + } + let mut namespace = fixture(ROOT_HEX)?; + namespace + .get_mut(40..42) + .ok_or("namespace absent")? + .copy_from_slice(&4_u16.to_be_bytes()); + namespace.truncate(42); + cases.push((Stage::Root, namespace)); + cases.extend(insufficient_count_prefixes()?); + let mut head = fixture(HEAD_HEX)?; + head.get_mut(32..40) + .ok_or("length absent")? + .copy_from_slice(&295_168_u64.to_be_bytes()); + head.truncate(39); + cases.push((Stage::Head, head)); + Ok(cases) +} + +fn insufficient_count_prefixes() -> Result, Box> { + let mut cases = Vec::new(); + // A zero three-byte count prefix admits at most 255 items, below the required 256. + for (stage, corpus, total) in [ + (Stage::Root, ROOT_HEX, 30_723_u64), + (Stage::Manifest, MANIFEST_HEX, 18_656), + ] { + let mut bytes = fixture(corpus)?; + bytes + .get_mut(24..32) + .ok_or("length absent")? + .copy_from_slice(&total.to_be_bytes()); + bytes.truncate(47); + cases.push((stage, bytes)); + } + Ok(cases) +} + +// Size: small. Oracle: every generated legal framing tuple supplies a completion for its prefixes. +// Delete only when stronger public assessment sweeps subsume legal counts and namespace boundaries. +#[test] +fn canonical_root_framing_remains_possible_at_every_header_prefix() -> Result<(), Box> { + let mut root = fixture(ROOT_HEX)?; + for count in 0_u32..=65_536 { + for namespace in [1_u16, 3, 118, 119, 120, 255] { + let total = 256_u64 + .checked_add(u64::from(namespace)) + .and_then(|n| n.checked_add(u64::from(count).checked_mul(119)?)) + .ok_or("root length overflow")?; + root.get_mut(24..32) + .ok_or("length absent")? + .copy_from_slice(&total.to_be_bytes()); + root.get_mut(40..42) + .ok_or("namespace absent")? + .copy_from_slice(&namespace.to_be_bytes()); + root.get_mut(44..48) + .ok_or("count absent")? + .copy_from_slice(&count.to_be_bytes()); + for end in 24..48 { + assert!( + matches!( + assess_root_stage(root.get(..end)), + RetentionStageAssessment::Truncated { .. } + ), + "legal root framing count {count}, namespace {namespace}, prefix {end} must remain possible" + ); + } + } + } + Ok(()) +} + +// Size: small. Oracle: the complete legal manifest-length domain supplies canonical completions. +// Delete only when stronger public assessment laws subsume manifest and head length feasibility. +#[test] +fn canonical_manifest_lengths_remain_possible_at_every_header_prefix() -> Result<(), Box> +{ + let mut manifest = fixture(MANIFEST_HEX)?; + let mut head = fixture(HEAD_HEX)?; + for count in 0_u32..=4_096 { + let total = 224_u64 + .checked_add( + u64::from(count) + .checked_mul(72) + .ok_or("entry length overflow")?, + ) + .ok_or("manifest length overflow")?; + manifest + .get_mut(24..32) + .ok_or("length absent")? + .copy_from_slice(&total.to_be_bytes()); + manifest + .get_mut(44..48) + .ok_or("count absent")? + .copy_from_slice(&count.to_be_bytes()); + head.get_mut(32..40) + .ok_or("length absent")? + .copy_from_slice(&total.to_be_bytes()); + for end in 24..48 { + assert!( + matches!( + assess_manifest_stage(manifest.get(..end)), + RetentionStageAssessment::Truncated { .. } + ), + "legal manifest count {count}, prefix {end} must remain possible" + ); + } + for end in 32..40 { + assert!( + matches!( + assess_head_stage(head.get(..end)), + RetentionStageAssessment::Truncated { .. } + ), + "legal head length {total}, prefix {end} must remain possible" + ); + } + } + Ok(()) +} + +fn require_cause(source: &(dyn Error + 'static), stage: Stage, length: usize) { + let observed = match stage { + Stage::Root => match source.downcast_ref::() { + Some(RetentionRootDecodeError::FramingPrefixImpossible { observed }) => Some(*observed), + _ => None, + }, + Stage::Manifest => match source.downcast_ref::() { + Some(RetentionManifestDecodeError::FramingPrefixImpossible { observed }) => { + Some(*observed) + } + _ => None, + }, + Stage::Head => match source.downcast_ref::() { + Some(RetentionHeadDecodeError::ManifestLengthPrefixImpossible { observed }) => { + Some(*observed) + } + _ => None, + }, + }; + assert_eq!( + observed, + Some(length), + "report impossible framing at the actual interrupted length: {source:?}" + ); +} diff --git a/src/adapters/retention/filesystem_retention_generation_refusal_tests.rs b/src/adapters/retention/filesystem_retention_generation_refusal_tests.rs new file mode 100644 index 00000000..9b8831b6 --- /dev/null +++ b/src/adapters/retention/filesystem_retention_generation_refusal_tests.rs @@ -0,0 +1,57 @@ +//! Complete invalid generation fields cannot authorize stage evidence deletion. +//! Size: medium; oracle: positive generation and mutation-free refusal contracts. +//! Delete only if a stronger public recovery test subsumes this preservation law. + +use std::error::Error; +use std::fs; + +use super::filesystem_retention_test_fixture::{ + HEAD_HEX, MANIFEST_HEX, ROOT_HEX, fixture, initial_preparation, open_authority, + retention_witness, +}; +use super::{FilesystemRetentionRecoveryError, RetentionFixedStage, RetentionRecoveryRefusal}; +use crate::execute_retention_publication; + +#[test] +fn invalid_complete_generations_refuse_recovery_without_changing_retained_bytes() +-> Result<(), Box> { + for (name, stage, hex, generation) in [ + ("root.next", RetentionFixedStage::Root, ROOT_HEX, 32..40), + ( + "manifest.next", + RetentionFixedStage::Manifest, + MANIFEST_HEX, + 32..40, + ), + ("head.next", RetentionFixedStage::Head, HEAD_HEX, 24..32), + ] { + let (sandbox, mut authority) = open_authority(&format!("zero-generation-{name}"))?; + let root_bytes = fixture(ROOT_HEX)?; + let _published = + execute_retention_publication(&mut authority, &initial_preparation(&root_bytes)?)?; + let mut bytes = fixture(hex)?; + let end = generation.end; + bytes + .get_mut(generation) + .ok_or("missing generation")? + .fill(0); + let prefix = bytes.get(..end).ok_or("missing stage prefix")?; + fs::write(sandbox.path().join("retention").join(name), prefix)?; + let before = retention_witness(sandbox.path())?; + + let result = authority.recover(); + + assert!( + matches!(result, Err(FilesystemRetentionRecoveryError::Plan { + source: RetentionRecoveryRefusal::StageCorrupt { stage: observed, .. } + }) if observed == stage), + "zero-generation {name} must refuse, observed {result:?}" + ); + assert_eq!( + retention_witness(sandbox.path())?, + before, + "generation refusal must preserve every retained byte for {name}" + ); + } + Ok(()) +} diff --git a/src/adapters/retention/filesystem_retention_incomplete_disposition_tests.rs b/src/adapters/retention/filesystem_retention_incomplete_disposition_tests.rs new file mode 100644 index 00000000..1fc7b492 --- /dev/null +++ b/src/adapters/retention/filesystem_retention_incomplete_disposition_tests.rs @@ -0,0 +1,175 @@ +//! This module owns preservation of incomplete stages under the bounded landing contract. + +use super::filesystem_retention_test_fixture::{ + HEAD_HEX, MANIFEST_HEX, ROOT_HEX, drive_publication, fixture, initial_preparation, + open_authority, retention_witness, +}; +use super::{ + FilesystemRetentionRecoveryError, RetentionCurrentStateRefusal, RetentionFixedStage as Stage, + RetentionPublicationStorage, RetentionRecoveryRefusal, +}; +use std::{error::Error, fs}; + +// Size: medium. Oracle: reserved lower-level capabilities cannot bypass decision A. +// Delete only when these public capabilities are removed or stronger bypass coverage replaces it. +#[test] +fn direct_discard_capabilities_refuse_without_mutation() -> Result<(), Box> { + use super::{RetentionRecordRefusal, RetentionRecoveryStorage, RetentionStorageError}; + let (sandbox, mut authority) = open_authority("direct-disposition-bypass")?; + for name in ["root.next", "manifest.next", "head.next"] { + fs::write( + sandbox.path().join("retention").join(name), + b"retained evidence", + )?; + } + let before = retention_witness(sandbox.path())?; + for result in [ + authority.discard_root_stage(), + authority.discard_manifest_stage(), + authority.discard_head_stage(), + ] { + assert!( + matches!( + result, + Err(RetentionStorageError::Operation { ref source, .. }) if matches!(source.as_ref(), RetentionStorageError::Refused { source: RetentionRecordRefusal::IncompleteDispositionRequired }) + ), + "reserved discard must refuse explicitly: {result:?}" + ); + assert_eq!( + retention_witness(sandbox.path())?, + before, + "reserved discard must preserve all bytes" + ); + } + Ok(()) +} + +// Size: medium. Oracle: demonstrated checksum corruption remains distinct from incompleteness. +// Delete only when stronger mixed-stage admission laws preserve the exact corruption diagnostic. +#[test] +fn known_corruption_precedes_incomplete_disposition() -> Result<(), Box> { + let (sandbox, mut authority) = open_authority("mixed-incomplete-corrupt")?; + let root = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root)?; + let mut corrupt = root.clone(); + *corrupt.last_mut().ok_or("empty root")? ^= 1; + fs::write(sandbox.path().join("retention/root.next"), corrupt)?; + fs::write(sandbox.path().join("retention/head.next"), b"")?; + let before = retention_witness(sandbox.path())?; + let result = authority.recover(); + assert!( + matches!(result, Err(FilesystemRetentionRecoveryError::Plan { source: RetentionRecoveryRefusal::StageCorrupt { stage: Stage::Root, ref source } }) if matches!(source.downcast_ref::(), Some(super::RetentionRootDecodeError::ChecksumMismatch { .. }))), + "known checksum corruption must remain precise: {result:?}" + ); + let error = authority + .verify_current(&preparation) + .err() + .ok_or("corruption admitted")?; + assert!( + matches!(error.get_ref().and_then(|cause| cause.downcast_ref::()), Some(RetentionCurrentStateRefusal::RecoveryRefused { source: RetentionRecoveryRefusal::StageCorrupt { stage: Stage::Root, source } }) if matches!(source.downcast_ref::(), Some(super::RetentionRootDecodeError::ChecksumMismatch { .. }))), + "publication preserves checksum cause: {error:?}" + ); + assert_eq!( + retention_witness(sandbox.path())?, + before, + "mixed-stage refusals preserve evidence" + ); + Ok(()) +} + +// Size: medium. Oracle: maintainer decision A forbids mutation before incomplete disposition. +// Behavior change: former incomplete-stage cleanup successes now refuse; delete only if superseded. +#[test] +fn incomplete_stages_block_direct_and_publication_recovery_without_effects() +-> Result<(), Box> { + for (stage, name, corpus) in [ + (Stage::Root, "root.next", ROOT_HEX), + (Stage::Manifest, "manifest.next", MANIFEST_HEX), + (Stage::Head, "head.next", HEAD_HEX), + ] { + for length in [0_usize, 20] { + let (sandbox, mut authority) = + open_authority(&format!("incomplete-{stage:?}-{length}"))?; + let root = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root)?; + // Leave the earlier root complete but unlinked. Even linking it is forbidden. + drive_publication(&mut authority, &preparation, 2)?; + fs::write( + sandbox.path().join("retention").join(name), + fixture(corpus)?.get(..length).ok_or("short corpus")?, + )?; + let before = retention_witness(sandbox.path())?; + let direct = authority.recover(); + assert!( + matches!(direct, Err(FilesystemRetentionRecoveryError::Plan { source: RetentionRecoveryRefusal::IncompleteStageRequiresDisposition { stage: actual, observed, .. } }) if actual == stage && observed == length), + "direct incomplete refusal must identify observed stage: {direct:?}" + ); + assert_eq!( + retention_witness(sandbox.path())?, + before, + "direct refusal must initiate no namespace effects" + ); + let error = authority + .verify_current(&preparation) + .err() + .ok_or("publication accepted incomplete stage")?; + assert!( + matches!(error.get_ref().and_then(|source| source.downcast_ref::()), Some(RetentionCurrentStateRefusal::RecoveryRefused { source: RetentionRecoveryRefusal::IncompleteStageRequiresDisposition { stage: actual, observed, .. } }) if *actual == stage && *observed == length), + "publication must preserve typed disposition requirement: {error:?}" + ); + assert_eq!( + retention_witness(sandbox.path())?, + before, + "publication refusal must initiate no namespace effects" + ); + } + } + Ok(()) +} + +// Size: medium. Oracle: unknown completion feasibility never authorizes destruction. +// Delete only when stronger recovery-boundary preservation covers this permanent counterexample. +#[test] +fn greatest_namespace_with_a_declared_successor_preserves_the_incomplete_manifest() +-> Result<(), Box> { + let (sandbox, mut authority) = open_authority("incomplete-max-namespace")?; + let root = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root)?; + drive_publication(&mut authority, &preparation, 7)?; + let mut manifest = fixture(MANIFEST_HEX)?; + manifest + .get_mut(24..32) + .ok_or("length absent")? + .copy_from_slice(&368_u64.to_be_bytes()); + manifest + .get_mut(44..48) + .ok_or("count absent")? + .copy_from_slice(&2_u32.to_be_bytes()); + manifest + .get_mut(160..192) + .ok_or("namespace absent")? + .fill(u8::MAX); + manifest.truncate(232); + fs::write(sandbox.path().join("retention/manifest.next"), manifest)?; + let before = retention_witness(sandbox.path())?; + let result = authority.recover(); + assert!( + matches!( + result, + Err(FilesystemRetentionRecoveryError::Plan { + source: RetentionRecoveryRefusal::IncompleteStageRequiresDisposition { + stage: Stage::Manifest, + expected: 368, + observed: 232 + } + }) + ), + "incomplete manifest must require disposition without claiming canonical completion: {result:?}" + ); + assert_eq!( + retention_witness(sandbox.path())?, + before, + "unknown completion must preserve evidence" + ); + Ok(()) +} diff --git a/src/adapters/retention/filesystem_retention_namespace.rs b/src/adapters/retention/filesystem_retention_namespace.rs index b6a03f08..75b31617 100644 --- a/src/adapters/retention/filesystem_retention_namespace.rs +++ b/src/adapters/retention/filesystem_retention_namespace.rs @@ -11,6 +11,14 @@ use super::filesystem_retention_pool_name as pool_name; use crate::{RetentionGenerationExpectation, RetentionManifest}; const CANONICAL_ENTRIES: [&str; 3] = [pool_name::HEAD, pool_name::ROOTS, pool_name::MANIFESTS]; +const RECOVERY_ENTRIES: [&str; 6] = [ + pool_name::HEAD, + pool_name::ROOTS, + pool_name::MANIFESTS, + pool_name::ROOT_STAGE, + pool_name::MANIFEST_STAGE, + pool_name::HEAD_STAGE, +]; /// Bounded observation of the admitted retention namespace. #[must_use] @@ -46,10 +54,31 @@ pub(super) fn admit( retention: &Dir, roots: &Dir, manifests: &Dir, +) -> io::Result { + admit_entries(retention, roots, manifests, &CANONICAL_ENTRIES) +} + +/// Admits recovery namespace names and pool kinds before observing stages. +/// +/// Only the three fixed stage names are additionally allowed. Their exact +/// kinds, bytes, and bounds remain the recovery observer's responsibility. +pub(super) fn admit_recovery( + retention: &Dir, + roots: &Dir, + manifests: &Dir, +) -> io::Result { + admit_entries(retention, roots, manifests, &RECOVERY_ENTRIES) +} + +fn admit_entries( + retention: &Dir, + roots: &Dir, + manifests: &Dir, + admitted_names: &[&str], ) -> io::Result { for entry in retention.entries()? { let name = entry?.file_name(); - if !CANONICAL_ENTRIES.iter().any(|canonical| name == *canonical) { + if !admitted_names.iter().any(|canonical| name == *canonical) { return Err(Refusal::UnknownRetentionEntry.into_io()); } } diff --git a/src/adapters/retention/filesystem_retention_observation_error_tests.rs b/src/adapters/retention/filesystem_retention_observation_error_tests.rs new file mode 100644 index 00000000..c882460e --- /dev/null +++ b/src/adapters/retention/filesystem_retention_observation_error_tests.rs @@ -0,0 +1,72 @@ +//! This module owns publication's recovery-observation error contract. + +use std::error::Error; +use std::fs; +use std::io; + +use super::filesystem_retention_test_fixture::{ + ROOT_HEX, fixture, initial_preparation, open_authority, +}; +use super::{RetentionCurrentStateRefusal as Refusal, RetentionPublicationError}; +use crate::execute_retention_publication; + +fn observation(error: &RetentionPublicationError) -> Result<&io::Error, Box> { + let RetentionPublicationError::CurrentVerification { source } = error else { + return Err(format!("observation must refuse current verification: {error:?}").into()); + }; + let Some(refusal @ Refusal::RecoveryObservationRefused { .. }) = source + .get_ref() + .and_then(|source| source.downcast_ref::()) + else { + return Err(format!("publication must identify recovery observation: {error:?}").into()); + }; + refusal + .source() + .and_then(|source| source.downcast_ref::()) + .ok_or_else(|| { + "observation must expose its original I/O source through Error::source".into() + }) +} + +// Size: medium. Oracle: publication identifies recovery observation and preserves its typed cause. +// Delete when this publication boundary is removed or stronger error-chain coverage subsumes it. +#[test] +fn publication_preserves_the_recovery_observation_refusal() -> Result<(), Box> { + let (sandbox, mut authority) = open_authority("retention-observation-typed-source")?; + fs::write(sandbox.path().join("retention/unknown"), b"evidence")?; + let root = fixture(ROOT_HEX)?; + let error = execute_retention_publication(&mut authority, &initial_preparation(&root)?) + .err() + .ok_or("unknown evidence was admitted")?; + let source = observation(&error)?; + assert!( + matches!( + source + .get_ref() + .and_then(|source| source.downcast_ref::()), + Some(Refusal::UnknownRetentionEntry) + ), + "original namespace refusal must survive: {error:?}" + ); + Ok(()) +} + +// Size: medium. Oracle: a no-follow stage-open failure preserves its exact OS cause through observation. +// Delete when stage observation is removed or stronger kernel-fault coverage subsumes it. +#[cfg(unix)] +#[test] +fn publication_preserves_the_recovery_observation_io_cause() -> Result<(), Box> { + let (sandbox, mut authority) = open_authority("retention-observation-io-source")?; + std::os::unix::fs::symlink("missing-stage", sandbox.path().join("retention/root.next"))?; + let root = fixture(ROOT_HEX)?; + let error = execute_retention_publication(&mut authority, &initial_preparation(&root)?) + .err() + .ok_or("symlink stage was admitted")?; + let source = observation(&error)?; + assert_eq!( + source.raw_os_error(), + Some(rustix::io::Errno::LOOP.raw_os_error()), + "original no-follow error must survive: {error:?}" + ); + Ok(()) +} diff --git a/src/adapters/retention/filesystem_retention_partial_anchor_tests.rs b/src/adapters/retention/filesystem_retention_partial_anchor_tests.rs new file mode 100644 index 00000000..e6e96ac9 --- /dev/null +++ b/src/adapters/retention/filesystem_retention_partial_anchor_tests.rs @@ -0,0 +1,264 @@ +//! This module owns preservation of contradictory coordinates inside partial root anchors. + +use super::filesystem_retention_test_fixture::{ + ROOT_HEX, fixture, open_authority, retention_witness, +}; +use super::{ + FilesystemRetentionRecoveryError, RetentionFixedStage, RetentionRecoveryRefusal, + RetentionRootDecodeError, +}; +use crate::LayoutIdBinaryParseError; +use std::{error::Error, fs}; + +#[derive(Debug)] +enum Fault { + Byte { + offset: usize, + expected: u8, + observed: u8, + }, + Length { + index: u32, + source: LayoutIdBinaryParseError, + }, + PrefixLength { + index: u32, + minimum: u64, + maximum: u64, + }, +} + +// Size: medium. Oracle: canonical embedded BlobId/LayoutId fixed bytes from the conformance root. +// Delete only when stronger recovery evidence subsumes each fixed field at its first available byte. +#[test] +fn contradictory_partial_anchor_fields_preserve_evidence() -> Result<(), Box> { + for index in [0_u32, 1] { + for relative in (0_usize..19).chain(59..79) { + let length = relative.checked_add(1).ok_or("length overflow")?; + let PartialAnchor { mut prefix, start } = partial_anchor(index, length)?; + let offset = start.checked_add(relative).ok_or("offset overflow")?; + let expected = *prefix.get(offset).ok_or("missing field byte")?; + let observed = expected ^ 1; + *prefix.get_mut(offset).ok_or("missing mutation byte")? = observed; + require_preservation( + &prefix, + &Fault::Byte { + offset, + expected, + observed, + }, + )?; + } + } + Ok(()) +} + +// Size: medium. Oracle: v1 layout lengths lie in 176..=46137520 and equal 176 modulo 44. +// Delete only when stronger recovery laws subsume early complete-length admission at both indices. +#[test] +fn invalid_lengths_in_partial_anchors_preserve_evidence() -> Result<(), Box> { + for index in [0_u32, 1] { + for length in [87, 118] { + for observed in [0_u64, 177, 46_137_521] { + let PartialAnchor { mut prefix, start } = partial_anchor(index, length)?; + let field_start = start.checked_add(79).ok_or("offset overflow")?; + let field_end = start.checked_add(87).ok_or("offset overflow")?; + prefix + .get_mut(field_start..field_end) + .ok_or("missing plan length")? + .copy_from_slice(&observed.to_be_bytes()); + let source = if observed == 177 { + LayoutIdBinaryParseError::PlanLengthNotCongruent { observed } + } else { + LayoutIdBinaryParseError::PlanLengthOutOfBounds { + minimum: 176, + maximum: 46_137_520, + observed, + } + }; + require_preservation(&prefix, &Fault::Length { index, source })?; + } + } + } + Ok(()) +} + +struct PartialAnchor { + prefix: Vec, + start: usize, +} + +fn partial_anchor(index: u32, length: usize) -> Result> { + let root = fixture(ROOT_HEX)?; + let mut prefix = root + .get(..195) + .ok_or("missing root header and namespace")? + .to_vec(); + let anchor = root.get(195..314).ok_or("missing conformance anchor")?; + for _ in 0..index { + prefix.extend_from_slice(anchor); + } + let start = prefix.len(); + prefix.extend_from_slice(anchor.get(..length).ok_or("invalid anchor prefix length")?); + let count = index.checked_add(1).ok_or("count overflow")?; + let total = u64::from(count) + .checked_mul(119) + .and_then(|size| size.checked_add(259)) + .ok_or("record length overflow")?; + prefix + .get_mut(24..32) + .ok_or("missing record length")? + .copy_from_slice(&total.to_be_bytes()); + prefix + .get_mut(44..48) + .ok_or("missing count")? + .copy_from_slice(&count.to_be_bytes()); + Ok(PartialAnchor { prefix, start }) +} + +fn require_preservation(prefix: &[u8], expected: &Fault) -> Result<(), Box> { + let (sandbox, mut authority) = open_authority(&format!("partial-anchor-{}", prefix.len()))?; + fs::write(sandbox.path().join("retention/root.next"), prefix)?; + let before = retention_witness(sandbox.path())?; + match authority.recover() { + Err(FilesystemRetentionRecoveryError::Plan { + source: RetentionRecoveryRefusal::StageCorrupt { stage, source }, + }) => { + assert_eq!( + stage, + RetentionFixedStage::Root, + "identify the corrupt partial-anchor stage" + ); + require_cause(source.as_ref(), expected); + } + result => { + return Err(format!("partial anchor must refuse {expected:?}: {result:?}").into()); + } + } + assert_eq!( + retention_witness(sandbox.path())?, + before, + "partial-anchor refusal must preserve retained evidence" + ); + Ok(()) +} + +fn require_cause(source: &(dyn Error + 'static), expected: &Fault) { + let exact = match (source.downcast_ref::(), expected) { + ( + Some(RetentionRootDecodeError::LayoutLengthPrefixAboveMaximum { + index, + minimum, + maximum, + }), + Fault::PrefixLength { + index: wanted_index, + minimum: wanted_minimum, + maximum: wanted_maximum, + }, + ) => index == wanted_index && minimum == wanted_minimum && maximum == wanted_maximum, + ( + Some(RetentionRootDecodeError::PrefixByteMismatch { + offset, + expected, + observed, + }), + Fault::Byte { + offset: wanted_offset, + expected: wanted_expected, + observed: wanted_observed, + }, + ) => offset == wanted_offset && expected == wanted_expected && observed == wanted_observed, + ( + Some(RetentionRootDecodeError::LayoutId { index, source }), + Fault::Length { + index: wanted_index, + source: wanted_source, + }, + ) => index == wanted_index && source == wanted_source, + _ => false, + }; + assert!( + exact, + "report exact partial-anchor cause: expected {expected:?}, observed {source:?}" + ); +} + +// Size: medium. Oracle: every completion of these big-endian prefixes exceeds 46137520. +// Delete only when stronger recovery laws subsume impossible partial numeric lengths. +#[test] +fn impossible_partial_layout_lengths_preserve_evidence() -> Result<(), Box> { + for index in [0_u32, 1] { + for available in 1_usize..8 { + require_length_prefix(index, available, 0xff00_0000_0000_0000)?; + } + require_length_prefix(index, 7, 46_137_600)?; + } + Ok(()) +} + +fn require_length_prefix(index: u32, available: usize, minimum: u64) -> Result<(), Box> { + let length = 79_usize.checked_add(available).ok_or("length overflow")?; + let PartialAnchor { mut prefix, start } = partial_anchor(index, length)?; + let field_start = start.checked_add(79).ok_or("offset overflow")?; + let field_end = field_start + .checked_add(available) + .ok_or("offset overflow")?; + prefix + .get_mut(field_start..field_end) + .ok_or("missing partial length")? + .copy_from_slice( + minimum + .to_be_bytes() + .get(..available) + .ok_or("missing source bytes")?, + ); + require_preservation( + &prefix, + &Fault::PrefixLength { + index, + minimum, + maximum: 46_137_520, + }, + ) +} + +// Size: small. Oracle: a canonical layout value witnesses a valid completion of every prefix. +// Delete only when stronger public assessment evidence subsumes these byte-boundary controls. +#[test] +fn canonical_layout_lengths_admit_every_partial_numeric_prefix() -> Result<(), Box> { + for count in std::iter::once(0_u64).chain((0..=20).filter_map(|bit| 1_u64.checked_shl(bit))) { + let length = count + .checked_mul(44) + .and_then(|size| size.checked_add(176)) + .ok_or("layout length overflow")?; + for available in 1_usize..8 { + let partial_length = 79_usize + .checked_add(available) + .ok_or("prefix length overflow")?; + let PartialAnchor { mut prefix, start } = partial_anchor(0, partial_length)?; + let field_start = start.checked_add(79).ok_or("offset overflow")?; + let field_end = field_start + .checked_add(available) + .ok_or("offset overflow")?; + prefix + .get_mut(field_start..field_end) + .ok_or("missing target bytes")? + .copy_from_slice( + length + .to_be_bytes() + .get(..available) + .ok_or("missing source bytes")?, + ); + let assessment = super::assess_root_stage(Some(&prefix)); + assert!( + matches!( + assessment, + super::RetentionStageAssessment::Truncated { .. } + ), + "valid layout length {length} with {available} available bytes must admit completion: {assessment:?}" + ); + } + } + Ok(()) +} diff --git a/src/adapters/retention/filesystem_retention_partial_entry_tests.rs b/src/adapters/retention/filesystem_retention_partial_entry_tests.rs new file mode 100644 index 00000000..8e3cc1dd --- /dev/null +++ b/src/adapters/retention/filesystem_retention_partial_entry_tests.rs @@ -0,0 +1,143 @@ +//! This module owns manifest partial-entry refusal and valid-completion laws. + +use super::filesystem_retention_test_fixture::{ + MANIFEST_HEX, ROOT_HEX, drive_publication, fixture, initial_preparation, open_authority, + retention_witness, +}; +use super::{ + FilesystemRetentionRecoveryError, RetentionFixedStage, RetentionManifestDecodeError, + RetentionRecoveryRefusal, RetentionStageAssessment, assess_manifest_stage, +}; +use crate::RootGenerationError; +use std::{error::Error, fs}; + +// Size: medium. Oracle: manifest namespace digests are strictly increasing. +// Delete only when stronger public recovery laws subsume each impossible partial-order class. +#[test] +fn impossible_partial_namespace_order_preserves_evidence() -> Result<(), Box> { + for length in 1..72 { + require_preservation(&entry_prefix([0x80; 32], [0x7f; 32], length)?, Fault::Order)?; + } + for length in 32..72 { + require_preservation(&entry_prefix([0x80; 32], [0x80; 32], length)?, Fault::Order)?; + } + for length in 1..32 { + require_preservation(&entry_prefix([0xff; 32], [0xff; 32], length)?, Fault::Order)?; + } + Ok(()) +} + +// Size: medium. Oracle: a complete root-generation field must be positive, even before its digest. +// Delete only when stronger public recovery evidence subsumes every remaining entry length. +#[test] +fn zero_generation_in_partial_manifest_entry_preserves_evidence() -> Result<(), Box> { + let mut bytes = fixture(MANIFEST_HEX)?; + bytes + .get_mut(192..200) + .ok_or("missing root generation")? + .fill(0); + for end in 200..232 { + require_preservation( + bytes.get(..end).ok_or("missing partial entry")?, + Fault::Generation, + )?; + } + Ok(()) +} + +// Size: small. Oracle: an ordered complete namespace supplies a witness for every prefix. +// Delete only when stronger public assessment coverage subsumes these possible-completion classes. +#[test] +fn ordered_manifest_entries_remain_possible_at_every_partial_length() -> Result<(), Box> +{ + let mut late_successor = [0x80; 32]; + *late_successor.last_mut().ok_or("empty namespace")? = 0x81; + for next in [[0x81; 32], late_successor] { + for length in 1..72 { + let prefix = entry_prefix([0x80; 32], next, length)?; + let assessment = assess_manifest_stage(Some(&prefix)); + assert!( + matches!(assessment, RetentionStageAssessment::Truncated { .. }), + "ordered namespace prefix length {length} must admit completion: {assessment:?}" + ); + } + } + Ok(()) +} + +fn entry_prefix(prior: [u8; 32], next: [u8; 32], length: usize) -> Result, Box> { + let bytes = fixture(MANIFEST_HEX)?; + let mut prefix = bytes.get(..232).ok_or("missing first entry")?.to_vec(); + prefix + .get_mut(24..32) + .ok_or("missing length")? + .copy_from_slice(&368_u64.to_be_bytes()); + prefix + .get_mut(44..48) + .ok_or("missing count")? + .copy_from_slice(&2_u32.to_be_bytes()); + prefix + .get_mut(160..192) + .ok_or("missing prior namespace")? + .copy_from_slice(&prior); + let mut entry = bytes.get(160..232).ok_or("missing fixture entry")?.to_vec(); + entry + .get_mut(..32) + .ok_or("missing next namespace")? + .copy_from_slice(&next); + prefix.extend_from_slice(entry.get(..length).ok_or("invalid entry prefix")?); + Ok(prefix) +} + +#[derive(Clone, Copy, Debug)] +enum Fault { + Order, + Generation, +} + +fn require_preservation(prefix: &[u8], fault: Fault) -> Result<(), Box> { + let (sandbox, mut authority) = + open_authority(&format!("partial-entry-{fault:?}-{}", prefix.len()))?; + let root = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root)?; + drive_publication(&mut authority, &preparation, 7)?; + fs::write(sandbox.path().join("retention/manifest.next"), prefix)?; + let before = retention_witness(sandbox.path())?; + match authority.recover() { + Err(FilesystemRetentionRecoveryError::Plan { + source: RetentionRecoveryRefusal::StageCorrupt { stage, source }, + }) => { + assert_eq!( + stage, + RetentionFixedStage::Manifest, + "identify the corrupt partial-entry stage" + ); + let exact = matches!( + (fault, source.downcast_ref::()), + ( + Fault::Order, + Some(RetentionManifestDecodeError::NonCanonicalEntryOrder { index: 1 }), + ) | ( + Fault::Generation, + Some(RetentionManifestDecodeError::RootGeneration { + index: 0, + source: RootGenerationError::Zero, + }), + ) + ); + assert!( + exact, + "report exact partial-entry {fault:?} cause: {source:?}" + ); + } + result => { + return Err(format!("partial manifest entry {fault:?} must refuse: {result:?}").into()); + } + } + assert_eq!( + retention_witness(sandbox.path())?, + before, + "partial-entry refusal must preserve retained evidence" + ); + Ok(()) +} diff --git a/src/adapters/retention/filesystem_retention_partial_history_tests.rs b/src/adapters/retention/filesystem_retention_partial_history_tests.rs new file mode 100644 index 00000000..12aa7be2 --- /dev/null +++ b/src/adapters/retention/filesystem_retention_partial_history_tests.rs @@ -0,0 +1,185 @@ +//! This module owns initial-record predecessor-prefix refusals and successor completion laws. + +use super::filesystem_retention_test_fixture::{ + HEAD_HEX, MANIFEST_HEX, ROOT_HEX, drive_publication, fixture, initial_preparation, + open_authority, retention_witness, +}; +use super::{ + FilesystemRetentionRecoveryError, RetentionFixedStage as Stage, RetentionHeadDecodeError, + RetentionManifestDecodeError, RetentionRecoveryRefusal, RetentionRootDecodeError, + RetentionStageAssessment, assess_head_stage, assess_manifest_stage, assess_root_stage, +}; +use std::{error::Error, fs}; + +struct Case { + stage: Stage, + name: &'static str, + corpus: &'static str, + phases: usize, + generation: usize, + predecessor: usize, +} +const CASES: [Case; 3] = [ + Case { + stage: Stage::Root, + name: "root.next", + corpus: ROOT_HEX, + phases: 1, + generation: 32, + predecessor: 116, + }, + Case { + stage: Stage::Manifest, + name: "manifest.next", + corpus: MANIFEST_HEX, + phases: 7, + generation: 32, + predecessor: 48, + }, + Case { + stage: Stage::Head, + name: "head.next", + corpus: HEAD_HEX, + phases: 11, + generation: 24, + predecessor: 72, + }, +]; + +// Size: medium. Oracle: initial records have no predecessor, encoded as all-zero digest bytes. +// Delete only when stronger public recovery laws subsume all record types and partial endpoints. +#[test] +fn nonzero_partial_initial_predecessors_preserve_evidence() -> Result<(), Box> { + for case in CASES { + let bytes = fixture(case.corpus)?; + for available in 1..32 { + let end = case + .predecessor + .checked_add(available) + .ok_or("prefix overflow")?; + let offset = end.checked_sub(1).ok_or("empty prefix")?; + let mut partial = bytes.get(..end).ok_or("missing prefix")?.to_vec(); + *partial.get_mut(offset).ok_or("missing predecessor byte")? = 7; + let (sandbox, mut authority) = + open_authority(&format!("partial-history-{}-{available}", case.name))?; + let root = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root)?; + drive_publication(&mut authority, &preparation, case.phases)?; + fs::write(sandbox.path().join("retention").join(case.name), partial)?; + let before = retention_witness(sandbox.path())?; + match authority.recover() { + Err(FilesystemRetentionRecoveryError::Plan { + source: RetentionRecoveryRefusal::StageCorrupt { stage, source }, + }) => { + assert_eq!( + stage, case.stage, + "identify the contradictory partial-history stage" + ); + require_cause(source.as_ref(), case.stage, offset); + } + result => { + return Err(format!( + "{} initial predecessor prefix {available} must refuse: {result:?}", + case.name + ) + .into()); + } + } + assert_eq!( + retention_witness(sandbox.path())?, + before, + "partial-history refusal must preserve retained evidence" + ); + } + } + Ok(()) +} + +// Size: small. Oracle: any partial successor predecessor, including zeros, can complete nonzero. +// Delete only when stronger public assessment laws subsume successor completion at all endpoints. +#[test] +fn successor_predecessors_remain_possible_before_the_digest_is_complete() +-> Result<(), Box> { + for case in CASES { + for (generation, byte) in [(2_u64, 0_u8), (2, 7), (u64::MAX, 0), (u64::MAX, 7)] { + let mut bytes = fixture(case.corpus)?; + let generation_end = case + .generation + .checked_add(8) + .ok_or("generation overflow")?; + bytes + .get_mut(case.generation..generation_end) + .ok_or("missing generation")? + .copy_from_slice(&generation.to_be_bytes()); + let predecessor_end = case + .predecessor + .checked_add(32) + .ok_or("predecessor overflow")?; + bytes + .get_mut(case.predecessor..predecessor_end) + .ok_or("missing predecessor")? + .fill(byte); + for available in 1..32 { + let end = case + .predecessor + .checked_add(available) + .ok_or("prefix overflow")?; + let partial = bytes.get(..end).ok_or("missing prefix")?; + let possible = match case.stage { + Stage::Root => matches!( + assess_root_stage(Some(partial)), + RetentionStageAssessment::Truncated { .. } + ), + Stage::Manifest => matches!( + assess_manifest_stage(Some(partial)), + RetentionStageAssessment::Truncated { .. } + ), + Stage::Head => matches!( + assess_head_stage(Some(partial)), + RetentionStageAssessment::Truncated { .. } + ), + }; + assert!( + possible, + "{} successor generation {generation}, predecessor prefix {available}, must admit nonzero completion", + case.name + ); + } + } + } + Ok(()) +} + +fn require_cause(source: &(dyn Error + 'static), stage: Stage, offset: usize) { + let actual = match stage { + Stage::Root => match source.downcast_ref::() { + Some(RetentionRootDecodeError::PrefixByteMismatch { + offset, + expected, + observed, + }) => Some((*offset, *expected, *observed)), + _ => None, + }, + Stage::Manifest => match source.downcast_ref::() { + Some(RetentionManifestDecodeError::PrefixByteMismatch { + offset, + expected, + observed, + }) => Some((*offset, *expected, *observed)), + _ => None, + }, + Stage::Head => match source.downcast_ref::() { + Some(RetentionHeadDecodeError::PrefixByteMismatch { + offset, + expected, + observed, + }) => Some((*offset, *expected, *observed)), + _ => None, + }, + }; + assert_eq!( + actual, + Some((offset, 0, 7)), + "report only the exact available predecessor contradiction: {source:?}" + ); +} diff --git a/src/adapters/retention/filesystem_retention_partial_integrity_tests.rs b/src/adapters/retention/filesystem_retention_partial_integrity_tests.rs new file mode 100644 index 00000000..bd035be2 --- /dev/null +++ b/src/adapters/retention/filesystem_retention_partial_integrity_tests.rs @@ -0,0 +1,164 @@ +//! This module owns preservation of interrupted records with provably corrupt integrity fields. + +use super::filesystem_retention_test_fixture::{ + MANIFEST_HEX, ROOT_HEX, drive_publication, fixture, initial_preparation, open_authority, + retention_witness, +}; +use super::{ + FilesystemRetentionRecoveryError, RetentionFixedStage as Stage, RetentionManifestDecodeError, + RetentionRecoveryRefusal, RetentionRootDecodeError, +}; +use std::{error::Error, fs}; + +// Size: medium. Oracle: canonical conformance digest/checksum bytes bind the unchanged preimage. +// Delete only when stronger filesystem recovery laws subsume both record trailers at every prefix. +#[test] +fn corrupt_partial_record_trailers_preserve_evidence() -> Result<(), Box> { + for (stage, corpus) in [(Stage::Root, ROOT_HEX), (Stage::Manifest, MANIFEST_HEX)] { + let bytes = fixture(corpus)?; + let digest_start = bytes.len().checked_sub(64).ok_or("missing trailer")?; + for offset in digest_start..bytes.len().checked_sub(1).ok_or("empty record")? { + let end = offset.checked_add(1).ok_or("offset overflow")?; + let mut partial = bytes.get(..end).ok_or("missing prefix")?.to_vec(); + let expected = *partial.get(offset).ok_or("missing trailer byte")?; + let observed = expected ^ 1; + *partial.get_mut(offset).ok_or("missing mutation byte")? = observed; + require_preservation( + stage, + &partial, + &Fault::Byte { + offset, + expected, + observed, + }, + )?; + } + } + Ok(()) +} + +// Size: medium. Oracle: the body-set digest is already determined when the complete body arrives. +// Delete only when stronger public recovery coverage subsumes both body-set digest refusals. +#[test] +fn corrupt_complete_body_digests_preserve_evidence() -> Result<(), Box> { + for (stage, corpus, offset) in [ + (Stage::Root, ROOT_HEX, 148_usize), + (Stage::Manifest, MANIFEST_HEX, 80), + ] { + let bytes = fixture(corpus)?; + let body_end = bytes.len().checked_sub(64).ok_or("missing trailer")?; + let end = offset.checked_add(32).ok_or("digest offset overflow")?; + let expected = <[u8; 32]>::try_from(bytes.get(offset..end).ok_or("missing digest")?)?; + let mut observed = expected; + *observed.first_mut().ok_or("empty digest")? ^= 1; + let mut partial = bytes.get(..body_end).ok_or("missing body")?.to_vec(); + partial + .get_mut(offset..end) + .ok_or("missing header digest")? + .copy_from_slice(&observed); + require_preservation(stage, &partial, &Fault::Set { expected, observed })?; + } + Ok(()) +} + +#[derive(Debug, Eq, PartialEq)] +enum Fault { + Byte { + offset: usize, + expected: u8, + observed: u8, + }, + Set { + expected: [u8; 32], + observed: [u8; 32], + }, +} + +fn require_preservation( + stage: Stage, + partial: &[u8], + expected: &Fault, +) -> Result<(), Box> { + let (name, phases) = match stage { + Stage::Root => ("root.next", 1), + _ => ("manifest.next", 7), + }; + let (sandbox, mut authority) = + open_authority(&format!("partial-integrity-{name}-{}", partial.len()))?; + let root = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root)?; + drive_publication(&mut authority, &preparation, phases)?; + fs::write(sandbox.path().join("retention").join(name), partial)?; + let before = retention_witness(sandbox.path())?; + match authority.recover() { + Err(FilesystemRetentionRecoveryError::Plan { + source: + RetentionRecoveryRefusal::StageCorrupt { + stage: actual, + source, + }, + }) => { + assert_eq!(actual, stage, "identify the corrupt integrity stage"); + require_cause(source.as_ref(), stage, expected); + } + result => { + return Err(format!( + "{name} integrity prefix {} must refuse: {result:?}", + partial.len() + ) + .into()); + } + } + assert_eq!( + retention_witness(sandbox.path())?, + before, + "corrupt integrity bytes must preserve retained evidence" + ); + Ok(()) +} + +fn require_cause(source: &(dyn Error + 'static), stage: Stage, expected: &Fault) { + let actual = match stage { + Stage::Root => match source.downcast_ref::() { + Some(RetentionRootDecodeError::PrefixByteMismatch { + offset, + expected, + observed, + }) => Some(Fault::Byte { + offset: *offset, + expected: *expected, + observed: *observed, + }), + Some(RetentionRootDecodeError::AnchorSetDigestMismatch { expected, observed }) => { + Some(Fault::Set { + expected: *expected, + observed: *observed, + }) + } + _ => None, + }, + _ => match source.downcast_ref::() { + Some(RetentionManifestDecodeError::PrefixByteMismatch { + offset, + expected, + observed, + }) => Some(Fault::Byte { + offset: *offset, + expected: *expected, + observed: *observed, + }), + Some(RetentionManifestDecodeError::EntrySetDigestMismatch { expected, observed }) => { + Some(Fault::Set { + expected: *expected, + observed: *observed, + }) + } + _ => None, + }, + }; + assert_eq!( + actual.as_ref(), + Some(expected), + "report exact available integrity contradiction: {source:?}" + ); +} diff --git a/src/adapters/retention/filesystem_retention_partial_profile_tests.rs b/src/adapters/retention/filesystem_retention_partial_profile_tests.rs new file mode 100644 index 00000000..f39e33ed --- /dev/null +++ b/src/adapters/retention/filesystem_retention_partial_profile_tests.rs @@ -0,0 +1,59 @@ +//! This module owns preservation of impossible interrupted profile prefixes. + +use super::filesystem_retention_test_fixture::{ + ROOT_HEX, fixture, open_authority, retention_witness, +}; +use super::{ + FilesystemRetentionRecoveryError, RetentionFixedStage, RetentionRecoveryRefusal, + RetentionRootDecodeError, +}; +use std::{error::Error, fs}; + +// Size: medium. Oracle: the registered v2 profile has one exact identity/version/digest encoding. +// Delete only when stronger public recovery coverage subsumes every partial profile field. +#[test] +fn contradictory_partial_root_profiles_preserve_evidence() -> Result<(), Box> { + let root = fixture(ROOT_HEX)?; + for end in 49_usize..88 { + let offset = end.checked_sub(1).ok_or("invalid prefix length")?; + let expected = *root.get(offset).ok_or("missing profile fixture byte")?; + let observed = expected ^ 1; + let mut partial = root.get(..end).ok_or("short fixture")?.to_vec(); + *partial.get_mut(offset).ok_or("missing mutation byte")? = observed; + let (sandbox, mut authority) = open_authority(&format!("partial-profile-{end}"))?; + fs::write(sandbox.path().join("retention/root.next"), partial)?; + let before = retention_witness(sandbox.path())?; + + match authority.recover() { + Err(FilesystemRetentionRecoveryError::Plan { + source: RetentionRecoveryRefusal::StageCorrupt { stage, source }, + }) => { + assert_eq!( + stage, + RetentionFixedStage::Root, + "identify the corrupt profile stage" + ); + assert!( + matches!(source.downcast_ref::(), + Some(RetentionRootDecodeError::PrefixByteMismatch { + offset: actual_offset, expected: actual_expected, observed: actual_observed, + }) if *actual_offset == offset && *actual_expected == expected && *actual_observed == observed + ), + "name the exact available profile contradiction at {offset}: {source:?}" + ); + } + result => { + return Err(format!( + "profile contradiction at {offset} must refuse recovery: {result:?}" + ) + .into()); + } + } + assert_eq!( + retention_witness(sandbox.path())?, + before, + "profile contradiction at {offset} must preserve retained evidence" + ); + } + Ok(()) +} diff --git a/src/adapters/retention/filesystem_retention_publication_closure_tests.rs b/src/adapters/retention/filesystem_retention_publication_closure_tests.rs new file mode 100644 index 00000000..8919f457 --- /dev/null +++ b/src/adapters/retention/filesystem_retention_publication_closure_tests.rs @@ -0,0 +1,79 @@ +//! These laws own live transitive admission after pure publication preflight. + +use std::error::Error; +use std::fs; +use std::io; + +use super::RetentionPublicationError; +use super::filesystem_retention_test_fixture::{ + ROOT_HEX, SEGMENT_NAME, fixture, initial_preparation, open_authority, retention_witness, +}; +use crate::adapters::{ + CatalogRestartError, CatalogRestartPhase, SegmentReadError, SegmentSealError, +}; +use crate::execute_retention_publication; + +#[derive(Clone, Copy, Debug)] +enum Damage { + Missing, + Corrupt, +} + +// Size: medium. Oracle: preflight is not proof of live reconstructability at publication. +// Delete only with this protocol or a stronger runtime-boundary replacement. +#[test] +fn publication_refuses_a_segment_removed_after_preflight() -> Result<(), Box> { + require_refusal(Damage::Missing) +} +#[test] +fn publication_refuses_a_segment_corrupted_after_preflight() -> Result<(), Box> { + require_refusal(Damage::Corrupt) +} + +fn require_refusal(damage: Damage) -> Result<(), Box> { + let (sandbox, mut authority) = open_authority(&format!("publication-live-closure-{damage:?}"))?; + let bytes = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&bytes)?; + let path = sandbox.path().join("segments").join(SEGMENT_NAME); + match damage { + Damage::Missing => fs::remove_file(path)?, + Damage::Corrupt => { + let mut bytes = fs::read(&path)?; + *bytes.last_mut().ok_or("empty segment")? ^= 1; + fs::write(path, bytes)?; + } + } + let before = retention_witness(sandbox.path())?; + let result = execute_retention_publication(&mut authority, &preparation); + assert_eq!( + retention_witness(sandbox.path())?, + before, + "live closure refusal must preserve retained bytes for {damage:?}: {result:?}" + ); + let error = match &result { + Err(RetentionPublicationError::CurrentVerification { source }) => source, + other => { + return Err(format!("live closure must refuse current verification: {other:?}").into()); + } + }; + let restart = error + .get_ref() + .and_then(|source| source.downcast_ref::()); + let exact = match (damage, restart) { + (Damage::Missing, Some(CatalogRestartError::Io { phase, source })) => { + *phase == CatalogRestartPhase::OpenSegment && source.kind() == io::ErrorKind::NotFound + } + (Damage::Corrupt, Some(CatalogRestartError::Segment { source, .. })) => matches!( + source.as_ref(), + SegmentReadError::Seal { + source: SegmentSealError::SealChecksumMismatch { .. } + } + ), + _ => false, + }; + assert!( + exact, + "live closure must retain exact {damage:?} refusal: {result:?}" + ); + Ok(()) +} diff --git a/src/adapters/retention/filesystem_retention_recovery.rs b/src/adapters/retention/filesystem_retention_recovery.rs new file mode 100644 index 00000000..60e843d3 --- /dev/null +++ b/src/adapters/retention/filesystem_retention_recovery.rs @@ -0,0 +1,365 @@ +//! This module owns filesystem execution of retention recovery under authority. + +use super::RetentionStorageError; +use super::{ + RetentionEffectDurability as Durability, RetentionNamespaceEffect as Effect, + RetentionStorageBoundary as Boundary, +}; +use std::io; + +use cap_fs_ext::DirExt; +use cap_std::fs::Dir; + +use super::filesystem_retention_authority::FilesystemRetentionPublicationAuthority; +use super::filesystem_retention_pool_name as pool_name; +use super::filesystem_retention_recovery_observation::RetentionRecoveryObservation; +use super::filesystem_retention_stage::{FilesystemRetentionStage, invalid_data}; +use super::filesystem_retention_storage::require_pinned_directories; +use super::{ + FilesystemRetentionRecoveryError as Error, RetentionRecoveryReceipt, RetentionRecoveryStorage, + RetentionStageAssessment, assess_head_stage, assess_manifest_stage, assess_root_stage, + execute_retention_recovery, plan_retention_recovery, +}; +use crate::adapters::CatalogRestartPolicy; +use crate::adapters::filesystem_catalog_artifact::synchronize_directory; + +#[cfg(test)] +#[path = "filesystem_retention_recovery_effect_tests.rs"] +mod effect_tests; +#[cfg(test)] +#[path = "filesystem_retention_recovery_storage_error_tests.rs"] +mod storage_error_tests; + +/// One retained stage as recovery holds it between steps. +pub(super) enum RecoveredStage { + /// A complete stage reopened and bound to its identity. + Complete { + stage: FilesystemRetentionStage, + pool_name: String, + namespace: Option, + }, + /// Incomplete evidence cannot authorize a filesystem mutation. + Incomplete, +} + +/// The retained stages one recovery run operates on. +pub(super) struct RetentionRecoveryContext { + root: Option, + manifest: Option, + head: Option, +} + +impl RetentionRecoveryContext { + fn reopen(retention: &Dir, observation: &RetentionRecoveryObservation) -> io::Result { + let root = observation + .root() + .map(|stage| -> io::Result { + match assess_root_stage(Some(&stage.bytes)) { + RetentionStageAssessment::Complete(admitted) => Ok(RecoveredStage::Complete { + stage: FilesystemRetentionStage::reopen( + retention, + pool_name::ROOT_STAGE, + &stage.bytes, + stage.identity, + )?, + pool_name: pool_name::root(admitted.root().generation(), admitted.digest()), + namespace: Some(pool_name::namespace(admitted.root().namespace().digest())), + }), + _ => Ok(RecoveredStage::Incomplete), + } + }) + .transpose()?; + let manifest = observation + .manifest() + .map(|stage| -> io::Result { + match assess_manifest_stage(Some(&stage.bytes)) { + RetentionStageAssessment::Complete(admitted) => Ok(RecoveredStage::Complete { + stage: FilesystemRetentionStage::reopen( + retention, + pool_name::MANIFEST_STAGE, + &stage.bytes, + stage.identity, + )?, + pool_name: pool_name::manifest( + admitted.manifest().generation(), + admitted.digest(), + ), + namespace: None, + }), + _ => Ok(RecoveredStage::Incomplete), + } + }) + .transpose()?; + let head = observation + .head() + .map(|stage| -> io::Result { + match assess_head_stage(Some(&stage.bytes)) { + RetentionStageAssessment::Complete(_) => Ok(RecoveredStage::Complete { + stage: FilesystemRetentionStage::reopen( + retention, + pool_name::HEAD_STAGE, + &stage.bytes, + stage.identity, + )?, + pool_name: pool_name::HEAD.to_owned(), + namespace: None, + }), + _ => Ok(RecoveredStage::Incomplete), + } + }) + .transpose()?; + Ok(Self { + root, + manifest, + head, + }) + } +} + +impl FilesystemRetentionPublicationAuthority { + /// Recovers fixed retention stages with an explicit catalog loading policy. + /// + /// The synchronous call runs under the retained writer lock. A clean store + /// returns an empty receipt; an incomplete stage requires disposition; a + /// complete stage is linked and retained as a recovery-protected orphan; + /// a complete head over linked stages is finalized. Any pending + /// publication attempt is discarded first. Publication calls this itself + /// as its first step; callers may also run it explicitly at restart. + /// Both entry points verify that protocol names still identify the pinned + /// directories before observing stages or executing recovery effects. + /// A complete root requires loading every head-selected catalog segment + /// into bounded retained memory, then replaying its anchors and closure + /// limits before any recovery effect. The supplied policy bounds aggregate + /// segment bytes and per-segment record admission; catalog bytes also have + /// their independent protocol bound. Clean and incomplete-root recovery + /// does not materialize the catalog. This call may allocate and block on I/O. + /// + /// # Errors + /// + /// Returns [`FilesystemRetentionRecoveryError`](super::FilesystemRetentionRecoveryError) + /// at the exact observation failure, planning refusal, or failed execution boundary. Execution failure may have + /// namespace effects; inspect its progress and reobserve before retry. + pub fn recover_with_catalog_policy( + &mut self, + policy: CatalogRestartPolicy, + ) -> Result { + require_pinned_directories(&self.root, &self.retention, &self.roots, &self.manifests) + .map_err(|source| Error::Observe { source })?; + self.attempt = None; + self.recovery = None; + let _census = super::filesystem_retention_namespace::admit_recovery( + &self.retention, + &self.roots, + &self.manifests, + ) + .map_err(|source| Error::Observe { source })?; + let observation = + RetentionRecoveryObservation::observe(&self.retention, &self.roots, &self.manifests) + .map_err(|source| Error::Observe { source })?; + let plan = plan_retention_recovery(observation.evidence()) + .map_err(|source| Error::Plan { source })?; + super::filesystem_retention_recovery_roots::admit(&self.roots, &observation.evidence()) + .map_err(|source| Error::Observe { source })?; + super::filesystem_retention_closure_admission::admit_recovery( + &self.root, + &observation.evidence(), + policy, + ) + .map_err(|source| Error::Observe { source })?; + self.recovery = Some( + RetentionRecoveryContext::reopen(&self.retention, &observation) + .map_err(|source| Error::Observe { source })?, + ); + let result = + execute_retention_recovery(self, &plan).map_err(|source| Error::Execute { source }); + self.recovery = None; + result + } +} + +fn no_recovery() -> RetentionStorageError { + RetentionStorageError::from(invalid_data("no retention recovery is in progress")) + .at(Boundary::RecoveryContext) +} + +fn take_complete( + slot: &mut Option, +) -> Result<(FilesystemRetentionStage, String, Option), RetentionStorageError> { + match slot.take() { + Some(RecoveredStage::Complete { + stage, + pool_name, + namespace, + }) => Ok((stage, pool_name, namespace)), + Some(other) => { + *slot = Some(other); + Err(RetentionStorageError::from(invalid_data( + "recovery step expected a complete stage", + )) + .at(Boundary::RecoveryContext)) + } + None => Err(no_recovery()), + } +} + +fn disposition_required() -> Result<(), RetentionStorageError> { + Err(RetentionStorageError::Refused { + source: super::RetentionRecordRefusal::IncompleteDispositionRequired, + } + .at(Boundary::RecoveryContext)) +} + +impl RetentionRecoveryStorage for FilesystemRetentionPublicationAuthority { + fn discard_head_stage(&mut self) -> Result<(), RetentionStorageError> { + disposition_required() + } + + fn discard_manifest_stage(&mut self) -> Result<(), RetentionStorageError> { + disposition_required() + } + + fn discard_root_stage(&mut self) -> Result<(), RetentionStorageError> { + disposition_required() + } + + fn link_root(&mut self) -> Result<(), RetentionStorageError> { + let context = self.recovery.as_ref().ok_or_else(no_recovery)?; + let Some(RecoveredStage::Complete { + stage, + pool_name: name, + namespace: Some(namespace), + }) = context.root.as_ref() + else { + return Err(RetentionStorageError::from(invalid_data( + "link_root expected a complete root stage", + )) + .at(Boundary::RecoveryContext)); + }; + stage.synchronize(&self.retention)?; + let created = match self.roots.create_dir(namespace) { + Ok(()) => true, + Err(source) if source.kind() == io::ErrorKind::AlreadyExists => false, + Err(source) => { + return Err(RetentionStorageError::from(source) + .at(Boundary::NamespaceCreation) + .uncertain(Effect::NamespaceCreated)); + } + }; + let mut durability = Durability::Unconfirmed; + let result = (|| { + let directory = self.roots.open_dir_nofollow(namespace).map_err(|source| { + RetentionStorageError::from(source).at(Boundary::NamespaceOpen) + })?; + synchronize_recovery_directory( + &self.roots, + Boundary::RootsSynchronization, + #[cfg(test)] + self.recovery_sync_failure, + )?; + durability = Durability::Synchronized; + let link = stage.link(&self.retention, &directory, name)?; + synchronize_recovery_directory( + &directory, + Boundary::PoolSynchronization, + #[cfg(test)] + self.recovery_sync_failure, + ) + .map_err(|error| link.report(error)) + })(); + result.map_err(|error| { + if created { + error.after(Effect::NamespaceCreated, durability) + } else { + error + } + }) + } + + fn link_manifest(&mut self) -> Result<(), RetentionStorageError> { + let context = self.recovery.as_ref().ok_or_else(no_recovery)?; + let Some(RecoveredStage::Complete { + stage, + pool_name: name, + .. + }) = context.manifest.as_ref() + else { + return Err(RetentionStorageError::from(invalid_data( + "link_manifest expected a complete manifest stage", + )) + .at(Boundary::RecoveryContext)); + }; + stage.synchronize(&self.retention)?; + let link = stage.link(&self.retention, &self.manifests, name)?; + synchronize_recovery_directory( + &self.manifests, + Boundary::PoolSynchronization, + #[cfg(test)] + self.recovery_sync_failure, + ) + .map_err(|error| link.report(error)) + } + + fn finalize_head(&mut self) -> Result<(), RetentionStorageError> { + let context = self.recovery.as_mut().ok_or_else(no_recovery)?; + let (stage, _name, _namespace) = take_complete(&mut context.head)?; + stage.synchronize(&self.retention)?; + stage.replace(&self.retention, pool_name::HEAD)?; + synchronize_recovery_directory( + &self.retention, + super::RetentionStorageBoundary::RetentionSynchronization, + #[cfg(test)] + self.recovery_sync_failure, + ) + .map_err(|error| error.after(Effect::HeadReplaced, Durability::Unconfirmed)) + } + + fn remove_root_stage(&mut self) -> Result<(), RetentionStorageError> { + let context = self.recovery.as_mut().ok_or_else(no_recovery)?; + let (stage, name, namespace) = take_complete(&mut context.root)?; + let namespace = namespace.ok_or_else(|| { + RetentionStorageError::from(invalid_data("root stage without a namespace")) + .at(Boundary::RecoveryContext) + })?; + let directory = self + .roots + .open_dir_nofollow(&namespace) + .map_err(|source| RetentionStorageError::from(source).at(Boundary::NamespaceOpen))?; + stage.remove(&self.retention, &directory, &name)?; + synchronize_recovery_directory( + &self.retention, + super::RetentionStorageBoundary::RetentionSynchronization, + #[cfg(test)] + self.recovery_sync_failure, + ) + .map_err(|error| error.after(Effect::StageRemoved, Durability::Unconfirmed)) + } + + fn remove_manifest_stage(&mut self) -> Result<(), RetentionStorageError> { + let context = self.recovery.as_mut().ok_or_else(no_recovery)?; + let (stage, name, _namespace) = take_complete(&mut context.manifest)?; + stage.remove(&self.retention, &self.manifests, &name)?; + synchronize_recovery_directory( + &self.retention, + super::RetentionStorageBoundary::RetentionSynchronization, + #[cfg(test)] + self.recovery_sync_failure, + ) + .map_err(|error| error.after(Effect::StageRemoved, Durability::Unconfirmed)) + } +} + +fn synchronize_recovery_directory( + directory: &Dir, + boundary: super::RetentionStorageBoundary, + #[cfg(test)] injected_failure: Option, +) -> Result<(), RetentionStorageError> { + #[cfg(test)] + if injected_failure == Some(boundary) { + return Err(RetentionStorageError::from(io::Error::from_raw_os_error( + rustix::io::Errno::IO.raw_os_error(), + )) + .at(boundary)); + } + synchronize_directory(directory) + .map_err(|source| RetentionStorageError::from(source).at(boundary)) +} diff --git a/src/adapters/retention/filesystem_retention_recovery_closure_law_tests.rs b/src/adapters/retention/filesystem_retention_recovery_closure_law_tests.rs new file mode 100644 index 00000000..a6738339 --- /dev/null +++ b/src/adapters/retention/filesystem_retention_recovery_closure_law_tests.rs @@ -0,0 +1,108 @@ +//! These laws own semantic closure replay at recovered head publication. + +use std::error::Error; + +use super::filesystem_retention_recovery_history_tests::install_history; +use super::filesystem_retention_test_fixture::{ + ROOT_HEX, drive_publication, fixture, initial_preparation, open_authority, retention_witness, +}; +use super::{AdmittedRetentionRoot, CanonicalRetentionRoot, FilesystemRetentionRecoveryError}; +use crate::{ + LayoutId, RetentionAnchor, RetentionClosureCounter, RetentionClosureLimits, + RetentionClosureVerificationError, RetentionPolicy, RetentionRoot, SegmentRecordIdentity, +}; + +// Size: medium. Oracle: every named closure member must exist in the current catalog. +// Delete only with this protocol or a stronger runtime-boundary replacement. +#[test] +fn recovered_head_refuses_an_absent_anchor_layout() -> Result<(), Box> { + let bytes = fixture(ROOT_HEX)?; + let template = AdmittedRetentionRoot::decode(&bytes)?; + let anchor = template + .root() + .anchors() + .first() + .ok_or("missing fixture anchor")?; + let mut text = anchor.layout_id().to_string(); + let last = text.pop().ok_or("empty layout identity")?; + text.push(if last == '0' { '1' } else { '0' }); + let absent: LayoutId = text.parse()?; + let root = RetentionRoot::new( + template.root().namespace().clone(), + template.root().generation(), + RetentionPolicy::new(template.root().profile(), template.root().limits()), + None, + vec![RetentionAnchor::new(anchor.blob_id(), absent)], + )?; + require_closure_refusal( + &CanonicalRetentionRoot::from_root(&root)?, + |error| { + matches!(error, RetentionClosureVerificationError::MissingMember { identity } + if *identity == SegmentRecordIdentity::Layout(absent)) + }, + "absent-layout", + ) +} + +// Size: medium. Oracle: stored physical closure limits apply during restart replay. +// Delete only with this protocol or a stronger runtime-boundary replacement. +#[test] +fn recovered_head_refuses_a_physical_closure_limit_violation() -> Result<(), Box> { + let bytes = fixture(ROOT_HEX)?; + let template = AdmittedRetentionRoot::decode(&bytes)?; + let limits = template.root().limits(); + let smaller = + RetentionClosureLimits::new(limits.nodes(), limits.depth(), limits.encoded_bytes(), 1)?; + let root = RetentionRoot::new( + template.root().namespace().clone(), + template.root().generation(), + RetentionPolicy::new(template.root().profile(), smaller), + None, + template.root().anchors().to_vec(), + )?; + require_closure_refusal( + &CanonicalRetentionRoot::from_root(&root)?, + |error| { + matches!(error, RetentionClosureVerificationError::LimitExceeded { + counter: RetentionClosureCounter::PhysicalBytes, maximum: 1, observed + } if *observed > 1) + }, + "physical-limit", + ) +} + +fn require_closure_refusal( + candidate: &CanonicalRetentionRoot, + oracle: impl FnOnce(&RetentionClosureVerificationError) -> bool, + label: &str, +) -> Result<(), Box> { + let (sandbox, mut authority) = open_authority(&format!("recovery-closure-law-{label}"))?; + let bytes = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&bytes)?; + drive_publication(&mut authority, &preparation, 13)?; + install_history(sandbox.path(), &preparation, candidate)?; + let before = retention_witness(sandbox.path())?; + let result = authority.recover(); + assert_eq!( + retention_witness(sandbox.path())?, + before, + "closure law {label} must preserve retained bytes: {result:?}" + ); + let error = match &result { + Err(FilesystemRetentionRecoveryError::Observe { source }) => source, + other => { + return Err( + format!("closure law {label} must refuse during observation: {other:?}").into(), + ); + } + }; + let closure = error + .get_ref() + .and_then(|source| source.downcast_ref::()) + .ok_or("closure refusal lost its typed source")?; + assert!( + oracle(closure), + "closure law {label} must preserve exact refusal: {result:?}" + ); + Ok(()) +} diff --git a/src/adapters/retention/filesystem_retention_recovery_closure_tests.rs b/src/adapters/retention/filesystem_retention_recovery_closure_tests.rs new file mode 100644 index 00000000..437adfdb --- /dev/null +++ b/src/adapters/retention/filesystem_retention_recovery_closure_tests.rs @@ -0,0 +1,144 @@ +//! These laws own transitive catalog admission before recovery publication. + +use std::error::Error; +use std::fs; +use std::io; + +use super::FilesystemRetentionRecoveryError; +use super::filesystem_retention_test_fixture::{ + CATALOG_NAME, ROOT_HEX, SEGMENT_NAME, drive_publication, fixture, initial_preparation, + open_authority, retention_witness, +}; +use crate::adapters::{ + CatalogDecodeError, CatalogRestartError, CatalogRestartPhase, SegmentReadError, + SegmentSealError, +}; + +#[derive(Clone, Copy, Debug)] +enum Damage { + MissingSegment, + CorruptSegment, + MissingCatalog, + CorruptCatalog, +} + +// Size: medium. Oracle: complete recovery stages require authenticated transitive members. +// Delete only with the retention protocol or a stronger runtime replacement. +#[test] +fn recovery_refuses_missing_segments_before_root_link() -> Result<(), Box> { + require_refusal(Damage::MissingSegment, 2) +} +#[test] +fn recovery_refuses_missing_segments_before_manifest_link() -> Result<(), Box> { + require_refusal(Damage::MissingSegment, 8) +} +#[test] +fn recovery_refuses_missing_segments_before_head_commit() -> Result<(), Box> { + require_refusal(Damage::MissingSegment, 13) +} +#[test] +fn recovery_refuses_corrupt_segments_before_root_link() -> Result<(), Box> { + require_refusal(Damage::CorruptSegment, 2) +} +#[test] +fn recovery_refuses_corrupt_segments_before_manifest_link() -> Result<(), Box> { + require_refusal(Damage::CorruptSegment, 8) +} +#[test] +fn recovery_refuses_corrupt_segments_before_head_commit() -> Result<(), Box> { + require_refusal(Damage::CorruptSegment, 13) +} +#[test] +fn recovery_refuses_missing_catalogs_before_root_link() -> Result<(), Box> { + require_refusal(Damage::MissingCatalog, 2) +} +#[test] +fn recovery_refuses_missing_catalogs_before_manifest_link() -> Result<(), Box> { + require_refusal(Damage::MissingCatalog, 8) +} +#[test] +fn recovery_refuses_missing_catalogs_before_head_commit() -> Result<(), Box> { + require_refusal(Damage::MissingCatalog, 13) +} +#[test] +fn recovery_refuses_corrupt_catalogs_before_root_link() -> Result<(), Box> { + require_refusal(Damage::CorruptCatalog, 2) +} +#[test] +fn recovery_refuses_corrupt_catalogs_before_manifest_link() -> Result<(), Box> { + require_refusal(Damage::CorruptCatalog, 8) +} +#[test] +fn recovery_refuses_corrupt_catalogs_before_head_commit() -> Result<(), Box> { + require_refusal(Damage::CorruptCatalog, 13) +} + +fn require_refusal(damage: Damage, prefix: usize) -> Result<(), Box> { + let label = format!("recovery-closure-{damage:?}-{prefix}"); + let (sandbox, mut authority) = open_authority(&label)?; + let bytes = fixture(ROOT_HEX)?; + drive_publication(&mut authority, &initial_preparation(&bytes)?, prefix)?; + let (directory, name) = match damage { + Damage::MissingSegment | Damage::CorruptSegment => ("segments", SEGMENT_NAME), + Damage::MissingCatalog | Damage::CorruptCatalog => ("catalogs", CATALOG_NAME), + }; + let path = sandbox.path().join(directory).join(name); + match damage { + Damage::MissingSegment | Damage::MissingCatalog => fs::remove_file(path)?, + Damage::CorruptSegment | Damage::CorruptCatalog => { + let mut bytes = fs::read(&path)?; + let offset = match damage { + Damage::CorruptCatalog => bytes.len().checked_sub(64), + _ => bytes.len().checked_sub(1), + } + .ok_or("artifact lacks its integrity trailer")?; + *bytes.get_mut(offset).ok_or("integrity trailer is absent")? ^= 1; + fs::write(path, bytes)?; + } + } + let before = retention_witness(sandbox.path())?; + let result = authority.recover(); + assert_eq!( + retention_witness(sandbox.path())?, + before, + "closure refusal must preserve retention evidence for {label}: {result:?}" + ); + let error = match &result { + Err(FilesystemRetentionRecoveryError::Observe { source }) => source, + other => { + return Err(format!( + "expected typed closure observation refusal for {label}: {other:?}" + ) + .into()); + } + }; + let restart = error + .get_ref() + .and_then(|source| source.downcast_ref::()); + assert!( + matches_damage(damage, restart), + "closure refusal must preserve exact {damage:?} source for {label}: {result:?}" + ); + Ok(()) +} + +fn matches_damage(damage: Damage, error: Option<&CatalogRestartError>) -> bool { + match (damage, error) { + (Damage::MissingSegment, Some(CatalogRestartError::Io { phase, source })) => { + *phase == CatalogRestartPhase::OpenSegment && source.kind() == io::ErrorKind::NotFound + } + (Damage::MissingCatalog, Some(CatalogRestartError::Io { phase, source })) => { + *phase == CatalogRestartPhase::OpenCatalog && source.kind() == io::ErrorKind::NotFound + } + (Damage::CorruptCatalog, Some(CatalogRestartError::Catalog { source })) => { + matches!(source, CatalogDecodeError::ChecksumMismatch { .. }) + } + (Damage::CorruptSegment, Some(CatalogRestartError::Segment { source, .. })) => matches!( + source.as_ref(), + SegmentReadError::Seal { + source: SegmentSealError::SealChecksumMismatch { .. } + } + ), + _ => false, + } +} diff --git a/src/adapters/retention/filesystem_retention_recovery_directory_tests.rs b/src/adapters/retention/filesystem_retention_recovery_directory_tests.rs new file mode 100644 index 00000000..8f5e941a --- /dev/null +++ b/src/adapters/retention/filesystem_retention_recovery_directory_tests.rs @@ -0,0 +1,67 @@ +//! These laws own directory-capability binding before explicit recovery. + +use std::error::Error; +use std::fs; +use std::io; + +use super::filesystem_retention_test_fixture::{ROOT_HEX, fixture, open_authority, refusal}; +use super::{FilesystemRetentionRecoveryError, RetentionCurrentStateRefusal}; + +#[test] +fn replaced_retention_directory_refuses_recovery_without_deleting_evidence() +-> Result<(), Box> { + assert_replacement_refuses("retention") +} + +#[test] +fn replaced_root_pool_refuses_recovery_without_deleting_evidence() -> Result<(), Box> { + assert_replacement_refuses("retention/roots") +} + +#[test] +fn replaced_manifest_pool_refuses_recovery_without_deleting_evidence() -> Result<(), Box> +{ + assert_replacement_refuses("retention/manifests") +} + +fn assert_replacement_refuses(name: &str) -> Result<(), Box> { + let fixture_name = format!("recovery-directory-binding-{}", name.replace('/', "-")); + let (sandbox, mut authority) = open_authority(&fixture_name)?; + let bytes = fixture(ROOT_HEX)?; + let prefix = bytes.get(..100).ok_or("root fixture too short")?; + let retention = sandbox.path().join("retention"); + let stage = retention.join("root.next"); + fs::write(&stage, prefix)?; + let replaced = sandbox.path().join(name); + let archived = replaced.with_extension("previous"); + fs::rename(&replaced, &archived)?; + fs::create_dir(&replaced)?; + let retained_stage = if name == "retention" { + fs::create_dir(replaced.join("roots"))?; + fs::create_dir(replaced.join("manifests"))?; + archived.join("root.next") + } else { + stage + }; + + let error = authority + .recover() + .err() + .ok_or("direct recovery mutated a replaced protocol directory")?; + assert_directory_refusal(error)?; + assert_eq!(fs::read(retained_stage)?, prefix); + assert!(!retention.join("HEAD").exists()); + Ok(()) +} + +fn assert_directory_refusal(error: FilesystemRetentionRecoveryError) -> Result<(), Box> { + let FilesystemRetentionRecoveryError::Observe { source } = error else { + return Err("directory replacement refused outside observation".into()); + }; + assert_eq!(source.kind(), io::ErrorKind::InvalidData); + assert!(matches!( + refusal(&source), + Some(RetentionCurrentStateRefusal::ProtocolDirectoryReplaced) + )); + Ok(()) +} diff --git a/src/adapters/retention/filesystem_retention_recovery_discard_prefix_tests.rs b/src/adapters/retention/filesystem_retention_recovery_discard_prefix_tests.rs new file mode 100644 index 00000000..4994cacb --- /dev/null +++ b/src/adapters/retention/filesystem_retention_recovery_discard_prefix_tests.rs @@ -0,0 +1,101 @@ +//! These laws own incomplete-stage preservation even when earlier evidence is missing. + +use std::error::Error; +use std::fs; + +use super::filesystem_retention_test_fixture::{ + ROOT_HEX, drive_publication, fixture, initial_preparation, manifest_pool_path, open_authority, + retention_witness, root_pool_path, +}; +use super::{FilesystemRetentionRecoveryError, RetentionFixedStage, RetentionRecoveryRefusal}; + +#[derive(Clone, Copy, Debug)] +enum Missing { + RootStage, + RootLink, + ManifestStage, + ManifestLink, +} + +// Size: medium. Oracle: a manifest write requires a complete linked root. +// Delete only with this protocol or a stronger runtime-boundary replacement. +#[test] +fn truncated_manifest_refuses_a_missing_root_stage() -> Result<(), Box> { + require_refusal(RetentionFixedStage::Manifest, Missing::RootStage) +} +#[test] +fn truncated_manifest_refuses_a_missing_root_link() -> Result<(), Box> { + require_refusal(RetentionFixedStage::Manifest, Missing::RootLink) +} + +// Size: medium. Oracle: a head write requires complete linked root and manifest. +// Delete only with this protocol or a stronger runtime-boundary replacement. +#[test] +fn truncated_head_refuses_a_missing_root_stage() -> Result<(), Box> { + require_refusal(RetentionFixedStage::Head, Missing::RootStage) +} +#[test] +fn truncated_head_refuses_a_missing_root_link() -> Result<(), Box> { + require_refusal(RetentionFixedStage::Head, Missing::RootLink) +} +#[test] +fn truncated_head_refuses_a_missing_manifest_stage() -> Result<(), Box> { + require_refusal(RetentionFixedStage::Head, Missing::ManifestStage) +} +#[test] +fn truncated_head_refuses_a_missing_manifest_link() -> Result<(), Box> { + require_refusal(RetentionFixedStage::Head, Missing::ManifestLink) +} + +fn require_refusal(stage: RetentionFixedStage, missing: Missing) -> Result<(), Box> { + let label = format!("recovery-impossible-discard-{stage:?}-{missing:?}"); + let (sandbox, mut authority) = open_authority(&label)?; + let bytes = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&bytes)?; + let (count, name) = match stage { + RetentionFixedStage::Manifest => (8, "manifest.next"), + RetentionFixedStage::Head => (13, "head.next"), + RetentionFixedStage::Root => return Err("root has no earlier stage".into()), + }; + drive_publication(&mut authority, &preparation, count)?; + let path = sandbox.path().join("retention").join(name); + let complete = fs::read(&path)?; + fs::write( + &path, + complete.get(..20).ok_or("stage shorter than prefix")?, + )?; + let earlier_stage = match missing { + Missing::RootStage => { + fs::remove_file(sandbox.path().join("retention/root.next"))?; + RetentionFixedStage::Root + } + Missing::RootLink => { + fs::remove_file(root_pool_path(sandbox.path(), preparation.candidate()))?; + RetentionFixedStage::Root + } + Missing::ManifestStage => { + fs::remove_file(sandbox.path().join("retention/manifest.next"))?; + RetentionFixedStage::Manifest + } + Missing::ManifestLink => { + fs::remove_file(manifest_pool_path(sandbox.path(), &preparation))?; + RetentionFixedStage::Manifest + } + }; + let before = retention_witness(sandbox.path())?; + let result = authority.recover(); + assert_eq!( + retention_witness(sandbox.path())?, + before, + "impossible {stage:?} discard must preserve retained bytes: {result:?}" + ); + assert!( + matches!(result, Err(FilesystemRetentionRecoveryError::Plan { + source: RetentionRecoveryRefusal::IncompleteStageRequiresDisposition { + stage: observed, observed: 20, .. + } + }) if observed == stage), + "incomplete {stage:?} requires disposition even without {earlier_stage:?}: {result:?}" + ); + Ok(()) +} diff --git a/src/adapters/retention/filesystem_retention_recovery_effect_tests.rs b/src/adapters/retention/filesystem_retention_recovery_effect_tests.rs new file mode 100644 index 00000000..9d6519c2 --- /dev/null +++ b/src/adapters/retention/filesystem_retention_recovery_effect_tests.rs @@ -0,0 +1,203 @@ +//! This module owns filesystem effects reported after recovery synchronization failures. + +use crate::adapters::FilesystemVersionTwoAdmission; +use crate::adapters::retention::filesystem_retention_test_fixture::{ + ROOT_HEX, drive_publication, fixture, initial_preparation, manifest_pool_path, open_authority, + retention_witness, root_pool_path, +}; +use crate::adapters::retention::{ + FilesystemRetentionPublicationAuthority, FilesystemRetentionRecoveryError, + RetentionEffectDurability as Durability, RetentionNamespaceEffect as Effect, + RetentionRecoveryStep as Step, RetentionStorageBoundary as Boundary, +}; +use std::{error::Error, fs, io}; + +struct Case { + phases: usize, + boundary: Boundary, + step: Step, + effects: &'static [(Effect, Durability)], +} + +// Size: medium. Oracle: successful namespace calls remain visible after an injected sync failure. +// Actual filesystem operations with deterministic pre-sync EIO; no claim of physical power loss. +// Delete only when stronger public failure/restart coverage subsumes each capability below. +#[test] +fn recovery_sync_failures_report_exact_known_effects_before_restart() -> Result<(), Box> +{ + for case in [ + Case { + phases: 2, + boundary: Boundary::RootsSynchronization, + step: Step::LinkRoot, + effects: &[(Effect::NamespaceCreated, Durability::Unconfirmed)], + }, + Case { + phases: 2, + boundary: Boundary::PoolSynchronization, + step: Step::LinkRoot, + effects: &[ + (Effect::NamespaceCreated, Durability::Synchronized), + (Effect::PoolLinkCreated, Durability::Unconfirmed), + ], + }, + Case { + phases: 8, + boundary: Boundary::PoolSynchronization, + step: Step::LinkManifest, + effects: &[(Effect::PoolLinkCreated, Durability::Unconfirmed)], + }, + Case { + phases: 13, + boundary: Boundary::RetentionSynchronization, + step: Step::FinalizeHead, + effects: &[(Effect::HeadReplaced, Durability::Unconfirmed)], + }, + Case { + phases: 16, + boundary: Boundary::RetentionSynchronization, + step: Step::RemoveManifestStage, + effects: &[(Effect::StageRemoved, Durability::Unconfirmed)], + }, + ] { + require_effects(&case)?; + } + Ok(()) +} + +fn require_effects(case: &Case) -> Result<(), Box> { + let (sandbox, mut authority) = open_authority(&format!( + "recovery-effect-{:?}-{:?}", + case.step, case.boundary + ))?; + let root = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root)?; + drive_publication(&mut authority, &preparation, case.phases)?; + let mut expected = retention_witness(sandbox.path())?; + authority.recovery_sync_failure = Some(case.boundary); + let error = match authority.recover() { + Err(FilesystemRetentionRecoveryError::Execute { source }) => source, + result => return Err(format!("expected sync failure: {result:?}").into()), + }; + require_progress(&error, case)?; + apply_expected_effects(case, sandbox.path(), &preparation, &root, &mut expected)?; + assert_eq!( + retention_witness(sandbox.path())?, + expected, + "only reported effects may change retained evidence" + ); + drop(authority); + let mut restarted = FilesystemRetentionPublicationAuthority::open( + FilesystemVersionTwoAdmission::reopen_unchecked_for_repository_tasks(sandbox.path())?, + )?; + let receipt = restarted.recover()?; + assert_eq!( + receipt.outcome(), + match case.step { + Step::LinkRoot => crate::RetentionRecoveryOutcome::Protected { + root_stage: true, + manifest_stage: false + }, + Step::LinkManifest => crate::RetentionRecoveryOutcome::Protected { + root_stage: true, + manifest_stage: true + }, + Step::FinalizeHead => crate::RetentionRecoveryOutcome::Committed, + _ => crate::RetentionRecoveryOutcome::Clean, + } + ); + assert_eq!( + fs::read(root_pool_path(sandbox.path(), preparation.candidate()))?, + root, + "restart retains exact root evidence" + ); + Ok(()) +} + +fn require_progress( + error: &crate::RetentionRecoveryError, + case: &Case, +) -> Result<(), Box> { + assert_eq!(error.step(), case.step); + assert!( + error.executed().is_empty(), + "effects belong to the first failing capability" + ); + let progress = error.progress().ok_or("missing effects")?; + assert_eq!(progress.boundary(), case.boundary); + assert_eq!( + progress + .known_effects() + .iter() + .map(|effect| (effect.effect(), effect.durability())) + .collect::>(), + case.effects + ); + assert_eq!(progress.uncertain_effect(), None); + assert_eq!( + io_cause(error).and_then(io::Error::raw_os_error), + Some(rustix::io::Errno::IO.raw_os_error()) + ); + Ok(()) +} + +fn apply_expected_effects( + case: &Case, + store: &std::path::Path, + preparation: &crate::RetentionPublicationPreparation<'_>, + root: &[u8], + expected: &mut std::collections::BTreeSet<(std::ffi::OsString, Vec)>, +) -> Result<(), Box> { + let publication = preparation.publication().ok_or("publication absent")?; + for &(effect, _) in case.effects { + match effect { + Effect::NamespaceCreated => assert!( + root_pool_path(store, preparation.candidate()) + .parent() + .ok_or("pool parent absent")? + .is_dir(), + "reported namespace creation must be visible" + ), + Effect::PoolLinkCreated => { + let (path, bytes) = if case.step == Step::LinkRoot { + ( + root_pool_path(store, preparation.candidate()), + root.to_vec(), + ) + } else { + ( + manifest_pool_path(store, preparation), + publication.manifest().encoded().to_vec(), + ) + }; + expected.insert((path.into_os_string(), bytes)); + } + Effect::HeadReplaced => { + expected.remove(&( + store.join("retention/head.next").into_os_string(), + publication.head().encoded().to_vec(), + )); + expected.insert(( + store.join("retention/HEAD").into_os_string(), + publication.head().encoded().to_vec(), + )); + } + Effect::StageRemoved => { + expected.remove(&( + store.join("retention/manifest.next").into_os_string(), + publication.manifest().encoded().to_vec(), + )); + } + } + } + Ok(()) +} + +fn io_cause<'a>(mut error: &'a (dyn Error + 'static)) -> Option<&'a io::Error> { + loop { + if let Some(source) = error.downcast_ref::() { + return Some(source); + } + error = error.source()?; + } +} diff --git a/src/adapters/retention/filesystem_retention_recovery_entry_set_tests.rs b/src/adapters/retention/filesystem_retention_recovery_entry_set_tests.rs new file mode 100644 index 00000000..369f7315 --- /dev/null +++ b/src/adapters/retention/filesystem_retention_recovery_entry_set_tests.rs @@ -0,0 +1,158 @@ +//! These laws own unrelated namespace preservation in recovered successors. + +use std::error::Error; +use std::fs; +use std::path::Path; + +use super::filesystem_retention_pool_name as pool_name; +use super::filesystem_retention_test_fixture::{ + ROOT_HEX, drive_publication, fixture, initial_preparation, initial_root, manifest_pool_path, + new_namespace_preparation, open_authority, retention_witness, successor_preparation, + successor_root, +}; +use super::{ + AdmittedRetentionManifest, AdmittedRetentionRoot, CanonicalRetentionHead, + CanonicalRetentionManifest, FilesystemRetentionRecoveryError, RetentionRecoveryRefusal, +}; +use crate::{ + RetentionHead, RetentionManifest, RetentionManifestEntry, RetentionManifestLength, + execute_retention_publication, +}; + +#[derive(Clone, Copy)] +enum Change { + Drop, + Alter, +} +#[derive(Clone, Copy)] +enum Point { + Manifest, + Head, +} + +// Size: medium. Oracle: only the staged root's namespace may change. +// Delete only with this protocol or a stronger boundary-level replacement. +#[test] +fn recovery_refuses_dropped_namespaces_before_manifest_link() -> Result<(), Box> { + require_refusal(Change::Drop, Point::Manifest) +} +#[test] +fn recovery_refuses_dropped_namespaces_before_head_commit() -> Result<(), Box> { + require_refusal(Change::Drop, Point::Head) +} +#[test] +fn recovery_refuses_altered_namespaces_before_manifest_link() -> Result<(), Box> { + require_refusal(Change::Alter, Point::Manifest) +} +#[test] +fn recovery_refuses_altered_namespaces_before_head_commit() -> Result<(), Box> { + require_refusal(Change::Alter, Point::Head) +} + +fn require_refusal(change: Change, point: Point) -> Result<(), Box> { + let label = match (change, point) { + (Change::Drop, Point::Manifest) => "drop-manifest", + (Change::Drop, Point::Head) => "drop-head", + (Change::Alter, Point::Manifest) => "alter-manifest", + (Change::Alter, Point::Head) => "alter-head", + }; + let (sandbox, mut authority) = open_authority(&format!("recovery-entry-set-{label}"))?; + let root = fixture(ROOT_HEX)?; + let admitted_root = AdmittedRetentionRoot::decode(&root)?; + let _first = execute_retention_publication(&mut authority, &initial_preparation(&root)?)?; + let first = authority + .observe_current()? + .ok_or("missing initial state")?; + let first_manifest = AdmittedRetentionManifest::decode(first.manifest_bytes())?; + let other = initial_root(b"unrelated", &admitted_root)?; + let addition = new_namespace_preparation(&first_manifest, other.encoded())?; + let _second = execute_retention_publication(&mut authority, &addition)?; + let current = authority + .observe_current()? + .ok_or("missing multi-namespace state")?; + let manifest = AdmittedRetentionManifest::decode(current.manifest_bytes())?; + let successor = successor_root(&admitted_root)?; + let preparation = successor_preparation(&admitted_root, &manifest, successor.encoded())?; + let prefix = match point { + Point::Manifest => 8, + Point::Head => 13, + }; + drive_publication(&mut authority, &preparation, prefix)?; + let publication = preparation.publication().ok_or("missing publication")?; + let staged = AdmittedRetentionManifest::decode(publication.manifest().encoded())?; + let candidate = preparation.candidate().root().namespace().digest(); + let entries = changed_entries(staged.manifest().entries(), candidate, change)?; + let changed = RetentionManifest::new( + staged.manifest().generation(), + staged.manifest().predecessor(), + entries, + )?; + let encoded = CanonicalRetentionManifest::from_manifest(&changed)?; + install_changed_stage(sandbox.path(), &preparation, &changed, &encoded, point)?; + let before = retention_witness(sandbox.path())?; + + let result = authority.recover(); + + assert_eq!( + retention_witness(sandbox.path())?, + before, + "unrelated namespace refusal must preserve every retained byte for {label}: {result:?}" + ); + assert!( + matches!( + result, + Err(FilesystemRetentionRecoveryError::Plan { + source: RetentionRecoveryRefusal::ManifestNotSuccessor + }) + ), + "unrelated namespace change must refuse before recovery effects: {result:?}" + ); + Ok(()) +} + +fn changed_entries( + entries: &[RetentionManifestEntry], + candidate: crate::RetentionNamespaceDigest, + change: Change, +) -> Result, Box> { + let mut changed = Vec::new(); + for entry in entries { + if entry.namespace() == candidate { + changed.push(*entry); + } else if matches!(change, Change::Alter) { + changed.push(RetentionManifestEntry::new( + entry.namespace(), + entry.root_generation().successor()?, + entry.root_digest(), + )); + } + } + Ok(changed) +} + +fn install_changed_stage( + root: &Path, + preparation: &super::RetentionPublicationPreparation<'_>, + manifest: &RetentionManifest, + encoded: &CanonicalRetentionManifest, + point: Point, +) -> Result<(), Box> { + let stage = root.join("retention/manifest.next"); + fs::write(&stage, encoded.encoded())?; + if matches!(point, Point::Head) { + fs::remove_file(manifest_pool_path(root, preparation))?; + let name = pool_name::manifest(manifest.generation(), encoded.digest()); + fs::hard_link(&stage, root.join("retention/manifests").join(name))?; + let head = RetentionHead::new( + manifest.generation(), + RetentionManifestLength::new(u64::try_from(encoded.encoded().len())?)?, + encoded.digest(), + manifest.predecessor(), + )?; + fs::write( + root.join("retention/head.next"), + CanonicalRetentionHead::from_head(&head).encoded(), + )?; + } + Ok(()) +} diff --git a/src/adapters/retention/filesystem_retention_recovery_error.rs b/src/adapters/retention/filesystem_retention_recovery_error.rs new file mode 100644 index 00000000..8746f340 --- /dev/null +++ b/src/adapters/retention/filesystem_retention_recovery_error.rs @@ -0,0 +1,48 @@ +//! This module owns the typed error of filesystem retention recovery. + +use std::error::Error; +use std::fmt; +use std::io; + +use super::{RetentionRecoveryError, RetentionRecoveryRefusal}; + +/// Why filesystem retention recovery did not reach a receipt. +#[derive(Debug)] +#[non_exhaustive] +pub enum FilesystemRetentionRecoveryError { + /// Reading the current state, a stage, or a pool entry failed. + Observe { + /// The exact filesystem or admission failure. + source: io::Error, + }, + /// The observed stages require disposition or contradict the recovery contract. + Plan { + /// The exact planning refusal. + source: RetentionRecoveryRefusal, + }, + /// A recovery step failed, possibly after its own effects and earlier completed steps. + Execute { + /// The failed step, completed prefix, failing-capability progress and original cause. + source: RetentionRecoveryError, + }, +} + +impl fmt::Display for FilesystemRetentionRecoveryError { + fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { + formatter.write_str(match self { + Self::Observe { .. } => "retention recovery could not observe the store", + Self::Plan { .. } => "retention recovery refused the observed stages", + Self::Execute { .. } => "a retention recovery step failed", + }) + } +} + +impl Error for FilesystemRetentionRecoveryError { + fn source(&self) -> Option<&(dyn Error + 'static)> { + match self { + Self::Observe { source } => Some(source), + Self::Plan { source } => Some(source), + Self::Execute { source } => Some(source), + } + } +} diff --git a/src/adapters/retention/filesystem_retention_recovery_head_binding_tests.rs b/src/adapters/retention/filesystem_retention_recovery_head_binding_tests.rs new file mode 100644 index 00000000..dd551ae9 --- /dev/null +++ b/src/adapters/retention/filesystem_retention_recovery_head_binding_tests.rs @@ -0,0 +1,131 @@ +//! These laws own staged head coordinates bound to the exact staged manifest. + +use std::error::Error; +use std::fs; +use std::io; +use std::path::Path; + +use super::filesystem_retention_test_fixture::{ + ROOT_HEX, drive_publication, fixture, head_path, initial_preparation, open_authority, + retention_witness, successor_preparation, successor_root, +}; +use super::{ + AdmittedRetentionManifest, AdmittedRetentionRoot, CanonicalRetentionHead, + ChecksummedRetentionHead, FilesystemRetentionPublicationAuthority, + FilesystemRetentionRecoveryError, RetentionRecoveryRefusal, +}; +use crate::{RetentionHead, RetentionManifestLength, execute_retention_publication}; + +#[derive(Clone, Copy)] +enum Binding { + Length, + Predecessor, +} + +// Size: medium. Oracle: head length must equal its selected manifest's bytes. +// Delete only with the protocol or a stronger boundary-level replacement. +#[test] +fn recovery_refuses_a_canonical_head_with_the_wrong_manifest_length() -> Result<(), Box> +{ + let (sandbox, mut authority) = open_authority("recovery-head-manifest-length")?; + let root = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root)?; + drive_publication(&mut authority, &preparation, 13)?; + let publication = preparation.publication().ok_or("missing publication")?; + let head = ChecksummedRetentionHead::decode(publication.head().encoded())?; + let mismatched = RetentionHead::new( + head.head().generation(), + RetentionManifestLength::MINIMUM, + head.head().manifest_digest(), + head.head().predecessor(), + )?; + fs::write( + sandbox.path().join("retention/head.next"), + CanonicalRetentionHead::from_head(&mismatched).encoded(), + )?; + + require_binding_refusal(&mut authority, sandbox.path(), Binding::Length) +} + +// Size: medium. Oracle: staged head and manifest must name one predecessor. +// Delete only with the protocol or a stronger boundary-level replacement. +#[test] +fn recovery_refuses_a_canonical_head_with_another_manifest_predecessor() +-> Result<(), Box> { + let (sandbox, mut authority) = open_authority("recovery-head-manifest-predecessor")?; + let root = fixture(ROOT_HEX)?; + let _published = execute_retention_publication(&mut authority, &initial_preparation(&root)?)?; + let current = authority + .observe_current()? + .ok_or("missing current generation")?; + let admitted_root = AdmittedRetentionRoot::decode(&root)?; + let manifest = AdmittedRetentionManifest::decode(current.manifest_bytes())?; + let successor = successor_root(&admitted_root)?; + let preparation = successor_preparation(&admitted_root, &manifest, successor.encoded())?; + drive_publication(&mut authority, &preparation, 13)?; + let publication = preparation + .publication() + .ok_or("missing successor publication")?; + let head = ChecksummedRetentionHead::decode(publication.head().encoded())?; + let mismatched = RetentionHead::new( + head.head().generation(), + head.head().manifest_length(), + head.head().manifest_digest(), + Some(publication.manifest().digest()), + )?; + fs::write( + sandbox.path().join("retention/head.next"), + CanonicalRetentionHead::from_head(&mismatched).encoded(), + )?; + + require_binding_refusal(&mut authority, sandbox.path(), Binding::Predecessor) +} + +fn require_binding_refusal( + authority: &mut FilesystemRetentionPublicationAuthority, + root: &Path, + binding: Binding, +) -> Result<(), Box> { + let head_before = read_head(root)?; + let before = retention_witness(root)?; + + let result = authority.recover(); + + assert_eq!( + read_head(root)?, + head_before, + "head/manifest binding refusal must preserve the published head: {result:?}" + ); + let correct = matches!( + (binding, &result), + ( + Binding::Length, + Err(FilesystemRetentionRecoveryError::Plan { + source: RetentionRecoveryRefusal::HeadStageNamesOtherManifest + }) + ) | ( + Binding::Predecessor, + Err(FilesystemRetentionRecoveryError::Plan { + source: RetentionRecoveryRefusal::HeadPredecessorMismatch + }) + ) + ); + assert!( + correct, + "head/manifest binding must refuse before execution: {result:?}" + ); + assert_eq!( + retention_witness(root)?, + before, + "head/manifest binding refusal must preserve every retained byte" + ); + Ok(()) +} + +fn read_head(root: &Path) -> io::Result>> { + match fs::read(head_path(root)) { + Ok(bytes) => Ok(Some(bytes)), + Err(source) if source.kind() == io::ErrorKind::NotFound => Ok(None), + Err(source) => Err(source), + } +} diff --git a/src/adapters/retention/filesystem_retention_recovery_history_domain_tests.rs b/src/adapters/retention/filesystem_retention_recovery_history_domain_tests.rs new file mode 100644 index 00000000..0b217630 --- /dev/null +++ b/src/adapters/retention/filesystem_retention_recovery_history_domain_tests.rs @@ -0,0 +1,66 @@ +//! These laws own a bounded candidate-generation domain at filesystem recovery. + +use std::error::Error; + +use super::filesystem_retention_recovery_history_tests::install_history; +use super::filesystem_retention_test_fixture::{ + ROOT_HEX, drive_publication, fixture, initial_preparation, open_authority, retention_witness, + successor_preparation, successor_root, +}; +use super::{ + AdmittedRetentionManifest, AdmittedRetentionRoot, CanonicalRetentionRoot, + FilesystemRetentionRecoveryError, RetentionRecoveryRefusal, +}; +use crate::{RetentionPolicy, RetentionRoot, RootGeneration, execute_retention_publication}; + +// Size: medium. Oracle: generation one admits only candidate generation two. +// Deterministic domain: 3..=16 plus u64::MAX, with ascending minimal counterexample. +// Replay: cargo test --lib --all-features recovered_heads_refuse_the_skipped_generation_domain +// Delete only with this protocol or a stronger runtime boundary replacement. +#[test] +fn recovered_heads_refuse_the_skipped_generation_domain() -> Result<(), Box> { + for generation in (3..=16).chain([u64::MAX]) { + require_generation_refusal(generation)?; + } + Ok(()) +} + +fn require_generation_refusal(generation: u64) -> Result<(), Box> { + let (sandbox, mut authority) = open_authority("recovery-skipped-generation-domain")?; + let bytes = fixture(ROOT_HEX)?; + let initial = AdmittedRetentionRoot::decode(&bytes)?; + let _published = execute_retention_publication(&mut authority, &initial_preparation(&bytes)?)?; + let current = authority + .observe_current()? + .ok_or("missing current state")?; + let manifest = AdmittedRetentionManifest::decode(current.manifest_bytes())?; + let successor = successor_root(&initial)?; + let preparation = successor_preparation(&initial, &manifest, successor.encoded())?; + drive_publication(&mut authority, &preparation, 13)?; + let skipped = RetentionRoot::new( + initial.root().namespace().clone(), + RootGeneration::new(generation)?, + RetentionPolicy::new(initial.root().profile(), initial.root().limits()), + Some(initial.digest()), + initial.root().anchors().to_vec(), + )?; + let encoded = CanonicalRetentionRoot::from_root(&skipped)?; + install_history(sandbox.path(), &preparation, &encoded)?; + let before = retention_witness(sandbox.path())?; + let result = authority.recover(); + assert_eq!( + retention_witness(sandbox.path())?, + before, + "candidate generation {generation} must preserve every retained byte: {result:?}" + ); + assert!( + matches!( + result, + Err(FilesystemRetentionRecoveryError::Plan { + source: RetentionRecoveryRefusal::RootNotSuccessor + }) + ), + "candidate generation {generation} must refuse with RootNotSuccessor: {result:?}" + ); + Ok(()) +} diff --git a/src/adapters/retention/filesystem_retention_recovery_history_tests.rs b/src/adapters/retention/filesystem_retention_recovery_history_tests.rs new file mode 100644 index 00000000..1de511e7 --- /dev/null +++ b/src/adapters/retention/filesystem_retention_recovery_history_tests.rs @@ -0,0 +1,182 @@ +//! These laws own root-history refusal before recovered head publication. + +use std::error::Error; +use std::fs; +use std::path::Path; + +use super::filesystem_retention_pool_name as pool_name; +use super::filesystem_retention_test_fixture::{ + ROOT_HEX, drive_publication, fixture, initial_preparation, initial_root, manifest_pool_path, + new_namespace_preparation, open_authority, retention_witness, root_pool_path, + successor_preparation, successor_root, +}; +use super::{ + AdmittedRetentionManifest, AdmittedRetentionRoot, CanonicalRetentionHead, + CanonicalRetentionManifest, CanonicalRetentionRoot, FilesystemRetentionRecoveryError, + RetentionPublicationPreparation, RetentionRecoveryRefusal, +}; +use crate::{ + RetentionHead, RetentionManifest, RetentionManifestEntry, RetentionManifestLength, + RetentionPolicy, RetentionRoot, execute_retention_publication, +}; + +// Size: medium. Oracle: a selected root admits only its exact next generation. +// Delete only with the protocol or a stronger runtime boundary replacement. +#[test] +fn recovered_head_refuses_a_skipped_root_generation() -> Result<(), Box> { + let (sandbox, mut authority) = open_authority("recovery-skipped-root-generation")?; + let bytes = fixture(ROOT_HEX)?; + let initial = AdmittedRetentionRoot::decode(&bytes)?; + let _published = execute_retention_publication(&mut authority, &initial_preparation(&bytes)?)?; + let current = authority + .observe_current()? + .ok_or("missing current state")?; + let manifest = AdmittedRetentionManifest::decode(current.manifest_bytes())?; + let successor = successor_root(&initial)?; + let preparation = successor_preparation(&initial, &manifest, successor.encoded())?; + drive_publication(&mut authority, &preparation, 13)?; + let skipped = RetentionRoot::new( + initial.root().namespace().clone(), + initial.root().generation().successor()?.successor()?, + RetentionPolicy::new(initial.root().profile(), initial.root().limits()), + Some(initial.digest()), + initial.root().anchors().to_vec(), + )?; + let encoded = CanonicalRetentionRoot::from_root(&skipped)?; + install_history(sandbox.path(), &preparation, &encoded)?; + let before = retention_witness(sandbox.path())?; + + let result = authority.recover(); + + assert_eq!( + retention_witness(sandbox.path())?, + before, + "skipped root generation must preserve every retained byte: {result:?}" + ); + assert!( + matches!( + result, + Err(FilesystemRetentionRecoveryError::Plan { + source: RetentionRecoveryRefusal::RootNotSuccessor + }) + ), + "skipped root generation must refuse before head publication: {result:?}" + ); + Ok(()) +} + +pub(super) fn install_history( + root: &Path, + preparation: &RetentionPublicationPreparation<'_>, + encoded: &CanonicalRetentionRoot, +) -> Result<(), Box> { + let old = preparation.candidate(); + let candidate = AdmittedRetentionRoot::decode(encoded.encoded())?; + let stage = root.join("retention/root.next"); + fs::write(&stage, encoded.encoded())?; + fs::remove_file(root_pool_path(root, old))?; + fs::hard_link(&stage, root_pool_path(root, &candidate))?; + let publication = preparation.publication().ok_or("missing publication")?; + let staged = AdmittedRetentionManifest::decode(publication.manifest().encoded())?; + let entries = staged + .manifest() + .entries() + .iter() + .map(|entry| { + if entry.namespace() == candidate.root().namespace().digest() { + RetentionManifestEntry::new( + entry.namespace(), + candidate.root().generation(), + candidate.digest(), + ) + } else { + *entry + } + }) + .collect(); + let manifest = RetentionManifest::new( + staged.manifest().generation(), + staged.manifest().predecessor(), + entries, + )?; + let canonical = CanonicalRetentionManifest::from_manifest(&manifest)?; + let stage = root.join("retention/manifest.next"); + fs::write(&stage, canonical.encoded())?; + fs::remove_file(manifest_pool_path(root, preparation))?; + fs::hard_link( + &stage, + root.join("retention/manifests").join(pool_name::manifest( + manifest.generation(), + canonical.digest(), + )), + )?; + let head = RetentionHead::new( + manifest.generation(), + RetentionManifestLength::new(u64::try_from(canonical.encoded().len())?)?, + canonical.digest(), + manifest.predecessor(), + )?; + fs::write( + root.join("retention/head.next"), + CanonicalRetentionHead::from_head(&head).encoded(), + )?; + Ok(()) +} + +// Size: medium. Oracle: the first publication starts at root generation one. +// Delete only with this protocol or a stronger runtime boundary replacement. +#[test] +fn recovered_initial_head_refuses_noninitial_root_history() -> Result<(), Box> { + let (sandbox, mut authority) = open_authority("recovery-noninitial-first-root")?; + let bytes = fixture(ROOT_HEX)?; + let initial = AdmittedRetentionRoot::decode(&bytes)?; + let preparation = initial_preparation(&bytes)?; + drive_publication(&mut authority, &preparation, 13)?; + let noninitial = successor_root(&initial)?; + install_history(sandbox.path(), &preparation, &noninitial)?; + require_history_refusal(sandbox.path(), &mut authority) +} + +// Size: medium. Oracle: a newly inserted namespace starts at root generation one. +// Delete only with this protocol or a stronger runtime boundary replacement. +#[test] +fn recovered_inserted_namespace_refuses_noninitial_root_history() -> Result<(), Box> { + let (sandbox, mut authority) = open_authority("recovery-noninitial-inserted-root")?; + let bytes = fixture(ROOT_HEX)?; + let template = AdmittedRetentionRoot::decode(&bytes)?; + let _published = execute_retention_publication(&mut authority, &initial_preparation(&bytes)?)?; + let current = authority + .observe_current()? + .ok_or("missing current state")?; + let manifest = AdmittedRetentionManifest::decode(current.manifest_bytes())?; + let initial = initial_root(b"new-namespace", &template)?; + let candidate = AdmittedRetentionRoot::decode(initial.encoded())?; + let preparation = new_namespace_preparation(&manifest, initial.encoded())?; + drive_publication(&mut authority, &preparation, 13)?; + let noninitial = successor_root(&candidate)?; + install_history(sandbox.path(), &preparation, &noninitial)?; + require_history_refusal(sandbox.path(), &mut authority) +} + +fn require_history_refusal( + root: &Path, + authority: &mut super::filesystem_retention_authority::FilesystemRetentionPublicationAuthority, +) -> Result<(), Box> { + let before = retention_witness(root)?; + let result = authority.recover(); + assert_eq!( + retention_witness(root)?, + before, + "noninitial namespace history must preserve every retained byte: {result:?}" + ); + assert!( + matches!( + result, + Err(FilesystemRetentionRecoveryError::Plan { + source: RetentionRecoveryRefusal::RootNotSuccessor + }) + ), + "noninitial namespace history must refuse before head publication: {result:?}" + ); + Ok(()) +} diff --git a/src/adapters/retention/filesystem_retention_recovery_identity_tests.rs b/src/adapters/retention/filesystem_retention_recovery_identity_tests.rs new file mode 100644 index 00000000..52605551 --- /dev/null +++ b/src/adapters/retention/filesystem_retention_recovery_identity_tests.rs @@ -0,0 +1,74 @@ +//! These laws own pool inode admission before retention recovery publication. + +use std::error::Error; +use std::fs; +use std::os::unix::fs::MetadataExt; + +use super::filesystem_retention_test_fixture::{ + ROOT_HEX, drive_publication, fixture, head_path, initial_preparation, manifest_pool_path, + open_authority, retention_witness, root_pool_path, +}; +use super::{FilesystemRetentionRecoveryError, RetentionPool, RetentionRecoveryRefusal}; + +// Size: medium. Oracle: a retained stage's pool link must name the same inode. +// Delete only with the protocol or a stronger public-boundary replacement. +#[test] +fn substituted_root_pool_refuses_before_recovery_commits_head() -> Result<(), Box> { + require_substitution_refusal(RetentionPool::Roots) +} + +// Size: medium. Oracle: identical bytes do not admit a substituted pool inode. +// Delete only with the protocol or a stronger public-boundary replacement. +#[test] +fn substituted_manifest_pool_refuses_before_recovery_commits_head() -> Result<(), Box> { + require_substitution_refusal(RetentionPool::Manifests) +} + +fn require_substitution_refusal(pool: RetentionPool) -> Result<(), Box> { + let label = match pool { + RetentionPool::Roots => "root", + RetentionPool::Manifests => "manifest", + }; + let (sandbox, mut authority) = open_authority(&format!("recovery-substituted-{label}"))?; + let root_bytes = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root_bytes)?; + drive_publication(&mut authority, &preparation, 13)?; + let target = match pool { + RetentionPool::Roots => root_pool_path(sandbox.path(), preparation.candidate()), + RetentionPool::Manifests => manifest_pool_path(sandbox.path(), &preparation), + }; + let stage = sandbox + .path() + .join("retention") + .join(format!("{label}.next")); + let replacement = target.with_extension("replacement"); + fs::copy(&target, &replacement)?; + fs::rename(replacement, &target)?; + let source = fs::metadata(&stage)?; + let replaced = fs::metadata(&target)?; + let stage_identity = (source.dev(), source.ino()); + let pool_identity = (replaced.dev(), replaced.ino()); + if stage_identity == pool_identity || fs::read(&stage)? != fs::read(&target)? { + return Err("fixture must contain equal bytes on distinct inodes".into()); + } + let before = retention_witness(sandbox.path())?; + + let result = authority.recover(); + + assert!( + !head_path(sandbox.path()).exists(), + "recovery must not commit HEAD over a substituted {label} pool inode: {result:?}" + ); + assert!( + matches!(result, Err(FilesystemRetentionRecoveryError::Plan { + source: RetentionRecoveryRefusal::PoolEntryDiffers { pool: observed } + }) if observed == pool), + "substituted {label} must refuse at pool admission: {result:?}" + ); + assert_eq!( + retention_witness(sandbox.path())?, + before, + "substituted {label} refusal must preserve every retained byte" + ); + Ok(()) +} diff --git a/src/adapters/retention/filesystem_retention_recovery_namespace_tests.rs b/src/adapters/retention/filesystem_retention_recovery_namespace_tests.rs new file mode 100644 index 00000000..a259ab91 --- /dev/null +++ b/src/adapters/retention/filesystem_retention_recovery_namespace_tests.rs @@ -0,0 +1,121 @@ +//! These laws own namespace refusal before recovery changes retained evidence. + +use std::error::Error; +use std::fs; +use std::io; + +use super::filesystem_retention_test_fixture::{ + HEAD_HEX, MANIFEST_HEX, ROOT_HEX, fixture, initial_preparation, open_authority, refusal, + retention_witness, +}; +use super::{ + FilesystemRetentionRecoveryError, RetentionCurrentStateRefusal, RetentionPublicationStorage, +}; + +#[derive(Clone, Copy)] +enum EntryPoint { + Recovery, + Publication, +} + +// Size: medium. Oracle: unknown namespace state refuses before any mutation. +// Delete only when the protocol is removed or stronger filesystem evidence subsumes this law. +#[test] +fn direct_recovery_preserves_stages_when_namespace_admission_refuses() -> Result<(), Box> +{ + require_namespace_refusal(EntryPoint::Recovery) +} + +// Size: medium. Oracle: publication-triggered recovery has the same admission law. +// Delete only with its behavior or a stronger public-boundary replacement. +#[test] +fn publication_recovery_preserves_stages_when_namespace_admission_refuses() +-> Result<(), Box> { + require_namespace_refusal(EntryPoint::Publication) +} + +fn require_namespace_refusal(entry_point: EntryPoint) -> Result<(), Box> { + let label = match entry_point { + EntryPoint::Recovery => "direct", + EntryPoint::Publication => "publication", + }; + for (stage, hex) in [ + ("root.next", ROOT_HEX), + ("manifest.next", MANIFEST_HEX), + ("head.next", HEAD_HEX), + ] { + for intruder in [ + "foreign.dat", + "roots/not-a-digest", + "manifests/bogus.manifest", + ] { + let name = format!("namespace-{label}-{stage}-{}", intruder.replace('/', "-")); + let (sandbox, mut authority) = open_authority(&name)?; + let retention = sandbox.path().join("retention"); + let bytes = fixture(hex)?; + fs::write( + retention.join(stage), + bytes.get(..8).ok_or("missing prefix")?, + )?; + fs::write(retention.join(intruder), b"unadmitted evidence")?; + let before = retention_witness(sandbox.path())?; + let root_bytes = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root_bytes)?; + + let error = match entry_point { + EntryPoint::Recovery => recovery_refusal(authority.recover())?, + EntryPoint::Publication => { + RetentionPublicationStorage::verify_current(&mut authority, &preparation) + .err() + .ok_or("publication admitted an unknown namespace")? + } + }; + + require_typed_refusal(&error, intruder); + assert_eq!( + retention_witness(sandbox.path())?, + before, + "namespace refusal must preserve all retained bytes for {stage} with {intruder}" + ); + } + } + Ok(()) +} + +fn recovery_refusal( + result: Result, +) -> Result> { + match result { + Err(FilesystemRetentionRecoveryError::Observe { source }) => Ok(source), + observed => { + Err(format!("namespace admission must refuse before recovery: {observed:?}").into()) + } + } +} + +fn require_typed_refusal(error: &io::Error, intruder: &str) { + assert_eq!( + error.kind(), + io::ErrorKind::InvalidData, + "namespace refusal for {intruder} must retain the InvalidData kind" + ); + let matches = matches!( + (intruder, refusal(error)), + ( + "foreign.dat", + Some(RetentionCurrentStateRefusal::UnknownRetentionEntry) + ) | ( + "roots/not-a-digest", + Some(RetentionCurrentStateRefusal::NonNamespaceEntry) + ) | ( + "manifests/bogus.manifest", + Some(RetentionCurrentStateRefusal::NoncanonicalPoolEntry { + pool: "manifest pool", + }), + ) + ); + assert!( + matches, + "exact namespace refusal missing for {intruder}: {error:?}" + ); +} diff --git a/src/adapters/retention/filesystem_retention_recovery_observation.rs b/src/adapters/retention/filesystem_retention_recovery_observation.rs new file mode 100644 index 00000000..b756f1fd --- /dev/null +++ b/src/adapters/retention/filesystem_retention_recovery_observation.rs @@ -0,0 +1,163 @@ +//! This module owns restart observation of the retention stages and pools. + +use std::io::{self, Read}; + +use cap_fs_ext::DirExt; +use cap_std::fs::Dir; + +use super::filesystem_retention_current::{self, ObservedRetentionState}; +use super::filesystem_retention_pool_name as pool_name; +use super::{ + RetentionPoolEntryObservation as Pool, RetentionPoolObservations, RetentionRecoveryEvidence, + RetentionStageAssessment, RetentionStageAssessments, assess_head_stage, assess_manifest_stage, + assess_root_stage, head_decoder, manifest_header_decoder, root_header_decoder, +}; +use crate::adapters::filesystem_exact_record::{ + self as exact_record, EntryIdentity, ExactRecordError, +}; + +/// The exact bytes and entry identity of one retained stage. +pub(super) struct StageBytes { + pub(super) bytes: Box<[u8]>, + pub(super) identity: EntryIdentity, +} + +/// Everything restart read under writer authority before planning recovery. +pub(super) struct RetentionRecoveryObservation { + current: Option, + root: Option, + manifest: Option, + head: Option, + pools: RetentionPoolObservations, +} + +impl RetentionRecoveryObservation { + /// Reads the current state, the three stages, and the pool entries the + /// complete stages name. Performs no mutation. + pub(super) fn observe(retention: &Dir, roots: &Dir, manifests: &Dir) -> io::Result { + let current = filesystem_retention_current::observe(retention, manifests)?; + let root = read_stage( + retention, + pool_name::ROOT_STAGE, + root_header_decoder::MAXIMUM_ENCODED_LENGTH, + )?; + let manifest = read_stage( + retention, + pool_name::MANIFEST_STAGE, + manifest_header_decoder::maximum_encoded_length() + .map_err(|source| io::Error::new(io::ErrorKind::InvalidData, source))?, + )?; + let head = read_stage( + retention, + pool_name::HEAD_STAGE, + head_decoder::ENCODED_LENGTH, + )?; + let root_pool = match root + .as_ref() + .map(|stage| (assess_root_stage(Some(&stage.bytes)), stage.identity)) + { + Some((RetentionStageAssessment::Complete(admitted), identity)) => { + let namespace = pool_name::namespace(admitted.root().namespace().digest()); + let name = pool_name::root(admitted.root().generation(), admitted.digest()); + match roots.open_dir_nofollow(namespace) { + Ok(directory) => pool_entry(&directory, &name, admitted.encoded(), identity)?, + Err(source) if source.kind() == io::ErrorKind::NotFound => Pool::Absent, + Err(source) => return Err(source), + } + } + _ => Pool::Absent, + }; + let manifest_pool = match manifest + .as_ref() + .map(|stage| (assess_manifest_stage(Some(&stage.bytes)), stage.identity)) + { + Some((RetentionStageAssessment::Complete(admitted), identity)) => { + let name = pool_name::manifest(admitted.manifest().generation(), admitted.digest()); + pool_entry(manifests, &name, admitted.encoded(), identity)? + } + _ => Pool::Absent, + }; + Ok(Self { + current, + root, + manifest, + head, + pools: RetentionPoolObservations { + root: root_pool, + manifest: manifest_pool, + }, + }) + } + + /// The pure evidence recovery plans from. + pub(super) fn evidence(&self) -> RetentionRecoveryEvidence<'_, '_> { + RetentionRecoveryEvidence::new( + self.current.as_ref(), + RetentionStageAssessments { + root: assess_root_stage(self.root.as_ref().map(|stage| &*stage.bytes)), + manifest: assess_manifest_stage(self.manifest.as_ref().map(|stage| &*stage.bytes)), + head: assess_head_stage(self.head.as_ref().map(|stage| &*stage.bytes)), + }, + self.pools, + ) + } + + pub(super) const fn root(&self) -> Option<&StageBytes> { + self.root.as_ref() + } + + pub(super) const fn manifest(&self) -> Option<&StageBytes> { + self.manifest.as_ref() + } + + pub(super) const fn head(&self) -> Option<&StageBytes> { + self.head.as_ref() + } +} + +/// Reads a stage's complete bytes up to one byte past `bound`. +/// +/// A stage longer than its format's maximum is returned in full up to that +/// point so assessment classifies it as corrupt rather than truncated. +fn read_stage(retention: &Dir, name: &str, bound: usize) -> io::Result> { + let mut file = match exact_record::open_read(retention, name) { + Ok(file) => file, + Err(source) if source.kind() == io::ErrorKind::NotFound => return Ok(None), + Err(source) => return Err(source), + }; + let metadata = file.metadata()?; + if !metadata.is_file() { + return Err(io::Error::new( + io::ErrorKind::InvalidData, + "retained retention stage is not a regular file", + )); + } + let identity = EntryIdentity::from(&metadata); + let limit = bound + .checked_add(1) + .and_then(|limit| u64::try_from(limit).ok()) + .ok_or_else(|| io::Error::new(io::ErrorKind::InvalidData, "stage bound overflowed"))?; + let mut bytes = Vec::new(); + file.by_ref().take(limit).read_to_end(&mut bytes)?; + Ok(Some(StageBytes { + bytes: bytes.into_boxed_slice(), + identity, + })) +} + +/// Whether the pool entry names the exact retained stage inode and bytes. +fn pool_entry( + directory: &Dir, + name: &str, + expected: &[u8], + identity: EntryIdentity, +) -> io::Result { + match exact_record::verify_named(directory, name, expected, identity) { + Ok(()) => Ok(Pool::Identical), + Err(ExactRecordError::Io(source)) if source.kind() == io::ErrorKind::NotFound => { + Ok(Pool::Absent) + } + Err(ExactRecordError::Refused(_)) => Ok(Pool::Different), + Err(ExactRecordError::Io(source)) => Err(source), + } +} diff --git a/src/adapters/retention/filesystem_retention_recovery_policy.rs b/src/adapters/retention/filesystem_retention_recovery_policy.rs new file mode 100644 index 00000000..cf2274cb --- /dev/null +++ b/src/adapters/retention/filesystem_retention_recovery_policy.rs @@ -0,0 +1,39 @@ +//! This module owns the default catalog resource policy for retention authority. + +use std::io; + +use super::filesystem_retention_authority::FilesystemRetentionPublicationAuthority; +use super::{FilesystemRetentionRecoveryError, RetentionRecoveryReceipt}; +use crate::adapters::segment_header::MAXIMUM_SEGMENT_LENGTH; +use crate::adapters::{CatalogRestartByteLimit, CatalogRestartPolicy, SegmentReadPolicy}; + +impl FilesystemRetentionPublicationAuthority { + /// Observes, plans, and executes fixed-stage recovery under writer authority. + /// + /// Complete staged roots materialize the current catalog and all selected + /// segments before closure verification. Aggregate retained segment bytes + /// are bounded to one protocol-maximum segment (1 GiB), independently of + /// the root's own closure limits; catalog and decoded indexes also retain + /// their protocol bounds. Larger catalog selections refuse without mutation. + /// Use [`Self::recover_with_catalog_policy`] for explicit caller-selected + /// loading bounds. Clean and incomplete-root recovery does not load segments. + /// The synchronous call may allocate and block on filesystem I/O. + /// + /// # Errors + /// + /// Returns the exact observation, planning, resource, closure, or execution + /// refusal; catalog and closure errors remain inspectable as error sources. + pub fn recover( + &mut self, + ) -> Result { + let policy = default_catalog_policy() + .map_err(|source| FilesystemRetentionRecoveryError::Observe { source })?; + self.recover_with_catalog_policy(policy) + } +} + +pub(super) fn default_catalog_policy() -> io::Result { + let bound = CatalogRestartByteLimit::new(MAXIMUM_SEGMENT_LENGTH) + .map_err(|source| io::Error::new(io::ErrorKind::InvalidInput, source))?; + Ok(CatalogRestartPolicy::new(SegmentReadPolicy::MAXIMUM, bound)) +} diff --git a/src/adapters/retention/filesystem_retention_recovery_policy_tests.rs b/src/adapters/retention/filesystem_retention_recovery_policy_tests.rs new file mode 100644 index 00000000..556ee079 --- /dev/null +++ b/src/adapters/retention/filesystem_retention_recovery_policy_tests.rs @@ -0,0 +1,75 @@ +//! These laws own explicit recovery catalog-memory admission at its boundary. + +use std::error::Error; +use std::fs; + +use super::filesystem_retention_test_fixture::{ + ROOT_HEX, SEGMENT_NAME, drive_publication, fixture, initial_preparation, open_authority, + retention_witness, +}; +use super::{FilesystemRetentionRecoveryError, RetentionRecoveryOutcome}; +use crate::adapters::{ + CatalogRestartByteLimit, CatalogRestartError, CatalogRestartPolicy, SegmentReadPolicy, +}; + +// Size: medium. Oracle: aggregate segment retention refuses before exceeding caller budget. +// Delete only with explicit recovery policy or a stronger runtime replacement. +#[test] +fn recovery_refuses_a_catalog_budget_smaller_than_selected_segments() -> Result<(), Box> +{ + let (sandbox, mut authority) = open_authority("recovery-catalog-budget-refusal")?; + let bytes = fixture(ROOT_HEX)?; + drive_publication(&mut authority, &initial_preparation(&bytes)?, 2)?; + let observed = fs::metadata(sandbox.path().join("segments").join(SEGMENT_NAME))?.len(); + let maximum = observed.checked_sub(1).ok_or("empty selected segment")?; + let policy = CatalogRestartPolicy::new( + SegmentReadPolicy::MAXIMUM, + CatalogRestartByteLimit::new(maximum)?, + ); + let before = retention_witness(sandbox.path())?; + let result = authority.recover_with_catalog_policy(policy); + assert_eq!( + retention_witness(sandbox.path())?, + before, + "catalog budget refusal must preserve retained evidence: {result:?}" + ); + let error = match &result { + Err(FilesystemRetentionRecoveryError::Observe { source }) => source, + other => { + return Err(format!("catalog budget must refuse during observation: {other:?}").into()); + } + }; + let restart = error + .get_ref() + .and_then(|source| source.downcast_ref::()); + assert!( + matches!(restart, Some(CatalogRestartError::RetainedSegmentBytes { + maximum: actual_maximum, observed: actual_observed + }) if *actual_maximum == maximum && *actual_observed == observed), + "catalog budget must name exact maximum {maximum} and observed {observed}: {result:?}" + ); + Ok(()) +} + +// Size: medium. Oracle: an inclusive aggregate byte budget admits exact-size selection. +// Delete only with explicit recovery policy or a stronger runtime replacement. +#[test] +fn recovery_admits_a_catalog_budget_equal_to_selected_segment_bytes() -> Result<(), Box> +{ + let (sandbox, mut authority) = open_authority("recovery-catalog-budget-equality")?; + let bytes = fixture(ROOT_HEX)?; + drive_publication(&mut authority, &initial_preparation(&bytes)?, 2)?; + let maximum = fs::metadata(sandbox.path().join("segments").join(SEGMENT_NAME))?.len(); + let policy = CatalogRestartPolicy::new( + SegmentReadPolicy::MAXIMUM, + CatalogRestartByteLimit::new(maximum)?, + ); + let result = authority.recover_with_catalog_policy(policy); + assert!( + matches!(result, Ok(ref receipt) if receipt.outcome() == RetentionRecoveryOutcome::Protected { + root_stage: true, manifest_stage: false + }), + "exact-size catalog budget must protect the admitted root: {result:?}" + ); + Ok(()) +} diff --git a/src/adapters/retention/filesystem_retention_recovery_predecessor_tests.rs b/src/adapters/retention/filesystem_retention_recovery_predecessor_tests.rs new file mode 100644 index 00000000..ec31e3ab --- /dev/null +++ b/src/adapters/retention/filesystem_retention_recovery_predecessor_tests.rs @@ -0,0 +1,136 @@ +//! These laws own selected predecessor admission before recovery effects. +//! Size: medium. Oracle: recovery must reopen the exact manifest-selected root. +//! Delete only with the protocol or a stronger public-boundary replacement. + +use std::error::Error; +use std::fs; + +use super::filesystem_retention_test_fixture::{ + ROOT_HEX, drive_publication, fixture, initial_preparation, initial_root, open_authority, + refusal, retention_witness, root_pool_path, successor_preparation, successor_root, +}; +use super::{ + AdmittedRetentionManifest, AdmittedRetentionRoot, FilesystemRetentionRecoveryError, + RetentionCurrentStateRefusal, +}; +use crate::execute_retention_publication; + +#[derive(Clone, Copy, Debug)] +enum Damage { + Missing, + Corrupt, + Substituted, +} + +#[test] +fn recovery_refuses_a_missing_predecessor_before_root_link() -> Result<(), Box> { + require_refusal(Damage::Missing, 2) +} + +#[test] +fn recovery_refuses_a_corrupt_predecessor_before_root_link() -> Result<(), Box> { + require_refusal(Damage::Corrupt, 2) +} + +#[test] +fn recovery_refuses_a_substituted_predecessor_before_root_link() -> Result<(), Box> { + require_refusal(Damage::Substituted, 2) +} + +#[test] +fn recovery_refuses_a_missing_predecessor_before_manifest_link() -> Result<(), Box> { + require_refusal(Damage::Missing, 8) +} + +#[test] +fn recovery_refuses_a_corrupt_predecessor_before_manifest_link() -> Result<(), Box> { + require_refusal(Damage::Corrupt, 8) +} + +#[test] +fn recovery_refuses_a_substituted_predecessor_before_manifest_link() -> Result<(), Box> { + require_refusal(Damage::Substituted, 8) +} + +#[test] +fn recovery_refuses_a_missing_predecessor_before_head_commit() -> Result<(), Box> { + require_refusal(Damage::Missing, 13) +} + +#[test] +fn recovery_refuses_a_corrupt_predecessor_before_head_commit() -> Result<(), Box> { + require_refusal(Damage::Corrupt, 13) +} + +#[test] +fn recovery_refuses_a_substituted_predecessor_before_head_commit() -> Result<(), Box> { + require_refusal(Damage::Substituted, 13) +} + +fn require_refusal(damage: Damage, prefix: usize) -> Result<(), Box> { + // Deterministic fault matrix: root link, manifest link, head replacement. + let label = format!("recovery-predecessor-{damage:?}-{prefix}"); + let (sandbox, mut authority) = open_authority(&label)?; + let root = fixture(ROOT_HEX)?; + let admitted = AdmittedRetentionRoot::decode(&root)?; + let _published = execute_retention_publication(&mut authority, &initial_preparation(&root)?)?; + let current = authority + .observe_current()? + .ok_or("missing published state")?; + let manifest = AdmittedRetentionManifest::decode(current.manifest_bytes())?; + let successor = successor_root(&admitted)?; + let preparation = successor_preparation(&admitted, &manifest, successor.encoded())?; + drive_publication(&mut authority, &preparation, prefix)?; + let selected = root_pool_path(sandbox.path(), &admitted); + damage_predecessor(&selected, &admitted, damage)?; + let before = retention_witness(sandbox.path())?; + + let result = authority.recover(); + + assert_eq!( + retention_witness(sandbox.path())?, + before, + "selected predecessor refusal must preserve retained bytes for {label}: {result:?}" + ); + let expected = match (&result, damage) { + (Err(FilesystemRetentionRecoveryError::Observe { source }), Damage::Missing) => { + matches!( + refusal(source), + Some(RetentionCurrentStateRefusal::PredecessorRootAbsent) + ) + } + (Err(FilesystemRetentionRecoveryError::Observe { source }), _) => { + matches!( + refusal(source), + Some(RetentionCurrentStateRefusal::PredecessorRootChanged) + ) + } + _ => false, + }; + assert!( + expected, + "selected predecessor must refuse at observation for {label}: {result:?}" + ); + Ok(()) +} + +fn damage_predecessor( + path: &std::path::Path, + root: &AdmittedRetentionRoot<'_>, + damage: Damage, +) -> Result<(), Box> { + match damage { + Damage::Missing => fs::remove_file(path)?, + Damage::Corrupt => { + let mut bytes = root.encoded().to_vec(); + let checksum = bytes.last_mut().ok_or("missing root checksum")?; + *checksum ^= 1; + fs::write(path, bytes)?; + } + Damage::Substituted => { + let replacement = initial_root(b"replacement", root)?; + fs::write(path, replacement.encoded())?; + } + } + Ok(()) +} diff --git a/src/adapters/retention/filesystem_retention_recovery_prefix_tests.rs b/src/adapters/retention/filesystem_retention_recovery_prefix_tests.rs new file mode 100644 index 00000000..4abba79d --- /dev/null +++ b/src/adapters/retention/filesystem_retention_recovery_prefix_tests.rs @@ -0,0 +1,287 @@ +//! Every publication crash prefix recovers to exactly one documented state. + +use std::error::Error; +use std::fs; +use std::path::Path; + +use super::filesystem_retention_test_fixture::{ + MANIFEST_HEX, PUBLICATION_PHASE_COUNT, ROOT_HEX, drive_publication, fixture, + initial_preparation, open_authority, refusal, successor_preparation, successor_root, +}; +use super::{ + AdmittedRetentionManifest, AdmittedRetentionRoot, RetentionCurrentStateRefusal, + RetentionPublicationError, RetentionPublicationOutcome, RetentionRecoveryOutcome as Outcome, + RetentionRecoveryStep as Step, +}; +use crate::execute_retention_publication; + +const PROTECTED_ROOT: Outcome = Outcome::Protected { + root_stage: true, + manifest_stage: false, +}; +const PROTECTED_BOTH: Outcome = Outcome::Protected { + root_stage: true, + manifest_stage: true, +}; + +/// The documented recovery of a crash after `count` phases, and the forward +/// retry's result afterwards. +fn expected(count: usize) -> (Vec, Outcome, Retry) { + match count { + 0 | 1 => (vec![], Outcome::Clean, Retry::Published), + 2..=5 => (vec![Step::LinkRoot], PROTECTED_ROOT, Retry::Refused), + 6 | 7 => (vec![], PROTECTED_ROOT, Retry::Refused), + 8 | 9 => (vec![Step::LinkManifest], PROTECTED_BOTH, Retry::Refused), + 10 | 11 => (vec![], PROTECTED_BOTH, Retry::Refused), + 12 | 13 => ( + vec![ + Step::FinalizeHead, + Step::RemoveRootStage, + Step::RemoveManifestStage, + ], + Outcome::Committed, + Retry::AlreadyCommitted, + ), + 14 | 15 => ( + vec![Step::RemoveRootStage, Step::RemoveManifestStage], + Outcome::Committed, + Retry::AlreadyCommitted, + ), + 16 => ( + vec![Step::RemoveManifestStage], + Outcome::Committed, + Retry::AlreadyCommitted, + ), + _ => (vec![], Outcome::Clean, Retry::AlreadyCommitted), + } +} + +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +enum Retry { + Published, + AlreadyCommitted, + Refused, +} + +fn stages_present(root: &Path) -> [bool; 3] { + let retention = root.join("retention"); + ["root.next", "manifest.next", "head.next"].map(|stage| retention.join(stage).exists()) +} + +fn stages_for(outcome: Outcome) -> [bool; 3] { + match outcome { + Outcome::Clean | Outcome::Committed => [false, false, false], + Outcome::Protected { + root_stage, + manifest_stage, + } => [root_stage, manifest_stage, false], + } +} + +#[test] +fn every_initial_publication_prefix_recovers_to_its_documented_state() -> Result<(), Box> +{ + for count in 0..=PUBLICATION_PHASE_COUNT { + let name = format!("filesystem-retention-recovery-prefix-{count}"); + let (sandbox, mut authority) = open_authority(&name)?; + let root_bytes = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root_bytes)?; + drive_publication(&mut authority, &preparation, count)?; + let (steps, outcome, retry) = expected(count); + + let receipt = authority + .recover() + .map_err(|error| format!("prefix {count}: {error}"))?; + + assert_eq!(receipt.executed(), steps, "prefix {count}: steps"); + assert_eq!(receipt.outcome(), outcome, "prefix {count}: outcome"); + assert_eq!( + stages_present(sandbox.path()), + stages_for(outcome), + "prefix {count}: stages" + ); + let second = authority + .recover() + .map_err(|error| format!("prefix {count} again: {error}"))?; + assert!( + second.executed().is_empty(), + "prefix {count}: recovery is idempotent" + ); + match ( + retry, + execute_retention_publication(&mut authority, &preparation), + ) { + (Retry::Published, Ok(receipt)) => { + assert_eq!(receipt.outcome(), RetentionPublicationOutcome::Published); + } + (Retry::AlreadyCommitted, Ok(receipt)) => { + assert_eq!( + receipt.outcome(), + RetentionPublicationOutcome::AlreadyCommitted + ); + } + (Retry::Refused, Err(RetentionPublicationError::CurrentVerification { source })) => { + assert!( + matches!( + refusal(&source), + Some(RetentionCurrentStateRefusal::RetainedStage) + ), + "prefix {count}: protected orphans refuse forward publication" + ); + } + (retry, result) => { + return Err(format!("prefix {count}: expected {retry:?}, got {result:?}").into()); + } + } + } + Ok(()) +} + +#[test] +fn a_crash_during_each_stage_write_requires_disposition_without_effects() +-> Result<(), Box> { + for (phase, stage, fixed) in [ + (2, "root.next", super::RetentionFixedStage::Root), + (8, "manifest.next", super::RetentionFixedStage::Manifest), + (12, "head.next", super::RetentionFixedStage::Head), + ] { + let name = format!("filesystem-retention-recovery-during-{phase}"); + let (sandbox, mut authority) = open_authority(&name)?; + let root_bytes = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root_bytes)?; + drive_publication(&mut authority, &preparation, phase - 1)?; + let publication = preparation.publication().ok_or("no publication")?; + let complete: &[u8] = match phase { + 2 => preparation.candidate().encoded(), + 8 => publication.manifest().encoded(), + _ => publication.head().encoded(), + }; + // 100 bytes is inside every record's framing: the head is 144 bytes, the + // manifest header 160, and the root header 192. + let partial = complete.get(..100).ok_or("record shorter than 100 bytes")?; + fs::write(sandbox.path().join("retention").join(stage), partial)?; + let before = super::filesystem_retention_test_fixture::retention_witness(sandbox.path())?; + let result = authority.recover(); + assert!( + matches!(result, Err(super::FilesystemRetentionRecoveryError::Plan { source: super::RetentionRecoveryRefusal::IncompleteStageRequiresDisposition { stage: actual, observed: 100, .. } }) if actual == fixed), + "during {phase}: typed disposition required: {result:?}" + ); + assert_eq!( + super::filesystem_retention_test_fixture::retention_witness(sandbox.path())?, + before, + "during {phase}: no recovery mutation" + ); + } + Ok(()) +} + +// Size: medium. Oracle: every ordered successor prefix preserves the old root until head finalization. +// Delete when stronger complete-prefix recovery evidence subsumes these runtime outcomes. +#[test] +fn successor_prefixes_recover_against_the_published_generation() -> Result<(), Box> { + for count in 0..=PUBLICATION_PHASE_COUNT { + let name = format!("filesystem-retention-recovery-successor-{count}"); + let (sandbox, mut authority) = open_authority(&name)?; + let root_bytes = fixture(ROOT_HEX)?; + let _published = + execute_retention_publication(&mut authority, &initial_preparation(&root_bytes)?)?; + let current_root = AdmittedRetentionRoot::decode(&root_bytes)?; + let manifest_bytes = fixture(MANIFEST_HEX)?; + let current_manifest = AdmittedRetentionManifest::decode(&manifest_bytes)?; + let candidate = successor_root(¤t_root)?; + let preparation = + successor_preparation(¤t_root, ¤t_manifest, candidate.encoded())?; + drive_publication(&mut authority, &preparation, count)?; + let (steps, outcome, _retry) = expected(count); + + let receipt = authority + .recover() + .map_err(|error| format!("successor {count}: {error}"))?; + + assert_eq!(receipt.executed(), steps, "successor {count}: steps"); + assert_eq!(receipt.outcome(), outcome, "successor {count}: outcome"); + let selected = if count < 12 { + root_bytes.as_slice() + } else { + candidate.encoded() + }; + require_successor_view(sandbox.path(), selected, count)?; + } + Ok(()) +} + +// Size: medium (owned filesystem); oracle: Core Law and recovery refusal +// contract. Delete only if the same public recovery refusal and exact-byte +// preservation are covered by a stronger test. +#[test] +fn noncanonical_short_stages_refuse_recovery_without_changing_retained_bytes() +-> Result<(), Box> { + use super::filesystem_retention_test_fixture::retention_witness; + use super::{FilesystemRetentionRecoveryError, RetentionFixedStage, RetentionRecoveryRefusal}; + + for (name, expected_stage) in [ + ("root.next", RetentionFixedStage::Root), + ("manifest.next", RetentionFixedStage::Manifest), + ("head.next", RetentionFixedStage::Head), + ] { + let (sandbox, mut authority) = open_authority(&format!("short-corrupt-{name}"))?; + let root_bytes = fixture(ROOT_HEX)?; + let _published = + execute_retention_publication(&mut authority, &initial_preparation(&root_bytes)?)?; + fs::write(sandbox.path().join("retention").join(name), b"invalid")?; + let before = retention_witness(sandbox.path())?; + + let result = authority.recover(); + + assert!( + matches!( + result, + Err(FilesystemRetentionRecoveryError::Plan { + source: RetentionRecoveryRefusal::StageCorrupt { stage, .. } + }) if stage == expected_stage + ), + "noncanonical {name} must refuse recovery, observed {result:?}" + ); + assert_eq!( + retention_witness(sandbox.path())?, + before, + "refusal must preserve every retained byte for {name}" + ); + } + Ok(()) +} + +/// Verifies persistent selected bytes independently of the recovery receipt. +fn require_successor_view( + root: &Path, + expected: &[u8], + prefix: usize, +) -> Result<(), Box> { + use crate::{ + CatalogRestartByteLimit, CatalogRestartPolicy, FilesystemRetentionSnapshot, + ReaderAttemptLimit, SegmentReadPolicy, + }; + let policy = CatalogRestartPolicy::new( + SegmentReadPolicy::MAXIMUM, + CatalogRestartByteLimit::new(1_048_576)?, + ); + let view = FilesystemRetentionSnapshot::load(root, policy, ReaderAttemptLimit::DEFAULT)?; + let expected_root = AdmittedRetentionRoot::decode(expected)?; + let head = view + .retention_head() + .ok_or("successor recovery lost the published head")?; + let expected_generation = if prefix < 12 { 1 } else { 2 }; + assert_eq!( + head.generation().get(), + expected_generation, + "successor prefix {prefix}: published generation" + ); + let selected = view + .retained_root(expected_root.root().namespace().digest())? + .ok_or("successor recovery lost the selected root")?; + assert_eq!( + &*selected, expected, + "successor prefix {prefix}: exact selected root bytes" + ); + Ok(()) +} diff --git a/src/adapters/retention/filesystem_retention_recovery_roots.rs b/src/adapters/retention/filesystem_retention_recovery_roots.rs new file mode 100644 index 00000000..ec949365 --- /dev/null +++ b/src/adapters/retention/filesystem_retention_recovery_roots.rs @@ -0,0 +1,41 @@ +//! This module owns selected root reopening before filesystem recovery effects. + +use std::io; + +use cap_std::fs::Dir; + +use super::filesystem_retention_current::{verify_committed, verify_predecessor}; +use super::{RetentionRecoveryEvidence, RetentionStageAssessment}; + +/// Reopens the published root selected for the complete staged root's namespace. +/// +/// A successor must authenticate the predecessor selected by the published +/// manifest, just as forward publication does. If that selection is already +/// the staged root, recovery is cleaning up a committed transition and instead +/// requires the selected bytes to equal that root. Initial namespaces have no +/// published predecessor. The pure plan has already refused contradictory +/// stage coordinates; this admission performs bounded reads and no mutation. +pub(super) fn admit(roots: &Dir, evidence: &RetentionRecoveryEvidence<'_, '_>) -> io::Result<()> { + let Some(current) = evidence.current() else { + return Ok(()); + }; + let RetentionStageAssessment::Complete(candidate) = &evidence.stages().root else { + return Ok(()); + }; + let namespace = candidate.root().namespace().digest(); + let entries = current.manifest().entries(); + let Some(entry) = entries + .binary_search_by_key(&namespace, |entry| entry.namespace()) + .ok() + .and_then(|index| entries.get(index)) + else { + return Ok(()); + }; + if entry.root_generation() == candidate.root().generation() + && entry.root_digest() == candidate.digest() + { + verify_committed(roots, current, candidate) + } else { + verify_predecessor(roots, current, candidate) + } +} diff --git a/src/adapters/retention/filesystem_retention_recovery_storage_error_tests.rs b/src/adapters/retention/filesystem_retention_recovery_storage_error_tests.rs new file mode 100644 index 00000000..bccfd390 --- /dev/null +++ b/src/adapters/retention/filesystem_retention_recovery_storage_error_tests.rs @@ -0,0 +1,286 @@ +//! This module owns typed recovery refusals after retained-stage admission. + +use std::error::Error; +use std::fs; +use std::io; + +use super::{RetentionRecoveryContext, RetentionRecoveryObservation}; +use crate::adapters::retention::filesystem_retention_test_fixture::{ + ROOT_HEX, drive_publication, fixture, initial_preparation, open_authority, retention_witness, +}; +use crate::adapters::retention::{ + RetentionRecordRefusal, RetentionRecoveryError, RetentionRecoveryStep, + execute_retention_recovery, plan_retention_recovery, +}; + +#[derive(Clone, Copy)] +enum Damage { + Bytes, + Identity, + Missing, +} + +// Size: medium. Oracle: byte disagreement remains distinguishable from inode substitution. +// Delete when recovery storage is removed or stronger public error laws subsume this scenario. +#[test] +fn changed_stage_bytes_preserve_the_recovery_record_refusal() -> Result<(), Box> { + let error = refused("recovery-error-bytes", Damage::Bytes)?; + assert_eq!( + cause::(&error), + Some(&RetentionRecordRefusal::Bytes), + "recovery must preserve the exact byte refusal: {error:?}" + ); + Ok(()) +} + +// Size: medium. Oracle: byte-identical inode substitution remains a typed identity refusal. +// Delete when recovery storage is removed or stronger public error laws subsume this scenario. +#[test] +fn substituted_stage_preserves_the_recovery_record_refusal() -> Result<(), Box> { + let error = refused("recovery-error-identity", Damage::Identity)?; + assert_eq!( + cause::(&error), + Some(&RetentionRecordRefusal::KindLengthOrIdentity), + "recovery must preserve the exact identity refusal: {error:?}" + ); + Ok(()) +} + +// Size: medium. Oracle: missing retained stages preserve the original filesystem error. +// Delete when recovery storage is removed or stronger public error laws subsume this scenario. +#[test] +fn a_missing_stage_preserves_the_recovery_io_cause() -> Result<(), Box> { + let error = refused("recovery-error-missing", Damage::Missing)?; + assert_eq!( + cause::(&error).map(io::Error::kind), + Some(io::ErrorKind::NotFound), + "recovery must preserve the missing-stage I/O error: {error:?}" + ); + assert_eq!( + cause::(&error).and_then(io::Error::raw_os_error), + Some(rustix::io::Errno::NOENT.raw_os_error()), + "recovery must preserve the original OS error code: {error:?}" + ); + Ok(()) +} + +fn refused(name: &str, damage: Damage) -> Result> { + let (sandbox, mut authority) = open_authority(name)?; + let bytes = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&bytes)?; + drive_publication(&mut authority, &preparation, 2)?; + let observation = RetentionRecoveryObservation::observe( + &authority.retention, + &authority.roots, + &authority.manifests, + )?; + let plan = plan_retention_recovery(observation.evidence())?; + authority.recovery = Some(RetentionRecoveryContext::reopen( + &authority.retention, + &observation, + )?); + let path = sandbox.path().join("retention/root.next"); + match damage { + Damage::Bytes => { + let mut changed = fs::read(&path)?; + *changed.last_mut().ok_or("empty root stage")? ^= 1; + fs::write(&path, changed)?; + } + Damage::Identity => { + let replacement = path.with_extension("replacement"); + fs::copy(&path, &replacement)?; + fs::rename(replacement, &path)?; + } + Damage::Missing => fs::remove_file(&path)?, + } + let before = retention_witness(sandbox.path())?; + + let error = execute_retention_recovery(&mut authority, &plan) + .err() + .ok_or("damaged recovery stage accepted")?; + + assert_eq!( + error.step(), + RetentionRecoveryStep::LinkRoot, + "the first root transition must refuse" + ); + assert_eq!( + retention_witness(sandbox.path())?, + before, + "recovery refusal must preserve all retained evidence" + ); + Ok(error) +} + +fn cause<'a, T: Error + 'static>(mut error: &'a (dyn Error + 'static)) -> Option<&'a T> { + loop { + if let Some(cause) = error.downcast_ref::() { + return Some(cause); + } + error = error.source()?; + } +} + +// Size: medium. Oracle: cleanup may remove only the exact retained source with proven pool evidence. +// Bug regression; delete only when a stronger runtime identity law subsumes this boundary. +#[test] +fn substituted_cleanup_source_refuses_before_unlink() -> Result<(), Box> { + let (sandbox, mut authority) = open_authority("cleanup-source-substitution")?; + let root = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root)?; + drive_publication(&mut authority, &preparation, 15)?; + let observation = RetentionRecoveryObservation::observe( + &authority.retention, + &authority.roots, + &authority.manifests, + )?; + let plan = plan_retention_recovery(observation.evidence())?; + authority.recovery = Some(RetentionRecoveryContext::reopen( + &authority.retention, + &observation, + )?); + let stage = sandbox.path().join("retention/root.next"); + let replacement = stage.with_extension("replacement"); + fs::copy(&stage, &replacement)?; + fs::rename(replacement, stage)?; + let before = retention_witness(sandbox.path())?; + let error = execute_retention_recovery(&mut authority, &plan) + .err() + .ok_or("cleanup deleted substituted source")?; + assert_eq!( + error.step(), + RetentionRecoveryStep::RemoveRootStage, + "source identity must be guarded before cleanup" + ); + assert_eq!( + cause::(&error), + Some(&RetentionRecordRefusal::KindLengthOrIdentity), + "source substitution must retain its typed cause" + ); + assert!( + error.executed().is_empty(), + "no preceding recovery step completed" + ); + let progress = error + .progress() + .ok_or("source refusal omitted pre-effect status")?; + assert_eq!( + progress.boundary(), + crate::RetentionStorageBoundary::SourceVerification + ); + assert!( + progress.known_effects().is_empty(), + "source guard precedes namespace effects" + ); + assert_eq!(progress.uncertain_effect(), None); + assert_eq!( + retention_witness(sandbox.path())?, + before, + "pre-effect source refusal must preserve source and pool evidence" + ); + Ok(()) +} + +// Size: medium. Oracle: failed directory synchronization cannot roll back an earlier unlink. +// Real filesystem unlink, deterministic injected EIO before directory sync; not a power-loss test. +// Delete only when stronger public recovery failure/restart laws subsume this exact effect boundary. +#[test] +fn cleanup_sync_failure_reports_removed_stage_and_preserved_pool() -> Result<(), Box> { + use crate::adapters::retention::{ + FilesystemRetentionRecoveryError, RetentionEffectDurability as Durability, + RetentionNamespaceEffect as Effect, RetentionStorageBoundary as Boundary, + }; + let (sandbox, mut authority) = open_authority("cleanup-sync-failure")?; + let root = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root)?; + drive_publication(&mut authority, &preparation, 15)?; + authority.recovery_sync_failure = Some(Boundary::RetentionSynchronization); + let error = match authority.recover() { + Err(FilesystemRetentionRecoveryError::Execute { source }) => source, + result => return Err(format!("expected cleanup sync failure: {result:?}").into()), + }; + assert_eq!(error.step(), RetentionRecoveryStep::RemoveRootStage); + assert!( + error.executed().is_empty(), + "no preceding step completed despite the failing capability's unlink" + ); + let progress = error + .progress() + .ok_or("missing failing-capability effects after unlink")?; + assert_eq!(progress.boundary(), Boundary::RetentionSynchronization); + assert_eq!( + progress + .known_effects() + .iter() + .map(|effect| (effect.effect(), effect.durability())) + .collect::>(), + [(Effect::StageRemoved, Durability::Unconfirmed)] + ); + assert_eq!(progress.uncertain_effect(), None); + assert_eq!( + cause::(&error).and_then(io::Error::raw_os_error), + Some(rustix::io::Errno::IO.raw_os_error()) + ); + assert!( + !sandbox.path().join("retention/root.next").exists(), + "unlink already occurred" + ); + let pool = crate::adapters::retention::filesystem_retention_test_fixture::root_pool_path( + sandbox.path(), + preparation.candidate(), + ); + assert_eq!( + fs::read(pool)?, + root, + "exact pool evidence survives cleanup failure" + ); + assert!( + sandbox.path().join("retention/manifest.next").exists(), + "later cleanup must not execute" + ); + drop(authority); + let admission = + crate::adapters::FilesystemVersionTwoAdmission::reopen_unchecked_for_repository_tasks( + sandbox.path(), + )?; + let mut restarted = + crate::adapters::retention::FilesystemRetentionPublicationAuthority::open(admission)?; + let retry = restarted.recover()?; + assert_eq!( + retry.executed(), + [RetentionRecoveryStep::RemoveManifestStage], + "retry must freshly observe the completed unlink" + ); + Ok(()) +} + +// Size: medium. Oracle: an observed stage identity remains binding when execution reopens it. +// Bug regression; delete only when stronger recovery observation-to-execution coverage subsumes it. +#[test] +fn substitution_before_reopening_refuses_without_rebinding_evidence() -> Result<(), Box> +{ + let (sandbox, mut authority) = open_authority("recovery-observation-substitution")?; + let root = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root)?; + drive_publication(&mut authority, &preparation, 2)?; + let observation = RetentionRecoveryObservation::observe( + &authority.retention, + &authority.roots, + &authority.manifests, + )?; + let stage = sandbox.path().join("retention/root.next"); + let replacement = stage.with_extension("replacement"); + fs::copy(&stage, &replacement)?; + fs::rename(replacement, stage)?; + let before = retention_witness(sandbox.path())?; + let error = RetentionRecoveryContext::reopen(&authority.retention, &observation) + .err() + .ok_or("reopening silently rebound observed stage identity")?; + assert_eq!( + cause::(&error), + Some(&RetentionRecordRefusal::KindLengthOrIdentity), + "observed substitution must preserve its identity diagnostic" + ); + assert_eq!(retention_witness(sandbox.path())?, before); + Ok(()) +} diff --git a/src/adapters/retention/filesystem_retention_recovery_tests.rs b/src/adapters/retention/filesystem_retention_recovery_tests.rs new file mode 100644 index 00000000..fb4ff774 --- /dev/null +++ b/src/adapters/retention/filesystem_retention_recovery_tests.rs @@ -0,0 +1,205 @@ +//! Filesystem retention recovery laws over real crash prefixes. + +use std::error::Error; +use std::fs; + +use super::filesystem_retention_test_fixture::{ + ROOT_HEX, drive_publication, fixture, head_path, initial_preparation, manifest_pool_path, + open_authority, root_pool_path, +}; +use super::{ + RetentionPublicationOutcome, RetentionRecoveryOutcome as Outcome, RetentionRecoveryStep as Step, +}; +use crate::execute_retention_publication; + +#[test] +fn a_clean_store_recovers_to_clean() -> Result<(), Box> { + let (_sandbox, mut authority) = open_authority("filesystem-retention-recovery-clean")?; + let receipt = authority.recover()?; + assert!(receipt.executed().is_empty()); + assert_eq!(receipt.outcome(), Outcome::Clean); + Ok(()) +} + +#[test] +fn a_written_root_stage_is_linked_and_protected() -> Result<(), Box> { + let (sandbox, mut authority) = open_authority("filesystem-retention-recovery-root")?; + let root_bytes = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root_bytes)?; + drive_publication(&mut authority, &preparation, 3)?; + + let receipt = authority.recover()?; + + assert_eq!(receipt.executed(), [Step::LinkRoot]); + assert_eq!( + receipt.outcome(), + Outcome::Protected { + root_stage: true, + manifest_stage: false + } + ); + assert_eq!( + fs::read(root_pool_path(sandbox.path(), preparation.candidate()))?, + root_bytes + ); + assert!(sandbox.path().join("retention").join("root.next").is_file()); + assert!(!head_path(sandbox.path()).exists()); + Ok(()) +} + +#[test] +fn a_synchronized_head_stage_is_finalized_and_the_retry_is_already_committed() +-> Result<(), Box> { + let (sandbox, mut authority) = open_authority("filesystem-retention-recovery-head")?; + let root_bytes = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root_bytes)?; + drive_publication(&mut authority, &preparation, 13)?; + let publication = preparation.publication().ok_or("no publication")?; + + let receipt = authority.recover()?; + + assert_eq!( + receipt.executed(), + [ + Step::FinalizeHead, + Step::RemoveRootStage, + Step::RemoveManifestStage + ] + ); + assert_eq!(receipt.outcome(), Outcome::Committed); + assert_eq!( + fs::read(head_path(sandbox.path()))?, + publication.head().encoded() + ); + assert_eq!( + fs::read(manifest_pool_path(sandbox.path(), &preparation))?, + publication.manifest().encoded() + ); + for stage in ["root.next", "manifest.next", "head.next"] { + assert!( + !sandbox.path().join("retention").join(stage).exists(), + "{stage} remained" + ); + } + let retry = execute_retention_publication(&mut authority, &preparation)?; + assert_eq!( + retry.outcome(), + RetentionPublicationOutcome::AlreadyCommitted + ); + Ok(()) +} + +#[test] +fn a_truncated_root_stage_is_preserved_for_disposition() -> Result<(), Box> { + let (sandbox, mut authority) = open_authority("filesystem-retention-recovery-truncated")?; + let root_bytes = fixture(ROOT_HEX)?; + let stage = sandbox.path().join("retention").join("root.next"); + fs::write( + &stage, + root_bytes + .get(..100) + .ok_or("root fixture shorter than 100 bytes")?, + )?; + + let result = authority.recover(); + assert!( + matches!( + result, + Err(super::FilesystemRetentionRecoveryError::Plan { + source: super::RetentionRecoveryRefusal::IncompleteStageRequiresDisposition { + stage: super::RetentionFixedStage::Root, + expected: 192, + observed: 100 + } + }) + ), + "incomplete root must require disposition: {result:?}" + ); + assert_eq!( + fs::read(stage)?, + root_bytes.get(..100).ok_or("missing prefix")? + ); + Ok(()) +} + +// Size: medium. Oracle: recovery names the missing root, preserving ambiguous evidence. +// Delete only when stronger public recovery evidence subsumes this diagnostic and preservation law. +#[test] +fn a_complete_manifest_without_its_root_reports_the_missing_root() -> Result<(), Box> { + use super::filesystem_retention_test_fixture::retention_witness; + use super::{FilesystemRetentionRecoveryError, RetentionRecoveryRefusal}; + + let (sandbox, mut authority) = open_authority("recovery-missing-root")?; + let root_bytes = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root_bytes)?; + drive_publication(&mut authority, &preparation, 13)?; + fs::remove_file(sandbox.path().join("retention/root.next"))?; + let before = retention_witness(sandbox.path())?; + + let result = authority.recover(); + + assert!( + matches!( + result, + Err(FilesystemRetentionRecoveryError::Plan { + source: RetentionRecoveryRefusal::ManifestStageWithoutRootStage, + }) + ), + "a complete manifest with no root stage must name the missing root: {result:?}" + ); + assert_eq!( + retention_witness(sandbox.path())?, + before, + "missing-root refusal must preserve all retained bytes" + ); + Ok(()) +} + +// Size: medium. Oracle: Core Law refuses checksum corruption and preserves the exact evidence. +// Delete only when stronger filesystem recovery coverage subsumes this corruption law. +#[test] +fn checksum_corrupt_root_stage_is_refused_without_changing_evidence() -> Result<(), Box> +{ + use super::filesystem_retention_test_fixture::retention_witness; + use super::{ + FilesystemRetentionRecoveryError, RetentionFixedStage, RetentionRecoveryRefusal, + RetentionRootDecodeError, + }; + + let (sandbox, mut authority) = open_authority("recovery-checksum-corrupt-root")?; + let root_bytes = fixture(ROOT_HEX)?; + let mut corrupt = root_bytes.clone(); + *corrupt.last_mut().ok_or("empty root fixture")? ^= 0x01; + let checksum_offset = root_bytes.len().checked_sub(32).ok_or("missing checksum")?; + let expected_checksum: [u8; 32] = root_bytes + .get(checksum_offset..) + .ok_or("missing checksum")? + .try_into()?; + let observed_checksum: [u8; 32] = corrupt + .get(checksum_offset..) + .ok_or("missing damaged checksum")? + .try_into()?; + fs::write(sandbox.path().join("retention/root.next"), &corrupt)?; + let before = retention_witness(sandbox.path())?; + + let result = authority.recover(); + + assert!( + matches!(&result, + Err(FilesystemRetentionRecoveryError::Plan { + source: RetentionRecoveryRefusal::StageCorrupt { + stage: RetentionFixedStage::Root, source, + }, + }) if matches!(source.downcast_ref::(), + Some(RetentionRootDecodeError::ChecksumMismatch { expected, observed }) + if *expected == expected_checksum && *observed == observed_checksum) + ), + "corrupt root stage must report its exact checksum refusal: {result:?}" + ); + assert_eq!( + retention_witness(sandbox.path())?, + before, + "corrupt-stage refusal must preserve all retained bytes" + ); + Ok(()) +} diff --git a/src/adapters/retention/filesystem_retention_refusal.rs b/src/adapters/retention/filesystem_retention_refusal.rs index be4d20d7..de8bf235 100644 --- a/src/adapters/retention/filesystem_retention_refusal.rs +++ b/src/adapters/retention/filesystem_retention_refusal.rs @@ -5,6 +5,7 @@ use std::fmt; use std::io; use super::{RetentionHeadDecodeError, RetentionManifestDecodeError}; +use super::{RetentionRecoveryError, RetentionRecoveryRefusal}; use crate::adapters::{CatalogDecodeError, PublicationHeadDecodeError}; use crate::{CatalogGeneration, LivenessGeneration, RetentionManifestDigest}; @@ -122,6 +123,21 @@ pub enum RetentionCurrentStateRefusal { /// A protocol directory named at admission (`retention`, `roots`, or /// `manifests`) no longer names the pinned directory that was admitted. ProtocolDirectoryReplaced, + /// Restart recovery could not observe or admit its retained evidence. + RecoveryObservationRefused { + /// The original observation failure, including its typed or OS cause. + source: io::Error, + }, + /// Restart recovery refused the retained stages as unrecoverable ambiguity. + RecoveryRefused { + /// The exact planning refusal. + source: RetentionRecoveryRefusal, + }, + /// A restart recovery step refused; the completed prefix remains. + RecoveryStepRefused { + /// The refused step, the completed prefix, and the storage error. + source: RetentionRecoveryError, + }, /// A record's kind or length disagreed with its declaration. RecordKindOrLength, /// A record carried bytes beyond its declared length. @@ -228,6 +244,11 @@ impl RetentionCurrentStateRefusal { Self::RecordKindOrLength => "retention record kind or length disagreed", Self::RecordTrailingBytes => "retention record carried trailing bytes", Self::RecordLengthOverflow => "retention record length exceeded the addressable range", + Self::RecoveryRefused { .. } => { + "restart recovery refused the retained retention stages" + } + Self::RecoveryStepRefused { .. } => "a restart recovery step refused", + Self::RecoveryObservationRefused { .. } => "restart recovery observation refused", Self::ProtocolDirectoryReplaced => { "a retention protocol directory was replaced after admission" } @@ -253,6 +274,9 @@ impl Error for RetentionCurrentStateRefusal { Self::ManifestRefused { source } => Some(source), Self::CatalogHeadRefused { source } => Some(source), Self::CatalogRefused { source } => Some(source.as_ref()), + Self::RecoveryRefused { source } => Some(source), + Self::RecoveryStepRefused { source } => Some(source), + Self::RecoveryObservationRefused { source } => Some(source), _ => None, } } diff --git a/src/adapters/retention/filesystem_retention_short_count_tests.rs b/src/adapters/retention/filesystem_retention_short_count_tests.rs new file mode 100644 index 00000000..5edfa28a --- /dev/null +++ b/src/adapters/retention/filesystem_retention_short_count_tests.rs @@ -0,0 +1,86 @@ +//! This module owns preservation of excessive counts in interrupted stage headers. + +use super::filesystem_retention_test_fixture::{ + MANIFEST_HEX, ROOT_HEX, drive_publication, fixture, initial_preparation, open_authority, + retention_witness, +}; +use super::{ + FilesystemRetentionRecoveryError, RetentionFixedStage as Stage, RetentionManifestDecodeError, + RetentionRecoveryRefusal, RetentionRootDecodeError, +}; +use std::{error::Error, fs}; + +// Size: medium. Oracle: version-two root/manifest count ceilings apply before stage discard. +// Delete only when stronger public recovery coverage subsumes excessive-count preservation. +#[test] +fn excessive_counts_in_short_headers_preserve_evidence() -> Result<(), Box> { + for (stage, name, corpus, phases, maximum, width) in [ + (Stage::Root, "root.next", ROOT_HEX, 1, 65_536_u32, 119_u64), + (Stage::Manifest, "manifest.next", MANIFEST_HEX, 7, 4_096, 72), + ] { + for observed in [maximum.checked_add(1).ok_or("count overflow")?, u32::MAX] { + let (sandbox, mut authority) = + open_authority(&format!("short-count-{name}-{observed}"))?; + let root = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root)?; + drive_publication(&mut authority, &preparation, phases)?; + let bytes = fixture(corpus)?; + // Both frozen fixtures contain one entry; keep the declared length consistent. + let length = u64::from(observed) + .checked_sub(1) + .and_then(|n| n.checked_mul(width)) + .and_then(|extra| u64::try_from(bytes.len()).ok()?.checked_add(extra)) + .ok_or("record length overflow")?; + let mut partial = bytes.get(..48).ok_or("short fixture header")?.to_vec(); + partial + .get_mut(24..32) + .ok_or("missing length")? + .copy_from_slice(&length.to_be_bytes()); + partial + .get_mut(44..48) + .ok_or("missing count")? + .copy_from_slice(&observed.to_be_bytes()); + fs::write(sandbox.path().join("retention").join(name), partial)?; + let before = retention_witness(sandbox.path())?; + + let result = authority.recover(); + + match result { + Err(FilesystemRetentionRecoveryError::Plan { + source: + RetentionRecoveryRefusal::StageCorrupt { + stage: actual, + source, + }, + }) => { + assert_eq!(actual, stage, "name the excessive-count stage"); + require_count_cause(source.as_ref(), stage, maximum, observed); + } + result => { + return Err(format!("{name} count {observed} must refuse: {result:?}").into()); + } + } + assert_eq!( + retention_witness(sandbox.path())?, + before, + "{name} count {observed} must preserve retained evidence" + ); + } + } + Ok(()) +} + +fn require_count_cause(source: &(dyn Error + 'static), stage: Stage, maximum: u32, observed: u32) { + let exact = match stage { + Stage::Root => matches!(source.downcast_ref::(), + Some(RetentionRootDecodeError::AnchorCountExceeded { maximum: m, observed: o }) + if *m == maximum && *o == observed), + _ => matches!(source.downcast_ref::(), + Some(RetentionManifestDecodeError::EntryCountExceeded { maximum: m, observed: o }) + if *m == maximum && *o == observed), + }; + assert!( + exact, + "{stage:?} must report count {observed} above maximum {maximum}: {source:?}" + ); +} diff --git a/src/adapters/retention/filesystem_retention_short_framing_tests.rs b/src/adapters/retention/filesystem_retention_short_framing_tests.rs new file mode 100644 index 00000000..7b1cb99f --- /dev/null +++ b/src/adapters/retention/filesystem_retention_short_framing_tests.rs @@ -0,0 +1,89 @@ +//! This module owns preservation of contradictory lengths in interrupted stage headers. + +use super::filesystem_retention_test_fixture::{ + MANIFEST_HEX, ROOT_HEX, drive_publication, fixture, initial_preparation, open_authority, + retention_witness, +}; +use super::{ + FilesystemRetentionRecoveryError, RetentionFixedStage as Stage, RetentionManifestDecodeError, + RetentionRecoveryRefusal, RetentionRootDecodeError, +}; +use std::{error::Error, fs}; + +// Size: medium. Oracle: the declared size equals the format size of the supplied golden record. +// Delete only when stronger public recovery laws preserve contradictory header framing. +#[test] +fn contradictory_lengths_in_short_stage_headers_preserve_evidence() -> Result<(), Box> { + for (stage, name, corpus, phases) in [ + (Stage::Root, "root.next", ROOT_HEX, 1), + (Stage::Manifest, "manifest.next", MANIFEST_HEX, 7), + ] { + let bytes = fixture(corpus)?; + let expected = u64::try_from(bytes.len())?; + for observed in [ + 0, + expected.checked_add(1).ok_or("length overflow")?, + u64::MAX, + ] { + let (sandbox, mut authority) = + open_authority(&format!("short-framing-{name}-{observed}"))?; + let root = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root)?; + drive_publication(&mut authority, &preparation, phases)?; + let mut partial = bytes.get(..48).ok_or("short fixture header")?.to_vec(); + partial + .get_mut(24..32) + .ok_or("missing declared length")? + .copy_from_slice(&observed.to_be_bytes()); + fs::write(sandbox.path().join("retention").join(name), partial)?; + let before = retention_witness(sandbox.path())?; + + let result = authority.recover(); + + match result { + Err(FilesystemRetentionRecoveryError::Plan { + source: + RetentionRecoveryRefusal::StageCorrupt { + stage: actual, + source, + }, + }) => { + assert_eq!(actual, stage, "name the contradictory stage"); + require_length_cause(source.as_ref(), stage, expected, observed); + } + result => { + return Err(format!( + "{name} length {observed} must refuse contradictory framing: {result:?}" + ) + .into()); + } + } + assert_eq!( + retention_witness(sandbox.path())?, + before, + "{name} length {observed} must preserve retained evidence" + ); + } + } + Ok(()) +} + +fn require_length_cause( + source: &(dyn Error + 'static), + stage: Stage, + expected: u64, + observed: u64, +) { + let matches_length = match stage { + Stage::Root => matches!(source.downcast_ref::(), + Some(RetentionRootDecodeError::DeclaredLengthMismatch { expected: e, observed: o }) + if *e == expected && *o == observed), + _ => matches!(source.downcast_ref::(), + Some(RetentionManifestDecodeError::DeclaredLengthMismatch { expected: e, observed: o }) + if *e == expected && *o == observed), + }; + assert!( + matches_length, + "{stage:?} must report expected {expected}, observed {observed}: {source:?}" + ); +} diff --git a/src/adapters/retention/filesystem_retention_short_head_tests.rs b/src/adapters/retention/filesystem_retention_short_head_tests.rs new file mode 100644 index 00000000..13a62a55 --- /dev/null +++ b/src/adapters/retention/filesystem_retention_short_head_tests.rs @@ -0,0 +1,170 @@ +//! This module owns preservation of semantically invalid interrupted head fields. + +use std::error::Error; +use std::fs; + +use super::filesystem_retention_test_fixture::{ + ROOT_HEX, drive_publication, fixture, initial_preparation, open_authority, retention_witness, +}; +use super::{ + FilesystemRetentionRecoveryError, RetentionFixedStage, RetentionHeadDecodeError, + RetentionRecoveryRefusal, +}; +use crate::RetentionManifestLengthError; + +// Size: medium. Oracle: the format's length bounds and entry alignment apply to available fields. +// Delete only if stronger public recovery evidence preserves every invalid-length class here. +#[test] +fn invalid_manifest_lengths_in_short_heads_preserve_corrupt_evidence() -> Result<(), Box> +{ + for value in [0, 223, 225, 295_137, u64::MAX] { + let expected = if value == 225 { + RetentionManifestLengthError::NotCongruent { observed: value } + } else { + RetentionManifestLengthError::OutOfBounds { + minimum: 224, + maximum: 295_136, + observed: value, + } + }; + let (sandbox, mut authority) = open_authority(&format!("short-head-length-{value}"))?; + let root = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root)?; + drive_publication(&mut authority, &preparation, 11)?; + let publication = preparation.publication().ok_or("missing publication")?; + let mut partial = publication + .head() + .encoded() + .get(..40) + .ok_or("short head fixture")? + .to_vec(); + partial + .get_mut(32..40) + .ok_or("missing length field")? + .copy_from_slice(&value.to_be_bytes()); + fs::write(sandbox.path().join("retention/head.next"), partial)?; + let before = retention_witness(sandbox.path())?; + + let result = authority.recover(); + + assert!( + matches!(&result, + Err(FilesystemRetentionRecoveryError::Plan { + source: RetentionRecoveryRefusal::StageCorrupt { stage: RetentionFixedStage::Head, source }, + }) if matches!(source.downcast_ref::(), + Some(RetentionHeadDecodeError::ManifestLength { source }) if *source == expected)), + "short head length {value} must refuse with {expected:?}: {result:?}" + ); + assert_eq!( + retention_witness(sandbox.path())?, + before, + "invalid short head length {value} must preserve retained evidence" + ); + } + Ok(()) +} + +// Size: medium. Oracle: initial heads have no predecessor; successors require one. +// Delete only when stronger public recovery evidence subsumes both history contradictions. +#[test] +fn contradictory_history_in_short_heads_preserves_evidence() -> Result<(), Box> { + use crate::{LivenessGeneration, RetentionHeadError, RetentionManifestDigest}; + for (generation, predecessor, expected) in [ + ( + 1_u64, + [7_u8; 32], + RetentionHeadError::InitialGenerationHasPredecessor { + observed: RetentionManifestDigest::from_hash([7; 32]), + }, + ), + ( + 2, + [0; 32], + RetentionHeadError::MissingPredecessor { + generation: LivenessGeneration::new(2)?, + }, + ), + ] { + let (sandbox, mut authority) = open_authority(&format!("short-head-history-{generation}"))?; + let root = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root)?; + drive_publication(&mut authority, &preparation, 11)?; + let publication = preparation.publication().ok_or("missing publication")?; + let mut partial = publication + .head() + .encoded() + .get(..104) + .ok_or("short head fixture")? + .to_vec(); + partial + .get_mut(24..32) + .ok_or("missing generation")? + .copy_from_slice(&generation.to_be_bytes()); + partial + .get_mut(72..104) + .ok_or("missing predecessor")? + .copy_from_slice(&predecessor); + fs::write(sandbox.path().join("retention/head.next"), partial)?; + let before = retention_witness(sandbox.path())?; + + let result = authority.recover(); + + assert!( + matches!(&result, + Err(FilesystemRetentionRecoveryError::Plan { + source: RetentionRecoveryRefusal::StageCorrupt { stage: RetentionFixedStage::Head, source }, + }) if matches!(source.downcast_ref::(), + Some(RetentionHeadDecodeError::Semantic { source }) if *source == expected)), + "short head generation {generation} must refuse with {expected:?}: {result:?}" + ); + assert_eq!( + retention_witness(sandbox.path())?, + before, + "contradictory short head generation {generation} must preserve retained evidence" + ); + } + Ok(()) +} + +// Size: medium. Oracle: every available checksum byte must match the golden head checksum. +// Delete only when stronger public recovery coverage subsumes interrupted-checksum corruption. +#[test] +fn contradictory_partial_head_checksums_preserve_evidence() -> Result<(), Box> { + for offset in 112_usize..143 { + let (sandbox, mut authority) = open_authority(&format!("short-head-checksum-{offset}"))?; + let root = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root)?; + drive_publication(&mut authority, &preparation, 11)?; + let publication = preparation.publication().ok_or("missing publication")?; + let end = offset.checked_add(1).ok_or("checksum prefix overflow")?; + let mut partial = publication + .head() + .encoded() + .get(..end) + .ok_or("missing checksum prefix")? + .to_vec(); + let byte = partial.get_mut(offset).ok_or("missing checksum byte")?; + let expected = *byte; + *byte ^= 1; + let observed = *byte; + fs::write(sandbox.path().join("retention/head.next"), partial)?; + let before = retention_witness(sandbox.path())?; + + let result = authority.recover(); + + assert!( + matches!(&result, + Err(FilesystemRetentionRecoveryError::Plan { + source: RetentionRecoveryRefusal::StageCorrupt { stage: RetentionFixedStage::Head, source }, + }) if source.downcast_ref::() == + Some(&RetentionHeadDecodeError::PrefixByteMismatch { offset, expected, observed })), + "checksum prefix ending at {end} must report the exact contradictory byte: {result:?}" + ); + assert_eq!( + retention_witness(sandbox.path())?, + before, + "corrupt checksum prefix ending at {end} must preserve retained evidence" + ); + } + Ok(()) +} diff --git a/src/adapters/retention/filesystem_retention_short_history_tests.rs b/src/adapters/retention/filesystem_retention_short_history_tests.rs new file mode 100644 index 00000000..752aa64f --- /dev/null +++ b/src/adapters/retention/filesystem_retention_short_history_tests.rs @@ -0,0 +1,101 @@ +//! This module owns preservation of contradictory interrupted root and manifest histories. + +use super::filesystem_retention_test_fixture::{ + MANIFEST_HEX, ROOT_HEX, drive_publication, fixture, initial_preparation, open_authority, + retention_witness, +}; +use super::{ + FilesystemRetentionRecoveryError, RetentionFixedStage as Stage, RetentionManifestDecodeError, + RetentionRecoveryRefusal, RetentionRootDecodeError, +}; +use crate::{RetentionManifestError, RetentionRootError}; +use std::{error::Error, fs}; + +// Size: medium. Oracle: initial records omit predecessors; successor records require them. +// Delete only when stronger public recovery coverage subsumes both record/history combinations. +#[test] +fn contradictory_short_root_and_manifest_histories_preserve_evidence() -> Result<(), Box> +{ + for (stage, name, corpus, phases, start, end) in [ + (Stage::Root, "root.next", ROOT_HEX, 1, 116, 148), + (Stage::Manifest, "manifest.next", MANIFEST_HEX, 7, 48, 80), + ] { + for (generation, predecessor) in [(1_u64, [7_u8; 32]), (2, [0; 32])] { + let (sandbox, mut authority) = + open_authority(&format!("short-history-{name}-{generation}"))?; + let root = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root)?; + drive_publication(&mut authority, &preparation, phases)?; + let bytes = fixture(corpus)?; + let mut partial = bytes.get(..end).ok_or("short fixture header")?.to_vec(); + partial + .get_mut(32..40) + .ok_or("missing generation")? + .copy_from_slice(&generation.to_be_bytes()); + partial + .get_mut(start..end) + .ok_or("missing predecessor")? + .copy_from_slice(&predecessor); + fs::write(sandbox.path().join("retention").join(name), partial)?; + let before = retention_witness(sandbox.path())?; + + let result = authority.recover(); + + match result { + Err(FilesystemRetentionRecoveryError::Plan { + source: + RetentionRecoveryRefusal::StageCorrupt { + stage: actual, + source, + }, + }) => { + assert_eq!(actual, stage, "name the contradictory history stage"); + require_history_cause(source.as_ref(), stage, generation); + } + result => return Err(format!( + "{name} generation {generation} must refuse contradictory history: {result:?}" + ) + .into()), + } + assert_eq!( + retention_witness(sandbox.path())?, + before, + "{name} generation {generation} must preserve retained evidence" + ); + } + } + Ok(()) +} + +fn require_history_cause(source: &(dyn Error + 'static), stage: Stage, generation: u64) { + let exact = match stage { + Stage::Root => match source.downcast_ref::() { + Some(RetentionRootDecodeError::Semantic { + source: RetentionRootError::InitialGenerationHasPredecessor { observed }, + }) => generation == 1 && observed.as_bytes() == &[7; 32], + Some(RetentionRootDecodeError::Semantic { + source: + RetentionRootError::MissingPredecessor { + generation: observed, + }, + }) => generation == 2 && observed.get() == generation, + _ => false, + }, + _ => match source.downcast_ref::() { + Some(RetentionManifestDecodeError::Semantic { + source: RetentionManifestError::InitialGenerationHasPredecessor { observed }, + }) => generation == 1 && observed.as_bytes() == &[7; 32], + Some(RetentionManifestDecodeError::Semantic { + source: + RetentionManifestError::MissingPredecessor { + generation: observed, + }, + }) => generation == 2 && observed.get() == generation, + _ => false, + }, + }; + assert!( + exact, + "{stage:?} generation {generation} must report its exact history contradiction: {source:?}" + ); +} diff --git a/src/adapters/retention/filesystem_retention_short_namespace_tests.rs b/src/adapters/retention/filesystem_retention_short_namespace_tests.rs new file mode 100644 index 00000000..12a77c33 --- /dev/null +++ b/src/adapters/retention/filesystem_retention_short_namespace_tests.rs @@ -0,0 +1,65 @@ +//! This module owns namespace-size refusal before interrupted-root discard. + +use super::filesystem_retention_test_fixture::{ + ROOT_HEX, fixture, open_authority, retention_witness, +}; +use super::{ + FilesystemRetentionRecoveryError, RetentionFixedStage, RetentionRecoveryRefusal, + RetentionRootDecodeError, +}; +use crate::RetentionNamespaceError; +use std::{error::Error, fs}; + +// Size: medium. Oracle: an opaque namespace has 1..=255 bytes regardless of missing payload. +// Delete only when stronger public recovery evidence covers these invalid size boundaries. +#[test] +fn impossible_namespace_sizes_in_short_roots_preserve_evidence() -> Result<(), Box> { + for size in [0_u16, 256, u16::MAX] { + let expected = if size == 0 { + RetentionNamespaceError::Empty + } else { + RetentionNamespaceError::TooLong { + maximum: 255, + observed: usize::from(size), + } + }; + for end in [42, 48] { + let (sandbox, mut authority) = + open_authority(&format!("short-namespace-{size}-{end}"))?; + let bytes = fixture(ROOT_HEX)?; + // The golden root contains a three-byte namespace; preserve consistent framing. + let length = u64::try_from(bytes.len())? + .checked_sub(3) + .and_then(|length| length.checked_add(u64::from(size))) + .ok_or("length overflow")?; + let mut partial = bytes.get(..end).ok_or("short root fixture")?.to_vec(); + partial + .get_mut(24..32) + .ok_or("missing record length")? + .copy_from_slice(&length.to_be_bytes()); + partial + .get_mut(40..42) + .ok_or("missing namespace length")? + .copy_from_slice(&size.to_be_bytes()); + fs::write(sandbox.path().join("retention/root.next"), partial)?; + let before = retention_witness(sandbox.path())?; + + let result = authority.recover(); + + assert!( + matches!(&result, + Err(FilesystemRetentionRecoveryError::Plan { + source: RetentionRecoveryRefusal::StageCorrupt { stage: RetentionFixedStage::Root, source }, + }) if matches!(source.downcast_ref::(), + Some(RetentionRootDecodeError::Namespace { source }) if *source == expected)), + "namespace size {size}, prefix {end} must report {expected:?}: {result:?}" + ); + assert_eq!( + retention_witness(sandbox.path())?, + before, + "namespace size {size}, prefix {end} must preserve retained evidence" + ); + } + } + Ok(()) +} diff --git a/src/adapters/retention/filesystem_retention_short_policy_tests.rs b/src/adapters/retention/filesystem_retention_short_policy_tests.rs new file mode 100644 index 00000000..6b39ea1f --- /dev/null +++ b/src/adapters/retention/filesystem_retention_short_policy_tests.rs @@ -0,0 +1,159 @@ +//! This module owns preservation of interrupted roots with invalid complete policies. + +use super::filesystem_retention_test_fixture::{ + ROOT_HEX, fixture, open_authority, retention_witness, +}; +use super::{ + FilesystemRetentionRecoveryError, RetentionFixedStage, RetentionRecoveryRefusal, + RetentionRootDecodeError, +}; +use crate::{ + RegisteredRetentionProfile, RetentionClosureLimit as Limit, + RetentionClosureLimitError as LimitError, RetentionProfileAdmissionError as ProfileError, +}; +use std::{error::Error, fs}; + +// Size: medium. Oracle: registered profile coordinates and digest are exact protocol requirements. +// Delete only when stronger public recovery coverage subsumes these malformed profiles. +#[test] +fn invalid_short_root_profiles_preserve_evidence() -> Result<(), Box> { + for (offset, replacement, expected) in [ + ( + 48, + 2_u32.to_be_bytes().to_vec(), + ProfileError::UnsupportedCoordinate { + expected_identity: 1, + expected_version: 1, + observed_identity: 2, + observed_version: 1, + }, + ), + ( + 52, + 2_u32.to_be_bytes().to_vec(), + ProfileError::UnsupportedCoordinate { + expected_identity: 1, + expected_version: 1, + observed_identity: 1, + observed_version: 2, + }, + ), + ( + 56, + vec![0; 32], + ProfileError::DefinitionDigestMismatch { + expected: *RegisteredRetentionProfile::SINGLE_CANONICAL_WITNESS_V1.digest(), + observed: [0; 32], + }, + ), + ] { + require_policy_refusal( + offset, + &replacement, + 88, + &RetentionRootDecodeError::Profile { source: expected }, + )?; + } + Ok(()) +} + +// Size: medium. Oracle: each closure resource is positive and bounded by the v2 format ceiling. +// Delete only when stronger public recovery coverage subsumes all resource violations. +#[test] +fn invalid_short_root_closure_limits_preserve_evidence() -> Result<(), Box> { + for (offset, width, limit, maximum) in [ + (88, 8, Limit::Nodes, 1_048_576_u64), + (96, 2, Limit::Depth, 8), + (100, 8, Limit::EncodedBytes, 16_777_216), + (108, 8, Limit::PhysicalBytes, 1_073_741_824), + ] { + for observed in [0, maximum.checked_add(1).ok_or("test boundary overflow")?] { + let bytes = observed.to_be_bytes(); + let start = bytes + .len() + .checked_sub(width) + .ok_or("invalid field width")?; + let replacement = bytes.get(start..).ok_or("missing replacement")?; + let source = if observed == 0 { + LimitError::Zero { limit } + } else { + LimitError::AboveMaximum { + limit, + maximum, + observed, + } + }; + require_policy_refusal( + offset, + replacement, + 116, + &RetentionRootDecodeError::ClosureLimit { source }, + )?; + } + } + Ok(()) +} + +fn require_policy_refusal( + offset: usize, + replacement: &[u8], + prefix_length: usize, + expected: &RetentionRootDecodeError, +) -> Result<(), Box> { + let (sandbox, mut authority) = + open_authority(&format!("short-root-policy-{offset}-{prefix_length}"))?; + let bytes = fixture(ROOT_HEX)?; + let mut partial = bytes.get(..prefix_length).ok_or("short fixture")?.to_vec(); + let end = offset + .checked_add(replacement.len()) + .ok_or("test offset overflow")?; + partial + .get_mut(offset..end) + .ok_or("missing policy field")? + .copy_from_slice(replacement); + fs::write(sandbox.path().join("retention/root.next"), partial)?; + let before = retention_witness(sandbox.path())?; + + match authority.recover() { + Err(FilesystemRetentionRecoveryError::Plan { + source: RetentionRecoveryRefusal::StageCorrupt { stage, source }, + }) => { + assert_eq!( + stage, + RetentionFixedStage::Root, + "identify the corrupt policy stage" + ); + require_policy_cause(source.as_ref(), expected); + } + result => { + return Err(format!( + "invalid root policy at {offset} must refuse recovery: {result:?}" + ) + .into()); + } + } + assert_eq!( + retention_witness(sandbox.path())?, + before, + "invalid policy must preserve retained evidence" + ); + Ok(()) +} + +fn require_policy_cause(source: &(dyn Error + 'static), expected: &RetentionRootDecodeError) { + let exact = match (source.downcast_ref::(), expected) { + ( + Some(RetentionRootDecodeError::Profile { source: actual }), + RetentionRootDecodeError::Profile { source: expected }, + ) => actual == expected, + ( + Some(RetentionRootDecodeError::ClosureLimit { source: actual }), + RetentionRootDecodeError::ClosureLimit { source: expected }, + ) => actual == expected, + _ => false, + }; + assert!( + exact, + "report exact root policy violation: expected {expected:?}, observed {source:?}" + ); +} diff --git a/src/adapters/retention/filesystem_retention_snapshot.rs b/src/adapters/retention/filesystem_retention_snapshot.rs new file mode 100644 index 00000000..751aebce --- /dev/null +++ b/src/adapters/retention/filesystem_retention_snapshot.rs @@ -0,0 +1,259 @@ +//! This module owns one fenced, double-collected reader view of a version-two store. + +use std::io; +use std::path::Path; + +use cap_fs_ext::DirExt; +use cap_std::fs::Dir; + +use super::filesystem_retention_current::{self, ObservedRetentionState}; +use super::filesystem_retention_pool_name as pool_name; +use super::{ + AdmittedRetentionRoot, FilesystemRetentionSnapshotError as Error, ReaderAttemptLimit, + ReaderFence, RetentionViewCoordinates, RetentionViewSource, collect_retention_view, + root_header_decoder, +}; +use crate::adapters::filesystem_exact_record::{self as exact_record, ExactRecordError}; +use crate::adapters::filesystem_platform_profile::root_identity; +use crate::adapters::filesystem_version_two_admission::require_root_identity; +use crate::adapters::{ + CatalogRestartError, CatalogRestartPolicy, ChecksummedPublicationHead, + FilesystemCatalogSnapshot, filesystem_initialization_namespace, filesystem_version_two_records, + publication_head_decoder, +}; +use crate::{RetentionHead, RetentionManifest, RetentionNamespaceDigest}; + +const HEAD_NAME: &str = "HEAD"; + +#[cfg(test)] +#[path = "filesystem_retention_snapshot_pinning_tests.rs"] +mod pinning_tests; + +#[cfg(test)] +#[path = "filesystem_retention_snapshot_moving_error_tests.rs"] +mod moving_error_tests; + +#[cfg(test)] +#[path = "filesystem_retention_snapshot_coordinate_tests.rs"] +mod coordinate_tests; + +/// One consistent reader view: the catalog snapshot, the retention head, and +/// the manifest it selects, all observed under one shared reader fence. +/// +/// The view holds the fence for its lifetime, so collection cannot delete the +/// roots or segments it names while it lives. Selected roots are read on +/// demand and verified against the manifest's digest before they are returned. +#[must_use] +pub struct FilesystemRetentionSnapshot { + _fence: ReaderFence, + roots: Dir, + catalog: FilesystemCatalogSnapshot, + retention: Option, +} + +struct View { + catalog: FilesystemCatalogSnapshot, + retention: Option, +} + +struct Source { + root: Dir, + retention: Dir, + manifests: Dir, + policy: CatalogRestartPolicy, +} + +impl RetentionViewSource for Source { + type View = Result; + + fn coordinates(&mut self) -> io::Result { + let catalog = filesystem_retention_current::read_exact_optional( + &self.root, + HEAD_NAME, + publication_head_decoder::ENCODED_LENGTH, + )? + .map(|bytes| { + ChecksummedPublicationHead::decode(&bytes) + .map(|head| { + ( + head.generation(), + head.catalog_length(), + head.catalog_digest(), + ) + }) + .map_err(|source| io::Error::new(io::ErrorKind::InvalidData, source)) + }) + .transpose()?; + let retention = filesystem_retention_current::observe(&self.retention, &self.manifests)? + .map(|state| *state.head()); + Ok(RetentionViewCoordinates { catalog, retention }) + } + + fn load(&mut self) -> io::Result { + let catalog = crate::adapters::catalog_restart_loader::load_from_directory( + &self.root, + HEAD_NAME, + self.policy, + ); + let catalog = match catalog { + Ok(catalog) => catalog, + Err(source) => return Ok(Err(source)), + }; + let retention = filesystem_retention_current::observe(&self.retention, &self.manifests)?; + Ok(Ok(View { catalog, retention })) + } +} + +impl FilesystemRetentionSnapshot { + /// Admits the root as version two, acquires the reader fence, and + /// double-collects one consistent view within `limit` attempts. + /// Admission requires the opened directory's restart-stable device and + /// inode to match the jointly admitted migration records before fencing. + /// + /// The call takes no writer authority and mutates nothing. It may block + /// while collection holds the fence exclusively. + /// + /// # Errors + /// + /// Returns [`FilesystemRetentionSnapshotError`](super::FilesystemRetentionSnapshotError) + /// at the exact admission, fence, collection, or catalog refusal. + /// A catalog admission result is returned only after both coordinate + /// reads agree; a moving head discards that result and retries. + pub fn load( + store_root: &Path, + policy: CatalogRestartPolicy, + limit: ReaderAttemptLimit, + ) -> Result { + let root = Dir::open_ambient_dir(store_root, cap_std::ambient_authority()) + .map_err(|source| Error::Admission { source })?; + filesystem_initialization_namespace::admit_version_two(&root) + .map_err(|source| Error::Admission { source })?; + let bound = filesystem_version_two_records::admit(&root) + .map_err(|source| Error::Admission { source })?; + let observed = root_identity(&root).map_err(|source| Error::Admission { source })?; + require_root_identity(bound, observed).map_err(|source| Error::Admission { + source: io::Error::new(io::ErrorKind::InvalidData, source), + })?; + let fence = ReaderFence::acquire(&root).map_err(|source| Error::Fence { source })?; + let retention = root + .open_dir_nofollow(pool_name::RETENTION) + .map_err(|source| Error::Admission { source })?; + let roots = retention + .open_dir_nofollow(pool_name::ROOTS) + .map_err(|source| Error::Admission { source })?; + let manifests = retention + .open_dir_nofollow(pool_name::MANIFESTS) + .map_err(|source| Error::Admission { source })?; + let mut source = Source { + root, + retention, + manifests, + policy, + }; + let view = collect_retention_view(&mut source, limit) + .map_err(|source| Error::View { source })? + .map_err(|source| Error::Catalog { source })?; + Ok(Self { + _fence: fence, + roots, + catalog: view.catalog, + retention: view.retention, + }) + } + + /// The catalog snapshot the view binds. + pub const fn catalog(&self) -> &FilesystemCatalogSnapshot { + &self.catalog + } + + /// The published retention head, or `None` when no generation is published. + #[must_use] + pub fn retention_head(&self) -> Option<&RetentionHead> { + self.retention.as_ref().map(ObservedRetentionState::head) + } + + /// The manifest the retention head selects, or `None` when none is published. + #[must_use] + pub fn manifest(&self) -> Option<&RetentionManifest> { + self.retention + .as_ref() + .map(ObservedRetentionState::manifest) + } + + /// Reads and verifies the root the manifest selects for `namespace`. + /// + /// Returns `None` when the manifest names no root for the namespace. The + /// pool entry is read without following links, bounded by the root + /// format's maximum length, decoded, and required to carry exactly the + /// generation and digest the manifest names. + /// + /// # Errors + /// + /// Returns [`FilesystemRetentionSnapshotError::Root`](super::FilesystemRetentionSnapshotError::Root) + /// when the entry is absent, unreadable, or not the selected root. + pub fn retained_root( + &self, + namespace: RetentionNamespaceDigest, + ) -> Result>, Error> { + let Some(manifest) = self.manifest() else { + return Ok(None); + }; + let entries = manifest.entries(); + let Some(entry) = entries + .binary_search_by_key(&namespace, |entry| entry.namespace()) + .ok() + .and_then(|index| entries.get(index).copied()) + else { + return Ok(None); + }; + let directory = self + .roots + .open_dir_nofollow(pool_name::namespace(namespace)) + .map_err(|source| Error::Root { source })?; + let name = pool_name::root(entry.root_generation(), entry.root_digest()); + let length = directory + .symlink_metadata(&name) + .and_then(|metadata| { + usize::try_from(metadata.len()).map_err(|_source| invalid("root length overflow")) + }) + .map_err(|source| Error::Root { source })?; + if length > root_header_decoder::MAXIMUM_ENCODED_LENGTH { + return Err(Error::Root { + source: invalid("selected root exceeds the format bound"), + }); + } + let bytes = match exact_record::read_exact_optional(&directory, &name, length) { + Ok(Some(bytes)) => bytes, + Ok(None) => { + return Err(Error::Root { + source: invalid("selected root is absent"), + }); + } + Err(ExactRecordError::Io(source)) => return Err(Error::Root { source }), + Err(ExactRecordError::Refused(refusal)) => { + return Err(Error::Root { + source: invalid_string(format!("selected root refused: {refusal}")), + }); + } + }; + let root = AdmittedRetentionRoot::decode(&bytes).map_err(|source| Error::Root { + source: io::Error::new(io::ErrorKind::InvalidData, source), + })?; + if root.digest() != entry.root_digest() + || root.root().generation() != entry.root_generation() + { + return Err(Error::Root { + source: invalid("selected root does not decode to the manifest's selection"), + }); + } + Ok(Some(bytes.into_boxed_slice())) + } +} + +fn invalid(message: &'static str) -> io::Error { + io::Error::new(io::ErrorKind::InvalidData, message) +} + +fn invalid_string(message: String) -> io::Error { + io::Error::new(io::ErrorKind::InvalidData, message) +} diff --git a/src/adapters/retention/filesystem_retention_snapshot_coordinate_tests.rs b/src/adapters/retention/filesystem_retention_snapshot_coordinate_tests.rs new file mode 100644 index 00000000..35ff35cd --- /dev/null +++ b/src/adapters/retention/filesystem_retention_snapshot_coordinate_tests.rs @@ -0,0 +1,88 @@ +//! This module owns reader refusal of a changed catalog-length coordinate. + +use std::error::Error; +use std::fs; +use std::io; +use std::num::NonZeroU32; + +use super::{Source, collect_retention_view, pool_name}; +use crate::adapters::retention::filesystem_retention_test_fixture::migrated_store; +use crate::adapters::retention::{ + ReaderAttemptLimit, ReaderFence, RetentionViewCoordinates, RetentionViewError, + RetentionViewSource, +}; +use crate::adapters::{ + CatalogRestartByteLimit, CatalogRestartPolicy, ChecksummedPublicationHead, SegmentReadPolicy, + publication_head_decoder, +}; +use cap_fs_ext::DirExt; +use cap_std::fs::Dir; + +struct RewriteHead { + source: Source, + rewrite: F, +} + +impl io::Result<()>> RetentionViewSource for RewriteHead { + type View = ::View; + + fn coordinates(&mut self) -> io::Result { + self.source.coordinates() + } + + fn load(&mut self) -> io::Result { + let view = self.source.load()?; + (self.rewrite)()?; + Ok(view) + } +} + +// Size: medium. Oracle: every validated HEAD coordinate must agree across collection. +// Delete when double collection is removed or a stronger filesystem law subsumes this schedule. +#[test] +fn a_changed_catalog_length_refuses_the_loaded_reader_view() -> Result<(), Box> { + let sandbox = migrated_store("reader-changed-catalog-length")?; + let root = Dir::open_ambient_dir(sandbox.path(), cap_std::ambient_authority())?; + let _fence = ReaderFence::acquire(&root)?; + let retention = root.open_dir_nofollow(pool_name::RETENTION)?; + let manifests = retention.open_dir_nofollow(pool_name::MANIFESTS)?; + let path = sandbox.path().join("HEAD"); + let mut head = fs::read(&path)?; + let decoded = ChecksummedPublicationHead::decode(&head)?; + let changed_length = decoded + .catalog_length() + .get() + .checked_add(160) + .ok_or("length overflow")?; + head.get_mut(32..40) + .ok_or("HEAD lacks catalog length")? + .copy_from_slice(&changed_length.to_be_bytes()); + let (covered, checksum) = head.split_at_mut(publication_head_decoder::CHECKSUM_INPUT_LENGTH); + checksum.copy_from_slice(&publication_head_decoder::checksum(covered)); + let _admitted_changed_head = ChecksummedPublicationHead::decode(&head)?; + let source = Source { + root, + retention, + manifests, + policy: CatalogRestartPolicy::new( + SegmentReadPolicy::MAXIMUM, + CatalogRestartByteLimit::new(1_048_576)?, + ), + }; + let mut scheduled = RewriteHead { + source, + rewrite: || fs::write(&path, &head), + }; + let limit = ReaderAttemptLimit::new(NonZeroU32::new(1).ok_or("zero attempts")?); + + let result = collect_retention_view(&mut scheduled, limit); + + assert!( + matches!( + result, + Err(RetentionViewError::AttemptsExhausted { attempts: 1 }) + ), + "a changed catalog length must refuse the previously loaded view" + ); + Ok(()) +} diff --git a/src/adapters/retention/filesystem_retention_snapshot_error.rs b/src/adapters/retention/filesystem_retention_snapshot_error.rs new file mode 100644 index 00000000..daa6a634 --- /dev/null +++ b/src/adapters/retention/filesystem_retention_snapshot_error.rs @@ -0,0 +1,63 @@ +//! This module owns the typed error of filesystem retention snapshot loading. + +use std::error::Error; +use std::fmt; +use std::io; + +use super::RetentionViewError; +use crate::adapters::CatalogRestartError; + +/// Why a reader could not bind one consistent version-two view. +#[derive(Debug)] +#[non_exhaustive] +pub enum FilesystemRetentionSnapshotError { + /// The root is not an exactly admitted version-two store. + Admission { + /// The exact namespace or record refusal. + source: io::Error, + }, + /// The reader fence could not be acquired. + Fence { + /// The exact filesystem failure. + source: io::Error, + }, + /// The heads never agreed, or a coordinate or retention read failed. + View { + /// The exact collection refusal. + source: RetentionViewError, + }, + /// Stable collected heads select a catalog that does not admit. + Catalog { + /// The exact restart refusal. + source: CatalogRestartError, + }, + /// A selected root pool entry is absent, unreadable, or not the manifest's. + Root { + /// The exact filesystem or decode refusal. + source: io::Error, + }, +} + +impl fmt::Display for FilesystemRetentionSnapshotError { + fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { + formatter.write_str(match self { + Self::Admission { .. } => "version-two reader admission refused", + Self::Fence { .. } => "reader fence acquisition failed", + Self::View { .. } => "reader view collection refused", + Self::Catalog { .. } => "catalog snapshot refused", + Self::Root { .. } => "selected retention root refused", + }) + } +} + +impl Error for FilesystemRetentionSnapshotError { + fn source(&self) -> Option<&(dyn Error + 'static)> { + match self { + Self::Admission { source } | Self::Fence { source } | Self::Root { source } => { + Some(source) + } + Self::View { source } => Some(source), + Self::Catalog { source } => Some(source), + } + } +} diff --git a/src/adapters/retention/filesystem_retention_snapshot_error_tests.rs b/src/adapters/retention/filesystem_retention_snapshot_error_tests.rs new file mode 100644 index 00000000..c74f9dbc --- /dev/null +++ b/src/adapters/retention/filesystem_retention_snapshot_error_tests.rs @@ -0,0 +1,119 @@ +//! This module owns exact public reader refusal boundaries. + +use std::error::Error; +use std::fs; +use std::io; +use std::path::Path; + +use super::filesystem_retention_test_fixture::{CATALOG_NAME, SEGMENT_NAME, migrated_store}; +use super::{ + FilesystemRetentionSnapshot, FilesystemRetentionSnapshotError as SnapshotError, + ReaderAttemptLimit, RetentionViewError, +}; +use crate::adapters::{ + CatalogDecodeError, CatalogRestartByteLimit, CatalogRestartError, CatalogRestartPhase, + CatalogRestartPolicy, PublicationHeadDecodeError, SegmentReadPolicy, +}; + +// Size: medium (owned filesystem). Oracle: selected-artifact restart failures +// have the documented catalog boundary and retain their exact typed source. +// Delete when the reader API is removed or stronger public laws subsume these cases. +#[test] +fn a_missing_selected_catalog_reports_its_exact_catalog_refusal() -> Result<(), Box> { + let sandbox = migrated_store("reader-error-missing-catalog")?; + fs::remove_file(sandbox.path().join("catalogs").join(CATALOG_NAME))?; + + let error = refusal(sandbox.path())?; + + assert!( + matches!(&error, Some(SnapshotError::Catalog { + source: CatalogRestartError::Io { phase: CatalogRestartPhase::OpenCatalog, source } + }) if source.kind() == io::ErrorKind::NotFound), + "missing selected catalog must preserve its catalog boundary, phase and I/O kind: {error:?}" + ); + assert!( + error + .as_ref() + .and_then(Error::source) + .and_then(|source| source.downcast_ref::()) + .is_some(), + "catalog refusal must expose CatalogRestartError directly as its source" + ); + Ok(()) +} + +// Size: medium. Oracle and deletion criterion: the public catalog-boundary law above. +#[test] +fn a_missing_selected_segment_reports_its_exact_catalog_refusal() -> Result<(), Box> { + let sandbox = migrated_store("reader-error-missing-segment")?; + fs::remove_file(sandbox.path().join("segments").join(SEGMENT_NAME))?; + + let error = refusal(sandbox.path())?; + + assert!( + matches!(&error, Some(SnapshotError::Catalog { + source: CatalogRestartError::Io { phase: CatalogRestartPhase::OpenSegment, source } + }) if source.kind() == io::ErrorKind::NotFound), + "missing selected segment must preserve its catalog boundary, phase and I/O kind: {error:?}" + ); + Ok(()) +} + +// Size: medium. Oracle and deletion criterion: the public catalog-boundary law above. +#[test] +fn a_corrupt_selected_catalog_reports_its_decoder_refusal() -> Result<(), Box> { + let sandbox = migrated_store("reader-error-corrupt-catalog")?; + let path = sandbox.path().join("catalogs").join(CATALOG_NAME); + let mut bytes = fs::read(&path)?; + let checksum = bytes + .len() + .checked_sub(64) + .ok_or("catalog lacks checksum")?; + *bytes.get_mut(checksum).ok_or("catalog checksum absent")? ^= 1; + fs::write(path, bytes)?; + + let error = refusal(sandbox.path())?; + + assert!( + matches!( + error, + Some(SnapshotError::Catalog { + source: CatalogRestartError::Catalog { + source: CatalogDecodeError::ChecksumMismatch { .. } + } + }) + ), + "corrupt selected catalog must preserve its exact decoder refusal: {error:?}" + ); + Ok(()) +} + +// Size: medium. Oracle: coordinate decoding remains a view failure, not a catalog load. +// Delete when the coordinate-read contract is removed or a stronger law subsumes this case. +#[test] +fn a_corrupt_catalog_head_reports_its_exact_view_refusal() -> Result<(), Box> { + let sandbox = migrated_store("reader-error-corrupt-head")?; + let path = sandbox.path().join("HEAD"); + let mut bytes = fs::read(&path)?; + *bytes.last_mut().ok_or("HEAD absent")? ^= 1; + fs::write(path, bytes)?; + + let error = refusal(sandbox.path())?; + + assert!( + matches!(&error, Some(SnapshotError::View { + source: RetentionViewError::Io { source } + }) if matches!(source.get_ref().and_then(|source| source.downcast_ref::()), + Some(PublicationHeadDecodeError::ChecksumMismatch { .. }))), + "corrupt coordinate HEAD must preserve its view boundary and decoder source: {error:?}" + ); + Ok(()) +} + +fn refusal(path: &Path) -> Result, Box> { + let policy = CatalogRestartPolicy::new( + SegmentReadPolicy::MAXIMUM, + CatalogRestartByteLimit::new(1_048_576)?, + ); + Ok(FilesystemRetentionSnapshot::load(path, policy, ReaderAttemptLimit::DEFAULT).err()) +} diff --git a/src/adapters/retention/filesystem_retention_snapshot_identity_tests.rs b/src/adapters/retention/filesystem_retention_snapshot_identity_tests.rs new file mode 100644 index 00000000..0570214e --- /dev/null +++ b/src/adapters/retention/filesystem_retention_snapshot_identity_tests.rs @@ -0,0 +1,53 @@ +//! These laws own reader admission against migration-bound physical root identity. + +use std::error::Error; +use std::fs; +use std::os::unix::fs::MetadataExt; + +use super::filesystem_retention_test_fixture::migrated_store; +use super::{FilesystemRetentionSnapshot, FilesystemRetentionSnapshotError, ReaderAttemptLimit}; +use crate::adapters::{ + CatalogRestartByteLimit, CatalogRestartPolicy, FilesystemPlatformAdmissionError, + SegmentReadPolicy, StoreRootIdentityCoordinate, +}; + +// Size: medium. Oracle: a reader must admit the root named by migration.intent. +// Delete only with migration binding or a stronger public reader-boundary law. +#[test] +fn a_reader_refuses_migration_records_bound_to_another_root() -> Result<(), Box> { + let donor = migrated_store("reader-migration-identity-donor")?; + let recipient = migrated_store("reader-migration-identity-recipient")?; + for name in ["FORMAT", "migration.intent", "migration.receipt"] { + fs::copy(donor.path().join(name), recipient.path().join(name))?; + } + let expected = fs::metadata(donor.path())?.ino(); + let observed = fs::metadata(recipient.path())?.ino(); + let policy = CatalogRestartPolicy::new( + SegmentReadPolicy::MAXIMUM, + CatalogRestartByteLimit::new(1_048_576)?, + ); + + let result = + FilesystemRetentionSnapshot::load(recipient.path(), policy, ReaderAttemptLimit::DEFAULT); + + let error = match result { + Err(FilesystemRetentionSnapshotError::Admission { source }) => source, + Ok(_) => { + return Err("reader accepted migration records naming another physical root".into()); + } + Err(other) => { + return Err(format!("reader root identity must refuse at admission: {other:?}").into()); + } + }; + let refusal = error + .get_ref() + .and_then(|source| source.downcast_ref::()); + assert!( + matches!(refusal, Some(FilesystemPlatformAdmissionError::RootIdentityChanged { + coordinate: StoreRootIdentityCoordinate::File, + expected: actual_expected, observed: actual_observed, + }) if *actual_expected == expected && *actual_observed == observed), + "reader must preserve exact bound inode {expected} and observed inode {observed}: {error:?}" + ); + Ok(()) +} diff --git a/src/adapters/retention/filesystem_retention_snapshot_moving_error_tests.rs b/src/adapters/retention/filesystem_retention_snapshot_moving_error_tests.rs new file mode 100644 index 00000000..c9239865 --- /dev/null +++ b/src/adapters/retention/filesystem_retention_snapshot_moving_error_tests.rs @@ -0,0 +1,101 @@ +//! This module owns retrying catalog admission when the collected heads move. + +use std::error::Error; +use std::fs; +use std::io; + +use super::{Source, collect_retention_view, pool_name}; +use crate::adapters::retention::filesystem_retention_test_fixture::{ + CATALOG_NAME, ROOT_HEX, fixture, initial_preparation, open_authority, +}; +use crate::adapters::retention::{ + ReaderAttemptLimit, ReaderFence, RetentionViewCoordinates, RetentionViewSource, +}; +use crate::adapters::{CatalogRestartByteLimit, CatalogRestartPolicy, SegmentReadPolicy}; +use crate::execute_retention_publication; +use cap_fs_ext::DirExt; +use cap_std::fs::Dir; + +struct AfterLoad { + source: Source, + action: Option, +} + +impl io::Result<()>> RetentionViewSource for AfterLoad { + type View = ::View; + + fn coordinates(&mut self) -> io::Result { + self.source.coordinates() + } + + fn load(&mut self) -> io::Result { + let view = self.source.load()?; + if let Some(action) = self.action.take() { + action()?; + } + Ok(view) + } +} + +// Size: medium. Oracle: a head change discards even a failed catalog admission; +// only a stable pair may select a catalog result, within the attempt budget. +// Delete when collected reader results are removed or a stronger law subsumes this schedule. +#[test] +fn a_head_change_discards_the_superseded_catalog_refusal() -> Result<(), Box> { + let (sandbox, mut authority) = open_authority("reader-moving-catalog-error")?; + let root = Dir::open_ambient_dir(sandbox.path(), cap_std::ambient_authority())?; + let _fence = ReaderFence::acquire(&root)?; + let retention = root.open_dir_nofollow(pool_name::RETENTION)?; + let manifests = retention.open_dir_nofollow(pool_name::MANIFESTS)?; + let mut source = Source { + root, + retention, + manifests, + policy: CatalogRestartPolicy::new( + SegmentReadPolicy::MAXIMUM, + CatalogRestartByteLimit::new(1_048_576)?, + ), + }; + let expected = source.coordinates()?.catalog.ok_or("catalog HEAD absent")?; + let path = sandbox.path().join("catalogs").join(CATALOG_NAME); + let bytes = fs::read(&path)?; + fs::remove_file(&path)?; + let root_bytes = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root_bytes)?; + let expected_retention = preparation.liveness_generation(); + let mut scheduled = AfterLoad { + source, + action: Some(|| { + fs::write(&path, &bytes)?; + let _published = execute_retention_publication(&mut authority, &preparation) + .map_err(io::Error::other)?; + Ok(()) + }), + }; + + let collected = collect_retention_view(&mut scheduled, ReaderAttemptLimit::DEFAULT); + assert!( + matches!(&collected, Ok(Ok(_))), + "moved heads must discard the superseded catalog refusal" + ); + let view = collected??; + + assert_eq!( + view.catalog.generation(), + expected.0, + "accepted catalog generation" + ); + assert_eq!( + view.catalog.catalog_digest(), + expected.2, + "accepted catalog digest" + ); + assert_eq!( + view.retention + .as_ref() + .map(|state| state.head().generation()), + Some(expected_retention), + "the returned reader view must bind the newly published retention head" + ); + Ok(()) +} diff --git a/src/adapters/retention/filesystem_retention_snapshot_pinning_tests.rs b/src/adapters/retention/filesystem_retention_snapshot_pinning_tests.rs new file mode 100644 index 00000000..7f5a8a58 --- /dev/null +++ b/src/adapters/retention/filesystem_retention_snapshot_pinning_tests.rs @@ -0,0 +1,58 @@ +//! This module owns catalog identity under ambient root replacement. + +use std::error::Error; +use std::fs; + +use super::{Source, collect_retention_view, pool_name}; +use crate::adapters::filesystem_test_sandbox::TestDirectory; +use crate::adapters::retention::filesystem_retention_test_fixture::migrated_store; +use crate::adapters::retention::{ReaderAttemptLimit, RetentionViewSource}; +use crate::adapters::{CatalogRestartByteLimit, CatalogRestartPolicy, SegmentReadPolicy}; +use cap_fs_ext::DirExt; +use cap_std::fs::Dir; + +// Size: medium (owned filesystem). Oracle: a pinned reader's catalog must +// match its original validated HEAD despite replacement of the ambient path. +// Delete only when pinned reader collection is removed or a stronger law subsumes it. +#[test] +fn replacing_the_ambient_root_preserves_the_pinned_catalog() -> Result<(), Box> { + let sandbox = migrated_store("reader-catalog-path-pinning")?; + let moved = TestDirectory::create("reader-catalog-path-pinning-moved")?; + let root = Dir::open_ambient_dir(sandbox.path(), cap_std::ambient_authority())?; + let retention = root.open_dir_nofollow(pool_name::RETENTION)?; + let manifests = retention.open_dir_nofollow(pool_name::MANIFESTS)?; + let mut source = Source { + root, + retention, + manifests, + policy: CatalogRestartPolicy::new( + SegmentReadPolicy::MAXIMUM, + CatalogRestartByteLimit::new(1_048_576)?, + ), + }; + let expected = source + .coordinates()? + .catalog + .ok_or("original HEAD absent")?; + + fs::rename(sandbox.path(), moved.path().join("pinned"))?; + fs::create_dir(sandbox.path())?; + let result = collect_retention_view(&mut source, ReaderAttemptLimit::DEFAULT)?; + assert!( + result.is_ok(), + "ambient replacement must not displace the pinned catalog" + ); + let view = result?; + + assert_eq!( + view.catalog.generation(), + expected.0, + "pinned catalog generation" + ); + assert_eq!( + view.catalog.catalog_digest(), + expected.2, + "pinned catalog digest" + ); + Ok(()) +} diff --git a/src/adapters/retention/filesystem_retention_snapshot_tests.rs b/src/adapters/retention/filesystem_retention_snapshot_tests.rs new file mode 100644 index 00000000..abbff4ac --- /dev/null +++ b/src/adapters/retention/filesystem_retention_snapshot_tests.rs @@ -0,0 +1,150 @@ +//! Reader fence and fenced snapshot laws over migrated stores. + +use std::error::Error; +use std::fs; + +use cap_std::fs::Dir; +use rustix::fs::{FlockOperation, flock}; +use rustix::io::Errno; + +use super::filesystem_retention_test_fixture::{ + ROOT_HEX, fixture, initial_preparation, migrated_store, open_authority, +}; +use super::{ + AdmittedRetentionRoot, FilesystemRetentionSnapshot, FilesystemRetentionSnapshotError, + ReaderAttemptLimit, ReaderFence, RetentionRootDecodeError, +}; +use crate::adapters::{ + CatalogRestartByteLimit, CatalogRestartPolicy, SegmentReadPolicy, SegmentRecordLimit, +}; +use crate::{LayoutEntryLimit, execute_retention_publication}; + +fn policy() -> Result> { + Ok(CatalogRestartPolicy::new( + SegmentReadPolicy::new(SegmentRecordLimit::MAXIMUM, LayoutEntryLimit::MAXIMUM), + CatalogRestartByteLimit::new(1_048_576)?, + )) +} + +#[test] +fn a_migrated_store_snapshot_binds_the_catalog_and_no_retention_head() -> Result<(), Box> +{ + let sandbox = migrated_store("filesystem-retention-snapshot-empty")?; + let snapshot = + FilesystemRetentionSnapshot::load(sandbox.path(), policy()?, ReaderAttemptLimit::DEFAULT)?; + assert_eq!(snapshot.catalog().generation().get(), 1); + assert!(snapshot.retention_head().is_none()); + assert!(snapshot.manifest().is_none()); + Ok(()) +} + +#[test] +fn a_published_generation_is_read_and_its_root_verified() -> Result<(), Box> { + let (sandbox, mut authority) = open_authority("filesystem-retention-snapshot-published")?; + let root_bytes = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root_bytes)?; + let _published = execute_retention_publication(&mut authority, &preparation)?; + drop(authority); + let candidate = AdmittedRetentionRoot::decode(&root_bytes)?; + let namespace = candidate.root().namespace().digest(); + + let snapshot = + FilesystemRetentionSnapshot::load(sandbox.path(), policy()?, ReaderAttemptLimit::DEFAULT)?; + + let head = snapshot + .retention_head() + .ok_or("no retention head in the view")?; + assert_eq!(head.generation(), preparation.liveness_generation()); + let root = snapshot + .retained_root(namespace)? + .ok_or("the manifest does not select the published namespace")?; + assert_eq!(&*root, root_bytes.as_slice()); + Ok(()) +} + +// Size: medium. Oracle: checksum damage reports the exact checksum mismatch against the golden bytes. +// Delete when selected-root reads are removed or stronger corruption evidence subsumes it. +#[test] +fn root_checksum_damage_reports_its_exact_mismatch() -> Result<(), Box> { + let (sandbox, mut authority) = open_authority("filesystem-retention-snapshot-substituted")?; + let root_bytes = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root_bytes)?; + let _published = execute_retention_publication(&mut authority, &preparation)?; + drop(authority); + let candidate = AdmittedRetentionRoot::decode(&root_bytes)?; + let path = super::filesystem_retention_test_fixture::root_pool_path(sandbox.path(), &candidate); + let mut corrupt = root_bytes.clone(); + *corrupt.last_mut().ok_or("empty root fixture")? ^= 0x01; + let checksum_offset = root_bytes.len().checked_sub(32).ok_or("missing checksum")?; + let expected_checksum: [u8; 32] = root_bytes + .get(checksum_offset..) + .ok_or("missing checksum")? + .try_into()?; + let observed_checksum: [u8; 32] = corrupt + .get(checksum_offset..) + .ok_or("missing damaged checksum")? + .try_into()?; + fs::write(&path, &corrupt)?; + + let snapshot = + FilesystemRetentionSnapshot::load(sandbox.path(), policy()?, ReaderAttemptLimit::DEFAULT)?; + let error = snapshot + .retained_root(candidate.root().namespace().digest()) + .err() + .ok_or("a corrupt root pool entry was returned")?; + + assert!( + matches!( + error, + FilesystemRetentionSnapshotError::Root { ref source } + if source.kind() == std::io::ErrorKind::InvalidData + && matches!(source.get_ref().and_then(|cause| cause.downcast_ref::()), + Some(RetentionRootDecodeError::ChecksumMismatch { expected, observed }) + if *expected == expected_checksum && *observed == observed_checksum) + ), + "checksum damage must retain its exact decoder cause and checksum coordinates: {error:?}" + ); + Ok(()) +} + +#[test] +// Size: medium. Oracle: live shared fences exclude a collector with WOULDBLOCK. +// Delete when reader fencing is removed or stronger lifetime evidence subsumes this law. +fn readers_share_the_fence_and_collection_cannot_take_it_exclusively() -> Result<(), Box> +{ + let sandbox = migrated_store("filesystem-retention-snapshot-fence")?; + let root = Dir::open_ambient_dir(sandbox.path(), cap_std::ambient_authority())?; + let first = ReaderFence::acquire(&root)?; + let second = ReaderFence::acquire(&root)?; + let collector = std::fs::File::open(sandbox.path().join("reader.lock"))?; + + let refused = flock(&collector, FlockOperation::NonBlockingLockExclusive); + + assert_eq!( + refused, + Err(Errno::WOULDBLOCK), + "an exclusive fence must wait for readers" + ); + drop(second); + drop(first); + flock(&collector, FlockOperation::NonBlockingLockExclusive)?; + Ok(()) +} + +#[test] +// Size: medium. Oracle: the reader.lock protocol requires an empty regular file. +// Delete when that protocol is removed or a stronger public-reader law subsumes this case. +fn a_non_empty_reader_lock_refuses_the_fence() -> Result<(), Box> { + let sandbox = migrated_store("filesystem-retention-snapshot-bad-fence")?; + fs::write(sandbox.path().join("reader.lock"), b"not empty")?; + let error = + FilesystemRetentionSnapshot::load(sandbox.path(), policy()?, ReaderAttemptLimit::DEFAULT) + .err() + .ok_or("a non-empty reader.lock was accepted as the fence")?; + assert!( + matches!(&error, FilesystemRetentionSnapshotError::Fence { source } + if source.kind() == std::io::ErrorKind::InvalidData), + "a nonempty lock must report Fence with InvalidData: {error:?}" + ); + Ok(()) +} diff --git a/src/adapters/retention/filesystem_retention_stage.rs b/src/adapters/retention/filesystem_retention_stage.rs index 35a24598..259c21e9 100644 --- a/src/adapters/retention/filesystem_retention_stage.rs +++ b/src/adapters/retention/filesystem_retention_stage.rs @@ -1,5 +1,10 @@ //! This module owns exact variable-length retention stage publication. +use super::{ + RetentionEffectDurability as Durability, RetentionNamespaceEffect as Effect, + RetentionStorageBoundary as Boundary, +}; +use super::{RetentionRecordRefusal, RetentionStorageError}; use std::io::{self, Write}; use cap_std::fs::{Dir, File}; @@ -9,6 +14,22 @@ use crate::adapters::filesystem_exact_record::{ self as exact_record, EntryIdentity, ExactRecordError, ExactRecordRefusal, }; +/// Whether a no-replacement link call created a namespace entry. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub(super) enum RetentionLinkOutcome { + Created, + Existing, +} + +impl RetentionLinkOutcome { + pub(super) fn report(self, error: RetentionStorageError) -> RetentionStorageError { + match self { + Self::Created => error.after(Effect::PoolLinkCreated, Durability::Unconfirmed), + Self::Existing => error, + } + } +} + /// One exclusively created, verified, and retained retention stage file. /// /// The stage retains its opened handle and recorded device and inode identity @@ -24,7 +45,11 @@ pub(super) struct FilesystemRetentionStage { impl FilesystemRetentionStage { /// Exclusively creates the named stage and writes its complete bytes. - pub(super) fn create(root: &Dir, name: &'static str, expected: &[u8]) -> io::Result { + pub(super) fn create( + root: &Dir, + name: &'static str, + expected: &[u8], + ) -> Result { let mut file = filesystem_catalog_artifact::create_exclusive(root, name)?; let identity = EntryIdentity::of_file(&file)?; file.write_all(expected)?; @@ -37,46 +62,125 @@ impl FilesystemRetentionStage { }) } + /// Reopens a retained stage whose exact bytes restart already read. + /// + /// The handle and the named entry are verified against `expected` and + /// bound to the observed entry's identity, so every later transition refuses a + /// substituted or replaced stage exactly as a freshly created one would. + pub(super) fn reopen( + root: &Dir, + name: &'static str, + expected: &[u8], + identity: EntryIdentity, + ) -> Result { + let file = exact_record::open_read(root, name)?; + verify_named_record(root, name, expected, identity)?; + Ok(Self { + name, + expected: Box::from(expected), + identity, + file, + }) + } + /// Synchronizes the complete stage and reverifies its exact bytes. - pub(super) fn synchronize(&self, root: &Dir) -> io::Result<()> { - self.require_handle()?; - self.file.sync_all()?; + pub(super) fn synchronize(&self, root: &Dir) -> Result<(), RetentionStorageError> { self.verify_stage(root) + .map_err(|error| error.at(Boundary::SourceVerification))?; + self.file.sync_all().map_err(|source| { + RetentionStorageError::from(source).at(Boundary::StageSynchronization) + })?; + self.verify_stage(root) + .map_err(|error| error.at(Boundary::SourceVerification)) } /// Links the verified stage into `target` under `name` without replacement. - pub(super) fn link(&self, root: &Dir, target: &Dir, name: &str) -> io::Result<()> { - self.verify_stage(root)?; - exact_record::link_without_replacement(root, self.name, target, name)?; - self.verify_stage(root)?; + pub(super) fn link( + &self, + root: &Dir, + target: &Dir, + name: &str, + ) -> Result { + self.verify_stage(root) + .map_err(|error| error.at(Boundary::SourceVerification))?; + let outcome = match root.hard_link(self.name, target, name) { + Ok(()) => RetentionLinkOutcome::Created, + Err(source) if source.kind() == io::ErrorKind::AlreadyExists => { + RetentionLinkOutcome::Existing + } + Err(source) => { + return Err(RetentionStorageError::from(source) + .at(Boundary::PoolLink) + .uncertain(Effect::PoolLinkCreated)); + } + }; + self.verify_stage(root) + .map_err(|error| outcome.report(error.at(Boundary::SourceVerification)))?; verify_named_record(target, name, &self.expected, self.identity) + .map_err(|error| outcome.report(error.at(Boundary::PoolVerification)))?; + Ok(outcome) } /// Removes only the retained stage after confirming its linked target. - pub(super) fn remove(self, root: &Dir, target: &Dir, name: &str) -> io::Result<()> { - verify_named_record(target, name, &self.expected, self.identity)?; - root.remove_file(self.name)?; - exact_record::require_absent(root, self.name).map_err(retention_error)?; + pub(super) fn remove( + &self, + root: &Dir, + target: &Dir, + name: &str, + ) -> Result<(), RetentionStorageError> { + self.verify_stage(root) + .map_err(|error| error.at(Boundary::SourceVerification))?; verify_named_record(target, name, &self.expected, self.identity) + .map_err(|error| error.at(Boundary::PoolVerification))?; + root.remove_file(self.name).map_err(|source| { + RetentionStorageError::from(source) + .at(Boundary::StageUnlink) + .uncertain(Effect::StageRemoved) + })?; + exact_record::require_absent(root, self.name).map_err(|error| { + retention_error(error) + .at(Boundary::StageAbsence) + .after(Effect::StageRemoved, Durability::Unconfirmed) + })?; + verify_named_record(target, name, &self.expected, self.identity).map_err(|error| { + error + .at(Boundary::PoolVerification) + .after(Effect::StageRemoved, Durability::Unconfirmed) + }) } /// Renames the verified stage onto `name`, replacing it atomically. - pub(super) fn replace(self, root: &Dir, name: &str) -> io::Result<()> { - self.verify_stage(root)?; - root.rename(self.name, root, name)?; - exact_record::require_absent(root, self.name).map_err(retention_error)?; - verify_named_record(root, name, &self.expected, self.identity) + pub(super) fn replace(&self, root: &Dir, name: &str) -> Result<(), RetentionStorageError> { + self.verify_stage(root) + .map_err(|error| error.at(Boundary::SourceVerification))?; + root.rename(self.name, root, name).map_err(|source| { + RetentionStorageError::from(source) + .at(Boundary::HeadRename) + .uncertain(Effect::HeadReplaced) + })?; + exact_record::require_absent(root, self.name).map_err(|error| { + retention_error(error) + .at(Boundary::StageAbsence) + .after(Effect::HeadReplaced, Durability::Unconfirmed) + })?; + verify_named_record(root, name, &self.expected, self.identity).map_err(|error| { + error + .at(Boundary::HeadVerification) + .after(Effect::HeadReplaced, Durability::Unconfirmed) + }) } - fn require_handle(&self) -> io::Result<()> { + fn require_handle(&self) -> Result<(), RetentionStorageError> { if EntryIdentity::of_file(&self.file)? == self.identity { Ok(()) } else { - Err(invalid_data("retention stage handle changed identity")) + Err(RetentionStorageError::Refused { + source: RetentionRecordRefusal::KindLengthOrIdentity, + }) } } - fn verify_stage(&self, root: &Dir) -> io::Result<()> { + fn verify_stage(&self, root: &Dir) -> Result<(), RetentionStorageError> { self.require_handle()?; verify_named_record(root, self.name, &self.expected, self.identity) } @@ -87,24 +191,26 @@ fn verify_named_record( name: &str, expected: &[u8], identity: EntryIdentity, -) -> io::Result<()> { +) -> Result<(), RetentionStorageError> { exact_record::verify_named(directory, name, expected, identity).map_err(retention_error) } -/// Maps a shared exact-record failure onto this protocol's refusal messages. -fn retention_error(error: ExactRecordError) -> io::Error { +/// Adapts exact-record failures without erasing their semantic distinctions. +pub(super) fn retention_error(error: ExactRecordError) -> RetentionStorageError { match error { - ExactRecordError::Io(source) => source, - ExactRecordError::Refused(refusal) => invalid_data(match refusal { - ExactRecordRefusal::LengthOverflow => "retention record length exceeded u64", - ExactRecordRefusal::KindOrLength | ExactRecordRefusal::KindLengthOrIdentity => { - "retention record kind, length, or identity disagreed" - } - ExactRecordRefusal::Bytes | ExactRecordRefusal::TrailingBytes => { - "retention record bytes disagreed" - } - ExactRecordRefusal::RemainedVisible => "removed retention stage remained visible", - }), + ExactRecordError::Io(source) => RetentionStorageError::Io { source }, + ExactRecordError::Refused(refusal) => RetentionStorageError::Refused { + source: match refusal { + ExactRecordRefusal::LengthOverflow => RetentionRecordRefusal::LengthOverflow, + ExactRecordRefusal::KindOrLength => RetentionRecordRefusal::KindOrLength, + ExactRecordRefusal::KindLengthOrIdentity => { + RetentionRecordRefusal::KindLengthOrIdentity + } + ExactRecordRefusal::Bytes => RetentionRecordRefusal::Bytes, + ExactRecordRefusal::TrailingBytes => RetentionRecordRefusal::TrailingBytes, + ExactRecordRefusal::RemainedVisible => RetentionRecordRefusal::RemainedVisible, + }, + }, } } diff --git a/src/adapters/retention/filesystem_retention_storage.rs b/src/adapters/retention/filesystem_retention_storage.rs index e5c897c9..a9f756c1 100644 --- a/src/adapters/retention/filesystem_retention_storage.rs +++ b/src/adapters/retention/filesystem_retention_storage.rs @@ -14,8 +14,9 @@ use super::filesystem_retention_pool_name as pool_name; use super::filesystem_retention_stage::FilesystemRetentionStage; use super::{ AdmittedRetentionRoot, CanonicalRetentionHead, CanonicalRetentionManifest, - RetentionCurrentStateRefusal, RetentionNamespaceAdmission, RetentionPublicationPreparation, - RetentionPublicationStorage, RetentionTransitionDisposition, + FilesystemRetentionRecoveryError, RetentionCurrentStateRefusal, RetentionNamespaceAdmission, + RetentionPublicationPreparation, RetentionPublicationStorage, RetentionRecoveryOutcome, + RetentionTransitionDisposition, }; use crate::RetentionGenerationExpectation; use crate::adapters::filesystem_catalog_artifact::synchronize_directory; @@ -27,7 +28,23 @@ impl RetentionPublicationStorage for FilesystemRetentionPublicationAuthority { preparation: &RetentionPublicationPreparation<'_>, ) -> io::Result { self.attempt = None; - require_pinned_directories(&self.root, &self.retention, &self.roots, &self.manifests)?; + let recovery = self.recover().map_err(|error| match error { + FilesystemRetentionRecoveryError::Observe { source } => { + RetentionCurrentStateRefusal::RecoveryObservationRefused { source }.into_io() + } + FilesystemRetentionRecoveryError::Plan { source } => { + RetentionCurrentStateRefusal::RecoveryRefused { source }.into_io() + } + FilesystemRetentionRecoveryError::Execute { source } => { + RetentionCurrentStateRefusal::RecoveryStepRefused { source }.into_io() + } + })?; + if matches!( + recovery.outcome(), + RetentionRecoveryOutcome::Protected { .. } + ) { + return Err(RetentionCurrentStateRefusal::RetainedStage.into_io()); + } require_no_retained_stage(&self.retention)?; let census = filesystem_retention_namespace::admit(&self.retention, &self.roots, &self.manifests)?; @@ -199,6 +216,7 @@ impl RetentionPublicationStorage for FilesystemRetentionPublicationAuthority { attempt::require_mut(&mut self.attempt)? .take_head_stage()? .replace(&self.retention, pool_name::HEAD) + .map_err(Into::into) } fn synchronize_retention_namespace(&mut self) -> io::Result<()> { @@ -208,21 +226,25 @@ impl RetentionPublicationStorage for FilesystemRetentionPublicationAuthority { fn remove_root_stage(&mut self) -> io::Result<()> { let attempt = attempt::require_mut(&mut self.attempt)?; let stage = attempt.take_root_stage()?; - stage.remove( - &self.retention, - attempt.namespace()?, - attempt.retained_root_name()?, - ) + stage + .remove( + &self.retention, + attempt.namespace()?, + attempt.retained_root_name()?, + ) + .map_err(Into::into) } fn remove_manifest_stage(&mut self) -> io::Result<()> { let attempt = attempt::require_mut(&mut self.attempt)?; let stage = attempt.take_manifest_stage()?; - stage.remove( - &self.retention, - &self.manifests, - attempt.retained_manifest_name()?, - ) + stage + .remove( + &self.retention, + &self.manifests, + attempt.retained_manifest_name()?, + ) + .map_err(Into::into) } fn synchronize_cleanup(&mut self) -> io::Result<()> { @@ -238,7 +260,7 @@ impl RetentionPublicationStorage for FilesystemRetentionPublicationAuthority { /// `roots`, or `manifests` entry renamed and replaced after admission means the /// store's namespace no longer describes the admitted state; publication /// refuses instead of writing into a directory no reader would find. -fn require_pinned_directories( +pub(super) fn require_pinned_directories( root: &Dir, retention: &Dir, roots: &Dir, diff --git a/src/adapters/retention/filesystem_retention_storage_tests.rs b/src/adapters/retention/filesystem_retention_storage_tests.rs index 3c132572..b1718c01 100644 --- a/src/adapters/retention/filesystem_retention_storage_tests.rs +++ b/src/adapters/retention/filesystem_retention_storage_tests.rs @@ -59,6 +59,8 @@ fn existing_root_stage_is_never_truncated() -> Result<(), Box> { Ok(()) } +// Size: medium. Oracle: a complete head without its manifest refuses with HeadStageWithoutManifestStage. +// Delete when this recovery state is removed or stronger public-boundary coverage subsumes it. #[test] fn retained_stage_refuses_publication_before_recovery() -> Result<(), Box> { let (sandbox, mut authority) = open_authority("filesystem-retention-recovery-required")?; @@ -77,6 +79,15 @@ fn retained_stage_refuses_publication_before_recovery() -> Result<(), Box io::Result)>> { } #[test] -fn retained_manifest_stage_refuses_publication_before_recovery() -> Result<(), Box> { - let (sandbox, mut authority) = - open_authority("filesystem-retention-recovery-required-manifest")?; +fn a_complete_orphan_root_stage_refuses_publication_until_disposition() -> Result<(), Box> +{ + let (sandbox, mut authority) = open_authority("filesystem-retention-recovery-required-orphan")?; let root_bytes = fixture(ROOT_HEX)?; let preparation = initial_preparation(&root_bytes)?; fs::write( - sandbox.path().join("retention").join("manifest.next"), - b"partial bytes left by a failed write", + sandbox.path().join("retention").join("root.next"), + &root_bytes, )?; - let before = retention_witness(sandbox.path())?; let error = execute_retention_publication(&mut authority, &preparation) .err() - .ok_or("retained manifest stage was unexpectedly published over")?; + .ok_or("a complete orphan root stage was unexpectedly published over")?; let RetentionPublicationError::CurrentVerification { source } = error else { - return Err("retained stage refused outside current-state verification".into()); + return Err("protected orphan refused outside current-state verification".into()); }; - assert_eq!(source.kind(), io::ErrorKind::InvalidData); - assert_eq!(retention_witness(sandbox.path())?, before); + assert!(matches!( + super::filesystem_retention_test_fixture::refusal(&source), + Some(super::RetentionCurrentStateRefusal::RetainedStage) + )); + assert!(sandbox.path().join("retention").join("root.next").is_file()); + assert_eq!( + fs::read(root_pool_path(sandbox.path(), preparation.candidate()))?, + root_bytes + ); + Ok(()) +} + +#[test] +fn a_truncated_stage_preserves_evidence_and_blocks_publication() -> Result<(), Box> { + let (sandbox, mut authority) = open_authority("filesystem-retention-recovered-stage")?; + let root_bytes = fixture(ROOT_HEX)?; + let preparation = initial_preparation(&root_bytes)?; + let stage = sandbox.path().join("retention").join("root.next"); + let prefix = root_bytes.get(..35).ok_or("no root prefix")?; + // Decision A preserves even a first-stage interruption for disposition. + fs::write(&stage, prefix)?; + + let before = retention_witness(sandbox.path())?; + let error = execute_retention_publication(&mut authority, &preparation) + .err() + .ok_or("incomplete stage admitted publication")?; + assert!( + matches!(error, RetentionPublicationError::CurrentVerification { ref source } if matches!(source.get_ref().and_then(|cause| cause.downcast_ref::()), Some(super::RetentionCurrentStateRefusal::RecoveryRefused { source: super::RetentionRecoveryRefusal::IncompleteStageRequiresDisposition { stage: super::RetentionFixedStage::Root, expected: 192, observed: 35 } }))), + "publication must retain the disposition cause: {error:?}" + ); + assert_eq!( + retention_witness(sandbox.path())?, + before, + "blocked publication must preserve every retained byte" + ); Ok(()) } diff --git a/src/adapters/retention/filesystem_retention_test_fixture.rs b/src/adapters/retention/filesystem_retention_test_fixture.rs index 9672c10d..76aa77ac 100644 --- a/src/adapters/retention/filesystem_retention_test_fixture.rs +++ b/src/adapters/retention/filesystem_retention_test_fixture.rs @@ -13,6 +13,7 @@ use super::{ AdmittedRetentionManifest, AdmittedRetentionRoot, CanonicalRetentionRoot, RetentionCurrentStateRefusal, RetentionPublicationPreparation, RetentionTransitionDisposition, }; +use super::{RetentionPublicationPhase as Phase, RetentionPublicationStorage}; use crate::LayoutEntryLimit; use crate::adapters::filesystem_test_sandbox::TestDirectory; use crate::adapters::test_support::decode_hex; @@ -44,7 +45,8 @@ const CATALOG_HEX: &str = const CATALOG_HEAD_HEX: &str = include_str!("../../../conformance/segment-store/v1/one-zero-bundle-head.hex"); -const SEGMENT_NAME: &str = "221f6745cd8a5221c9a87c3707593608479282b54a4a74d0e753fd76f70e8db2.seg"; +pub(super) const SEGMENT_NAME: &str = + "221f6745cd8a5221c9a87c3707593608479282b54a4a74d0e753fd76f70e8db2.seg"; pub(super) const CATALOG_NAME: &str = "0000000000000001-0b7cad1b6de663d34beacbc214db7497f2e36ab6b08dfbd5febbc8d06a418811.cat"; @@ -57,7 +59,8 @@ pub(super) fn open_authority( name: &str, ) -> Result<(TestDirectory, FilesystemRetentionPublicationAuthority), Box> { let sandbox = migrated_store(name)?; - let admission = FilesystemVersionTwoAdmission::reopen_unchecked_for_tests(sandbox.path())?; + let admission = + FilesystemVersionTwoAdmission::reopen_unchecked_for_repository_tasks(sandbox.path())?; let authority = FilesystemRetentionPublicationAuthority::open(admission)?; Ok((sandbox, authority)) } @@ -169,11 +172,18 @@ pub(super) fn successor_root( CanonicalRetentionRoot::from_root(&root).map_err(Into::into) } -/// Extracts the typed current-state refusal carried by a verification error. +/// Extracts the underlying current-state cause, traversing recovery observation. +/// Tests of the observation boundary itself inspect the outer error directly. pub(super) fn refusal(source: &io::Error) -> Option<&RetentionCurrentStateRefusal> { - source + let refusal = source .get_ref() - .and_then(|inner| inner.downcast_ref::()) + .and_then(|inner| inner.downcast_ref::())?; + match refusal { + RetentionCurrentStateRefusal::RecoveryObservationRefused { source } => { + self::refusal(source) + } + other => Some(other), + } } pub(super) fn head_path(root: &Path) -> PathBuf { @@ -261,3 +271,48 @@ fn write_version_one(sandbox: &TestDirectory) -> Result<(), Box> { const fn maximum_policy() -> SegmentReadPolicy { SegmentReadPolicy::new(SegmentRecordLimit::MAXIMUM, LayoutEntryLimit::MAXIMUM) } + +/// The number of storage-port phases one publication executes. +pub(super) const PUBLICATION_PHASE_COUNT: usize = Phase::ALL.len() + 1; + +/// Executes publication phases 1 through `count` and stops, like a crash there. +/// +/// Phase 1 is current-state verification; 2 through 18 are the storage-port +/// phases in `RetentionPublicationPhase::ALL` order, so `count` selects the +/// exact prefix a process death after that phase would leave behind. +pub(super) fn drive_publication( + authority: &mut FilesystemRetentionPublicationAuthority, + preparation: &RetentionPublicationPreparation<'_>, + count: usize, +) -> Result<(), Box> { + let publication = preparation + .publication() + .ok_or("preparation carries no publication")?; + let root = preparation.candidate(); + let Some(storage_count) = count.checked_sub(1) else { + return Ok(()); + }; + let _verification = authority.verify_current(preparation)?; + for phase in Phase::ALL.into_iter().take(storage_count) { + match phase { + Phase::WriteRootStage => authority.write_root_stage(root), + Phase::SynchronizeRootStage => authority.synchronize_root_stage(), + Phase::AdmitRootNamespace => authority.admit_root_namespace(root).map(|_| ()), + Phase::SynchronizeRootsAfterNamespace => authority.synchronize_roots_after_namespace(), + Phase::LinkRoot => authority.link_root(root), + Phase::SynchronizeRootNamespace => authority.synchronize_root_namespace(root), + Phase::WriteManifestStage => authority.write_manifest_stage(publication.manifest()), + Phase::SynchronizeManifestStage => authority.synchronize_manifest_stage(), + Phase::LinkManifest => authority.link_manifest(publication.manifest()), + Phase::SynchronizeManifestPool => authority.synchronize_manifest_pool(), + Phase::WriteHeadStage => authority.write_head_stage(publication.head()), + Phase::SynchronizeHeadStage => authority.synchronize_head_stage(), + Phase::ReplaceHead => authority.replace_head(), + Phase::SynchronizeRetentionNamespace => authority.synchronize_retention_namespace(), + Phase::RemoveRootStage => authority.remove_root_stage(), + Phase::RemoveManifestStage => authority.remove_manifest_stage(), + Phase::SynchronizeCleanup => authority.synchronize_cleanup(), + }?; + } + Ok(()) +} diff --git a/src/adapters/retention/filesystem_version_two_admission_tests.rs b/src/adapters/retention/filesystem_version_two_admission_tests.rs index 9649bef5..98fe3a28 100644 --- a/src/adapters/retention/filesystem_version_two_admission_tests.rs +++ b/src/adapters/retention/filesystem_version_two_admission_tests.rs @@ -28,9 +28,10 @@ fn version_two_reopen_refuses_a_corrupt_format_marker() -> Result<(), Box Result<(), Box Result<(), Box> { let sandbox = migrated_store("version-two-admission-exact")?; - let admission = FilesystemVersionTwoAdmission::reopen_unchecked_for_tests(sandbox.path())?; + let admission = + FilesystemVersionTwoAdmission::reopen_unchecked_for_repository_tasks(sandbox.path())?; drop(admission); Ok(()) @@ -160,9 +164,10 @@ fn refuses_namespace( let sandbox = migrated_store(name)?; mutate(sandbox.path())?; - let error = FilesystemVersionTwoAdmission::reopen_unchecked_for_tests(sandbox.path()) - .err() - .ok_or_else(|| format!("{name}: version-two root was unexpectedly admitted"))?; + let error = + FilesystemVersionTwoAdmission::reopen_unchecked_for_repository_tasks(sandbox.path()) + .err() + .ok_or_else(|| format!("{name}: version-two root was unexpectedly admitted"))?; assert!( matches!(error, FilesystemPlatformAdmissionError::Namespace { .. }), @@ -211,7 +216,8 @@ fn a_protocol_directory_replaced_after_reopen_is_neither_opened_nor_published_in let preparation = initial_preparation(&root_bytes)?; let _published = execute_retention_publication(&mut first, &preparation)?; drop(first); - let admission = FilesystemVersionTwoAdmission::reopen_unchecked_for_tests(sandbox.path())?; + let admission = + FilesystemVersionTwoAdmission::reopen_unchecked_for_repository_tasks(sandbox.path())?; fs::rename( sandbox.path().join("retention"), sandbox.path().join("retention.moved"), diff --git a/src/adapters/retention/head_decode_error.rs b/src/adapters/retention/head_decode_error.rs index dee202b4..bbc877e9 100644 --- a/src/adapters/retention/head_decode_error.rs +++ b/src/adapters/retention/head_decode_error.rs @@ -5,6 +5,11 @@ use crate::{LivenessGenerationError, RetentionHeadError, RetentionManifestLength /// Failure to decode and admit one version-2 retention head. #[derive(Clone, Copy, Debug, Eq, PartialEq)] pub enum RetentionHeadDecodeError { + /// Available size-field bytes admit no canonical completion. + ManifestLengthPrefixImpossible { + /// Actual byte length of the interrupted stage. + observed: usize, + }, /// The input was not exactly one complete fixed-width head. WrongLength { /// Required fixed width. @@ -12,6 +17,15 @@ pub enum RetentionHeadDecodeError { /// Observed input width. observed: usize, }, + /// An available byte contradicts a fixed field or checksum in an interrupted stage. + PrefixByteMismatch { + /// Absolute byte offset in the observed stage. + offset: usize, + /// The canonical byte required at this offset. + expected: u8, + /// The byte actually present at this offset. + observed: u8, + }, /// The fixed record magic was not canonical. InvalidMagic { /// Observed 16 magic bytes. diff --git a/src/adapters/retention/head_decode_error_display.rs b/src/adapters/retention/head_decode_error_display.rs index 601e2c90..a3ff16fe 100644 --- a/src/adapters/retention/head_decode_error_display.rs +++ b/src/adapters/retention/head_decode_error_display.rs @@ -7,10 +7,22 @@ use super::RetentionHeadDecodeError; impl fmt::Display for RetentionHeadDecodeError { fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { match self { + Self::ManifestLengthPrefixImpossible { observed } => write!( + formatter, + "retention head size fields have no canonical completion at {observed} bytes" + ), Self::WrongLength { expected, observed } => write!( formatter, "retention head has {observed} bytes; expected {expected}" ), + Self::PrefixByteMismatch { + offset, + expected, + observed, + } => write!( + formatter, + "retention stage byte {offset} is {observed:#04x}; expected {expected:#04x}" + ), Self::InvalidMagic { observed } => { write!(formatter, "invalid retention head magic {observed:02x?}") } @@ -57,6 +69,8 @@ impl Error for RetentionHeadDecodeError { Self::ManifestLength { source } => Some(source), Self::Semantic { source } => Some(source), Self::WrongLength { .. } + | Self::PrefixByteMismatch { .. } + | Self::ManifestLengthPrefixImpossible { .. } | Self::InvalidMagic { .. } | Self::UnsupportedVersion { .. } | Self::InvalidRecordLength { .. } diff --git a/src/adapters/retention/head_decoder.rs b/src/adapters/retention/head_decoder.rs index 76d9462d..c61395dc 100644 --- a/src/adapters/retention/head_decoder.rs +++ b/src/adapters/retention/head_decoder.rs @@ -16,15 +16,20 @@ pub(super) fn decode( require_length(encoded)?; validate_fixed_fields(encoded)?; verify_checksum(encoded)?; + let head = admit_fields(encoded)?; + Ok(ChecksummedRetentionHead::admitted(encoded, head)) +} + +/// Admits complete semantic fields, including those preceding an interrupted checksum. +pub(super) fn admit_fields(encoded: &[u8]) -> Result { let generation = LivenessGeneration::new(read_u64(encoded, 24)?) .map_err(|source| RetentionHeadDecodeError::LivenessGeneration { source })?; let manifest_length = RetentionManifestLength::new(read_u64(encoded, 32)?) .map_err(|source| RetentionHeadDecodeError::ManifestLength { source })?; let manifest_digest = RetentionManifestDigest::from_hash(read_array(encoded, 40)?); let predecessor = predecessor(read_array(encoded, 72)?); - let head = RetentionHead::new(generation, manifest_length, manifest_digest, predecessor) - .map_err(|source| RetentionHeadDecodeError::Semantic { source })?; - Ok(ChecksummedRetentionHead::admitted(encoded, head)) + RetentionHead::new(generation, manifest_length, manifest_digest, predecessor) + .map_err(|source| RetentionHeadDecodeError::Semantic { source }) } fn validate_fixed_fields(encoded: &[u8]) -> Result<(), RetentionHeadDecodeError> { @@ -107,7 +112,7 @@ fn read_u32(encoded: &[u8], offset: usize) -> Result Result { +pub(super) fn read_u64(encoded: &[u8], offset: usize) -> Result { read_array(encoded, offset).map(u64::from_be_bytes) } diff --git a/src/adapters/retention/manifest_decode_error.rs b/src/adapters/retention/manifest_decode_error.rs index 4f00dc85..384e0bd9 100644 --- a/src/adapters/retention/manifest_decode_error.rs +++ b/src/adapters/retention/manifest_decode_error.rs @@ -7,6 +7,11 @@ use crate::{LivenessGenerationError, RetentionManifestError, RootGenerationError /// Failure to decode and admit one version-2 retention manifest. #[derive(Debug)] pub enum RetentionManifestDecodeError { + /// Available size-field bytes admit no canonical completion. + FramingPrefixImpossible { + /// Actual byte length of the interrupted stage. + observed: usize, + }, /// The byte string ended before its required exact length. Truncated { /// Required byte length. @@ -21,6 +26,15 @@ pub enum RetentionManifestDecodeError { /// Observed byte length. observed: usize, }, + /// An available byte contradicts a fixed or computed integrity field in an interrupted stage. + PrefixByteMismatch { + /// Absolute byte offset in the observed stage. + offset: usize, + /// The canonical byte required at this offset. + expected: u8, + /// The byte actually present at this offset. + observed: u8, + }, /// The fixed record magic was not canonical. InvalidMagic { /// Observed 16 magic bytes. diff --git a/src/adapters/retention/manifest_decode_error_display.rs b/src/adapters/retention/manifest_decode_error_display.rs index 76a81dd6..98c6ee5f 100644 --- a/src/adapters/retention/manifest_decode_error_display.rs +++ b/src/adapters/retention/manifest_decode_error_display.rs @@ -7,6 +7,10 @@ use super::RetentionManifestDecodeError; impl fmt::Display for RetentionManifestDecodeError { fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { match self { + Self::FramingPrefixImpossible { observed } => write!( + formatter, + "retention manifest size fields have no canonical completion at {observed} bytes" + ), Self::Truncated { expected, observed } => write!( formatter, "retention manifest has {observed} bytes; expected {expected}" @@ -15,6 +19,14 @@ impl fmt::Display for RetentionManifestDecodeError { formatter, "retention manifest has trailing data: expected {expected} bytes, observed {observed}" ), + Self::PrefixByteMismatch { + offset, + expected, + observed, + } => write!( + formatter, + "retention stage byte {offset} is {observed:#04x}; expected {expected:#04x}" + ), Self::InvalidMagic { observed } => { write!( formatter, @@ -91,6 +103,8 @@ impl Error for RetentionManifestDecodeError { Self::Semantic { source } => Some(source), Self::Truncated { .. } | Self::TrailingData { .. } + | Self::PrefixByteMismatch { .. } + | Self::FramingPrefixImpossible { .. } | Self::InvalidMagic { .. } | Self::UnsupportedVersion { .. } | Self::InvalidHeaderLength { .. } diff --git a/src/adapters/retention/manifest_entry_decoder.rs b/src/adapters/retention/manifest_entry_decoder.rs index 46d09fb9..4db1f081 100644 --- a/src/adapters/retention/manifest_entry_decoder.rs +++ b/src/adapters/retention/manifest_entry_decoder.rs @@ -26,21 +26,9 @@ pub(super) fn decode( for (position, bytes) in encoded.chunks_exact(ENTRY_WIDTH).enumerate() { let index = u32::try_from(position).map_err(|_| RetentionManifestDecodeError::LengthOverflow)?; - let namespace = RetentionNamespaceDigest::from_hash(read_array(bytes, 0)?); - let root_generation = RootGeneration::new(read_u64(bytes, 32)?) - .map_err(|source| RetentionManifestDecodeError::RootGeneration { index, source })?; - let root_digest = RetentionRootDigest::from_hash(read_array(bytes, 40)?); - if let Some(prior) = previous - && namespace <= prior - { - return Err(RetentionManifestDecodeError::NonCanonicalEntryOrder { index }); - } - entries.push(RetentionManifestEntry::new( - namespace, - root_generation, - root_digest, - )); - previous = Some(namespace); + let entry = admit_entry(bytes, index, previous)?; + previous = Some(entry.namespace()); + entries.push(entry); } Ok(entries) } @@ -67,3 +55,71 @@ fn read_array( observed: encoded.len(), }) } + +/// Admits complete entries and constraints decidable from the partial suffix. +pub(super) fn admit_prefix(encoded: &[u8]) -> Result<(), RetentionManifestDecodeError> { + let mut previous = None; + for (position, bytes) in encoded.chunks(ENTRY_WIDTH).enumerate() { + let index = + u32::try_from(position).map_err(|_| RetentionManifestDecodeError::LengthOverflow)?; + if bytes.len() < ENTRY_WIDTH { + return admit_partial_entry(bytes, index, previous); + } + previous = Some(admit_entry(bytes, index, previous)?.namespace()); + } + Ok(()) +} + +fn admit_entry( + bytes: &[u8], + index: u32, + previous: Option, +) -> Result { + let namespace = RetentionNamespaceDigest::from_hash(read_array(bytes, 0)?); + let root_generation = admit_generation(bytes, index)?; + let root_digest = RetentionRootDigest::from_hash(read_array(bytes, 40)?); + admit_order(namespace.as_bytes(), index, previous)?; + Ok(RetentionManifestEntry::new( + namespace, + root_generation, + root_digest, + )) +} + +fn admit_partial_entry( + bytes: &[u8], + index: u32, + previous: Option, +) -> Result<(), RetentionManifestDecodeError> { + if bytes.len() >= 40 { + let _generation = admit_generation(bytes, index)?; + } + // All unknown namespace suffix bytes set to 0xff give the greatest possible + // completion. If even this cannot follow the predecessor, no completion can. + let mut maximum = [u8::MAX; 32]; + for (target, observed) in maximum.iter_mut().zip(bytes) { + *target = *observed; + } + admit_order(&maximum, index, previous) +} + +fn admit_generation( + bytes: &[u8], + index: u32, +) -> Result { + RootGeneration::new(read_u64(bytes, 32)?) + .map_err(|source| RetentionManifestDecodeError::RootGeneration { index, source }) +} + +fn admit_order( + namespace: &[u8; 32], + index: u32, + previous: Option, +) -> Result<(), RetentionManifestDecodeError> { + if let Some(prior) = previous + && namespace <= prior.as_bytes() + { + return Err(RetentionManifestDecodeError::NonCanonicalEntryOrder { index }); + } + Ok(()) +} diff --git a/src/adapters/retention/manifest_header_decoder.rs b/src/adapters/retention/manifest_header_decoder.rs index d577564f..5d890a45 100644 --- a/src/adapters/retention/manifest_header_decoder.rs +++ b/src/adapters/retention/manifest_header_decoder.rs @@ -6,8 +6,13 @@ use super::manifest_field_decoder::{ }; pub(super) const HEADER_LENGTH: usize = 160; -const ENTRY_WIDTH: usize = 72; -const TRAILER_LENGTH: usize = 64; +pub(super) const ENTRY_WIDTH: usize = 72; +pub(super) const TRAILER_LENGTH: usize = 64; + +/// Derives the read bound from the same framing admitted by this decoder. +pub(super) fn maximum_encoded_length() -> Result { + canonical_length(crate::RetentionManifest::MAXIMUM_ENTRY_COUNT) +} pub(super) struct DecodedManifestHeader { pub(super) generation: u64, @@ -65,6 +70,13 @@ fn validate_fixed_fields(encoded: &[u8]) -> Result<(), RetentionManifestDecodeEr require_zero(encoded, 112, 48, "trailing header") } +/// Checks framing once the complete size fields of an interrupted header exist. +pub(super) fn admit_prefix_length(encoded: &[u8]) -> Result<(), RetentionManifestDecodeError> { + let entry_count = read_u32(encoded, 44)?; + super::manifest_semantic_header::admit_count(entry_count)?; + require_declared_length(encoded, canonical_length(entry_count)?) +} + fn canonical_length(entry_count: u32) -> Result { let entries = usize::try_from(entry_count) .map_err(|_| RetentionManifestDecodeError::LengthOverflow)? diff --git a/src/adapters/retention/manifest_integrity.rs b/src/adapters/retention/manifest_integrity.rs index 9e4e7413..1c9a2a11 100644 --- a/src/adapters/retention/manifest_integrity.rs +++ b/src/adapters/retention/manifest_integrity.rs @@ -80,3 +80,32 @@ fn hash(domain: &[u8], bytes: &[u8]) -> [u8; 32] { hasher.update(bytes); *hasher.finalize().as_bytes() } + +/// Verifies only digest bytes whose entire preimage has arrived. +pub(super) fn verify_prefix( + encoded: &[u8], + digest_offset: usize, + checksum_offset: usize, +) -> Result<(), RetentionManifestDecodeError> { + for (offset, domain) in [ + ( + checksum_offset, + b"keep.retention-manifest-checksum/v2\0".as_slice(), + ), + (digest_offset, b"keep.retention-manifest/v2\0".as_slice()), + ] { + if let Some(preimage) = encoded.get(..offset) { + let expected = hash(domain, preimage); + super::stage_fixed_field_admission::admit( + encoded, + &[(offset, &expected)], + |offset, expected, observed| RetentionManifestDecodeError::PrefixByteMismatch { + offset, + expected, + observed, + }, + )?; + } + } + Ok(()) +} diff --git a/src/adapters/retention/manifest_semantic_header.rs b/src/adapters/retention/manifest_semantic_header.rs index 8135a08b..c22dc235 100644 --- a/src/adapters/retention/manifest_semantic_header.rs +++ b/src/adapters/retention/manifest_semantic_header.rs @@ -1,4 +1,4 @@ -//! This boundary module owns post-integrity retention manifest header admission. +//! This boundary module owns retention manifest header semantic admission. use super::RetentionManifestDecodeError; use super::manifest_header_decoder::DecodedManifestHeader; @@ -12,12 +12,7 @@ pub(super) struct AdmittedManifestHeader { pub(super) fn admit( header: &DecodedManifestHeader, ) -> Result { - if header.entry_count > RetentionManifest::MAXIMUM_ENTRY_COUNT { - return Err(RetentionManifestDecodeError::EntryCountExceeded { - maximum: RetentionManifest::MAXIMUM_ENTRY_COUNT, - observed: header.entry_count, - }); - } + admit_count(header.entry_count)?; let generation = LivenessGeneration::new(header.generation) .map_err(|source| RetentionManifestDecodeError::LivenessGeneration { source })?; Ok(AdmittedManifestHeader { @@ -33,3 +28,14 @@ fn predecessor(bytes: [u8; 32]) -> Option { Some(RetentionManifestDigest::from_hash(bytes)) } } + +/// Admits a complete count before either interrupted-stage discard or body allocation. +pub(super) const fn admit_count(entry_count: u32) -> Result<(), RetentionManifestDecodeError> { + if entry_count > RetentionManifest::MAXIMUM_ENTRY_COUNT { + return Err(RetentionManifestDecodeError::EntryCountExceeded { + maximum: RetentionManifest::MAXIMUM_ENTRY_COUNT, + observed: entry_count, + }); + } + Ok(()) +} diff --git a/src/adapters/retention/reader_attempt_limit.rs b/src/adapters/retention/reader_attempt_limit.rs new file mode 100644 index 00000000..67b6afdf --- /dev/null +++ b/src/adapters/retention/reader_attempt_limit.rs @@ -0,0 +1,25 @@ +//! This module owns the bounded retry limit of reader view collection. + +use std::num::NonZeroU32; + +/// How many times a reader may re-collect before refusing a moving store. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +#[must_use] +pub struct ReaderAttemptLimit(NonZeroU32); + +impl ReaderAttemptLimit { + /// Three attempts: one publication may land between any two reads, and a + /// store that moves faster than a reader can double-collect is refused. + pub const DEFAULT: Self = Self(NonZeroU32::MIN.saturating_add(2)); + + /// Admits an explicit positive attempt count. + pub const fn new(attempts: NonZeroU32) -> Self { + Self(attempts) + } + + /// The admitted attempt count. + #[must_use] + pub const fn get(self) -> u32 { + self.0.get() + } +} diff --git a/src/adapters/retention/reader_fence.rs b/src/adapters/retention/reader_fence.rs new file mode 100644 index 00000000..9218aff3 --- /dev/null +++ b/src/adapters/retention/reader_fence.rs @@ -0,0 +1,74 @@ +//! This module owns the shared reader fence over one version-two store root. + +use std::io; + +use cap_fs_ext::MetadataExt; +use cap_std::fs::{Dir, File, Metadata}; +use rustix::fs::{FlockOperation, flock}; + +use crate::adapters::filesystem_exact_record; + +const READER_LOCK: &str = "reader.lock"; + +#[cfg(test)] +#[path = "reader_fence_tests.rs"] +mod tests; + +/// A shared kernel lock on `reader.lock` held for one snapshot's lifetime. +/// +/// Collection acquires the store writer authority and then an exclusive lock +/// on the same file, so while any fence is held no published segment, root, +/// or manifest can be deleted. Publication proceeds beside fences because it +/// only adds immutable successors. Dropping the fence releases only the +/// kernel lock; the persistent file is never deleted. +#[must_use] +pub(super) struct ReaderFence { + _file: File, +} + +impl ReaderFence { + /// Acquires the shared fence, waiting while collection holds it exclusively. + /// + /// `reader.lock` must be a regular zero-length file reached without + /// following links; its identity is verified after the open so a swapped + /// entry refuses. + pub(super) fn acquire(root: &Dir) -> io::Result { + Self::acquire_with(root, || Ok(())) + } + + // Keep the acquisition protocol shared with deterministic replacement + // schedules, without ambient hooks or changes to production lock policy. + fn acquire_with( + root: &Dir, + after_verified_open: impl FnOnce() -> io::Result<()>, + ) -> io::Result { + let file = filesystem_exact_record::open_read(root, READER_LOCK)?; + verify(root, &file)?; + after_verified_open()?; + flock(&file, FlockOperation::LockShared)?; + verify(root, &file)?; + Ok(Self { _file: file }) + } +} + +fn verify(root: &Dir, file: &File) -> io::Result<()> { + let handle = file.metadata()?; + let entry = root.symlink_metadata(READER_LOCK)?; + if handle.is_file() + && entry.is_file() + && handle.len() == 0 + && entry.len() == 0 + && identity(&handle) == identity(&entry) + { + Ok(()) + } else { + Err(io::Error::new( + io::ErrorKind::InvalidData, + "reader fence kind, length, or identity disagreed", + )) + } +} + +fn identity(metadata: &Metadata) -> (u64, u64) { + (metadata.dev(), metadata.ino()) +} diff --git a/src/adapters/retention/reader_fence_tests.rs b/src/adapters/retention/reader_fence_tests.rs new file mode 100644 index 00000000..b89fbfb8 --- /dev/null +++ b/src/adapters/retention/reader_fence_tests.rs @@ -0,0 +1,33 @@ +//! This module owns reader-fence identity refusal during acquisition. + +use std::error::Error; +use std::fs; +use std::io; + +use super::ReaderFence; +use crate::adapters::retention::filesystem_retention_test_fixture::migrated_store; +use cap_std::fs::Dir; + +// Size: medium (owned filesystem). Oracle: a fence must lock the same inode +// its directory entry names after acquisition, even when both files are empty. +// Delete when reader fencing is removed or stronger scheduling evidence subsumes this law. +#[test] +fn replacing_the_reader_lock_during_acquisition_refuses_the_fence() -> Result<(), Box> { + let sandbox = migrated_store("reader-fence-replaced-during-acquisition")?; + let root = Dir::open_ambient_dir(sandbox.path(), cap_std::ambient_authority())?; + let path = sandbox.path().join("reader.lock"); + + let error = ReaderFence::acquire_with(&root, || { + fs::remove_file(&path)?; + let _replacement = fs::File::create_new(&path)?; + Ok(()) + }) + .err(); + + assert_eq!( + error.as_ref().map(io::Error::kind), + Some(io::ErrorKind::InvalidData), + "a fence acquired on the displaced inode must refuse: {error:?}" + ); + Ok(()) +} diff --git a/src/adapters/retention/recovery_evidence.rs b/src/adapters/retention/recovery_evidence.rs new file mode 100644 index 00000000..4d85aa32 --- /dev/null +++ b/src/adapters/retention/recovery_evidence.rs @@ -0,0 +1,97 @@ +//! This module owns the complete evidence retention recovery plans from. + +use super::{ + ObservedRetentionState, RetentionHeadStageAssessment, RetentionManifestStageAssessment, + RetentionRootStageAssessment, +}; + +/// Whether an immutable pool already holds the entry a complete stage names. +/// +/// The observation is meaningful only for a `Complete` stage: a truncated or +/// corrupt stage names no canonical entry, and the adapter reports `Absent`. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub enum RetentionPoolEntryObservation { + /// No entry exists under the canonical name. + Absent, + /// The entry is admitted as the exact stage object with matching bytes. + /// Filesystem adapters must also bind the retained stage's device and inode. + Identical, + /// The entry exists with other bytes, another kind, or another stage identity. + Different, +} + +/// The three fixed-stage assessments read at restart. +#[derive(Debug)] +pub struct RetentionStageAssessments<'bytes> { + /// Assessment of `retention/root.next`. + pub root: RetentionRootStageAssessment<'bytes>, + /// Assessment of `retention/manifest.next`. + pub manifest: RetentionManifestStageAssessment<'bytes>, + /// Assessment of `retention/head.next`. + pub head: RetentionHeadStageAssessment<'bytes>, +} + +/// Whether each immutable pool holds the entry its complete stage names. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub struct RetentionPoolObservations { + /// The root pool entry the complete root stage names. + pub root: RetentionPoolEntryObservation, + /// The manifest pool entry the complete manifest stage names. + pub manifest: RetentionPoolEntryObservation, +} + +/// Everything restart observed before planning retention recovery. +/// +/// The evidence is read under exclusive writer authority and performs no +/// mutation; planning over it is pure. +#[derive(Debug)] +pub struct RetentionRecoveryEvidence<'bytes, 'state> { + current: Option<&'state ObservedRetentionState>, + stages: RetentionStageAssessments<'bytes>, + pools: RetentionPoolObservations, +} + +impl<'bytes, 'state> RetentionRecoveryEvidence<'bytes, 'state> { + /// Binds the observed current state, the three stage assessments, and the + /// pool observations for the entries the complete stages name. + #[must_use] + pub const fn new( + current: Option<&'state ObservedRetentionState>, + stages: RetentionStageAssessments<'bytes>, + pools: RetentionPoolObservations, + ) -> Self { + Self { + current, + stages, + pools, + } + } + + pub(super) fn into_parts( + self, + ) -> ( + Option<&'state ObservedRetentionState>, + RetentionStageAssessments<'bytes>, + RetentionPoolObservations, + ) { + (self.current, self.stages, self.pools) + } + + /// The published head and manifest, or `None` when no head is published. + #[must_use] + pub const fn current(&self) -> Option<&'state ObservedRetentionState> { + self.current + } + + /// The three stage assessments. + #[must_use] + pub const fn stages(&self) -> &RetentionStageAssessments<'bytes> { + &self.stages + } + + /// The pool observations for the entries the complete stages name. + #[must_use] + pub const fn pools(&self) -> RetentionPoolObservations { + self.pools + } +} diff --git a/src/adapters/retention/recovery_execution.rs b/src/adapters/retention/recovery_execution.rs new file mode 100644 index 00000000..2c3b0023 --- /dev/null +++ b/src/adapters/retention/recovery_execution.rs @@ -0,0 +1,127 @@ +//! This module owns ordered execution of one retention recovery plan. + +use super::RetentionStorageError; +use std::error::Error; +use std::fmt; + +use super::{ + RetentionRecoveryOutcome, RetentionRecoveryPlan, RetentionRecoveryStep, + RetentionRecoveryStorage, +}; + +/// The complete record of one executed retention recovery plan. +#[derive(Clone, Debug, Eq, PartialEq)] +#[must_use] +pub struct RetentionRecoveryReceipt { + executed: Vec, + outcome: RetentionRecoveryOutcome, +} + +impl RetentionRecoveryReceipt { + /// Every step that executed, in order. + #[must_use] + pub fn executed(&self) -> &[RetentionRecoveryStep] { + &self.executed + } + + /// The state the store is in now. + #[must_use] + pub const fn outcome(&self) -> RetentionRecoveryOutcome { + self.outcome + } +} + +/// One failed recovery step and the steps that completed before it. +#[derive(Debug)] +pub struct RetentionRecoveryError { + step: RetentionRecoveryStep, + executed: Vec, + source: RetentionStorageError, +} + +impl RetentionRecoveryError { + /// Effects reported by the failing capability, independently of preceding completed steps. + /// + /// `None` means that adapter did not report its effects; it does not mean no mutation. + #[must_use] + pub const fn progress(&self) -> Option<&super::RetentionStorageProgress> { + self.source.progress() + } + /// The precise storage failure, without dynamic downcasting. + #[must_use] + pub const fn storage_error(&self) -> &RetentionStorageError { + &self.source + } + + /// The step that failed, possibly after effects. + #[must_use] + pub const fn step(&self) -> RetentionRecoveryStep { + self.step + } + + /// Every step that completed before the failure, in order. This excludes effects of the failing step. + #[must_use] + pub fn executed(&self) -> &[RetentionRecoveryStep] { + &self.executed + } +} + +impl fmt::Display for RetentionRecoveryError { + fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { + write!( + formatter, + "retention recovery step {:?} failed after {} completed step(s)", + self.step, + self.executed.len() + ) + } +} + +impl Error for RetentionRecoveryError { + fn source(&self) -> Option<&(dyn Error + 'static)> { + Some(&self.source) + } +} + +/// Executes `plan` against `storage` in order, stopping at the first error. +/// +/// Each step calls exactly one storage capability. A failed step leaves the +/// completed steps' effects in place, names the step, and returns; the caller +/// re-observes and re-plans rather than continuing from stale evidence. The +/// failing capability may also have effects: inspect its reported progress. +/// Missing progress is uncertainty, never a claim that nothing changed. +/// +/// # Errors +/// +/// Returns [`RetentionRecoveryError`] with the failed step, the completed +/// steps, and the storage's own error as source. +pub fn execute_retention_recovery( + storage: &mut S, + plan: &RetentionRecoveryPlan, +) -> Result { + let mut executed = Vec::with_capacity(plan.steps().len()); + for &step in plan.steps() { + let result = match step { + RetentionRecoveryStep::DiscardHeadStage => storage.discard_head_stage(), + RetentionRecoveryStep::DiscardManifestStage => storage.discard_manifest_stage(), + RetentionRecoveryStep::DiscardRootStage => storage.discard_root_stage(), + RetentionRecoveryStep::LinkRoot => storage.link_root(), + RetentionRecoveryStep::LinkManifest => storage.link_manifest(), + RetentionRecoveryStep::FinalizeHead => storage.finalize_head(), + RetentionRecoveryStep::RemoveRootStage => storage.remove_root_stage(), + RetentionRecoveryStep::RemoveManifestStage => storage.remove_manifest_stage(), + }; + if let Err(source) = result { + return Err(RetentionRecoveryError { + step, + executed, + source, + }); + } + executed.push(step); + } + Ok(RetentionRecoveryReceipt { + executed, + outcome: plan.outcome(), + }) +} diff --git a/src/adapters/retention/recovery_execution_tests.rs b/src/adapters/retention/recovery_execution_tests.rs new file mode 100644 index 00000000..1d6fe9cd --- /dev/null +++ b/src/adapters/retention/recovery_execution_tests.rs @@ -0,0 +1,197 @@ +//! Retention recovery execution laws against a recording fake storage. + +use super::RetentionStorageError; +use std::error::Error; +use std::io; + +use super::{ + RetentionRecoveryError, RetentionRecoveryOutcome, RetentionRecoveryPlan, + RetentionRecoveryStep as Step, RetentionRecoveryStorage, execute_retention_recovery, +}; + +#[derive(Default)] +struct Recording { + calls: Vec, + refuse_at: Option, + failure: Option, +} + +impl Recording { + fn record(&mut self, step: Step) -> Result<(), RetentionStorageError> { + if self.refuse_at == Some(step) { + return Err(self + .failure + .take() + .unwrap_or_else(|| io::Error::other("injected refusal").into())); + } + self.calls.push(step); + Ok(()) + } +} + +impl RetentionRecoveryStorage for Recording { + fn discard_head_stage(&mut self) -> Result<(), RetentionStorageError> { + self.record(Step::DiscardHeadStage) + } + fn discard_manifest_stage(&mut self) -> Result<(), RetentionStorageError> { + self.record(Step::DiscardManifestStage) + } + fn discard_root_stage(&mut self) -> Result<(), RetentionStorageError> { + self.record(Step::DiscardRootStage) + } + fn link_root(&mut self) -> Result<(), RetentionStorageError> { + self.record(Step::LinkRoot) + } + fn link_manifest(&mut self) -> Result<(), RetentionStorageError> { + self.record(Step::LinkManifest) + } + fn finalize_head(&mut self) -> Result<(), RetentionStorageError> { + self.record(Step::FinalizeHead) + } + fn remove_root_stage(&mut self) -> Result<(), RetentionStorageError> { + self.record(Step::RemoveRootStage) + } + fn remove_manifest_stage(&mut self) -> Result<(), RetentionStorageError> { + self.record(Step::RemoveManifestStage) + } +} + +const FINALIZE: [Step; 3] = [ + Step::FinalizeHead, + Step::RemoveRootStage, + Step::RemoveManifestStage, +]; + +// Size: small. Oracle: the public executor stops at its first refused storage capability. +// The port records successful effects, not executor internals; none may follow failed finalization. +// Delete when a stronger recovery-boundary law subsumes first-step refusal and its empty receipt. +#[test] +fn refused_finalization_reports_no_completed_steps() -> Result<(), Box> { + let plan = RetentionRecoveryPlan::new(FINALIZE.to_vec(), RetentionRecoveryOutcome::Committed); + let mut storage = Recording { + calls: Vec::new(), + refuse_at: Some(Step::FinalizeHead), + ..Recording::default() + }; + + let error = execute_retention_recovery(&mut storage, &plan) + .err() + .ok_or("refused finalization was reported as success")?; + + assert_eq!( + error.step(), + Step::FinalizeHead, + "name the first refused capability" + ); + assert!( + error.executed().is_empty(), + "no operation completed before finalization refused" + ); + assert!( + error.progress().is_none(), + "unreported effects must remain unknown" + ); + assert!( + storage.calls.is_empty(), + "a finalization refusal must prevent cleanup effects" + ); + Ok(()) +} + +#[test] +fn every_step_calls_exactly_its_capability_in_plan_order() -> Result<(), Box> { + let all = [ + Step::DiscardHeadStage, + Step::DiscardManifestStage, + Step::DiscardRootStage, + Step::LinkRoot, + Step::LinkManifest, + Step::FinalizeHead, + Step::RemoveRootStage, + Step::RemoveManifestStage, + ]; + let plan = RetentionRecoveryPlan::new(all.to_vec(), RetentionRecoveryOutcome::Committed); + let mut storage = Recording::default(); + + let receipt = execute_retention_recovery(&mut storage, &plan)?; + + assert_eq!(storage.calls, all); + assert_eq!(receipt.executed(), all); + assert_eq!(receipt.outcome(), RetentionRecoveryOutcome::Committed); + Ok(()) +} + +#[test] +fn an_empty_plan_touches_nothing_and_reports_its_outcome() -> Result<(), Box> { + let plan = RetentionRecoveryPlan::new(Vec::new(), RetentionRecoveryOutcome::Clean); + let mut storage = Recording::default(); + + let receipt = execute_retention_recovery(&mut storage, &plan)?; + + assert!(storage.calls.is_empty()); + assert!(receipt.executed().is_empty()); + assert_eq!(receipt.outcome(), RetentionRecoveryOutcome::Clean); + Ok(()) +} + +#[test] +fn a_refused_step_stops_execution_and_names_the_completed_prefix() -> Result<(), Box> { + let plan = RetentionRecoveryPlan::new(FINALIZE.to_vec(), RetentionRecoveryOutcome::Committed); + let mut storage = Recording { + calls: Vec::new(), + refuse_at: Some(Step::RemoveRootStage), + ..Recording::default() + }; + + let error: RetentionRecoveryError = execute_retention_recovery(&mut storage, &plan) + .err() + .ok_or("an injected refusal was reported as success")?; + + assert_eq!(error.step(), Step::RemoveRootStage); + assert_eq!(error.executed(), [Step::FinalizeHead]); + assert_eq!(storage.calls, [Step::FinalizeHead]); + assert!(error.source().is_some()); + Ok(()) +} + +// Size: small. Oracle: an uncertain effect is not promoted to a known effect or erased. +// Port-level fault simulation; real namespace/synchronization effects are tested separately. +// Delete when a stronger public execution law subsumes uncertain cause propagation and stop-on-error. +#[test] +fn uncertain_failed_effect_survives_execution_without_later_cleanup() -> Result<(), Box> +{ + use super::{RetentionNamespaceEffect as Effect, RetentionStorageBoundary as Boundary}; + let plan = RetentionRecoveryPlan::new(FINALIZE.to_vec(), RetentionRecoveryOutcome::Committed); + let failure = RetentionStorageError::from(io::Error::from_raw_os_error( + rustix::io::Errno::IO.raw_os_error(), + )) + .at(Boundary::HeadRename) + .uncertain(Effect::HeadReplaced); + let mut storage = Recording { + refuse_at: Some(Step::FinalizeHead), + failure: Some(failure), + ..Recording::default() + }; + let error = execute_retention_recovery(&mut storage, &plan) + .err() + .ok_or("uncertain failure accepted")?; + let progress = error.progress().ok_or("uncertain effect omitted")?; + assert_eq!(progress.boundary(), Boundary::HeadRename); + assert!( + progress.known_effects().is_empty(), + "an uncertain rename is not a known rename" + ); + assert_eq!(progress.uncertain_effect(), Some(Effect::HeadReplaced)); + assert!(storage.calls.is_empty(), "no later capability may execute"); + let RetentionStorageError::Operation { source, .. } = error.storage_error() else { + return Err("operation cause absent".into()); + }; + let RetentionStorageError::Io { source } = source.as_ref() else { + return Err("original I/O cause absent".into()); + }; + assert_eq!( + source.raw_os_error(), + Some(rustix::io::Errno::IO.raw_os_error()) + ); + Ok(()) +} diff --git a/src/adapters/retention/recovery_head_length_tests.rs b/src/adapters/retention/recovery_head_length_tests.rs new file mode 100644 index 00000000..630a495d --- /dev/null +++ b/src/adapters/retention/recovery_head_length_tests.rs @@ -0,0 +1,95 @@ +//! These laws own the planner's exact head-to-manifest length contract. + +use std::error::Error; + +use super::filesystem_retention_test_fixture::{MANIFEST_HEX, ROOT_HEX, fixture}; +use super::{ + AdmittedRetentionManifest, CanonicalRetentionHead, CanonicalRetentionManifest, + RetentionPoolEntryObservation as Pool, RetentionPoolObservations, RetentionRecoveryEvidence, + RetentionRecoveryOutcome, RetentionRecoveryRefusal, RetentionStageAssessments, + assess_head_stage, assess_manifest_stage, assess_root_stage, plan_retention_recovery, +}; +use crate::{RetentionHead, RetentionManifest, RetentionManifestLength}; + +// Size: small. Oracle: every admitted head length must equal the manifest bytes. +// Exhaustive finite input sweep; the first failing length is a minimal reproducer. +// Delete only when a stronger public planner property subsumes this contract. +#[test] +fn recovery_binds_head_length_across_the_complete_canonical_length_domain() +-> Result<(), Box> { + let root = fixture(ROOT_HEX)?; + let manifest_bytes = fixture(MANIFEST_HEX)?; + let manifest = AdmittedRetentionManifest::decode(&manifest_bytes)?; + let actual = u64::try_from(manifest_bytes.len())?; + let (minimum, entry_width) = encoded_length_basis(manifest.manifest())?; + for entries in 0..=RetentionManifest::MAXIMUM_ENTRY_COUNT { + let length = u64::from(entries) + .checked_mul(entry_width) + .and_then(|bytes| minimum.checked_add(bytes)) + .ok_or("canonical length sweep overflowed")?; + let head = RetentionHead::new( + manifest.manifest().generation(), + RetentionManifestLength::new(length)?, + manifest.digest(), + manifest.manifest().predecessor(), + )?; + let encoded = CanonicalRetentionHead::from_head(&head); + let result = plan_retention_recovery(RetentionRecoveryEvidence::new( + None, + RetentionStageAssessments { + root: assess_root_stage(Some(&root)), + manifest: assess_manifest_stage(Some(&manifest_bytes)), + head: assess_head_stage(Some(encoded.encoded())), + }, + RetentionPoolObservations { + root: Pool::Identical, + manifest: Pool::Identical, + }, + )); + if length == actual { + let plan = result?; + assert_eq!( + plan.outcome(), + RetentionRecoveryOutcome::Committed, + "exact manifest length {length} must remain recoverable" + ); + } else { + assert!( + matches!( + result, + Err(RetentionRecoveryRefusal::HeadStageNamesOtherManifest) + ), + "head length {length} must refuse manifest length {actual}: {result:?}" + ); + } + } + Ok(()) +} + +/// Derives generated input lengths through the public encoding boundary. +/// The expected planner result still compares against the independent fixture bytes. +fn encoded_length_basis(manifest: &RetentionManifest) -> Result<(u64, u64), Box> { + let entry = manifest + .entries() + .first() + .copied() + .ok_or("fixture has no entry")?; + let empty = RetentionManifest::new(manifest.generation(), manifest.predecessor(), vec![])?; + let single = + RetentionManifest::new(manifest.generation(), manifest.predecessor(), vec![entry])?; + let minimum = u64::try_from( + CanonicalRetentionManifest::from_manifest(&empty)? + .encoded() + .len(), + )?; + let single_length = u64::try_from( + CanonicalRetentionManifest::from_manifest(&single)? + .encoded() + .len(), + )?; + let width = single_length + .checked_sub(minimum) + .filter(|width| *width > 0) + .ok_or("one manifest entry must increase encoded length")?; + Ok((minimum, width)) +} diff --git a/src/adapters/retention/recovery_manifest_entries.rs b/src/adapters/retention/recovery_manifest_entries.rs new file mode 100644 index 00000000..63393bc8 --- /dev/null +++ b/src/adapters/retention/recovery_manifest_entries.rs @@ -0,0 +1,25 @@ +//! This module owns preservation of unrelated namespaces in recovery successors. + +use super::{AdmittedRetentionManifest, AdmittedRetentionRoot, ObservedRetentionState}; + +/// Requires every entry outside the staged root namespace to remain exact. +/// +/// Both admitted slices are canonically ordered and duplicate-free, so stream +/// equality rejects omissions, additions, and coordinate changes without +/// allocation. An absent current state permits only the staged namespace. +pub(super) fn preserves_unrelated( + current: Option<&ObservedRetentionState>, + manifest: &AdmittedRetentionManifest<'_>, + root: &AdmittedRetentionRoot<'_>, +) -> bool { + let namespace = root.root().namespace().digest(); + let before = current.map_or(&[][..], |state| state.manifest().entries()); + before + .iter() + .filter(|entry| entry.namespace() != namespace) + .eq(manifest + .manifest() + .entries() + .iter() + .filter(|entry| entry.namespace() != namespace)) +} diff --git a/src/adapters/retention/recovery_plan.rs b/src/adapters/retention/recovery_plan.rs new file mode 100644 index 00000000..5a96fa27 --- /dev/null +++ b/src/adapters/retention/recovery_plan.rs @@ -0,0 +1,70 @@ +//! This module owns the typed retention recovery plan and its outcome. + +/// One ordered recovery effect. Each maps to exactly one storage capability. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub enum RetentionRecoveryStep { + /// Reserved head disposition step; never emitted by the current planner. + DiscardHeadStage, + /// Reserved manifest disposition step; never emitted by the current planner. + DiscardManifestStage, + /// Reserved root disposition step; never emitted by the current planner. + DiscardRootStage, + /// Admit the namespace directory and link the complete root stage into it. + LinkRoot, + /// Link the complete manifest stage into the manifest pool. + LinkManifest, + /// Replace `retention/HEAD` with the complete head stage and synchronize. + FinalizeHead, + /// Remove the retained root stage after its pool link is proven. + RemoveRootStage, + /// Remove the retained manifest stage after its pool link is proven. + RemoveManifestStage, +} + +/// The state recovery leaves once every step has executed. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub enum RetentionRecoveryOutcome { + /// No stage remains; forward publication may proceed. + Clean, + /// The staged generation is (or was already) the published head and its + /// stages are removed; forward publication may proceed. + Committed, + /// Complete stages remain linked and retained as valid orphans. They are + /// recovery-protected until explicit disposition; forward publication + /// refuses meanwhile. + Protected { + /// `root.next` remains retained. + root_stage: bool, + /// `manifest.next` remains retained. + manifest_stage: bool, + }, +} + +/// The ordered effects recovery must execute and the state they produce. +#[derive(Clone, Debug, Eq, PartialEq)] +#[must_use] +pub struct RetentionRecoveryPlan { + steps: Vec, + outcome: RetentionRecoveryOutcome, +} + +impl RetentionRecoveryPlan { + pub(super) const fn new( + steps: Vec, + outcome: RetentionRecoveryOutcome, + ) -> Self { + Self { steps, outcome } + } + + /// The effects in execution order; empty when nothing must change. + #[must_use] + pub fn steps(&self) -> &[RetentionRecoveryStep] { + &self.steps + } + + /// The state the store is in after every step executes. + #[must_use] + pub const fn outcome(&self) -> RetentionRecoveryOutcome { + self.outcome + } +} diff --git a/src/adapters/retention/recovery_planner.rs b/src/adapters/retention/recovery_planner.rs new file mode 100644 index 00000000..b6fb24cd --- /dev/null +++ b/src/adapters/retention/recovery_planner.rs @@ -0,0 +1,308 @@ +//! This module owns pure planning of retention recovery from restart evidence. + +use super::{ + AdmittedRetentionManifest, AdmittedRetentionRoot, ChecksummedRetentionHead, + ObservedRetentionState, RetentionFixedStage, RetentionPool, + RetentionPoolEntryObservation as Pool, RetentionPoolObservations, RetentionRecoveryEvidence, + RetentionRecoveryOutcome, RetentionRecoveryPlan, RetentionRecoveryRefusal as Refusal, + RetentionRecoveryStep as Step, RetentionStageAssessment as Stage, +}; +use crate::{LivenessGeneration, RootGeneration}; + +type Current<'state> = Option<&'state ObservedRetentionState>; + +/// Plans retention recovery from complete restart evidence. +/// +/// The call performs no I/O. It applies the documented classification: a +/// incomplete stage requires explicit disposition before any effects; a complete +/// root or manifest stage is linked into its pool and retained as a +/// recovery-protected orphan; a complete head stage naming the staged +/// manifest is finalized and both retained stages are removed; a staged +/// generation the published head already names is cleaned up. Any other +/// combination is unrecoverable ambiguity and refuses before any effect. +/// +/// # Errors +/// +/// Returns [`RetentionRecoveryRefusal`](super::RetentionRecoveryRefusal) naming +/// the exact ambiguity. +pub fn plan_retention_recovery( + evidence: RetentionRecoveryEvidence<'_, '_>, +) -> Result { + let (current, stages, pools) = evidence.into_parts(); + // Preserve known corruption even when another stage is merely incomplete. + if let Stage::Corrupt(source) = stages.head { + return Err(Refusal::corrupt_head(source)); + } + if let Stage::Corrupt(source) = stages.manifest { + return Err(Refusal::corrupt_manifest(source)); + } + if let Stage::Corrupt(source) = stages.root { + return Err(Refusal::corrupt_root(source)); + } + let steps = Vec::new(); + let head = match stages.head { + Stage::Absent => None, + Stage::Truncated { expected, observed } => { + return Err(Refusal::IncompleteStageRequiresDisposition { + stage: RetentionFixedStage::Head, + expected, + observed, + }); + } + Stage::Corrupt(source) => return Err(Refusal::corrupt_head(source)), + Stage::Complete(head) => Some(head), + }; + let manifest = match stages.manifest { + Stage::Absent => None, + Stage::Truncated { expected, observed } => { + return Err(Refusal::IncompleteStageRequiresDisposition { + stage: RetentionFixedStage::Manifest, + expected, + observed, + }); + } + Stage::Corrupt(source) => return Err(Refusal::corrupt_manifest(source)), + Stage::Complete(manifest) => Some(manifest), + }; + let root = match stages.root { + Stage::Absent => None, + Stage::Truncated { expected, observed } => { + return Err(Refusal::IncompleteStageRequiresDisposition { + stage: RetentionFixedStage::Root, + expected, + observed, + }); + } + Stage::Corrupt(source) => return Err(Refusal::corrupt_root(source)), + Stage::Complete(root) => Some(root), + }; + if root.is_some() && pools.root == Pool::Different { + return Err(Refusal::PoolEntryDiffers { + pool: RetentionPool::Roots, + }); + } + if manifest.is_some() && pools.manifest == Pool::Different { + return Err(Refusal::PoolEntryDiffers { + pool: RetentionPool::Manifests, + }); + } + match (head, manifest, root) { + (Some(head), Some(manifest), Some(root)) => finalize_head( + current, + CompleteStages { + head: &head, + manifest: &manifest, + root: &root, + }, + pools, + steps, + ), + (Some(_), None, _) => Err(Refusal::HeadStageWithoutManifestStage), + (Some(_), Some(_), None) => Err(Refusal::ManifestStageWithoutRootStage), + (None, Some(manifest), root) => { + plan_manifest(current, &manifest, root.as_ref(), pools, steps) + } + (None, None, Some(root)) => plan_root(current, &root, pools.root, steps), + (None, None, None) => Ok(RetentionRecoveryPlan::new( + steps, + RetentionRecoveryOutcome::Clean, + )), + } +} + +/// The three complete stages a head finalization is planned from. +#[derive(Clone, Copy)] +struct CompleteStages<'a, 'bytes> { + head: &'a ChecksummedRetentionHead<'bytes>, + manifest: &'a AdmittedRetentionManifest<'bytes>, + root: &'a AdmittedRetentionRoot<'bytes>, +} + +fn finalize_head( + current: Current<'_>, + stages: CompleteStages<'_, '_>, + pools: RetentionPoolObservations, + mut steps: Vec, +) -> Result { + let CompleteStages { + head, + manifest, + root, + } = stages; + let head = head.head(); + if head.manifest_digest() != manifest.digest() + || head.generation() != manifest.manifest().generation() + || head.manifest_length().get() + != u64::try_from(manifest.encoded().len()) + .map_err(|_| Refusal::HeadStageNamesOtherManifest)? + { + return Err(Refusal::HeadStageNamesOtherManifest); + } + if head.predecessor() != manifest.manifest().predecessor() { + return Err(Refusal::HeadPredecessorMismatch); + } + if !manifest_names_root(manifest, root) { + return Err(Refusal::ManifestStageNamesOtherRoot); + } + if !is_committed(current, manifest) && !root_succeeds(current, root) { + return Err(Refusal::RootNotSuccessor); + } + if !is_committed(current, manifest) + && !super::recovery_manifest_entries::preserves_unrelated(current, manifest, root) + { + return Err(Refusal::ManifestNotSuccessor); + } + if pools.root != Pool::Identical { + return Err(Refusal::RootNotLinkedBeforeHead); + } + if pools.manifest != Pool::Identical { + return Err(Refusal::ManifestNotLinkedBeforeHead); + } + if !is_committed(current, manifest) && !manifest_succeeds(current, manifest) { + return Err(Refusal::HeadPredecessorMismatch); + } + steps.extend([ + Step::FinalizeHead, + Step::RemoveRootStage, + Step::RemoveManifestStage, + ]); + Ok(RetentionRecoveryPlan::new( + steps, + RetentionRecoveryOutcome::Committed, + )) +} + +fn plan_manifest( + current: Current<'_>, + manifest: &AdmittedRetentionManifest<'_>, + root: Option<&AdmittedRetentionRoot<'_>>, + pools: RetentionPoolObservations, + mut steps: Vec, +) -> Result { + if is_committed(current, manifest) { + if let Some(root) = root { + if !manifest_names_root(manifest, root) { + return Err(Refusal::ManifestStageNamesOtherRoot); + } + if pools.root != Pool::Identical { + return Err(Refusal::RootNotLinkedBeforeHead); + } + steps.push(Step::RemoveRootStage); + } + steps.push(Step::RemoveManifestStage); + return Ok(RetentionRecoveryPlan::new( + steps, + RetentionRecoveryOutcome::Committed, + )); + } + let root = root.ok_or(Refusal::ManifestStageWithoutRootStage)?; + if !manifest_names_root(manifest, root) { + return Err(Refusal::ManifestStageNamesOtherRoot); + } + if !manifest_succeeds(current, manifest) { + return Err(Refusal::ManifestNotSuccessor); + } + if !super::recovery_manifest_entries::preserves_unrelated(current, manifest, root) { + return Err(Refusal::ManifestNotSuccessor); + } + if !root_succeeds(current, root) { + return Err(Refusal::RootNotSuccessor); + } + if pools.root == Pool::Absent { + steps.push(Step::LinkRoot); + } + if pools.manifest == Pool::Absent { + steps.push(Step::LinkManifest); + } + Ok(RetentionRecoveryPlan::new( + steps, + RetentionRecoveryOutcome::Protected { + root_stage: true, + manifest_stage: true, + }, + )) +} + +fn plan_root( + current: Current<'_>, + root: &AdmittedRetentionRoot<'_>, + root_pool: Pool, + mut steps: Vec, +) -> Result { + if !root_succeeds(current, root) { + return Err(Refusal::RootNotSuccessor); + } + if root_pool == Pool::Absent { + steps.push(Step::LinkRoot); + } + Ok(RetentionRecoveryPlan::new( + steps, + RetentionRecoveryOutcome::Protected { + root_stage: true, + manifest_stage: false, + }, + )) +} + +/// Whether the published head already names the staged manifest. +fn is_committed(current: Current<'_>, manifest: &AdmittedRetentionManifest<'_>) -> bool { + current.is_some_and(|current| { + current.head().manifest_digest() == manifest.digest() + && current.head().generation() == manifest.manifest().generation() + }) +} + +fn manifest_names_root( + manifest: &AdmittedRetentionManifest<'_>, + root: &AdmittedRetentionRoot<'_>, +) -> bool { + let namespace = root.root().namespace().digest(); + let entries = manifest.manifest().entries(); + entries + .binary_search_by_key(&namespace, |entry| entry.namespace()) + .ok() + .and_then(|index| entries.get(index)) + .is_some_and(|entry| { + entry.root_generation() == root.root().generation() + && entry.root_digest() == root.digest() + }) +} + +fn manifest_succeeds(current: Current<'_>, manifest: &AdmittedRetentionManifest<'_>) -> bool { + let manifest = manifest.manifest(); + current.map_or_else( + || manifest.predecessor().is_none() && manifest.generation() == LivenessGeneration::INITIAL, + |current| { + manifest.predecessor() == Some(current.head().manifest_digest()) + && current + .head() + .generation() + .successor() + .is_ok_and(|successor| successor == manifest.generation()) + }, + ) +} + +fn root_succeeds(current: Current<'_>, root: &AdmittedRetentionRoot<'_>) -> bool { + let namespace = root.root().namespace().digest(); + let entry = current.and_then(|current| { + let entries = current.manifest().entries(); + entries + .binary_search_by_key(&namespace, |entry| entry.namespace()) + .ok() + .and_then(|index| entries.get(index).copied()) + }); + entry.map_or_else( + || { + root.root().predecessor().is_none() + && root.root().generation() == RootGeneration::INITIAL + }, + |entry| { + root.root().predecessor() == Some(entry.root_digest()) + && entry + .root_generation() + .successor() + .is_ok_and(|successor| successor == root.root().generation()) + }, + ) +} diff --git a/src/adapters/retention/recovery_planner_tests.rs b/src/adapters/retention/recovery_planner_tests.rs new file mode 100644 index 00000000..fe510d00 --- /dev/null +++ b/src/adapters/retention/recovery_planner_tests.rs @@ -0,0 +1,374 @@ +//! Retention recovery planning laws over the golden version-two records. + +use std::error::Error; + +use super::filesystem_retention_test_fixture::{ + HEAD_HEX, MANIFEST_HEX, ROOT_HEX, fixture, initial_preparation, successor_preparation, +}; +use super::{ + AdmittedRetentionManifest, AdmittedRetentionRoot, ObservedRetentionState, RetentionFixedStage, + RetentionPool, RetentionPoolEntryObservation as Pool, RetentionPoolObservations, + RetentionRecoveryEvidence, RetentionRecoveryOutcome as Outcome, RetentionRecoveryRefusal, + RetentionRecoveryStep as Step, RetentionStageAssessments, assess_head_stage, + assess_manifest_stage, assess_root_stage, plan_retention_recovery, +}; + +struct Records { + root: Vec, + manifest: Vec, + head: Vec, +} + +fn generation_one() -> Result> { + Ok(Records { + root: fixture(ROOT_HEX)?, + manifest: fixture(MANIFEST_HEX)?, + head: fixture(HEAD_HEX)?, + }) +} + +type StageBytes<'b> = (Option<&'b [u8]>, Option<&'b [u8]>, Option<&'b [u8]>); + +fn evidence<'b, 's>( + current: Option<&'s ObservedRetentionState>, + (root, manifest, head): StageBytes<'b>, + (root_pool, manifest_pool): (Pool, Pool), +) -> RetentionRecoveryEvidence<'b, 's> { + RetentionRecoveryEvidence::new( + current, + RetentionStageAssessments { + root: assess_root_stage(root), + manifest: assess_manifest_stage(manifest), + head: assess_head_stage(head), + }, + RetentionPoolObservations { + root: root_pool, + manifest: manifest_pool, + }, + ) +} + +fn corrupt(bytes: &[u8]) -> Vec { + let mut bytes = bytes.to_vec(); + if let Some(last) = bytes.last_mut() { + *last ^= 0x01; + } + bytes +} + +#[test] +fn a_clean_store_needs_nothing() -> Result<(), Box> { + let plan = plan_retention_recovery(evidence( + None, + (None, None, None), + (Pool::Absent, Pool::Absent), + ))?; + assert!(plan.steps().is_empty()); + assert_eq!(plan.outcome(), Outcome::Clean); + Ok(()) +} + +#[test] +fn a_truncated_root_stage_requires_disposition() -> Result<(), Box> { + let records = generation_one()?; + let partial = records + .root + .get(..100) + .ok_or("root fixture shorter than 100 bytes")?; + let result = plan_retention_recovery(evidence( + None, + (Some(partial), None, None), + (Pool::Absent, Pool::Absent), + )); + assert!( + matches!( + result, + Err( + RetentionRecoveryRefusal::IncompleteStageRequiresDisposition { + stage: RetentionFixedStage::Root, + expected: 192, + observed: 100 + } + ) + ), + "incomplete root requires disposition: {result:?}" + ); + Ok(()) +} + +#[test] +fn a_truncated_stage_with_a_later_effect_refuses() -> Result<(), Box> { + let records = generation_one()?; + let partial = records + .manifest + .get(..100) + .ok_or("manifest fixture shorter than 100 bytes")?; + let error = plan_retention_recovery(evidence( + None, + (Some(&records.root), Some(partial), None), + (Pool::Identical, Pool::Identical), + )) + .err() + .ok_or("a truncated manifest with a pool link was discarded")?; + assert!(matches!( + error, + RetentionRecoveryRefusal::IncompleteStageRequiresDisposition { + stage: RetentionFixedStage::Manifest, + expected: 160, + observed: 100 + } + )); + Ok(()) +} + +#[test] +fn a_corrupt_stage_refuses_with_its_decode_error() -> Result<(), Box> { + let records = generation_one()?; + let corrupt_root = corrupt(&records.root); + let error = plan_retention_recovery(evidence( + None, + (Some(&corrupt_root), None, None), + (Pool::Absent, Pool::Absent), + )) + .err() + .ok_or("a corrupt root stage was planned")?; + assert!(matches!( + error, + RetentionRecoveryRefusal::StageCorrupt { + stage: RetentionFixedStage::Root, + .. + } + )); + assert!(error.source().is_some()); + Ok(()) +} + +#[test] +fn a_complete_root_stage_is_linked_and_protected() -> Result<(), Box> { + let records = generation_one()?; + let unlinked = plan_retention_recovery(evidence( + None, + (Some(&records.root), None, None), + (Pool::Absent, Pool::Absent), + ))?; + let linked = plan_retention_recovery(evidence( + None, + (Some(&records.root), None, None), + (Pool::Identical, Pool::Absent), + ))?; + let differs = plan_retention_recovery(evidence( + None, + (Some(&records.root), None, None), + (Pool::Different, Pool::Absent), + )) + .err() + .ok_or("a conflicting root pool entry was planned over")?; + assert_eq!(unlinked.steps(), [Step::LinkRoot]); + assert!(linked.steps().is_empty()); + for plan in [&unlinked, &linked] { + assert_eq!( + plan.outcome(), + Outcome::Protected { + root_stage: true, + manifest_stage: false + } + ); + } + assert!(matches!( + differs, + RetentionRecoveryRefusal::PoolEntryDiffers { + pool: RetentionPool::Roots + } + )); + Ok(()) +} + +#[test] +fn complete_root_and_manifest_stages_are_linked_and_protected() -> Result<(), Box> { + let records = generation_one()?; + let plan = plan_retention_recovery(evidence( + None, + (Some(&records.root), Some(&records.manifest), None), + (Pool::Absent, Pool::Absent), + ))?; + assert_eq!(plan.steps(), [Step::LinkRoot, Step::LinkManifest]); + assert_eq!( + plan.outcome(), + Outcome::Protected { + root_stage: true, + manifest_stage: true + } + ); + Ok(()) +} + +#[test] +fn a_complete_head_stage_over_linked_stages_is_finalized() -> Result<(), Box> { + let records = generation_one()?; + let plan = plan_retention_recovery(evidence( + None, + ( + Some(&records.root), + Some(&records.manifest), + Some(&records.head), + ), + (Pool::Identical, Pool::Identical), + ))?; + assert_eq!( + plan.steps(), + [ + Step::FinalizeHead, + Step::RemoveRootStage, + Step::RemoveManifestStage + ] + ); + assert_eq!(plan.outcome(), Outcome::Committed); + let unlinked = plan_retention_recovery(evidence( + None, + ( + Some(&records.root), + Some(&records.manifest), + Some(&records.head), + ), + (Pool::Absent, Pool::Identical), + )) + .err() + .ok_or("a head stage over an unlinked root was finalized")?; + assert!(matches!( + unlinked, + RetentionRecoveryRefusal::RootNotLinkedBeforeHead + )); + Ok(()) +} + +#[test] +fn a_truncated_head_stage_requires_disposition_despite_complete_earlier_evidence() +-> Result<(), Box> { + let records = generation_one()?; + let partial = records + .head + .get(..40) + .ok_or("head fixture shorter than 40 bytes")?; + let result = plan_retention_recovery(evidence( + None, + (Some(&records.root), Some(&records.manifest), Some(partial)), + (Pool::Identical, Pool::Identical), + )); + assert!( + matches!( + result, + Err( + RetentionRecoveryRefusal::IncompleteStageRequiresDisposition { + stage: RetentionFixedStage::Head, + expected: 144, + observed: 40 + } + ) + ), + "incomplete head requires disposition: {result:?}" + ); + Ok(()) +} + +#[test] +fn a_head_stage_without_the_staged_manifest_refuses() -> Result<(), Box> { + let records = generation_one()?; + let error = plan_retention_recovery(evidence( + None, + (Some(&records.root), None, Some(&records.head)), + (Pool::Identical, Pool::Absent), + )) + .err() + .ok_or("a head stage without a manifest stage was finalized")?; + assert!(matches!( + error, + RetentionRecoveryRefusal::HeadStageWithoutManifestStage + )); + Ok(()) +} + +#[test] +fn stages_the_published_head_already_names_are_cleaned_up() -> Result<(), Box> { + let records = generation_one()?; + let current = ObservedRetentionState::for_tests(&records.head, &records.manifest)?; + let both = plan_retention_recovery(evidence( + Some(¤t), + (Some(&records.root), Some(&records.manifest), None), + (Pool::Identical, Pool::Identical), + ))?; + let manifest_only = plan_retention_recovery(evidence( + Some(¤t), + (None, Some(&records.manifest), None), + (Pool::Absent, Pool::Identical), + ))?; + let head_too = plan_retention_recovery(evidence( + Some(¤t), + ( + Some(&records.root), + Some(&records.manifest), + Some(&records.head), + ), + (Pool::Identical, Pool::Identical), + ))?; + assert_eq!( + both.steps(), + [Step::RemoveRootStage, Step::RemoveManifestStage] + ); + assert_eq!(manifest_only.steps(), [Step::RemoveManifestStage]); + assert_eq!( + head_too.steps(), + [ + Step::FinalizeHead, + Step::RemoveRootStage, + Step::RemoveManifestStage + ] + ); + for plan in [&both, &manifest_only, &head_too] { + assert_eq!(plan.outcome(), Outcome::Committed); + } + Ok(()) +} + +#[test] +fn a_successor_generation_is_planned_against_the_published_state() -> Result<(), Box> { + let records = generation_one()?; + let current = ObservedRetentionState::for_tests(&records.head, &records.manifest)?; + let current_root = AdmittedRetentionRoot::decode(&records.root)?; + let current_manifest = AdmittedRetentionManifest::decode(&records.manifest)?; + let candidate = super::filesystem_retention_test_fixture::successor_root(¤t_root)?; + let preparation = successor_preparation(¤t_root, ¤t_manifest, candidate.encoded())?; + let publication = preparation + .publication() + .ok_or("successor preparation carries no publication")?; + let manifest = publication.manifest().encoded().to_vec(); + let head = publication.head().encoded().to_vec(); + + let finalize = plan_retention_recovery(evidence( + Some(¤t), + (Some(candidate.encoded()), Some(&manifest), Some(&head)), + (Pool::Identical, Pool::Identical), + ))?; + let stale = plan_retention_recovery(evidence( + None, + (Some(candidate.encoded()), Some(&manifest), None), + (Pool::Absent, Pool::Absent), + )) + .err() + .ok_or("a successor manifest over an absent head was planned")?; + + assert_eq!( + finalize.steps(), + [ + Step::FinalizeHead, + Step::RemoveRootStage, + Step::RemoveManifestStage + ] + ); + assert_eq!(finalize.outcome(), Outcome::Committed); + assert!(matches!( + stale, + RetentionRecoveryRefusal::ManifestNotSuccessor + )); + drop(initial_preparation(&records.root)?); + Ok(()) +} diff --git a/src/adapters/retention/recovery_refusal.rs b/src/adapters/retention/recovery_refusal.rs new file mode 100644 index 00000000..b33cf577 --- /dev/null +++ b/src/adapters/retention/recovery_refusal.rs @@ -0,0 +1,207 @@ +//! This module owns the typed refusals of retention recovery planning. + +use std::error::Error; +use std::fmt; + +use super::{RetentionHeadDecodeError, RetentionManifestDecodeError, RetentionRootDecodeError}; + +/// One of the three fixed retention stage names. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub enum RetentionFixedStage { + /// `retention/root.next`. + Root, + /// `retention/manifest.next`. + Manifest, + /// `retention/head.next`. + Head, +} + +/// One of the two immutable retention pools. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub enum RetentionPool { + /// `retention/roots/`. + Roots, + /// `retention/manifests`. + Manifests, +} + +/// Why the observed stages are unrecoverable ambiguity rather than a plan. +/// +/// Every variant leaves the store untouched: recovery refuses before any +/// effect, and the evidence stays for explicit disposition. +#[derive(Debug)] +#[non_exhaustive] +pub enum RetentionRecoveryRefusal { + /// A retained stage is incomplete; automatic disposition is not supported. + IncompleteStageRequiresDisposition { + /// The stage requiring explicit disposition. + stage: RetentionFixedStage, + /// The minimum boundary currently required by decoding, not a completion proof. + expected: usize, + /// Bytes actually observed. + observed: usize, + }, + /// A stage is complete enough to judge and fails a canonical law. + StageCorrupt { + /// The stage that failed. + stage: RetentionFixedStage, + /// The exact decode refusal. + source: Box, + }, + /// A truncated stage has a later-ordered effect, so it is not pre-effect. + TruncatedStageWithLaterEffect { + /// The truncated stage. + stage: RetentionFixedStage, + }, + /// A truncated stage lacks a complete, linked earlier stage. + TruncatedStageWithoutEarlierEvidence { + /// The truncated stage that cannot be discarded. + stage: RetentionFixedStage, + /// The earlier stage whose complete record or pool link is missing. + earlier_stage: RetentionFixedStage, + }, + /// A complete stage names a pool entry whose bytes or stage identity differ. + PoolEntryDiffers { + /// The pool holding the conflicting entry. + pool: RetentionPool, + }, + /// A complete head stage exists without a complete manifest stage. + HeadStageWithoutManifestStage, + /// The head stage's digest, generation, or length disagrees with the manifest. + HeadStageNamesOtherManifest, + /// The head stage's predecessor disagrees with staged or published history. + HeadPredecessorMismatch, + /// The head stage exists but the staged manifest was never linked. + ManifestNotLinkedBeforeHead, + /// The head stage exists but the staged root was never linked. + RootNotLinkedBeforeHead, + /// A complete manifest stage exists without the root stage it introduces + /// and is not the published manifest. + ManifestStageWithoutRootStage, + /// The manifest stage does not select the staged root. + ManifestStageNamesOtherRoot, + /// The manifest stage is not the exact successor of the published manifest. + ManifestNotSuccessor, + /// The root stage is not the exact successor of the namespace's current root. + RootNotSuccessor, +} + +impl RetentionRecoveryRefusal { + pub(super) fn corrupt_root(source: RetentionRootDecodeError) -> Self { + Self::StageCorrupt { + stage: RetentionFixedStage::Root, + source: Box::new(source), + } + } + + pub(super) fn corrupt_manifest(source: RetentionManifestDecodeError) -> Self { + Self::StageCorrupt { + stage: RetentionFixedStage::Manifest, + source: Box::new(source), + } + } + + pub(super) fn corrupt_head(source: RetentionHeadDecodeError) -> Self { + Self::StageCorrupt { + stage: RetentionFixedStage::Head, + source: Box::new(source), + } + } +} + +impl fmt::Display for RetentionFixedStage { + fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { + formatter.write_str(match self { + Self::Root => "root.next", + Self::Manifest => "manifest.next", + Self::Head => "head.next", + }) + } +} + +impl fmt::Display for RetentionPool { + fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { + formatter.write_str(match self { + Self::Roots => "retention/roots", + Self::Manifests => "retention/manifests", + }) + } +} + +impl fmt::Display for RetentionRecoveryRefusal { + fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Self::IncompleteStageRequiresDisposition { + stage, + expected, + observed, + } => write!( + formatter, + "retention stage {stage} requires disposition: {observed} bytes, decoder requires {expected}" + ), + Self::StageCorrupt { stage, .. } => { + write!(formatter, "retention stage {stage} is corrupt") + } + Self::TruncatedStageWithLaterEffect { stage } => write!( + formatter, + "truncated retention stage {stage} has a later-ordered effect" + ), + Self::PoolEntryDiffers { pool } => { + write!( + formatter, + "{pool} holds a different entry under the staged name" + ) + } + Self::TruncatedStageWithoutEarlierEvidence { + stage, + earlier_stage, + } => write!( + formatter, + "truncated retention stage {stage} lacks complete linked {earlier_stage} evidence" + ), + other => formatter.write_str(other.message()), + } + } +} + +impl RetentionRecoveryRefusal { + const fn message(&self) -> &'static str { + match self { + Self::HeadStageWithoutManifestStage => { + "head.next exists without a complete manifest.next" + } + Self::HeadStageNamesOtherManifest => { + "head.next names a manifest other than manifest.next" + } + Self::HeadPredecessorMismatch => "head.next does not succeed the published manifest", + Self::ManifestNotLinkedBeforeHead => { + "head.next exists but manifest.next was never linked" + } + Self::RootNotLinkedBeforeHead => "head.next exists but root.next was never linked", + Self::ManifestStageWithoutRootStage => { + "manifest.next exists without root.next and is not the published manifest" + } + Self::ManifestStageNamesOtherRoot => "manifest.next does not select root.next", + Self::ManifestNotSuccessor => { + "manifest.next is not the successor of the published manifest" + } + Self::RootNotSuccessor => { + "root.next is not the successor of its namespace's current root" + } + Self::StageCorrupt { .. } + | Self::IncompleteStageRequiresDisposition { .. } + | Self::TruncatedStageWithLaterEffect { .. } + | Self::TruncatedStageWithoutEarlierEvidence { .. } + | Self::PoolEntryDiffers { .. } => "retention recovery refused", + } + } +} + +impl Error for RetentionRecoveryRefusal { + fn source(&self) -> Option<&(dyn Error + 'static)> { + match self { + Self::StageCorrupt { source, .. } => Some(source.as_ref()), + _ => None, + } + } +} diff --git a/src/adapters/retention/recovery_stage_assessment.rs b/src/adapters/retention/recovery_stage_assessment.rs new file mode 100644 index 00000000..b0a9c17f --- /dev/null +++ b/src/adapters/retention/recovery_stage_assessment.rs @@ -0,0 +1,107 @@ +//! This module owns restart assessment of the three fixed retention stages. + +use super::{ + AdmittedRetentionManifest, AdmittedRetentionRoot, ChecksummedRetentionHead, + RetentionHeadDecodeError, RetentionManifestDecodeError, RetentionRootDecodeError, +}; + +/// One fixed retention stage as assessed from its exact bytes at restart. +/// +/// `Truncated` means decoding needs more bytes and the available checks found +/// no contradiction. It does not prove that a canonical completion exists or +/// authorize disposal. Recovery preserves these bytes for explicit disposition. +/// Demonstrated fixed-field, checksum, digest or semantic contradictions remain +/// `Corrupt` with their precise diagnostic. +#[derive(Debug)] +pub enum RetentionStageAssessment { + /// No entry exists under the stage name. + Absent, + /// The bytes decode as one canonical record. + Complete(Record), + /// The bytes end before the declared record boundary. + Truncated { + /// The next minimum length required by decoding; not a completion proof. + expected: usize, + /// The length that was present. + observed: usize, + }, + /// The bytes are complete enough to judge and fail a canonical law. + Corrupt(Error), +} + +/// Assessment of `retention/root.next`. +pub type RetentionRootStageAssessment<'bytes> = + RetentionStageAssessment, RetentionRootDecodeError>; +/// Assessment of `retention/manifest.next`. +pub type RetentionManifestStageAssessment<'bytes> = + RetentionStageAssessment, RetentionManifestDecodeError>; +/// Assessment of `retention/head.next`. +pub type RetentionHeadStageAssessment<'bytes> = + RetentionStageAssessment, RetentionHeadDecodeError>; + +impl RetentionStageAssessment { + /// Returns whether an entry exists under the stage name. + #[must_use] + pub const fn is_present(&self) -> bool { + !matches!(self, Self::Absent) + } +} + +/// Assesses the bytes found under `retention/root.next`, if any. +#[must_use] +pub fn assess_root_stage(bytes: Option<&[u8]>) -> RetentionRootStageAssessment<'_> { + let Some(bytes) = bytes else { + return RetentionStageAssessment::Absent; + }; + match AdmittedRetentionRoot::decode(bytes) { + Ok(root) => RetentionStageAssessment::Complete(root), + Err(RetentionRootDecodeError::Truncated { expected, observed }) => { + match super::stage_prefix_admission::root(bytes) { + Ok(()) => RetentionStageAssessment::Truncated { expected, observed }, + Err(source) => RetentionStageAssessment::Corrupt(source), + } + } + Err(source) => RetentionStageAssessment::Corrupt(source), + } +} + +/// Assesses the bytes found under `retention/manifest.next`, if any. +#[must_use] +pub fn assess_manifest_stage(bytes: Option<&[u8]>) -> RetentionManifestStageAssessment<'_> { + let Some(bytes) = bytes else { + return RetentionStageAssessment::Absent; + }; + match AdmittedRetentionManifest::decode(bytes) { + Ok(manifest) => RetentionStageAssessment::Complete(manifest), + Err(RetentionManifestDecodeError::Truncated { expected, observed }) => { + match super::stage_prefix_admission::manifest(bytes) { + Ok(()) => RetentionStageAssessment::Truncated { expected, observed }, + Err(source) => RetentionStageAssessment::Corrupt(source), + } + } + Err(source) => RetentionStageAssessment::Corrupt(source), + } +} + +/// Assesses the bytes found under `retention/head.next`, if any. +/// +/// The head is one fixed 144-byte record. Fewer bytes with canonical available +/// fixed fields are a truncation; contradictory bytes or extra bytes are corruption. +#[must_use] +pub fn assess_head_stage(bytes: Option<&[u8]>) -> RetentionHeadStageAssessment<'_> { + let Some(bytes) = bytes else { + return RetentionStageAssessment::Absent; + }; + match ChecksummedRetentionHead::decode(bytes) { + Ok(head) => RetentionStageAssessment::Complete(head), + Err(RetentionHeadDecodeError::WrongLength { expected, observed }) + if observed < expected => + { + match super::stage_prefix_admission::head(bytes) { + Ok(()) => RetentionStageAssessment::Truncated { expected, observed }, + Err(source) => RetentionStageAssessment::Corrupt(source), + } + } + Err(source) => RetentionStageAssessment::Corrupt(source), + } +} diff --git a/src/adapters/retention/recovery_storage.rs b/src/adapters/retention/recovery_storage.rs new file mode 100644 index 00000000..ee330506 --- /dev/null +++ b/src/adapters/retention/recovery_storage.rs @@ -0,0 +1,77 @@ +//! This module owns the blocking storage capability port for retention recovery. + +use super::RetentionStorageError; + +/// Durable capabilities retention recovery executes, one per plan step. +/// +/// Each capability owns its complete effect and the synchronization that makes +/// it durable, so an implementation cannot report a step as done before its +/// evidence would survive process death. Every capability is called at most +/// once per plan, in plan order, and never after a failed capability. +/// +/// Failure is not rollback: report the failing boundary and known/uncertain +/// namespace effects through `RetentionStorageError::Operation` when available. +/// Missing progress is unreported effects, never proof of no mutation. A caller +/// must obtain fresh observation before another attempt. +pub trait RetentionRecoveryStorage { + /// Reserved incomplete-head disposition capability; automatic disposal is deferred. + /// + /// # Errors + /// + /// Must refuse without mutation. The filesystem adapter returns `IncompleteDispositionRequired`. + fn discard_head_stage(&mut self) -> Result<(), RetentionStorageError>; + + /// Reserved incomplete-manifest disposition capability; automatic disposal is deferred. + /// + /// # Errors + /// + /// Must refuse without mutation. The filesystem adapter returns `IncompleteDispositionRequired`. + fn discard_manifest_stage(&mut self) -> Result<(), RetentionStorageError>; + + /// Reserved incomplete-root disposition capability; automatic disposal is deferred. + /// + /// # Errors + /// + /// Must refuse without mutation. The filesystem adapter returns `IncompleteDispositionRequired`. + fn discard_root_stage(&mut self) -> Result<(), RetentionStorageError>; + + /// Admits the staged root's namespace directory, links the complete root + /// stage into it without replacement, and synchronizes both directories. + /// + /// # Errors + /// + /// Returns the exact filesystem failure or a refusal of a conflicting entry. + fn link_root(&mut self) -> Result<(), RetentionStorageError>; + + /// Links the complete manifest stage into the manifest pool without + /// replacement and synchronizes the pool. + /// + /// # Errors + /// + /// Returns the exact filesystem failure or a refusal of a conflicting entry. + fn link_manifest(&mut self) -> Result<(), RetentionStorageError>; + + /// Replaces `retention/HEAD` with the complete head stage atomically and + /// synchronizes `retention`. + /// + /// # Errors + /// + /// Returns the exact filesystem failure. + fn finalize_head(&mut self) -> Result<(), RetentionStorageError>; + + /// Removes the retained root stage after proving its pool link and + /// synchronizes `retention`. + /// + /// # Errors + /// + /// Returns the exact filesystem failure or a refusal when the link is not proven. + fn remove_root_stage(&mut self) -> Result<(), RetentionStorageError>; + + /// Removes the retained manifest stage after proving its pool link and + /// synchronizes `retention`. + /// + /// # Errors + /// + /// Returns the exact filesystem failure or a refusal when the link is not proven. + fn remove_manifest_stage(&mut self) -> Result<(), RetentionStorageError>; +} diff --git a/src/adapters/retention/retention_model_refusal.rs b/src/adapters/retention/retention_model_refusal.rs new file mode 100644 index 00000000..98b7b64f --- /dev/null +++ b/src/adapters/retention/retention_model_refusal.rs @@ -0,0 +1,102 @@ +//! This module owns the model's exact rejected-operation diagnostic oracles. + +use std::error::Error; + +use super::super::{ + AdmittedRetentionManifest, AdmittedRetentionRoot, RetentionCurrentStateRefusal, + RetentionPublicationError, RetentionPublicationPreparationError, +}; +use crate::{ + LivenessGeneration, RetentionManifestDigest, RetentionNamespaceDigest, RetentionRootDigest, + RootGeneration, +}; + +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub(super) enum Refusal { + RepeatedInitial { + namespace: RetentionNamespaceDigest, + generation: RootGeneration, + digest: RetentionRootDigest, + }, + Superseded { + generation: LivenessGeneration, + digest: RetentionManifestDigest, + }, +} + +impl Refusal { + pub(super) fn repeated_initial( + candidate: &[u8], + manifest: Option<&[u8]>, + ) -> Result> { + let namespace = AdmittedRetentionRoot::decode(candidate)? + .root() + .namespace() + .digest(); + let manifest = AdmittedRetentionManifest::decode(manifest.ok_or("model manifest absent")?)?; + let entry = manifest + .manifest() + .entries() + .iter() + .find(|entry| entry.namespace() == namespace) + .ok_or("model namespace absent from pre-operation manifest")?; + Ok(Self::RepeatedInitial { + namespace, + generation: entry.root_generation(), + digest: entry.root_digest(), + }) + } + + pub(super) fn superseded(manifest: Option<&[u8]>) -> Result> { + let manifest = AdmittedRetentionManifest::decode(manifest.ok_or("model manifest absent")?)?; + Ok(Self::Superseded { + generation: manifest.manifest().generation(), + digest: manifest.digest(), + }) + } + + pub(super) fn verify(self, error: &(dyn Error + 'static)) -> Result<(), Box> { + match self { + Self::RepeatedInitial { + namespace, + generation, + digest, + } => { + assert!( + matches!( + error.downcast_ref::(), + Some(RetentionPublicationPreparationError::ManifestSuccessorMismatch { + namespace: observed_namespace, + current_generation, + current_digest, + candidate_generation: RootGeneration::INITIAL, + candidate_predecessor: None, + }) if (*observed_namespace, *current_generation, *current_digest) + == (namespace, generation, digest) + ), + "repeated initial must report its exact manifest successor mismatch: {error:?}" + ); + } + Self::Superseded { generation, digest } => { + let Some(RetentionPublicationError::CurrentVerification { source }) = + error.downcast_ref::() + else { + return Err(format!( + "superseded publication must fail current verification: {error:?}" + ) + .into()); + }; + assert!( + matches!( + source.get_ref().and_then(|source| source.downcast_ref::()), + Some(RetentionCurrentStateRefusal::Superseded { + current_generation, current_digest, + }) if (*current_generation, *current_digest) == (generation, digest) + ), + "stale initial must report the exact superseding manifest: {error:?}" + ); + } + } + Ok(()) + } +} diff --git a/src/adapters/retention/retention_model_tests.rs b/src/adapters/retention/retention_model_tests.rs new file mode 100644 index 00000000..85c4cf88 --- /dev/null +++ b/src/adapters/retention/retention_model_tests.rs @@ -0,0 +1,379 @@ +//! Model-based retention laws: every operation sequence agrees with a +//! deterministic namespace-to-anchor-set map, and a refused operation leaves +//! the fenced reader view exactly where it was. + +use std::collections::BTreeMap; +use std::error::Error; +use std::path::PathBuf; + +#[path = "retention_model_refusal.rs"] +mod refusal; +use refusal::Refusal; + +use super::filesystem_retention_test_fixture::{ + ROOT_HEX, fixture, initial_preparation, initial_root, new_namespace_preparation, + open_authority, successor_preparation, successor_root, +}; +use super::{ + AdmittedRetentionManifest, AdmittedRetentionRoot, FilesystemRetentionPublicationAuthority, + FilesystemRetentionSnapshot, ReaderAttemptLimit, RetentionPublicationOutcome, + RetentionPublicationPreparation, +}; +use crate::adapters::{ + CatalogRestartByteLimit, CatalogRestartPolicy, SegmentReadPolicy, SegmentRecordLimit, +}; +use crate::{ + LayoutEntryLimit, RetentionAnchor, RetentionNamespaceDigest, execute_retention_publication, +}; + +const NAMESPACE_B: &[u8] = b"model-namespace-b"; + +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +enum Namespace { + A, + B, +} + +#[derive(Clone, Copy, Debug)] +enum Operation { + /// Publish generation one of the namespace from a fresh view. + Initial(Namespace), + /// Publish the exact successor of namespace A from a fresh view. + Successor, + /// Replay the last accepted publication byte for byte. + RetryLast, + /// Publish generation one of namespace A from a view that predates it. + StaleInitial, +} + +const OPERATIONS: [Operation; 5] = [ + Operation::Initial(Namespace::A), + Operation::Initial(Namespace::B), + Operation::Successor, + Operation::RetryLast, + Operation::StaleInitial, +]; + +/// The byte ingredients of one preparation; rebuilding it is byte-identical. +#[derive(Clone)] +enum Recipe { + Initial { + candidate: Vec, + manifest: Option>, + }, + Successor { + current_root: Vec, + manifest: Vec, + candidate: Vec, + }, +} + +impl Recipe { + fn candidate(&self) -> &[u8] { + match self { + Self::Initial { candidate, .. } | Self::Successor { candidate, .. } => candidate, + } + } + + fn publish( + &self, + authority: &mut FilesystemRetentionPublicationAuthority, + ) -> Result> { + let preparation: RetentionPublicationPreparation<'_> = match self { + Self::Initial { + candidate, + manifest: None, + } => initial_preparation(candidate)?, + Self::Initial { + candidate, + manifest: Some(manifest), + } => { + new_namespace_preparation(&AdmittedRetentionManifest::decode(manifest)?, candidate)? + } + Self::Successor { + current_root, + manifest, + candidate, + } => successor_preparation( + &AdmittedRetentionRoot::decode(current_root)?, + &AdmittedRetentionManifest::decode(manifest)?, + candidate, + )?, + }; + Ok(execute_retention_publication(authority, &preparation)?.outcome()) + } +} + +/// The reference model: namespace digest to (generation, anchors), plus the +/// liveness generation, which counts accepted publications. +#[derive(Default)] +struct Model { + namespaces: BTreeMap)>, + liveness: u64, +} + +struct Store { + authority: FilesystemRetentionPublicationAuthority, + path: PathBuf, + template: Vec, + last_accepted: Option, +} + +fn policy() -> Result> { + Ok(CatalogRestartPolicy::new( + SegmentReadPolicy::new(SegmentRecordLimit::MAXIMUM, LayoutEntryLimit::MAXIMUM), + CatalogRestartByteLimit::new(1_048_576)?, + )) +} + +fn snapshot(store: &Store) -> Result> { + Ok(FilesystemRetentionSnapshot::load( + &store.path, + policy()?, + ReaderAttemptLimit::DEFAULT, + )?) +} + +fn digest_of(bytes: &[u8]) -> Result> { + Ok(AdmittedRetentionRoot::decode(bytes)? + .root() + .namespace() + .digest()) +} + +fn candidate_bytes(store: &Store, namespace: Namespace) -> Result, Box> { + match namespace { + Namespace::A => Ok(store.template.clone()), + Namespace::B => { + let template = AdmittedRetentionRoot::decode(&store.template)?; + Ok(initial_root(NAMESPACE_B, &template)?.encoded().to_vec()) + } + } +} + +/// A recipe and the answer the model expects for it. +type Planned = (Recipe, Expected); + +/// What the model says the store must answer. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +enum Expected { + Published, + AlreadyCommitted, + Refused(Refusal), +} + +/// Builds the recipe an operation would publish and the model's expected +/// answer. `None` means the operation has no candidate. +fn recipe( + store: &Store, + model: &Model, + operation: Operation, +) -> Result, Box> { + let fresh_manifest = store + .authority + .observe_current()? + .map(|state| state.manifest_bytes().to_vec()); + Ok(match operation { + Operation::Initial(namespace) => { + let candidate = candidate_bytes(store, namespace)?; + let expected = if model.namespaces.contains_key(&digest_of(&candidate)?) { + Expected::Refused(Refusal::repeated_initial( + &candidate, + fresh_manifest.as_deref(), + )?) + } else { + Expected::Published + }; + Some(( + Recipe::Initial { + candidate, + manifest: fresh_manifest, + }, + expected, + )) + } + Operation::StaleInitial => { + // The stale caller's preparation stages liveness generation one, + // so it is byte-identical to the accepted publication only while + // that publication is still the whole history; any later + // publication supersedes it. + let candidate = candidate_bytes(store, Namespace::A)?; + let expected = match ( + model.namespaces.get(&digest_of(&candidate)?), + model.liveness, + ) { + (None, 0) => Expected::Published, + (Some(_), 1) => Expected::AlreadyCommitted, + _ => Expected::Refused(Refusal::superseded(fresh_manifest.as_deref())?), + }; + Some(( + Recipe::Initial { + candidate, + manifest: None, + }, + expected, + )) + } + Operation::Successor => { + let digest = digest_of(&store.template)?; + if !model.namespaces.contains_key(&digest) { + return Ok(None); + } + let current_root = snapshot(store)? + .retained_root(digest)? + .ok_or("model root absent on disk")? + .to_vec(); + let candidate = successor_root(&AdmittedRetentionRoot::decode(¤t_root)?)? + .encoded() + .to_vec(); + Some(( + Recipe::Successor { + current_root, + manifest: fresh_manifest.ok_or("successor over no manifest")?, + candidate, + }, + Expected::Published, + )) + } + Operation::RetryLast => store + .last_accepted + .clone() + .map(|recipe| (recipe, Expected::AlreadyCommitted)), + }) +} + +/// Applies one operation to the store and the model. +fn apply(store: &mut Store, model: &mut Model, operation: Operation) -> Result<(), Box> { + let Some((recipe, expected)) = recipe(store, model, operation)? else { + return Ok(()); + }; + match (expected, recipe.publish(&mut store.authority)) { + (Expected::AlreadyCommitted, Ok(RetentionPublicationOutcome::AlreadyCommitted)) => {} + (Expected::Refused(refusal), Err(error)) => refusal.verify(error.as_ref())?, + (Expected::Published, Ok(RetentionPublicationOutcome::Published)) => { + let candidate = AdmittedRetentionRoot::decode(recipe.candidate())?; + model.namespaces.insert( + candidate.root().namespace().digest(), + ( + candidate.root().generation().get(), + candidate.root().anchors().to_vec(), + ), + ); + model.liveness = model.liveness.saturating_add(1); + store.last_accepted = Some(recipe); + } + (expected, result) => { + return Err( + format!("{operation:?}: model expects {expected:?}, store {result:?}").into(), + ); + } + } + Ok(()) +} + +/// Requires the fenced reader view to agree with the model exactly. +fn verify(store: &Store, model: &Model) -> Result<(), Box> { + let snapshot = snapshot(store)?; + let observed: BTreeMap<_, _> = snapshot + .manifest() + .map(|manifest| { + manifest + .entries() + .iter() + .map(|entry| (entry.namespace(), entry.root_generation().get())) + .collect() + }) + .unwrap_or_default(); + let expected: BTreeMap<_, _> = model + .namespaces + .iter() + .map(|(namespace, (generation, _))| (*namespace, *generation)) + .collect(); + assert_eq!(observed, expected, "manifest disagrees with the model"); + let liveness = snapshot + .retention_head() + .map_or(0, |head| head.generation().get()); + assert_eq!( + liveness, model.liveness, + "liveness generation disagrees with the model" + ); + for (namespace, (generation, anchors)) in &model.namespaces { + let bytes = snapshot + .retained_root(*namespace)? + .ok_or("model namespace has no root on disk")?; + let root = AdmittedRetentionRoot::decode(&bytes)?; + assert_eq!(root.root().generation().get(), *generation); + assert_eq!( + root.root().anchors(), + anchors.as_slice(), + "anchor set disagrees with the model" + ); + } + Ok(()) +} + +/// Runs every three-operation sequence that starts with `first` in a fresh +/// migrated store each, checking the fenced view against the model after +/// every step. +fn run_sequences(first: Operation, label: &str) -> Result<(), Box> { + let template = fixture(ROOT_HEX)?; + let mut sequences = 0_u32; + for second in OPERATIONS { + for third in OPERATIONS { + let name = format!("filesystem-retention-model-{label}-{sequences}"); + let (sandbox, authority) = open_authority(&name)?; + let mut store = Store { + authority, + path: sandbox.path().to_path_buf(), + template: template.clone(), + last_accepted: None, + }; + let mut model = Model::default(); + for operation in [first, second, third] { + apply(&mut store, &mut model, operation) + .map_err(|error| format!("{first:?} {second:?} {third:?}: {error}"))?; + verify(&store, &model) + .map_err(|error| format!("{first:?} {second:?} {third:?}: {error}"))?; + } + sequences = sequences.saturating_add(1); + } + } + Ok(()) +} + +// Size: medium. Oracle: namespace state model plus exact rejected-operation contract. +// Delete only when stronger scenario exploration subsumes these histories and diagnostics. +#[test] +fn sequences_starting_with_an_initial_publication_of_a_agree_with_the_model() +-> Result<(), Box> { + run_sequences(Operation::Initial(Namespace::A), "initial-a") +} + +// Size: medium. Oracle: namespace state model plus exact rejected-operation contract. +// Delete only when stronger scenario exploration subsumes these histories and diagnostics. +#[test] +fn sequences_starting_with_an_initial_publication_of_b_agree_with_the_model() +-> Result<(), Box> { + run_sequences(Operation::Initial(Namespace::B), "initial-b") +} + +// Size: medium. Oracle: namespace state model plus exact rejected-operation contract. +// Delete only when stronger scenario exploration subsumes these histories and diagnostics. +#[test] +fn sequences_starting_with_a_successor_agree_with_the_model() -> Result<(), Box> { + run_sequences(Operation::Successor, "successor") +} + +// Size: medium. Oracle: namespace state model plus exact rejected-operation contract. +// Delete only when stronger scenario exploration subsumes these histories and diagnostics. +#[test] +fn sequences_starting_with_a_retry_agree_with_the_model() -> Result<(), Box> { + run_sequences(Operation::RetryLast, "retry") +} + +// Size: medium. Oracle: namespace state model plus exact rejected-operation contract. +// Delete only when stronger scenario exploration subsumes these histories and diagnostics. +#[test] +fn sequences_starting_with_a_stale_initial_agree_with_the_model() -> Result<(), Box> { + run_sequences(Operation::StaleInitial, "stale") +} diff --git a/src/adapters/retention/retention_record_refusal.rs b/src/adapters/retention/retention_record_refusal.rs new file mode 100644 index 00000000..5ee3f488 --- /dev/null +++ b/src/adapters/retention/retention_record_refusal.rs @@ -0,0 +1,42 @@ +//! This module owns semantic refusals of exact retention records. + +use std::error::Error; +use std::fmt; + +/// Why a retention record cannot be used for a storage transition. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +#[non_exhaustive] +pub enum RetentionRecordRefusal { + /// Automatic disposition of incomplete stages is not supported. + IncompleteDispositionRequired, + /// The expected byte length exceeds the filesystem's range. + LengthOverflow, + /// The entry is not a regular file of the expected length. + KindOrLength, + /// The entry's kind, length, or identity differs from the retained record. + KindLengthOrIdentity, + /// The record differs from the admitted bytes. + Bytes, + /// The record has bytes beyond its expected end. + TrailingBytes, + /// A removed entry remains visible. + RemainedVisible, +} + +impl fmt::Display for RetentionRecordRefusal { + fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { + formatter.write_str(match self { + Self::IncompleteDispositionRequired => { + "incomplete retention stage requires explicit disposition" + } + Self::LengthOverflow => "retention record length exceeded the filesystem's range", + Self::KindOrLength => "retention record kind or length disagreed", + Self::KindLengthOrIdentity => "retention record kind, length, or identity disagreed", + Self::Bytes => "retention record bytes disagreed", + Self::TrailingBytes => "retention record carried trailing bytes", + Self::RemainedVisible => "removed retention record remained visible", + }) + } +} + +impl Error for RetentionRecordRefusal {} diff --git a/src/adapters/retention/retention_storage_error.rs b/src/adapters/retention/retention_storage_error.rs new file mode 100644 index 00000000..c4626cef --- /dev/null +++ b/src/adapters/retention/retention_storage_error.rs @@ -0,0 +1,127 @@ +//! This module owns typed failures at the retention storage boundary. + +use std::error::Error; +use std::fmt; +use std::io; + +use super::{ + RetentionEffectDurability, RetentionKnownEffect, RetentionNamespaceEffect, + RetentionRecordRefusal, RetentionStorageBoundary, RetentionStorageProgress, +}; + +/// An operational failure or exact-record refusal during retention storage. +#[derive(Debug)] +#[non_exhaustive] +pub enum RetentionStorageError { + /// An operation stopped at a known boundary, possibly after namespace effects. + Operation { + /// The original typed storage cause, never stringified. + source: Box, + /// Known effects and uncertainty within this failing capability. + progress: RetentionStorageProgress, + }, + /// A filesystem operation failed. + Io { + /// The original operational error, including its OS code and source. + source: io::Error, + }, + /// The record did not match its admitted evidence. + Refused { + /// The precise record refusal. + source: RetentionRecordRefusal, + }, +} + +impl fmt::Display for RetentionStorageError { + fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Self::Operation { progress, .. } => write!( + formatter, + "retention storage stopped at {:?}; known effects {:?}, uncertain effect {:?}", + progress.boundary, progress.known, progress.uncertain + ), + Self::Io { .. } => formatter.write_str("retention storage operation failed"), + Self::Refused { source } => fmt::Display::fmt(source, formatter), + } + } +} + +impl Error for RetentionStorageError { + fn source(&self) -> Option<&(dyn Error + 'static)> { + match self { + Self::Operation { source, .. } => Some(source.as_ref()), + Self::Io { source } => Some(source), + Self::Refused { source } => Some(source), + } + } +} + +impl From for RetentionStorageError { + fn from(source: io::Error) -> Self { + Self::Io { source } + } +} + +impl From for io::Error { + fn from(error: RetentionStorageError) -> Self { + match error { + RetentionStorageError::Io { source } => source, + refused @ (RetentionStorageError::Refused { .. } + | RetentionStorageError::Operation { .. }) => Self::new(refused.io_kind(), refused), + } + } +} + +impl RetentionStorageError { + fn io_kind(&self) -> io::ErrorKind { + match self { + Self::Io { source } => source.kind(), + Self::Refused { .. } => io::ErrorKind::InvalidData, + Self::Operation { source, .. } => source.io_kind(), + } + } + + pub(super) fn at(self, boundary: RetentionStorageBoundary) -> Self { + match self { + Self::Operation { .. } => self, + source => Self::Operation { + source: Box::new(source), + progress: RetentionStorageProgress { + boundary, + known: Vec::new(), + uncertain: None, + }, + }, + } + } + + pub(super) fn after( + mut self, + effect: RetentionNamespaceEffect, + durability: RetentionEffectDurability, + ) -> Self { + if let Self::Operation { progress, .. } = &mut self { + progress + .known + .insert(0, RetentionKnownEffect { effect, durability }); + } + self + } + + pub(super) const fn uncertain(mut self, effect: RetentionNamespaceEffect) -> Self { + if let Self::Operation { progress, .. } = &mut self { + progress.uncertain = Some(effect); + } + self + } + /// Progress within the failing capability, if its adapter reports it. + /// + /// `None` means unreported effects, not absence of effects. Callers must reobserve. + #[must_use] + pub const fn progress(&self) -> Option<&RetentionStorageProgress> { + match self { + Self::Operation { progress, .. } => Some(progress), + _ => None, + } + } +} diff --git a/src/adapters/retention/retention_storage_error_law_tests.rs b/src/adapters/retention/retention_storage_error_law_tests.rs new file mode 100644 index 00000000..c8db9a50 --- /dev/null +++ b/src/adapters/retention/retention_storage_error_law_tests.rs @@ -0,0 +1,82 @@ +//! This module owns recovery executor preservation of typed port refusals. + +use std::error::Error; + +use super::{ + RetentionRecordRefusal as Refusal, RetentionRecoveryOutcome, RetentionRecoveryPlan, + RetentionRecoveryStep as Step, RetentionRecoveryStorage, RetentionStorageError as StorageError, + execute_retention_recovery, +}; + +struct Refusing(Refusal); + +impl Refusing { + fn refuse(&self) -> Result<(), StorageError> { + Err(StorageError::Refused { source: self.0 }) + } +} + +impl RetentionRecoveryStorage for Refusing { + fn discard_head_stage(&mut self) -> Result<(), StorageError> { + self.refuse() + } + fn discard_manifest_stage(&mut self) -> Result<(), StorageError> { + self.refuse() + } + fn discard_root_stage(&mut self) -> Result<(), StorageError> { + self.refuse() + } + fn link_root(&mut self) -> Result<(), StorageError> { + self.refuse() + } + fn link_manifest(&mut self) -> Result<(), StorageError> { + self.refuse() + } + fn finalize_head(&mut self) -> Result<(), StorageError> { + self.refuse() + } + fn remove_root_stage(&mut self) -> Result<(), StorageError> { + self.refuse() + } + fn remove_manifest_stage(&mut self) -> Result<(), StorageError> { + self.refuse() + } +} + +// Size: small. Oracle: the public executor preserves a port's precise refusal +// for every capability; this law makes no filesystem fault-coverage claim. +// Delete when the port is removed or stronger cheaper propagation evidence subsumes it. +#[test] +fn every_recovery_capability_preserves_its_supplied_record_refusal() -> Result<(), Box> { + let steps = [ + Step::DiscardHeadStage, + Step::DiscardManifestStage, + Step::DiscardRootStage, + Step::LinkRoot, + Step::LinkManifest, + Step::FinalizeHead, + Step::RemoveRootStage, + Step::RemoveManifestStage, + ]; + let refusals = [ + Refusal::LengthOverflow, + Refusal::KindOrLength, + Refusal::KindLengthOrIdentity, + Refusal::Bytes, + Refusal::TrailingBytes, + Refusal::RemainedVisible, + ]; + for step in steps { + for refusal in refusals { + let plan = RetentionRecoveryPlan::new(vec![step], RetentionRecoveryOutcome::Committed); + let error = execute_retention_recovery(&mut Refusing(refusal), &plan) + .err() + .ok_or("refused storage was reported as committed")?; + assert!( + matches!(error.storage_error(), StorageError::Refused { source } if *source == refusal), + "{step:?} must preserve {refusal:?}: {error:?}" + ); + } + } + Ok(()) +} diff --git a/src/adapters/retention/retention_storage_progress.rs b/src/adapters/retention/retention_storage_progress.rs new file mode 100644 index 00000000..adeb5281 --- /dev/null +++ b/src/adapters/retention/retention_storage_progress.rs @@ -0,0 +1,106 @@ +//! This module owns reported namespace effects of a failing retention storage operation. + +/// The fallible boundary at which a retention storage capability stopped. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +#[non_exhaustive] +pub enum RetentionStorageBoundary { + /// No active recovery context or required stage was available. + RecoveryContext, + /// The retained source handle and pathname were checked. + SourceVerification, + /// Staged file contents were synchronized. + StageSynchronization, + /// A root namespace directory was created. + NamespaceCreation, + /// The root namespace directory was opened without following links. + NamespaceOpen, + /// The roots directory was synchronized after namespace creation. + RootsSynchronization, + /// A pool hard link was attempted without replacement. + PoolLink, + /// The pool target's kind, identity and exact bytes were checked. + PoolVerification, + /// A pool directory was synchronized. + PoolSynchronization, + /// The source stage pathname was unlinked. + StageUnlink, + /// The source stage pathname was checked for absence. + StageAbsence, + /// The head stage was renamed onto the published head. + HeadRename, + /// The replaced head's identity and exact bytes were checked. + HeadVerification, + /// The retention directory was synchronized after a namespace effect. + RetentionSynchronization, +} + +/// A namespace effect whose occurrence or attempted occurrence is reported. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub enum RetentionNamespaceEffect { + /// A new root namespace directory was created. + NamespaceCreated, + /// A new immutable pool link was created. + PoolLinkCreated, + /// The published head pathname was replaced. + HeadReplaced, + /// A retained stage pathname was removed. + StageRemoved, +} + +/// Whether the directory synchronization covering an observed effect completed. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub enum RetentionEffectDurability { + /// The covering directory synchronization has not completed successfully. + Unconfirmed, + /// The covering directory synchronization completed successfully. + Synchronized, +} + +/// One effect positively observed by the failing capability. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub struct RetentionKnownEffect { + pub(super) effect: RetentionNamespaceEffect, + pub(super) durability: RetentionEffectDurability, +} + +impl RetentionKnownEffect { + /// The namespace effect known to have occurred. + #[must_use] + pub const fn effect(self) -> RetentionNamespaceEffect { + self.effect + } + /// The observed synchronization status; failure never implies rollback. + #[must_use] + pub const fn durability(self) -> RetentionEffectDurability { + self.durability + } +} + +/// What one failing capability established before returning its error. +/// +/// Only namespace effects are listed; this is not a complete syscall trace. +/// An empty known list does not exclude the separately reported uncertain effect. +#[derive(Debug)] +pub struct RetentionStorageProgress { + pub(super) boundary: RetentionStorageBoundary, + pub(super) known: Vec, + pub(super) uncertain: Option, +} + +impl RetentionStorageProgress { + /// The exact failing boundary within the capability. + #[must_use] + pub const fn boundary(&self) -> RetentionStorageBoundary { + self.boundary + } + /// Positively observed namespace effects, with their synchronization status. + #[must_use] + pub fn known_effects(&self) -> &[RetentionKnownEffect] { + &self.known + } + /// An attempted effect whose occurrence could not be established from the failed operation. + #[must_use] + pub const fn uncertain_effect(&self) -> Option { + self.uncertain + } +} diff --git a/src/adapters/retention/retention_view_collector.rs b/src/adapters/retention/retention_view_collector.rs new file mode 100644 index 00000000..6c64456e --- /dev/null +++ b/src/adapters/retention/retention_view_collector.rs @@ -0,0 +1,118 @@ +//! This module owns storage-independent double collection of one reader view. + +use std::error::Error; +use std::fmt; +use std::io; + +use super::ReaderAttemptLimit; +use crate::{CatalogDigest, CatalogGeneration, CatalogLength, RetentionHead}; + +#[cfg(test)] +#[path = "retention_view_coordinate_law_tests.rs"] +mod coordinate_law_tests; + +/// The coordinates both heads name at one instant. +/// +/// A view is accepted only when the coordinates read before loading it equal +/// the coordinates read after, so the view belongs to one catalog generation +/// and one liveness generation. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub struct RetentionViewCoordinates { + /// The catalog `HEAD` coordinate, or `None` when no catalog is published. + pub catalog: Option<(CatalogGeneration, CatalogLength, CatalogDigest)>, + /// The `retention/HEAD` coordinate, or `None` when no retention head is published. + pub retention: Option, +} + +/// The reads one reader view needs, in the order the collector calls them. +pub trait RetentionViewSource { + /// The complete view loaded between two coordinate reads. + type View; + + /// Reads both head coordinates without loading anything they select. + /// + /// # Errors + /// + /// Returns the exact read or decode failure. + fn coordinates(&mut self) -> io::Result; + + /// Loads the complete view the current heads select. + /// + /// # Errors + /// + /// Returns the exact load failure. + fn load(&mut self) -> io::Result; +} + +/// Why a reader view could not be collected. +#[derive(Debug)] +#[non_exhaustive] +pub enum RetentionViewError { + /// No catalog `HEAD` is published, so no view exists to collect. + CatalogAbsent, + /// The heads moved between every collection within the attempt limit. + AttemptsExhausted { + /// The attempts that were made. + attempts: u32, + }, + /// A read or load failed. + Io { + /// The exact failure. + source: io::Error, + }, +} + +impl fmt::Display for RetentionViewError { + fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Self::CatalogAbsent => formatter.write_str("no catalog head is published"), + Self::AttemptsExhausted { attempts } => write!( + formatter, + "the store moved between every one of {attempts} view collections" + ), + Self::Io { .. } => formatter.write_str("reader view collection failed"), + } + } +} + +impl Error for RetentionViewError { + fn source(&self) -> Option<&(dyn Error + 'static)> { + match self { + Self::Io { source } => Some(source), + Self::CatalogAbsent | Self::AttemptsExhausted { .. } => None, + } + } +} + +/// Collects one view whose head coordinates agree before and after loading. +/// +/// Each attempt reads the coordinates, loads the view, and reads the +/// coordinates again; a view is accepted only when both reads agree. A +/// validated generation, length, digest, or predecessor change discards the +/// view and retries until `limit` is exhausted, which refuses. Invalid head +/// encodings, including checksum failures, return the source's read error. +/// +/// # Errors +/// +/// Returns [`RetentionViewError`] for an absent catalog, an exhausted limit, +/// or the source's own failure. +pub fn collect_retention_view( + source: &mut S, + limit: ReaderAttemptLimit, +) -> Result { + let io = |source| RetentionViewError::Io { source }; + for _attempt in 0..limit.get() { + let before = source.coordinates().map_err(io)?; + if before.catalog.is_none() { + return Err(RetentionViewError::CatalogAbsent); + } + let view = source.load().map_err(io)?; + let after = source.coordinates().map_err(io)?; + if before == after { + return Ok(view); + } + } + Err(RetentionViewError::AttemptsExhausted { + attempts: limit.get(), + }) +} diff --git a/src/adapters/retention/retention_view_collector_tests.rs b/src/adapters/retention/retention_view_collector_tests.rs new file mode 100644 index 00000000..160b4075 --- /dev/null +++ b/src/adapters/retention/retention_view_collector_tests.rs @@ -0,0 +1,215 @@ +//! This module owns reader collection outcomes at the public source port. + +use std::collections::VecDeque; +use std::error::Error; +use std::io; +use std::num::NonZeroU32; + +use super::{ + ReaderAttemptLimit, RetentionViewCoordinates, RetentionViewError, RetentionViewSource, + collect_retention_view, +}; +use crate::{ + CatalogDigest, CatalogGeneration, CatalogLength, LivenessGeneration, RetentionHead, + RetentionManifestDigest, RetentionManifestLength, +}; + +struct Scripted { + coordinates: VecDeque>, + views: VecDeque>, +} + +impl RetentionViewSource for Scripted { + type View = &'static str; + + fn coordinates(&mut self) -> io::Result { + self.coordinates + .pop_front() + .ok_or_else(|| io::Error::other("coordinate script exhausted"))? + } + + fn load(&mut self) -> io::Result { + self.views + .pop_front() + .ok_or_else(|| io::Error::other("view script exhausted"))? + } +} + +fn published( + generation: u64, + digest: [u8; 32], +) -> Result> { + Ok(RetentionViewCoordinates { + catalog: Some(( + CatalogGeneration::new(generation)?, + CatalogLength::MINIMUM, + CatalogDigest::from_validated(digest), + )), + retention: None, + }) +} + +const ABSENT: RetentionViewCoordinates = RetentionViewCoordinates { + catalog: None, + retention: None, +}; + +// Size: small. Oracle: stable coordinates return the loaded payload. +// Delete when the collector contract is retired or subsumed by stronger coverage. +#[test] +fn a_stable_store_returns_its_loaded_view() -> Result<(), Box> { + let stable = published(1, [0; 32])?; + let mut source = Scripted { + coordinates: [Ok(stable), Ok(stable)].into(), + views: [Ok("stable payload")].into(), + }; + assert_eq!( + collect_retention_view(&mut source, ReaderAttemptLimit::DEFAULT)?, + "stable payload" + ); + Ok(()) +} + +fn require_retry( + before: RetentionViewCoordinates, + after: RetentionViewCoordinates, +) -> Result<(), Box> { + let mut source = Scripted { + coordinates: [Ok(before), Ok(after), Ok(after), Ok(after)].into(), + views: [Ok("superseded payload"), Ok("stable payload")].into(), + }; + assert_eq!( + collect_retention_view(&mut source, ReaderAttemptLimit::DEFAULT)?, + "stable payload", + "changed coordinates must discard the superseded payload: {before:?} -> {after:?}" + ); + Ok(()) +} + +// Size: small. Oracle: a changed generation invalidates the loaded payload. +// Delete when the collector contract is retired or subsumed by stronger coverage. +#[test] +fn a_publication_between_reads_discards_the_superseded_view() -> Result<(), Box> { + require_retry(published(1, [0; 32])?, published(2, [0; 32])?) +} + +// Size: small. Oracle: digest equality is required even at the same generation. +// Domain: every nonzero first byte; remaining digest bytes stay zero. +// Delete when a stronger digest-domain law subsumes this sweep. +#[test] +fn changed_catalog_digests_discard_the_superseded_view() -> Result<(), Box> { + for byte in 1..=u8::MAX { + let mut digest = [0; 32]; + *digest.first_mut().ok_or("digest absent")? = byte; + require_retry(published(1, [0; 32])?, published(1, digest)?)?; + } + Ok(()) +} + +// Size: small. Oracle: a selected retention digest change invalidates the loaded view. +// Domain: every nonzero first byte with other digest bytes fixed at zero. +// Delete when a stronger digest-domain law subsumes this sweep. +#[test] +fn changed_retention_digests_discard_the_superseded_view() -> Result<(), Box> { + let mut before = published(1, [0; 32])?; + before.retention = Some(RetentionHead::new( + LivenessGeneration::new(1)?, + RetentionManifestLength::MINIMUM, + RetentionManifestDigest::from_hash([0; 32]), + None, + )?); + for byte in 1..=u8::MAX { + let mut digest = [0; 32]; + *digest.first_mut().ok_or("digest absent")? = byte; + let mut after = before; + after.retention = Some(RetentionHead::new( + LivenessGeneration::new(1)?, + RetentionManifestLength::MINIMUM, + RetentionManifestDigest::from_hash(digest), + None, + )?); + require_retry(before, after)?; + } + Ok(()) +} + +// Size: small. Oracle: exhausted attempts refuse instead of returning a moving view. +// Delete when the bounded collector contract is retired or subsumed. +#[test] +fn a_store_that_never_settles_exhausts_the_limit() -> Result<(), Box> { + let mut source = Scripted { + coordinates: (1..=4) + .map(|generation| published(generation, [0; 32]).map(Ok)) + .collect::>()?, + views: [Ok("superseded one"), Ok("superseded two")].into(), + }; + let limit = ReaderAttemptLimit::new(NonZeroU32::new(2).ok_or("zero")?); + assert!(matches!( + collect_retention_view(&mut source, limit), + Err(RetentionViewError::AttemptsExhausted { attempts: 2 }) + )); + Ok(()) +} + +// Size: small. Oracle: an absent catalog cannot produce a reader view. +// Delete when catalog absence is removed from the contract or subsumed. +#[test] +fn an_absent_catalog_refuses_collection() { + let mut source = Scripted { + coordinates: [Ok(ABSENT)].into(), + views: [Ok("unselected payload")].into(), + }; + assert!(matches!( + collect_retention_view(&mut source, ReaderAttemptLimit::DEFAULT), + Err(RetentionViewError::CatalogAbsent) + )); +} + +fn require_io(mut source: Scripted) -> Result<(), Box> { + let error = collect_retention_view(&mut source, ReaderAttemptLimit::DEFAULT) + .err() + .ok_or("read failure returned a view")?; + let RetentionViewError::Io { source } = error else { + return Err(format!("source I/O failure was reclassified: {error:?}").into()); + }; + assert_eq!( + source.raw_os_error(), + Some(123), + "collector must preserve the original source I/O error" + ); + Ok(()) +} + +// Size: small. Oracle: initial-coordinate I/O failure is returned intact. +// Delete when the source port is removed or stronger propagation coverage subsumes it. +#[test] +fn initial_coordinate_failure_preserves_its_io_cause() -> Result<(), Box> { + require_io(Scripted { + coordinates: [Err(io::Error::from_raw_os_error(123))].into(), + views: [Ok("unselected payload")].into(), + }) +} + +// Size: small. Oracle: load I/O failure is returned intact. +// Delete when the source port is removed or stronger propagation coverage subsumes it. +#[test] +fn view_load_failure_preserves_its_io_cause() -> Result<(), Box> { + require_io(Scripted { + coordinates: [Ok(published(1, [0; 32])?)].into(), + views: [Err(io::Error::from_raw_os_error(123))].into(), + }) +} + +// Size: small. Oracle: final-coordinate I/O failure invalidates the loaded view. +// Delete when the source port is removed or stronger propagation coverage subsumes it. +#[test] +fn final_coordinate_failure_preserves_its_io_cause() -> Result<(), Box> { + require_io(Scripted { + coordinates: [ + Ok(published(1, [0; 32])?), + Err(io::Error::from_raw_os_error(123)), + ] + .into(), + views: [Ok("unverified payload")].into(), + }) +} diff --git a/src/adapters/retention/retention_view_coordinate_law_tests.rs b/src/adapters/retention/retention_view_coordinate_law_tests.rs new file mode 100644 index 00000000..c15991b1 --- /dev/null +++ b/src/adapters/retention/retention_view_coordinate_law_tests.rs @@ -0,0 +1,103 @@ +//! This module owns semantic reader retries for complete retention coordinates. + +use std::collections::VecDeque; +use std::error::Error; +use std::io; + +use super::{RetentionViewCoordinates, RetentionViewSource, collect_retention_view}; +use crate::adapters::retention::ReaderAttemptLimit; +use crate::{ + CatalogDigest, CatalogGeneration, CatalogLength, LivenessGeneration, RetentionHead, + RetentionManifestDigest, RetentionManifestLength, +}; + +struct Views { + heads: VecDeque, + values: VecDeque<&'static str>, +} + +impl RetentionViewSource for Views { + type View = &'static str; + + fn coordinates(&mut self) -> io::Result { + self.heads + .pop_front() + .ok_or_else(|| io::Error::other("coordinate script exhausted")) + } + + fn load(&mut self) -> io::Result { + self.values + .pop_front() + .ok_or_else(|| io::Error::other("view script exhausted")) + } +} + +fn head( + length: RetentionManifestLength, + predecessor: [u8; 32], +) -> Result> { + Ok(RetentionHead::new( + LivenessGeneration::new(2)?, + length, + RetentionManifestDigest::from_hash([255; 32]), + Some(RetentionManifestDigest::from_hash(predecessor)), + )?) +} + +fn retry(before: RetentionHead, after: RetentionHead) -> Result<(), Box> { + let catalog = Some(( + CatalogGeneration::new(1)?, + CatalogLength::MINIMUM, + CatalogDigest::from_validated([0; 32]), + )); + let before = RetentionViewCoordinates { + catalog, + retention: Some(before), + }; + let after = RetentionViewCoordinates { + catalog, + retention: Some(after), + }; + let mut source = Views { + heads: [before, after, after, after].into(), + values: ["superseded", "stable"].into(), + }; + let view = collect_retention_view(&mut source, ReaderAttemptLimit::DEFAULT)?; + assert_eq!( + view, "stable", + "a changed retention coordinate must discard the superseded view: {before:?} -> {after:?}" + ); + Ok(()) +} + +// Size: small. Oracle: changing a selected manifest length invalidates that attempt. +// Domain: every admitted length above the minimum; no snapshots or harness-count oracle. +// Delete when full-head collection is removed or a stronger cheaper law subsumes this domain. +#[test] +fn every_changed_manifest_length_discards_the_superseded_view() -> Result<(), Box> { + let before = head(RetentionManifestLength::MINIMUM, [0; 32])?; + for length in (RetentionManifestLength::MINIMUM.get()..=RetentionManifestLength::MAXIMUM.get()) + .step_by(72) + .skip(1) + { + retry( + before, + head(RetentionManifestLength::new(length)?, [0; 32])?, + )?; + } + Ok(()) +} + +// Size: small. Oracle: changing a predecessor invalidates that collection attempt. +// Domain: every nonzero first byte, with other predecessor bytes held at zero. +// Delete when full-head collection is removed or a stronger cheaper law subsumes this domain. +#[test] +fn changed_predecessor_bytes_discard_the_superseded_view() -> Result<(), Box> { + let before = head(RetentionManifestLength::MINIMUM, [0; 32])?; + for byte in 1..=u8::MAX { + let mut predecessor = [0; 32]; + *predecessor.first_mut().ok_or("predecessor absent")? = byte; + retry(before, head(RetentionManifestLength::MINIMUM, predecessor)?)?; + } + Ok(()) +} diff --git a/src/adapters/retention/root_anchor_decoder.rs b/src/adapters/retention/root_anchor_decoder.rs index 44d26253..419c4dfa 100644 --- a/src/adapters/retention/root_anchor_decoder.rs +++ b/src/adapters/retention/root_anchor_decoder.rs @@ -20,17 +20,7 @@ pub(super) fn decode( for (position, bytes) in encoded.chunks_exact(ANCHOR_WIDTH).enumerate() { let index = u32::try_from(position).map_err(|_| RetentionRootDecodeError::LengthOverflow)?; - let (blob_bytes, layout_bytes) = bytes.split_at(BLOB_ID_WIDTH); - let blob_id = BlobId::parse_binary(blob_bytes) - .map_err(|source| RetentionRootDecodeError::BlobId { index, source })?; - let layout_id = LayoutId::parse_binary(layout_bytes) - .map_err(|source| RetentionRootDecodeError::LayoutId { index, source })?; - let observed = RetentionAnchor::new(blob_id, layout_id); - if let Some(prior) = previous - && observed <= prior - { - return Err(RetentionRootDecodeError::NonCanonicalAnchorOrder { index }); - } + let observed = admit_anchor(bytes, index, previous)?; anchors.push(observed); previous = Some(observed); } @@ -45,3 +35,44 @@ pub(super) fn decode( }) } } + +/// Admits complete anchors and available coordinates in the partial suffix. +pub(super) fn admit_prefix( + encoded: &[u8], + body_start: usize, +) -> Result<(), RetentionRootDecodeError> { + let mut previous = None; + for (position, bytes) in encoded.chunks(ANCHOR_WIDTH).enumerate() { + let index = + u32::try_from(position).map_err(|_| RetentionRootDecodeError::LengthOverflow)?; + if bytes.len() < ANCHOR_WIDTH { + let start = position + .checked_mul(ANCHOR_WIDTH) + .and_then(|offset| body_start.checked_add(offset)) + .ok_or(RetentionRootDecodeError::LengthOverflow)?; + super::root_anchor_prefix::admit(bytes, index, start)?; + return super::root_anchor_order_prefix::admit(bytes, index, previous); + } + previous = Some(admit_anchor(bytes, index, previous)?); + } + Ok(()) +} + +pub(super) fn admit_anchor( + bytes: &[u8], + index: u32, + previous: Option, +) -> Result { + let (blob_bytes, layout_bytes) = bytes.split_at(BLOB_ID_WIDTH); + let blob_id = BlobId::parse_binary(blob_bytes) + .map_err(|source| RetentionRootDecodeError::BlobId { index, source })?; + let layout_id = LayoutId::parse_binary(layout_bytes) + .map_err(|source| RetentionRootDecodeError::LayoutId { index, source })?; + let observed = RetentionAnchor::new(blob_id, layout_id); + if let Some(prior) = previous + && observed <= prior + { + return Err(RetentionRootDecodeError::NonCanonicalAnchorOrder { index }); + } + Ok(observed) +} diff --git a/src/adapters/retention/root_anchor_order_prefix.rs b/src/adapters/retention/root_anchor_order_prefix.rs new file mode 100644 index 00000000..336d374a --- /dev/null +++ b/src/adapters/retention/root_anchor_order_prefix.rs @@ -0,0 +1,52 @@ +//! This boundary module owns ordering feasibility of an incomplete canonical anchor. + +use super::RetentionRootDecodeError as Error; +use crate::{LayoutRecordLength, RetentionAnchor}; + +pub(super) fn admit( + bytes: &[u8], + index: u32, + previous: Option, +) -> Result<(), Error> { + let Some(prior) = previous else { + return Ok(()); + }; + let mut greatest = greatest_coordinate(prior)?; + for (target, observed) in greatest.iter_mut().zip(bytes) { + *target = *observed; + } + let upper = super::root_field_decoder::read_u64(&greatest, 79)?; + let length = LayoutRecordLength::greatest_at_most(upper) + .ok_or(Error::NonCanonicalAnchorOrder { index })?; + greatest + .get_mut(79..87) + .ok_or(Error::LengthOverflow)? + .copy_from_slice(&length.get().to_be_bytes()); + // Clipping and alignment must never replace an available byte. The bound + // is only a witness for possibility, never returned as an observed anchor. + if !greatest.starts_with(bytes) { + return Err(Error::NonCanonicalAnchorOrder { index }); + } + super::root_anchor_decoder::admit_anchor(&greatest, index, previous).map(|_| ()) +} + +fn greatest_coordinate(prior: RetentionAnchor) -> Result<[u8; 119], Error> { + let mut encoded = [u8::MAX; 119]; + encoded + .get_mut(..59) + .ok_or(Error::LengthOverflow)? + .copy_from_slice(&prior.blob_id().encode_binary()); + encoded + .get_mut(59..) + .ok_or(Error::LengthOverflow)? + .copy_from_slice(&prior.layout_id().encode_binary()); + encoded + .get_mut(19..59) + .ok_or(Error::LengthOverflow)? + .fill(u8::MAX); + encoded + .get_mut(79..119) + .ok_or(Error::LengthOverflow)? + .fill(u8::MAX); + Ok(encoded) +} diff --git a/src/adapters/retention/root_anchor_prefix.rs b/src/adapters/retention/root_anchor_prefix.rs new file mode 100644 index 00000000..691f6f80 --- /dev/null +++ b/src/adapters/retention/root_anchor_prefix.rs @@ -0,0 +1,54 @@ +//! This boundary module owns coordinate admission within an interrupted root anchor. + +use super::RetentionRootDecodeError as Error; +use crate::adapters::{blob_id_binary, layout_id_binary}; + +pub(super) fn admit(bytes: &[u8], index: u32, start: usize) -> Result<(), Error> { + super::stage_fixed_field_admission::admit( + bytes, + &[ + (0, &blob_id_binary::BINARY_MAGIC), + (16, &blob_id_binary::IDENTITY_VERSION.to_be_bytes()), + (18, &[blob_id_binary::HASH_ALGORITHM]), + (59, &layout_id_binary::BINARY_MAGIC), + (75, &layout_id_binary::IDENTITY_VERSION.to_be_bytes()), + (77, &layout_id_binary::LAYOUT_CODEC.to_be_bytes()), + ], + |offset, expected, observed| { + start + .checked_add(offset) + .map_or(Error::LengthOverflow, |offset| Error::PrefixByteMismatch { + offset, + expected, + observed, + }) + }, + )?; + admit_length(bytes, index) +} + +fn admit_length(bytes: &[u8], index: u32) -> Result<(), Error> { + if bytes.len() >= 87 { + let observed = super::root_field_decoder::read_u64(bytes, 79)?; + let _length = layout_id_binary::validate_plan_length(observed) + .map_err(|source| Error::LayoutId { index, source })?; + return Ok(()); + } + let Some(available) = bytes.get(79..) else { + return Ok(()); + }; + let mut completion = [0_u8; 8]; + for (target, observed) in completion.iter_mut().zip(available) { + *target = *observed; + } + let minimum = u64::from_be_bytes(completion); + let maximum = crate::LayoutRecordLength::MAXIMUM.get(); + if minimum > maximum { + return Err(Error::LayoutLengthPrefixAboveMaximum { + index, + minimum, + maximum, + }); + } + Ok(()) +} diff --git a/src/adapters/retention/root_decode_error.rs b/src/adapters/retention/root_decode_error.rs index 49653856..79ab52a4 100644 --- a/src/adapters/retention/root_decode_error.rs +++ b/src/adapters/retention/root_decode_error.rs @@ -3,14 +3,19 @@ use std::collections::TryReserveError; use crate::{ - BlobIdBinaryParseError, LayoutIdBinaryParseError, RetentionClosureLimitError, - RetentionNamespaceError, RetentionProfileAdmissionError, RetentionRootError, - RootGenerationError, + BlobIdBinaryParseError, LayoutIdBinaryParseError, RetentionClosureLimit, + RetentionClosureLimitError, RetentionNamespaceError, RetentionProfileAdmissionError, + RetentionRootError, RootGenerationError, }; /// Failure to decode and admit one version-2 retention root. #[derive(Debug)] pub enum RetentionRootDecodeError { + /// Available size-field bytes admit no canonical completion. + FramingPrefixImpossible { + /// Actual byte length of the interrupted stage. + observed: usize, + }, /// The byte string ended before its required exact length. Truncated { /// Required byte length. @@ -25,6 +30,15 @@ pub enum RetentionRootDecodeError { /// Observed byte length. observed: usize, }, + /// An available byte contradicts a fixed or computed integrity field in an interrupted stage. + PrefixByteMismatch { + /// Absolute byte offset in the observed stage. + offset: usize, + /// The canonical byte required at this offset. + expected: u8, + /// The byte actually present at this offset. + observed: u8, + }, /// The fixed record magic was not canonical. InvalidMagic { /// Observed 16 magic bytes. @@ -97,6 +111,15 @@ pub enum RetentionRootDecodeError { /// Preserved limit failure. source: RetentionClosureLimitError, }, + /// An interrupted limit cannot be completed within its resource ceiling. + ClosureLimitPrefixAboveMaximum { + /// Resource whose available bytes already exceed its ceiling. + limit: RetentionClosureLimit, + /// Smallest value possible when all unavailable bytes are zero. + minimum: u64, + /// Largest admitted value for this resource. + maximum: u64, + }, /// One anchor contained a malformed `BlobId`. BlobId { /// Zero-based anchor index. @@ -104,6 +127,15 @@ pub enum RetentionRootDecodeError { /// Preserved coordinate failure. source: BlobIdBinaryParseError, }, + /// An incomplete layout length cannot fit within its format ceiling. + LayoutLengthPrefixAboveMaximum { + /// Zero-based anchor index. + index: u32, + /// Smallest completion of the available big-endian length bytes. + minimum: u64, + /// Maximum admitted layout record length. + maximum: u64, + }, /// One anchor contained a malformed `LayoutId`. LayoutId { /// Zero-based anchor index. diff --git a/src/adapters/retention/root_decode_error_display.rs b/src/adapters/retention/root_decode_error_display.rs index f95e69f8..017e9ec9 100644 --- a/src/adapters/retention/root_decode_error_display.rs +++ b/src/adapters/retention/root_decode_error_display.rs @@ -6,6 +6,10 @@ use super::RetentionRootDecodeError; impl fmt::Display for RetentionRootDecodeError { fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { match self { + Self::FramingPrefixImpossible { observed } => write!( + formatter, + "retention root size fields have no canonical completion at {observed} bytes" + ), Self::Truncated { expected, observed } => { write!( formatter, @@ -16,6 +20,11 @@ impl fmt::Display for RetentionRootDecodeError { formatter, "retention root has trailing data: expected {expected} bytes, observed {observed}" ), + Self::PrefixByteMismatch { + offset, + expected, + observed, + } => prefix_failure(formatter, *offset, *expected, *observed), Self::InvalidMagic { observed } => { write!(formatter, "invalid retention root magic {observed:02x?}") } @@ -58,17 +67,25 @@ impl fmt::Display for RetentionRootDecodeError { Self::ClosureLimit { source } => { write!(formatter, "invalid root closure limit: {source}") } - Self::BlobId { index, source } => { - write!( - formatter, - "invalid BlobId in retention anchor {index}: {source}" - ) - } + Self::ClosureLimitPrefixAboveMaximum { + limit, + minimum, + maximum, + } => write!( + formatter, + "retention root {limit} prefix requires at least {minimum}; maximum is {maximum}" + ), + Self::BlobId { index, source } => identity_failure(formatter, "BlobId", *index, source), + Self::LayoutLengthPrefixAboveMaximum { + index, + minimum, + maximum, + } => write!( + formatter, + "retention anchor {index} layout-length prefix requires at least {minimum}; maximum is {maximum}" + ), Self::LayoutId { index, source } => { - write!( - formatter, - "invalid LayoutId in retention anchor {index}: {source}" - ) + identity_failure(formatter, "LayoutId", *index, source) } Self::NonCanonicalAnchorOrder { index, .. } => write!( formatter, @@ -91,6 +108,30 @@ impl fmt::Display for RetentionRootDecodeError { } } +fn prefix_failure( + formatter: &mut fmt::Formatter<'_>, + offset: usize, + expected: u8, + observed: u8, +) -> fmt::Result { + write!( + formatter, + "retention stage byte {offset} is {observed:#04x}; expected {expected:#04x}" + ) +} + +fn identity_failure( + formatter: &mut fmt::Formatter<'_>, + kind: &str, + index: u32, + source: &dyn Error, +) -> fmt::Result { + write!( + formatter, + "invalid {kind} in retention anchor {index}: {source}" + ) +} + impl Error for RetentionRootDecodeError { fn source(&self) -> Option<&(dyn Error + 'static)> { match self { @@ -104,6 +145,10 @@ impl Error for RetentionRootDecodeError { Self::Semantic { source } => Some(source), Self::Truncated { .. } | Self::TrailingData { .. } + | Self::PrefixByteMismatch { .. } + | Self::FramingPrefixImpossible { .. } + | Self::ClosureLimitPrefixAboveMaximum { .. } + | Self::LayoutLengthPrefixAboveMaximum { .. } | Self::InvalidMagic { .. } | Self::UnsupportedVersion { .. } | Self::InvalidHeaderLength { .. } diff --git a/src/adapters/retention/root_header_decoder.rs b/src/adapters/retention/root_header_decoder.rs index 7e0f5e9a..1b62f3e4 100644 --- a/src/adapters/retention/root_header_decoder.rs +++ b/src/adapters/retention/root_header_decoder.rs @@ -7,8 +7,8 @@ use super::root_field_decoder::{ }; pub(super) const HEADER_LENGTH: usize = 192; -const ANCHOR_WIDTH: usize = 119; -const TRAILER_LENGTH: usize = 64; +pub(super) const ANCHOR_WIDTH: usize = 119; +pub(super) const TRAILER_LENGTH: usize = 64; /// Longest canonical root: the header, a namespace at /// `RetentionNamespace::MAXIMUM_BYTE_LENGTH` (255), `RetentionRoot:: /// MAXIMUM_ANCHOR_COUNT` (65,536) anchors, and the trailer. Pinned against the @@ -87,6 +87,14 @@ fn validate_fixed_fields(encoded: &[u8]) -> Result<(), RetentionRootDecodeError> require_zero(encoded, 180, 12, "trailing header") } +/// Checks framing once the complete size fields of an interrupted header exist. +pub(super) fn admit_prefix_length(encoded: &[u8]) -> Result<(), RetentionRootDecodeError> { + let namespace_length = usize::from(read_u16(encoded, 40)?); + let anchor_count = read_u32(encoded, 44)?; + super::root_semantic_header::admit_count(anchor_count)?; + require_declared_length(encoded, canonical_length(namespace_length, anchor_count)?) +} + fn canonical_length( namespace_length: usize, anchor_count: u32, diff --git a/src/adapters/retention/root_integrity.rs b/src/adapters/retention/root_integrity.rs index d7537516..7fb11a11 100644 --- a/src/adapters/retention/root_integrity.rs +++ b/src/adapters/retention/root_integrity.rs @@ -81,3 +81,32 @@ fn hash(domain: &[u8], bytes: &[u8]) -> [u8; 32] { hasher.update(bytes); *hasher.finalize().as_bytes() } + +/// Verifies only digest bytes whose entire preimage has arrived. +pub(super) fn verify_prefix( + encoded: &[u8], + digest_offset: usize, + checksum_offset: usize, +) -> Result<(), RetentionRootDecodeError> { + for (offset, domain) in [ + ( + checksum_offset, + b"keep.retention-root-checksum/v2\0".as_slice(), + ), + (digest_offset, b"keep.retention-root/v2\0".as_slice()), + ] { + if let Some(preimage) = encoded.get(..offset) { + let expected = hash(domain, preimage); + super::stage_fixed_field_admission::admit( + encoded, + &[(offset, &expected)], + |offset, expected, observed| RetentionRootDecodeError::PrefixByteMismatch { + offset, + expected, + observed, + }, + )?; + } + } + Ok(()) +} diff --git a/src/adapters/retention/root_semantic_header.rs b/src/adapters/retention/root_semantic_header.rs index 17e9d9df..726af32f 100644 --- a/src/adapters/retention/root_semantic_header.rs +++ b/src/adapters/retention/root_semantic_header.rs @@ -1,4 +1,4 @@ -//! This boundary module owns post-integrity retention header admission. +//! This boundary module owns retention root header semantic admission. use super::RetentionRootDecodeError; use super::root_header_decoder::DecodedRootHeader; @@ -17,12 +17,7 @@ pub(super) struct AdmittedRootHeader { pub(super) fn admit( header: &DecodedRootHeader, ) -> Result { - if header.anchor_count > RetentionRoot::MAXIMUM_ANCHOR_COUNT { - return Err(RetentionRootDecodeError::AnchorCountExceeded { - maximum: RetentionRoot::MAXIMUM_ANCHOR_COUNT, - observed: header.anchor_count, - }); - } + admit_count(header.anchor_count)?; let generation = RootGeneration::new(header.generation) .map_err(|source| RetentionRootDecodeError::Generation { source })?; let profile = RegisteredRetentionProfile::admit( @@ -53,3 +48,14 @@ fn predecessor(bytes: [u8; 32]) -> Option { Some(RetentionRootDigest::from_hash(bytes)) } } + +/// Admits a complete count before either interrupted-stage discard or body allocation. +pub(super) const fn admit_count(anchor_count: u32) -> Result<(), RetentionRootDecodeError> { + if anchor_count > RetentionRoot::MAXIMUM_ANCHOR_COUNT { + return Err(RetentionRootDecodeError::AnchorCountExceeded { + maximum: RetentionRoot::MAXIMUM_ANCHOR_COUNT, + observed: anchor_count, + }); + } + Ok(()) +} diff --git a/src/adapters/retention/stage_closure_limit_admission.rs b/src/adapters/retention/stage_closure_limit_admission.rs new file mode 100644 index 00000000..5f1757e7 --- /dev/null +++ b/src/adapters/retention/stage_closure_limit_admission.rs @@ -0,0 +1,44 @@ +//! This boundary module owns admission of available big-endian closure-limit prefixes. + +use super::RetentionRootDecodeError as Error; +use crate::{RetentionClosureLimit as Limit, RetentionClosureLimitError, RetentionClosureLimits}; + +pub(super) fn admit(encoded: &[u8]) -> Result<(), Error> { + admit_field(encoded, 88, 8, Limit::Nodes)?; + admit_field(encoded, 96, 2, Limit::Depth)?; + admit_field(encoded, 100, 8, Limit::EncodedBytes)?; + admit_field(encoded, 108, 8, Limit::PhysicalBytes) +} + +fn admit_field(encoded: &[u8], offset: usize, width: usize, limit: Limit) -> Result<(), Error> { + let available = encoded.get(offset..).unwrap_or_default(); + let length = available.len().min(width); + let mut completed = [0_u8; 8]; + let padding = completed + .len() + .checked_sub(width) + .ok_or(Error::LengthOverflow)?; + let target = completed.get_mut(padding..).ok_or(Error::LengthOverflow)?; + for (target, source) in target.iter_mut().zip(available) { + *target = *source; + } + let minimum = u64::from_be_bytes(completed); + let admission = RetentionClosureLimits::admit_limit(limit, minimum); + if length == width { + return admission + .map(|_| ()) + .map_err(|source| Error::ClosureLimit { source }); + } + // An unfinished all-zero field still admits a positive completion. For + // any other prefix, its zero-filled minimum is a possible completion. + match admission { + Err(RetentionClosureLimitError::AboveMaximum { maximum, .. }) => { + Err(Error::ClosureLimitPrefixAboveMaximum { + limit, + minimum, + maximum, + }) + } + Ok(_) | Err(RetentionClosureLimitError::Zero { .. }) => Ok(()), + } +} diff --git a/src/adapters/retention/stage_fixed_field_admission.rs b/src/adapters/retention/stage_fixed_field_admission.rs new file mode 100644 index 00000000..5f927554 --- /dev/null +++ b/src/adapters/retention/stage_fixed_field_admission.rs @@ -0,0 +1,18 @@ +//! This boundary module owns bytewise admission of available fixed stage fields. + +pub(super) fn admit( + encoded: &[u8], + fields: &[(usize, &[u8])], + mismatch: impl Fn(usize, u8, u8) -> E, +) -> Result<(), E> { + for &(start, canonical) in fields { + for (&expected, (offset, &observed)) in + canonical.iter().zip(encoded.iter().enumerate().skip(start)) + { + if expected != observed { + return Err(mismatch(offset, expected, observed)); + } + } + } + Ok(()) +} diff --git a/src/adapters/retention/stage_framing_prefix.rs b/src/adapters/retention/stage_framing_prefix.rs new file mode 100644 index 00000000..ff43755b --- /dev/null +++ b/src/adapters/retention/stage_framing_prefix.rs @@ -0,0 +1,106 @@ +//! This boundary module owns feasibility of incomplete retention size-field combinations. + +use super::{manifest_header_decoder as manifest, root_header_decoder as root}; +use crate::{RetentionManifest, RetentionNamespace, RetentionRoot}; + +pub(super) fn root(encoded: &[u8]) -> Option<()> { + let namespace = Bounds::read::<2>(encoded, 40)? + .intersect(1, u64::from(RetentionNamespace::MAXIMUM_BYTE_LENGTH))?; + let count = Bounds::read::<4>(encoded, 44)? + .intersect(0, u64::from(RetentionRoot::MAXIMUM_ANCHOR_COUNT))?; + possible_length( + Bounds::read::<8>(encoded, 24)?, + namespace, + count, + u64::try_from(root::HEADER_LENGTH.checked_add(root::TRAILER_LENGTH)?).ok()?, + u64::try_from(root::ANCHOR_WIDTH).ok()?, + ) +} + +pub(super) fn manifest(encoded: &[u8]) -> Option<()> { + let count = Bounds::read::<4>(encoded, 44)? + .intersect(0, u64::from(RetentionManifest::MAXIMUM_ENTRY_COUNT))?; + manifest_length(Bounds::read::<8>(encoded, 24)?, count) +} + +pub(super) fn head(encoded: &[u8]) -> Option<()> { + manifest_length( + Bounds::read::<8>(encoded, 32)?, + Bounds { + minimum: 0, + maximum: u64::from(RetentionManifest::MAXIMUM_ENTRY_COUNT), + }, + ) +} + +fn manifest_length(length: Bounds, count: Bounds) -> Option<()> { + possible_length( + length, + Bounds { + minimum: 0, + maximum: 0, + }, + count, + u64::try_from(manifest::HEADER_LENGTH.checked_add(manifest::TRAILER_LENGTH)?).ok()?, + u64::try_from(manifest::ENTRY_WIDTH).ok()?, + ) +} + +/// Chooses the greatest count whose smallest record fits the length ceiling. +/// If its largest possible record cannot reach the length floor, no smaller +/// count can. Bounds are feasibility witnesses, never decoded observations. +fn possible_length( + length: Bounds, + extra: Bounds, + count: Bounds, + base: u64, + stride: u64, +) -> Option<()> { + let minimum_base = base.checked_add(extra.minimum)?; + let greatest_count = length + .maximum + .checked_sub(minimum_base)? + .checked_div(stride)? + .min(count.maximum); + if greatest_count < count.minimum { + return None; + } + let greatest_length = base + .checked_add(extra.maximum)? + .checked_add(greatest_count.checked_mul(stride)?)?; + (greatest_length >= length.minimum).then_some(()) +} + +#[derive(Clone, Copy)] +struct Bounds { + minimum: u64, + maximum: u64, +} + +impl Bounds { + fn read(encoded: &[u8], offset: usize) -> Option { + let available = encoded.get(offset..).unwrap_or_default(); + let mut minimum = [0_u8; WIDTH]; + let mut maximum = [u8::MAX; WIDTH]; + for ((low, high), observed) in minimum.iter_mut().zip(&mut maximum).zip(available) { + *low = *observed; + *high = *observed; + } + Some(Self { + minimum: integer(&minimum)?, + maximum: integer(&maximum)?, + }) + } + + fn intersect(self, minimum: u64, maximum: u64) -> Option { + let minimum = self.minimum.max(minimum); + let maximum = self.maximum.min(maximum); + (minimum <= maximum).then_some(Self { minimum, maximum }) + } +} + +fn integer(bytes: &[u8]) -> Option { + bytes.iter().try_fold(0_u64, |value, byte| { + value.checked_mul(256)?.checked_add(u64::from(*byte)) + }) +} diff --git a/src/adapters/retention/stage_generation_admission.rs b/src/adapters/retention/stage_generation_admission.rs new file mode 100644 index 00000000..e9b7b178 --- /dev/null +++ b/src/adapters/retention/stage_generation_admission.rs @@ -0,0 +1,31 @@ +//! This module owns admission of complete generation fields in interrupted stages. + +use super::{RetentionHeadDecodeError, RetentionManifestDecodeError, RetentionRootDecodeError}; +use crate::{LivenessGeneration, RootGeneration}; + +pub(super) fn root(encoded: &[u8]) -> Result<(), RetentionRootDecodeError> { + if encoded.len() >= 40 { + let value = super::root_field_decoder::read_u64(encoded, 32)?; + let _generation = RootGeneration::new(value) + .map_err(|source| RetentionRootDecodeError::Generation { source })?; + } + Ok(()) +} + +pub(super) fn manifest(encoded: &[u8]) -> Result<(), RetentionManifestDecodeError> { + if encoded.len() >= 40 { + let value = super::manifest_field_decoder::read_u64(encoded, 32)?; + let _generation = LivenessGeneration::new(value) + .map_err(|source| RetentionManifestDecodeError::LivenessGeneration { source })?; + } + Ok(()) +} + +pub(super) fn head(encoded: &[u8]) -> Result<(), RetentionHeadDecodeError> { + if encoded.len() >= 32 { + let value = super::head_decoder::read_u64(encoded, 24)?; + let _generation = LivenessGeneration::new(value) + .map_err(|source| RetentionHeadDecodeError::LivenessGeneration { source })?; + } + Ok(()) +} diff --git a/src/adapters/retention/stage_history_admission.rs b/src/adapters/retention/stage_history_admission.rs new file mode 100644 index 00000000..32f0c993 --- /dev/null +++ b/src/adapters/retention/stage_history_admission.rs @@ -0,0 +1,72 @@ +//! This boundary module owns admission of available interrupted-record history fields. + +use super::{RetentionHeadDecodeError, RetentionManifestDecodeError, RetentionRootDecodeError}; +use crate::{ + LivenessGeneration, RetentionManifest, RetentionManifestDigest, RetentionRoot, + RetentionRootDigest, RootGeneration, +}; + +pub(super) fn root(encoded: &[u8]) -> Result<(), RetentionRootDecodeError> { + let generation = RootGeneration::new(super::root_field_decoder::read_u64(encoded, 32)?) + .map_err(|source| RetentionRootDecodeError::Generation { source })?; + if encoded.len() < 148 { + if generation != RootGeneration::INITIAL { + return Ok(()); + } + return absent_predecessor(encoded, 116, |offset, expected, observed| { + RetentionRootDecodeError::PrefixByteMismatch { + offset, + expected, + observed, + } + }); + } + let bytes = super::root_field_decoder::read_array(encoded, 116)?; + let predecessor = (bytes != [0; 32]).then(|| RetentionRootDigest::from_hash(bytes)); + RetentionRoot::admit_predecessor(generation, predecessor) + .map_err(|source| RetentionRootDecodeError::Semantic { source }) +} + +pub(super) fn manifest(encoded: &[u8]) -> Result<(), RetentionManifestDecodeError> { + let generation = LivenessGeneration::new(super::manifest_field_decoder::read_u64(encoded, 32)?) + .map_err(|source| RetentionManifestDecodeError::LivenessGeneration { source })?; + if encoded.len() < 80 { + if generation != LivenessGeneration::INITIAL { + return Ok(()); + } + return absent_predecessor(encoded, 48, |offset, expected, observed| { + RetentionManifestDecodeError::PrefixByteMismatch { + offset, + expected, + observed, + } + }); + } + let bytes = super::manifest_field_decoder::read_array(encoded, 48)?; + let predecessor = (bytes != [0; 32]).then(|| RetentionManifestDigest::from_hash(bytes)); + RetentionManifest::admit_predecessor(generation, predecessor) + .map_err(|source| RetentionManifestDecodeError::Semantic { source }) +} + +/// Admits an incomplete predecessor after the complete head generation arrived. +pub(super) fn head_prefix(encoded: &[u8]) -> Result<(), RetentionHeadDecodeError> { + let generation = super::head_decoder::read_u64(encoded, 24)?; + if generation != LivenessGeneration::INITIAL.get() { + return Ok(()); + } + absent_predecessor(encoded, 72, |offset, expected, observed| { + RetentionHeadDecodeError::PrefixByteMismatch { + offset, + expected, + observed, + } + }) +} + +fn absent_predecessor( + encoded: &[u8], + start: usize, + mismatch: impl Fn(usize, u8, u8) -> E, +) -> Result<(), E> { + super::stage_fixed_field_admission::admit(encoded, &[(start, &[0; 32])], mismatch) +} diff --git a/src/adapters/retention/stage_prefix_admission.rs b/src/adapters/retention/stage_prefix_admission.rs new file mode 100644 index 00000000..665e7882 --- /dev/null +++ b/src/adapters/retention/stage_prefix_admission.rs @@ -0,0 +1,120 @@ +//! This module owns admission of available retention stage fields. + +use super::stage_fixed_field_admission::admit; +use super::{RetentionHeadDecodeError, RetentionManifestDecodeError, RetentionRootDecodeError}; + +pub(super) fn root(encoded: &[u8]) -> Result<(), RetentionRootDecodeError> { + admit( + encoded, + &[ + (0, b"KEEP:RET:ROOT2\0\0"), + (16, &2_u16.to_be_bytes()), + (18, &192_u16.to_be_bytes()), + (20, &[0; 4]), + (42, &119_u16.to_be_bytes()), + (98, &[0; 2]), + (180, &[0; 12]), + ], + |offset, expected, observed| RetentionRootDecodeError::PrefixByteMismatch { + offset, + expected, + observed, + }, + )?; + super::stage_generation_admission::root(encoded)?; + super::stage_root_policy_admission::admit(encoded)?; + if encoded.len() >= 42 { + let length = usize::from(super::root_field_decoder::read_u16(encoded, 40)?); + let _length = crate::RetentionNamespace::admit_length(length) + .map_err(|source| RetentionRootDecodeError::Namespace { source })?; + } + if encoded.len() >= 48 { + super::root_header_decoder::admit_prefix_length(encoded)?; + } else if super::stage_framing_prefix::root(encoded) != Some(()) { + return Err(RetentionRootDecodeError::FramingPrefixImpossible { + observed: encoded.len(), + }); + } + if encoded.len() >= 116 { + super::stage_history_admission::root(encoded)?; + } + super::stage_record_integrity::root(encoded) +} + +pub(super) fn manifest(encoded: &[u8]) -> Result<(), RetentionManifestDecodeError> { + admit( + encoded, + &[ + (0, b"KEEP:RET:LIVE2\0\0"), + (16, &2_u16.to_be_bytes()), + (18, &160_u16.to_be_bytes()), + (20, &[0; 4]), + (40, &72_u16.to_be_bytes()), + (42, &[0; 2]), + (112, &[0; 48]), + ], + |offset, expected, observed| RetentionManifestDecodeError::PrefixByteMismatch { + offset, + expected, + observed, + }, + )?; + super::stage_generation_admission::manifest(encoded)?; + if encoded.len() >= 48 { + super::manifest_header_decoder::admit_prefix_length(encoded)?; + } else if super::stage_framing_prefix::manifest(encoded) != Some(()) { + return Err(RetentionManifestDecodeError::FramingPrefixImpossible { + observed: encoded.len(), + }); + } + if encoded.len() >= 48 { + super::stage_history_admission::manifest(encoded)?; + } + super::stage_record_integrity::manifest(encoded) +} + +pub(super) fn head(encoded: &[u8]) -> Result<(), RetentionHeadDecodeError> { + admit( + encoded, + &[ + (0, &super::head_decoder::MAGIC), + (16, &super::head_decoder::VERSION.to_be_bytes()), + (18, &super::head_decoder::RECORD_LENGTH.to_be_bytes()), + (20, &[0; 4]), + (104, &[0; 8]), + ], + |offset, expected, observed| RetentionHeadDecodeError::PrefixByteMismatch { + offset, + expected, + observed, + }, + )?; + super::stage_generation_admission::head(encoded)?; + if encoded.len() >= 40 { + let value = super::head_decoder::read_u64(encoded, 32)?; + let _length = crate::RetentionManifestLength::new(value) + .map_err(|source| RetentionHeadDecodeError::ManifestLength { source })?; + } else if super::stage_framing_prefix::head(encoded) != Some(()) { + return Err(RetentionHeadDecodeError::ManifestLengthPrefixImpossible { + observed: encoded.len(), + }); + } + if encoded.len() >= 104 { + let _head = super::head_decoder::admit_fields(encoded)?; + } else if encoded.len() >= 72 { + super::stage_history_admission::head_prefix(encoded)?; + } + if let Some(preimage) = encoded.get(..super::head_decoder::CHECKSUM_OFFSET) { + let checksum = super::head_decoder::checksum(preimage); + admit( + encoded, + &[(super::head_decoder::CHECKSUM_OFFSET, &checksum)], + |offset, expected, observed| RetentionHeadDecodeError::PrefixByteMismatch { + offset, + expected, + observed, + }, + )?; + } + Ok(()) +} diff --git a/src/adapters/retention/stage_record_integrity.rs b/src/adapters/retention/stage_record_integrity.rs new file mode 100644 index 00000000..dd8a7daa --- /dev/null +++ b/src/adapters/retention/stage_record_integrity.rs @@ -0,0 +1,58 @@ +//! This boundary module owns integrity admission after interrupted record framing admits. + +use super::{RetentionManifestDecodeError as ManifestError, RetentionRootDecodeError as RootError}; + +pub(super) fn root(encoded: &[u8]) -> Result<(), RootError> { + if encoded.len() < super::root_header_decoder::HEADER_LENGTH { + return Ok(()); + } + let total = usize::try_from(super::root_field_decoder::read_u64(encoded, 24)?) + .map_err(|_| RootError::LengthOverflow)?; + let checksum_offset = total.checked_sub(32).ok_or(RootError::LengthOverflow)?; + let digest_offset = checksum_offset + .checked_sub(32) + .ok_or(RootError::LengthOverflow)?; + super::root_integrity::verify_prefix(encoded, digest_offset, checksum_offset)?; + let namespace_length = usize::from(super::root_field_decoder::read_u16(encoded, 40)?); + let body_start = super::root_header_decoder::HEADER_LENGTH + .checked_add(namespace_length) + .ok_or(RootError::LengthOverflow)?; + if let Some(anchors) = encoded.get(body_start..digest_offset) { + let _digest = super::root_integrity::verify_anchor_set( + super::root_field_decoder::read_u32(encoded, 44)?, + anchors, + super::root_field_decoder::read_array(encoded, 148)?, + )?; + } + if let Some(anchors) = encoded.get(body_start..encoded.len().min(digest_offset)) { + super::root_anchor_decoder::admit_prefix(anchors, body_start)?; + } + Ok(()) +} + +pub(super) fn manifest(encoded: &[u8]) -> Result<(), ManifestError> { + if encoded.len() < super::manifest_header_decoder::HEADER_LENGTH { + return Ok(()); + } + let total = usize::try_from(super::manifest_field_decoder::read_u64(encoded, 24)?) + .map_err(|_| ManifestError::LengthOverflow)?; + let checksum_offset = total.checked_sub(32).ok_or(ManifestError::LengthOverflow)?; + let digest_offset = checksum_offset + .checked_sub(32) + .ok_or(ManifestError::LengthOverflow)?; + super::manifest_integrity::verify_prefix(encoded, digest_offset, checksum_offset)?; + if let Some(entries) = encoded.get(super::manifest_header_decoder::HEADER_LENGTH..digest_offset) + { + super::manifest_integrity::verify_entry_set( + super::manifest_field_decoder::read_u32(encoded, 44)?, + entries, + super::manifest_field_decoder::read_array(encoded, 80)?, + )?; + } + if let Some(entries) = + encoded.get(super::manifest_header_decoder::HEADER_LENGTH..encoded.len().min(digest_offset)) + { + super::manifest_entry_decoder::admit_prefix(entries)?; + } + Ok(()) +} diff --git a/src/adapters/retention/stage_root_policy_admission.rs b/src/adapters/retention/stage_root_policy_admission.rs new file mode 100644 index 00000000..a8328ef7 --- /dev/null +++ b/src/adapters/retention/stage_root_policy_admission.rs @@ -0,0 +1,39 @@ +//! This boundary module owns admission of policy groups and partial profile fields in interrupted roots. + +use super::RetentionRootDecodeError; +use super::root_field_decoder::{read_array, read_u32}; +use crate::RegisteredRetentionProfile; + +pub(super) fn admit(encoded: &[u8]) -> Result<(), RetentionRootDecodeError> { + if encoded.len() < 88 { + admit_profile_prefix(encoded)?; + } + if encoded.len() >= 88 { + let _profile = RegisteredRetentionProfile::admit( + read_u32(encoded, 48)?, + read_u32(encoded, 52)?, + read_array(encoded, 56)?, + ) + .map_err(|source| RetentionRootDecodeError::Profile { source })?; + } + super::stage_closure_limit_admission::admit(encoded) +} + +fn admit_profile_prefix(encoded: &[u8]) -> Result<(), RetentionRootDecodeError> { + // The registry is currently closed to this one exact profile; unavailable + // bytes remain unknown, while every available byte must admit a completion. + let profile = RegisteredRetentionProfile::SINGLE_CANONICAL_WITNESS_V1; + super::stage_fixed_field_admission::admit( + encoded, + &[ + (48, &profile.identity().to_be_bytes()), + (52, &profile.version().to_be_bytes()), + (56, profile.digest()), + ], + |offset, expected, observed| RetentionRootDecodeError::PrefixByteMismatch { + offset, + expected, + observed, + }, + ) +} diff --git a/src/adapters/store_migration/filesystem_migration_remount_tests.rs b/src/adapters/store_migration/filesystem_migration_remount_tests.rs index 208c6dd9..3a9a3860 100644 --- a/src/adapters/store_migration/filesystem_migration_remount_tests.rs +++ b/src/adapters/store_migration/filesystem_migration_remount_tests.rs @@ -127,5 +127,5 @@ fn reopen(root: &Path) -> Result Result { - FilesystemVersionTwoAdmission::reopen_unchecked_for_tests(root) + FilesystemVersionTwoAdmission::reopen_unchecked_for_repository_tasks(root) } diff --git a/src/adapters/store_migration/filesystem_migration_repository_tasks.rs b/src/adapters/store_migration/filesystem_migration_repository_tasks.rs index 3d91ff05..e7579f3e 100644 --- a/src/adapters/store_migration/filesystem_migration_repository_tasks.rs +++ b/src/adapters/store_migration/filesystem_migration_repository_tasks.rs @@ -28,14 +28,21 @@ impl FilesystemStoreMigrationAuthority { /// /// # Errors /// - /// Returns the same admission and pool failures as [`Self::open`]. + /// Returns [`Error::Namespace`] when the root capability cannot be cloned, + /// [`Error::RootIdentity`] when its identity cannot be observed, or the + /// inventory failures from [`Self::open`]. No platform admission is performed. #[doc(hidden)] pub fn open_unchecked_for_repository_tasks( lock: FilesystemWriterLock, policy: SegmentReadPolicy, ) -> Result { - let admission = FilesystemPlatformAdmission::unchecked_for_repository_tasks(lock) - .map_err(|source| Error::Platform { source })?; + let root = lock + .clone_directory() + .map_err(|source| Error::Namespace { source })?; + let identity = crate::adapters::filesystem_platform_profile::root_identity_lenient(&root) + .map_err(|source| Error::RootIdentity { source })?; + drop(root); + let admission = FilesystemPlatformAdmission::initialized(lock, identity); let inventory = FilesystemStoreMigrationInventoryReader::open(admission, policy) .map_err(|source| Error::Inventory { source })?; Ok(Self::with_policy( diff --git a/src/layout/record_length.rs b/src/layout/record_length.rs index d01c1209..2d65f033 100644 --- a/src/layout/record_length.rs +++ b/src/layout/record_length.rs @@ -32,6 +32,15 @@ impl LayoutRecordLength { Some(Self(value)) } + /// Returns the greatest canonical length no greater than a boundary's bound. + pub(crate) fn greatest_at_most(upper: u64) -> Option { + let bounded = upper.min(Self::MAXIMUM.get()); + let remainder = bounded + .checked_sub(HEADER_AND_CHECKSUM_LENGTH)? + .checked_rem(ENTRY_LENGTH)?; + Self::from_wire(bounded.checked_sub(remainder)?) + } + /// Returns the exact encoded byte count. #[must_use] pub const fn get(self) -> u64 { diff --git a/src/lib.rs b/src/lib.rs index 2e0ffcef..1127d095 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -137,15 +137,27 @@ pub use adapters::{ AdmittedRetentionManifest, AdmittedRetentionRoot, CanonicalRetentionHead, CanonicalRetentionManifest, CanonicalRetentionRoot, ChecksummedRetentionHead, FilesystemRetentionAuthorityError, FilesystemRetentionPublicationAuthority, - ObservedRetentionState, PreparedRetentionPublication, RetentionAuthorityDirectory, - RetentionClosureVerificationError, RetentionCurrentStateRefusal, RetentionHeadDecodeError, - RetentionManifestDecodeError, RetentionManifestEncodeError, RetentionNamespaceAdmission, - RetentionPublicationError, RetentionPublicationOutcome, RetentionPublicationPhase, - RetentionPublicationPreparation, RetentionPublicationPreparationError, - RetentionPublicationReceipt, RetentionPublicationStorage, RetentionRootDecodeError, - RetentionRootEncodeError, RetentionTransitionDisposition, RetentionTransitionError, - RetentionTransitionPreflight, RetentionTransitionPreflightError, RetentionTransitionReadiness, - VerifiedRetentionClosure, execute_retention_publication, plan_retention_transition, + FilesystemRetentionRecoveryError, FilesystemRetentionSnapshot, + FilesystemRetentionSnapshotError, ObservedRetentionState, PreparedRetentionPublication, + ReaderAttemptLimit, RetentionAuthorityDirectory, RetentionClosureVerificationError, + RetentionCurrentStateRefusal, RetentionEffectDurability, RetentionFixedStage, + RetentionHeadDecodeError, RetentionHeadStageAssessment, RetentionKnownEffect, + RetentionManifestDecodeError, RetentionManifestEncodeError, RetentionManifestStageAssessment, + RetentionNamespaceAdmission, RetentionNamespaceEffect, RetentionPool, + RetentionPoolEntryObservation, RetentionPoolObservations, RetentionPublicationError, + RetentionPublicationOutcome, RetentionPublicationPhase, RetentionPublicationPreparation, + RetentionPublicationPreparationError, RetentionPublicationReceipt, RetentionPublicationStorage, + RetentionRecordRefusal, RetentionRecoveryError, RetentionRecoveryEvidence, + RetentionRecoveryOutcome, RetentionRecoveryPlan, RetentionRecoveryReceipt, + RetentionRecoveryRefusal, RetentionRecoveryStep, RetentionRecoveryStorage, + RetentionRootDecodeError, RetentionRootEncodeError, RetentionRootStageAssessment, + RetentionStageAssessment, RetentionStageAssessments, RetentionStorageBoundary, + RetentionStorageError, RetentionStorageProgress, RetentionTransitionDisposition, + RetentionTransitionError, RetentionTransitionPreflight, RetentionTransitionPreflightError, + RetentionTransitionReadiness, RetentionViewCoordinates, RetentionViewError, + RetentionViewSource, VerifiedRetentionClosure, assess_head_stage, assess_manifest_stage, + assess_root_stage, collect_retention_view, execute_retention_publication, + execute_retention_recovery, plan_retention_recovery, plan_retention_transition, preflight_retention_transition, prepare_retention_publication, verify_retention_closure, }; pub use adapters::{ diff --git a/src/retention/closure_limits.rs b/src/retention/closure_limits.rs index ad73135f..bdbefb54 100644 --- a/src/retention/closure_limits.rs +++ b/src/retention/closure_limits.rs @@ -35,18 +35,11 @@ impl RetentionClosureLimits { encoded_bytes: u64, physical_bytes: u64, ) -> Result { - let nodes = admit_u64(RetentionClosureLimit::Nodes, nodes, Self::MAXIMUM_NODES)?; + let nodes = Self::admit_limit(RetentionClosureLimit::Nodes, nodes)?; let depth = admit_depth(depth)?; - let encoded_bytes = admit_u64( - RetentionClosureLimit::EncodedBytes, - encoded_bytes, - Self::MAXIMUM_ENCODED_BYTES, - )?; - let physical_bytes = admit_u64( - RetentionClosureLimit::PhysicalBytes, - physical_bytes, - Self::MAXIMUM_PHYSICAL_BYTES, - )?; + let encoded_bytes = Self::admit_limit(RetentionClosureLimit::EncodedBytes, encoded_bytes)?; + let physical_bytes = + Self::admit_limit(RetentionClosureLimit::PhysicalBytes, physical_bytes)?; Ok(Self { nodes, depth, @@ -55,6 +48,20 @@ impl RetentionClosureLimits { }) } + /// Admits one resource independently for interrupted boundary records. + pub(crate) fn admit_limit( + limit: RetentionClosureLimit, + observed: u64, + ) -> Result { + let maximum = match limit { + RetentionClosureLimit::Nodes => Self::MAXIMUM_NODES, + RetentionClosureLimit::Depth => u64::from(Self::MAXIMUM_DEPTH), + RetentionClosureLimit::EncodedBytes => Self::MAXIMUM_ENCODED_BYTES, + RetentionClosureLimit::PhysicalBytes => Self::MAXIMUM_PHYSICAL_BYTES, + }; + admit_u64(limit, observed, maximum) + } + /// Returns the positive closure node limit. #[must_use] pub const fn nodes(self) -> u64 { @@ -98,13 +105,6 @@ fn admit_u64( fn admit_depth(observed: u16) -> Result { let limit = RetentionClosureLimit::Depth; - let value = NonZeroU16::new(observed).ok_or(RetentionClosureLimitError::Zero { limit })?; - if observed > RetentionClosureLimits::MAXIMUM_DEPTH { - return Err(RetentionClosureLimitError::AboveMaximum { - limit, - maximum: u64::from(RetentionClosureLimits::MAXIMUM_DEPTH), - observed: u64::from(observed), - }); - } - Ok(value) + let _admitted = RetentionClosureLimits::admit_limit(limit, u64::from(observed))?; + NonZeroU16::new(observed).ok_or(RetentionClosureLimitError::Zero { limit }) } diff --git a/src/retention/manifest.rs b/src/retention/manifest.rs index 5095cdb9..a665436c 100644 --- a/src/retention/manifest.rs +++ b/src/retention/manifest.rs @@ -33,7 +33,7 @@ impl RetentionManifest { predecessor: Option, mut entries: Vec, ) -> Result { - admit_predecessor(generation, predecessor)?; + Self::admit_predecessor(generation, predecessor)?; let observed = entries.len(); let entry_count = u32::try_from(observed).map_err(|_| RetentionManifestError::EntryCountExceeded { @@ -77,19 +77,22 @@ impl RetentionManifest { } } -fn admit_predecessor( - generation: LivenessGeneration, - predecessor: Option, -) -> Result<(), RetentionManifestError> { - if generation.get() == 1 { - return predecessor.map_or(Ok(()), |observed| { - Err(RetentionManifestError::InitialGenerationHasPredecessor { observed }) - }); - } - if predecessor.is_some() { - Ok(()) - } else { - Err(RetentionManifestError::MissingPredecessor { generation }) +impl RetentionManifest { + /// Admits generation history independently of an unavailable record body. + pub(crate) fn admit_predecessor( + generation: LivenessGeneration, + predecessor: Option, + ) -> Result<(), RetentionManifestError> { + if generation.get() == 1 { + return predecessor.map_or(Ok(()), |observed| { + Err(RetentionManifestError::InitialGenerationHasPredecessor { observed }) + }); + } + if predecessor.is_some() { + Ok(()) + } else { + Err(RetentionManifestError::MissingPredecessor { generation }) + } } } diff --git a/src/retention/namespace.rs b/src/retention/namespace.rs index 01063b6d..bf04228a 100644 --- a/src/retention/namespace.rs +++ b/src/retention/namespace.rs @@ -44,7 +44,8 @@ impl RetentionNamespace { RetentionNamespaceDigest::from_hash(*hasher.finalize().as_bytes()) } - fn admit_length(observed: usize) -> Result { + /// Admits a declared namespace length without allocating or requiring its bytes. + pub(crate) fn admit_length(observed: usize) -> Result { let length = u8::try_from(observed).map_err(|_| RetentionNamespaceError::TooLong { maximum: Self::MAXIMUM_BYTE_LENGTH, observed, diff --git a/src/retention/root.rs b/src/retention/root.rs index 964876c7..a642b31d 100644 --- a/src/retention/root.rs +++ b/src/retention/root.rs @@ -40,7 +40,7 @@ impl RetentionRoot { predecessor: Option, mut anchors: Vec, ) -> Result { - admit_predecessor(generation, predecessor)?; + Self::admit_predecessor(generation, predecessor)?; let observed = anchors.len(); let anchor_count = u32::try_from(observed).map_err(|_| RetentionRootError::AnchorCountExceeded { @@ -103,16 +103,19 @@ impl RetentionRoot { } } -const fn admit_predecessor( - generation: RootGeneration, - predecessor: Option, -) -> Result<(), RetentionRootError> { - match (generation.get(), predecessor) { - (1, Some(observed)) => { - Err(RetentionRootError::InitialGenerationHasPredecessor { observed }) +impl RetentionRoot { + /// Admits generation history independently of an unavailable record body. + pub(crate) const fn admit_predecessor( + generation: RootGeneration, + predecessor: Option, + ) -> Result<(), RetentionRootError> { + match (generation.get(), predecessor) { + (1, Some(observed)) => { + Err(RetentionRootError::InitialGenerationHasPredecessor { observed }) + } + (1, None) | (_, Some(_)) => Ok(()), + (_, None) => Err(RetentionRootError::MissingPredecessor { generation }), } - (1, None) | (_, Some(_)) => Ok(()), - (_, None) => Err(RetentionRootError::MissingPredecessor { generation }), } } diff --git a/tests/migration_descriptor_exhaustion.rs b/tests/migration_descriptor_exhaustion.rs new file mode 100644 index 00000000..fb1af637 --- /dev/null +++ b/tests/migration_descriptor_exhaustion.rs @@ -0,0 +1,85 @@ +//! This module owns migration admission diagnostics under descriptor exhaustion. + +#![cfg(all(feature = "repository-tasks", target_os = "linux"))] + +#[path = "segment_filesystem_stage/sandbox.rs"] +pub mod sandbox; + +use std::error::Error; +use std::fs::File; +use std::process::Command; + +use keep::{ + FilesystemMigrationAuthorityError as AuthorityError, FilesystemStoreMigrationAuthority, + RepositoryInitializationStorage, SegmentReadPolicy, initialize_store, +}; + +const CHILD: &str = "KEEP_MIGRATION_DESCRIPTOR_CHILD"; +const TEST: &str = "a_root_clone_failure_reports_the_namespace_boundary"; + +// Size: medium. Oracle: capability duplication failure is Namespace with its original EMFILE. +// The isolated child owns a 64-descriptor ceiling and a 20-second execution ceiling. +// Delete when repository-task migration admission is removed or stronger kernel-fault coverage subsumes it. +#[test] +fn a_root_clone_failure_reports_the_namespace_boundary() -> Result<(), Box> { + if std::env::var_os(CHILD).is_some() { + return exhaust_descriptors(); + } + let output = Command::new("timeout") + .args([ + "20s", + "/bin/sh", + "-c", + "ulimit -n 64; exec \"$@\"", + "migration-descriptor-test", + ]) + .arg(std::env::current_exe()?) + .args(["--exact", TEST, "--nocapture", "--test-threads=1"]) + .env(CHILD, "1") + .output()?; + assert!( + output.status.success(), + "isolated migration clone-failure law failed: {:?}\n{}\n{}", + output.status.code(), + String::from_utf8_lossy(&output.stdout), + String::from_utf8_lossy(&output.stderr) + ); + Ok(()) +} + +fn exhaust_descriptors() -> Result<(), Box> { + let sandbox = sandbox::TestDirectory::create("migration-clone-descriptor-exhaustion")?; + let mut storage = RepositoryInitializationStorage::admit_unchecked(sandbox.path())?; + let _initialized = initialize_store(&mut storage)?; + let lock = storage.into_writer_lock()?; + let descriptors = fill_descriptors()?; + let result = FilesystemStoreMigrationAuthority::open_unchecked_for_repository_tasks( + lock, + SegmentReadPolicy::MAXIMUM, + ); + drop(descriptors); + let error = result + .err() + .ok_or("descriptor exhaustion unexpectedly admitted migration")?; + assert!( + matches!(error, AuthorityError::Namespace { ref source } + if source.raw_os_error() == Some(rustix::io::Errno::MFILE.raw_os_error())), + "root clone failure must retain Namespace and EMFILE: {error:?}" + ); + sandbox.remove()?; + Ok(()) +} + +fn fill_descriptors() -> Result, Box> { + let mut descriptors = Vec::with_capacity(64); + for _ in 0..128 { + match File::open("/dev/null") { + Ok(file) => descriptors.push(file), + Err(error) if error.raw_os_error() == Some(rustix::io::Errno::MFILE.raw_os_error()) => { + return Ok(descriptors); + } + Err(error) => return Err(error.into()), + } + } + Err("the child descriptor ceiling was not enforced".into()) +} diff --git a/tests/retention_core_architecture_contract.rs b/tests/retention_core_architecture_contract.rs new file mode 100644 index 00000000..3b75430d --- /dev/null +++ b/tests/retention_core_architecture_contract.rs @@ -0,0 +1,61 @@ +//! The retention core admits no caller identity, path, clock, or application policy. + +use std::error::Error; +use std::fs; +use std::path::Path; + +/// Tokens that would let identity, paths, clocks, or environment into a +/// transition decision. The filesystem adapters own paths; the core does not. +const FORBIDDEN: [&str; 8] = [ + "SystemTime", + "Instant", + "std::env", + "std::path", + "std::fs", + "getuid", + "hostname", + "username", +]; + +/// Storage-independent retention modules outside `src/retention/`. +const CORE_ADAPTERS: [&str; 8] = [ + "src/adapters/retention/transition_planner.rs", + "src/adapters/retention/transition_preflight.rs", + "src/adapters/retention/publication_preparation.rs", + "src/adapters/retention/publication_execution.rs", + "src/adapters/retention/publication_storage.rs", + "src/adapters/retention/recovery_planner.rs", + "src/adapters/retention/recovery_execution.rs", + "src/adapters/retention/retention_view_collector.rs", +]; + +#[test] +fn the_retention_core_admits_no_identity_path_clock_or_policy() -> Result<(), Box> { + let root = Path::new(env!("CARGO_MANIFEST_DIR")); + let mut sources = Vec::new(); + for entry in fs::read_dir(root.join("src/retention"))? { + let path = entry?.path(); + if path.extension().is_some_and(|extension| extension == "rs") { + sources.push(path); + } + } + for adapter in CORE_ADAPTERS { + let path = root.join(adapter); + assert!( + path.is_file(), + "{adapter} is missing; update the contract list" + ); + sources.push(path); + } + for path in sources { + let source = fs::read_to_string(&path)?; + for token in FORBIDDEN { + assert!( + !source.contains(token), + "{} names `{token}`; the retention core decides from evidence alone", + path.display() + ); + } + } + Ok(()) +} diff --git a/tests/retention_stage_generation.rs b/tests/retention_stage_generation.rs new file mode 100644 index 00000000..b93f6816 --- /dev/null +++ b/tests/retention_stage_generation.rs @@ -0,0 +1,82 @@ +//! Interrupted retention stages cannot admit a complete zero generation. +//! Size: small; oracle: positive root/liveness generation domain contracts. +//! Delete only if a stronger public assessment test subsumes these refusals. + +mod support; + +use std::error::Error; + +use keep::{ + LivenessGenerationError, RetentionHeadDecodeError, RetentionManifestDecodeError, + RetentionRootDecodeError, RetentionStageAssessment, RootGenerationError, assess_head_stage, + assess_manifest_stage, assess_root_stage, +}; + +const ROOT: &str = include_str!("../conformance/segment-store/v2/one-anchor-root.hex"); +const MANIFEST: &str = include_str!("../conformance/segment-store/v2/one-root-manifest.hex"); +const HEAD: &str = include_str!("../conformance/segment-store/v2/one-root-head.hex"); + +#[test] +fn a_complete_zero_root_generation_refuses_every_later_interrupted_prefix() +-> Result<(), Box> { + let mut bytes = support::decode_hex(ROOT.trim_end())?; + bytes.get_mut(32..40).ok_or("missing generation")?.fill(0); + for end in 40..bytes.len() { + let prefix = bytes.get(..end).ok_or("missing root prefix")?; + let assessment = assess_root_stage(Some(prefix)); + assert!( + matches!( + assessment, + RetentionStageAssessment::Corrupt(RetentionRootDecodeError::Generation { + source: RootGenerationError::Zero + }) + ), + "zero root generation at prefix {end} must refuse, observed {assessment:?}" + ); + } + Ok(()) +} + +#[test] +fn a_complete_zero_manifest_generation_refuses_every_later_interrupted_prefix() +-> Result<(), Box> { + let mut bytes = support::decode_hex(MANIFEST.trim_end())?; + bytes.get_mut(32..40).ok_or("missing generation")?.fill(0); + for end in 40..bytes.len() { + let prefix = bytes.get(..end).ok_or("missing manifest prefix")?; + let assessment = assess_manifest_stage(Some(prefix)); + assert!( + matches!( + assessment, + RetentionStageAssessment::Corrupt( + RetentionManifestDecodeError::LivenessGeneration { + source: LivenessGenerationError::Zero + } + ) + ), + "zero manifest generation at prefix {end} must refuse, observed {assessment:?}" + ); + } + Ok(()) +} + +#[test] +fn a_complete_zero_head_generation_refuses_every_later_interrupted_prefix() +-> Result<(), Box> { + let mut bytes = support::decode_hex(HEAD.trim_end())?; + bytes.get_mut(24..32).ok_or("missing generation")?.fill(0); + for end in 32..bytes.len() { + let prefix = bytes.get(..end).ok_or("missing head prefix")?; + let assessment = assess_head_stage(Some(prefix)); + assert!( + matches!( + assessment, + RetentionStageAssessment::Corrupt(RetentionHeadDecodeError::LivenessGeneration { + source: LivenessGenerationError::Zero + }) + ), + "zero head generation at prefix {end} must refuse, observed {assessment:?}" + ); + } + Ok(()) +} diff --git a/tests/retention_stage_prefix.rs b/tests/retention_stage_prefix.rs new file mode 100644 index 00000000..356ffff8 --- /dev/null +++ b/tests/retention_stage_prefix.rs @@ -0,0 +1,219 @@ +//! Recovery assessment obeys the version-two retention fixed-field grammar. +//! Size: small; oracle: conformance records and the retention format grammar. +//! Delete only when a stronger public-contract test subsumes these prefix laws. + +mod support; + +use std::error::Error; + +use keep::{RetentionStageAssessment, assess_head_stage, assess_manifest_stage, assess_root_stage}; + +const ROOT: &str = include_str!("../conformance/segment-store/v2/one-anchor-root.hex"); +const MANIFEST: &str = include_str!("../conformance/segment-store/v2/one-root-manifest.hex"); +const HEAD: &str = include_str!("../conformance/segment-store/v2/one-root-head.hex"); + +#[test] +fn canonical_root_prefixes_remain_recoverable_interrupted_writes() -> Result<(), Box> { + let bytes = support::decode_hex(ROOT.trim_end())?; + for end in 0..bytes.len() { + let prefix = bytes.get(..end).ok_or("missing root prefix")?; + let assessment = assess_root_stage(Some(prefix)); + assert!( + matches!(assessment, RetentionStageAssessment::Truncated { .. }), + "root prefix {end} must be truncated, observed {assessment:?}" + ); + } + Ok(()) +} + +#[test] +fn canonical_manifest_prefixes_remain_recoverable_interrupted_writes() -> Result<(), Box> +{ + let bytes = support::decode_hex(MANIFEST.trim_end())?; + for end in 0..bytes.len() { + let prefix = bytes.get(..end).ok_or("missing manifest prefix")?; + let assessment = assess_manifest_stage(Some(prefix)); + assert!( + matches!(assessment, RetentionStageAssessment::Truncated { .. }), + "manifest prefix {end} must be truncated, observed {assessment:?}" + ); + } + Ok(()) +} + +#[test] +fn canonical_head_prefixes_remain_recoverable_interrupted_writes() -> Result<(), Box> { + let bytes = support::decode_hex(HEAD.trim_end())?; + for end in 0..bytes.len() { + let prefix = bytes.get(..end).ok_or("missing head prefix")?; + let assessment = assess_head_stage(Some(prefix)); + assert!( + matches!(assessment, RetentionStageAssessment::Truncated { .. }), + "head prefix {end} must be truncated, observed {assessment:?}" + ); + } + Ok(()) +} + +#[test] +fn contradictory_root_fixed_bytes_are_corrupt_at_every_interrupted_length() +-> Result<(), Box> { + let bytes = support::decode_hex(ROOT.trim_end())?; + let mut witnesses = 0; + for offset in (0..24).chain(42..44).chain(98..100).chain(180..192) { + let mut altered = bytes.clone(); + *altered.get_mut(offset).ok_or("missing root fixed byte")? ^= 1; + for end in offset.checked_add(1).ok_or("prefix overflow")?..bytes.len() { + let prefix = altered.get(..end).ok_or("missing root prefix")?; + let assessment = assess_root_stage(Some(prefix)); + assert!( + matches!(assessment, RetentionStageAssessment::Corrupt(_)), + "root byte {offset}, prefix {end} must be corrupt, observed {assessment:?}" + ); + witnesses += 1; + } + } + assert!( + witnesses > 0, + "root corruption law must examine actual prefixes" + ); + Ok(()) +} + +#[test] +fn contradictory_manifest_fixed_bytes_are_corrupt_at_every_interrupted_length() +-> Result<(), Box> { + let bytes = support::decode_hex(MANIFEST.trim_end())?; + let mut witnesses = 0; + for offset in (0..24).chain(40..44).chain(112..160) { + let mut altered = bytes.clone(); + *altered + .get_mut(offset) + .ok_or("missing manifest fixed byte")? ^= 1; + for end in offset.checked_add(1).ok_or("prefix overflow")?..bytes.len() { + let prefix = altered.get(..end).ok_or("missing manifest prefix")?; + let assessment = assess_manifest_stage(Some(prefix)); + assert!( + matches!(assessment, RetentionStageAssessment::Corrupt(_)), + "manifest byte {offset}, prefix {end} must be corrupt, observed {assessment:?}" + ); + witnesses += 1; + } + } + assert!( + witnesses > 0, + "manifest corruption law must examine actual prefixes" + ); + Ok(()) +} + +#[test] +fn contradictory_head_fixed_bytes_are_corrupt_at_every_interrupted_length() +-> Result<(), Box> { + let bytes = support::decode_hex(HEAD.trim_end())?; + let mut witnesses = 0; + for offset in (0..24).chain(104..112) { + let mut altered = bytes.clone(); + *altered.get_mut(offset).ok_or("missing head fixed byte")? ^= 1; + for end in offset.checked_add(1).ok_or("prefix overflow")?..bytes.len() { + let prefix = altered.get(..end).ok_or("missing head prefix")?; + let assessment = assess_head_stage(Some(prefix)); + assert!( + matches!(assessment, RetentionStageAssessment::Corrupt(_)), + "head byte {offset}, prefix {end} must be corrupt, observed {assessment:?}" + ); + witnesses += 1; + } + } + assert!( + witnesses > 0, + "head corruption law must examine actual prefixes" + ); + Ok(()) +} + +#[test] +fn short_root_refusal_reports_only_observed_bytes() { + let assessment = assess_root_stage(Some(b"X")); + assert!( + matches!( + assessment, + RetentionStageAssessment::Corrupt(keep::RetentionRootDecodeError::PrefixByteMismatch { + offset: 0, + expected: b'K', + observed: b'X', + }) + ), + "one-byte root refusal must report offset 0, K and X; observed {assessment:?}" + ); +} + +#[test] +fn short_manifest_refusal_reports_only_observed_bytes() { + let assessment = assess_manifest_stage(Some(b"X")); + assert!( + matches!( + assessment, + RetentionStageAssessment::Corrupt( + keep::RetentionManifestDecodeError::PrefixByteMismatch { + offset: 0, + expected: b'K', + observed: b'X', + } + ) + ), + "one-byte manifest refusal must report offset 0, K and X; observed {assessment:?}" + ); +} + +#[test] +fn short_head_refusal_reports_only_observed_bytes() { + let assessment = assess_head_stage(Some(b"X")); + assert!( + matches!( + assessment, + RetentionStageAssessment::Corrupt(keep::RetentionHeadDecodeError::PrefixByteMismatch { + offset: 0, + expected: b'K', + observed: b'X', + }) + ), + "one-byte head refusal must report offset 0, K and X; observed {assessment:?}" + ); +} + +#[test] +fn admitted_closure_values_remain_possible_at_every_prefix() -> Result<(), Box> { + let fixture = support::decode_hex(ROOT.trim_end())?; + for (offset, width, maximum) in [ + (88_usize, 8_usize, 1_048_576_u64), + (96, 2, 8), + (100, 8, 16_777_216), + (108, 8, 1_073_741_824), + ] { + // Deterministically cover every admitted high-bit boundary, plus the ceiling. + for value in (0..64) + .filter_map(|bit| 1_u64.checked_shl(bit)) + .filter(|value| *value <= maximum) + .chain([maximum]) + { + let raw = value.to_be_bytes(); + let start = raw.len().checked_sub(width).ok_or("field width overflow")?; + let field = raw.get(start..).ok_or("missing encoded field")?; + for length in 1..=width { + let end = offset.checked_add(length).ok_or("prefix overflow")?; + let mut prefix = fixture.get(..end).ok_or("missing fixture")?.to_vec(); + prefix + .get_mut(offset..end) + .ok_or("missing target")? + .copy_from_slice(field.get(..length).ok_or("missing source")?); + let assessment = assess_root_stage(Some(&prefix)); + assert!( + matches!(assessment, RetentionStageAssessment::Truncated { .. }), + "admitted closure value {value} at {offset}, prefix length {length}, must admit completion: {assessment:?}" + ); + } + } + } + Ok(()) +} diff --git a/xtask/src/durability_crash_matrix/production_protocol.rs b/xtask/src/durability_crash_matrix/production_protocol.rs index 45fe2600..0b47f3bc 100644 --- a/xtask/src/durability_crash_matrix/production_protocol.rs +++ b/xtask/src/durability_crash_matrix/production_protocol.rs @@ -10,6 +10,8 @@ mod publication; mod publication_storage; mod recovery; mod recovery_storage; +pub(super) mod retention; +mod retention_storage; mod segment_stage; use std::error::Error; @@ -43,6 +45,9 @@ pub(super) fn run( DurabilityCrashSequence::RecoveryDiscard => { recovery::run(&store_root, &mut control)?; } + DurabilityCrashSequence::Retention => { + retention::run(&store_root, &mut control)?; + } DurabilityCrashSequence::Migration => { migration::run(&store_root, &mut control)?; } diff --git a/xtask/src/durability_crash_matrix/production_protocol/fixture.rs b/xtask/src/durability_crash_matrix/production_protocol/fixture.rs index 728f7dde..4f6f1b08 100644 --- a/xtask/src/durability_crash_matrix/production_protocol/fixture.rs +++ b/xtask/src/durability_crash_matrix/production_protocol/fixture.rs @@ -11,6 +11,20 @@ const SEGMENT_HEX: &str = const CATALOG_HEX: &str = include_str!("../../../../conformance/segment-store/v1/one-zero-catalog.hex"); const HEAD_HEX: &str = include_str!("../../../../conformance/segment-store/v1/one-zero-head.hex"); +const BUNDLE_SEGMENT_HEX: &str = + include_str!("../../../../conformance/segment-store/v1/one-zero-bundle-segment.hex"); +const BUNDLE_CATALOG_HEX: &str = + include_str!("../../../../conformance/segment-store/v1/one-zero-bundle-catalog.hex"); +const BUNDLE_HEAD_HEX: &str = + include_str!("../../../../conformance/segment-store/v1/one-zero-bundle-head.hex"); +const RETENTION_ROOT_HEX: &str = + include_str!("../../../../conformance/segment-store/v2/one-anchor-root.hex"); +/// Pool name of the bundle segment the retention root's closure references. +pub(in crate::durability_crash_matrix) const BUNDLE_SEGMENT_NAME: &str = + "221f6745cd8a5221c9a87c3707593608479282b54a4a74d0e753fd76f70e8db2.seg"; +/// Pool name of the generation-one bundle catalog. +pub(in crate::durability_crash_matrix) const BUNDLE_CATALOG_NAME: &str = + "0000000000000001-0b7cad1b6de663d34beacbc214db7497f2e36ab6b08dfbd5febbc8d06a418811.cat"; pub(in crate::durability_crash_matrix) const SEGMENT_POOL_PATH: &str = "segments/b7542dced2ab770894a14d1d04b066e3a899942602c5986d35ba6df6c1a35cfc.seg"; @@ -31,6 +45,26 @@ impl GoldenFixture { Self::decode("catalog", CATALOG_HEX, 352) } + pub(in crate::durability_crash_matrix) fn bundle_segment() + -> Result { + Self::decode("bundle segment", BUNDLE_SEGMENT_HEX, 701) + } + + pub(in crate::durability_crash_matrix) fn bundle_catalog() + -> Result { + Self::decode("bundle catalog", BUNDLE_CATALOG_HEX, 512) + } + + pub(in crate::durability_crash_matrix) fn bundle_head() + -> Result { + Self::decode("bundle head", BUNDLE_HEAD_HEX, 128) + } + + pub(in crate::durability_crash_matrix) fn retention_root() + -> Result { + Self::decode("retention root", RETENTION_ROOT_HEX, 378) + } + pub(in crate::durability_crash_matrix) fn head() -> Result { Self::decode("head", HEAD_HEX, 128) } diff --git a/xtask/src/durability_crash_matrix/production_protocol/initialization.rs b/xtask/src/durability_crash_matrix/production_protocol/initialization.rs index d75d1e17..32c81178 100644 --- a/xtask/src/durability_crash_matrix/production_protocol/initialization.rs +++ b/xtask/src/durability_crash_matrix/production_protocol/initialization.rs @@ -25,16 +25,23 @@ pub(super) fn run( .map_err(|source| verification("execute production store initialization", source)) } -pub(super) fn publisher( +/// Initializes a fresh store and returns its retained writer lock. +pub(super) fn initialized_lock( store_root: &Path, -) -> Result { +) -> Result { let mut storage = RepositoryInitializationStorage::admit_unchecked(store_root) .map_err(|source| DurabilityCrashMatrixError::io("open initialization storage", source))?; let _receipt = initialize_store(&mut storage) .map_err(|source| verification("initialize production crash store", source))?; - let lock = storage.into_writer_lock().map_err(|source| { - DurabilityCrashMatrixError::io("retain initialized writer lock", source) - })?; + storage + .into_writer_lock() + .map_err(|source| DurabilityCrashMatrixError::io("retain initialized writer lock", source)) +} + +pub(super) fn publisher( + store_root: &Path, +) -> Result { + let lock = initialized_lock(store_root)?; FilesystemCatalogPublisher::open_unchecked_for_repository_tasks(lock, restart_policy()?) .map_err(|source| DurabilityCrashMatrixError::io("open crash catalog publisher", source)) } diff --git a/xtask/src/durability_crash_matrix/production_protocol/retention.rs b/xtask/src/durability_crash_matrix/production_protocol/retention.rs new file mode 100644 index 00000000..486f54ba --- /dev/null +++ b/xtask/src/durability_crash_matrix/production_protocol/retention.rs @@ -0,0 +1,114 @@ +//! This module owns execution of the production retention publication protocol. + +use std::fs; +use std::path::Path; + +use keep::{ + AdmittedCatalog, AdmittedRetentionRoot, AdmittedSegment, ChecksummedCatalog, + ChecksummedPublicationHead, FilesystemRetentionPublicationAuthority, + FilesystemStoreMigrationAuthority, FilesystemVersionTwoAdmission, + RetentionGenerationExpectation, RetentionPublicationPreparation, execute_retention_publication, + execute_store_migration, preflight_retention_transition, prepare_retention_publication, +}; + +use super::control::CrashControl; +use super::fixture::{BUNDLE_CATALOG_NAME, BUNDLE_SEGMENT_NAME, GoldenFixture}; +use super::initialization; +use super::retention_storage::CrashRetentionStorage; +use super::{DurabilityCrashMatrixError, verification}; + +/// Migrates a fresh bundle store and publishes retention generation one, +/// dying at the selected coordinate. +pub(super) fn run( + store_root: &Path, + control: &mut CrashControl, +) -> Result<(), DurabilityCrashMatrixError> { + let authority = migrated_authority(store_root)?; + let root = GoldenFixture::retention_root()?; + let preparation = preparation(root.bytes())?; + let mut storage = CrashRetentionStorage::new(authority, control, store_root); + execute_retention_publication(&mut storage, &preparation) + .map(|_receipt| ()) + .map_err(|source| verification("execute production retention publication", source)) +} + +/// Initializes, populates, and migrates the bundle store, then reopens it as +/// version two and returns retention authority over it. +fn migrated_authority( + store_root: &Path, +) -> Result { + let lock = initialization::initialized_lock(store_root)?; + write_bundle(store_root)?; + let mut migration = FilesystemStoreMigrationAuthority::open_unchecked_for_repository_tasks( + lock, + initialization::segment_policy(), + ) + .map_err(|source| verification("open crash migration authority", source))?; + let intent = migration + .observe_intent() + .map_err(|source| verification("observe crash migration intent", source))?; + let _receipt = execute_store_migration(&mut migration, &intent) + .map_err(|source| verification("execute crash store migration", source))?; + drop(migration); + reopened_authority(store_root) +} + +/// Reopens the migrated store and returns retention authority over it. +pub(in crate::durability_crash_matrix) fn reopened_authority( + store_root: &Path, +) -> Result { + let admission = + FilesystemVersionTwoAdmission::reopen_unchecked_for_repository_tasks(store_root) + .map_err(|source| verification("reopen crash store as version two", source))?; + FilesystemRetentionPublicationAuthority::open(admission) + .map_err(|source| verification("open crash retention authority", source)) +} + +fn write_bundle(store_root: &Path) -> Result<(), DurabilityCrashMatrixError> { + let segment = GoldenFixture::bundle_segment()?; + let catalog = GoldenFixture::bundle_catalog()?; + let head = GoldenFixture::bundle_head()?; + for (relative, bytes) in [ + (format!("segments/{BUNDLE_SEGMENT_NAME}"), segment.bytes()), + (format!("catalogs/{BUNDLE_CATALOG_NAME}"), catalog.bytes()), + ("HEAD".to_owned(), head.bytes()), + ] { + fs::write(store_root.join(&relative), bytes) + .map_err(|source| DurabilityCrashMatrixError::io("write bundle corpus", source))?; + } + Ok(()) +} + +/// Prepares the frozen generation-one root as an initial publication against +/// the bundle catalog snapshot. +pub(in crate::durability_crash_matrix) fn preparation( + root_bytes: &[u8], +) -> Result, DurabilityCrashMatrixError> { + let segment_fixture = GoldenFixture::bundle_segment()?; + let catalog_fixture = GoldenFixture::bundle_catalog()?; + let head_fixture = GoldenFixture::bundle_head()?; + let candidate = AdmittedRetentionRoot::decode(root_bytes) + .map_err(|source| verification("decode crash retention root", source))?; + let segment = + AdmittedSegment::decode(segment_fixture.bytes(), initialization::segment_policy()) + .map_err(|source| verification("admit bundle segment", source))?; + let segments = [segment]; + let catalog: AdmittedCatalog<'_, '_> = ChecksummedCatalog::decode(catalog_fixture.bytes()) + .map_err(|source| verification("decode bundle catalog", source))? + .admit(&segments) + .map_err(|source| verification("admit bundle catalog", source))?; + let head = ChecksummedPublicationHead::decode(head_fixture.bytes()) + .map_err(|source| verification("decode bundle head", source))?; + let snapshot = head + .admit(catalog) + .map_err(|source| verification("admit bundle snapshot", source))?; + let preflight = preflight_retention_transition( + RetentionGenerationExpectation::Absent, + None, + candidate, + &snapshot, + ) + .map_err(|source| verification("preflight crash retention transition", source))?; + prepare_retention_publication(preflight, None) + .map_err(|source| verification("prepare crash retention publication", source)) +} diff --git a/xtask/src/durability_crash_matrix/production_protocol/retention_storage.rs b/xtask/src/durability_crash_matrix/production_protocol/retention_storage.rs new file mode 100644 index 00000000..3d8bfa2a --- /dev/null +++ b/xtask/src/durability_crash_matrix/production_protocol/retention_storage.rs @@ -0,0 +1,229 @@ +//! This module owns crash injection around production retention publication. + +use std::fs::OpenOptions; +use std::io::{self, Write}; +use std::path::{Path, PathBuf}; + +use keep::{ + AdmittedRetentionRoot, CanonicalRetentionHead, CanonicalRetentionManifest, + FilesystemRetentionPublicationAuthority, RetentionNamespaceAdmission, + RetentionPublicationPreparation, RetentionPublicationStorage, RetentionTransitionDisposition, +}; +use xtask::{DurabilityCrashPoint, DurabilityCrashPosition}; + +use super::control::{CrashControl, DuringTiming}; + +/// Bytes an interrupted stage write leaves behind: inside every record's +/// fixed framing, so restart classifies the stage as truncated. +const STAGE_INTERRUPTION: usize = 100; + +pub(super) struct CrashRetentionStorage<'control> { + inner: FilesystemRetentionPublicationAuthority, + control: &'control mut CrashControl, + retention: PathBuf, +} + +impl<'control> CrashRetentionStorage<'control> { + pub(super) fn new( + inner: FilesystemRetentionPublicationAuthority, + control: &'control mut CrashControl, + store_root: &Path, + ) -> Self { + Self { + inner, + control, + retention: store_root.join("retention"), + } + } + + fn execute( + &mut self, + point: DurabilityCrashPoint, + during: DuringTiming, + operation: impl FnOnce(&mut FilesystemRetentionPublicationAuthority) -> io::Result, + ) -> io::Result { + self.control.before(point, during)?; + let result = operation(&mut self.inner)?; + self.control.after(point, during)?; + Ok(result) + } + + fn execute_write( + &mut self, + point: DurabilityCrashPoint, + stage: &str, + bytes: &[u8], + complete: impl FnOnce(&mut FilesystemRetentionPublicationAuthority) -> io::Result<()>, + ) -> io::Result<()> { + match self.control.position(point) { + None => complete(&mut self.inner), + Some(DurabilityCrashPosition::Before) => self.control.await_process_death(), + Some(DurabilityCrashPosition::During) => { + let partial = bytes.get(..STAGE_INTERRUPTION).ok_or_else(|| { + io::Error::other("retention record shorter than the interruption prefix") + })?; + let mut file = OpenOptions::new() + .write(true) + .create_new(true) + .open(self.retention.join(stage))?; + file.write_all(partial)?; + self.control.await_process_death() + } + Some(DurabilityCrashPosition::After) => { + complete(&mut self.inner)?; + self.control.await_process_death() + } + } + } +} + +impl RetentionPublicationStorage for CrashRetentionStorage<'_> { + fn verify_current( + &mut self, + preparation: &RetentionPublicationPreparation<'_>, + ) -> io::Result { + self.inner.verify_current(preparation) + } + + fn write_root_stage(&mut self, root: &AdmittedRetentionRoot<'_>) -> io::Result<()> { + self.execute_write( + DurabilityCrashPoint::WriteRootStage, + "root.next", + root.encoded(), + |inner| inner.write_root_stage(root), + ) + } + + fn synchronize_root_stage(&mut self) -> io::Result<()> { + self.execute( + DurabilityCrashPoint::SynchronizeRootStage, + DuringTiming::Before, + FilesystemRetentionPublicationAuthority::synchronize_root_stage, + ) + } + + fn admit_root_namespace( + &mut self, + root: &AdmittedRetentionRoot<'_>, + ) -> io::Result { + self.execute( + DurabilityCrashPoint::AdmitRootNamespace, + DuringTiming::After, + |inner| inner.admit_root_namespace(root), + ) + } + + fn synchronize_roots_after_namespace(&mut self) -> io::Result<()> { + self.execute( + DurabilityCrashPoint::SynchronizeRootsAfterNamespace, + DuringTiming::Before, + FilesystemRetentionPublicationAuthority::synchronize_roots_after_namespace, + ) + } + + fn link_root(&mut self, root: &AdmittedRetentionRoot<'_>) -> io::Result<()> { + self.execute( + DurabilityCrashPoint::LinkRoot, + DuringTiming::After, + |inner| inner.link_root(root), + ) + } + + fn synchronize_root_namespace(&mut self, root: &AdmittedRetentionRoot<'_>) -> io::Result<()> { + self.execute( + DurabilityCrashPoint::SynchronizeRootNamespace, + DuringTiming::Before, + |inner| inner.synchronize_root_namespace(root), + ) + } + + fn write_manifest_stage(&mut self, manifest: &CanonicalRetentionManifest) -> io::Result<()> { + self.execute_write( + DurabilityCrashPoint::WriteManifestStage, + "manifest.next", + manifest.encoded(), + |inner| inner.write_manifest_stage(manifest), + ) + } + + fn synchronize_manifest_stage(&mut self) -> io::Result<()> { + self.execute( + DurabilityCrashPoint::SynchronizeManifestStage, + DuringTiming::Before, + FilesystemRetentionPublicationAuthority::synchronize_manifest_stage, + ) + } + + fn link_manifest(&mut self, manifest: &CanonicalRetentionManifest) -> io::Result<()> { + self.execute( + DurabilityCrashPoint::LinkManifest, + DuringTiming::After, + |inner| inner.link_manifest(manifest), + ) + } + + fn synchronize_manifest_pool(&mut self) -> io::Result<()> { + self.execute( + DurabilityCrashPoint::SynchronizeManifestPool, + DuringTiming::Before, + FilesystemRetentionPublicationAuthority::synchronize_manifest_pool, + ) + } + + fn write_head_stage(&mut self, head: &CanonicalRetentionHead) -> io::Result<()> { + self.execute_write( + DurabilityCrashPoint::WriteHeadStage, + "head.next", + head.encoded(), + |inner| inner.write_head_stage(head), + ) + } + + fn synchronize_head_stage(&mut self) -> io::Result<()> { + self.execute( + DurabilityCrashPoint::SynchronizeHeadStage, + DuringTiming::Before, + FilesystemRetentionPublicationAuthority::synchronize_head_stage, + ) + } + + fn replace_head(&mut self) -> io::Result<()> { + self.execute( + DurabilityCrashPoint::ReplaceRetentionHead, + DuringTiming::After, + FilesystemRetentionPublicationAuthority::replace_head, + ) + } + + fn synchronize_retention_namespace(&mut self) -> io::Result<()> { + self.execute( + DurabilityCrashPoint::SynchronizeRetentionNamespace, + DuringTiming::Before, + FilesystemRetentionPublicationAuthority::synchronize_retention_namespace, + ) + } + + fn remove_root_stage(&mut self) -> io::Result<()> { + self.execute( + DurabilityCrashPoint::RemoveRootStage, + DuringTiming::After, + FilesystemRetentionPublicationAuthority::remove_root_stage, + ) + } + + fn remove_manifest_stage(&mut self) -> io::Result<()> { + self.execute( + DurabilityCrashPoint::RemoveManifestStage, + DuringTiming::After, + FilesystemRetentionPublicationAuthority::remove_manifest_stage, + ) + } + + fn synchronize_cleanup(&mut self) -> io::Result<()> { + self.execute( + DurabilityCrashPoint::SynchronizeRetentionCleanup, + DuringTiming::Before, + FilesystemRetentionPublicationAuthority::synchronize_cleanup, + ) + } +} diff --git a/xtask/src/durability_crash_matrix/restart.rs b/xtask/src/durability_crash_matrix/restart.rs index 8ee7c51a..2c2a6834 100644 --- a/xtask/src/durability_crash_matrix/restart.rs +++ b/xtask/src/durability_crash_matrix/restart.rs @@ -3,6 +3,9 @@ mod expectation; mod migration; mod migration_expectation; +mod retention; +mod retention_incomplete; +mod retention_snapshot; mod semantic; use std::collections::BTreeSet; @@ -19,6 +22,9 @@ pub(super) fn verify( store_root: &Path, case: DurabilityCrashCase, ) -> Result<(), DurabilityCrashMatrixError> { + if case.point().sequence() == xtask::DurabilityCrashSequence::Retention { + return retention::verify(store_root, case); + } if case.point().sequence() == DurabilityCrashSequence::Migration { return migration::verify(store_root, case); } diff --git a/xtask/src/durability_crash_matrix/restart/expectation.rs b/xtask/src/durability_crash_matrix/restart/expectation.rs index c4a779a0..e54c7961 100644 --- a/xtask/src/durability_crash_matrix/restart/expectation.rs +++ b/xtask/src/durability_crash_matrix/restart/expectation.rs @@ -64,7 +64,7 @@ impl ExpectedStoreState { DurabilityCrashSequence::Head => sequence::head(case), DurabilityCrashSequence::RecoveryDiscard => sequence::recovery(case), DurabilityCrashSequence::Initialization => sequence::initialization(case), - DurabilityCrashSequence::Migration => { + DurabilityCrashSequence::Retention | DurabilityCrashSequence::Migration => { Err(DurabilityCrashMatrixError::PointSequenceMismatch { point: case.point(), }) diff --git a/xtask/src/durability_crash_matrix/restart/retention.rs b/xtask/src/durability_crash_matrix/restart/retention.rs new file mode 100644 index 00000000..9538059b --- /dev/null +++ b/xtask/src/durability_crash_matrix/restart/retention.rs @@ -0,0 +1,164 @@ +//! This module owns post-process-death verification of retention publication. +//! +//! After the child dies at its coordinate, restart reopens the migrated store +//! through the same admission a production caller would use, runs retention +//! recovery, and requires the documented steps and outcome for that exact +//! prefix; then it independently reads the persistent snapshot before requiring +//! the forward retry to report the outcome recovery predicts. + +use std::io; +use std::path::Path; + +use keep::{ + RetentionCurrentStateRefusal, RetentionPublicationError, RetentionPublicationOutcome, + RetentionRecoveryOutcome as Outcome, RetentionRecoveryStep as Step, + execute_retention_publication, +}; +use xtask::{DurabilityCrashCase, DurabilityCrashPoint, DurabilityCrashPosition}; + +use super::super::DurabilityCrashMatrixError; +use super::super::production_protocol::fixture::GoldenFixture; +use super::super::production_protocol::retention::{preparation, reopened_authority}; +use super::super::production_protocol::verification; + +const PROTECTED_ROOT: Outcome = Outcome::Protected { + root_stage: true, + manifest_stage: false, +}; +const PROTECTED_BOTH: Outcome = Outcome::Protected { + root_stage: true, + manifest_stage: true, +}; + +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +enum Retry { + Published, + AlreadyCommitted, + Refused, +} + +pub(super) fn verify( + store_root: &Path, + case: DurabilityCrashCase, +) -> Result<(), DurabilityCrashMatrixError> { + if let Some(stage) = prefix(case).1 { + return super::retention_incomplete::verify(store_root, stage); + } + let (steps, outcome, retry) = expected(case); + let mut authority = reopened_authority(store_root)?; + let receipt = authority + .recover() + .map_err(|source| verification("recover crash retention stages", source))?; + if receipt.executed() != steps.as_slice() || receipt.outcome() != outcome { + return Err(mismatch(format!( + "{} {:?}: expected {steps:?} -> {outcome:?}, recovered {:?} -> {:?}", + case.point().identifier(), + case.position(), + receipt.executed(), + receipt.outcome() + ))); + } + let root = GoldenFixture::retention_root()?; + let generation = (prefix(case).0 >= 12).then_some(1); + super::retention_snapshot::verify(store_root, root.bytes(), generation)?; + let preparation = preparation(root.bytes())?; + match ( + retry, + execute_retention_publication(&mut authority, &preparation), + ) { + (Retry::Published, Ok(receipt)) + if receipt.outcome() == RetentionPublicationOutcome::Published => {} + (Retry::AlreadyCommitted, Ok(receipt)) + if receipt.outcome() == RetentionPublicationOutcome::AlreadyCommitted => {} + (Retry::Refused, Err(RetentionPublicationError::CurrentVerification { source })) + if source + .get_ref() + .and_then(|refusal| refusal.downcast_ref::()) + .is_some_and(|refusal| { + matches!(refusal, RetentionCurrentStateRefusal::RetainedStage) + }) => {} + (retry, result) => { + return Err(mismatch(format!( + "{} {:?}: expected forward retry {retry:?}, got {result:?}", + case.point().identifier(), + case.position() + ))); + } + } + Ok(()) +} + +fn mismatch(message: String) -> DurabilityCrashMatrixError { + verification("verify crash retention recovery", io::Error::other(message)) +} + +/// The number of completed publication phases and any truncated stage the +/// coordinate leaves behind. +fn prefix(case: DurabilityCrashCase) -> (usize, Option) { + let index = DurabilityCrashPoint::ALL + .iter() + .position(|point| *point == case.point()) + .and_then(|index| index.checked_sub(35)) + .unwrap_or(0); + // Phase 1 is current-state verification; point n is phase n + 2. + let phase = index.saturating_add(2); + let write = match case.point() { + DurabilityCrashPoint::WriteRootStage => Some(Step::DiscardRootStage), + DurabilityCrashPoint::WriteManifestStage => Some(Step::DiscardManifestStage), + DurabilityCrashPoint::WriteHeadStage => Some(Step::DiscardHeadStage), + _ => None, + }; + match case.position() { + DurabilityCrashPosition::After => (phase, None), + DurabilityCrashPosition::During if write.is_some() => (phase.saturating_sub(1), write), + DurabilityCrashPosition::During if atomic(case.point()) => (phase, None), + DurabilityCrashPosition::Before | DurabilityCrashPosition::During => { + (phase.saturating_sub(1), None) + } + } +} + +const fn atomic(point: DurabilityCrashPoint) -> bool { + matches!( + point, + DurabilityCrashPoint::AdmitRootNamespace + | DurabilityCrashPoint::LinkRoot + | DurabilityCrashPoint::LinkManifest + | DurabilityCrashPoint::ReplaceRetentionHead + | DurabilityCrashPoint::RemoveRootStage + | DurabilityCrashPoint::RemoveManifestStage + ) +} + +/// The documented recovery for the prefix a coordinate leaves behind. +fn expected(case: DurabilityCrashCase) -> (Vec, Outcome, Retry) { + let (count, _) = prefix(case); + let (steps, outcome, retry) = match count { + 0 | 1 => (vec![], Outcome::Clean, Retry::Published), + 2..=5 => (vec![Step::LinkRoot], PROTECTED_ROOT, Retry::Refused), + 6 | 7 => (vec![], PROTECTED_ROOT, Retry::Refused), + 8 | 9 => (vec![Step::LinkManifest], PROTECTED_BOTH, Retry::Refused), + 10 | 11 => (vec![], PROTECTED_BOTH, Retry::Refused), + 12 | 13 => ( + vec![ + Step::FinalizeHead, + Step::RemoveRootStage, + Step::RemoveManifestStage, + ], + Outcome::Committed, + Retry::AlreadyCommitted, + ), + 14 | 15 => ( + vec![Step::RemoveRootStage, Step::RemoveManifestStage], + Outcome::Committed, + Retry::AlreadyCommitted, + ), + 16 => ( + vec![Step::RemoveManifestStage], + Outcome::Committed, + Retry::AlreadyCommitted, + ), + _ => (vec![], Outcome::Clean, Retry::AlreadyCommitted), + }; + (steps, outcome, retry) +} diff --git a/xtask/src/durability_crash_matrix/restart/retention_incomplete.rs b/xtask/src/durability_crash_matrix/restart/retention_incomplete.rs new file mode 100644 index 00000000..9d80d4bf --- /dev/null +++ b/xtask/src/durability_crash_matrix/restart/retention_incomplete.rs @@ -0,0 +1,80 @@ +//! This module owns evidence preservation after a process dies during a retention stage write. + +use super::super::{ + DurabilityCrashMatrixError, + production_protocol::{ + fixture::GoldenFixture, + retention::{preparation, reopened_authority}, + verification, + }, +}; +use keep::{ + FilesystemRetentionRecoveryError, RetentionCurrentStateRefusal, RetentionFixedStage as Stage, + RetentionPublicationError, RetentionRecoveryRefusal, RetentionRecoveryStep as Step, + execute_retention_publication, +}; +use std::{collections::BTreeMap, fs, io, path::Path}; + +pub(super) fn verify(store: &Path, interrupted: Step) -> Result<(), DurabilityCrashMatrixError> { + let (stage, expected) = match interrupted { + Step::DiscardRootStage => (Stage::Root, 192), + Step::DiscardManifestStage => (Stage::Manifest, 160), + Step::DiscardHeadStage => (Stage::Head, 144), + _ => return Err(failure("unexpected incomplete-stage coordinate")), + }; + let before = witness(store)?; + let mut authority = reopened_authority(store)?; + let result = authority.recover(); + if !matches!(result, Err(FilesystemRetentionRecoveryError::Plan { source: RetentionRecoveryRefusal::IncompleteStageRequiresDisposition { stage: actual, expected: minimum, observed: 100 } }) if actual == stage && minimum == expected) + { + return Err(failure(&format!( + "expected disposition for {stage:?}, got {result:?}" + ))); + } + if witness(store)? != before { + return Err(failure( + "incomplete-stage recovery changed retained evidence", + )); + } + let root = GoldenFixture::retention_root()?; + super::retention_snapshot::verify(store, root.bytes(), None)?; + let preparation = preparation(root.bytes())?; + let retry = execute_retention_publication(&mut authority, &preparation); + if !matches!(retry, Err(RetentionPublicationError::CurrentVerification { ref source }) if matches!(source.get_ref().and_then(|cause| cause.downcast_ref::()), Some(RetentionCurrentStateRefusal::RecoveryRefused { source: RetentionRecoveryRefusal::IncompleteStageRequiresDisposition { stage: actual, expected: minimum, observed: 100 } }) if *actual == stage && *minimum == expected)) + { + return Err(failure(&format!( + "publication did not preserve disposition requirement: {retry:?}" + ))); + } + if witness(store)? != before { + return Err(failure("publication retry changed retained evidence")); + } + Ok(()) +} + +fn witness(store: &Path) -> Result>>, DurabilityCrashMatrixError> { + super::inventory(store)? + .into_iter() + .map(|name| { + let path = store.join(&name); + let metadata = fs::symlink_metadata(&path) + .map_err(|source| verification("inspect retained crash evidence", source))?; + let bytes = if metadata.is_file() { + Some( + fs::read(path) + .map_err(|source| verification("read retained crash evidence", source))?, + ) + } else { + None + }; + Ok((name, bytes)) + }) + .collect() +} + +fn failure(message: &str) -> DurabilityCrashMatrixError { + verification( + "verify incomplete retention restart", + io::Error::other(message.to_owned()), + ) +} diff --git a/xtask/src/durability_crash_matrix/restart/retention_snapshot.rs b/xtask/src/durability_crash_matrix/restart/retention_snapshot.rs new file mode 100644 index 00000000..1b875f67 --- /dev/null +++ b/xtask/src/durability_crash_matrix/restart/retention_snapshot.rs @@ -0,0 +1,52 @@ +//! This module owns the independent reader oracle after retention crash recovery. + +use std::io; +use std::path::Path; + +use keep::{ + AdmittedRetentionRoot, CatalogRestartByteLimit, CatalogRestartPolicy, + FilesystemRetentionSnapshot, ReaderAttemptLimit, SegmentReadPolicy, +}; + +use super::super::DurabilityCrashMatrixError; +use super::super::production_protocol::verification; + +/// Reads product output independently of recovery receipts and forward retries. +/// The coordinate's publication prefix supplies the expected generation; +/// the frozen publication input supplies the namespace and exact root bytes. +pub(super) fn verify( + store_root: &Path, + expected_root: &[u8], + expected_generation: Option, +) -> Result<(), DurabilityCrashMatrixError> { + let limit = CatalogRestartByteLimit::new(1_048_576) + .map_err(|source| verification("bound recovered catalog read", source))?; + let policy = CatalogRestartPolicy::new(SegmentReadPolicy::MAXIMUM, limit); + let view = FilesystemRetentionSnapshot::load(store_root, policy, ReaderAttemptLimit::DEFAULT) + .map_err(|source| verification("load recovered retention snapshot", source))?; + let observed_generation = view.retention_head().map(|head| head.generation().get()); + if observed_generation != expected_generation { + return Err(verification( + "verify recovered retention generation", + io::Error::other(format!( + "expected {expected_generation:?}, observed {observed_generation:?}" + )), + )); + } + let root = AdmittedRetentionRoot::decode(expected_root) + .map_err(|source| verification("decode expected recovered root", source))?; + let observed = view + .retained_root(root.root().namespace().digest()) + .map_err(|source| verification("read recovered selected root", source))?; + let expected = expected_generation.map(|_| expected_root); + if observed.as_deref() != expected { + return Err(verification( + "verify recovered selected root bytes", + io::Error::other(format!( + "expected {expected:?}, observed {:?}", + observed.as_deref() + )), + )); + } + Ok(()) +} diff --git a/xtask/src/durability_crash_point.rs b/xtask/src/durability_crash_point.rs index 555ed7ec..0cd6ad04 100644 --- a/xtask/src/durability_crash_point.rs +++ b/xtask/src/durability_crash_point.rs @@ -13,18 +13,21 @@ pub enum DurabilityCrashSequence { RecoveryDiscard, /// Writer-locked store initialization. Initialization, + /// Version-two retention publication, `KEEP-CRASH-036` through `052`. + Retention, /// One-way version-1 to version-2 store migration. Migration, } impl DurabilityCrashSequence { /// Every sequence in stable protocol order. - pub const ALL: [Self; 6] = [ + pub const ALL: [Self; 7] = [ Self::Segment, Self::Catalog, Self::Head, Self::RecoveryDiscard, Self::Initialization, + Self::Retention, Self::Migration, ]; @@ -37,6 +40,7 @@ impl DurabilityCrashSequence { Self::Head => "head", Self::RecoveryDiscard => "recovery-discard", Self::Initialization => "initialization", + Self::Retention => "retention", Self::Migration => "migration", } } @@ -123,6 +127,40 @@ pub enum DurabilityCrashPoint { CreateCatalogPoolDirectory, /// Synchronize the store root after initialization. SynchronizeRootAfterInitialization, + /// Retention root stage write. + WriteRootStage, + /// Retention root stage synchronization. + SynchronizeRootStage, + /// New namespace-directory creation or exact admission. + AdmitRootNamespace, + /// Namespace-pool synchronization after creation. + SynchronizeRootsAfterNamespace, + /// Immutable root link. + LinkRoot, + /// Root namespace-directory synchronization. + SynchronizeRootNamespace, + /// Retention manifest stage write. + WriteManifestStage, + /// Retention manifest stage synchronization. + SynchronizeManifestStage, + /// Immutable manifest link. + LinkManifest, + /// Manifest pool synchronization. + SynchronizeManifestPool, + /// Retention-head stage write. + WriteHeadStage, + /// Retention-head stage synchronization. + SynchronizeHeadStage, + /// Retention-head atomic replacement. + ReplaceRetentionHead, + /// Committed retention namespace synchronization. + SynchronizeRetentionNamespace, + /// Retained root-stage removal. + RemoveRootStage, + /// Retained manifest-stage removal. + RemoveManifestStage, + /// Retention cleanup synchronization. + SynchronizeRetentionCleanup, /// Write the complete canonical `migration.intent.next`. MigrationWriteIntentStage, /// Synchronize `migration.intent.next`. @@ -170,7 +208,7 @@ pub enum DurabilityCrashPoint { impl DurabilityCrashPoint { /// Every crash boundary in stable protocol order. - pub const ALL: [Self; 56] = [ + pub const ALL: [Self; 73] = [ Self::CreateSegmentStage, Self::WriteSegmentHeader, Self::AppendSegmentRecord, @@ -206,6 +244,23 @@ impl DurabilityCrashPoint { Self::CreateSegmentPoolDirectory, Self::CreateCatalogPoolDirectory, Self::SynchronizeRootAfterInitialization, + Self::WriteRootStage, + Self::SynchronizeRootStage, + Self::AdmitRootNamespace, + Self::SynchronizeRootsAfterNamespace, + Self::LinkRoot, + Self::SynchronizeRootNamespace, + Self::WriteManifestStage, + Self::SynchronizeManifestStage, + Self::LinkManifest, + Self::SynchronizeManifestPool, + Self::WriteHeadStage, + Self::SynchronizeHeadStage, + Self::ReplaceRetentionHead, + Self::SynchronizeRetentionNamespace, + Self::RemoveRootStage, + Self::RemoveManifestStage, + Self::SynchronizeRetentionCleanup, Self::MigrationWriteIntentStage, Self::MigrationSynchronizeIntentStage, Self::MigrationLinkIntent, @@ -306,6 +361,23 @@ impl DurabilityCrashPoint { | Self::CreateSegmentPoolDirectory | Self::CreateCatalogPoolDirectory | Self::SynchronizeRootAfterInitialization => DurabilityCrashSequence::Initialization, + Self::WriteRootStage + | Self::SynchronizeRootStage + | Self::AdmitRootNamespace + | Self::SynchronizeRootsAfterNamespace + | Self::LinkRoot + | Self::SynchronizeRootNamespace + | Self::WriteManifestStage + | Self::SynchronizeManifestStage + | Self::LinkManifest + | Self::SynchronizeManifestPool + | Self::WriteHeadStage + | Self::SynchronizeHeadStage + | Self::ReplaceRetentionHead + | Self::SynchronizeRetentionNamespace + | Self::RemoveRootStage + | Self::RemoveManifestStage + | Self::SynchronizeRetentionCleanup => DurabilityCrashSequence::Retention, Self::MigrationWriteIntentStage | Self::MigrationSynchronizeIntentStage | Self::MigrationLinkIntent diff --git a/xtask/src/durability_crash_point_identity.rs b/xtask/src/durability_crash_point_identity.rs index 7959b43b..168af5a7 100644 --- a/xtask/src/durability_crash_point_identity.rs +++ b/xtask/src/durability_crash_point_identity.rs @@ -42,6 +42,23 @@ impl DurabilityCrashPoint { Self::CreateSegmentPoolDirectory => "KEEP-CRASH-033", Self::CreateCatalogPoolDirectory => "KEEP-CRASH-034", Self::SynchronizeRootAfterInitialization => "KEEP-CRASH-035", + Self::WriteRootStage => "KEEP-CRASH-036", + Self::SynchronizeRootStage => "KEEP-CRASH-037", + Self::AdmitRootNamespace => "KEEP-CRASH-038", + Self::SynchronizeRootsAfterNamespace => "KEEP-CRASH-039", + Self::LinkRoot => "KEEP-CRASH-040", + Self::SynchronizeRootNamespace => "KEEP-CRASH-041", + Self::WriteManifestStage => "KEEP-CRASH-042", + Self::SynchronizeManifestStage => "KEEP-CRASH-043", + Self::LinkManifest => "KEEP-CRASH-044", + Self::SynchronizeManifestPool => "KEEP-CRASH-045", + Self::WriteHeadStage => "KEEP-CRASH-046", + Self::SynchronizeHeadStage => "KEEP-CRASH-047", + Self::ReplaceRetentionHead => "KEEP-CRASH-048", + Self::SynchronizeRetentionNamespace => "KEEP-CRASH-049", + Self::RemoveRootStage => "KEEP-CRASH-050", + Self::RemoveManifestStage => "KEEP-CRASH-051", + Self::SynchronizeRetentionCleanup => "KEEP-CRASH-052", Self::MigrationWriteIntentStage => "KEEP-CRASH-053", Self::MigrationSynchronizeIntentStage => "KEEP-CRASH-054", Self::MigrationLinkIntent => "KEEP-CRASH-055", diff --git a/xtask/tests/durability_crash_case_contract.rs b/xtask/tests/durability_crash_case_contract.rs index 755ac076..c062a5eb 100644 --- a/xtask/tests/durability_crash_case_contract.rs +++ b/xtask/tests/durability_crash_case_contract.rs @@ -6,45 +6,9 @@ use std::error::Error; use xtask::{ DurabilityCrashCase, DurabilityCrashCaseError, DurabilityCrashOccurrence, DurabilityCrashPoint, - DurabilityCrashPosition, DurabilityCrashSequence, + DurabilityCrashPosition, }; -#[test] -fn every_crash_point_has_three_ordered_positions_and_one_during_case_per_occurrence() --> Result<(), Box> { - let cases: Vec<_> = DurabilityCrashCase::all().collect(); - let mut expected = Vec::new(); - for point in DurabilityCrashPoint::ALL { - for position in DurabilityCrashPosition::ALL { - let occurrences = if position == DurabilityCrashPosition::During { - point.during_occurrences() - } else { - 1 - }; - for ordinal in 0..occurrences { - let occurrence = point - .occurrence_counted() - .then_some(DurabilityCrashOccurrence::new(ordinal)); - expected.push(DurabilityCrashCase::new(point, position, occurrence)?); - } - } - } - - assert_eq!(cases, expected); - // 56 boundaries at three positions, plus five extra namespace-prefix - // lengths for `KEEP-CRASH-060`. - assert_eq!(cases.len(), 173); - let migration: Vec<_> = - DurabilityCrashCase::in_sequence(DurabilityCrashSequence::Migration).collect(); - assert_eq!(migration.len(), 68); - assert!( - migration - .iter() - .all(|case| case.point().sequence() == DurabilityCrashSequence::Migration) - ); - Ok(()) -} - #[test] fn occurrence_coordinates_exist_only_for_counted_boundaries() -> Result<(), Box> { let occurrence = DurabilityCrashOccurrence::new(7); diff --git a/xtask/tests/durability_crash_point_contract.rs b/xtask/tests/durability_crash_point_contract.rs index 06c21cea..57b4c26c 100644 --- a/xtask/tests/durability_crash_point_contract.rs +++ b/xtask/tests/durability_crash_point_contract.rs @@ -4,7 +4,9 @@ use xtask::{DurabilityCrashPoint, DurabilityCrashSequence}; -use DurabilityCrashSequence::{Catalog, Head, Initialization, Migration, RecoveryDiscard, Segment}; +use DurabilityCrashSequence::{ + Catalog, Head, Initialization, Migration, RecoveryDiscard, Retention, Segment, +}; const EXPECTED: &[(DurabilityCrashPoint, &str, DurabilityCrashSequence)] = &[ ( @@ -162,6 +164,87 @@ const EXPECTED: &[(DurabilityCrashPoint, &str, DurabilityCrashSequence)] = &[ "KEEP-CRASH-035", Initialization, ), + ( + DurabilityCrashPoint::WriteRootStage, + "KEEP-CRASH-036", + Retention, + ), + ( + DurabilityCrashPoint::SynchronizeRootStage, + "KEEP-CRASH-037", + Retention, + ), + ( + DurabilityCrashPoint::AdmitRootNamespace, + "KEEP-CRASH-038", + Retention, + ), + ( + DurabilityCrashPoint::SynchronizeRootsAfterNamespace, + "KEEP-CRASH-039", + Retention, + ), + (DurabilityCrashPoint::LinkRoot, "KEEP-CRASH-040", Retention), + ( + DurabilityCrashPoint::SynchronizeRootNamespace, + "KEEP-CRASH-041", + Retention, + ), + ( + DurabilityCrashPoint::WriteManifestStage, + "KEEP-CRASH-042", + Retention, + ), + ( + DurabilityCrashPoint::SynchronizeManifestStage, + "KEEP-CRASH-043", + Retention, + ), + ( + DurabilityCrashPoint::LinkManifest, + "KEEP-CRASH-044", + Retention, + ), + ( + DurabilityCrashPoint::SynchronizeManifestPool, + "KEEP-CRASH-045", + Retention, + ), + ( + DurabilityCrashPoint::WriteHeadStage, + "KEEP-CRASH-046", + Retention, + ), + ( + DurabilityCrashPoint::SynchronizeHeadStage, + "KEEP-CRASH-047", + Retention, + ), + ( + DurabilityCrashPoint::ReplaceRetentionHead, + "KEEP-CRASH-048", + Retention, + ), + ( + DurabilityCrashPoint::SynchronizeRetentionNamespace, + "KEEP-CRASH-049", + Retention, + ), + ( + DurabilityCrashPoint::RemoveRootStage, + "KEEP-CRASH-050", + Retention, + ), + ( + DurabilityCrashPoint::RemoveManifestStage, + "KEEP-CRASH-051", + Retention, + ), + ( + DurabilityCrashPoint::SynchronizeRetentionCleanup, + "KEEP-CRASH-052", + Retention, + ), ( DurabilityCrashPoint::MigrationWriteIntentStage, "KEEP-CRASH-053", @@ -331,13 +414,37 @@ fn migration_boundaries_follow_the_twenty_one_phases_in_order() { ); } +// Subject: repository-tool runtime, not Keep recovery acceptance evidence. +// Size: small; oracle: the exact sequence names documented in README.md. +// Delete if this CLI is removed or a stronger command-boundary test subsumes it. +#[test] +fn crash_matrix_cli_admits_its_stable_sequence_names() { + for (name, expected) in [ + ("segment", Segment), + ("catalog", Catalog), + ("head", Head), + ("recovery-discard", RecoveryDiscard), + ("initialization", Initialization), + ("retention", Retention), + ("migration", Migration), + ] { + assert_eq!( + DurabilityCrashSequence::from_identifier(name), + Some(expected), + "CLI sequence {name} must select {expected:?}" + ); + } +} + +// Subject: repository-tool runtime. Size: small; oracle: exact CLI vocabulary. +// Delete if this CLI is removed or a stronger command-boundary test subsumes it. #[test] -fn sequences_round_trip_their_command_line_identifiers() { - for sequence in DurabilityCrashSequence::ALL { +fn crash_matrix_cli_refuses_names_outside_its_exact_vocabulary() { + for name in ["", "unknown", "Retention", "retentions", "retention "] { assert_eq!( - DurabilityCrashSequence::from_identifier(sequence.identifier()), - Some(sequence) + DurabilityCrashSequence::from_identifier(name), + None, + "unsupported CLI sequence {name:?} must refuse" ); } - assert_eq!(DurabilityCrashSequence::from_identifier("retention"), None); } diff --git a/xtask/tests/retention_recovery_sync.rs b/xtask/tests/retention_recovery_sync.rs new file mode 100644 index 00000000..af5f87b3 --- /dev/null +++ b/xtask/tests/retention_recovery_sync.rs @@ -0,0 +1,55 @@ +//! Linux OS-boundary evidence for recovered retention file durability. + +#![cfg(all(feature = "repository-tasks", target_os = "linux"))] + +use std::error::Error; +use std::process::Command; + +// Size: medium (local subprocesses and filesystem). +// Oracle: retention recovery must sync exact stage contents before publication. +// Delete only when this protocol is removed or stronger OS fault evidence subsumes it. +#[test] +fn recovered_root_contents_are_synchronized_before_pool_publication() -> Result<(), Box> +{ + require_stage_sync("KEEP-CRASH-036", "root.next", "linkat(") +} + +#[test] +fn recovered_manifest_contents_are_synchronized_before_pool_publication() +-> Result<(), Box> { + require_stage_sync("KEEP-CRASH-042", "manifest.next", "linkat(") +} + +#[test] +fn recovered_head_contents_are_synchronized_before_head_publication() -> Result<(), Box> +{ + require_stage_sync("KEEP-CRASH-046", "head.next", "renameat(") +} + +fn require_stage_sync(point: &str, stage: &str, publication: &str) -> Result<(), Box> { + let output = Command::new("strace") + .args(["-f", "-yy", "-e", "trace=fsync,linkat,renameat,renameat2"]) + .arg(env!("CARGO_BIN_EXE_xtask")) + .args(["durability-crash-matrix", "--case", point, "after"]) + .output()?; + assert!( + output.status.success(), + "traced recovery failed: {output:?}" + ); + let trace = std::str::from_utf8(&output.stderr)?; + // The killed writer has a [pid ...] prefix. Only the surviving parent + // performs restart recovery; its unprefixed syscalls are the observation. + let source = format!("\"{stage}\""); + let boundary = trace + .lines() + .position(|line| line.starts_with(publication) && line.contains(&source)) + .ok_or_else(|| format!("no recovery publication witness for {stage}: {trace}"))?; + let file = format!("/retention/{stage}>"); + assert!( + trace.lines().take(boundary).any(|line| { + line.starts_with("fsync(") && line.contains(&file) && line.ends_with("= 0") + }), + "recovery must successfully synchronize {stage} contents before publication: {trace}" + ); + Ok(()) +} diff --git a/xtask/tests/retention_store_v2_protocol_contract.rs b/xtask/tests/retention_store_v2_protocol_contract.rs index d3e7fbe9..7926925a 100644 --- a/xtask/tests/retention_store_v2_protocol_contract.rs +++ b/xtask/tests/retention_store_v2_protocol_contract.rs @@ -52,6 +52,7 @@ fn version_two_is_one_routed_protocol() -> Result<(), Box "successor to `keep.segment-store/v1`", "[Retention records](retention.md)", "[Retention publication](retention-publication.md)", + "[Retention publication recovery](retention-recovery.md)", "[Closure verification](closure.md)", "[Closure corruption boundary](closure-corruption.md)", "[GC and disposition records](gc.md)", @@ -138,33 +139,6 @@ fn gc_records_are_bounded_before_their_implementation() -> Result<(), Box Result<(), Box> { - let requirements = normalized(&read(&format!("{FORMAT_ROOT}/requirements.md"))?); - - for required in [ - "`KEEP-RETENTION-001`", - "`KEEP-RETENTION-010`", - "`KEEP-MIGRATION-001`", - "`KEEP-MIGRATION-008`", - "`KEEP-GC-001`", - "Planned in #19", - "Planned in #21", - "golden-format", - "model-based", - "corruption", - "crash-injection", - "fuzz", - ] { - assert!( - requirements.contains(required), - "segment-store v2 requirement ledger omits `{required}`" - ); - } - Ok(()) -} - #[test] fn version_two_pages_stay_within_the_review_threshold() -> Result<(), Box> { for name in [ @@ -179,6 +153,7 @@ fn version_two_pages_stay_within_the_review_threshold() -> Result<(), Box Result<(), Box Result<(), Box