From 7b5af5e2501ecab5d06bfb78e83ed58d0d2c48f1 Mon Sep 17 00:00:00 2001 From: Huang-Ming Huang Date: Fri, 11 Sep 2026 17:49:19 +0000 Subject: [PATCH 01/16] Add WIRE-385 schedule recovery scenario Change-Id: I274aebcb28ed8a569d50d0d7e2e436ee8fa62136 --- .../contracts/sysio/OpregContractSteps.ts | 58 +- .../sysio/OpregContractSteps.test.ts | 18 + .../package.json | 2 +- .../src/TerminationScenario.ts | 745 +++++++++++++++++- .../src/TerminationScenarioConstants.ts | 59 +- 5 files changed, 870 insertions(+), 12 deletions(-) diff --git a/packages/cluster-tool/src/orchestration/steps/contracts/sysio/OpregContractSteps.ts b/packages/cluster-tool/src/orchestration/steps/contracts/sysio/OpregContractSteps.ts index e3ed48381..e64f50342 100644 --- a/packages/cluster-tool/src/orchestration/steps/contracts/sysio/OpregContractSteps.ts +++ b/packages/cluster-tool/src/orchestration/steps/contracts/sysio/OpregContractSteps.ts @@ -7,7 +7,7 @@ import { } from "../../../ClusterBuildStep.js" import type { StepInput } from "../../../StepRunner.js" -const { SysioContractName } = SysioContracts +const { SysioContractAccount, SysioContractName } = SysioContracts /** Steps for `sysio.opreg` (operator registry) actions. */ export namespace OpregContractSteps { @@ -18,7 +18,9 @@ export namespace OpregContractSteps { } /** `sysio.opreg::setconfig` — availability caps, termination thresholds, collateral minimums. */ - export function planSetconfig( + export function planSetconfig< + C extends ClusterBuildContext = ClusterBuildContext + >( actor: Report.Actor, name: string, description: string, @@ -54,7 +56,9 @@ export namespace OpregContractSteps { } /** `sysio.opreg::regoperator` — register a batch operator / underwriter / producer. */ - export function planRegoperator( + export function planRegoperator< + C extends ClusterBuildContext = ClusterBuildContext + >( actor: Report.Actor, name: string, description: string, @@ -82,4 +86,52 @@ export namespace OpregContractSteps { .getSysioContract(SysioContractName.opreg) .actions.regoperator.invoke(input.data) } + + /** Input for {@link planSlash} — the generated `opreg::slash` data. */ + export interface SlashInput extends StepInput { + readonly kind: "OpregContractSteps.SlashInput" + readonly data: SysioContracts.SysioOpregSlashAction + } + + /** + * `sysio.opreg::slash` — mark an operator slashed under the challenge + * contract authority required by the on-chain action. + */ + export function planSlash< + C extends ClusterBuildContext = ClusterBuildContext + >( + actor: Report.Actor, + name: string, + description: string, + options: ClusterBuildStepOptions, + data: SysioContracts.SysioOpregSlashAction + ): ClusterBuildStep { + return ClusterBuildStep.create( + actor, + name, + description, + options, + { kind: "OpregContractSteps.SlashInput", data }, + runSlash + ) + } + + /** Named runner — `sysio.opreg::slash` authorized by `sysio.chalg`. */ + export async function runSlash( + ctx: C, + input: SlashInput, + signal: AbortSignal + ): Promise { + signal.throwIfAborted() + await ctx.wire + .getSysioContract(SysioContractName.opreg) + .actions.slash.invoke(input.data, { + authorization: [ + { + actor: SysioContractAccount[SysioContractName.chalg], + permission: "active" + } + ] + }) + } } diff --git a/packages/cluster-tool/tests/orchestration/steps/contracts/sysio/OpregContractSteps.test.ts b/packages/cluster-tool/tests/orchestration/steps/contracts/sysio/OpregContractSteps.test.ts index ec9119a93..62f0569da 100644 --- a/packages/cluster-tool/tests/orchestration/steps/contracts/sysio/OpregContractSteps.test.ts +++ b/packages/cluster-tool/tests/orchestration/steps/contracts/sysio/OpregContractSteps.test.ts @@ -49,4 +49,22 @@ describe("Steps.contracts.sysio.opreg", () => { expect(step.input.data.is_bootstrapped).toBe(true) expect(typeof step.runner).toBe("function") }) + + it("slash carries the opreg::slash data", () => { + const data: SysioContracts.SysioOpregSlashAction = { + account: "batchop.a", + reason: "schedule recovery regression" + } + const step = Steps.contracts.sysio.opreg.planSlash( + Report.Actor.Sysio, + "slash-current-operator", + "slash one operator from the current group", + {}, + data + ) + expect(step.actor).toBe(Report.Actor.Sysio) + expect(step.input.kind).toBe("OpregContractSteps.SlashInput") + expect(step.input.data).toBe(data) + expect(typeof step.runner).toBe("function") + }) }) diff --git a/packages/flow-batch-operator-termination/package.json b/packages/flow-batch-operator-termination/package.json index 17730193a..13e458dfb 100644 --- a/packages/flow-batch-operator-termination/package.json +++ b/packages/flow-batch-operator-termination/package.json @@ -3,7 +3,7 @@ "version": "0.1.17", "private": true, "type": "commonjs", - "description": "Flow: Batch Operator Termination via Delivery Underperformance", + "description": "Flow: Batch Operator Termination, Schedule Freeze, and Recovery", "scripts": { "build": "tsc -b tsconfig.json", "test": "node lib/index.js", diff --git a/packages/flow-batch-operator-termination/src/TerminationScenario.ts b/packages/flow-batch-operator-termination/src/TerminationScenario.ts index 009af509b..d669a735f 100644 --- a/packages/flow-batch-operator-termination/src/TerminationScenario.ts +++ b/packages/flow-batch-operator-termination/src/TerminationScenario.ts @@ -5,14 +5,18 @@ import { SysioContracts } from "@wireio/sdk-core" import { OperatorType } from "@wireio/opp-typescript-models" import { ClusterBuildPhase, + ClusterBuildStep, + ClusterConfigProvider, EthereumCollateralTool, FlowScenario, + OperatorDaemonTool, Report, SolanaCollateralTool, SolanaOutpostBootstrapper, SolanaOutpostProgramTool, Steps, WireOperatorProvisioningTool, + contractView, getLogger, matchesProtoEnum, outputKey, @@ -28,10 +32,57 @@ import { TerminationScenarioConstants as Constants } from "./TerminationScenario const log = getLogger(__filename) -const { SysioContractName, SysioOpregActiontype, SysioOpregOperatorstatus } = - SysioContracts +const { + SysioContractName, + SysioOpregActiontype, + SysioOpregOperatortype, + SysioOpregOperatorstatus +} = SysioContracts const { Actor } = Report +/** Minimal Ethereum inbound surface needed to prove sequential acceptance. */ +interface EthereumInboundView { + activeGroupIndex(): Promise + batchOpGroups(groupIndex: number, memberIndex: number): Promise + epochDeliveries(epochIndex: number, operator: string): Promise + nextEpochIndex(): Promise +} + +/** Anchor-decoded subset of the Solana outpost configuration account. */ +interface SolanaOutpostConfigAccount { + nextEpochIndex: BN +} + +/** State captured after the first incomplete schedule window is withheld. */ +interface WithheldScheduleCheckpoint { + epochIndex: number + activeGroup: string[] + ethereumNextEpoch: number + solanaNextEpoch: number +} + +/** State captured after a replacement makes the held window publishable. */ +interface RepairedScheduleCheckpoint extends WithheldScheduleCheckpoint { + nextGroup: string[] +} + +const WithheldScheduleCheckpointKey = outputKey( + "TerminationScenario.withheldScheduleCheckpoint", + "depot duty and outpost epoch cursors after the first withheld schedule window" +) +const RepairedScheduleCheckpointKey = outputKey( + "TerminationScenario.repairedScheduleCheckpoint", + "held duty, next group, and outpost cursors after the repaired window is published" +) +const HeldSlashAccountKey = outputKey( + "TerminationScenario.heldSlashAccount", + "operator removed from the announced group while schedule duty is held" +) +const HeldSlashEpochKey = outputKey( + "TerminationScenario.heldSlashEpoch", + "depot epoch in which a member of the held group was slashed" +) + /** * Post-deposit snapshot of the doomed operator's ETH wallet balance (wei), * captured after BOTH bonds landed but BEFORE termination begins. The remit @@ -128,12 +179,252 @@ interface SolanaCollateralLedgerEntry { /** The slice of the SOL outpost's `OperatorRegistry` PDA account this flow reads. */ interface SolanaOperatorRegistryAccount { + activeGroupIndex: number collateralByCode: SolanaCollateralLedgerEntry[] + groupCount: number + groups: SolanaOperatorGroup[] +} + +/** One fixed-capacity group in the zero-copy Solana operator registry. */ +interface SolanaOperatorGroup { + memberCount: number + members: PublicKey[] +} + +/** Signer records retained for one accepted Solana inbound epoch. */ +interface SolanaOperatorDelivery { + operator: PublicKey +} + +/** Signer records retained for one accepted Solana inbound epoch. */ +interface SolanaEpochDeliveriesAccount { + deliveries: SolanaOperatorDelivery[] } /** Anchor account-client surface for a runtime-loaded IDL (untyped `Program` namespace). */ interface SolanaAccountClient { fetch(address: PublicKey): Promise + fetchNullable(address: PublicKey): Promise +} + +/** Bound Solana OPP account readers and their program addresses. */ +interface SolanaOppAccounts { + accounts: Record + configAddress: PublicKey + programId: PublicKey +} + +/** Cross-chain signing identities for one replacement operator. */ +interface ReplacementAddresses { + ethereum: string + solana: PublicKey +} + +/** Compare ordered operator groups without depending on array identity. */ +function sameGroup(left: readonly string[], right: readonly string[]): boolean { + return ( + left.length === right.length && + left.every((value, index) => value === right[index]) + ) +} + +/** Require the full disjoint schedule used by the recovery regression. */ +function assertCompleteSchedule(groups: readonly string[][]): void { + Assert.equal( + groups.length, + Constants.BatchOperatorGroups, + "schedule group count changed" + ) + const members = new Set() + for (const group of groups) { + Assert.equal( + group.length, + Constants.OperatorsPerEpoch, + "schedule contains a short group" + ) + for (const member of group) { + Assert.ok( + !members.has(member), + `${member} is seated in more than one group` + ) + members.add(member) + } + } + Assert.equal( + members.size, + Constants.BatchOperatorCount, + "schedule window is not full" + ) +} + +/** Read the Ethereum outpost's next sequential inbound epoch. */ +async function readEthereumNextEpoch( + ctx: ClusterBuildContext +): Promise { + return Number(await loadEthereumInbound(ctx).nextEpochIndex()) +} + +/** Bind the deployed Ethereum inbound contract to its read-only flow surface. */ +function loadEthereumInbound(ctx: ClusterBuildContext): EthereumInboundView { + const deploymentsPath = ClusterConfigProvider.ethereumDeploymentsPath( + ctx.config + ) + const addresses = EthereumCollateralTool.loadOutpostAddresses(deploymentsPath) + return contractView( + addresses.OPPInbound, + EthereumCollateralTool.loadOutpostAbi( + ctx.config.ethereumPath, + "OPPInbound" + ), + ctx.ethereum.wallet.signer + ) +} + +/** Read the Solana outpost's next sequential inbound epoch. */ +async function readSolanaNextEpoch(ctx: ClusterBuildContext): Promise { + const { accounts, configAddress } = loadSolanaOppAccounts(ctx) + const config = (await accounts[ + Constants.SolanaOutpostConfigAccountName + ].fetch(configAddress)) as SolanaOutpostConfigAccount + return Number(config.nextEpochIndex.toString()) +} + +/** Load the Solana OPP program and the account namespace used by flow reads. */ +function loadSolanaOppAccounts(ctx: ClusterBuildContext): SolanaOppAccounts { + const reader = ctx.keyStore.assertOperator( + Constants.RecoverySolanaReaderLabel + ) + const program = SolanaCollateralTool.loadOppOutpostProgram( + ctx, + solanaKeypair(reader.solana) + ) + const configAddress = SolanaOutpostProgramTool.derivePda( + program.programId, + Buffer.from(SolanaOutpostBootstrapper.PdaSeed.OutpostConfig) + ) + const accounts: Record = program.account + return { accounts, configAddress, programId: program.programId } +} + +/** Cross-chain signing addresses for one WIRE operator account. */ +function replacementAddresses( + ctx: ClusterBuildContext, + account: string +): ReplacementAddresses { + const operator = ctx.keyStore.operators.find( + entry => entry.account === account + ) + Assert.ok(operator != null, `${account} is absent from the cluster key store`) + Assert.ok(operator.ethereum != null, `${account} has no Ethereum identity`) + Assert.ok(operator.solana != null, `${account} has no Solana identity`) + return { + ethereum: operator.ethereum.address, + solana: solanaKeypair(operator.solana).publicKey + } +} + +/** + * Prove one replacement signed accepted deliveries for the same duty epoch on + * both outposts. Each contract records a signer only after its active-group + * admission check, so this is also direct evidence that the replacement's + * propagated address was seated rather than merely present in the depot group. + */ +async function replacementDeliveredOnBothOutposts( + ctx: ClusterBuildContext, + account: string, + epochIndex: number +): Promise { + const addresses = replacementAddresses(ctx, account) + const ethereumDigest = await loadEthereumInbound(ctx).epochDeliveries( + epochIndex, + addresses.ethereum + ) + if (/^0x0{64}$/i.test(ethereumDigest)) return false + + const { accounts, programId } = loadSolanaOppAccounts(ctx) + const epochBytes = Buffer.alloc(4) + epochBytes.writeUInt32LE(epochIndex) + const deliveriesAddress = SolanaOutpostProgramTool.derivePda( + programId, + Buffer.from("epoch_deliveries"), + epochBytes + ) + const deliveries = (await accounts.epochDeliveries.fetchNullable( + deliveriesAddress + )) as SolanaEpochDeliveriesAccount | null + return ( + deliveries != null && + deliveries.deliveries.some(entry => entry.operator.equals(addresses.solana)) + ) +} + +/** Read whether an operator is SLASHED in the depot registry. */ +async function operatorIsSlashed( + ctx: ClusterBuildContext, + account: string +): Promise { + const { rows } = await ctx.wire + .getSysioContract(SysioContractName.opreg) + .tables.operators.query({ limit: Constants.OperatorsQueryLimit }) + const target = rows.find(row => row.account === account) + return ( + target != null && + matchesProtoEnum( + target.status, + SysioOpregOperatorstatus, + SysioOpregOperatorstatus.OPERATOR_STATUS_SLASHED + ) + ) +} + +/** Remove one eligible member from the group whose duty is currently held. */ +async function runSlashHeldGroupMember( + ctx: ClusterBuildContext, + _input: null, + signal: AbortSignal +): Promise { + signal.throwIfAborted() + const checkpoint = ctx.outputs.assert(WithheldScheduleCheckpointKey) + const { rows } = await ctx.wire + .getSysioContract(SysioContractName.opreg) + .tables.operators.query({ limit: Constants.OperatorsQueryLimit }) + const candidate = checkpoint.activeGroup.find(account => { + if ( + account === Constants.RecoverySlashTargetAccount || + account === Constants.RecoverySolanaReaderLabel + ) { + return false + } + const row = rows.find(operator => operator.account === account) + return ( + row != null && + matchesProtoEnum( + row.type, + SysioOpregOperatortype, + SysioOpregOperatortype.OPERATOR_TYPE_BATCH + ) && + matchesProtoEnum( + row.status, + SysioOpregOperatorstatus, + SysioOpregOperatorstatus.OPERATOR_STATUS_ACTIVE + ) + ) + }) + Assert.ok(candidate != null, "held group has no eligible slash candidate") + const state = await Steps.contracts.sysio.epoch.readEpochState(ctx) + await Steps.contracts.sysio.opreg.runSlash( + ctx, + { + kind: "OpregContractSteps.SlashInput", + data: { + account: candidate, + reason: "WIRE-385 held-duty recovery regression" + } + }, + signal + ) + ctx.outputs.set(HeldSlashAccountKey, candidate) + ctx.outputs.set(HeldSlashEpochKey, Number(state.current_epoch_index)) } /** The SOL outpost's on-chain collateral ledger from the `OperatorRegistry` PDA (a read). */ @@ -187,15 +478,22 @@ async function readSolanaCollateralLedger( * outpost's escrow ledger returns to 0, and each wallet is credited the * exact bond amount (wei/lamport-exact — any drift means the outpost decoded * a different amount than the depot encoded). + * 9. **Schedule recovery** — remove an announced operator at the exact roster + * floor, prove the depot freezes that duty while both outposts keep accepting + * sequential epochs, remove another member while frozen, add two replacements, + * and prove publication and rotation resume through replacement-backed duty. */ export class TerminationScenario extends FlowScenario { readonly name = "flow-batch-operator-termination" readonly description = - "Non-bootstrapped batch operator bonds ETH + SOL, misses its scheduled deliveries, is terminated, and both bonds are remitted back" + "Terminate and remit a non-bootstrapped operator, then freeze, repair, and resume a depleted batch schedule across Ethereum and Solana" override readonly defaults: ClusterBuildOptions = { epochDurationSec: Constants.EpochDurationSec, batchOperatorCount: Constants.BatchOperatorCount, + operatorsPerEpoch: Constants.OperatorsPerEpoch, + batchOpGroups: Constants.BatchOperatorGroups, + adHocCount: Constants.RecoveryAdHocDaemonCount, terminateMaxConsecutiveMisses: Constants.TerminateMaxConsecutiveMisses, // Depot must enforce "ACTIVE requires the minimum on EVERY registered // outpost chain" — otherwise the operator flips ACTIVE on an empty @@ -708,5 +1006,446 @@ export class TerminationScenario extends FlowScenario { quickStepOptions ) ) + + // ── 9. Reach the exact roster floor with the fixed slash target on duty ── + ClusterBuildPhase.create( + cluster, + "PrepareScheduleRecovery", + "The terminated test operator is gone and batchop.a reaches current duty in a complete window" + ).push( + verifyStep( + Actor.Sysio, + "exact-minimum-window-ready", + "three disjoint groups of three remain, with batchop.a in the current group", + async ctx => { + await pollUntil( + `${Constants.RecoverySlashTargetAccount} reaches current duty in a complete window`, + async () => { + const state = + await Steps.contracts.sysio.epoch.readEpochState(ctx) + const groups = state.batch_op_groups + const complete = + groups.length === Constants.BatchOperatorGroups && + groups.every( + group => group.length === Constants.OperatorsPerEpoch + ) + if (!complete) return false + assertCompleteSchedule(groups) + const current = groups[state.current_batch_op_group] ?? [] + return current.includes(Constants.RecoverySlashTargetAccount) + }, + Constants.recoveryDeadlineMs(Constants.BatchOperatorGroups + 2), + Constants.PollIntervalMs + ) + }, + { + timeoutMs: Constants.recoveryDeadlineMs( + Constants.BatchOperatorGroups + 2 + ) + } + ) + ) + + // ── 10. Remove one current member at the exact floor → withhold ── + ClusterBuildPhase.create( + cluster, + "StarveScheduleWindow", + "Slashing one seated operator makes the next tail one seat short" + ).push( + Steps.contracts.sysio.opreg.planSlash( + Actor.Sysio, + "slash-current-member", + `slash ${Constants.RecoverySlashTargetAccount} to exercise withheld-window recovery`, + {}, + { + account: Constants.RecoverySlashTargetAccount, + reason: "WIRE-385 schedule recovery regression" + } + ), + verifyStep( + Actor.Sysio, + "capture-withheld-window", + "the first slide enters the announced group and persists an incomplete window", + async ctx => { + let checkpoint: WithheldScheduleCheckpoint | null = null + await pollUntil( + "an incomplete schedule window is persisted after the target is slashed", + async () => { + const state = + await Steps.contracts.sysio.epoch.readEpochState(ctx) + const current = + state.batch_op_groups[state.current_batch_op_group] ?? [] + const incomplete = state.batch_op_groups.some( + group => group.length !== Constants.OperatorsPerEpoch + ) + if ( + !(await operatorIsSlashed( + ctx, + Constants.RecoverySlashTargetAccount + )) || + !incomplete || + current.length === 0 + ) { + return false + } + const [ethereumNextEpoch, solanaNextEpoch] = await Promise.all([ + readEthereumNextEpoch(ctx), + readSolanaNextEpoch(ctx) + ]) + const expectedNextEpoch = Number(state.current_epoch_index) + 1 + if ( + ethereumNextEpoch < expectedNextEpoch || + solanaNextEpoch < expectedNextEpoch + ) { + return false + } + checkpoint = { + epochIndex: Number(state.current_epoch_index), + activeGroup: [...current], + ethereumNextEpoch, + solanaNextEpoch + } + return true + }, + Constants.recoveryDeadlineMs(4), + Constants.PollIntervalMs + ) + Assert.ok(checkpoint != null, "withheld checkpoint was not captured") + ctx.outputs.set(WithheldScheduleCheckpointKey, checkpoint) + }, + { timeoutMs: Constants.recoveryDeadlineMs(4) } + ) + ) + + // ── 11. Keep announced duty while both outposts accept later epochs ── + ClusterBuildPhase.create( + cluster, + "HoldAnnouncedDuty", + "The current group remains fixed while the short future window is withheld" + ).push( + verifyStep( + Actor.Sysio, + "held-duty-remains-live", + "two epochs land on Ethereum and Solana without changing the announced current group", + async ctx => { + const checkpoint = ctx.outputs.assert(WithheldScheduleCheckpointKey) + await pollUntil( + "both outposts advance twice while the depot duty remains held", + async () => { + const state = + await Steps.contracts.sysio.epoch.readEpochState(ctx) + const current = + state.batch_op_groups[state.current_batch_op_group] ?? [] + if ( + Number(state.current_epoch_index) > checkpoint.epochIndex && + !sameGroup(current, checkpoint.activeGroup) + ) { + throw new Error( + `duty rotated before repair: ${checkpoint.activeGroup.join(",")} -> ${current.join(",")}` + ) + } + Assert.ok( + state.batch_op_groups.some( + group => group.length !== Constants.OperatorsPerEpoch + ), + "schedule became complete before a replacement was provisioned" + ) + return ( + Number(state.current_epoch_index) >= + checkpoint.epochIndex + Constants.RecoveryHeldEpochAdvances && + (await readEthereumNextEpoch(ctx)) >= + checkpoint.ethereumNextEpoch + + Constants.RecoveryHeldEpochAdvances && + (await readSolanaNextEpoch(ctx)) >= + checkpoint.solanaNextEpoch + + Constants.RecoveryHeldEpochAdvances + ) + }, + Constants.recoveryDeadlineMs( + Constants.RecoveryHeldEpochAdvances + 4 + ), + Constants.PollIntervalMs + ) + }, + { + timeoutMs: Constants.recoveryDeadlineMs( + Constants.RecoveryHeldEpochAdvances + 4 + ) + } + ) + ) + + // ── 12. Lose a member of held duty without changing its announced seats ── + ClusterBuildPhase.create( + cluster, + "DegradeHeldDuty", + "An announced seat becomes ineligible while the incomplete future window is held" + ).push( + ClusterBuildStep.create( + Actor.Sysio, + "slash-held-member", + "slash one eligible held-group member selected from live chain state", + {}, + null, + runSlashHeldGroupMember + ), + verifyStep( + Actor.Sysio, + "held-seat-preserved", + "the next epoch retains the announced seat as a denominator placeholder", + async ctx => { + const withheld = ctx.outputs.assert(WithheldScheduleCheckpointKey) + const slashedAccount = ctx.outputs.assert(HeldSlashAccountKey) + const slashEpoch = ctx.outputs.assert(HeldSlashEpochKey) + await pollUntil( + "held duty survives a member becoming ineligible", + async () => { + const state = + await Steps.contracts.sysio.epoch.readEpochState(ctx) + const current = + state.batch_op_groups[state.current_batch_op_group] ?? [] + if (!sameGroup(current, withheld.activeGroup)) { + throw new Error( + `held duty changed after ${slashedAccount} was slashed: ${withheld.activeGroup.join(",")} -> ${current.join(",")}` + ) + } + Assert.ok( + state.batch_op_groups.some( + group => group.length !== Constants.OperatorsPerEpoch + ), + "schedule became complete before replacements were provisioned" + ) + if ( + Number(state.current_epoch_index) <= slashEpoch || + !(await operatorIsSlashed(ctx, slashedAccount)) + ) { + return false + } + const [ethereumNextEpoch, solanaNextEpoch] = await Promise.all([ + readEthereumNextEpoch(ctx), + readSolanaNextEpoch(ctx) + ]) + const expectedNextEpoch = Number(state.current_epoch_index) + 1 + return ( + ethereumNextEpoch >= expectedNextEpoch && + solanaNextEpoch >= expectedNextEpoch + ) + }, + Constants.recoveryDeadlineMs(4), + Constants.PollIntervalMs + ) + }, + { timeoutMs: Constants.recoveryDeadlineMs(4) } + ) + ) + + // ── 13. Add ACTIVE standbys and their daemons for the two roster losses ── + WireOperatorProvisioningTool.planOperatorAccountProvisioning( + cluster, + "ProvisionScheduleReplacement", + "Provision two bootstrapped batch operators to repair both roster losses", + {}, + Constants.RecoveryOperatorLabels.map((label, index) => ({ + label, + type: OperatorType.BATCH, + ethereumHdIndex: Constants.RecoveryOperatorEthereumHdIndices[index], + isBootstrapped: true + })) + ) + + ClusterBuildPhase.create( + cluster, + "StartScheduleReplacementDaemons", + "Start both replacement batch-operator daemons before they enter rotation" + ).push( + ...Constants.RecoveryOperatorLabels.map(label => + OperatorDaemonTool.planDaemonStart( + Actor.BatchOperator, + `start-${label}-daemon`, + `start ${label}'s batch-operator daemon`, + {}, + label + ) + ) + ) + + // ── 14. Repair and publish the future window without moving held duty ── + ClusterBuildPhase.create( + cluster, + "RepairScheduleWindow", + "The replacement completes and publishes lookahead without moving current duty" + ).push( + verifyStep( + Actor.Sysio, + "complete-window-published", + "the held window becomes full and names the next group while current duty is unchanged", + async ctx => { + const withheld = ctx.outputs.assert(WithheldScheduleCheckpointKey) + const replacements = Constants.RecoveryOperatorLabels.map( + label => ctx.keyStore.assertOperator(label).account + ) + let checkpoint: RepairedScheduleCheckpoint | null = null + await pollUntil( + "the replacement fills the held schedule window", + async () => { + const state = + await Steps.contracts.sysio.epoch.readEpochState(ctx) + const groups = state.batch_op_groups + const current = groups[state.current_batch_op_group] ?? [] + const complete = + groups.length === Constants.BatchOperatorGroups && + groups.every( + group => group.length === Constants.OperatorsPerEpoch + ) + if ( + !complete || + !groups.some(group => + group.some(member => replacements.includes(member)) + ) + ) { + return false + } + Assert.ok( + sameGroup(current, withheld.activeGroup), + "current duty moved before the repaired lookahead was published" + ) + assertCompleteSchedule(groups) + const nextGroup = + groups[state.current_batch_op_group + 1] ?? current + const [ethereumNextEpoch, solanaNextEpoch] = await Promise.all([ + readEthereumNextEpoch(ctx), + readSolanaNextEpoch(ctx) + ]) + const expectedNextEpoch = Number(state.current_epoch_index) + 1 + if ( + ethereumNextEpoch < expectedNextEpoch || + solanaNextEpoch < expectedNextEpoch + ) { + return false + } + checkpoint = { + epochIndex: Number(state.current_epoch_index), + activeGroup: [...current], + nextGroup: [...nextGroup], + ethereumNextEpoch, + solanaNextEpoch + } + return true + }, + Constants.recoveryDeadlineMs(4), + Constants.PollIntervalMs + ) + Assert.ok(checkpoint != null, "repaired checkpoint was not captured") + ctx.outputs.set(RepairedScheduleCheckpointKey, checkpoint) + }, + { timeoutMs: Constants.recoveryDeadlineMs(4) } + ) + ) + + // ── 15. Rotate only after the repaired lookahead has been published ── + ClusterBuildPhase.create( + cluster, + "ResumeScheduleRotation", + "The next published group takes duty and both outposts remain sequential" + ).push( + verifyStep( + Actor.Sysio, + "rotation-resumes-after-publication", + "the announced next group serves an epoch accepted by Ethereum and Solana", + async ctx => { + const repaired = ctx.outputs.assert(RepairedScheduleCheckpointKey) + await pollUntil( + "the repaired next group serves an epoch accepted by both outposts", + async () => { + const state = + await Steps.contracts.sysio.epoch.readEpochState(ctx) + const current = + state.batch_op_groups[state.current_batch_op_group] ?? [] + const [ethereumNextEpoch, solanaNextEpoch] = await Promise.all([ + readEthereumNextEpoch(ctx), + readSolanaNextEpoch(ctx) + ]) + const expectedNextEpoch = Number(state.current_epoch_index) + 1 + return ( + Number(state.current_epoch_index) > repaired.epochIndex && + sameGroup(current, repaired.nextGroup) && + ethereumNextEpoch >= expectedNextEpoch && + solanaNextEpoch >= expectedNextEpoch + ) + }, + Constants.recoveryDeadlineMs(4), + Constants.PollIntervalMs + ) + }, + { timeoutMs: Constants.recoveryDeadlineMs(4) } + ) + ) + + // ── 16. Prove each replacement is seated and delivers on both outposts ── + ClusterBuildPhase.create( + cluster, + "ExerciseScheduleReplacement", + "Each replacement signs an accepted duty epoch on Ethereum and Solana" + ).push( + verifyStep( + Actor.BatchOperator, + "replacement-duty-serves", + "each replacement is admitted as an active-group signer on Ethereum and Solana", + async ctx => { + const replacements = Constants.RecoveryOperatorLabels.map( + label => ctx.keyStore.assertOperator(label).account + ) + const dutyEpochs = new Map() + await pollUntil( + "each replacement signs an accepted duty epoch on both outposts", + async () => { + const state = + await Steps.contracts.sysio.epoch.readEpochState(ctx) + const current = + state.batch_op_groups[state.current_batch_op_group] ?? [] + const currentEpoch = Number(state.current_epoch_index) + for (const replacement of replacements) { + if ( + current.includes(replacement) && + !dutyEpochs.has(replacement) + ) { + dutyEpochs.set(replacement, currentEpoch) + } + } + assertCompleteSchedule(state.batch_op_groups) + if (dutyEpochs.size !== replacements.length) return false + + const [ethereumNextEpoch, solanaNextEpoch] = await Promise.all([ + readEthereumNextEpoch(ctx), + readSolanaNextEpoch(ctx) + ]) + for (const replacement of replacements) { + const dutyEpoch = dutyEpochs.get(replacement) + Assert.ok(dutyEpoch != null) + if ( + ethereumNextEpoch < dutyEpoch + 1 || + solanaNextEpoch < dutyEpoch + 1 || + !(await replacementDeliveredOnBothOutposts( + ctx, + replacement, + dutyEpoch + )) + ) { + return false + } + } + return true + }, + Constants.recoveryDeadlineMs(Constants.BatchOperatorGroups + 4), + Constants.PollIntervalMs + ) + }, + { + timeoutMs: Constants.recoveryDeadlineMs( + Constants.BatchOperatorGroups + 4 + ) + } + ) + ) } } diff --git a/packages/flow-batch-operator-termination/src/TerminationScenarioConstants.ts b/packages/flow-batch-operator-termination/src/TerminationScenarioConstants.ts index 44cbdf789..e17a0bc87 100644 --- a/packages/flow-batch-operator-termination/src/TerminationScenarioConstants.ts +++ b/packages/flow-batch-operator-termination/src/TerminationScenarioConstants.ts @@ -35,6 +35,10 @@ export namespace TerminationScenarioConstants { * still cover consensus majority on every group. */ export const BatchOperatorCount = 9 + /** Number of disjoint groups retained in the rolling schedule window. */ + export const BatchOperatorGroups = 3 + /** Operators per group; three keeps majority consensus unambiguous. */ + export const OperatorsPerEpoch = 3 /** * Override for `terminate_max_consecutive_misses` so `termcheck` fires inside * the flow's budget: 2 consecutive missed scheduled epochs flip TERMINATED. @@ -96,29 +100,74 @@ export namespace TerminationScenarioConstants { export const OperatorsQueryLimit = 100 /** Anchor account-namespace name of the SOL outpost's `OperatorRegistry` PDA account. */ export const SolanaOperatorRegistryAccountName = "operatorRegistry" + /** Anchor account namespace for the Solana outpost configuration PDA. */ + export const SolanaOutpostConfigAccountName = "outpostConfig" + + /** Bootstrapped operator removed at the exact nine-operator roster floor. */ + export const RecoverySlashTargetAccount = "batchop.a" + /** Healthy operator key used for read-only Solana account access. */ + export const RecoverySolanaReaderLabel = "batchop.b" + /** Harness labels for operators that repair two independent roster losses. */ + export const RecoveryOperatorLabels = ["recoverya", "recoveryb"] as const + /** Anvil HD slots beyond the bootstrapped roster and underwriters. */ + export const RecoveryOperatorEthereumHdIndices = [36, 37] as const + /** Ad-hoc daemons started for the two flow-provisioned replacements. */ + export const RecoveryAdHocDaemonCount = 2 + /** Epoch advances required while the announced duty remains frozen. */ + export const RecoveryHeldEpochAdvances = 2 /** Deadline for the ETH deposit to credit the depot balance row. */ export function ethereumDepositDeadlineMs(): number { - return ProtocolTiming.effectiveEpochSec(EpochDurationSec) * EthereumDepositRelayEpochs * MsPerSecond + return ( + ProtocolTiming.effectiveEpochSec(EpochDurationSec) * + EthereumDepositRelayEpochs * + MsPerSecond + ) } /** Deadline for the SOL deposit to land and the ACTIVE flip to follow. */ export function solanaActivationDeadlineMs(): number { - return ProtocolTiming.effectiveEpochSec(EpochDurationSec) * SolanaActivationEpochs * MsPerSecond + return ( + ProtocolTiming.effectiveEpochSec(EpochDurationSec) * + SolanaActivationEpochs * + MsPerSecond + ) } /** Deadline for the operator to appear in `epochstate.batch_op_groups`. */ export function scheduleWindowDeadlineMs(): number { - return ProtocolTiming.effectiveEpochSec(EpochDurationSec) * ScheduleWindowEpochs * MsPerSecond + return ( + ProtocolTiming.effectiveEpochSec(EpochDurationSec) * + ScheduleWindowEpochs * + MsPerSecond + ) } /** Deadline for the miss window to accumulate and `termcheck` to flip TERMINATED. */ export function terminationDeadlineMs(): number { - return ProtocolTiming.effectiveEpochSec(EpochDurationSec) * MissAccumulationEpochs * MsPerSecond + return ( + ProtocolTiming.effectiveEpochSec(EpochDurationSec) * + MissAccumulationEpochs * + MsPerSecond + ) } /** Deadline for the post-termination WITHDRAW_REMIT effects on either outpost. */ export function remitDeadlineMs(): number { - return ProtocolTiming.effectiveEpochSec(EpochDurationSec) * RemitPropagationEpochs * MsPerSecond + return ( + ProtocolTiming.effectiveEpochSec(EpochDurationSec) * + RemitPropagationEpochs * + MsPerSecond + ) + } + + /** Deadline for a recovery condition spanning the supplied epoch count. */ + export function recoveryDeadlineMs(epochCount: number): number { + return ( + ProtocolTiming.effectiveEpochSec(EpochDurationSec) * + epochCount * + MsPerSecond + + PollDeadlineBufferMs + ) } } From 82c56a607a0d2e3d8766d80b4e98cdb05d1dff3f Mon Sep 17 00:00:00 2001 From: Huang-Ming Huang Date: Fri, 11 Sep 2026 19:27:12 +0000 Subject: [PATCH 02/16] Address WIRE-385 recovery scenario feedback Change-Id: I5fda20e3c43fb9d2912410cd953c8bca4e4ba047 --- .../src/TerminationScenario.ts | 128 +++++++++++++++--- .../src/TerminationScenarioConstants.ts | 2 - 2 files changed, 109 insertions(+), 21 deletions(-) diff --git a/packages/flow-batch-operator-termination/src/TerminationScenario.ts b/packages/flow-batch-operator-termination/src/TerminationScenario.ts index d669a735f..1983b3f20 100644 --- a/packages/flow-batch-operator-termination/src/TerminationScenario.ts +++ b/packages/flow-batch-operator-termination/src/TerminationScenario.ts @@ -82,6 +82,10 @@ const HeldSlashEpochKey = outputKey( "TerminationScenario.heldSlashEpoch", "depot epoch in which a member of the held group was slashed" ) +const RecoverySlashAccountKey = outputKey( + "TerminationScenario.recoverySlashAccount", + "active current-group operator selected for the first recovery slash" +) /** * Post-deposit snapshot of the doomed operator's ETH wallet balance (wei), @@ -377,6 +381,85 @@ async function operatorIsSlashed( ) } +/** Read the ACTIVE batch-operator account names from the live registry. */ +async function readActiveBatchOperatorAccounts( + ctx: ClusterBuildContext +): Promise> { + const { rows } = await ctx.wire + .getSysioContract(SysioContractName.opreg) + .tables.operators.query({ limit: Constants.OperatorsQueryLimit }) + return new Set( + rows + .filter( + row => + matchesProtoEnum( + row.type, + SysioOpregOperatortype, + SysioOpregOperatortype.OPERATOR_TYPE_BATCH + ) && + matchesProtoEnum( + row.status, + SysioOpregOperatorstatus, + SysioOpregOperatorstatus.OPERATOR_STATUS_ACTIVE + ) + ) + .map(row => row.account) + ) +} + +/** Select and slash an active member of the live current group without crossing an epoch boundary. */ +async function runSlashRecoveryTarget( + ctx: ClusterBuildContext, + _input: null, + signal: AbortSignal +): Promise { + signal.throwIfAborted() + const before = await Steps.contracts.sysio.epoch.readEpochState(ctx) + assertCompleteSchedule(before.batch_op_groups) + const current = before.batch_op_groups[before.current_batch_op_group] ?? [] + const activeAccounts = await readActiveBatchOperatorAccounts(ctx) + Assert.equal( + activeAccounts.size, + Constants.BatchOperatorCount, + "recovery slash did not start at the exact active roster floor" + ) + Assert.ok( + current.every(account => activeAccounts.has(account)), + "current duty contains an inactive historical placeholder" + ) + const readerAccount = ctx.keyStore.assertOperator( + Constants.RecoverySolanaReaderLabel + ).account + const account = current.find(member => member !== readerAccount) + Assert.ok( + account != null, + "current duty has no eligible recovery slash target" + ) + await Steps.contracts.sysio.opreg.runSlash( + ctx, + { + kind: "OpregContractSteps.SlashInput", + data: { + account, + reason: "WIRE-385 schedule recovery regression" + } + }, + signal + ) + const after = await Steps.contracts.sysio.epoch.readEpochState(ctx) + const afterCurrent = after.batch_op_groups[after.current_batch_op_group] ?? [] + Assert.equal( + Number(after.current_epoch_index), + Number(before.current_epoch_index), + "epoch advanced across the recovery slash" + ) + Assert.ok( + sameGroup(afterCurrent, current), + "current duty changed across the recovery slash" + ) + ctx.outputs.set(RecoverySlashAccountKey, account) +} + /** Remove one eligible member from the group whose duty is currently held. */ async function runSlashHeldGroupMember( ctx: ClusterBuildContext, @@ -388,11 +471,12 @@ async function runSlashHeldGroupMember( const { rows } = await ctx.wire .getSysioContract(SysioContractName.opreg) .tables.operators.query({ limit: Constants.OperatorsQueryLimit }) + const firstSlashAccount = ctx.outputs.assert(RecoverySlashAccountKey) + const readerAccount = ctx.keyStore.assertOperator( + Constants.RecoverySolanaReaderLabel + ).account const candidate = checkpoint.activeGroup.find(account => { - if ( - account === Constants.RecoverySlashTargetAccount || - account === Constants.RecoverySolanaReaderLabel - ) { + if (account === firstSlashAccount || account === readerAccount) { return false } const row = rows.find(operator => operator.account === account) @@ -1007,19 +1091,19 @@ export class TerminationScenario extends FlowScenario { ) ) - // ── 9. Reach the exact roster floor with the fixed slash target on duty ── + // ── 9. Reach the exact roster floor with a fully active group on duty ── ClusterBuildPhase.create( cluster, "PrepareScheduleRecovery", - "The terminated test operator is gone and batchop.a reaches current duty in a complete window" + "The terminated test operator is gone and a complete window reaches active current duty" ).push( verifyStep( Actor.Sysio, "exact-minimum-window-ready", - "three disjoint groups of three remain, with batchop.a in the current group", + "three disjoint groups of three remain with a fully active current group", async ctx => { await pollUntil( - `${Constants.RecoverySlashTargetAccount} reaches current duty in a complete window`, + "a complete three-by-three window has a fully active current group", async () => { const state = await Steps.contracts.sysio.epoch.readEpochState(ctx) @@ -1032,7 +1116,17 @@ export class TerminationScenario extends FlowScenario { if (!complete) return false assertCompleteSchedule(groups) const current = groups[state.current_batch_op_group] ?? [] - return current.includes(Constants.RecoverySlashTargetAccount) + const activeAccounts = await readActiveBatchOperatorAccounts(ctx) + if (activeAccounts.size !== Constants.BatchOperatorCount) { + return false + } + const readerAccount = ctx.keyStore.assertOperator( + Constants.RecoverySolanaReaderLabel + ).account + return ( + current.every(account => activeAccounts.has(account)) && + current.some(account => account !== readerAccount) + ) }, Constants.recoveryDeadlineMs(Constants.BatchOperatorGroups + 2), Constants.PollIntervalMs @@ -1052,15 +1146,13 @@ export class TerminationScenario extends FlowScenario { "StarveScheduleWindow", "Slashing one seated operator makes the next tail one seat short" ).push( - Steps.contracts.sysio.opreg.planSlash( + ClusterBuildStep.create( Actor.Sysio, "slash-current-member", - `slash ${Constants.RecoverySlashTargetAccount} to exercise withheld-window recovery`, + "slash the selected active current-group member to exercise withheld-window recovery", {}, - { - account: Constants.RecoverySlashTargetAccount, - reason: "WIRE-385 schedule recovery regression" - } + null, + runSlashRecoveryTarget ), verifyStep( Actor.Sysio, @@ -1078,11 +1170,9 @@ export class TerminationScenario extends FlowScenario { const incomplete = state.batch_op_groups.some( group => group.length !== Constants.OperatorsPerEpoch ) + const slashTarget = ctx.outputs.assert(RecoverySlashAccountKey) if ( - !(await operatorIsSlashed( - ctx, - Constants.RecoverySlashTargetAccount - )) || + !(await operatorIsSlashed(ctx, slashTarget)) || !incomplete || current.length === 0 ) { diff --git a/packages/flow-batch-operator-termination/src/TerminationScenarioConstants.ts b/packages/flow-batch-operator-termination/src/TerminationScenarioConstants.ts index e17a0bc87..5daf807b7 100644 --- a/packages/flow-batch-operator-termination/src/TerminationScenarioConstants.ts +++ b/packages/flow-batch-operator-termination/src/TerminationScenarioConstants.ts @@ -103,8 +103,6 @@ export namespace TerminationScenarioConstants { /** Anchor account namespace for the Solana outpost configuration PDA. */ export const SolanaOutpostConfigAccountName = "outpostConfig" - /** Bootstrapped operator removed at the exact nine-operator roster floor. */ - export const RecoverySlashTargetAccount = "batchop.a" /** Healthy operator key used for read-only Solana account access. */ export const RecoverySolanaReaderLabel = "batchop.b" /** Harness labels for operators that repair two independent roster losses. */ From c8d83dd5985e5fc01834a6de40c1348fe52f6db3 Mon Sep 17 00:00:00 2001 From: Huang-Ming Huang Date: Fri, 11 Sep 2026 21:46:27 +0000 Subject: [PATCH 03/16] Verify WIRE-385 explicit schedule publication and activation Change-Id: I6373a6739601afad27340ae94f90983e7a583f17 --- .../contracts/sysio/EpochContractSteps.ts | 4 +- .../src/TerminationScenario.ts | 58 +++++++++++-------- 2 files changed, 37 insertions(+), 25 deletions(-) diff --git a/packages/cluster-tool/src/orchestration/steps/contracts/sysio/EpochContractSteps.ts b/packages/cluster-tool/src/orchestration/steps/contracts/sysio/EpochContractSteps.ts index 5b3a2588e..75403408e 100644 --- a/packages/cluster-tool/src/orchestration/steps/contracts/sysio/EpochContractSteps.ts +++ b/packages/cluster-tool/src/orchestration/steps/contracts/sysio/EpochContractSteps.ts @@ -125,8 +125,8 @@ export namespace EpochContractSteps { } /** - * The depot's whole sliding-window batch-operator schedule — every group, - * `[current, next, next+1]` at the default `batch_op_groups` of 3. + * The depot's last activated batch-operator window, including historical groups. + * Current duty is selected by its cursor; `next_batch_op_groups` is separate. * * @param ctx - The build context. * @returns The schedule groups (empty when the epoch state has no row yet). diff --git a/packages/flow-batch-operator-termination/src/TerminationScenario.ts b/packages/flow-batch-operator-termination/src/TerminationScenario.ts index 1983b3f20..3f95ff28f 100644 --- a/packages/flow-batch-operator-termination/src/TerminationScenario.ts +++ b/packages/flow-batch-operator-termination/src/TerminationScenario.ts @@ -53,7 +53,7 @@ interface SolanaOutpostConfigAccount { nextEpochIndex: BN } -/** State captured after the first incomplete schedule window is withheld. */ +/** State captured after the first schedule candidate is withheld. */ interface WithheldScheduleCheckpoint { epochIndex: number activeGroup: string[] @@ -416,6 +416,7 @@ async function runSlashRecoveryTarget( signal.throwIfAborted() const before = await Steps.contracts.sysio.epoch.readEpochState(ctx) assertCompleteSchedule(before.batch_op_groups) + assertCompleteSchedule(before.next_batch_op_groups) const current = before.batch_op_groups[before.current_batch_op_group] ?? [] const activeAccounts = await readActiveBatchOperatorAccounts(ctx) Assert.equal( @@ -1114,6 +1115,16 @@ export class TerminationScenario extends FlowScenario { group => group.length === Constants.OperatorsPerEpoch ) if (!complete) return false + const published = state.next_batch_op_groups + if ( + published.length !== Constants.BatchOperatorGroups || + published.some( + group => group.length !== Constants.OperatorsPerEpoch + ) + ) { + return false + } + assertCompleteSchedule(published) assertCompleteSchedule(groups) const current = groups[state.current_batch_op_group] ?? [] const activeAccounts = await readActiveBatchOperatorAccounts(ctx) @@ -1157,27 +1168,30 @@ export class TerminationScenario extends FlowScenario { verifyStep( Actor.Sysio, "capture-withheld-window", - "the first slide enters the announced group and persists an incomplete window", + "the first advance enters announced duty and discards an incomplete candidate", async ctx => { let checkpoint: WithheldScheduleCheckpoint | null = null await pollUntil( - "an incomplete schedule window is persisted after the target is slashed", + "no next window is published after the target is slashed", async () => { const state = await Steps.contracts.sysio.epoch.readEpochState(ctx) const current = state.batch_op_groups[state.current_batch_op_group] ?? [] - const incomplete = state.batch_op_groups.some( - group => group.length !== Constants.OperatorsPerEpoch - ) + assertCompleteSchedule(state.batch_op_groups) + const withheld = state.next_batch_op_groups.length === 0 const slashTarget = ctx.outputs.assert(RecoverySlashAccountKey) if ( !(await operatorIsSlashed(ctx, slashTarget)) || - !incomplete || + !withheld || current.length === 0 ) { return false } + Assert.ok( + !current.includes(slashTarget), + "withheld duty did not enter the announced successor" + ) const [ethereumNextEpoch, solanaNextEpoch] = await Promise.all([ readEthereumNextEpoch(ctx), readSolanaNextEpoch(ctx) @@ -1211,7 +1225,7 @@ export class TerminationScenario extends FlowScenario { ClusterBuildPhase.create( cluster, "HoldAnnouncedDuty", - "The current group remains fixed while the short future window is withheld" + "The serving window remains intact while candidate publication is withheld" ).push( verifyStep( Actor.Sysio, @@ -1234,11 +1248,11 @@ export class TerminationScenario extends FlowScenario { `duty rotated before repair: ${checkpoint.activeGroup.join(",")} -> ${current.join(",")}` ) } - Assert.ok( - state.batch_op_groups.some( - group => group.length !== Constants.OperatorsPerEpoch - ), - "schedule became complete before a replacement was provisioned" + assertCompleteSchedule(state.batch_op_groups) + Assert.equal( + state.next_batch_op_groups.length, + 0, + "a candidate was published before a replacement was provisioned" ) return ( Number(state.current_epoch_index) >= @@ -1269,7 +1283,7 @@ export class TerminationScenario extends FlowScenario { ClusterBuildPhase.create( cluster, "DegradeHeldDuty", - "An announced seat becomes ineligible while the incomplete future window is held" + "An announced seat becomes ineligible while no next window is published" ).push( ClusterBuildStep.create( Actor.Sysio, @@ -1300,9 +1314,7 @@ export class TerminationScenario extends FlowScenario { ) } Assert.ok( - state.batch_op_groups.some( - group => group.length !== Constants.OperatorsPerEpoch - ), + state.next_batch_op_groups.length === 0, "schedule became complete before replacements were provisioned" ) if ( @@ -1368,7 +1380,7 @@ export class TerminationScenario extends FlowScenario { verifyStep( Actor.Sysio, "complete-window-published", - "the held window becomes full and names the next group while current duty is unchanged", + "a complete candidate is published while the serving window is unchanged", async ctx => { const withheld = ctx.outputs.assert(WithheldScheduleCheckpointKey) const replacements = Constants.RecoveryOperatorLabels.map( @@ -1376,12 +1388,13 @@ export class TerminationScenario extends FlowScenario { ) let checkpoint: RepairedScheduleCheckpoint | null = null await pollUntil( - "the replacement fills the held schedule window", + "the replacement enables a complete next-window announcement", async () => { const state = await Steps.contracts.sysio.epoch.readEpochState(ctx) - const groups = state.batch_op_groups - const current = groups[state.current_batch_op_group] ?? [] + const groups = state.next_batch_op_groups + const current = + state.batch_op_groups[state.current_batch_op_group] ?? [] const complete = groups.length === Constants.BatchOperatorGroups && groups.every( @@ -1400,8 +1413,7 @@ export class TerminationScenario extends FlowScenario { "current duty moved before the repaired lookahead was published" ) assertCompleteSchedule(groups) - const nextGroup = - groups[state.current_batch_op_group + 1] ?? current + const nextGroup = groups[1] ?? groups[0] const [ethereumNextEpoch, solanaNextEpoch] = await Promise.all([ readEthereumNextEpoch(ctx), readSolanaNextEpoch(ctx) From 4b760862191d03a44ba90a8851c1353154aa258a Mon Sep 17 00:00:00 2001 From: Huang-Ming Huang Date: Sat, 12 Sep 2026 00:04:35 +0000 Subject: [PATCH 04/16] Fund replacement operators before recovery delivery Change-Id: I2a4165064a36ac08dbb66a2b1b9e0819883f34c8 --- .../src/TerminationScenario.ts | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/packages/flow-batch-operator-termination/src/TerminationScenario.ts b/packages/flow-batch-operator-termination/src/TerminationScenario.ts index 3f95ff28f..df23f51dd 100644 --- a/packages/flow-batch-operator-termination/src/TerminationScenario.ts +++ b/packages/flow-batch-operator-termination/src/TerminationScenario.ts @@ -1351,7 +1351,9 @@ export class TerminationScenario extends FlowScenario { label, type: OperatorType.BATCH, ethereumHdIndex: Constants.RecoveryOperatorEthereumHdIndices[index], - isBootstrapped: true + isBootstrapped: true, + airdropSolanaLamports: + WireOperatorProvisioningTool.DefaultSolanaAirdropLamports })) ) From d63e0e76681b4a684722d46c0055697df52e5392 Mon Sep 17 00:00:00 2001 From: Huang-Ming Huang Date: Sat, 12 Sep 2026 14:45:28 +0000 Subject: [PATCH 05/16] Observe later replacement duties after quorum races Change-Id: I94bcf0f8b1f42c96925f33881467dbe21846d254 --- .../src/TerminationScenario.ts | 47 ++++++++++--------- 1 file changed, 26 insertions(+), 21 deletions(-) diff --git a/packages/flow-batch-operator-termination/src/TerminationScenario.ts b/packages/flow-batch-operator-termination/src/TerminationScenario.ts index df23f51dd..ce9ed65ce 100644 --- a/packages/flow-batch-operator-termination/src/TerminationScenario.ts +++ b/packages/flow-batch-operator-termination/src/TerminationScenario.ts @@ -1499,7 +1499,8 @@ export class TerminationScenario extends FlowScenario { const replacements = Constants.RecoveryOperatorLabels.map( label => ctx.keyStore.assertOperator(label).account ) - const dutyEpochs = new Map() + const dutyEpochs = new Map>() + const delivered = new Set() await pollUntil( "each replacement signs an accepted duty epoch on both outposts", async () => { @@ -1509,36 +1510,40 @@ export class TerminationScenario extends FlowScenario { state.batch_op_groups[state.current_batch_op_group] ?? [] const currentEpoch = Number(state.current_epoch_index) for (const replacement of replacements) { - if ( - current.includes(replacement) && - !dutyEpochs.has(replacement) - ) { - dutyEpochs.set(replacement, currentEpoch) + if (current.includes(replacement) && !delivered.has(replacement)) { + const epochs = dutyEpochs.get(replacement) ?? new Set() + epochs.add(currentEpoch) + dutyEpochs.set(replacement, epochs) } } assertCompleteSchedule(state.batch_op_groups) - if (dutyEpochs.size !== replacements.length) return false - const [ethereumNextEpoch, solanaNextEpoch] = await Promise.all([ readEthereumNextEpoch(ctx), readSolanaNextEpoch(ctx) ]) for (const replacement of replacements) { - const dutyEpoch = dutyEpochs.get(replacement) - Assert.ok(dutyEpoch != null) - if ( - ethereumNextEpoch < dutyEpoch + 1 || - solanaNextEpoch < dutyEpoch + 1 || - !(await replacementDeliveredOnBothOutposts( - ctx, - replacement, - dutyEpoch - )) - ) { - return false + if (delivered.has(replacement)) continue + // A delivery arriving after quorum is a benign no-op. Keep + // later observed duties eligible instead of pinning the first. + for (const dutyEpoch of dutyEpochs.get(replacement) ?? []) { + if ( + ethereumNextEpoch >= dutyEpoch + 1 && + solanaNextEpoch >= dutyEpoch + 1 && + (await replacementDeliveredOnBothOutposts( + ctx, + replacement, + dutyEpoch + )) + ) { + delivered.add(replacement) + log.info( + `${replacement} signed accepted deliveries on both outposts for duty epoch ${dutyEpoch}` + ) + break + } } } - return true + return delivered.size === replacements.length }, Constants.recoveryDeadlineMs(Constants.BatchOperatorGroups + 4), Constants.PollIntervalMs From 8d53c83ab7ea81f94a25eac6c98070422b6e7dc1 Mon Sep 17 00:00:00 2001 From: Huang-Ming Huang Date: Sat, 12 Sep 2026 19:05:33 +0000 Subject: [PATCH 06/16] Require replacement signers in recovery flow quorum Change-Id: I9e54653f069116b4e2c26e46b5bdd0255f279c6d --- jest.config.ts | 3 +- .../jest.config.ts | 22 +++++ .../src/ReplacementQuorum.ts | 20 +++++ .../src/TerminationScenario.ts | 87 ++++++++++++++++++- .../tests/ReplacementQuorum.test.ts | 67 ++++++++++++++ 5 files changed, 197 insertions(+), 2 deletions(-) create mode 100644 packages/flow-batch-operator-termination/jest.config.ts create mode 100644 packages/flow-batch-operator-termination/src/ReplacementQuorum.ts create mode 100644 packages/flow-batch-operator-termination/tests/ReplacementQuorum.test.ts diff --git a/jest.config.ts b/jest.config.ts index ac2aed1d2..ca3aeb889 100644 --- a/jest.config.ts +++ b/jest.config.ts @@ -10,7 +10,7 @@ const config: Config = { // `ClusterConfigProvider.resolve` claims every daemon port (each TCP-probed, // UDP-role ones probed twice) and `findAvailableRange` sweeps a 64-port // window — ~15s per test even with the suite running ALONE. Under the full - // 8-project run that comfortably exceeds a 30s ceiling. + // multi-project run that comfortably exceeds a 30s ceiling. // // An undershot ceiling does NOT fail cleanly here, which is why this is // sized generously rather than trimmed: a test killed mid-`withFileLock` @@ -26,6 +26,7 @@ const config: Config = { "packages/cluster-tool-shared", "packages/cluster-tool", "packages/flow-batch-operator-slashing", + "packages/flow-batch-operator-termination", "packages/debugging-shared", "packages/debugging-server", "packages/debugging-client-shared", diff --git a/packages/flow-batch-operator-termination/jest.config.ts b/packages/flow-batch-operator-termination/jest.config.ts new file mode 100644 index 000000000..aed8d4aff --- /dev/null +++ b/packages/flow-batch-operator-termination/jest.config.ts @@ -0,0 +1,22 @@ +const config = { + displayName: "flow-batch-operator-termination", + testEnvironment: "node", + roots: ["/tests"], + testMatch: ["**/*.test.ts"], + transform: { + "^.+\\.ts$": [ + "ts-jest", + { + tsconfig: "/../../etc/tsconfig/tsconfig.base.jest.json" + } + ] + }, + moduleNameMapper: { + "^(\\.{1,2}/.*)\\.js$": "$1", + "^@wireio/test-flow-batch-operator-termination/(.*)\\.js$": + "/src/$1", + "^@wireio/test-flow-batch-operator-termination/(.*)$": "/src/$1" + } +} + +export default config diff --git a/packages/flow-batch-operator-termination/src/ReplacementQuorum.ts b/packages/flow-batch-operator-termination/src/ReplacementQuorum.ts new file mode 100644 index 000000000..d875dc491 --- /dev/null +++ b/packages/flow-batch-operator-termination/src/ReplacementQuorum.ts @@ -0,0 +1,20 @@ +import Assert from "node:assert" +import { TerminationScenarioConstants as Constants } from "./TerminationScenarioConstants.js" + +/** Select an original peer whose absence makes the replacement necessary for quorum. */ +export function assertReplacementQuorumPeer( + groups: readonly (readonly string[])[], + replacements: readonly string[], + replacement: string +): string { + Assert.equal(replacements.length, Constants.RecoveryOperatorLabels.length) + Assert.equal(new Set(replacements).size, replacements.length) + Assert.ok(replacements.includes(replacement), "unknown replacement") + const group = groups.find(members => members.includes(replacement)) + Assert.ok(group != null, `${replacement} is not seated`) + Assert.equal(group.length, Constants.OperatorsPerEpoch) + Assert.equal(new Set(group).size, group.length) + const peer = group.find(account => !replacements.includes(account)) + Assert.ok(peer != null, "replacement duty has no original peer") + return peer +} diff --git a/packages/flow-batch-operator-termination/src/TerminationScenario.ts b/packages/flow-batch-operator-termination/src/TerminationScenario.ts index ce9ed65ce..9e808efc7 100644 --- a/packages/flow-batch-operator-termination/src/TerminationScenario.ts +++ b/packages/flow-batch-operator-termination/src/TerminationScenario.ts @@ -9,6 +9,8 @@ import { ClusterConfigProvider, EthereumCollateralTool, FlowScenario, + NodeConfig, + NodeRole, OperatorDaemonTool, Report, SolanaCollateralTool, @@ -26,9 +28,11 @@ import { verifyStep, type ClusterBuild, type ClusterBuildContext, - type ClusterBuildOptions + type ClusterBuildOptions, + type StepInput } from "@wireio/cluster-tool" import { TerminationScenarioConstants as Constants } from "./TerminationScenarioConstants.js" +import { assertReplacementQuorumPeer } from "./ReplacementQuorum.js" const log = getLogger(__filename) @@ -362,6 +366,45 @@ async function replacementDeliveredOnBothOutposts( ) } +interface StopReplacementPeerInput extends StepInput { + readonly kind: "TerminationScenario.StopReplacementPeerInput" + readonly label: string +} + +/** Stop one original signer so this replacement is necessary for quorum. */ +async function runStopReplacementPeer( + ctx: ClusterBuildContext, + input: StopReplacementPeerInput, + signal: AbortSignal +): Promise { + signal.throwIfAborted() + const state = await Steps.contracts.sysio.epoch.readEpochState(ctx) + const replacements = Constants.RecoveryOperatorLabels.map( + label => ctx.keyStore.assertOperator(label).account + ) + const replacement = ctx.keyStore.assertOperator(input.label).account + const account = assertReplacementQuorumPeer( + state.batch_op_groups, + replacements, + replacement + ) + const operator = ctx.keyStore.operators.find(entry => entry.account === account) + Assert.ok(operator != null, `${account} has no operator identity`) + const node = NodeConfig.plan(ctx.config).find( + entry => + entry.role === NodeRole.batch_operator && + entry.batchOperatorLabel === operator.label + ) + Assert.ok(node != null, `${account} has no planned batch-operator daemon`) + const daemon = ctx.processManager.get(node.name) + Assert.ok(daemon != null, `${node.name} is not registered`) + // Replacements in the same group select the same peer. stop() is idempotent. + await daemon.stop(signal) + log.info( + `stopped ${account}'s daemon; ${replacement} is required for quorum` + ) +} + /** Read whether an operator is SLASHED in the depot registry. */ async function operatorIsSlashed( ctx: ClusterBuildContext, @@ -1491,6 +1534,48 @@ export class TerminationScenario extends FlowScenario { "ExerciseScheduleReplacement", "Each replacement signs an accepted duty epoch on Ethereum and Solana" ).push( + verifyStep( + Actor.Sysio, + "replacement-groups-ready", + "both replacements are seated in a complete, fully active window", + async ctx => { + const replacements = Constants.RecoveryOperatorLabels.map( + label => ctx.keyStore.assertOperator(label).account + ) + await pollUntil( + "historical vacancies leave the activated window", + async () => { + const state = + await Steps.contracts.sysio.epoch.readEpochState(ctx) + const active = await readActiveBatchOperatorAccounts(ctx) + assertCompleteSchedule(state.batch_op_groups) + const members = state.batch_op_groups.flat() + return ( + members.every(account => active.has(account)) && + replacements.every(account => members.includes(account)) + ) + }, + Constants.recoveryDeadlineMs(Constants.BatchOperatorGroups + 1), + Constants.PollIntervalMs + ) + }, + { + timeoutMs: Constants.recoveryDeadlineMs(Constants.BatchOperatorGroups + 1) + } + ), + // A late third signature is a valid no-op after quorum. Stop one original + // peer per replacement group so each replacement must be admitted. + // The groups may be shared or distinct; every group retains two signers. + ...Constants.RecoveryOperatorLabels.map(label => + ClusterBuildStep.create( + Actor.BatchOperator, + `stop-${label}-peer`, + `stop an original group peer so ${label} is required for quorum`, + {}, + { kind: "TerminationScenario.StopReplacementPeerInput", label }, + runStopReplacementPeer + ) + ), verifyStep( Actor.BatchOperator, "replacement-duty-serves", diff --git a/packages/flow-batch-operator-termination/tests/ReplacementQuorum.test.ts b/packages/flow-batch-operator-termination/tests/ReplacementQuorum.test.ts new file mode 100644 index 000000000..d52a1e124 --- /dev/null +++ b/packages/flow-batch-operator-termination/tests/ReplacementQuorum.test.ts @@ -0,0 +1,67 @@ +import { assertReplacementQuorumPeer } from "@wireio/test-flow-batch-operator-termination/ReplacementQuorum.js" + +const Replacements = ["replacementa", "replacementb"] +const OtherGroup = ["othera", "otherb", "otherc"] + +describe("assertReplacementQuorumPeer", () => { + test("selects the same original peer when both replacements share a group", () => { + const groups = [OtherGroup, ["replacementb", "original", "replacementa"]] + for (const replacement of Replacements) { + expect( + assertReplacementQuorumPeer(groups, Replacements, replacement) + ).toBe("original") + } + }) + + test("selects one original peer from each distinct replacement group", () => { + const groups = [ + ["replacementa", "originala", "originalb"], + ["replacementb", "originalc", "originald"] + ] + expect( + assertReplacementQuorumPeer(groups, Replacements, "replacementa") + ).toBe("originala") + expect( + assertReplacementQuorumPeer(groups, Replacements, "replacementb") + ).toBe("originalc") + }) + + test("refuses a missing replacement instead of stopping an unrelated peer", () => { + expect(() => + assertReplacementQuorumPeer([OtherGroup], Replacements, "replacementa") + ).toThrow("replacementa is not seated") + }) + + test("refuses a target outside the replacement set", () => { + expect(() => + assertReplacementQuorumPeer([OtherGroup], Replacements, "othera") + ).toThrow("unknown replacement") + }) + + test("refuses extra peers that would allow quorum without the replacement", () => { + expect(() => + assertReplacementQuorumPeer( + [["replacementa", "originala", "originalb", "originalc"]], + Replacements, + "replacementa" + ) + ).toThrow() + }) + + test("refuses duplicate seats and duplicate replacement identities", () => { + expect(() => + assertReplacementQuorumPeer( + [[...Replacements, "replacementa"]], + Replacements, + "replacementa" + ) + ).toThrow() + expect(() => + assertReplacementQuorumPeer( + [["replacementa", "originala", "originalb"]], + ["replacementa", "replacementa"], + "replacementa" + ) + ).toThrow() + }) +}) From dac51ad36f062200cc926d1b729b0c16ffe85a6d Mon Sep 17 00:00:00 2001 From: Huang-Ming Huang Date: Tue, 15 Sep 2026 14:33:31 +0000 Subject: [PATCH 07/16] Address PR review feedback Change-Id: Idac130820e1541d788243ab0db4deeb0dcc47029 --- .../src/StandingSpare.ts | 42 +++ .../src/TerminationScenario.ts | 319 +++++++++++++++--- .../src/TerminationScenarioConstants.ts | 15 +- .../tests/StandingSpare.test.ts | 50 +++ 4 files changed, 370 insertions(+), 56 deletions(-) create mode 100644 packages/flow-batch-operator-termination/src/StandingSpare.ts create mode 100644 packages/flow-batch-operator-termination/tests/StandingSpare.test.ts diff --git a/packages/flow-batch-operator-termination/src/StandingSpare.ts b/packages/flow-batch-operator-termination/src/StandingSpare.ts new file mode 100644 index 000000000..8a1cc3589 --- /dev/null +++ b/packages/flow-batch-operator-termination/src/StandingSpare.ts @@ -0,0 +1,42 @@ +import Assert from "node:assert" +import { TerminationScenarioConstants as Constants } from "./TerminationScenarioConstants.js" + +/** Require a full, disjoint nine-seat schedule, regardless of spare count. */ +export function assertCompleteSchedule( + groups: readonly (readonly string[])[] +): void { + Assert.equal(groups.length, Constants.BatchOperatorGroups, "schedule group count changed") + const members = new Set() + for (const group of groups) { + Assert.equal(group.length, Constants.OperatorsPerEpoch, "schedule contains a short group") + for (const member of group) { + Assert.ok(!members.has(member), `${member} is seated in more than one group`) + members.add(member) + } + } + Assert.equal(members.size, Constants.ScheduleSeatCount, "schedule window is not full") +} + +/** Choose an ACTIVE operator outside a complete activated schedule window. */ +export function assertStandingSpare( + activeAccounts: ReadonlySet, + groups: readonly (readonly string[])[], + expectedSpareCount: number +): string { + assertCompleteSchedule(groups) + const seated = groups.flat() + Assert.ok( + seated.every(account => activeAccounts.has(account)), + "activated window contains an inactive member" + ) + const spares = [...activeAccounts] + .filter(account => !seated.includes(account)) + .sort() + Assert.equal( + spares.length, + expectedSpareCount, + "standing-spare pool changed before recovery" + ) + Assert.ok(spares.length > 0, "no standing spare is available") + return spares[0] +} diff --git a/packages/flow-batch-operator-termination/src/TerminationScenario.ts b/packages/flow-batch-operator-termination/src/TerminationScenario.ts index 9e808efc7..29f56aad4 100644 --- a/packages/flow-batch-operator-termination/src/TerminationScenario.ts +++ b/packages/flow-batch-operator-termination/src/TerminationScenario.ts @@ -33,6 +33,7 @@ import { } from "@wireio/cluster-tool" import { TerminationScenarioConstants as Constants } from "./TerminationScenarioConstants.js" import { assertReplacementQuorumPeer } from "./ReplacementQuorum.js" +import { assertCompleteSchedule, assertStandingSpare } from "./StandingSpare.js" const log = getLogger(__filename) @@ -70,6 +71,20 @@ interface RepairedScheduleCheckpoint extends WithheldScheduleCheckpoint { nextGroup: string[] } +/** Complete lookahead published after the doomed operator is terminated. */ +interface AbsorbedRemovalCheckpoint { + epochIndex: number + servingGroup: string[] + nextGroups: string[][] + ethereumNextEpoch: number + solanaNextEpoch: number +} + +const AbsorbedRemovalCheckpointKey = outputKey( + "TerminationScenario.absorbedRemovalCheckpoint", + "complete lookahead and outpost cursors after the termination is absorbed" +) + const WithheldScheduleCheckpointKey = outputKey( "TerminationScenario.withheldScheduleCheckpoint", "depot duty and outpost epoch cursors after the first withheld schedule window" @@ -236,35 +251,6 @@ function sameGroup(left: readonly string[], right: readonly string[]): boolean { ) } -/** Require the full disjoint schedule used by the recovery regression. */ -function assertCompleteSchedule(groups: readonly string[][]): void { - Assert.equal( - groups.length, - Constants.BatchOperatorGroups, - "schedule group count changed" - ) - const members = new Set() - for (const group of groups) { - Assert.equal( - group.length, - Constants.OperatorsPerEpoch, - "schedule contains a short group" - ) - for (const member of group) { - Assert.ok( - !members.has(member), - `${member} is seated in more than one group` - ) - members.add(member) - } - } - Assert.equal( - members.size, - Constants.BatchOperatorCount, - "schedule window is not full" - ) -} - /** Read the Ethereum outpost's next sequential inbound epoch. */ async function readEthereumNextEpoch( ctx: ClusterBuildContext @@ -464,7 +450,7 @@ async function runSlashRecoveryTarget( const activeAccounts = await readActiveBatchOperatorAccounts(ctx) Assert.equal( activeAccounts.size, - Constants.BatchOperatorCount, + Constants.ScheduleSeatCount, "recovery slash did not start at the exact active roster floor" ) Assert.ok( @@ -504,6 +490,61 @@ async function runSlashRecoveryTarget( ctx.outputs.set(RecoverySlashAccountKey, account) } +interface SlashStandingSpareInput extends StepInput { + readonly kind: "TerminationScenario.SlashStandingSpareInput" + readonly ordinal: number +} + +/** Slash one verified standing spare, leaving every activated seat available. */ +async function runSlashStandingSpare( + ctx: ClusterBuildContext, + input: SlashStandingSpareInput, + signal: AbortSignal +): Promise { + signal.throwIfAborted() + let spare: string | null = null + await pollUntil( + `standing spare ${input.ordinal + 1} is outside a full active window`, + async () => { + const state = await Steps.contracts.sysio.epoch.readEpochState(ctx) + const active = await readActiveBatchOperatorAccounts(ctx) + if ( + active.size !== Constants.BatchOperatorCount - input.ordinal || + state.batch_op_groups.length !== Constants.BatchOperatorGroups || + state.batch_op_groups.some( + group => group.length !== Constants.OperatorsPerEpoch + ) + ) { + return false + } + try { + spare = assertStandingSpare( + active, + state.batch_op_groups, + Constants.StandingSpareCount - input.ordinal + ) + return true + } catch { + return false + } + }, + Constants.recoveryDeadlineMs(Constants.BatchOperatorGroups + 2), + Constants.PollIntervalMs + ) + Assert.ok(spare != null, "no safe standing spare was selected") + await Steps.contracts.sysio.opreg.runSlash( + ctx, + { + kind: "OpregContractSteps.SlashInput", + data: { + account: spare, + reason: `WIRE-385 exhaust standing spare ${input.ordinal + 1}` + } + }, + signal + ) +} + /** Remove one eligible member from the group whose duty is currently held. */ async function runSlashHeldGroupMember( ctx: ClusterBuildContext, @@ -555,26 +596,25 @@ async function runSlashHeldGroupMember( ctx.outputs.set(HeldSlashEpochKey, Number(state.current_epoch_index)) } -/** The SOL outpost's on-chain collateral ledger from the `OperatorRegistry` PDA (a read). */ -async function readSolanaCollateralLedger( +/** Read the SOL outpost's zero-copy operator registry and schedule. */ +async function readSolanaOperatorRegistry( ctx: ClusterBuildContext -): Promise { - const operator = ctx.keyStore.assertOperator(Constants.DoomedOperatorLabel) - const program = SolanaCollateralTool.loadOppOutpostProgram( - ctx, - solanaKeypair(operator.solana) - ) +): Promise { + const { accounts, programId } = loadSolanaOppAccounts(ctx) const [registryAddress] = PublicKey.findProgramAddressSync( [Buffer.from(SolanaOutpostBootstrapper.PdaSeed.OperatorRegistry)], - program.programId + programId ) - // Anchor types `Program.account` per-IDL; for a runtime-loaded IDL the - // account clients are reached by name — one assertion to the string-keyed view. - const accounts: Record = program.account - const registryAccount = (await accounts[ + return (await accounts[ Constants.SolanaOperatorRegistryAccountName ].fetch(registryAddress)) as SolanaOperatorRegistryAccount - return registryAccount.collateralByCode ?? [] +} + +/** The SOL outpost's on-chain collateral ledger from the operator registry. */ +async function readSolanaCollateralLedger( + ctx: ClusterBuildContext +): Promise { + return (await readSolanaOperatorRegistry(ctx)).collateralByCode ?? [] } /** @@ -606,10 +646,10 @@ async function readSolanaCollateralLedger( * outpost's escrow ledger returns to 0, and each wallet is credited the * exact bond amount (wei/lamport-exact — any drift means the outpost decoded * a different amount than the depot encoded). - * 9. **Schedule recovery** — remove an announced operator at the exact roster - * floor, prove the depot freezes that duty while both outposts keep accepting - * sequential epochs, remove another member while frozen, add two replacements, - * and prove publication and rotation resume through replacement-backed duty. + * 9. **Schedule recovery** — prove two standing spares absorb the terminated + * operator without a hold, remove those spares and one announced operator + * to reach the freeze, then repair two independent roster losses and prove + * publication and rotation resume across both outposts. */ export class TerminationScenario extends FlowScenario { readonly name = "flow-batch-operator-termination" @@ -1005,7 +1045,160 @@ export class TerminationScenario extends FlowScenario { ) ) - // ── 8. Depot auto-remits the full bond on termination — both outposts ── + // ── 8. Two standing spares absorb the loss without a schedule hold ── + ClusterBuildPhase.create( + cluster, + "AbsorbTerminatedOperator", + "A complete successor reaches both outposts and the next duty rotates" + ).push( + verifyStep( + Actor.Sysio, + "standing-spare-successor-accepted", + "termination leaves eleven ACTIVE operators; both outposts seat a complete live successor", + async ctx => { + const doomed = doomedOperatorAccount(ctx) + let checkpoint: AbsorbedRemovalCheckpoint | null = null + await pollUntil( + "both outposts accept the complete post-termination lookahead", + async () => { + const state = await Steps.contracts.sysio.epoch.readEpochState(ctx) + const groups = state.next_batch_op_groups + const active = await readActiveBatchOperatorAccounts(ctx) + if ( + active.size !== Constants.BatchOperatorCount || + groups.length !== Constants.BatchOperatorGroups || + groups.some( + group => group.length !== Constants.OperatorsPerEpoch + ) + ) { + return false + } + assertCompleteSchedule(groups) + const liveGroups = groups.slice(Constants.SuccessorGroupIndex) + Assert.ok( + liveGroups.flat().every(account => active.has(account)), + "successor contains an inactive active/future seat" + ) + Assert.ok( + !liveGroups.flat().includes(doomed), + "terminated operator remains in an active/future seat" + ) + const oldMembers = new Set(state.batch_op_groups.flat()) + Assert.ok( + liveGroups.flat().some(account => !oldMembers.has(account)), + "no standing operator joined the successor" + ) + + const ethereum = loadEthereumInbound(ctx) + const solana = await readSolanaOperatorRegistry(ctx) + if ( + Number(await ethereum.activeGroupIndex()) !== Constants.SuccessorGroupIndex || + solana.activeGroupIndex !== Constants.SuccessorGroupIndex || + solana.groupCount !== Constants.BatchOperatorGroups + ) { + return false + } + for (let groupIndex = Constants.SuccessorGroupIndex; groupIndex < groups.length; ++groupIndex) { + const solanaGroup = solana.groups[groupIndex] + if (solanaGroup?.memberCount !== Constants.OperatorsPerEpoch) { + return false + } + for (let memberIndex = 0; memberIndex < Constants.OperatorsPerEpoch; ++memberIndex) { + const addresses = replacementAddresses( + ctx, + groups[groupIndex][memberIndex] + ) + if ( + (await ethereum.batchOpGroups(groupIndex, memberIndex)).toLowerCase() !== + addresses.ethereum.toLowerCase() || + !solanaGroup.members[memberIndex]?.equals(addresses.solana) + ) { + return false + } + } + } + const [ethereumNextEpoch, solanaNextEpoch] = await Promise.all([ + readEthereumNextEpoch(ctx), + readSolanaNextEpoch(ctx) + ]) + if ( + ethereumNextEpoch < Number(state.current_epoch_index) + 1 || + solanaNextEpoch < Number(state.current_epoch_index) + 1 + ) { + return false + } + checkpoint = { + epochIndex: Number(state.current_epoch_index), + servingGroup: [ + ...(state.batch_op_groups[state.current_batch_op_group] ?? []) + ], + nextGroups: groups.map(group => [...group]), + ethereumNextEpoch, + solanaNextEpoch + } + return true + }, + Constants.recoveryDeadlineMs(Constants.BatchOperatorGroups + 3), + Constants.PollIntervalMs + ) + Assert.ok(checkpoint != null, "absorbed-removal checkpoint was not captured") + ctx.outputs.set(AbsorbedRemovalCheckpointKey, checkpoint) + }, + { + timeoutMs: Constants.recoveryDeadlineMs(Constants.BatchOperatorGroups + 3) + } + ), + verifyStep( + Actor.Sysio, + "standing-spare-duty-rotates", + "the announced live group takes duty and both outpost cursors keep advancing", + async ctx => { + const checkpoint = ctx.outputs.assert(AbsorbedRemovalCheckpointKey) + const doomed = doomedOperatorAccount(ctx) + let sawActivation = false + await pollUntil( + "the complete successor becomes the activated schedule", + async () => { + const state = await Steps.contracts.sysio.epoch.readEpochState(ctx) + const current = state.batch_op_groups[state.current_batch_op_group] ?? [] + if ( + Number(state.current_epoch_index) <= checkpoint.epochIndex || + state.current_batch_op_group !== Constants.SuccessorGroupIndex || + !state.batch_op_groups.every((group, index) => + sameGroup(group, checkpoint.nextGroups[index] ?? []) + ) || + !sameGroup(current, checkpoint.nextGroups[Constants.SuccessorGroupIndex]) + ) { + if (!sawActivation) return false + } else { + Assert.ok( + !sameGroup(current, checkpoint.servingGroup), + "serving duty did not rotate to the announced successor" + ) + Assert.ok(!current.includes(doomed), "terminated operator returned to duty") + sawActivation = true + } + const [ethereumNextEpoch, solanaNextEpoch] = await Promise.all([ + readEthereumNextEpoch(ctx), + readSolanaNextEpoch(ctx) + ]) + return ( + sawActivation && + ethereumNextEpoch > checkpoint.ethereumNextEpoch && + solanaNextEpoch > checkpoint.solanaNextEpoch + ) + }, + Constants.recoveryDeadlineMs(Constants.BatchOperatorGroups + 3), + Constants.PollIntervalMs + ) + }, + { + timeoutMs: Constants.recoveryDeadlineMs(Constants.BatchOperatorGroups + 3) + } + ) + ) + + // ── 9. Depot auto-remits the full bond on termination — both outposts ── ClusterBuildPhase.create( cluster, "RemitBonds", @@ -1135,7 +1328,29 @@ export class TerminationScenario extends FlowScenario { ) ) - // ── 9. Reach the exact roster floor with a fully active group on duty ── + // ── 10. Exhaust both standing spares before requesting a freeze ── + ClusterBuildPhase.create( + cluster, + "ExhaustStandingSpares", + "Each standing spare is slashed in its own reported contract step" + ).push( + ...Array.from({ length: Constants.StandingSpareCount }, (_, ordinal) => + ClusterBuildStep.create( + Actor.Sysio, + `slash-standing-spare-${ordinal + 1}`, + "slash one ACTIVE operator outside the complete activated window", + { + timeoutMs: Constants.recoveryDeadlineMs( + Constants.BatchOperatorGroups + 2 + ) + }, + { kind: "TerminationScenario.SlashStandingSpareInput", ordinal }, + runSlashStandingSpare + ) + ) + ) + + // ── 11. Reach the exact nine-seat roster floor ── ClusterBuildPhase.create( cluster, "PrepareScheduleRecovery", @@ -1171,7 +1386,7 @@ export class TerminationScenario extends FlowScenario { assertCompleteSchedule(groups) const current = groups[state.current_batch_op_group] ?? [] const activeAccounts = await readActiveBatchOperatorAccounts(ctx) - if (activeAccounts.size !== Constants.BatchOperatorCount) { + if (activeAccounts.size !== Constants.ScheduleSeatCount) { return false } const readerAccount = ctx.keyStore.assertOperator( @@ -1194,7 +1409,7 @@ export class TerminationScenario extends FlowScenario { ) ) - // ── 10. Remove one current member at the exact floor → withhold ── + // ── 12. Remove one current member at the exact floor → withhold ── ClusterBuildPhase.create( cluster, "StarveScheduleWindow", diff --git a/packages/flow-batch-operator-termination/src/TerminationScenarioConstants.ts b/packages/flow-batch-operator-termination/src/TerminationScenarioConstants.ts index 5daf807b7..459a40146 100644 --- a/packages/flow-batch-operator-termination/src/TerminationScenarioConstants.ts +++ b/packages/flow-batch-operator-termination/src/TerminationScenarioConstants.ts @@ -30,15 +30,22 @@ export namespace TerminationScenarioConstants { /** Epoch duration (s) — the bare-cluster working baseline (`sysio.epoch::setconfig` floor is 60). */ export const EpochDurationSec = 60 /** - * Bootstrapped batch operators stood up by the harness. 9 → 3 odd-sized - * groups of 3; with the doomed operator never delivering, the remaining 8 - * still cover consensus majority on every group. + * Bootstrapped batch operators stood up by the harness. Eleven operators + * cover nine schedule seats plus two standing spares. The separately + * provisioned doomed operator never delivers, but its two group peers can + * still reach consensus majority. */ - export const BatchOperatorCount = 9 + export const BatchOperatorCount = 11 /** Number of disjoint groups retained in the rolling schedule window. */ export const BatchOperatorGroups = 3 /** Operators per group; three keeps majority consensus unambiguous. */ export const OperatorsPerEpoch = 3 + /** Number of seats in one full schedule window. */ + export const ScheduleSeatCount = BatchOperatorGroups * OperatorsPerEpoch + /** Spares remaining after the doomed operator terminates. */ + export const StandingSpareCount = BatchOperatorCount - ScheduleSeatCount + /** Published lookahead serves group one after a three-group window activates. */ + export const SuccessorGroupIndex = 1 /** * Override for `terminate_max_consecutive_misses` so `termcheck` fires inside * the flow's budget: 2 consecutive missed scheduled epochs flip TERMINATED. diff --git a/packages/flow-batch-operator-termination/tests/StandingSpare.test.ts b/packages/flow-batch-operator-termination/tests/StandingSpare.test.ts new file mode 100644 index 000000000..7a0add073 --- /dev/null +++ b/packages/flow-batch-operator-termination/tests/StandingSpare.test.ts @@ -0,0 +1,50 @@ +import { + assertCompleteSchedule, + assertStandingSpare +} from "@wireio/test-flow-batch-operator-termination/StandingSpare.js" + +const Groups = [ + ["a", "b", "c"], + ["d", "e", "f"], + ["g", "h", "i"] +] + +describe("assertStandingSpare", () => { + test("selects one of two active operators outside the nine-seat window", () => { + expect(assertStandingSpare(new Set([...Groups.flat(), "k", "j"]), Groups, 2)).toBe("j") + }) + + test("rejects an incomplete or inactive schedule instead of slashing a seat", () => { + expect(() => + assertStandingSpare(new Set([...Groups.flat(), "j"]), Groups, 2) + ).toThrow("standing-spare pool changed") + expect(() => + assertStandingSpare(new Set([...Groups.flat().slice(1), "j", "k"]), Groups, 2) + ).toThrow("inactive member") + }) + + test("rejects duplicate seats and an empty spare pool", () => { + expect(() => + assertStandingSpare(new Set([...Groups.flat(), "j", "k"]), + [Groups[0], Groups[1], ["g", "h", "h"]], 2) + ).toThrow() + expect(() => + assertStandingSpare(new Set(Groups.flat()), Groups, 0) + ).toThrow("no standing spare") + }) +}) + +describe("assertCompleteSchedule", () => { + test("accepts nine disjoint seats with two additional standing spares", () => { + expect(() => assertCompleteSchedule(Groups)).not.toThrow() + }) + + test("rejects a short group or duplicate seat", () => { + expect(() => assertCompleteSchedule([Groups[0], Groups[1], ["g", "h"]])).toThrow( + "short group" + ) + expect(() => assertCompleteSchedule([Groups[0], Groups[1], ["g", "h", "h"]])).toThrow( + "more than one group" + ) + }) +}) From 1f53681b2b21c394deb1836bfd2fe2731f4cae7b Mon Sep 17 00:00:00 2001 From: Huang-Ming Huang Date: Tue, 15 Sep 2026 16:23:42 +0000 Subject: [PATCH 08/16] Classify revoked operator rejections in flow heartbeat Change-Id: I79aa3c8250781321c1fdfe2164d0ae3f65d5f9a5 --- scripts/flow-heartbeat-monitor.mjs | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/scripts/flow-heartbeat-monitor.mjs b/scripts/flow-heartbeat-monitor.mjs index 695acbc5d..2c682fb74 100755 --- a/scripts/flow-heartbeat-monitor.mjs +++ b/scripts/flow-heartbeat-monitor.mjs @@ -174,14 +174,16 @@ const EchoWrapperExcludePattern = `${TrxEchoExcludePattern}|signaled NACK|bad pa * learns via the NEXT envelope's OPERATORS attestation (1–2 epochs), and until * that dispatches the underwriter plugin's commit retries bounce off the * outpost's status gate: SOL opp-outpost `0x1795` (OperatorNotActive), ETH - * `OPP_NotActiveOperator`. Forensically verified (2026-07-04, + * `OPP_NotActiveOperator`. Anvil prints the latter as its custom-error selector + * `0xabc01454`; a deliberately revoked batch member gets the same rejection. + * Forensically verified (2026-07-04, * flow-swap-from-wire): the epoch-2 envelope carried ACTIVE and dispatched 24s * after the first bounce — the ~5s retry loop heals on the next attempt, so * bailing on growth here kills a healthy flow. The liveness probes (epoch * advance, opp delta, per-direction growth) remain the bail gates. */ const RegistrySyncLagPattern = - "custom program error: 0x1795|OPP_NotActiveOperator" + "custom program error: 0x1795|OPP_NotActiveOperator|execution reverted: custom error 0xabc01454" /** * Expected NEGATIVE-TEST reverts — a flow deliberately submits an invalid action From e678da4095b3adac79a7df2e1b9e180cfcd5b326 Mon Sep 17 00:00:00 2001 From: Huang-Ming Huang Date: Mon, 21 Sep 2026 17:00:05 +0000 Subject: [PATCH 09/16] Fix clean-install schedule recovery build Change-Id: I5a3363163ac54c8f9d820cce6e0e96242a50bddd --- .../src/TerminationScenario.ts | 56 ++++++++++++++----- 1 file changed, 41 insertions(+), 15 deletions(-) diff --git a/packages/flow-batch-operator-termination/src/TerminationScenario.ts b/packages/flow-batch-operator-termination/src/TerminationScenario.ts index 3162490cf..e3901887c 100644 --- a/packages/flow-batch-operator-termination/src/TerminationScenario.ts +++ b/packages/flow-batch-operator-termination/src/TerminationScenario.ts @@ -81,6 +81,32 @@ interface AbsorbedRemovalCheckpoint { solanaNextEpoch: number } +/** Epoch state returned by the companion WIRE-385 SYSIO schema. */ +interface ScheduleRecoveryEpochState + extends SysioContracts.SysioEpochEpochStateType { + next_batch_op_groups: string[][] +} + +/** + * Read the epoch state through the WIRE-385 schema boundary. + * + * A clean Tools checkout may still resolve the last published SDK while the + * companion Libraries PR is pending. Keep that temporary type lag isolated + * here, and verify the deployed contract response before the flow uses it. + */ +async function readScheduleRecoveryEpochState( + ctx: ClusterBuildContext +): Promise { + const state = await Steps.contracts.sysio.epoch.readEpochState(ctx) + const nextGroups = (state as Partial | undefined) + ?.next_batch_op_groups + Assert.ok( + Array.isArray(nextGroups), + "epoch state does not expose next_batch_op_groups; deploy the WIRE-385 SYSIO schema before running this flow" + ) + return state as ScheduleRecoveryEpochState +} + const AbsorbedRemovalCheckpointKey = outputKey( "TerminationScenario.absorbedRemovalCheckpoint", "complete lookahead and outpost cursors after the termination is absorbed" @@ -365,7 +391,7 @@ async function runStopReplacementPeer( signal: AbortSignal ): Promise { signal.throwIfAborted() - const state = await Steps.contracts.sysio.epoch.readEpochState(ctx) + const state = await readScheduleRecoveryEpochState(ctx) const replacements = Constants.RecoveryOperatorLabels.map( label => ctx.keyStore.assertOperator(label).account ) @@ -444,7 +470,7 @@ async function runSlashRecoveryTarget( signal: AbortSignal ): Promise { signal.throwIfAborted() - const before = await Steps.contracts.sysio.epoch.readEpochState(ctx) + const before = await readScheduleRecoveryEpochState(ctx) assertCompleteSchedule(before.batch_op_groups) assertCompleteSchedule(before.next_batch_op_groups) const current = before.batch_op_groups[before.current_batch_op_group] ?? [] @@ -477,7 +503,7 @@ async function runSlashRecoveryTarget( }, signal ) - const after = await Steps.contracts.sysio.epoch.readEpochState(ctx) + const after = await readScheduleRecoveryEpochState(ctx) const afterCurrent = after.batch_op_groups[after.current_batch_op_group] ?? [] Assert.equal( Number(after.current_epoch_index), @@ -507,7 +533,7 @@ async function runSlashStandingSpare( await pollUntil( `standing spare ${input.ordinal + 1} is outside a full active window`, async () => { - const state = await Steps.contracts.sysio.epoch.readEpochState(ctx) + const state = await readScheduleRecoveryEpochState(ctx) const active = await readActiveBatchOperatorAccounts(ctx) if ( active.size !== Constants.BatchOperatorCount - input.ordinal || @@ -581,7 +607,7 @@ async function runSlashHeldGroupMember( ) }) Assert.ok(candidate != null, "held group has no eligible slash candidate") - const state = await Steps.contracts.sysio.epoch.readEpochState(ctx) + const state = await readScheduleRecoveryEpochState(ctx) await Steps.contracts.sysio.opreg.runSlash( ctx, { @@ -1062,7 +1088,7 @@ export class TerminationScenario extends FlowScenario { await pollUntil( "both outposts accept the complete post-termination lookahead", async () => { - const state = await Steps.contracts.sysio.epoch.readEpochState(ctx) + const state = await readScheduleRecoveryEpochState(ctx) const groups = state.next_batch_op_groups const active = await readActiveBatchOperatorAccounts(ctx) if ( @@ -1160,7 +1186,7 @@ export class TerminationScenario extends FlowScenario { await pollUntil( "the complete successor becomes the activated schedule", async () => { - const state = await Steps.contracts.sysio.epoch.readEpochState(ctx) + const state = await readScheduleRecoveryEpochState(ctx) const current = state.batch_op_groups[state.current_batch_op_group] ?? [] if ( Number(state.current_epoch_index) <= checkpoint.epochIndex || @@ -1366,7 +1392,7 @@ export class TerminationScenario extends FlowScenario { "a complete three-by-three window has a fully active current group", async () => { const state = - await Steps.contracts.sysio.epoch.readEpochState(ctx) + await readScheduleRecoveryEpochState(ctx) const groups = state.batch_op_groups const complete = groups.length === Constants.BatchOperatorGroups && @@ -1434,7 +1460,7 @@ export class TerminationScenario extends FlowScenario { "no next window is published after the target is slashed", async () => { const state = - await Steps.contracts.sysio.epoch.readEpochState(ctx) + await readScheduleRecoveryEpochState(ctx) const current = state.batch_op_groups[state.current_batch_op_group] ?? [] assertCompleteSchedule(state.batch_op_groups) @@ -1496,7 +1522,7 @@ export class TerminationScenario extends FlowScenario { "both outposts advance twice while the depot duty remains held", async () => { const state = - await Steps.contracts.sysio.epoch.readEpochState(ctx) + await readScheduleRecoveryEpochState(ctx) const current = state.batch_op_groups[state.current_batch_op_group] ?? [] if ( @@ -1564,7 +1590,7 @@ export class TerminationScenario extends FlowScenario { "held duty survives a member becoming ineligible", async () => { const state = - await Steps.contracts.sysio.epoch.readEpochState(ctx) + await readScheduleRecoveryEpochState(ctx) const current = state.batch_op_groups[state.current_batch_op_group] ?? [] if (!sameGroup(current, withheld.activeGroup)) { @@ -1652,7 +1678,7 @@ export class TerminationScenario extends FlowScenario { "the replacement enables a complete next-window announcement", async () => { const state = - await Steps.contracts.sysio.epoch.readEpochState(ctx) + await readScheduleRecoveryEpochState(ctx) const groups = state.next_batch_op_groups const current = state.batch_op_groups[state.current_batch_op_group] ?? [] @@ -1721,7 +1747,7 @@ export class TerminationScenario extends FlowScenario { "the repaired next group serves an epoch accepted by both outposts", async () => { const state = - await Steps.contracts.sysio.epoch.readEpochState(ctx) + await readScheduleRecoveryEpochState(ctx) const current = state.batch_op_groups[state.current_batch_op_group] ?? [] const [ethereumNextEpoch, solanaNextEpoch] = await Promise.all([ @@ -1762,7 +1788,7 @@ export class TerminationScenario extends FlowScenario { "historical vacancies leave the activated window", async () => { const state = - await Steps.contracts.sysio.epoch.readEpochState(ctx) + await readScheduleRecoveryEpochState(ctx) const active = await readActiveBatchOperatorAccounts(ctx) assertCompleteSchedule(state.batch_op_groups) const members = state.batch_op_groups.flat() @@ -1806,7 +1832,7 @@ export class TerminationScenario extends FlowScenario { "each replacement signs an accepted duty epoch on both outposts", async () => { const state = - await Steps.contracts.sysio.epoch.readEpochState(ctx) + await readScheduleRecoveryEpochState(ctx) const current = state.batch_op_groups[state.current_batch_op_group] ?? [] const currentEpoch = Number(state.current_epoch_index) From e741d8c1e0bf8afd74a19e31b2d5063d323d3aef Mon Sep 17 00:00:00 2001 From: Huang-Ming Huang Date: Mon, 21 Sep 2026 17:46:07 +0000 Subject: [PATCH 10/16] Seed Ethereum bootstrap from depot schedule Change-Id: Ifa65e64377a92a2a51ea84eae70b3fc2c05fba18 --- .../src/orchestration/ClusterBuildDefaults.ts | 167 ++++++++++-------- .../ethereum/EthereumOutpostBootstrapper.ts | 26 ++- .../ethereum/EthereumOutpostSteps.ts | 53 +++++- ...ClusterBuildDefaultsEpochBootstrap.test.ts | 26 +-- .../EthereumOutpostBootstrapper.test.ts | 36 +++- .../ethereum/EthereumOutpostSteps.test.ts | 47 +++++ 6 files changed, 268 insertions(+), 87 deletions(-) diff --git a/packages/cluster-tool/src/orchestration/ClusterBuildDefaults.ts b/packages/cluster-tool/src/orchestration/ClusterBuildDefaults.ts index bab806704..72b24a83e 100644 --- a/packages/cluster-tool/src/orchestration/ClusterBuildDefaults.ts +++ b/packages/cluster-tool/src/orchestration/ClusterBuildDefaults.ts @@ -796,8 +796,13 @@ export namespace ClusterBuildDefaults { ) ) - // ── outpost deploys (own the run anvil + validator) — OR, in external mode, - // verify the already-running remote outpost endpoints instead ── + // ── outpost process bring-up / external materialization ── + // Local Ethereum contract deployment is intentionally deferred until after + // operator provisioning + schbatchgps. Its initializer must receive the + // depot's actual randomized-account schedule, or every legitimate epoch-1 + // signer is rejected as inactive. + // + // External mode instead verifies the already-running endpoints here. if (isExternalOutpost) { ClusterBuildPhase.create( prerequisites, @@ -829,25 +834,13 @@ export namespace ClusterBuildDefaults { ClusterBuildPhase.create( prerequisites, "EthereumOutpost", - "Deploy the Ethereum outpost" + "Start the Ethereum outpost process" ).push( Steps.processes.anvil.planStart( Actor.EthereumOutpost, "start-anvil", "start the run-time anvil (instamine)", {} - ), - Steps.ethereumOutpost.planDeploy( - Actor.EthereumOutpost, - "deploy-ethereum", - "deploy + seed the Ethereum outpost", - { timeoutMs: 900_000 } - ), - Steps.processes.anvil.planEnableIntervalMining( - Actor.EthereumOutpost, - "enable-interval-mining", - "switch anvil to interval mining", - {} ) ) ClusterBuildPhase.create( @@ -870,9 +863,87 @@ export namespace ClusterBuildDefaults { ) } - // ── registry + optional mock reserves + underwriter config ── + // ═══ Cluster Operator Bootstrap — operators, schedule, outpost, nodes, epoch ═══ + const postContractDeployment = ClusterBuildPhaseGroup.create( + cluster, + "Cluster Operator Bootstrap", + "Provision operators, seed the outposts, start operator nodes, and bootstrap the first epoch" + ) + + // Bootstrapped batch operators + underwriters via the ONE mechanism. Fee-payer + // funding only — deposit flows provision their own non-bootstrapped ops with + // collateral funding on top. + const isSSM = config.signatureProvider.type === SignatureProviderType.SSM + WireOperatorProvisioningTool.planOperatorAccountProvisioning( + postContractDeployment, + "Create batchops & uws", + "Provision the bootstrapped batch operators + underwriters", + {}, + [ + ...batchOperators.map((label, index) => ({ + label, + type: OperatorType.BATCH, + ethereumHdIndex: index + 1, + isBootstrapped: true, + // Fee-payer funding for the daemon's per-epoch deliveries on BOTH + // chains. ETH is SSM-only: under KEY the EM keys come off the anvil + // mnemonic and are prefunded, under SSM they come off a generated + // mnemonic anvil never funded. See BatchOperatorEthereumFundingWei. + airdropSolanaLamports: BatchOperatorAirdropLamports, + ...(isSSM ? { fundEthereumWei: BatchOperatorEthereumFundingWei } : {}) + })), + ...underwriters.map((label, index) => ({ + label, + type: OperatorType.UNDERWRITER, + ethereumHdIndex: config.batchOperatorCount + index + 1, + isBootstrapped: false, + ...(isSSM ? { fundEthereumWei: BatchOperatorEthereumFundingWei } : {}) + })) + ] + ) + + // Materialize the one schedule both local outposts authorize for epoch 1. + // The generated WIRE account names only exist after provisioning above. ClusterBuildPhase.create( - prerequisites, + postContractDeployment, + "InitialBatchOperatorSchedule", + "Build the initial batch-operator schedule" + ).push( + Steps.contracts.sysio.epoch.planSchbatchgps( + Actor.Sysio, + "schedule-batch-groups", + "build the initial batch-operator schedule", + {} + ) + ) + + if (!isExternalOutpost) { + ClusterBuildPhase.create( + postContractDeployment, + "DeployEthereumOutpost", + "Deploy the Ethereum outpost with the depot's initial schedule" + ).push( + Steps.ethereumOutpost.planDeploy( + Actor.EthereumOutpost, + "deploy-ethereum", + "deploy + seed the Ethereum outpost with the initial operator schedule", + { timeoutMs: 900_000 } + ), + Steps.processes.anvil.planEnableIntervalMining( + Actor.EthereumOutpost, + "enable-interval-mining", + "switch anvil to interval mining", + {} + ) + ) + } + + // Registry token rows consume the local deployment artifacts, so this + // follows the schedule-seeded Ethereum deploy. It remains before the first + // epoch and before daemon artifact publication in both local and external + // modes. + ClusterBuildPhase.create( + postContractDeployment, "Registry", "Seed chains + tokens" ).push( @@ -890,14 +961,14 @@ export namespace ClusterBuildDefaults { // 0→1) could never seed them. if (config.enableMockReserves) { Steps.registry.planMockReserves( - prerequisites, + postContractDeployment, "MockReserves", "Seed the 8 mock (chain, token) PRIMARY reserves", {} ) } ClusterBuildPhase.create( - prerequisites, + postContractDeployment, "UnderwriterConfig", "Configure sysio.uwrit" ).push( @@ -917,7 +988,7 @@ export namespace ClusterBuildDefaults { ) ) ClusterBuildPhase.create( - prerequisites, + postContractDeployment, "ReserveConfig", "Configure sysio.reserv fee routing" ).push( @@ -930,16 +1001,11 @@ export namespace ClusterBuildDefaults { ) ) - // ═══ Cluster Post Contract Deployment — batch/uw operators, nodes, first epoch ═══ - const postContractDeployment = ClusterBuildPhaseGroup.create( - cluster, - "Cluster Post Contract Deployment", - "Provision batch operators + underwriters, start operator nodes, bootstrap the first epoch" - ) - // The operator daemons' shared prerequisites: the in-process OPP debugging // sink (external_debugging_plugin posts every envelope there) + the deploy // artifacts (ETH ABIs with addresses, SOL program id + IDL) their args reference. + // In local mode this must follow DeployEthereumOutpost so the ABI artifacts + // carry the addresses from the schedule-seeded deployment. ClusterBuildPhase.create( postContractDeployment, "OperatorDaemonPrerequisites", @@ -975,38 +1041,6 @@ export namespace ClusterBuildDefaults { ) ) - // Bootstrapped batch operators + underwriters via the ONE mechanism. Fee-payer - // funding only — deposit flows provision their own non-bootstrapped ops with - // collateral funding on top. - const isSSM = config.signatureProvider.type === SignatureProviderType.SSM - WireOperatorProvisioningTool.planOperatorAccountProvisioning( - postContractDeployment, - "Create batchops & uws", - "Provision the bootstrapped batch operators + underwriters", - {}, - [ - ...batchOperators.map((label, index) => ({ - label, - type: OperatorType.BATCH, - ethereumHdIndex: index + 1, - isBootstrapped: true, - // Fee-payer funding for the daemon's per-epoch deliveries on BOTH - // chains. ETH is SSM-only: under KEY the EM keys come off the anvil - // mnemonic and are prefunded, under SSM they come off a generated - // mnemonic anvil never funded. See BatchOperatorEthereumFundingWei. - airdropSolanaLamports: BatchOperatorAirdropLamports, - ...(isSSM ? { fundEthereumWei: BatchOperatorEthereumFundingWei } : {}) - })), - ...underwriters.map((label, index) => ({ - label, - type: OperatorType.UNDERWRITER, - ethereumHdIndex: config.batchOperatorCount + index + 1, - isBootstrapped: false, - ...(isSSM ? { fundEthereumWei: BatchOperatorEthereumFundingWei } : {}) - })) - ] - ) - // SSM mode: publish the just-provisioned operator keys BEFORE the operator // daemons start — their wire/ethereum/solana `--signature-provider ...SSM:` // specs fetch the private keys from SSM at nodeop startup. @@ -1057,20 +1091,13 @@ export namespace ClusterBuildDefaults { // scenarios that need a real restart. // ── first epoch ── - // Step ORDER is load-bearing: the roster seed reads the schedule - // `schbatchgps` just materialized, and must land before `msgch::bootstrap` - // delivers the first envelope. + // Step ORDER is load-bearing: both local outposts were seeded from the + // schedule materialized above, and Solana's transient roster seed must land + // before `msgch::bootstrap` delivers the first envelope. const epochBootstrap = ClusterBuildPhase.create( postContractDeployment, "EpochBootstrap", - "Schedule groups + bootstrap epoch 0 → 1" - ).push( - Steps.contracts.sysio.epoch.planSchbatchgps( - Actor.Sysio, - "schedule-batch-groups", - "build the initial batch-operator schedule", - {} - ) + "Bootstrap epoch 0 → 1" ) // SOL-376: seed the LOCAL Solana outpost's operator registry with the // depot's epoch-1 batch-operator group (just materialized by schbatchgps) diff --git a/packages/cluster-tool/src/orchestration/ethereum/EthereumOutpostBootstrapper.ts b/packages/cluster-tool/src/orchestration/ethereum/EthereumOutpostBootstrapper.ts index d5ec638b9..9cddb88f6 100644 --- a/packages/cluster-tool/src/orchestration/ethereum/EthereumOutpostBootstrapper.ts +++ b/packages/cluster-tool/src/orchestration/ethereum/EthereumOutpostBootstrapper.ts @@ -44,6 +44,12 @@ export interface EthereumOutpostBootstrapperOptions { * sharing `/.local/deployments/` wiped each other mid-run). */ deploymentsPath: string + /** Depot schedule groups mapped to their Ethereum signing addresses. */ + initialOperatorGroups: string[][] + /** Active group cursor in the depot schedule at deployment time. */ + initialActiveGroupIndex: number + /** Depot epoch duration, used by the outpost's initial roster window. */ + epochDurationSec: number /** * Number of deterministic accounts to generate — MUST match the run anvil's * `--accounts` (default: {@link AnvilProcess.AccountCount}) so every generated @@ -88,6 +94,21 @@ export class EthereumOutpostBootstrapper { options.deploymentsPath, "EthereumOutpostBootstrapper: deploymentsPath is required" ) + Assert.ok( + options.initialOperatorGroups?.length > 0 && + options.initialOperatorGroups.every(group => group.length > 0), + "EthereumOutpostBootstrapper: initialOperatorGroups must contain non-empty groups" + ) + Assert.ok( + Number.isInteger(options.initialActiveGroupIndex) && + options.initialActiveGroupIndex >= 0 && + options.initialActiveGroupIndex < options.initialOperatorGroups.length, + "EthereumOutpostBootstrapper: initialActiveGroupIndex is out of range" + ) + Assert.ok( + Number.isInteger(options.epochDurationSec) && options.epochDurationSec > 0, + "EthereumOutpostBootstrapper: epochDurationSec must be positive" + ) this.config = defaults( { ...options }, EthereumOutpostBootstrapper.createDefaultOptions() @@ -179,7 +200,10 @@ export class EthereumOutpostBootstrapper { key: deployerPrivateKey, addressFile: Path.join(localDir, "outpost-addrs.json"), gasLimitFile: Path.join(localDir, "outpost-gas-limits.json"), - useMockAggregator: true + useMockAggregator: true, + initialOperatorGroups: this.config.initialOperatorGroups, + initialActiveGroupIndex: this.config.initialActiveGroupIndex, + epochDurationSec: this.config.epochDurationSec } Fs.writeFileSync( Path.join(localDir, "liqeth.json"), diff --git a/packages/cluster-tool/src/orchestration/ethereum/EthereumOutpostSteps.ts b/packages/cluster-tool/src/orchestration/ethereum/EthereumOutpostSteps.ts index b78191355..22728f372 100644 --- a/packages/cluster-tool/src/orchestration/ethereum/EthereumOutpostSteps.ts +++ b/packages/cluster-tool/src/orchestration/ethereum/EthereumOutpostSteps.ts @@ -1,4 +1,6 @@ +import Assert from "node:assert" import Path from "node:path" +import { OperatorType } from "@wireio/opp-typescript-models" import { Report } from "../../report/Report.js" import { toDialAddress, toURL } from "../../utils/netUtils.js" import { ClusterBuildContext } from "../ClusterBuildContext.js" @@ -8,6 +10,8 @@ import { } from "../ClusterBuildStep.js" import { EthereumOutpostBootstrapper } from "./EthereumOutpostBootstrapper.js" import { ClusterConfigProvider } from "../../config/ClusterConfigProvider.js" +import { OperatorAccount } from "../outputs/OperatorAccount.js" +import { EpochContractSteps } from "../steps/contracts/sysio/EpochContractSteps.js" /** Steps that deploy + seed the Ethereum (anvil) outpost. */ export namespace EthereumOutpostSteps { @@ -46,6 +50,15 @@ export namespace EthereumOutpostSteps { signal: AbortSignal ): Promise { signal.throwIfAborted() + const epochState = await EpochContractSteps.readEpochState(ctx) + Assert.ok( + epochState?.batch_op_groups?.length > 0, + "runDeploy: initial batch-operator schedule is empty" + ) + const initialOperatorGroups = resolveInitialOperatorGroups( + ctx.keyStore.operatorsByType(OperatorType.BATCH), + epochState.batch_op_groups + ) // Same derivation as AnvilProcess.rpcUrl — the run anvil was bound to this // exact port by Steps.processes.anvil.start, so they cannot diverge. await new EthereumOutpostBootstrapper({ @@ -55,7 +68,45 @@ export namespace EthereumOutpostSteps { ctx.config.bind.anvil.port, toDialAddress(ctx.config.bind.anvil.address) ), - deploymentsPath: ClusterConfigProvider.ethereumDeploymentsPath(ctx.config) + deploymentsPath: ClusterConfigProvider.ethereumDeploymentsPath(ctx.config), + initialOperatorGroups, + initialActiveGroupIndex: epochState.current_batch_op_group, + epochDurationSec: ctx.config.epochDurationSec }).bootstrap() } + + /** + * Map the depot's materialized schedule to the exact Ethereum keys its + * operator daemons use. Account names are generated during provisioning, so + * this mapping must be resolved from the live key store after + * `schbatchgps`; deriving it from labels or HD indexes can authorize the + * wrong first-epoch signers. + */ + export function resolveInitialOperatorGroups( + batchOperators: OperatorAccount[], + scheduleGroups: string[][] + ): string[][] { + Assert.ok(scheduleGroups.length > 0, "resolveInitialOperatorGroups: schedule is empty") + const operatorByAccount = new Map( + batchOperators.map(operator => [operator.account, operator]) + ) + return scheduleGroups.map((group, groupIndex) => { + Assert.ok( + group.length > 0, + `resolveInitialOperatorGroups: schedule group ${groupIndex} is empty` + ) + return group.map(accountName => { + const operator = operatorByAccount.get(accountName) + Assert.ok( + operator, + `resolveInitialOperatorGroups: schedule member ${accountName} not found among provisioned batch operators` + ) + Assert.ok( + operator.ethereum?.address, + `resolveInitialOperatorGroups: schedule member ${accountName} has no Ethereum address` + ) + return operator.ethereum.address + }) + }) + } } diff --git a/packages/cluster-tool/tests/orchestration/ClusterBuildDefaultsEpochBootstrap.test.ts b/packages/cluster-tool/tests/orchestration/ClusterBuildDefaultsEpochBootstrap.test.ts index 089587949..558df5735 100644 --- a/packages/cluster-tool/tests/orchestration/ClusterBuildDefaultsEpochBootstrap.test.ts +++ b/packages/cluster-tool/tests/orchestration/ClusterBuildDefaultsEpochBootstrap.test.ts @@ -8,8 +8,9 @@ import { import { collectStepNames } from "./clusterBuildFixture.js" -/** The three EpochBootstrap steps, in the order the depot requires them. */ +/** Bootstrap steps whose global order is load-bearing. */ const ScheduleBatchGroupsStep = "schedule-batch-groups" +const DeployEthereumStep = "deploy-ethereum" const SeedSolanaRosterStep = "seed-solana-roster" const BootstrapEpochStep = "bootstrap-epoch" @@ -45,17 +46,19 @@ describe("ClusterBuildDefaults — EpochBootstrap step order", () => { } } - it("seeds the Solana roster BETWEEN schbatchgps and msgch::bootstrap", async () => { - // Load-bearing order: `opp_bootstrap` reads the schedule `schbatchgps` just - // materialized, and the SOL outpost's `epoch_in` refuses to finalize the - // first envelope `msgch::bootstrap` delivers until the roster is seeded. + it("seeds both local outposts from schbatchgps before msgch::bootstrap", async () => { + // Ethereum's initializer and Solana's opp_bootstrap both read the schedule + // schbatchgps materialized. Neither may follow the first envelope. const cluster = await ClusterBuildDefaults.create(baseOptions()) const names = collectStepNames(cluster.children) - expect(names.indexOf(SeedSolanaRosterStep)).toBe( - names.indexOf(ScheduleBatchGroupsStep) + 1 + expect(names.indexOf(DeployEthereumStep)).toBeGreaterThan( + names.indexOf(ScheduleBatchGroupsStep) ) - expect(names.indexOf(BootstrapEpochStep)).toBe( - names.indexOf(SeedSolanaRosterStep) + 1 + expect(names.indexOf(SeedSolanaRosterStep)).toBeGreaterThan( + names.indexOf(DeployEthereumStep) + ) + expect(names.indexOf(BootstrapEpochStep)).toBeGreaterThan( + names.indexOf(SeedSolanaRosterStep) ) }) @@ -70,8 +73,9 @@ describe("ClusterBuildDefaults — EpochBootstrap step order", () => { }) const names = collectStepNames(cluster.children) expect(names).not.toContain(SeedSolanaRosterStep) - expect(names.indexOf(BootstrapEpochStep)).toBe( - names.indexOf(ScheduleBatchGroupsStep) + 1 + expect(names).not.toContain(DeployEthereumStep) + expect(names.indexOf(BootstrapEpochStep)).toBeGreaterThan( + names.indexOf(ScheduleBatchGroupsStep) ) }) }) diff --git a/packages/cluster-tool/tests/orchestration/ethereum/EthereumOutpostBootstrapper.test.ts b/packages/cluster-tool/tests/orchestration/ethereum/EthereumOutpostBootstrapper.test.ts index 43484720e..881f6c49c 100644 --- a/packages/cluster-tool/tests/orchestration/ethereum/EthereumOutpostBootstrapper.test.ts +++ b/packages/cluster-tool/tests/orchestration/ethereum/EthereumOutpostBootstrapper.test.ts @@ -6,6 +6,7 @@ import { toURL } from "@wireio/cluster-tool/utils" const AnvilAccount0Address = "0xf39Fd6e51aad88F6F4ce6aB8827279cffFb92266" const AnvilAccount0PrivateKey = "0xac0974bec39a17e36ba4a6b4d238ff944bacb478cbed5efcae784d7bf4f2ff80" +const InitialRoster = [[AnvilAccount0Address]] describe("EthereumOutpostBootstrapper.generateAccounts", () => { it("generates the requested count deterministically from anvil's mnemonic", () => { @@ -40,7 +41,10 @@ describe("EthereumOutpostBootstrapper constructor", () => { ethereumPath: "", anvilDataPath: "/tmp/anvil", rpcUrl, - deploymentsPath + deploymentsPath, + initialOperatorGroups: InitialRoster, + initialActiveGroupIndex: 0, + epochDurationSec: 60 }) ).toThrow(/ethereumPath is required/) }) @@ -52,7 +56,10 @@ describe("EthereumOutpostBootstrapper constructor", () => { ethereumPath: "/repo/eth", anvilDataPath: "", rpcUrl, - deploymentsPath + deploymentsPath, + initialOperatorGroups: InitialRoster, + initialActiveGroupIndex: 0, + epochDurationSec: 60 }) ).toThrow(/anvilDataPath is required/) }) @@ -64,7 +71,10 @@ describe("EthereumOutpostBootstrapper constructor", () => { ethereumPath: "/repo/eth", anvilDataPath: "/tmp/anvil", rpcUrl: "", - deploymentsPath + deploymentsPath, + initialOperatorGroups: InitialRoster, + initialActiveGroupIndex: 0, + epochDurationSec: 60 }) ).toThrow(/rpcUrl is required/) }) @@ -76,8 +86,26 @@ describe("EthereumOutpostBootstrapper constructor", () => { ethereumPath: "/repo/eth", anvilDataPath: "/tmp/anvil", rpcUrl, - deploymentsPath: "" + deploymentsPath: "", + initialOperatorGroups: InitialRoster, + initialActiveGroupIndex: 0, + epochDurationSec: 60 }) ).toThrow(/deploymentsPath is required/) }) + + it("throws when the initial roster is empty", () => { + expect( + () => + new EthereumOutpostBootstrapper({ + ethereumPath: "/repo/eth", + anvilDataPath: "/tmp/anvil", + rpcUrl, + deploymentsPath, + initialOperatorGroups: [], + initialActiveGroupIndex: 0, + epochDurationSec: 60 + }) + ).toThrow(/initialOperatorGroups must contain non-empty groups/) + }) }) diff --git a/packages/cluster-tool/tests/orchestration/ethereum/EthereumOutpostSteps.test.ts b/packages/cluster-tool/tests/orchestration/ethereum/EthereumOutpostSteps.test.ts index d03c3fad5..4e9ec2fd8 100644 --- a/packages/cluster-tool/tests/orchestration/ethereum/EthereumOutpostSteps.test.ts +++ b/packages/cluster-tool/tests/orchestration/ethereum/EthereumOutpostSteps.test.ts @@ -1,5 +1,8 @@ import { Steps } from "@wireio/cluster-tool/orchestration" import { Report } from "@wireio/cluster-tool/report" +import { OperatorType } from "@wireio/opp-typescript-models" + +import { fixtureOperatorAccount } from "../outputs/operatorAccountFixture.js" describe("Steps.ethereumOutpost.deploy", () => { it("builds an input-less deploy step with a runner", () => { @@ -14,3 +17,47 @@ describe("Steps.ethereumOutpost.deploy", () => { expect(typeof step.runner).toBe("function") }) }) + +describe("Steps.ethereumOutpost.resolveInitialOperatorGroups", () => { + it("maps every depot schedule group to Ethereum addresses in schedule order", () => { + const fixtures = [ + fixtureOperatorAccount("batchop.a", OperatorType.BATCH, "wireno.alpha"), + fixtureOperatorAccount("batchop.b", OperatorType.BATCH, "wireno.bravo"), + fixtureOperatorAccount("batchop.c", OperatorType.BATCH, "wireno.charlie") + ], + operators = fixtures.map((operator, index) => ({ + ...operator, + ethereum: { + ...operator.ethereum, + address: `0x${String(index + 1).padStart(40, "0")}` + } + })) + const groups = Steps.ethereumOutpost.resolveInitialOperatorGroups(operators, [ + ["wireno.charlie", "wireno.alpha"], + ["wireno.bravo"] + ]) + + expect(groups).toEqual([ + [operators[2].ethereum.address, operators[0].ethereum.address], + [operators[1].ethereum.address] + ]) + }) + + it("rejects schedule members missing from the provisioned batch roster", () => { + const operators = [ + fixtureOperatorAccount("batchop.a", OperatorType.BATCH, "wireno.alpha") + ] + expect(() => + Steps.ethereumOutpost.resolveInitialOperatorGroups(operators, [["wireno.ghost"]]) + ).toThrow(/wireno\.ghost.*not found among provisioned batch operators/) + }) + + it("rejects empty schedules and groups", () => { + expect(() => Steps.ethereumOutpost.resolveInitialOperatorGroups([], [])).toThrow( + /schedule is empty/ + ) + expect(() => Steps.ethereumOutpost.resolveInitialOperatorGroups([], [[]])).toThrow( + /schedule group 0 is empty/ + ) + }) +}) From 17ae20ebb9186575bcda7793ae1a5bd300fcc116 Mon Sep 17 00:00:00 2001 From: Huang-Ming Huang Date: Tue, 22 Sep 2026 20:48:04 +0000 Subject: [PATCH 11/16] Use generated epoch state in the recovery flow Change-Id: Ic0c0ab55561899c47baf557a246aa1cbf9d6c46b --- .../src/TerminationScenario.ts | 23 ++++--------------- 1 file changed, 5 insertions(+), 18 deletions(-) diff --git a/packages/flow-batch-operator-termination/src/TerminationScenario.ts b/packages/flow-batch-operator-termination/src/TerminationScenario.ts index e3901887c..22023f214 100644 --- a/packages/flow-batch-operator-termination/src/TerminationScenario.ts +++ b/packages/flow-batch-operator-termination/src/TerminationScenario.ts @@ -81,30 +81,17 @@ interface AbsorbedRemovalCheckpoint { solanaNextEpoch: number } -/** Epoch state returned by the companion WIRE-385 SYSIO schema. */ -interface ScheduleRecoveryEpochState - extends SysioContracts.SysioEpochEpochStateType { - next_batch_op_groups: string[][] -} - -/** - * Read the epoch state through the WIRE-385 schema boundary. - * - * A clean Tools checkout may still resolve the last published SDK while the - * companion Libraries PR is pending. Keep that temporary type lag isolated - * here, and verify the deployed contract response before the flow uses it. - */ +/** Read the generated epoch state and verify the deployed schedule capability. */ async function readScheduleRecoveryEpochState( ctx: ClusterBuildContext -): Promise { +): Promise { const state = await Steps.contracts.sysio.epoch.readEpochState(ctx) - const nextGroups = (state as Partial | undefined) - ?.next_batch_op_groups + Assert.ok(state, "epoch state is not initialized") Assert.ok( - Array.isArray(nextGroups), + Array.isArray(state.next_batch_op_groups), "epoch state does not expose next_batch_op_groups; deploy the WIRE-385 SYSIO schema before running this flow" ) - return state as ScheduleRecoveryEpochState + return state } const AbsorbedRemovalCheckpointKey = outputKey( From f3237b01a68a7ae00b15a6ba1d791acdf4e91551 Mon Sep 17 00:00:00 2001 From: Huang-Ming Huang Date: Tue, 22 Sep 2026 21:55:07 +0000 Subject: [PATCH 12/16] Keep WIRE-385 bootstrap coverage and defer recovery scenarios Change-Id: Ife8c43b291260367d861734bc268237913e4ad79 --- jest.config.ts | 3 +- .../contracts/sysio/EpochContractSteps.ts | 4 +- .../contracts/sysio/OpregContractSteps.ts | 58 +- .../sysio/OpregContractSteps.test.ts | 18 - .../jest.config.ts | 22 - .../package.json | 2 +- .../src/ReplacementQuorum.ts | 20 - .../src/StandingSpare.ts | 42 - .../src/TerminationScenario.ts | 1193 +---------------- .../src/TerminationScenarioConstants.ts | 72 +- .../tests/ReplacementQuorum.test.ts | 67 - .../tests/StandingSpare.test.ts | 50 - scripts/flow-heartbeat-monitor.mjs | 6 +- 13 files changed, 34 insertions(+), 1523 deletions(-) delete mode 100644 packages/flow-batch-operator-termination/jest.config.ts delete mode 100644 packages/flow-batch-operator-termination/src/ReplacementQuorum.ts delete mode 100644 packages/flow-batch-operator-termination/src/StandingSpare.ts delete mode 100644 packages/flow-batch-operator-termination/tests/ReplacementQuorum.test.ts delete mode 100644 packages/flow-batch-operator-termination/tests/StandingSpare.test.ts diff --git a/jest.config.ts b/jest.config.ts index ca3aeb889..ac2aed1d2 100644 --- a/jest.config.ts +++ b/jest.config.ts @@ -10,7 +10,7 @@ const config: Config = { // `ClusterConfigProvider.resolve` claims every daemon port (each TCP-probed, // UDP-role ones probed twice) and `findAvailableRange` sweeps a 64-port // window — ~15s per test even with the suite running ALONE. Under the full - // multi-project run that comfortably exceeds a 30s ceiling. + // 8-project run that comfortably exceeds a 30s ceiling. // // An undershot ceiling does NOT fail cleanly here, which is why this is // sized generously rather than trimmed: a test killed mid-`withFileLock` @@ -26,7 +26,6 @@ const config: Config = { "packages/cluster-tool-shared", "packages/cluster-tool", "packages/flow-batch-operator-slashing", - "packages/flow-batch-operator-termination", "packages/debugging-shared", "packages/debugging-server", "packages/debugging-client-shared", diff --git a/packages/cluster-tool/src/orchestration/steps/contracts/sysio/EpochContractSteps.ts b/packages/cluster-tool/src/orchestration/steps/contracts/sysio/EpochContractSteps.ts index 75403408e..5b3a2588e 100644 --- a/packages/cluster-tool/src/orchestration/steps/contracts/sysio/EpochContractSteps.ts +++ b/packages/cluster-tool/src/orchestration/steps/contracts/sysio/EpochContractSteps.ts @@ -125,8 +125,8 @@ export namespace EpochContractSteps { } /** - * The depot's last activated batch-operator window, including historical groups. - * Current duty is selected by its cursor; `next_batch_op_groups` is separate. + * The depot's whole sliding-window batch-operator schedule — every group, + * `[current, next, next+1]` at the default `batch_op_groups` of 3. * * @param ctx - The build context. * @returns The schedule groups (empty when the epoch state has no row yet). diff --git a/packages/cluster-tool/src/orchestration/steps/contracts/sysio/OpregContractSteps.ts b/packages/cluster-tool/src/orchestration/steps/contracts/sysio/OpregContractSteps.ts index e64f50342..e3ed48381 100644 --- a/packages/cluster-tool/src/orchestration/steps/contracts/sysio/OpregContractSteps.ts +++ b/packages/cluster-tool/src/orchestration/steps/contracts/sysio/OpregContractSteps.ts @@ -7,7 +7,7 @@ import { } from "../../../ClusterBuildStep.js" import type { StepInput } from "../../../StepRunner.js" -const { SysioContractAccount, SysioContractName } = SysioContracts +const { SysioContractName } = SysioContracts /** Steps for `sysio.opreg` (operator registry) actions. */ export namespace OpregContractSteps { @@ -18,9 +18,7 @@ export namespace OpregContractSteps { } /** `sysio.opreg::setconfig` — availability caps, termination thresholds, collateral minimums. */ - export function planSetconfig< - C extends ClusterBuildContext = ClusterBuildContext - >( + export function planSetconfig( actor: Report.Actor, name: string, description: string, @@ -56,9 +54,7 @@ export namespace OpregContractSteps { } /** `sysio.opreg::regoperator` — register a batch operator / underwriter / producer. */ - export function planRegoperator< - C extends ClusterBuildContext = ClusterBuildContext - >( + export function planRegoperator( actor: Report.Actor, name: string, description: string, @@ -86,52 +82,4 @@ export namespace OpregContractSteps { .getSysioContract(SysioContractName.opreg) .actions.regoperator.invoke(input.data) } - - /** Input for {@link planSlash} — the generated `opreg::slash` data. */ - export interface SlashInput extends StepInput { - readonly kind: "OpregContractSteps.SlashInput" - readonly data: SysioContracts.SysioOpregSlashAction - } - - /** - * `sysio.opreg::slash` — mark an operator slashed under the challenge - * contract authority required by the on-chain action. - */ - export function planSlash< - C extends ClusterBuildContext = ClusterBuildContext - >( - actor: Report.Actor, - name: string, - description: string, - options: ClusterBuildStepOptions, - data: SysioContracts.SysioOpregSlashAction - ): ClusterBuildStep { - return ClusterBuildStep.create( - actor, - name, - description, - options, - { kind: "OpregContractSteps.SlashInput", data }, - runSlash - ) - } - - /** Named runner — `sysio.opreg::slash` authorized by `sysio.chalg`. */ - export async function runSlash( - ctx: C, - input: SlashInput, - signal: AbortSignal - ): Promise { - signal.throwIfAborted() - await ctx.wire - .getSysioContract(SysioContractName.opreg) - .actions.slash.invoke(input.data, { - authorization: [ - { - actor: SysioContractAccount[SysioContractName.chalg], - permission: "active" - } - ] - }) - } } diff --git a/packages/cluster-tool/tests/orchestration/steps/contracts/sysio/OpregContractSteps.test.ts b/packages/cluster-tool/tests/orchestration/steps/contracts/sysio/OpregContractSteps.test.ts index 62f0569da..ec9119a93 100644 --- a/packages/cluster-tool/tests/orchestration/steps/contracts/sysio/OpregContractSteps.test.ts +++ b/packages/cluster-tool/tests/orchestration/steps/contracts/sysio/OpregContractSteps.test.ts @@ -49,22 +49,4 @@ describe("Steps.contracts.sysio.opreg", () => { expect(step.input.data.is_bootstrapped).toBe(true) expect(typeof step.runner).toBe("function") }) - - it("slash carries the opreg::slash data", () => { - const data: SysioContracts.SysioOpregSlashAction = { - account: "batchop.a", - reason: "schedule recovery regression" - } - const step = Steps.contracts.sysio.opreg.planSlash( - Report.Actor.Sysio, - "slash-current-operator", - "slash one operator from the current group", - {}, - data - ) - expect(step.actor).toBe(Report.Actor.Sysio) - expect(step.input.kind).toBe("OpregContractSteps.SlashInput") - expect(step.input.data).toBe(data) - expect(typeof step.runner).toBe("function") - }) }) diff --git a/packages/flow-batch-operator-termination/jest.config.ts b/packages/flow-batch-operator-termination/jest.config.ts deleted file mode 100644 index aed8d4aff..000000000 --- a/packages/flow-batch-operator-termination/jest.config.ts +++ /dev/null @@ -1,22 +0,0 @@ -const config = { - displayName: "flow-batch-operator-termination", - testEnvironment: "node", - roots: ["/tests"], - testMatch: ["**/*.test.ts"], - transform: { - "^.+\\.ts$": [ - "ts-jest", - { - tsconfig: "/../../etc/tsconfig/tsconfig.base.jest.json" - } - ] - }, - moduleNameMapper: { - "^(\\.{1,2}/.*)\\.js$": "$1", - "^@wireio/test-flow-batch-operator-termination/(.*)\\.js$": - "/src/$1", - "^@wireio/test-flow-batch-operator-termination/(.*)$": "/src/$1" - } -} - -export default config diff --git a/packages/flow-batch-operator-termination/package.json b/packages/flow-batch-operator-termination/package.json index 13e458dfb..17730193a 100644 --- a/packages/flow-batch-operator-termination/package.json +++ b/packages/flow-batch-operator-termination/package.json @@ -3,7 +3,7 @@ "version": "0.1.17", "private": true, "type": "commonjs", - "description": "Flow: Batch Operator Termination, Schedule Freeze, and Recovery", + "description": "Flow: Batch Operator Termination via Delivery Underperformance", "scripts": { "build": "tsc -b tsconfig.json", "test": "node lib/index.js", diff --git a/packages/flow-batch-operator-termination/src/ReplacementQuorum.ts b/packages/flow-batch-operator-termination/src/ReplacementQuorum.ts deleted file mode 100644 index d875dc491..000000000 --- a/packages/flow-batch-operator-termination/src/ReplacementQuorum.ts +++ /dev/null @@ -1,20 +0,0 @@ -import Assert from "node:assert" -import { TerminationScenarioConstants as Constants } from "./TerminationScenarioConstants.js" - -/** Select an original peer whose absence makes the replacement necessary for quorum. */ -export function assertReplacementQuorumPeer( - groups: readonly (readonly string[])[], - replacements: readonly string[], - replacement: string -): string { - Assert.equal(replacements.length, Constants.RecoveryOperatorLabels.length) - Assert.equal(new Set(replacements).size, replacements.length) - Assert.ok(replacements.includes(replacement), "unknown replacement") - const group = groups.find(members => members.includes(replacement)) - Assert.ok(group != null, `${replacement} is not seated`) - Assert.equal(group.length, Constants.OperatorsPerEpoch) - Assert.equal(new Set(group).size, group.length) - const peer = group.find(account => !replacements.includes(account)) - Assert.ok(peer != null, "replacement duty has no original peer") - return peer -} diff --git a/packages/flow-batch-operator-termination/src/StandingSpare.ts b/packages/flow-batch-operator-termination/src/StandingSpare.ts deleted file mode 100644 index 8a1cc3589..000000000 --- a/packages/flow-batch-operator-termination/src/StandingSpare.ts +++ /dev/null @@ -1,42 +0,0 @@ -import Assert from "node:assert" -import { TerminationScenarioConstants as Constants } from "./TerminationScenarioConstants.js" - -/** Require a full, disjoint nine-seat schedule, regardless of spare count. */ -export function assertCompleteSchedule( - groups: readonly (readonly string[])[] -): void { - Assert.equal(groups.length, Constants.BatchOperatorGroups, "schedule group count changed") - const members = new Set() - for (const group of groups) { - Assert.equal(group.length, Constants.OperatorsPerEpoch, "schedule contains a short group") - for (const member of group) { - Assert.ok(!members.has(member), `${member} is seated in more than one group`) - members.add(member) - } - } - Assert.equal(members.size, Constants.ScheduleSeatCount, "schedule window is not full") -} - -/** Choose an ACTIVE operator outside a complete activated schedule window. */ -export function assertStandingSpare( - activeAccounts: ReadonlySet, - groups: readonly (readonly string[])[], - expectedSpareCount: number -): string { - assertCompleteSchedule(groups) - const seated = groups.flat() - Assert.ok( - seated.every(account => activeAccounts.has(account)), - "activated window contains an inactive member" - ) - const spares = [...activeAccounts] - .filter(account => !seated.includes(account)) - .sort() - Assert.equal( - spares.length, - expectedSpareCount, - "standing-spare pool changed before recovery" - ) - Assert.ok(spares.length > 0, "no standing spare is available") - return spares[0] -} diff --git a/packages/flow-batch-operator-termination/src/TerminationScenario.ts b/packages/flow-batch-operator-termination/src/TerminationScenario.ts index 22023f214..983b2eb68 100644 --- a/packages/flow-batch-operator-termination/src/TerminationScenario.ts +++ b/packages/flow-batch-operator-termination/src/TerminationScenario.ts @@ -5,20 +5,14 @@ import { SysioContracts } from "@wireio/sdk-core" import { OperatorType } from "@wireio/opp-typescript-models" import { ClusterBuildPhase, - ClusterBuildStep, - ClusterConfigProvider, EthereumCollateralTool, FlowScenario, - NodeConfig, - NodeRole, - OperatorDaemonTool, Report, SolanaCollateralTool, SolanaOutpostBootstrapper, SolanaOutpostProgramTool, Steps, WireOperatorProvisioningTool, - contractView, getLogger, matchesProtoEnum, outputKey, @@ -29,97 +23,16 @@ import { verifyStep, type ClusterBuild, type ClusterBuildContext, - type ClusterBuildOptions, - type StepInput + type ClusterBuildOptions } from "@wireio/cluster-tool" import { TerminationScenarioConstants as Constants } from "./TerminationScenarioConstants.js" -import { assertReplacementQuorumPeer } from "./ReplacementQuorum.js" -import { assertCompleteSchedule, assertStandingSpare } from "./StandingSpare.js" const log = getLogger(__filename) -const { - SysioContractName, - SysioOpregActiontype, - SysioOpregOperatortype, - SysioOpregOperatorstatus -} = SysioContracts +const { SysioContractName, SysioOpregActiontype, SysioOpregOperatorstatus } = + SysioContracts const { Actor } = Report -/** Minimal Ethereum inbound surface needed to prove sequential acceptance. */ -interface EthereumInboundView { - activeGroupIndex(): Promise - batchOpGroups(groupIndex: number, memberIndex: number): Promise - epochDeliveries(epochIndex: number, operator: string): Promise - nextEpochIndex(): Promise -} - -/** Anchor-decoded subset of the Solana outpost configuration account. */ -interface SolanaOutpostConfigAccount { - nextEpochIndex: BN -} - -/** State captured after the first schedule candidate is withheld. */ -interface WithheldScheduleCheckpoint { - epochIndex: number - activeGroup: string[] - ethereumNextEpoch: number - solanaNextEpoch: number -} - -/** State captured after a replacement makes the held window publishable. */ -interface RepairedScheduleCheckpoint extends WithheldScheduleCheckpoint { - nextGroup: string[] -} - -/** Complete lookahead published after the doomed operator is terminated. */ -interface AbsorbedRemovalCheckpoint { - epochIndex: number - servingGroup: string[] - nextGroups: string[][] - ethereumNextEpoch: number - solanaNextEpoch: number -} - -/** Read the generated epoch state and verify the deployed schedule capability. */ -async function readScheduleRecoveryEpochState( - ctx: ClusterBuildContext -): Promise { - const state = await Steps.contracts.sysio.epoch.readEpochState(ctx) - Assert.ok(state, "epoch state is not initialized") - Assert.ok( - Array.isArray(state.next_batch_op_groups), - "epoch state does not expose next_batch_op_groups; deploy the WIRE-385 SYSIO schema before running this flow" - ) - return state -} - -const AbsorbedRemovalCheckpointKey = outputKey( - "TerminationScenario.absorbedRemovalCheckpoint", - "complete lookahead and outpost cursors after the termination is absorbed" -) - -const WithheldScheduleCheckpointKey = outputKey( - "TerminationScenario.withheldScheduleCheckpoint", - "depot duty and outpost epoch cursors after the first withheld schedule window" -) -const RepairedScheduleCheckpointKey = outputKey( - "TerminationScenario.repairedScheduleCheckpoint", - "held duty, next group, and outpost cursors after the repaired window is published" -) -const HeldSlashAccountKey = outputKey( - "TerminationScenario.heldSlashAccount", - "operator removed from the announced group while schedule duty is held" -) -const HeldSlashEpochKey = outputKey( - "TerminationScenario.heldSlashEpoch", - "depot epoch in which a member of the held group was slashed" -) -const RecoverySlashAccountKey = outputKey( - "TerminationScenario.recoverySlashAccount", - "active current-group operator selected for the first recovery slash" -) - /** * Post-deposit snapshot of the doomed operator's ETH wallet balance (wei), * captured after BOTH bonds landed but BEFORE termination begins. The remit @@ -216,419 +129,34 @@ interface SolanaCollateralLedgerEntry { /** The slice of the SOL outpost's `OperatorRegistry` PDA account this flow reads. */ interface SolanaOperatorRegistryAccount { - activeGroupIndex: number collateralByCode: SolanaCollateralLedgerEntry[] - groupCount: number - groups: SolanaOperatorGroup[] -} - -/** One fixed-capacity group in the zero-copy Solana operator registry. */ -interface SolanaOperatorGroup { - memberCount: number - members: PublicKey[] -} - -/** Signer records retained for one accepted Solana inbound epoch. */ -interface SolanaOperatorDelivery { - operator: PublicKey -} - -/** Signer records retained for one accepted Solana inbound epoch. */ -interface SolanaEpochDeliveriesAccount { - deliveries: SolanaOperatorDelivery[] } /** Anchor account-client surface for a runtime-loaded IDL (untyped `Program` namespace). */ interface SolanaAccountClient { fetch(address: PublicKey): Promise - fetchNullable(address: PublicKey): Promise -} - -/** Bound Solana OPP account readers and their program addresses. */ -interface SolanaOppAccounts { - accounts: Record - configAddress: PublicKey - programId: PublicKey -} - -/** Cross-chain signing identities for one replacement operator. */ -interface ReplacementAddresses { - ethereum: string - solana: PublicKey -} - -/** Compare ordered operator groups without depending on array identity. */ -function sameGroup(left: readonly string[], right: readonly string[]): boolean { - return ( - left.length === right.length && - left.every((value, index) => value === right[index]) - ) } -/** Read the Ethereum outpost's next sequential inbound epoch. */ -async function readEthereumNextEpoch( +/** The SOL outpost's on-chain collateral ledger from the `OperatorRegistry` PDA (a read). */ +async function readSolanaCollateralLedger( ctx: ClusterBuildContext -): Promise { - return Number(await loadEthereumInbound(ctx).nextEpochIndex()) -} - -/** Bind the deployed Ethereum inbound contract to its read-only flow surface. */ -function loadEthereumInbound(ctx: ClusterBuildContext): EthereumInboundView { - const deploymentsPath = ClusterConfigProvider.ethereumDeploymentsPath( - ctx.config - ) - const addresses = EthereumCollateralTool.loadOutpostAddresses(deploymentsPath) - return contractView( - addresses.OPPInbound, - EthereumCollateralTool.loadOutpostAbi( - ctx.config.ethereumPath, - "OPPInbound" - ), - ctx.ethereum.wallet.signer - ) -} - -/** Read the Solana outpost's next sequential inbound epoch. */ -async function readSolanaNextEpoch(ctx: ClusterBuildContext): Promise { - const { accounts, configAddress } = loadSolanaOppAccounts(ctx) - const config = (await accounts[ - Constants.SolanaOutpostConfigAccountName - ].fetch(configAddress)) as SolanaOutpostConfigAccount - return Number(config.nextEpochIndex.toString()) -} - -/** Load the Solana OPP program and the account namespace used by flow reads. */ -function loadSolanaOppAccounts(ctx: ClusterBuildContext): SolanaOppAccounts { - const reader = ctx.keyStore.assertOperator( - Constants.RecoverySolanaReaderLabel - ) +): Promise { + const operator = ctx.keyStore.assertOperator(Constants.DoomedOperatorLabel) const program = SolanaCollateralTool.loadOppOutpostProgram( ctx, - solanaKeypair(reader.solana) - ) - const configAddress = SolanaOutpostProgramTool.derivePda( - program.programId, - Buffer.from(SolanaOutpostBootstrapper.PdaSeed.OutpostConfig) - ) - const accounts: Record = program.account - return { accounts, configAddress, programId: program.programId } -} - -/** Cross-chain signing addresses for one WIRE operator account. */ -function replacementAddresses( - ctx: ClusterBuildContext, - account: string -): ReplacementAddresses { - const operator = ctx.keyStore.operators.find( - entry => entry.account === account - ) - Assert.ok(operator != null, `${account} is absent from the cluster key store`) - Assert.ok(operator.ethereum != null, `${account} has no Ethereum identity`) - Assert.ok(operator.solana != null, `${account} has no Solana identity`) - return { - ethereum: operator.ethereum.address, - solana: solanaKeypair(operator.solana).publicKey - } -} - -/** - * Prove one replacement signed accepted deliveries for the same duty epoch on - * both outposts. Each contract records a signer only after its active-group - * admission check, so this is also direct evidence that the replacement's - * propagated address was seated rather than merely present in the depot group. - */ -async function replacementDeliveredOnBothOutposts( - ctx: ClusterBuildContext, - account: string, - epochIndex: number -): Promise { - const addresses = replacementAddresses(ctx, account) - const ethereumDigest = await loadEthereumInbound(ctx).epochDeliveries( - epochIndex, - addresses.ethereum - ) - if (/^0x0{64}$/i.test(ethereumDigest)) return false - - const { accounts, programId } = loadSolanaOppAccounts(ctx) - const epochBytes = Buffer.alloc(4) - epochBytes.writeUInt32LE(epochIndex) - const deliveriesAddress = SolanaOutpostProgramTool.derivePda( - programId, - Buffer.from("epoch_deliveries"), - epochBytes - ) - const deliveries = (await accounts.epochDeliveries.fetchNullable( - deliveriesAddress - )) as SolanaEpochDeliveriesAccount | null - return ( - deliveries != null && - deliveries.deliveries.some(entry => entry.operator.equals(addresses.solana)) - ) -} - -interface StopReplacementPeerInput extends StepInput { - readonly kind: "TerminationScenario.StopReplacementPeerInput" - readonly label: string -} - -/** Stop one original signer so this replacement is necessary for quorum. */ -async function runStopReplacementPeer( - ctx: ClusterBuildContext, - input: StopReplacementPeerInput, - signal: AbortSignal -): Promise { - signal.throwIfAborted() - const state = await readScheduleRecoveryEpochState(ctx) - const replacements = Constants.RecoveryOperatorLabels.map( - label => ctx.keyStore.assertOperator(label).account - ) - const replacement = ctx.keyStore.assertOperator(input.label).account - const account = assertReplacementQuorumPeer( - state.batch_op_groups, - replacements, - replacement - ) - const operator = ctx.keyStore.operators.find(entry => entry.account === account) - Assert.ok(operator != null, `${account} has no operator identity`) - const node = NodeConfig.plan(ctx.config).find( - entry => - entry.role === NodeRole.batch_operator && - entry.batchOperatorLabel === operator.label - ) - Assert.ok(node != null, `${account} has no planned batch-operator daemon`) - const daemon = ctx.processManager.get(node.name) - Assert.ok(daemon != null, `${node.name} is not registered`) - // Replacements in the same group select the same peer. stop() is idempotent. - await daemon.stop(signal) - log.info( - `stopped ${account}'s daemon; ${replacement} is required for quorum` - ) -} - -/** Read whether an operator is SLASHED in the depot registry. */ -async function operatorIsSlashed( - ctx: ClusterBuildContext, - account: string -): Promise { - const { rows } = await ctx.wire - .getSysioContract(SysioContractName.opreg) - .tables.operators.query({ limit: Constants.OperatorsQueryLimit }) - const target = rows.find(row => row.account === account) - return ( - target != null && - matchesProtoEnum( - target.status, - SysioOpregOperatorstatus, - SysioOpregOperatorstatus.OPERATOR_STATUS_SLASHED - ) - ) -} - -/** Read the ACTIVE batch-operator account names from the live registry. */ -async function readActiveBatchOperatorAccounts( - ctx: ClusterBuildContext -): Promise> { - const { rows } = await ctx.wire - .getSysioContract(SysioContractName.opreg) - .tables.operators.query({ limit: Constants.OperatorsQueryLimit }) - return new Set( - rows - .filter( - row => - matchesProtoEnum( - row.type, - SysioOpregOperatortype, - SysioOpregOperatortype.OPERATOR_TYPE_BATCH - ) && - matchesProtoEnum( - row.status, - SysioOpregOperatorstatus, - SysioOpregOperatorstatus.OPERATOR_STATUS_ACTIVE - ) - ) - .map(row => row.account) - ) -} - -/** Select and slash an active member of the live current group without crossing an epoch boundary. */ -async function runSlashRecoveryTarget( - ctx: ClusterBuildContext, - _input: null, - signal: AbortSignal -): Promise { - signal.throwIfAborted() - const before = await readScheduleRecoveryEpochState(ctx) - assertCompleteSchedule(before.batch_op_groups) - assertCompleteSchedule(before.next_batch_op_groups) - const current = before.batch_op_groups[before.current_batch_op_group] ?? [] - const activeAccounts = await readActiveBatchOperatorAccounts(ctx) - Assert.equal( - activeAccounts.size, - Constants.ScheduleSeatCount, - "recovery slash did not start at the exact active roster floor" - ) - Assert.ok( - current.every(account => activeAccounts.has(account)), - "current duty contains an inactive historical placeholder" - ) - const readerAccount = ctx.keyStore.assertOperator( - Constants.RecoverySolanaReaderLabel - ).account - const account = current.find(member => member !== readerAccount) - Assert.ok( - account != null, - "current duty has no eligible recovery slash target" - ) - await Steps.contracts.sysio.opreg.runSlash( - ctx, - { - kind: "OpregContractSteps.SlashInput", - data: { - account, - reason: "WIRE-385 schedule recovery regression" - } - }, - signal - ) - const after = await readScheduleRecoveryEpochState(ctx) - const afterCurrent = after.batch_op_groups[after.current_batch_op_group] ?? [] - Assert.equal( - Number(after.current_epoch_index), - Number(before.current_epoch_index), - "epoch advanced across the recovery slash" - ) - Assert.ok( - sameGroup(afterCurrent, current), - "current duty changed across the recovery slash" - ) - ctx.outputs.set(RecoverySlashAccountKey, account) -} - -interface SlashStandingSpareInput extends StepInput { - readonly kind: "TerminationScenario.SlashStandingSpareInput" - readonly ordinal: number -} - -/** Slash one verified standing spare, leaving every activated seat available. */ -async function runSlashStandingSpare( - ctx: ClusterBuildContext, - input: SlashStandingSpareInput, - signal: AbortSignal -): Promise { - signal.throwIfAborted() - let spare: string | null = null - await pollUntil( - `standing spare ${input.ordinal + 1} is outside a full active window`, - async () => { - const state = await readScheduleRecoveryEpochState(ctx) - const active = await readActiveBatchOperatorAccounts(ctx) - if ( - active.size !== Constants.BatchOperatorCount - input.ordinal || - state.batch_op_groups.length !== Constants.BatchOperatorGroups || - state.batch_op_groups.some( - group => group.length !== Constants.OperatorsPerEpoch - ) - ) { - return false - } - try { - spare = assertStandingSpare( - active, - state.batch_op_groups, - Constants.StandingSpareCount - input.ordinal - ) - return true - } catch { - return false - } - }, - Constants.recoveryDeadlineMs(Constants.BatchOperatorGroups + 2), - Constants.PollIntervalMs - ) - Assert.ok(spare != null, "no safe standing spare was selected") - await Steps.contracts.sysio.opreg.runSlash( - ctx, - { - kind: "OpregContractSteps.SlashInput", - data: { - account: spare, - reason: `WIRE-385 exhaust standing spare ${input.ordinal + 1}` - } - }, - signal + solanaKeypair(operator.solana) ) -} - -/** Remove one eligible member from the group whose duty is currently held. */ -async function runSlashHeldGroupMember( - ctx: ClusterBuildContext, - _input: null, - signal: AbortSignal -): Promise { - signal.throwIfAborted() - const checkpoint = ctx.outputs.assert(WithheldScheduleCheckpointKey) - const { rows } = await ctx.wire - .getSysioContract(SysioContractName.opreg) - .tables.operators.query({ limit: Constants.OperatorsQueryLimit }) - const firstSlashAccount = ctx.outputs.assert(RecoverySlashAccountKey) - const readerAccount = ctx.keyStore.assertOperator( - Constants.RecoverySolanaReaderLabel - ).account - const candidate = checkpoint.activeGroup.find(account => { - if (account === firstSlashAccount || account === readerAccount) { - return false - } - const row = rows.find(operator => operator.account === account) - return ( - row != null && - matchesProtoEnum( - row.type, - SysioOpregOperatortype, - SysioOpregOperatortype.OPERATOR_TYPE_BATCH - ) && - matchesProtoEnum( - row.status, - SysioOpregOperatorstatus, - SysioOpregOperatorstatus.OPERATOR_STATUS_ACTIVE - ) - ) - }) - Assert.ok(candidate != null, "held group has no eligible slash candidate") - const state = await readScheduleRecoveryEpochState(ctx) - await Steps.contracts.sysio.opreg.runSlash( - ctx, - { - kind: "OpregContractSteps.SlashInput", - data: { - account: candidate, - reason: "WIRE-385 held-duty recovery regression" - } - }, - signal - ) - ctx.outputs.set(HeldSlashAccountKey, candidate) - ctx.outputs.set(HeldSlashEpochKey, Number(state.current_epoch_index)) -} - -/** Read the SOL outpost's zero-copy operator registry and schedule. */ -async function readSolanaOperatorRegistry( - ctx: ClusterBuildContext -): Promise { - const { accounts, programId } = loadSolanaOppAccounts(ctx) const [registryAddress] = PublicKey.findProgramAddressSync( [Buffer.from(SolanaOutpostBootstrapper.PdaSeed.OperatorRegistry)], - programId + program.programId ) - return (await accounts[ + // Anchor types `Program.account` per-IDL; for a runtime-loaded IDL the + // account clients are reached by name — one assertion to the string-keyed view. + const accounts: Record = program.account + const registryAccount = (await accounts[ Constants.SolanaOperatorRegistryAccountName ].fetch(registryAddress)) as SolanaOperatorRegistryAccount -} - -/** The SOL outpost's on-chain collateral ledger from the operator registry. */ -async function readSolanaCollateralLedger( - ctx: ClusterBuildContext -): Promise { - return (await readSolanaOperatorRegistry(ctx)).collateralByCode ?? [] + return registryAccount.collateralByCode ?? [] } /** @@ -660,22 +188,15 @@ async function readSolanaCollateralLedger( * outpost's escrow ledger returns to 0, and each wallet is credited the * exact bond amount (wei/lamport-exact — any drift means the outpost decoded * a different amount than the depot encoded). - * 9. **Schedule recovery** — prove two standing spares absorb the terminated - * operator without a hold, remove those spares and one announced operator - * to reach the freeze, then repair two independent roster losses and prove - * publication and rotation resume across both outposts. */ export class TerminationScenario extends FlowScenario { readonly name = "flow-batch-operator-termination" readonly description = - "Terminate and remit a non-bootstrapped operator, then freeze, repair, and resume a depleted batch schedule across Ethereum and Solana" + "Non-bootstrapped batch operator bonds ETH + SOL, misses its scheduled deliveries, is terminated, and both bonds are remitted back" override readonly defaults: ClusterBuildOptions = { epochDurationSec: Constants.EpochDurationSec, batchOperatorCount: Constants.BatchOperatorCount, - operatorsPerEpoch: Constants.OperatorsPerEpoch, - batchOpGroups: Constants.BatchOperatorGroups, - adHocCount: Constants.RecoveryAdHocDaemonCount, terminateMaxConsecutiveMisses: Constants.TerminateMaxConsecutiveMisses, // Depot must enforce "ACTIVE requires the minimum on EVERY registered // outpost chain" — otherwise the operator flips ACTIVE on an empty @@ -1059,160 +580,7 @@ export class TerminationScenario extends FlowScenario { ) ) - // ── 8. Two standing spares absorb the loss without a schedule hold ── - ClusterBuildPhase.create( - cluster, - "AbsorbTerminatedOperator", - "A complete successor reaches both outposts and the next duty rotates" - ).push( - verifyStep( - Actor.Sysio, - "standing-spare-successor-accepted", - "termination leaves eleven ACTIVE operators; both outposts seat a complete live successor", - async ctx => { - const doomed = doomedOperatorAccount(ctx) - let checkpoint: AbsorbedRemovalCheckpoint | null = null - await pollUntil( - "both outposts accept the complete post-termination lookahead", - async () => { - const state = await readScheduleRecoveryEpochState(ctx) - const groups = state.next_batch_op_groups - const active = await readActiveBatchOperatorAccounts(ctx) - if ( - active.size !== Constants.BatchOperatorCount || - groups.length !== Constants.BatchOperatorGroups || - groups.some( - group => group.length !== Constants.OperatorsPerEpoch - ) - ) { - return false - } - assertCompleteSchedule(groups) - const liveGroups = groups.slice(Constants.SuccessorGroupIndex) - Assert.ok( - liveGroups.flat().every(account => active.has(account)), - "successor contains an inactive active/future seat" - ) - Assert.ok( - !liveGroups.flat().includes(doomed), - "terminated operator remains in an active/future seat" - ) - const oldMembers = new Set(state.batch_op_groups.flat()) - Assert.ok( - liveGroups.flat().some(account => !oldMembers.has(account)), - "no standing operator joined the successor" - ) - - const ethereum = loadEthereumInbound(ctx) - const solana = await readSolanaOperatorRegistry(ctx) - if ( - Number(await ethereum.activeGroupIndex()) !== Constants.SuccessorGroupIndex || - solana.activeGroupIndex !== Constants.SuccessorGroupIndex || - solana.groupCount !== Constants.BatchOperatorGroups - ) { - return false - } - for (let groupIndex = Constants.SuccessorGroupIndex; groupIndex < groups.length; ++groupIndex) { - const solanaGroup = solana.groups[groupIndex] - if (solanaGroup?.memberCount !== Constants.OperatorsPerEpoch) { - return false - } - for (let memberIndex = 0; memberIndex < Constants.OperatorsPerEpoch; ++memberIndex) { - const addresses = replacementAddresses( - ctx, - groups[groupIndex][memberIndex] - ) - if ( - (await ethereum.batchOpGroups(groupIndex, memberIndex)).toLowerCase() !== - addresses.ethereum.toLowerCase() || - !solanaGroup.members[memberIndex]?.equals(addresses.solana) - ) { - return false - } - } - } - const [ethereumNextEpoch, solanaNextEpoch] = await Promise.all([ - readEthereumNextEpoch(ctx), - readSolanaNextEpoch(ctx) - ]) - if ( - ethereumNextEpoch < Number(state.current_epoch_index) + 1 || - solanaNextEpoch < Number(state.current_epoch_index) + 1 - ) { - return false - } - checkpoint = { - epochIndex: Number(state.current_epoch_index), - servingGroup: [ - ...(state.batch_op_groups[state.current_batch_op_group] ?? []) - ], - nextGroups: groups.map(group => [...group]), - ethereumNextEpoch, - solanaNextEpoch - } - return true - }, - Constants.recoveryDeadlineMs(Constants.BatchOperatorGroups + 3), - Constants.PollIntervalMs - ) - Assert.ok(checkpoint != null, "absorbed-removal checkpoint was not captured") - ctx.outputs.set(AbsorbedRemovalCheckpointKey, checkpoint) - }, - { - timeoutMs: Constants.recoveryDeadlineMs(Constants.BatchOperatorGroups + 3) - } - ), - verifyStep( - Actor.Sysio, - "standing-spare-duty-rotates", - "the announced live group takes duty and both outpost cursors keep advancing", - async ctx => { - const checkpoint = ctx.outputs.assert(AbsorbedRemovalCheckpointKey) - const doomed = doomedOperatorAccount(ctx) - let sawActivation = false - await pollUntil( - "the complete successor becomes the activated schedule", - async () => { - const state = await readScheduleRecoveryEpochState(ctx) - const current = state.batch_op_groups[state.current_batch_op_group] ?? [] - if ( - Number(state.current_epoch_index) <= checkpoint.epochIndex || - state.current_batch_op_group !== Constants.SuccessorGroupIndex || - !state.batch_op_groups.every((group, index) => - sameGroup(group, checkpoint.nextGroups[index] ?? []) - ) || - !sameGroup(current, checkpoint.nextGroups[Constants.SuccessorGroupIndex]) - ) { - if (!sawActivation) return false - } else { - Assert.ok( - !sameGroup(current, checkpoint.servingGroup), - "serving duty did not rotate to the announced successor" - ) - Assert.ok(!current.includes(doomed), "terminated operator returned to duty") - sawActivation = true - } - const [ethereumNextEpoch, solanaNextEpoch] = await Promise.all([ - readEthereumNextEpoch(ctx), - readSolanaNextEpoch(ctx) - ]) - return ( - sawActivation && - ethereumNextEpoch > checkpoint.ethereumNextEpoch && - solanaNextEpoch > checkpoint.solanaNextEpoch - ) - }, - Constants.recoveryDeadlineMs(Constants.BatchOperatorGroups + 3), - Constants.PollIntervalMs - ) - }, - { - timeoutMs: Constants.recoveryDeadlineMs(Constants.BatchOperatorGroups + 3) - } - ) - ) - - // ── 9. Depot auto-remits the full bond on termination — both outposts ── + // ── 8. Depot auto-remits the full bond on termination — both outposts ── ClusterBuildPhase.create( cluster, "RemitBonds", @@ -1341,534 +709,5 @@ export class TerminationScenario extends FlowScenario { quickStepOptions ) ) - - // ── 10. Exhaust both standing spares before requesting a freeze ── - ClusterBuildPhase.create( - cluster, - "ExhaustStandingSpares", - "Each standing spare is slashed in its own reported contract step" - ).push( - ...Array.from({ length: Constants.StandingSpareCount }, (_, ordinal) => - ClusterBuildStep.create( - Actor.Sysio, - `slash-standing-spare-${ordinal + 1}`, - "slash one ACTIVE operator outside the complete activated window", - { - timeoutMs: Constants.recoveryDeadlineMs( - Constants.BatchOperatorGroups + 2 - ) - }, - { kind: "TerminationScenario.SlashStandingSpareInput", ordinal }, - runSlashStandingSpare - ) - ) - ) - - // ── 11. Reach the exact nine-seat roster floor ── - ClusterBuildPhase.create( - cluster, - "PrepareScheduleRecovery", - "The terminated test operator is gone and a complete window reaches active current duty" - ).push( - verifyStep( - Actor.Sysio, - "exact-minimum-window-ready", - "three disjoint groups of three remain with a fully active current group", - async ctx => { - await pollUntil( - "a complete three-by-three window has a fully active current group", - async () => { - const state = - await readScheduleRecoveryEpochState(ctx) - const groups = state.batch_op_groups - const complete = - groups.length === Constants.BatchOperatorGroups && - groups.every( - group => group.length === Constants.OperatorsPerEpoch - ) - if (!complete) return false - const published = state.next_batch_op_groups - if ( - published.length !== Constants.BatchOperatorGroups || - published.some( - group => group.length !== Constants.OperatorsPerEpoch - ) - ) { - return false - } - assertCompleteSchedule(published) - assertCompleteSchedule(groups) - const current = groups[state.current_batch_op_group] ?? [] - const activeAccounts = await readActiveBatchOperatorAccounts(ctx) - if (activeAccounts.size !== Constants.ScheduleSeatCount) { - return false - } - const readerAccount = ctx.keyStore.assertOperator( - Constants.RecoverySolanaReaderLabel - ).account - return ( - current.every(account => activeAccounts.has(account)) && - current.some(account => account !== readerAccount) - ) - }, - Constants.recoveryDeadlineMs(Constants.BatchOperatorGroups + 2), - Constants.PollIntervalMs - ) - }, - { - timeoutMs: Constants.recoveryDeadlineMs( - Constants.BatchOperatorGroups + 2 - ) - } - ) - ) - - // ── 12. Remove one current member at the exact floor → withhold ── - ClusterBuildPhase.create( - cluster, - "StarveScheduleWindow", - "Slashing one seated operator makes the next tail one seat short" - ).push( - ClusterBuildStep.create( - Actor.Sysio, - "slash-current-member", - "slash the selected active current-group member to exercise withheld-window recovery", - {}, - null, - runSlashRecoveryTarget - ), - verifyStep( - Actor.Sysio, - "capture-withheld-window", - "the first advance enters announced duty and discards an incomplete candidate", - async ctx => { - let checkpoint: WithheldScheduleCheckpoint | null = null - await pollUntil( - "no next window is published after the target is slashed", - async () => { - const state = - await readScheduleRecoveryEpochState(ctx) - const current = - state.batch_op_groups[state.current_batch_op_group] ?? [] - assertCompleteSchedule(state.batch_op_groups) - const withheld = state.next_batch_op_groups.length === 0 - const slashTarget = ctx.outputs.assert(RecoverySlashAccountKey) - if ( - !(await operatorIsSlashed(ctx, slashTarget)) || - !withheld || - current.length === 0 - ) { - return false - } - Assert.ok( - !current.includes(slashTarget), - "withheld duty did not enter the announced successor" - ) - const [ethereumNextEpoch, solanaNextEpoch] = await Promise.all([ - readEthereumNextEpoch(ctx), - readSolanaNextEpoch(ctx) - ]) - const expectedNextEpoch = Number(state.current_epoch_index) + 1 - if ( - ethereumNextEpoch < expectedNextEpoch || - solanaNextEpoch < expectedNextEpoch - ) { - return false - } - checkpoint = { - epochIndex: Number(state.current_epoch_index), - activeGroup: [...current], - ethereumNextEpoch, - solanaNextEpoch - } - return true - }, - Constants.recoveryDeadlineMs(4), - Constants.PollIntervalMs - ) - Assert.ok(checkpoint != null, "withheld checkpoint was not captured") - ctx.outputs.set(WithheldScheduleCheckpointKey, checkpoint) - }, - { timeoutMs: Constants.recoveryDeadlineMs(4) } - ) - ) - - // ── 11. Keep announced duty while both outposts accept later epochs ── - ClusterBuildPhase.create( - cluster, - "HoldAnnouncedDuty", - "The serving window remains intact while candidate publication is withheld" - ).push( - verifyStep( - Actor.Sysio, - "held-duty-remains-live", - "two epochs land on Ethereum and Solana without changing the announced current group", - async ctx => { - const checkpoint = ctx.outputs.assert(WithheldScheduleCheckpointKey) - await pollUntil( - "both outposts advance twice while the depot duty remains held", - async () => { - const state = - await readScheduleRecoveryEpochState(ctx) - const current = - state.batch_op_groups[state.current_batch_op_group] ?? [] - if ( - Number(state.current_epoch_index) > checkpoint.epochIndex && - !sameGroup(current, checkpoint.activeGroup) - ) { - throw new Error( - `duty rotated before repair: ${checkpoint.activeGroup.join(",")} -> ${current.join(",")}` - ) - } - assertCompleteSchedule(state.batch_op_groups) - Assert.equal( - state.next_batch_op_groups.length, - 0, - "a candidate was published before a replacement was provisioned" - ) - return ( - Number(state.current_epoch_index) >= - checkpoint.epochIndex + Constants.RecoveryHeldEpochAdvances && - (await readEthereumNextEpoch(ctx)) >= - checkpoint.ethereumNextEpoch + - Constants.RecoveryHeldEpochAdvances && - (await readSolanaNextEpoch(ctx)) >= - checkpoint.solanaNextEpoch + - Constants.RecoveryHeldEpochAdvances - ) - }, - Constants.recoveryDeadlineMs( - Constants.RecoveryHeldEpochAdvances + 4 - ), - Constants.PollIntervalMs - ) - }, - { - timeoutMs: Constants.recoveryDeadlineMs( - Constants.RecoveryHeldEpochAdvances + 4 - ) - } - ) - ) - - // ── 12. Lose a member of held duty without changing its announced seats ── - ClusterBuildPhase.create( - cluster, - "DegradeHeldDuty", - "An announced seat becomes ineligible while no next window is published" - ).push( - ClusterBuildStep.create( - Actor.Sysio, - "slash-held-member", - "slash one eligible held-group member selected from live chain state", - {}, - null, - runSlashHeldGroupMember - ), - verifyStep( - Actor.Sysio, - "held-seat-preserved", - "the next epoch retains the announced seat as a denominator placeholder", - async ctx => { - const withheld = ctx.outputs.assert(WithheldScheduleCheckpointKey) - const slashedAccount = ctx.outputs.assert(HeldSlashAccountKey) - const slashEpoch = ctx.outputs.assert(HeldSlashEpochKey) - await pollUntil( - "held duty survives a member becoming ineligible", - async () => { - const state = - await readScheduleRecoveryEpochState(ctx) - const current = - state.batch_op_groups[state.current_batch_op_group] ?? [] - if (!sameGroup(current, withheld.activeGroup)) { - throw new Error( - `held duty changed after ${slashedAccount} was slashed: ${withheld.activeGroup.join(",")} -> ${current.join(",")}` - ) - } - Assert.ok( - state.next_batch_op_groups.length === 0, - "schedule became complete before replacements were provisioned" - ) - if ( - Number(state.current_epoch_index) <= slashEpoch || - !(await operatorIsSlashed(ctx, slashedAccount)) - ) { - return false - } - const [ethereumNextEpoch, solanaNextEpoch] = await Promise.all([ - readEthereumNextEpoch(ctx), - readSolanaNextEpoch(ctx) - ]) - const expectedNextEpoch = Number(state.current_epoch_index) + 1 - return ( - ethereumNextEpoch >= expectedNextEpoch && - solanaNextEpoch >= expectedNextEpoch - ) - }, - Constants.recoveryDeadlineMs(4), - Constants.PollIntervalMs - ) - }, - { timeoutMs: Constants.recoveryDeadlineMs(4) } - ) - ) - - // ── 13. Add ACTIVE standbys and their daemons for the two roster losses ── - WireOperatorProvisioningTool.planOperatorAccountProvisioning( - cluster, - "ProvisionScheduleReplacement", - "Provision two bootstrapped batch operators to repair both roster losses", - {}, - Constants.RecoveryOperatorLabels.map((label, index) => ({ - label, - type: OperatorType.BATCH, - ethereumHdIndex: Constants.RecoveryOperatorEthereumHdIndices[index], - isBootstrapped: true, - airdropSolanaLamports: - WireOperatorProvisioningTool.DefaultSolanaAirdropLamports - })) - ) - - ClusterBuildPhase.create( - cluster, - "StartScheduleReplacementDaemons", - "Start both replacement batch-operator daemons before they enter rotation" - ).push( - ...Constants.RecoveryOperatorLabels.map(label => - OperatorDaemonTool.planDaemonStart( - Actor.BatchOperator, - `start-${label}-daemon`, - `start ${label}'s batch-operator daemon`, - {}, - label - ) - ) - ) - - // ── 14. Repair and publish the future window without moving held duty ── - ClusterBuildPhase.create( - cluster, - "RepairScheduleWindow", - "The replacement completes and publishes lookahead without moving current duty" - ).push( - verifyStep( - Actor.Sysio, - "complete-window-published", - "a complete candidate is published while the serving window is unchanged", - async ctx => { - const withheld = ctx.outputs.assert(WithheldScheduleCheckpointKey) - const replacements = Constants.RecoveryOperatorLabels.map( - label => ctx.keyStore.assertOperator(label).account - ) - let checkpoint: RepairedScheduleCheckpoint | null = null - await pollUntil( - "the replacement enables a complete next-window announcement", - async () => { - const state = - await readScheduleRecoveryEpochState(ctx) - const groups = state.next_batch_op_groups - const current = - state.batch_op_groups[state.current_batch_op_group] ?? [] - const complete = - groups.length === Constants.BatchOperatorGroups && - groups.every( - group => group.length === Constants.OperatorsPerEpoch - ) - if ( - !complete || - !groups.some(group => - group.some(member => replacements.includes(member)) - ) - ) { - return false - } - Assert.ok( - sameGroup(current, withheld.activeGroup), - "current duty moved before the repaired lookahead was published" - ) - assertCompleteSchedule(groups) - const nextGroup = groups[1] ?? groups[0] - const [ethereumNextEpoch, solanaNextEpoch] = await Promise.all([ - readEthereumNextEpoch(ctx), - readSolanaNextEpoch(ctx) - ]) - const expectedNextEpoch = Number(state.current_epoch_index) + 1 - if ( - ethereumNextEpoch < expectedNextEpoch || - solanaNextEpoch < expectedNextEpoch - ) { - return false - } - checkpoint = { - epochIndex: Number(state.current_epoch_index), - activeGroup: [...current], - nextGroup: [...nextGroup], - ethereumNextEpoch, - solanaNextEpoch - } - return true - }, - Constants.recoveryDeadlineMs(4), - Constants.PollIntervalMs - ) - Assert.ok(checkpoint != null, "repaired checkpoint was not captured") - ctx.outputs.set(RepairedScheduleCheckpointKey, checkpoint) - }, - { timeoutMs: Constants.recoveryDeadlineMs(4) } - ) - ) - - // ── 15. Rotate only after the repaired lookahead has been published ── - ClusterBuildPhase.create( - cluster, - "ResumeScheduleRotation", - "The next published group takes duty and both outposts remain sequential" - ).push( - verifyStep( - Actor.Sysio, - "rotation-resumes-after-publication", - "the announced next group serves an epoch accepted by Ethereum and Solana", - async ctx => { - const repaired = ctx.outputs.assert(RepairedScheduleCheckpointKey) - await pollUntil( - "the repaired next group serves an epoch accepted by both outposts", - async () => { - const state = - await readScheduleRecoveryEpochState(ctx) - const current = - state.batch_op_groups[state.current_batch_op_group] ?? [] - const [ethereumNextEpoch, solanaNextEpoch] = await Promise.all([ - readEthereumNextEpoch(ctx), - readSolanaNextEpoch(ctx) - ]) - const expectedNextEpoch = Number(state.current_epoch_index) + 1 - return ( - Number(state.current_epoch_index) > repaired.epochIndex && - sameGroup(current, repaired.nextGroup) && - ethereumNextEpoch >= expectedNextEpoch && - solanaNextEpoch >= expectedNextEpoch - ) - }, - Constants.recoveryDeadlineMs(4), - Constants.PollIntervalMs - ) - }, - { timeoutMs: Constants.recoveryDeadlineMs(4) } - ) - ) - - // ── 16. Prove each replacement is seated and delivers on both outposts ── - ClusterBuildPhase.create( - cluster, - "ExerciseScheduleReplacement", - "Each replacement signs an accepted duty epoch on Ethereum and Solana" - ).push( - verifyStep( - Actor.Sysio, - "replacement-groups-ready", - "both replacements are seated in a complete, fully active window", - async ctx => { - const replacements = Constants.RecoveryOperatorLabels.map( - label => ctx.keyStore.assertOperator(label).account - ) - await pollUntil( - "historical vacancies leave the activated window", - async () => { - const state = - await readScheduleRecoveryEpochState(ctx) - const active = await readActiveBatchOperatorAccounts(ctx) - assertCompleteSchedule(state.batch_op_groups) - const members = state.batch_op_groups.flat() - return ( - members.every(account => active.has(account)) && - replacements.every(account => members.includes(account)) - ) - }, - Constants.recoveryDeadlineMs(Constants.BatchOperatorGroups + 1), - Constants.PollIntervalMs - ) - }, - { - timeoutMs: Constants.recoveryDeadlineMs(Constants.BatchOperatorGroups + 1) - } - ), - // A late third signature is a valid no-op after quorum. Stop one original - // peer per replacement group so each replacement must be admitted. - // The groups may be shared or distinct; every group retains two signers. - ...Constants.RecoveryOperatorLabels.map(label => - ClusterBuildStep.create( - Actor.BatchOperator, - `stop-${label}-peer`, - `stop an original group peer so ${label} is required for quorum`, - {}, - { kind: "TerminationScenario.StopReplacementPeerInput", label }, - runStopReplacementPeer - ) - ), - verifyStep( - Actor.BatchOperator, - "replacement-duty-serves", - "each replacement is admitted as an active-group signer on Ethereum and Solana", - async ctx => { - const replacements = Constants.RecoveryOperatorLabels.map( - label => ctx.keyStore.assertOperator(label).account - ) - const dutyEpochs = new Map>() - const delivered = new Set() - await pollUntil( - "each replacement signs an accepted duty epoch on both outposts", - async () => { - const state = - await readScheduleRecoveryEpochState(ctx) - const current = - state.batch_op_groups[state.current_batch_op_group] ?? [] - const currentEpoch = Number(state.current_epoch_index) - for (const replacement of replacements) { - if (current.includes(replacement) && !delivered.has(replacement)) { - const epochs = dutyEpochs.get(replacement) ?? new Set() - epochs.add(currentEpoch) - dutyEpochs.set(replacement, epochs) - } - } - assertCompleteSchedule(state.batch_op_groups) - const [ethereumNextEpoch, solanaNextEpoch] = await Promise.all([ - readEthereumNextEpoch(ctx), - readSolanaNextEpoch(ctx) - ]) - for (const replacement of replacements) { - if (delivered.has(replacement)) continue - // A delivery arriving after quorum is a benign no-op. Keep - // later observed duties eligible instead of pinning the first. - for (const dutyEpoch of dutyEpochs.get(replacement) ?? []) { - if ( - ethereumNextEpoch >= dutyEpoch + 1 && - solanaNextEpoch >= dutyEpoch + 1 && - (await replacementDeliveredOnBothOutposts( - ctx, - replacement, - dutyEpoch - )) - ) { - delivered.add(replacement) - log.info( - `${replacement} signed accepted deliveries on both outposts for duty epoch ${dutyEpoch}` - ) - break - } - } - } - return delivered.size === replacements.length - }, - Constants.recoveryDeadlineMs(Constants.BatchOperatorGroups + 4), - Constants.PollIntervalMs - ) - }, - { - timeoutMs: Constants.recoveryDeadlineMs( - Constants.BatchOperatorGroups + 4 - ) - } - ) - ) } } diff --git a/packages/flow-batch-operator-termination/src/TerminationScenarioConstants.ts b/packages/flow-batch-operator-termination/src/TerminationScenarioConstants.ts index 459a40146..44cbdf789 100644 --- a/packages/flow-batch-operator-termination/src/TerminationScenarioConstants.ts +++ b/packages/flow-batch-operator-termination/src/TerminationScenarioConstants.ts @@ -30,22 +30,11 @@ export namespace TerminationScenarioConstants { /** Epoch duration (s) — the bare-cluster working baseline (`sysio.epoch::setconfig` floor is 60). */ export const EpochDurationSec = 60 /** - * Bootstrapped batch operators stood up by the harness. Eleven operators - * cover nine schedule seats plus two standing spares. The separately - * provisioned doomed operator never delivers, but its two group peers can - * still reach consensus majority. + * Bootstrapped batch operators stood up by the harness. 9 → 3 odd-sized + * groups of 3; with the doomed operator never delivering, the remaining 8 + * still cover consensus majority on every group. */ - export const BatchOperatorCount = 11 - /** Number of disjoint groups retained in the rolling schedule window. */ - export const BatchOperatorGroups = 3 - /** Operators per group; three keeps majority consensus unambiguous. */ - export const OperatorsPerEpoch = 3 - /** Number of seats in one full schedule window. */ - export const ScheduleSeatCount = BatchOperatorGroups * OperatorsPerEpoch - /** Spares remaining after the doomed operator terminates. */ - export const StandingSpareCount = BatchOperatorCount - ScheduleSeatCount - /** Published lookahead serves group one after a three-group window activates. */ - export const SuccessorGroupIndex = 1 + export const BatchOperatorCount = 9 /** * Override for `terminate_max_consecutive_misses` so `termcheck` fires inside * the flow's budget: 2 consecutive missed scheduled epochs flip TERMINATED. @@ -107,72 +96,29 @@ export namespace TerminationScenarioConstants { export const OperatorsQueryLimit = 100 /** Anchor account-namespace name of the SOL outpost's `OperatorRegistry` PDA account. */ export const SolanaOperatorRegistryAccountName = "operatorRegistry" - /** Anchor account namespace for the Solana outpost configuration PDA. */ - export const SolanaOutpostConfigAccountName = "outpostConfig" - - /** Healthy operator key used for read-only Solana account access. */ - export const RecoverySolanaReaderLabel = "batchop.b" - /** Harness labels for operators that repair two independent roster losses. */ - export const RecoveryOperatorLabels = ["recoverya", "recoveryb"] as const - /** Anvil HD slots beyond the bootstrapped roster and underwriters. */ - export const RecoveryOperatorEthereumHdIndices = [36, 37] as const - /** Ad-hoc daemons started for the two flow-provisioned replacements. */ - export const RecoveryAdHocDaemonCount = 2 - /** Epoch advances required while the announced duty remains frozen. */ - export const RecoveryHeldEpochAdvances = 2 /** Deadline for the ETH deposit to credit the depot balance row. */ export function ethereumDepositDeadlineMs(): number { - return ( - ProtocolTiming.effectiveEpochSec(EpochDurationSec) * - EthereumDepositRelayEpochs * - MsPerSecond - ) + return ProtocolTiming.effectiveEpochSec(EpochDurationSec) * EthereumDepositRelayEpochs * MsPerSecond } /** Deadline for the SOL deposit to land and the ACTIVE flip to follow. */ export function solanaActivationDeadlineMs(): number { - return ( - ProtocolTiming.effectiveEpochSec(EpochDurationSec) * - SolanaActivationEpochs * - MsPerSecond - ) + return ProtocolTiming.effectiveEpochSec(EpochDurationSec) * SolanaActivationEpochs * MsPerSecond } /** Deadline for the operator to appear in `epochstate.batch_op_groups`. */ export function scheduleWindowDeadlineMs(): number { - return ( - ProtocolTiming.effectiveEpochSec(EpochDurationSec) * - ScheduleWindowEpochs * - MsPerSecond - ) + return ProtocolTiming.effectiveEpochSec(EpochDurationSec) * ScheduleWindowEpochs * MsPerSecond } /** Deadline for the miss window to accumulate and `termcheck` to flip TERMINATED. */ export function terminationDeadlineMs(): number { - return ( - ProtocolTiming.effectiveEpochSec(EpochDurationSec) * - MissAccumulationEpochs * - MsPerSecond - ) + return ProtocolTiming.effectiveEpochSec(EpochDurationSec) * MissAccumulationEpochs * MsPerSecond } /** Deadline for the post-termination WITHDRAW_REMIT effects on either outpost. */ export function remitDeadlineMs(): number { - return ( - ProtocolTiming.effectiveEpochSec(EpochDurationSec) * - RemitPropagationEpochs * - MsPerSecond - ) - } - - /** Deadline for a recovery condition spanning the supplied epoch count. */ - export function recoveryDeadlineMs(epochCount: number): number { - return ( - ProtocolTiming.effectiveEpochSec(EpochDurationSec) * - epochCount * - MsPerSecond + - PollDeadlineBufferMs - ) + return ProtocolTiming.effectiveEpochSec(EpochDurationSec) * RemitPropagationEpochs * MsPerSecond } } diff --git a/packages/flow-batch-operator-termination/tests/ReplacementQuorum.test.ts b/packages/flow-batch-operator-termination/tests/ReplacementQuorum.test.ts deleted file mode 100644 index d52a1e124..000000000 --- a/packages/flow-batch-operator-termination/tests/ReplacementQuorum.test.ts +++ /dev/null @@ -1,67 +0,0 @@ -import { assertReplacementQuorumPeer } from "@wireio/test-flow-batch-operator-termination/ReplacementQuorum.js" - -const Replacements = ["replacementa", "replacementb"] -const OtherGroup = ["othera", "otherb", "otherc"] - -describe("assertReplacementQuorumPeer", () => { - test("selects the same original peer when both replacements share a group", () => { - const groups = [OtherGroup, ["replacementb", "original", "replacementa"]] - for (const replacement of Replacements) { - expect( - assertReplacementQuorumPeer(groups, Replacements, replacement) - ).toBe("original") - } - }) - - test("selects one original peer from each distinct replacement group", () => { - const groups = [ - ["replacementa", "originala", "originalb"], - ["replacementb", "originalc", "originald"] - ] - expect( - assertReplacementQuorumPeer(groups, Replacements, "replacementa") - ).toBe("originala") - expect( - assertReplacementQuorumPeer(groups, Replacements, "replacementb") - ).toBe("originalc") - }) - - test("refuses a missing replacement instead of stopping an unrelated peer", () => { - expect(() => - assertReplacementQuorumPeer([OtherGroup], Replacements, "replacementa") - ).toThrow("replacementa is not seated") - }) - - test("refuses a target outside the replacement set", () => { - expect(() => - assertReplacementQuorumPeer([OtherGroup], Replacements, "othera") - ).toThrow("unknown replacement") - }) - - test("refuses extra peers that would allow quorum without the replacement", () => { - expect(() => - assertReplacementQuorumPeer( - [["replacementa", "originala", "originalb", "originalc"]], - Replacements, - "replacementa" - ) - ).toThrow() - }) - - test("refuses duplicate seats and duplicate replacement identities", () => { - expect(() => - assertReplacementQuorumPeer( - [[...Replacements, "replacementa"]], - Replacements, - "replacementa" - ) - ).toThrow() - expect(() => - assertReplacementQuorumPeer( - [["replacementa", "originala", "originalb"]], - ["replacementa", "replacementa"], - "replacementa" - ) - ).toThrow() - }) -}) diff --git a/packages/flow-batch-operator-termination/tests/StandingSpare.test.ts b/packages/flow-batch-operator-termination/tests/StandingSpare.test.ts deleted file mode 100644 index 7a0add073..000000000 --- a/packages/flow-batch-operator-termination/tests/StandingSpare.test.ts +++ /dev/null @@ -1,50 +0,0 @@ -import { - assertCompleteSchedule, - assertStandingSpare -} from "@wireio/test-flow-batch-operator-termination/StandingSpare.js" - -const Groups = [ - ["a", "b", "c"], - ["d", "e", "f"], - ["g", "h", "i"] -] - -describe("assertStandingSpare", () => { - test("selects one of two active operators outside the nine-seat window", () => { - expect(assertStandingSpare(new Set([...Groups.flat(), "k", "j"]), Groups, 2)).toBe("j") - }) - - test("rejects an incomplete or inactive schedule instead of slashing a seat", () => { - expect(() => - assertStandingSpare(new Set([...Groups.flat(), "j"]), Groups, 2) - ).toThrow("standing-spare pool changed") - expect(() => - assertStandingSpare(new Set([...Groups.flat().slice(1), "j", "k"]), Groups, 2) - ).toThrow("inactive member") - }) - - test("rejects duplicate seats and an empty spare pool", () => { - expect(() => - assertStandingSpare(new Set([...Groups.flat(), "j", "k"]), - [Groups[0], Groups[1], ["g", "h", "h"]], 2) - ).toThrow() - expect(() => - assertStandingSpare(new Set(Groups.flat()), Groups, 0) - ).toThrow("no standing spare") - }) -}) - -describe("assertCompleteSchedule", () => { - test("accepts nine disjoint seats with two additional standing spares", () => { - expect(() => assertCompleteSchedule(Groups)).not.toThrow() - }) - - test("rejects a short group or duplicate seat", () => { - expect(() => assertCompleteSchedule([Groups[0], Groups[1], ["g", "h"]])).toThrow( - "short group" - ) - expect(() => assertCompleteSchedule([Groups[0], Groups[1], ["g", "h", "h"]])).toThrow( - "more than one group" - ) - }) -}) diff --git a/scripts/flow-heartbeat-monitor.mjs b/scripts/flow-heartbeat-monitor.mjs index 2c682fb74..695acbc5d 100755 --- a/scripts/flow-heartbeat-monitor.mjs +++ b/scripts/flow-heartbeat-monitor.mjs @@ -174,16 +174,14 @@ const EchoWrapperExcludePattern = `${TrxEchoExcludePattern}|signaled NACK|bad pa * learns via the NEXT envelope's OPERATORS attestation (1–2 epochs), and until * that dispatches the underwriter plugin's commit retries bounce off the * outpost's status gate: SOL opp-outpost `0x1795` (OperatorNotActive), ETH - * `OPP_NotActiveOperator`. Anvil prints the latter as its custom-error selector - * `0xabc01454`; a deliberately revoked batch member gets the same rejection. - * Forensically verified (2026-07-04, + * `OPP_NotActiveOperator`. Forensically verified (2026-07-04, * flow-swap-from-wire): the epoch-2 envelope carried ACTIVE and dispatched 24s * after the first bounce — the ~5s retry loop heals on the next attempt, so * bailing on growth here kills a healthy flow. The liveness probes (epoch * advance, opp delta, per-direction growth) remain the bail gates. */ const RegistrySyncLagPattern = - "custom program error: 0x1795|OPP_NotActiveOperator|execution reverted: custom error 0xabc01454" + "custom program error: 0x1795|OPP_NotActiveOperator" /** * Expected NEGATIVE-TEST reverts — a flow deliberately submits an invalid action From f3c5b71c41934453749dfd3fde72993f93c9870a Mon Sep 17 00:00:00 2001 From: Huang-Ming Huang Date: Wed, 23 Sep 2026 01:46:50 +0000 Subject: [PATCH 13/16] Fix collateral flow withdrawal eligibility Change-Id: I7046a9b4d14ba0cf3ce647ab295a1826684bfbe4 --- .../src/CollateralLifecycleScenario.ts | 45 ++++++++++++++++--- .../CollateralLifecycleScenarioConstants.ts | 9 ++-- 2 files changed, 45 insertions(+), 9 deletions(-) diff --git a/packages/flow-operator-collateral-deposit/src/CollateralLifecycleScenario.ts b/packages/flow-operator-collateral-deposit/src/CollateralLifecycleScenario.ts index 31b289b37..021a833b9 100644 --- a/packages/flow-operator-collateral-deposit/src/CollateralLifecycleScenario.ts +++ b/packages/flow-operator-collateral-deposit/src/CollateralLifecycleScenario.ts @@ -58,7 +58,7 @@ async function readWithdrawQueueRows( * schedule prefers non-bootstrapped operators, and its group must relay). * 3. **DepositEthereum** — bond on the ETH outpost → depot credits the balance row. * 4. **DepositSolana** — bond on the SOL outpost → all-chain rule met → ACTIVE. - * 5. **WithdrawRequest** — release half the ETH bond → depot queues it. + * 5. **WithdrawRequest** — release half the ETH bond; verify reserved collateral leaves the operator ACTIVE. * 6. **WaitAndFlush** — the wait window elapses; `flushwtdw` drains the queue. * 7. **ProcessRemit** — WITHDRAW_REMIT lands on the ETH outpost; escrow decrements. */ @@ -78,12 +78,12 @@ export class CollateralLifecycleScenario extends FlowScenario { { chainCode: Constants.EthereumChainCode, tokenCode: Constants.EthereumTokenCode, - minimumBond: Number(Constants.BondAmount) + minimumBond: Number(Constants.MinimumBond) }, { chainCode: Constants.SolanaChainCode, tokenCode: Constants.SolanaTokenCode, - minimumBond: Number(Constants.BondAmount) + minimumBond: Number(Constants.MinimumBond) } ] } @@ -210,11 +210,11 @@ export class CollateralLifecycleScenario extends FlowScenario { ) ) - // ── 5. Withdraw half the ETH bond → depot queues it ── + // ── 5. Withdraw excess ETH collateral while retaining relay eligibility ── ClusterBuildPhase.create( cluster, "WithdrawRequest", - "Release half the ETH bond; depot enqueues wtdwqueue" + "Release half the ETH bond; depot enqueues wtdwqueue and retains eligibility" ).push( EthereumCollateralTool.planWithdrawal( Actor.User, @@ -246,6 +246,41 @@ export class CollateralLifecycleScenario extends FlowScenario { ) }, stepOptions + ), + verifyStep( + Actor.Sysio, + "depot-status-active-after-withdraw", + "reserved withdrawal retains the minimum ETH collateral and ACTIVE status", + async ctx => { + const operator = await readDepositorRow(ctx), + requests = await readWithdrawQueueRows(ctx), + ethBalance = operator?.balances.find( + balance => + slugValue(balance.chain_code) === Constants.EthereumChainCode && + slugValue(balance.token_code) === Constants.EthereumTokenCode + ), + reservedAmount = requests + .filter( + request => + slugValue(request.chain_code) === Constants.EthereumChainCode && + slugValue(request.token_code) === Constants.EthereumTokenCode + ) + .reduce((sum, request) => sum + BigInt(request.amount), 0n) + if ( + ethBalance == null || + BigInt(ethBalance.balance) - reservedAmount < Constants.MinimumBond || + !matchesProtoEnum( + operator.status, + SysioOpregOperatorstatus, + SysioOpregOperatorstatus.OPERATOR_STATUS_ACTIVE + ) + ) { + throw new Error( + `Withdrawing excess ETH collateral must retain the minimum bond and ACTIVE status; balance=${ethBalance?.balance}, reserved=${reservedAmount}, status=${operator?.status}` + ) + } + }, + stepOptions ) ) diff --git a/packages/flow-operator-collateral-deposit/src/CollateralLifecycleScenarioConstants.ts b/packages/flow-operator-collateral-deposit/src/CollateralLifecycleScenarioConstants.ts index 43d257e3c..7010ae481 100644 --- a/packages/flow-operator-collateral-deposit/src/CollateralLifecycleScenarioConstants.ts +++ b/packages/flow-operator-collateral-deposit/src/CollateralLifecycleScenarioConstants.ts @@ -2,10 +2,9 @@ import { SlugName } from "@wireio/sdk-core" import { ProtocolTiming } from "@wireio/cluster-tool" /** - * Constants for the collateral-lifecycle flow. Amounts + epoch budgets carry - * over from the previously-validated flow run (2026-06): the bond is deposited + * Constants for the collateral-lifecycle flow. Twice the minimum bond is deposited * on BOTH outpost chains (all-chain collateral invariant), half the ETH bond is - * withdrawn mid-flow, and every poll deadline derives from extension-inclusive + * withdrawn while retaining the minimum. Every poll deadline uses extension-inclusive * epochs ({@link ProtocolTiming.effectiveEpochSec}) so the flow scales with the * epoch duration and survives extended epochs. */ @@ -29,9 +28,11 @@ export namespace CollateralLifecycleScenarioConstants { */ export const AdHocDaemonCount = 1 + /** Minimum collateral retained per chain to keep the depositor eligible to relay. */ + export const MinimumBond = 1_000_000n /** Collateral bonded per chain (raw outpost units — wei / lamports). */ export const BondAmount = 2_000_000n - /** ETH bond released mid-flow (half — stays above the minimum on the rest). */ + /** ETH bond released mid-flow (half — leaves exactly the required minimum). */ export const WithdrawAmount = 1_000_000n /** Escrow expected on the ETH outpost after the withdraw remit. */ export const ExpectedRemainingBalance = BondAmount - WithdrawAmount From 59e9cf6ff6a327790e6464a5e77b6854cf93b6b4 Mon Sep 17 00:00:00 2001 From: Huang-Ming Huang Date: Thu, 24 Sep 2026 15:47:21 +0000 Subject: [PATCH 14/16] Address review with post-termination rotation coverage Use the upstream bootstrap restored by Ethereum #205 and keep WIRE-385 tooling changes limited to operator lifecycle flow validation. Change-Id: I4012e2b59685961e9f6d1dddb03b652c015c32cb --- .../src/orchestration/ClusterBuildDefaults.ts | 207 ++++++------ .../ethereum/EthereumOutpostBootstrapper.ts | 295 ++++++++++++++++-- .../ethereum/EthereumOutpostSteps.ts | 289 ++++++++++++++--- ...ClusterBuildDefaultsEpochBootstrap.test.ts | 33 +- .../EthereumOutpostBootstrapper.test.ts | 196 ++++++++++-- .../ethereum/EthereumOutpostSteps.test.ts | 248 +++++++++++++-- .../src/TerminationScenario.ts | 160 +++++++++- 7 files changed, 1172 insertions(+), 256 deletions(-) diff --git a/packages/cluster-tool/src/orchestration/ClusterBuildDefaults.ts b/packages/cluster-tool/src/orchestration/ClusterBuildDefaults.ts index 805c62102..ed84ba154 100644 --- a/packages/cluster-tool/src/orchestration/ClusterBuildDefaults.ts +++ b/packages/cluster-tool/src/orchestration/ClusterBuildDefaults.ts @@ -796,13 +796,8 @@ export namespace ClusterBuildDefaults { ) ) - // ── outpost process bring-up / external materialization ── - // Local Ethereum contract deployment is intentionally deferred until after - // operator provisioning + schbatchgps. Its initializer must receive the - // depot's actual randomized-account schedule, or every legitimate epoch-1 - // signer is rejected as inactive. - // - // External mode instead verifies the already-running endpoints here. + // ── outpost deploys (own the run anvil + validator) — OR, in external mode, + // verify the already-running remote outpost endpoints instead ── if (isExternalOutpost) { ClusterBuildPhase.create( prerequisites, @@ -834,13 +829,25 @@ export namespace ClusterBuildDefaults { ClusterBuildPhase.create( prerequisites, "EthereumOutpost", - "Start the Ethereum outpost process" + "Deploy the Ethereum outpost" ).push( Steps.processes.anvil.planStart( Actor.EthereumOutpost, "start-anvil", "start the run-time anvil (instamine)", {} + ), + Steps.ethereumOutpost.planDeploy( + Actor.EthereumOutpost, + "deploy-ethereum", + "deploy + seed the Ethereum outpost", + { timeoutMs: 900_000 } + ), + Steps.processes.anvil.planEnableIntervalMining( + Actor.EthereumOutpost, + "enable-interval-mining", + "switch anvil to interval mining", + {} ) ) ClusterBuildPhase.create( @@ -863,93 +870,9 @@ export namespace ClusterBuildDefaults { ) } - // ═══ Cluster Operator Bootstrap — operators, schedule, outpost, nodes, epoch ═══ - const postContractDeployment = ClusterBuildPhaseGroup.create( - cluster, - "Cluster Operator Bootstrap", - "Provision operators, seed the outposts, start operator nodes, and bootstrap the first epoch" - ) - - // Bootstrapped batch operators + underwriters via the ONE mechanism. Fee-payer - // funding only — deposit flows provision their own non-bootstrapped ops with - // collateral funding on top. - const isSSM = config.signatureProvider.type === SignatureProviderType.SSM - WireOperatorProvisioningTool.planOperatorAccountProvisioning( - postContractDeployment, - "Create batchops & uws", - "Provision the bootstrapped batch operators + underwriters", - {}, - [ - ...batchOperators.map((label, index) => ({ - label, - type: OperatorType.BATCH, - ethereumHdIndex: Constants.batchOperatorEthereumHdIndex(index), - isBootstrapped: true, - // Fee-payer funding for the daemon's per-epoch deliveries on BOTH - // chains. ETH is SSM-only: under KEY the EM keys come off the anvil - // mnemonic and are prefunded, under SSM they come off a generated - // mnemonic anvil never funded. See BatchOperatorEthereumFundingWei. - airdropSolanaLamports: BatchOperatorAirdropLamports, - ...(isSSM ? { fundEthereumWei: BatchOperatorEthereumFundingWei } : {}) - })), - ...underwriters.map((label, index) => ({ - label, - type: OperatorType.UNDERWRITER, - // Use the filtered batch-operator list that owns the preceding HD - // indices; the raw config count can include entries this plan did - // not provision. - ethereumHdIndex: Constants.underwriterEthereumHdIndex( - batchOperators.length, - index - ), - isBootstrapped: false, - ...(isSSM ? { fundEthereumWei: BatchOperatorEthereumFundingWei } : {}) - })) - ] - ) - - // Materialize the one schedule both local outposts authorize for epoch 1. - // The generated WIRE account names only exist after provisioning above. - ClusterBuildPhase.create( - postContractDeployment, - "InitialBatchOperatorSchedule", - "Build the initial batch-operator schedule" - ).push( - Steps.contracts.sysio.epoch.planSchbatchgps( - Actor.Sysio, - "schedule-batch-groups", - "build the initial batch-operator schedule", - {} - ) - ) - - if (!isExternalOutpost) { - ClusterBuildPhase.create( - postContractDeployment, - "DeployEthereumOutpost", - "Deploy the Ethereum outpost with the depot's initial schedule" - ).push( - Steps.ethereumOutpost.planDeploy( - Actor.EthereumOutpost, - "deploy-ethereum", - "deploy + seed the Ethereum outpost with the initial operator schedule", - { timeoutMs: 900_000 } - ), - Steps.processes.anvil.planEnableIntervalMining( - Actor.EthereumOutpost, - "enable-interval-mining", - "switch anvil to interval mining", - {} - ) - ) - } - - // Registry token rows consume the local deployment artifacts, so this - // follows the schedule-seeded Ethereum deploy. It remains before the first - // epoch and before daemon artifact publication in both local and external - // modes. + // ── registry + optional mock reserves + underwriter config ── ClusterBuildPhase.create( - postContractDeployment, + prerequisites, "Registry", "Seed chains + tokens" ).push( @@ -967,14 +890,14 @@ export namespace ClusterBuildDefaults { // 0→1) could never seed them. if (config.enableMockReserves) { Steps.registry.planMockReserves( - postContractDeployment, + prerequisites, "MockReserves", "Seed the 8 mock (chain, token) PRIMARY reserves", {} ) } ClusterBuildPhase.create( - postContractDeployment, + prerequisites, "UnderwriterConfig", "Configure sysio.uwrit" ).push( @@ -994,7 +917,7 @@ export namespace ClusterBuildDefaults { ) ) ClusterBuildPhase.create( - postContractDeployment, + prerequisites, "ReserveConfig", "Configure sysio.reserv fee routing" ).push( @@ -1007,11 +930,16 @@ export namespace ClusterBuildDefaults { ) ) + // ═══ Cluster Post Contract Deployment — batch/uw operators, nodes, first epoch ═══ + const postContractDeployment = ClusterBuildPhaseGroup.create( + cluster, + "Cluster Post Contract Deployment", + "Provision batch operators + underwriters, start operator nodes, bootstrap the first epoch" + ) + // The operator daemons' shared prerequisites: the in-process OPP debugging // sink (external_debugging_plugin posts every envelope there) + the deploy // artifacts (ETH ABIs with addresses, SOL program id + IDL) their args reference. - // In local mode this must follow DeployEthereumOutpost so the ABI artifacts - // carry the addresses from the schedule-seeded deployment. ClusterBuildPhase.create( postContractDeployment, "OperatorDaemonPrerequisites", @@ -1047,6 +975,46 @@ export namespace ClusterBuildDefaults { ) ) + // Bootstrapped batch operators + underwriters via the ONE mechanism. Fee-payer + // funding only — deposit flows provision their own non-bootstrapped ops with + // collateral funding on top. + const isSSM = config.signatureProvider.type === SignatureProviderType.SSM + WireOperatorProvisioningTool.planOperatorAccountProvisioning( + postContractDeployment, + "Create batchops & uws", + "Provision the bootstrapped batch operators + underwriters", + {}, + [ + ...batchOperators.map((label, index) => ({ + label, + type: OperatorType.BATCH, + ethereumHdIndex: Constants.batchOperatorEthereumHdIndex(index), + isBootstrapped: true, + // Fee-payer funding for the daemon's per-epoch deliveries on BOTH + // chains. ETH is SSM-only: under KEY the EM keys come off the anvil + // mnemonic and are prefunded, under SSM they come off a generated + // mnemonic anvil never funded. See BatchOperatorEthereumFundingWei. + airdropSolanaLamports: BatchOperatorAirdropLamports, + ...(isSSM ? { fundEthereumWei: BatchOperatorEthereumFundingWei } : {}) + })), + ...underwriters.map((label, index) => ({ + label, + type: OperatorType.UNDERWRITER, + // The offset is the length of the PREFIX-FILTERED batch-operator + // array, not `config.batchOperatorCount`: only a list already + // narrowed to `batchop.` entries guarantees the ordinal the HD-index + // rule is defined against. `index` is that same filtered array's + // iterator index for underwriters. + ethereumHdIndex: Constants.underwriterEthereumHdIndex( + batchOperators.length, + index + ), + isBootstrapped: false, + ...(isSSM ? { fundEthereumWei: BatchOperatorEthereumFundingWei } : {}) + })) + ] + ) + // SSM mode: publish the just-provisioned operator keys BEFORE the operator // daemons start — their wire/ethereum/solana `--signature-provider ...SSM:` // specs fetch the private keys from SSM at nodeop startup. @@ -1097,21 +1065,44 @@ export namespace ClusterBuildDefaults { // scenarios that need a real restart. // ── first epoch ── - // Step ORDER is load-bearing: both local outposts were seeded from the - // schedule materialized above, and Solana's transient roster seed must land - // before `msgch::bootstrap` delivers the first envelope. + // Step ORDER is load-bearing: both roster seeds read the schedule + // `schbatchgps` just materialized, and must land before `msgch::bootstrap` + // delivers the first envelope. const epochBootstrap = ClusterBuildPhase.create( postContractDeployment, "EpochBootstrap", - "Bootstrap epoch 0 → 1" + "Schedule groups + bootstrap epoch 0 → 1" + ).push( + Steps.contracts.sysio.epoch.planSchbatchgps( + Actor.Sysio, + "schedule-batch-groups", + "build the initial batch-operator schedule", + {} + ) ) - // SOL-376: seed the LOCAL Solana outpost's operator registry with the - // depot's epoch-1 batch-operator group (just materialized by schbatchgps) - // BEFORE the first envelope is delivered — `epoch_in` refuses to finalize - // until `opp_bootstrap` runs. External outposts are seeded by their own - // operators, out of band. + // Seed BOTH local outposts with the depot's schedule (just materialized by + // schbatchgps) BEFORE the first envelope is delivered. External outposts + // are seeded by their own operators, out of band. + // + // Ethereum: `OPPInbound.initialize` ran in Cluster Prerequisites with a + // PROVISIONAL roster — the depot's window did not exist yet, and its + // membership is ordered by account names generated later — while under + // WNE-27 epoch 1 is deliverable only by the group mapped to it. + // `installInitialRoster` (the SOL-376 shape) replaces the provisional + // roster with the depot's window; it refuses once an epoch-1 delivery has + // been counted, which is why it sits before `bootstrap-epoch`. + // + // Solana (SOL-376): the outpost's operator registry starts empty and + // `epoch_in` refuses to finalize until `opp_bootstrap` seeds the depot's + // epoch-1 group. if (!isExternalOutpost) epochBootstrap.push( + Steps.ethereumOutpost.planOppBootstrap( + Actor.EthereumOutpost, + "seed-ethereum-roster", + "seed the Ethereum outpost batch-operator roster from the depot schedule (installInitialRoster)", + {} + ), Steps.solanaOutpost.planOppBootstrap( Actor.SolanaOutpost, "seed-solana-roster", diff --git a/packages/cluster-tool/src/orchestration/ethereum/EthereumOutpostBootstrapper.ts b/packages/cluster-tool/src/orchestration/ethereum/EthereumOutpostBootstrapper.ts index 9cddb88f6..edca6e2dd 100644 --- a/packages/cluster-tool/src/orchestration/ethereum/EthereumOutpostBootstrapper.ts +++ b/packages/cluster-tool/src/orchestration/ethereum/EthereumOutpostBootstrapper.ts @@ -6,6 +6,7 @@ import { promisify } from "node:util" import { ethers } from "ethers" import { defaults, range } from "lodash" import Assert from "node:assert" +import { ChainKind } from "@wireio/opp-typescript-models" import { AnvilProcess } from "../../cluster/processes/AnvilProcess.js" import { getLogger } from "../../logging/Logger.js" import { StepExtraRecorder } from "../../report/tools/StepExtraRecorder.js" @@ -15,6 +16,7 @@ import { withFileLock } from "../../utils/fsUtils.js" import { scaleTimeoutMs } from "../../utils/asyncUtils.js" +import { EvmAddressPattern, loadOutpostContract } from "../../utils/ethereumUtils.js" const log = getLogger(__filename) const execFileAsync = promisify(execFile) @@ -28,6 +30,29 @@ export interface EthereumAccount { usedFor: string } +/** + * A batch-operator roster for `OPPInbound` — what `initialize` installs at + * construction (WNE-41) and what `installInitialRoster` replaces it with once + * the depot's schedule exists (SOL-376 shape, see + * {@link EthereumOutpostBootstrapper.oppBootstrap}). + * + * `OPPInbound.isActiveOperator` is FAIL-CLOSED: an uninitialized roster + * authorizes nobody, and on a WIRE cluster the addresses that send `epochIn` + * are the batch-operator DAEMONS' own EOAs — not the deployer. Without this the + * first envelope is refused and the epoch never advances. + */ +export interface EthereumOutpostInitialRoster { + /** + * Batch-operator ETH addresses, one group per depot window slot in slot + * order: `groups[k]` serves epoch `1 + k` and its length is that epoch's + * consensus threshold (WNE-27). The construction-time roster can only + * reproduce the depot's SHAPE; the seeded one IS the depot's window. + */ + groups: string[][] + /** The depot's global `epoch_duration_sec`; must be positive. */ + epochDurationSec: number +} + /** Caller options for {@link EthereumOutpostBootstrapper}. */ export interface EthereumOutpostBootstrapperOptions { /** Path to the `wire-ethereum` repo root. */ @@ -44,12 +69,12 @@ export interface EthereumOutpostBootstrapperOptions { * sharing `/.local/deployments/` wiped each other mid-run). */ deploymentsPath: string - /** Depot schedule groups mapped to their Ethereum signing addresses. */ - initialOperatorGroups: string[][] - /** Active group cursor in the depot schedule at deployment time. */ - initialActiveGroupIndex: number - /** Depot epoch duration, used by the outpost's initial roster window. */ - epochDurationSec: number + /** + * WNE-41 initial batch-operator roster for `OPPInbound.initialize`. Required: + * a cluster deployed without one has an outpost whose `epochIn` is callable + * by nobody, and `initialize` is one-shot. + */ + initialRoster: EthereumOutpostInitialRoster /** * Number of deterministic accounts to generate — MUST match the run anvil's * `--accounts` (default: {@link AnvilProcess.AccountCount}) so every generated @@ -94,20 +119,16 @@ export class EthereumOutpostBootstrapper { options.deploymentsPath, "EthereumOutpostBootstrapper: deploymentsPath is required" ) + // WNE-41 — fail HERE, before anvil is touched. An empty or zero-duration + // roster is refused by `OPPInbound` itself (`OPP_InvalidInitialRoster`), + // and at construction that revert surfaces as a failed deploy mid-run. Assert.ok( - options.initialOperatorGroups?.length > 0 && - options.initialOperatorGroups.every(group => group.length > 0), - "EthereumOutpostBootstrapper: initialOperatorGroups must contain non-empty groups" - ) - Assert.ok( - Number.isInteger(options.initialActiveGroupIndex) && - options.initialActiveGroupIndex >= 0 && - options.initialActiveGroupIndex < options.initialOperatorGroups.length, - "EthereumOutpostBootstrapper: initialActiveGroupIndex is out of range" + options.initialRoster?.groups?.some(group => group.length > 0), + "EthereumOutpostBootstrapper: initialRoster needs at least one batch-operator address" ) Assert.ok( - Number.isInteger(options.epochDurationSec) && options.epochDurationSec > 0, - "EthereumOutpostBootstrapper: epochDurationSec must be positive" + options.initialRoster.epochDurationSec > 0, + "EthereumOutpostBootstrapper: initialRoster.epochDurationSec must be positive" ) this.config = defaults( { ...options }, @@ -195,16 +216,25 @@ export class EthereumOutpostBootstrapper { rewardCooldown: 100, withdrawalDelay: 50 } - const outpostConfig = { - url: rpcUrl, - key: deployerPrivateKey, - addressFile: Path.join(localDir, "outpost-addrs.json"), - gasLimitFile: Path.join(localDir, "outpost-gas-limits.json"), - useMockAggregator: true, - initialOperatorGroups: this.config.initialOperatorGroups, - initialActiveGroupIndex: this.config.initialActiveGroupIndex, - epochDurationSec: this.config.epochDurationSec - } + const { initialRoster } = this.config, + outpostConfig = { + url: rpcUrl, + key: deployerPrivateKey, + addressFile: Path.join(localDir, "outpost-addrs.json"), + gasLimitFile: Path.join(localDir, "outpost-gas-limits.json"), + useMockAggregator: true, + // WNE-41: consumed by `deployLocal.ts`'s OutpostLocalDeploy, which + // hands them to `OPPInbound.initialize`. The deployer is deliberately + // NOT among them — on a cluster the batch-operator daemons sign + // `epochIn` with their own keys. + initialOperatorGroups: initialRoster.groups, + epochDurationSec: initialRoster.epochDurationSec + } + log.info( + `[ethereum] initial batch-operator roster: ${initialRoster.groups + .map(group => `[${group.join(", ")}]`) + .join(" ")} (epochDurationSec=${initialRoster.epochDurationSec})` + ) Fs.writeFileSync( Path.join(localDir, "liqeth.json"), JSON.stringify(liqEthConfig, null, 2) @@ -398,9 +428,220 @@ export class EthereumOutpostBootstrapper { } log.info("[ethereum] seedReserveManager complete") } + + /** + * Seed the ETH outpost's batch-operator roster from the depot's REAL schedule + * via `OPPInbound.installInitialRoster` — the SOL-376 `opp_bootstrap` shape. + * + * `initialize` ran in Cluster Prerequisites, before `sysio.epoch::schbatchgps` + * existed, so the roster it seated could only reproduce the depot's shape, + * not its membership; under WNE-27 epoch 1 is deliverable solely by the + * group mapped to it, so that provisional roster leaves the outpost + * undeliverable whenever the depot's name-ordered schedule seats a different + * operator. This call must therefore land AFTER `schbatchgps` and BEFORE the + * depot's first envelope — `installInitialRoster` refuses once an epoch-1 + * delivery has been counted (`OPP_BootstrapWindowClosed`). The seed is + * transient: the depot's first `BATCH_OPERATOR_GROUPS` attestation replaces + * it under consensus. + * + * Routed through `OutpostManager.execute` as the deployer (anvil HD index 0, + * the manager's post-handoff admin), exactly like `deployLocal.ts` wires the + * other `restricted` setters. Every window slot is installed, so slot `k` + * serves epoch `1 + k` as a delivered window would. + * + * @param seed - the depot's window as EVM addresses + the slot serving epoch 1. + */ + async oppBootstrap(seed: EthereumOutpostBootstrapper.OppBootstrapSeed): Promise { + const { ethereumPath, deploymentsPath, rpcUrl } = this.config, + initialGroups = EthereumOutpostBootstrapper.initialBatchOperatorGroups(seed), + outpostAddressesFile = Path.join( + deploymentsPath, + EthereumOutpostBootstrapper.OutpostAddressesFile + ) + Assert.ok( + Fs.existsSync(outpostAddressesFile), + `oppBootstrap: ${outpostAddressesFile} is missing — the Ethereum outpost deploy must precede the roster seed` + ) + const outpostAddresses: Record = JSON.parse( + Fs.readFileSync(outpostAddressesFile, "utf-8") + ), + provider = new ethers.JsonRpcProvider(rpcUrl), + // The deployer is `deployLocal.ts`'s `owner`: the OutpostManager admin + // after handoff, and the ONE signer allowed through `manager.execute`. + deployer = new ethers.Wallet( + EthereumOutpostBootstrapper.generateAccounts( + EthereumOutpostBootstrapper.DeployerAccountIndex + 1 + )[EthereumOutpostBootstrapper.DeployerAccountIndex].privateKey, + provider + ), + manager = loadOutpostContract( + ethereumPath, + outpostAddresses, + EthereumOutpostBootstrapper.OutpostManagerContractName, + [...EthereumOutpostBootstrapper.OutpostArtifactSubpath], + deployer + ), + oppInbound = loadOutpostContract( + ethereumPath, + outpostAddresses, + EthereumOutpostBootstrapper.OppInboundContractName, + [...EthereumOutpostBootstrapper.OutpostArtifactSubpath], + deployer + ), + oppInboundAddress = await oppInbound.getAddress(), + activeGroup = seed.window.groups[seed.activeGroupIndex] + + log.info( + `[ethereum] installInitialRoster: seeding ${seed.window.groups.length} window slot(s), ` + + `epoch-1 slot ${seed.activeGroupIndex} = [${activeGroup.join(", ")}], ` + + `epoch_duration=${seed.window.epochDurationSec}s (deployer=${deployer.address})` + ) + try { + const transaction = await manager.execute( + oppInboundAddress, + oppInbound.interface.encodeFunctionData( + EthereumOutpostBootstrapper.InstallInitialRosterFunction, + [initialGroups] + ) + ) + await transaction.wait() + + // Read the seat back: the outpost must now authorize the depot's epoch-1 + // operators and nobody from the provisional roster it replaced. + const seated = await Promise.all( + activeGroup.map(member => oppInbound.isActiveOperator(member)) + ) + Assert.ok( + seated.every(Boolean), + `oppBootstrap: OPPInbound does not authorize every epoch-1 member after installInitialRoster ` + + `([${activeGroup.join(", ")}] → [${seated.join(", ")}])` + ) + } finally { + provider.destroy() + } + log.info("[ethereum] installInitialRoster: ETH outpost roster seeded on the depot's schedule") + } } export namespace EthereumOutpostBootstrapper { + /** + * SOL-376 seed for `OPPInbound.installInitialRoster`: the depot's whole + * schedule window as EVM addresses plus the slot the depot serves epoch 1 + * from (`epochstate.current_batch_op_group`). + */ + export interface OppBootstrapSeed { + /** Every window group in slot order, with the depot's epoch duration. */ + window: EthereumOutpostInitialRoster + /** Index of the window group that serves epoch 1. */ + activeGroupIndex: number + } + + /** One `ChainAddress` as `OPPInbound.installInitialRoster` takes it. */ + export interface InitialChainAddress { + kind: ChainKind + address_: string + } + + /** One `BatchOperatorGroup` of the roster tuple. */ + export interface InitialBatchOperatorGroup { + operators: InitialChainAddress[] + } + + /** The `BatchOperatorGroups` tuple `OPPInbound.installInitialRoster` takes. */ + export interface InitialBatchOperatorGroups { + activeGroupIndex: number + epochIndex: number + groups: InitialBatchOperatorGroup[] + epochDurationSec: number + } + + /** The `OutpostManager` surface the seed drives — the post-handoff admin path. */ + export interface OutpostManagerExecuteView { + execute(target: string, data: string): Promise + } + + /** The `OPPInbound` surface the seed reads back through. */ + export interface OppInboundRosterView { + isActiveOperator(operator: string): Promise + } + + /** `deployLocal.ts`'s outpost address map, under the cluster's deployments dir. */ + export const OutpostAddressesFile = "outpost-addrs.json" + /** Artifact dir segments under `/artifacts/contracts` for the OPP contracts. */ + export const OutpostArtifactSubpath = ["outpost"] as const + /** `outpost-addrs.json` key + artifact basename of the manager. */ + export const OutpostManagerContractName = "OutpostManager" + /** `outpost-addrs.json` key + artifact basename of the inbound endpoint. */ + export const OppInboundContractName = "OPPInbound" + /** The roster seed entry point on `OPPInbound`. */ + export const InstallInitialRosterFunction = "installInitialRoster" + /** + * `epochIndex` the seeded window is anchored at: the active slot serves + * epoch `0 + 1`, the first inbound epoch. The contract pins the anchor + * itself; the tuple carries it for shape. + */ + export const InitialRosterAnchorEpochIndex = 0 + + /** + * Build the `BatchOperatorGroups` tuple for `installInitialRoster`, + * validating here what `OPPInbound._installInitialRoster` validates on-chain + * so a bad seed fails with a readable message instead of an + * `OPP_InvalidInitialRoster` revert: at least one group, no empty group, only + * well-formed non-zero EVM addresses, no repeat WITHIN a group (a repeat + * inflates that group's consensus threshold past the operators able to + * deliver; across groups it is one operator serving consecutive epochs), a + * positive epoch duration, and an in-range active slot. + * + * @param seed - the window + the slot serving epoch 1. + * @return the tuple, ready to ABI-encode as the call's one argument. + * @throws on any of the rejections above. + */ + export function initialBatchOperatorGroups(seed: OppBootstrapSeed): InitialBatchOperatorGroups { + const { window, activeGroupIndex } = seed, + { groups, epochDurationSec } = window + Assert.ok(groups.length > 0, "initial roster: at least one batch-operator group is required") + Assert.ok( + Number.isSafeInteger(epochDurationSec) && epochDurationSec > 0, + `initial roster: epochDurationSec must be a positive integer (got ${epochDurationSec})` + ) + Assert.ok( + Number.isSafeInteger(activeGroupIndex) && activeGroupIndex >= 0 && activeGroupIndex < groups.length, + `initial roster: activeGroupIndex ${activeGroupIndex} is out of range for ${groups.length} group(s)` + ) + return { + activeGroupIndex, + epochIndex: InitialRosterAnchorEpochIndex, + groups: groups.map((members, groupIndex) => { + Assert.ok( + members.length > 0, + `initial roster: group ${groupIndex} is empty — an empty active group leaves epoch 1 undeliverable by anyone` + ) + const seen = new Set() + return { + operators: members.map(member => { + Assert.ok( + EvmAddressPattern.test(member), + `initial roster: group ${groupIndex} member '${member}' is not an EVM address` + ) + const normalized = ethers.getAddress(member) + Assert.ok( + normalized !== ethers.ZeroAddress, + `initial roster: group ${groupIndex} contains the zero address` + ) + Assert.ok( + !seen.has(normalized), + `initial roster: ${normalized} appears twice in group ${groupIndex} — a duplicate inflates ` + + `the consensus threshold beyond the operators able to meet it` + ) + seen.add(normalized) + return { kind: ChainKind.EVM, address_: normalized } + }) + } + }), + epochDurationSec + } + } + /** Annotated accounts filename written under the anvil data path. */ export const AccountsFile = "accounts.json" /** Anvil's default deterministic mnemonic. */ diff --git a/packages/cluster-tool/src/orchestration/ethereum/EthereumOutpostSteps.ts b/packages/cluster-tool/src/orchestration/ethereum/EthereumOutpostSteps.ts index 22728f372..8be3af55d 100644 --- a/packages/cluster-tool/src/orchestration/ethereum/EthereumOutpostSteps.ts +++ b/packages/cluster-tool/src/orchestration/ethereum/EthereumOutpostSteps.ts @@ -1,23 +1,154 @@ import Assert from "node:assert" import Path from "node:path" +import { range } from "lodash" +import { KeyType } from "@wireio/sdk-core" import { OperatorType } from "@wireio/opp-typescript-models" +import { Constants } from "../../Constants.js" +import { KeyGenerator } from "../../clients/wire/KeyGenerator.js" +import { BatchOperatorSchedule } from "../../config/BatchOperatorSchedule.js" import { Report } from "../../report/Report.js" +import { mapSeries } from "../../utils/asyncUtils.js" import { toDialAddress, toURL } from "../../utils/netUtils.js" import { ClusterBuildContext } from "../ClusterBuildContext.js" import { ClusterBuildStep, type ClusterBuildStepOptions } from "../ClusterBuildStep.js" -import { EthereumOutpostBootstrapper } from "./EthereumOutpostBootstrapper.js" -import { ClusterConfigProvider } from "../../config/ClusterConfigProvider.js" -import { OperatorAccount } from "../outputs/OperatorAccount.js" +import type { OperatorAccount } from "../outputs/OperatorAccount.js" import { EpochContractSteps } from "../steps/contracts/sysio/EpochContractSteps.js" +import { KeySteps } from "../steps/KeySteps.js" +import { + EthereumOutpostBootstrapper, + type EthereumOutpostBootstrapperOptions, + type EthereumOutpostInitialRoster +} from "./EthereumOutpostBootstrapper.js" +import { ClusterConfigProvider } from "../../config/ClusterConfigProvider.js" /** Steps that deploy + seed the Ethereum (anvil) outpost. */ export namespace EthereumOutpostSteps { /** Subpath (under the cluster data dir) for the annotated accounts file. */ const AnvilDataSubpath = "anvil" + /** + * Derive the PROVISIONAL batch-operator roster the outpost is constructed with + * (WNE-41: `OPPInbound.initialize` refuses to seat an empty one) — the ETH + * addresses of every bootstrapped batch operator, grouped in the depot's + * SHAPE. The depot's real membership is installed later by + * {@link planOppBootstrap} (`seed-ethereum-roster`), once + * `sysio.epoch::schbatchgps` has produced it. + * + * The addresses are DERIVED rather than read off provisioned operators + * because the outpost deploys in **Cluster Prerequisites**, while the batch + * operators are provisioned later in **Cluster Post Contract Deployment**. + * The (mnemonic, HD index) pair is fully determined by then, and both sides + * read it from the SAME two authorities — {@link KeySteps.ethereumMnemonic} + * and {@link Constants.batchOperatorEthereumHdIndex} — so the provisional + * roster names exactly the keys the daemons later sign `epochIn` with. + * + * Group `k` serves epoch `1 + k` and its length is that epoch's consensus + * threshold (WNE-27), so the grouping is sized by the depot's own + * `operators_per_epoch` / `batch_op_groups` from + * {@link BatchOperatorSchedule.resolve}. What it CANNOT reproduce is which + * operator lands in which group: the depot orders by on-chain account name, + * generated by `roa::newuser` from a nonce and the block number — which is + * why the seed step exists. + * + * @param ctx - The build context. + * @returns The provisional roster for `OPPInbound.initialize`. + */ + export async function resolveInitialRoster( + ctx: C + ): Promise { + const { config } = ctx, + { operatorsPerEpoch, batchOpGroups, batchOperatorMinimumActive } = + BatchOperatorSchedule.resolve(config), + keyContext = KeyGenerator.context( + config.executables.clio, + config.buildPath, + KeySteps.ethereumMnemonic(ctx) + ), + // Walk the BATCH-OPERATOR LABEL list — an array already narrowed to + // `batchop.` entries — and take that filtered array's iterator index. + // The provisioning side (`ClusterBuildDefaults`) assigns each operator's + // HD index the same way, so the roster authorizes exactly the addresses + // the daemons later sign `epochIn` with. Indexing a bare + // `range(batchOperatorCount)` would agree only by coincidence of + // ordering; neither the position nor the sort order of an operator in a + // persisted `cluster-keys.json` is guaranteed, so the ordinal is only + // meaningful once the list has been filtered by label prefix. + batchOperatorLabels = range(config.batchOperatorCount).map(index => + Constants.batchOperatorLabel(index) + ), + addresses = await mapSeries( + batchOperatorLabels, + async (label, index) => + ( + await KeyGenerator.create(KeyType.EM, keyContext, { + ethereumHdIndex: Constants.batchOperatorEthereumHdIndex(index), + purpose: `ethereum-outpost initial batch-operator roster (${label})` + }) + ).address + ) + return { + groups: partitionLikeDepot( + addresses, + operatorsPerEpoch, + batchOpGroups, + batchOperatorMinimumActive + ), + epochDurationSec: config.epochDurationSec + } + } + + /** + * Partition roster addresses into the SHAPE of the window + * `sysio.epoch::schbatchgps` builds: TRIM the pool to + * `batch_operator_minimum_active`, EVEN/ODD INTERLEAVE it, then partition + * into `batch_op_groups` groups of `operators_per_epoch`. + * + * Every group's length is the consensus threshold of the epoch its slot + * serves (WNE-27), so the sizes must be the depot's. Operators past the + * scheduled window are left OUT, exactly as the depot leaves them ungrouped: + * a slot they rode would misrepresent the window and authorize them for an + * epoch the depot never scheduled them for. Membership within the window is + * label order here and account-name order on the depot — the seed step + * ({@link planOppBootstrap}) installs the depot's own before the first + * envelope. + * + * @param addresses - Roster addresses in batch-operator label order. + * @param operatorsPerEpoch - Depot `operators_per_epoch` (group SIZE). + * @param batchOpGroups - Depot `batch_op_groups` (group COUNT). + * @param batchOperatorMinimumActive - Depot `batch_operator_minimum_active`. + * @returns At most `batchOpGroups` non-empty groups. + */ + export function partitionLikeDepot( + addresses: string[], + operatorsPerEpoch: number, + batchOpGroups: number, + batchOperatorMinimumActive: number + ): string[][] { + // TRIM — the depot schedules only `batch_operator_minimum_active` of the + // ACTIVE pool; anything past it stays ACTIVE but ungrouped there. + const scheduled = addresses.slice(0, batchOperatorMinimumActive), + // INTERLEAVE — evens then odds, exactly as `schbatchgps` shuffles. + interleaved = [ + ...scheduled.filter((_address, index) => index % 2 === 0), + ...scheduled.filter((_address, index) => index % 2 === 1) + ] + return ( + range(batchOpGroups) + .map(group => + interleaved.slice( + group * operatorsPerEpoch, + group * operatorsPerEpoch + operatorsPerEpoch + ) + ) + // The contract rejects an EMPTY group, so a short pool yields fewer + // groups rather than padded ones. + .filter(group => group.length > 0) + ) + } + /** * Deploy the Ethereum outpost against the already-running run anvil * (`Steps.processes.anvil.start` must precede this in the phase): deploy the @@ -43,6 +174,29 @@ export namespace EthereumOutpostSteps { ) } + /** + * The bootstrapper options every Ethereum-outpost runner constructs from the + * context — paths, the run anvil's endpoint (same derivation as + * `AnvilProcess.rpcUrl`: the anvil was bound to this exact port by + * `Steps.processes.anvil.start`, so they cannot diverge), and the roster the + * call installs. + */ + function bootstrapperOptions( + ctx: C, + initialRoster: EthereumOutpostInitialRoster + ): EthereumOutpostBootstrapperOptions { + return { + ethereumPath: ctx.config.ethereumPath, + anvilDataPath: Path.join(ctx.config.dataPath, AnvilDataSubpath), + rpcUrl: toURL( + ctx.config.bind.anvil.port, + toDialAddress(ctx.config.bind.anvil.address) + ), + deploymentsPath: ClusterConfigProvider.ethereumDeploymentsPath(ctx.config), + initialRoster + } + } + /** Named runner — `EthereumOutpostBootstrapper.bootstrap` against the run anvil. */ export async function runDeploy( ctx: C, @@ -50,63 +204,98 @@ export namespace EthereumOutpostSteps { signal: AbortSignal ): Promise { signal.throwIfAborted() + await new EthereumOutpostBootstrapper( + bootstrapperOptions(ctx, await resolveInitialRoster(ctx)) + ).bootstrap() + } + + /** + * Seed the Ethereum outpost's batch-operator roster from the depot's REAL + * schedule via `OPPInbound.installInitialRoster` — the counterpart of + * `Steps.solanaOutpost.planOppBootstrap` (SOL-376). One write, one step. + * + * Runs in `EpochBootstrap` AFTER `schedule-batch-groups` (the schedule must + * exist to be read) and BEFORE `bootstrap-epoch` (the contract closes the + * bootstrap window on the first counted epoch-1 delivery). Input-less — the + * window is read off `sysio.epoch::epochstate` and the operators off + * `ctx.keyStore` at run time. + */ + export function planOppBootstrap( + actor: Report.Actor, + name: string, + description: string, + options: ClusterBuildStepOptions + ): ClusterBuildStep { + return ClusterBuildStep.create(actor, name, description, options, null, runOppBootstrap) + } + + /** Named runner — `EthereumOutpostBootstrapper.oppBootstrap`. */ + export async function runOppBootstrap( + ctx: C, + _input: null, + signal: AbortSignal + ): Promise { + signal.throwIfAborted() + const epochState = await EpochContractSteps.readEpochState(ctx) Assert.ok( epochState?.batch_op_groups?.length > 0, - "runDeploy: initial batch-operator schedule is empty" + "runOppBootstrap: the depot has no batch-operator schedule yet — schbatchgps must precede the roster seed" ) - const initialOperatorGroups = resolveInitialOperatorGroups( + const seed = resolveOppBootstrapSeed( ctx.keyStore.operatorsByType(OperatorType.BATCH), - epochState.batch_op_groups + epochState.batch_op_groups, + epochState.current_batch_op_group, + ctx.config.epochDurationSec ) - // Same derivation as AnvilProcess.rpcUrl — the run anvil was bound to this - // exact port by Steps.processes.anvil.start, so they cannot diverge. - await new EthereumOutpostBootstrapper({ - ethereumPath: ctx.config.ethereumPath, - anvilDataPath: Path.join(ctx.config.dataPath, AnvilDataSubpath), - rpcUrl: toURL( - ctx.config.bind.anvil.port, - toDialAddress(ctx.config.bind.anvil.address) - ), - deploymentsPath: ClusterConfigProvider.ethereumDeploymentsPath(ctx.config), - initialOperatorGroups, - initialActiveGroupIndex: epochState.current_batch_op_group, - epochDurationSec: ctx.config.epochDurationSec - }).bootstrap() + await new EthereumOutpostBootstrapper(bootstrapperOptions(ctx, seed.window)).oppBootstrap(seed) } /** - * Map the depot's materialized schedule to the exact Ethereum keys its - * operator daemons use. Account names are generated during provisioning, so - * this mapping must be resolved from the live key store after - * `schbatchgps`; deriving it from labels or HD indexes can authorize the - * wrong first-epoch signers. + * Build the `installInitialRoster` seed from the provisioned batch operators + * and the depot's schedule window: every window group, in slot order, mapped + * to its members' EVM addresses — the addresses their daemons sign `epochIn` + * with — plus the slot the depot serves epoch 1 from. Mirrors + * `SolanaOutpostSteps.resolveOppBootstrapSeed`, minus Solana's single-group + * packet cap: `installInitialRoster` takes the whole window, so slot `k` + * serves epoch `1 + k` exactly as a delivered `BATCH_OPERATOR_GROUPS` would. + * + * @param batchOperators - every provisioned batch operator (from `ctx.keyStore`). + * @param window - the depot's `epochstate.batch_op_groups` (account names per slot). + * @param activeGroupIndex - the depot's `epochstate.current_batch_op_group`. + * @param epochDurationSec - the depot's epoch duration. + * @return the seed for {@link EthereumOutpostBootstrapper.oppBootstrap}. + * @throws if a scheduled account is not among the provisioned batch operators + * or its operator lacks an Ethereum key. */ - export function resolveInitialOperatorGroups( + export function resolveOppBootstrapSeed( batchOperators: OperatorAccount[], - scheduleGroups: string[][] - ): string[][] { - Assert.ok(scheduleGroups.length > 0, "resolveInitialOperatorGroups: schedule is empty") - const operatorByAccount = new Map( - batchOperators.map(operator => [operator.account, operator]) - ) - return scheduleGroups.map((group, groupIndex) => { - Assert.ok( - group.length > 0, - `resolveInitialOperatorGroups: schedule group ${groupIndex} is empty` - ) - return group.map(accountName => { - const operator = operatorByAccount.get(accountName) - Assert.ok( - operator, - `resolveInitialOperatorGroups: schedule member ${accountName} not found among provisioned batch operators` - ) - Assert.ok( - operator.ethereum?.address, - `resolveInitialOperatorGroups: schedule member ${accountName} has no Ethereum address` - ) - return operator.ethereum.address - }) - }) + window: string[][], + activeGroupIndex: number, + epochDurationSec: number + ): EthereumOutpostBootstrapper.OppBootstrapSeed { + const operatorByAccount = new Map(batchOperators.map(operator => [operator.account, operator])) + return { + window: { + groups: window.map(accountNames => + accountNames.map(accountName => { + const operator = operatorByAccount.get(accountName) + Assert.ok( + operator, + `resolveOppBootstrapSeed: scheduled batch operator ${accountName} not found among provisioned batch operators` + ) + Assert.ok( + operator.ethereum, + `resolveOppBootstrapSeed: scheduled batch operator ${accountName} has no Ethereum key` + ) + // The EM pair carries its address (`KeyPairAddress`) — the same + // `msg.sender` the operator's daemon signs `epochIn` with. + return operator.ethereum.address + }) + ), + epochDurationSec + }, + activeGroupIndex + } } } diff --git a/packages/cluster-tool/tests/orchestration/ClusterBuildDefaultsEpochBootstrap.test.ts b/packages/cluster-tool/tests/orchestration/ClusterBuildDefaultsEpochBootstrap.test.ts index 558df5735..b43db597a 100644 --- a/packages/cluster-tool/tests/orchestration/ClusterBuildDefaultsEpochBootstrap.test.ts +++ b/packages/cluster-tool/tests/orchestration/ClusterBuildDefaultsEpochBootstrap.test.ts @@ -8,9 +8,9 @@ import { import { collectStepNames } from "./clusterBuildFixture.js" -/** Bootstrap steps whose global order is load-bearing. */ +/** The four EpochBootstrap steps, in the order the depot requires them. */ const ScheduleBatchGroupsStep = "schedule-batch-groups" -const DeployEthereumStep = "deploy-ethereum" +const SeedEthereumRosterStep = "seed-ethereum-roster" const SeedSolanaRosterStep = "seed-solana-roster" const BootstrapEpochStep = "bootstrap-epoch" @@ -46,23 +46,26 @@ describe("ClusterBuildDefaults — EpochBootstrap step order", () => { } } - it("seeds both local outposts from schbatchgps before msgch::bootstrap", async () => { - // Ethereum's initializer and Solana's opp_bootstrap both read the schedule - // schbatchgps materialized. Neither may follow the first envelope. + it("seeds BOTH outpost rosters BETWEEN schbatchgps and msgch::bootstrap", async () => { + // Load-bearing order: both seeds read the schedule `schbatchgps` just + // materialized. Ethereum's `installInitialRoster` closes its bootstrap + // window on the first counted epoch-1 delivery, and the SOL outpost's + // `epoch_in` refuses to finalize the first envelope `msgch::bootstrap` + // delivers until `opp_bootstrap` has seeded it. const cluster = await ClusterBuildDefaults.create(baseOptions()) const names = collectStepNames(cluster.children) - expect(names.indexOf(DeployEthereumStep)).toBeGreaterThan( - names.indexOf(ScheduleBatchGroupsStep) + expect(names.indexOf(SeedEthereumRosterStep)).toBe( + names.indexOf(ScheduleBatchGroupsStep) + 1 ) - expect(names.indexOf(SeedSolanaRosterStep)).toBeGreaterThan( - names.indexOf(DeployEthereumStep) + expect(names.indexOf(SeedSolanaRosterStep)).toBe( + names.indexOf(SeedEthereumRosterStep) + 1 ) - expect(names.indexOf(BootstrapEpochStep)).toBeGreaterThan( - names.indexOf(SeedSolanaRosterStep) + expect(names.indexOf(BootstrapEpochStep)).toBe( + names.indexOf(SeedSolanaRosterStep) + 1 ) }) - it("omits the roster seed in external-outpost mode, keeping the rest in order", async () => { + it("omits both roster seeds in external-outpost mode, keeping the rest in order", async () => { // External outposts are seeded by their own operators, out of band. const cluster = await ClusterBuildDefaults.create({ ...baseOptions(), @@ -72,10 +75,10 @@ describe("ClusterBuildDefaults — EpochBootstrap step order", () => { underwriterCount: 0 }) const names = collectStepNames(cluster.children) + expect(names).not.toContain(SeedEthereumRosterStep) expect(names).not.toContain(SeedSolanaRosterStep) - expect(names).not.toContain(DeployEthereumStep) - expect(names.indexOf(BootstrapEpochStep)).toBeGreaterThan( - names.indexOf(ScheduleBatchGroupsStep) + expect(names.indexOf(BootstrapEpochStep)).toBe( + names.indexOf(ScheduleBatchGroupsStep) + 1 ) }) }) diff --git a/packages/cluster-tool/tests/orchestration/ethereum/EthereumOutpostBootstrapper.test.ts b/packages/cluster-tool/tests/orchestration/ethereum/EthereumOutpostBootstrapper.test.ts index 881f6c49c..f97c0f636 100644 --- a/packages/cluster-tool/tests/orchestration/ethereum/EthereumOutpostBootstrapper.test.ts +++ b/packages/cluster-tool/tests/orchestration/ethereum/EthereumOutpostBootstrapper.test.ts @@ -1,12 +1,27 @@ -import { EthereumOutpostBootstrapper } from "@wireio/cluster-tool/orchestration" -import { BindConfigProvider } from "@wireio/cluster-tool/config" +import Fs from "node:fs" +import Os from "node:os" +import Path from "node:path" +import { ethers } from "ethers" +import { ChainKind } from "@wireio/opp-typescript-models" +import { + EthereumOutpostBootstrapper, + type EthereumOutpostInitialRoster +} from "@wireio/cluster-tool/orchestration" +import { + BindConfigProvider, + ClusterConfigProvider +} from "@wireio/cluster-tool/config" import { toURL } from "@wireio/cluster-tool/utils" /** anvil/hardhat account 0 from the `test test … junk` mnemonic — well-known + stable. */ const AnvilAccount0Address = "0xf39Fd6e51aad88F6F4ce6aB8827279cffFb92266" const AnvilAccount0PrivateKey = "0xac0974bec39a17e36ba4a6b4d238ff944bacb478cbed5efcae784d7bf4f2ff80" -const InitialRoster = [[AnvilAccount0Address]] +/** anvil accounts 1 and 2 — distinct roster members for the seed cases. */ +const AnvilAccount1Address = "0x70997970C51812dc3A010C7d01b50e0d17dc79C8" +const AnvilAccount2Address = "0x3C44CdDdB6a900fa2b585dd299e03d12FA4293BC" +/** The depot's epoch duration for the seed cases. */ +const SeedEpochDurationSec = 60 describe("EthereumOutpostBootstrapper.generateAccounts", () => { it("generates the requested count deterministically from anvil's mnemonic", () => { @@ -27,7 +42,12 @@ describe("EthereumOutpostBootstrapper.generateAccounts", () => { describe("EthereumOutpostBootstrapper constructor", () => { let rpcUrl: string - const deploymentsPath = "/tmp/cluster/data/ethereum-deployments" + const deploymentsPath = "/tmp/cluster/data/ethereum-deployments", + /** A valid WNE-41 initial roster — one operator, one positive duration. */ + initialRoster: EthereumOutpostInitialRoster = { + groups: [[AnvilAccount0Address]], + epochDurationSec: ClusterConfigProvider.DefaultEpochDurationSec + } beforeAll(async () => { rpcUrl = toURL( await BindConfigProvider.findAvailable(BindConfigProvider.DefaultAnvil) @@ -42,9 +62,7 @@ describe("EthereumOutpostBootstrapper constructor", () => { anvilDataPath: "/tmp/anvil", rpcUrl, deploymentsPath, - initialOperatorGroups: InitialRoster, - initialActiveGroupIndex: 0, - epochDurationSec: 60 + initialRoster }) ).toThrow(/ethereumPath is required/) }) @@ -57,9 +75,7 @@ describe("EthereumOutpostBootstrapper constructor", () => { anvilDataPath: "", rpcUrl, deploymentsPath, - initialOperatorGroups: InitialRoster, - initialActiveGroupIndex: 0, - epochDurationSec: 60 + initialRoster }) ).toThrow(/anvilDataPath is required/) }) @@ -72,9 +88,7 @@ describe("EthereumOutpostBootstrapper constructor", () => { anvilDataPath: "/tmp/anvil", rpcUrl: "", deploymentsPath, - initialOperatorGroups: InitialRoster, - initialActiveGroupIndex: 0, - epochDurationSec: 60 + initialRoster }) ).toThrow(/rpcUrl is required/) }) @@ -87,14 +101,15 @@ describe("EthereumOutpostBootstrapper constructor", () => { anvilDataPath: "/tmp/anvil", rpcUrl, deploymentsPath: "", - initialOperatorGroups: InitialRoster, - initialActiveGroupIndex: 0, - epochDurationSec: 60 + initialRoster }) ).toThrow(/deploymentsPath is required/) }) - it("throws when the initial roster is empty", () => { + // WNE-41 — `OPPInbound.initialize` is one-shot and `isActiveOperator` is + // fail-closed, so both of these would otherwise produce an outpost whose + // `epochIn` no address can ever call. + it("throws when the initial roster carries no operator", () => { expect( () => new EthereumOutpostBootstrapper({ @@ -102,10 +117,149 @@ describe("EthereumOutpostBootstrapper constructor", () => { anvilDataPath: "/tmp/anvil", rpcUrl, deploymentsPath, - initialOperatorGroups: [], - initialActiveGroupIndex: 0, - epochDurationSec: 60 + initialRoster: { + groups: [[]], + epochDurationSec: ClusterConfigProvider.DefaultEpochDurationSec + } }) - ).toThrow(/initialOperatorGroups must contain non-empty groups/) + ).toThrow(/at least one batch-operator address/) + }) + + it("throws when the initial epochDurationSec is not positive", () => { + expect( + () => + new EthereumOutpostBootstrapper({ + ethereumPath: "/repo/eth", + anvilDataPath: "/tmp/anvil", + rpcUrl, + deploymentsPath, + initialRoster: { groups: [[AnvilAccount0Address]], epochDurationSec: 0 } + }) + ).toThrow(/epochDurationSec must be positive/) + }) +}) + +describe("EthereumOutpostBootstrapper.initialBatchOperatorGroups", () => { + /** A seed over `groups`, epoch-1 slot `activeGroupIndex`. */ + const seed = ( + groups: string[][], + activeGroupIndex = 0, + epochDurationSec = SeedEpochDurationSec + ): EthereumOutpostBootstrapper.OppBootstrapSeed => ({ + window: { groups, epochDurationSec }, + activeGroupIndex + }) + + it("builds the BatchOperatorGroups tuple: EVM-kinded, checksummed, anchored at epoch 0", () => { + const tuple = EthereumOutpostBootstrapper.initialBatchOperatorGroups( + seed([[AnvilAccount1Address.toLowerCase()], [AnvilAccount2Address, AnvilAccount1Address]], 1) + ) + expect(tuple).toEqual({ + activeGroupIndex: 1, + epochIndex: EthereumOutpostBootstrapper.InitialRosterAnchorEpochIndex, + groups: [ + { operators: [{ kind: ChainKind.EVM, address_: AnvilAccount1Address }] }, + { + operators: [ + { kind: ChainKind.EVM, address_: AnvilAccount2Address }, + { kind: ChainKind.EVM, address_: AnvilAccount1Address } + ] + } + ], + epochDurationSec: SeedEpochDurationSec + }) + expect(EthereumOutpostBootstrapper.InitialRosterAnchorEpochIndex).toBe(0) + }) + + it("allows the same operator in DIFFERENT slots — one operator serving consecutive epochs", () => { + const tuple = EthereumOutpostBootstrapper.initialBatchOperatorGroups( + seed([[AnvilAccount1Address], [AnvilAccount1Address]]) + ) + expect(tuple.groups).toHaveLength(2) + }) + + it("rejects what OPPInbound._installInitialRoster rejects, with a readable reason", () => { + for (const [groups, reason] of [ + [[], /at least one batch-operator group/], + [[[]], /group 0 is empty/], + [[[AnvilAccount1Address], []], /group 1 is empty/], + [[["not-an-address"]], /group 0 member 'not-an-address' is not an EVM address/], + [[[ethers.ZeroAddress]], /group 0 contains the zero address/], + [ + [[AnvilAccount1Address, AnvilAccount1Address.toLowerCase()]], + /appears twice in group 0/ + ] + ] as const) { + expect(() => + EthereumOutpostBootstrapper.initialBatchOperatorGroups(seed([...groups.map(group => [...group])])) + ).toThrow(reason) + } + }) + + it("rejects a non-positive epoch duration and an out-of-range active slot", () => { + expect(() => + EthereumOutpostBootstrapper.initialBatchOperatorGroups(seed([[AnvilAccount1Address]], 0, 0)) + ).toThrow(/epochDurationSec must be a positive integer/) + expect(() => + EthereumOutpostBootstrapper.initialBatchOperatorGroups(seed([[AnvilAccount1Address]], 1)) + ).toThrow(/activeGroupIndex 1 is out of range for 1 group/) + }) +}) + +describe("EthereumOutpostBootstrapper.oppBootstrap", () => { + let rpcUrl: string, deploymentsPath: string + beforeAll(async () => { + rpcUrl = toURL( + await BindConfigProvider.findAvailable(BindConfigProvider.DefaultAnvil) + ) + }) + beforeEach(() => { + deploymentsPath = Fs.mkdtempSync(Path.join(Os.tmpdir(), "eth-opp-bootstrap-")) + }) + afterEach(() => { + Fs.rmSync(deploymentsPath, { recursive: true, force: true }) + }) + + it("refuses to seed before the outpost deploy has written its address map", async () => { + // The seed runs in EpochBootstrap, long after `deploy-ethereum`; a missing + // address map means the phases were reordered, not that there is nothing + // to seed — fail loudly rather than skip like the reserve seeding does. + const window: EthereumOutpostInitialRoster = { + groups: [[AnvilAccount1Address]], + epochDurationSec: SeedEpochDurationSec + } + const bootstrapper = new EthereumOutpostBootstrapper({ + ethereumPath: "/repo/eth", + anvilDataPath: Path.join(deploymentsPath, "anvil"), + rpcUrl, + deploymentsPath, + initialRoster: window + }) + await expect( + bootstrapper.oppBootstrap({ window, activeGroupIndex: 0 }) + ).rejects.toThrow(/outpost-addrs\.json is missing — the Ethereum outpost deploy must precede the roster seed/) + }) + + it("validates the seed before touching the chain", async () => { + Fs.writeFileSync( + Path.join(deploymentsPath, EthereumOutpostBootstrapper.OutpostAddressesFile), + JSON.stringify({ OutpostManager: AnvilAccount0Address, OPPInbound: AnvilAccount1Address }) + ) + const window: EthereumOutpostInitialRoster = { + groups: [[AnvilAccount1Address]], + epochDurationSec: SeedEpochDurationSec + } + const bootstrapper = new EthereumOutpostBootstrapper({ + ethereumPath: "/repo/eth", + anvilDataPath: Path.join(deploymentsPath, "anvil"), + rpcUrl, + deploymentsPath, + initialRoster: window + }) + // An out-of-range active slot fails in the tuple builder — no provider, + // no artifact read, no RPC. + await expect( + bootstrapper.oppBootstrap({ window, activeGroupIndex: 3 }) + ).rejects.toThrow(/activeGroupIndex 3 is out of range/) }) }) diff --git a/packages/cluster-tool/tests/orchestration/ethereum/EthereumOutpostSteps.test.ts b/packages/cluster-tool/tests/orchestration/ethereum/EthereumOutpostSteps.test.ts index 4e9ec2fd8..a111b37d7 100644 --- a/packages/cluster-tool/tests/orchestration/ethereum/EthereumOutpostSteps.test.ts +++ b/packages/cluster-tool/tests/orchestration/ethereum/EthereumOutpostSteps.test.ts @@ -1,9 +1,40 @@ -import { Steps } from "@wireio/cluster-tool/orchestration" -import { Report } from "@wireio/cluster-tool/report" +import { ethers } from "ethers" +import { getLogger } from "@wireio/shared" import { OperatorType } from "@wireio/opp-typescript-models" - +import { Constants } from "@wireio/cluster-tool" +import { + ClusterBuildContext, + EthereumOutpostBootstrapper, + EthereumOutpostSteps, + Steps +} from "@wireio/cluster-tool/orchestration" +import { Report } from "@wireio/cluster-tool/report" +import { fixtureConfig } from "../../config/clusterConfigFixture.js" import { fixtureOperatorAccount } from "../outputs/operatorAccountFixture.js" +/** A batch `OperatorAccount` under an explicit chain account name, with its OWN EM key. */ +const batchOperator = (label: string, account: string, ethereumHdIndex: number) => + fixtureOperatorAccount(label, OperatorType.BATCH, account, ethereumHdIndex) + +/** The depot's epoch duration for the seed cases. */ +const SeedEpochDurationSec = 60 + +/** A context over the persisted fixture, optionally reshaped for a case. */ +const context = (overrides: Parameters[0] = {}) => + new ClusterBuildContext(fixtureConfig(overrides), getLogger("eth-outpost-test")) + +/** + * The address the harness will later generate the operator's EM key for — + * derived here INDEPENDENTLY (straight from ethers) so the test pins the + * mapping rather than re-running the code under test. + */ +const expectedOperatorAddress = (index: number): string => + ethers.HDNodeWallet.fromPhrase( + EthereumOutpostBootstrapper.AnvilMnemonic, + undefined, + `${EthereumOutpostBootstrapper.DerivationPath}${Constants.batchOperatorEthereumHdIndex(index)}` + ).address + describe("Steps.ethereumOutpost.deploy", () => { it("builds an input-less deploy step with a runner", () => { const step = Steps.ethereumOutpost.planDeploy( @@ -18,46 +49,197 @@ describe("Steps.ethereumOutpost.deploy", () => { }) }) -describe("Steps.ethereumOutpost.resolveInitialOperatorGroups", () => { - it("maps every depot schedule group to Ethereum addresses in schedule order", () => { - const fixtures = [ - fixtureOperatorAccount("batchop.a", OperatorType.BATCH, "wireno.alpha"), - fixtureOperatorAccount("batchop.b", OperatorType.BATCH, "wireno.bravo"), - fixtureOperatorAccount("batchop.c", OperatorType.BATCH, "wireno.charlie") - ], - operators = fixtures.map((operator, index) => ({ - ...operator, - ethereum: { - ...operator.ethereum, - address: `0x${String(index + 1).padStart(40, "0")}` - } - })) - const groups = Steps.ethereumOutpost.resolveInitialOperatorGroups(operators, [ - ["wireno.charlie", "wireno.alpha"], - ["wireno.bravo"] +describe("EthereumOutpostSteps.resolveInitialRoster", () => { + it("carries the depot's epoch duration", async () => { + const roster = await EthereumOutpostSteps.resolveInitialRoster( + context({ epochDurationSec: 45 }) + ) + expect(roster.epochDurationSec).toBe(45) + }) + + it("seats EVERY batch operator, at the address its own EM key will derive to", async () => { + // The whole point of WNE-41 on a cluster: these are the addresses the + // daemons sign `epochIn` with, and `isActiveOperator` is fail-closed. + const roster = await EthereumOutpostSteps.resolveInitialRoster( + context({ batchOperatorCount: 3, operatorsPerEpoch: 3, batchOpGroups: 1 }) + ) + // Order is the depot's business — `schbatchgps` interleaves before + // partitioning — so this asserts MEMBERSHIP, which is what authorization + // depends on. + expect(roster.groups.flat().sort()).toEqual( + [ + expectedOperatorAddress(0), + expectedOperatorAddress(1), + expectedOperatorAddress(2) + ].sort() + ) + }) + + it("excludes the deploy owner — deployment privilege is not delivery privilege", async () => { + const roster = await EthereumOutpostSteps.resolveInitialRoster(context()), + deployer = EthereumOutpostBootstrapper.generateAccounts(1)[0].address + expect(roster.groups.flat()).not.toContain(deployer) + }) + + it("sizes group 0 by the depot's operators-per-epoch (the consensus threshold)", async () => { + // `OPPInbound` derives its threshold from `batchOpGroups[0].length`, so a + // group 0 wider than the depot's active group would demand deliveries that + // never come and stall the epoch. + const roster = await EthereumOutpostSteps.resolveInitialRoster( + context({ batchOperatorCount: 9, operatorsPerEpoch: 3, batchOpGroups: 3 }) + ) + expect(roster.groups).toHaveLength(3) + roster.groups.forEach(group => expect(group).toHaveLength(3)) + }) + + it("seats every scheduled operator across the window's slots, once", async () => { + // 9 operators, 3 groups of 3: the whole roster is scheduled, so every + // address lands in exactly one slot — slot `k` serving epoch `1 + k`. + const roster = await EthereumOutpostSteps.resolveInitialRoster( + context({ batchOperatorCount: 9, operatorsPerEpoch: 3, batchOpGroups: 3 }) + ) + expect(roster.groups.flat()).toContain(expectedOperatorAddress(8)) + expect(new Set(roster.groups.flat()).size).toBe(9) + }) + + it("interleaves the pool the way `schbatchgps` does", async () => { + // The depot shuffles evens-then-odds before partitioning, so group 0 is + // [0, 2, 4] rather than [0, 1, 2]. Reproducing it keeps the initial + // grouping the same SHAPE the depot installs on its first attestation. + const roster = await EthereumOutpostSteps.resolveInitialRoster( + context({ batchOperatorCount: 9, operatorsPerEpoch: 3, batchOpGroups: 3 }) + ) + expect(roster.groups[0]).toEqual([ + expectedOperatorAddress(0), + expectedOperatorAddress(2), + expectedOperatorAddress(4) ]) + }) +}) + +describe("EthereumOutpostSteps.partitionLikeDepot", () => { + const pool = (count: number) => + Array.from({ length: count }, (_unused, index) => `0x${index}`) + + it("trims to batch_operator_minimum_active before grouping", () => { + // 9 addresses but the depot only schedules 3: group 0's length is the + // consensus threshold, so the untrimmed pool must not widen it. + const groups = EthereumOutpostSteps.partitionLikeDepot(pool(9), 3, 1, 3) + expect(groups[0]).toHaveLength(3) + }) + + it("leaves operators past the scheduled window OUT — the depot leaves them ungrouped too", () => { + // Under WNE-27 a slot authorizes its members for the epoch it serves, so a + // seat for an unscheduled operator would authorize it for an epoch the + // depot never scheduled it for. It stays ACTIVE and ungrouped, as on the + // depot, until a rotation schedules it. + const groups = EthereumOutpostSteps.partitionLikeDepot(pool(5), 3, 1, 3) + expect(groups).toHaveLength(1) + expect(groups[0]).toHaveLength(3) + expect(groups.flat()).not.toContain("0x3") + expect(groups.flat()).not.toContain("0x4") + }) + + it("emits exactly batch_op_groups groups when the pool fills them", () => { + const groups = EthereumOutpostSteps.partitionLikeDepot(pool(9), 3, 3, 9) + expect(groups).toHaveLength(3) + expect(groups[0]).toEqual(["0x0", "0x2", "0x4"]) + }) + + it("never emits an EMPTY group — the contract rejects one", () => { + // A pool short of `batchOpGroups * operatorsPerEpoch` yields fewer groups + // rather than padded ones; `_installInitialRoster` reverts on an empty one. + const groups = EthereumOutpostSteps.partitionLikeDepot(pool(3), 3, 3, 3) + expect(groups.every(group => group.length > 0)).toBe(true) + expect(groups.flat()).toHaveLength(3) + }) +}) - expect(groups).toEqual([ - [operators[2].ethereum.address, operators[0].ethereum.address], +describe("Steps.ethereumOutpost.oppBootstrap", () => { + it("builds an input-less installInitialRoster step with a runner", () => { + const step = Steps.ethereumOutpost.planOppBootstrap( + Report.Actor.EthereumOutpost, + "seed-ethereum-roster", + "seed the Ethereum outpost batch-operator roster", + {} + ) + expect(step.actor).toBe(Report.Actor.EthereumOutpost) + expect(step.input).toBeNull() + expect(typeof step.runner).toBe("function") + }) + + it("maps every window slot, in slot order, to its members' EM addresses", () => { + const operators = [ + batchOperator("batchop.a", "wireno.aaaaa", 1), + batchOperator("batchop.b", "wireno.bbbbb", 2), + batchOperator("batchop.c", "wireno.ccccc", 3) + ] + // The depot's window orders by ACCOUNT NAME, which need not follow the + // harness's label order — exactly the gap the seed exists to close. + const seed = EthereumOutpostSteps.resolveOppBootstrapSeed( + operators, + [["wireno.ccccc"], ["wireno.aaaaa"], ["wireno.bbbbb"]], + 0, + SeedEpochDurationSec + ) + + expect(seed.window.groups).toEqual([ + [operators[2].ethereum.address], + [operators[0].ethereum.address], [operators[1].ethereum.address] ]) + expect(new Set(seed.window.groups.flat()).size).toBe(3) + expect(seed.window.epochDurationSec).toBe(SeedEpochDurationSec) + expect(seed.activeGroupIndex).toBe(0) }) - it("rejects schedule members missing from the provisioned batch roster", () => { + it("keeps member order within a multi-member slot and carries the depot's active slot", () => { const operators = [ - fixtureOperatorAccount("batchop.a", OperatorType.BATCH, "wireno.alpha") + batchOperator("batchop.a", "wireno.aaaaa", 1), + batchOperator("batchop.b", "wireno.bbbbb", 2), + batchOperator("batchop.c", "wireno.ccccc", 3) ] + const seed = EthereumOutpostSteps.resolveOppBootstrapSeed( + operators, + [["wireno.bbbbb", "wireno.aaaaa", "wireno.ccccc"]], + 0, + SeedEpochDurationSec + ) + expect(seed.window.groups).toEqual([ + [operators[1].ethereum.address, operators[0].ethereum.address, operators[2].ethereum.address] + ]) + // A rotated depot cursor rides through untouched. + expect( + EthereumOutpostSteps.resolveOppBootstrapSeed( + operators, + [["wireno.aaaaa"], ["wireno.bbbbb"]], + 1, + SeedEpochDurationSec + ).activeGroupIndex + ).toBe(1) + }) + + it("throws when a scheduled account is not a provisioned batch operator", () => { + const operators = [batchOperator("batchop.a", "wireno.aaaaa", 1)] expect(() => - Steps.ethereumOutpost.resolveInitialOperatorGroups(operators, [["wireno.ghost"]]) - ).toThrow(/wireno\.ghost.*not found among provisioned batch operators/) + EthereumOutpostSteps.resolveOppBootstrapSeed( + operators, + [["wireno.aaaaa"], ["wireno.zzzzz"]], + 0, + SeedEpochDurationSec + ) + ).toThrow(/wireno\.zzzzz not found among provisioned batch operators/) }) - it("rejects empty schedules and groups", () => { - expect(() => Steps.ethereumOutpost.resolveInitialOperatorGroups([], [])).toThrow( - /schedule is empty/ - ) - expect(() => Steps.ethereumOutpost.resolveInitialOperatorGroups([], [[]])).toThrow( - /schedule group 0 is empty/ - ) + it("throws a DISTINCT error when a scheduled operator carries no Ethereum key", () => { + const keyless = { ...batchOperator("batchop.a", "wireno.aaaaa", 1), ethereum: undefined } + expect(() => + EthereumOutpostSteps.resolveOppBootstrapSeed( + [keyless], + [["wireno.aaaaa"]], + 0, + SeedEpochDurationSec + ) + ).toThrow(/wireno\.aaaaa has no Ethereum key/) }) }) diff --git a/packages/flow-batch-operator-termination/src/TerminationScenario.ts b/packages/flow-batch-operator-termination/src/TerminationScenario.ts index 983b2eb68..1d2155222 100644 --- a/packages/flow-batch-operator-termination/src/TerminationScenario.ts +++ b/packages/flow-batch-operator-termination/src/TerminationScenario.ts @@ -2,9 +2,15 @@ import Assert from "node:assert" import { PublicKey } from "@solana/web3.js" import type { BN } from "@coral-xyz/anchor" import { SysioContracts } from "@wireio/sdk-core" -import { OperatorType } from "@wireio/opp-typescript-models" +import { + AttestationType, + BatchOperatorGroups, + Envelope, + OperatorType +} from "@wireio/opp-typescript-models" import { ClusterBuildPhase, + ClusterConfigProvider, EthereumCollateralTool, FlowScenario, Report, @@ -14,6 +20,7 @@ import { Steps, WireOperatorProvisioningTool, getLogger, + loadOutpostContract, matchesProtoEnum, outputKey, packedSlugValue, @@ -137,6 +144,15 @@ interface SolanaAccountClient { fetch(address: PublicKey): Promise } +/** The outpost cursors advance only after an inbound envelope is accepted. */ +interface EthereumInboundView { + nextEpochIndex(): Promise +} + +interface SolanaOutpostConfigAccount { + nextEpochIndex: number +} + /** The SOL outpost's on-chain collateral ledger from the `OperatorRegistry` PDA (a read). */ async function readSolanaCollateralLedger( ctx: ClusterBuildContext @@ -188,11 +204,14 @@ async function readSolanaCollateralLedger( * outpost's escrow ledger returns to 0, and each wallet is credited the * exact bond amount (wei/lamport-exact — any drift means the outpost decoded * a different amount than the depot encoded). + * 9. **ContinuedRotation** — standing operators fill the vacated seat, every + * observed published window excludes the terminated operator, and both + * outposts accept another complete rotation after the remits land. */ export class TerminationScenario extends FlowScenario { readonly name = "flow-batch-operator-termination" readonly description = - "Non-bootstrapped batch operator bonds ETH + SOL, misses its scheduled deliveries, is terminated, and both bonds are remitted back" + "Batch operator termination remits both bonds and standing operators keep both outposts advancing" override readonly defaults: ClusterBuildOptions = { epochDurationSec: Constants.EpochDurationSec, @@ -709,5 +728,142 @@ export class TerminationScenario extends FlowScenario { quickStepOptions ) ) + + // ── 9. Remittance is not enough: the following duty groups must deliver ── + ClusterBuildPhase.create( + cluster, + "ContinuedRotation", + "Standing operators absorb the termination and both outposts complete another rotation" + ).push( + verifyStep( + Actor.Sysio, + "post-remit-rotation", + "published groups exclude the terminated operator and both outposts advance through a full window", + async ctx => { + const operator = ctx.keyStore.assertOperator( + Constants.DoomedOperatorLabel + ), + addresses = EthereumCollateralTool.loadOutpostAddresses( + ClusterConfigProvider.ethereumDeploymentsPath(ctx.config) + ), + ethereum = loadOutpostContract( + ctx.config.ethereumPath, + addresses, + "OPPInbound", + ["outpost"], + ctx.ethereum.wallet.signer + ), + program = SolanaCollateralTool.loadOppOutpostProgram( + ctx, + solanaKeypair(operator.solana) + ), + configAddress = SolanaOutpostProgramTool.derivePda( + program.programId, + Buffer.from(SolanaOutpostBootstrapper.PdaSeed.OutpostConfig) + ), + accounts: Record = program.account, + readCursors = async () => { + const [ethNext, solConfig] = await Promise.all([ + ethereum.nextEpochIndex(), + accounts.outpostConfig.fetch(configAddress) + ]) + return [ + Number(ethNext), + Number((solConfig as SolanaOutpostConfigAccount).nextEpochIndex) + ] + }, + baseline = await readCursors(), + start = await Steps.contracts.sysio.epoch.readEpochState(ctx), + standing = new Set( + ctx.keyStore.operators + .filter( + entry => + entry.type === OperatorType.BATCH && + entry.account !== operator.account + ) + .map(entry => entry.account) + ), + groupCount = start.batch_op_groups.length, + groupSize = ctx.config.operatorsPerEpoch, + targetEpoch = + Math.max(Number(start.current_epoch_index), ...baseline) + + groupCount, + chains = [Constants.EthereumChainCode, Constants.SolanaChainCode] + + Assert.ok( + groupCount > 1, + "termination regression requires multiple duty groups" + ) + await pollUntil( + `both outposts accept post-remit epochs through ${targetEpoch}`, + async () => { + const { rows } = await ctx.wire.getOutboundEnvelopes() + for (const chain of chains) { + const row = rows.find( + entry => packedSlugValue(entry.chain_code) === chain + ) + Assert.ok( + row != null, + `missing outbound envelope for chain ${chain}` + ) + const envelope = Envelope.fromBinary( + Buffer.from(row.raw_envelope, "hex") + ), + announcements = envelope.messages.flatMap(message => + (message.payload?.attestations ?? []) + .filter( + entry => + entry.type === AttestationType.BATCH_OPERATOR_GROUPS + ) + .map(entry => BatchOperatorGroups.fromBinary(entry.data)) + ) + Assert.ok( + announcements.length > 0, + `no published group window at epoch ${row.epoch_index}` + ) + for (const announcement of announcements) { + const groups = announcement.groups.map(group => + group.operators.map(member => + Buffer.from(member.address).toString("utf8") + ) + ), + members = groups.flat() + Assert.equal( + groups.length, + groupCount, + "published window lost a group" + ) + Assert.ok( + groups.every(group => group.length === groupSize), + "standing operators did not fill every seat" + ) + Assert.ok( + !members.includes(operator.account), + "terminated operator re-entered a published group" + ) + Assert.equal( + new Set(members).size, + groupCount * groupSize, + "published groups repeat an operator" + ) + Assert.ok( + members.every(account => standing.has(account)), + "replacement is not a standing operator" + ) + } + } + const cursors = await readCursors() + log.info( + `[${this.name}] post-remit ETH/SOL next epochs=${cursors.join("/")}; target>${targetEpoch}` + ) + return cursors.every(epoch => epoch > targetEpoch) + }, + Constants.remitDeadlineMs(), + Constants.PollIntervalMs + ) + }, + remitStepOptions + ) + ) } } From 082d21a161cbcc8adb957b3b9e3519a21d66dadc Mon Sep 17 00:00:00 2001 From: Huang-Ming Huang Date: Thu, 24 Sep 2026 18:19:19 +0000 Subject: [PATCH 15/16] Fix termination flow default group-size assertion Change-Id: Id6d1c5678e2894b5aba6b7d64f6a04b9781a21df --- .../src/TerminationScenario.ts | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/packages/flow-batch-operator-termination/src/TerminationScenario.ts b/packages/flow-batch-operator-termination/src/TerminationScenario.ts index 1d2155222..227e260cd 100644 --- a/packages/flow-batch-operator-termination/src/TerminationScenario.ts +++ b/packages/flow-batch-operator-termination/src/TerminationScenario.ts @@ -9,6 +9,7 @@ import { OperatorType } from "@wireio/opp-typescript-models" import { + BatchOperatorSchedule, ClusterBuildPhase, ClusterConfigProvider, EthereumCollateralTool, @@ -784,7 +785,9 @@ export class TerminationScenario extends FlowScenario { .map(entry => entry.account) ), groupCount = start.batch_op_groups.length, - groupSize = ctx.config.operatorsPerEpoch, + groupSize = BatchOperatorSchedule.resolve( + ctx.config + ).operatorsPerEpoch, targetEpoch = Math.max(Number(start.current_epoch_index), ...baseline) + groupCount, From d50fc7847f11867245bca5432871dadd88f42e21 Mon Sep 17 00:00:00 2001 From: Huang-Ming Huang Date: Fri, 25 Sep 2026 20:25:09 +0000 Subject: [PATCH 16/16] Clarify read-only Solana epoch cursor projection Change-Id: I697691ea336494c4bde147818a84c06c31ae1159 --- .../flow-batch-operator-termination/src/TerminationScenario.ts | 1 + 1 file changed, 1 insertion(+) diff --git a/packages/flow-batch-operator-termination/src/TerminationScenario.ts b/packages/flow-batch-operator-termination/src/TerminationScenario.ts index 227e260cd..77d99a679 100644 --- a/packages/flow-batch-operator-termination/src/TerminationScenario.ts +++ b/packages/flow-batch-operator-termination/src/TerminationScenario.ts @@ -150,6 +150,7 @@ interface EthereumInboundView { nextEpochIndex(): Promise } +/** Read-only projection of the Solana outpost configuration's inbound epoch cursor. */ interface SolanaOutpostConfigAccount { nextEpochIndex: number }