From c408b4f390c749b4c2f1440510a51d0e18252fa7 Mon Sep 17 00:00:00 2001 From: J-Dog Date: Thu, 17 Sep 2026 09:49:18 -0700 Subject: [PATCH 01/35] ci: sibling checkouts follow the pull request's base branch, keeping push, release and hotfix refs unchanged --- .github/workflows/ci.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index d63f8ce..1549262 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -42,7 +42,7 @@ jobs: uses: actions/checkout@v4 with: repository: XChain-Platform/xchain-hub - ref: ${{ github.ref == 'refs/heads/master' && 'master' || 'develop' }} + ref: ${{ github.base_ref || (github.ref == 'refs/heads/master' && 'master' || 'develop') }} ssh-key: ${{ secrets.XCHAIN_HUB_DEPLOY_KEY }} path: xchain-hub From 829d2f7a20c2b05ca9c507a6fd21baab2d381ab2 Mon Sep 17 00:00:00 2001 From: J-Dog Date: Thu, 17 Sep 2026 10:43:10 -0700 Subject: [PATCH 02/35] refactor(node): extract migration scan helpers Move migration scanning and ledger reads behind the service facade. Preserve migration guard behavior while reducing oversized file and function counts. --- .../migration_precondition_service.js | 280 ++---------------- .../migration_scan.js | 264 +++++++++++++++++ 2 files changed, 294 insertions(+), 250 deletions(-) create mode 100644 src/services/migration_precondition_service/migration_scan.js diff --git a/src/services/migration_precondition_service.js b/src/services/migration_precondition_service.js index e6f8eb6..3a3eed7 100644 --- a/src/services/migration_precondition_service.js +++ b/src/services/migration_precondition_service.js @@ -52,17 +52,21 @@ * schema from src/sql at the current widths and can never be behind. ********************************************************************/ -const fs = require('fs') - -const { XChainService, EXTERNAL_DB } = require('../config') -const { tableCountSql, tableExistsSql } = require('../db/information_schema') -const { appliedMigrationsSql } = require('../db/migrations') -const { migrationsDirOf, migrationFiles } = require('../utils/migration_files') +const { XChainService } = require('../config') +const { migrationsDirOf } = require('../utils/migration_files') const { getModuleTmpDir, getModuleDatabaseName, getDockerContainerImageName } = require('./config_service') -const { readMigrateCli, migrateCliPathFor } = require('../utils/indexer_migrate_cli') const config = require('../config'); const { getLogger } = require('../observability/logger'); const logger = getLogger(); +const { + migrationDeclaresDeployPrecondition, + migrationMode, + pendingManualMigrations, + runningBuildSupportsPerFileMigrations, + listDeployPreconditionMigrations, + readAppliedMigrations, + refusalMessage +} = require('./migration_precondition_service/migration_scan') // Only these modules ship a migrations directory, so everything else skips the // guard entirely and costs the update path nothing. @@ -72,7 +76,6 @@ const MIGRATION_BEARING_MODULES = [ ] const SKIP_ENV = 'XCHAIN_NODE_SKIP_MIGRATION_PRECONDITION' -const LEDGER_TABLE = 'schema_migrations' function guardSkipped() { // Read BY NAME, not through SKIP_ENV, even though the constant is right @@ -84,231 +87,24 @@ function guardSkipped() { return v === '1' || v === 'true' || v === 'yes' } -/** - * Does this migration file's header declare itself a deploy precondition? - * - * Prologue-anchored: the scan stops at the first non-blank, non-comment line, so - * the token can only arm the flag from the leading comment block and never from - * body prose or a data literal. Widening that to the whole file is how a - * migration that merely DISCUSSES the convention would start refusing deploys. - * - * Twin of xchain-indexer's Database.migrationDeclaresDeployPrecondition. It is - * duplicated rather than shared because this tool reads these files out of a - * source tree it has only cloned, with that tree's dependencies uninstalled, so - * requiring the module is not available to it. Keep the two in step. - */ -function migrationDeclaresDeployPrecondition(raw) { - const prologue = [] - for (const line of String(raw).split('\n')) { - const trimmed = line.trim() - if (trimmed === '' || trimmed.startsWith('--')) { prologue.push(line); continue } - break - } - return /^\s*--\s*xchain:migration\b[^\n]*\bdeploy-precondition\s*=\s*required\b/im.test(prologue.join('\n')) -} - -/** - * The `mode=` a migration header declares, or null when it declares none. - * Prologue-anchored exactly like migrationDeclaresDeployPrecondition, so a token - * in body prose or a data literal cannot answer for the file. - * - * Twin of the modules' own Database._migrationMode, duplicated for the reason - * given above: this tool reads a cloned tree whose dependencies are not - * installed. Keep them in step. - */ -function migrationMode(raw) { - const prologue = [] - for (const line of String(raw).split('\n')) { - const trimmed = line.trim() - if (trimmed === '' || trimmed.startsWith('--')) { prologue.push(line); continue } - break - } - const m = prologue.join('\n').match(/^\s*--\s*xchain:migration\b[^\n]*\bmode\s*=\s*([A-Za-z]+)/im) - return m ? m[1].toLowerCase() : null -} - -/** - * Every gated (mode=manual) migration in `dir` that the ledger has not recorded, - * sorted. This is the blast radius of an UNSCOPED migrate run against that - * database: the runner applies every pending manual file, not just the one an - * operator names. The refusal names that whole set, so the consequence is on - * screen rather than left for the operator to discover. - */ -function pendingManualMigrations(dir, applied) { - return migrationFiles(dir).filter(([f, file]) => { - if (applied && applied.has(f)) return false - try { - return migrationMode(fs.readFileSync(file, 'utf8')) === 'manual' - } catch { - return false - } - }).map(([f]) => f) +function emptyDatabaseResult(module, dbName, required) { + logger.warn(`Migration precondition guard: ${dbName} holds no tables yet, so ${module}'s gated ` + + `migrations (${required.join(', ')}) cannot be outstanding on it; proceeding.`) + return { checked: true, required, ok: true, reason: 'empty-database' } } -/** - * Does the build CURRENTLY RUNNING in the target container understand per-file - * migration targeting (`--file`)? - * - * This matters because the remedy an operator is about to run executes inside - * that container, on its build, not on the one being deployed. A build without - * the flag does not reject it: it ignores it and applies every pending manual - * migration, which on a live database can mean a data backfill and a - * dedup-then-unique nobody authorised. - * - * Returns true, false, or null when the container could not be read at all - * (stopped, absent, docker unreachable). Callers must treat null like false: - * an unverified capability is not a capability, and the cost of being wrong is - * asymmetric. - */ -async function runningBuildSupportsPerFileMigrations(container, deps = {}) { - try { - const cat = deps.getDockerContainerFileCat || require('./docker_service').getDockerContainerFileCat - const found = await readMigrateCli(cat, container) - return found ? /['"]--file['"]/.test(found.source) : null - } catch { - return null - } -} - -/** - * Every migration filename in `dir` whose header declares a deploy precondition, - * sorted. A missing directory yields [] - a module (or a ref) with no migrations - * declares no preconditions, which is not an error. - */ -function listDeployPreconditionMigrations(dir) { - return migrationFiles(dir).filter(([, file]) => { - try { - return migrationDeclaresDeployPrecondition(fs.readFileSync(file, 'utf8')) - } catch { - return false - } - }).map(([f]) => f) -} - -/** - * Read the applied-migration ledger of one module database. - * - * Returns { state: 'ledger', applied: Set } when the ledger was read, - * { state: 'empty-database' } when the schema holds no tables at all (or does - * not exist yet), and { state: 'unreadable', reason } for everything else. The - * three are deliberately distinct: only the middle one is safe to proceed on. - * - * WHY ROOT AND NOT THE MODULE'S OWN ACCOUNT - * ----------------------------------------- - * The first cut of this guard connected with the module's generated credentials - * (INDEXER_DB_USER/PASS) over the published port. Run against the live regtest - * stack it produced `Access denied for user 'xchain_indexer_bitcoin_regtest'`, - * i.e. an unknown-state REFUSAL of a perfectly deployable update - the sidecar - * password and the live account had drifted, which is a documented recurring - * condition here and has nothing to do with migrations. A guard that fails - * closed on a routine credential drift blocks every deploy and gets switched - * off. So this uses the same root-credential runner every other DB read in - * xchain-node uses (see clearHubPriceIngestWatermark), which the update path - * already resolves non-interactively for its own credential parity pass. - */ -async function defaultReadAppliedMigrations({ database, coin, network }, deps = {}) { - // The name comes from getModuleDatabaseName, but it reaches SQL as text (an - // identifier cannot be bound), so gate it on the same allowlist the - // provisioning DDL uses rather than trusting its provenance. - if (!/^[A-Za-z0-9_]+$/.test(String(database))) { - return { state: 'unreadable', reason: 'refusing to query a database name that is not a plain identifier' } - } - const literal = "'" + database + "'" - - let runner = deps.runner - try { - if (!runner) { - const { - getExternalDbConfig, executeNativeMariaDbCommand, - executeDockerMariaDbCommand, askMariadbRootPassword, getDatabaseContainerId - } = require('./database_service') - if (EXTERNAL_DB) { - const cfg = await getExternalDbConfig() - runner = (sql) => executeNativeMariaDbCommand(cfg, sql, '-B -N') - } else { - const containerId = await getDatabaseContainerId() - if (!containerId) return { state: 'unreadable', reason: 'no MariaDB container found on this host' } - const rootPassword = await askMariadbRootPassword(coin, network) - runner = (sql) => executeDockerMariaDbCommand(containerId, rootPassword, sql, '-B -N') - } - } - - const rawTableCount = String(await runner(tableCountSql(literal))).trim() - const tableCount = parseInt(rawTableCount, 10) - // An unreadable or non-numeric count (empty output, a driver notice, NaN) - // is not the same fact as a genuinely empty schema: `!tableCount` is true - // for both 0 and NaN, and collapsing them here is exactly the outage this - // guard exists to prevent - an unknown migration state waved through as - // "empty" instead of refused. Only a real, parseable zero counts as empty. - if (Number.isNaN(tableCount)) { - return { state: 'unreadable', reason: 'could not read a table count for ' + database + ' (got ' + JSON.stringify(rawTableCount) + ')' } - } - // No tables at all: either the database does not exist yet or it is - // untouched. A fresh install builds its schema from src/sql, which already - // carries the post-migration widths, so it cannot be behind. - if (tableCount === 0) return { state: 'empty-database' } - - const rawHasLedger = String(await runner( - tableExistsSql(literal, "'" + LEDGER_TABLE + "'"))).trim() - const hasLedger = parseInt(rawHasLedger, 10) - // Same collapse shape applies to the ledger-presence count: an unreadable - // or NaN result must refuse, not be read as "no ledger table". - if (Number.isNaN(hasLedger)) { - return { state: 'unreadable', reason: 'could not read whether ' + database + ' has a ' + LEDGER_TABLE + ' ledger (got ' + JSON.stringify(rawHasLedger) + ')' } - } - // Tables but no ledger: this database predates the migration runner, or is - // not the database we think it is. Either way its migration state is - // unknowable, which is the case this guard must not wave through. - if (!hasLedger) { - return { state: 'unreadable', reason: database + ' holds ' + tableCount + ' table(s) but no ' + LEDGER_TABLE + ' ledger' } - } - - const out = String(await runner(appliedMigrationsSql(database, LEDGER_TABLE))) - const applied = new Set(out.split('\n').map(s => s.trim()).filter(Boolean)) - return { state: 'ledger', applied } - } catch (err) { - return { state: 'unreadable', reason: (err && err.message) ? err.message : String(err) } - } -} - -function refusalMessage(module, coin, network, dbName, missing, remedy = {}) { - const container = getDockerContainerImageName(module, coin, network) - const files = missing.join(', ') - const plural = missing.length > 1 - - // The remedy runs on the build inside the container, which is the one being - // REPLACED. Only name the scoped command when that build was confirmed to - // honour --file; otherwise the command would quietly widen to every pending - // manual migration, so state that instead of printing it. - let instructions - if (remedy.supportsPerFile === true) { - instructions = 'apply ' + (plural ? 'them' : 'it') + - ' deliberately, with the writer quiesced, then re-run the update:\n' + - missing.map(f => ' docker exec -i ' + container + ' node ' + migrateCliPathFor(container) + ' --file ' + f).join('\n') - } else { - const wouldApply = (remedy.pendingManual && remedy.pendingManual.length) - ? remedy.pendingManual - : missing - instructions = 'DO NOT run `node ' + migrateCliPathFor(container) + '` inside ' + container + '. ' + - (remedy.supportsPerFile === false - ? 'That container runs a build with no per-file targeting: it ignores --file' - : 'Whether that container\'s build honours --file could not be read, and an unverified capability is not one: it may ignore --file') + - ' and apply EVERY pending manual migration on ' + dbName + ', which is ' + - wouldApply.length + ' file(s):\n' + - wouldApply.map(f => ' ' + f + (missing.includes(f) ? ' (the one you need)' : '')).join('\n') + '\n' + - ' Apply ' + (plural ? 'the needed files' : 'the needed file') + ' with a build that supports ' + - '--file, or apply the statement by hand with the writer quiesced, then re-run the update.' - } - - return 'update refused: the ' + module + ' source about to be deployed asserts migration' + - (plural ? 's' : '') + ' ' + files + ' at startup, but ' + dbName + - ' has not applied ' + (plural ? 'them' : 'it') + '. Deploying now replaces a working ' + - 'container with one that crash-loops on boot (the 2026-08-09 mainnet halt: all three indexers went to ' + - 'Restarting(1) on exactly this). These migrations are operator-gated on purpose - ' + - instructions + '\n' + - ' Take a fresh backup first: DEPLOY-ORDER.md says so for every migration-bearing deploy, ' + - 'and the coin boxes back up only WEEKLY. ' + - 'Set ' + SKIP_ENV + '=1 to override.' +function assertMigrationStateReadable(result, module, dbName, required) { + if (result.state === 'ledger') return + // Say which situation this is. "Apply the migration" would be advice this + // branch cannot justify: what failed is reading the ledger, not the ledger + // reporting a gap. + throw new Error( + `update refused: ${module} asserts migration(s) ${required.join(', ')} at startup, and whether ` + + `${dbName} has applied ${required.length > 1 ? 'them' : 'it'} could NOT be determined ` + + `(${result.reason}). This is an unknown-state refusal, not a known-missing migration. Check the ` + + `database is up and that this host can reach it, then re-run; set ${SKIP_ENV}=1 to override once ` + + `you know the schema is current.` + ) } /** @@ -357,24 +153,8 @@ async function assertRequiredMigrationsApplied(module, coin, network, branch = n const dbName = getModuleDatabaseName(module, coin, network) const result = await readApplied({ database: dbName, coin, network }) - if (result.state === 'empty-database') { - logger.warn(`Migration precondition guard: ${dbName} holds no tables yet, so ${module}'s gated ` + - `migrations (${required.join(', ')}) cannot be outstanding on it; proceeding.`) - return { checked: true, required, ok: true, reason: 'empty-database' } - } - - if (result.state !== 'ledger') { - // Say which situation this is. "Apply the migration" would be advice this - // branch cannot justify: what failed is reading the ledger, not the ledger - // reporting a gap. - throw new Error( - `update refused: ${module} asserts migration(s) ${required.join(', ')} at startup, and whether ` + - `${dbName} has applied ${required.length > 1 ? 'them' : 'it'} could NOT be determined ` + - `(${result.reason}). This is an unknown-state refusal, not a known-missing migration. Check the ` + - `database is up and that this host can reach it, then re-run; set ${SKIP_ENV}=1 to override once ` + - `you know the schema is current.` - ) - } + if (result.state === 'empty-database') return emptyDatabaseResult(module, dbName, required) + assertMigrationStateReadable(result, module, dbName, required) const missing = required.filter(f => !result.applied.has(f)) if (missing.length) { @@ -409,6 +189,6 @@ module.exports = { // Exported for the unit suite: the refusal path hinges on an unreachable // database returning `unreadable` rather than throwing past the guard, and // that is a property of the real driver call, not of a stub. - readAppliedMigrations: defaultReadAppliedMigrations, + readAppliedMigrations, assertRequiredMigrationsApplied } diff --git a/src/services/migration_precondition_service/migration_scan.js b/src/services/migration_precondition_service/migration_scan.js new file mode 100644 index 0000000..6c5adae --- /dev/null +++ b/src/services/migration_precondition_service/migration_scan.js @@ -0,0 +1,264 @@ +/********************************************************************* + * + * Copyright © 2025–2026 Dankest, LLC + * Based on XChain Platform by Dankest, LLC – https://dankest.llc + * + * SPDX-License-Identifier: AGPL-3.0-or-later + * + * This file is part of XChain Platform. Licensed under the GNU Affero + * General Public License v3.0 or later; see LICENSE.md. A commercial + * license (without AGPL source-disclosure terms) is available - + * contact legal@dankest.llc. + * + ********************************************************************/ + +const fs = require('fs') + +const { EXTERNAL_DB } = require('../../config') +const { tableCountSql, tableExistsSql } = require('../../db/information_schema') +const { appliedMigrationsSql } = require('../../db/migrations') +const { migrationFiles } = require('../../utils/migration_files') +const { getDockerContainerImageName } = require('../config_service') +const { readMigrateCli, migrateCliPathFor } = require('../../utils/indexer_migrate_cli') + +const SKIP_ENV = 'XCHAIN_NODE_SKIP_MIGRATION_PRECONDITION' +const LEDGER_TABLE = 'schema_migrations' + +/** + * Does this migration file's header declare itself a deploy precondition? + * + * Prologue-anchored: the scan stops at the first non-blank, non-comment line, so + * the token can only arm the flag from the leading comment block and never from + * body prose or a data literal. Widening that to the whole file is how a + * migration that merely DISCUSSES the convention would start refusing deploys. + * + * Twin of xchain-indexer's Database.migrationDeclaresDeployPrecondition. It is + * duplicated rather than shared because this tool reads these files out of a + * source tree it has only cloned, with that tree's dependencies uninstalled, so + * requiring the module is not available to it. Keep the two in step. + */ +function migrationDeclaresDeployPrecondition(raw) { + const prologue = [] + for (const line of String(raw).split('\n')) { + const trimmed = line.trim() + if (trimmed === '' || trimmed.startsWith('--')) { prologue.push(line); continue } + break + } + return /^\s*--\s*xchain:migration\b[^\n]*\bdeploy-precondition\s*=\s*required\b/im.test(prologue.join('\n')) +} + +/** + * The `mode=` a migration header declares, or null when it declares none. + * Prologue-anchored exactly like migrationDeclaresDeployPrecondition, so a token + * in body prose or a data literal cannot answer for the file. + * + * Twin of the modules' own Database._migrationMode, duplicated for the reason + * given above: this tool reads a cloned tree whose dependencies are not + * installed. Keep them in step. + */ +function migrationMode(raw) { + const prologue = [] + for (const line of String(raw).split('\n')) { + const trimmed = line.trim() + if (trimmed === '' || trimmed.startsWith('--')) { prologue.push(line); continue } + break + } + const m = prologue.join('\n').match(/^\s*--\s*xchain:migration\b[^\n]*\bmode\s*=\s*([A-Za-z]+)/im) + return m ? m[1].toLowerCase() : null +} + +/** + * Every gated (mode=manual) migration in `dir` that the ledger has not recorded, + * sorted. This is the blast radius of an UNSCOPED migrate run against that + * database: the runner applies every pending manual file, not just the one an + * operator names. The refusal names that whole set, so the consequence is on + * screen rather than left for the operator to discover. + */ +function pendingManualMigrations(dir, applied) { + return migrationFiles(dir).filter(([f, file]) => { + if (applied && applied.has(f)) return false + try { + return migrationMode(fs.readFileSync(file, 'utf8')) === 'manual' + } catch { + return false + } + }).map(([f]) => f) +} + +/** + * Does the build CURRENTLY RUNNING in the target container understand per-file + * migration targeting (`--file`)? + * + * This matters because the remedy an operator is about to run executes inside + * that container, on its build, not on the one being deployed. A build without + * the flag does not reject it: it ignores it and applies every pending manual + * migration, which on a live database can mean a data backfill and a + * dedup-then-unique nobody authorised. + * + * Returns true, false, or null when the container could not be read at all + * (stopped, absent, docker unreachable). Callers must treat null like false: + * an unverified capability is not a capability, and the cost of being wrong is + * asymmetric. + */ +async function runningBuildSupportsPerFileMigrations(container, deps = {}) { + try { + const cat = deps.getDockerContainerFileCat || require('../docker_service').getDockerContainerFileCat + const found = await readMigrateCli(cat, container) + return found ? /['"]--file['"]/.test(found.source) : null + } catch { + return null + } +} + +/** + * Every migration filename in `dir` whose header declares a deploy precondition, + * sorted. A missing directory yields [] - a module (or a ref) with no migrations + * declares no preconditions, which is not an error. + */ +function listDeployPreconditionMigrations(dir) { + return migrationFiles(dir).filter(([, file]) => { + try { + return migrationDeclaresDeployPrecondition(fs.readFileSync(file, 'utf8')) + } catch { + return false + } + }).map(([f]) => f) +} + +function databaseLiteral(database) { + // The name comes from getModuleDatabaseName, but it reaches SQL as text (an + // identifier cannot be bound), so gate it on the same allowlist the + // provisioning DDL uses rather than trusting its provenance. + return /^[A-Za-z0-9_]+$/.test(String(database)) ? "'" + database + "'" : null +} + +/** + * Read the applied-migration ledger of one module database. + * + * Returns { state: 'ledger', applied: Set } when the ledger was read, + * { state: 'empty-database' } when the schema holds no tables at all (or does + * not exist yet), and { state: 'unreadable', reason } for everything else. The + * three are deliberately distinct: only the middle one is safe to proceed on. + * + * WHY ROOT AND NOT THE MODULE'S OWN ACCOUNT + * ----------------------------------------- + * The first cut of this guard connected with the module's generated credentials + * (INDEXER_DB_USER/PASS) over the published port. Run against the live regtest + * stack it produced `Access denied for user 'xchain_indexer_bitcoin_regtest'`, + * i.e. an unknown-state REFUSAL of a perfectly deployable update - the sidecar + * password and the live account had drifted, which is a documented recurring + * condition here and has nothing to do with migrations. A guard that fails + * closed on a routine credential drift blocks every deploy and gets switched + * off. So this uses the same root-credential runner every other DB read in + * xchain-node uses (see clearHubPriceIngestWatermark), which the update path + * already resolves non-interactively for its own credential parity pass. + */ +async function readAppliedMigrations({ database, coin, network }, deps = {}) { + const literal = databaseLiteral(database) + if (!literal) return { state: 'unreadable', reason: 'refusing to query a database name that is not a plain identifier' } + + let runner = deps.runner + try { + if (!runner) { + const { + getExternalDbConfig, executeNativeMariaDbCommand, + executeDockerMariaDbCommand, askMariadbRootPassword, getDatabaseContainerId + } = require('../database_service') + if (EXTERNAL_DB) { + const cfg = await getExternalDbConfig() + runner = (sql) => executeNativeMariaDbCommand(cfg, sql, '-B -N') + } else { + const containerId = await getDatabaseContainerId() + if (!containerId) return { state: 'unreadable', reason: 'no MariaDB container found on this host' } + const rootPassword = await askMariadbRootPassword(coin, network) + runner = (sql) => executeDockerMariaDbCommand(containerId, rootPassword, sql, '-B -N') + } + } + + const rawTableCount = String(await runner(tableCountSql(literal))).trim() + const tableCount = parseInt(rawTableCount, 10) + // An unreadable or non-numeric count (empty output, a driver notice, NaN) + // is not the same fact as a genuinely empty schema: `!tableCount` is true + // for both 0 and NaN, and collapsing them here is exactly the outage this + // guard exists to prevent - an unknown migration state waved through as + // "empty" instead of refused. Only a real, parseable zero counts as empty. + if (Number.isNaN(tableCount)) { + return { state: 'unreadable', reason: 'could not read a table count for ' + database + ' (got ' + JSON.stringify(rawTableCount) + ')' } + } + // No tables at all: either the database does not exist yet or it is + // untouched. A fresh install builds its schema from src/sql, which already + // carries the post-migration widths, so it cannot be behind. + if (tableCount === 0) return { state: 'empty-database' } + + const rawHasLedger = String(await runner( + tableExistsSql(literal, "'" + LEDGER_TABLE + "'"))).trim() + const hasLedger = parseInt(rawHasLedger, 10) + // Same collapse shape applies to the ledger-presence count: an unreadable + // or NaN result must refuse, not be read as "no ledger table". + if (Number.isNaN(hasLedger)) { + return { state: 'unreadable', reason: 'could not read whether ' + database + ' has a ' + LEDGER_TABLE + ' ledger (got ' + JSON.stringify(rawHasLedger) + ')' } + } + // Tables but no ledger: this database predates the migration runner, or is + // not the database we think it is. Either way its migration state is + // unknowable, which is the case this guard must not wave through. + if (!hasLedger) { + return { state: 'unreadable', reason: database + ' holds ' + tableCount + ' table(s) but no ' + LEDGER_TABLE + ' ledger' } + } + + const out = String(await runner(appliedMigrationsSql(database, LEDGER_TABLE))) + const applied = new Set(out.split('\n').map(s => s.trim()).filter(Boolean)) + return { state: 'ledger', applied } + } catch (err) { + return { state: 'unreadable', reason: (err && err.message) ? err.message : String(err) } + } +} + +function refusalMessage(module, coin, network, dbName, missing, remedy = {}) { + const container = getDockerContainerImageName(module, coin, network) + const files = missing.join(', ') + const plural = missing.length > 1 + + // The remedy runs on the build inside the container, which is the one being + // REPLACED. Only name the scoped command when that build was confirmed to + // honour --file; otherwise the command would quietly widen to every pending + // manual migration, so state that instead of printing it. + let instructions + if (remedy.supportsPerFile === true) { + instructions = 'apply ' + (plural ? 'them' : 'it') + + ' deliberately, with the writer quiesced, then re-run the update:\n' + + missing.map(f => ' docker exec -i ' + container + ' node ' + migrateCliPathFor(container) + ' --file ' + f).join('\n') + } else { + const wouldApply = (remedy.pendingManual && remedy.pendingManual.length) + ? remedy.pendingManual + : missing + instructions = 'DO NOT run `node ' + migrateCliPathFor(container) + '` inside ' + container + '. ' + + (remedy.supportsPerFile === false + ? 'That container runs a build with no per-file targeting: it ignores --file' + : 'Whether that container\'s build honours --file could not be read, and an unverified capability is not one: it may ignore --file') + + ' and apply EVERY pending manual migration on ' + dbName + ', which is ' + + wouldApply.length + ' file(s):\n' + + wouldApply.map(f => ' ' + f + (missing.includes(f) ? ' (the one you need)' : '')).join('\n') + '\n' + + ' Apply ' + (plural ? 'the needed files' : 'the needed file') + ' with a build that supports ' + + '--file, or apply the statement by hand with the writer quiesced, then re-run the update.' + } + + return 'update refused: the ' + module + ' source about to be deployed asserts migration' + + (plural ? 's' : '') + ' ' + files + ' at startup, but ' + dbName + + ' has not applied ' + (plural ? 'them' : 'it') + '. Deploying now replaces a working ' + + 'container with one that crash-loops on boot (the 2026-08-09 mainnet halt: all three indexers went to ' + + 'Restarting(1) on exactly this). These migrations are operator-gated on purpose - ' + + instructions + '\n' + + ' Take a fresh backup first: DEPLOY-ORDER.md says so for every migration-bearing deploy, ' + + 'and the coin boxes back up only WEEKLY. ' + + 'Set ' + SKIP_ENV + '=1 to override.' +} + +module.exports = { + migrationDeclaresDeployPrecondition, + migrationMode, + pendingManualMigrations, + runningBuildSupportsPerFileMigrations, + listDeployPreconditionMigrations, + readAppliedMigrations, + refusalMessage +} From 90de5a1beab5302ae37abc56127f01ebb8e69953 Mon Sep 17 00:00:00 2001 From: J-Dog Date: Thu, 17 Sep 2026 11:28:13 -0700 Subject: [PATCH 03/35] fix(migrations): restore default reader wiring Use the migration scan reader when callers omit injected dependencies. Add a regression test that exercises the production default path. --- .../migration_precondition_service.js | 2 +- .../migration_precondition_service.test.js | 24 +++++++++++++++++++ 2 files changed, 25 insertions(+), 1 deletion(-) diff --git a/src/services/migration_precondition_service.js b/src/services/migration_precondition_service.js index 3a3eed7..945d220 100644 --- a/src/services/migration_precondition_service.js +++ b/src/services/migration_precondition_service.js @@ -131,7 +131,7 @@ async function assertRequiredMigrationsApplied(module, coin, network, branch = n const cloneGitDep = deps.cloneGit || require('./module_service').cloneGit const listRequired = deps.listDeployPreconditionMigrations || listDeployPreconditionMigrations - const readApplied = deps.readAppliedMigrations || defaultReadAppliedMigrations + const readApplied = deps.readAppliedMigrations || readAppliedMigrations // Clone the target source and read ITS migrations: the constraint must come // from the code that is about to run. The tmp tree is NOT reused from the skew diff --git a/test/unit/migration_precondition_service.test.js b/test/unit/migration_precondition_service.test.js index 097514b..807100a 100644 --- a/test/unit/migration_precondition_service.test.js +++ b/test/unit/migration_precondition_service.test.js @@ -13,6 +13,7 @@ const fs = require('fs') const os = require('os') const path = require('path') +const proxyquire = require('proxyquire').noCallThru() const sinon = require('sinon') const { expect } = require('chai') @@ -279,6 +280,29 @@ describe('MigrationPreconditionService', () => { describe('assertRequiredMigrationsApplied', () => { + it('uses the moved default migration reader when no deps are passed', async () => { + const cloneGit = sinon.stub().resolves() + const listRequired = sinon.stub().returns([GATED]) + const readApplied = sinon.stub().resolves({ state: 'ledger', applied: new Set([GATED]) }) + const migrationScan = require('../../src/services/migration_precondition_service/migration_scan') + const service = proxyquire('../../src/services/migration_precondition_service', { + './module_service': { cloneGit }, + './migration_precondition_service/migration_scan': { + ...migrationScan, + listDeployPreconditionMigrations: listRequired, + readAppliedMigrations: readApplied + } + }) + + const res = await service.assertRequiredMigrationsApplied( + XChainService.XCHAIN_INDEXER, 'bitcoin', 'mainnet', 'master') + + expect(res.ok).to.equal(true) + expect(cloneGit.calledOnce).to.equal(true) + expect(listRequired.calledOnce).to.equal(true) + expect(readApplied.calledOnce).to.equal(true) + }) + it('is inert for a module that ships no migrations', async () => { const deps = makeDeps() const res = await assertRequiredMigrationsApplied(XChainService.XCHAIN_ENCODER, 'bitcoin', 'mainnet', 'master', deps) From 2aaba8ff78adb3562d5090bbd835c248d3766beb Mon Sep 17 00:00:00 2001 From: J-Dog Date: Thu, 17 Sep 2026 10:53:01 -0700 Subject: [PATCH 04/35] refactor(operations): split module operation seams --- src/operations/module_operations.js | 1565 +---------------- .../module_operations/module_controls.js | 277 +++ .../module_operations/recreate_modules.js | 89 + src/operations/module_operations/repo_refs.js | 170 ++ .../module_operations/reset_modules.js | 387 ++++ .../module_operations/shared_services.js | 119 ++ .../module_operations/uninstall_modules.js | 105 ++ .../module_operations/update_modules.js | 236 +++ 8 files changed, 1451 insertions(+), 1497 deletions(-) create mode 100644 src/operations/module_operations/module_controls.js create mode 100644 src/operations/module_operations/recreate_modules.js create mode 100644 src/operations/module_operations/repo_refs.js create mode 100644 src/operations/module_operations/reset_modules.js create mode 100644 src/operations/module_operations/shared_services.js create mode 100644 src/operations/module_operations/uninstall_modules.js create mode 100644 src/operations/module_operations/update_modules.js diff --git a/src/operations/module_operations.js b/src/operations/module_operations.js index 3c905bf..2f72364 100644 --- a/src/operations/module_operations.js +++ b/src/operations/module_operations.js @@ -12,9 +12,11 @@ * ********************************************************************** * XChain Node - Module Operations - * Bulk operations over lists of modules (install, start, stop, etc.) + * Public facade for bulk operations over lists of modules. ********************************************************************/ +'use strict' + const path = require('path') const fs = require('fs') const readline = require('readline') @@ -24,8 +26,8 @@ const execFileAsync = promisify(execFile) const { NODE_MODULE_NAME, DB_MODULE_NAME, HUB_MODULE_NAME, EXPLORER_MODULE_NAME, SYNC_MODULE_NAME, XChainService, SEP, dataDir, EXTERNAL_DB, Coin, CoinTickerSymbol, Network, DEFAULT_MODULE_BRANCH } = require('../config') const { db } = require('../state') const { sleep } = require('../utils/helpers') -const { getDockerContainerImageName, getUtxoTrackerVolumeName, filterCommandParameters, getDockerNetwork } = require('../services/config_service') -const { createDockerNetwork, killContainer, removeContainer, probeContainerPresenceByName, stopContainer, stopContainerByName, startContainer, restartContainer, execContainer, shellContainer, logContainer, startDockerMonitor, waitContainer, saveContainerLogs, getContainerBindMounts } = require('../services/docker_service') +const { getDockerContainerImageName, getUtxoTrackerVolumeName, getDockerNetwork } = require('../services/config_service') +const { createDockerNetwork, probeContainerPresenceByName, stopContainer, stopContainerByName, startContainer, restartContainer, execContainer, shellContainer, logContainer, startDockerMonitor, waitContainer, saveContainerLogs, getContainerBindMounts, removeContainer } = require('../services/docker_service') const { stopModuleContainer } = require('../services/stop_budget_service') const { buildDatabaseModule, resetDatabases, clearHubPriceIngestWatermark, purgeHubCrossChainRows, manualHubCrossChainPurgeStatements, getDatabaseContainerId, pingExternalDatabase } = require('../services/database_service') const { getModuleBranch, installModule, uninstallModule } = require('../services/module_service') @@ -33,10 +35,10 @@ const { assertHubNotBehind } = require('../services/skew_guard_service') const { assertRequiredMigrationsApplied } = require('../services/migration_precondition_service') const { statusChanged } = require('../services/status_service') const { reindexAffectedModules, recordReindex } = require('../services/bootstrap_republish_ledger') -const config = require('../config'); -// The services the operations below reach into. Each is bound here as a module -// object and destructured inside the function that uses it, so a call reads the -// export as it stands at that moment, exactly as a require at the call site does. +const config = require('../config') + +// These service objects stay intact so tests and callers that replace an export +// after this facade loads are observed at the original call sites. const bootstrapService = require('../services/bootstrap_service') const databaseService = require('../services/database_service') const explorerService = require('../services/explorer_service') @@ -49,1496 +51,65 @@ const stateModule = require('../state') const validatorService = require('../services/validator_service') const versionService = require('../services/version_service') -// Resolve the operator's single ref slot into an install target and publish it -// for the duration of the run, so every module clone and every bundled-library -// staging inside it resolves against ONE decision (release-management spec -// section 11). Cleared in a finally, or a later branch install in the same -// process would inherit a stale pin. -async function withInstallTarget(ref, run, { fallbackToBranch = true } = {}) { - const { - resolveInstallTarget, setActiveTarget, clearActiveTarget - } = releaseManifestService - const { recordInstallTarget } = installTargetService - - const target = await resolveInstallTarget(ref, { defaultBranch: DEFAULT_MODULE_BRANCH, fallbackToBranch }) - - if (target.kind === 'release') { - console.log(`Installing XChain ${target.tag} (${target.resolvedFrom}); every component is manifest-pinned.`) - } else { - console.log(`Installing from branch '${target.ref}' (UNRELEASED: tracking install, no version pinning).`) - } - - // What this node is on, for the next no-ref `update` to converge on. - recordInstallTarget(target) - setActiveTarget(target) - try { - return await run(target) - } finally { - clearActiveTarget() - } -} - -/** - * Install every requested module and REPORT what was actually built. - * - * installModule returns false for a module it decided not to touch (already - * installed, or a singleton container that a previous coin/network pass in this - * same run already created). That return used to be dropped on the floor, so - * "built six containers" and "built nothing" printed the same and exited the - * same. Unlike `update`, a no-op install is NOT a failure - the desired state - * already holds, and `install` is run idempotently by scripts and harnesses - - * so the report is printed rather than turned into a non-zero exit. - * - * @returns {Promise<{installed: Array, skipped: Array}>} - */ -async function installModules(servicesList, ref = null) { - return withInstallTarget(ref, async (target) => { - // A release install passes no branch: resolveComponentRef inside - // installModule supplies the pinned ref per component. A branch install - // passes the branch, exactly as before. - const branch = target.kind === 'release' ? null : target.ref - const outcome = { installed: [], skipped: [] } - // Per-run, so a second install in the same process reports its own - // restores rather than replaying the first one's. - bootstrapService.resetBootstrapOutcomes() - - try { - for (const nextCoin in servicesList) { - for (const nextNetwork in servicesList[nextCoin]) { - if (nextCoin && nextNetwork) { - await createDockerNetwork(getDockerNetwork(nextCoin, nextNetwork)) - await buildDatabaseModule(nextCoin, nextNetwork) - } - for (const nextModule of servicesList[nextCoin][nextNetwork]) { - const result = await installModule(nextModule, nextCoin, nextNetwork, false, null, false, branch) - if (result === false) { - outcome.skipped.push({ module: nextModule, coin: nextCoin, network: nextNetwork, reason: 'already-installed' }) - } else { - outcome.installed.push({ module: nextModule, coin: nextCoin, network: nextNetwork }) - } - } - } - } - - if (outcome.skipped.length > 0) { - console.log('install: nothing to do for ' + outcome.skipped - .map(s => `${s.module} (${s.coin} ${s.network})`).join(', ') - + ' - already installed. Use `update` to rebuild.') - } - } finally { - // In a finally because a run that throws is the one whose summary - // matters most: it leaves some services restored and some facing - // hours of resync, and the error alone does not say which. - bootstrapService.reportBootstrapOutcomes() - } - - // The explorer is installed in the shared bucket, which runs BEFORE the - // coin stacks, and it learns its coins by polling the hub. So a run that - // installed a coin leaves it serving 503 for up to a poll interval after - // this loop ends. Returning there hands every caller a stack that reports - // installed and answers nothing; the first one to be bitten was the e2e - // gate, whose suite starts the moment install returns. - return outcome - }) -} - -// Make the coins this run installed usable before the command returns. -// -// updateHub and updateExplorer push coin config to the hub and JOIN the hub and -// explorer containers to each coin's docker network. They run in preCheck, which -// fires BEFORE the action, so an install that creates brand-new coin stacks ends -// without either shared service having heard about them: the explorer sits on no -// network from which the hub is reachable, never populates a DB pool, and answers -// 503 until some later command's preCheck happens to fix it. Measured on a clean -// host, it stayed degraded through a full 150-second readiness wait. -// -// This is a COMMAND-level step, not part of the install primitive: it reconciles -// against live docker, and installModules is also driven directly by suites whose -// container registry is fixture data that such a reconcile would purge. -// -// Returns whether the stack is usable. The modules are installed either way, but -// reporting success for a stack whose explorer serves 503 makes every later -// failure land on the caller's first read instead of here. -async function syncSharedServicesAfterInstall(outcome) { - if (!outcome || !outcome.installed.some(i => i.coin && i.network)) return true - - const { updateHub } = hubService - const { updateExplorer, waitForExplorerReady } = explorerService - - try { await updateHub() } catch (err) { console.warn('install: could not push config to the hub: ' + err) } - try { await updateExplorer() } catch (err) { console.warn('install: could not attach the explorer to the new coin networks: ' + err) } - - if (await waitForExplorerReady()) return true - - console.warn('install: the xchain-explorer is still not serving coin data.' + - ' The stack is installed; the explorer either cannot reach the hub or the hub' + - ' has no config for these coins yet. Check it before running anything that reads it.') - - // Escape hatch for the install-then-fix flows: the modules ARE installed, so a - // caller that intends to repair the explorer by hand can still treat this as success. - if (allowDegradedExplorer()) { - console.warn('install: continuing anyway (XCHAIN_NODE_ALLOW_DEGRADED_EXPLORER is set).') - return true - } - return false -} - -// Opt-out for callers that knowingly accept a stack whose explorer serves no coins. -function allowDegradedExplorer() { - return ['1', 'true', 'yes'].includes(String(config.XCHAIN_NODE_ALLOW_DEGRADED_EXPLORER).toLowerCase()) -} - -/** - * Update the requested modules. - * - * The ref decides the mode, the same way it does for `install`: - * - a release ref (`vX.Y.Z`): pinned update to that train; - * - a branch name: tracking update, every module moved to that branch's tip; - * - no ref: whatever kind of node this is. A release node (the operator - * path, and the only kind a default `install` produces) moves to the - * LATEST published release, fully pinned; a branch node stays on its - * branch and takes newer commits. - * - * Until 2026-09 a no-ref update re-read each module's git branch. A pinned - * checkout is detached, so that read answered `HEAD` and every release - * node failed its own documented upgrade command. `update all` is the - * command operators are told to run, so it has to mean "take me to the - * newest release" on the node an operator has. - * - * `opts.all` marks a run that expanded from `all`: the shared services join - * it (hub first, then sync) and a coin node whose pinned binary has not - * changed is left running rather than rebuilt. - */ -async function updateModules(servicesList, ref = null, opts = {}) { - const { isReleaseRef } = releaseManifestService - const { recordInstallTarget, resolveUpdateTarget } = installTargetService - - const list = opts.all ? includeSharedServicesForUpdate(servicesList) : servicesList - const runOpts = { skipCurrentNode: !!opts.all, quietNotInstalled: !!opts.all } - - await repairValidatorConfigBeforeHubUpdate(list) - - if (isReleaseRef(ref)) { - return withInstallTarget(ref, async () => updateModulesOnBranch(list, null, runOpts)) - } - - if (ref) { - // An explicitly named branch is a decision about what this node is. - recordInstallTarget({ kind: 'branch', ref }) - return updateModulesOnBranch(list, ref, runOpts) - } - - // No ref: the update target is remembered from the last install/update, - // or classified from the checkouts on a node an older CLI installed. - const target = process.env.XCHAIN_NODE_UPDATE_TARGET - ? { kind: 'release', ref: process.env.XCHAIN_NODE_UPDATE_TARGET, inferred: false } - : await resolveUpdateTarget() - - if (target.kind === 'branch') { - console.log(`This node tracks branch '${target.ref}'${target.inferred ? ' (classified from its checkouts)' : ''}; updating to its newest commits. Name a release (e.g. \`update all v0.15.2\`) to move it onto a release.`) - recordInstallTarget({ kind: 'branch', ref: target.ref }) - return updateModulesOnBranch(list, target.ref, runOpts) - } - - // A release node with no ref: the LATEST release (the recorded tag is - // where the node is, not where it is going), never a branch fallback. A - // lookup failure stops the run with nothing changed. The re-executed - // child of a CLI self-update already knows the tag its parent resolved. - const releaseRef = process.env.XCHAIN_NODE_UPDATE_TARGET || null - return withInstallTarget(releaseRef, async () => updateModulesOnBranch(list, null, runOpts), { fallbackToBranch: false }) -} - -/** - * On a validator, an update that rebuilds the hub first re-runs the validator - * repair path (`validator init` over an initialized node): additive only, it - * fills in what a newer version added (a recorded network, wallets, the - * publisher config) and never touches the signing key, the stake or an - * existing hub API key. The hub mounts that config, so this runs BEFORE the - * rebuild. This replaces the manual `validator init` re-run the docs asked - * for after every upgrade. - * - * A repair failure is reported and does not stop the update: the hub still - * boots on the config it has, which is what it ran on before. - */ -async function repairValidatorConfigBeforeHubUpdate(servicesList) { - const shared = (servicesList[""] && servicesList[""][""]) || [] - if (!shared.includes(HUB_MODULE_NAME)) return false - const { isInitialized, initValidator } = validatorService - let initialized = false - try { initialized = isInitialized() } catch { return false } - if (!initialized) return false - try { - console.log('This node is a validator; checking its config for anything a newer version added...') - await initValidator({}) - return true - } catch (err) { - console.warn(`Could not repair the validator config (${err && err.message ? err.message : err}); the hub is updated on its existing config.`) - return false - } -} - -/** - * `all` for an UPDATE includes the shared services, hub first. - * - * filterCommandParameters leaves the hub and sync out of `all` because the - * same expansion serves install/start/stop, where "all" has never meant the - * hub. For update it must: the hub is the first thing a release moves, and - * the docs' promise that "the hub is updated first, automatically" was only - * ever true of a hub that was missing (preCheck installs one) and never of - * a hub that was running. Rebuilt with the shared bucket first so iteration - * order IS the deploy order: hub, sync, explorer, then the coin stacks. - */ -function includeSharedServicesForUpdate(servicesList) { - const shared = (servicesList[""] && servicesList[""][""]) || [] - const ordered = [HUB_MODULE_NAME, SYNC_MODULE_NAME, ...shared.filter(m => m !== HUB_MODULE_NAME && m !== SYNC_MODULE_NAME)] - const rest = {} - for (const coin of Object.keys(servicesList)) { - if (coin !== "") rest[coin] = servicesList[coin] - } - return { "": { "": ordered }, ...rest } -} - -/** - * Record ONE installModule call in an update outcome. - * - * installModule returns false when it declined to touch the module (its - * early-return paths) and a container id / true when it built one. Counting - * every call as "updated" regardless - which is what the loop used to do with - * that return value - is how a run that rebuilt nothing still reported a - * landed deploy and exited 0. - */ -function recordInstallOutcome(outcome, result, module, coin, network) { - if (result === false) { - console.warn(`update: ${module} (${coin} ${network}) was not rebuilt; nothing changed for it.`) - outcome.skipped.push({ module, coin, network, reason: 'no-op' }) - } else { - outcome.updated.push({ module, coin, network }) - } -} - -/** - * Runs the update over every requested module and REPORTS what it did. - * - * The report exists because the old `return true` made "updated three - * containers" and "matched nothing at all" indistinguishable to the caller, so - * a run that changed nothing still exited 0 and read as a landed deploy. The - * caller (cli `update`) turns an empty `updated` list into a non-zero exit. - * - * @returns {Promise<{updated: Array, skipped: Array}>} - */ -async function updateModulesOnBranch(servicesList, branch = null, { skipCurrentNode = false, quietNotInstalled = false } = {}) { - const outcome = { updated: [], skipped: [] } - try { - await updateModulesInto(outcome, servicesList, branch, { skipCurrentNode, quietNotInstalled }) - } finally { - // Under `all`, a service that is not installed is the ordinary case (a - // validator has a hub and nothing else), so the per-service warnings - // collapse into one line; the outcome still lists every one of them. - const absent = outcome.skipped.filter(s => s.reason === 'not-installed') - if (quietNotInstalled && absent.length > 0) { - console.log(`update: skipped ${absent.length} service${absent.length === 1 ? '' : 's'} not installed on this node.`) - } - } - return outcome -} - -async function updateModulesInto(outcome, servicesList, branch, { skipCurrentNode, quietNotInstalled }) { - for (const nextCoin in servicesList) { - for (const nextNetwork in servicesList[nextCoin]) { - for (const nextModule of servicesList[nextCoin][nextNetwork]) { - if (nextModule === DB_MODULE_NAME) { - // `update` cannot rebuild the database. Its container is created by - // buildDatabaseModule from a pinned mariadb image, not from module - // source, and the existing-container branch there does nothing at - // all - yet the DB branch of installModule answered a hard `true`, - // which recordInstallOutcome counts as an updated module. So - // `update database` exited 0 reporting a landed upgrade over an - // untouched container. Refuse it here, where the update contract - // lives, and state the remediation uninstallModule already names. - console.warn(`update: ${nextModule} (${nextCoin} ${nextNetwork}) is not rebuilt by update; the database container must be removed manually and reinstalled.`) - outcome.skipped.push({ module: nextModule, coin: nextCoin, network: nextNetwork, reason: 'not-updatable' }) - continue - } - const moduleContainerId = await db.getModuleContainer(nextModule, nextCoin, nextNetwork) - if (nextModule === NODE_MODULE_NAME) { - // The running node is deliberately left alone here. buildCryptoNode - // stops it gracefully (SIGTERM with a flush budget) and force-removes - // the stopped carcass right before its `docker run --name`, so the - // daemon keeps serving through the download and image build and its - // block index is flushed before it goes. An up-front `docker rm -f` - // at this point was SIGKILL: the killed daemon came back at its last - // flushed index (16 regtest blocks lost, 2026-09-03), and it also - // hid the old container from buildCryptoNode's bind-mount drift guard. - // - // Recreate even when the container was missing from the registry: - // the old `if (!moduleContainerId) continue` made `update node` a - // silent no-op (exit 0, nothing created) once the node had crashed or - // been removed; only `install master node` could bring it back. - // installModule's remoteUpdate path rebuilds it from local source. - // - // That recreate-when-missing rule is for a TARGETED `update node`. - // Under `all`, a coin/network with no node is simply not installed - // here, like any other absent service: measured on a hub-only - // sandbox 2026-09-08, `update all` otherwise set about installing - // a Bitcoin mainnet daemon and failed on its missing network. - if (!moduleContainerId && quietNotInstalled) { - outcome.skipped.push({ module: nextModule, coin: nextCoin, network: nextNetwork, reason: 'not-installed' }) - continue - } - // - // Under `update all` a coin node whose pinned binary has not - // changed is left running. Rebuilding it anyway restarts a - // daemon that was serving fine, and one such rebuild broke a - // relocated datadir's mounts and halted a tracker. - // A targeted `update node ` still rebuilds. - if (skipCurrentNode && moduleContainerId && await coinNodeIsCurrent(nextCoin, nextNetwork)) { - console.log(`update: ${nextModule} (${nextCoin} ${nextNetwork}) already runs the pinned daemon version; left running.`) - outcome.skipped.push({ module: nextModule, coin: nextCoin, network: nextNetwork, reason: 'current' }) - continue - } - const built = await installModule(nextModule, nextCoin, nextNetwork, true, null) - recordInstallOutcome(outcome, built, nextModule, nextCoin, nextNetwork) - } else { - if (!moduleContainerId) { - // Skipping is still right for `update all` on a partly - // installed stack, but skipping SILENTLY is what let a - // targeted `update ` print nothing, - // change nothing and exit 0. Say it, and record it so - // the caller can fail a run that updated nothing. - if (!quietNotInstalled) { - console.warn(`update: ${nextModule} (${nextCoin} ${nextNetwork}) has no registered container; nothing to update (install it first if you expected it here).`) - } - outcome.skipped.push({ module: nextModule, coin: nextCoin, network: nextNetwork, reason: 'not-installed' }) - continue - } - let moduleBranch = branch - if (!moduleBranch) { - try { moduleBranch = await getModuleBranch(nextModule) } catch { /* use default */ } - // A pinned checkout is detached and answers `HEAD`, which - // is not a branch anything can clone. Under a release - // update the manifest pin decides the ref anyway; for a - // component the manifest does not carry, null means the - // default branch, the same thing `install` would do. - if (moduleBranch === 'HEAD') moduleBranch = null - } - // remoteUpdate=true so installModule actually rebuilds the - // container. Without it, the `if (!containerNodeVersion || - // remoteUpdate)` guard short-circuits for any already-installed - // service and `update` becomes a silent no-op. - // - // Version-skew guard: a hub-dependent service whose new - // source declares `xchainRequiresHub` in its package.json is - // REFUSED when the installed hub is behind that version, before - // anything is torn down. Throws out of updateModules so the - // update fails closed with nothing modified for this module. - // Under a pinned update the guard must read the PINNED - // source's package.json, not the branch tip: it clones into - // a tmp tree to find `xchainRequiresHub`, and reading that - // from a different ref than the one about to be installed - // is how a skew guard blesses a version it never saw. - const { resolveComponentRef } = releaseManifestService - const pin = resolveComponentRef(nextModule, moduleBranch) - await assertHubNotBehind(nextModule, pin.ref) - // Migration-precondition guard: a service whose new source asserts a - // GATED (mode=manual) migration at startup is REFUSED when the database - // it will use has not applied that migration, before anything is torn - // down. Without it the only thing that discovers the requirement is the - // recreated container crash-looping - which is exactly how a routine - // indexer deploy took all three mainnet indexers down on 2026-08-09. - // Reads the same PINNED ref as the skew guard above, for the same - // reason: a precondition read from a different ref than the one being - // installed is a check that blessed a version it never saw. - await assertRequiredMigrationsApplied(nextModule, nextCoin, nextNetwork, pin.ref) - // moduleBranch MUST be threaded through: installModule re-clones the - // module on the remoteUpdate path (cloneGit with this `branch`), so a - // null branch here re-clones the default branch and clobbers the branch - // the operator asked for (the cause of `update - // ` silently deploying master). installModule does the clone, so - // no separate cloneGit is needed here. - const rebuilt = await installModule(nextModule, nextCoin, nextNetwork, true, moduleContainerId, false, moduleBranch) - recordInstallOutcome(outcome, rebuilt, nextModule, nextCoin, nextNetwork) - } - } - } - } - return outcome -} - -/** - * Does the running coin node already carry the daemon version an update - * would install? The container writes its version file at build time (a - * bare `28.1`); the pinned release comes back tagged (`v28.1`). Any doubt - * answers false, so the rebuild the operator could always get still happens. - */ -async function coinNodeIsCurrent(coin, network) { - try { - const { getLastStatus, getRemoteModuleVersions } = stateModule - const { checkRemoteNodeVersion } = versionService - const running = getLastStatus()?.[coin]?.[network]?.[NODE_MODULE_NAME]?.["container_version"] - if (!running) return false - if (!(NODE_MODULE_NAME + SEP + coin in getRemoteModuleVersions())) { - await checkRemoteNodeVersion(coin) - } - const pinned = getRemoteModuleVersions()[NODE_MODULE_NAME + SEP + coin]?.["tag_name"] - if (!pinned) return false - const strip = v => String(v).trim().replace(/^v/, '') - return strip(running) === strip(pinned) - } catch { - return false - } -} - -// Modules whose container is not created from the config map by buildAndUp: the -// crypto node goes through buildCryptoNode and the database container through -// buildDatabaseModule, so neither has config env for this verb to re-stamp. -const RECREATE_UNSUPPORTED_MODULES = [NODE_MODULE_NAME, DB_MODULE_NAME] - -/** - * Re-stamp a service's container from the CURRENT config without touching its image. - * - * A container freezes its env at `docker run`, so a config value it got wrong (a DB - * password from another install's config store) cannot be corrected in place. - * `update` corrects it only by also re-cloning from GitHub and rebuilding, which turns - * a credential repair into an unreviewed version change on a live venue. This keeps the - * image byte-identical and changes only what the config map now says. - * - * Reports what it recreated, for the same reason `update` does: an unsupported - * module (node, database) was logged and skipped while the command still - * exited 0, so `recreate node && echo ok` printed ok having recreated nothing. - * - * @param {Object} servicesList - * @returns {Promise<{recreated: Array, skipped: Array}>} - */ -async function recreateModules(servicesList) { - const { buildAndUp } = moduleService - const { setDatabaseParameters, setHubDatabaseParameters } = databaseService - - const outcome = { recreated: [], skipped: [] } - const failures = [] - let touchedDbModule = false - let touchedHubModule = false - for (const nextCoin in servicesList) { - for (const nextNetwork in servicesList[nextCoin]) { - for (const nextModule of servicesList[nextCoin][nextNetwork]) { - if (RECREATE_UNSUPPORTED_MODULES.includes(nextModule)) { - // Still a continue: `recreate all` legitimately sweeps past the - // node and the database. What changed is that the skip is now - // recorded, so a run that recreated NOTHING can be reported as - // the failed request it is instead of exiting 0. - // The database has no `update` to redirect to either: that verb - // refuses it for the same reason (no container built from the - // config map, no in-place image upgrade). Say the real remedy. - const remedy = nextModule === DB_MODULE_NAME - ? "; the database container must be removed manually and reinstalled" - : "; use `update " + nextModule + "` instead" - console.log("recreate does not apply to " + nextModule + remedy) - outcome.skipped.push({ module: nextModule, coin: nextCoin, network: nextNetwork, reason: 'not-recreatable' }) - continue - } - const moduleContainerId = await db.getModuleContainer(nextModule, nextCoin, nextNetwork) - if (!moduleContainerId) { - // No registry row is TWO different states and this verb must not - // conflate them. Registry drift (row lost, container still up) has - // to keep recreating: dropping it was what made `update node` a - // silent no-op. An explicitly uninstalled venue must NOT: uninstall - // removes the container and the row but leaves the image tag, so - // buildAndUp's reuseImage check passes, `null` reads as "nothing to - // tear down", and the operator gets back a service they tore down, - // built from a stale image and re-stamped into the registry that - // status/precheck/autoheal trust. Discriminate on a POSITIVE docker - // answer only: 'unknown' is a daemon hiccup, not an absence. - const presence = await probeContainerPresenceByName( - getDockerContainerImageName(nextModule, nextCoin, nextNetwork)) - if (presence === 'gone') { - console.warn(`recreate: ${nextModule} (${nextCoin} ${nextNetwork}) has no container; nothing to recreate. Install it first.`) - outcome.skipped.push({ module: nextModule, coin: nextCoin, network: nextNetwork, reason: 'not-installed' }) - continue - } - } - try { - await buildAndUp(nextModule, nextCoin, nextNetwork, moduleContainerId, false, null, { reuseImage: true }) - } catch (err) { - // Visiting the rest of the sweep after one venue fails follows - // uninstallModules: `recreate all` was already half-applied by the - // time it threw, and stopping there hid which venues had been - // touched behind one flat `recreate failed:`. The run still REJECTS - // below, naming every venue - a failure never becomes a skip. - const why = (err && err.message) ? err.message : String(err) - console.error(`recreate: ${nextModule} (${nextCoin} ${nextNetwork}) FAILED: ${why}`) - failures.push({ module: nextModule, coin: nextCoin, network: nextNetwork, reason: why }) - continue - } - outcome.recreated.push({ module: nextModule, coin: nextCoin, network: nextNetwork }) - if (nextModule === XChainService.XCHAIN_DECODER || nextModule === XChainService.XCHAIN_INDEXER) { - touchedDbModule = true - } - if (nextModule === HUB_MODULE_NAME) { - touchedHubModule = true - } - } - } - } - - // Provision AFTER every container is back on the config values, so the drift guard - // in setDatabaseParameters sees the state we just converged rather than the one - // that made the recreate necessary. - if (touchedDbModule) await setDatabaseParameters() - // Same rule for the SHARED hub account, and it matters most on this verb: the - // recreated hub starts on the config store's HUB_DB_PASS, so without rotating - // the live 'xchain_hub'@'%' account to match, `recreate xchain-hub` hands the - // hub a password MariaDB never received and it crash-loops on ER_ACCESS_DENIED. - // The `update` path rotates here for the same reason (ModuleService installModule). - if (touchedHubModule) await setHubDatabaseParameters() - await statusChanged() - if (failures.length) { - // Rejecting AFTER provisioning is deliberate: the venues that did come back - // start on the config store's password and would crash-loop on - // ER_ACCESS_DENIED if the run bailed before rotating their accounts. - throw new Error('recreate failed for ' + failures.length + ' module(s): ' - + failures.map(f => `${f.module} (${f.coin} ${f.network}): ${f.reason}`).join('; ')) - } - return outcome -} - -/** - * Uninstall every requested module, then FAIL if any of them failed. - * - * Visiting the rest of the list after one module fails is deliberate and stays: - * an operator tearing down a stack wants the other containers gone. What was - * wrong is that the per-module `catch` swallowed the error and the function - * returned true regardless, so `uninstall all` reported a clean teardown while - * leaving containers running - the exact "did nothing, said success" shape the - * `update` no-op fix removed elsewhere. - * - * Shared services (database, hub, explorer, sync) are installed ONCE and serve - * every coin/network on the box, so they are ordered LAST and only removed when - * nothing is left to serve. `--include-shared` is a request, not an override: with - * bitcoin still installed, `uninstall all dogecoin mainnet --include-shared` used - * to take the explorer down for bitcoin too. Now the shared pass runs after the - * per-coin pass (so a genuine full teardown still reaches them, the remaining set - * being empty by then) and skips with a reason naming what is still installed. - * - * @returns {Promise<{uninstalled: Array, skipped: Array}>} on full success - * @throws {Error} listing every module that failed, after all were attempted - */ -async function uninstallModules(servicesList, includeShared = false) { - const sharedModules = [DB_MODULE_NAME, HUB_MODULE_NAME, EXPLORER_MODULE_NAME, SYNC_MODULE_NAME] - const outcome = { uninstalled: [], skipped: [] } - const failures = [] - const deferredShared = [] - - const uninstallOne = async (nextModule, nextCoin, nextNetwork) => { - const moduleContainerId = await db.getModuleContainer(nextModule, nextCoin, nextNetwork) - if (!moduleContainerId) { - outcome.skipped.push({ module: nextModule, coin: nextCoin, network: nextNetwork, reason: 'not-installed' }) - return - } - try { - await uninstallModule(nextCoin, nextNetwork, nextModule) - outcome.uninstalled.push({ module: nextModule, coin: nextCoin, network: nextNetwork }) - } catch (err) { - const why = (err && err.message) ? err.message : String(err) - console.error(`uninstall: ${nextModule} (${nextCoin} ${nextNetwork}) FAILED: ${why}`) - failures.push({ module: nextModule, coin: nextCoin, network: nextNetwork, reason: why }) - } - } - - for (const nextCoin in servicesList) { - for (const nextNetwork in servicesList[nextCoin]) { - for (const nextModule of servicesList[nextCoin][nextNetwork]) { - if (sharedModules.includes(nextModule)) { - if (!includeShared) { - outcome.skipped.push({ module: nextModule, coin: nextCoin, network: nextNetwork, reason: 'shared' }) - } else { - deferredShared.push({ module: nextModule, coin: nextCoin, network: nextNetwork }) - } - continue - } - await uninstallOne(nextModule, nextCoin, nextNetwork) - } - } - } - - // Shared pass. `remaining` is read AFTER the per-coin pass above, so a full - // teardown finds it empty and still removes them. A coin/network module is any - // registry row carrying a coin; shared services are registered under ''/''. - if (deferredShared.length > 0) { - let remaining = [] - try { - remaining = (await db.getAllModuleContainers(null, null)).filter(r => r.coin) - } catch (err) { - // The registry is the only thing that can answer "is anything still - // being served". Unreadable, we refuse rather than guess: leaving a - // shared service up costs an operator one more command, tearing it - // down under a live coin costs every other coin its explorer/hub. - const why = (err && err.message) ? err.message : String(err) - for (const s of deferredShared) - outcome.skipped.push({ ...s, reason: `shared, module registry unreadable (${why})` }) - deferredShared.length = 0 - } - const stillServed = [...new Set(remaining.map(r => `${r.coin} ${r.network}`))].sort() - for (const s of deferredShared) { - if (stillServed.length > 0) { - const reason = `shared, still serving ${stillServed.join(', ')}` - console.warn(`uninstall: keeping ${s.module}; it is ${reason}.`) - outcome.skipped.push({ ...s, reason }) - continue - } - await uninstallOne(s.module, s.coin, s.network) - } - } - - if (failures.length > 0) { - const detail = failures.map(f => `${f.module} (${f.coin} ${f.network}): ${f.reason}`).join('; ') - const err = new Error(`uninstall failed for ${failures.length} module${failures.length === 1 ? '' : 's'}: ${detail}`) - err.failures = failures - err.uninstalled = outcome.uninstalled - throw err - } - return outcome -} - -async function logModules(servicesList, follow = true) { - const moduleContainerIds = [] - for (const nextCoin in servicesList) { - for (const nextNetwork in servicesList[nextCoin]) { - for (const nextModule of servicesList[nextCoin][nextNetwork]) { - const containerId = await db.getModuleContainer(nextModule, nextCoin, nextNetwork) - if (!containerId) continue - moduleContainerIds.push({ - name: getDockerContainerImageName(nextModule, nextCoin, nextNetwork), - id: containerId - }) - } - } - } - - if (moduleContainerIds.length > 0) { - if (follow) { - // A single interleaved TTY stream only makes sense for one - // container; warn instead of silently dropping the rest so the - // operator knows N-1 services are omitted from `tail all`. - if (moduleContainerIds.length > 1) { - const omitted = moduleContainerIds.slice(1).map(c => c["name"]).join(", ") - console.log("Following only " + moduleContainerIds[0]["name"] + "; omitted: " + omitted) - } - const moduleName = moduleContainerIds[0]["name"] - console.log("") - console.log("") - console.log("####" + moduleName + " LOGS####") - console.log("") - await logContainer(moduleContainerIds[0]["id"], follow) - } else { - // Non-follow dumps can safely iterate every selected service in - // sequence (no shared TTY to interleave). - for (const container of moduleContainerIds) { - console.log("") - console.log("") - console.log("####" + container["name"] + " LOGS####") - console.log("") - await logContainer(container["id"], follow) - } - } - } else { - console.log("No service was selected") - } - return true -} - -async function monitorModules(servicesList, follow = true) { - const moduleContainerIds = [] - for (const nextCoin in servicesList) { - for (const nextNetwork in servicesList[nextCoin]) { - for (const nextModule of servicesList[nextCoin][nextNetwork]) { - const containerId = await db.getModuleContainer(nextModule, nextCoin, nextNetwork) - if (!containerId) continue - moduleContainerIds.push({ - name: getDockerContainerImageName(nextModule, nextCoin, nextNetwork), - id: containerId - }) - } - } - } - await startDockerMonitor(moduleContainerIds, follow) - return true -} - -async function restartModules(servicesList) { - for (const nextCoin in servicesList) { - for (const nextNetwork in servicesList[nextCoin]) { - for (const nextModule of servicesList[nextCoin][nextNetwork]) { - const containerId = await db.getModuleContainer(nextModule, nextCoin, nextNetwork) - if (!containerId) continue - try { - await restartContainer(containerId) - await statusChanged() - } catch (err) { - console.log(err) - } - } - } - } - return true -} - -async function stopModules(servicesList) { - for (const nextCoin in servicesList) { - for (const nextNetwork in servicesList[nextCoin]) { - for (const nextModule of servicesList[nextCoin][nextNetwork]) { - const containerId = await db.getModuleContainer(nextModule, nextCoin, nextNetwork) - if (!containerId) continue - try { - // With the service's budget, not docker's ten seconds: a bare - // `docker stop` on a container created before the budget was - // stamped on it is a coin flip for a service mid-block. - await stopModuleContainer(stopContainerByName, nextModule, nextCoin, nextNetwork, containerId) - } catch (err) { - console.log(err) - } - } - } - } - return true -} - -async function startModules(servicesList) { - for (const nextCoin in servicesList) { - for (const nextNetwork in servicesList[nextCoin]) { - for (const nextModule of servicesList[nextCoin][nextNetwork]) { - const containerId = await db.getModuleContainer(nextModule, nextCoin, nextNetwork) - if (!containerId) continue - try { - await startContainer(containerId) - } catch (err) { - console.log(err) - } - } - } - } - return true -} - -// Audited clear of a decoder's durable REORG_HALT marker, run inside the decoder -// container so it uses the service's own DB credentials and code -// (xchain-decoder/src/clear-reorg-halt.js checks the database is intact, then -// records the clear as an events row with the reason). One decoder per -// coin/network; `servicesList` is the filtered map the CLI builds. Returns true -// only when every targeted decoder answered exit 0. -async function clearDecoderReorgHalt(servicesList, { reason, force = false, dryRun = false } = {}) { - // A dry run writes nothing, so it runs without a reason; a real clear records one. - const reasonText = typeof reason === 'string' ? reason.trim() : '' - if (!dryRun && reasonText.length < 8) { - console.log('clear-reorg-halt: --reason must say, in at least 8 characters, why this database is known good; it is recorded with the clear.') - return false - } - const args = ['node', 'src/clear-reorg-halt.js'] - if (reasonText) args.push('--reason', reasonText) - if (force) args.push('--force') - if (dryRun) args.push('--dry-run') - let targeted = 0 - let ok = true - for (const nextCoin in servicesList) { - for (const nextNetwork in servicesList[nextCoin]) { - if (!servicesList[nextCoin][nextNetwork].includes(XChainService.XCHAIN_DECODER)) continue - const containerId = await db.getModuleContainer(XChainService.XCHAIN_DECODER, nextCoin, nextNetwork) - if (!containerId) { - console.log('clear-reorg-halt: no xchain-decoder container is installed for ' + nextCoin + ' ' + nextNetwork) - ok = false - continue - } - targeted++ - try { - const out = await execContainer(containerId, args) - if (out) console.log(out) - } catch (err) { - // The script prints its refusal on stderr and exits non-zero; docker - // exec surfaces that as an error whose stdout/stderr carry the text. - const text = [err && err.stdout, err && err.stderr].filter(Boolean).join('\n').trim() - console.log(text || ('clear-reorg-halt: failed for ' + nextCoin + ' ' + nextNetwork + ': ' + (err && err.message))) - ok = false - } - } - } - if (targeted === 0 && ok) { - console.log('clear-reorg-halt: no xchain-decoder matched the given chain and network') - return false - } - return ok -} - -async function execModules(servicesList, command) { - const commandArgs = command.split(/\s+/) - for (const nextCoin in servicesList) { - for (const nextNetwork in servicesList[nextCoin]) { - for (const nextModule of servicesList[nextCoin][nextNetwork]) { - const containerId = await db.getModuleContainer(nextModule, nextCoin, nextNetwork) - if (!containerId) continue - try { - const execStdOut = await execContainer(containerId, commandArgs) - console.log(execStdOut) - } catch (err) { - console.log(err) - } - } - } - } - return true -} - -async function shellModule(servicesList) { - for (const nextCoin in servicesList) { - for (const nextNetwork in servicesList[nextCoin]) { - for (const nextModule of servicesList[nextCoin][nextNetwork]) { - const containerId = await db.getModuleContainer(nextModule, nextCoin, nextNetwork) - if (!containerId) continue - try { - await shellContainer(containerId) - return true - } catch (err) { - console.log(err) - } - } - } - } - return true -} - -// `ref` is the ref to clone the e2e-test suite at, normally the same one the -// stack under test was installed at. Null keeps the default-branch behaviour -// every caller had before the option existed. -async function runE2ETest(coin, network, testName = null, grep = null, script = null, ref = null) { - let dockerCmdArgs = null - if (script) { - // Run an arbitrary e2e npm script (e.g. test:security, test:perf:budget) so CI - // can drive the stack-dependent suites beyond the default action suite. Takes - // precedence over testName; the e2e-test image carries these scripts. - dockerCmdArgs = ['npm', 'run', script] - } else if (testName) { - dockerCmdArgs = ['npx', 'mocha', '--timeout', '0', '--exit', - '--require', './test/initialCheck.test.js', - `test/actions/${testName}.test.js`] - if (grep) dockerCmdArgs.push('--grep', grep) - } - const containerId = await installModule(XChainService.XCHAIN_E2E_TEST, coin, network, true, null, true, ref, dockerCmdArgs) - - console.log("Running e2e tests, please wait...") - const exitCode = await waitContainer(containerId) - - const now = new Date() - const timestamp = now.toISOString().replace(/[:.]/g, '-').slice(0, 19) - const logFile = path.join(dataDir, 'e2e-logs', `${coin}-${network}-${timestamp}.log`) - - await saveContainerLogs(containerId, logFile) - await removeContainer(containerId) - - return { logFile, exitCode } -} - -// Prompts the operator to type "yes" before a destructive reset proceeds. -// Reused instead of duplicated so every call site aborts the exact same way -// on a non-affirmative answer. Not called at all when the caller passes -// force=true (CI/scripted resets). -async function confirmDestructiveReset(coin, network, targets) { - if (!process.stdin.isTTY) { - throw new Error( - 'reset: refusing to run a destructive reset on a non-interactive terminal without --yes. ' + - 'Re-run with --yes to confirm.' - ) - } - console.warn(`\nWARNING: this will IRREVERSIBLY destroy ${coin} ${network} data.`) - console.warn(` Affected stores: ${targets.join(', ')}`) - console.warn(' This forces a full resync afterward. There is no undo.\n') - const rl = readline.createInterface({ input: process.stdin, output: process.stdout }) - const answer = await new Promise((resolve) => { - rl.question('Type "yes" to confirm: ', resolve) - }) - rl.close() - return answer.trim().toLowerCase() === 'yes' -} - -// True when a docker error means the container is already gone. Matched the -// same way DockerService.removeContainer matches it; execFile puts docker's -// stderr into the error message, and stopContainer can also reject a plain -// string, which is a real failure and must not be read as a miss here. -function isNoSuchContainerError(err) { - if (!err || typeof err === 'string') return false - return /no such container/i.test(String(err.message || err.stderr || '')) -} - -// True when a docker error means the named volume is already gone. Same rule -// isNoSuchContainerError uses, and the one DockerService.probeContainerPresenceByName -// states: docker SAYING "no such volume" is the only thing that means absent; -// every other failure is unknown and must be treated as possibly-present. -function isNoSuchVolumeError(err) { - if (!err || typeof err === 'string') return false - return /no such volume/i.test(String(err.message || err.stderr || '')) -} - -// The operator-facing reason for a rejected docker or registry call. Handles the -// bare-string rejection stopContainer can produce as well as an Error. -function failureReason(err) { - return (err && err.message) || String(err) -} - -// Put back the services an aborted reset already stopped, so the abort leaves -// the stack as it found it rather than half torn down. Returns the modules that -// could not be restarted, for the operator message. -// -// Reads the registry strictly: getModuleContainer answers null on a SQL error as -// well as on a miss, so a rollback run during the very registry outage that -// caused the abort restarted nothing and still reported zero failures -// (uuid:846cc40d). A lookup that fails now lands in `failed` and is named in the -// STILL DOWN line. -async function restartStoppedModules(modules, coin, network) { - const failed = [] - for (const module of modules) { - try { - const containerId = await db.getModuleContainerStrict(module, coin, network) - if (!containerId) continue - await startContainer(containerId) - } catch { - failed.push(module) - } - } - return failed -} - -/** - * Resolve the HOST directory that holds this chain's node datadir, asking the - * node container itself first. - * - * `dataDir` is env-derived (XCHAIN_NODE_DATA_DIR, else the in-repo data/), so a - * shell that never sourced the operator's profile resolves a path the stack has - * never used. The wipe was guarded on fs.existsSync of that path, so the guard - * went silently false and `reset all` reported success with the chain untouched. - * The container name is already resolved deterministically from the prefix and - * coin/network, so use that same key to read the datadir off the container's own - * bind mounts: whatever the daemon actually writes to is what a reset must wipe. - * - * Falls back to the env-derived path only when it really is on disk. Returns - * path=null when neither answer exists, and the caller fails closed on that - * rather than skipping the wipe. - * - * @returns {Promise<{path: (string|null), resolvedFrom: (string|null), configuredPath: string, containerName: string}>} - */ -async function resolveNodeDataPath(coin, network) { - const containerName = getDockerContainerImageName(NODE_MODULE_NAME, coin, network) - const configuredPath = path.join(dataDir, NODE_MODULE_NAME, coin, network) - - let mounts = [] - try { - mounts = await getContainerBindMounts(containerName) - } catch { /* no container, or docker unreachable: fall through to the configured path */ } - const dataMount = (Array.isArray(mounts) ? mounts : []) - .find(m => m && m.destination === `/root/.${coin}` && m.source) - if (dataMount) { - return { - path: dataMount.source, - resolvedFrom: `the /root/.${coin} bind mount of container ${containerName}`, - configuredPath, - containerName - } - } - - if (fs.existsSync(configuredPath)) { - return { path: configuredPath, resolvedFrom: 'the configured data dir', configuredPath, containerName } - } - - return { path: null, resolvedFrom: null, configuredPath, containerName } -} - -// The service names `reset` can act on. `reset` is the only destructive CLI path -// and the only one that bypasses resolveArgs/filterCommandParameters, so it must -// validate its own raw args: without this an unrecognised service (a typo, or a -// non-resettable module like xchain-encoder) leaves every reset flag false, so -// `targets` is empty, no branch fires, and resetModules returns true - the CLI -// exits 0 reporting success after resetting nothing. Fail loud instead, -// matching resolveArgs/rollback and the "fail fast BEFORE any destructive wipe" -// convention this file already follows. -const RESETTABLE_SERVICES = [ - 'all', - NODE_MODULE_NAME, - XChainService.XCHAIN_UTXO_TRACKER, - XChainService.XCHAIN_DECODER, - XChainService.XCHAIN_INDEXER -] - -async function resetModules(service, coin, network, force = false, withIndexer = false) { - if (!RESETTABLE_SERVICES.includes(service)) { - throw new Error("reset: unknown service '" + service + "'; expected one of " - + RESETTABLE_SERVICES.join(', ')) - } - if (!Object.values(Coin).includes(coin)) { - throw new Error("reset: unknown coin '" + coin + "'; expected one of " - + Object.values(Coin).join(', ')) - } - if (!Object.values(Network).includes(network)) { - throw new Error("reset: unknown network '" + network + "'; expected one of " - + Object.values(Network).join(', ')) - } - const resetAll = service === 'all' - const resetNode = resetAll || service === NODE_MODULE_NAME - const resetUtxoTracker = resetAll || service === XChainService.XCHAIN_UTXO_TRACKER - const resetDecoder = resetAll || service === XChainService.XCHAIN_DECODER - const resetIndexer = resetAll || service === XChainService.XCHAIN_INDEXER || (withIndexer && resetDecoder) - - // The indexer's rollback cursor IS a decoder `events` id, and the decoder - // never deletes those rows, so wiping the decoder alone restarts the ids - // under a cursor that now points past them: the indexer fails RE-1 and stops - // committing. The pair only has a coherent state when both move together. - // Asymmetric on purpose - resetting the indexer alone re-derives it from an - // intact decoder, which is an ordinary reindex and stays allowed. - if (resetDecoder && !resetIndexer) { - let indexerInstalled = null - try { - // Strict read: getModuleContainer answers null on a SQL error and on an - // unopened pool as well as on a genuine miss, so a registry blip read as - // "no indexer installed" and waved through the one wipe this guard exists - // to stop (uuid:7cbafa08). A lookup that FAILS is not evidence of absence. - // Only an empty result set still means "not installed", which stays allowed. - indexerInstalled = await db.getModuleContainerStrict(XChainService.XCHAIN_INDEXER, coin, network) - } catch (err) { - // abortBeforeAnyWipe is declared further down this function and nothing has - // been stopped yet, so the plain refusal is the correct shape here. - console.log(`Aborted: cannot read the ${XChainService.XCHAIN_INDEXER} registry row ` - + `(${failureReason(err)}), so it is not known whether resetting ` - + `${XChainService.XCHAIN_DECODER} alone would strand it. No data was touched.`) - return false - } - if (indexerInstalled) { - console.log(`Aborted: resetting ${XChainService.XCHAIN_DECODER} alone would leave ` - + `${XChainService.XCHAIN_INDEXER} incoherent. No data was touched.`) - console.log(" The indexer tracks reorgs by a decoder event id. Wiping the decoder restarts") - console.log(" those ids, so the indexer would abort with a reorg-cursor error (RE-1) and stop") - console.log(" committing blocks until both are rebuilt together.") - console.log(` Reset the pair: xchain-node reset ${XChainService.XCHAIN_DECODER} ${coin} ${network} --with-indexer`) - console.log(` Or the whole stack (also re-syncs the chain): xchain-node reset all ${coin} ${network}`) - return false - } - } - - // Relocated blocks/txindex host paths (XCHAIN_NODE_BLOCKS_DIR mode): these - // live OUTSIDE the in-datadir path the node wipe clears, so they must be - // wiped explicitly and named in the confirmation, else a reset restarts the - // daemon over a stale blocks dir + stale txindex (uuid:90630038). - // Env-first with config/node.local fallback: a reset from a - // profile-less shell must still see the relocated stores, or it restarts - // the daemon over stale out-of-datadir chain data. - const { resolveBlocksDir } = nodeService - const blocksDir = await resolveBlocksDir() - const blocksHostPath = blocksDir ? `${blocksDir}/${coin}/${network}` : null - const txindexHostPath = blocksDir ? `${blocksDir}/${coin}/${network}-txindex` : null - - // Resolve the node datadir BEFORE anything is stopped or confirmed, and - // refuse the whole reset by name when it cannot be resolved. The - // old code re-derived the path from XCHAIN_NODE_DATA_DIR at the wipe site - // and skipped the wipe whenever that path was absent, so a reset run from a - // profile-less shell wiped the decoder/indexer DBs, left the chain in place, - // and exited 0; the missing "Clearing node data" line was the only tell. - // "Not installed" stays a legitimate skip, and is stated out loud. - let nodeDataPath = null - if (resetNode) { - let nodeInstalled = null - let registryReadable = true - try { - nodeInstalled = await db.getModuleContainer(NODE_MODULE_NAME, coin, network) - } catch { registryReadable = false } - - const resolved = await resolveNodeDataPath(coin, network) - if (resolved.path) { - nodeDataPath = resolved.path - if (path.resolve(nodeDataPath) !== path.resolve(resolved.configuredPath)) { - console.log(`Node datadir resolved from ${resolved.resolvedFrom}: ${nodeDataPath}`) - console.log(` (XCHAIN_NODE_DATA_DIR in this shell would have pointed at ${resolved.configuredPath})`) - } - } else if (registryReadable && !nodeInstalled) { - console.log(`No ${NODE_MODULE_NAME} container is installed for ${coin} ${network}; there is no node data to clear.`) - } else { - const envState = config.XCHAIN_NODE_DATA_DIR && config.XCHAIN_NODE_DATA_DIR.trim() !== '' - ? `set to ${process.env.XCHAIN_NODE_DATA_DIR}` - : 'UNSET in this shell (non-interactive shells do not source the profile)' - console.log(`Aborted: cannot resolve the ${coin} ${network} node datadir. No data was touched.`) - console.log(` Container ${resolved.containerName} reported no /root/.${coin} bind mount ` - + '(it is absent, or docker is unreachable from here).') - console.log(` The configured path ${resolved.configuredPath} does not exist either.`) - console.log(` XCHAIN_NODE_DATA_DIR is ${envState}.`) - console.log(' Set XCHAIN_NODE_DATA_DIR to this stack\'s data root (or make docker reachable so the') - console.log(' node container can be inspected) and re-run. Refusing rather than resetting the') - console.log(' databases around a chain that would stay untouched.') - return false - } - } - - if (!force) { - const targets = [] - if (resetNode) { - if (nodeDataPath) targets.push(`node datadir (${nodeDataPath})`) - if (blocksDir) { - targets.push(`relocated blocks dir (${blocksHostPath})`) - targets.push(`relocated txindex dir (${txindexHostPath})`) - } - } - if (resetUtxoTracker) targets.push(`xchain-utxo-tracker Docker volume (${getUtxoTrackerVolumeName(coin, network)})`) - if (resetDecoder) targets.push('xchain-decoder database') - if (resetIndexer) { - targets.push('xchain-indexer database') - // Named in the confirmation because it is a change to the HUB's state, not - // this chain's: the operator should see that the reset reaches across. - targets.push('hub price ingest fence row for this chain (price_ingest_watermarks)') - } - // A regtest chain reset is a RE-GENESIS (the datadir goes, the chain - // comes back from block 0), which invalidates every hub row anchored on - // the dead chain's blocks. Named for the same reason the price fence is: - // it is a change to the HUB's state, not just this chain's, and the - // operator should see that the reset reaches across. - if (resetNode && network === Network.REGTEST) { - targets.push('hub cross-chain relic rows for this network ' - + '(cross_chain_matches, cross_chain_calls, capability_snapshots)') - } - const confirmed = await confirmDestructiveReset(coin, network, targets) - if (!confirmed) { - console.log('Aborted: reset was not confirmed. No data was touched.') - return false - } - } - - // Fail fast BEFORE any destructive wipe: a DB reset needs a working MariaDB, - // and resetDatabases is not reached until AFTER the stop loop and every wipe - // below, so discovering the problem there half-destroys the stack. In docker - // mode the failure is `docker exec null` with the container gone - // (uuid:6f6584dc); in EXTERNAL_DB mode it is an unreachable host, or a - // getExternalDbConfig throw on a partial env, and NOTHING probed for it - // (uuid:41887889). Both modes are probed here. It sits ahead of the stop - // loop, not after it: this abort returns before the restart pass, so probing - // later left every already-stopped service DOWN while still reporting that - // no data was touched (uuid:bb190060). - const dbResetNeeded = resetDecoder || resetIndexer - if (dbResetNeeded) { - if (EXTERNAL_DB) { - const probe = await pingExternalDatabase() - if (!probe.ok) { - console.log(`Aborted: cannot reach the external MariaDB at ${probe.host}:${probe.port}` - + ` (${probe.error}). No data was touched.`) - return false - } - } else { - const dbContainerId = await getDatabaseContainerId() - if (!dbContainerId) { - console.log('Aborted: MariaDB container not found; install the database first. No data was touched.') - return false - } - } - } - - const modulesToStop = [] - if (resetNode) modulesToStop.push(NODE_MODULE_NAME) - if (resetUtxoTracker) modulesToStop.push(XChainService.XCHAIN_UTXO_TRACKER) - if (resetDecoder) modulesToStop.push(XChainService.XCHAIN_DECODER) - if (resetIndexer) modulesToStop.push(XChainService.XCHAIN_INDEXER) - if (resetAll) modulesToStop.push(XChainService.XCHAIN_REGTEST_MINER) - - console.log(`Stopping ${coin} ${network} services...`) - // Abort before any wipe when a target will not stop, and put back whatever - // was already stopped. The bare catch this replaced swallowed EVERY - // stopContainer rejection as "not installed", so a daemon that failed to - // stop kept reading and writing the store while the wipes below deleted it - // (uuid:9c88cfe6). Only a "no such container" miss is still a legitimate - // skip; the registry miss is already handled by the null check. - const stoppedModules = [] - // Refuse the whole reset, put back whatever this run stopped, and report it. - // Only reachable while nothing has been wiped yet, which is why it may still - // promise that no data was touched. - const abortBeforeAnyWipe = async (reason) => { - const restartFailures = await restartStoppedModules(stoppedModules, coin, network) - console.log(`Aborted: ${reason}. No data was touched.`) - if (stoppedModules.length > 0) { - console.log(` Restarted ${stoppedModules.length - restartFailures.length} of ` - + `${stoppedModules.length} already-stopped service(s).`) - } - if (restartFailures.length > 0) { - console.log(` STILL DOWN, start by hand: ${restartFailures.join(', ')}`) - } - return false - } - for (const module of modulesToStop) { - let containerId = null - try { - containerId = await db.getModuleContainerStrict(module, coin, network) - } catch (err) { - // A registry read that FAILED is not evidence the module is absent. - // The swallowing read this replaced answered null on any SQL error, so - // a blip after the reachability precheck made a live indexer look - // uninstalled: the loop skipped stopping it and the wipes below, - // resetDatabases included, ran underneath it (uuid:846cc40d). - return await abortBeforeAnyWipe( - `cannot read the ${module} registry row (${failureReason(err)}), so it is not known ` - + 'whether that service is running') - } - // A SUCCESSFUL read with no row is still a legitimate "not installed" - // skip; without this check stopContainer(null) fails and ABORTS the reset - // (uuid:fd7cc224 sibling site). - if (!containerId) continue - try { - await stopContainer(containerId) - stoppedModules.push(module) - } catch (err) { - if (isNoSuchContainerError(err)) continue - return await abortBeforeAnyWipe(`${module} failed to stop (${failureReason(err)})`) - } - } - - // Classify the tracker volume's presence BEFORE the first wipe. The wipe - // below swallowed every failure as "the volume may not exist", so a - // permission error, an unreachable daemon or a failed alpine pull left stale - // tracker data behind while the decoder/indexer databases were dropped and - // the run reported success (uuid:e24c98d4). Only docker SAYING "no such - // volume" is absence; anything else refuses here, while the stack is whole. - let utxoVolumeName = null - let utxoVolumePresent = false - if (resetUtxoTracker) { - // Name the volume through the shared helper so it carries NODE_PREFIX - // (uuid:7523dd94): an unprefixed name resolves to the DEFAULT_NODE_PREFIX - // stack's volume and wipes that one instead of the intended target. - utxoVolumeName = getUtxoTrackerVolumeName(coin, network) - try { - await execFileAsync('docker', ['volume', 'inspect', utxoVolumeName]) - utxoVolumePresent = true - } catch (err) { - if (!isNoSuchVolumeError(err)) { - return await abortBeforeAnyWipe( - `cannot determine whether the Docker volume ${utxoVolumeName} exists ` - + `(${failureReason(err)})`) - } - console.log(`No Docker volume ${utxoVolumeName} to clear.`) - } - } - - // Tracks whether anything irreversible has happened yet, so a later abort - // reports the stack's real state instead of promising an untouched one. - let nodeDataWiped = false - - if (resetNode) { - // No existsSync guard here any more: the path was resolved (and the - // reset refused, or the "not installed" skip announced) up top, so an - // unresolvable datadir can no longer read as a silent no-op. - if (nodeDataPath) { - console.log(`Clearing node data at ${nodeDataPath}...`) - await execFileAsync('docker', ['run', '--rm', '-v', `${nodeDataPath}:/data`, 'alpine', 'sh', '-c', 'find /data -mindepth 1 -delete']) - nodeDataWiped = true - } - // Relocated blocks/txindex (XCHAIN_NODE_BLOCKS_DIR) live outside the - // datadir, so wipe them here too or the daemon restarts over stale - // chain data (uuid:90630038). - for (const relocated of [blocksHostPath, txindexHostPath]) { - if (relocated && fs.existsSync(relocated)) { - console.log(`Clearing relocated node data at ${relocated}...`) - await execFileAsync('docker', ['run', '--rm', '-v', `${relocated}:/data`, 'alpine', 'sh', '-c', 'find /data -mindepth 1 -delete']) - nodeDataWiped = true - } - } - } - - if (resetUtxoTracker && utxoVolumePresent) { - try { - console.log(`Clearing Docker volume ${utxoVolumeName}...`) - await execFileAsync('docker', ['run', '--rm', '-v', `${utxoVolumeName}:/data`, 'alpine', 'sh', '-c', 'find /data -mindepth 1 -delete']) - } catch (err) { - // The volume exists and the wipe failed, so the tracker still holds - // its old store. Falling through would drop the decoder/indexer - // databases around retained tracker data and still return true. - const reason = `clearing the Docker volume ${utxoVolumeName} failed (${failureReason(err)})` - if (nodeDataWiped) { - // Node data is already gone, so this reset is half done and cannot - // claim otherwise. Starting the services again would run a - // resynced chain under decoder and indexer stores that still - // describe the old one, so they stay down until the operator - // re-runs the same reset. - console.log(`Aborted: ${reason}.`) - console.log(' The node data for this stack WAS already cleared; the decoder/indexer') - console.log(' databases were NOT touched, and the stopped services are left down.') - console.log(' Fix the volume problem and re-run the same reset command.') - return false - } - return await abortBeforeAnyWipe(reason) - } - } - - const dbModulesToReset = [ - ...(resetDecoder ? [XChainService.XCHAIN_DECODER] : []), - ...(resetIndexer ? [XChainService.XCHAIN_INDEXER] : []), - ] - if (dbModulesToReset.length > 0) { - await resetDatabases(coin, network, dbModulesToReset) - } - - // A wiped indexer DB restarts push_generations at 0, which the hub's price - // ingest fence reads as a stale replay and DROPS, killing this chain's price - // rail (and the native-fee / XCHAIN-USD path) with no error. Clear the fence - // row here, while the indexer is still stopped, so the first push after the - // restart below lands. Never fatal: the wipe already happened, so a failure - // must not abort the restart pass and leave the stack down. It is reported - // loudly instead, with the statement to run by hand. - if (resetIndexer) { - try { - await clearHubPriceIngestWatermark(coin, network) - } catch (err) { - console.warn('WARNING: clearing the hub price ingest fence failed: ' + (err && err.message ? err.message : err)) - console.warn(" Run on the hub DB before the indexer catches up:") - // Network-scoped, matching the statement DatabaseService prints on its - // own failure branches. The '' bucket is the legacy/unset scope that - // pre-migration rows sit in. - console.warn(" DELETE FROM price_ingest_watermarks WHERE source_chain = '" - + (CoinTickerSymbol[coin] || coin) + "' AND network IN ('" - + String(network || '').trim().toLowerCase() + "', '');") - console.warn(" Keep the network clause: it is what leaves every OTHER network's fence for this chain in place.") - } - } - - // A re-genesised regtest chain leaves the hub's cross-chain rows behind: - // they are keyed by `network` and a BTC-anchored snapshot_block, and nothing - // in them names the chain INSTANCE, so the mirror bootstrap hands every - // fresh indexer the dead chain's finalized matches and it refuses them at - // every block for as long as it runs. Purge them here, while the indexer is - // stopped, so the rebuilt mirror never sees them. Regtest only: no other - // network has a re-genesis path, and there these rows are live federation - // history. Never fatal, like the price fence above: the wipe already - // happened, so a failure must not abort the restart pass and leave the stack - // down. It is reported loudly instead, with the statements to run by hand. - if (resetNode && network === Network.REGTEST) { - try { - await purgeHubCrossChainRows(coin, network) - } catch (err) { - console.warn('WARNING: purging the hub cross-chain relic rows failed: ' - + ((err && err.message) ? err.message : err)) - console.warn(' A fresh indexer will mirror the dead chain\'s matches back in. Run on the hub DB:') - for (const statement of manualHubCrossChainPurgeStatements(CoinTickerSymbol[coin] || coin)) { - console.warn(' ' + statement) - } - console.warn(` Then restart the hub: xchain-node restart ${HUB_MODULE_NAME}`) - } - } - - // A reset is a REINDEX: from here the wiped stores rebuild on a new lineage, - // and every bootstrap archive already published for these combos describes - // the old one. Without a marker nothing forces a republish, so the - // stale-lineage archive stays newest until the next scheduled run (up to a - // week for a tracker, which is opt-in besides) and no age check catches it, - // because the file is hours old and simply wrong. Mark the combos DUE so - // the publisher pulls them into its next plan regardless of schedule or - // tracker opt-in. - // - // Best-effort by design: the wipes already happened, so a bookkeeping - // failure must never abort the restart pass and leave the stack down. It is - // reported loudly instead, with the command to publish by hand. - const reindexedModules = reindexAffectedModules({ - node: resetNode, utxoTracker: resetUtxoTracker, decoder: resetDecoder, indexer: resetIndexer - }) - if (reindexedModules.length > 0) { - try { - const marked = recordReindex(reindexedModules, coin, network, { reason: `reset ${service}` }) - if (marked.length > 0) { - console.log(`Marked ${marked.length} bootstrap combo(s) for republish after this reindex: ${marked.join(', ')}`) - } else { - throw new Error('the republish ledger could not be written') - } - } catch (err) { - console.warn('WARNING: could not record this reindex in the bootstrap republish ledger: ' - + ((err && err.message) ? err.message : err)) - console.warn(' The published archives for these combos are now from the PRE-reset lineage and') - console.warn(' nothing will force a republish. Republish by hand once the stack has caught up:') - for (const module of reindexedModules) { - console.warn(` xchain-node bootstrap create ${module} ${coin} ${network}`) - } - } - } - - console.log(`Restarting ${coin} ${network} services...`) - // Track restart failures instead of swallowing them: a silent skip here - // left a wiped stack DOWN (node never restarted, every dependent service - // crash-looped) while `reset` still reported success. "Not installed" - // (registry miss) stays a legitimate skip; a failed docker start gets one - // retry, then is reported loudly at the end. - const startFailures = [] - for (const module of modulesToStop) { - let containerId = null - try { - containerId = await db.getModuleContainerStrict(module, coin, network) - } catch (err) { - // Strict, like the stop loop: the swallowing read answered null on a - // SQL error too, so a registry blip here left a just-wiped service - // DOWN and still reported a clean reset (uuid:846cc40d). Nothing can - // be undone at this point, so report it with the other start - // failures rather than aborting. - startFailures.push({ module, error: `registry lookup failed (${failureReason(err)})` }) - continue - } - // A SUCCESSFUL read with no row is "not installed" and stays a skip. - // Without this check startContainer(null) fails on every branch, and - // every `reset all` on mainnet/testnet (where the regtest-only miner has - // no registry row) reports a false failure after the reset actually - // succeeded (uuid:fd7cc224). - if (!containerId) continue /* not installed, skip */ - try { - await startContainer(containerId) - } catch (firstErr) { - console.warn(`Failed to start ${module} (${firstErr && firstErr.message}); retrying in 3s...`) - await sleep(3000) - try { - await startContainer(containerId) - } catch (retryErr) { - startFailures.push({ module, error: (retryErr && retryErr.message) || String(retryErr) }) - } - } - } - - // Workaround for a known race: the decoder + indexer's initial pool - // connections sometimes lose the connection mid-startup right after a - // DROP DATABASE / CREATE DATABASE cycle (their inner retry-on-connect - // helps but doesn't fully cover the case where Node throws before the - // retry loop is reached). A simple "settle then bounce" of decoder + - // indexer after the first start pass is empirically deterministic and - // costs ~5s on the happy path. - const bounceCandidates = [XChainService.XCHAIN_DECODER, XChainService.XCHAIN_INDEXER] - .filter((m) => modulesToStop.includes(m)) - if (bounceCandidates.length > 0) { - await sleep(5000) - for (const module of bounceCandidates) { - try { - const containerId = await db.getModuleContainer(module, coin, network) - // Latent today only because the catch below hides a null-arg - // failure; guard explicitly so a future narrower catch stays - // correct (uuid:fd7cc224 sibling site). - if (!containerId) continue - // restartContainer = docker stop + docker start; sufficient to - // re-enter Node's bootstrap with the freshly-created DB ready. - await restartContainer(containerId) - } catch { /* not installed, skip */ } - } - } - - await statusChanged() - - if (startFailures.length > 0) { - const detail = startFailures.map((f) => `${f.module}: ${f.error}`).join('; ') - throw new Error(`reset completed but ${startFailures.length} service(s) failed to restart: ${detail}. Start them manually (docker start) or re-run reset.`) - } - return true -} +const repoRefs = require('./module_operations/repo_refs') +const sharedServices = require('./module_operations/shared_services') +const updateOperations = require('./module_operations/update_modules') +const recreateOperations = require('./module_operations/recreate_modules') +const uninstallOperations = require('./module_operations/uninstall_modules') +const moduleControls = require('./module_operations/module_controls') +const resetOperations = require('./module_operations/reset_modules') + +const dependencies = { + path, fs, readline, execFileAsync, + NODE_MODULE_NAME, DB_MODULE_NAME, HUB_MODULE_NAME, EXPLORER_MODULE_NAME, + SYNC_MODULE_NAME, XChainService, SEP, dataDir, EXTERNAL_DB, Coin, + CoinTickerSymbol, Network, DEFAULT_MODULE_BRANCH, db, sleep, + getDockerContainerImageName, getUtxoTrackerVolumeName, getDockerNetwork, + createDockerNetwork, probeContainerPresenceByName, stopContainer, + stopContainerByName, startContainer, restartContainer, execContainer, + shellContainer, logContainer, startDockerMonitor, waitContainer, + saveContainerLogs, getContainerBindMounts, removeContainer, + stopModuleContainer, buildDatabaseModule, resetDatabases, + clearHubPriceIngestWatermark, purgeHubCrossChainRows, + manualHubCrossChainPurgeStatements, getDatabaseContainerId, + pingExternalDatabase, getModuleBranch, installModule, uninstallModule, + assertHubNotBehind, assertRequiredMigrationsApplied, statusChanged, + reindexAffectedModules, recordReindex, config, bootstrapService, + databaseService, explorerService, hubService, installTargetService, + moduleService, nodeService, releaseManifestService, stateModule, + validatorService, versionService +} + +repoRefs.configure(dependencies) +dependencies.withInstallTarget = repoRefs.withInstallTarget +dependencies.confirmDestructiveReset = repoRefs.confirmDestructiveReset +dependencies.isNoSuchContainerError = repoRefs.isNoSuchContainerError +dependencies.isNoSuchVolumeError = repoRefs.isNoSuchVolumeError +dependencies.failureReason = repoRefs.failureReason +dependencies.restartStoppedModules = repoRefs.restartStoppedModules +dependencies.resolveNodeDataPath = repoRefs.resolveNodeDataPath +sharedServices.configure(dependencies) +updateOperations.configure(dependencies) +recreateOperations.configure(dependencies) +uninstallOperations.configure(dependencies) +moduleControls.configure(dependencies) +dependencies.restartResetModules = moduleControls.restartResetModules +resetOperations.configure(dependencies) module.exports = { - installModules, - syncSharedServicesAfterInstall, - updateModules, - recreateModules, - uninstallModules, - logModules, - monitorModules, - restartModules, - stopModules, - startModules, - execModules, - clearDecoderReorgHalt, - shellModule, - runE2ETest, - resetModules + installModules: sharedServices.installModules, + syncSharedServicesAfterInstall: sharedServices.syncSharedServicesAfterInstall, + updateModules: updateOperations.updateModules, + recreateModules: recreateOperations.recreateModules, + uninstallModules: uninstallOperations.uninstallModules, + logModules: moduleControls.logModules, + monitorModules: moduleControls.monitorModules, + restartModules: moduleControls.restartModules, + stopModules: moduleControls.stopModules, + startModules: moduleControls.startModules, + execModules: moduleControls.execModules, + clearDecoderReorgHalt: moduleControls.clearDecoderReorgHalt, + shellModule: moduleControls.shellModule, + runE2ETest: repoRefs.runE2ETest, + resetModules: resetOperations.resetModules } diff --git a/src/operations/module_operations/module_controls.js b/src/operations/module_operations/module_controls.js new file mode 100644 index 0000000..53cc674 --- /dev/null +++ b/src/operations/module_operations/module_controls.js @@ -0,0 +1,277 @@ +'use strict' + +let XChainService, failureReason, sleep, db, execContainer, getDockerContainerImageName, logContainer, restartContainer, shellContainer, startContainer, startDockerMonitor, statusChanged, stopContainerByName, stopModuleContainer + +function configure(dependencies) { + ({ XChainService, failureReason, sleep, db, execContainer, getDockerContainerImageName, logContainer, restartContainer, shellContainer, startContainer, startDockerMonitor, statusChanged, stopContainerByName, stopModuleContainer } = dependencies) +} + +async function logModules(servicesList, follow = true) { + const moduleContainerIds = [] + for (const nextCoin in servicesList) { + for (const nextNetwork in servicesList[nextCoin]) { + for (const nextModule of servicesList[nextCoin][nextNetwork]) { + const containerId = await db.getModuleContainer(nextModule, nextCoin, nextNetwork) + if (!containerId) continue + moduleContainerIds.push({ + name: getDockerContainerImageName(nextModule, nextCoin, nextNetwork), + id: containerId + }) + } + } + } + + if (moduleContainerIds.length > 0) { + if (follow) { + // A single interleaved TTY stream only makes sense for one + // container; warn instead of silently dropping the rest so the + // operator knows N-1 services are omitted from `tail all`. + if (moduleContainerIds.length > 1) { + const omitted = moduleContainerIds.slice(1).map(c => c["name"]).join(", ") + console.log("Following only " + moduleContainerIds[0]["name"] + "; omitted: " + omitted) + } + const moduleName = moduleContainerIds[0]["name"] + console.log("") + console.log("") + console.log("####" + moduleName + " LOGS####") + console.log("") + await logContainer(moduleContainerIds[0]["id"], follow) + } else { + // Non-follow dumps can safely iterate every selected service in + // sequence (no shared TTY to interleave). + for (const container of moduleContainerIds) { + console.log("") + console.log("") + console.log("####" + container["name"] + " LOGS####") + console.log("") + await logContainer(container["id"], follow) + } + } + } else { + console.log("No service was selected") + } + return true +} + +async function monitorModules(servicesList, follow = true) { + const moduleContainerIds = [] + for (const nextCoin in servicesList) { + for (const nextNetwork in servicesList[nextCoin]) { + for (const nextModule of servicesList[nextCoin][nextNetwork]) { + const containerId = await db.getModuleContainer(nextModule, nextCoin, nextNetwork) + if (!containerId) continue + moduleContainerIds.push({ + name: getDockerContainerImageName(nextModule, nextCoin, nextNetwork), + id: containerId + }) + } + } + } + await startDockerMonitor(moduleContainerIds, follow) + return true +} + +async function restartModules(servicesList) { + for (const nextCoin in servicesList) { + for (const nextNetwork in servicesList[nextCoin]) { + for (const nextModule of servicesList[nextCoin][nextNetwork]) { + const containerId = await db.getModuleContainer(nextModule, nextCoin, nextNetwork) + if (!containerId) continue + try { + await restartContainer(containerId) + await statusChanged() + } catch (err) { + console.log(err) + } + } + } + } + return true +} + +async function stopModules(servicesList) { + for (const nextCoin in servicesList) { + for (const nextNetwork in servicesList[nextCoin]) { + for (const nextModule of servicesList[nextCoin][nextNetwork]) { + const containerId = await db.getModuleContainer(nextModule, nextCoin, nextNetwork) + if (!containerId) continue + try { + // With the service's budget, not docker's ten seconds: a bare + // `docker stop` on a container created before the budget was + // stamped on it is a coin flip for a service mid-block. + await stopModuleContainer(stopContainerByName, nextModule, nextCoin, nextNetwork, containerId) + } catch (err) { + console.log(err) + } + } + } + } + return true +} + +async function startModules(servicesList) { + for (const nextCoin in servicesList) { + for (const nextNetwork in servicesList[nextCoin]) { + for (const nextModule of servicesList[nextCoin][nextNetwork]) { + const containerId = await db.getModuleContainer(nextModule, nextCoin, nextNetwork) + if (!containerId) continue + try { + await startContainer(containerId) + } catch (err) { + console.log(err) + } + } + } + } + return true +} + +// Audited clear of a decoder's durable REORG_HALT marker, run inside the decoder +// container so it uses the service's own DB credentials and code +// (xchain-decoder/src/clear-reorg-halt.js checks the database is intact, then +// records the clear as an events row with the reason). One decoder per +// coin/network; `servicesList` is the filtered map the CLI builds. Returns true +// only when every targeted decoder answered exit 0. +async function clearDecoderReorgHalt(servicesList, { reason, force = false, dryRun = false } = {}) { + // A dry run writes nothing, so it runs without a reason; a real clear records one. + const reasonText = typeof reason === 'string' ? reason.trim() : '' + if (!dryRun && reasonText.length < 8) { + console.log('clear-reorg-halt: --reason must say, in at least 8 characters, why this database is known good; it is recorded with the clear.') + return false + } + const args = ['node', 'src/clear-reorg-halt.js'] + if (reasonText) args.push('--reason', reasonText) + if (force) args.push('--force') + if (dryRun) args.push('--dry-run') + let targeted = 0 + let ok = true + for (const nextCoin in servicesList) { + for (const nextNetwork in servicesList[nextCoin]) { + if (!servicesList[nextCoin][nextNetwork].includes(XChainService.XCHAIN_DECODER)) continue + const containerId = await db.getModuleContainer(XChainService.XCHAIN_DECODER, nextCoin, nextNetwork) + if (!containerId) { + console.log('clear-reorg-halt: no xchain-decoder container is installed for ' + nextCoin + ' ' + nextNetwork) + ok = false + continue + } + targeted++ + try { + const out = await execContainer(containerId, args) + if (out) console.log(out) + } catch (err) { + // The script prints its refusal on stderr and exits non-zero; docker + // exec surfaces that as an error whose stdout/stderr carry the text. + const text = [err && err.stdout, err && err.stderr].filter(Boolean).join('\n').trim() + console.log(text || ('clear-reorg-halt: failed for ' + nextCoin + ' ' + nextNetwork + ': ' + (err && err.message))) + ok = false + } + } + } + if (targeted === 0 && ok) { + console.log('clear-reorg-halt: no xchain-decoder matched the given chain and network') + return false + } + return ok +} + +async function execModules(servicesList, command) { + const commandArgs = command.split(/\s+/) + for (const nextCoin in servicesList) { + for (const nextNetwork in servicesList[nextCoin]) { + for (const nextModule of servicesList[nextCoin][nextNetwork]) { + const containerId = await db.getModuleContainer(nextModule, nextCoin, nextNetwork) + if (!containerId) continue + try { + const execStdOut = await execContainer(containerId, commandArgs) + console.log(execStdOut) + } catch (err) { + console.log(err) + } + } + } + } + return true +} + +async function shellModule(servicesList) { + for (const nextCoin in servicesList) { + for (const nextNetwork in servicesList[nextCoin]) { + for (const nextModule of servicesList[nextCoin][nextNetwork]) { + const containerId = await db.getModuleContainer(nextModule, nextCoin, nextNetwork) + if (!containerId) continue + try { + await shellContainer(containerId) + return true + } catch (err) { + console.log(err) + } + } + } + } + return true +} + +async function restartResetModules(context) { + const { coin, modulesToStop, network } = context + console.log(`Restarting ${coin} ${network} services...`) + // Track restart failures instead of swallowing them: a silent skip here left a wiped stack DOWN (node never restarted, every dependent service crash-looped) while `reset` still reported success. + // "Not installed" (registry miss) stays a legitimate skip; a failed docker start gets one retry, then is reported loudly at the end. + const startFailures = [] + for (const module of modulesToStop) { + let containerId = null + try { + containerId = await db.getModuleContainerStrict(module, coin, network) + } catch (err) { + // Strict, like the stop loop: the swallowing read answered null on a SQL error too, so a registry blip here left a just-wiped service DOWN and still reported a clean reset + // (uuid:846cc40d). Nothing can be undone at this point, so report it with the other start failures rather than aborting. + startFailures.push({ module, error: `registry lookup failed (${failureReason(err)})` }) + continue + } + // A SUCCESSFUL read with no row is "not installed" and stays a skip. Without this check startContainer(null) fails on every branch, and every `reset all` on mainnet/testnet (where the + // regtest-only miner has no registry row) reports a false failure after the reset actually succeeded (uuid:fd7cc224). + if (!containerId) continue /* not installed, skip */ + try { + await startContainer(containerId) + } catch (firstErr) { + console.warn(`Failed to start ${module} (${firstErr && firstErr.message}); retrying in 3s...`) + await sleep(3000) + try { + await startContainer(containerId) + } catch (retryErr) { + startFailures.push({ module, error: (retryErr && retryErr.message) || String(retryErr) }) + } + } + } + return bounceResetModules({ ...context, startFailures }) +} + +async function bounceResetModules(context) { + const { coin, modulesToStop, network, startFailures } = context + // Workaround for a known race: the decoder + indexer's initial pool connections sometimes lose the connection mid-startup right after a DROP DATABASE / CREATE DATABASE cycle (their inner + // retry-on-connect helps but doesn't fully cover the case where Node throws before the retry loop is reached). A simple "settle then bounce" of decoder + indexer after the first start pass is + // empirically deterministic and costs ~5s on the happy path. + const bounceCandidates = [XChainService.XCHAIN_DECODER, XChainService.XCHAIN_INDEXER] + .filter((m) => modulesToStop.includes(m)) + if (bounceCandidates.length > 0) { + await sleep(5000) + for (const module of bounceCandidates) { + try { + const containerId = await db.getModuleContainer(module, coin, network) + // Latent today only because the catch below hides a null-arg failure; guard explicitly so a future narrower catch stays correct (uuid:fd7cc224 sibling site). + if (!containerId) continue + // restartContainer = docker stop + docker start; sufficient to re-enter Node's bootstrap with the freshly-created DB ready. + await restartContainer(containerId) + } catch { /* not installed, skip */ } + } + } + + await statusChanged() + + if (startFailures.length > 0) { + const detail = startFailures.map((f) => `${f.module}: ${f.error}`).join('; ') + throw new Error(`reset completed but ${startFailures.length} service(s) failed to restart: ${detail}. Start them manually (docker start) or re-run reset.`) + } + return true +} + +module.exports = { configure, logModules, monitorModules, restartModules, stopModules, startModules, clearDecoderReorgHalt, execModules, shellModule, restartResetModules } diff --git a/src/operations/module_operations/recreate_modules.js b/src/operations/module_operations/recreate_modules.js new file mode 100644 index 0000000..fe53606 --- /dev/null +++ b/src/operations/module_operations/recreate_modules.js @@ -0,0 +1,89 @@ +'use strict' + +let DB_MODULE_NAME, HUB_MODULE_NAME, NODE_MODULE_NAME, XChainService, databaseService, db, getDockerContainerImageName, moduleService, probeContainerPresenceByName, statusChanged +let RECREATE_UNSUPPORTED_MODULES + +function configure(dependencies) { + ({ DB_MODULE_NAME, HUB_MODULE_NAME, NODE_MODULE_NAME, XChainService, databaseService, db, getDockerContainerImageName, moduleService, probeContainerPresenceByName, statusChanged } = dependencies) + RECREATE_UNSUPPORTED_MODULES = [NODE_MODULE_NAME, DB_MODULE_NAME] +} + +// Modules whose container is not created from the config map by buildAndUp: the crypto node goes through buildCryptoNode and the database container through buildDatabaseModule, so neither has config env for this verb to re-stamp. +/** + * Re-stamp a service's container from the CURRENT config without touching its image. + * + * A container freezes its env at `docker run`, so a config value it got wrong (a DB + * password from another install's config store) cannot be corrected in place. + * `update` corrects it only by also re-cloning from GitHub and rebuilding, which turns + * a credential repair into an unreviewed version change on a live venue. This keeps the + * image byte-identical and changes only what the config map now says. + * + * Reports what it recreated, for the same reason `update` does: an unsupported + * module (node, database) was logged and skipped while the command still + * exited 0, so `recreate node && echo ok` printed ok having recreated nothing. + * + * @param {Object} servicesList + * @returns {Promise<{recreated: Array, skipped: Array}>} + */ +async function recreateModules(servicesList) { + const { buildAndUp } = moduleService + const { setDatabaseParameters, setHubDatabaseParameters } = databaseService + const outcome = { recreated: [], skipped: [] }, failures = [] + let touchedDbModule = false + let touchedHubModule = false + for (const nextCoin in servicesList) { + for (const nextNetwork in servicesList[nextCoin]) { + for (const nextModule of servicesList[nextCoin][nextNetwork]) { + if (RECREATE_UNSUPPORTED_MODULES.includes(nextModule)) { + // Still a continue: `recreate all` legitimately sweeps past the node and the database. What changed is that the skip is now recorded, so a run that recreated NOTHING can be reported as the failed request it is instead of exiting 0. The database has no `update` to redirect to either: that verb refuses it for the same reason (no container built from the config map, no in-place image upgrade). Say the real remedy. + const remedy = nextModule === DB_MODULE_NAME + ? "; the database container must be removed manually and reinstalled" + : "; use `update " + nextModule + "` instead" + console.log("recreate does not apply to " + nextModule + remedy) + outcome.skipped.push({ module: nextModule, coin: nextCoin, network: nextNetwork, reason: 'not-recreatable' }) + continue + } + const moduleContainerId = await db.getModuleContainer(nextModule, nextCoin, nextNetwork) + if (!moduleContainerId) { + // No registry row is TWO different states and this verb must not conflate them. Registry drift (row lost, container still up) has to keep recreating: dropping it was what made `update node` a silent no-op. An explicitly uninstalled venue must NOT: uninstall removes the container and the row but leaves the image tag, so buildAndUp's reuseImage check passes, `null` reads as "nothing to tear down", and the operator gets back a service they tore down, built from a stale image and re-stamped into the registry that status/precheck/autoheal trust. Discriminate on a POSITIVE docker answer only: 'unknown' is a daemon hiccup, not an absence. + const presence = await probeContainerPresenceByName( + getDockerContainerImageName(nextModule, nextCoin, nextNetwork)) + if (presence === 'gone') { + console.warn(`recreate: ${nextModule} (${nextCoin} ${nextNetwork}) has no container; nothing to recreate. Install it first.`) + outcome.skipped.push({ module: nextModule, coin: nextCoin, network: nextNetwork, reason: 'not-installed' }) + continue + } + } + try { + await buildAndUp(nextModule, nextCoin, nextNetwork, moduleContainerId, false, null, { reuseImage: true }) + } catch (err) { + // Visiting the rest of the sweep after one venue fails follows uninstallModules: `recreate all` was already half-applied by the time it threw, and stopping there hid which venues had been touched behind one flat `recreate failed:`. The run still REJECTS below, naming every venue - a failure never becomes a skip. + const why = (err && err.message) ? err.message : String(err) + console.error(`recreate: ${nextModule} (${nextCoin} ${nextNetwork}) FAILED: ${why}`) + failures.push({ module: nextModule, coin: nextCoin, network: nextNetwork, reason: why }) + continue + } + outcome.recreated.push({ module: nextModule, coin: nextCoin, network: nextNetwork }) + if (nextModule === XChainService.XCHAIN_DECODER || nextModule === XChainService.XCHAIN_INDEXER) { + touchedDbModule = true + } + if (nextModule === HUB_MODULE_NAME) { + touchedHubModule = true + } + } + } + } + // Provision AFTER every container is back on the config values, so the drift guard in setDatabaseParameters sees the state we just converged rather than the one that made the recreate necessary. + if (touchedDbModule) await setDatabaseParameters() + // Same rule for the SHARED hub account, and it matters most on this verb: the recreated hub starts on the config store's HUB_DB_PASS, so without rotating the live 'xchain_hub'@'%' account to match, `recreate xchain-hub` hands the hub a password MariaDB never received and it crash-loops on ER_ACCESS_DENIED. The `update` path rotates here for the same reason (ModuleService installModule). + if (touchedHubModule) await setHubDatabaseParameters() + await statusChanged() + if (failures.length) { + // Rejecting AFTER provisioning is deliberate: the venues that did come back start on the config store's password and would crash-loop on ER_ACCESS_DENIED if the run bailed before rotating their accounts. + throw new Error('recreate failed for ' + failures.length + ' module(s): ' + + failures.map(f => `${f.module} (${f.coin} ${f.network}): ${f.reason}`).join('; ')) + } + return outcome +} + +module.exports = { configure, recreateModules } diff --git a/src/operations/module_operations/repo_refs.js b/src/operations/module_operations/repo_refs.js new file mode 100644 index 0000000..9e5d1f1 --- /dev/null +++ b/src/operations/module_operations/repo_refs.js @@ -0,0 +1,170 @@ +'use strict' + +let NODE_MODULE_NAME, db, fs, getContainerBindMounts, getDockerContainerImageName, readline, startContainer, DEFAULT_MODULE_BRANCH, XChainService, dataDir, installModule, installTargetService, path, releaseManifestService, removeContainer, saveContainerLogs, waitContainer + +function configure(dependencies) { + ({ NODE_MODULE_NAME, db, fs, getContainerBindMounts, getDockerContainerImageName, readline, startContainer, DEFAULT_MODULE_BRANCH, XChainService, dataDir, installModule, installTargetService, path, releaseManifestService, removeContainer, saveContainerLogs, waitContainer } = dependencies) +} + +// Resolve the operator's single ref slot into an install target and publish it +// for the duration of the run, so every module clone and every bundled-library +// staging inside it resolves against ONE decision (release-management spec +// section 11). Cleared in a finally, or a later branch install in the same +// process would inherit a stale pin. +async function withInstallTarget(ref, run, { fallbackToBranch = true } = {}) { + const { + resolveInstallTarget, setActiveTarget, clearActiveTarget + } = releaseManifestService + const { recordInstallTarget } = installTargetService + + const target = await resolveInstallTarget(ref, { defaultBranch: DEFAULT_MODULE_BRANCH, fallbackToBranch }) + + if (target.kind === 'release') { + console.log(`Installing XChain ${target.tag} (${target.resolvedFrom}); every component is manifest-pinned.`) + } else { + console.log(`Installing from branch '${target.ref}' (UNRELEASED: tracking install, no version pinning).`) + } + + // What this node is on, for the next no-ref `update` to converge on. + recordInstallTarget(target) + setActiveTarget(target) + try { + return await run(target) + } finally { + clearActiveTarget() + } +} +// `ref` is the ref to clone the e2e-test suite at, normally the same one the +// stack under test was installed at. Null keeps the default-branch behaviour +// every caller had before the option existed. +async function runE2ETest(coin, network, testName = null, grep = null, script = null, ref = null) { + let dockerCmdArgs = null + if (script) { + // Run an arbitrary e2e npm script (e.g. test:security, test:perf:budget) so CI + // can drive the stack-dependent suites beyond the default action suite. Takes + // precedence over testName; the e2e-test image carries these scripts. + dockerCmdArgs = ['npm', 'run', script] + } else if (testName) { + dockerCmdArgs = ['npx', 'mocha', '--timeout', '0', '--exit', + '--require', './test/initialCheck.test.js', + `test/actions/${testName}.test.js`] + if (grep) dockerCmdArgs.push('--grep', grep) + } + const containerId = await installModule(XChainService.XCHAIN_E2E_TEST, coin, network, true, null, true, ref, dockerCmdArgs) + + console.log("Running e2e tests, please wait...") + const exitCode = await waitContainer(containerId) + + const now = new Date() + const timestamp = now.toISOString().replace(/[:.]/g, '-').slice(0, 19) + const logFile = path.join(dataDir, 'e2e-logs', `${coin}-${network}-${timestamp}.log`) + + await saveContainerLogs(containerId, logFile) + await removeContainer(containerId) + + return { logFile, exitCode } +} + +// Prompts the operator to type "yes" before a destructive reset proceeds. Reused instead of duplicated so every call site aborts the exact same way on a non-affirmative answer. Not called at all when +// the caller passes force=true (CI/scripted resets). +async function confirmDestructiveReset(coin, network, targets) { + if (!process.stdin.isTTY) { + throw new Error( + 'reset: refusing to run a destructive reset on a non-interactive terminal without --yes. ' + + 'Re-run with --yes to confirm.' + ) + } + console.warn(`\nWARNING: this will IRREVERSIBLY destroy ${coin} ${network} data.`) + console.warn(` Affected stores: ${targets.join(', ')}`) + console.warn(' This forces a full resync afterward. There is no undo.\n') + const rl = readline.createInterface({ input: process.stdin, output: process.stdout }) + const answer = await new Promise((resolve) => { + rl.question('Type "yes" to confirm: ', resolve) + }) + rl.close() + return answer.trim().toLowerCase() === 'yes' +} + +// True when a docker error means the container is already gone. Matched the same way DockerService.removeContainer matches it; execFile puts docker's stderr into the error message, and stopContainer +// can also reject a plain string, which is a real failure and must not be read as a miss here. +function isNoSuchContainerError(err) { + if (!err || typeof err === 'string') return false + return /no such container/i.test(String(err.message || err.stderr || '')) +} + +// True when a docker error means the named volume is already gone. Same rule isNoSuchContainerError uses, and the one DockerService.probeContainerPresenceByName states: docker SAYING "no such volume" +// is the only thing that means absent; every other failure is unknown and must be treated as possibly-present. +function isNoSuchVolumeError(err) { + if (!err || typeof err === 'string') return false + return /no such volume/i.test(String(err.message || err.stderr || '')) +} + +// The operator-facing reason for a rejected docker or registry call. Handles the bare-string rejection stopContainer can produce as well as an Error. +function failureReason(err) { + return (err && err.message) || String(err) +} + +// Put back the services an aborted reset already stopped, so the abort leaves the stack as it found it rather than half torn down. Returns the modules that could not be restarted, for the operator +// message. +// +// Reads the registry strictly: getModuleContainer answers null on a SQL error as well as on a miss, so a rollback run during the very registry outage that caused the abort restarted nothing and still +// reported zero failures (uuid:846cc40d). A lookup that fails now lands in `failed` and is named in the STILL DOWN line. +async function restartStoppedModules(modules, coin, network) { + const failed = [] + for (const module of modules) { + try { + const containerId = await db.getModuleContainerStrict(module, coin, network) + if (!containerId) continue + await startContainer(containerId) + } catch { + failed.push(module) + } + } + return failed +} + +/** + * Resolve the HOST directory that holds this chain's node datadir, asking the + * node container itself first. + * + * `dataDir` is env-derived (XCHAIN_NODE_DATA_DIR, else the in-repo data/), so a + * shell that never sourced the operator's profile resolves a path the stack has + * never used. The wipe was guarded on fs.existsSync of that path, so the guard + * went silently false and `reset all` reported success with the chain untouched. + * The container name is already resolved deterministically from the prefix and + * coin/network, so use that same key to read the datadir off the container's own + * bind mounts: whatever the daemon actually writes to is what a reset must wipe. + * + * Falls back to the env-derived path only when it really is on disk. Returns + * path=null when neither answer exists, and the caller fails closed on that + * rather than skipping the wipe. + * + * @returns {Promise<{path: (string|null), resolvedFrom: (string|null), configuredPath: string, containerName: string}>} + */ +async function resolveNodeDataPath(coin, network) { + const containerName = getDockerContainerImageName(NODE_MODULE_NAME, coin, network) + const configuredPath = path.join(dataDir, NODE_MODULE_NAME, coin, network) + + let mounts = [] + try { + mounts = await getContainerBindMounts(containerName) + } catch { /* no container, or docker unreachable: fall through to the configured path */ } + const dataMount = (Array.isArray(mounts) ? mounts : []) + .find(m => m && m.destination === `/root/.${coin}` && m.source) + if (dataMount) { + return { + path: dataMount.source, + resolvedFrom: `the /root/.${coin} bind mount of container ${containerName}`, + configuredPath, + containerName + } + } + + if (fs.existsSync(configuredPath)) { + return { path: configuredPath, resolvedFrom: 'the configured data dir', configuredPath, containerName } + } + + return { path: null, resolvedFrom: null, configuredPath, containerName } +} + +module.exports = { configure, withInstallTarget, runE2ETest, confirmDestructiveReset, isNoSuchContainerError, isNoSuchVolumeError, failureReason, restartStoppedModules, resolveNodeDataPath } diff --git a/src/operations/module_operations/reset_modules.js b/src/operations/module_operations/reset_modules.js new file mode 100644 index 0000000..8f7aa6a --- /dev/null +++ b/src/operations/module_operations/reset_modules.js @@ -0,0 +1,387 @@ +'use strict' + +let confirmDestructiveReset, failureReason, isNoSuchContainerError, isNoSuchVolumeError, resolveNodeDataPath, restartResetModules, restartStoppedModules, Coin, CoinTickerSymbol, EXTERNAL_DB, HUB_MODULE_NAME, Network, NODE_MODULE_NAME, XChainService, clearHubPriceIngestWatermark, config, dataDir, db, execFileAsync, fs, getContainerBindMounts, getDatabaseContainerId, getDockerContainerImageName, getUtxoTrackerVolumeName, manualHubCrossChainPurgeStatements, nodeService, path, pingExternalDatabase, purgeHubCrossChainRows, readline, recordReindex, reindexAffectedModules, resetDatabases, restartContainer, sleep, startContainer, statusChanged, stopContainer +let RESETTABLE_SERVICES + +function configure(dependencies) { + ({ confirmDestructiveReset, failureReason, isNoSuchContainerError, isNoSuchVolumeError, resolveNodeDataPath, restartResetModules, restartStoppedModules, Coin, CoinTickerSymbol, EXTERNAL_DB, HUB_MODULE_NAME, Network, NODE_MODULE_NAME, XChainService, clearHubPriceIngestWatermark, config, dataDir, db, execFileAsync, fs, getContainerBindMounts, getDatabaseContainerId, getDockerContainerImageName, getUtxoTrackerVolumeName, manualHubCrossChainPurgeStatements, nodeService, path, pingExternalDatabase, purgeHubCrossChainRows, readline, recordReindex, reindexAffectedModules, resetDatabases, restartContainer, sleep, startContainer, statusChanged, stopContainer } = dependencies) + RESETTABLE_SERVICES = [ + 'all', + NODE_MODULE_NAME, + XChainService.XCHAIN_UTXO_TRACKER, + XChainService.XCHAIN_DECODER, + XChainService.XCHAIN_INDEXER + ] +} + +// The service names `reset` can act on. `reset` is the only destructive CLI path and the only one that bypasses resolveArgs/filterCommandParameters, so it must validate its own raw args: without this +// an unrecognised service (a typo, or a non-resettable module like xchain-encoder) leaves every reset flag false, so `targets` is empty, no branch fires, and resetModules returns true - the CLI exits +// 0 reporting success after resetting nothing. Fail loud instead, matching resolveArgs/rollback and the "fail fast BEFORE any destructive wipe" convention this file already follows. + +async function resetModules(service, coin, network, force = false, withIndexer = false) { + if (!RESETTABLE_SERVICES.includes(service)) { + throw new Error("reset: unknown service '" + service + "'; expected one of " + + RESETTABLE_SERVICES.join(', ')) + } + if (!Object.values(Coin).includes(coin)) { + throw new Error("reset: unknown coin '" + coin + "'; expected one of " + + Object.values(Coin).join(', ')) + } + if (!Object.values(Network).includes(network)) { + throw new Error("reset: unknown network '" + network + "'; expected one of " + + Object.values(Network).join(', ')) + } + const resetAll = service === 'all' + const resetNode = resetAll || service === NODE_MODULE_NAME + const resetUtxoTracker = resetAll || service === XChainService.XCHAIN_UTXO_TRACKER + const resetDecoder = resetAll || service === XChainService.XCHAIN_DECODER + const resetIndexer = resetAll || service === XChainService.XCHAIN_INDEXER || (withIndexer && resetDecoder) + return validateDecoderPair({ service, coin, network, force, resetAll, resetNode, resetUtxoTracker, resetDecoder, resetIndexer }) +} + +async function validateDecoderPair(context) { + const { coin, network, resetDecoder, resetIndexer } = context + // The indexer's rollback cursor IS a decoder `events` id, and the decoder never deletes those rows, so wiping the decoder alone restarts the ids under a cursor that now points past them: the + // indexer fails RE-1 and stops committing. The pair only has a coherent state when both move together. Asymmetric on purpose - resetting the indexer alone re-derives it from an intact decoder, + // which is an ordinary reindex and stays allowed. + if (resetDecoder && !resetIndexer) { + let indexerInstalled = null + try { + // Strict read: getModuleContainer answers null on a SQL error and on an unopened pool as well as on a genuine miss, so a registry blip read as "no indexer installed" and waved through the + // one wipe this guard exists to stop (uuid:7cbafa08). A lookup that FAILS is not evidence of absence. Only an empty result set still means "not installed", which stays allowed. + indexerInstalled = await db.getModuleContainerStrict(XChainService.XCHAIN_INDEXER, coin, network) + } catch (err) { + // abortBeforeAnyWipe is declared further down this function and nothing has been stopped yet, so the plain refusal is the correct shape here. + console.log(`Aborted: cannot read the ${XChainService.XCHAIN_INDEXER} registry row ` + + `(${failureReason(err)}), so it is not known whether resetting ` + + `${XChainService.XCHAIN_DECODER} alone would strand it. No data was touched.`) + return false + } + if (indexerInstalled) { + console.log(`Aborted: resetting ${XChainService.XCHAIN_DECODER} alone would leave ` + + `${XChainService.XCHAIN_INDEXER} incoherent. No data was touched.`) + console.log(" The indexer tracks reorgs by a decoder event id. Wiping the decoder restarts") + console.log(" those ids, so the indexer would abort with a reorg-cursor error (RE-1) and stop") + console.log(" committing blocks until both are rebuilt together.") + console.log(` Reset the pair: xchain-node reset ${XChainService.XCHAIN_DECODER} ${coin} ${network} --with-indexer`) + console.log(` Or the whole stack (also re-syncs the chain): xchain-node reset all ${coin} ${network}`) + return false + } + } + return resolveResetPaths(context) +} + +async function resolveResetPaths(context) { + const { coin, network, resetNode } = context + // Relocated blocks/txindex host paths (XCHAIN_NODE_BLOCKS_DIR mode): these live OUTSIDE the in-datadir path the node wipe clears, so they must be wiped explicitly and named in the confirmation, + // else a reset restarts the daemon over a stale blocks dir + stale txindex (uuid:90630038). Env-first with config/node.local fallback: a reset from a profile-less shell must still see the + // relocated stores, or it restarts the daemon over stale out-of-datadir chain data. + const { resolveBlocksDir } = nodeService + const blocksDir = await resolveBlocksDir() + const blocksHostPath = blocksDir ? `${blocksDir}/${coin}/${network}` : null + const txindexHostPath = blocksDir ? `${blocksDir}/${coin}/${network}-txindex` : null + + // Resolve the node datadir BEFORE anything is stopped or confirmed, and refuse the whole reset by name when it cannot be resolved. The old code re-derived the path from XCHAIN_NODE_DATA_DIR at + // the wipe site and skipped the wipe whenever that path was absent, so a reset run from a profile-less shell wiped the decoder/indexer DBs, left the chain in place, and exited 0; the missing + // "Clearing node data" line was the only tell. "Not installed" stays a legitimate skip, and is stated out loud. + let nodeDataPath = null + if (resetNode) { + let nodeInstalled = null + let registryReadable = true + try { + nodeInstalled = await db.getModuleContainer(NODE_MODULE_NAME, coin, network) + } catch { registryReadable = false } + + const resolved = await resolveNodeDataPath(coin, network) + if (resolved.path) { + nodeDataPath = resolved.path + if (path.resolve(nodeDataPath) !== path.resolve(resolved.configuredPath)) { + console.log(`Node datadir resolved from ${resolved.resolvedFrom}: ${nodeDataPath}`) + console.log(` (XCHAIN_NODE_DATA_DIR in this shell would have pointed at ${resolved.configuredPath})`) + } + } else if (registryReadable && !nodeInstalled) { + console.log(`No ${NODE_MODULE_NAME} container is installed for ${coin} ${network}; there is no node data to clear.`) + } else { + const envState = config.XCHAIN_NODE_DATA_DIR && config.XCHAIN_NODE_DATA_DIR.trim() !== '' + ? `set to ${process.env.XCHAIN_NODE_DATA_DIR}` + : 'UNSET in this shell (non-interactive shells do not source the profile)' + console.log(`Aborted: cannot resolve the ${coin} ${network} node datadir. No data was touched.`) + console.log(` Container ${resolved.containerName} reported no /root/.${coin} bind mount ` + + '(it is absent, or docker is unreachable from here).') + console.log(` The configured path ${resolved.configuredPath} does not exist either.`) + console.log(` XCHAIN_NODE_DATA_DIR is ${envState}.`) + console.log(' Set XCHAIN_NODE_DATA_DIR to this stack\'s data root (or make docker reachable so the') + console.log(' node container can be inspected) and re-run. Refusing rather than resetting the') + console.log(' databases around a chain that would stay untouched.') + return false + } + } + + return confirmReset({ ...context, blocksDir, blocksHostPath, txindexHostPath, nodeDataPath }) +} + +async function confirmReset(context) { + const { blocksDir, blocksHostPath, coin, force, network, nodeDataPath, resetDecoder, resetIndexer, resetNode, resetUtxoTracker, txindexHostPath } = context + if (!force) { + const targets = [] + if (resetNode) { + if (nodeDataPath) targets.push(`node datadir (${nodeDataPath})`) + if (blocksDir) { + targets.push(`relocated blocks dir (${blocksHostPath})`) + targets.push(`relocated txindex dir (${txindexHostPath})`) + } + } + if (resetUtxoTracker) targets.push(`xchain-utxo-tracker Docker volume (${getUtxoTrackerVolumeName(coin, network)})`) + if (resetDecoder) targets.push('xchain-decoder database') + if (resetIndexer) { + targets.push('xchain-indexer database') + // Named in the confirmation because it is a change to the HUB's state, not this chain's: the operator should see that the reset reaches across. + targets.push('hub price ingest fence row for this chain (price_ingest_watermarks)') + } + // A regtest chain reset is a RE-GENESIS (the datadir goes, the chain comes back from block 0), which invalidates every hub row anchored on the dead chain's blocks. Named for the same reason + // the price fence is: it is a change to the HUB's state, not just this chain's, and the operator should see that the reset reaches across. + if (resetNode && network === Network.REGTEST) { + targets.push('hub cross-chain relic rows for this network ' + + '(cross_chain_matches, cross_chain_calls, capability_snapshots)') + } + const confirmed = await confirmDestructiveReset(coin, network, targets) + if (!confirmed) { + console.log('Aborted: reset was not confirmed. No data was touched.') + return false + } + } + return checkResetDatabase(context) +} + +async function checkResetDatabase(context) { + const { coin, network, resetAll, resetDecoder, resetIndexer, resetNode, resetUtxoTracker } = context + // Fail fast BEFORE any destructive wipe: a DB reset needs a working MariaDB, and resetDatabases is not reached until AFTER the stop loop and every wipe below, so discovering the problem there + // half-destroys the stack. In docker mode the failure is `docker exec null` with the container gone (uuid:6f6584dc); in EXTERNAL_DB mode it is an unreachable host, or a getExternalDbConfig throw + // on a partial env, and NOTHING probed for it (uuid:41887889). Both modes are probed here. It sits ahead of the stop loop, not after it: this abort returns before the restart pass, so probing + // later left every already-stopped service DOWN while still reporting that no data was touched (uuid:bb190060). + const dbResetNeeded = resetDecoder || resetIndexer + if (dbResetNeeded) { + if (EXTERNAL_DB) { + const probe = await pingExternalDatabase() + if (!probe.ok) { + console.log(`Aborted: cannot reach the external MariaDB at ${probe.host}:${probe.port}` + + ` (${probe.error}). No data was touched.`) + return false + } + } else { + const dbContainerId = await getDatabaseContainerId() + if (!dbContainerId) { + console.log('Aborted: MariaDB container not found; install the database first. No data was touched.') + return false + } + } + } + + const modulesToStop = [] + if (resetNode) modulesToStop.push(NODE_MODULE_NAME) + if (resetUtxoTracker) modulesToStop.push(XChainService.XCHAIN_UTXO_TRACKER) + if (resetDecoder) modulesToStop.push(XChainService.XCHAIN_DECODER) + if (resetIndexer) modulesToStop.push(XChainService.XCHAIN_INDEXER) + if (resetAll) modulesToStop.push(XChainService.XCHAIN_REGTEST_MINER) + return stopResetModules({ ...context, modulesToStop }) +} + +async function stopResetModules(context) { + const { coin, network, modulesToStop } = context + console.log(`Stopping ${coin} ${network} services...`) + // Abort before any wipe when a target will not stop, and put back whatever was already stopped. The bare catch this replaced swallowed EVERY stopContainer rejection as "not installed", so a + // daemon that failed to stop kept reading and writing the store while the wipes below deleted it (uuid:9c88cfe6). Only a "no such container" miss is still a legitimate skip; the registry miss is + // already handled by the null check. + const stoppedModules = [] + // Refuse the whole reset, put back whatever this run stopped, and report it. Only reachable while nothing has been wiped yet, which is why it may still promise that no data was touched. + const abortBeforeAnyWipe = async (reason) => { + const restartFailures = await restartStoppedModules(stoppedModules, coin, network) + console.log(`Aborted: ${reason}. No data was touched.`) + if (stoppedModules.length > 0) { + console.log(` Restarted ${stoppedModules.length - restartFailures.length} of ` + + `${stoppedModules.length} already-stopped service(s).`) + } + if (restartFailures.length > 0) { + console.log(` STILL DOWN, start by hand: ${restartFailures.join(', ')}`) + } + return false + } + for (const module of modulesToStop) { + let containerId = null + try { + containerId = await db.getModuleContainerStrict(module, coin, network) + } catch (err) { + // A registry read that FAILED is not evidence the module is absent. The swallowing read this replaced answered null on any SQL error, so a blip after the reachability precheck made a live + // indexer look uninstalled: the loop skipped stopping it and the wipes below, resetDatabases included, ran underneath it (uuid:846cc40d). + return await abortBeforeAnyWipe( + `cannot read the ${module} registry row (${failureReason(err)}), so it is not known ` + + 'whether that service is running') + } + // A SUCCESSFUL read with no row is still a legitimate "not installed" skip; without this check stopContainer(null) fails and ABORTS the reset (uuid:fd7cc224 sibling site). + if (!containerId) continue + try { + await stopContainer(containerId) + stoppedModules.push(module) + } catch (err) { + if (isNoSuchContainerError(err)) continue + return await abortBeforeAnyWipe(`${module} failed to stop (${failureReason(err)})`) + } + } + return classifyResetVolume({ ...context, stoppedModules, abortBeforeAnyWipe }) +} + +async function classifyResetVolume(context) { + const { abortBeforeAnyWipe, coin, network, resetUtxoTracker } = context + // Classify the tracker volume's presence BEFORE the first wipe. The wipe below swallowed every failure as "the volume may not exist", so a permission error, an unreachable daemon or a failed + // alpine pull left stale tracker data behind while the decoder/indexer databases were dropped and the run reported success (uuid:e24c98d4). Only docker SAYING "no such volume" is absence; + // anything else refuses here, while the stack is whole. + let utxoVolumeName = null + let utxoVolumePresent = false + if (resetUtxoTracker) { + // Name the volume through the shared helper so it carries NODE_PREFIX (uuid:7523dd94): an unprefixed name resolves to the DEFAULT_NODE_PREFIX stack's volume and wipes that one instead of the + // intended target. + utxoVolumeName = getUtxoTrackerVolumeName(coin, network) + try { + await execFileAsync('docker', ['volume', 'inspect', utxoVolumeName]) + utxoVolumePresent = true + } catch (err) { + if (!isNoSuchVolumeError(err)) { + return await abortBeforeAnyWipe( + `cannot determine whether the Docker volume ${utxoVolumeName} exists ` + + `(${failureReason(err)})`) + } + console.log(`No Docker volume ${utxoVolumeName} to clear.`) + } + } + return wipeResetStores({ ...context, utxoVolumeName, utxoVolumePresent }) +} + +async function wipeResetStores(context) { + const { abortBeforeAnyWipe, blocksHostPath, coin, network, nodeDataPath, resetDecoder, resetIndexer, resetNode, resetUtxoTracker, txindexHostPath, utxoVolumeName, utxoVolumePresent } = context + // Tracks whether anything irreversible has happened yet, so a later abort reports the stack's real state instead of promising an untouched one. + let nodeDataWiped = false + + if (resetNode) { + // No existsSync guard here any more: the path was resolved (and the reset refused, or the "not installed" skip announced) up top, so an unresolvable datadir can no longer read as a silent + // no-op. + if (nodeDataPath) { + console.log(`Clearing node data at ${nodeDataPath}...`) + await execFileAsync('docker', ['run', '--rm', '-v', `${nodeDataPath}:/data`, 'alpine', 'sh', '-c', 'find /data -mindepth 1 -delete']) + nodeDataWiped = true + } + // Relocated blocks/txindex (XCHAIN_NODE_BLOCKS_DIR) live outside the datadir, so wipe them here too or the daemon restarts over stale chain data (uuid:90630038). + for (const relocated of [blocksHostPath, txindexHostPath]) { + if (relocated && fs.existsSync(relocated)) { + console.log(`Clearing relocated node data at ${relocated}...`) + await execFileAsync('docker', ['run', '--rm', '-v', `${relocated}:/data`, 'alpine', 'sh', '-c', 'find /data -mindepth 1 -delete']) + nodeDataWiped = true + } + } + } + + if (resetUtxoTracker && utxoVolumePresent) { + try { + console.log(`Clearing Docker volume ${utxoVolumeName}...`) + await execFileAsync('docker', ['run', '--rm', '-v', `${utxoVolumeName}:/data`, 'alpine', 'sh', '-c', 'find /data -mindepth 1 -delete']) + } catch (err) { + // The volume exists and the wipe failed, so the tracker still holds its old store. Falling through would drop the decoder/indexer databases around retained tracker data and still return + // true. + const reason = `clearing the Docker volume ${utxoVolumeName} failed (${failureReason(err)})` + if (nodeDataWiped) { + // Node data is already gone, so this reset is half done and cannot claim otherwise. Starting the services again would run a resynced chain under decoder and indexer stores that still + // describe the old one, so they stay down until the operator re-runs the same reset. + console.log(`Aborted: ${reason}.`) + console.log(' The node data for this stack WAS already cleared; the decoder/indexer') + console.log(' databases were NOT touched, and the stopped services are left down.') + console.log(' Fix the volume problem and re-run the same reset command.') + return false + } + return await abortBeforeAnyWipe(reason) + } + } + + const dbModulesToReset = [ + ...(resetDecoder ? [XChainService.XCHAIN_DECODER] : []), + ...(resetIndexer ? [XChainService.XCHAIN_INDEXER] : []), + ] + if (dbModulesToReset.length > 0) { + await resetDatabases(coin, network, dbModulesToReset) + } + + return cleanResetMetadata(context) +} + +async function cleanResetMetadata(context) { + const { coin, network, resetIndexer, resetNode } = context + // A wiped indexer DB restarts push_generations at 0, which the hub's price ingest fence reads as a stale replay and DROPS, killing this chain's price rail (and the native-fee / XCHAIN-USD path) + // with no error. Clear the fence row here, while the indexer is still stopped, so the first push after the restart below lands. Never fatal: the wipe already happened, so a failure must not abort + // the restart pass and leave the stack down. It is reported loudly instead, with the statement to run by hand. + if (resetIndexer) { + try { + await clearHubPriceIngestWatermark(coin, network) + } catch (err) { + console.warn('WARNING: clearing the hub price ingest fence failed: ' + (err && err.message ? err.message : err)) + console.warn(" Run on the hub DB before the indexer catches up:") + // Network-scoped, matching the statement DatabaseService prints on its own failure branches. The '' bucket is the legacy/unset scope that pre-migration rows sit in. + console.warn(" DELETE " + "FROM price_ingest_watermarks WHERE source_chain = '" + + (CoinTickerSymbol[coin] || coin) + "' AND network IN ('" + + String(network || '').trim().toLowerCase() + "', '');") + console.warn(" Keep the network clause: it is what leaves every OTHER network's fence for this chain in place.") + } + } + + // A re-genesised regtest chain leaves the hub's cross-chain rows behind: they are keyed by `network` and a BTC-anchored snapshot_block, and nothing in them names the chain INSTANCE, so the mirror + // bootstrap hands every fresh indexer the dead chain's finalized matches and it refuses them at every block for as long as it runs. Purge them here, while the indexer is stopped, so the rebuilt + // mirror never sees them. Regtest only: no other network has a re-genesis path, and there these rows are live federation history. Never fatal, like the price fence above: the wipe already + // happened, so a failure must not abort the restart pass and leave the stack down. It is reported loudly instead, with the statements to run by hand. + if (resetNode && network === Network.REGTEST) { + try { + await purgeHubCrossChainRows(coin, network) + } catch (err) { + console.warn('WARNING: purging the hub cross-chain relic rows failed: ' + + ((err && err.message) ? err.message : err)) + console.warn(' A fresh indexer will mirror the dead chain\'s matches back in. Run on the hub DB:') + for (const statement of manualHubCrossChainPurgeStatements(CoinTickerSymbol[coin] || coin)) { + console.warn(' ' + statement) + } + console.warn(` Then restart the hub: xchain-node restart ${HUB_MODULE_NAME}`) + } + } + + return recordResetReindex(context) +} + +function recordResetReindex(context) { + const { coin, network, resetDecoder, resetIndexer, resetNode, resetUtxoTracker, service } = context + // A reset is a REINDEX: from here the wiped stores rebuild on a new lineage, and every bootstrap archive already published for these combos describes the old one. Without a marker nothing forces + // a republish, so the stale-lineage archive stays newest until the next scheduled run (up to a week for a tracker, which is opt-in besides) and no age check catches it, because the file is hours + // old and simply wrong. Mark the combos DUE so the publisher pulls them into its next plan regardless of schedule or tracker opt-in. + // + // Best-effort by design: the wipes already happened, so a bookkeeping failure must never abort the restart pass and leave the stack down. It is reported loudly instead, with the command to + // publish by hand. + const reindexedModules = reindexAffectedModules({ + node: resetNode, utxoTracker: resetUtxoTracker, decoder: resetDecoder, indexer: resetIndexer + }) + if (reindexedModules.length > 0) { + try { + const marked = recordReindex(reindexedModules, coin, network, { reason: `reset ${service}` }) + if (marked.length > 0) { + console.log(`Marked ${marked.length} bootstrap combo(s) for republish after this reindex: ${marked.join(', ')}`) + } else { + throw new Error('the republish ledger could not be written') + } + } catch (err) { + console.warn('WARNING: could not record this reindex in the bootstrap republish ledger: ' + + ((err && err.message) ? err.message : err)) + console.warn(' The published archives for these combos are now from the PRE-reset lineage and') + console.warn(' nothing will force a republish. Republish by hand once the stack has caught up:') + for (const module of reindexedModules) { + console.warn(` xchain-node bootstrap create ${module} ${coin} ${network}`) + } + } + } + return restartResetModules(context) +} + + +module.exports = { configure, resetModules } diff --git a/src/operations/module_operations/shared_services.js b/src/operations/module_operations/shared_services.js new file mode 100644 index 0000000..c40b236 --- /dev/null +++ b/src/operations/module_operations/shared_services.js @@ -0,0 +1,119 @@ +'use strict' + +let bootstrapService, buildDatabaseModule, config, createDockerNetwork, explorerService, getDockerNetwork, hubService, installModule, withInstallTarget + +function configure(dependencies) { + ({ bootstrapService, buildDatabaseModule, config, createDockerNetwork, explorerService, getDockerNetwork, hubService, installModule, withInstallTarget } = dependencies) +} + +/** + * Install every requested module and REPORT what was actually built. + * + * installModule returns false for a module it decided not to touch (already + * installed, or a singleton container that a previous coin/network pass in this + * same run already created). Dropping that return on the floor makes + * "built six containers" and "built nothing" print the same and exit the + * same. Unlike `update`, a no-op install is NOT a failure - the desired state + * already holds, and `install` is run idempotently by scripts and harnesses - + * so the report is printed rather than turned into a non-zero exit. + * + * @returns {Promise<{installed: Array, skipped: Array}>} + */ +async function installModules(servicesList, ref = null) { + return withInstallTarget(ref, async (target) => { + // A release install passes no branch: resolveComponentRef inside + // installModule supplies the pinned ref per component. A branch install + // passes the branch, exactly as before. + const branch = target.kind === 'release' ? null : target.ref + const outcome = { installed: [], skipped: [] } + // Per-run, so a second install in the same process reports its own + // restores rather than replaying the first one's. + bootstrapService.resetBootstrapOutcomes() + + try { + for (const nextCoin in servicesList) { + for (const nextNetwork in servicesList[nextCoin]) { + if (nextCoin && nextNetwork) { + await createDockerNetwork(getDockerNetwork(nextCoin, nextNetwork)) + await buildDatabaseModule(nextCoin, nextNetwork) + } + for (const nextModule of servicesList[nextCoin][nextNetwork]) { + const result = await installModule(nextModule, nextCoin, nextNetwork, false, null, false, branch) + if (result === false) { + outcome.skipped.push({ module: nextModule, coin: nextCoin, network: nextNetwork, reason: 'already-installed' }) + } else { + outcome.installed.push({ module: nextModule, coin: nextCoin, network: nextNetwork }) + } + } + } + } + + if (outcome.skipped.length > 0) { + console.log('install: nothing to do for ' + outcome.skipped + .map(s => `${s.module} (${s.coin} ${s.network})`).join(', ') + + ' - already installed. Use `update` to rebuild.') + } + } finally { + // In a finally because a run that throws is the one whose summary + // matters most: it leaves some services restored and some facing + // hours of resync, and the error alone does not say which. + bootstrapService.reportBootstrapOutcomes() + } + + // The explorer is installed in the shared bucket, which runs BEFORE the + // coin stacks, and it learns its coins by polling the hub. So a run that + // installed a coin leaves it serving 503 for up to a poll interval after + // this loop ends. Returning there hands every caller a stack that reports + // installed and answers nothing; the first one to be bitten was the e2e + // gate, whose suite starts the moment install returns. + return outcome + }) +} + +// Make the coins this run installed usable before the command returns. +// +// updateHub and updateExplorer push coin config to the hub and JOIN the hub and +// explorer containers to each coin's docker network. They run in preCheck, which +// fires BEFORE the action, so an install that creates brand-new coin stacks ends +// without either shared service having heard about them: the explorer sits on no +// network from which the hub is reachable, never populates a DB pool, and answers +// 503 until some later command's preCheck happens to fix it. Measured on a clean +// host, it stayed degraded through a full 150-second readiness wait. +// +// This is a COMMAND-level step, not part of the install primitive: it reconciles +// against live docker, and installModules is also driven directly by suites whose +// container registry is fixture data that such a reconcile would purge. +// +// Returns whether the stack is usable. The modules are installed either way, but +// reporting success for a stack whose explorer serves 503 makes every later +// failure land on the caller's first read instead of here. +async function syncSharedServicesAfterInstall(outcome) { + if (!outcome || !outcome.installed.some(i => i.coin && i.network)) return true + + const { updateHub } = hubService + const { updateExplorer, waitForExplorerReady } = explorerService + + try { await updateHub() } catch (err) { console.warn('install: could not push config to the hub: ' + err) } + try { await updateExplorer() } catch (err) { console.warn('install: could not attach the explorer to the new coin networks: ' + err) } + + if (await waitForExplorerReady()) return true + + console.warn('install: the xchain-explorer is still not serving coin data.' + + ' The stack is installed; the explorer either cannot reach the hub or the hub' + + ' has no config for these coins yet. Check it before running anything that reads it.') + + // Escape hatch for the install-then-fix flows: the modules ARE installed, so a + // caller that intends to repair the explorer by hand can still treat this as success. + if (allowDegradedExplorer()) { + console.warn('install: continuing anyway (XCHAIN_NODE_ALLOW_DEGRADED_EXPLORER is set).') + return true + } + return false +} + +// Opt-out for callers that knowingly accept a stack whose explorer serves no coins. +function allowDegradedExplorer() { + return ['1', 'true', 'yes'].includes(String(config.XCHAIN_NODE_ALLOW_DEGRADED_EXPLORER).toLowerCase()) +} + +module.exports = { configure, installModules, syncSharedServicesAfterInstall } diff --git a/src/operations/module_operations/uninstall_modules.js b/src/operations/module_operations/uninstall_modules.js new file mode 100644 index 0000000..eb2b39c --- /dev/null +++ b/src/operations/module_operations/uninstall_modules.js @@ -0,0 +1,105 @@ +'use strict' + +let DB_MODULE_NAME, EXPLORER_MODULE_NAME, HUB_MODULE_NAME, SYNC_MODULE_NAME, db, uninstallModule + +function configure(dependencies) { + ({ DB_MODULE_NAME, EXPLORER_MODULE_NAME, HUB_MODULE_NAME, SYNC_MODULE_NAME, db, uninstallModule } = dependencies) +} + +/** + * Uninstall every requested module, then FAIL if any of them failed. + * + * Visiting the rest of the list after one module fails is deliberate and stays: + * an operator tearing down a stack wants the other containers gone. What was + * wrong is that the per-module `catch` swallowed the error and the function + * returned true regardless, so `uninstall all` reported a clean teardown while + * leaving containers running - the exact "did nothing, said success" shape the + * `update` no-op fix removed elsewhere. + * + * Shared services (database, hub, explorer, sync) are installed ONCE and serve + * every coin/network on the box, so they are ordered LAST and only removed when + * nothing is left to serve. `--include-shared` is a request, not an override: with + * bitcoin still installed, `uninstall all dogecoin mainnet --include-shared` must + * not take the explorer down for bitcoin too. The shared pass runs after the + * per-coin pass (so a genuine full teardown still reaches them, the remaining set + * being empty by then) and skips with a reason naming what is still installed. + * + * @returns {Promise<{uninstalled: Array, skipped: Array}>} on full success + * @throws {Error} listing every module that failed, after all were attempted + */ +function createUninstallOne(outcome, failures) { + return async (nextModule, nextCoin, nextNetwork) => { + const moduleContainerId = await db.getModuleContainer(nextModule, nextCoin, nextNetwork) + if (!moduleContainerId) { + outcome.skipped.push({ module: nextModule, coin: nextCoin, network: nextNetwork, reason: 'not-installed' }) + return + } + try { + await uninstallModule(nextCoin, nextNetwork, nextModule) + outcome.uninstalled.push({ module: nextModule, coin: nextCoin, network: nextNetwork }) + } catch (err) { + const why = (err && err.message) ? err.message : String(err) + console.error(`uninstall: ${nextModule} (${nextCoin} ${nextNetwork}) FAILED: ${why}`) + failures.push({ module: nextModule, coin: nextCoin, network: nextNetwork, reason: why }) + } + } +} + +async function uninstallModules(servicesList, includeShared = false) { + const sharedModules = [DB_MODULE_NAME, HUB_MODULE_NAME, EXPLORER_MODULE_NAME, SYNC_MODULE_NAME] + const outcome = { uninstalled: [], skipped: [] } + const failures = [] + const deferredShared = [] + const uninstallOne = createUninstallOne(outcome, failures) + + for (const nextCoin in servicesList) { + for (const nextNetwork in servicesList[nextCoin]) { + for (const nextModule of servicesList[nextCoin][nextNetwork]) { + if (sharedModules.includes(nextModule)) { + if (!includeShared) { + outcome.skipped.push({ module: nextModule, coin: nextCoin, network: nextNetwork, reason: 'shared' }) + } else { + deferredShared.push({ module: nextModule, coin: nextCoin, network: nextNetwork }) + } + continue + } + await uninstallOne(nextModule, nextCoin, nextNetwork) + } + } + } + + // Shared pass. `remaining` is read AFTER the per-coin pass above, so a full teardown finds it empty and still removes them. A coin/network module is any registry row carrying a coin; shared services are registered under ''/''. + if (deferredShared.length > 0) { + let remaining = [] + try { + remaining = (await db.getAllModuleContainers(null, null)).filter(r => r.coin) + } catch (err) { + // The registry is the only thing that can answer "is anything still being served". Unreadable, we refuse rather than guess: leaving a shared service up costs an operator one more command, tearing it down under a live coin costs every other coin its explorer/hub. + const why = (err && err.message) ? err.message : String(err) + for (const s of deferredShared) + outcome.skipped.push({ ...s, reason: `shared, module registry unreadable (${why})` }) + deferredShared.length = 0 + } + const stillServed = [...new Set(remaining.map(r => `${r.coin} ${r.network}`))].sort() + for (const s of deferredShared) { + if (stillServed.length > 0) { + const reason = `shared, still serving ${stillServed.join(', ')}` + console.warn(`uninstall: keeping ${s.module}; it is ${reason}.`) + outcome.skipped.push({ ...s, reason }) + continue + } + await uninstallOne(s.module, s.coin, s.network) + } + } + + if (failures.length > 0) { + const detail = failures.map(f => `${f.module} (${f.coin} ${f.network}): ${f.reason}`).join('; ') + const err = new Error(`uninstall failed for ${failures.length} module${failures.length === 1 ? '' : 's'}: ${detail}`) + err.failures = failures + err.uninstalled = outcome.uninstalled + throw err + } + return outcome +} + +module.exports = { configure, uninstallModules } diff --git a/src/operations/module_operations/update_modules.js b/src/operations/module_operations/update_modules.js new file mode 100644 index 0000000..7dba921 --- /dev/null +++ b/src/operations/module_operations/update_modules.js @@ -0,0 +1,236 @@ +'use strict' + +let DB_MODULE_NAME, HUB_MODULE_NAME, NODE_MODULE_NAME, SEP, SYNC_MODULE_NAME, assertHubNotBehind, assertRequiredMigrationsApplied, db, getModuleBranch, installModule, installTargetService, releaseManifestService, stateModule, validatorService, versionService, withInstallTarget + +function configure(dependencies) { + ({ DB_MODULE_NAME, HUB_MODULE_NAME, NODE_MODULE_NAME, SEP, SYNC_MODULE_NAME, assertHubNotBehind, assertRequiredMigrationsApplied, db, getModuleBranch, installModule, installTargetService, releaseManifestService, stateModule, validatorService, versionService, withInstallTarget } = dependencies) +} + +/** + * Update the requested modules. + * + * The ref decides the mode, the same way it does for `install`: + * - a release ref (`vX.Y.Z`): pinned update to that train; + * - a branch name: tracking update, every module moved to that branch's tip; + * - no ref: whatever kind of node this is. A release node (the operator + * path, and the only kind a default `install` produces) moves to the + * LATEST published release, fully pinned; a branch node stays on its + * branch and takes newer commits. + * + * Until 2026-09 a no-ref update re-read each module's git branch. A pinned + * checkout is detached, so that read answered `HEAD` and every release + * node failed its own documented upgrade command. `update all` is the + * command operators are told to run, so it has to mean "take me to the + * newest release" on the node an operator has. + * + * `opts.all` marks a run that expanded from `all`: the shared services join + * it (hub first, then sync) and a coin node whose pinned binary has not + * changed is left running rather than rebuilt. + */ +async function updateModules(servicesList, ref = null, opts = {}) { + const { isReleaseRef } = releaseManifestService + const { recordInstallTarget, resolveUpdateTarget } = installTargetService + + const list = opts.all ? includeSharedServicesForUpdate(servicesList) : servicesList + const runOpts = { skipCurrentNode: !!opts.all, quietNotInstalled: !!opts.all } + + await repairValidatorConfigBeforeHubUpdate(list) + + if (isReleaseRef(ref)) { + return withInstallTarget(ref, async () => updateModulesOnBranch(list, null, runOpts)) + } + + if (ref) { + // An explicitly named branch is a decision about what this node is. + recordInstallTarget({ kind: 'branch', ref }) + return updateModulesOnBranch(list, ref, runOpts) + } + + // No ref: the update target is remembered from the last install/update, or classified from the checkouts on a node an older CLI installed. + const target = process.env.XCHAIN_NODE_UPDATE_TARGET + ? { kind: 'release', ref: process.env.XCHAIN_NODE_UPDATE_TARGET, inferred: false } + : await resolveUpdateTarget() + + if (target.kind === 'branch') { + console.log(`This node tracks branch '${target.ref}'${target.inferred ? ' (classified from its checkouts)' : ''}; updating to its newest commits. Name a release (e.g. \`update all v0.15.2\`) to move it onto a release.`) + recordInstallTarget({ kind: 'branch', ref: target.ref }) + return updateModulesOnBranch(list, target.ref, runOpts) + } + + // A release node with no ref: the LATEST release (the recorded tag is where the node is, not where it is going), never a branch fallback. A lookup failure stops the run with nothing changed. The re-executed child of a CLI self-update already knows the tag its parent resolved. + const releaseRef = process.env.XCHAIN_NODE_UPDATE_TARGET || null + return withInstallTarget(releaseRef, async () => updateModulesOnBranch(list, null, runOpts), { fallbackToBranch: false }) +} + +/** + * On a validator, an update that rebuilds the hub first re-runs the validator + * repair path (`validator init` over an initialized node): additive only, it + * fills in what a newer version added (a recorded network, wallets, the + * publisher config) and never touches the signing key, the stake or an + * existing hub API key. The hub mounts that config, so this runs BEFORE the + * rebuild. This replaces the manual `validator init` re-run the docs asked + * for after every upgrade. + * + * A repair failure is reported and does not stop the update: the hub still + * boots on the config it has, which is what it ran on before. + */ +async function repairValidatorConfigBeforeHubUpdate(servicesList) { + const shared = (servicesList[""] && servicesList[""][""]) || [] + if (!shared.includes(HUB_MODULE_NAME)) return false + const { isInitialized, initValidator } = validatorService + let initialized = false + try { initialized = isInitialized() } catch { return false } + if (!initialized) return false + try { + console.log('This node is a validator; checking its config for anything a newer version added...') + await initValidator({}) + return true + } catch (err) { + console.warn(`Could not repair the validator config (${err && err.message ? err.message : err}); the hub is updated on its existing config.`) + return false + } +} + +/** + * `all` for an UPDATE includes the shared services, hub first. + * + * filterCommandParameters leaves the hub and sync out of `all` because the + * same expansion serves install/start/stop, where "all" has never meant the + * hub. For update it must: the hub is the first thing a release moves, and + * the docs' promise that "the hub is updated first, automatically" was only + * ever true of a hub that was missing (preCheck installs one) and never of + * a hub that was running. Rebuilt with the shared bucket first so iteration + * order IS the deploy order: hub, sync, explorer, then the coin stacks. + */ +function includeSharedServicesForUpdate(servicesList) { + const shared = (servicesList[""] && servicesList[""][""]) || [] + const ordered = [HUB_MODULE_NAME, SYNC_MODULE_NAME, ...shared.filter(m => m !== HUB_MODULE_NAME && m !== SYNC_MODULE_NAME)] + const rest = {} + for (const coin of Object.keys(servicesList)) { + if (coin !== "") rest[coin] = servicesList[coin] + } + return { "": { "": ordered }, ...rest } +} + +/** + * Record ONE installModule call in an update outcome. + * + * installModule returns false when it declined to touch the module (its + * early-return paths) and a container id / true when it built one. Counting + * every call as "updated" regardless of that return value is how a run that + * rebuilt nothing still reports a landed deploy and exits 0. + */ +function recordInstallOutcome(outcome, result, module, coin, network) { + if (result === false) { + console.warn(`update: ${module} (${coin} ${network}) was not rebuilt; nothing changed for it.`) + outcome.skipped.push({ module, coin, network, reason: 'no-op' }) + } else { + outcome.updated.push({ module, coin, network }) + } +} + +/** + * Runs the update over every requested module and REPORTS what it did. + * + * The report exists because the old `return true` made "updated three + * containers" and "matched nothing at all" indistinguishable to the caller, so + * a run that changed nothing still exited 0 and read as a landed deploy. The + * caller (cli `update`) turns an empty `updated` list into a non-zero exit. + * + * @returns {Promise<{updated: Array, skipped: Array}>} + */ +async function updateModulesOnBranch(servicesList, branch = null, { skipCurrentNode = false, quietNotInstalled = false } = {}) { + const outcome = { updated: [], skipped: [] } + try { + await updateModulesInto(outcome, servicesList, branch, { skipCurrentNode, quietNotInstalled }) + } finally { + // Under `all`, a service that is not installed is the ordinary case (a validator has a hub and nothing else), so the per-service warnings collapse into one line; the outcome still lists every one of them. + const absent = outcome.skipped.filter(s => s.reason === 'not-installed') + if (quietNotInstalled && absent.length > 0) { + console.log(`update: skipped ${absent.length} service${absent.length === 1 ? '' : 's'} not installed on this node.`) + } + } + return outcome +} + +async function updateModulesInto(outcome, servicesList, branch, { skipCurrentNode, quietNotInstalled }) { + for (const nextCoin in servicesList) { + for (const nextNetwork in servicesList[nextCoin]) { + for (const nextModule of servicesList[nextCoin][nextNetwork]) { + if (nextModule === DB_MODULE_NAME) { + // `update` cannot rebuild the database. Its container is created by buildDatabaseModule from a pinned mariadb image, not from module source, and the existing-container branch there does nothing at all - yet the DB branch of installModule answered a hard `true`, which recordInstallOutcome counts as an updated module. So `update database` exited 0 reporting a landed upgrade over an untouched container. Refuse it here, where the update contract lives, and state the remediation uninstallModule already names. + console.warn(`update: ${nextModule} (${nextCoin} ${nextNetwork}) is not rebuilt by update; the database container must be removed manually and reinstalled.`) + outcome.skipped.push({ module: nextModule, coin: nextCoin, network: nextNetwork, reason: 'not-updatable' }) + continue + } + const moduleContainerId = await db.getModuleContainer(nextModule, nextCoin, nextNetwork) + if (nextModule === NODE_MODULE_NAME) { + // The running node is deliberately left alone here. buildCryptoNode stops it gracefully (SIGTERM with a flush budget) and force-removes the stopped carcass right before its `docker run --name`, so the daemon keeps serving through the download and image build and its block index is flushed before it goes. An up-front `docker rm -f` at this point was SIGKILL: the killed daemon came back at its last flushed index (16 regtest blocks lost, 2026-09-03), and it also hid the old container from buildCryptoNode's bind-mount drift guard. Recreate even when the container was missing from the registry: the old `if (!moduleContainerId) continue` made `update node` a silent no-op (exit 0, nothing created) once the node had crashed or been removed; only `install master node` could bring it back. installModule's remoteUpdate path rebuilds it from local source. That recreate-when-missing rule is for a TARGETED `update node`. Under `all`, a coin/network with no node is simply not installed here, like any other absent service: measured on a hub-only sandbox 2026-09-08, `update all` otherwise set about installing a Bitcoin mainnet daemon and failed on its missing network. + if (!moduleContainerId && quietNotInstalled) { + outcome.skipped.push({ module: nextModule, coin: nextCoin, network: nextNetwork, reason: 'not-installed' }) + continue + } + // Under `update all` a coin node whose pinned binary has not changed is left running. Rebuilding it anyway restarts a daemon that was serving fine, and one such rebuild broke a relocated datadir's mounts and halted a tracker. A targeted `update node ` still rebuilds. + if (skipCurrentNode && moduleContainerId && await coinNodeIsCurrent(nextCoin, nextNetwork)) { + console.log(`update: ${nextModule} (${nextCoin} ${nextNetwork}) already runs the pinned daemon version; left running.`) + outcome.skipped.push({ module: nextModule, coin: nextCoin, network: nextNetwork, reason: 'current' }) + continue + } + const built = await installModule(nextModule, nextCoin, nextNetwork, true, null) + recordInstallOutcome(outcome, built, nextModule, nextCoin, nextNetwork) + } else { + if (!moduleContainerId) { + // Skipping is still right for `update all` on a partly installed stack, but skipping SILENTLY is what let a targeted `update ` print nothing, change nothing and exit 0. Say it, and record it so the caller can fail a run that updated nothing. + if (!quietNotInstalled) { + console.warn(`update: ${nextModule} (${nextCoin} ${nextNetwork}) has no registered container; nothing to update (install it first if you expected it here).`) + } + outcome.skipped.push({ module: nextModule, coin: nextCoin, network: nextNetwork, reason: 'not-installed' }) + continue + } + let moduleBranch = branch + if (!moduleBranch) { + try { moduleBranch = await getModuleBranch(nextModule) } catch { /* use default */ } + // A pinned checkout is detached and answers `HEAD`, which is not a branch anything can clone. Under a release update the manifest pin decides the ref anyway; for a component the manifest does not carry, null means the default branch, the same thing `install` would do. + if (moduleBranch === 'HEAD') moduleBranch = null + } + // remoteUpdate=true so installModule actually rebuilds the container. Without it, the `if (!containerNodeVersion || remoteUpdate)` guard short-circuits for any already-installed service and `update` becomes a silent no-op. Version-skew guard: a hub-dependent service whose new source declares `xchainRequiresHub` in its package.json is REFUSED when the installed hub is behind that version, before anything is torn down. Throws out of updateModules so the update fails closed with nothing modified for this module. Under a pinned update the guard must read the PINNED source's package.json, not the branch tip: it clones into a tmp tree to find `xchainRequiresHub`, and reading that from a different ref than the one about to be installed is how a skew guard blesses a version it never saw. + const { resolveComponentRef } = releaseManifestService + const pin = resolveComponentRef(nextModule, moduleBranch) + await assertHubNotBehind(nextModule, pin.ref) + // Migration-precondition guard: a service whose new source asserts a GATED (mode=manual) migration at startup is REFUSED when the database it will use has not applied that migration, before anything is torn down. Without it the only thing that discovers the requirement is the recreated container crash-looping - which is exactly how a routine indexer deploy took all three mainnet indexers down on 2026-08-09. Reads the same PINNED ref as the skew guard above, for the same reason: a precondition read from a different ref than the one being installed is a check that blessed a version it never saw. + await assertRequiredMigrationsApplied(nextModule, nextCoin, nextNetwork, pin.ref) + // moduleBranch MUST be threaded through: installModule re-clones the module on the remoteUpdate path (cloneGit with this `branch`), so a null branch here re-clones the default branch and clobbers the branch the operator asked for (the cause of `update ` silently deploying master). installModule does the clone, so no separate cloneGit is needed here. + const rebuilt = await installModule(nextModule, nextCoin, nextNetwork, true, moduleContainerId, false, moduleBranch) + recordInstallOutcome(outcome, rebuilt, nextModule, nextCoin, nextNetwork) + } + } + } + } + return outcome +} + +/** + * Does the running coin node already carry the daemon version an update + * would install? The container writes its version file at build time (a + * bare `28.1`); the pinned release comes back tagged (`v28.1`). Any doubt + * answers false, so the rebuild the operator could always get still happens. + */ +async function coinNodeIsCurrent(coin, network) { + try { + const { getLastStatus, getRemoteModuleVersions } = stateModule + const { checkRemoteNodeVersion } = versionService + const running = getLastStatus()?.[coin]?.[network]?.[NODE_MODULE_NAME]?.["container_version"] + if (!running) return false + if (!(NODE_MODULE_NAME + SEP + coin in getRemoteModuleVersions())) { + await checkRemoteNodeVersion(coin) + } + const pinned = getRemoteModuleVersions()[NODE_MODULE_NAME + SEP + coin]?.["tag_name"] + if (!pinned) return false + const strip = v => String(v).trim().replace(/^v/, '') + return strip(running) === strip(pinned) + } catch { + return false + } +} + +module.exports = { configure, updateModules } From bec75a3011969403888b7e28f53c4237a36dd6bc Mon Sep 17 00:00:00 2001 From: J-Dog Date: Thu, 17 Sep 2026 10:48:50 -0700 Subject: [PATCH 05/35] refactor(cli): split command modules --- bin/capture_cli_prints.js | 11 +- src/cli.js | 813 +------------------------------------- src/cli/commands.js | 234 +++++++++++ src/cli/dispatch.js | 178 +++++++++ src/cli/errors.js | 194 +++++++++ src/cli/options.js | 193 +++++++++ src/cli/output.js | 299 ++++++++++++++ src/cli/parse_command.js | 48 +++ 8 files changed, 1171 insertions(+), 799 deletions(-) create mode 100644 src/cli/commands.js create mode 100644 src/cli/dispatch.js create mode 100644 src/cli/errors.js create mode 100644 src/cli/options.js create mode 100644 src/cli/output.js create mode 100644 src/cli/parse_command.js diff --git a/bin/capture_cli_prints.js b/bin/capture_cli_prints.js index 9320f7e..9ed541c 100644 --- a/bin/capture_cli_prints.js +++ b/bin/capture_cli_prints.js @@ -13,7 +13,7 @@ * * The CLI print contract: what this tool exists to prove. * - * The four user-facing paths below print straight to the console on purpose, + * The user-facing paths below print straight to the console on purpose, * because their console IS the product a human reads. A restructure that * moves those files, renames their methods or routes anything through a logger * must leave every one of those prints saying exactly what it said before. A @@ -46,9 +46,9 @@ const { execFileSync } = require('child_process'); const REPO = path.resolve(__dirname, '..'); -// The four user-facing paths whose prints go to the console, not the logger. +// The user-facing paths whose prints go to the console, not the logger. // A path that is a directory contributes every .js file under it. -const PRINT_PATHS = ['src/ui', 'src/cli.js', 'src/operations', 'src/precheck.js']; +const PRINT_PATHS = ['src/ui', 'src/cli.js', 'src/cli', 'src/operations', 'src/precheck.js']; const CONSOLE_METHOD = /console\s*\.\s*([A-Za-z]+)\s*\(/g; @@ -180,7 +180,10 @@ function collectPrints() { * each in a child process and keep the bytes. */ function driveHelp() { - const cliSrc = fs.readFileSync(path.join(REPO, 'src/cli.js'), 'utf8'); + const cliSrc = ['src/cli.js', 'src/cli'] + .flatMap(expandPath) + .map(rel => fs.readFileSync(path.join(REPO, rel), 'utf8')) + .join('\n'); const names = [...cliSrc.matchAll(/\.command\(\s*'([a-z0-9:-]+)'/g)].map(x => x[1]); const targets = ['--help', ...[...new Set(names)].sort().map(n => `${n} --help`)]; const chunks = []; diff --git a/src/cli.js b/src/cli.js index 423a086..b8a3601 100644 --- a/src/cli.js +++ b/src/cli.js @@ -14,7 +14,6 @@ * XChain Node - CLI * Commander setup and command definitions ********************************************************************/ - const { Command } = require('commander') const { version } = require('../package.json') const { preCheck } = require('./precheck') @@ -52,37 +51,9 @@ const { stakeValidator, unstakeValidator } = require('./services/validator_stake const { restoreBootstrapInterface, startInterface } = require('./ui/menu') const { acquireCommandLock } = require('./utils/command_lock') const { noticeNewerRelease } = require('./services/self_update_service') - -// Commander's action handlers are async, but program.parse() is synchronous: -// anything an action rejects with escapes as an unhandled rejection, which Node -// prints as an ERR_UNHANDLED_REJECTION stack. Several services reject with a -// plain string (cloneGit, buildAndUp), and a string reason turns that stack into -// noise with the actual message buried in it. Register one backstop -// that prints the reason readably and exits non-zero, keeping the stack when the -// reason is a real Error so genuine bugs stay debuggable. -function installUnhandledRejectionHandler() { - process.on('unhandledRejection', (reason) => { - const detail = reason instanceof Error - ? (reason.stack || reason.message) - : String(reason) - console.error('xchain-node: command failed: ' + detail) - process.exit(1) - }) -} - -// Same backstop as above, for a synchronous throw that escapes an action -// handler instead of a rejection. Without this, node prints its own uncaught -// exception dump and the process state past that point is unknown, so this -// still exits non-zero rather than letting the CLI continue. -function installUncaughtExceptionHandler() { - process.on('uncaughtException', (err) => { - const detail = err instanceof Error - ? (err.stack || err.message) - : String(err) - console.error('xchain-node: command failed: ' + detail) - process.exit(1) - }) -} +const { runParseCommand } = require('./cli/parse_command') +const { installUnhandledRejectionHandler, installUncaughtExceptionHandler } = require('./cli/errors') +const loadModule = require // Which ref, if any, the command about to run will install its modules at. // @@ -214,769 +185,21 @@ async function maybeSelfUpdateBeforeUpdate(args, deps = {}) { } async function parseCommand() { - installUnhandledRejectionHandler() - installUncaughtExceptionHandler() - const program = new Command() - - const commandsNeedingVersions = ['install', 'update', 'reinstall'] - // Read-only commands only display state and never change which services - // are installed/running, so they don't need to push local config to the - // hub/explorer. Skipping the push keeps them fast and avoids the lengthy - // updateconfig round-trip on multi-coin nodes. Any command NOT listed here - // (install, update, start, stop, restart, uninstall, reset, sync, …) still - // pushes; the default is to sync, so a new/unknown command stays safe. - const readOnlyCommands = ['ps', 'tail', 'logs', 'monitor', 'tailmonitor', 'bootstrap-combos', 'bootstrap-republish-due'] - // Commands that mutate stack state (containers, images, DBs, config - // pushes). Two of these interleaving from concurrent shells can corrupt an - // install mid-flight, so they serialize on a pidfile lock; a second - // invocation is refused with a clear message instead of interleaving. - // `e2etest` is included: its action does its own docker build/run/rm, so it - // must stay serialized against install/update the same way the others are. - const mutatingCommands = ['install', 'update', 'recreate', 'reinstall', 'uninstall', 'reset', 'bootstrap', 'sync', 'start', 'stop', 'restart', 'rollback', 'autoheal', 'e2etest'] - // How long a non-mutating command blocks for a lock-holding mutator before - // giving up (bounded so a read-only command pauses, then errors clearly, - // rather than corrupting the stack by provisioning concurrently). Tunable. - const LOCK_WAIT_MS = parseInt(process.env.XCHAIN_NODE_LOCK_WAIT_MS || '15000', 10) || 15000 - // How long a MUTATING command blocks for a lock holder before refusing. Zero - // keeps the interactive contract below; an unattended caller sets it so a - // scheduled run waits out a deploy instead of losing its work. - const MUTATING_LOCK_WAIT_MS = parseInt(process.env.XCHAIN_NODE_MUTATING_LOCK_WAIT_MS || '0', 10) || 0 - program.hook('preAction', async (thisCommand, actionCommand) => { - setVerbose(thisCommand.opts().verbose ?? false) - if (thisCommand.opts().verbose) console.log("Checking xchain-node structure") - const commandName = actionCommand.name() - // `validator` subcommands are offline (key generation + local config - // file writes). They must NOT trigger the Docker/MariaDB precheck, so an - // operator can prepare their validator identity before any stack is up. - const parentName = actionCommand.parent && actionCommand.parent.name() - if (commandName === 'validator' || parentName === 'validator') return - // `rollback` is declared but unimplemented: its action only names the - // reset-and-restore recovery path and exits non-zero. Provisioning - // Docker/MariaDB/hub and taking the mutating lock to reach a two-line - // refusal is what made it read as a hang. Measured 2026-08-30 while - // repairing a regtest indexer: an operator reached for `rollback` - // mid-incident and waited ~10 minutes on a command that printed nothing. - // It stays listed in mutatingCommands above so that a real - // implementation, which would drop this early return, is serialized. - if (commandName === 'rollback') return - // `bootstrap-republish-due` reads one local JSON file and prints it. It - // must not provision Docker/MariaDB or take the command lock: the - // publisher asks it on EVERY run, and a read-only command that waits out - // a lock holder exits non-zero, which the publisher would read as "no - // combo is due" and silently drop the forced republish this whole - // mechanism exists to guarantee. - if (commandName === 'bootstrap-republish-due') return - - // The CLI moves itself BEFORE anything else runs for a release update: - // ahead of the lock (the re-executed child takes it), ahead of preCheck - // (the child's preCheck is the one that should run, at the new code). - // Nothing to do answers quickly and the command continues here. - if (commandName === 'update') { - try { - await maybeSelfUpdateBeforeUpdate(actionCommand.args || []) - } catch (err) { - console.error('update failed: ' + redactSecrets(err && err.message ? err.message : err)) - return process.exit(1) - } - } - - // preCheck provisions shared containers/DB/hub (buildDatabaseModule, - // ensureXchainNodeAccess, scanAndRegisterModules, installHubModule) for - // EVERY non-validator command, not just the mutating ones. Running that - // provisioning unlocked lets a concurrent `ps`/`e2etest`/`exec` tear down - // and rebuild the hub out from under a lock-holding `update` mid docker - // build. So acquire the lock around preCheck for every command. - // - // A mutating command keeps the lock through its whole action (released on - // process exit, since actions terminate via process.exit()) and refuses - // a held lock unless asked to wait. A non-mutating command holds - // the lock only across preCheck, releasing it right after, so a - // long-running `monitor`/`tail`/`logs` does not pin the lock for its - // lifetime; it waits a bounded time for a busy mutator, then errors. - const holdThroughAction = mutatingCommands.includes(commandName) - let release - try { - release = acquireCommandLock({ - command: commandName, - waitMs: holdThroughAction ? MUTATING_LOCK_WAIT_MS : LOCK_WAIT_MS - }) - } catch (err) { - console.error(err.message) - return process.exit(1) - } - if (holdThroughAction) { - // Release only on process exit (also covers throws and SIGINT/SIGTERM - // via the default handlers ending the process). - process.on('exit', release) - process.on('SIGINT', () => process.exit(130)) - process.on('SIGTERM', () => process.exit(143)) - } - try { - await preCheck( - commandsNeedingVersions.includes(commandName), - !readOnlyCommands.includes(commandName), - // The ref the action is about to install at, so the hub preCheck - // provisions is staged from it too. Read with the SAME classifier - // the action uses rather than "the first positional", because the - // args are order-independent and only resolveArgs knows which one - // is a ref (`install regtest` names a network, not a branch). A - // command that names no ref, or an arg shape resolveArgs refuses, - // yields null and the previous default-branch behaviour. - refForPreCheck(commandName, actionCommand), - // Whether this command could bring a crash-looping hub back, which - // is the only thing that lets preCheck's config push degrade to a - // warning instead of aborting the command that would fix the hub. - commandRepairsHub(commandName, actionCommand) - ) - } finally { - // Non-mutating commands hand the lock back as soon as provisioning is - // done; mutating commands keep it (released on exit) for their action. - if (!holdThroughAction) release() - } - // Anonymous usage telemetry (default-on, opt-out). Fire-and-forget: - // a failure here must never block or break the command being run. - try { - const optOut = thisCommand.opts().telemetry === false - await maybeReportTelemetry(actionCommand.name(), optOut) - } catch { /* telemetry is best-effort */ } - // One line when a newer release exists, on every command that reached - // this point. `update` is the command it recommends, so it says nothing - // there. Cached an hour, silent offline, never throws. - if (commandName !== 'update') { - await noticeNewerRelease() - } + runParseCommand({ + Command, version, preCheck, setVerbose, filterCommandParameters, resolveArgs, + HUB_MODULE_NAME, redactSecrets, installModules, syncSharedServicesAfterInstall, + updateModules, recreateModules, uninstallModules, logModules, monitorModules, + restartModules, stopModules, startModules, execModules, clearDecoderReorgHalt, + shellModule, runE2ETest, resetModules, getStatus, scanAndRegisterModules, + maybeReportTelemetry, makeBootstrap, listServedBootstrapCombos, listRepublishDue, + initValidator, getValidatorSettings, isInitialized, getCapabilityConfigHostPath, + readWallets, publicWalletInfo, getSignerMountDir, COIN_NETWORKS, WALLETS_FILE, + getRollcallStatus, capabilityDriftReport, capabilityDriftExitCode, + formatCapabilityDrift, stakeValidator, unstakeValidator, + restoreBootstrapInterface, startInterface, acquireCommandLock, + noticeNewerRelease, refForPreCheck, commandRepairsHub, + maybeSelfUpdateBeforeUpdate, loadModule }) - - program - .name('xchain-node') - .version(version, '-V, --version', 'Shows xchain-node version') - .option('-v, --verbose', 'Print precheck progress messages') - .option('-i, --interactive', 'Interactive mode') - .option('--no-bootstrap', 'Do not download bootstrap files (full parse)') - .option('--no-explorer', 'Do not install xchain-explorer') - .option('--no-telemetry', 'Disable anonymous usage telemetry (see Privacy & Telemetry docs)') - .action(async (options) => { - if (options.interactive) { - return startInterface() - } - program.help() - }) - - program - .command('install') - .description('Installs XChain services') - // ONE ref slot, order-independent, classified by shape (release-management - // spec section 11): a vX.Y.Z argument is a RELEASE and installs that - // train's exact manifest-pinned component set; anything else is a branch - // and installs a tracking (unreleased) checkout. Omitting it entirely - // resolves the latest published xchain-node release, which is why this is - // no longer a required argument. - .argument('[ref]', '(a release like v0.9.0, or a branch like master/develop; omit for the latest release)') - .argument('[service]', '(node, xchain-encoder, xchain-decoder, xchain-utxo-tracker, xchain-indexer, xchain-explorer, all)') - .argument('[chain]', '(bitcoin, litecoin, dogecoin, all)') - .argument('[network]', '(mainnet, testnet, regtest, all)') - .action(async (branch, service, chain, network) => { - // Honor the global `--no-bootstrap` flag (defined on the root program above): - // commander assigns a flag matching a global option to the global, so reading - // the install command's own opts would always see the default. Skips the - // auto-download/restore and syncs from scratch. - if (program.opts().bootstrap === false) process.env.XCHAIN_NODE_NO_BOOTSTRAP = '1' - // defaultBranch null: an absent ref must reach installModules as null - // so it resolves the latest release. Substituting 'master' here would - // make the documented default install a branch install forever. - const resolved = resolveArgs([branch, service, chain, network], { expectBranch: true, defaultBranch: null }) - const serviceList = filterCommandParameters(null, resolved.service, resolved.chain, resolved.network) - const installed = await installModules(serviceList, resolved.branch) - // A coin installed by THIS run is unknown to the hub and explorer until - // something tells them, and the thing that does runs in preCheck, ahead - // of this action. Without it the command returns a stack whose explorer - // serves 503 to everything. - // Exit non-zero when the explorer never came up serving coins, so the - // caller stops here rather than at its first read of a 503 stack. - const usable = await syncSharedServicesAfterInstall(installed) - return process.exit(usable ? 0 : 1) - }) - - program - .command('uninstall') - .description('Uninstall XChain services') - .argument('', '(node, xchain-encoder, xchain-decoder, xchain-utxo-tracker, xchain-indexer, xchain-explorer, all)') - .argument('[chain]', '(bitcoin, litecoin, dogecoin, all)') - .argument('[network]', '(mainnet, testnet, regtest, all)') - .option('--include-shared', 'Also uninstall shared services (database, xchain-hub, xchain-explorer, xchain-sync)') - .action(async (service, chain, network, options) => { - const serviceList = filterCommandParameters(null, service, chain, network) - // A module that failed to uninstall used to be printed and forgotten, - // leaving the command exiting 0 with containers still running. The - // remaining modules are still attempted (uninstallModules finishes the - // list first); only the exit status changes. - try { - await uninstallModules(serviceList, options.includeShared) - } catch (err) { - console.error('uninstall failed: ' + redactSecrets(err && err.message ? err.message : err)) - return process.exit(1) - } - return process.exit(0) - }) - - program - .command('update') - .description('Update XChain services (and the CLI itself) to the latest release, or to a named release or branch') - // Every positional is optional: `xchain-node update` alone is the - // documented upgrade and means `update all`. The ref keeps its one - // order-independent slot, classified by shape like `install`. - .argument('[service]', '(node, xchain-hub, xchain-sync, xchain-encoder, xchain-decoder, xchain-utxo-tracker, xchain-indexer, xchain-explorer, all)') - .argument('[chain]', '(bitcoin, litecoin, dogecoin, all)') - .argument('[network]', '(mainnet, testnet, regtest, all)') - .argument('[ref]', '(a release like v0.9.0 for a pinned update, or any branch name; a branch is resolved on the module\'s remote, so push it first, or point the module at a local path with XCHAIN_NODE_MODULES_URLS_OVERRIDE. Omit to move a release node to the latest release, or a branch node to its newest commits)') - .action(async (service, chain, network, branch) => { - const resolved = resolveArgs([service, chain, network, branch], { expectBranch: true, defaultBranch: null }) - const serviceList = filterCommandParameters(null, resolved.service, resolved.chain, resolved.network) - // Report a failed update as an error and exit non-zero instead of - // letting it escape as an unhandled rejection. The deploy checkout - // is left intact by cloneGit, so the message is the whole outcome: - // nothing to roll back by hand. - let outcome - try { - outcome = await updateModules(serviceList, resolved.branch, { all: resolved.service === 'all' }) - } catch (err) { - console.error('update failed: ' + redactSecrets(err && err.message ? err.message : err)) - return process.exit(1) - } - // An update that touched nothing is a FAILED deploy, not a - // successful one: the operator asked for new code to be running and - // the old code still is. Exiting 0 here is what let scripts and - // `&& echo ok` treat a no-op as a landed redeploy. - if (outcome && Array.isArray(outcome.updated) && outcome.updated.length === 0) { - const why = (outcome.skipped || []) - .map(s => `${s.module} (${s.coin} ${s.network}): ${s.reason}`) - .join('; ') - console.error('update failed: nothing was updated' + (why ? ' - ' + why : ' (no requested service matched an installed container)')) - return process.exit(1) - } - return process.exit(0) - }) - - program - .command('recreate') - .description('Recreate a service container from the current config, reusing its existing image (no rebuild, no re-clone)') - .argument('', '(xchain-encoder, xchain-decoder, xchain-utxo-tracker, xchain-indexer, xchain-explorer, all)') - .argument('[chain]', '(bitcoin, litecoin, dogecoin, all)') - .argument('[network]', '(mainnet, testnet, regtest, all)') - .action(async (service, chain, network) => { - const serviceList = filterCommandParameters(null, service, chain, network) - let outcome - try { - outcome = await recreateModules(serviceList) - } catch (err) { - console.error('recreate failed: ' + redactSecrets(err && err.message ? err.message : err)) - return process.exit(1) - } - // Same rule as `update`: a run that recreated NO container did not do - // what the operator asked, whatever it printed on the way. `recreate - // node` (unsupported) took this path and still exited 0. - if (!outcome || !Array.isArray(outcome.recreated) || outcome.recreated.length === 0) { - const why = ((outcome && outcome.skipped) || []) - .map(s => `${s.module} (${s.coin} ${s.network}): ${s.reason}`) - .join('; ') - console.error('recreate failed: nothing was recreated' - + (why ? ' - ' + why : ' (no requested service can be recreated from the config map)')) - return process.exit(1) - } - // The operator's next move is always to check the container came back, - // so print the status here instead of making them ask for it. - await getStatus(null, null, true) - return process.exit(0) - }) - - program - .command('ps') - .description('List installed XChain services and status') - .action(async () => { - await getStatus(null, null, true) - return process.exit(0) - }) - - program - .command('bootstrap-combos') - .description('List served :: combos, one per line (scriptable)') - .addHelpText('after', ` -Reads the module registry, not live containers, so a STOPPED or crash-looping -combo is still listed. scripts/publish-bootstraps.sh --all builds its plan from -this: detecting from \`docker ps\` dropped stopped combos before the source-health -gate could report them, so the cron exited 0 while a consumer archive went stale.`) - .action(async () => { - const combos = await listServedBootstrapCombos() - for (const combo of combos) console.log(combo) - return process.exit(0) - }) - - program - .command('bootstrap-republish-due') - .description('List combos whose published bootstrap predates their last reindex (scriptable)') - .option('--json', 'emit the full records (reindexedAt, publishedAt, reason) instead of bare combos') - .addHelpText('after', ` -A reset wipes a store and rebuilds it on a NEW lineage, so every bootstrap -already published for that combo describes the old one: a fresh install that -takes it restores pre-reindex state and halts. Nothing forced a republish, and -no age check catches it, because the stale-lineage archive is hours old and -simply wrong. - -\`reset\` records the combos it wiped, \`bootstrap create\` records what it -re-derived, and this lists the difference. scripts/publish-bootstraps.sh reads -it to pull due combos into its plan even when the schedule or the tracker -opt-in would have skipped them.`) - .action(async (options) => { - const due = listRepublishDue() - if (options && options.json) { - console.log(JSON.stringify(due, null, 2)) - } else { - for (const entry of due) console.log(entry.combo) - } - return process.exit(0) - }) - - program - .command('sync') - .description('Scan Docker for xchain-node containers and register any missing in the database') - .action(async () => { - const added = await scanAndRegisterModules() - console.log(added === 0 ? "Nothing to add (already in sync)" : `Registered ${added} module(s)`) - return process.exit(0) - }) - - program - .command('start') - .description('Start XChain service') - .argument('', '(node, xchain-encoder, xchain-decoder, xchain-utxo-tracker, xchain-indexer, xchain-explorer, all)') - .argument('[chain]', '(bitcoin, litecoin, dogecoin, all)') - .argument('[network]', '(mainnet, testnet, regtest, all)') - .action(async (service, chain, network) => { - const serviceList = filterCommandParameters(null, service, chain, network) - await startModules(serviceList) - return process.exit(0) - }) - - program - .command('stop') - .description('Stop XChain service') - .argument('', '(node, xchain-encoder, xchain-decoder, xchain-utxo-tracker, xchain-indexer, xchain-explorer, all)') - .argument('[chain]', '(bitcoin, litecoin, dogecoin, all)') - .argument('[network]', '(mainnet, testnet, regtest, all)') - .action(async (service, chain, network) => { - const serviceList = filterCommandParameters(null, service, chain, network) - await stopModules(serviceList) - return process.exit(0) - }) - - program - .command('restart') - .description('Restart XChain service') - .argument('', '(node, xchain-encoder, xchain-decoder, xchain-utxo-tracker, xchain-indexer, xchain-explorer, all)') - .argument('[chain]', '(bitcoin, litecoin, dogecoin, all)') - .argument('[network]', '(mainnet, testnet, regtest, all)') - .action(async (service, chain, network) => { - const serviceList = filterCommandParameters(null, service, chain, network) - await restartModules(serviceList) - return process.exit(0) - }) - - program - .command('autoheal') - .description('Restart containers stuck in the Docker "unhealthy" state (opt-in per service); one-shot, cron/timer safe') - .option('--dry-run', 'report restart candidates without acting') - .action(async (options) => { - const { runAutoheal } = require('./services/autoheal_service') - const result = await runAutoheal({ dryRun: options.dryRun ?? false }) - // Exit non-zero ONLY when a restart was attempted and failed, so a - // timer unit can alert on real remediation failures without paging - // on "nothing to do" passes. - return process.exit(result.failed.length > 0 ? 1 : 0) - }) - - program - .command('tail') - .description('Tail XChain service logs') - .argument('[service]', '(node, database, xchain-hub, xchain-encoder, xchain-decoder, xchain-utxo-tracker, xchain-indexer, xchain-explorer, all)') - .argument('[chain]', '(bitcoin, litecoin, dogecoin, all)') - .argument('[network]', '(mainnet, testnet, regtest, all)') - .action(async (service, chain, network) => { - const serviceList = filterCommandParameters(null, service, chain, network) - await logModules(serviceList) - return process.exit(0) - }) - - program - .command('logs') - .description('Display full XChain service logs') - .argument('[service]', '(node, database, xchain-hub, xchain-encoder, xchain-decoder, xchain-utxo-tracker, xchain-indexer, xchain-explorer, all)') - .argument('[chain]', '(bitcoin, litecoin, dogecoin, all)') - .argument('[network]', '(mainnet, testnet, regtest, all)') - .action(async (service, chain, network) => { - const serviceList = filterCommandParameters(null, service, chain, network) - await logModules(serviceList, false) - return process.exit(0) - }) - - program - .command('monitor') - .description('Display service logs in split spaces on the screen') - .argument('[service]', '(node, database, xchain-hub, xchain-encoder, xchain-decoder, xchain-utxo-tracker, xchain-indexer, xchain-explorer, all)') - .argument('[chain]', '(bitcoin, litecoin, dogecoin, all)') - .argument('[network]', '(mainnet, testnet, regtest, all)') - .action(async (service, chain, network) => { - const serviceList = filterCommandParameters(null, service, chain, network) - await monitorModules(serviceList, false) - return process.exit(0) - }) - - program - .command('tailmonitor') - .description('Display service logs in split spaces on the screen (follow)') - .argument('[service]', '(node, database, xchain-hub, xchain-encoder, xchain-decoder, xchain-utxo-tracker, xchain-indexer, xchain-explorer, all)') - .argument('[chain]', '(bitcoin, litecoin, dogecoin, all)') - .argument('[network]', '(mainnet, testnet, regtest, all)') - .action(async (service, chain, network) => { - const serviceList = filterCommandParameters(null, service, chain, network) - await monitorModules(serviceList) - return process.exit(0) - }) - - program - .command('exec') - .description('Execute command on XChain service container') - .argument('', '(node, database, xchain-hub, xchain-encoder, xchain-decoder, xchain-utxo-tracker, xchain-indexer, xchain-explorer)') - .argument('', '(bitcoin, litecoin, dogecoin)') - .argument('', '(mainnet, testnet, regtest)') - .argument('', 'The shell command to execute') - .action(async (service, chain, network, command) => { - const serviceList = filterCommandParameters(null, service, chain, network) - await execModules(serviceList, command) - return process.exit(0) - }) - - program - .command('clear-reorg-halt') - .description('Clear a decoder\'s durable REORG_HALT marker after verifying the database is intact; the reason is recorded in its events table') - .argument('', '(bitcoin, litecoin, dogecoin)') - .argument('', '(mainnet, testnet, regtest)') - .option('--reason ', 'Why this database is known good (recorded with the clear; required unless --dry-run)') - .option('--force', 'Clear a database that has held dispenser state; you have compared its dispensers table against a known-good replica') - .option('--dry-run', 'Run the checks and report the verdict without writing the clear') - .action(async (chain, network, options) => { - if (chain === 'all' || network === 'all') { - console.log("clear-reorg-halt takes one chain and one network; 'all' is invalid") - return process.exit(1) - } - const serviceList = filterCommandParameters(null, 'xchain-decoder', chain, network) - const ok = await clearDecoderReorgHalt(serviceList, { - reason: options.reason, force: !!options.force, dryRun: !!options.dryRun - }) - return process.exit(ok ? 0 : 1) - }) - - program - .command('shell') - .description('Shell into a XChain service container') - .argument('', '(node, database, xchain-hub, xchain-encoder, xchain-decoder, xchain-utxo-tracker, xchain-indexer, xchain-explorer)') - .argument('', '(bitcoin, litecoin, dogecoin)') - .argument('', '(mainnet, testnet, regtest)') - .action(async (service, chain, network) => { - if (service === "all" || chain === "all" || network === "all") { - console.log("The shell command can't be used for multiple containers, 'all' is invalid") - return process.exit(0) - } - const serviceList = filterCommandParameters(null, service, chain, network) - await shellModule(serviceList) - return process.exit(0) - }) - - program - .command('e2etest') - .description('Run E2E tests on a regtest network') - .argument('', '(bitcoin, litecoin, dogecoin)') - .argument('[testName]', 'optional test file name (e.g. "order", "issue"); runs only that suite') - .option('--grep ', 'only run tests matching this pattern (passed to mocha --grep)') - .option('--script ', 'run a specific e2e npm script (e.g. test:security) instead of the default suite') - // The suite is CODE, cloned like any other module, and it defaulted to - // xchain-e2e-test's default branch no matter which ref the stack under it - // was installed at. For the release ceremony's freeze gate that means - // master's suites grading a release stack: a suite added or corrected on - // the release branch never runs, and one deleted there runs anyway. - // Omitted, the previous default-branch behaviour is unchanged. - .option('--ref ', 'clone the e2e-test suite at this ref (match the ref the stack was installed at)') - .action(async (chain, testName, options) => { - const { logFile, exitCode } = await runE2ETest(chain, 'regtest', testName, options.grep, options.script, options.ref || null) - console.log("E2E tests finished with exit code " + exitCode) - console.log("Logs saved to: " + logFile) - // Propagate the suite's real exit code so CI (and run-multichain-e2e.sh) - // can gate on $? natively instead of scraping the line above. - return process.exit(exitCode) - }) - - program - .command('reset') - .description('Reset data for a specific service or all services of a coin/network') - .argument('', '(node, xchain-utxo-tracker, xchain-decoder, xchain-indexer, all)') - .argument('', '(bitcoin, litecoin, dogecoin)') - .argument('', '(mainnet, testnet, regtest)') - .option('--yes', 'Skip the destructive-reset confirmation prompt (for CI/scripted resets)') - .option('--with-indexer', 'Reset xchain-indexer alongside xchain-decoder; the pair is only coherent when both move together') - .action(async (service, chain, network, options) => { - const confirmed = await resetModules(service, chain, network, - !!(options && options.yes), !!(options && options.withIndexer)) - return process.exit(confirmed ? 0 : 1) - }) - - program - .command('rollback') - .description('NOT IMPLEMENTED - prints the reset + bootstrap restore path for recovering a service to a block_index') - .argument('', 'The index of the last known good block') - .argument('', '(xchain-decoder, xchain-utxo-tracker, xchain-indexer, all)') - .argument('', '(bitcoin, litecoin, dogecoin)') - .argument('', '(mainnet, testnet, regtest)') - .action(async (blockIndex, service, chain, network) => { - // Not yet implemented. Fail loudly instead of silently doing nothing, - // so operators don't believe a rollback occurred. This is reached - // during an incident, so it prints the runnable recovery path with - // the operator's own arguments already substituted in, and exits - // through process.exit(): setting process.exitCode alone left the - // process alive on whatever handles were open, which is how a - // command that had already printed its answer still looked hung. - console.error('`rollback` is not yet implemented; nothing was rolled back.') - console.error(`To recover ${service} (${chain} ${network}) to block ${blockIndex}, use reset followed by a bootstrap restore:`) - console.error(` xchain-node reset ${service} ${chain} ${network}`) - console.error(` xchain-node bootstrap restore ${service} ${chain} ${network}`) - console.error('Restore rewinds to the newest bootstrap at or before that block, then the service re-parses forward.') - return process.exit(1) - }) - - program - .command('bootstrap') - .description('Create / Restore XChain service bootstraps') - .argument('', '(create, restore)') - .argument('', '(xchain-decoder, xchain-utxo-tracker, xchain-indexer, xchain-hub)') - .argument('', '(bitcoin, litecoin, dogecoin)') - .argument('', '(mainnet, testnet, regtest)') - .option('--latest', 'restore the newest bootstrap without prompting (scriptable)') - .option('--file ', 'restore this exact bootstrap archive without prompting') - .addHelpText('after', ` -Notes: - bootstrap create - generates a bootstrap file - bootstrap restore - restores a bootstrap file - - restore prompts only on a TTY. With --latest, --file, or no TTY it - resolves non-interactively, so a script cannot wedge on an unanswerable - menu while holding the command lock.`) - .action(async (action, service, chain, network, options) => { - if (action === "create") { - try { - await makeBootstrap(chain, network, service) - } catch (err) { - // A source-health refusal is an expected, actionable outcome, - // not a crash: print the reasons and exit non-zero so the - // cron publisher can classify it, with no stack trace to - // read past. Anything else keeps its stack. - if (err && err.name === 'BootstrapSourceUnhealthyError') { - console.error(redactSecrets(err.message)) - return process.exit(1) - } - throw err - } - } else { - try { - await restoreBootstrapInterface(chain, network, service, { - latest: options.latest === true, - file: options.file || null, - }) - } catch (err) { - // Mirror the create path above: an integrity/provenance - // refusal (bad signature, unsigned archive, inner-checksum - // mismatch) is the supply-chain gate doing its job, so print - // the reason and exit non-zero. Leaving it uncaught printed a - // stack trace that reads as a tool crash and invites a retry - // of a restore that must never succeed. - if (err && err.name === 'BootstrapIntegrityError') { - console.error(redactSecrets(err.message)) - return process.exit(1) - } - throw err - } - } - return process.exit(0) - }) - - const validator = program - .command('validator') - .description('Validator-mode setup for the xchain-hub (key generation + config)') - - validator - .command('init') - .description('Generate a validator signing key + config so the hub runs in validator mode') - .option('--seed-nodes ', 'comma-separated peer addresses (host:port,host:port)') - .option('--p2p-addr ', 'this validator\'s public address (host:port)') - .option('--p2p-port ', 'P2P listen port (10002 testnet, 10001 mainnet; default 10001)') - .option('--network ', 'federation to join: testnet or mainnet (default: implied by --p2p-port)') - .option('--oracle-epoch-start ', 'shared oracle epoch start (unix ms); defaults to the known federation value') - .option('--capabilities ', 'enabled capabilities (default price,cross_chain,oracle_publish,attestation)') - .option('--import-stake-key', 'use your own BTC stake key: prompts for the WIF (or set XCHAIN_NODE_STAKE_WIF)') - .option('--import-doge-key', 'use your own DOGE publisher key: prompts for the WIF (or set XCHAIN_NODE_DOGE_WIF)') - .option('--no-wallets', 'skip wallet generation (you run your own signer via XCHAIN_NODE_HUB_SIGNER_DIR)') - .option('--mint-hub-api-key', 'on a re-run, generate a HUB_API_KEY if this host has none (401s every consumer that carries no key)') - .option('--force', 'overwrite existing validator config (generates a NEW signing key; wallets are kept)') - .option('--force-wallets', 'also replace existing wallets (the old addresses and any coin at them are abandoned)') - .action(async (opts) => { - try { - await initValidator(opts) - } catch (e) { - console.error('\nERROR: ' + e.message + '\n') - return process.exit(1) - } - return process.exit(0) - }) - - validator - .command('stake') - .description('Mint XCHAIN if short (testnet) and broadcast the STAKE naming this validator\'s pubkey; dry run without --broadcast') - .option('--amount ', 'amount to stake (default 25000, clears every capability floor)') - .option('--broadcast', 'actually send the transactions (default: print the plan only)') - .option('--no-wait', 'return once the STAKE is broadcast instead of waiting for it to index') - .option('--serialize', 'send one action per block (default: chained back to back into one block)') - .option('--fee-per-kb ', 'fee rate in coin per kB (default: the encoder\'s estimate)') - .option('--timeout ', 'how long to wait for the stake to index (default 120)') - .action(async (opts) => { - try { - await stakeValidator(opts) - } catch (e) { - console.error('\nERROR: ' + e.message + '\n') - return process.exit(1) - } - return process.exit(0) - }) - - validator - .command('unstake') - .description('Withdraw this validator\'s stake and leave the active set; dry run without --broadcast') - .option('--broadcast', 'actually send the transaction (default: print the plan only)') - .option('--no-wait', 'return once broadcast instead of waiting for it to index') - .option('--fee-per-kb ', 'fee rate in coin per kB (default: the encoder\'s estimate)') - .option('--timeout ', 'how long to wait for it to index (default 120)') - .action(async (opts) => { - try { - await unstakeValidator(opts) - } catch (e) { - console.error('\nERROR: ' + e.message + '\n') - return process.exit(1) - } - return process.exit(0) - }) - - validator - .command('status') - .description('Show this node\'s validator configuration (pubkey, wallets, peers, capabilities)') - .action(async () => { - const s = getValidatorSettings() - if (!s) { - console.log(isInitialized() - ? 'Validator is initialized but disabled.' - : 'No validator configured. Run: xchain-node validator init') - } else { - console.log('Validator enabled.') - console.log(' pubkey : ' + s.pubkey) - console.log(' network : ' + (s.network || '(unset; set HUB_NETWORK in .env)')) - console.log(' p2p address : ' + s.P2P_VALIDATOR_ADDR) - console.log(' seed nodes : ' + ((s.SEED_NODES || []).join(', ') || '(none)')) - console.log(' oracle epoch : ' + (s.ORACLE_EPOCH_START || '(unset, required before oracle runs)')) - console.log(' capabilities : ' + ((s.capabilities || []).join(', ') || '(none)')) - // full_node is a possession-proof tier, not an opt-in capability, and - // it ships inert on every network (reward share zero, no verifier - // set) until its activation flag day. Said here so an operator whose - // stake clears its floor does not go looking for how to earn it. - console.log(' full_node : not active on this network yet (tier turns on with a flag day; nothing to configure)') - // Print the live path: it moved into its own directory (so the hub's - // bind mount cannot break `docker cp`), and this is where an operator - // coming from an older install finds it after the migration. - console.log(' caps config : ' + (getCapabilityConfigHostPath() || '(missing; re-run validator init)')) - // Addresses only. The keys stay in the 0600 file. - const w = publicWalletInfo(readWallets()) - const coins = COIN_NETWORKS[(w && w.network) || s.network] || { stakeCoin: 'coin', dogeCoin: 'DOGE' } - if (w) { - console.log(' stake wallet : ' + w.stakeAddress + ' (' + coins.stakeCoin + ' for fees, holds the XCHAIN stake)') - console.log(' DOGE wallet : ' + w.dogeAddress + ' (' + coins.dogeCoin + ' for price rounds and anchors)') - console.log(' keys file : ' + WALLETS_FILE + ' (mode 0600; back it up)') - console.log(' DOGE signer : ' + (process.env.XCHAIN_NODE_HUB_SIGNER_DIR - ? process.env.XCHAIN_NODE_HUB_SIGNER_DIR + ' (operator-supplied, XCHAIN_NODE_HUB_SIGNER_DIR)' - : (getSignerMountDir() || '(missing; re-run validator init)'))) - } else { - console.log(' wallets : (none; re-run validator init, or run your own signer via XCHAIN_NODE_HUB_SIGNER_DIR)') - } - // ROLLCALL reporting: the DOGE runway (reusing the address read above, - // never a second fetch), whether the configured signer can PUBLISH a - // roll call rather than only sign one, and this key's BTC-side absence - // streak. Each degrades to its own "unavailable" line instead of - // crashing the whole command or printing a reassuring zero. - const rollcall = await getRollcallStatus(w, (w && w.network) || s.network) - if (rollcall.doge) { - console.log(rollcall.doge.unavailable - ? ' DOGE runway : unavailable (could not read the DOGE wallet balance' + - (rollcall.doge.error ? ': ' + rollcall.doge.error : '') + ')' - : ' DOGE runway : ' + rollcall.doge.balance + ' ' + coins.dogeCoin + ' confirmed, ~' + - rollcall.doge.rollcalls + ' roll call(s) of runway (~0.006 ' + coins.dogeCoin + - ' each: two ~0.003 ' + coins.dogeCoin + ' transactions)') - } - if (rollcall.broadcast) { - console.log(rollcall.broadcast.exportsBroadcast - ? ' roll call : this signer exports broadcast, so it can publish roll calls' - : ' roll call : NO broadcast export in ' + rollcall.broadcast.file + - ' - it can SIGN a roll call but never PUBLISH one, silently. Add broadcast(payload) ' + - 'or use the CLI-generated signer.') - } - if (rollcall.absences) { - if (rollcall.absences.unavailable) { - console.log(' roll call absences (BTC): unavailable (' + - (rollcall.absences.error || rollcall.absences.reason || 'indexer read failed') + - '); check the explorer before assuming this key is safe') - } else if (rollcall.absences.streak === 0) { - console.log(' roll call absences (BTC): none on record') - } else if (rollcall.absences.streak === 1) { - console.log(' roll call absences (BTC): 1 (warning shot; one more consecutive miss evicts this stake)') - } else { - console.log(' roll call absences (BTC): ' + rollcall.absences.streak + - (rollcall.absences.evictedNow ? ' EVICTED - dropped from every capability set' : '')) - } - } - console.log('') - console.log(' On-chain membership: xchain-node validator stake (dry run shows balances and the stake)') - console.log(' Capability drift : xchain-node validator drift (compares this against the indexer)') - } - return process.exit(0) - }) - - // The probe half of the mispointed-config-dir defect: what this host RESOLVES - // as its capability set against what the indexer ANSWERS for the same key. - // Exits 1 on drift and 2 when the comparison could not be made, so a deploy - // check can branch on it rather than parse the text. - validator - .command('drift') - .description('Compare this host\'s resolved capability set against the indexer\'s validator sets') - .option('--pubkey ', 'identity to look up when this host has none (a mispointed config dir resolves standalone)') - .option('--network ', 'network to query: testnet or mainnet (default: this host\'s recorded network)') - .option('--block ', 'settle membership at this block instead of the indexer\'s tip') - .action(async (opts) => { - let report - try { - report = await capabilityDriftReport( - { expectPubkey: opts.pubkey, network: opts.network }, - opts.block === undefined ? {} : { blockIndex: Number(opts.block) }) - } catch (e) { - console.error('\nERROR: ' + e.message + '\n') - return process.exit(2) - } - for (const line of formatCapabilityDrift(report)) console.log(line) - return process.exit(capabilityDriftExitCode(report)) - }) - - program.parse(process.argv) } // installUnhandledRejectionHandler and installUncaughtExceptionHandler are @@ -997,6 +220,6 @@ module.exports = { parseCommand, installUnhandledRejectionHandler, installUncaug // Running it directly otherwise silently does nothing, because program.parse() // lives inside parseCommand() and would never be called. if (require.main === module) { - require('dotenv').config() + loadModule('dotenv').config() parseCommand() } diff --git a/src/cli/commands.js b/src/cli/commands.js new file mode 100644 index 0000000..117bd72 --- /dev/null +++ b/src/cli/commands.js @@ -0,0 +1,234 @@ +/********************************************************************* + * + * Copyright © 2025–2026 Dankest, LLC + * Based on XChain Platform by Dankest, LLC – https://dankest.llc + * + * SPDX-License-Identifier: AGPL-3.0-or-later + * + * This file is part of XChain Platform. Licensed under the GNU Affero + * General Public License v3.0 or later; see LICENSE.md. A commercial + * license (without AGPL source-disclosure terms) is available - + * contact legal@dankest.llc. + * + ********************************************************************** + * XChain Node - CLI + * Commander setup and command definitions + ********************************************************************/ +function registerInstall(program, deps) { + const { filterCommandParameters, resolveArgs, installModules, syncSharedServicesAfterInstall } = deps + program + .command('install') + .description('Installs XChain services') + // ONE ref slot, order-independent, classified by shape (release-management + // spec section 11): a vX.Y.Z argument is a RELEASE and installs that + // train's exact manifest-pinned component set; anything else is a branch + // and installs a tracking (unreleased) checkout. Omitting it entirely + // resolves the latest published xchain-node release, which is why this is + // no longer a required argument. + .argument('[ref]', '(a release like v0.9.0, or a branch like master/develop; omit for the latest release)') + .argument('[service]', '(node, xchain-encoder, xchain-decoder, xchain-utxo-tracker, xchain-indexer, xchain-explorer, all)') + .argument('[chain]', '(bitcoin, litecoin, dogecoin, all)') + .argument('[network]', '(mainnet, testnet, regtest, all)') + .action(async (branch, service, chain, network) => { + // Honor the global `--no-bootstrap` flag (defined on the root program above): + // commander assigns a flag matching a global option to the global, so reading + // the install command's own opts would always see the default. Skips the + // auto-download/restore and syncs from scratch. + if (program.opts().bootstrap === false) process.env.XCHAIN_NODE_NO_BOOTSTRAP = '1' + // defaultBranch null: an absent ref must reach installModules as null + // so it resolves the latest release. Substituting 'master' here would + // make the documented default install a branch install forever. + const resolved = resolveArgs([branch, service, chain, network], { expectBranch: true, defaultBranch: null }) + const serviceList = filterCommandParameters(null, resolved.service, resolved.chain, resolved.network) + const installed = await installModules(serviceList, resolved.branch) + // A coin installed by THIS run is unknown to the hub and explorer until + // something tells them, and the thing that does runs in preCheck, ahead + // of this action. Without it the command returns a stack whose explorer + // serves 503 to everything. + // Exit non-zero when the explorer never came up serving coins, so the + // caller stops here rather than at its first read of a 503 stack. + const usable = await syncSharedServicesAfterInstall(installed) + return process.exit(usable ? 0 : 1) + }) + +} + +function registerUninstall(program, deps) { + const { filterCommandParameters, uninstallModules, redactSecrets } = deps + program + .command('uninstall') + .description('Uninstall XChain services') + .argument('', '(node, xchain-encoder, xchain-decoder, xchain-utxo-tracker, xchain-indexer, xchain-explorer, all)') + .argument('[chain]', '(bitcoin, litecoin, dogecoin, all)') + .argument('[network]', '(mainnet, testnet, regtest, all)') + .option('--include-shared', 'Also uninstall shared services (database, xchain-hub, xchain-explorer, xchain-sync)') + .action(async (service, chain, network, options) => { + const serviceList = filterCommandParameters(null, service, chain, network) + // A module that failed to uninstall must not be printed and forgotten, + // leaving the command exiting 0 with containers still running. The + // remaining modules are still attempted (uninstallModules finishes the + // list first); only the exit status changes. + try { + await uninstallModules(serviceList, options.includeShared) + } catch (err) { + console.error('uninstall failed: ' + redactSecrets(err && err.message ? err.message : err)) + return process.exit(1) + } + return process.exit(0) + }) + +} + +function registerUpdate(program, deps) { + const { filterCommandParameters, resolveArgs, updateModules, redactSecrets } = deps + program + .command('update') + .description('Update XChain services (and the CLI itself) to the latest release, or to a named release or branch') + // Every positional is optional: `xchain-node update` alone is the + // documented upgrade and means `update all`. The ref keeps its one + // order-independent slot, classified by shape like `install`. + .argument('[service]', '(node, xchain-hub, xchain-sync, xchain-encoder, xchain-decoder, xchain-utxo-tracker, xchain-indexer, xchain-explorer, all)') + .argument('[chain]', '(bitcoin, litecoin, dogecoin, all)') + .argument('[network]', '(mainnet, testnet, regtest, all)') + .argument('[ref]', '(a release like v0.9.0 for a pinned update, or any branch name; a branch is resolved on the module\'s remote, so push it first, or point the module at a local path with XCHAIN_NODE_MODULES_URLS_OVERRIDE. Omit to move a release node to the latest release, or a branch node to its newest commits)') + .action(async (service, chain, network, branch) => { + const resolved = resolveArgs([service, chain, network, branch], { expectBranch: true, defaultBranch: null }) + const serviceList = filterCommandParameters(null, resolved.service, resolved.chain, resolved.network) + // Report a failed update as an error and exit non-zero instead of + // letting it escape as an unhandled rejection. The deploy checkout + // is left intact by cloneGit, so the message is the whole outcome: + // nothing to roll back by hand. + let outcome + try { + outcome = await updateModules(serviceList, resolved.branch, { all: resolved.service === 'all' }) + } catch (err) { + console.error('update failed: ' + redactSecrets(err && err.message ? err.message : err)) + return process.exit(1) + } + // An update that touched nothing is a FAILED deploy, not a + // successful one: the operator asked for new code to be running and + // the old code still is. Exiting 0 here is what let scripts and + // `&& echo ok` treat a no-op as a landed redeploy. + if (outcome && Array.isArray(outcome.updated) && outcome.updated.length === 0) { + const why = (outcome.skipped || []) + .map(s => `${s.module} (${s.coin} ${s.network}): ${s.reason}`) + .join('; ') + console.error('update failed: nothing was updated' + (why ? ' - ' + why : ' (no requested service matched an installed container)')) + return process.exit(1) + } + return process.exit(0) + }) + +} + +function registerRecreate(program, deps) { + const { filterCommandParameters, recreateModules, getStatus, redactSecrets } = deps + program + .command('recreate') + .description('Recreate a service container from the current config, reusing its existing image (no rebuild, no re-clone)') + .argument('', '(xchain-encoder, xchain-decoder, xchain-utxo-tracker, xchain-indexer, xchain-explorer, all)') + .argument('[chain]', '(bitcoin, litecoin, dogecoin, all)') + .argument('[network]', '(mainnet, testnet, regtest, all)') + .action(async (service, chain, network) => { + const serviceList = filterCommandParameters(null, service, chain, network) + let outcome + try { + outcome = await recreateModules(serviceList) + } catch (err) { + console.error('recreate failed: ' + redactSecrets(err && err.message ? err.message : err)) + return process.exit(1) + } + // Same rule as `update`: a run that recreated NO container did not do + // what the operator asked, whatever it printed on the way. `recreate + // node` (unsupported) took this path and still exited 0. + if (!outcome || !Array.isArray(outcome.recreated) || outcome.recreated.length === 0) { + const why = ((outcome && outcome.skipped) || []) + .map(s => `${s.module} (${s.coin} ${s.network}): ${s.reason}`) + .join('; ') + console.error('recreate failed: nothing was recreated' + + (why ? ' - ' + why : ' (no requested service can be recreated from the config map)')) + return process.exit(1) + } + // The operator's next move is always to check the container came back, + // so print the status here instead of making them ask for it. + await getStatus(null, null, true) + return process.exit(0) + }) + +} + +function registerExec(program, deps) { + const { filterCommandParameters, execModules } = deps + program + .command('exec') + .description('Execute command on XChain service container') + .argument('', '(node, database, xchain-hub, xchain-encoder, xchain-decoder, xchain-utxo-tracker, xchain-indexer, xchain-explorer)') + .argument('', '(bitcoin, litecoin, dogecoin)') + .argument('', '(mainnet, testnet, regtest)') + .argument('', 'The shell command to execute') + .action(async (service, chain, network, command) => { + const serviceList = filterCommandParameters(null, service, chain, network) + await execModules(serviceList, command) + return process.exit(0) + }) + +} + +function registerClearReorgHalt(program, deps) { + const { filterCommandParameters, clearDecoderReorgHalt } = deps + program + .command('clear-reorg-halt') + .description('Clear a decoder\'s durable REORG_HALT marker after verifying the database is intact; the reason is recorded in its events table') + .argument('', '(bitcoin, litecoin, dogecoin)') + .argument('', '(mainnet, testnet, regtest)') + .option('--reason ', 'Why this database is known good (recorded with the clear; required unless --dry-run)') + .option('--force', 'Clear a database that has held dispenser state; you have compared its dispensers table against a known-good replica') + .option('--dry-run', 'Run the checks and report the verdict without writing the clear') + .action(async (chain, network, options) => { + if (chain === 'all' || network === 'all') { + console.log("clear-reorg-halt takes one chain and one network; 'all' is invalid") + return process.exit(1) + } + const serviceList = filterCommandParameters(null, 'xchain-decoder', chain, network) + const ok = await clearDecoderReorgHalt(serviceList, { + reason: options.reason, force: !!options.force, dryRun: !!options.dryRun + }) + return process.exit(ok ? 0 : 1) + }) + +} + +function registerShell(program, deps) { + const { filterCommandParameters, shellModule } = deps + program + .command('shell') + .description('Shell into a XChain service container') + .argument('', '(node, database, xchain-hub, xchain-encoder, xchain-decoder, xchain-utxo-tracker, xchain-indexer, xchain-explorer)') + .argument('', '(bitcoin, litecoin, dogecoin)') + .argument('', '(mainnet, testnet, regtest)') + .action(async (service, chain, network) => { + if (service === "all" || chain === "all" || network === "all") { + console.log("The shell command can't be used for multiple containers, 'all' is invalid") + return process.exit(0) + } + const serviceList = filterCommandParameters(null, service, chain, network) + await shellModule(serviceList) + return process.exit(0) + }) + +} + +function registerPrimaryCommands(program, deps) { + registerInstall(program, deps) + registerUninstall(program, deps) + registerUpdate(program, deps) + registerRecreate(program, deps) +} + +function registerToolCommands(program, deps) { + registerExec(program, deps) + registerClearReorgHalt(program, deps) + registerShell(program, deps) +} + +module.exports = { registerPrimaryCommands, registerToolCommands } diff --git a/src/cli/dispatch.js b/src/cli/dispatch.js new file mode 100644 index 0000000..a5b204a --- /dev/null +++ b/src/cli/dispatch.js @@ -0,0 +1,178 @@ +/********************************************************************* + * + * Copyright © 2025–2026 Dankest, LLC + * Based on XChain Platform by Dankest, LLC – https://dankest.llc + * + * SPDX-License-Identifier: AGPL-3.0-or-later + * + * This file is part of XChain Platform. Licensed under the GNU Affero + * General Public License v3.0 or later; see LICENSE.md. A commercial + * license (without AGPL source-disclosure terms) is available - + * contact legal@dankest.llc. + * + ********************************************************************** + * XChain Node - CLI + * Commander setup and command definitions + ********************************************************************/ + +function dispatchSettings() { + const commandsNeedingVersions = ['install', 'update', 'reinstall'] + // Read-only commands only display state and never change which services + // are installed/running, so they don't need to push local config to the + // hub/explorer. Skipping the push keeps them fast and avoids the lengthy + // updateconfig round-trip on multi-coin nodes. Any command NOT listed here + // (install, update, start, stop, restart, uninstall, reset, sync, …) still + // pushes; the default is to sync, so a new/unknown command stays safe. + const readOnlyCommands = ['ps', 'tail', 'logs', 'monitor', 'tailmonitor', 'bootstrap-combos', 'bootstrap-republish-due'] + // Commands that mutate stack state (containers, images, DBs, config + // pushes). Two of these interleaving from concurrent shells can corrupt an + // install mid-flight, so they serialize on a pidfile lock; a second + // invocation is refused with a clear message instead of interleaving. + // `e2etest` is included: its action does its own docker build/run/rm, so it + // must stay serialized against install/update the same way the others are. + const mutatingCommands = ['install', 'update', 'recreate', 'reinstall', 'uninstall', 'reset', 'bootstrap', 'sync', 'start', 'stop', 'restart', 'rollback', 'autoheal', 'e2etest'] + // How long a non-mutating command blocks for a lock-holding mutator before + // giving up (bounded so a read-only command pauses, then errors clearly, + // rather than corrupting the stack by provisioning concurrently). Tunable. + const LOCK_WAIT_MS = parseInt(process.env.XCHAIN_NODE_LOCK_WAIT_MS || '15000', 10) || 15000 + // How long a MUTATING command blocks for a lock holder before refusing. Zero + // keeps the interactive contract below; an unattended caller sets it so a + // scheduled run waits out a deploy instead of losing its work. + const MUTATING_LOCK_WAIT_MS = parseInt(process.env.XCHAIN_NODE_MUTATING_LOCK_WAIT_MS || '0', 10) || 0 + return { commandsNeedingVersions, readOnlyCommands, mutatingCommands, LOCK_WAIT_MS, MUTATING_LOCK_WAIT_MS } +} + +function skipsPreAction(actionCommand) { + const commandName = actionCommand.name() + // `validator` subcommands are offline (key generation + local config + // file writes). They must NOT trigger the Docker/MariaDB precheck, so an + // operator can prepare their validator identity before any stack is up. + const parentName = actionCommand.parent && actionCommand.parent.name() + if (commandName === 'validator' || parentName === 'validator') return true + // `rollback` is declared but unimplemented: its action only names the + // reset-and-restore recovery path and exits non-zero. Provisioning + // Docker/MariaDB/hub and taking the mutating lock to reach a two-line + // refusal is what made it read as a hang. Measured 2026-08-30 while + // repairing a regtest indexer: an operator reached for `rollback` + // mid-incident and waited ~10 minutes on a command that printed nothing. + // It stays listed in mutatingCommands above so that a real + // implementation, which would drop this early return, is serialized. + if (commandName === 'rollback') return true + // `bootstrap-republish-due` reads one local JSON file and prints it. It + // must not provision Docker/MariaDB or take the command lock: the + // publisher asks it on EVERY run, and a read-only command that waits out + // a lock holder exits non-zero, which the publisher would read as "no + // combo is due" and silently drop the forced republish this whole + // mechanism exists to guarantee. + return commandName === 'bootstrap-republish-due' +} + +// preCheck provisions shared containers/DB/hub (buildDatabaseModule, +// ensureXchainNodeAccess, scanAndRegisterModules, installHubModule) for +// EVERY non-validator command, not just the mutating ones. Running that +// provisioning unlocked lets a concurrent `ps`/`e2etest`/`exec` tear down +// and rebuild the hub out from under a lock-holding `update` mid docker +// build. So acquire the lock around preCheck for every command. +// +// A mutating command keeps the lock through its whole action (released on +// process exit, since actions terminate via process.exit()) and refuses +// a held lock unless asked to wait. A non-mutating command holds +// the lock only across preCheck, releasing it right after, so a +// long-running `monitor`/`tail`/`logs` does not pin the lock for its +// lifetime; it waits a bounded time for a busy mutator, then errors. +function commandLock(commandName, settings, acquireCommandLock) { + const holdThroughAction = settings.mutatingCommands.includes(commandName) + let release + try { + release = acquireCommandLock({ + command: commandName, + waitMs: holdThroughAction ? settings.MUTATING_LOCK_WAIT_MS : settings.LOCK_WAIT_MS + }) + } catch (err) { + console.error(err.message) + process.exit(1) + return null + } + if (holdThroughAction) { + // Release only on process exit (also covers throws and SIGINT/SIGTERM + // via the default handlers ending the process). + process.on('exit', release) + process.on('SIGINT', () => process.exit(130)) + process.on('SIGTERM', () => process.exit(143)) + } + return { holdThroughAction, release } +} + +// The CLI moves itself BEFORE anything else runs for a release update: +// ahead of the lock (the re-executed child takes it), ahead of preCheck +// (the child's preCheck is the one that should run, at the new code). +// Nothing to do answers quickly and the command continues here. +// +// The ref the action is about to install at, so the hub preCheck +// provisions is staged from it too. Read with the SAME classifier +// the action uses rather than "the first positional", because the +// args are order-independent and only resolveArgs knows which one +// is a ref (`install regtest` names a network, not a branch). A +// command that names no ref, or an arg shape resolveArgs refuses, +// yields null and the previous default-branch behaviour. +// +// Whether this command could bring a crash-looping hub back, which +// is the only thing that lets preCheck's config push degrade to a +// warning instead of aborting the command that would fix the hub. +// +// Anonymous usage telemetry (default-on, opt-out). Fire-and-forget: +// a failure here must never block or break the command being run. +// +// One line when a newer release exists, on every command that reached +// this point. `update` is the command it recommends, so it says nothing +// there. Cached an hour, silent offline, never throws. +async function beforeAction(thisCommand, actionCommand, settings, deps) { + const { + setVerbose, maybeSelfUpdateBeforeUpdate, redactSecrets, + acquireCommandLock, preCheck, refForPreCheck, commandRepairsHub, + maybeReportTelemetry, noticeNewerRelease + } = deps + setVerbose(thisCommand.opts().verbose ?? false) + if (thisCommand.opts().verbose) console.log("Checking xchain-node structure") + const commandName = actionCommand.name() + if (skipsPreAction(actionCommand)) return + + if (commandName === 'update') { + try { + await maybeSelfUpdateBeforeUpdate(actionCommand.args || []) + } catch (err) { + console.error('update failed: ' + redactSecrets(err && err.message ? err.message : err)) + return process.exit(1) + } + } + + const lock = commandLock(commandName, settings, acquireCommandLock) + if (!lock) return + try { + await preCheck( + settings.commandsNeedingVersions.includes(commandName), + !settings.readOnlyCommands.includes(commandName), + refForPreCheck(commandName, actionCommand), + commandRepairsHub(commandName, actionCommand) + ) + } finally { + // Non-mutating commands hand the lock back as soon as provisioning is + // done; mutating commands keep it (released on exit) for their action. + if (!lock.holdThroughAction) lock.release() + } + try { + const optOut = thisCommand.opts().telemetry === false + await maybeReportTelemetry(actionCommand.name(), optOut) + } catch { /* telemetry is best-effort */ } + if (commandName !== 'update') { + await noticeNewerRelease() + } +} + +function installDispatch(program, deps) { + const settings = dispatchSettings() + program.hook('preAction', (thisCommand, actionCommand) => + beforeAction(thisCommand, actionCommand, settings, deps)) +} + +module.exports = { installDispatch } diff --git a/src/cli/errors.js b/src/cli/errors.js new file mode 100644 index 0000000..d3589f4 --- /dev/null +++ b/src/cli/errors.js @@ -0,0 +1,194 @@ +/********************************************************************* + * + * Copyright © 2025–2026 Dankest, LLC + * Based on XChain Platform by Dankest, LLC – https://dankest.llc + * + * SPDX-License-Identifier: AGPL-3.0-or-later + * + * This file is part of XChain Platform. Licensed under the GNU Affero + * General Public License v3.0 or later; see LICENSE.md. A commercial + * license (without AGPL source-disclosure terms) is available - + * contact legal@dankest.llc. + * + ********************************************************************** + * XChain Node - CLI + * Commander setup and command definitions + ********************************************************************/ +// Commander's action handlers are async, but program.parse() is synchronous: +// anything an action rejects with escapes as an unhandled rejection, which Node +// prints as an ERR_UNHANDLED_REJECTION stack. Several services reject with a +// plain string (cloneGit, buildAndUp), and a string reason turns that stack into +// noise with the actual message buried in it. Register one backstop +// that prints the reason readably and exits non-zero, keeping the stack when the +// reason is a real Error so genuine bugs stay debuggable. +function installUnhandledRejectionHandler() { + process.on('unhandledRejection', (reason) => { + const detail = reason instanceof Error + ? (reason.stack || reason.message) + : String(reason) + console.error('xchain-node: command failed: ' + detail) + process.exit(1) + }) +} + +// Same backstop as above, for a synchronous throw that escapes an action +// handler instead of a rejection. Without this, node prints its own uncaught +// exception dump and the process state past that point is unknown, so this +// still exits non-zero rather than letting the CLI continue. +function installUncaughtExceptionHandler() { + process.on('uncaughtException', (err) => { + const detail = err instanceof Error + ? (err.stack || err.message) + : String(err) + console.error('xchain-node: command failed: ' + detail) + process.exit(1) + }) +} + + +function registerE2eTest(program, deps) { + const { runE2ETest } = deps + program + .command('e2etest') + .description('Run E2E tests on a regtest network') + .argument('', '(bitcoin, litecoin, dogecoin)') + .argument('[testName]', 'optional test file name (e.g. "order", "issue"); runs only that suite') + .option('--grep ', 'only run tests matching this pattern (passed to mocha --grep)') + .option('--script ', 'run a specific e2e npm script (e.g. test:security) instead of the default suite') + // The suite is CODE, cloned like any other module, and it defaulted to + // xchain-e2e-test's default branch no matter which ref the stack under it + // was installed at. For the release ceremony's freeze gate that means + // master's suites grading a release stack: a suite added or corrected on + // the release branch never runs, and one deleted there runs anyway. + // Omitted, the previous default-branch behaviour is unchanged. + .option('--ref ', 'clone the e2e-test suite at this ref (match the ref the stack was installed at)') + .action(async (chain, testName, options) => { + const { logFile, exitCode } = await runE2ETest(chain, 'regtest', testName, options.grep, options.script, options.ref || null) + console.log("E2E tests finished with exit code " + exitCode) + console.log("Logs saved to: " + logFile) + // Propagate the suite's real exit code so CI (and run-multichain-e2e.sh) + // can gate on $? natively instead of scraping the line above. + return process.exit(exitCode) + }) + +} + +function registerReset(program, deps) { + const { resetModules } = deps + program + .command('reset') + .description('Reset data for a specific service or all services of a coin/network') + .argument('', '(node, xchain-utxo-tracker, xchain-decoder, xchain-indexer, all)') + .argument('', '(bitcoin, litecoin, dogecoin)') + .argument('', '(mainnet, testnet, regtest)') + .option('--yes', 'Skip the destructive-reset confirmation prompt (for CI/scripted resets)') + .option('--with-indexer', 'Reset xchain-indexer alongside xchain-decoder; the pair is only coherent when both move together') + .action(async (service, chain, network, options) => { + const confirmed = await resetModules(service, chain, network, + !!(options && options.yes), !!(options && options.withIndexer)) + return process.exit(confirmed ? 0 : 1) + }) + +} + +function registerRollback(program) { + program + .command('rollback') + .description('NOT IMPLEMENTED - prints the reset + bootstrap restore path for recovering a service to a block_index') + .argument('', 'The index of the last known good block') + .argument('', '(xchain-decoder, xchain-utxo-tracker, xchain-indexer, all)') + .argument('', '(bitcoin, litecoin, dogecoin)') + .argument('', '(mainnet, testnet, regtest)') + .action(async (blockIndex, service, chain, network) => { + // Not yet implemented. Fail loudly instead of silently doing nothing, + // so operators don't believe a rollback occurred. This is reached + // during an incident, so it prints the runnable recovery path with + // the operator's own arguments already substituted in, and exits + // through process.exit(): setting process.exitCode alone left the + // process alive on whatever handles were open, which is how a + // command that had already printed its answer still looked hung. + console.error('`rollback` is not yet implemented; nothing was rolled back.') + console.error(`To recover ${service} (${chain} ${network}) to block ${blockIndex}, use reset followed by a bootstrap restore:`) + console.error(` xchain-node reset ${service} ${chain} ${network}`) + console.error(` xchain-node bootstrap restore ${service} ${chain} ${network}`) + console.error('Restore rewinds to the newest bootstrap at or before that block, then the service re-parses forward.') + return process.exit(1) + }) + +} + +function registerBootstrap(program, deps) { + const { makeBootstrap, restoreBootstrapInterface, redactSecrets } = deps + program + .command('bootstrap') + .description('Create / Restore XChain service bootstraps') + .argument('', '(create, restore)') + .argument('', '(xchain-decoder, xchain-utxo-tracker, xchain-indexer, xchain-hub)') + .argument('', '(bitcoin, litecoin, dogecoin)') + .argument('', '(mainnet, testnet, regtest)') + .option('--latest', 'restore the newest bootstrap without prompting (scriptable)') + .option('--file ', 'restore this exact bootstrap archive without prompting') + .addHelpText('after', ` +Notes: + bootstrap create - generates a bootstrap file + bootstrap restore - restores a bootstrap file + + restore prompts only on a TTY. With --latest, --file, or no TTY it + resolves non-interactively, so a script cannot wedge on an unanswerable + menu while holding the command lock.`) + .action(async (action, service, chain, network, options) => { + if (action === "create") { + try { + await makeBootstrap(chain, network, service) + } catch (err) { + // A source-health refusal is an expected, actionable outcome, + // not a crash: print the reasons and exit non-zero so the + // cron publisher can classify it, with no stack trace to + // read past. Anything else keeps its stack. + if (err && err.name === 'BootstrapSourceUnhealthyError') { + console.error(redactSecrets(err.message)) + return process.exit(1) + } + throw err + } + } else { + try { + await restoreBootstrapInterface(chain, network, service, { + latest: options.latest === true, + file: options.file || null, + }) + } catch (err) { + // Mirror the create path above: an integrity/provenance + // refusal (bad signature, unsigned archive, inner-checksum + // mismatch) is the supply-chain gate doing its job, so print + // the reason and exit non-zero. Leaving it uncaught printed a + // stack trace that reads as a tool crash and invites a retry + // of a restore that must never succeed. + if (err && err.name === 'BootstrapIntegrityError') { + console.error(redactSecrets(err.message)) + return process.exit(1) + } + throw err + } + } + return process.exit(0) + }) + +} + +function registerPreRecoveryCommands(program, deps) { + registerE2eTest(program, deps) + registerReset(program, deps) +} + +function registerRecoveryCommands(program, deps) { + registerRollback(program) + registerBootstrap(program, deps) +} + +module.exports = { + installUnhandledRejectionHandler, + installUncaughtExceptionHandler, + registerPreRecoveryCommands, + registerRecoveryCommands +} diff --git a/src/cli/options.js b/src/cli/options.js new file mode 100644 index 0000000..ccce9d2 --- /dev/null +++ b/src/cli/options.js @@ -0,0 +1,193 @@ +/********************************************************************* + * + * Copyright © 2025–2026 Dankest, LLC + * Based on XChain Platform by Dankest, LLC – https://dankest.llc + * + * SPDX-License-Identifier: AGPL-3.0-or-later + * + * This file is part of XChain Platform. Licensed under the GNU Affero + * General Public License v3.0 or later; see LICENSE.md. A commercial + * license (without AGPL source-disclosure terms) is available - + * contact legal@dankest.llc. + * + ********************************************************************** + * XChain Node - CLI + * Commander setup and command definitions + ********************************************************************/ +function configureRootOptions(program, deps) { + const { version, startInterface } = deps + program + .name('xchain-node') + .version(version, '-V, --version', 'Shows xchain-node version') + .option('-v, --verbose', 'Print precheck progress messages') + .option('-i, --interactive', 'Interactive mode') + .option('--no-bootstrap', 'Do not download bootstrap files (full parse)') + .option('--no-explorer', 'Do not install xchain-explorer') + .option('--no-telemetry', 'Disable anonymous usage telemetry (see Privacy & Telemetry docs)') + .action(async (options) => { + if (options.interactive) { + return startInterface() + } + program.help() + }) + +} + +function registerSync(program, deps) { + const { scanAndRegisterModules } = deps + program + .command('sync') + .description('Scan Docker for xchain-node containers and register any missing in the database') + .action(async () => { + const added = await scanAndRegisterModules() + console.log(added === 0 ? "Nothing to add (already in sync)" : `Registered ${added} module(s)`) + return process.exit(0) + }) + +} + +function registerStart(program, deps) { + const { filterCommandParameters, startModules } = deps + program + .command('start') + .description('Start XChain service') + .argument('', '(node, xchain-encoder, xchain-decoder, xchain-utxo-tracker, xchain-indexer, xchain-explorer, all)') + .argument('[chain]', '(bitcoin, litecoin, dogecoin, all)') + .argument('[network]', '(mainnet, testnet, regtest, all)') + .action(async (service, chain, network) => { + const serviceList = filterCommandParameters(null, service, chain, network) + await startModules(serviceList) + return process.exit(0) + }) + +} + +function registerStop(program, deps) { + const { filterCommandParameters, stopModules } = deps + program + .command('stop') + .description('Stop XChain service') + .argument('', '(node, xchain-encoder, xchain-decoder, xchain-utxo-tracker, xchain-indexer, xchain-explorer, all)') + .argument('[chain]', '(bitcoin, litecoin, dogecoin, all)') + .argument('[network]', '(mainnet, testnet, regtest, all)') + .action(async (service, chain, network) => { + const serviceList = filterCommandParameters(null, service, chain, network) + await stopModules(serviceList) + return process.exit(0) + }) + +} + +function registerRestart(program, deps) { + const { filterCommandParameters, restartModules } = deps + program + .command('restart') + .description('Restart XChain service') + .argument('', '(node, xchain-encoder, xchain-decoder, xchain-utxo-tracker, xchain-indexer, xchain-explorer, all)') + .argument('[chain]', '(bitcoin, litecoin, dogecoin, all)') + .argument('[network]', '(mainnet, testnet, regtest, all)') + .action(async (service, chain, network) => { + const serviceList = filterCommandParameters(null, service, chain, network) + await restartModules(serviceList) + return process.exit(0) + }) + +} + +function registerAutoheal(program, deps) { + const { loadModule } = deps + program + .command('autoheal') + .description('Restart containers stuck in the Docker "unhealthy" state (opt-in per service); one-shot, cron/timer safe') + .option('--dry-run', 'report restart candidates without acting') + .action(async (options) => { + const { runAutoheal } = loadModule('./services/autoheal_service') + const result = await runAutoheal({ dryRun: options.dryRun ?? false }) + // Exit non-zero ONLY when a restart was attempted and failed, so a + // timer unit can alert on real remediation failures without paging + // on "nothing to do" passes. + return process.exit(result.failed.length > 0 ? 1 : 0) + }) + +} + +function registerTail(program, deps) { + const { filterCommandParameters, logModules } = deps + program + .command('tail') + .description('Tail XChain service logs') + .argument('[service]', '(node, database, xchain-hub, xchain-encoder, xchain-decoder, xchain-utxo-tracker, xchain-indexer, xchain-explorer, all)') + .argument('[chain]', '(bitcoin, litecoin, dogecoin, all)') + .argument('[network]', '(mainnet, testnet, regtest, all)') + .action(async (service, chain, network) => { + const serviceList = filterCommandParameters(null, service, chain, network) + await logModules(serviceList) + return process.exit(0) + }) + +} + +function registerLogs(program, deps) { + const { filterCommandParameters, logModules } = deps + program + .command('logs') + .description('Display full XChain service logs') + .argument('[service]', '(node, database, xchain-hub, xchain-encoder, xchain-decoder, xchain-utxo-tracker, xchain-indexer, xchain-explorer, all)') + .argument('[chain]', '(bitcoin, litecoin, dogecoin, all)') + .argument('[network]', '(mainnet, testnet, regtest, all)') + .action(async (service, chain, network) => { + const serviceList = filterCommandParameters(null, service, chain, network) + await logModules(serviceList, false) + return process.exit(0) + }) + +} + +function registerMonitor(program, deps) { + const { filterCommandParameters, monitorModules } = deps + program + .command('monitor') + .description('Display service logs in split spaces on the screen') + .argument('[service]', '(node, database, xchain-hub, xchain-encoder, xchain-decoder, xchain-utxo-tracker, xchain-indexer, xchain-explorer, all)') + .argument('[chain]', '(bitcoin, litecoin, dogecoin, all)') + .argument('[network]', '(mainnet, testnet, regtest, all)') + .action(async (service, chain, network) => { + const serviceList = filterCommandParameters(null, service, chain, network) + await monitorModules(serviceList, false) + return process.exit(0) + }) + +} + +function registerTailMonitor(program, deps) { + const { filterCommandParameters, monitorModules } = deps + program + .command('tailmonitor') + .description('Display service logs in split spaces on the screen (follow)') + .argument('[service]', '(node, database, xchain-hub, xchain-encoder, xchain-decoder, xchain-utxo-tracker, xchain-indexer, xchain-explorer, all)') + .argument('[chain]', '(bitcoin, litecoin, dogecoin, all)') + .argument('[network]', '(mainnet, testnet, regtest, all)') + .action(async (service, chain, network) => { + const serviceList = filterCommandParameters(null, service, chain, network) + await monitorModules(serviceList) + return process.exit(0) + }) + +} + +function registerLifecycleCommands(program, deps) { + registerSync(program, deps) + registerStart(program, deps) + registerStop(program, deps) + registerRestart(program, deps) + registerAutoheal(program, deps) +} + +function registerStreamCommands(program, deps) { + registerTail(program, deps) + registerLogs(program, deps) + registerMonitor(program, deps) + registerTailMonitor(program, deps) +} + +module.exports = { configureRootOptions, registerLifecycleCommands, registerStreamCommands } diff --git a/src/cli/output.js b/src/cli/output.js new file mode 100644 index 0000000..b700ee4 --- /dev/null +++ b/src/cli/output.js @@ -0,0 +1,299 @@ +/********************************************************************* + * + * Copyright © 2025–2026 Dankest, LLC + * Based on XChain Platform by Dankest, LLC – https://dankest.llc + * + * SPDX-License-Identifier: AGPL-3.0-or-later + * + * This file is part of XChain Platform. Licensed under the GNU Affero + * General Public License v3.0 or later; see LICENSE.md. A commercial + * license (without AGPL source-disclosure terms) is available - + * contact legal@dankest.llc. + * + ********************************************************************** + * XChain Node - CLI + * Commander setup and command definitions + ********************************************************************/ +function registerPs(program, deps) { + const { getStatus } = deps + program + .command('ps') + .description('List installed XChain services and status') + .action(async () => { + await getStatus(null, null, true) + return process.exit(0) + }) + +} + +function registerBootstrapCombos(program, deps) { + const { listServedBootstrapCombos } = deps + program + .command('bootstrap-combos') + .description('List served :: combos, one per line (scriptable)') + .addHelpText('after', ` +Reads the module registry, not live containers, so a STOPPED or crash-looping +combo is still listed. scripts/publish-bootstraps.sh --all builds its plan from +this: detecting from \`docker ps\` dropped stopped combos before the source-health +gate could report them, so the cron exited 0 while a consumer archive went stale.`) + .action(async () => { + const combos = await listServedBootstrapCombos() + for (const combo of combos) console.log(combo) + return process.exit(0) + }) + +} + +function registerBootstrapRepublishDue(program, deps) { + const { listRepublishDue } = deps + program + .command('bootstrap-republish-due') + .description('List combos whose published bootstrap predates their last reindex (scriptable)') + .option('--json', 'emit the full records (reindexedAt, publishedAt, reason) instead of bare combos') + .addHelpText('after', ` +A reset wipes a store and rebuilds it on a NEW lineage, so every bootstrap +already published for that combo describes the old one: a fresh install that +takes it restores pre-reindex state and halts. Nothing forced a republish, and +no age check catches it, because the stale-lineage archive is hours old and +simply wrong. + +\`reset\` records the combos it wiped, \`bootstrap create\` records what it +re-derived, and this lists the difference. scripts/publish-bootstraps.sh reads +it to pull due combos into its plan even when the schedule or the tracker +opt-in would have skipped them.`) + .action(async (options) => { + const due = listRepublishDue() + if (options && options.json) { + console.log(JSON.stringify(due, null, 2)) + } else { + for (const entry of due) console.log(entry.combo) + } + return process.exit(0) + }) + +} + +function registerStatusCommands(program, deps) { + registerPs(program, deps) + registerBootstrapCombos(program, deps) + registerBootstrapRepublishDue(program, deps) +} + +function registerValidatorRoot(program) { + const validator = program + .command('validator') + .description('Validator-mode setup for the xchain-hub (key generation + config)') + + return validator +} + +function registerValidatorInit(validator, deps) { + const { initValidator } = deps + validator + .command('init') + .description('Generate a validator signing key + config so the hub runs in validator mode') + .option('--seed-nodes ', 'comma-separated peer addresses (host:port,host:port)') + .option('--p2p-addr ', 'this validator\'s public address (host:port)') + .option('--p2p-port ', 'P2P listen port (10002 testnet, 10001 mainnet; default 10001)') + .option('--network ', 'federation to join: testnet or mainnet (default: implied by --p2p-port)') + .option('--oracle-epoch-start ', 'shared oracle epoch start (unix ms); defaults to the known federation value') + .option('--capabilities ', 'enabled capabilities (default price,cross_chain,oracle_publish,attestation)') + .option('--import-stake-key', 'use your own BTC stake key: prompts for the WIF (or set XCHAIN_NODE_STAKE_WIF)') + .option('--import-doge-key', 'use your own DOGE publisher key: prompts for the WIF (or set XCHAIN_NODE_DOGE_WIF)') + .option('--no-wallets', 'skip wallet generation (you run your own signer via XCHAIN_NODE_HUB_SIGNER_DIR)') + .option('--mint-hub-api-key', 'on a re-run, generate a HUB_API_KEY if this host has none (401s every consumer that carries no key)') + .option('--force', 'overwrite existing validator config (generates a NEW signing key; wallets are kept)') + .option('--force-wallets', 'also replace existing wallets (the old addresses and any coin at them are abandoned)') + .action(async (opts) => { + try { + await initValidator(opts) + } catch (e) { + console.error('\nERROR: ' + e.message + '\n') + return process.exit(1) + } + return process.exit(0) + }) + +} + +function registerValidatorStake(validator, deps) { + const { stakeValidator } = deps + validator + .command('stake') + .description('Mint XCHAIN if short (testnet) and broadcast the STAKE naming this validator\'s pubkey; dry run without --broadcast') + .option('--amount ', 'amount to stake (default 25000, clears every capability floor)') + .option('--broadcast', 'actually send the transactions (default: print the plan only)') + .option('--no-wait', 'return once the STAKE is broadcast instead of waiting for it to index') + .option('--serialize', 'send one action per block (default: chained back to back into one block)') + .option('--fee-per-kb ', 'fee rate in coin per kB (default: the encoder\'s estimate)') + .option('--timeout ', 'how long to wait for the stake to index (default 120)') + .action(async (opts) => { + try { + await stakeValidator(opts) + } catch (e) { + console.error('\nERROR: ' + e.message + '\n') + return process.exit(1) + } + return process.exit(0) + }) + +} + +function registerValidatorUnstake(validator, deps) { + const { unstakeValidator } = deps + validator + .command('unstake') + .description('Withdraw this validator\'s stake and leave the active set; dry run without --broadcast') + .option('--broadcast', 'actually send the transaction (default: print the plan only)') + .option('--no-wait', 'return once broadcast instead of waiting for it to index') + .option('--fee-per-kb ', 'fee rate in coin per kB (default: the encoder\'s estimate)') + .option('--timeout ', 'how long to wait for it to index (default 120)') + .action(async (opts) => { + try { + await unstakeValidator(opts) + } catch (e) { + console.error('\nERROR: ' + e.message + '\n') + return process.exit(1) + } + return process.exit(0) + }) + +} + +function printValidatorConfiguration(s, deps) { + const { + getCapabilityConfigHostPath, readWallets, publicWalletInfo, + getSignerMountDir, COIN_NETWORKS, WALLETS_FILE + } = deps + console.log('Validator enabled.') + console.log(' pubkey : ' + s.pubkey) + console.log(' network : ' + (s.network || '(unset; set HUB_NETWORK in .env)')) + console.log(' p2p address : ' + s.P2P_VALIDATOR_ADDR) + console.log(' seed nodes : ' + ((s.SEED_NODES || []).join(', ') || '(none)')) + console.log(' oracle epoch : ' + (s.ORACLE_EPOCH_START || '(unset, required before oracle runs)')) + console.log(' capabilities : ' + ((s.capabilities || []).join(', ') || '(none)')) + // full_node is a possession-proof tier, not an opt-in capability, and + // it ships inert on every network (reward share zero, no verifier + // set) until its activation flag day. Said here so an operator whose + // stake clears its floor does not go looking for how to earn it. + console.log(' full_node : not active on this network yet (tier turns on with a flag day; nothing to configure)') + // Print the live path: it moved into its own directory (so the hub's + // bind mount cannot break `docker cp`), and this is where an operator + // coming from an older install finds it after the migration. + console.log(' caps config : ' + (getCapabilityConfigHostPath() || '(missing; re-run validator init)')) + // Addresses only. The keys stay in the 0600 file. + const w = publicWalletInfo(readWallets()) + const coins = COIN_NETWORKS[(w && w.network) || s.network] || { stakeCoin: 'coin', dogeCoin: 'DOGE' } + if (w) { + console.log(' stake wallet : ' + w.stakeAddress + ' (' + coins.stakeCoin + ' for fees, holds the XCHAIN stake)') + console.log(' DOGE wallet : ' + w.dogeAddress + ' (' + coins.dogeCoin + ' for price rounds and anchors)') + console.log(' keys file : ' + WALLETS_FILE + ' (mode 0600; back it up)') + console.log(' DOGE signer : ' + (process.env.XCHAIN_NODE_HUB_SIGNER_DIR + ? process.env.XCHAIN_NODE_HUB_SIGNER_DIR + ' (operator-supplied, XCHAIN_NODE_HUB_SIGNER_DIR)' + : (getSignerMountDir() || '(missing; re-run validator init)'))) + } else { + console.log(' wallets : (none; re-run validator init, or run your own signer via XCHAIN_NODE_HUB_SIGNER_DIR)') + } + return { w, coins } +} + + // ROLLCALL reporting: the DOGE runway (reusing the address read above, + // never a second fetch), whether the configured signer can PUBLISH a + // roll call rather than only sign one, and this key's BTC-side absence + // streak. Each degrades to its own "unavailable" line instead of + // crashing the whole command or printing a reassuring zero. +function printRollcallStatus(rollcall, coins) { + if (rollcall.doge) { + console.log(rollcall.doge.unavailable + ? ' DOGE runway : unavailable (could not read the DOGE wallet balance' + + (rollcall.doge.error ? ': ' + rollcall.doge.error : '') + ')' + : ' DOGE runway : ' + rollcall.doge.balance + ' ' + coins.dogeCoin + ' confirmed, ~' + + rollcall.doge.rollcalls + ' roll call(s) of runway (~0.006 ' + coins.dogeCoin + + ' each: two ~0.003 ' + coins.dogeCoin + ' transactions)') + } + if (rollcall.broadcast) { + console.log(rollcall.broadcast.exportsBroadcast + ? ' roll call : this signer exports broadcast, so it can publish roll calls' + : ' roll call : NO broadcast export in ' + rollcall.broadcast.file + + ' - it can SIGN a roll call but never PUBLISH one, silently. Add broadcast(payload) ' + + 'or use the CLI-generated signer.') + } + if (rollcall.absences) { + if (rollcall.absences.unavailable) { + console.log(' roll call absences (BTC): unavailable (' + + (rollcall.absences.error || rollcall.absences.reason || 'indexer read failed') + + '); check the explorer before assuming this key is safe') + } else if (rollcall.absences.streak === 0) { + console.log(' roll call absences (BTC): none on record') + } else if (rollcall.absences.streak === 1) { + console.log(' roll call absences (BTC): 1 (warning shot; one more consecutive miss evicts this stake)') + } else { + console.log(' roll call absences (BTC): ' + rollcall.absences.streak + + (rollcall.absences.evictedNow ? ' EVICTED - dropped from every capability set' : '')) + } + } +} + +async function showValidatorStatus(deps) { + const { getValidatorSettings, isInitialized, getRollcallStatus } = deps + const s = getValidatorSettings() + if (!s) { + console.log(isInitialized() + ? 'Validator is initialized but disabled.' + : 'No validator configured. Run: xchain-node validator init') + } else { + const { w, coins } = printValidatorConfiguration(s, deps) + const rollcall = await getRollcallStatus(w, (w && w.network) || s.network) + printRollcallStatus(rollcall, coins) + console.log('') + console.log(' On-chain membership: xchain-node validator stake (dry run shows balances and the stake)') + console.log(' Capability drift : xchain-node validator drift (compares this against the indexer)') + } + return process.exit(0) +} + +function registerValidatorStatus(validator, deps) { + validator + .command('status') + .description('Show this node\'s validator configuration (pubkey, wallets, peers, capabilities)') + .action(() => showValidatorStatus(deps)) +} + +function registerValidatorDrift(validator, deps) { + const { capabilityDriftReport, capabilityDriftExitCode, formatCapabilityDrift } = deps + // The probe half of the mispointed-config-dir defect: what this host RESOLVES + // as its capability set against what the indexer ANSWERS for the same key. + // Exits 1 on drift and 2 when the comparison could not be made, so a deploy + // check can branch on it rather than parse the text. + validator + .command('drift') + .description('Compare this host\'s resolved capability set against the indexer\'s validator sets') + .option('--pubkey ', 'identity to look up when this host has none (a mispointed config dir resolves standalone)') + .option('--network ', 'network to query: testnet or mainnet (default: this host\'s recorded network)') + .option('--block ', 'settle membership at this block instead of the indexer\'s tip') + .action(async (opts) => { + let report + try { + report = await capabilityDriftReport( + { expectPubkey: opts.pubkey, network: opts.network }, + opts.block === undefined ? {} : { blockIndex: Number(opts.block) }) + } catch (e) { + console.error('\nERROR: ' + e.message + '\n') + return process.exit(2) + } + for (const line of formatCapabilityDrift(report)) console.log(line) + return process.exit(capabilityDriftExitCode(report)) + }) + +} + +function registerValidatorCommands(program, deps) { + const validator = registerValidatorRoot(program) + registerValidatorInit(validator, deps) + registerValidatorStake(validator, deps) + registerValidatorUnstake(validator, deps) + registerValidatorStatus(validator, deps) + registerValidatorDrift(validator, deps) +} + +module.exports = { registerStatusCommands, registerValidatorCommands } diff --git a/src/cli/parse_command.js b/src/cli/parse_command.js new file mode 100644 index 0000000..d2bbd36 --- /dev/null +++ b/src/cli/parse_command.js @@ -0,0 +1,48 @@ +/********************************************************************* + * + * Copyright © 2025–2026 Dankest, LLC + * Based on XChain Platform by Dankest, LLC – https://dankest.llc + * + * SPDX-License-Identifier: AGPL-3.0-or-later + * + * This file is part of XChain Platform. Licensed under the GNU Affero + * General Public License v3.0 or later; see LICENSE.md. A commercial + * license (without AGPL source-disclosure terms) is available - + * contact legal@dankest.llc. + * + ********************************************************************** + * XChain Node - CLI + * Commander setup and command definitions + ********************************************************************/ + +const { installDispatch } = require('./dispatch') +const { configureRootOptions, registerLifecycleCommands, registerStreamCommands } = require('./options') +const { registerPrimaryCommands, registerToolCommands } = require('./commands') +const { registerStatusCommands, registerValidatorCommands } = require('./output') +const { + installUnhandledRejectionHandler, + installUncaughtExceptionHandler, + registerPreRecoveryCommands, + registerRecoveryCommands +} = require('./errors') + +function runParseCommand(deps) { + installUnhandledRejectionHandler() + installUncaughtExceptionHandler() + const program = new deps.Command() + + installDispatch(program, deps) + configureRootOptions(program, deps) + registerPrimaryCommands(program, deps) + registerStatusCommands(program, deps) + registerLifecycleCommands(program, deps) + registerStreamCommands(program, deps) + registerToolCommands(program, deps) + registerPreRecoveryCommands(program, deps) + registerRecoveryCommands(program, deps) + registerValidatorCommands(program, deps) + + program.parse(process.argv) +} + +module.exports = { runParseCommand } From 72fb4510b3174439a49387ce7741ea47b43bf3d7 Mon Sep 17 00:00:00 2001 From: J-Dog Date: Thu, 17 Sep 2026 10:51:40 -0700 Subject: [PATCH 06/35] refactor(config): split config service responsibilities --- src/services/config_service.js | 1665 +-------------------- src/services/config_service/arguments.js | 79 + src/services/config_service/coins.js | 123 ++ src/services/config_service/database.js | 142 ++ src/services/config_service/defaults.js | 343 +++++ src/services/config_service/filters.js | 87 ++ src/services/config_service/networks.js | 387 +++++ src/services/config_service/services.js | 307 ++++ src/services/config_service/sidecars.js | 292 ++++ src/services/config_service/validation.js | 30 + 10 files changed, 1873 insertions(+), 1582 deletions(-) create mode 100644 src/services/config_service/arguments.js create mode 100644 src/services/config_service/coins.js create mode 100644 src/services/config_service/database.js create mode 100644 src/services/config_service/defaults.js create mode 100644 src/services/config_service/filters.js create mode 100644 src/services/config_service/networks.js create mode 100644 src/services/config_service/services.js create mode 100644 src/services/config_service/sidecars.js create mode 100644 src/services/config_service/validation.js diff --git a/src/services/config_service.js b/src/services/config_service.js index 9aa8897..a86d013 100644 --- a/src/services/config_service.js +++ b/src/services/config_service.js @@ -24,19 +24,28 @@ const { NODE_PREFIX, DEFAULT_NODE_PREFIX, SEP, DB_SEP, NODE_MODULE_NAME, DB_MODULE_NAME, HUB_MODULE_NAME, EXPLORER_MODULE_NAME, SYNC_MODULE_NAME, Coin, Network, XChainService, CoinTickerSymbol, REGTEST_MODULES, - moduleDir, tmpDir, cryptoNodesDir, dataDir, bootstrapDir, configDir, - EXTERNAL_DB, EXTERNAL_DB_HOST, EXTERNAL_DB_PORT + moduleDir, tmpDir, cryptoNodesDir, bootstrapDir, configDir, EXTERNAL_DB } = require('../config') const { stringToCoin } = require('../utils/helpers') const { preferredSecretEnvName, foldSecretEnvAliases, readSecretHostEnv, deprecatedSecretEnvNames } = require('../config/secret_env') const { getCoinConfigByFullName } = require('../coins') -const config = require('../config'); +const config = require('../config') // Destructured where they are used, so each call reads the export at that moment. const releaseManifestService = require('./release_manifest_service') -const stateModule = require('../state') -const peers = require('./peer_services').bindPeerServices(require) -const { getLogger } = require('../observability/logger'); -const logger = getLogger(); +const stateModule = require('../state') +const peers = require('./peer_services').bindPeerServices(require) +const { getLogger } = require('../observability/logger') +const logger = getLogger() + +const defaults = require('./config_service/defaults') +const coins = require('./config_service/coins') +const services = require('./config_service/services') +const sidecars = require('./config_service/sidecars') +const database = require('./config_service/database') +const networks = require('./config_service/networks') +const filters = require('./config_service/filters') +const argumentsService = require('./config_service/arguments') +const { validatePort } = require('./config_service/validation') function getModuleDir(module) { return moduleDir + "/" + module @@ -144,1611 +153,103 @@ function getModuleDatabaseName(module, coin, network) { + moduleName.charAt(0).toUpperCase() + moduleName.slice(1) } -function validatePort(value) { - if (typeof value === 'number') { - return Number.isInteger(value) && value >= 1 && value <= 65535 - } - if (typeof value === 'string' && /^\d+$/.test(value)) { - const port = parseInt(value, 10) - return port >= 1 && port <= 65535 - } - return false -} - -// Persist RPC credentials to the untracked -.local sidecar. A fresh sidecar -// is created with writeFileSync; an existing one is appended to, unless overwrite is set -// (used by the legacy-migration path to replace it outright). The sidecar holds the live -// node RPC user/password, so it is forced to 0600 the same way credentials.json is: without -// an explicit mode the file lands at the process umask (commonly 0644), leaving the RPC -// credentials readable by any local user on the host. -function persistSidecarCreds(localFilePath, creds, { overwrite = false } = {}) { - const body = Object.keys(creds).map(k => `${k}=${creds[k]}`).join("\n") + "\n" - if (overwrite || !fs.existsSync(localFilePath)) { - const dir = path.dirname(localFilePath) - if (!fs.existsSync(dir)) fs.mkdirSync(dir, { recursive: true }) - fs.writeFileSync(localFilePath, body, { mode: 0o600 }) - } else { - let needsLeadingNewline = false - try { - const size = fs.statSync(localFilePath).size - if (size > 0) { - const fd = fs.openSync(localFilePath, 'r') - try { - const buf = Buffer.alloc(1) - fs.readSync(fd, buf, 0, 1, size - 1) - needsLeadingNewline = buf.toString('utf8') !== '\n' - } finally { - fs.closeSync(fd) - } - } - } catch { - needsLeadingNewline = false - } - fs.appendFileSync(localFilePath, needsLeadingNewline ? '\n' + body : body) - } - // chmod unconditionally: writeFileSync's mode only applies on create, and an - // already-existing sidecar (append path, or one written before this fix) keeps its - // old permissions otherwise. - try { fs.chmodSync(localFilePath, 0o600) } catch {} -} - -// Update specific KEY=VALUE entries in a sidecar while PRESERVING all other keys. Unlike -// persistSidecarCreds({overwrite:true}) (which rewrites the file with only the keys it is -// given), this reads the existing sidecar, overlays the supplied values, and rewrites it -// 0600. Used by the DB-password rotation to set DECODER_DB_PASS/INDEXER_DB_PASS (or -// HUB_DB_PASS) without clobbering the NODE_USER/NODE_PASSWORD already in the sidecar. -function upsertSidecarValues(localFilePath, values) { - const merged = {} - if (fs.existsSync(localFilePath)) { - for (const line of fs.readFileSync(localFilePath, "utf8").split(/\r?\n/)) { - const eqIndex = line.indexOf("=") - if (eqIndex > 0) merged[line.substring(0, eqIndex)] = line.substring(eqIndex + 1) - } - } - for (const k in values) { - // Respect the naming this sidecar already uses. A file the operator - // renamed to the redaction-safe `*_SECRET` form must not sprout the legacy - // twin again on the next rotation: two names for one credential is exactly - // the ambiguity foldSecretEnvAliases() refuses to guess through, so the - // rotation would leave the stack unable to start. - const alias = preferredSecretEnvName(k) - merged[alias && alias in merged ? alias : k] = values[k] - } - persistSidecarCreds(localFilePath, merged, { overwrite: true }) -} - -// Read a single KEY=VALUE from a sidecar file, or undefined if the file or key is absent. -// Uses the same createReadStream + readline path as the main config reader above. -// -// Secret-bearing keys are also accepted under their redaction-safe `*_SECRET` name, -// which wins over the legacy name when both are present and non-empty. Keys -// with no alias (XCHAIN_NODE_BLOCKS_DIR and friends) are unaffected. -async function readSidecarValue(localFilePath, key) { - if (!fs.existsSync(localFilePath)) return undefined - const alias = preferredSecretEnvName(key) - let legacyValue = undefined - let aliasValue = undefined - const stream = fs.createReadStream(localFilePath) - const rl = readline.createInterface({ input: stream, crlfDelay: Infinity }) - for await (const line of rl) { - const eqIndex = line.indexOf("=") - if (eqIndex <= 0) continue - const lineKey = line.substring(0, eqIndex) - if (lineKey === key) legacyValue = line.substring(eqIndex + 1) - else if (alias && lineKey === alias) aliasValue = line.substring(eqIndex + 1) - } - if (aliasValue !== undefined && aliasValue !== '') return aliasValue - return legacyValue -} - -// Tell the operator, once per config load, which of their secret-bearing keys sit under a -// name automatic redaction does not match. Better to hear it from your own node -// than from a transcript that printed the value. -function warnDeprecatedSecretNames(config, filePath) { - for (const { legacy, preferred } of deprecatedSecretEnvNames(config)) { - logger.warn(`Warning: ${legacy} is a deprecated name that automatic secret redaction does not match; ` + - `rename it to ${preferred} in ${filePath} (its value prints in full whenever the file is read)`) - } -} - -// Whether a per-install random DB password can actually be APPLIED to the live MariaDB -// account on the next provision. Rotation runs either via the external-DB path (EXTERNAL_DB) -// or by exec-ing into a local MariaDB container; on a native, non-container host neither -// runs, so a generated password would never reach the DB and would desync the sidecar from a -// DB still on the old password (the 2026-06-26 indexer outage). Returns false on any error -// (e.g. docker absent), the safe direction: prefer the static default over a password we -// cannot apply. DatabaseService requires this file at load, so it is read through peers. -async function dbPasswordCanRotate() { - if (EXTERNAL_DB) return true - try { - const { getDatabaseContainerId } = peers.databaseService - return !!(await getDatabaseContainerId()) - } catch { - return false - } -} - -// HUB_DB_PASS is a SHARED-service credential: the hub and every coin/network stack that -// connects to the hub DB must present the SAME password, so it cannot be generated per -// coin/network. Resolve it once from a shared 0600 sidecar (config/hub.local), generating -// and persisting it on first use. getDefaultConfig() calls are sequential in the installer, -// so the read-or-generate is not racy in practice. -async function getOrCreateHubDbPass() { - const hubLocalPath = hubSidecarPath() - let pass = await readSidecarValue(hubLocalPath, "HUB_DB_PASS") - if (!pass) { - // Only mint a random shared password where the rotation can apply it; otherwise use - // the static default so the sidecar never diverges from a DB the rotation cannot reach. - if (await dbPasswordCanRotate()) { - pass = crypto.randomBytes(24).toString('hex') - upsertSidecarValues(hubLocalPath, { HUB_DB_PASS: pass }) - } else { - pass = "xchain" + SEP + "password" - } - } - return pass -} - -// Path of the shared hub sidecar. Not per coin/network: the hub is one service on the -// host and its credentials are shared by every stack that talks to it. -function hubSidecarPath() { - return path.resolve(configDir, "hub.local") -} - -/** - * Make sure this host has a HUB_API_KEY on disk, generating one on first use. - * - * A hub refuses to boot with no key unless keyless operation is explicitly declared, so - * `validator init` has to leave a credential behind or the documented onboarding path - * ends in a node that cannot start. Read-or-generate against the same shared 0600 - * sidecar the hub DB password uses: one host, one hub credential, and every service - * that authenticates to this hub reads it from that file. - * - * An existing key is REUSED and never rotated, including under `validator init --force` - * (which regenerates the signing key). The API key is already configured into indexers - * and explorers that write to this hub, so minting a new one behind their back would - * 401 all of them. - * - * Returns the sidecar PATH and whether it just generated, never the key itself: callers - * report where the credential lives, and no caller has a reason to print it. - * - * @returns {Promise<{path: string, generated: boolean}>} - */ -async function ensureHubApiKey() { - const sidecarPath = hubSidecarPath() - const existing = await readSidecarValue(sidecarPath, "HUB_API_KEY") - if (existing) return { path: sidecarPath, generated: false } - // 32 bytes: the strength the runbook told operators to mint by hand. - upsertSidecarValues(sidecarPath, { HUB_API_KEY: crypto.randomBytes(32).toString('hex') }) - return { path: sidecarPath, generated: true } -} - -/** - * Report whether this host already holds a HUB_API_KEY, WITHOUT ever minting one. - * - * A credential APPEARING is as breaking as one disappearing. A hub deployed with no key - * runs keyless (HUB_ALLOW_UNAUTHENTICATED), and every indexer, explorer and shared service - * pointed at it carries no key either; a key landing in this sidecar flips the hub to - * authenticated on its next deploy and 401s all of them at once, while the hub itself still - * looks healthy. So the callers that only need to SAY where the credential lives (a re-run - * of `validator init` over an already-provisioned node) read through here, and generation - * stays with the fresh-install path in ensureHubApiKey. - * - * @returns {Promise<{path: string, present: boolean}>} - */ -async function readHubApiKey() { - const sidecarPath = hubSidecarPath() - const existing = await readSidecarValue(sidecarPath, "HUB_API_KEY") - return { path: sidecarPath, present: !!existing } -} - -// Fill in HUB_API_KEY from the shared sidecar when the host env did not supply one. -// The hub, the co-located indexer and the shared services must all present the SAME -// value or their writes 401 against each other, so they resolve it from one file. -// Host env still wins, and this NEVER mints: generation belongs to `validator init`, -// so a standalone install with no validator stays keyless exactly as before. -async function applyHubApiKeyFromSidecar(target) { - if (target["HUB_API_KEY"] !== undefined && target["HUB_API_KEY"] !== "") return - const key = await readSidecarValue(hubSidecarPath(), "HUB_API_KEY") - if (key) target["HUB_API_KEY"] = key -} - -// A command composes the shared hub's config many times, so each deploy-time warning -// about it is said once per process rather than once per composition. -const warnedHubConfigKeys = new Set() -function warnHubConfigOnce(key, message) { - if (warnedHubConfigKeys.has(key)) return - warnedHubConfigKeys.add(key) - logger.warn(message) -} - -// The coin/network stacks this deployment runs, from the module registry. Returns [] -// when the registry is unreadable (no pool yet), which the callers treat as "unknown" -// rather than "none". -async function getRegisteredCoinStacks() { - try { - const { db } = stateModule - const rows = await db.getAllModuleContainers(null, null) - return (rows || []).filter(r => r && r.coin && r.network - && Object.values(Coin).includes(r.coin) && Object.values(Network).includes(r.network)) - } catch { - return [] - } -} +sidecars.configure({ + crypto, fs, path, readline, configDir, preferredSecretEnvName, + foldSecretEnvAliases, deprecatedSecretEnvNames, logger +}) +database.configure({ crypto, EXTERNAL_DB, SEP, XChainService, peers, sidecars }) +networks.configure({ + stateModule, Coin, Network, XChainService, logger, config, peers, readSecretHostEnv, + HUB_MODULE_NAME, getDockerContainerImageName +}) +defaults.configure({ + config, Network, Coin, CoinTickerSymbol, XChainService, DB_SEP, SEP, bootstrapDir, + NODE_MODULE_NAME, HUB_MODULE_NAME, EXPLORER_MODULE_NAME, SYNC_MODULE_NAME, + getDockerContainerImageName, getModuleDatabaseName +}) +coins.configure({ config, CoinTickerSymbol, Network, XChainService, getCoinConfigByFullName, getDockerContainerImageName }) +services.configure({ + config, peers, logger, readSecretHostEnv, Coin, Network, CoinTickerSymbol, XChainService, + HUB_MODULE_NAME, EXPLORER_MODULE_NAME, getDockerContainerImageName +}) +filters.configure({ + Coin, Network, XChainService, REGTEST_MODULES, + NODE_MODULE_NAME, DB_MODULE_NAME, HUB_MODULE_NAME, EXPLORER_MODULE_NAME, SYNC_MODULE_NAME +}) +argumentsService.configure({ + Coin, Network, XChainService, NODE_MODULE_NAME, DB_MODULE_NAME, + HUB_MODULE_NAME, EXPLORER_MODULE_NAME, SYNC_MODULE_NAME, + releaseManifestService, serviceAliases: filters.serviceAliases +}) -// The coin and network tokens of the command being run. A first install registers no -// coin stack until AFTER preCheck has deployed the shared hub, so the operator's own -// arguments are the only source the hub's env can be composed from on a fresh host. -function getCommandCoinsAndNetworks() { - const argv = process.argv.slice(2) - return { - coins: [...new Set(argv.filter(t => Object.values(Coin).includes(t)))], - networks: [...new Set(argv.filter(t => Object.values(Network).includes(t)))] - } -} - -// The network a standalone hub should declare: the one every stack of this deployment -// runs on. Ambiguous (several networks, or none named) leaves it unset, because one -// hub declaring the wrong network mis-gates the ingest rules it is being set for. -async function resolveDeploymentHubNetwork() { - const registered = new Set((await getRegisteredCoinStacks()).map(r => r.network)) - const networks = registered.size > 0 ? registered : new Set(getCommandCoinsAndNetworks().networks) - if (networks.size === 1) return [...networks][0] - if (networks.size > 1) { - warnHubConfigOnce("HUB_NETWORK_AMBIGUOUS", - "WARNING: HUB_NETWORK is not set and this deployment runs stacks on " + - [...networks].sort().join(", ") + ", so the shared hub cannot derive one network. " + - "Its network-keyed ingest gates (PRICE batch validation) stay closed until " + - "HUB_NETWORK is set in the host env.") - } - return null -} - -// Whether this deployment runs a BTC indexer for `network`, counting the one the -// running command is installing right now: the hub is deployed before it exists, and -// the composed URL names the container that install creates. -async function hasBitcoinIndexer(network) { - const registered = await getRegisteredCoinStacks() - if (registered.some(r => r.coin === Coin.BITCOIN && r.network === network - && r.module === XChainService.XCHAIN_INDEXER)) return true - const command = getCommandCoinsAndNetworks() - return command.coins.includes(Coin.BITCOIN) && command.networks.includes(network) -} +const { + persistSidecarCreds, upsertSidecarValues, readSidecarValue, + ensureHubApiKey, applyHubApiKeyFromSidecar, readHubApiKey +} = sidecars +const { filterCommandParameters } = filters +const { resolveArgs } = argumentsService async function getDefaultConfig(module, coin, network) { - let defaultValues = null - - if (coin && network) { - defaultValues = { - "NETWORK": network, - // The coin node's CONTAINER NAME, never the bare `node` alias every coin - // node also carries. The indexer joins its sibling coins' networks for - // cross-chain reads (ModuleService.crossChainNetworksFor) and the hub - // joins every stack, so from either container docker DNS answers `node` - // with whichever sibling network sorts first (bitcoin), and a dogecoin - // stack's RPC credentials then hit the bitcoin node: HTTP 401 by name, - // 200 by IP. Measured on regtest 2026-09-11 and reported by a testnet - // operator the same day. The container name resolves on any shared - // network and is unique per coin/network, like every other *_URL here. - "NODE_URL": getDockerContainerImageName(NODE_MODULE_NAME, coin, network), - "NODE_PORT": (network === Network.MAINNET ? 8332 : (network === Network.TESTNET ? 18332 : 18444)), - "NODE_USER": "rpc", - "NODE_PASSWORD": "rpc", - "UTXO_TRACKER_URL": getDockerContainerImageName(XChainService.XCHAIN_UTXO_TRACKER, coin, network), - "UTXO_TRACKER_API_PORT": 3001, - "UTXO_TRACKER_PORT": 3001, - "UTXO_TRACKER_BOOTSTRAP_VOLUME": bootstrapDir + "/" + coin + "/" + network + "/" + module + "/bootstrap/", - "DECODER_DB_NAME": getModuleDatabaseName(XChainService.XCHAIN_DECODER, coin, network), - "DECODER_DB_HOST": "mariadb", - "DECODER_DB_PORT": 3306, - "DECODER_DB_USER": "xchain" + DB_SEP + "decoder" + DB_SEP + coin + DB_SEP + network, - "DECODER_DB_PASS": "xchain" + SEP + "password", - "DECODER_URL": getDockerContainerImageName(XChainService.XCHAIN_DECODER, coin, network), - "DECODER_API_PORT": 3002, - "DECODER_PORT": 3002, - "DECODER_BOOTSTRAP_VOLUME": bootstrapDir + "/" + coin + "/" + network + "/" + module + "/bootstrap/", - "INDEXER_BOOTSTRAP_VOLUME": bootstrapDir + "/" + coin + "/" + network + "/" + module + "/bootstrap/", - "ENCODER_URL": getDockerContainerImageName(XChainService.XCHAIN_ENCODER, coin, network), - "ENCODER_API_PORT": 3003, - "ENCODER_PORT": 3003, - "INDEXER_URL": getDockerContainerImageName(XChainService.XCHAIN_INDEXER, coin, network), - "INDEXER_API_PORT": 3004, - "INDEXER_PORT": 3004, - "INDEXER_COIN": CoinTickerSymbol[coin], - "INDEXER_NETWORK": network, - "INDEXER_DB_HOST": "mariadb", - "INDEXER_DB_PORT": 3306, - "INDEXER_DB_NAME": getModuleDatabaseName(XChainService.XCHAIN_INDEXER, coin, network), - "INDEXER_DB_USER": "xchain" + DB_SEP + "indexer" + DB_SEP + coin + DB_SEP + network, - "INDEXER_DB_PASS": "xchain" + SEP + "password", - // The e2e-test harness reaches the indexer DB via DATABASE_URL/DATABASE_PORT - // (test/initialCheck.test.js), not INDEXER_DB_HOST/PORT. Default them here so - // the EXTERNAL_DB rewrite below can repoint them; on a host-native-DB box the - // docker DNS name "mariadb" doesn't resolve and the suite fails at bootstrap. - "DATABASE_URL": "mariadb", - "DATABASE_PORT": 3306, - "HUB_HOST": "0.0.0.0", - "HUB_API_HOST": getDockerContainerImageName(HUB_MODULE_NAME, "", ""), - "HUB_PORT": 10000, - // The indexer's hub client keys ENTIRELY off HUB_API_URL (hub_client.js: - // `this.enabled = !!this.hubUrl`). HUB_API_HOST above is set but read by - // nothing in xchain-indexer, so without this the client stayed disabled on - // every installed stack and no push ever left the indexer: chain tips, - // config, and in particular the PRICE v1 oracle_price pushes that a FIAT - // dispenser later prices against. Prod sets HUB_API_URL by hand in the - // per-coin config file, which is why this went unnoticed. - // - // Composed from the same container name + port as HUB_API_HOST/HUB_PORT, so - // it resolves on the docker network exactly as the sibling *_API_HOST vars - // do. An operator config-file value still wins, so prod's explicit - // cross-host URL is unaffected. - // - // This enables the push half only. The read-back half (hub tables mirrored - // into the indexer) additionally needs HUB_DB_NAME + HUB_DB_SYNC_ENABLED, - // which regtest deliberately leaves unset (see the network !== "regtest" - // block below), so turning this on cannot start a mirror bootstrap and - // cannot re-page price_snapshots out from under a seeded test venue. - "HUB_API_URL": "http://" + getDockerContainerImageName(HUB_MODULE_NAME, "", "") + ":10000", - // The e2e federation suites (test:federation / test:attestation:llm) boot - // in-process MultiValidatorHubs that create + drop XChain___MVH_* - // databases, so the container needs the hub DB credentials. DatabaseService - // grants this user CREATE/DROP on the XChain_%_MVH_% pattern when the hub - // module is installed. Omitting these made requireFederationEnv loud-fail. - "HUB_DB_HOST": "mariadb", - "HUB_DB_PORT": 3306, - "HUB_DB_USER": "xchain" + DB_SEP + "hub", - "HUB_DB_PASS": "xchain" + SEP + "password", - // Explorer is a SHARED service (no coin/network suffix). Like the hub - // above, the e2e-test container needs to reach it on the docker network; - // omitting it left EXPLORER_URL/EXPLORER_API_PORT unset, which failed the - // e2e harness's checkAllEnvironmentalVariables() → broken hub-config fallback. - "EXPLORER_URL": getDockerContainerImageName(EXPLORER_MODULE_NAME, "", ""), - "EXPLORER_API_PORT": 8080, - "EXPLORER_PORT": 8080 - } - - if (network === "regtest") { - defaultValues["REGTEST_MINER_URL"] = getDockerContainerImageName(XChainService.XCHAIN_REGTEST_MINER, coin, network) - defaultValues["REGTEST_MINER_API_PORT"] = 3005 - defaultValues["REGTEST_MINER_PORT"] = 3005 - // Encoder's express-rate-limit defaults to 60 RPM, which the e2e - // suite blows past whenever the stale-UTXO retry shim fires (up - // to 15 retries per failing tx, easily 100+ RPM during the - // order/swap blocks). Production-safe defaults stay at 60; we - // raise it for regtest where load is by-design bursty. - defaultValues["ENCODER_RATE_LIMIT_RPM"] = 99999 - // The browser wallet calls the encoder cross-origin (create_tx, - // ping). The encoder disables CORS unless CORS_ORIGIN is set, so a fresh - // regtest stack blocks every browser request and the wallet reports the - // chain "degraded". regtest is a local single-operator dev venue (same - // reasoning as INDEXER_ALLOW_UNAUTHENTICATED below), so default it open; - // a host CORS_ORIGIN or config-file value still wins. mainnet/testnet keep - // the fail-safe default (CORS off unless the operator opts in). - defaultValues["CORS_ORIGIN"] = config.CORS_ORIGIN || "*" - } - - // Encoder passthrough. A production encoder sits behind a reverse proxy on - // ANOTHER box, reached over a public address, so its default trust-proxy - // setting (loopback, uniquelocal) never honours X-Forwarded-For and the - // per-IP limiter keys every visitor on the proxy's egress address: one - // bucket per encoder for the whole world. ENCODER_TRUST_PROXY names that - // egress address so the container recovers the real client. - // ENCODER_RATE_LIMIT_RPM rides the same passthrough, placed after the - // regtest block above so a host value wins over the 99999 regtest literal - // and survives update/recreate. Read BY NAME, same as the explorer - // passthrough below: a computed process.env read is invisible to the - // platform's env-var coverage gate, which is what turns an undocumented - // variable into a silent one. - if (module === XChainService.XCHAIN_ENCODER) { - const encoderPassthroughVars = ["ENCODER_TRUST_PROXY", "ENCODER_RATE_LIMIT_RPM"] - for (const key of encoderPassthroughVars) { - const value = { - ENCODER_TRUST_PROXY: config.ENCODER_TRUST_PROXY, - ENCODER_RATE_LIMIT_RPM: config.ENCODER_RATE_LIMIT_RPM - }[key] - if (value === undefined || value === "") continue - defaultValues[key] = value - } - } - - // Native-coin protocol fee destination (per coin/network). Defaults from the vendored - // canonical coin registry (src/coins), so a stock install provisions the decoder's - // FEE_DESTINATION (fee-output capture into transaction_outputs) and the indexer's - // XCHAIN_FEE_DESTINATION__ without any operator env. Previously these - // were host-env-only, so default installs left the decoder capturing nothing and - // native-fee validation failing closed on every fee-bearing LTC/DOGE action (audit - // F-11). getCoinConfigByFullName applies the per-coin host-env override itself - // (ignored + warned on mainnet: fee acceptance is consensus and must not depend on - // operator env); the generic FEE_DESTINATION host var is honored on non-mainnet only. - // Injected as a default, so a value in the - config file still wins. - const feeDestEnvName = 'XCHAIN_FEE_DESTINATION_' + CoinTickerSymbol[coin] + '_' + network.toUpperCase() - const registryFeeDestination = getCoinConfigByFullName(coin, network).addresses.FEE_DESTINATION - const feeDestination = (network !== Network.MAINNET && !config.FEE_DESTINATION_ENV[feeDestEnvName] && config.FEE_DESTINATION) - ? config.FEE_DESTINATION - : registryFeeDestination - if (feeDestination) { - defaultValues['FEE_DESTINATION'] = feeDestination - defaultValues[feeDestEnvName] = feeDestination - } - - // Address-deriving modules (encoder/decoder/utxo-tracker) resolve their bitcoinjs - // network via CryptoNetworks, which keys on the COIN-PREFIXED name (e.g. - // "bitcoin-regtest"). A bare network ("regtest") matches no case → getBitcoinJsNetwork - // returns undefined → bitcoinjs-lib falls back to MAINNET → addresses derive with - // mainnet version bytes (a regtest "m..."/"n..." source comes out as "1..."), which then - // never matches on-chain balances (e.g. an issuance fee-check reads the wrong address and - // fails "insufficient funds (FEE)"). Those modules need the coin-prefixed network; the - // indexer/node keep the bare network (protocol-change matching / bitcoind conf). - if (module === XChainService.XCHAIN_ENCODER || module === XChainService.XCHAIN_DECODER || module === XChainService.XCHAIN_UTXO_TRACKER) { - defaultValues["NETWORK"] = coin + "-" + network - } - - // LevelDB tuning passthrough (xchain-utxo-tracker only). src/store/level_up_db.js reads - // LEVELDB_CACHE_BYTES (documented default 4 GiB, components/utxo-tracker/configuration.md:41) - // and LEVELDB_WRITE_BUFFER_BYTES from process.env inside the container, but - // getDefaultConfig never forwarded either host var into the tracker's - // defaultValues, so an operator exporting LEVELDB_CACHE_BYTES before - // install/update got silence: the container never saw it and LevelUpDb fell - // back to its in-container default every time. Mirrors hubPassthroughVars / - // genesisPassthroughVars: only set, non-empty host vars are injected, so an - // unset env leaves the tracker's own default untouched. - // Read by name rather than through a loop: the env-var doc-coverage gate can - // only see a variable it can name, and a computed process.env[varName] read - // widens its blind spot. - if (module === XChainService.XCHAIN_UTXO_TRACKER) { - const levelDbPassthrough = { - LEVELDB_CACHE_BYTES: config.LEVELDB_CACHE_BYTES, - LEVELDB_WRITE_BUFFER_BYTES: config.LEVELDB_WRITE_BUFFER_BYTES - } - for (const [varName, value] of Object.entries(levelDbPassthrough)) { - if (value !== undefined && value !== "") { - defaultValues[varName] = value - } - } - } - - // e2e-test also derives addresses (test/cryptoHelper.js) and resolves its - // bitcoinjs network from COIN+NETWORK. initialCheck.test.js reads - // process.env.COIN and only splits NETWORK when COIN is absent; without COIN - // it mis-splits the bare network ("regtest" → COIN="regtest", NETWORK=undefined) - // → getBitcoinJsNetwork returns undefined → bitcoinjs falls back to MAINNET - // ("1..." addresses) and funded txs never confirm on regtest. Inject COIN so - // the resolution is correct while NETWORK stays bare for other env consumers. - // - // The contract-template suites (test:sdk/*Template) load their source from - // xchain-contracts. LIBRARY_BUNDLES stages it into the e2e-test build context - // and the Dockerfile COPYs it to /XChainE2ETest/xchain-contracts, so point the - // resolver there. Without the bundle present the suites skip (they no longer - // abort the run). - if (module === XChainService.XCHAIN_E2E_TEST) { - defaultValues["COIN"] = coin - defaultValues["XCHAIN_CONTRACTS_DIR"] = "/XChainE2ETest/xchain-contracts" - - // The validator-onboarding suite STAKEs the hub's own signing pubkey and - // asserts the indexer then admits it to each capability set, so it needs - // to know which key the hub actually runs as. It read VALIDATOR_PUBKEY - // from the env and skipped when unset, which meant the only way to run it - // was for an operator to hand-copy the hex out of `validator status` into - // the coin config - so it skipped everywhere nobody had, including CI. - // Derive it from the same settings file the hub's own env comes from - // (getValidatorEnv above), so the two can never name different keys. - // - // PUBLIC half only. The seed stays in signing.key / SIGNING_PRIVKEY_HEX - // and goes to the hub alone; the test needs the pubkey and nothing else. - // - // A standalone node has no validator, so this is absent and the suite - // still skips - correctly, because there is no identity to onboard. - const { getValidatorSettings } = peers.validatorService - const validatorSettings = getValidatorSettings() - if (validatorSettings && validatorSettings.pubkey) { - defaultValues["VALIDATOR_PUBKEY"] = validatorSettings.pubkey - } - - // The harness discovers every rail's node and indexer credentials through - // the hub's getallconfigs (test/helpers/chainRail.js), and a keyed hub - // gates that read behind HUB_API_KEY. A validator-mode host is keyed - // (`validator init` mints the key into the hub sidecar), so without this - // passthrough the e2e container was the one hub client on the host still - // calling keyless: the litecoin and dogecoin matrix legs 401'd in - // initialCheck's beforeAll (`[chainRail] hub has no config for - // bitcoin/regtest`, run 35120852486) while the standalone bitcoin leg, - // whose hub has no key, never noticed. Host env first, then the sidecar, - // exactly as the indexer and the shared services resolve it; a keyless - // host stays keyless. - if (config.HUB_API_KEY !== undefined && config.HUB_API_KEY !== "") { - defaultValues.HUB_API_KEY = config.HUB_API_KEY - } - await applyHubApiKeyFromSidecar(defaultValues) - } - - // Genesis-ledger bootstrap env (xchain-indexer only). The indexer binds its - // consensus-critical genesis parameters from the container environment: mainnet/testnet - // are frozen-pinned in the indexer's configs/.js, but regtest reads the activation - // block + ledger/dump hashes from env so an operator can dry-run genesis at a current - // regtest block. Without this passthrough those host vars never reach the container, so - // genesis can't be enabled on a regtest/dev stack. Mirrors hubPassthroughVars: only set, - // non-empty host vars are injected (and a config-file value still wins), so an unset env - // leaves GENESIS_BLOCK at its 0/default and genesis stays off. The path vars point at - // in-container files; override them only when a custom CSV/dump is volume-mounted. - if (module === XChainService.XCHAIN_INDEXER) { - // The GENESIS_AIRDROP_* members carry the XCP/XDP airdrop leg. The indexer honors - // them on regtest ONLY: off regtest the armed bucket set comes from the - // pinned coin bundle, so passing them through cannot arm anything on a mainnet or - // testnet stack, only on the regtest dry-run this passthrough exists for. - const genesisPassthroughVars = [ - "XCHAIN_GENESIS_BLOCK", "XCHAIN_GENESIS_LEDGER_HASH", "XCHAIN_GENESIS_DUMP_HASH", - "GENESIS_LEDGER_PATH", "GENESIS_DUMP_PATH", - "GENESIS_BLOCK_TIMEOUT_MS", "GENESIS_DUMP_TIMEOUT_MS", - "GENESIS_AIRDROP_PATHS", "GENESIS_AIRDROP_HASHES", "GENESIS_AIRDROP_AMOUNTS", - "GENESIS_AIRDROP_SNAPSHOT_BLOCK", "GENESIS_AIRDROP_SET_HASH" - ] - for (const varName of genesisPassthroughVars) { - if (config.INDEXER_GENESIS_ENV[varName] !== undefined && config.INDEXER_GENESIS_ENV[varName] !== "") { - defaultValues[varName] = config.INDEXER_GENESIS_ENV[varName] - } - } - // ROLLCALL rail env (xchain-indexer only). Two separate things, both of which a - // deployed indexer needs before an epoch close can do anything at all. - // - // 1. DOGE_INDEXER_API_URL / DOGE_INDEXER_API_KEY, on EVERY network. Roll calls - // land on DOGECOIN and the BTC indexer is the only place the close runs, so - // rollcall_proof_client.js (and anchor_proof_client.js beside it) has to be - // able to ask a DOGE indexer. With no URL the close returns - // `{decided:false, reason:'DOGE indexer not configured'}` and the BTC indexer - // DEFERS the block forever, which is exactly how a single-coin venue wedges. - // Sourced from host env so the pair survives an `update` instead of needing - // to be hand-set on the container after every deploy. - // - // 2. XC_ROLLCALL_REGTEST_ACTIVATION, on REGTEST ONLY. This is the one value a - // regtest venue owns: the no-tunable-input rule is scoped to shared-ledger - // networks, because two regtest venues cannot fork each other. It is gated on - // the network here as well as in the indexer's own rollcall_activation.js, - // which is structurally unable to reach the environment for mainnet or - // testnet - two independent gates, so neither one being edited alone can arm - // a shared ledger from a host variable. - // - // 3. HUB_SYNC_ANCHOR_ATTEST_GRACE_S, on REGTEST ONLY, for the same reason as - // (2) and with the same two independent gates: the indexer's own - // resolveWatermarkGrace IGNORES it off regtest with a warning, because a - // watermark grace is a consensus input and a per-node value forks - // settlement. - // - // WHY A REGTEST VENUE NEEDS IT AT ALL. The anchor-reward attestation - // barrier holds a block until `streamWatermark >= blockTime + 120`. Off - // regtest that is free: blocks are ten minutes apart, so by the time one is - // processed the watermark is long past it. On regtest, blocks are stamped at - // about wall clock and the watermark tracks wall clock too, so a freshly - // mined block can NEVER be 120s behind the watermark and the barrier is - // unsatisfiable by construction. Every affected block then burns the full - // 60s timeout before proceeding anyway. - // - // MEASURED, on the 2026-09-06 release matrix: the BTC leg parsed 367 blocks - // in six hours and was killed by the job budget, against 2013 blocks in 1h52m - // on the pre-mirror build - 160 deferrals at 60s each, about 2.7 hours spent - // waiting for a condition that could not arrive. The other two coins were - // unaffected because this barrier is BTC-only. Nothing was wrong with the - // product: the venue was simply running a shared-ledger constant on a chain - // whose block cadence it was never sized for. - // 4. HUB_PRICE_SYNC_TIMEOUT_MS, on REGTEST ONLY here even though the value - // itself is not a consensus input. It bounds ONE mirror-barrier ATTEMPT: - // on expiry the block is DEFERRED and retried, never committed - // uncertified, which XChainIndexer states outright ("purely operational: - // it opens no barrier and commits no block"). So shortening it trades - // nothing away; it only makes a failed attempt cheaper. - // - // WHY A FAST VENUE NEEDS IT. Where the mirror legitimately lags the - // chain, every affected block waits the full attempt before deferring. - // Measured on the 2026-09-06 release matrix: 119 anchor-attest deferrals - // at the 60s default burned 119 minutes of a 289-minute BTC leg, 41% of - // the wall clock, and the indexer fell far enough behind that thirty - // e2e waits gave up on rows that had not landed yet. The barrier is - // doing its job; the cost per attempt is what a fast venue cannot afford. - // - // Gated on regtest anyway, because a shared ledger wants the long - // attempt: there a lagging mirror is a real fault worth waiting on, not - // a cadence mismatch. - // 5. XCHAIN_COINPAY_EXPIRATION_S, on REGTEST ONLY, for the same reason as (2) - // and (3) and with the same two independent gates: the indexer's own - // resolveCoinpayExpiration IGNORES it off regtest with a warning, because - // the window is added to a match's BLOCK_TIME and STORED as the - // obligation's deadline, so a per-node value expires the same escrow at - // different blocks and forks the ledger. - // - // WHY A REGTEST VENUE NEEDS IT. The e2e COINPay expiry case cannot wait out - // a two-hour deadline, so it freezes the node clock past the deadline and - // mines. That stamps the mined blocks two hours into the FUTURE, and the - // anchor-attest barrier in (3) compares a block's own timestamp against a - // wall-clock watermark, so the indexer then waits those two hours in real - // time on that one block. - // - // MEASURED, on the 2026-09-06 release matrix run 34015867460: all 119 - // deferrals in the BTC leg named the SAME block, held 2h08m50s, while the - // watermark tracked wall clock throughout (1-6s behind, advancing at 0.9999 - // of real time) and the hub logged no late heartbeat and no backpressure. - // Nothing was lagging. Shortening the window on regtest removes the clock - // jump that causes it, rather than teaching every barrier to special-case a - // future-stamped block. - const rollcallPassthroughVars = ["DOGE_INDEXER_API_URL", "DOGE_INDEXER_API_KEY"] - // XC_ROLLCALL_GATES_REGTEST_ACTIVATION follows XC_ROLLCALL_REGTEST_ACTIVATION's - // same env-derived regtest shape (D84): it arms ROLLCALL v1 and the rules-aware - // attestation set separately from the rail, so a venue can drive v0 as its control. - // - // XC_MIRROR_ADMISSION_ACTIVATION rides the same regtest-only shape: without a - // path here the indexer side of the admission-map mirror can never be armed on - // regtest (row 24x), and it must arm together with the hub's copy above or the - // admission-era canonical refuses a legacy-map row and halts the block loop. - if (network === Network.REGTEST) rollcallPassthroughVars.push("XC_ROLLCALL_REGTEST_ACTIVATION", - "XC_ROLLCALL_GATES_REGTEST_ACTIVATION", - "HUB_SYNC_ANCHOR_ATTEST_GRACE_S", - "HUB_PRICE_SYNC_TIMEOUT_MS", - "XCHAIN_COINPAY_EXPIRATION_S", - "XC_MIRROR_ADMISSION_ACTIVATION") - for (const varName of rollcallPassthroughVars) { - if (config.INDEXER_ROLLCALL_ENV[varName] !== undefined && config.INDEXER_ROLLCALL_ENV[varName] !== "") { - defaultValues[varName] = config.INDEXER_ROLLCALL_ENV[varName] - } - } - - // The indexer pushes chain tips / config to the hub (HUB_API_URL); when that - // hub enforces HUB_API_KEY, the indexer must present the same key or its writes - // 401. Sourced from host env (.env) so it persists across `update`, then from the - // shared hub sidecar so an indexer co-located with a validator hub picks up the - // key `validator init` generated. Neither set leaves the indexer sending no key - // (keyless, the prior default). - if (config.HUB_API_KEY !== undefined && config.HUB_API_KEY !== "") { - defaultValues.HUB_API_KEY = config.HUB_API_KEY - } - await applyHubApiKeyFromSidecar(defaultValues) - - // The hub authenticates to each indexer's federation API (attestation, stake polling, - // capability snapshots) with _INDEXER_API_KEY; the indexer fails closed unless its - // INDEXER_API_KEY matches. Source from host env so it persists across `update`, mirroring - // HUB_API_KEY above. Unset leaves the indexer fail-closed (keyless reads rejected). - if (config.INDEXER_API_KEY !== undefined && config.INDEXER_API_KEY !== "") { - defaultValues.INDEXER_API_KEY = config.INDEXER_API_KEY - } else if (network === Network.REGTEST) { - // With no key configured the indexer fails closed: every gated method - // (feequotedryrun, the federation reads the staking e2e family asserts - // against) 401s, so a fresh regtest install can never pass those suites - // (audit F-10). Regtest is a local single-operator venue, so default the - // indexer's own documented keyless escape hatch on. A config-file value - // or host INDEXER_API_KEY still wins; mainnet/testnet stay fail-closed. - defaultValues.INDEXER_ALLOW_UNAUTHENTICATED = "true" - } - - // Point the indexer's hub-DB connection (its price_snapshots/oracle_prices source) - // at its OWN database on mainnet/testnet: in this single-box topology HubDbSync - // mirrors those hub tables into the indexer DB, so the indexer reads prices from - // itself using its own DB account. Without HUB_DB_NAME the connection is never made - // and the mainnet native-fee price-source gate (XChainIndexer.start) fails closed. - // This mirrors the proven prod per-coin override; operator config overrides still - // win. HUB_DB_PASS is reconciled after the per-install DB password is resolved - // (see below); HUB_DB_HOST/PORT are already set above. - // - // WHY regtest WAS EXCLUDED UNTIL NOW, corrected 2026-07-26 then armed 2026-09-03 - // (the regtest mirror wedge). The old note here said "regtest has no hub to sync from", which - // stopped being true at 336a7d5 (HUB_API_URL is now composed for regtest too, and - // a regtest indexer does reach the hub: enabling this on litecoin-regtest - // bootstrapped 3 real rows into oracle_prices). The exclusion stood for a - // different and harder reason: turning the mirror on ARMS the block-loop - // watermark barriers (price, oracle, and now the ATTEST response mirror), and - // each one only opens once the mirror's stream watermark clears the row's time - // plus that barrier's grace. Production block timestamps LAG wall clock, so the - // watermark runs ahead and the escape fires; regtest blocks are stamped at ~now, - // so a real-network grace can NEVER be satisfied and every freshly mined block - // defers forever. Observed live: block 1479 deferred on a 60s timeout, repeatedly, - // until this was reverted. - // - // Armed unconditionally now (mainnet/testnet keep the exact same assignment they - // always had) because leaving the mirror off on regtest silently defeats every - // reader that expects hub state to reach the indexer, not just PRICE but - // the ATTEST response mirror this arms for too. Arming the pointer alone would - // reproduce the price wedge above, so every watermark grace this mirror gates is - // defaulted to 0 on regtest in the SAME step below: a config-file value or a host - // env override for any one of them still wins (resolveWatermarkGrace in - // hub_db_sync.js honours an override on regtest only, so the default below is - // exactly the value that seam already expects). Do not widen these off regtest: - // a per-node grace forks settlement. - defaultValues.HUB_DB_NAME = defaultValues.INDEXER_DB_NAME - defaultValues.HUB_DB_USER = defaultValues.INDEXER_DB_USER - defaultValues.HUB_DB_SYNC_ENABLED = "true" - - if (network === Network.REGTEST) { - // The three barrier graces the armed regtest mirror must clear to avoid the - // wedge above. HUB_SYNC_ATTEST_RESPONSE_GRACE_S is the passthrough this row - // adds (xchain-indexer/src/hub/hub_db_sync.js:615 reads it via resolveWatermarkGrace); - // HUB_SYNC_PRICE_GRACE_S / HUB_SYNC_ORACLE_GRACE_S are the pair the regtest mirror wedge already - // requires be set to 0 alongside it. A host env value always wins over the - // regtest default so an e2e drill can still exercise a nonzero grace. - // The list is EVERY watermark grace hub_db_sync.js resolves, not the three - // that first wedged: each mirrored table has its own barrier, and any one - // left at its frozen default holds every block up to 60s while that - // table's mirror watermark stands still, which on a three-rail regtest - // venue idle for hours is every block; an SDK drive's 120s index wait - // then dies on the second block. Measured 2026-09-09 on the match - // barrier, then again on the anchor-reward attestation barrier once - // match was cleared (hub_db_sync.js reads all of them through - // resolveWatermarkGrace, regtest-overridable only). - const hubSyncRegtestGraceVars = [ - "HUB_SYNC_PRICE_GRACE_S", "HUB_SYNC_ORACLE_GRACE_S", "HUB_SYNC_ATTEST_RESPONSE_GRACE_S", - "HUB_SYNC_MATCH_GRACE_S", "HUB_SYNC_CALL_GRACE_S", "HUB_SYNC_ANCHOR_ATTEST_GRACE_S" - ] - for (const varName of hubSyncRegtestGraceVars) { - defaultValues[varName] = (config.HUB_SYNC_GRACE_ENV[varName] !== undefined && config.HUB_SYNC_GRACE_ENV[varName] !== "") - ? config.HUB_SYNC_GRACE_ENV[varName] - : "0" - } - } - } + let defaultValues; if (coin && network) { + defaultValues = defaults.createCoinDefaults(module, coin, network); coins.applyCoinDefaults(defaultValues, module, coin, network) + services.configureE2e(defaultValues, module, coin) + if (module === XChainService.XCHAIN_E2E_TEST) await applyHubApiKeyFromSidecar(defaultValues) + services.configureIndexerBeforeHubKey(defaultValues, module, network) + if (module === XChainService.XCHAIN_INDEXER) await applyHubApiKeyFromSidecar(defaultValues) + services.configureIndexerAfterHubKey(defaultValues, module, network) } else { - defaultValues = { - "HUB_HOST": "0.0.0.0", - "HUB_API_HOST": getDockerContainerImageName(HUB_MODULE_NAME, "", ""), - "HUB_PORT": 10000, - "HUB_DB_HOST": "mariadb", - "HUB_DB_PORT": 3306, - "HUB_DB_NAME": "XChain" + DB_SEP + "Hub", - "HUB_DB_USER": "xchain" + DB_SEP + "hub", - "HUB_DB_PASS": "xchain" + SEP + "password", - "EXPLORER_HOST": "127.0.0.1", - "EXPLORER_PORT": 18080, - "EXPLORER_API_HOST": getDockerContainerImageName(EXPLORER_MODULE_NAME, "", ""), - "EXPLORER_API_USER": false, - "EXPLORER_API_PASS": false, - "EXPLORER_API_PORT_HTTP": 8080, - "EXPLORER_PORT_HTTP": 18080, - "EXPLORER_API_PORT_HTTPS": 8081, - "EXPLORER_PORT_HTTPS": 18081, - "SYNC_MODE": "server", - "SYNC_API_PORT": 3006, - "SYNC_PORT": 3006, - "SYNC_API_HOST": getDockerContainerImageName(SYNC_MODULE_NAME, "", ""), - "HUB_API_HOST_SYNC": getDockerContainerImageName(HUB_MODULE_NAME, "", "") - } - - // Allow the operator to override the shared hub's port via host env. - // The hub has no per-coin config file, so host env is the injection point, same - // as the explorer override below. constants.js's HUB_MODULE_NAME.docker.ports - // entry maps BOTH the published host port and the container-internal port from - // this one HUB_PORT value (`-p ${HUB_PORT}:${HUB_PORT}`), and every hub client - // in this file (buildCheckpointConfig, updateHubOrExplorer, ...) reads the - // computed defaultValues.HUB_PORT rather than the constants.js default, so - // overriding it here keeps the published port, the container's own listener, - // and every in-process caller in agreement. Motivating case: a second - // co-located xchain-node install (e.g. verifying `install master xchain-hub` - // boots correctly) needs its hub reachable on a host port distinct from a - // standing shared hub's 10000, which had no override at all and so could only - // be tested by tearing the shared hub down or standing up a whole separate - // Docker daemon. - if (config.HUB_PORT_OVERRIDE !== undefined && config.HUB_PORT_OVERRIDE !== "") { - defaultValues.HUB_PORT = config.HUB_PORT_OVERRIDE - } - - // Allow the operator to override the explorer's published HOST ports via host - // env. Shared services (explorer/hub) have no per-coin config file, so host env - // is the injection point (same pattern as the hub passthrough vars below). The - // motivating case: a second co-located xchain-node install (e.g. a federation - // stack alongside the primary node) must publish the explorer on a non-default - // port to avoid colliding with the primary's 18080/18081. Container-internal - // ports (EXPLORER_API_PORT_HTTP/HTTPS) are unchanged. - for (const k of ["EXPLORER_PORT_HTTP", "EXPLORER_PORT_HTTPS", "EXPLORER_PORT"]) { - if (config.EXPLORER_PORT_ENV[k] !== undefined && config.EXPLORER_PORT_ENV[k] !== "") { - defaultValues[k] = config.EXPLORER_PORT_ENV[k] - } - } - - // The explorer hard-requires a co-located hub-mirror DB (state_checkpoints / - // capability_snapshots / cross_chain_matches) per serving coin, since - // xchain-sync never replicates those tables. Regtest and dev stacks don't run - // that replication, so the explorer would crash-loop on startup there. Let the - // operator opt out via host env (ALLOW_NO_COLOCATED_HUB_DB=1): the hub-mirrored - // endpoints then fail loud per-request instead of blocking startup. Unset on - // mainnet/testnet so the missing-DB guard still catches a real misconfiguration. - if (config.ALLOW_NO_COLOCATED_HUB_DB !== undefined && config.ALLOW_NO_COLOCATED_HUB_DB !== "") { - defaultValues.ALLOW_NO_COLOCATED_HUB_DB = config.ALLOW_NO_COLOCATED_HUB_DB - } - - // Shared services that call the hub as clients (the sync server's config - // discovery via getallconfigs, the explorer's hub reads) must present the - // hub's API key once the hub enforces its sensitive-read tier: getallconfigs - // 401s keyless when HUB_API_KEY is set hub-side. Sourced from host env so it - // persists across `update`, then from the shared hub sidecar, mirroring the - // indexer's passthrough above. Neither set keeps the prior keyless behavior - // (fine against a keyless hub). - if (config.HUB_API_KEY !== undefined && config.HUB_API_KEY !== "") { - defaultValues.HUB_API_KEY = config.HUB_API_KEY - } + defaultValues = defaults.createSharedDefaults(); defaults.configureSharedBeforeHubKey(defaultValues) await applyHubApiKeyFromSidecar(defaultValues) - - // The browser wallet calls the hub cross-origin (ping, config reads). - // The hub disables CORS unless CORS_ORIGIN is set, so a browser wallet is - // blocked and reports the chain "degraded". The hub is a shared, network- - // agnostic service (it may front mainnet), so unlike the per-network encoder - // above it is NOT auto-defaulted open: the operator opts in via host env. - // On a local regtest dev box set CORS_ORIGIN=* when installing the hub. - if (config.CORS_ORIGIN !== undefined && config.CORS_ORIGIN !== "") { - defaultValues.CORS_ORIGIN = config.CORS_ORIGIN - } - - if (module === EXPLORER_MODULE_NAME) { - // Self-synced hub-mirror checkpoint schema (row 39, #4138 decoupling): - // HubMirrorSyncManager needs the hub's own REST base URL to pull - // state_checkpoints / capability_snapshots / cross_chain_matches, which - // is a DIFFERENT thing from HUB_API_HOST/HUB_PORT above (those feed the - // explorer's ordinary getallconfigs config poll, not the mirror writer). - // Opt-in via host env EXPLORER_CHECKPOINT_SELF_SYNC, read directly by - // HubService.buildHubModuleConfig's checkpoint injection (see there for - // why this stays a second knob instead of piggybacking - // ALLOW_NO_COLOCATED_HUB_DB: that flag only downgrades the fatal - // startup assertion to a warning and says nothing about whether a - // local mirror should be provisioned). Emitted only when opted in, so - // a deployment that never uses self-sync carries no unused hub URL. - // - // This env is no longer the mirror writer's ONLY source: the same URL - // now ships inside the checkpoint config block beside self_sync - // (HubService.buildCheckpointConfig), because these two lived on - // different delivery paths - container env written at install time - // versus the hub's config push - and opting in after the container - // existed left the explorer self-syncing with nowhere to sync from. - // Kept for the explorer's other hub reads (HubOperationalCache) and as - // the fallback for hand-written config.json deployments. - if (config.EXPLORER_CHECKPOINT_SELF_SYNC !== undefined && config.EXPLORER_CHECKPOINT_SELF_SYNC !== "") { - defaultValues.HUB_API_URL = config.HUB_API_URL || - ("http://" + getDockerContainerImageName(HUB_MODULE_NAME, "", "") + ":" + defaultValues.HUB_PORT) - } - - // Read Contract simulation (contract.html #contract-read-card) is - // default-off; the readers test for the exact STRING 'true', not any - // truthy value, so pass the host env through verbatim rather than - // coercing it. Sourced from host env so it persists across `update`/ - // `recreate`, mirroring the other explorer passthroughs here. - if (config.EXPLORER_VM_QUERY_ENABLED !== undefined && config.EXPLORER_VM_QUERY_ENABLED !== "") { - defaultValues.EXPLORER_VM_QUERY_ENABLED = config.EXPLORER_VM_QUERY_ENABLED - } - - // Serving limits, same host-env injection point as the knobs above, - // because every one of these defaults is tuned for a PUBLIC explorer - // and is wrong for a private venue: - // EXPLORER_*RATE_LIMIT_RPM - the nine request budgets, per IP: the - // app-wide cap, the quote/pre-flight caps, and the six per-route - // caps (checkpoint-list, checkpoint-verify, action-proof, - // validator-set-proof, vm-query, batch). A dev box reaches the explorer - // through one tunnel, so every browser and every test run shares a - // single bucket, and a browser-driven suite sustains far more than - // any one of these caps on its own. All nine are now reachable - // from the host env; the five per-route caps were unreachable on a - // node-managed explorer (the regtest venue), which could raise only - // the app-wide and fee-quote caps before this change. - // EXPLORER_TIP_MAX_AGE_S - 6h by default, and 0 disables it. A - // regtest chain has no block cadence: it advances only when someone - // mines, so an idle one crosses the age gate and the explorer delists - // a chain that is perfectly healthy (503 COIN_DATA_STALE on every - // read, lag 0 on /status). - // Read BY NAME rather than by scanning process.env for a pattern: a - // computed read is invisible to the platform's env-var coverage gate, - // which is what turns an undocumented variable into a silent one. The - // per-coin EXPLORER_TIP_MAX_AGE_S_ form is deliberately NOT - // carried here - the explorer honours it directly, and the global knob - // already covers the case this passthrough exists for (an instance - // serving nothing but a private venue). - for (const key of [ - "EXPLORER_RATE_LIMIT_RPM", - "EXPLORER_FEE_QUOTE_RATE_LIMIT_RPM", - "EXPLORER_PREFLIGHT_POST_RATE_LIMIT_RPM", - "EXPLORER_TIP_MAX_AGE_S", - "EXPLORER_CHECKPOINT_LIST_RATE_LIMIT_RPM", - "EXPLORER_CHECKPOINT_VERIFY_RATE_LIMIT_RPM", - "EXPLORER_ACTION_PROOF_RATE_LIMIT_RPM", - "EXPLORER_VALIDATOR_SET_PROOF_RATE_LIMIT_RPM", - "EXPLORER_VM_QUERY_RATE_LIMIT_RPM", - "EXPLORER_BATCH_RATE_LIMIT_RPM" - ]) { - const value = { - EXPLORER_RATE_LIMIT_RPM: config.EXPLORER_RATE_LIMIT_RPM, - EXPLORER_FEE_QUOTE_RATE_LIMIT_RPM: config.EXPLORER_FEE_QUOTE_RATE_LIMIT_RPM, - EXPLORER_PREFLIGHT_POST_RATE_LIMIT_RPM: config.EXPLORER_PREFLIGHT_POST_RATE_LIMIT_RPM, - EXPLORER_TIP_MAX_AGE_S: config.EXPLORER_TIP_MAX_AGE_S, - EXPLORER_CHECKPOINT_LIST_RATE_LIMIT_RPM: config.EXPLORER_CHECKPOINT_LIST_RATE_LIMIT_RPM, - EXPLORER_CHECKPOINT_VERIFY_RATE_LIMIT_RPM: config.EXPLORER_CHECKPOINT_VERIFY_RATE_LIMIT_RPM, - EXPLORER_ACTION_PROOF_RATE_LIMIT_RPM: config.EXPLORER_ACTION_PROOF_RATE_LIMIT_RPM, - EXPLORER_BATCH_RATE_LIMIT_RPM: config.EXPLORER_BATCH_RATE_LIMIT_RPM, - EXPLORER_VALIDATOR_SET_PROOF_RATE_LIMIT_RPM: config.EXPLORER_VALIDATOR_SET_PROOF_RATE_LIMIT_RPM, - EXPLORER_VM_QUERY_RATE_LIMIT_RPM: config.EXPLORER_VM_QUERY_RATE_LIMIT_RPM - }[key] - if (value === undefined || value === "") continue - defaultValues[key] = value - } - } - - // The explorer resolves each coin's utxo-tracker and decoder from - // UTXO_TRACKER_URL_ (e.g. UTXO_TRACKER_URL_RBTC) and - // DECODER_API_URL__ (e.g. DECODER_API_URL_BTC_REGTEST). - // The explorer is a shared service with no per-venue config file, so these - // were never emitted anywhere: every coin reported tracker_available:false - // and decoder_health 'unconfigured', which blanks address balances/UTXOs - // (the wallet's balance source). Emit the pair for every coin/network at - // the venue containers' internal ports; entries for venues not installed - // on this host are inert because the explorer only probes coins it serves. - if (module === EXPLORER_MODULE_NAME) { - const networkCodePrefix = { [Network.MAINNET]: "", [Network.TESTNET]: "T", [Network.REGTEST]: "R" } - for (const coinName of Object.values(Coin)) { - for (const net of Object.values(Network)) { - const tick = CoinTickerSymbol[coinName] - defaultValues["UTXO_TRACKER_URL_" + networkCodePrefix[net] + tick] = - "http://" + getDockerContainerImageName(XChainService.XCHAIN_UTXO_TRACKER, coinName, net) + ":3001" - defaultValues["DECODER_API_URL_" + tick + "_" + net.toUpperCase()] = - "http://" + getDockerContainerImageName(XChainService.XCHAIN_DECODER, coinName, net) + ":3002" - - // Same omission, one service later. The explorer's - // quote/pre-flight proxies (/api/preflight, /api/feequote, - // /api/oraclefeequote) resolve their upstream from - // INDEXER_API_URL__, which was never emitted - // here, so every one of them answered INDEXER_NOT_CONFIGURED - // on a container install. That silently reduced the wallet's - // pre-flight to its client-side tier: the confirm surface - // still renders a verdict, just never the indexer's own. - // - // Unlike the two above, this one yields to the host env. The - // explorer is also run natively (systemd) on hosts where the - // indexers live on OTHER boxes, and those set this var by - // hand to a remote address; a container-local - // default that overrode it would point a working production - // explorer at a hostname that does not resolve. - const indexerVar = "INDEXER_API_URL_" + tick + "_" + net.toUpperCase() - defaultValues[indexerVar] = config.INDEXER_API_URL_ENV[indexerVar] - || "http://" + getDockerContainerImageName(XChainService.XCHAIN_INDEXER, coinName, net) + ":3004" - } - } - } + defaults.configureSharedAfterHubKey(defaultValues, module) } - - // Usage-telemetry env is only meaningful to the hub (the telemetry collector). - // TELEMETRY_IP_SALT is read from the host environment (e.g. xchain-node's .env) so - // the IP-hash salt stays out of source and config files; without it the hub records - // country/region but leaves ip_hash null. The hub is a shared service (no per - // coin/network config file), so the host env is the injection point. + networks.configureHubBeforeKey(defaultValues, module) if (module === HUB_MODULE_NAME) { - defaultValues["TELEMETRY_ENABLED"] = config.TELEMETRY_ENABLED - defaultValues["TELEMETRY_RETENTION_DAYS"] = config.TELEMETRY_RETENTION_DAYS - defaultValues["TELEMETRY_IP_SALT"] = config.TELEMETRY_IP_SALT - // Gate for the per-install detail endpoint (GET /telemetry/operators). Like the - // salt, sourced from host env so the secret stays out of source/config files; - // unset leaves the endpoint fail-closed (401 for everyone). - defaultValues["TELEMETRY_ADMIN_KEY"] = config.TELEMETRY_ADMIN_KEY - - // BTC indexer JSON-RPC URL for the validator-mode price oracle's block-height - // anchor (hub.getlatestblock). Sourced from host env so a hub NOT co-located with - // a BTC indexer (e.g. the master hub box, where the BTC stack lives elsewhere) can - // point at a reachable indexer. Empty default ⇒ the hub falls back to its configs - // table, so co-located standalone/validator installs are unaffected. Left empty - // here, it is composed from the co-located BTC indexer further down. - defaultValues["BTC_INDEXER_API_URL"] = config.BTC_INDEXER_API_URL - - // State-checkpoint engine + ANCHOR publisher (validator mode). The hub is a - // shared service (no per coin/network config file), so like the telemetry - // salt and BTC_INDEXER_API_URL above, the host env is the injection point. - // Per-coin _INDEXER_URLs feed getblockhashes (checkpoint state reads); - // DOGE_* configures the on-chain ANCHOR/price publisher signer pipeline; - // XDEX_* are the shared single-validator/regtest seams. Only set values are - // injected, so unset host env leaves the hub's own defaults untouched. - const hubPassthroughVars = [ - // HUB_API_KEY gates the hub's consensus-affecting write methods. Sourced from - // host env (.env) so it persists across `update` (a hand-set container value is - // dropped on rebuild). Set it on the publicly-fronted master hub so writes are - // authenticated; unset leaves the hub keyless (the prior default). - "HUB_API_KEY", - "BTC_INDEXER_URL", "LTC_INDEXER_URL", "DOGE_INDEXER_URL", - "BTC_INDEXER_API_KEY", "LTC_INDEXER_API_KEY", "DOGE_INDEXER_API_KEY", - "CHECKPOINT_ENABLED", "CHECKPOINT_INTERVAL_BLOCKS", "CHECKPOINT_CONFIRMATIONS", - "CHECKPOINT_POLL_MS", "CHECKPOINT_ROUND_TIMEOUT_MS", "CHECKPOINT_CHAINS", - "ANCHOR_ENABLED", "ANCHOR_INTERVAL_MS", "ANCHOR_MATCH_BATCH_SIZE", - "ANCHOR_MAX_BATCH", "ANCHOR_CHUNK_MAX_BYTES", "ANCHOR_ROUND_TIMEOUT_MS", - // ANCHOR_CHUNK_RETRY_MS must outlast the utxo-tracker's mempool poll - // (60s on mainnet) or back-to-back same-wallet anchor broadcasts - // exhaust their retries on a stale UTXO view (txn-mempool-conflict). - "ANCHOR_CHUNK_RETRY_MS", - "ANCHOR_ELECTION_TOLERANCE_BLOCKS", "ANCHOR_REWARD_PER_PUBLISH", - // Anchor every Nth checkpoint_seq on-chain (off-multiples stay in the - // free off-chain mirror); decouples DOGE spend from checkpoint cadence. - "ANCHOR_CHECKPOINT_EVERY_N", - "DOGE_ENCODER_URL", "DOGE_ENCODER_API_KEY", "DOGE_ADDRESS", - "DOGE_PUBKEY_HEX", "DOGE_LOW_BALANCE_THRESHOLD", - "XDEX_SEED_LOCAL_VALIDATOR", "XDEX_SNAPSHOT_BLOCK", - // Per-coin confirmation depth the hub's cross-chain engines wait for - // before proposing a source leg (coins/index.js resolveConfirmations). - // A regtest venue pins these to 1 so a bridge lock finalizes on the - // next block instead of six BTC blocks nothing is mining (the nightly - // two-stack legs sat on "not proposing BTC:3 (below depth 6)" until - // the 120 s credit wait gave up). Inert on mainnet and testnet: the - // hub clamps a value below the per-coin default UP to that default - // off regtest, so this can only raise the depth on a real network. - "XCHAIN_CONFIRMATIONS_BTC", "XCHAIN_CONFIRMATIONS_LTC", "XCHAIN_CONFIRMATIONS_DOGE", - // Reverse-proxy trust for the hub's express API (rate-limiter IP - // keying). Default 'loopback' suits the Apache-on-same-host prod - // topology; containerized hubs see the docker bridge as the peer, - // so an operator fronting the container with a proxy sets this. - "HUB_TRUST_PROXY", - // Deployment network for the hub's consensus gates (notably - // STAKE_WEIGHTED_QUORUM, whose activation height is per-network). - // REQUIRED by the hub in validator mode (it fails loud on a - // blank/invalid value; no silent default here either) and must - // match the INDEXER_NETWORK of the chains this hub federates. - "HUB_NETWORK", - // Oracle price-round finalization threshold. Defaults to 2 in the hub - // (a 2-hub diversity floor so a lone external source never becomes a - // federation-signed price). Single-host prod / regtest deployments must - // set ORACLE_MIN_SUBMISSIONS=1 explicitly or no round ever finalizes, - // which stalls every indexer's oracle price-sync barrier. Passed through - // here so the host env survives a hub container regenerate. - "ORACLE_MIN_SUBMISSIONS", - // Oracle round cadence. CONSENSUS-UNIFORM: every hub in a federation must - // share these or round numbering and the submission cutoff diverge. Passed - // through for single-validator regtest/e2e venues, where short rounds keep - // a live drill from waiting 10 minutes per finalization; real networks - // leave them unset and take the hub defaults. - "ORACLE_ROUND_INTERVAL", "ORACLE_SUBMISSION_WINDOW", - // PRICE batch-publisher knobs (window length, finalization grace, - // co-sign timeout, buffer cap). Deliberately NOT consensus-grouped in - // HubConsensusEnvGuard: batch validation is range-agnostic, so two hubs - // running different window sizes just elect different leaders and may - // double-publish overlapping windows, which is idempotent at ingest and - // is the same posture today's publisher failover already has. They - // change what a leader PROPOSES, never what any node ACCEPTS. Passed - // through so the host env survives a hub container regenerate. - "ORACLE_BATCH_WINDOW_ROUNDS", "ORACLE_BATCH_GRACE_MS", - "ORACLE_BATCH_SIGN_TIMEOUT_MS", "ORACLE_BATCH_BUFFER_MAX_ROUNDS", - // Same family: the time budgeted between a window closing and its batch - // being readable on chain (assembly, co-signing, broadcast, one DOGE - // confirmation). The publisher subtracts it from the fee-price staleness - // bound to derive the window ceiling, so a venue whose - // landing latency differs from the fleet's tunes it here rather than - // being clamped to a window that does not suit it. - "ORACLE_BATCH_LANDING_RESERVE_MS", - // ATTEST response mirror regtest-only overrides (the attest response mirror design). Both are - // honoured by the receiving hub module ONLY when HUB_NETWORK=regtest (a warn- - // and-ignore off regtest, the same posture resolveWatermarkGrace takes on the - // indexer side), so passing them through here unconditionally mirrors the - // ORACLE_BATCH_* family above: they cannot arm anything off regtest by any path - // in this file, the real gate lives at the point of consumption. - // - // ATTEST_RESPONSE_FORWARD_S_OVERRIDE lets a regtest venue's leader pick a short - // effective_time margin instead of the real 120s ATTEST_RESPONSE_FORWARD_S, so a - // response can bind within the same short block cadence a regtest drill runs at - // (xchain-hub/src/attestation/attest_response_timing.js). - "ATTEST_RESPONSE_FORWARD_S_OVERRIDE", - // ATTEST_BATCH_WINDOW_S_OVERRIDE is the same seam for the batch cadence: - // AttestationBatchPublisher (row 20, not yet built) will read it on the same - // regtest-only pattern as the forward override above, so the passthrough is - // wired ahead of that publisher rather than after it. - "ATTEST_BATCH_WINDOW_S_OVERRIDE", - // Per-IP request/min cap on the hub's express API (default 100). Too low - // for legitimate multi-indexer re-bootstrap: every indexer on a box shares - // one source IP, so a fleet bootstrapping HubDbSync tables (oracle_prices, - // price_snapshots, cross_chain_calls, capability_snapshots, state_checkpoints) - // collectively blows 100/min and gets 429'd, so the heartbeat gate then stays - // closed and the chain stalls. Raise for prod fleets. Passed through so the - // host env survives a hub container regenerate. - // - // The hub exempts loopback and private-range callers from - // that cap by default, which covers the case above: the indexers reach the hub - // container over the bridge network this compose file creates, so a managed - // node no longer needs the limit raised to rebuild price history from the chain. - // HUB_RATE_LIMIT_EXEMPT_LOCAL=false turns the exemption off and restores the - // old behavior for an operator who wants the cap enforced on every caller; - // passed through for the same container-regenerate reason. - "HUB_RATE_LIMIT_RPM", "HUB_RATE_LIMIT_EXEMPT_LOCAL", - // XCHAIN derived-price source. XCHAIN is listed on no exchange, so - // a validator computes XCHAIN/USD from realized fills in its OWN BTC indexer - // database instead of fetching it. Every native-coin fee decision on LTC and - // DOGE needs that pair, and without it those chains cannot price a fee at all. - // - // Read-only access; unset means the hub simply abstains from the pair and - // submits the 36 API pairs exactly as before, which is a supported state. - // Passed through here so the values survive a hub container regenerate - a - // config file alone never reaches the container. - "XCHAIN_PRICE_INDEXER_DB_HOST", "XCHAIN_PRICE_INDEXER_DB_PORT", - "XCHAIN_PRICE_INDEXER_DB_NAME", "XCHAIN_PRICE_INDEXER_DB_USER", - "XCHAIN_PRICE_INDEXER_DB_PASS", "XCHAIN_PRICE_INDEXER_DB_COIN", - // Consensus-uniform derivation parameters (window length, confirmation - // buffer, bootstrap price). Overrides exist for regtest and e2e only: a hub - // running different values computes a different XCHAIN/BTC leg and lands - // outside the co-sign deviation band, so on a real network leave them unset - // and move them only by a coordinated flag-day. - "XCHAIN_PRICE_WINDOW_BLOCKS", "XCHAIN_PRICE_CONFIRMATION_BUFFER", - "XCHAIN_PRICE_BOOTSTRAP_SATS", - // D2 supersession threshold override. The shipped constant keeps - // supersession disabled (bootstrap carry-forward only); regtest/e2e drills - // set '0' so any realized volume supersedes, which is what lets a live - // proof distinguish a derived print from the carry-forward it would - // otherwise silently match. - "XCHAIN_PRICE_MIN_BTC_VOLUME", - // Hub API authentication. Without these two passed through, VALIDATOR MODE - // IS UNREACHABLE: the hub refuses to boot in validator mode unless one of - // them is set ("HUB_API_KEY is not set in validator mode. Write methods - // would be UNAUTHENTICATED"), and neither could previously reach the - // container at all. - // - // That failure also WEDGES the installer, so it is worth more than a - // one-line fix: once P2P_VALIDATOR_ADDR is baked into the container env the - // hub crash-loops, and `update xchain-hub` then fails because its own - // precheck tries to restart the container that cannot start. Recovery is - // `docker rm -f` the container and update again. - // - // HUB_ALLOW_UNAUTHENTICATED=true is the documented keyless escape hatch and - // suits a single-host regtest venue that already ran open; a real network - // sets HUB_API_KEY instead. - "HUB_API_KEY", "HUB_ALLOW_UNAUTHENTICATED", - // The regtest ROLLCALL arming opt-in. The hub carries a byte-twin of - // the indexer's rollcall_activation.js, and ROLLCALL_ACTIVATION is one of - // consensus_rules_digest.js's SHARED_GATES, so an indexer armed against an - // inert container hub reports a rules MISMATCH on the venue. Both sides take - // the same variable, so a venue arms as a unit. - // - // Passed through with no network gate, unlike the indexer's copy above: the - // hub is a shared service and getDefaultConfig is called for it as - // (module, null, null), so there is no network here to gate on. That is safe - // because the real gate is in the hub's own rollcall_activation.js, which can - // reach the environment for regtest and for nothing else - mainnet and testnet - // are literal there and unreachable from env by any path in the file. On a - // mainnet or testnet hub this variable is therefore inert, not dangerous. - "XC_ROLLCALL_REGTEST_ACTIVATION", - // XC_ROLLCALL_GATES_REGTEST_ACTIVATION rides beside it with the same - // no-network-gate reasoning: it arms ROLLCALL v1 and the rules-aware - // attestation set separately from the rail, so a venue can drive v0 as its - // control, and the hub's own rollcall_gates_activation.js gates it for real. - "XC_ROLLCALL_GATES_REGTEST_ACTIVATION", - // XC_MIRROR_ADMISSION_ACTIVATION follows the same no-network-gate shape - // (D84 precedent): it arms the admission-map mirror and its consumer and - // barrier gates together, so a venue arms as a unit; the hub's own - // mirror-admission gate module gates it for real. - "XC_MIRROR_ADMISSION_ACTIVATION" - ] - for (const varName of hubPassthroughVars) { - // Secret-bearing names in this list (XCHAIN_PRICE_INDEXER_DB_PASS) are also - // accepted from the host env under their redaction-safe `*_SECRET` spelling; - // everything else resolves to a plain process.env read. - const value = readSecretHostEnv(varName) - if (value !== undefined && value !== "") { - defaultValues[varName] = value - } - } - - // `validator init` leaves a generated key in the shared hub sidecar so the - // onboarding path produces a hub that BOOTS. Read it here (host env still wins), - // before the keyless declaration below: a node that has a credential must deploy - // authenticated rather than be handed the escape hatch it no longer needs. await applyHubApiKeyFromSidecar(defaultValues) - - // The hub now REFUSES to boot when HUB_API_KEY is unset unless - // keyless operation is declared with HUB_ALLOW_UNAUTHENTICATED. A managed - // deploy with no key in the host env is a legitimate posture (single-host - // regtest, a hub reachable only on a private network), so make the - // declaration here rather than letting the container crash-loop: the point - // of the hub-side change is that keyless is a stated choice, and the - // deployer is what states it. An operator who wants the refusal instead - // sets HUB_ALLOW_UNAUTHENTICATED=false in the host env, which the - // passthrough above preserves. `xchain-node go-live` still refuses a - // keyless mainnet hub outright (GoLiveGate). - // MAINNET is the exception: there the review's "invert the defaults, fail - // closed" applies with real funds behind it, so we do NOT declare keyless - // on the operator's behalf and the hub's own refusal stands. (A mainnet - // VALIDATOR hub is already covered: the hub has refused keyless validator - // boots since before this change, so no running one can be keyless and - // undeclared. This only reaches a mainnet config-only hub.) - const hubNetworkIsMainnet = String(defaultValues["HUB_NETWORK"] || "").toLowerCase() === Network.MAINNET - if (!defaultValues["HUB_API_KEY"] && defaultValues["HUB_ALLOW_UNAUTHENTICATED"] === undefined - && !hubNetworkIsMainnet) { - defaultValues["HUB_ALLOW_UNAUTHENTICATED"] = "true" - logger.warn("WARNING: HUB_API_KEY is not set, so this hub is deployed with an UNAUTHENTICATED " + - "write surface (HUB_ALLOW_UNAUTHENTICATED=true). Anyone who can reach the hub port can drive " + - "updateconfig / registervalidator / reportreorg. Set HUB_API_KEY in the host env before " + - "exposing this hub beyond a trusted network.") - } - - // A hub with no HUB_NETWORK resolves its network to '', which fails every - // network-keyed ingest gate closed, so a non-validator install can never - // validate an on-chain PRICE batch. Host env still wins; unresolved stays unset. + networks.configureHubAccess(defaultValues) if (!defaultValues["HUB_NETWORK"]) { - const deploymentNetwork = await resolveDeploymentHubNetwork() - if (deploymentNetwork) defaultValues["HUB_NETWORK"] = deploymentNetwork + const deploymentNetwork = await networks.resolveDeploymentHubNetwork(); if (deploymentNetwork) defaultValues["HUB_NETWORK"] = deploymentNetwork } - - // Capability snapshots are read off a BTC indexer, and with none reachable the - // hub refuses every on-chain PRICE batch for insufficient signer stake. Compose - // the co-located one; say so when this deployment has none to compose. if (!defaultValues["BTC_INDEXER_API_URL"] && defaultValues["HUB_NETWORK"]) { const btcNetwork = String(defaultValues["HUB_NETWORK"]).toLowerCase() - if (await hasBitcoinIndexer(btcNetwork)) { - defaultValues["BTC_INDEXER_API_URL"] = "http://" + - getDockerContainerImageName(XChainService.XCHAIN_INDEXER, Coin.BITCOIN, btcNetwork) + ":3004" + if (await networks.hasBitcoinIndexer(btcNetwork)) { + defaultValues["BTC_INDEXER_API_URL"] = "http://" + getDockerContainerImageName(XChainService.XCHAIN_INDEXER, Coin.BITCOIN, btcNetwork) + ":3004" } else { - warnHubConfigOnce("BTC_INDEXER_MISSING", - "WARNING: this hub has no BTC indexer (BTC_INDEXER_API_URL is not set in the " + - "host env and this deployment runs no bitcoin " + btcNetwork + " stack), so it cannot read " + - "capability snapshots: every on-chain PRICE batch its indexer parses is recorded invalid " + - "for insufficient signer stake. Install a bitcoin " + btcNetwork + " stack, or set " + - "BTC_INDEXER_API_URL to a reachable BTC indexer.") + networks.warnHubConfigOnce("BTC_INDEXER_MISSING", "WARNING: this hub has no BTC indexer (BTC_INDEXER_API_URL is not set in the host env and this deployment runs no bitcoin " + btcNetwork + " stack), so it cannot read capability snapshots: every on-chain PRICE batch its indexer parses is recorded invalid for insufficient signer stake. Install a bitcoin " + btcNetwork + " stack, or set BTC_INDEXER_API_URL to a reachable BTC indexer.") } } - - // Operator signer for the on-chain DOGE publishers: when the host sets - // XCHAIN_NODE_HUB_SIGNER_DIR, ModuleService mounts that directory - // read-only at /XChainHub/operator-signer and the hub loads - // /signer.js via HUB_SIGNER_MODULE (see xchain-hub - // examples/doge-signer.example.js for the module contract). - if (config.XCHAIN_NODE_HUB_SIGNER_DIR) { - defaultValues["HUB_SIGNER_MODULE"] = "/XChainHub/operator-signer/signer.js" - } - - // Validator mode: when `xchain-node validator init` has been run, inject the - // P2P / signing-key / capability-config env so the hub starts as a full - // validator. Returns {} (no change) for a standalone node, so the standalone - // install path is unaffected. - const { getValidatorEnv, validatorModeReport } = peers.validatorService - Object.assign(defaultValues, getValidatorEnv()) - - // State the resolved mode and the directory it came from: an empty validator - // env means standalone, disabled, or a configDir carrying no validator/, and - // a deploy cannot tell those apart. A statement, never a refusal. - const validator = validatorModeReport() - if (validator.mode === 'validator') { - logger.info("xchain-node: this hub deploys in VALIDATOR mode, from " + validator.dir) - } else if (validator.mode === 'incomplete') { - warnHubConfigOnce("VALIDATOR_STATE_INCOMPLETE", - "WARNING: the validator state under " + validator.dir + " is HALF PRESENT (missing " + - validator.missing.join(", ") + "), so this hub deploys STANDALONE: no P2P_VALIDATOR_ADDR, " + - "no SIGNING_PRIVKEY_HEX, no capability mount, and its anchor publisher will never run. " + - "Half a validator state is never a standalone node, so this is a broken install rather " + - "than a choice: restore the missing file, or point XCHAIN_NODE_CONFIG_DIR at the config " + - "directory that holds the complete set.") - } else if (validator.mode === 'disabled') { - logger.info("xchain-node: this hub deploys STANDALONE because the validator state at " + - validator.dir + " records enabled:false.") - } else { - logger.info("xchain-node: this hub deploys STANDALONE (no validator state under " + - validator.dir + "). If this host IS meant to be a validator, XCHAIN_NODE_CONFIG_DIR is " + - "resolving to the wrong config directory and the real one holds validator/.") - } + networks.configureHubValidator(defaultValues) } - - // Read the config file for this coin/network pair. Non-secret operator overrides live - // in the main config file; runtime RPC credentials live in a separate, untracked - // -.local sidecar so the main file can be diffed/shared without ever - // carrying rpcuser/rpcpassword. const defaultConfig = {} if (coin && network && coin !== "" && network !== "") { - const configFilePath = path.resolve(configDir, `${coin}-${network}`) - if (!configFilePath.startsWith(path.resolve(configDir) + path.sep) && configFilePath !== path.resolve(configDir)) { - throw new Error('Config path traversal detected') - } - const localFilePath = configFilePath + ".local" - - // Track whether the main file still carries credentials so legacy installs can be migrated. + const { configFilePath, localFilePath } = sidecars.coinConfigPaths(coin, network) let mainFileHasCreds = false - - if (!fs.existsSync(configFilePath)) { - logger.warn("Warning: config file not found: " + configFilePath + " (using defaults)") - } else { - const configFileStream = fs.createReadStream(configFilePath) - const rl = readline.createInterface({ input: configFileStream, crlfDelay: Infinity }) - for await (const line of rl) { - const eqIndex = line.indexOf("=") - if (eqIndex > 0) { - const key = line.substring(0, eqIndex) - let value = line.substring(eqIndex + 1) - // Recover RPC credentials glued onto the tail of a preceding setting by - // an older appender that wrote NODE_USER=/NODE_PASSWORD= with no leading - // newline (e.g. `DUST_AMOUNT=546NODE_USER=`). Peel each credential - // off the value tail, password first so a double-glue - // `...NODE_USER=NODE_PASSWORD=

` resolves cleanly, so the real value - // is uncorrupted and mainFileHasCreds arms the migration below, which - // relocates the credential to the sidecar and strips it from this file. - for (const credKey of ["NODE_PASSWORD", "NODE_USER"]) { - const at = value.indexOf(credKey + "=") - if (at >= 0) { - defaultConfig[credKey] = value.substring(at + credKey.length + 1) - value = value.substring(0, at) - mainFileHasCreds = true - } - } - defaultConfig[key] = value - // NODE_SECRET is the redaction-safe spelling of NODE_PASSWORD. - // It arms the same migration: a credential in the MAIN config file gets - // relocated to the sidecar whichever name it arrived under. - if (key === "NODE_USER" || key === "NODE_PASSWORD" || key === "NODE_SECRET") mainFileHasCreds = true - } - } + if (!fs.existsSync(configFilePath)) logger.warn("Warning: config file not found: " + configFilePath + " (using defaults)") + else { const rl = readline.createInterface({ input: fs.createReadStream(configFilePath), crlfDelay: Infinity }) + for await (const line of rl) mainFileHasCreds = sidecars.readMainConfigLine(defaultConfig, line) || mainFileHasCreds } - - // Accept every secret-bearing key under its redaction-safe `*_SECRET` name and - // fold it onto the canonical legacy name here, at the one place config enters - // the process, so nothing downstream (container env, DB provisioner, RPC - // connectors) has to learn a second spelling. Runs BEFORE the migration - // and generation steps below, which key off the canonical names. - // - // PER FILE, not on the merged result. Merging first would make a renamed key in - // the sidecar and the legacy key left behind in the main config file look like - // one file contradicting itself, and the "both names, different values" refusal - // would fire on what is really just the ordinary sidecar-wins precedence. - warnDeprecatedSecretNames(defaultConfig, configFilePath) - foldSecretEnvAliases(defaultConfig) - - // Credentials from the sidecar take precedence over anything in the main file. + sidecars.normalizeMainConfig(defaultConfig, configFilePath) if (fs.existsSync(localFilePath)) { const sidecarConfig = {} - const localStream = fs.createReadStream(localFilePath) - const rlLocal = readline.createInterface({ input: localStream, crlfDelay: Infinity }) - for await (const line of rlLocal) { - const eqIndex = line.indexOf("=") - if (eqIndex > 0) { - sidecarConfig[line.substring(0, eqIndex)] = line.substring(eqIndex + 1) - } - } - warnDeprecatedSecretNames(sidecarConfig, localFilePath) - Object.assign(defaultConfig, foldSecretEnvAliases(sidecarConfig)) - } - - // One-time migration for legacy installs: older versions appended NODE_USER / - // NODE_PASSWORD into the main config file alongside non-secret settings. Move the - // credentials into the sidecar and strip them from the main file so the two never - // share a file again. Existing creds keep working; they are simply relocated. - if (mainFileHasCreds) { - const creds = {} - if ("NODE_USER" in defaultConfig) creds["NODE_USER"] = defaultConfig["NODE_USER"] - if ("NODE_PASSWORD" in defaultConfig) creds["NODE_PASSWORD"] = defaultConfig["NODE_PASSWORD"] - persistSidecarCreds(localFilePath, creds, { overwrite: true }) - const remaining = [] - for (const key in defaultConfig) { - if (key !== "NODE_USER" && key !== "NODE_PASSWORD") remaining.push(`${key}=${defaultConfig[key]}`) - } - fs.writeFileSync(configFilePath, remaining.length ? remaining.join("\n") + "\n" : "") - } - - // Generate and persist to the sidecar whichever RPC credential is missing, - // evaluated PER KEY. The old both-absent (&&) guard meant a partial sidecar (one - // key present, one absent) generated nothing, and the missing half then silently - // resolved to the static "rpc" default via the merge below, leaving a well-known - // default credential on a live stack with no operator signal. - const genRpcCreds = {} - if (!("NODE_USER" in defaultConfig)) { - const nodeUser = crypto.randomBytes(12).toString('hex') - defaultConfig["NODE_USER"] = nodeUser - genRpcCreds["NODE_USER"] = nodeUser - } - if (!("NODE_PASSWORD" in defaultConfig)) { - const nodePassword = crypto.randomBytes(24).toString('hex') - defaultConfig["NODE_PASSWORD"] = nodePassword - genRpcCreds["NODE_PASSWORD"] = nodePassword - } - if (Object.keys(genRpcCreds).length) persistSidecarCreds(localFilePath, genRpcCreds) - - // Generate and persist a per-install password for each per-coin/network DB account - // (decoder, indexer) on first provision, so installs no longer share the static - // default. An operator override in the main config file or the sidecar wins (the merge - // above already loaded those into defaultConfig). Only generate where the next provision - // can rotate the live account to the new value (EXTERNAL_DB or a DB container); on a - // native, non-container host the rotation no-ops, so generating would desync the sidecar - // from the DB and lock the service out (the 2026-06-26 indexer outage). There these fall - // through to the static default in defaultValues, matching the un-rotatable live account. - const freshDbCreds = {} - if (await dbPasswordCanRotate()) { - for (const k of ["DECODER_DB_PASS", "INDEXER_DB_PASS"]) { - if (!(k in defaultConfig)) { - const val = crypto.randomBytes(24).toString('hex') - defaultConfig[k] = val - freshDbCreds[k] = val - } - } - } - if (Object.keys(freshDbCreds).length) upsertSidecarValues(localFilePath, freshDbCreds) - - // The indexer's hub-DB connection reuses its OWN DB account (HUB_DB_NAME/USER are set - // to the indexer's in the indexer block above, on every network including regtest - // since the regtest mirror is armed too), so its hub-DB password must be the - // INDEXER_DB_PASS the container will actually get, not the shared hub password. Set - // it here, before the shared HUB_DB_PASS fallback below, so that fallback sees the - // key already present and skips. An operator override (already in defaultConfig) - // wins. On the non-rotatable path (dbPasswordCanRotate() false, the 2026-06-26 - // outage fallback) INDEXER_DB_PASS is still absent here and only lands via the - // static-defaults merge below; mirror that same static default instead of copying - // `undefined`, which would both mismatch the account AND occupy the key so the - // fallback/merge never repaired it (HubDbSync ER_ACCESS_DENIED lockout, #2246). Not - // network-gated: leaving regtest out here while HUB_DB_NAME/USER above point at the - // indexer's own account would hand the armed mirror the WRONG password (the shared - // hub password against the indexer's own DB user), so the mirror this row arms would - // never actually connect. - if (module === XChainService.XCHAIN_INDEXER && !("HUB_DB_PASS" in defaultConfig)) { - defaultConfig["HUB_DB_PASS"] = defaultConfig["INDEXER_DB_PASS"] !== undefined - ? defaultConfig["INDEXER_DB_PASS"] - : defaultValues["INDEXER_DB_PASS"] - } + const rlLocal = readline.createInterface({ input: fs.createReadStream(localFilePath), crlfDelay: Infinity }) + for await (const line of rlLocal) sidecars.readSidecarConfigLine(sidecarConfig, line) + sidecars.mergeSidecarConfig(defaultConfig, sidecarConfig, localFilePath) + } + if (mainFileHasCreds) sidecars.migrateMainCredentials(defaultConfig, configFilePath, localFilePath) + sidecars.generateRpcCredentials(defaultConfig, localFilePath) + if (await database.dbPasswordCanRotate()) database.generateDatabaseCredentials(defaultConfig, localFilePath) + database.setIndexerHubPassword(defaultConfig, defaultValues, module) } - - // HUB_DB_PASS is shared across the hub and every coin/network stack (decoder/indexer - // connect to the hub DB). Resolve it from the shared sidecar (generate on first use) - // unless an operator override already supplied it. Applies to both the coin/network - // callers (HUB_DB_PASS in their defaults) and the shared-service hub caller. if (("HUB_DB_PASS" in defaultValues) && !("HUB_DB_PASS" in defaultConfig)) { - defaultConfig["HUB_DB_PASS"] = await getOrCreateHubDbPass() - } - - for (const key in defaultValues) { - if (!(key in defaultConfig)) { - defaultConfig[key] = defaultValues[key] - } + defaultConfig["HUB_DB_PASS"] = await database.getOrCreateHubDbPass() } - - // EXTERNAL_DB: rewrite the DB host/port keys so containerized services - // reach the host-native MariaDB via the bridge gateway instead of the - // docker DNS name "mariadb" (which no longer resolves once the bundled - // container is decommissioned). The DB name/user/pass keys stay as-is; - // those are about the credentials, not the network location. + database.mergeDefaults(defaultConfig, defaultValues) if (EXTERNAL_DB) { - // Resolve the real external host/port via getExternalDbConfig() (env → - // saved credentials.json → prompt) rather than the load-time - // EXTERNAL_DB_HOST/PORT constants, which only reflect env vars or the - // 127.0.0.1:3306 defaults. Otherwise a host/port saved at the first-run - // prompt is ignored and provisioned containers get *_DB_HOST=127.0.0.1 - // (their own loopback), unreachable to the real DB (uuid:52c5b5f1). - // DatabaseService requires this file at load, so it is read through peers. - const { getExternalDbConfig } = peers.databaseService - const extCfg = await getExternalDbConfig() - const extHost = extCfg.host - const extPort = extCfg.port - const dbHostKeys = ['HUB_DB_HOST', 'DECODER_DB_HOST', 'INDEXER_DB_HOST', 'DATABASE_URL'] - const dbPortKeys = ['HUB_DB_PORT', 'DECODER_DB_PORT', 'INDEXER_DB_PORT', 'DATABASE_PORT'] - for (const k of dbHostKeys) { - if (k in defaultConfig) defaultConfig[k] = extHost - } - for (const k of dbPortKeys) { - if (k in defaultConfig) defaultConfig[k] = extPort - } + const externalConfig = await peers.databaseService.getExternalDbConfig() + database.applyExternalDatabase(defaultConfig, externalConfig) } - return defaultConfig } -// Short names operators actually type for the shared services, whose canonical -// names carry an `xchain-` prefix that is easy to omit. 'explorer' already had -// this treatment; the others did not, so `recreate hub` matched no container and -// looked like the hub was simply unsupported. -const SERVICE_ALIASES = { - hub: HUB_MODULE_NAME, - sync: SYNC_MODULE_NAME, - db: DB_MODULE_NAME -} - -function filterCommandParameters(branch, modules, coins, networks) { - const servicesList = {} - let addExplorer = false - - // Callers that bypass resolveArgs (recreate, start/stop/restart, logs) hand - // the operator's raw token straight through, so the alias map has to apply - // here too. - if (modules && SERVICE_ALIASES[modules]) modules = SERVICE_ALIASES[modules] - - if (coins && coins !== "all") { - coins = [coins] - } else { - coins = Object.values(Coin) - } - - if (networks && networks !== "all") { - networks = [networks] - } else { - networks = Object.values(Network) - } - - // Shared services (hub / explorer / db / sync) are registered under a single - // empty coin+network key, not per-coin. A bare `update xchain-hub` must resolve - // to that ""/"" container; otherwise it gets fanned out across real coins where - // it matches nothing and the command silently no-ops. - const SHARED_SERVICES = [HUB_MODULE_NAME, EXPLORER_MODULE_NAME, DB_MODULE_NAME, SYNC_MODULE_NAME] - - if (modules === "all") { - // The coin node leads the per-chain list so `install all` creates it - // before the services that poll it. With the node last, a mainnet - // install created the decoder four and a half hours before the node - // existed (a 151 GiB tracker restore sat between them); the decoder - // spent that time logging ENOTFOUND for a name the network did not - // carry yet, and the node's own initial sync, the slowest step on the - // box, had not even started. Every other command that expands `all` - // (update, start, stop, uninstall) tolerates either order. - modules = [NODE_MODULE_NAME, ...Object.values(XChainService).filter(m => m !== XChainService.XCHAIN_E2E_TEST)] - addExplorer = true - } else if (modules === "explorer") { - addExplorer = true - coins = [] - } else if (SHARED_SERVICES.includes(modules)) { - // Explicitly-named shared service → emit under the empty ""/"" key only. - return { "": { "": [modules] } } - } else if (modules === "node") { - modules = [NODE_MODULE_NAME] - } else if (modules) { - modules = [modules] - } - - if (addExplorer) { - servicesList[""] = { "": [EXPLORER_MODULE_NAME] } - } - - for (const nextCoin of coins) { - if (!(nextCoin in servicesList)) servicesList[nextCoin] = {} - - for (const nextNetwork of networks) { - if (!(nextNetwork in servicesList[nextCoin])) servicesList[nextCoin][nextNetwork] = [] - - for (const nextModule of modules) { - if (!REGTEST_MODULES.includes(nextModule) || nextNetwork === Network.REGTEST) { - servicesList[nextCoin][nextNetwork].push(nextModule) - } - } - } - } - - return servicesList -} - -function resolveArgs(args, { expectBranch = false, defaultBranch = 'master' } = {}) { - let service = 'all', chain = 'all', network = 'all', branch = null - - const knownServices = [ - ...Object.values(XChainService), - NODE_MODULE_NAME, DB_MODULE_NAME, HUB_MODULE_NAME, - EXPLORER_MODULE_NAME, SYNC_MODULE_NAME, 'explorer' - ] - const knownChains = Object.values(Coin) - const knownNetworks = Object.values(Network) - - for (const arg of args) { - if (!arg || arg === 'all') continue - - // 'xchain-node' is the CLI itself, not an installable service. Without this - // guard the loop silently drops it and leaves service='all', so e.g. - // `install master xchain-node` would expand to EVERY service. Fail loudly. - if (arg === 'xchain-node') { - throw new Error("'xchain-node' is the CLI itself, not an installable service. Omit it to operate on all services, or name a specific one (e.g. xchain-indexer, xchain-decoder, xchain-hub).") - } - - const aliased = SERVICE_ALIASES[arg] - - if (knownChains.includes(arg)) { - chain = arg - } else if (knownNetworks.includes(arg)) { - network = arg - } else if (knownServices.includes(arg)) { - service = arg - } else if (aliased) { - service = aliased - } else if (expectBranch && !branch) { - branch = arg - } else { - // Nothing claimed this token. The 'xchain-node' guard above exists - // because a dropped token leaves service='all', and that trap is not - // specific to that one name: `install master hub` silently expanded to - // EVERY service on every coin and network rather than refusing. An - // operator reaching for one service must never get all of them. - throw new Error(`Unrecognized argument '${arg}'. Valid services: ` + - knownServices.filter(s => s !== 'explorer').sort().join(', ') + '. ' + - `Valid coins: ${knownChains.join(', ')}. Networks: ${knownNetworks.join(', ')}.`) - } - } - - if (expectBranch && !branch) branch = defaultBranch - - if (branch && !/^[a-zA-Z0-9._\-\/]+$/.test(branch)) { - throw new Error("Invalid branch name: " + branch + " (branch names may only contain letters, numbers, dots, hyphens, underscores, and slashes)") - } - - // Classify the ref slot by shape rather than adding a second CLI field - // (operator-confirmed 2026-08-13, release-management spec section 11). The - // caller decides what to do with it; resolveArgs only reports the shape, so - // this stays a pure function with no network in it. - const { isReleaseRef } = releaseManifestService - - return { service, chain, network, branch, isRelease: isReleaseRef(branch) } -} - module.exports = { getModuleDir, getModuleTmpDir, diff --git a/src/services/config_service/arguments.js b/src/services/config_service/arguments.js new file mode 100644 index 0000000..41961cb --- /dev/null +++ b/src/services/config_service/arguments.js @@ -0,0 +1,79 @@ +/********************************************************************* + * + * Copyright © 2025–2026 Dankest, LLC + * Based on XChain Platform by Dankest, LLC – https://dankest.llc + * + * SPDX-License-Identifier: AGPL-3.0-or-later + * + * This file is part of XChain Platform. Licensed under the GNU Affero + * General Public License v3.0 or later; see LICENSE.md. A commercial + * license (without AGPL source-disclosure terms) is available - + * contact legal@dankest.llc. + * + ********************************************************************** + * Config Service Arguments + ********************************************************************/ + +'use strict' + +let Coin, Network, XChainService, NODE_MODULE_NAME, DB_MODULE_NAME +let HUB_MODULE_NAME, EXPLORER_MODULE_NAME, SYNC_MODULE_NAME, releaseManifestService, serviceAliases + +function configure(dependencies) { + ({ Coin, Network, XChainService, NODE_MODULE_NAME, DB_MODULE_NAME, + HUB_MODULE_NAME, EXPLORER_MODULE_NAME, SYNC_MODULE_NAME, releaseManifestService, serviceAliases } = dependencies) +} + +// 'xchain-node' is the CLI itself, not an installable service. Without this +// guard the loop silently drops it and leaves service='all', so e.g. +// `install master xchain-node` would expand to EVERY service. Fail loudly. +// Nothing claimed this token. The 'xchain-node' guard above exists +// because a dropped token leaves service='all', and that trap is not +// specific to that one name: `install master hub` silently expanded to +// EVERY service on every coin and network rather than refusing. An +// operator reaching for one service must never get all of them. +// Classify the ref slot by shape rather than adding a second CLI field +// (operator-confirmed 2026-08-13, release-management spec section 11). The +// caller decides what to do with it; resolveArgs only reports the shape, so +// this stays a pure function with no network in it. +function resolveArgs(args, { expectBranch = false, defaultBranch = 'master' } = {}) { + const SERVICE_ALIASES = serviceAliases() + let service = 'all', chain = 'all', network = 'all', branch = null + const knownServices = [ + ...Object.values(XChainService), + NODE_MODULE_NAME, DB_MODULE_NAME, HUB_MODULE_NAME, + EXPLORER_MODULE_NAME, SYNC_MODULE_NAME, 'explorer' + ] + const knownChains = Object.values(Coin) + const knownNetworks = Object.values(Network) + for (const arg of args) { + if (!arg || arg === 'all') continue + if (arg === 'xchain-node') { + throw new Error("'xchain-node' is the CLI itself, not an installable service. Omit it to operate on all services, or name a specific one (e.g. xchain-indexer, xchain-decoder, xchain-hub).") + } + const aliased = SERVICE_ALIASES[arg] + if (knownChains.includes(arg)) { + chain = arg + } else if (knownNetworks.includes(arg)) { + network = arg + } else if (knownServices.includes(arg)) { + service = arg + } else if (aliased) { + service = aliased + } else if (expectBranch && !branch) { + branch = arg + } else { + throw new Error(`Unrecognized argument '${arg}'. Valid services: ` + + knownServices.filter(s => s !== 'explorer').sort().join(', ') + '. ' + + `Valid coins: ${knownChains.join(', ')}. Networks: ${knownNetworks.join(', ')}.`) + } + } + if (expectBranch && !branch) branch = defaultBranch + if (branch && !/^[a-zA-Z0-9._\-\/]+$/.test(branch)) { + throw new Error("Invalid branch name: " + branch + " (branch names may only contain letters, numbers, dots, hyphens, underscores, and slashes)") + } + const { isReleaseRef } = releaseManifestService + return { service, chain, network, branch, isRelease: isReleaseRef(branch) } +} + +module.exports = { configure, resolveArgs } diff --git a/src/services/config_service/coins.js b/src/services/config_service/coins.js new file mode 100644 index 0000000..dcce77c --- /dev/null +++ b/src/services/config_service/coins.js @@ -0,0 +1,123 @@ +/********************************************************************* + * + * Copyright © 2025–2026 Dankest, LLC + * Based on XChain Platform by Dankest, LLC – https://dankest.llc + * + * SPDX-License-Identifier: AGPL-3.0-or-later + * + * This file is part of XChain Platform. Licensed under the GNU Affero + * General Public License v3.0 or later; see LICENSE.md. A commercial + * license (without AGPL source-disclosure terms) is available - + * contact legal@dankest.llc. + * + ********************************************************************** + * Config Service Coin Defaults + ********************************************************************/ + +'use strict' + +let config, CoinTickerSymbol, Network, XChainService, getCoinConfigByFullName, getDockerContainerImageName + +function configure(dependencies) { + ({ config, CoinTickerSymbol, Network, XChainService, getCoinConfigByFullName, getDockerContainerImageName } = dependencies) +} + +// Encoder's express-rate-limit defaults to 60 RPM, which the e2e +// suite blows past whenever the stale-UTXO retry shim fires (up +// to 15 retries per failing tx, easily 100+ RPM during the +// order/swap blocks). Production-safe defaults stay at 60; we +// raise it for regtest where load is by-design bursty. +// The browser wallet calls the encoder cross-origin (create_tx, +// ping). The encoder disables CORS unless CORS_ORIGIN is set, so a fresh +// regtest stack blocks every browser request and the wallet reports the +// chain "degraded". regtest is a local single-operator dev venue (same +// reasoning as INDEXER_ALLOW_UNAUTHENTICATED below), so default it open; +// a host CORS_ORIGIN or config-file value still wins. mainnet/testnet keep +// the fail-safe default (CORS off unless the operator opts in). +// Encoder passthrough. A production encoder sits behind a reverse proxy on +// ANOTHER box, reached over a public address, so its default trust-proxy +// setting (loopback, uniquelocal) never honours X-Forwarded-For and the +// per-IP limiter keys every visitor on the proxy's egress address: one +// bucket per encoder for the whole world. ENCODER_TRUST_PROXY names that +// egress address so the container recovers the real client. +// ENCODER_RATE_LIMIT_RPM rides the same passthrough, placed after the +// regtest block above so a host value wins over the 99999 regtest literal +// and survives update/recreate. Read BY NAME, same as the explorer +// passthrough below: a computed process.env read is invisible to the +// platform's env-var coverage gate, which is what turns an undocumented +// variable into a silent one. +// Native-coin protocol fee destination (per coin/network). Defaults from the vendored +// canonical coin registry (src/coins), so a stock install provisions the decoder's +// FEE_DESTINATION (fee-output capture into transaction_outputs) and the indexer's +// XCHAIN_FEE_DESTINATION__ without any operator env. Previously these +// were host-env-only, so default installs left the decoder capturing nothing and +// native-fee validation failing closed on every fee-bearing LTC/DOGE action (audit +// F-11). getCoinConfigByFullName applies the per-coin host-env override itself +// (ignored + warned on mainnet: fee acceptance is consensus and must not depend on +// operator env); the generic FEE_DESTINATION host var is honored on non-mainnet only. +// Injected as a default, so a value in the - config file still wins. +// Address-deriving modules (encoder/decoder/utxo-tracker) resolve their bitcoinjs +// network via CryptoNetworks, which keys on the COIN-PREFIXED name (e.g. +// "bitcoin-regtest"). A bare network ("regtest") matches no case → getBitcoinJsNetwork +// returns undefined → bitcoinjs-lib falls back to MAINNET → addresses derive with +// mainnet version bytes (a regtest "m..."/"n..." source comes out as "1..."), which then +// never matches on-chain balances (e.g. an issuance fee-check reads the wrong address and +// fails "insufficient funds (FEE)"). Those modules need the coin-prefixed network; the +// indexer/node keep the bare network (protocol-change matching / bitcoind conf). +// LevelDB tuning passthrough (xchain-utxo-tracker only). src/store/level_up_db.js reads +// LEVELDB_CACHE_BYTES (documented default 4 GiB, components/utxo-tracker/configuration.md:41) +// and LEVELDB_WRITE_BUFFER_BYTES from process.env inside the container, but +// getDefaultConfig never forwarded either host var into the tracker's +// defaultValues, so an operator exporting LEVELDB_CACHE_BYTES before +// install/update got silence: the container never saw it and LevelUpDb fell +// back to its in-container default every time. Mirrors hubPassthroughVars / +// genesisPassthroughVars: only set, non-empty host vars are injected, so an +// unset env leaves the tracker's own default untouched. +// Read by name rather than through a loop: the env-var doc-coverage gate can +// only see a variable it can name, and a computed process.env[varName] read +// widens its blind spot. +function applyCoinDefaults(defaultValues, module, coin, network) { + if (network === "regtest") { + defaultValues["REGTEST_MINER_URL"] = getDockerContainerImageName(XChainService.XCHAIN_REGTEST_MINER, coin, network) + defaultValues["REGTEST_MINER_API_PORT"] = 3005 + defaultValues["REGTEST_MINER_PORT"] = 3005 + defaultValues["ENCODER_RATE_LIMIT_RPM"] = 99999 + defaultValues["CORS_ORIGIN"] = config.CORS_ORIGIN || "*" + } + if (module === XChainService.XCHAIN_ENCODER) { + const encoderPassthroughVars = ["ENCODER_TRUST_PROXY", "ENCODER_RATE_LIMIT_RPM"] + for (const key of encoderPassthroughVars) { + const value = { + ENCODER_TRUST_PROXY: config.ENCODER_TRUST_PROXY, + ENCODER_RATE_LIMIT_RPM: config.ENCODER_RATE_LIMIT_RPM + }[key] + if (value === undefined || value === "") continue + defaultValues[key] = value + } + } + const feeDestEnvName = 'XCHAIN_FEE_DESTINATION_' + CoinTickerSymbol[coin] + '_' + network.toUpperCase() + const registryFeeDestination = getCoinConfigByFullName(coin, network).addresses.FEE_DESTINATION + const feeDestination = (network !== Network.MAINNET && !config.FEE_DESTINATION_ENV[feeDestEnvName] && config.FEE_DESTINATION) + ? config.FEE_DESTINATION + : registryFeeDestination + if (feeDestination) { + defaultValues['FEE_DESTINATION'] = feeDestination + defaultValues[feeDestEnvName] = feeDestination + } + if (module === XChainService.XCHAIN_ENCODER || module === XChainService.XCHAIN_DECODER || module === XChainService.XCHAIN_UTXO_TRACKER) { + defaultValues["NETWORK"] = coin + "-" + network + } + if (module === XChainService.XCHAIN_UTXO_TRACKER) { + const levelDbPassthrough = { + LEVELDB_CACHE_BYTES: config.LEVELDB_CACHE_BYTES, + LEVELDB_WRITE_BUFFER_BYTES: config.LEVELDB_WRITE_BUFFER_BYTES + } + for (const [varName, value] of Object.entries(levelDbPassthrough)) { + if (value !== undefined && value !== "") { + defaultValues[varName] = value + } + } + } +} + +module.exports = { configure, applyCoinDefaults } diff --git a/src/services/config_service/database.js b/src/services/config_service/database.js new file mode 100644 index 0000000..b1fb6ba --- /dev/null +++ b/src/services/config_service/database.js @@ -0,0 +1,142 @@ +/********************************************************************* + * + * Copyright © 2025–2026 Dankest, LLC + * Based on XChain Platform by Dankest, LLC – https://dankest.llc + * + * SPDX-License-Identifier: AGPL-3.0-or-later + * + * This file is part of XChain Platform. Licensed under the GNU Affero + * General Public License v3.0 or later; see LICENSE.md. A commercial + * license (without AGPL source-disclosure terms) is available - + * contact legal@dankest.llc. + * + ********************************************************************** + * Config Service Database Defaults + ********************************************************************/ + +'use strict' + +let crypto, EXTERNAL_DB, SEP, XChainService, peers, sidecars + +function configure(dependencies) { + ({ crypto, EXTERNAL_DB, SEP, XChainService, peers, sidecars } = dependencies) +} + +// Whether a per-install random DB password can actually be APPLIED to the live MariaDB +// account on the next provision. Rotation runs either via the external-DB path (EXTERNAL_DB) +// or by exec-ing into a local MariaDB container; on a native, non-container host neither +// runs, so a generated password would never reach the DB and would desync the sidecar from a +// DB still on the old password (the 2026-06-26 indexer outage). Returns false on any error +// (e.g. docker absent), the safe direction: prefer the static default over a password we +// cannot apply. DatabaseService requires this file at load, so it is read through peers. +async function dbPasswordCanRotate() { + if (EXTERNAL_DB) return true + try { + const { getDatabaseContainerId } = peers.databaseService + return !!(await getDatabaseContainerId()) + } catch { + return false + } +} + +// HUB_DB_PASS is a SHARED-service credential: the hub and every coin/network stack that +// connects to the hub DB must present the SAME password, so it cannot be generated per +// coin/network. Resolve it once from a shared 0600 sidecar (config/hub.local), generating +// and persisting it on first use. getDefaultConfig() calls are sequential in the installer, +// so the read-or-generate is not racy in practice. +async function getOrCreateHubDbPass() { + const hubLocalPath = sidecars.hubSidecarPath() + let pass = await sidecars.readSidecarValue(hubLocalPath, "HUB_DB_PASS") + if (!pass) { + // Only mint a random shared password where the rotation can apply it; otherwise use + // the static default so the sidecar never diverges from a DB the rotation cannot reach. + if (await dbPasswordCanRotate()) { + pass = crypto.randomBytes(24).toString('hex') + sidecars.upsertSidecarValues(hubLocalPath, { HUB_DB_PASS: pass }) + } else { + pass = "xchain" + SEP + "password" + } + } + return pass +} + +// Generate and persist a per-install password for each per-coin/network DB account +// (decoder, indexer) on first provision, so installs no longer share the static +// default. An operator override in the main config file or the sidecar wins (the merge +// above already loaded those into defaultConfig). Only generate where the next provision +// can rotate the live account to the new value (EXTERNAL_DB or a DB container); on a +// native, non-container host the rotation no-ops, so generating would desync the sidecar +// from the DB and lock the service out (the 2026-06-26 indexer outage). There these fall +// through to the static default in defaultValues, matching the un-rotatable live account. +// The indexer's hub-DB connection reuses its OWN DB account (HUB_DB_NAME/USER are set +// to the indexer's in the indexer block above, on every network including regtest +// since the regtest mirror is armed too), so its hub-DB password must be the +// INDEXER_DB_PASS the container will actually get, not the shared hub password. Set +// it here, before the shared HUB_DB_PASS fallback below, so that fallback sees the +// key already present and skips. An operator override (already in defaultConfig) +// wins. On the non-rotatable path (dbPasswordCanRotate() false, the 2026-06-26 +// outage fallback) INDEXER_DB_PASS is still absent here and only lands via the +// static-defaults merge below; mirror that same static default instead of copying +// `undefined`, which would both mismatch the account AND occupy the key so the +// fallback/merge never repaired it (HubDbSync ER_ACCESS_DENIED lockout, #2246). Not +// network-gated: leaving regtest out here while HUB_DB_NAME/USER above point at the +// indexer's own account would hand the armed mirror the WRONG password (the shared +// hub password against the indexer's own DB user), so the mirror this row arms would +// never actually connect. +// HUB_DB_PASS is shared across the hub and every coin/network stack (decoder/indexer +// connect to the hub DB). Resolve it from the shared sidecar (generate on first use) +// unless an operator override already supplied it. Applies to both the coin/network +// callers (HUB_DB_PASS in their defaults) and the shared-service hub caller. +// EXTERNAL_DB: rewrite the DB host/port keys so containerized services +// reach the host-native MariaDB via the bridge gateway instead of the +// docker DNS name "mariadb" (which no longer resolves once the bundled +// container is decommissioned). The DB name/user/pass keys stay as-is; +// those are about the credentials, not the network location. +// Resolve the real external host/port via getExternalDbConfig() (env → +// saved credentials.json → prompt) rather than the load-time +// EXTERNAL_DB_HOST/PORT constants, which only reflect env vars or the +// 127.0.0.1:3306 defaults. Otherwise a host/port saved at the first-run +// prompt is ignored and provisioned containers get *_DB_HOST=127.0.0.1 +// (their own loopback), unreachable to the real DB (uuid:52c5b5f1). +// DatabaseService requires this file at load, so it is read through peers. +function generateDatabaseCredentials(defaultConfig, localFilePath) { + const freshDbCreds = {} + for (const key of ["DECODER_DB_PASS", "INDEXER_DB_PASS"]) { + if (!(key in defaultConfig)) { + const value = crypto.randomBytes(24).toString('hex') + defaultConfig[key] = value + freshDbCreds[key] = value + } + } + if (Object.keys(freshDbCreds).length) sidecars.upsertSidecarValues(localFilePath, freshDbCreds) +} + +function setIndexerHubPassword(defaultConfig, defaultValues, module) { + if (module === XChainService.XCHAIN_INDEXER && !("HUB_DB_PASS" in defaultConfig)) { + defaultConfig["HUB_DB_PASS"] = defaultConfig["INDEXER_DB_PASS"] !== undefined + ? defaultConfig["INDEXER_DB_PASS"] + : defaultValues["INDEXER_DB_PASS"] + } +} + +function mergeDefaults(defaultConfig, defaultValues) { + for (const key in defaultValues) { + if (!(key in defaultConfig)) defaultConfig[key] = defaultValues[key] + } +} + +function applyExternalDatabase(defaultConfig, externalConfig) { + const dbHostKeys = ['HUB_DB_HOST', 'DECODER_DB_HOST', 'INDEXER_DB_HOST', 'DATABASE_URL'] + const dbPortKeys = ['HUB_DB_PORT', 'DECODER_DB_PORT', 'INDEXER_DB_PORT', 'DATABASE_PORT'] + for (const key of dbHostKeys) { + if (key in defaultConfig) defaultConfig[key] = externalConfig.host + } + for (const key of dbPortKeys) { + if (key in defaultConfig) defaultConfig[key] = externalConfig.port + } +} + +module.exports = { + configure, dbPasswordCanRotate, getOrCreateHubDbPass, generateDatabaseCredentials, + setIndexerHubPassword, mergeDefaults, applyExternalDatabase +} diff --git a/src/services/config_service/defaults.js b/src/services/config_service/defaults.js new file mode 100644 index 0000000..6b93e25 --- /dev/null +++ b/src/services/config_service/defaults.js @@ -0,0 +1,343 @@ +/********************************************************************* + * + * Copyright © 2025–2026 Dankest, LLC + * Based on XChain Platform by Dankest, LLC – https://dankest.llc + * + * SPDX-License-Identifier: AGPL-3.0-or-later + * + * This file is part of XChain Platform. Licensed under the GNU Affero + * General Public License v3.0 or later; see LICENSE.md. A commercial + * license (without AGPL source-disclosure terms) is available - + * contact legal@dankest.llc. + * + ********************************************************************** + * Config Service Defaults + ********************************************************************/ + +'use strict' + +let config, Network, Coin, CoinTickerSymbol, XChainService, DB_SEP, SEP, bootstrapDir +let NODE_MODULE_NAME, HUB_MODULE_NAME, EXPLORER_MODULE_NAME, SYNC_MODULE_NAME +let getDockerContainerImageName, getModuleDatabaseName + +function configure(dependencies) { + ({ config, Network, Coin, CoinTickerSymbol, XChainService, DB_SEP, SEP, bootstrapDir, + NODE_MODULE_NAME, HUB_MODULE_NAME, EXPLORER_MODULE_NAME, SYNC_MODULE_NAME, + getDockerContainerImageName, getModuleDatabaseName } = dependencies) +} + +// The coin node's CONTAINER NAME, never the bare `node` alias every coin +// node also carries. The indexer joins its sibling coins' networks for +// cross-chain reads (ModuleService.crossChainNetworksFor) and the hub +// joins every stack, so from either container docker DNS answers `node` +// with whichever sibling network sorts first (bitcoin), and a dogecoin +// stack's RPC credentials then hit the bitcoin node: HTTP 401 by name, +// 200 by IP. Measured on regtest 2026-09-11 and reported by a testnet +// operator the same day. The container name resolves on any shared +// network and is unique per coin/network, like every other *_URL here. +// The e2e-test harness reaches the indexer DB via DATABASE_URL/DATABASE_PORT +// (test/initialCheck.test.js), not INDEXER_DB_HOST/PORT. Default them here so +// the EXTERNAL_DB rewrite below can repoint them; on a host-native-DB box the +// docker DNS name "mariadb" doesn't resolve and the suite fails at bootstrap. +// The indexer's hub client keys ENTIRELY off HUB_API_URL (hub_client.js: +// `this.enabled = !!this.hubUrl`). HUB_API_HOST above is set but read by +// nothing in xchain-indexer, so without this the client stayed disabled on +// every installed stack and no push ever left the indexer: chain tips, +// config, and in particular the PRICE v1 oracle_price pushes that a FIAT +// dispenser later prices against. Prod sets HUB_API_URL by hand in the +// per-coin config file, which is why this went unnoticed. +// +// Composed from the same container name + port as HUB_API_HOST/HUB_PORT, so +// it resolves on the docker network exactly as the sibling *_API_HOST vars +// do. An operator config-file value still wins, so prod's explicit +// cross-host URL is unaffected. +// +// This enables the push half only. The read-back half (hub tables mirrored +// into the indexer) additionally needs HUB_DB_NAME + HUB_DB_SYNC_ENABLED, +// which regtest deliberately leaves unset (see the network !== "regtest" +// block below), so turning this on cannot start a mirror bootstrap and +// cannot re-page price_snapshots out from under a seeded test venue. +// The e2e federation suites (test:federation / test:attestation:llm) boot +// in-process MultiValidatorHubs that create + drop XChain___MVH_* +// databases, so the container needs the hub DB credentials. DatabaseService +// grants this user CREATE/DROP on the XChain_%_MVH_% pattern when the hub +// module is installed. Omitting these made requireFederationEnv loud-fail. +// Explorer is a SHARED service (no coin/network suffix). Like the hub +// above, the e2e-test container needs to reach it on the docker network; +// omitting it left EXPLORER_URL/EXPLORER_API_PORT unset, which failed the +// e2e harness's checkAllEnvironmentalVariables() → broken hub-config fallback. +function createCoinDefaults(module, coin, network) { + return { + "NETWORK": network, + "NODE_URL": getDockerContainerImageName(NODE_MODULE_NAME, coin, network), + "NODE_PORT": (network === Network.MAINNET ? 8332 : (network === Network.TESTNET ? 18332 : 18444)), + "NODE_USER": "rpc", + "NODE_PASSWORD": "rpc", + "UTXO_TRACKER_URL": getDockerContainerImageName(XChainService.XCHAIN_UTXO_TRACKER, coin, network), + "UTXO_TRACKER_API_PORT": 3001, + "UTXO_TRACKER_PORT": 3001, + "UTXO_TRACKER_BOOTSTRAP_VOLUME": bootstrapDir + "/" + coin + "/" + network + "/" + module + "/bootstrap/", + "DECODER_DB_NAME": getModuleDatabaseName(XChainService.XCHAIN_DECODER, coin, network), + "DECODER_DB_HOST": "mariadb", + "DECODER_DB_PORT": 3306, + "DECODER_DB_USER": "xchain" + DB_SEP + "decoder" + DB_SEP + coin + DB_SEP + network, + "DECODER_DB_PASS": "xchain" + SEP + "password", + "DECODER_URL": getDockerContainerImageName(XChainService.XCHAIN_DECODER, coin, network), + "DECODER_API_PORT": 3002, + "DECODER_PORT": 3002, + "DECODER_BOOTSTRAP_VOLUME": bootstrapDir + "/" + coin + "/" + network + "/" + module + "/bootstrap/", + "INDEXER_BOOTSTRAP_VOLUME": bootstrapDir + "/" + coin + "/" + network + "/" + module + "/bootstrap/", + "ENCODER_URL": getDockerContainerImageName(XChainService.XCHAIN_ENCODER, coin, network), + "ENCODER_API_PORT": 3003, + "ENCODER_PORT": 3003, + "INDEXER_URL": getDockerContainerImageName(XChainService.XCHAIN_INDEXER, coin, network), + "INDEXER_API_PORT": 3004, + "INDEXER_PORT": 3004, + "INDEXER_COIN": CoinTickerSymbol[coin], + "INDEXER_NETWORK": network, + "INDEXER_DB_HOST": "mariadb", + "INDEXER_DB_PORT": 3306, + "INDEXER_DB_NAME": getModuleDatabaseName(XChainService.XCHAIN_INDEXER, coin, network), + "INDEXER_DB_USER": "xchain" + DB_SEP + "indexer" + DB_SEP + coin + DB_SEP + network, + "INDEXER_DB_PASS": "xchain" + SEP + "password", + "DATABASE_URL": "mariadb", + "DATABASE_PORT": 3306, + "HUB_HOST": "0.0.0.0", + "HUB_API_HOST": getDockerContainerImageName(HUB_MODULE_NAME, "", ""), + "HUB_PORT": 10000, + "HUB_API_URL": "http://" + getDockerContainerImageName(HUB_MODULE_NAME, "", "") + ":10000", + "HUB_DB_HOST": "mariadb", + "HUB_DB_PORT": 3306, + "HUB_DB_USER": "xchain" + DB_SEP + "hub", + "HUB_DB_PASS": "xchain" + SEP + "password", + "EXPLORER_URL": getDockerContainerImageName(EXPLORER_MODULE_NAME, "", ""), + "EXPLORER_API_PORT": 8080, + "EXPLORER_PORT": 8080 + } +} + + +function createSharedDefaults() { + return { + "HUB_HOST": "0.0.0.0", + "HUB_API_HOST": getDockerContainerImageName(HUB_MODULE_NAME, "", ""), + "HUB_PORT": 10000, + "HUB_DB_HOST": "mariadb", + "HUB_DB_PORT": 3306, + "HUB_DB_NAME": "XChain" + DB_SEP + "Hub", + "HUB_DB_USER": "xchain" + DB_SEP + "hub", + "HUB_DB_PASS": "xchain" + SEP + "password", + "EXPLORER_HOST": "127.0.0.1", + "EXPLORER_PORT": 18080, + "EXPLORER_API_HOST": getDockerContainerImageName(EXPLORER_MODULE_NAME, "", ""), + "EXPLORER_API_USER": false, + "EXPLORER_API_PASS": false, + "EXPLORER_API_PORT_HTTP": 8080, + "EXPLORER_PORT_HTTP": 18080, + "EXPLORER_API_PORT_HTTPS": 8081, + "EXPLORER_PORT_HTTPS": 18081, + "SYNC_MODE": "server", + "SYNC_API_PORT": 3006, + "SYNC_PORT": 3006, + "SYNC_API_HOST": getDockerContainerImageName(SYNC_MODULE_NAME, "", ""), + "HUB_API_HOST_SYNC": getDockerContainerImageName(HUB_MODULE_NAME, "", "") + } +} + +// Allow the operator to override the shared hub's port via host env. +// The hub has no per-coin config file, so host env is the injection point, same +// as the explorer override below. constants.js's HUB_MODULE_NAME.docker.ports +// entry maps BOTH the published host port and the container-internal port from +// this one HUB_PORT value (`-p ${HUB_PORT}:${HUB_PORT}`), and every hub client +// in this file (buildCheckpointConfig, updateHubOrExplorer, ...) reads the +// computed defaultValues.HUB_PORT rather than the constants.js default, so +// overriding it here keeps the published port, the container's own listener, +// and every in-process caller in agreement. Motivating case: a second +// co-located xchain-node install (e.g. verifying `install master xchain-hub` +// boots correctly) needs its hub reachable on a host port distinct from a +// standing shared hub's 10000, which had no override at all and so could only +// be tested by tearing the shared hub down or standing up a whole separate +// Docker daemon. +// Allow the operator to override the explorer's published HOST ports via host +// env. Shared services (explorer/hub) have no per-coin config file, so host env +// is the injection point (same pattern as the hub passthrough vars below). The +// motivating case: a second co-located xchain-node install (e.g. a federation +// stack alongside the primary node) must publish the explorer on a non-default +// port to avoid colliding with the primary's 18080/18081. Container-internal +// ports (EXPLORER_API_PORT_HTTP/HTTPS) are unchanged. +// The explorer hard-requires a co-located hub-mirror DB (state_checkpoints / +// capability_snapshots / cross_chain_matches) per serving coin, since +// xchain-sync never replicates those tables. Regtest and dev stacks don't run +// that replication, so the explorer would crash-loop on startup there. Let the +// operator opt out via host env (ALLOW_NO_COLOCATED_HUB_DB=1): the hub-mirrored +// endpoints then fail loud per-request instead of blocking startup. Unset on +// mainnet/testnet so the missing-DB guard still catches a real misconfiguration. +// Shared services that call the hub as clients (the sync server's config +// discovery via getallconfigs, the explorer's hub reads) must present the +// hub's API key once the hub enforces its sensitive-read tier: getallconfigs +// 401s keyless when HUB_API_KEY is set hub-side. Sourced from host env so it +// persists across `update`, then from the shared hub sidecar, mirroring the +// indexer's passthrough above. Neither set keeps the prior keyless behavior +// (fine against a keyless hub). +function configureSharedBeforeHubKey(defaultValues) { + if (config.HUB_PORT_OVERRIDE !== undefined && config.HUB_PORT_OVERRIDE !== "") { + defaultValues.HUB_PORT = config.HUB_PORT_OVERRIDE + } + for (const k of ["EXPLORER_PORT_HTTP", "EXPLORER_PORT_HTTPS", "EXPLORER_PORT"]) { + if (config.EXPLORER_PORT_ENV[k] !== undefined && config.EXPLORER_PORT_ENV[k] !== "") { + defaultValues[k] = config.EXPLORER_PORT_ENV[k] + } + } + if (config.ALLOW_NO_COLOCATED_HUB_DB !== undefined && config.ALLOW_NO_COLOCATED_HUB_DB !== "") { + defaultValues.ALLOW_NO_COLOCATED_HUB_DB = config.ALLOW_NO_COLOCATED_HUB_DB + } + if (config.HUB_API_KEY !== undefined && config.HUB_API_KEY !== "") { + defaultValues.HUB_API_KEY = config.HUB_API_KEY + } +} + +// The browser wallet calls the hub cross-origin (ping, config reads). +// The hub disables CORS unless CORS_ORIGIN is set, so a browser wallet is +// blocked and reports the chain "degraded". The hub is a shared, network- +// agnostic service (it may front mainnet), so unlike the per-network encoder +// above it is NOT auto-defaulted open: the operator opts in via host env. +// On a local regtest dev box set CORS_ORIGIN=* when installing the hub. +// Self-synced hub-mirror checkpoint schema (row 39, #4138 decoupling): +// HubMirrorSyncManager needs the hub's own REST base URL to pull +// state_checkpoints / capability_snapshots / cross_chain_matches, which +// is a DIFFERENT thing from HUB_API_HOST/HUB_PORT above (those feed the +// explorer's ordinary getallconfigs config poll, not the mirror writer). +// Opt-in via host env EXPLORER_CHECKPOINT_SELF_SYNC, read directly by +// HubService.buildHubModuleConfig's checkpoint injection (see there for +// why this stays a second knob instead of piggybacking +// ALLOW_NO_COLOCATED_HUB_DB: that flag only downgrades the fatal +// startup assertion to a warning and says nothing about whether a +// local mirror should be provisioned). Emitted only when opted in, so +// a deployment that never uses self-sync carries no unused hub URL. +// +// This env is no longer the mirror writer's ONLY source: the same URL +// now ships inside the checkpoint config block beside self_sync +// (HubService.buildCheckpointConfig), because these two lived on +// different delivery paths - container env written at install time +// versus the hub's config push - and opting in after the container +// existed left the explorer self-syncing with nowhere to sync from. +// Kept for the explorer's other hub reads (HubOperationalCache) and as +// the fallback for hand-written config.json deployments. +// Read Contract simulation (contract.html #contract-read-card) is +// default-off; the readers test for the exact STRING 'true', not any +// truthy value, so pass the host env through verbatim rather than +// coercing it. Sourced from host env so it persists across `update`/ +// `recreate`, mirroring the other explorer passthroughs here. +// Serving limits, same host-env injection point as the knobs above, +// because every one of these defaults is tuned for a PUBLIC explorer +// and is wrong for a private venue: +// EXPLORER_*RATE_LIMIT_RPM - the nine request budgets, per IP: the +// app-wide cap, the quote/pre-flight caps, and the six per-route +// caps (checkpoint-list, checkpoint-verify, action-proof, +// validator-set-proof, vm-query, batch). A dev box reaches the explorer +// through one tunnel, so every browser and every test run shares a +// single bucket, and a browser-driven suite sustains far more than +// any one of these caps on its own. All nine are now reachable +// from the host env; the five per-route caps were unreachable on a +// node-managed explorer (the regtest venue), which could raise only +// the app-wide and fee-quote caps before this change. +// EXPLORER_TIP_MAX_AGE_S - 6h by default, and 0 disables it. A +// regtest chain has no block cadence: it advances only when someone +// mines, so an idle one crosses the age gate and the explorer delists +// a chain that is perfectly healthy (503 COIN_DATA_STALE on every +// read, lag 0 on /status). +// Read BY NAME rather than by scanning process.env for a pattern: a +// computed read is invisible to the platform's env-var coverage gate, +// which is what turns an undocumented variable into a silent one. The +// per-coin EXPLORER_TIP_MAX_AGE_S_ form is deliberately NOT +// carried here - the explorer honours it directly, and the global knob +// already covers the case this passthrough exists for (an instance +// serving nothing but a private venue). +// The explorer resolves each coin's utxo-tracker and decoder from +// UTXO_TRACKER_URL_ (e.g. UTXO_TRACKER_URL_RBTC) and +// DECODER_API_URL__ (e.g. DECODER_API_URL_BTC_REGTEST). +// The explorer is a shared service with no per-venue config file, so these +// were never emitted anywhere: every coin reported tracker_available:false +// and decoder_health 'unconfigured', which blanks address balances/UTXOs +// (the wallet's balance source). Emit the pair for every coin/network at +// the venue containers' internal ports; entries for venues not installed +// on this host are inert because the explorer only probes coins it serves. +// Same omission, one service later. The explorer's +// quote/pre-flight proxies (/api/preflight, /api/feequote, +// /api/oraclefeequote) resolve their upstream from +// INDEXER_API_URL__, which was never emitted +// here, so every one of them answered INDEXER_NOT_CONFIGURED +// on a container install. That silently reduced the wallet's +// pre-flight to its client-side tier: the confirm surface +// still renders a verdict, just never the indexer's own. +// +// Unlike the two above, this one yields to the host env. The +// explorer is also run natively (systemd) on hosts where the +// indexers live on OTHER boxes, and those set this var by +// hand to a remote address; a container-local +// default that overrode it would point a working production +// explorer at a hostname that does not resolve. +function configureSharedAfterHubKey(defaultValues, module) { + if (config.CORS_ORIGIN !== undefined && config.CORS_ORIGIN !== "") { + defaultValues.CORS_ORIGIN = config.CORS_ORIGIN + } + if (module === EXPLORER_MODULE_NAME) { + if (config.EXPLORER_CHECKPOINT_SELF_SYNC !== undefined && config.EXPLORER_CHECKPOINT_SELF_SYNC !== "") { + defaultValues.HUB_API_URL = config.HUB_API_URL || + ("http://" + getDockerContainerImageName(HUB_MODULE_NAME, "", "") + ":" + defaultValues.HUB_PORT) + } + if (config.EXPLORER_VM_QUERY_ENABLED !== undefined && config.EXPLORER_VM_QUERY_ENABLED !== "") { + defaultValues.EXPLORER_VM_QUERY_ENABLED = config.EXPLORER_VM_QUERY_ENABLED + } + for (const key of [ + "EXPLORER_RATE_LIMIT_RPM", + "EXPLORER_FEE_QUOTE_RATE_LIMIT_RPM", + "EXPLORER_PREFLIGHT_POST_RATE_LIMIT_RPM", + "EXPLORER_TIP_MAX_AGE_S", + "EXPLORER_CHECKPOINT_LIST_RATE_LIMIT_RPM", + "EXPLORER_CHECKPOINT_VERIFY_RATE_LIMIT_RPM", + "EXPLORER_ACTION_PROOF_RATE_LIMIT_RPM", + "EXPLORER_VALIDATOR_SET_PROOF_RATE_LIMIT_RPM", + "EXPLORER_VM_QUERY_RATE_LIMIT_RPM", + "EXPLORER_BATCH_RATE_LIMIT_RPM" + ]) { + const value = { + EXPLORER_RATE_LIMIT_RPM: config.EXPLORER_RATE_LIMIT_RPM, + EXPLORER_FEE_QUOTE_RATE_LIMIT_RPM: config.EXPLORER_FEE_QUOTE_RATE_LIMIT_RPM, + EXPLORER_PREFLIGHT_POST_RATE_LIMIT_RPM: config.EXPLORER_PREFLIGHT_POST_RATE_LIMIT_RPM, + EXPLORER_TIP_MAX_AGE_S: config.EXPLORER_TIP_MAX_AGE_S, + EXPLORER_CHECKPOINT_LIST_RATE_LIMIT_RPM: config.EXPLORER_CHECKPOINT_LIST_RATE_LIMIT_RPM, + EXPLORER_CHECKPOINT_VERIFY_RATE_LIMIT_RPM: config.EXPLORER_CHECKPOINT_VERIFY_RATE_LIMIT_RPM, + EXPLORER_ACTION_PROOF_RATE_LIMIT_RPM: config.EXPLORER_ACTION_PROOF_RATE_LIMIT_RPM, + EXPLORER_BATCH_RATE_LIMIT_RPM: config.EXPLORER_BATCH_RATE_LIMIT_RPM, + EXPLORER_VALIDATOR_SET_PROOF_RATE_LIMIT_RPM: config.EXPLORER_VALIDATOR_SET_PROOF_RATE_LIMIT_RPM, + EXPLORER_VM_QUERY_RATE_LIMIT_RPM: config.EXPLORER_VM_QUERY_RATE_LIMIT_RPM + }[key] + if (value === undefined || value === "") continue + defaultValues[key] = value + } + } + if (module === EXPLORER_MODULE_NAME) { + const networkCodePrefix = { [Network.MAINNET]: "", [Network.TESTNET]: "T", [Network.REGTEST]: "R" } + for (const coinName of Object.values(Coin)) { + for (const net of Object.values(Network)) { + const tick = CoinTickerSymbol[coinName] + defaultValues["UTXO_TRACKER_URL_" + networkCodePrefix[net] + tick] = + "http://" + getDockerContainerImageName(XChainService.XCHAIN_UTXO_TRACKER, coinName, net) + ":3001" + defaultValues["DECODER_API_URL_" + tick + "_" + net.toUpperCase()] = + "http://" + getDockerContainerImageName(XChainService.XCHAIN_DECODER, coinName, net) + ":3002" + const indexerVar = "INDEXER_API_URL_" + tick + "_" + net.toUpperCase() + defaultValues[indexerVar] = config.INDEXER_API_URL_ENV[indexerVar] + || "http://" + getDockerContainerImageName(XChainService.XCHAIN_INDEXER, coinName, net) + ":3004" + } + } + } +} + +module.exports = { + configure, + createCoinDefaults, + createSharedDefaults, + configureSharedBeforeHubKey, + configureSharedAfterHubKey +} diff --git a/src/services/config_service/filters.js b/src/services/config_service/filters.js new file mode 100644 index 0000000..ffcbf2c --- /dev/null +++ b/src/services/config_service/filters.js @@ -0,0 +1,87 @@ +/********************************************************************* + * + * Copyright © 2025–2026 Dankest, LLC + * Based on XChain Platform by Dankest, LLC – https://dankest.llc + * + * SPDX-License-Identifier: AGPL-3.0-or-later + * + * This file is part of XChain Platform. Licensed under the GNU Affero + * General Public License v3.0 or later; see LICENSE.md. A commercial + * license (without AGPL source-disclosure terms) is available - + * contact legal@dankest.llc. + * + ********************************************************************** + * Config Service Command Filters + ********************************************************************/ + +'use strict' + +let Coin, Network, XChainService, REGTEST_MODULES +let NODE_MODULE_NAME, DB_MODULE_NAME, HUB_MODULE_NAME, EXPLORER_MODULE_NAME, SYNC_MODULE_NAME + +function configure(dependencies) { + ({ Coin, Network, XChainService, REGTEST_MODULES, + NODE_MODULE_NAME, DB_MODULE_NAME, HUB_MODULE_NAME, EXPLORER_MODULE_NAME, SYNC_MODULE_NAME } = dependencies) +} + +// Short names operators actually type for the shared services, whose canonical +// names carry an `xchain-` prefix that is easy to omit. 'explorer' already had +// this treatment; the others did not, so `recreate hub` matched no container and +// looked like the hub was simply unsupported. +function serviceAliases() { + return { hub: HUB_MODULE_NAME, sync: SYNC_MODULE_NAME, db: DB_MODULE_NAME } +} + +// Callers that bypass resolveArgs (recreate, start/stop/restart, logs) hand +// the operator's raw token straight through, so the alias map has to apply +// here too. +// Shared services (hub / explorer / db / sync) are registered under a single +// empty coin+network key, not per-coin. A bare `update xchain-hub` must resolve +// to that ""/"" container; otherwise it gets fanned out across real coins where +// it matches nothing and the command silently no-ops. +// The coin node leads the per-chain list so `install all` creates it +// before the services that poll it. With the node last, a mainnet +// install created the decoder four and a half hours before the node +// existed (a 151 GiB tracker restore sat between them); the decoder +// spent that time logging ENOTFOUND for a name the network did not +// carry yet, and the node's own initial sync, the slowest step on the +// box, had not even started. Every other command that expands `all` +// (update, start, stop, uninstall) tolerates either order. +// Explicitly-named shared service → emit under the empty ""/"" key only. +function filterCommandParameters(branch, modules, coins, networks) { + const servicesList = {} + let addExplorer = false + const aliases = serviceAliases() + if (modules && aliases[modules]) modules = aliases[modules] + coins = coins && coins !== "all" ? [coins] : Object.values(Coin) + networks = networks && networks !== "all" ? [networks] : Object.values(Network) + const sharedServices = [HUB_MODULE_NAME, EXPLORER_MODULE_NAME, DB_MODULE_NAME, SYNC_MODULE_NAME] + if (modules === "all") { + modules = [NODE_MODULE_NAME, ...Object.values(XChainService).filter(m => m !== XChainService.XCHAIN_E2E_TEST)] + addExplorer = true + } else if (modules === "explorer") { + addExplorer = true + coins = [] + } else if (sharedServices.includes(modules)) { + return { "": { "": [modules] } } + } else if (modules === "node") { + modules = [NODE_MODULE_NAME] + } else if (modules) { + modules = [modules] + } + if (addExplorer) servicesList[""] = { "": [EXPLORER_MODULE_NAME] } + for (const nextCoin of coins) { + if (!(nextCoin in servicesList)) servicesList[nextCoin] = {} + for (const nextNetwork of networks) { + if (!(nextNetwork in servicesList[nextCoin])) servicesList[nextCoin][nextNetwork] = [] + for (const nextModule of modules) { + if (!REGTEST_MODULES.includes(nextModule) || nextNetwork === Network.REGTEST) { + servicesList[nextCoin][nextNetwork].push(nextModule) + } + } + } + } + return servicesList +} + +module.exports = { configure, serviceAliases, filterCommandParameters } diff --git a/src/services/config_service/networks.js b/src/services/config_service/networks.js new file mode 100644 index 0000000..88d309d --- /dev/null +++ b/src/services/config_service/networks.js @@ -0,0 +1,387 @@ +/********************************************************************* + * + * Copyright © 2025–2026 Dankest, LLC + * Based on XChain Platform by Dankest, LLC – https://dankest.llc + * + * SPDX-License-Identifier: AGPL-3.0-or-later + * + * This file is part of XChain Platform. Licensed under the GNU Affero + * General Public License v3.0 or later; see LICENSE.md. A commercial + * license (without AGPL source-disclosure terms) is available - + * contact legal@dankest.llc. + * + ********************************************************************** + * Config Service Network Defaults + ********************************************************************/ + +'use strict' + +let stateModule, Coin, Network, XChainService, logger, config, peers, readSecretHostEnv +let HUB_MODULE_NAME, getDockerContainerImageName +let warnedHubConfigKeys = new Set() + +function configure(dependencies) { + ({ stateModule, Coin, Network, XChainService, logger, config, peers, readSecretHostEnv, + HUB_MODULE_NAME, getDockerContainerImageName } = dependencies) + warnedHubConfigKeys = new Set() +} + +// A command composes the shared hub's config many times, so each deploy-time warning +// about it is said once per process rather than once per composition. +function warnHubConfigOnce(key, message) { + if (warnedHubConfigKeys.has(key)) return + warnedHubConfigKeys.add(key) + logger.warn(message) +} + +// The coin/network stacks this deployment runs, from the module registry. Returns [] +// when the registry is unreadable (no pool yet), which the callers treat as "unknown" +// rather than "none". +async function getRegisteredCoinStacks() { + try { + const { db } = stateModule + const rows = await db.getAllModuleContainers(null, null) + return (rows || []).filter(r => r && r.coin && r.network + && Object.values(Coin).includes(r.coin) && Object.values(Network).includes(r.network)) + } catch { + return [] + } +} + +// The coin and network tokens of the command being run. A first install registers no +// coin stack until AFTER preCheck has deployed the shared hub, so the operator's own +// arguments are the only source the hub's env can be composed from on a fresh host. +function getCommandCoinsAndNetworks() { + const argv = process.argv.slice(2) + return { + coins: [...new Set(argv.filter(t => Object.values(Coin).includes(t)))], + networks: [...new Set(argv.filter(t => Object.values(Network).includes(t)))] + } +} + +// The network a standalone hub should declare: the one every stack of this deployment +// runs on. Ambiguous (several networks, or none named) leaves it unset, because one +// hub declaring the wrong network mis-gates the ingest rules it is being set for. +async function resolveDeploymentHubNetwork() { + const registered = new Set((await getRegisteredCoinStacks()).map(r => r.network)) + const networks = registered.size > 0 ? registered : new Set(getCommandCoinsAndNetworks().networks) + if (networks.size === 1) return [...networks][0] + if (networks.size > 1) { + warnHubConfigOnce("HUB_NETWORK_AMBIGUOUS", + "WARNING: HUB_NETWORK is not set and this deployment runs stacks on " + + [...networks].sort().join(", ") + ", so the shared hub cannot derive one network. " + + "Its network-keyed ingest gates (PRICE batch validation) stay closed until " + + "HUB_NETWORK is set in the host env.") + } + return null +} + +// Whether this deployment runs a BTC indexer for `network`, counting the one the +// running command is installing right now: the hub is deployed before it exists, and +// the composed URL names the container that install creates. +async function hasBitcoinIndexer(network) { + const registered = await getRegisteredCoinStacks() + if (registered.some(r => r.coin === Coin.BITCOIN && r.network === network + && r.module === XChainService.XCHAIN_INDEXER)) return true + const command = getCommandCoinsAndNetworks() + return command.coins.includes(Coin.BITCOIN) && command.networks.includes(network) +} + +// Usage-telemetry env is only meaningful to the hub (the telemetry collector). +// TELEMETRY_IP_SALT is read from the host environment (e.g. xchain-node's .env) so +// the IP-hash salt stays out of source and config files; without it the hub records +// country/region but leaves ip_hash null. The hub is a shared service (no per +// coin/network config file), so the host env is the injection point. +// Gate for the per-install detail endpoint (GET /telemetry/operators). Like the +// salt, sourced from host env so the secret stays out of source/config files; +// unset leaves the endpoint fail-closed (401 for everyone). +// BTC indexer JSON-RPC URL for the validator-mode price oracle's block-height +// anchor (hub.getlatestblock). Sourced from host env so a hub NOT co-located with +// a BTC indexer (e.g. the master hub box, where the BTC stack lives elsewhere) can +// point at a reachable indexer. Empty default ⇒ the hub falls back to its configs +// table, so co-located standalone/validator installs are unaffected. Left empty +// here, it is composed from the co-located BTC indexer further down. +// State-checkpoint engine + ANCHOR publisher (validator mode). The hub is a +// shared service (no per coin/network config file), so like the telemetry +// salt and BTC_INDEXER_API_URL above, the host env is the injection point. +// Per-coin _INDEXER_URLs feed getblockhashes (checkpoint state reads); +// DOGE_* configures the on-chain ANCHOR/price publisher signer pipeline; +// XDEX_* are the shared single-validator/regtest seams. Only set values are +// injected, so unset host env leaves the hub's own defaults untouched. +// HUB_API_KEY gates the hub's consensus-affecting write methods. Sourced from +// host env (.env) so it persists across `update` (a hand-set container value is +// dropped on rebuild). Set it on the publicly-fronted master hub so writes are +// authenticated; unset leaves the hub keyless (the prior default). +// ANCHOR_CHUNK_RETRY_MS must outlast the utxo-tracker's mempool poll +// (60s on mainnet) or back-to-back same-wallet anchor broadcasts +// exhaust their retries on a stale UTXO view (txn-mempool-conflict). +// Anchor every Nth checkpoint_seq on-chain (off-multiples stay in the +// free off-chain mirror); decouples DOGE spend from checkpoint cadence. +// Per-coin confirmation depth the hub's cross-chain engines wait for +// before proposing a source leg (coins/index.js resolveConfirmations). +// A regtest venue pins these to 1 so a bridge lock finalizes on the +// next block instead of six BTC blocks nothing is mining (the nightly +// two-stack legs sat on "not proposing BTC:3 (below depth 6)" until +// the 120 s credit wait gave up). Inert on mainnet and testnet: the +// hub clamps a value below the per-coin default UP to that default +// off regtest, so this can only raise the depth on a real network. +// Reverse-proxy trust for the hub's express API (rate-limiter IP +// keying). Default 'loopback' suits the Apache-on-same-host prod +// topology; containerized hubs see the docker bridge as the peer, +// so an operator fronting the container with a proxy sets this. +// Deployment network for the hub's consensus gates (notably +// STAKE_WEIGHTED_QUORUM, whose activation height is per-network). +// REQUIRED by the hub in validator mode (it fails loud on a +// blank/invalid value; no silent default here either) and must +// match the INDEXER_NETWORK of the chains this hub federates. +// Oracle price-round finalization threshold. Defaults to 2 in the hub +// (a 2-hub diversity floor so a lone external source never becomes a +// federation-signed price). Single-host prod / regtest deployments must +// set ORACLE_MIN_SUBMISSIONS=1 explicitly or no round ever finalizes, +// which stalls every indexer's oracle price-sync barrier. Passed through +// here so the host env survives a hub container regenerate. +// Oracle round cadence. CONSENSUS-UNIFORM: every hub in a federation must +// share these or round numbering and the submission cutoff diverge. Passed +// through for single-validator regtest/e2e venues, where short rounds keep +// a live drill from waiting 10 minutes per finalization; real networks +// leave them unset and take the hub defaults. +// PRICE batch-publisher knobs (window length, finalization grace, +// co-sign timeout, buffer cap). Deliberately NOT consensus-grouped in +// HubConsensusEnvGuard: batch validation is range-agnostic, so two hubs +// running different window sizes just elect different leaders and may +// double-publish overlapping windows, which is idempotent at ingest and +// is the same posture today's publisher failover already has. They +// change what a leader PROPOSES, never what any node ACCEPTS. Passed +// through so the host env survives a hub container regenerate. +// Same family: the time budgeted between a window closing and its batch +// being readable on chain (assembly, co-signing, broadcast, one DOGE +// confirmation). The publisher subtracts it from the fee-price staleness +// bound to derive the window ceiling, so a venue whose +// landing latency differs from the fleet's tunes it here rather than +// being clamped to a window that does not suit it. +// ATTEST response mirror regtest-only overrides (the attest response mirror design). Both are +// honoured by the receiving hub module ONLY when HUB_NETWORK=regtest (a warn- +// and-ignore off regtest, the same posture resolveWatermarkGrace takes on the +// indexer side), so passing them through here unconditionally mirrors the +// ORACLE_BATCH_* family above: they cannot arm anything off regtest by any path +// in this file, the real gate lives at the point of consumption. +// +// ATTEST_RESPONSE_FORWARD_S_OVERRIDE lets a regtest venue's leader pick a short +// effective_time margin instead of the real 120s ATTEST_RESPONSE_FORWARD_S, so a +// response can bind within the same short block cadence a regtest drill runs at +// (xchain-hub/src/attestation/attest_response_timing.js). +// ATTEST_BATCH_WINDOW_S_OVERRIDE is the same seam for the batch cadence: +// AttestationBatchPublisher (row 20, not yet built) will read it on the same +// regtest-only pattern as the forward override above, so the passthrough is +// wired ahead of that publisher rather than after it. +// Per-IP request/min cap on the hub's express API (default 100). Too low +// for legitimate multi-indexer re-bootstrap: every indexer on a box shares +// one source IP, so a fleet bootstrapping HubDbSync tables (oracle_prices, +// price_snapshots, cross_chain_calls, capability_snapshots, state_checkpoints) +// collectively blows 100/min and gets 429'd, so the heartbeat gate then stays +// closed and the chain stalls. Raise for prod fleets. Passed through so the +// host env survives a hub container regenerate. +// +// The hub exempts loopback and private-range callers from +// that cap by default, which covers the case above: the indexers reach the hub +// container over the bridge network this compose file creates, so a managed +// node no longer needs the limit raised to rebuild price history from the chain. +// HUB_RATE_LIMIT_EXEMPT_LOCAL=false turns the exemption off and restores the +// old behavior for an operator who wants the cap enforced on every caller; +// passed through for the same container-regenerate reason. +// XCHAIN derived-price source. XCHAIN is listed on no exchange, so +// a validator computes XCHAIN/USD from realized fills in its OWN BTC indexer +// database instead of fetching it. Every native-coin fee decision on LTC and +// DOGE needs that pair, and without it those chains cannot price a fee at all. +// +// Read-only access; unset means the hub simply abstains from the pair and +// submits the 36 API pairs exactly as before, which is a supported state. +// Passed through here so the values survive a hub container regenerate - a +// config file alone never reaches the container. +// Consensus-uniform derivation parameters (window length, confirmation +// buffer, bootstrap price). Overrides exist for regtest and e2e only: a hub +// running different values computes a different XCHAIN/BTC leg and lands +// outside the co-sign deviation band, so on a real network leave them unset +// and move them only by a coordinated flag-day. +// D2 supersession threshold override. The shipped constant keeps +// supersession disabled (bootstrap carry-forward only); regtest/e2e drills +// set '0' so any realized volume supersedes, which is what lets a live +// proof distinguish a derived print from the carry-forward it would +// otherwise silently match. +// Hub API authentication. Without these two passed through, VALIDATOR MODE +// IS UNREACHABLE: the hub refuses to boot in validator mode unless one of +// them is set ("HUB_API_KEY is not set in validator mode. Write methods +// would be UNAUTHENTICATED"), and without this passthrough neither reaches +// the container at all. +// +// That failure also WEDGES the installer, so it is worth more than a +// one-line fix: once P2P_VALIDATOR_ADDR is baked into the container env the +// hub crash-loops, and `update xchain-hub` then fails because its own +// precheck tries to restart the container that cannot start. Recovery is +// `docker rm -f` the container and update again. +// +// HUB_ALLOW_UNAUTHENTICATED=true is the documented keyless escape hatch and +// suits a single-host regtest venue that already ran open; a real network +// sets HUB_API_KEY instead. +// The regtest ROLLCALL arming opt-in. The hub carries a byte-twin of +// the indexer's rollcall_activation.js, and ROLLCALL_ACTIVATION is one of +// consensus_rules_digest.js's SHARED_GATES, so an indexer armed against an +// inert container hub reports a rules MISMATCH on the venue. Both sides take +// the same variable, so a venue arms as a unit. +// +// Passed through with no network gate, unlike the indexer's copy above: the +// hub is a shared service and getDefaultConfig is called for it as +// (module, null, null), so there is no network here to gate on. That is safe +// because the real gate is in the hub's own rollcall_activation.js, which can +// reach the environment for regtest and for nothing else - mainnet and testnet +// are literal there and unreachable from env by any path in the file. On a +// mainnet or testnet hub this variable is therefore inert, not dangerous. +// XC_ROLLCALL_GATES_REGTEST_ACTIVATION rides beside it with the same +// no-network-gate reasoning: it arms ROLLCALL v1 and the rules-aware +// attestation set separately from the rail, so a venue can drive v0 as its +// control, and the hub's own rollcall_gates_activation.js gates it for real. +// XC_MIRROR_ADMISSION_ACTIVATION follows the same no-network-gate shape +// (D84 precedent): it arms the admission-map mirror and its consumer and +// barrier gates together, so a venue arms as a unit; the hub's own +// mirror-admission gate module gates it for real. +// Secret-bearing names in this list (XCHAIN_PRICE_INDEXER_DB_PASS) are also +// accepted from the host env under their redaction-safe `*_SECRET` spelling; +// everything else resolves to a plain process.env read. +function configureHubBeforeKey(defaultValues, module) { + if (module === HUB_MODULE_NAME) { + defaultValues["TELEMETRY_ENABLED"] = config.TELEMETRY_ENABLED + defaultValues["TELEMETRY_RETENTION_DAYS"] = config.TELEMETRY_RETENTION_DAYS + defaultValues["TELEMETRY_IP_SALT"] = config.TELEMETRY_IP_SALT + defaultValues["TELEMETRY_ADMIN_KEY"] = config.TELEMETRY_ADMIN_KEY + defaultValues["BTC_INDEXER_API_URL"] = config.BTC_INDEXER_API_URL + const hubPassthroughVars = [ + "HUB_API_KEY", + "BTC_INDEXER_URL", "LTC_INDEXER_URL", "DOGE_INDEXER_URL", + "BTC_INDEXER_API_KEY", "LTC_INDEXER_API_KEY", "DOGE_INDEXER_API_KEY", + "CHECKPOINT_ENABLED", "CHECKPOINT_INTERVAL_BLOCKS", "CHECKPOINT_CONFIRMATIONS", + "CHECKPOINT_POLL_MS", "CHECKPOINT_ROUND_TIMEOUT_MS", "CHECKPOINT_CHAINS", + "ANCHOR_ENABLED", "ANCHOR_INTERVAL_MS", "ANCHOR_MATCH_BATCH_SIZE", + "ANCHOR_MAX_BATCH", "ANCHOR_CHUNK_MAX_BYTES", "ANCHOR_ROUND_TIMEOUT_MS", + "ANCHOR_CHUNK_RETRY_MS", + "ANCHOR_ELECTION_TOLERANCE_BLOCKS", "ANCHOR_REWARD_PER_PUBLISH", + "ANCHOR_CHECKPOINT_EVERY_N", + "DOGE_ENCODER_URL", "DOGE_ENCODER_API_KEY", "DOGE_ADDRESS", + "DOGE_PUBKEY_HEX", "DOGE_LOW_BALANCE_THRESHOLD", + "XDEX_SEED_LOCAL_VALIDATOR", "XDEX_SNAPSHOT_BLOCK", + "XCHAIN_CONFIRMATIONS_BTC", "XCHAIN_CONFIRMATIONS_LTC", "XCHAIN_CONFIRMATIONS_DOGE", + "HUB_TRUST_PROXY", + "HUB_NETWORK", + "ORACLE_MIN_SUBMISSIONS", + "ORACLE_ROUND_INTERVAL", "ORACLE_SUBMISSION_WINDOW", + "ORACLE_BATCH_WINDOW_ROUNDS", "ORACLE_BATCH_GRACE_MS", + "ORACLE_BATCH_SIGN_TIMEOUT_MS", "ORACLE_BATCH_BUFFER_MAX_ROUNDS", + "ORACLE_BATCH_LANDING_RESERVE_MS", + "ATTEST_RESPONSE_FORWARD_S_OVERRIDE", + "ATTEST_BATCH_WINDOW_S_OVERRIDE", + "HUB_RATE_LIMIT_RPM", "HUB_RATE_LIMIT_EXEMPT_LOCAL", + "XCHAIN_PRICE_INDEXER_DB_HOST", "XCHAIN_PRICE_INDEXER_DB_PORT", + "XCHAIN_PRICE_INDEXER_DB_NAME", "XCHAIN_PRICE_INDEXER_DB_USER", + "XCHAIN_PRICE_INDEXER_DB_PASS", "XCHAIN_PRICE_INDEXER_DB_COIN", + "XCHAIN_PRICE_WINDOW_BLOCKS", "XCHAIN_PRICE_CONFIRMATION_BUFFER", + "XCHAIN_PRICE_BOOTSTRAP_SATS", + "XCHAIN_PRICE_MIN_BTC_VOLUME", + "HUB_API_KEY", "HUB_ALLOW_UNAUTHENTICATED", + "XC_ROLLCALL_REGTEST_ACTIVATION", + "XC_ROLLCALL_GATES_REGTEST_ACTIVATION", + "XC_MIRROR_ADMISSION_ACTIVATION" + ] + for (const varName of hubPassthroughVars) { + const value = readSecretHostEnv(varName) + if (value !== undefined && value !== "") { + defaultValues[varName] = value + } + } + } +} + +// The hub now REFUSES to boot when HUB_API_KEY is unset unless +// keyless operation is declared with HUB_ALLOW_UNAUTHENTICATED. A managed +// deploy with no key in the host env is a legitimate posture (single-host +// regtest, a hub reachable only on a private network), so make the +// declaration here rather than letting the container crash-loop: the point +// of the hub-side change is that keyless is a stated choice, and the +// deployer is what states it. An operator who wants the refusal instead +// sets HUB_ALLOW_UNAUTHENTICATED=false in the host env, which the +// passthrough above preserves. `xchain-node go-live` still refuses a +// keyless mainnet hub outright (GoLiveGate). +// MAINNET is the exception: there the review's "invert the defaults, fail +// closed" applies with real funds behind it, so we do NOT declare keyless +// on the operator's behalf and the hub's own refusal stands. (A mainnet +// VALIDATOR hub is already covered: the hub has refused keyless validator +// boots since before this change, so no running one can be keyless and +// undeclared. This only reaches a mainnet config-only hub.) +// `validator init` leaves a generated key in the shared hub sidecar so the +// onboarding path produces a hub that BOOTS. Read it here (host env still wins), +// before the keyless declaration below: a node that has a credential must deploy +// authenticated rather than be handed the escape hatch it no longer needs. +function configureHubAccess(defaultValues) { + const hubNetworkIsMainnet = String(defaultValues["HUB_NETWORK"] || "").toLowerCase() === Network.MAINNET + if (!defaultValues["HUB_API_KEY"] && defaultValues["HUB_ALLOW_UNAUTHENTICATED"] === undefined + && !hubNetworkIsMainnet) { + defaultValues["HUB_ALLOW_UNAUTHENTICATED"] = "true" + logger.warn("WARNING: HUB_API_KEY is not set, so this hub is deployed with an UNAUTHENTICATED " + + "write surface (HUB_ALLOW_UNAUTHENTICATED=true). Anyone who can reach the hub port can drive " + + "updateconfig / registervalidator / reportreorg. Set HUB_API_KEY in the host env before " + + "exposing this hub beyond a trusted network.") + } +} + +// Operator signer for the on-chain DOGE publishers: when the host sets +// XCHAIN_NODE_HUB_SIGNER_DIR, ModuleService mounts that directory +// read-only at /XChainHub/operator-signer and the hub loads +//

/signer.js via HUB_SIGNER_MODULE (see xchain-hub +// examples/doge-signer.example.js for the module contract). +// Validator mode: when `xchain-node validator init` has been run, inject the +// P2P / signing-key / capability-config env so the hub starts as a full +// validator. Returns {} (no change) for a standalone node, so the standalone +// install path is unaffected. +// State the resolved mode and the directory it came from: an empty validator +// env means standalone, disabled, or a configDir carrying no validator/, and +// a deploy cannot tell those apart. A statement, never a refusal. +function configureHubValidator(defaultValues) { + if (config.XCHAIN_NODE_HUB_SIGNER_DIR) { + defaultValues["HUB_SIGNER_MODULE"] = "/XChainHub/operator-signer/signer.js" + } + const { getValidatorEnv, validatorModeReport } = peers.validatorService + Object.assign(defaultValues, getValidatorEnv()) + const validator = validatorModeReport() + if (validator.mode === 'validator') { + logger.info("xchain-node: this hub deploys in VALIDATOR mode, from " + validator.dir) + } else if (validator.mode === 'incomplete') { + warnHubConfigOnce("VALIDATOR_STATE_INCOMPLETE", + "WARNING: the validator state under " + validator.dir + " is HALF PRESENT (missing " + + validator.missing.join(", ") + "), so this hub deploys STANDALONE: no P2P_VALIDATOR_ADDR, " + + "no SIGNING_PRIVKEY_HEX, no capability mount, and its anchor publisher will never run. " + + "Half a validator state is never a standalone node, so this is a broken install rather " + + "than a choice: restore the missing file, or point XCHAIN_NODE_CONFIG_DIR at the config " + + "directory that holds the complete set.") + } else if (validator.mode === 'disabled') { + logger.info("xchain-node: this hub deploys STANDALONE because the validator state at " + + validator.dir + " records enabled:false.") + } else { + logger.info("xchain-node: this hub deploys STANDALONE (no validator state under " + + validator.dir + "). If this host IS meant to be a validator, XCHAIN_NODE_CONFIG_DIR is " + + "resolving to the wrong config directory and the real one holds validator/.") + } +} + +module.exports = { + // A hub with no HUB_NETWORK resolves its network to '', which fails every + // network-keyed ingest gate closed, so a non-validator install can never + // validate an on-chain PRICE batch. Host env still wins; unresolved stays unset. + // Capability snapshots are read off a BTC indexer, and with none reachable the + // hub refuses every on-chain PRICE batch for insufficient signer stake. Compose + // the co-located one; say so when this deployment has none to compose. + configure, + warnHubConfigOnce, + resolveDeploymentHubNetwork, + hasBitcoinIndexer, + configureHubBeforeKey, + configureHubAccess, + configureHubValidator +} diff --git a/src/services/config_service/services.js b/src/services/config_service/services.js new file mode 100644 index 0000000..70559c1 --- /dev/null +++ b/src/services/config_service/services.js @@ -0,0 +1,307 @@ +/********************************************************************* + * + * Copyright © 2025–2026 Dankest, LLC + * Based on XChain Platform by Dankest, LLC – https://dankest.llc + * + * SPDX-License-Identifier: AGPL-3.0-or-later + * + * This file is part of XChain Platform. Licensed under the GNU Affero + * General Public License v3.0 or later; see LICENSE.md. A commercial + * license (without AGPL source-disclosure terms) is available - + * contact legal@dankest.llc. + * + ********************************************************************** + * Config Service Service Defaults + ********************************************************************/ + +'use strict' + +let config, peers, logger, readSecretHostEnv, Coin, Network, CoinTickerSymbol, XChainService +let HUB_MODULE_NAME, EXPLORER_MODULE_NAME, getDockerContainerImageName + +function configure(dependencies) { + ({ config, peers, logger, readSecretHostEnv, Coin, Network, CoinTickerSymbol, XChainService, + HUB_MODULE_NAME, EXPLORER_MODULE_NAME, getDockerContainerImageName } = dependencies) +} + +// The validator-onboarding suite STAKEs the hub's own signing pubkey and +// asserts the indexer then admits it to each capability set, so it needs +// to know which key the hub actually runs as. It read VALIDATOR_PUBKEY +// from the env and skipped when unset, which meant the only way to run it +// was for an operator to hand-copy the hex out of `validator status` into +// the coin config - so it skipped everywhere nobody had, including CI. +// Derive it from the same settings file the hub's own env comes from +// (getValidatorEnv above), so the two can never name different keys. +// +// PUBLIC half only. The seed stays in signing.key / SIGNING_PRIVKEY_HEX +// and goes to the hub alone; the test needs the pubkey and nothing else. +// +// A standalone node has no validator, so this is absent and the suite +// still skips - correctly, because there is no identity to onboard. +// The harness discovers every rail's node and indexer credentials through +// the hub's getallconfigs (test/helpers/chainRail.js), and a keyed hub +// gates that read behind HUB_API_KEY. A validator-mode host is keyed +// (`validator init` mints the key into the hub sidecar), so without this +// passthrough the e2e container was the one hub client on the host still +// calling keyless: the litecoin and dogecoin matrix legs 401'd in +// initialCheck's beforeAll (`[chainRail] hub has no config for +// bitcoin/regtest`, run 35120852486) while the standalone bitcoin leg, +// whose hub has no key, never noticed. Host env first, then the sidecar, +// exactly as the indexer and the shared services resolve it; a keyless +// host stays keyless. +// e2e-test also derives addresses (test/cryptoHelper.js) and resolves its +// bitcoinjs network from COIN+NETWORK. initialCheck.test.js reads +// process.env.COIN and only splits NETWORK when COIN is absent; without COIN +// it mis-splits the bare network ("regtest" → COIN="regtest", NETWORK=undefined) +// → getBitcoinJsNetwork returns undefined → bitcoinjs falls back to MAINNET +// ("1..." addresses) and funded txs never confirm on regtest. Inject COIN so +// the resolution is correct while NETWORK stays bare for other env consumers. +// +// The contract-template suites (test:sdk/*Template) load their source from +// xchain-contracts. LIBRARY_BUNDLES stages it into the e2e-test build context +// and the Dockerfile COPYs it to /XChainE2ETest/xchain-contracts, so point the +// resolver there. Without the bundle present the suites skip (they no longer +// abort the run). +function configureE2e(defaultValues, module, coin) { + if (module === XChainService.XCHAIN_E2E_TEST) { + defaultValues["COIN"] = coin + defaultValues["XCHAIN_CONTRACTS_DIR"] = "/XChainE2ETest/xchain-contracts" + const { getValidatorSettings } = peers.validatorService + const validatorSettings = getValidatorSettings() + if (validatorSettings && validatorSettings.pubkey) { + defaultValues["VALIDATOR_PUBKEY"] = validatorSettings.pubkey + } + if (config.HUB_API_KEY !== undefined && config.HUB_API_KEY !== "") { + defaultValues.HUB_API_KEY = config.HUB_API_KEY + } + } +} + +// Genesis-ledger bootstrap env (xchain-indexer only). The indexer binds its +// consensus-critical genesis parameters from the container environment: mainnet/testnet +// are frozen-pinned in the indexer's configs/.js, but regtest reads the activation +// block + ledger/dump hashes from env so an operator can dry-run genesis at a current +// regtest block. Without this passthrough those host vars never reach the container, so +// genesis can't be enabled on a regtest/dev stack. Mirrors hubPassthroughVars: only set, +// non-empty host vars are injected (and a config-file value still wins), so an unset env +// leaves GENESIS_BLOCK at its 0/default and genesis stays off. The path vars point at +// in-container files; override them only when a custom CSV/dump is volume-mounted. +// The GENESIS_AIRDROP_* members carry the XCP/XDP airdrop leg. The indexer honors +// them on regtest ONLY: off regtest the armed bucket set comes from the +// pinned coin bundle, so passing them through cannot arm anything on a mainnet or +// testnet stack, only on the regtest dry-run this passthrough exists for. +// ROLLCALL rail env (xchain-indexer only). Two separate things, both of which a +// deployed indexer needs before an epoch close can do anything at all. +// +// 1. DOGE_INDEXER_API_URL / DOGE_INDEXER_API_KEY, on EVERY network. Roll calls +// land on DOGECOIN and the BTC indexer is the only place the close runs, so +// rollcall_proof_client.js (and anchor_proof_client.js beside it) has to be +// able to ask a DOGE indexer. With no URL the close returns +// `{decided:false, reason:'DOGE indexer not configured'}` and the BTC indexer +// DEFERS the block forever, which is exactly how a single-coin venue wedges. +// Sourced from host env so the pair survives an `update` instead of needing +// to be hand-set on the container after every deploy. +// +// 2. XC_ROLLCALL_REGTEST_ACTIVATION, on REGTEST ONLY. This is the one value a +// regtest venue owns: the no-tunable-input rule is scoped to shared-ledger +// networks, because two regtest venues cannot fork each other. It is gated on +// the network here as well as in the indexer's own rollcall_activation.js, +// which is structurally unable to reach the environment for mainnet or +// testnet - two independent gates, so neither one being edited alone can arm +// a shared ledger from a host variable. +// +// 3. HUB_SYNC_ANCHOR_ATTEST_GRACE_S, on REGTEST ONLY, for the same reason as +// (2) and with the same two independent gates: the indexer's own +// resolveWatermarkGrace IGNORES it off regtest with a warning, because a +// watermark grace is a consensus input and a per-node value forks +// settlement. +// +// WHY A REGTEST VENUE NEEDS IT AT ALL. The anchor-reward attestation +// barrier holds a block until `streamWatermark >= blockTime + 120`. Off +// regtest that is free: blocks are ten minutes apart, so by the time one is +// processed the watermark is long past it. On regtest, blocks are stamped at +// about wall clock and the watermark tracks wall clock too, so a freshly +// mined block can NEVER be 120s behind the watermark and the barrier is +// unsatisfiable by construction. Every affected block then burns the full +// 60s timeout before proceeding anyway. +// +// MEASURED, on the 2026-09-06 release matrix: the BTC leg parsed 367 blocks +// in six hours and was killed by the job budget, against 2013 blocks in 1h52m +// on the pre-mirror build - 160 deferrals at 60s each, about 2.7 hours spent +// waiting for a condition that could not arrive. The other two coins were +// unaffected because this barrier is BTC-only. Nothing was wrong with the +// product: the venue was simply running a shared-ledger constant on a chain +// whose block cadence it was never sized for. +// 4. HUB_PRICE_SYNC_TIMEOUT_MS, on REGTEST ONLY here even though the value +// itself is not a consensus input. It bounds ONE mirror-barrier ATTEMPT: +// on expiry the block is DEFERRED and retried, never committed +// uncertified, which XChainIndexer states outright ("purely operational: +// it opens no barrier and commits no block"). So shortening it trades +// nothing away; it only makes a failed attempt cheaper. +// +// WHY A FAST VENUE NEEDS IT. Where the mirror legitimately lags the +// chain, every affected block waits the full attempt before deferring. +// Measured on the 2026-09-06 release matrix: 119 anchor-attest deferrals +// at the 60s default burned 119 minutes of a 289-minute BTC leg, 41% of +// the wall clock, and the indexer fell far enough behind that thirty +// e2e waits gave up on rows that had not landed yet. The barrier is +// doing its job; the cost per attempt is what a fast venue cannot afford. +// +// Gated on regtest anyway, because a shared ledger wants the long +// attempt: there a lagging mirror is a real fault worth waiting on, not +// a cadence mismatch. +// 5. XCHAIN_COINPAY_EXPIRATION_S, on REGTEST ONLY, for the same reason as (2) +// and (3) and with the same two independent gates: the indexer's own +// resolveCoinpayExpiration IGNORES it off regtest with a warning, because +// the window is added to a match's BLOCK_TIME and STORED as the +// obligation's deadline, so a per-node value expires the same escrow at +// different blocks and forks the ledger. +// +// WHY A REGTEST VENUE NEEDS IT. The e2e COINPay expiry case cannot wait out +// a two-hour deadline, so it freezes the node clock past the deadline and +// mines. That stamps the mined blocks two hours into the FUTURE, and the +// anchor-attest barrier in (3) compares a block's own timestamp against a +// wall-clock watermark, so the indexer then waits those two hours in real +// time on that one block. +// +// MEASURED, on the 2026-09-06 release matrix run 34015867460: all 119 +// deferrals in the BTC leg named the SAME block, held 2h08m50s, while the +// watermark tracked wall clock throughout (1-6s behind, advancing at 0.9999 +// of real time) and the hub logged no late heartbeat and no backpressure. +// Nothing was lagging. Shortening the window on regtest removes the clock +// jump that causes it, rather than teaching every barrier to special-case a +// future-stamped block. +// XC_ROLLCALL_GATES_REGTEST_ACTIVATION follows XC_ROLLCALL_REGTEST_ACTIVATION's +// same env-derived regtest shape (D84): it arms ROLLCALL v1 and the rules-aware +// attestation set separately from the rail, so a venue can drive v0 as its control. +// +// XC_MIRROR_ADMISSION_ACTIVATION rides the same regtest-only shape: without a +// path here the indexer side of the admission-map mirror can never be armed on +// regtest (row 24x), and it must arm together with the hub's copy above or the +// admission-era canonical refuses a legacy-map row and halts the block loop. +// The indexer pushes chain tips / config to the hub (HUB_API_URL); when that +// hub enforces HUB_API_KEY, the indexer must present the same key or its writes +// 401. Sourced from host env (.env) so it persists across `update`, then from the +// shared hub sidecar so an indexer co-located with a validator hub picks up the +// key `validator init` generated. Neither set leaves the indexer sending no key +// (keyless, the prior default). +function configureIndexerBeforeHubKey(defaultValues, module, network) { + if (module === XChainService.XCHAIN_INDEXER) { + const genesisPassthroughVars = [ + "XCHAIN_GENESIS_BLOCK", "XCHAIN_GENESIS_LEDGER_HASH", "XCHAIN_GENESIS_DUMP_HASH", + "GENESIS_LEDGER_PATH", "GENESIS_DUMP_PATH", + "GENESIS_BLOCK_TIMEOUT_MS", "GENESIS_DUMP_TIMEOUT_MS", + "GENESIS_AIRDROP_PATHS", "GENESIS_AIRDROP_HASHES", "GENESIS_AIRDROP_AMOUNTS", + "GENESIS_AIRDROP_SNAPSHOT_BLOCK", "GENESIS_AIRDROP_SET_HASH" + ] + for (const varName of genesisPassthroughVars) { + if (config.INDEXER_GENESIS_ENV[varName] !== undefined && config.INDEXER_GENESIS_ENV[varName] !== "") { + defaultValues[varName] = config.INDEXER_GENESIS_ENV[varName] + } + } + const rollcallPassthroughVars = ["DOGE_INDEXER_API_URL", "DOGE_INDEXER_API_KEY"] + if (network === Network.REGTEST) rollcallPassthroughVars.push("XC_ROLLCALL_REGTEST_ACTIVATION", + "XC_ROLLCALL_GATES_REGTEST_ACTIVATION", + "HUB_SYNC_ANCHOR_ATTEST_GRACE_S", + "HUB_PRICE_SYNC_TIMEOUT_MS", + "XCHAIN_COINPAY_EXPIRATION_S", + "XC_MIRROR_ADMISSION_ACTIVATION") + for (const varName of rollcallPassthroughVars) { + if (config.INDEXER_ROLLCALL_ENV[varName] !== undefined && config.INDEXER_ROLLCALL_ENV[varName] !== "") { + defaultValues[varName] = config.INDEXER_ROLLCALL_ENV[varName] + } + } + if (config.HUB_API_KEY !== undefined && config.HUB_API_KEY !== "") { + defaultValues.HUB_API_KEY = config.HUB_API_KEY + } + } +} + +// The hub authenticates to each indexer's federation API (attestation, stake polling, +// capability snapshots) with _INDEXER_API_KEY; the indexer fails closed unless its +// INDEXER_API_KEY matches. Source from host env so it persists across `update`, mirroring +// HUB_API_KEY above. Unset leaves the indexer fail-closed (keyless reads rejected). +// With no key configured the indexer fails closed: every gated method +// (feequotedryrun, the federation reads the staking e2e family asserts +// against) 401s, so a fresh regtest install can never pass those suites +// (audit F-10). Regtest is a local single-operator venue, so default the +// indexer's own documented keyless escape hatch on. A config-file value +// or host INDEXER_API_KEY still wins; mainnet/testnet stay fail-closed. +// Point the indexer's hub-DB connection (its price_snapshots/oracle_prices source) +// at its OWN database on mainnet/testnet: in this single-box topology HubDbSync +// mirrors those hub tables into the indexer DB, so the indexer reads prices from +// itself using its own DB account. Without HUB_DB_NAME the connection is never made +// and the mainnet native-fee price-source gate (XChainIndexer.start) fails closed. +// This mirrors the proven prod per-coin override; operator config overrides still +// win. HUB_DB_PASS is reconciled after the per-install DB password is resolved +// (see below); HUB_DB_HOST/PORT are already set above. +// +// WHY regtest WAS EXCLUDED UNTIL NOW, corrected 2026-07-26 then armed 2026-09-03 +// (the regtest mirror wedge). The old note here said "regtest has no hub to sync from", which +// stopped being true at 336a7d5 (HUB_API_URL is now composed for regtest too, and +// a regtest indexer does reach the hub: enabling this on litecoin-regtest +// bootstrapped 3 real rows into oracle_prices). The exclusion stood for a +// different and harder reason: turning the mirror on ARMS the block-loop +// watermark barriers (price, oracle, and now the ATTEST response mirror), and +// each one only opens once the mirror's stream watermark clears the row's time +// plus that barrier's grace. Production block timestamps LAG wall clock, so the +// watermark runs ahead and the escape fires; regtest blocks are stamped at ~now, +// so a real-network grace can NEVER be satisfied and every freshly mined block +// defers forever. Observed live: block 1479 deferred on a 60s timeout, repeatedly, +// until this was reverted. +// +// Armed unconditionally now (mainnet/testnet keep the exact same assignment they +// always had) because leaving the mirror off on regtest silently defeats every +// reader that expects hub state to reach the indexer, not just PRICE but +// the ATTEST response mirror this arms for too. Arming the pointer alone would +// reproduce the price wedge above, so every watermark grace this mirror gates is +// defaulted to 0 on regtest in the SAME step below: a config-file value or a host +// env override for any one of them still wins (resolveWatermarkGrace in +// hub_db_sync.js honours an override on regtest only, so the default below is +// exactly the value that seam already expects). Do not widen these off regtest: +// a per-node grace forks settlement. +// The three barrier graces the armed regtest mirror must clear to avoid the +// wedge above. HUB_SYNC_ATTEST_RESPONSE_GRACE_S is the passthrough this row +// adds (xchain-indexer/src/hub/hub_db_sync.js:615 reads it via resolveWatermarkGrace); +// HUB_SYNC_PRICE_GRACE_S / HUB_SYNC_ORACLE_GRACE_S are the pair the regtest mirror wedge already +// requires be set to 0 alongside it. A host env value always wins over the +// regtest default so an e2e drill can still exercise a nonzero grace. +// The list is EVERY watermark grace hub_db_sync.js resolves, not the three +// that first wedged: each mirrored table has its own barrier, and any one +// left at its frozen default holds every block up to 60s while that +// table's mirror watermark stands still, which on a three-rail regtest +// venue idle for hours is every block; an SDK drive's 120s index wait +// then dies on the second block. Measured 2026-09-09 on the match +// barrier, then again on the anchor-reward attestation barrier once +// match was cleared (hub_db_sync.js reads all of them through +// resolveWatermarkGrace, regtest-overridable only). +function configureIndexerAfterHubKey(defaultValues, module, network) { + if (module === XChainService.XCHAIN_INDEXER) { + if (config.INDEXER_API_KEY !== undefined && config.INDEXER_API_KEY !== "") { + defaultValues.INDEXER_API_KEY = config.INDEXER_API_KEY + } else if (network === Network.REGTEST) { + defaultValues.INDEXER_ALLOW_UNAUTHENTICATED = "true" + } + defaultValues.HUB_DB_NAME = defaultValues.INDEXER_DB_NAME + defaultValues.HUB_DB_USER = defaultValues.INDEXER_DB_USER + defaultValues.HUB_DB_SYNC_ENABLED = "true" + if (network === Network.REGTEST) { + const hubSyncRegtestGraceVars = [ + "HUB_SYNC_PRICE_GRACE_S", "HUB_SYNC_ORACLE_GRACE_S", "HUB_SYNC_ATTEST_RESPONSE_GRACE_S", + "HUB_SYNC_MATCH_GRACE_S", "HUB_SYNC_CALL_GRACE_S", "HUB_SYNC_ANCHOR_ATTEST_GRACE_S" + ] + for (const varName of hubSyncRegtestGraceVars) { + defaultValues[varName] = (config.HUB_SYNC_GRACE_ENV[varName] !== undefined && config.HUB_SYNC_GRACE_ENV[varName] !== "") + ? config.HUB_SYNC_GRACE_ENV[varName] + : "0" + } + } + } +} + +module.exports = { + configure, + configureE2e, + configureIndexerBeforeHubKey, + configureIndexerAfterHubKey +} diff --git a/src/services/config_service/sidecars.js b/src/services/config_service/sidecars.js new file mode 100644 index 0000000..b22468f --- /dev/null +++ b/src/services/config_service/sidecars.js @@ -0,0 +1,292 @@ +/********************************************************************* + * + * Copyright © 2025–2026 Dankest, LLC + * Based on XChain Platform by Dankest, LLC – https://dankest.llc + * + * SPDX-License-Identifier: AGPL-3.0-or-later + * + * This file is part of XChain Platform. Licensed under the GNU Affero + * General Public License v3.0 or later; see LICENSE.md. A commercial + * license (without AGPL source-disclosure terms) is available - + * contact legal@dankest.llc. + * + ********************************************************************** + * Config Service Sidecars + ********************************************************************/ + +'use strict' + +let crypto, fs, path, readline, configDir, preferredSecretEnvName +let foldSecretEnvAliases, deprecatedSecretEnvNames, logger + +function configure(dependencies) { + ({ crypto, fs, path, readline, configDir, preferredSecretEnvName, + foldSecretEnvAliases, deprecatedSecretEnvNames, logger } = dependencies) +} + + +// Persist RPC credentials to the untracked -.local sidecar. A fresh sidecar +// is created with writeFileSync; an existing one is appended to, unless overwrite is set +// (used by the legacy-migration path to replace it outright). The sidecar holds the live +// node RPC user/password, so it is forced to 0600 the same way credentials.json is: without +// an explicit mode the file lands at the process umask (commonly 0644), leaving the RPC +// credentials readable by any local user on the host. +function persistSidecarCreds(localFilePath, creds, { overwrite = false } = {}) { + const body = Object.keys(creds).map(k => `${k}=${creds[k]}`).join("\n") + "\n" + if (overwrite || !fs.existsSync(localFilePath)) { + const dir = path.dirname(localFilePath) + if (!fs.existsSync(dir)) fs.mkdirSync(dir, { recursive: true }) + fs.writeFileSync(localFilePath, body, { mode: 0o600 }) + } else { + let needsLeadingNewline = false + try { + const size = fs.statSync(localFilePath).size + if (size > 0) { + const fd = fs.openSync(localFilePath, 'r') + try { + const buf = Buffer.alloc(1) + fs.readSync(fd, buf, 0, 1, size - 1) + needsLeadingNewline = buf.toString('utf8') !== '\n' + } finally { + fs.closeSync(fd) + } + } + } catch { + needsLeadingNewline = false + } + fs.appendFileSync(localFilePath, needsLeadingNewline ? '\n' + body : body) + } + // chmod unconditionally: writeFileSync's mode only applies on create, and an + // already-existing sidecar (append path, or one written before this fix) keeps its + // old permissions otherwise. + try { fs.chmodSync(localFilePath, 0o600) } catch {} +} + +// Update specific KEY=VALUE entries in a sidecar while PRESERVING all other keys. Unlike +// persistSidecarCreds({overwrite:true}) (which rewrites the file with only the keys it is +// given), this reads the existing sidecar, overlays the supplied values, and rewrites it +// 0600. Used by the DB-password rotation to set DECODER_DB_PASS/INDEXER_DB_PASS (or +// HUB_DB_PASS) without clobbering the NODE_USER/NODE_PASSWORD already in the sidecar. +function upsertSidecarValues(localFilePath, values) { + const merged = {} + if (fs.existsSync(localFilePath)) { + for (const line of fs.readFileSync(localFilePath, "utf8").split(/\r?\n/)) { + const eqIndex = line.indexOf("=") + if (eqIndex > 0) merged[line.substring(0, eqIndex)] = line.substring(eqIndex + 1) + } + } + for (const k in values) { + // Respect the naming this sidecar already uses. A file the operator + // renamed to the redaction-safe `*_SECRET` form must not sprout the legacy + // twin again on the next rotation: two names for one credential is exactly + // the ambiguity foldSecretEnvAliases() refuses to guess through, so the + // rotation would leave the stack unable to start. + const alias = preferredSecretEnvName(k) + merged[alias && alias in merged ? alias : k] = values[k] + } + persistSidecarCreds(localFilePath, merged, { overwrite: true }) +} + +// Read a single KEY=VALUE from a sidecar file, or undefined if the file or key is absent. +// Uses the same createReadStream + readline path as the main config reader above. +// +// Secret-bearing keys are also accepted under their redaction-safe `*_SECRET` name, +// which wins over the legacy name when both are present and non-empty. Keys +// with no alias (XCHAIN_NODE_BLOCKS_DIR and friends) are unaffected. +async function readSidecarValue(localFilePath, key) { + if (!fs.existsSync(localFilePath)) return undefined + const alias = preferredSecretEnvName(key) + let legacyValue = undefined + let aliasValue = undefined + const stream = fs.createReadStream(localFilePath) + const rl = readline.createInterface({ input: stream, crlfDelay: Infinity }) + for await (const line of rl) { + const eqIndex = line.indexOf("=") + if (eqIndex <= 0) continue + const lineKey = line.substring(0, eqIndex) + if (lineKey === key) legacyValue = line.substring(eqIndex + 1) + else if (alias && lineKey === alias) aliasValue = line.substring(eqIndex + 1) + } + if (aliasValue !== undefined && aliasValue !== '') return aliasValue + return legacyValue +} + +// Tell the operator, once per config load, which of their secret-bearing keys sit under a +// name automatic redaction does not match. Better to hear it from your own node +// than from a transcript that printed the value. +function warnDeprecatedSecretNames(config, filePath) { + for (const { legacy, preferred } of deprecatedSecretEnvNames(config)) { + logger.warn(`Warning: ${legacy} is a deprecated name that automatic secret redaction does not match; ` + + `rename it to ${preferred} in ${filePath} (its value prints in full whenever the file is read)`) + } +} + +// Path of the shared hub sidecar. Not per coin/network: the hub is one service on the +// host and its credentials are shared by every stack that talks to it. +function hubSidecarPath() { + return path.resolve(configDir, "hub.local") +} + +/** + * Make sure this host has a HUB_API_KEY on disk, generating one on first use. + * + * A hub refuses to boot with no key unless keyless operation is explicitly declared, so + * `validator init` has to leave a credential behind or the documented onboarding path + * ends in a node that cannot start. Read-or-generate against the same shared 0600 + * sidecar the hub DB password uses: one host, one hub credential, and every service + * that authenticates to this hub reads it from that file. + * + * An existing key is REUSED and never rotated, including under `validator init --force` + * (which regenerates the signing key). The API key is already configured into indexers + * and explorers that write to this hub, so minting a new one behind their back would + * 401 all of them. + * + * Returns the sidecar PATH and whether it just generated, never the key itself: callers + * report where the credential lives, and no caller has a reason to print it. + * + * @returns {Promise<{path: string, generated: boolean}>} + */ +async function ensureHubApiKey() { + const sidecarPath = hubSidecarPath() + const existing = await readSidecarValue(sidecarPath, "HUB_API_KEY") + if (existing) return { path: sidecarPath, generated: false } + // 32 bytes: the strength the runbook told operators to mint by hand. + upsertSidecarValues(sidecarPath, { HUB_API_KEY: crypto.randomBytes(32).toString('hex') }) + return { path: sidecarPath, generated: true } +} + +/** + * Report whether this host already holds a HUB_API_KEY, WITHOUT ever minting one. + * + * A credential APPEARING is as breaking as one disappearing. A hub deployed with no key + * runs keyless (HUB_ALLOW_UNAUTHENTICATED), and every indexer, explorer and shared service + * pointed at it carries no key either; a key landing in this sidecar flips the hub to + * authenticated on its next deploy and 401s all of them at once, while the hub itself still + * looks healthy. So the callers that only need to SAY where the credential lives (a re-run + * of `validator init` over an already-provisioned node) read through here, and generation + * stays with the fresh-install path in ensureHubApiKey. + * + * @returns {Promise<{path: string, present: boolean}>} + */ +async function readHubApiKey() { + const sidecarPath = hubSidecarPath() + const existing = await readSidecarValue(sidecarPath, "HUB_API_KEY") + return { path: sidecarPath, present: !!existing } +} + +// Fill in HUB_API_KEY from the shared sidecar when the host env did not supply one. +// The hub, the co-located indexer and the shared services must all present the SAME +// value or their writes 401 against each other, so they resolve it from one file. +// Host env still wins, and this NEVER mints: generation belongs to `validator init`, +// so a standalone install with no validator stays keyless exactly as before. +async function applyHubApiKeyFromSidecar(target) { + if (target["HUB_API_KEY"] !== undefined && target["HUB_API_KEY"] !== "") return + const key = await readSidecarValue(hubSidecarPath(), "HUB_API_KEY") + if (key) target["HUB_API_KEY"] = key +} + +// Read the config file for this coin/network pair. Non-secret operator overrides live +// in the main config file; runtime RPC credentials live in a separate, untracked +// -.local sidecar so the main file can be diffed/shared without ever +// carrying rpcuser/rpcpassword. +// Track whether the main file still carries credentials so legacy installs can be migrated. +// Recover RPC credentials glued onto the tail of a preceding setting by +// an older appender that wrote NODE_USER=/NODE_PASSWORD= with no leading +// newline (e.g. `DUST_AMOUNT=546NODE_USER=`). Peel each credential +// off the value tail, password first so a double-glue +// `...NODE_USER=NODE_PASSWORD=

` resolves cleanly, so the real value +// is uncorrupted and mainFileHasCreds arms the migration below, which +// relocates the credential to the sidecar and strips it from this file. +// NODE_SECRET is the redaction-safe spelling of NODE_PASSWORD. +// It arms the same migration: a credential in the MAIN config file gets +// relocated to the sidecar whichever name it arrived under. +// Accept every secret-bearing key under its redaction-safe `*_SECRET` name and +// fold it onto the canonical legacy name here, at the one place config enters +// the process, so nothing downstream (container env, DB provisioner, RPC +// connectors) has to learn a second spelling. Runs BEFORE the migration +// and generation steps below, which key off the canonical names. +// +// PER FILE, not on the merged result. Merging first would make a renamed key in +// the sidecar and the legacy key left behind in the main config file look like +// one file contradicting itself, and the "both names, different values" refusal +// would fire on what is really just the ordinary sidecar-wins precedence. +// Credentials from the sidecar take precedence over anything in the main file. +// One-time migration for legacy installs: older versions appended NODE_USER / +// NODE_PASSWORD into the main config file alongside non-secret settings. Move the +// credentials into the sidecar and strip them from the main file so the two never +// share a file again. Existing creds keep working; they are simply relocated. +// Generate and persist to the sidecar whichever RPC credential is missing, +// evaluated PER KEY. The old both-absent (&&) guard meant a partial sidecar (one +// key present, one absent) generated nothing, and the missing half then silently +// resolved to the static "rpc" default via the merge below, leaving a well-known +// default credential on a live stack with no operator signal. +function coinConfigPaths(coin, network) { + const configFilePath = path.resolve(configDir, `${coin}-${network}`) + if (!configFilePath.startsWith(path.resolve(configDir) + path.sep) && configFilePath !== path.resolve(configDir)) { + throw new Error('Config path traversal detected') + } + return { configFilePath, localFilePath: configFilePath + ".local" } +} + +function readMainConfigLine(defaultConfig, line) { + const eqIndex = line.indexOf("=") + if (eqIndex <= 0) return false + const key = line.substring(0, eqIndex) + let value = line.substring(eqIndex + 1) + let hasCredentials = false + for (const credKey of ["NODE_PASSWORD", "NODE_USER"]) { + const at = value.indexOf(credKey + "=") + if (at >= 0) { + defaultConfig[credKey] = value.substring(at + credKey.length + 1) + value = value.substring(0, at) + hasCredentials = true + } + } + defaultConfig[key] = value + return hasCredentials || key === "NODE_USER" || key === "NODE_PASSWORD" || key === "NODE_SECRET" +} + +function normalizeMainConfig(defaultConfig, configFilePath) { + warnDeprecatedSecretNames(defaultConfig, configFilePath) + foldSecretEnvAliases(defaultConfig) +} + +function readSidecarConfigLine(sidecarConfig, line) { + const eqIndex = line.indexOf("=") + if (eqIndex > 0) sidecarConfig[line.substring(0, eqIndex)] = line.substring(eqIndex + 1) +} + +function mergeSidecarConfig(defaultConfig, sidecarConfig, localFilePath) { + warnDeprecatedSecretNames(sidecarConfig, localFilePath) + Object.assign(defaultConfig, foldSecretEnvAliases(sidecarConfig)) +} + +function migrateMainCredentials(defaultConfig, configFilePath, localFilePath) { + const creds = {} + if ("NODE_USER" in defaultConfig) creds["NODE_USER"] = defaultConfig["NODE_USER"] + if ("NODE_PASSWORD" in defaultConfig) creds["NODE_PASSWORD"] = defaultConfig["NODE_PASSWORD"] + persistSidecarCreds(localFilePath, creds, { overwrite: true }) + const remaining = [] + for (const key in defaultConfig) { + if (key !== "NODE_USER" && key !== "NODE_PASSWORD") remaining.push(`${key}=${defaultConfig[key]}`) + } + fs.writeFileSync(configFilePath, remaining.length ? remaining.join("\n") + "\n" : "") +} + +function generateRpcCredentials(defaultConfig, localFilePath) { + const generated = {} + if (!("NODE_USER" in defaultConfig)) { + generated["NODE_USER"] = defaultConfig["NODE_USER"] = crypto.randomBytes(12).toString('hex') + } + if (!("NODE_PASSWORD" in defaultConfig)) { + generated["NODE_PASSWORD"] = defaultConfig["NODE_PASSWORD"] = crypto.randomBytes(24).toString('hex') + } + if (Object.keys(generated).length) persistSidecarCreds(localFilePath, generated) +} + +module.exports = { + configure, persistSidecarCreds, upsertSidecarValues, readSidecarValue, + warnDeprecatedSecretNames, hubSidecarPath, ensureHubApiKey, readHubApiKey, + applyHubApiKeyFromSidecar, coinConfigPaths, readMainConfigLine, normalizeMainConfig, + readSidecarConfigLine, mergeSidecarConfig, migrateMainCredentials, generateRpcCredentials +} diff --git a/src/services/config_service/validation.js b/src/services/config_service/validation.js new file mode 100644 index 0000000..0e58b2e --- /dev/null +++ b/src/services/config_service/validation.js @@ -0,0 +1,30 @@ +/********************************************************************* + * + * Copyright © 2025–2026 Dankest, LLC + * Based on XChain Platform by Dankest, LLC – https://dankest.llc + * + * SPDX-License-Identifier: AGPL-3.0-or-later + * + * This file is part of XChain Platform. Licensed under the GNU Affero + * General Public License v3.0 or later; see LICENSE.md. A commercial + * license (without AGPL source-disclosure terms) is available - + * contact legal@dankest.llc. + * + ********************************************************************** + * Config Service Validation + ********************************************************************/ + +'use strict' + +function validatePort(value) { + if (typeof value === 'number') { + return Number.isInteger(value) && value >= 1 && value <= 65535 + } + if (typeof value === 'string' && /^\d+$/.test(value)) { + const port = parseInt(value, 10) + return port >= 1 && port <= 65535 + } + return false +} + +module.exports = { validatePort } From 1a04b226a8e321573ead57e79d80414bd56a8f32 Mon Sep 17 00:00:00 2001 From: J-Dog Date: Thu, 17 Sep 2026 13:11:22 -0700 Subject: [PATCH 07/35] fix(config): separate hub config polling from feed routing Render a private HUB_CONFIG_URL independently of the indexer feed endpoint. Preserve the private hub key for config polls while allowing per-stack credential overrides. --- src/services/config_service/defaults.js | 13 ++++--- src/services/config_service/services.js | 15 ++++---- .../service_routes.test.js | 36 +++++++++++++++++++ 3 files changed, 51 insertions(+), 13 deletions(-) diff --git a/src/services/config_service/defaults.js b/src/services/config_service/defaults.js index 6b93e25..d8de5b3 100644 --- a/src/services/config_service/defaults.js +++ b/src/services/config_service/defaults.js @@ -39,13 +39,11 @@ function configure(dependencies) { // (test/initialCheck.test.js), not INDEXER_DB_HOST/PORT. Default them here so // the EXTERNAL_DB rewrite below can repoint them; on a host-native-DB box the // docker DNS name "mariadb" doesn't resolve and the suite fails at bootstrap. -// The indexer's hub client keys ENTIRELY off HUB_API_URL (hub_client.js: -// `this.enabled = !!this.hubUrl`). HUB_API_HOST above is set but read by -// nothing in xchain-indexer, so without this the client stayed disabled on -// every installed stack and no push ever left the indexer: chain tips, -// config, and in particular the PRICE v1 oracle_price pushes that a FIAT -// dispenser later prices against. Prod sets HUB_API_URL by hand in the -// per-coin config file, which is why this went unnoticed. +// HUB_API_URL is the indexer's write endpoint. An operator may point it at a +// validator feed, so the config poll gets its own URL for the managed private +// hub. Otherwise getallconfigs follows the feed override to a port that does +// not expose private methods. A per-coin HUB_CONFIG_URL still overrides this +// default during the config-file merge below. // // Composed from the same container name + port as HUB_API_HOST/HUB_PORT, so // it resolves on the docker network exactly as the sibling *_API_HOST vars @@ -106,6 +104,7 @@ function createCoinDefaults(module, coin, network) { "HUB_API_HOST": getDockerContainerImageName(HUB_MODULE_NAME, "", ""), "HUB_PORT": 10000, "HUB_API_URL": "http://" + getDockerContainerImageName(HUB_MODULE_NAME, "", "") + ":10000", + "HUB_CONFIG_URL": "http://" + getDockerContainerImageName(HUB_MODULE_NAME, "", "") + ":10000", "HUB_DB_HOST": "mariadb", "HUB_DB_PORT": 3306, "HUB_DB_USER": "xchain" + DB_SEP + "hub", diff --git a/src/services/config_service/services.js b/src/services/config_service/services.js index 70559c1..085c607 100644 --- a/src/services/config_service/services.js +++ b/src/services/config_service/services.js @@ -179,12 +179,12 @@ function configureE2e(defaultValues, module, coin) { // path here the indexer side of the admission-map mirror can never be armed on // regtest (row 24x), and it must arm together with the hub's copy above or the // admission-era canonical refuses a legacy-map row and halts the block loop. -// The indexer pushes chain tips / config to the hub (HUB_API_URL); when that -// hub enforces HUB_API_KEY, the indexer must present the same key or its writes -// 401. Sourced from host env (.env) so it persists across `update`, then from the -// shared hub sidecar so an indexer co-located with a validator hub picks up the -// key `validator init` generated. Neither set leaves the indexer sending no key -// (keyless, the prior default). +// The indexer pushes chain tips to HUB_API_URL; when that hub enforces +// HUB_API_KEY, the indexer must present the same key or its writes 401. +// Sourced from host env so it persists across `update`, then from the shared +// hub sidecar so an indexer co-located with a private hub picks up its key. +// Preserve that private credential separately for getallconfigs before a +// per-coin sidecar can override HUB_API_KEY with the feed credential. function configureIndexerBeforeHubKey(defaultValues, module, network) { if (module === XChainService.XCHAIN_INDEXER) { const genesisPassthroughVars = [ @@ -277,6 +277,9 @@ function configureIndexerBeforeHubKey(defaultValues, module, network) { // resolveWatermarkGrace, regtest-overridable only). function configureIndexerAfterHubKey(defaultValues, module, network) { if (module === XChainService.XCHAIN_INDEXER) { + if (defaultValues.HUB_API_KEY) { + defaultValues.HUB_CONFIG_API_KEY = defaultValues.HUB_API_KEY + } if (config.INDEXER_API_KEY !== undefined && config.INDEXER_API_KEY !== "") { defaultValues.INDEXER_API_KEY = config.INDEXER_API_KEY } else if (network === Network.REGTEST) { diff --git a/test/unit/config_service.test/service_routes.test.js b/test/unit/config_service.test/service_routes.test.js index f3077f4..d504760 100644 --- a/test/unit/config_service.test/service_routes.test.js +++ b/test/unit/config_service.test/service_routes.test.js @@ -76,6 +76,36 @@ function serviceRoutes1() { }) } +function hubSplitRoutes() { + it('keeps config polling on the private hub when HUB_API_URL points at a feed', async function () { + const cs = makeServiceWithConfig('HUB_API_URL=http://feed:10002\n') + const config = await cs.getDefaultConfig('xchain-indexer', 'bitcoin', 'testnet') + expect(config['HUB_API_URL']).to.equal('http://feed:10002') + expect(config['HUB_CONFIG_URL']).to.equal( + 'http://' + config['HUB_API_HOST'] + ':' + config['HUB_PORT']) + }) + + it('uses the private hub key for config polling while preserving a feed key override', async function () { + const { cs } = makeMemoryConfigService({ + [coinMain]: 'HUB_API_URL=http://feed:10002\n', + [coinSidecar]: 'HUB_API_KEY=feed-key-fixture\n', + [hubSidecar]: 'HUB_API_KEY=private-key-fixture\n' + }) + const config = await cs.getDefaultConfig('xchain-indexer', 'bitcoin', 'mainnet') + expect(config['HUB_API_KEY']).to.equal('feed-key-fixture') + expect(config['HUB_CONFIG_API_KEY']).to.equal('private-key-fixture') + }) + + it('honours an explicit config-poll key from the credential sidecar', async function () { + const { cs } = makeMemoryConfigService({ + [coinSidecar]: 'HUB_API_KEY=feed-key-fixture\nHUB_CONFIG_API_KEY=config-key-fixture\n', + [hubSidecar]: 'HUB_API_KEY=private-key-fixture\n' + }) + const config = await cs.getDefaultConfig('xchain-indexer', 'bitcoin', 'mainnet') + expect(config['HUB_CONFIG_API_KEY']).to.equal('config-key-fixture') + }) +} + function serviceRoutes2() { it('does not include REGTEST_MINER_URL for testnet', async function () { const cs = makeServiceWithConfig('') @@ -134,3 +164,9 @@ describe('ConfigService', function () { describe('with coin and network (coin-specific config)', serviceRoutes2) }) }) + +describe('ConfigService', function () { + describe('getDefaultConfig()', function () { + describe('hub feed and private config routes', hubSplitRoutes) + }) +}) From 2b2577716519d341f34b008b1b1364a94b59552e Mon Sep 17 00:00:00 2001 From: J-Dog Date: Thu, 17 Sep 2026 10:44:43 -0700 Subject: [PATCH 08/35] refactor: split validator stake operations --- src/services/validator_stake_service.js | 255 +----------------- .../stake_operations.js | 205 ++++++++++++++ .../unstake_operations.js | 117 ++++++++ 3 files changed, 329 insertions(+), 248 deletions(-) create mode 100644 src/services/validator_stake_service/stake_operations.js create mode 100644 src/services/validator_stake_service/unstake_operations.js diff --git a/src/services/validator_stake_service.js b/src/services/validator_stake_service.js index 4f99fc3..d59953c 100644 --- a/src/services/validator_stake_service.js +++ b/src/services/validator_stake_service.js @@ -291,256 +291,15 @@ function openValidatorSession(opts, deps) { return { settings, network, coins, pubkey, sdk, session, address: session.address } } -/** - * Run the unstake command: withdraw this validator's stake and leave the set. - * - * The counterpart to staking, and it matters more than it looks. Membership is - * derived from chain stake alone, so a validator that has staked but is not - * running COUNTS toward every capability's N while contributing nothing, which - * raises the federation's quorum threshold (CapabilitySnapshot.getQuorum) and - * puts a hub that cannot answer into publisher elections. Standing down is how - * an operator stops being that. - */ -async function unstakeValidator(opts = {}, deps = {}) { - const log = deps.log || console.log - const { network, coins, pubkey, sdk, session, address } = openValidatorSession(opts, deps) - const timing = stakeTiming(coins, network) - - // Read the set rather than a per-pubkey lookup (see readChainState): the - // SDK exposes no getValidator, and a lookup failure must not read as - // "nothing staked" when that is the very thing being acted on. - let active = null - try { - const v = await sdk.explorer.getValidators() - active = ((v && v.data) || []).find(r => r && r.status === 'valid' && - String(r.signing_pubkey || '').toLowerCase() === pubkey) || null - } catch (e) { - throw fail('could not read the validator set (' + e.message + '), so this run cannot tell ' + - 'whether there is a stake to withdraw. Nothing was sent.') - } - - log('') - log('Validator unstake plan') - log(' signing pubkey : ' + pubkey) - log(' stake address : ' + address) - - if (!active) { - log('') - log(' This pubkey carries no valid stake. Nothing to withdraw.') - log('') - return { unstaked: false, nothingStaked: true } - } - - log(' active stake : ' + active.amount + ' ' + STAKE_TICK + - ' (action ' + active.action_index + ', activated at block ' + active.activation_block + ')') - log('') - log(' Steps:') - log(' UNSTAKE v0: withdraw the full stake for this pubkey') - log('') - // Two clocks, printed together on purpose. Leaving the active set and getting - // the coins back are different events an order of magnitude or two apart, and - // an operator told only the first one plans an hour and waits a week. - log(' Two clocks start at the block this lands in, and they are far apart:') - log(' active set: ' + timing.activationBlocks + ' more blocks' + paren(timing.activationFor) + - '. Until then the stake keeps') - log(' counting toward every capability; then it drops out.') - log(' cooldown : ' + timing.cooldownBlocks + ' blocks' + paren(timing.cooldownFor) + - '. The ' + active.amount + ' ' + STAKE_TICK + ' stays locked') - log(' until the cooldown sweep credits it back, and is NOT') - log(' spendable before then.') - - if (!opts.broadcast) { - log('') - log(' Dry run: nothing sent. Re-run with --broadcast to withdraw.') - log('') - return { unstaked: false, dryRun: true, active } - } - - const timeoutMin = Number(opts.timeout) - const timeoutMs = (Number.isFinite(timeoutMin) && timeoutMin > 0 ? timeoutMin : 120) * 60 * 1000 - const enc = {} - if (opts.feePerKb) enc.feePerKb = Number(opts.feePerKb) - - log('') - log(' Sending UNSTAKE v0...') - const r = await session.submit({ action: 'UNSTAKE', params: { VERSION: 0, SIGNING_PUBKEY: pubkey } }, enc, - { waitForIndexer: opts.wait !== false, timeout: timeoutMs, pollInterval: 15000 }) - log(' txid ' + r.txid + (opts.wait !== false ? ' (indexed)' : ' (broadcast)')) - log('') - log(' Unstaked. You leave the active set ' + timing.activationBlocks + ' blocks' + - paren(timing.activationFor) + ' after the block') - log(' this landed in; until then the federation still counts you, which is why standing') - log(' down is not instant.') - log(' Your ' + active.amount + ' ' + STAKE_TICK + ' stays locked for ' + timing.cooldownBlocks + - ' blocks' + paren(timing.cooldownFor) + ' from that same') - log(' block, then the cooldown sweep credits it back and it is spendable. Do not plan') - log(' around having it sooner.') - log(' Watch it at ' + explorerUrl(coins, 'validator/' + pubkey)) - log('') - return { unstaked: true, txid: r.txid } -} - -/** - * Run the stake command. `deps` lets tests inject an SDK factory and a - * logger; production uses the real SDK and console. - */ -async function stakeValidator(opts = {}, deps = {}) { - const log = deps.log || console.log - const { network, coins, pubkey, sdk, session, address } = openValidatorSession(opts, deps) - const amount = parseInt(opts.amount) || DEFAULT_STAKE_AMOUNT - const timing = stakeTiming(coins, network) - - const state = await readChainState(sdk, address, pubkey) - const plan = planMints(network, state.tokenBal, amount, state.mintMax, state.mintAddressMax) +const { createStakeValidator } = require('./validator_stake_service/stake_operations') +const { createUnstakeValidator } = require('./validator_stake_service/unstake_operations') - log('') - log('Validator stake plan (' + network + ')') - log(' signing pubkey : ' + pubkey) - log(' stake address : ' + address) - log(' ' + coins.stakeCoin.padEnd(15) + ': ' + state.coinBal + ' confirmed' + - (state.coinPending ? ' (+' + state.coinPending + ' pending)' : '') + ' (pays the transaction fees)') - log(' ' + STAKE_TICK.padEnd(15) + ': ' + state.tokenBal + ' held, ' + amount + ' to stake' + - (plan.short ? ', short ' + plan.short : '')) - if (state.mintMax) log(' faucet caps : ' + state.mintMax + ' per MINT, ' + (state.mintAddressMax || 'no') + ' per address') - - if (state.existing) { - log('') - log(' This pubkey already carries a valid STAKE of ' + state.existing.amount + - ' (action ' + state.existing.action_index + ', activates at block ' + state.existing.activation_block + ').') - log(' Nothing to do. Check it at ' + explorerUrl(coins, 'validator/' + pubkey)) - log('') - return { staked: false, existing: state.existing } - } - - if (state.existingUnknown) { - log('') - log(' WARNING: could not read the validator set (' + state.existingUnknown + '),') - log(' so this run cannot tell whether the pubkey is already staked. Check') - log(' ' + explorerUrl(coins, 'validator/' + pubkey) + ' before broadcasting.') - } - - const steps = plan.mints.map((a, i) => 'MINT ' + (i + 1) + '/' + plan.mints.length + ': ' + a + ' ' + STAKE_TICK) - steps.push('STAKE v1: ' + amount + ' ' + STAKE_TICK + ' to ' + pubkey) - log('') - log(' Steps:') - for (const s of steps) log(' ' + s) - - // The exit cost, stated before the money moves rather than after. Getting the - // stake back is the cooldown clock, not the activation clock, and it is the - // one that decides whether this XCHAIN is reachable next week. - log('') - log(' The ' + amount + ' ' + STAKE_TICK + ' is escrowed for as long as you stay staked. Standing down') - log(' later frees it only after a cooldown of ' + timing.cooldownBlocks + ' blocks' + - paren(timing.cooldownFor) + ', on top of the') - log(' ' + timing.activationBlocks + ' blocks it takes to leave the active set. Do not stake ' + - STAKE_TICK + ' you may') - log(' need before then.') - - const blockers = [] - if (plan.reason) blockers.push(plan.reason) - if (state.coinBal <= 0) blockers.push('no confirmed ' + coins.stakeCoin + ' at ' + address + ' to pay fees; fund it first.') - if (blockers.length) { - log('') - for (const b of blockers) log(' BLOCKED: ' + b) - log('') - return { staked: false, plan, blockers } - } - - if (!opts.broadcast) { - log('') - log(' Dry run: nothing sent. Re-run with --broadcast to send the ' + steps.length + ' transaction(s) above.') - log(' They go out back to back, each funded by the one before it, so the run confirms in the') - log(' next block or two rather than costing a block per step. (--serialize sends them a block') - log(' apart instead.)') - log('') - return { staked: false, plan, dryRun: true } - } - - // Long waits, deliberately: the indexer sees an action only once its block - // is mined, and a testnet block can take twenty minutes. Parsed as a real - // number rather than an integer, because parseInt('0.5') is 0, which would - // silently fall through to the 120-minute default for every sub-minute - // value instead of honouring it. - const timeoutMin = Number(opts.timeout) - const timeoutMs = (Number.isFinite(timeoutMin) && timeoutMin > 0 ? timeoutMin : 120) * 60 * 1000 - const baseEncoder = {} - if (opts.feePerKb) baseEncoder.feePerKb = Number(opts.feePerKb) - - // Every action is sent back to back and funded from the one before it, so - // the whole run lands in a single block (see chainedInputs). --serialize - // restores the old one-action-per-block behaviour, which costs a block per - // step and is only worth it if chaining ever misbehaves on a venue. - const chain = !opts.serialize - const sent = [] - let prevTxid = null - let chainBroken = false - - async function send(kind, params, isLast) { - const enc = Object.assign({}, baseEncoder) - if (chain && prevTxid) { - const inputs = await chainedInputs(sdk, address, prevTxid, opts.chainTimeoutMs) - if (inputs) enc.utxos = inputs - else { - // Nothing spendable came back from the previous transaction, so - // the ordering guarantee is gone. Say so rather than broadcast a - // STAKE that a miner may place ahead of its own funding. - chainBroken = true - log(' (no spendable output from ' + prevTxid.slice(0, 16) + '..., cannot chain)') - } - } - // Only the final action waits for the indexer: the ones before it are - // ordered by the funding chain, so waiting on them buys nothing but a - // block of latency. - const wait = isLast && opts.wait !== false - const r = await session.submit({ action: kind, params }, enc, - { waitForIndexer: wait, timeout: timeoutMs, pollInterval: 15000 }) - if (chain && prevTxid && !(r.spentInputs || []).some(i => i.txid === prevTxid)) chainBroken = true - prevTxid = r.txid - sent.push({ step: kind, txid: r.txid }) - log(' txid ' + r.txid + (wait ? ' (indexed)' : ' (broadcast)')) - return r - } - - for (let i = 0; i < plan.mints.length; i++) { - log('') - log(' Sending MINT ' + (i + 1) + '/' + plan.mints.length + ' (' + plan.mints[i] + ' ' + STAKE_TICK + ')...') - await send('MINT', { VERSION: 0, TICK: STAKE_TICK, AMOUNT: String(plan.mints[i]) }, opts.serialize === true) - } - - // A broken chain means in-block order is the miner's choice, and a STAKE - // evaluated before its own mints is rejected for insufficient funds. Wait - // the mints out instead: once they are indexed, ordering stops mattering. - if (plan.mints.length && (chainBroken || opts.serialize)) { - log('') - log(' Waiting for the mints to be indexed before staking' + - (chainBroken ? ' (the funding chain broke, so in-block order is not guaranteed)' : '') + '...') - const ok = await waitForBalance(sdk, address, amount, timeoutMs, log, opts.balancePollMs) - if (!ok) { - log('') - log(' The mints have not indexed within the timeout. They are broadcast and will confirm;') - log(' re-run this command to send the STAKE once they do.') - log('') - return { staked: false, sent, pendingMints: true } - } - prevTxid = null // fund the STAKE freely; ordering no longer matters - } - - log('') - log(' Sending STAKE v1 (' + amount + ' ' + STAKE_TICK + ' to ' + pubkey + ')...') - const r = await send('STAKE', { VERSION: 1, AMOUNT: String(amount), SIGNING_PUBKEY: pubkey }, true) - log('') - if (opts.wait === false) { - log(' Broadcast. Watch it land at ' + explorerUrl(coins, 'validator/' + pubkey)) - } else { - log(' Staked. The stake activates ' + timing.activationBlocks + ' blocks' + paren(timing.activationFor) + - ' after it is indexed; peers') - log(' admit you on their next signer-set refresh after that.') - log(' Watch it at ' + explorerUrl(coins, 'validator/' + pubkey)) - } - log(' Next: xchain-node install master xchain-hub') - log('') - return { staked: true, sent, txid: r.txid, chained: chain && !chainBroken } +const operationHelpers = { + openValidatorSession, stakeTiming, readChainState, planMints, explorerUrl, + chainedInputs, waitForBalance, fail, paren, STAKE_TICK, DEFAULT_STAKE_AMOUNT } +const stakeValidator = createStakeValidator(operationHelpers) +const unstakeValidator = createUnstakeValidator(operationHelpers) module.exports = { stakeValidator, unstakeValidator, planMints, stakeTiming diff --git a/src/services/validator_stake_service/stake_operations.js b/src/services/validator_stake_service/stake_operations.js new file mode 100644 index 0000000..5b8ead3 --- /dev/null +++ b/src/services/validator_stake_service/stake_operations.js @@ -0,0 +1,205 @@ +/********************************************************************* + * + * Copyright © 2025–2026 Dankest, LLC + * Based on XChain Platform by Dankest, LLC – https://dankest.llc + * + * SPDX-License-Identifier: AGPL-3.0-or-later + * + * This file is part of XChain Platform. Licensed under the GNU Affero + * General Public License v3.0 or later; see LICENSE.md. A commercial + * license (without AGPL source-disclosure terms) is available - + * contact legal@dankest.llc. + * + ********************************************************************** + * XChain Node - Validator Stake Operations + ********************************************************************/ + +function logStakeBalances(log, network, coins, pubkey, address, amount, state, plan, STAKE_TICK) { + log('') + log('Validator stake plan (' + network + ')') + log(' signing pubkey : ' + pubkey) + log(' stake address : ' + address) + log(' ' + coins.stakeCoin.padEnd(15) + ': ' + state.coinBal + ' confirmed' + + (state.coinPending ? ' (+' + state.coinPending + ' pending)' : '') + ' (pays the transaction fees)') + log(' ' + STAKE_TICK.padEnd(15) + ': ' + state.tokenBal + ' held, ' + amount + ' to stake' + + (plan.short ? ', short ' + plan.short : '')) + if (state.mintMax) log(' faucet caps : ' + state.mintMax + ' per MINT, ' + (state.mintAddressMax || 'no') + ' per address') +} + +function logStakeSteps(log, amount, pubkey, plan, timing, STAKE_TICK, paren) { + const steps = plan.mints.map((a, i) => 'MINT ' + (i + 1) + '/' + plan.mints.length + ': ' + a + ' ' + STAKE_TICK) + steps.push('STAKE v1: ' + amount + ' ' + STAKE_TICK + ' to ' + pubkey) + log('') + log(' Steps:') + for (const s of steps) log(' ' + s) + + // The exit cost, stated before the money moves rather than after. Getting the + // stake back is the cooldown clock, not the activation clock, and it is the + // one that decides whether this XCHAIN is reachable next week. + log('') + log(' The ' + amount + ' ' + STAKE_TICK + ' is escrowed for as long as you stay staked. Standing down') + log(' later frees it only after a cooldown of ' + timing.cooldownBlocks + ' blocks' + + paren(timing.cooldownFor) + ', on top of the') + log(' ' + timing.activationBlocks + ' blocks it takes to leave the active set. Do not stake ' + + STAKE_TICK + ' you may') + log(' need before then.') + return steps +} + +function stakeBlockers(state, plan, coins, address) { + const blockers = [] + if (plan.reason) blockers.push(plan.reason) + if (state.coinBal <= 0) blockers.push('no confirmed ' + coins.stakeCoin + ' at ' + address + ' to pay fees; fund it first.') + return blockers +} + +async function sendStakeAction(context, progress, kind, params, isLast) { + const { sdk, session, address, opts, log, timeoutMs, baseEncoder, chainedInputs } = context + const enc = Object.assign({}, baseEncoder) + if (progress.chain && progress.prevTxid) { + const inputs = await chainedInputs(sdk, address, progress.prevTxid, opts.chainTimeoutMs) + if (inputs) enc.utxos = inputs + else { + // Nothing spendable came back from the previous transaction, so + // the ordering guarantee is gone. Say so rather than broadcast a + // STAKE that a miner may place ahead of its own funding. + progress.chainBroken = true + log(' (no spendable output from ' + progress.prevTxid.slice(0, 16) + '..., cannot chain)') + } + } + // Only the final action waits for the indexer: the ones before it are + // ordered by the funding chain, so waiting on them buys nothing but a + // block of latency. + const wait = isLast && opts.wait !== false + const r = await session.submit({ action: kind, params }, enc, + { waitForIndexer: wait, timeout: timeoutMs, pollInterval: 15000 }) + if (progress.chain && progress.prevTxid && !(r.spentInputs || []).some(i => i.txid === progress.prevTxid)) { + progress.chainBroken = true + } + progress.prevTxid = r.txid + progress.sent.push({ step: kind, txid: r.txid }) + log(' txid ' + r.txid + (wait ? ' (indexed)' : ' (broadcast)')) + return r +} + +async function broadcastStake(context) { + const { opts, sdk, address, amount, plan, log, waitForBalance, STAKE_TICK } = context + // Long waits, deliberately: the indexer sees an action only once its block + // is mined, and a testnet block can take twenty minutes. Parsed as a real + // number rather than an integer, because parseInt('0.5') is 0, which would + // silently fall through to the 120-minute default for every sub-minute + // value instead of honouring it. + const timeoutMin = Number(opts.timeout) + context.timeoutMs = (Number.isFinite(timeoutMin) && timeoutMin > 0 ? timeoutMin : 120) * 60 * 1000 + context.baseEncoder = {} + if (opts.feePerKb) context.baseEncoder.feePerKb = Number(opts.feePerKb) + + // Every action is sent back to back and funded from the one before it, so + // the whole run lands in a single block (see chainedInputs). --serialize + // restores the old one-action-per-block behaviour, which costs a block per + // step and is only worth it if chaining ever misbehaves on a venue. + const progress = { chain: !opts.serialize, sent: [], prevTxid: null, chainBroken: false } + for (let i = 0; i < plan.mints.length; i++) { + log('') + log(' Sending MINT ' + (i + 1) + '/' + plan.mints.length + ' (' + plan.mints[i] + ' ' + STAKE_TICK + ')...') + await sendStakeAction(context, progress, 'MINT', + { VERSION: 0, TICK: STAKE_TICK, AMOUNT: String(plan.mints[i]) }, opts.serialize === true) + } + + // A broken chain means in-block order is the miner's choice, and a STAKE + // evaluated before its own mints is rejected for insufficient funds. Wait + // the mints out instead: once they are indexed, ordering stops mattering. + if (plan.mints.length && (progress.chainBroken || opts.serialize)) { + log('') + log(' Waiting for the mints to be indexed before staking' + + (progress.chainBroken ? ' (the funding chain broke, so in-block order is not guaranteed)' : '') + '...') + const ok = await waitForBalance(sdk, address, amount, context.timeoutMs, log, opts.balancePollMs) + if (!ok) { + log('') + log(' The mints have not indexed within the timeout. They are broadcast and will confirm;') + log(' re-run this command to send the STAKE once they do.') + log('') + return { staked: false, sent: progress.sent, pendingMints: true } + } + progress.prevTxid = null // fund the STAKE freely; ordering no longer matters + } + + return finishStake(context, progress) +} + +async function finishStake(context, progress) { + const { opts, coins, pubkey, amount, log, timing, explorerUrl, STAKE_TICK, paren } = context + log('') + log(' Sending STAKE v1 (' + amount + ' ' + STAKE_TICK + ' to ' + pubkey + ')...') + const r = await sendStakeAction(context, progress, 'STAKE', + { VERSION: 1, AMOUNT: String(amount), SIGNING_PUBKEY: pubkey }, true) + log('') + if (opts.wait === false) { + log(' Broadcast. Watch it land at ' + explorerUrl(coins, 'validator/' + pubkey)) + } else { + log(' Staked. The stake activates ' + timing.activationBlocks + ' blocks' + paren(timing.activationFor) + + ' after it is indexed; peers') + log(' admit you on their next signer-set refresh after that.') + log(' Watch it at ' + explorerUrl(coins, 'validator/' + pubkey)) + } + log(' Next: xchain-node install master xchain-hub') + log('') + return { staked: true, sent: progress.sent, txid: r.txid, chained: progress.chain && !progress.chainBroken } +} + +function createStakeValidator(helpers) { + const { openValidatorSession, stakeTiming, readChainState, planMints, explorerUrl, + chainedInputs, waitForBalance, paren, STAKE_TICK, DEFAULT_STAKE_AMOUNT } = helpers + + /** + * Run the stake command. `deps` lets tests inject an SDK factory and a + * logger; production uses the real SDK and console. + */ + return async function stakeValidator(opts = {}, deps = {}) { + const log = deps.log || console.log + const { network, coins, pubkey, sdk, session, address } = openValidatorSession(opts, deps) + const amount = parseInt(opts.amount) || DEFAULT_STAKE_AMOUNT + const timing = stakeTiming(coins, network) + const state = await readChainState(sdk, address, pubkey) + const plan = planMints(network, state.tokenBal, amount, state.mintMax, state.mintAddressMax) + + logStakeBalances(log, network, coins, pubkey, address, amount, state, plan, STAKE_TICK) + if (state.existing) { + log('') + log(' This pubkey already carries a valid STAKE of ' + state.existing.amount + + ' (action ' + state.existing.action_index + ', activates at block ' + state.existing.activation_block + ').') + log(' Nothing to do. Check it at ' + explorerUrl(coins, 'validator/' + pubkey)) + log('') + return { staked: false, existing: state.existing } + } + if (state.existingUnknown) { + log('') + log(' WARNING: could not read the validator set (' + state.existingUnknown + '),') + log(' so this run cannot tell whether the pubkey is already staked. Check') + log(' ' + explorerUrl(coins, 'validator/' + pubkey) + ' before broadcasting.') + } + + const steps = logStakeSteps(log, amount, pubkey, plan, timing, STAKE_TICK, paren) + const blockers = stakeBlockers(state, plan, coins, address) + if (blockers.length) { + log('') + for (const b of blockers) log(' BLOCKED: ' + b) + log('') + return { staked: false, plan, blockers } + } + if (!opts.broadcast) { + log('') + log(' Dry run: nothing sent. Re-run with --broadcast to send the ' + steps.length + ' transaction(s) above.') + log(' They go out back to back, each funded by the one before it, so the run confirms in the') + log(' next block or two rather than costing a block per step. (--serialize sends them a block') + log(' apart instead.)') + log('') + return { staked: false, plan, dryRun: true } + } + + return broadcastStake({ opts, sdk, session, address, coins, pubkey, amount, plan, log, timing, + explorerUrl, chainedInputs, waitForBalance, paren, STAKE_TICK }) + } +} + +module.exports = { createStakeValidator } diff --git a/src/services/validator_stake_service/unstake_operations.js b/src/services/validator_stake_service/unstake_operations.js new file mode 100644 index 0000000..1044e77 --- /dev/null +++ b/src/services/validator_stake_service/unstake_operations.js @@ -0,0 +1,117 @@ +/********************************************************************* + * + * Copyright © 2025–2026 Dankest, LLC + * Based on XChain Platform by Dankest, LLC – https://dankest.llc + * + * SPDX-License-Identifier: AGPL-3.0-or-later + * + * This file is part of XChain Platform. Licensed under the GNU Affero + * General Public License v3.0 or later; see LICENSE.md. A commercial + * license (without AGPL source-disclosure terms) is available - + * contact legal@dankest.llc. + * + ********************************************************************** + * XChain Node - Validator Unstake Operations + ********************************************************************/ + +function logUnstakePlan(log, pubkey, address, active, timing, STAKE_TICK, paren) { + log('') + log('Validator unstake plan') + log(' signing pubkey : ' + pubkey) + log(' stake address : ' + address) + if (!active) return + + log(' active stake : ' + active.amount + ' ' + STAKE_TICK + + ' (action ' + active.action_index + ', activated at block ' + active.activation_block + ')') + log('') + log(' Steps:') + log(' UNSTAKE v0: withdraw the full stake for this pubkey') + log('') + // Two clocks, printed together on purpose. Leaving the active set and getting + // the coins back are different events an order of magnitude or two apart, and + // an operator told only the first one plans an hour and waits a week. + log(' Two clocks start at the block this lands in, and they are far apart:') + log(' active set: ' + timing.activationBlocks + ' more blocks' + paren(timing.activationFor) + + '. Until then the stake keeps') + log(' counting toward every capability; then it drops out.') + log(' cooldown : ' + timing.cooldownBlocks + ' blocks' + paren(timing.cooldownFor) + + '. The ' + active.amount + ' ' + STAKE_TICK + ' stays locked') + log(' until the cooldown sweep credits it back, and is NOT') + log(' spendable before then.') +} + +function logUnstakeSuccess(log, active, timing, coins, pubkey, STAKE_TICK, paren, explorerUrl) { + log('') + log(' Unstaked. You leave the active set ' + timing.activationBlocks + ' blocks' + + paren(timing.activationFor) + ' after the block') + log(' this landed in; until then the federation still counts you, which is why standing') + log(' down is not instant.') + log(' Your ' + active.amount + ' ' + STAKE_TICK + ' stays locked for ' + timing.cooldownBlocks + + ' blocks' + paren(timing.cooldownFor) + ' from that same') + log(' block, then the cooldown sweep credits it back and it is spendable. Do not plan') + log(' around having it sooner.') + log(' Watch it at ' + explorerUrl(coins, 'validator/' + pubkey)) + log('') +} + +function createUnstakeValidator({ + openValidatorSession, stakeTiming, fail, paren, explorerUrl, STAKE_TICK +}) { + /** + * Run the unstake command: withdraw this validator's stake and leave the set. + * + * The counterpart to staking, and it matters more than it looks. Membership is + * derived from chain stake alone, so a validator that has staked but is not + * running COUNTS toward every capability's N while contributing nothing, which + * raises the federation's quorum threshold (CapabilitySnapshot.getQuorum) and + * puts a hub that cannot answer into publisher elections. Standing down is how + * an operator stops being that. + */ + return async function unstakeValidator(opts = {}, deps = {}) { + const log = deps.log || console.log + const { network, coins, pubkey, sdk, session, address } = openValidatorSession(opts, deps) + const timing = stakeTiming(coins, network) + + // Read the set rather than a per-pubkey lookup (see readChainState): the + // SDK exposes no getValidator, and a lookup failure must not read as + // "nothing staked" when that is the very thing being acted on. + let active = null + try { + const v = await sdk.explorer.getValidators() + active = ((v && v.data) || []).find(r => r && r.status === 'valid' && + String(r.signing_pubkey || '').toLowerCase() === pubkey) || null + } catch (e) { + throw fail('could not read the validator set (' + e.message + '), so this run cannot tell ' + + 'whether there is a stake to withdraw. Nothing was sent.') + } + + logUnstakePlan(log, pubkey, address, active, timing, STAKE_TICK, paren) + if (!active) { + log('') + log(' This pubkey carries no valid stake. Nothing to withdraw.') + log('') + return { unstaked: false, nothingStaked: true } + } + if (!opts.broadcast) { + log('') + log(' Dry run: nothing sent. Re-run with --broadcast to withdraw.') + log('') + return { unstaked: false, dryRun: true, active } + } + + const timeoutMin = Number(opts.timeout) + const timeoutMs = (Number.isFinite(timeoutMin) && timeoutMin > 0 ? timeoutMin : 120) * 60 * 1000 + const enc = {} + if (opts.feePerKb) enc.feePerKb = Number(opts.feePerKb) + + log('') + log(' Sending UNSTAKE v0...') + const r = await session.submit({ action: 'UNSTAKE', params: { VERSION: 0, SIGNING_PUBKEY: pubkey } }, enc, + { waitForIndexer: opts.wait !== false, timeout: timeoutMs, pollInterval: 15000 }) + log(' txid ' + r.txid + (opts.wait !== false ? ' (indexed)' : ' (broadcast)')) + logUnstakeSuccess(log, active, timing, coins, pubkey, STAKE_TICK, paren, explorerUrl) + return { unstaked: true, txid: r.txid } + } +} + +module.exports = { createUnstakeValidator } From fcd528fa3d80cec4c2b264d1d612e8d23c6f1c79 Mon Sep 17 00:00:00 2001 From: J-Dog Date: Thu, 17 Sep 2026 13:24:12 -0700 Subject: [PATCH 09/35] test: split migration precondition suite Move migration precondition registrations into focused suite parts while preserving shared hooks. Keep title pin coverage and async execution order unchanged. --- bin/pins/suite-title-splits.json | 7 + .../migration_precondition_service.test.js | 395 +----------------- .../01_migration_metadata.test.js | 65 +++ .../02_migration_inventory.test.js | 107 +++++ .../03_applied_ledger.test.js | 112 +++++ .../04_guard_outcomes.test.js | 204 +++++++++ 6 files changed, 514 insertions(+), 376 deletions(-) create mode 100644 test/unit/migration_precondition_service.test/01_migration_metadata.test.js create mode 100644 test/unit/migration_precondition_service.test/02_migration_inventory.test.js create mode 100644 test/unit/migration_precondition_service.test/03_applied_ledger.test.js create mode 100644 test/unit/migration_precondition_service.test/04_guard_outcomes.test.js diff --git a/bin/pins/suite-title-splits.json b/bin/pins/suite-title-splits.json index df8e629..f0c035a 100644 --- a/bin/pins/suite-title-splits.json +++ b/bin/pins/suite-title-splits.json @@ -398,6 +398,13 @@ "test/unit/module_operations.test/17_module_error_paths.test.js", "test/unit/module_operations.test/18_sync_shared_services_after_install.test.js", "test/unit/module_operations.test/19_e2e_image_siblings.test.js" + ], + "test/unit/migration_precondition_service.test.js": [ + "test/unit/migration_precondition_service.test.js", + "test/unit/migration_precondition_service.test/01_migration_metadata.test.js", + "test/unit/migration_precondition_service.test/02_migration_inventory.test.js", + "test/unit/migration_precondition_service.test/03_applied_ledger.test.js", + "test/unit/migration_precondition_service.test/04_guard_outcomes.test.js" ] } } diff --git a/test/unit/migration_precondition_service.test.js b/test/unit/migration_precondition_service.test.js index 807100a..2c0d285 100644 --- a/test/unit/migration_precondition_service.test.js +++ b/test/unit/migration_precondition_service.test.js @@ -13,7 +13,6 @@ const fs = require('fs') const os = require('os') const path = require('path') -const proxyquire = require('proxyquire').noCallThru() const sinon = require('sinon') const { expect } = require('chai') @@ -28,6 +27,10 @@ const { readAppliedMigrations, assertRequiredMigrationsApplied } = require('../../src/services/migration_precondition_service') +const registerMigrationMetadata = require('./migration_precondition_service.test/01_migration_metadata.test') +const registerMigrationInventory = require('./migration_precondition_service.test/02_migration_inventory.test') +const registerAppliedLedger = require('./migration_precondition_service.test/03_applied_ledger.test') +const registerGuardOutcomes = require('./migration_precondition_service.test/04_guard_outcomes.test') const GATED = '2026-07-24-pubkeys-widen-uncompressed.sql' @@ -60,384 +63,24 @@ function makeDeps({ required = [GATED], applied = [GATED], state = 'ledger', rea describe('MigrationPreconditionService', () => { - let warnStub - beforeEach(() => { warnStub = sinon.stub(console, 'warn') }) + beforeEach(() => { registerGuardOutcomes.setWarnStub(sinon.stub(console, 'warn')) }) afterEach(() => { - warnStub.restore() + registerGuardOutcomes.restoreWarnStub() delete process.env[SKIP_ENV] }) - describe('migrationDeclaresDeployPrecondition', () => { - it('reads the tag off the xchain:migration directive line', () => { - expect(migrationDeclaresDeployPrecondition(TAGGED)).to.equal(true) - }) - it('tolerates spacing around the token', () => { - expect(migrationDeclaresDeployPrecondition('-- xchain:migration mode = manual deploy-precondition = required\nALTER TABLE t;')).to.equal(true) - }) - it('is false for an ordinary migration and for an empty file', () => { - expect(migrationDeclaresDeployPrecondition(UNTAGGED)).to.equal(false) - expect(migrationDeclaresDeployPrecondition('')).to.equal(false) - }) - it('ignores the token once the SQL body has started', () => { - // Prologue anchoring: a migration that merely DISCUSSES the convention in a - // trailing comment must not start refusing every deploy. - expect(migrationDeclaresDeployPrecondition('ALTER TABLE t;\n-- xchain:migration mode=manual deploy-precondition=required\n')).to.equal(false) - }) - it('ignores the token on a comment line that is not the directive', () => { - expect(migrationDeclaresDeployPrecondition('-- deploy-precondition=required, see the other file\nALTER TABLE t;')).to.equal(false) - }) - it('sees the tag through a long license banner', () => { - const banner = Array(30).fill('-- license line').join('\n') - expect(migrationDeclaresDeployPrecondition(banner + '\n\n' + TAGGED)).to.equal(true) - }) - }) - - describe('migrationMode', () => { - - it('reads the mode a header declares', () => { - expect(migrationMode(TAGGED)).to.equal('manual') - expect(migrationMode('-- xchain:migration mode=auto\nSELECT 1;\n')).to.equal('auto') - }) - - it('returns null when no mode is declared', () => { - expect(migrationMode('-- just a comment\nSELECT 1;\n')).to.equal(null) - }) - - it('ignores a mode token that appears after the prologue', () => { - // Body prose and data literals must not be able to answer for the file. - const body = '-- header\nSELECT 1;\n-- xchain:migration mode=auto\n' - expect(migrationMode(body)).to.equal(null) - }) - }) - - describe('pendingManualMigrations', () => { - - let dir - beforeEach(() => { - dir = fs.mkdtempSync(path.join(os.tmpdir(), 'xc-pending-')) - fs.writeFileSync(path.join(dir, 'a-manual.sql'), TAGGED) - fs.writeFileSync(path.join(dir, 'b-auto.sql'), '-- xchain:migration mode=auto\nSELECT 1;\n') - fs.writeFileSync(path.join(dir, 'c-manual.sql'), '-- xchain:migration mode=manual\nSELECT 1;\n') - }) - afterEach(() => { fs.rmSync(dir, { recursive: true, force: true }) }) - - it('lists only manual migrations the ledger has not recorded', () => { - expect(pendingManualMigrations(dir, new Set())).to.deep.equal(['a-manual.sql', 'c-manual.sql']) - }) - - it('excludes what the ledger already carries', () => { - expect(pendingManualMigrations(dir, new Set(['a-manual.sql']))).to.deep.equal(['c-manual.sql']) - }) - - it('yields nothing for a missing directory rather than throwing', () => { - expect(pendingManualMigrations(path.join(dir, 'nope'), new Set())).to.deep.equal([]) - }) - }) - - describe('listDeployPreconditionMigrations', () => { - let dir - beforeEach(() => { - dir = fs.mkdtempSync(path.join(os.tmpdir(), 'xcn-mig-')) - }) - afterEach(() => { fs.rmSync(dir, { recursive: true, force: true }) }) - - it('returns only the tagged .sql files, sorted', () => { - fs.writeFileSync(path.join(dir, '2026-07-24-b.sql'), TAGGED) - fs.writeFileSync(path.join(dir, '2026-07-01-a.sql'), TAGGED) - fs.writeFileSync(path.join(dir, '2026-07-30-c.sql'), UNTAGGED) - fs.writeFileSync(path.join(dir, 'notes.txt'), TAGGED) - expect(listDeployPreconditionMigrations(dir)).to.deep.equal(['2026-07-01-a.sql', '2026-07-24-b.sql']) - }) - - it('returns [] for a missing directory (a ref with no migrations declares nothing)', () => { - expect(listDeployPreconditionMigrations(path.join(dir, 'nope'))).to.deep.equal([]) - }) - - it('reads the REAL indexer tree and finds the migration behind the 2026-08-09 halt', function () { - // Guards the whole contract end to end: if the tag is ever dropped from the - // committed file, or the migrations path moves, this guard must stop the - // suite rather than pass silently on a directory that no longer exists. - if (!INDEXER_PRESENT) { - if (REQUIRE_SIBLINGS) - throw new Error('XCHAIN_REQUIRE_SIBLINGS=1 but xchain-indexer is not checked out at ' + INDEXER_DIR) - return this.skip() // sibling repo not checked out - } - expect(fs.existsSync(INDEXER_MIGRATIONS), 'xchain-indexer is checked out at ' + INDEXER_DIR - + ' but has no migrations directory at ' + INDEXER_MIGRATIONS).to.equal(true) - expect(listDeployPreconditionMigrations(INDEXER_MIGRATIONS)).to.include(GATED) - }) - - it('reads the REAL indexer tree and finds the bridge-tables migration', function () { - // The bridge build lands `2026-09-12-bridge-tables.sql` (mode=manual, - // deploy-precondition=required) beside the token-bridge-fields migration - // (mode=auto, no precondition). Nothing in xchain-node had to change for - // either to be covered: this guard is a directory scan of whatever the - // target tree carries, so a new deploy-precondition migration is wired in - // the moment it lands on the indexer, no xchain-node release required - // (the same "no coupled release" property MigrationPreconditionService's - // header describes for the contract as a whole). This test is the proof. - if (!INDEXER_PRESENT) { - if (REQUIRE_SIBLINGS) - throw new Error('XCHAIN_REQUIRE_SIBLINGS=1 but xchain-indexer is not checked out at ' + INDEXER_DIR) - return this.skip() // sibling repo not checked out - } - expect(fs.existsSync(INDEXER_MIGRATIONS), 'xchain-indexer is checked out at ' + INDEXER_DIR - + ' but has no migrations directory at ' + INDEXER_MIGRATIONS).to.equal(true) - const required = listDeployPreconditionMigrations(INDEXER_MIGRATIONS) - expect(required).to.include('2026-09-12-bridge-tables.sql') - expect(required).to.not.include('2026-09-12-token-bridge-fields.sql') - }) - }) - - describe('readAppliedMigrations', () => { - - const target = { database: 'XChain_BTC_Mainnet_Indexer', coin: 'bitcoin', network: 'mainnet' } - - // Fake mariadb batch-mode output: one value per COUNT query, newline-joined - // names for the ledger read - the exact shapes `-B -N` produces. - function runnerFor({ tables = 40, ledger = 1, names = [GATED] }) { - return async (sql) => { - if (/TABLE_NAME = 'schema_migrations'/.test(sql)) return String(ledger) - if (/COUNT\(\*\)/.test(sql)) return String(tables) - return names.join('\n') - } - } - - it('reads the ledger into a set of applied names', async () => { - const res = await readAppliedMigrations(target, { runner: runnerFor({ names: [GATED, '2026-08-11-attests-relay-identity-index.sql'] }) }) - expect(res.state).to.equal('ledger') - expect([...res.applied]).to.have.members([GATED, '2026-08-11-attests-relay-identity-index.sql']) - }) - - it('tolerates a ledger read that comes back empty', async () => { - const res = await readAppliedMigrations(target, { runner: runnerFor({ names: [] }) }) - expect(res.state).to.equal('ledger') - expect(res.applied.size).to.equal(0) - }) - - it('calls a database with no tables empty, not unreadable', async () => { - const res = await readAppliedMigrations(target, { runner: runnerFor({ tables: 0 }) }) - expect(res.state).to.equal('empty-database') - }) - - it('calls a populated database with no ledger table UNREADABLE, never empty', async () => { - // Waving this through would be the whole outage again: a real schema whose - // migration state nobody can see. - const res = await readAppliedMigrations(target, { runner: runnerFor({ tables: 40, ledger: 0 }) }) - expect(res.state).to.equal('unreadable') - expect(res.reason).to.contain('schema_migrations') - }) - - it('refuses an unreadable table count instead of collapsing it into empty-database', async () => { - // A count that fails to parse (a driver notice, an empty batch-mode - // reply, garbage) yields NaN. `!NaN` is true just like `!0`, so a naive - // falsy check reads an unreadable count as the same "empty database" - // verdict as a genuinely empty one - the exact bug this guards against. - const res = await readAppliedMigrations(target, { - runner: async () => 'ERROR 2013 (HY000): Lost connection to MySQL server' - }) - expect(res.state).to.equal('unreadable') - expect(res.reason).to.contain(target.database) - }) - - it('refuses an empty-string table count instead of reading it as zero', async () => { - const res = await readAppliedMigrations(target, { runner: async () => '' }) - expect(res.state).to.equal('unreadable') - }) - - it('refuses an unreadable ledger-presence count instead of reading it as "no ledger"', async () => { - const res = await readAppliedMigrations(target, { - runner: async (sql) => { - if (/TABLE_NAME = 'schema_migrations'/.test(sql)) return 'ERROR: connection reset' - if (/COUNT\(\*\)/.test(sql)) return '40' - return '' - } - }) - expect(res.state).to.equal('unreadable') - expect(res.reason).to.contain('schema_migrations') - }) - - it('turns a driver failure into unreadable instead of throwing past the guard', async () => { - // A throw here would escape assertRequiredMigrationsApplied as an opaque - // driver error, and the operator would read ECONNREFUSED with no idea a - // migration was at stake. - const res = await readAppliedMigrations(target, { - runner: async () => { throw new Error('ECONNREFUSED 127.0.0.1:13306') } - }) - expect(res.state).to.equal('unreadable') - expect(res.reason).to.contain('ECONNREFUSED') - }) - - it('refuses a database name that is not a plain identifier, before any query runs', async () => { - let called = false - const res = await readAppliedMigrations( - { database: 'x`; DROP DATABASE y; -- ', coin: 'bitcoin', network: 'mainnet' }, - { runner: async () => { called = true; return '0' } }) - expect(res.state).to.equal('unreadable') - expect(called, 'nothing may reach SQL').to.equal(false) - }) - }) - - describe('assertRequiredMigrationsApplied', () => { - - it('uses the moved default migration reader when no deps are passed', async () => { - const cloneGit = sinon.stub().resolves() - const listRequired = sinon.stub().returns([GATED]) - const readApplied = sinon.stub().resolves({ state: 'ledger', applied: new Set([GATED]) }) - const migrationScan = require('../../src/services/migration_precondition_service/migration_scan') - const service = proxyquire('../../src/services/migration_precondition_service', { - './module_service': { cloneGit }, - './migration_precondition_service/migration_scan': { - ...migrationScan, - listDeployPreconditionMigrations: listRequired, - readAppliedMigrations: readApplied - } - }) - - const res = await service.assertRequiredMigrationsApplied( - XChainService.XCHAIN_INDEXER, 'bitcoin', 'mainnet', 'master') - - expect(res.ok).to.equal(true) - expect(cloneGit.calledOnce).to.equal(true) - expect(listRequired.calledOnce).to.equal(true) - expect(readApplied.calledOnce).to.equal(true) - }) - - it('is inert for a module that ships no migrations', async () => { - const deps = makeDeps() - const res = await assertRequiredMigrationsApplied(XChainService.XCHAIN_ENCODER, 'bitcoin', 'mainnet', 'master', deps) - expect(res).to.deep.equal({ checked: false, reason: 'no-migrations' }) - expect(deps.cloneGit.called).to.equal(false) - }) - - it('covers the indexer and the decoder, the two migration-bearing modules', () => { - expect(MIGRATION_BEARING_MODULES).to.have.members([XChainService.XCHAIN_INDEXER, XChainService.XCHAIN_DECODER]) - }) - - it('proceeds, loudly, when the skip env is set', async () => { - process.env[SKIP_ENV] = '1' - const deps = makeDeps({ applied: [] }) - const res = await assertRequiredMigrationsApplied(XChainService.XCHAIN_INDEXER, 'bitcoin', 'mainnet', 'master', deps) - expect(res.reason).to.equal('skipped-by-env') - expect(warnStub.called).to.equal(true) - }) - - it('refuses when the target DB has not applied a declared precondition', async () => { - const deps = makeDeps({ applied: ['2026-07-21-anchor-reward-attestations-table.sql'] }) - let err = null - try { - await assertRequiredMigrationsApplied(XChainService.XCHAIN_INDEXER, 'bitcoin', 'mainnet', 'master', deps) - } catch (e) { err = e } - expect(err, 'the deploy must be refused').to.not.equal(null) - expect(err.message).to.contain(GATED) - expect(err.message).to.contain('XChain_BTC_Mainnet_Indexer') - // Only safe to print because this build was confirmed to honour --file. - expect(err.message).to.contain('--file ' + GATED) - }) - - it('does NOT print a scoped command the running build would ignore', async () => { - // A build without per-file targeting does not reject --file, it ignores - // it and applies every pending manual migration, so printing the command - // hands the operator a wider action than the one it describes. - const deps = makeDeps({ - applied: [], - supportsPerFile: false, - pendingManual: [GATED, '2026-08-10-action-data-utf8mb4.sql', '2026-06-13-dispensers-expiration-bigint.sql'] - }) - let err = null - try { - await assertRequiredMigrationsApplied(XChainService.XCHAIN_INDEXER, 'bitcoin', 'mainnet', 'master', deps) - } catch (e) { err = e } - expect(err, 'the deploy must be refused').to.not.equal(null) - expect(err.message).to.not.contain('docker exec') - expect(err.message).to.contain('DO NOT run') - // The whole blast radius is named, not just the file that is needed. - expect(err.message).to.contain('3 file(s)') - expect(err.message).to.contain('2026-08-10-action-data-utf8mb4.sql') - expect(err.message).to.contain('2026-06-13-dispensers-expiration-bigint.sql') - expect(err.message).to.contain('(the one you need)') - }) - - it('treats an unreadable container as lacking the capability', async () => { - // An unverified capability is not a capability: the cost of guessing - // wrong is an unauthorised migration on a live database. - const deps = makeDeps({ applied: [], supportsPerFile: null }) - let err = null - try { - await assertRequiredMigrationsApplied(XChainService.XCHAIN_INDEXER, 'bitcoin', 'mainnet', 'master', deps) - } catch (e) { err = e } - expect(err, 'the deploy must be refused').to.not.equal(null) - expect(err.message).to.not.contain('docker exec') - expect(err.message).to.contain('could not be read') - }) - - it('still refuses when the capability probe itself throws', async () => { - const deps = makeDeps({ applied: [] }) - deps.runningBuildSupportsPerFileMigrations = sinon.stub().rejects(new Error('docker unreachable')) - let err = null - try { - await assertRequiredMigrationsApplied(XChainService.XCHAIN_INDEXER, 'bitcoin', 'mainnet', 'master', deps) - } catch (e) { err = e } - expect(err, 'the deploy must still be refused').to.not.equal(null) - expect(err.message).to.contain('update refused') - expect(err.message).to.not.contain('docker exec') - }) - - it('reads the source tree about to be deployed, at the pinned ref', async () => { - const deps = makeDeps() - await assertRequiredMigrationsApplied(XChainService.XCHAIN_INDEXER, 'bitcoin', 'mainnet', 'release-1.2.3', deps) - expect(deps.cloneGit.calledWith(XChainService.XCHAIN_INDEXER, false, true, 'release-1.2.3')).to.equal(true) - }) - - it('passes when every declared precondition is in the ledger', async () => { - const deps = makeDeps() - const res = await assertRequiredMigrationsApplied(XChainService.XCHAIN_INDEXER, 'bitcoin', 'mainnet', 'master', deps) - expect(res.ok).to.equal(true) - expect(res.missing).to.deep.equal([]) - }) - - it('does not touch the database when the target source declares no preconditions', async () => { - const deps = makeDeps({ required: [] }) - const res = await assertRequiredMigrationsApplied(XChainService.XCHAIN_INDEXER, 'bitcoin', 'mainnet', 'master', deps) - expect(res).to.deep.equal({ checked: false, reason: 'no-preconditions' }) - expect(deps.readAppliedMigrations.called).to.equal(false) - }) - - it('proceeds on a genuinely empty database (a fresh install cannot be behind)', async () => { - const deps = makeDeps({ state: 'empty-database' }) - const res = await assertRequiredMigrationsApplied(XChainService.XCHAIN_INDEXER, 'bitcoin', 'mainnet', 'master', deps) - expect(res.ok).to.equal(true) - expect(res.reason).to.equal('empty-database') - }) - - it('refuses when the migration state cannot be read, and says so is not the same as missing', async () => { - const deps = makeDeps({ state: 'unreadable', reason: 'ECONNREFUSED 127.0.0.1:13306' }) - let err = null - try { - await assertRequiredMigrationsApplied(XChainService.XCHAIN_INDEXER, 'bitcoin', 'mainnet', 'master', deps) - } catch (e) { err = e } - expect(err, 'an unknown migration state must fail closed').to.not.equal(null) - expect(err.message).to.contain('could NOT be determined') - expect(err.message).to.contain('ECONNREFUSED') - expect(err.message).to.contain(SKIP_ENV) - }) - - it('proceeds with a warning when the source itself cannot be cloned', async () => { - // The update is about to fail on the same clone; adding a second failure - // mode here would only obscure the real one. - const deps = makeDeps({ cloneErr: new Error('network down') }) - const res = await assertRequiredMigrationsApplied(XChainService.XCHAIN_INDEXER, 'bitcoin', 'mainnet', 'master', deps) - expect(res).to.deep.equal({ checked: false, reason: 'source-unreadable' }) - expect(warnStub.called).to.equal(true) - }) + const deps = { + fs, os, path, sinon, expect, XChainService, + MIGRATION_BEARING_MODULES, SKIP_ENV, + migrationDeclaresDeployPrecondition, migrationMode, + listDeployPreconditionMigrations, pendingManualMigrations, + readAppliedMigrations, assertRequiredMigrationsApplied, + GATED, TAGGED, UNTAGGED, INDEXER_DIR, INDEXER_MIGRATIONS, + INDEXER_PRESENT, REQUIRE_SIBLINGS, makeDeps + } - it('checks the database belonging to the module, coin and network being updated', async () => { - const deps = makeDeps() - await assertRequiredMigrationsApplied(XChainService.XCHAIN_DECODER, 'litecoin', 'testnet', 'master', deps) - const arg = deps.readAppliedMigrations.firstCall.args[0] - expect(arg.database).to.equal('XChain_LTC_Testnet_Decoder') - expect(arg.coin).to.equal('litecoin') - expect(arg.network).to.equal('testnet') - }) - }) + registerMigrationMetadata(deps) + registerMigrationInventory(deps) + registerAppliedLedger(deps) + registerGuardOutcomes(deps) }) diff --git a/test/unit/migration_precondition_service.test/01_migration_metadata.test.js b/test/unit/migration_precondition_service.test/01_migration_metadata.test.js new file mode 100644 index 0000000..b720f14 --- /dev/null +++ b/test/unit/migration_precondition_service.test/01_migration_metadata.test.js @@ -0,0 +1,65 @@ +'use strict' + +// Copyright © 2025–2026 Dankest, LLC +// Based on XChain Platform by Dankest, LLC – https://dankest.llc +// +// SPDX-License-Identifier: AGPL-3.0-or-later +// +// This file is part of XChain Platform. Licensed under the GNU Affero +// General Public License v3.0 or later; see LICENSE.md. A commercial +// license (without AGPL source-disclosure terms) is available - +// contact legal@dankest.llc. + +function registerMigrationDirectiveTests({ expect, migrationDeclaresDeployPrecondition, TAGGED, UNTAGGED }) { + describe('migrationDeclaresDeployPrecondition', () => { + it('reads the tag off the xchain:migration directive line', () => { + expect(migrationDeclaresDeployPrecondition(TAGGED)).to.equal(true) + }) + it('tolerates spacing around the token', () => { + expect(migrationDeclaresDeployPrecondition('-- xchain:migration mode = manual deploy-precondition = required\nALTER TABLE t;')).to.equal(true) + }) + it('is false for an ordinary migration and for an empty file', () => { + expect(migrationDeclaresDeployPrecondition(UNTAGGED)).to.equal(false) + expect(migrationDeclaresDeployPrecondition('')).to.equal(false) + }) + it('ignores the token once the SQL body has started', () => { + // Prologue anchoring: a migration that merely DISCUSSES the convention in a + // trailing comment must not start refusing every deploy. + expect(migrationDeclaresDeployPrecondition('ALTER TABLE t;\n-- xchain:migration mode=manual deploy-precondition=required\n')).to.equal(false) + }) + it('ignores the token on a comment line that is not the directive', () => { + expect(migrationDeclaresDeployPrecondition('-- deploy-precondition=required, see the other file\nALTER TABLE t;')).to.equal(false) + }) + it('sees the tag through a long license banner', () => { + const banner = Array(30).fill('-- license line').join('\n') + expect(migrationDeclaresDeployPrecondition(banner + '\n\n' + TAGGED)).to.equal(true) + }) + }) +} + +function registerMigrationModeTests({ expect, migrationMode, TAGGED }) { + describe('migrationMode', () => { + + it('reads the mode a header declares', () => { + expect(migrationMode(TAGGED)).to.equal('manual') + expect(migrationMode('-- xchain:migration mode=auto\nSELECT 1;\n')).to.equal('auto') + }) + + it('returns null when no mode is declared', () => { + expect(migrationMode('-- just a comment\nSELECT 1;\n')).to.equal(null) + }) + + it('ignores a mode token that appears after the prologue', () => { + // Body prose and data literals must not be able to answer for the file. + const body = '-- header\nSELECT 1;\n-- xchain:migration mode=auto\n' + expect(migrationMode(body)).to.equal(null) + }) + }) +} + +function registerMigrationMetadata(deps) { + registerMigrationDirectiveTests(deps) + registerMigrationModeTests(deps) +} + +module.exports = registerMigrationMetadata diff --git a/test/unit/migration_precondition_service.test/02_migration_inventory.test.js b/test/unit/migration_precondition_service.test/02_migration_inventory.test.js new file mode 100644 index 0000000..dec2817 --- /dev/null +++ b/test/unit/migration_precondition_service.test/02_migration_inventory.test.js @@ -0,0 +1,107 @@ +'use strict' + +// Copyright © 2025–2026 Dankest, LLC +// Based on XChain Platform by Dankest, LLC – https://dankest.llc +// +// SPDX-License-Identifier: AGPL-3.0-or-later +// +// This file is part of XChain Platform. Licensed under the GNU Affero +// General Public License v3.0 or later; see LICENSE.md. A commercial +// license (without AGPL source-disclosure terms) is available - +// contact legal@dankest.llc. + +function registerPendingManualMigrations({ fs, os, path, expect, pendingManualMigrations, TAGGED }) { + describe('pendingManualMigrations', () => { + + let dir + beforeEach(() => { + dir = fs.mkdtempSync(path.join(os.tmpdir(), 'xc-pending-')) + fs.writeFileSync(path.join(dir, 'a-manual.sql'), TAGGED) + fs.writeFileSync(path.join(dir, 'b-auto.sql'), '-- xchain:migration mode=auto\nSELECT 1;\n') + fs.writeFileSync(path.join(dir, 'c-manual.sql'), '-- xchain:migration mode=manual\nSELECT 1;\n') + }) + afterEach(() => { fs.rmSync(dir, { recursive: true, force: true }) }) + + it('lists only manual migrations the ledger has not recorded', () => { + expect(pendingManualMigrations(dir, new Set())).to.deep.equal(['a-manual.sql', 'c-manual.sql']) + }) + + it('excludes what the ledger already carries', () => { + expect(pendingManualMigrations(dir, new Set(['a-manual.sql']))).to.deep.equal(['c-manual.sql']) + }) + + it('yields nothing for a missing directory rather than throwing', () => { + expect(pendingManualMigrations(path.join(dir, 'nope'), new Set())).to.deep.equal([]) + }) + }) +} + +function registerRealIndexerInventory({ fs, expect, listDeployPreconditionMigrations, GATED, + INDEXER_DIR, INDEXER_MIGRATIONS, INDEXER_PRESENT, REQUIRE_SIBLINGS }) { + it('reads the REAL indexer tree and finds the migration behind the 2026-08-09 halt', function () { + // Guards the whole contract end to end: if the tag is ever dropped from the + // committed file, or the migrations path moves, this guard must stop the + // suite rather than pass silently on a directory that no longer exists. + if (!INDEXER_PRESENT) { + if (REQUIRE_SIBLINGS) + throw new Error('XCHAIN_REQUIRE_SIBLINGS=1 but xchain-indexer is not checked out at ' + INDEXER_DIR) + return this.skip() // sibling repo not checked out + } + expect(fs.existsSync(INDEXER_MIGRATIONS), 'xchain-indexer is checked out at ' + INDEXER_DIR + + ' but has no migrations directory at ' + INDEXER_MIGRATIONS).to.equal(true) + expect(listDeployPreconditionMigrations(INDEXER_MIGRATIONS)).to.include(GATED) + }) + + it('reads the REAL indexer tree and finds the bridge-tables migration', function () { + // The bridge build lands `2026-09-12-bridge-tables.sql` (mode=manual, + // deploy-precondition=required) beside the token-bridge-fields migration + // (mode=auto, no precondition). Nothing in xchain-node had to change for + // either to be covered: this guard is a directory scan of whatever the + // target tree carries, so a new deploy-precondition migration is wired in + // the moment it lands on the indexer, no xchain-node release required + // (the same "no coupled release" property MigrationPreconditionService's + // header describes for the contract as a whole). This test is the proof. + if (!INDEXER_PRESENT) { + if (REQUIRE_SIBLINGS) + throw new Error('XCHAIN_REQUIRE_SIBLINGS=1 but xchain-indexer is not checked out at ' + INDEXER_DIR) + return this.skip() // sibling repo not checked out + } + expect(fs.existsSync(INDEXER_MIGRATIONS), 'xchain-indexer is checked out at ' + INDEXER_DIR + + ' but has no migrations directory at ' + INDEXER_MIGRATIONS).to.equal(true) + const required = listDeployPreconditionMigrations(INDEXER_MIGRATIONS) + expect(required).to.include('2026-09-12-bridge-tables.sql') + expect(required).to.not.include('2026-09-12-token-bridge-fields.sql') + }) +} + +function registerListDeployPreconditionMigrations(deps) { + const { fs, os, path, expect, listDeployPreconditionMigrations, TAGGED, UNTAGGED } = deps + describe('listDeployPreconditionMigrations', () => { + let dir + beforeEach(() => { + dir = fs.mkdtempSync(path.join(os.tmpdir(), 'xcn-mig-')) + }) + afterEach(() => { fs.rmSync(dir, { recursive: true, force: true }) }) + + it('returns only the tagged .sql files, sorted', () => { + fs.writeFileSync(path.join(dir, '2026-07-24-b.sql'), TAGGED) + fs.writeFileSync(path.join(dir, '2026-07-01-a.sql'), TAGGED) + fs.writeFileSync(path.join(dir, '2026-07-30-c.sql'), UNTAGGED) + fs.writeFileSync(path.join(dir, 'notes.txt'), TAGGED) + expect(listDeployPreconditionMigrations(dir)).to.deep.equal(['2026-07-01-a.sql', '2026-07-24-b.sql']) + }) + + it('returns [] for a missing directory (a ref with no migrations declares nothing)', () => { + expect(listDeployPreconditionMigrations(path.join(dir, 'nope'))).to.deep.equal([]) + }) + + registerRealIndexerInventory(deps) + }) +} + +function registerMigrationInventory(deps) { + registerPendingManualMigrations(deps) + registerListDeployPreconditionMigrations(deps) +} + +module.exports = registerMigrationInventory diff --git a/test/unit/migration_precondition_service.test/03_applied_ledger.test.js b/test/unit/migration_precondition_service.test/03_applied_ledger.test.js new file mode 100644 index 0000000..9c27a57 --- /dev/null +++ b/test/unit/migration_precondition_service.test/03_applied_ledger.test.js @@ -0,0 +1,112 @@ +'use strict' + +// Copyright © 2025–2026 Dankest, LLC +// Based on XChain Platform by Dankest, LLC – https://dankest.llc +// +// SPDX-License-Identifier: AGPL-3.0-or-later +// +// This file is part of XChain Platform. Licensed under the GNU Affero +// General Public License v3.0 or later; see LICENSE.md. A commercial +// license (without AGPL source-disclosure terms) is available - +// contact legal@dankest.llc. + +function registerReadableLedgerTests({ expect, readAppliedMigrations, GATED }, target, runnerFor) { + it('reads the ledger into a set of applied names', async () => { + const res = await readAppliedMigrations(target, { runner: runnerFor({ names: [GATED, '2026-08-11-attests-relay-identity-index.sql'] }) }) + expect(res.state).to.equal('ledger') + expect([...res.applied]).to.have.members([GATED, '2026-08-11-attests-relay-identity-index.sql']) + }) + + it('tolerates a ledger read that comes back empty', async () => { + const res = await readAppliedMigrations(target, { runner: runnerFor({ names: [] }) }) + expect(res.state).to.equal('ledger') + expect(res.applied.size).to.equal(0) + }) + + it('calls a database with no tables empty, not unreadable', async () => { + const res = await readAppliedMigrations(target, { runner: runnerFor({ tables: 0 }) }) + expect(res.state).to.equal('empty-database') + }) + + it('calls a populated database with no ledger table UNREADABLE, never empty', async () => { + // Waving this through would be the whole outage again: a real schema whose + // migration state nobody can see. + const res = await readAppliedMigrations(target, { runner: runnerFor({ tables: 40, ledger: 0 }) }) + expect(res.state).to.equal('unreadable') + expect(res.reason).to.contain('schema_migrations') + }) +} + +function registerUnreadableLedgerTests({ expect, readAppliedMigrations }, target) { + it('refuses an unreadable table count instead of collapsing it into empty-database', async () => { + // A count that fails to parse (a driver notice, an empty batch-mode + // reply, garbage) yields NaN. `!NaN` is true just like `!0`, so a naive + // falsy check reads an unreadable count as the same "empty database" + // verdict as a genuinely empty one - the exact bug this guards against. + const res = await readAppliedMigrations(target, { + runner: async () => 'ERROR 2013 (HY000): Lost connection to MySQL server' + }) + expect(res.state).to.equal('unreadable') + expect(res.reason).to.contain(target.database) + }) + + it('refuses an empty-string table count instead of reading it as zero', async () => { + const res = await readAppliedMigrations(target, { runner: async () => '' }) + expect(res.state).to.equal('unreadable') + }) + + it('refuses an unreadable ledger-presence count instead of reading it as "no ledger"', async () => { + const res = await readAppliedMigrations(target, { + runner: async (sql) => { + if (/TABLE_NAME = 'schema_migrations'/.test(sql)) return 'ERROR: connection reset' + if (/COUNT\(\*\)/.test(sql)) return '40' + return '' + } + }) + expect(res.state).to.equal('unreadable') + expect(res.reason).to.contain('schema_migrations') + }) + + it('turns a driver failure into unreadable instead of throwing past the guard', async () => { + // A throw here would escape assertRequiredMigrationsApplied as an opaque + // driver error, and the operator would read ECONNREFUSED with no idea a + // migration was at stake. + const res = await readAppliedMigrations(target, { + runner: async () => { throw new Error('ECONNREFUSED 127.0.0.1:13306') } + }) + expect(res.state).to.equal('unreadable') + expect(res.reason).to.contain('ECONNREFUSED') + }) + + it('refuses a database name that is not a plain identifier, before any query runs', async () => { + let called = false + const res = await readAppliedMigrations( + { database: 'x`; DROP DATABASE y; -- ', coin: 'bitcoin', network: 'mainnet' }, + { runner: async () => { called = true; return '0' } }) + expect(res.state).to.equal('unreadable') + expect(called, 'nothing may reach SQL').to.equal(false) + }) +} + +function registerAppliedLedger(deps) { + const { GATED } = deps + describe('readAppliedMigrations', () => { + + const target = { database: 'XChain_BTC_Mainnet_Indexer', coin: 'bitcoin', network: 'mainnet' } + + // Fake mariadb batch-mode output: one value per COUNT query, newline-joined + // names for the ledger read - the exact shapes `-B -N` produces. + function runnerFor({ tables = 40, ledger = 1, names = [GATED] }) { + return async (sql) => { + if (/TABLE_NAME = 'schema_migrations'/.test(sql)) return String(ledger) + if (/COUNT\(\*\)/.test(sql)) return String(tables) + return names.join('\n') + } + } + + registerReadableLedgerTests(deps, target, runnerFor) + registerUnreadableLedgerTests(deps, target) + }) +} + +module.exports = registerAppliedLedger diff --git a/test/unit/migration_precondition_service.test/04_guard_outcomes.test.js b/test/unit/migration_precondition_service.test/04_guard_outcomes.test.js new file mode 100644 index 0000000..148cd10 --- /dev/null +++ b/test/unit/migration_precondition_service.test/04_guard_outcomes.test.js @@ -0,0 +1,204 @@ +'use strict' + +// Copyright © 2025–2026 Dankest, LLC +// Based on XChain Platform by Dankest, LLC – https://dankest.llc +// +// SPDX-License-Identifier: AGPL-3.0-or-later +// +// This file is part of XChain Platform. Licensed under the GNU Affero +// General Public License v3.0 or later; see LICENSE.md. A commercial +// license (without AGPL source-disclosure terms) is available - +// contact legal@dankest.llc. + +const proxyquire = require('proxyquire').noCallThru() + +let warnStub + +function registerBasicGuardOutcomes({ expect, sinon, XChainService, MIGRATION_BEARING_MODULES, SKIP_ENV, + assertRequiredMigrationsApplied, makeDeps, GATED }) { + it('uses the moved default migration reader when no deps are passed', async () => { + const cloneGit = sinon.stub().resolves() + const listRequired = sinon.stub().returns([GATED]) + const readApplied = sinon.stub().resolves({ state: 'ledger', applied: new Set([GATED]) }) + const migrationScan = require('../../../src/services/migration_precondition_service/migration_scan') + const service = proxyquire('../../../src/services/migration_precondition_service', { + './module_service': { cloneGit }, + './migration_precondition_service/migration_scan': { + ...migrationScan, + listDeployPreconditionMigrations: listRequired, + readAppliedMigrations: readApplied + } + }) + + const res = await service.assertRequiredMigrationsApplied( + XChainService.XCHAIN_INDEXER, 'bitcoin', 'mainnet', 'master') + + expect(res.ok).to.equal(true) + expect(cloneGit.calledOnce).to.equal(true) + expect(listRequired.calledOnce).to.equal(true) + expect(readApplied.calledOnce).to.equal(true) + }) + + it('is inert for a module that ships no migrations', async () => { + const deps = makeDeps() + const res = await assertRequiredMigrationsApplied(XChainService.XCHAIN_ENCODER, 'bitcoin', 'mainnet', 'master', deps) + expect(res).to.deep.equal({ checked: false, reason: 'no-migrations' }) + expect(deps.cloneGit.called).to.equal(false) + }) + + it('covers the indexer and the decoder, the two migration-bearing modules', () => { + expect(MIGRATION_BEARING_MODULES).to.have.members([XChainService.XCHAIN_INDEXER, XChainService.XCHAIN_DECODER]) + }) + + it('proceeds, loudly, when the skip env is set', async () => { + process.env[SKIP_ENV] = '1' + const deps = makeDeps({ applied: [] }) + const res = await assertRequiredMigrationsApplied(XChainService.XCHAIN_INDEXER, 'bitcoin', 'mainnet', 'master', deps) + expect(res.reason).to.equal('skipped-by-env') + expect(warnStub.called).to.equal(true) + }) +} + +function registerMissingMigrationOutcomes({ expect, XChainService, assertRequiredMigrationsApplied, + GATED, makeDeps }) { + it('refuses when the target DB has not applied a declared precondition', async () => { + const deps = makeDeps({ applied: ['2026-07-21-anchor-reward-attestations-table.sql'] }) + let err = null + try { + await assertRequiredMigrationsApplied(XChainService.XCHAIN_INDEXER, 'bitcoin', 'mainnet', 'master', deps) + } catch (e) { err = e } + expect(err, 'the deploy must be refused').to.not.equal(null) + expect(err.message).to.contain(GATED) + expect(err.message).to.contain('XChain_BTC_Mainnet_Indexer') + // Only safe to print because this build was confirmed to honour --file. + expect(err.message).to.contain('--file ' + GATED) + }) + + it('does NOT print a scoped command the running build would ignore', async () => { + // A build without per-file targeting does not reject --file, it ignores + // it and applies every pending manual migration, so printing the command + // hands the operator a wider action than the one it describes. + const deps = makeDeps({ + applied: [], + supportsPerFile: false, + pendingManual: [GATED, '2026-08-10-action-data-utf8mb4.sql', '2026-06-13-dispensers-expiration-bigint.sql'] + }) + let err = null + try { + await assertRequiredMigrationsApplied(XChainService.XCHAIN_INDEXER, 'bitcoin', 'mainnet', 'master', deps) + } catch (e) { err = e } + expect(err, 'the deploy must be refused').to.not.equal(null) + expect(err.message).to.not.contain('docker exec') + expect(err.message).to.contain('DO NOT run') + // The whole blast radius is named, not just the file that is needed. + expect(err.message).to.contain('3 file(s)') + expect(err.message).to.contain('2026-08-10-action-data-utf8mb4.sql') + expect(err.message).to.contain('2026-06-13-dispensers-expiration-bigint.sql') + expect(err.message).to.contain('(the one you need)') + }) +} + +function registerCapabilityProbeOutcomes({ sinon, expect, XChainService, + assertRequiredMigrationsApplied, makeDeps }) { + it('treats an unreadable container as lacking the capability', async () => { + // An unverified capability is not a capability: the cost of guessing + // wrong is an unauthorised migration on a live database. + const deps = makeDeps({ applied: [], supportsPerFile: null }) + let err = null + try { + await assertRequiredMigrationsApplied(XChainService.XCHAIN_INDEXER, 'bitcoin', 'mainnet', 'master', deps) + } catch (e) { err = e } + expect(err, 'the deploy must be refused').to.not.equal(null) + expect(err.message).to.not.contain('docker exec') + expect(err.message).to.contain('could not be read') + }) + + it('still refuses when the capability probe itself throws', async () => { + const deps = makeDeps({ applied: [] }) + deps.runningBuildSupportsPerFileMigrations = sinon.stub().rejects(new Error('docker unreachable')) + let err = null + try { + await assertRequiredMigrationsApplied(XChainService.XCHAIN_INDEXER, 'bitcoin', 'mainnet', 'master', deps) + } catch (e) { err = e } + expect(err, 'the deploy must still be refused').to.not.equal(null) + expect(err.message).to.contain('update refused') + expect(err.message).to.not.contain('docker exec') + }) +} + +function registerSuccessfulGuardOutcomes({ expect, XChainService, assertRequiredMigrationsApplied, makeDeps }) { + it('reads the source tree about to be deployed, at the pinned ref', async () => { + const deps = makeDeps() + await assertRequiredMigrationsApplied(XChainService.XCHAIN_INDEXER, 'bitcoin', 'mainnet', 'release-1.2.3', deps) + expect(deps.cloneGit.calledWith(XChainService.XCHAIN_INDEXER, false, true, 'release-1.2.3')).to.equal(true) + }) + + it('passes when every declared precondition is in the ledger', async () => { + const deps = makeDeps() + const res = await assertRequiredMigrationsApplied(XChainService.XCHAIN_INDEXER, 'bitcoin', 'mainnet', 'master', deps) + expect(res.ok).to.equal(true) + expect(res.missing).to.deep.equal([]) + }) + + it('does not touch the database when the target source declares no preconditions', async () => { + const deps = makeDeps({ required: [] }) + const res = await assertRequiredMigrationsApplied(XChainService.XCHAIN_INDEXER, 'bitcoin', 'mainnet', 'master', deps) + expect(res).to.deep.equal({ checked: false, reason: 'no-preconditions' }) + expect(deps.readAppliedMigrations.called).to.equal(false) + }) + + it('proceeds on a genuinely empty database (a fresh install cannot be behind)', async () => { + const deps = makeDeps({ state: 'empty-database' }) + const res = await assertRequiredMigrationsApplied(XChainService.XCHAIN_INDEXER, 'bitcoin', 'mainnet', 'master', deps) + expect(res.ok).to.equal(true) + expect(res.reason).to.equal('empty-database') + }) +} + +function registerFailureAndScopeOutcomes({ expect, XChainService, SKIP_ENV, + assertRequiredMigrationsApplied, makeDeps }) { + it('refuses when the migration state cannot be read, and says so is not the same as missing', async () => { + const deps = makeDeps({ state: 'unreadable', reason: 'ECONNREFUSED 127.0.0.1:13306' }) + let err = null + try { + await assertRequiredMigrationsApplied(XChainService.XCHAIN_INDEXER, 'bitcoin', 'mainnet', 'master', deps) + } catch (e) { err = e } + expect(err, 'an unknown migration state must fail closed').to.not.equal(null) + expect(err.message).to.contain('could NOT be determined') + expect(err.message).to.contain('ECONNREFUSED') + expect(err.message).to.contain(SKIP_ENV) + }) + + it('proceeds with a warning when the source itself cannot be cloned', async () => { + // The update is about to fail on the same clone; adding a second failure + // mode here would only obscure the real one. + const deps = makeDeps({ cloneErr: new Error('network down') }) + const res = await assertRequiredMigrationsApplied(XChainService.XCHAIN_INDEXER, 'bitcoin', 'mainnet', 'master', deps) + expect(res).to.deep.equal({ checked: false, reason: 'source-unreadable' }) + expect(warnStub.called).to.equal(true) + }) + + it('checks the database belonging to the module, coin and network being updated', async () => { + const deps = makeDeps() + await assertRequiredMigrationsApplied(XChainService.XCHAIN_DECODER, 'litecoin', 'testnet', 'master', deps) + const arg = deps.readAppliedMigrations.firstCall.args[0] + expect(arg.database).to.equal('XChain_LTC_Testnet_Decoder') + expect(arg.coin).to.equal('litecoin') + expect(arg.network).to.equal('testnet') + }) +} + +function registerGuardOutcomes(deps) { + describe('assertRequiredMigrationsApplied', () => { + registerBasicGuardOutcomes(deps) + registerMissingMigrationOutcomes(deps) + registerCapabilityProbeOutcomes(deps) + registerSuccessfulGuardOutcomes(deps) + registerFailureAndScopeOutcomes(deps) + }) +} + +registerGuardOutcomes.setWarnStub = (stub) => { warnStub = stub } +registerGuardOutcomes.restoreWarnStub = () => { warnStub.restore() } + +module.exports = registerGuardOutcomes From 9310e07021c5584c7853b072bd83fe7e1ed0c613 Mon Sep 17 00:00:00 2001 From: J-Dog Date: Thu, 17 Sep 2026 13:16:39 -0700 Subject: [PATCH 10/35] test: split nightly workflow registrations Extract module-scope helpers for shared, per-coin, validator, and bitcoin checks. Preserve hook count, test order, fixtures, assertions, and synchronous execution. --- .../two_stack_legs.test.js | 118 ++++++++++-------- 1 file changed, 65 insertions(+), 53 deletions(-) diff --git a/test/unit/nightly_e2e_workflow.test/two_stack_legs.test.js b/test/unit/nightly_e2e_workflow.test/two_stack_legs.test.js index c40c7ac..529ba3e 100644 --- a/test/unit/nightly_e2e_workflow.test/two_stack_legs.test.js +++ b/test/unit/nightly_e2e_workflow.test/two_stack_legs.test.js @@ -86,10 +86,9 @@ function runStep(step, env) { return { dir, calls, stdout } } -describe('nightly-e2e.yml two-stack legs (litecoin and dogecoin gas in over the bitcoin rail)', function () { - let steps - before(function () { steps = loadSteps() }) +let steps +function registerSharedWorkflowChecks() { it('gates the ports step off the bitcoin leg, whose single-stack shape stays as it was', function () { expect(steps.ports.if).to.equal("env.COIN != 'bitcoin'") }) @@ -101,63 +100,65 @@ describe('nightly-e2e.yml two-stack legs (litecoin and dogecoin gas in over the expect(m, 'docker run mariadb:11 --max-connections=N').to.not.equal(null) expect(parseInt(m[1], 10)).to.be.at.least(400) }) +} - for (const coin of ['litecoin', 'dogecoin']) { - describe(coin + ' leg', function () { - const env = { COIN: coin, STACK_REF: 'release/vX.Y.Z', XCHAIN_NODE_EXTERNAL_DB_HOST: '172.17.0.1' } - - it('writes the coin and bitcoin config files with the chainRail port blocks', function () { - const { dir } = runStep(steps.ports, env) - const own = parseConfigFile(path.join(dir, 'config', coin + '-regtest')) - const btc = parseConfigFile(path.join(dir, 'config', 'bitcoin-regtest')) - for (const [key, port] of Object.entries(CHAIN_RAIL_DEFAULT_PORTS[coin])) expect(own[key], coin + ' ' + key).to.equal(String(port)) - for (const [key, port] of Object.entries(CHAIN_RAIL_DEFAULT_PORTS.bitcoin)) expect(btc[key], 'bitcoin ' + key).to.equal(String(port)) - // Same host block twice would collide at the second install's port check. - expect(new Set([...Object.values(own), ...Object.values(btc)].filter(v => /^\d+$/.test(v))).size).to.equal(12) - }) +function registerCoinLegChecks(coin) { + describe(coin + ' leg', function () { + const env = { COIN: coin, STACK_REF: 'release/vX.Y.Z', XCHAIN_NODE_EXTERNAL_DB_HOST: '172.17.0.1' } + + it('writes the coin and bitcoin config files with the chainRail port blocks', function () { + const { dir } = runStep(steps.ports, env) + const own = parseConfigFile(path.join(dir, 'config', coin + '-regtest')) + const btc = parseConfigFile(path.join(dir, 'config', 'bitcoin-regtest')) + for (const [key, port] of Object.entries(CHAIN_RAIL_DEFAULT_PORTS[coin])) expect(own[key], coin + ' ' + key).to.equal(String(port)) + for (const [key, port] of Object.entries(CHAIN_RAIL_DEFAULT_PORTS.bitcoin)) expect(btc[key], 'bitcoin ' + key).to.equal(String(port)) + // Same host block twice would collide at the second install's port check. + expect(new Set([...Object.values(own), ...Object.values(btc)].filter(v => /^\d+$/.test(v))).size).to.equal(12) + }) - it('routes the e2e container to the bitcoin rail through the docker bridge gateway', function () { - const { dir } = runStep(steps.ports, env) - const own = parseConfigFile(path.join(dir, 'config', coin + '-regtest')) - const btc = parseConfigFile(path.join(dir, 'config', 'bitcoin-regtest')) - expect(own.BTC_SERVICE_HOST).to.equal('172.17.0.1') - // The bitcoin stack's own containers need no such route. - expect(btc).to.not.have.property('BTC_SERVICE_HOST') - }) + it('routes the e2e container to the bitcoin rail through the docker bridge gateway', function () { + const { dir } = runStep(steps.ports, env) + const own = parseConfigFile(path.join(dir, 'config', coin + '-regtest')) + const btc = parseConfigFile(path.join(dir, 'config', 'bitcoin-regtest')) + expect(own.BTC_SERVICE_HOST).to.equal('172.17.0.1') + // The bitcoin stack's own containers need no such route. + expect(btc).to.not.have.property('BTC_SERVICE_HOST') + }) - it('routes the coin indexer to the bitcoin indexer for the bridge escrow proof', function () { - // The destination indexer fetches the escrow proof from the origin - // chain's indexer at BTC_INDEXER_API_URL before it credits a bridged - // transfer, and holds the block at the proof barrier when nothing is - // wired (run 35140173657: 900 s at bridge_proof_barrier, 143 blocks - // behind). The bitcoin indexer joins the coin's docker network, so - // its container name on the indexer's own port is the route. - const { dir } = runStep(steps.ports, env) - const own = parseConfigFile(path.join(dir, 'config', coin + '-regtest')) - const btc = parseConfigFile(path.join(dir, 'config', 'bitcoin-regtest')) - expect(own.BTC_INDEXER_API_URL).to.equal('http://xchain-node-bitcoin-regtest-xchain-indexer:3004') - expect(btc).to.not.have.property('BTC_INDEXER_API_URL') - }) + it('routes the coin indexer to the bitcoin indexer for the bridge escrow proof', function () { + // The destination indexer fetches the escrow proof from the origin + // chain's indexer at BTC_INDEXER_API_URL before it credits a bridged + // transfer, and holds the block at the proof barrier when nothing is + // wired (run 35140173657: 900 s at bridge_proof_barrier, 143 blocks + // behind). The bitcoin indexer joins the coin's docker network, so + // its container name on the indexer's own port is the route. + const { dir } = runStep(steps.ports, env) + const own = parseConfigFile(path.join(dir, 'config', coin + '-regtest')) + const btc = parseConfigFile(path.join(dir, 'config', 'bitcoin-regtest')) + expect(own.BTC_INDEXER_API_URL).to.equal('http://xchain-node-bitcoin-regtest-xchain-indexer:3004') + expect(btc).to.not.have.property('BTC_INDEXER_API_URL') + }) - it('never writes a credential into either file (the install generates those into the .local sidecars)', function () { - const { dir } = runStep(steps.ports, env) - for (const file of [coin + '-regtest', 'bitcoin-regtest']) { - const keys = Object.keys(parseConfigFile(path.join(dir, 'config', file))) - expect(keys.filter(k => /USER|PASS|SECRET|KEY/.test(k)), file).to.deep.equal([]) - } - }) + it('never writes a credential into either file (the install generates those into the .local sidecars)', function () { + const { dir } = runStep(steps.ports, env) + for (const file of [coin + '-regtest', 'bitcoin-regtest']) { + const keys = Object.keys(parseConfigFile(path.join(dir, 'config', file))) + expect(keys.filter(k => /USER|PASS|SECRET|KEY/.test(k)), file).to.deep.equal([]) + } + }) - it('installs the coin under test first, then the bitcoin gas rail, both at the same ref', function () { - const { calls, dir } = runStep(steps.boot, Object.assign({ XCHAIN_NODE_DATA_DIR: path.join(os.tmpdir(), 'nightly-e2e-data-' + process.pid) }, env)) - expect(calls).to.deep.equal([ - 'src/index.js install release/vX.Y.Z all ' + coin + ' regtest', - 'src/index.js install release/vX.Y.Z all bitcoin regtest', - ]) - expect(dir).to.be.a('string') - }) + it('installs the coin under test first, then the bitcoin gas rail, both at the same ref', function () { + const { calls, dir } = runStep(steps.boot, Object.assign({ XCHAIN_NODE_DATA_DIR: path.join(os.tmpdir(), 'nightly-e2e-data-' + process.pid) }, env)) + expect(calls).to.deep.equal([ + 'src/index.js install release/vX.Y.Z all ' + coin + ' regtest', + 'src/index.js install release/vX.Y.Z all bitcoin regtest', + ]) + expect(dir).to.be.a('string') }) - } + }) +} +function registerValidatorModeChecks() { // The bridged credit needs a hub that FINALIZES transfers, which a standalone // hub never does: startCrossChain returns before constructing // CrossChainBridgeEngine without a peerManager, and even with an identity the @@ -215,7 +216,9 @@ describe('nightly-e2e.yml two-stack legs (litecoin and dogecoin gas in over the expect(exported).to.deep.equal({ HUB_NETWORK: 'regtest', ORACLE_MIN_SUBMISSIONS: '1' }) }) }) +} +function registerBitcoinLegChecks() { describe('bitcoin leg', function () { it('boots exactly one stack, unchanged from the single-stack shape', function () { const { calls } = runStep(steps.boot, { @@ -225,4 +228,13 @@ describe('nightly-e2e.yml two-stack legs (litecoin and dogecoin gas in over the expect(calls).to.deep.equal(['src/index.js install develop all bitcoin regtest']) }) }) +} + +describe('nightly-e2e.yml two-stack legs (litecoin and dogecoin gas in over the bitcoin rail)', function () { + before(function () { steps = loadSteps() }) + + registerSharedWorkflowChecks() + for (const coin of ['litecoin', 'dogecoin']) registerCoinLegChecks(coin) + registerValidatorModeChecks() + registerBitcoinLegChecks() }) From 1b89fb73177dc25e874a079f08bf8b30929d7ecc Mon Sep 17 00:00:00 2001 From: J-Dog Date: Thu, 17 Sep 2026 15:40:33 -0700 Subject: [PATCH 11/35] test(node): split indexer migration CLI registrations Extract path precedence, capability detection, and refusal test registration into synchronous module-scope helpers. Preserve test callback text, title order, and await behavior while lowering the oversized-function count. --- test/unit/utils/indexer_migrate_cli.test.js | 13 +++++++++++-- 1 file changed, 11 insertions(+), 2 deletions(-) diff --git a/test/unit/utils/indexer_migrate_cli.test.js b/test/unit/utils/indexer_migrate_cli.test.js index 3c00c98..f2fa740 100644 --- a/test/unit/utils/indexer_migrate_cli.test.js +++ b/test/unit/utils/indexer_migrate_cli.test.js @@ -43,8 +43,7 @@ function catFor(files) { }) } -describe('indexer migrate CLI location', () => { - +function registerPathPrecedenceTests() { // Pins the full candidate list, newest first, so a future indexer layout // move that edits MIGRATE_CLI_PATHS without adding the new path (or drops // an old one a still-supported build carries) fails here first, rather @@ -88,7 +87,9 @@ describe('indexer migrate CLI location', () => { expect(migrateCliPathFor('c-gone')).to.equal(MIGRATE_CLI_PATHS[0]) }) }) +} +function registerCapabilityDetectionTests() { describe('runningBuildSupportsPerFileMigrations', () => { it('sees --file support on a build from before the move', async () => { const deps = { getDockerContainerFileCat: catFor({ [OLD_PATH]: WITH_FILE }) } @@ -110,7 +111,9 @@ describe('indexer migrate CLI location', () => { expect(await runningBuildSupportsPerFileMigrations('c-probe-none', deps)).to.equal(null) }) }) +} +function registerRefusalBehaviorTests() { describe('refusal remedy', () => { let warnStub beforeEach(() => { warnStub = sinon.stub(console, 'warn') }) @@ -162,4 +165,10 @@ describe('indexer migrate CLI location', () => { expect(message).to.contain('node ' + NEWEST_PATH) }) }) +} + +describe('indexer migrate CLI location', () => { + registerPathPrecedenceTests() + registerCapabilityDetectionTests() + registerRefusalBehaviorTests() }) From 3048e912307e6fcd84d8415b1b24e5e67001afb8 Mon Sep 17 00:00:00 2001 From: J-Dog Date: Fri, 18 Sep 2026 17:29:41 -0700 Subject: [PATCH 12/35] Route runtime environment access through config Move CLI and operation environment reads behind live config accessors while preserving names, defaults, coercion, and timing. Keep the decoupled structure gate at the existing signer-only baseline without widening exemptions. --- src/cli.js | 9 +++++---- src/cli/commands.js | 4 ++-- src/cli/dispatch.js | 8 ++++---- src/cli/output.js | 4 ++-- src/config/index.js | 6 ++++++ src/operations/module_operations/reset_modules.js | 2 +- src/operations/module_operations/update_modules.js | 10 +++++----- src/precheck.js | 4 ++-- 8 files changed, 27 insertions(+), 20 deletions(-) diff --git a/src/cli.js b/src/cli.js index b8a3601..812786e 100644 --- a/src/cli.js +++ b/src/cli.js @@ -19,7 +19,8 @@ const { version } = require('../package.json') const { preCheck } = require('./precheck') const { setVerbose } = require('./state') const { filterCommandParameters, resolveArgs } = require('./services/config_service') -const { HUB_MODULE_NAME } = require('./config') +const config = require('./config') +const { HUB_MODULE_NAME } = config const { redactSecrets } = require('./utils/helpers') const { installModules, @@ -147,7 +148,7 @@ async function maybeSelfUpdateBeforeUpdate(args, deps = {}) { } catch { return { moved: false, reason: 'unparsed-args' } // the action reports it } - if (process.env.XCHAIN_NODE_UPDATE_TARGET) return { moved: false, reason: 'already-reexecuted' } + if (config.XCHAIN_NODE_UPDATE_TARGET) return { moved: false, reason: 'already-reexecuted' } if (selfUpdate.selfUpdateDisabled()) return { moved: false, reason: 'disabled' } let tag = null @@ -180,7 +181,7 @@ async function maybeSelfUpdateBeforeUpdate(args, deps = {}) { } // The run continues in this process at the resolved tag: hand it on so // updateModules does not resolve the latest release a second time. - if (outcome && !outcome.moved) process.env.XCHAIN_NODE_UPDATE_TARGET = tag + if (outcome && !outcome.moved) config.XCHAIN_NODE_UPDATE_TARGET = tag return outcome } @@ -198,7 +199,7 @@ async function parseCommand() { formatCapabilityDrift, stakeValidator, unstakeValidator, restoreBootstrapInterface, startInterface, acquireCommandLock, noticeNewerRelease, refForPreCheck, commandRepairsHub, - maybeSelfUpdateBeforeUpdate, loadModule + maybeSelfUpdateBeforeUpdate, loadModule, config }) } diff --git a/src/cli/commands.js b/src/cli/commands.js index 117bd72..9fbbb36 100644 --- a/src/cli/commands.js +++ b/src/cli/commands.js @@ -15,7 +15,7 @@ * Commander setup and command definitions ********************************************************************/ function registerInstall(program, deps) { - const { filterCommandParameters, resolveArgs, installModules, syncSharedServicesAfterInstall } = deps + const { filterCommandParameters, resolveArgs, installModules, syncSharedServicesAfterInstall, config } = deps program .command('install') .description('Installs XChain services') @@ -34,7 +34,7 @@ function registerInstall(program, deps) { // commander assigns a flag matching a global option to the global, so reading // the install command's own opts would always see the default. Skips the // auto-download/restore and syncs from scratch. - if (program.opts().bootstrap === false) process.env.XCHAIN_NODE_NO_BOOTSTRAP = '1' + if (program.opts().bootstrap === false) config.XCHAIN_NODE_NO_BOOTSTRAP = '1' // defaultBranch null: an absent ref must reach installModules as null // so it resolves the latest release. Substituting 'master' here would // make the documented default install a branch install forever. diff --git a/src/cli/dispatch.js b/src/cli/dispatch.js index a5b204a..d9234ed 100644 --- a/src/cli/dispatch.js +++ b/src/cli/dispatch.js @@ -15,7 +15,7 @@ * Commander setup and command definitions ********************************************************************/ -function dispatchSettings() { +function dispatchSettings(config) { const commandsNeedingVersions = ['install', 'update', 'reinstall'] // Read-only commands only display state and never change which services // are installed/running, so they don't need to push local config to the @@ -34,11 +34,11 @@ function dispatchSettings() { // How long a non-mutating command blocks for a lock-holding mutator before // giving up (bounded so a read-only command pauses, then errors clearly, // rather than corrupting the stack by provisioning concurrently). Tunable. - const LOCK_WAIT_MS = parseInt(process.env.XCHAIN_NODE_LOCK_WAIT_MS || '15000', 10) || 15000 + const LOCK_WAIT_MS = parseInt(config.XCHAIN_NODE_LOCK_WAIT_MS || '15000', 10) || 15000 // How long a MUTATING command blocks for a lock holder before refusing. Zero // keeps the interactive contract below; an unattended caller sets it so a // scheduled run waits out a deploy instead of losing its work. - const MUTATING_LOCK_WAIT_MS = parseInt(process.env.XCHAIN_NODE_MUTATING_LOCK_WAIT_MS || '0', 10) || 0 + const MUTATING_LOCK_WAIT_MS = parseInt(config.XCHAIN_NODE_MUTATING_LOCK_WAIT_MS || '0', 10) || 0 return { commandsNeedingVersions, readOnlyCommands, mutatingCommands, LOCK_WAIT_MS, MUTATING_LOCK_WAIT_MS } } @@ -170,7 +170,7 @@ async function beforeAction(thisCommand, actionCommand, settings, deps) { } function installDispatch(program, deps) { - const settings = dispatchSettings() + const settings = dispatchSettings(deps.config) program.hook('preAction', (thisCommand, actionCommand) => beforeAction(thisCommand, actionCommand, settings, deps)) } diff --git a/src/cli/output.js b/src/cli/output.js index b700ee4..8ebc74a 100644 --- a/src/cli/output.js +++ b/src/cli/output.js @@ -188,8 +188,8 @@ function printValidatorConfiguration(s, deps) { console.log(' stake wallet : ' + w.stakeAddress + ' (' + coins.stakeCoin + ' for fees, holds the XCHAIN stake)') console.log(' DOGE wallet : ' + w.dogeAddress + ' (' + coins.dogeCoin + ' for price rounds and anchors)') console.log(' keys file : ' + WALLETS_FILE + ' (mode 0600; back it up)') - console.log(' DOGE signer : ' + (process.env.XCHAIN_NODE_HUB_SIGNER_DIR - ? process.env.XCHAIN_NODE_HUB_SIGNER_DIR + ' (operator-supplied, XCHAIN_NODE_HUB_SIGNER_DIR)' + console.log(' DOGE signer : ' + (deps.config.XCHAIN_NODE_HUB_SIGNER_DIR + ? deps.config.XCHAIN_NODE_HUB_SIGNER_DIR + ' (operator-supplied, XCHAIN_NODE_HUB_SIGNER_DIR)' : (getSignerMountDir() || '(missing; re-run validator init)'))) } else { console.log(' wallets : (none; re-run validator init, or run your own signer via XCHAIN_NODE_HUB_SIGNER_DIR)') diff --git a/src/config/index.js b/src/config/index.js index ea10ad9..7028545 100644 --- a/src/config/index.js +++ b/src/config/index.js @@ -236,7 +236,9 @@ module.exports = { get XCHAIN_NODE_GO_LIVE() { return process.env.XCHAIN_NODE_GO_LIVE }, get XCHAIN_NODE_GPG_BIN() { return process.env.XCHAIN_NODE_GPG_BIN || 'gpg' }, get XCHAIN_NODE_HUB_SIGNER_DIR() { return process.env.XCHAIN_NODE_HUB_SIGNER_DIR }, + get XCHAIN_NODE_LOCK_WAIT_MS() { return process.env.XCHAIN_NODE_LOCK_WAIT_MS }, get XCHAIN_NODE_LOCK_DIR() { return process.env.XCHAIN_NODE_LOCK_DIR }, + get XCHAIN_NODE_MUTATING_LOCK_WAIT_MS() { return process.env.XCHAIN_NODE_MUTATING_LOCK_WAIT_MS }, get XCHAIN_NODE_NO_BOOTSTRAP() { return process.env.XCHAIN_NODE_NO_BOOTSTRAP }, get XCHAIN_NODE_NO_TELEMETRY() { return process.env.XCHAIN_NODE_NO_TELEMETRY || '' }, get XCHAIN_NODE_REINDEX_LEDGER_DIR() { return process.env.XCHAIN_NODE_REINDEX_LEDGER_DIR }, @@ -248,6 +250,10 @@ module.exports = { get XCHAIN_NODE_STAKE_WIF() { return process.env.XCHAIN_NODE_STAKE_WIF }, get XCHAIN_NODE_STOP_TIMEOUT_SECONDS() { return process.env.XCHAIN_NODE_STOP_TIMEOUT_SECONDS }, get XCHAIN_NODE_TELEMETRY_URL() { return process.env.XCHAIN_NODE_TELEMETRY_URL }, + get XCHAIN_NODE_UPDATE_TARGET() { return process.env.XCHAIN_NODE_UPDATE_TARGET }, + set XCHAIN_NODE_NO_BOOTSTRAP(value) { process.env.XCHAIN_NODE_NO_BOOTSTRAP = value }, + set XCHAIN_NODE_UPDATE_TARGET(value) { process.env.XCHAIN_NODE_UPDATE_TARGET = value }, + hostEnv() { return process.env }, ...require('./env_views').bindEnvViews({ read: (name) => process.env[name], copy: () => ({ ...process.env }) }), // Below this line, one entry per environment variable this service reads. // They are passed straight through rather than parsed, because almost all diff --git a/src/operations/module_operations/reset_modules.js b/src/operations/module_operations/reset_modules.js index 8f7aa6a..c9be512 100644 --- a/src/operations/module_operations/reset_modules.js +++ b/src/operations/module_operations/reset_modules.js @@ -103,7 +103,7 @@ async function resolveResetPaths(context) { console.log(`No ${NODE_MODULE_NAME} container is installed for ${coin} ${network}; there is no node data to clear.`) } else { const envState = config.XCHAIN_NODE_DATA_DIR && config.XCHAIN_NODE_DATA_DIR.trim() !== '' - ? `set to ${process.env.XCHAIN_NODE_DATA_DIR}` + ? `set to ${config.XCHAIN_NODE_DATA_DIR}` : 'UNSET in this shell (non-interactive shells do not source the profile)' console.log(`Aborted: cannot resolve the ${coin} ${network} node datadir. No data was touched.`) console.log(` Container ${resolved.containerName} reported no /root/.${coin} bind mount ` diff --git a/src/operations/module_operations/update_modules.js b/src/operations/module_operations/update_modules.js index 7dba921..30f815f 100644 --- a/src/operations/module_operations/update_modules.js +++ b/src/operations/module_operations/update_modules.js @@ -1,9 +1,9 @@ 'use strict' -let DB_MODULE_NAME, HUB_MODULE_NAME, NODE_MODULE_NAME, SEP, SYNC_MODULE_NAME, assertHubNotBehind, assertRequiredMigrationsApplied, db, getModuleBranch, installModule, installTargetService, releaseManifestService, stateModule, validatorService, versionService, withInstallTarget +let DB_MODULE_NAME, HUB_MODULE_NAME, NODE_MODULE_NAME, SEP, SYNC_MODULE_NAME, assertHubNotBehind, assertRequiredMigrationsApplied, config, db, getModuleBranch, installModule, installTargetService, releaseManifestService, stateModule, validatorService, versionService, withInstallTarget function configure(dependencies) { - ({ DB_MODULE_NAME, HUB_MODULE_NAME, NODE_MODULE_NAME, SEP, SYNC_MODULE_NAME, assertHubNotBehind, assertRequiredMigrationsApplied, db, getModuleBranch, installModule, installTargetService, releaseManifestService, stateModule, validatorService, versionService, withInstallTarget } = dependencies) + ({ DB_MODULE_NAME, HUB_MODULE_NAME, NODE_MODULE_NAME, SEP, SYNC_MODULE_NAME, assertHubNotBehind, assertRequiredMigrationsApplied, config, db, getModuleBranch, installModule, installTargetService, releaseManifestService, stateModule, validatorService, versionService, withInstallTarget } = dependencies) } /** @@ -47,8 +47,8 @@ async function updateModules(servicesList, ref = null, opts = {}) { } // No ref: the update target is remembered from the last install/update, or classified from the checkouts on a node an older CLI installed. - const target = process.env.XCHAIN_NODE_UPDATE_TARGET - ? { kind: 'release', ref: process.env.XCHAIN_NODE_UPDATE_TARGET, inferred: false } + const target = config.XCHAIN_NODE_UPDATE_TARGET + ? { kind: 'release', ref: config.XCHAIN_NODE_UPDATE_TARGET, inferred: false } : await resolveUpdateTarget() if (target.kind === 'branch') { @@ -58,7 +58,7 @@ async function updateModules(servicesList, ref = null, opts = {}) { } // A release node with no ref: the LATEST release (the recorded tag is where the node is, not where it is going), never a branch fallback. A lookup failure stops the run with nothing changed. The re-executed child of a CLI self-update already knows the tag its parent resolved. - const releaseRef = process.env.XCHAIN_NODE_UPDATE_TARGET || null + const releaseRef = config.XCHAIN_NODE_UPDATE_TARGET || null return withInstallTarget(releaseRef, async () => updateModulesOnBranch(list, null, runOpts), { fallbackToBranch: false }) } diff --git a/src/precheck.js b/src/precheck.js index bc6874d..fbe50d9 100644 --- a/src/precheck.js +++ b/src/precheck.js @@ -18,7 +18,7 @@ const fs = require('fs') const { dataDir, moduleDir, tmpDir, containersFilesDir, - EXTERNAL_DB } = require('./config') + EXTERNAL_DB, hostEnv } = require('./config') const { db, isVerbose } = require('./state') const { redactSecrets } = require('./utils/helpers') const { checkDockerInstalledAndReachable, createDockerNetwork, checkContainerdDataRootRelocation, checkMemoryLimitSupport } = require('./services/docker_service') @@ -308,7 +308,7 @@ async function preCheck(checkVersions = false, syncHubConfig = true, moduleRef = // state-changing command after it did the same. Same precedence as the container // env: a host-env HUB_API_KEY still wins, the sidecar only fills an empty one, and // this never mints (a host with no sidecar stays keyless exactly as before). - await applyHubApiKeyFromSidecar(process.env) + await applyHubApiKeyFromSidecar(hostEnv()) await installHubAtRef(moduleRef) From 2c7b238a143dcefd8a89c3671a2e5729a98442bf Mon Sep 17 00:00:00 2001 From: J-Dog Date: Sat, 19 Sep 2026 13:51:21 -0700 Subject: [PATCH 13/35] fix(ci): add an all-coin dispatch option to the nightly e2e matrix MT3 needs one dispatched run that covers bitcoin, litecoin and dogecoin against a named ref. The schedule trigger already covers all three coins but never takes a ref, and a dispatch could name a ref but only one coin. Add coin: all to the workflow_dispatch choice input and extend the matrix expression so all expands to the same three-coin array the schedule uses, while every existing single-coin choice and the scheduled behaviour are unchanged. --- .github/workflows/nightly-e2e.yml | 20 ++++++++++++++++---- 1 file changed, 16 insertions(+), 4 deletions(-) diff --git a/.github/workflows/nightly-e2e.yml b/.github/workflows/nightly-e2e.yml index 2f51b26..8716333 100644 --- a/.github/workflows/nightly-e2e.yml +++ b/.github/workflows/nightly-e2e.yml @@ -13,7 +13,7 @@ name: E2E (regtest) # "which chain". A suite filter is appended only when one is set, so the common # full-suite run stays short. run-name: >- - E2E ${{ github.event_name == 'schedule' && 'all coins' || inputs.coin }} regtest @ ${{ inputs.ref || 'develop' }}${{ inputs.suite && format(' [{0} only]', inputs.suite) || '' }} + E2E ${{ (github.event_name == 'schedule' || inputs.coin == 'all') && 'all coins' || inputs.coin }} regtest @ ${{ inputs.ref || 'develop' }}${{ inputs.suite && format(' [{0} only]', inputs.suite) || '' }} # Cross-component integration gate. Boots the FULL XChain stack on regtest via # xchain-node - which clones every service repo at the REF THIS RUN NAMES and @@ -68,9 +68,15 @@ on: workflow_dispatch: inputs: coin: - description: 'Coin to test' + # 'all' expands to the same three-coin array the schedule uses, so a + # dispatch can name a ref (release/vX.Y.Z) AND cover all three coins in + # one run - the schedule can name all three coins but never a ref, and + # a single-coin dispatch can name a ref but never more than one coin; + # this is the third combination, needed only for the release-freeze + # acceptance gate (MT3), which the other two cannot satisfy together. + description: 'Coin to test, or "all" for bitcoin+litecoin+dogecoin in one run' type: choice - options: [bitcoin, litecoin, dogecoin] + options: [bitcoin, litecoin, dogecoin, all] default: bitcoin suite: # ONE suite instead of the whole action set, so fixing a defect costs a boot @@ -140,10 +146,16 @@ jobs: # chosen coin, because a subset proves a fix and must not pretend to be a train. # The three legs are independent stacks, so fail-fast would throw away two # answers to report one; a release needs all three verdicts, not the first. + # + # coin: 'all' is the third case: a dispatch names a ref a schedule can never + # name, and needs to cover all three coins the way a schedule does. It reuses + # the identical three-coin literal so the expanded matrix is byte-identical + # to the schedule case; anything else falls through to the single-coin shape, + # defaulting to bitcoin the same way an absent input always has. strategy: fail-fast: false matrix: - coin: ${{ fromJSON(github.event_name == 'schedule' && '["bitcoin","litecoin","dogecoin"]' || format('["{0}"]', github.event.inputs.coin || 'bitcoin')) }} + coin: ${{ fromJSON(github.event_name == 'schedule' && '["bitcoin","litecoin","dogecoin"]' || github.event.inputs.coin == 'all' && '["bitcoin","litecoin","dogecoin"]' || format('["{0}"]', github.event.inputs.coin || 'bitcoin')) }} # A BTC full action suite alone runs ~1h50m of wall clock, and the security # and performance suites are sequenced AFTER it, so at 120 the two of them # shared whatever minutes the action suite happened to leave - usually none. From acf63fd34010f5e6465f4db7e6b75439d70c943c Mon Sep 17 00:00:00 2001 From: J-Dog Date: Sat, 19 Sep 2026 21:19:46 -0700 Subject: [PATCH 14/35] fix(module-operations): exec the decoder's clear script at its snake_case path xchain-decoder renamed src/clear-reorg-halt.js to src/clear_reorg_halt.js; the clear-reorg-halt command runs that file inside the decoder container by path. Re-applied at the current location of the function and its test, module_operations/module_controls.js, after the module_operations split moved both off the file this fix originally touched. --- src/operations/module_operations/module_controls.js | 4 ++-- test/unit/module_operations.test/03_module_controls.test.js | 6 +++--- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/src/operations/module_operations/module_controls.js b/src/operations/module_operations/module_controls.js index 53cc674..abda88a 100644 --- a/src/operations/module_operations/module_controls.js +++ b/src/operations/module_operations/module_controls.js @@ -128,7 +128,7 @@ async function startModules(servicesList) { // Audited clear of a decoder's durable REORG_HALT marker, run inside the decoder // container so it uses the service's own DB credentials and code -// (xchain-decoder/src/clear-reorg-halt.js checks the database is intact, then +// (xchain-decoder/src/clear_reorg_halt.js checks the database is intact, then // records the clear as an events row with the reason). One decoder per // coin/network; `servicesList` is the filtered map the CLI builds. Returns true // only when every targeted decoder answered exit 0. @@ -139,7 +139,7 @@ async function clearDecoderReorgHalt(servicesList, { reason, force = false, dryR console.log('clear-reorg-halt: --reason must say, in at least 8 characters, why this database is known good; it is recorded with the clear.') return false } - const args = ['node', 'src/clear-reorg-halt.js'] + const args = ['node', 'src/clear_reorg_halt.js'] if (reasonText) args.push('--reason', reasonText) if (force) args.push('--force') if (dryRun) args.push('--dry-run') diff --git a/test/unit/module_operations.test/03_module_controls.test.js b/test/unit/module_operations.test/03_module_controls.test.js index 59926c5..af02b98 100644 --- a/test/unit/module_operations.test/03_module_controls.test.js +++ b/test/unit/module_operations.test/03_module_controls.test.js @@ -207,7 +207,7 @@ describe('moduleOperations', function () { expect(ok).to.be.true expect(stubs.db.getModuleContainer.calledWith('xchain-decoder', 'bitcoin', 'mainnet')).to.be.true expect(stubs.execContainer.calledWith('container-id-123', - ['node', 'src/clear-reorg-halt.js', '--reason', REASON])).to.be.true + ['node', 'src/clear_reorg_halt.js', '--reason', REASON])).to.be.true }) it('passes --force and --dry-run through', async function () { @@ -215,7 +215,7 @@ describe('moduleOperations', function () { const ops = loadOperations(stubs) await ops.clearDecoderReorgHalt({ bitcoin: { mainnet: ['xchain-decoder'] } }, { reason: REASON, force: true, dryRun: true }) expect(stubs.execContainer.firstCall.args[1]).to.deep.equal( - ['node', 'src/clear-reorg-halt.js', '--reason', REASON, '--force', '--dry-run']) + ['node', 'src/clear_reorg_halt.js', '--reason', REASON, '--force', '--dry-run']) }) it('refuses a trivial reason without touching any container', async function () { @@ -246,7 +246,7 @@ describe('moduleOperations', function () { const ok = await ops.clearDecoderReorgHalt({ bitcoin: { mainnet: ['xchain-decoder'] } }, { dryRun: true }) expect(ok).to.be.true expect(stubs.execContainer.firstCall.args[1]).to.deep.equal( - ['node', 'src/clear-reorg-halt.js', '--dry-run']) + ['node', 'src/clear_reorg_halt.js', '--dry-run']) }) it('refuses a real clear with no reason at all without touching any container', async function () { From 7b0072e38702bfb7f9163089ab6a710f27d83f25 Mon Sep 17 00:00:00 2001 From: J-Dog Date: Sun, 20 Sep 2026 11:12:57 -0700 Subject: [PATCH 15/35] test(secret-env): hard-fail missing hub pin --- test/unit/secret_env.test.js | 22 ++++------------------ 1 file changed, 4 insertions(+), 18 deletions(-) diff --git a/test/unit/secret_env.test.js b/test/unit/secret_env.test.js index 9d6ec33..003a4dd 100644 --- a/test/unit/secret_env.test.js +++ b/test/unit/secret_env.test.js @@ -70,13 +70,8 @@ const CONTAINER_ID = 'a'.repeat(64) const coinSidecar = path.resolve(configDir, 'bitcoin-mainnet') + '.local' const coinMain = path.resolve(configDir, 'bitcoin-mainnet') -// xchain-hub as a checkout, not just the one file this suite reads from it: a -// present repo missing the pinned table is a moved or renamed file, while an -// absent repo is a standalone install with no sibling to compare against. -const HUB_DIR = path.join(__dirname, '../../../xchain-hub') -const HUB_TABLE = path.join(HUB_DIR, 'src/secret_env.js') -const HUB_PRESENT = fs.existsSync(HUB_DIR) -const REQUIRE_SIBLINGS = process.env.XCHAIN_REQUIRE_SIBLINGS === '1' +// Hard pin: a missing sibling or table path is a test failure. +const HUB_TABLE = path.join(__dirname, '../../../xchain-hub/src/secret_env.js') // The platform checkout as a directory, not just the one script this suite // reads from it: a present checkout missing the pinned tool is a moved or @@ -129,17 +124,8 @@ describe('secret-env', function () { it('agrees with the xchain-hub table on every key both own', function () { // xchain-node composes the hub container's env, so if the two tables // disagreed on a name the hub would boot without its DB password. - if (!HUB_PRESENT) { - if (REQUIRE_SIBLINGS) - throw new Error('XCHAIN_REQUIRE_SIBLINGS=1 but xchain-hub is not checked out at ' + HUB_DIR) - this.skip() // sibling repo not checked out - return - } - // A checked-out hub with no table at the pinned path is a moved or - // renamed file, not a missing sibling, and skipping here would hide - // exactly the drift this test exists to catch. - expect(fs.existsSync(HUB_TABLE), 'xchain-hub is checked out at ' + HUB_DIR - + ' but has no secret-env table at ' + HUB_TABLE).to.equal(true) + expect(fs.existsSync(HUB_TABLE), + 'xchain-hub secret-env table is missing at pinned path ' + HUB_TABLE).to.equal(true) const hubAliases = require(HUB_TABLE).SECRET_ENV_ALIASES for (const [legacy, preferred] of Object.entries(hubAliases)) { expect(secretEnv.SECRET_ENV_ALIASES[legacy], 'xchain-node is missing ' + legacy) From cb2c8dab4c8cf9b850fe1d7b37ff61fa62108972 Mon Sep 17 00:00:00 2001 From: J-Dog Date: Sun, 20 Sep 2026 15:31:55 -0700 Subject: [PATCH 16/35] Remove obsolete indexer migrate CLI path --- src/utils/indexer_migrate_cli.js | 12 ++++---- test/unit/utils/indexer_migrate_cli.test.js | 33 +++++---------------- 2 files changed, 14 insertions(+), 31 deletions(-) diff --git a/src/utils/indexer_migrate_cli.js b/src/utils/indexer_migrate_cli.js index 4a9655c..4ed911a 100644 --- a/src/utils/indexer_migrate_cli.js +++ b/src/utils/indexer_migrate_cli.js @@ -17,12 +17,12 @@ * found. ********************************************************************/ -// Newest first. The indexer has moved the CLI twice: from the top of src/ to -// src/migration/, then (v0.19.0) into src/db/migration/ alongside the rest of -// the db layer. The deploy guard reads the container being REPLACED, which -// can run a build from any of the three layouts, so every spelling stays -// readable for as long as a supported indexer build carries it. -const MIGRATE_CLI_PATHS = ['src/db/migration/migrate.js', 'src/migration/migrate.js', 'src/migrate.js'] +// Newest first. In v0.19.0 the indexer moved the CLI from the top of src/ into +// src/db/migration/ alongside the rest of the db layer. The deploy guard reads +// the container being REPLACED, which can run a build from either layout, so +// both spellings stay readable for as long as a supported indexer build +// carries them. +const MIGRATE_CLI_PATHS = ['src/db/migration/migrate.js', 'src/migrate.js'] // Container name -> the path its last successful read answered at. The remedy // the refusal prints runs on THAT build, so it has to name the path the read diff --git a/test/unit/utils/indexer_migrate_cli.test.js b/test/unit/utils/indexer_migrate_cli.test.js index f2fa740..24942df 100644 --- a/test/unit/utils/indexer_migrate_cli.test.js +++ b/test/unit/utils/indexer_migrate_cli.test.js @@ -10,9 +10,9 @@ // Where the indexer's operator migration CLI is read from inside a container. // The deploy guard reads the container being REPLACED, which can run an indexer -// build from any of three layouts (top of src/, src/migration/, or the -// v0.19.0+ src/db/migration/), so the probe has to answer for all of them and -// the refusal has to print the path that running build really carries. +// build from either layout (top of src/ or the v0.19.0+ src/db/migration/), so +// the probe has to answer for both and the refusal has to print the path that +// running build really carries. const sinon = require('sinon') const { expect } = require('chai') @@ -26,7 +26,6 @@ const { const GATED = '2026-07-24-pubkeys-widen-uncompressed.sql' const NEWEST_PATH = 'src/db/migration/migrate.js' -const NEW_PATH = 'src/migration/migrate.js' const OLD_PATH = 'src/migrate.js' // Source text for a CLI that parses --file, and for one that predates it. @@ -50,26 +49,18 @@ function registerPathPrecedenceTests() { // than silently reappearing as the 'could not be read' refusal on the // next roll. it('pins every known CLI layout, newest first', () => { - expect(MIGRATE_CLI_PATHS).to.deep.equal([NEWEST_PATH, NEW_PATH, OLD_PATH]) + expect(MIGRATE_CLI_PATHS).to.deep.equal([NEWEST_PATH, OLD_PATH]) }) describe('readMigrateCli', () => { - it('reads the v0.19.0+ db/migration layout before either older path', async () => { - const cat = catFor({ [NEWEST_PATH]: WITH_FILE, [NEW_PATH]: WITHOUT_FILE, [OLD_PATH]: WITHOUT_FILE }) + it('reads the v0.19.0+ db/migration layout before the older path', async () => { + const cat = catFor({ [NEWEST_PATH]: WITH_FILE, [OLD_PATH]: WITHOUT_FILE }) const found = await readMigrateCli(cat, 'c-newest') expect(found).to.deep.equal({ cliPath: NEWEST_PATH, source: WITH_FILE }) expect(cat.firstCall.args[1]).to.equal(NEWEST_PATH) expect(migrateCliPathFor('c-newest')).to.equal(NEWEST_PATH) }) - it('reads the moved CLI before the pre-move path on a build without the newest layout', async () => { - const cat = catFor({ [NEW_PATH]: WITH_FILE, [OLD_PATH]: WITHOUT_FILE }) - const found = await readMigrateCli(cat, 'c-both') - expect(found).to.deep.equal({ cliPath: NEW_PATH, source: WITH_FILE }) - expect(cat.firstCall.args[1]).to.equal(NEWEST_PATH) - expect(migrateCliPathFor('c-both')).to.equal(NEW_PATH) - }) - it('falls back to the pre-move path on a build that still carries it', async () => { const found = await readMigrateCli(catFor({ [OLD_PATH]: WITH_FILE }), 'c-old') expect(found).to.deep.equal({ cliPath: OLD_PATH, source: WITH_FILE }) @@ -77,7 +68,7 @@ function registerPathPrecedenceTests() { }) it('counts an empty read as absent, as the single-path probe did', async () => { - const found = await readMigrateCli(catFor({ [NEW_PATH]: '', [OLD_PATH]: WITH_FILE }), 'c-empty') + const found = await readMigrateCli(catFor({ [NEWEST_PATH]: '', [OLD_PATH]: WITH_FILE }), 'c-empty') expect(found.cliPath).to.equal(OLD_PATH) }) @@ -97,7 +88,7 @@ function registerCapabilityDetectionTests() { }) it('sees --file support on a build from after the move', async () => { - const deps = { getDockerContainerFileCat: catFor({ [NEW_PATH]: WITH_FILE }) } + const deps = { getDockerContainerFileCat: catFor({ [NEWEST_PATH]: WITH_FILE }) } expect(await runningBuildSupportsPerFileMigrations('c-probe-new', deps)).to.equal(true) }) @@ -143,18 +134,10 @@ function registerRefusalBehaviorTests() { expect(message).to.contain('node ' + OLD_PATH + ' --file ' + GATED) }) - it('names the moved CLI path when the running build carries that', async () => { - const message = await refusalFor({ [NEW_PATH]: WITH_FILE }) - expect(message, 'the deploy must be refused').to.not.equal(null) - expect(message).to.contain('node ' + NEW_PATH + ' --file ' + GATED) - expect(message).to.not.contain('node ' + OLD_PATH) - }) - it('names the v0.19.0+ db/migration CLI path when the running build carries that', async () => { const message = await refusalFor({ [NEWEST_PATH]: WITH_FILE }) expect(message, 'the deploy must be refused').to.not.equal(null) expect(message).to.contain('node ' + NEWEST_PATH + ' --file ' + GATED) - expect(message).to.not.contain('node ' + NEW_PATH) expect(message).to.not.contain('node ' + OLD_PATH) }) From 0aea8e5eb51f5318811d7beaaa905920ea730e44 Mon Sep 17 00:00:00 2001 From: J-Dog Date: Sun, 20 Sep 2026 17:27:37 -0700 Subject: [PATCH 17/35] Add regtest validator P2P defaults --- .github/workflows/nightly-e2e.yml | 1 + src/cli/output.js | 4 +-- .../validator_service/validator_paths.js | 6 ++-- .../two_stack_legs.test.js | 3 +- .../12_coin_wallets.test.js | 28 +++++++++++++++++++ 5 files changed, 36 insertions(+), 6 deletions(-) diff --git a/.github/workflows/nightly-e2e.yml b/.github/workflows/nightly-e2e.yml index 8716333..f84bd4b 100644 --- a/.github/workflows/nightly-e2e.yml +++ b/.github/workflows/nightly-e2e.yml @@ -373,6 +373,7 @@ jobs: # federation; this venue is a single hub, so the run's own clock is fine. run: | node src/index.js validator init \ + --network regtest \ --oracle-epoch-start "$(date -u +%s)000" \ --capabilities price,cross_chain,oracle_publish,attestation { diff --git a/src/cli/output.js b/src/cli/output.js index 8ebc74a..73f541e 100644 --- a/src/cli/output.js +++ b/src/cli/output.js @@ -94,8 +94,8 @@ function registerValidatorInit(validator, deps) { .description('Generate a validator signing key + config so the hub runs in validator mode') .option('--seed-nodes ', 'comma-separated peer addresses (host:port,host:port)') .option('--p2p-addr ', 'this validator\'s public address (host:port)') - .option('--p2p-port ', 'P2P listen port (10002 testnet, 10001 mainnet; default 10001)') - .option('--network ', 'federation to join: testnet or mainnet (default: implied by --p2p-port)') + .option('--p2p-port ', 'P2P listen port (10003 regtest, 10002 testnet, 10001 mainnet; default 10001)') + .option('--network ', 'federation to join: regtest, testnet, or mainnet (default: implied by --p2p-port)') .option('--oracle-epoch-start ', 'shared oracle epoch start (unix ms); defaults to the known federation value') .option('--capabilities ', 'enabled capabilities (default price,cross_chain,oracle_publish,attestation)') .option('--import-stake-key', 'use your own BTC stake key: prompts for the WIF (or set XCHAIN_NODE_STAKE_WIF)') diff --git a/src/services/validator_service/validator_paths.js b/src/services/validator_service/validator_paths.js index 489d617..549379a 100644 --- a/src/services/validator_service/validator_paths.js +++ b/src/services/validator_service/validator_paths.js @@ -115,10 +115,10 @@ let SIGNER_MODULES_DIR = path.join(SIGNER_DIR, 'node_modules') const SIGNER_CONTAINER_DIR = '/XChainHub/operator-signer' const SIGNER_CONTAINER_PATH = SIGNER_CONTAINER_DIR + '/signer.js' -// The P2P port names the federation: one host can serve both networks later, +// The P2P port names the federation: one host can serve multiple networks later, // so the port is the declared split rather than anything the protocol enforces. -const NETWORK_BY_P2P_PORT = { 10001: 'mainnet', 10002: 'testnet' } -const P2P_PORT_BY_NETWORK = { mainnet: 10001, testnet: 10002 } +const NETWORK_BY_P2P_PORT = { 10001: 'mainnet', 10002: 'testnet', 10003: 'regtest' } +const P2P_PORT_BY_NETWORK = { mainnet: 10001, testnet: 10002, regtest: 10003 } // Oracle round-numbering anchor per federation. A hub with a different value // computes different round numbers and its submissions never line up, so the diff --git a/test/unit/nightly_e2e_workflow.test/two_stack_legs.test.js b/test/unit/nightly_e2e_workflow.test/two_stack_legs.test.js index 529ba3e..93ca875 100644 --- a/test/unit/nightly_e2e_workflow.test/two_stack_legs.test.js +++ b/test/unit/nightly_e2e_workflow.test/two_stack_legs.test.js @@ -189,7 +189,7 @@ function registerValidatorModeChecks() { for (const coin of ['litecoin', 'dogecoin']) { it(coin + ': inits a cross_chain-capable identity and exports the seed so the hub finalizes the gas lock', function () { const { calls, exported } = runValidatorStep(coin) - expect(calls[0]).to.match(/^src\/index\.js validator init --oracle-epoch-start \d+ --capabilities [a-z_,]+$/) + expect(calls[0]).to.match(/^src\/index\.js validator init --network regtest --oracle-epoch-start \d+ --capabilities [a-z_,]+$/) expect(calls[0].split('--capabilities ')[1].split(',')).to.include('cross_chain') expect(calls[calls.length - 1]).to.equal('src/index.js validator status') // The indexer URLs ride the same export: the hub's cross-chain @@ -212,6 +212,7 @@ function registerValidatorModeChecks() { it('bitcoin (opt-in): keeps the identity but never seeds, so the input changes nothing beyond the price regime it documents', function () { const { calls, exported } = runValidatorStep('bitcoin') + expect(calls[0]).to.match(/^src\/index\.js validator init --network regtest --oracle-epoch-start \d+ --capabilities [a-z_,]+$/) expect(calls[calls.length - 1]).to.equal('src/index.js validator status') expect(exported).to.deep.equal({ HUB_NETWORK: 'regtest', ORACLE_MIN_SUBMISSIONS: '1' }) }) diff --git a/test/unit/validator_service.test/12_coin_wallets.test.js b/test/unit/validator_service.test/12_coin_wallets.test.js index d1cff9f..798fa6e 100644 --- a/test/unit/validator_service.test/12_coin_wallets.test.js +++ b/test/unit/validator_service.test/12_coin_wallets.test.js @@ -15,9 +15,14 @@ const { configStub } = require('../../helpers/config_stub') const { expect } = require('chai') const proxyquire = require('proxyquire').noCallThru() const path = require('path') +// ValidatorService loads the SDK lazily, but this suite exercises real wallet +// generation. Pay the SDK's module-load cost while Mocha loads the test file, +// outside the timed test body, so the 10-second verify timeout measures init. +require('@dankest-llc/xchain-sdk') // Fake config dir (never touches the real filesystem) const FAKE_CONFIG_DIR = '/tmp/test-xchain-config' const FAKE_VALIDATOR_DIR = path.join(FAKE_CONFIG_DIR, 'validator') +const FAKE_SETTINGS_FILE = path.join(FAKE_VALIDATOR_DIR, 'validator.json') // The capability config lives in its OWN directory: that directory is what the // hub container bind-mounts, and a single-FILE bind mount breaks `docker cp` // against the container for every path. signing.key must stay outside it. @@ -185,6 +190,29 @@ describe('ValidatorService', function () { expect(result.network).to.equal('testnet') expect(result.SEED_NODES).to.deep.equal(['01','02','03','04','05'].map(n => 'ws://validator' + n + '.xchain.io:10002')) }) + + it('records regtest on its local port without mainnet federation seeds', async function () { + const fs = makeFs() + const vs = loadValidatorService(fs) + const result = await vs.initValidator({ network: 'regtest', wallets: false }) + expect(result.network).to.equal('regtest') + expect(result.P2P_PORT).to.equal(10003) + expect(result.P2P_VALIDATOR_ADDR).to.equal('0.0.0.0:10003') + expect(result.SEED_NODES).to.deep.equal([]) + const write = fs.writeFileSync.getCalls().find(c => c.args[0] === FAKE_SETTINGS_FILE) + const recorded = JSON.parse(write.args[1]) + expect(recorded.network).to.equal('regtest') + expect(recorded.P2P_PORT).to.equal(10003) + expect(recorded.P2P_VALIDATOR_ADDR).to.equal('0.0.0.0:10003') + expect(recorded.SEED_NODES).to.deep.equal([]) + }) + + it('recognizes P2P port 10003 as regtest without federation seeds', async function () { + const vs = loadValidatorService(makeFs()) + const result = await vs.initValidator({ p2pPort: '10003', wallets: false }) + expect(result.network).to.equal('regtest') + expect(result.SEED_NODES).to.deep.equal([]) + }) }) }) From cb4ebc1ec0c2ee943b467caf552093f28fe08485 Mon Sep 17 00:00:00 2001 From: J-Dog Date: Sun, 20 Sep 2026 18:06:08 -0700 Subject: [PATCH 18/35] test: isolate registry signer config with fixture --- test/unit/service_registry.test.js | 54 ++++++++++++++++++++---------- 1 file changed, 37 insertions(+), 17 deletions(-) diff --git a/test/unit/service_registry.test.js b/test/unit/service_registry.test.js index 3201ee9..e568577 100644 --- a/test/unit/service_registry.test.js +++ b/test/unit/service_registry.test.js @@ -20,6 +20,10 @@ const sinon = require('sinon') const { expect } = require('chai') const proxyquire = require('proxyquire').noCallThru() +const fs = require('fs') +const os = require('os') +const path = require('path') +const { configStub } = require('../helpers/config_stub') const { SERVICE_REGISTRY, XChainService, @@ -32,25 +36,26 @@ const { // via ModuleService's export, and the hub descriptor via SERVICE_REGISTRY // data + a tiny local re-implementation mirror is NOT used; instead we assert // the registry shape directly plus drive the real docker builder. -function loadModuleService() { +function loadValidatorFixture(configDir) { + return proxyquire('../../src/services/validator_service', { + '../config': configStub({ configDir }), + './config_service': { + ensureHubApiKey: async () => ({ generated: false }), + readHubApiKey: async () => ({ present: false }) + } + }) +} + +function loadModuleService(validatorService) { return proxyquire('../../src/services/module_service', { // ModuleService only pulls ValidatorService in lazily (hub caps), and // its top-level requires resolve fine without a live DB when we don't // call installModule. buildModuleDockerArgs itself has no side effects. // - // ValidatorService is stubbed to a machine with NO validator state. Left - // unstubbed, the lazy require reads config/validator/ off the REAL - // filesystem, and on any box that has run `validator init` the hub - // volume assertions below then see that machine's signer mount (the - // "no static volumes when unconfigured" case failed exactly that way on - // an operator checkout, 2026-09-11) while CI, which has no such - // directory, passes. Tests that WANT a mount stub their own. - './validator_service': { - getCapabilityConfigMountDir: () => null, - getSignerMountDir: () => null, - CAPS_CONTAINER_DIR: '/validator', - SIGNER_CONTAINER_DIR: '/XChainHub/operator-signer' - }, + // Use the real validator config reader against the test fixture. This + // keeps an initialized config directory on the test host out of the + // unconfigured hub case while still exercising signer discovery. + './validator_service': validatorService, './config_service': { getModuleDir: (m) => '/modules/' + m, getModuleTmpDir: (m) => '/tmp/' + m, @@ -88,11 +93,24 @@ const ENV = { } let ms +let validatorFixtureDir +let validatorFixtureService function reloadModuleService() { - ms = loadModuleService() + validatorFixtureDir = fs.mkdtempSync(path.join(os.tmpdir(), 'xchain-service-registry-')) + validatorFixtureService = loadValidatorFixture(validatorFixtureDir) + ms = loadModuleService(validatorFixtureService) +} + +function removeValidatorFixture() { + if (!validatorFixtureDir) return + fs.rmSync(validatorFixtureDir, { recursive: true, force: true }) + validatorFixtureDir = null + validatorFixtureService = null } +afterEach(removeValidatorFixture) + describe('SERVICE_REGISTRY', function () { describe('coverage parity with canonical service enums', function () { @@ -200,10 +218,13 @@ describe('SERVICE_REGISTRY', function () { }) it('hub: singleton, unconditional single port, no static volumes when unconfigured', function () { + const signerLookup = sinon.spy(validatorFixtureService, 'getSignerMountDir') const r = ms.buildModuleDockerArgs(HUB_MODULE_NAME, ENV, 'bitcoin', 'mainnet') expect(r.singleton).to.equal(true) expect(r.portArgs).to.deep.equal(['-p', '10000:10000']) - // No HUB_CAPABILITY_CONFIG in env and no signer dir env => no volumes. + expect(validatorFixtureService.VALIDATOR_DIR).to.equal(path.join(validatorFixtureDir, 'validator')) + expect(signerLookup.calledOnce).to.equal(true) + // No HUB_CAPABILITY_CONFIG in env and no signer in the fixture means no volumes. expect(r.volumeArgs).to.deep.equal([]) }) }) @@ -250,7 +271,6 @@ describe('SERVICE_REGISTRY', function () { beforeEach(reloadModuleService) it('hub: mounts the generated DOGE signer (ro) with this package\'s node_modules beside it', function () { - const path = require('path') const ms2 = proxyquire('../../src/services/module_service', { './config_service': { getUtxoTrackerVolumeName: () => 'v', getModuleDir: (m) => '/m/' + m, From 4c2429c24949eab59f9882aa1999921167f256da Mon Sep 17 00:00:00 2001 From: J-Dog Date: Sun, 20 Sep 2026 18:15:52 -0700 Subject: [PATCH 19/35] test(validator): split regtest init into its own describe The two regtest cases pushed the coin-wallets describe past the 60-line readability limit, which the code-structure gate refuses on push. Same assertions, own block. --- .../validator_service.test/12_coin_wallets.test.js | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/test/unit/validator_service.test/12_coin_wallets.test.js b/test/unit/validator_service.test/12_coin_wallets.test.js index 798fa6e..2a4c981 100644 --- a/test/unit/validator_service.test/12_coin_wallets.test.js +++ b/test/unit/validator_service.test/12_coin_wallets.test.js @@ -191,6 +191,16 @@ describe('ValidatorService', function () { expect(result.SEED_NODES).to.deep.equal(['01','02','03','04','05'].map(n => 'ws://validator' + n + '.xchain.io:10002')) }) + }) +}) + +// Its own block, not a fifth case inside 'coin wallets': that describe was +// already at the readability limit for a single function, and regtest init is +// a separate federation shape rather than another wallet assertion. +describe('ValidatorService', function () { + + describe('regtest initialization', function () { + it('records regtest on its local port without mainnet federation seeds', async function () { const fs = makeFs() const vs = loadValidatorService(fs) From a747a3b677a293ce0e83329965d965ea917882e5 Mon Sep 17 00:00:00 2001 From: J-Dog Date: Mon, 21 Sep 2026 11:55:48 -0700 Subject: [PATCH 20/35] docs(precheck): restore hub failure debugging context --- src/precheck.js | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/precheck.js b/src/precheck.js index fbe50d9..ae1d1d9 100644 --- a/src/precheck.js +++ b/src/precheck.js @@ -220,7 +220,7 @@ async function installHubAtRef(moduleRef) { // Preserve the cause. A bare `catch {}` here would discard the ONLY description // of what actually went wrong and replace it with a message that names no // reason, so every hub install failure would look identical and be undebuggable - // without editing this file first. + // without editing this file first. That failure mode cost two debugging cycles. // Secrets are redacted because installHubModule handles DB credentials. throw new Error("There was an error trying to install the hub module: " + redactSecrets(err), { cause: err }) } From 7c74e1aae9a7793f9568fc8f4f4bddccefb73751 Mon Sep 17 00:00:00 2001 From: J-Dog Date: Mon, 21 Sep 2026 14:05:55 -0700 Subject: [PATCH 21/35] test(hub): prove HUB_PORT_OVERRIDE moves the hub's published and container port --- .../xchain_hub_docker_run_command.test.js | 23 ++++++++++++++++++- 1 file changed, 22 insertions(+), 1 deletion(-) diff --git a/test/integration/docker_commands.test/xchain_hub_docker_run_command.test.js b/test/integration/docker_commands.test/xchain_hub_docker_run_command.test.js index d67169f..cb97d13 100644 --- a/test/integration/docker_commands.test/xchain_hub_docker_run_command.test.js +++ b/test/integration/docker_commands.test/xchain_hub_docker_run_command.test.js @@ -12,6 +12,7 @@ const { expect } = require('chai') const { dockerSuite } = require('./support/fixture') +const realConfig = require('../../../src/config') // Hub Docker command (shared service) dockerSuite('xchain-hub Docker run command', function (fixture) { @@ -23,7 +24,7 @@ dockerSuite('xchain-hub Docker run command', function (fixture) { const { ModuleService } = makeBuildAndUp() await ModuleService.buildAndUp('xchain-hub', null, null, null, true) - const buildCmd = capture.findCommands(/docker build/)[0].command + const buildCmd = capture.findCommands(/docker build /)[0].command expect(buildCmd).to.include('-t xchain-node-xchain-hub') const runCmd = capture.findCommands(/docker run/)[0].command @@ -31,4 +32,24 @@ dockerSuite('xchain-hub Docker run command', function (fixture) { expect(runCmd).to.include('--network xchain-node') expect(runCmd).to.include('-p 10000:10000') }) + + it('publishes and binds the hub on HUB_PORT_OVERRIDE for a second co-located install', async function () { + const { env, capture, makeBuildAndUp } = fixture() + env.createFakeModule('xchain-hub') + + const original = Object.getOwnPropertyDescriptor(realConfig, 'HUB_PORT_OVERRIDE') + Object.defineProperty(realConfig, 'HUB_PORT_OVERRIDE', { + value: '10500', configurable: true, enumerable: true, writable: true + }) + try { + const { ModuleService } = makeBuildAndUp() + await ModuleService.buildAndUp('xchain-hub', null, null, null, true) + } finally { + Object.defineProperty(realConfig, 'HUB_PORT_OVERRIDE', original) + } + + const runCmd = capture.findCommands(/docker run/)[0].command + expect(runCmd).to.include('-p 10500:10500') + expect(runCmd).to.not.include('10000:10000') + }) }) From 49e658ea1d108391393ae3e58fa96b7927eaeeca Mon Sep 17 00:00:00 2001 From: J-Dog Date: Mon, 21 Sep 2026 14:20:47 -0700 Subject: [PATCH 22/35] Replace service console fallbacks with logger adapters --- src/services/release_signature_service.js | 12 +- src/services/self_update_service.js | 14 +- .../stake_operations.js | 11 +- .../unstake_operations.js | 9 +- test/unit/logging/adapter_contract.test.js | 272 ++++++++++++++++++ 5 files changed, 312 insertions(+), 6 deletions(-) create mode 100644 test/unit/logging/adapter_contract.test.js diff --git a/src/services/release_signature_service.js b/src/services/release_signature_service.js index 5b397b2..4466f31 100644 --- a/src/services/release_signature_service.js +++ b/src/services/release_signature_service.js @@ -51,6 +51,7 @@ const crypto = require('crypto') const config = require('../config'); +const { getLogger } = require('../observability/logger') const { PLATFORM_KEY_FINGERPRINT, KEY_PATH, @@ -65,6 +66,15 @@ const { assertStatusIsGood } = require('./release_signature_service/gpg_verification.js') +function defaultLogger() { + const logger = getLogger() + return { + log: logger.info.bind(logger), + warn: logger.warn.bind(logger), + error: logger.error.bind(logger) + } +} + function signatureCheckDisabled() { return /^(0|false|no)$/i.test(config.XCHAIN_NODE_REQUIRE_SIGNED_RELEASE) } @@ -167,7 +177,7 @@ function assertDigestMatches({ sumsText, bytes, name }) { * @returns {Promise<{verified: boolean, fingerprint?: string, reason?: string}>} */ async function verifyManifestForTag({ - tag, manifestBytes, fetchAsset, logger = console, + tag, manifestBytes, fetchAsset, logger = defaultLogger(), keyPath = KEY_PATH, fingerprint = PLATFORM_KEY_FINGERPRINT }) { const disabled = signatureCheckDisabled() diff --git a/src/services/self_update_service.js b/src/services/self_update_service.js index 6b71ecc..818b501 100644 --- a/src/services/self_update_service.js +++ b/src/services/self_update_service.js @@ -47,6 +47,7 @@ const { promisify } = require('util') const execFileAsync = promisify(execFile) const { dataDir, SELF_UPDATE_ENV, childProcessEnv } = require('../config') +const { getLogger } = require('../observability/logger') const CARRIER_ROOT = path.join(__dirname, '../..') const TARGET_ENV = 'XCHAIN_NODE_UPDATE_TARGET' @@ -55,6 +56,15 @@ const CHECK_CACHE_FILE = 'release-check.json' const CHECK_TTL_MS = 60 * 60 * 1000 const CHECK_TIMEOUT_MS = 5000 +function defaultLogger() { + const logger = getLogger() + return { + log: logger.info.bind(logger), + warn: logger.warn.bind(logger), + error: logger.error.bind(logger) + } +} + function currentVersion() { return require('../../package.json').version } @@ -147,7 +157,7 @@ async function installCarrierDependencies(deps) { */ async function selfUpdateAndReexec({ tag, childArgs, deps = {} }) { const env = deps.env || SELF_UPDATE_ENV - const logger = deps.logger || console + const logger = deps.logger || defaultLogger() const version = deps.currentVersion ? deps.currentVersion() : currentVersion() if (env[TARGET_ENV]) return { moved: false, reason: 'already-reexecuted' } @@ -271,7 +281,7 @@ async function latestReleaseTagCached(deps = {}) { */ async function noticeNewerRelease(deps = {}) { const env = deps.env || SELF_UPDATE_ENV - const logger = deps.logger || console + const logger = deps.logger || defaultLogger() if (env[TARGET_ENV]) return null try { const version = deps.currentVersion ? deps.currentVersion() : currentVersion() diff --git a/src/services/validator_stake_service/stake_operations.js b/src/services/validator_stake_service/stake_operations.js index 5b8ead3..d71deb2 100644 --- a/src/services/validator_stake_service/stake_operations.js +++ b/src/services/validator_stake_service/stake_operations.js @@ -14,6 +14,13 @@ * XChain Node - Validator Stake Operations ********************************************************************/ +const { getLogger } = require('../../observability/logger') + +function defaultLog() { + const logger = getLogger() + return logger.info.bind(logger) +} + function logStakeBalances(log, network, coins, pubkey, address, amount, state, plan, STAKE_TICK) { log('') log('Validator stake plan (' + network + ')') @@ -153,10 +160,10 @@ function createStakeValidator(helpers) { /** * Run the stake command. `deps` lets tests inject an SDK factory and a - * logger; production uses the real SDK and console. + * logger; production uses the real SDK and service logger. */ return async function stakeValidator(opts = {}, deps = {}) { - const log = deps.log || console.log + const log = deps.log || defaultLog() const { network, coins, pubkey, sdk, session, address } = openValidatorSession(opts, deps) const amount = parseInt(opts.amount) || DEFAULT_STAKE_AMOUNT const timing = stakeTiming(coins, network) diff --git a/src/services/validator_stake_service/unstake_operations.js b/src/services/validator_stake_service/unstake_operations.js index 1044e77..24a0602 100644 --- a/src/services/validator_stake_service/unstake_operations.js +++ b/src/services/validator_stake_service/unstake_operations.js @@ -14,6 +14,13 @@ * XChain Node - Validator Unstake Operations ********************************************************************/ +const { getLogger } = require('../../observability/logger') + +function defaultLog() { + const logger = getLogger() + return logger.info.bind(logger) +} + function logUnstakePlan(log, pubkey, address, active, timing, STAKE_TICK, paren) { log('') log('Validator unstake plan') @@ -68,7 +75,7 @@ function createUnstakeValidator({ * an operator stops being that. */ return async function unstakeValidator(opts = {}, deps = {}) { - const log = deps.log || console.log + const log = deps.log || defaultLog() const { network, coins, pubkey, sdk, session, address } = openValidatorSession(opts, deps) const timing = stakeTiming(coins, network) diff --git a/test/unit/logging/adapter_contract.test.js b/test/unit/logging/adapter_contract.test.js new file mode 100644 index 0000000..4e20ee3 --- /dev/null +++ b/test/unit/logging/adapter_contract.test.js @@ -0,0 +1,272 @@ +'use strict' + +// Copyright © 2025–2026 Dankest, LLC +// Based on XChain Platform by Dankest, LLC – https://dankest.llc +// +// SPDX-License-Identifier: AGPL-3.0-or-later +// +// This file is part of XChain Platform. Licensed under the GNU Affero +// General Public License v3.0 or later; see LICENSE.md. A commercial +// license (without AGPL source-disclosure terms) is available - +// contact legal@dankest.llc. + +const { EventEmitter } = require('events') +const { expect } = require('chai') +const sinon = require('sinon') + +const LOGGER_PATH = require.resolve('../../../src/observability/logger.js') +const CONFIG_PATH = require.resolve('../../../src/config/index.js') +const RELEASE_PATH = require.resolve('../../../src/services/release_signature_service.js') +const UPDATE_PATH = require.resolve('../../../src/services/self_update_service.js') +const STAKE_PATH = require.resolve('../../../src/services/validator_stake_service/stake_operations.js') +const UNSTAKE_PATH = require.resolve('../../../src/services/validator_stake_service/unstake_operations.js') + +let loggerCacheEntry +let configCacheEntry +let cachedModules +let logger +let getLogger +let output + +function loadService(modulePath) { + if (!cachedModules.has(modulePath)) cachedModules.set(modulePath, require.cache[modulePath]) + delete require.cache[modulePath] + return require(modulePath) +} + +function expectNoGlobalOutput() { + expect(output.log.called, 'global log sink').to.equal(false) + expect(output.warn.called, 'global warn sink').to.equal(false) + expect(output.error.called, 'global error sink').to.equal(false) +} + +function operationHelpers() { + const pubkey = 'ab'.repeat(32) + const sdk = { + explorer: { + getValidators: sinon.stub().resolves({ data: [] }) + } + } + return { + openValidatorSession: () => ({ + network: 'testnet', + coins: { stakeCoin: 'TBTC' }, + pubkey, + sdk, + session: {}, + address: 'mStakeAddress' + }), + stakeTiming: () => ({ + activationBlocks: 6, + cooldownBlocks: 1000, + activationFor: 'roughly 60 minutes', + cooldownFor: 'roughly 7 days' + }), + readChainState: sinon.stub().resolves({ + coinBal: 1, + coinPending: 0, + tokenBal: 25000, + mintMax: 10000, + mintAddressMax: 50000, + existing: null, + existingUnknown: null + }), + planMints: () => ({ short: 0, mints: [], reason: null }), + explorerUrl: () => '/validator/' + pubkey, + chainedInputs: sinon.stub(), + waitForBalance: sinon.stub(), + fail: message => new Error(message), + paren: text => text ? ` (${text})` : '', + STAKE_TICK: 'XCHAIN', + DEFAULT_STAKE_AMOUNT: 25000 + } +} + +function installStubs() { + cachedModules = new Map() + logger = { + info: sinon.spy(), + warn: sinon.spy(), + error: sinon.spy() + } + getLogger = sinon.stub().returns(logger) + + loggerCacheEntry = require.cache[LOGGER_PATH] + configCacheEntry = require.cache[CONFIG_PATH] + require.cache[LOGGER_PATH] = { + id: LOGGER_PATH, + filename: LOGGER_PATH, + loaded: true, + exports: { getLogger } + } + + output = { + log: sinon.spy(console, 'log'), + warn: sinon.spy(console, 'warn'), + error: sinon.spy(console, 'error') + } +} + +function restoreStubs() { + delete process.env.XCHAIN_NODE_REQUIRE_SIGNED_RELEASE + for (const [modulePath, entry] of cachedModules) { + delete require.cache[modulePath] + if (entry) require.cache[modulePath] = entry + } + delete require.cache[LOGGER_PATH] + if (loggerCacheEntry) require.cache[LOGGER_PATH] = loggerCacheEntry + delete require.cache[CONFIG_PATH] + if (configCacheEntry) require.cache[CONFIG_PATH] = configCacheEntry + sinon.restore() +} + +function useStubs() { + beforeEach(installStubs) + afterEach(restoreStubs) +} + +describe('release signature adapter', function () { + useStubs() + + it('routes the unsigned-release fallback through getLogger().warn', async function () { + process.env.XCHAIN_NODE_REQUIRE_SIGNED_RELEASE = '0' + const { verifyManifestForTag } = loadService(RELEASE_PATH) + const callsBefore = getLogger.callCount + + const result = await verifyManifestForTag({ + tag: 'v1.2.3', + manifestBytes: Buffer.from('{}'), + fetchAsset: sinon.stub().resolves(null) + }) + + expect(result.verified).to.equal(false) + expect(getLogger.callCount).to.equal(callsBefore + 1) + expect(logger.warn.calledWithMatch(/WITHOUT release signature verification/)).to.equal(true) + expectNoGlobalOutput() + }) +}) + +describe('self update adapter', function () { + useStubs() + + it('routes self-update progress, warning and spawn failure through getLogger()', async function () { + const { selfUpdateAndReexec } = loadService(UPDATE_PATH) + const callsBefore = getLogger.callCount + const child = new EventEmitter() + const spawn = sinon.stub().callsFake(() => { + setImmediate(() => child.emit('error', new Error('spawn failed'))) + return child + }) + + await selfUpdateAndReexec({ + tag: 'v1.2.3', + childArgs: ['update', 'all', 'all', 'all', 'v1.2.3'], + deps: { + env: {}, + currentVersion: () => '1.2.2', + describeCarrier: sinon.stub().resolves({ + isRepo: true, + commit: 'c'.repeat(40), + dirty: [] + }), + execFile: sinon.stub().resolves({ stdout: '' }), + verifyGitTagSignature: sinon.stub().throws(new Error('signature unavailable')), + signatureCheckDisabled: () => true, + spawn, + exit: sinon.stub() + } + }) + + expect(getLogger.callCount).to.equal(callsBefore + 1) + expect(logger.info.calledWithMatch(/Updating the xchain-node CLI/)).to.equal(true) + expect(logger.warn.calledWithMatch(/WITHOUT verifying its tag/)).to.equal(true) + expect(logger.error.calledWithMatch(/Could not re-run xchain-node/)).to.equal(true) + expectNoGlobalOutput() + }) + + it('routes the newer-release notice through getLogger().info', async function () { + const { noticeNewerRelease } = loadService(UPDATE_PATH) + const callsBefore = getLogger.callCount + + const tag = await noticeNewerRelease({ + env: {}, + currentVersion: () => '1.2.2', + resolveLatestReleaseTag: sinon.stub().resolves('v1.2.3'), + readCheckCache: () => null, + writeCheckCache: sinon.stub() + }) + + expect(tag).to.equal('v1.2.3') + expect(getLogger.callCount).to.equal(callsBefore + 1) + expect(logger.info.calledWithMatch(/A newer XChain release is available/)).to.equal(true) + expectNoGlobalOutput() + }) +}) + +describe('injected logger seams', function () { + useStubs() + + it('preserves injected object loggers without consulting the default adapter', async function () { + process.env.XCHAIN_NODE_REQUIRE_SIGNED_RELEASE = '0' + const { verifyManifestForTag } = loadService(RELEASE_PATH) + const { noticeNewerRelease } = loadService(UPDATE_PATH) + const releaseLogger = { log: sinon.spy(), warn: sinon.spy(), error: sinon.spy() } + const updateLogger = { log: sinon.spy(), warn: sinon.spy(), error: sinon.spy() } + const callsBefore = getLogger.callCount + + await verifyManifestForTag({ + tag: 'v1.2.3', + manifestBytes: Buffer.from('{}'), + fetchAsset: sinon.stub().resolves(null), + logger: releaseLogger + }) + await noticeNewerRelease({ + env: {}, + logger: updateLogger, + currentVersion: () => '1.2.2', + resolveLatestReleaseTag: sinon.stub().resolves('v1.2.3'), + readCheckCache: () => null, + writeCheckCache: sinon.stub() + }) + + expect(releaseLogger.warn.calledOnce).to.equal(true) + expect(updateLogger.log.calledOnce).to.equal(true) + expect(getLogger.callCount).to.equal(callsBefore) + expectNoGlobalOutput() + }) + + it('preserves the stake and unstake deps.log callback seam', async function () { + const { createStakeValidator } = loadService(STAKE_PATH) + const { createUnstakeValidator } = loadService(UNSTAKE_PATH) + const helpers = operationHelpers() + const stakeLog = sinon.spy() + const unstakeLog = sinon.spy() + const callsBefore = getLogger.callCount + + await createStakeValidator(helpers)({}, { log: stakeLog }) + await createUnstakeValidator(helpers)({}, { log: unstakeLog }) + + expect(stakeLog.called).to.equal(true) + expect(unstakeLog.called).to.equal(true) + expect(getLogger.callCount).to.equal(callsBefore) + expectNoGlobalOutput() + }) +}) + +describe('stake default logger', function () { + useStubs() + + it('routes stake and unstake defaults through getLogger().info', async function () { + const { createStakeValidator } = loadService(STAKE_PATH) + const { createUnstakeValidator } = loadService(UNSTAKE_PATH) + const helpers = operationHelpers() + const callsBefore = getLogger.callCount + + await createStakeValidator(helpers)({}, {}) + await createUnstakeValidator(helpers)({}, {}) + + expect(getLogger.callCount).to.equal(callsBefore + 2) + expect(logger.info.called).to.equal(true) + expectNoGlobalOutput() + }) +}) From 5d16f9481ddd8d7ec07237ff3c1677c5608ed3fe Mon Sep 17 00:00:00 2001 From: J-Dog Date: Mon, 21 Sep 2026 20:15:34 -0700 Subject: [PATCH 23/35] docs: clarify buildx package names --- INSTALL.md | 9 ++++----- 1 file changed, 4 insertions(+), 5 deletions(-) diff --git a/INSTALL.md b/INSTALL.md index 5040097..903885e 100644 --- a/INSTALL.md +++ b/INSTALL.md @@ -38,11 +38,10 @@ sudo apt update sudo apt install docker-ce docker-ce-cli containerd.io docker-buildx-plugin docker-compose-plugin -y ``` -`docker-buildx-plugin` is required, not optional: the module images only build -under BuildKit, and `xchain-node install` refuses to build without it. Ubuntu's -own `docker.io` package ships without the plugin, so a Docker installed from the -Ubuntu archive needs `sudo apt install docker-buildx-plugin` before any module -install. +Buildx is required: the module images only build under BuildKit, and +`xchain-node install` refuses to build without it. Install `docker-buildx-plugin` +when using Docker's repository as above, or install `docker-buildx` when using +the Ubuntu archive's `docker.io` package. ### Add your user to the docker group ``` From 686123428f211b53fc6d30d56350834e538b3759 Mon Sep 17 00:00:00 2001 From: J-Dog Date: Mon, 21 Sep 2026 20:15:46 -0700 Subject: [PATCH 24/35] test(node): isolate reset confirmation dependencies --- .../15_reset_regtest_hub_purge.test.js | 14 ++++++++------ .../unit/module_operations.test/helpers/harness.js | 8 ++++++++ 2 files changed, 16 insertions(+), 6 deletions(-) diff --git a/test/unit/module_operations.test/15_reset_regtest_hub_purge.test.js b/test/unit/module_operations.test/15_reset_regtest_hub_purge.test.js index 37ec80f..fd4814f 100644 --- a/test/unit/module_operations.test/15_reset_regtest_hub_purge.test.js +++ b/test/unit/module_operations.test/15_reset_regtest_hub_purge.test.js @@ -138,28 +138,30 @@ describe('moduleOperations', function () { describe('the regtest re-genesis hub purge', function () { it('names the hub rows in the confirmation for a regtest node reset', async function () { - const readline = require('readline') const isTTYDescriptor = Object.getOwnPropertyDescriptor(process.stdin, 'isTTY') Object.defineProperty(process.stdin, 'isTTY', { value: true, configurable: true }) - const createInterface = sinon.stub(readline, 'createInterface').returns({ - question: (_q, cb) => cb('yes'), - close() {} - }) const warned = [] const warn = sinon.stub(console, 'warn').callsFake((...a) => warned.push(a.join(' '))) try { const stubs = makeStubs() + stubs.readline.createInterface.returns({ + question: (_q, cb) => cb('yes'), + close() {} + }) stubs.execFile.callsFake((cmd, args, cb) => cb(null, '', '')) const ops = loadOperations(stubs) expect(await ops.resetModules('node', 'bitcoin', 'regtest', false)).to.be.true const mainnetStubs = makeStubs() + mainnetStubs.readline.createInterface.returns({ + question: (_q, cb) => cb('yes'), + close() {} + }) mainnetStubs.execFile.callsFake((cmd, args, cb) => cb(null, '', '')) const mainnetOps = loadOperations(mainnetStubs) expect(await mainnetOps.resetModules('node', 'bitcoin', 'mainnet', false)).to.be.true } finally { warn.restore() - createInterface.restore() if (isTTYDescriptor) Object.defineProperty(process.stdin, 'isTTY', isTTYDescriptor) else delete process.stdin.isTTY } diff --git a/test/unit/module_operations.test/helpers/harness.js b/test/unit/module_operations.test/helpers/harness.js index da61f14..dbbe35e 100644 --- a/test/unit/module_operations.test/helpers/harness.js +++ b/test/unit/module_operations.test/helpers/harness.js @@ -91,6 +91,10 @@ function makeStubs() { assertHubNotBehind: sinon.stub().resolves({ checked: false, reason: 'not-hub-dependent' }), assertRequiredMigrationsApplied: sinon.stub().resolves({ checked: false, reason: 'no-migrations' }), statusChanged: sinon.stub().resolves(), + readline: { + createInterface: sinon.stub() + }, + resolveBlocksDir: sinon.stub().resolves(null), execFile: sinon.stub(), fs: { existsSync: sinon.stub().returns(false) @@ -183,12 +187,16 @@ function loadOperations(stubs, constantsOverrides = null) { '../services/status_service': { statusChanged: stubs.statusChanged }, + '../services/node_service': { + resolveBlocksDir: stubs.resolveBlocksDir + }, '../services/bootstrap_service': stubs.bootstrapService, '../services/bootstrap_republish_ledger': { reindexAffectedModules: stubs.republishLedger.reindexAffectedModules, recordReindex: stubs.republishLedger.recordReindex }, 'child_process': { execFile: stubs.execFile }, + 'readline': stubs.readline, 'fs': stubs.fs, 'util': { promisify: (fn) => async (...args) => { From 5092cc92c239e51357a9540bac7feade67c0da10 Mon Sep 17 00:00:00 2001 From: J-Dog Date: Mon, 21 Sep 2026 21:20:08 -0700 Subject: [PATCH 25/35] fix: preserve branch installs and report update refusals --- src/operations/module_operations.js | 98 ++++++++++++++++++++++++++++- 1 file changed, 95 insertions(+), 3 deletions(-) diff --git a/src/operations/module_operations.js b/src/operations/module_operations.js index 2f72364..f864bd5 100644 --- a/src/operations/module_operations.js +++ b/src/operations/module_operations.js @@ -22,6 +22,7 @@ const fs = require('fs') const readline = require('readline') const { execFile } = require('child_process') const { promisify } = require('util') +const { AsyncLocalStorage } = require('async_hooks') const execFileAsync = promisify(execFile) const { NODE_MODULE_NAME, DB_MODULE_NAME, HUB_MODULE_NAME, EXPLORER_MODULE_NAME, SYNC_MODULE_NAME, XChainService, SEP, dataDir, EXTERNAL_DB, Coin, CoinTickerSymbol, Network, DEFAULT_MODULE_BRANCH } = require('../config') const { db } = require('../state') @@ -59,6 +60,97 @@ const uninstallOperations = require('./module_operations/uninstall_modules') const moduleControls = require('./module_operations/module_controls') const resetOperations = require('./module_operations/reset_modules') +const updateAllProgress = new AsyncLocalStorage() + +function moduleKey(module, coin, network) { + return JSON.stringify([module, coin, network]) +} + +function moduleLabel(module, coin, network) { + return coin && network ? `${module} (${coin} ${network})` : module +} + +function updateAllEntries(servicesList) { + const shared = (servicesList[''] && servicesList['']['']) || [] + const orderedShared = [ + HUB_MODULE_NAME, + SYNC_MODULE_NAME, + ...shared.filter(module => module !== HUB_MODULE_NAME && module !== SYNC_MODULE_NAME) + ] + const entries = orderedShared.map(module => ({ module, coin: '', network: '' })) + for (const coin of Object.keys(servicesList)) { + if (coin === '') continue + for (const network of Object.keys(servicesList[coin])) { + for (const module of servicesList[coin][network]) { + entries.push({ module, coin, network }) + } + } + } + return entries +} + +function moveUnmoveSummary(entries, progress) { + const moved = entries.filter(entry => progress.moved.has(moduleKey(entry.module, entry.coin, entry.network))) + const unmoved = entries.filter(entry => !progress.moved.has(moduleKey(entry.module, entry.coin, entry.network))) + return '\nupdate all moved: ' + (moved.length ? moved.map(entry => moduleLabel(entry.module, entry.coin, entry.network)).join(', ') : 'none') + + '\nupdate all unmoved: ' + (unmoved.length ? unmoved.map(entry => moduleLabel(entry.module, entry.coin, entry.network)).join(', ') : 'none') +} + +function updateAllRefused(result) { + if (result == null || result === false) return true + if (result.refused === true || result.ok === false || result.success === false) return true + return !Array.isArray(result.updated) || result.updated.length === 0 +} + +function installMoved(result) { + return result === true || (typeof result === 'string' && result.length > 0) +} + +async function installModuleWithProgress(...args) { + const result = await installModule(...args) + const progress = updateAllProgress.getStore() + if (progress && args[3] === true) { + if (installMoved(result)) { + progress.moved.add(moduleKey(args[0], args[1], args[2])) + } else { + progress.refused = true + } + } + return result +} + +async function installModules(servicesList, ref = null) { + let installRef = ref + if (!installRef) { + const target = await installTargetService.resolveUpdateTarget() + if (target.kind === 'branch') installRef = target.ref + } + return sharedServices.installModules(servicesList, installRef) +} + +async function updateModules(servicesList, ref = null, opts = {}) { + if (!opts.all) return updateOperations.updateModules(servicesList, ref, opts) + + const entries = updateAllEntries(servicesList) + const progress = { moved: new Set(), refused: false } + let result + try { + result = await updateAllProgress.run(progress, () => updateOperations.updateModules(servicesList, ref, opts)) + } catch (err) { + const summary = moveUnmoveSummary(entries, progress) + if (err && typeof err === 'object' && typeof err.message === 'string') { + err.message += summary + throw err + } + throw new Error(String(err) + summary) + } + + if (entries.length > 0 && (progress.refused || updateAllRefused(result))) { + console.log(moveUnmoveSummary(entries, progress)) + } + return result +} + const dependencies = { path, fs, readline, execFileAsync, NODE_MODULE_NAME, DB_MODULE_NAME, HUB_MODULE_NAME, EXPLORER_MODULE_NAME, @@ -72,7 +164,7 @@ const dependencies = { stopModuleContainer, buildDatabaseModule, resetDatabases, clearHubPriceIngestWatermark, purgeHubCrossChainRows, manualHubCrossChainPurgeStatements, getDatabaseContainerId, - pingExternalDatabase, getModuleBranch, installModule, uninstallModule, + pingExternalDatabase, getModuleBranch, installModule: installModuleWithProgress, uninstallModule, assertHubNotBehind, assertRequiredMigrationsApplied, statusChanged, reindexAffectedModules, recordReindex, config, bootstrapService, databaseService, explorerService, hubService, installTargetService, @@ -97,9 +189,9 @@ dependencies.restartResetModules = moduleControls.restartResetModules resetOperations.configure(dependencies) module.exports = { - installModules: sharedServices.installModules, + installModules, syncSharedServicesAfterInstall: sharedServices.syncSharedServicesAfterInstall, - updateModules: updateOperations.updateModules, + updateModules, recreateModules: recreateOperations.recreateModules, uninstallModules: uninstallOperations.uninstallModules, logModules: moduleControls.logModules, From 583b5e429708514c6bcb31bad124ef2c47825f8f Mon Sep 17 00:00:00 2001 From: J-Dog Date: Tue, 22 Sep 2026 09:18:46 -0700 Subject: [PATCH 26/35] build(ci): grade the fast tier on a push, defer the heavy tiers to the sweep Generated block: do not hand-edit it. Change the tier config and re-run the tier wirer, which owns both this file and the hook so the two cannot drift. The pre-push venue gate ran the whole GitHub transcript on every push, coverage re-runs and perf scenarios included. That is the right set to grade a release with and the wrong set to pay for on every push, especially while three venues serve every repo on the platform: a long gate does not just cost its own minutes, it forms a queue behind itself for every other session pushing that hour. A push now grades the fast tier. The tiers named in the generated block move to the scheduled full sweep, which already runs against every repo every three hours and before any release or deploy, so nothing stops being graded. The verdict line is rewritten with it, which is the part that matters: a fast run can no longer print "all tiers green (same set GitHub CI runs)". It prints which tiers were deferred and states plainly that they were NOT graded there, so a fast green can never be mistaken for a full one. The dispatcher's verdict cache is keyed on repo + sha + cmd, so a fast green cannot stand in for a full one either. --- bin/ci-full.sh | 42 +++++++++++++++++++++++++++++++++++++++++- 1 file changed, 41 insertions(+), 1 deletion(-) diff --git a/bin/ci-full.sh b/bin/ci-full.sh index 74dc594..9534faa 100755 --- a/bin/ci-full.sh +++ b/bin/ci-full.sh @@ -39,7 +39,35 @@ SELF="$(pwd)" SIB="$(cd .. && pwd)" FAILED="" +# >>> ci-tier (generated block; re-run the tier wirer to update) >>> +# Tier classes. A push grades the FAST tier only: the unit job, the pin and +# drift guards, and the structure and hygiene checks the hook runs before it +# dispatches. The tiers named below (coverage re-runs, perf scenarios) are +# skipped when the gate sets CI_TIER=fast, and each skip is recorded so the +# closing verdict can never claim a green it did not earn. Nothing stops +# being graded: a scheduled sweep re-runs this same script with CI_TIER=full +# on every repo every three hours and before any release or deploy, and a +# red there is tracked down and fixed first. CI_TIER is unset for a hand +# run, so a bare `npm run ci:full` still runs every tier as it always did. +CI_TIER_FULL_ONLY=( + "coverage ratchet (coverage:check)" +) +DEFERRED="" +ci_tier_deferred() { + [ "${CI_TIER:-full}" = "fast" ] || return 1 + local t + for t in ${CI_TIER_FULL_ONLY[@]+"${CI_TIER_FULL_ONLY[@]}"}; do + if [ "$t" = "$1" ]; then + DEFERRED="$DEFERRED [$1]" + echo; echo "ci:full ===== $1 DEFERRED (CI_TIER=fast, runs in the full sweep) =====" + return 0 + fi + done + return 1 +} +# <<< ci-tier <<< run_tier() { + ci_tier_deferred "$1" && return 0 # ci-tier guard (generated) local name="$1"; shift echo; echo "ci:full ===== $name =====" if "$@"; then @@ -110,8 +138,20 @@ run_tier "identity pin (vendored coin bytes)" node bin/pin_identity.js --compare run_tier "coverage ratchet (coverage:check)" npm run coverage:check echo +# >>> ci-tier summary (generated) >>> +echo "ci:full: tier class ${CI_TIER:-full}" +if [ -n "${DEFERRED:-}" ]; then + echo "ci:full: DEFERRED to the full sweep:$DEFERRED" +fi +# <<< ci-tier summary <<< if [ -n "$FAILED" ]; then echo "ci:full: RED tiers:$FAILED" exit 1 fi -echo "ci:full: all tiers green (same set GitHub CI runs)" +# >>> ci-tier verdict (generated) >>> +if [ "${CI_TIER:-full}" = "fast" ]; then + echo "ci:full: all FAST tiers green; the DEFERRED tiers above were NOT graded here" +else + echo "ci:full: all tiers green (same set GitHub CI runs)" +fi +# <<< ci-tier verdict <<< From 96e62026d320d7d86cb75bfd24d90e8dc2b2dc58 Mon Sep 17 00:00:00 2001 From: J-Dog Date: Tue, 22 Sep 2026 13:50:31 -0700 Subject: [PATCH 27/35] test(node): add sql_safety unit coverage --- test/unit/utils/sql_safety.test.js | 34 ++++++++++++++++++++++++++++++ 1 file changed, 34 insertions(+) create mode 100644 test/unit/utils/sql_safety.test.js diff --git a/test/unit/utils/sql_safety.test.js b/test/unit/utils/sql_safety.test.js new file mode 100644 index 0000000..8e5b8de --- /dev/null +++ b/test/unit/utils/sql_safety.test.js @@ -0,0 +1,34 @@ +'use strict' + +// Copyright © 2025-2026 Dankest, LLC +// Based on XChain Platform by Dankest, LLC - https://dankest.llc +// +// SPDX-License-Identifier: AGPL-3.0-or-later + +const { expect } = require('chai') + +const { assertSafeDbIdentifier, escapeSqlStringLiteral } = require('../../../src/utils/sql_safety') + +describe('sql_safety', () => { + describe('assertSafeDbIdentifier', () => { + it('returns a safe identifier unchanged', () => { + expect(assertSafeDbIdentifier('xchain_btc_mainnet')).to.equal('xchain_btc_mainnet') + }) + + it('throws naming the kind for an unsafe identifier', () => { + expect(() => assertSafeDbIdentifier("evil'; DROP", 'database name')) + .to.throw(/Unsafe MariaDB database name/) + }) + }) + + describe('escapeSqlStringLiteral', () => { + it('backslash-escapes a quote and returns one quoted literal', () => { + expect(escapeSqlStringLiteral("x'; DROP DATABASE d; --")) + .to.equal('\'x\\\'; DROP DATABASE d; --\'') + }) + + it('escapes the NUL, the newline and a backslash per the switch table', () => { + expect(escapeSqlStringLiteral('a\0b\nc')).to.equal('\'a\\0b\\nc\'') + }) + }) +}) From db8815bf2eb49264616b30e3632a6cb62b5925d7 Mon Sep 17 00:00:00 2001 From: J-Dog Date: Tue, 22 Sep 2026 19:24:29 -0700 Subject: [PATCH 28/35] fix(node): report a non-zero drain exit instead of calling it a clean stop stopContainerByName now carries the container's exit code out, and a non-zero exit other than a kill warns and names the service's own drain timer, so an overrun drain that exits 1 no longer reads as clean. The killed-branch text explains that raising the per-service stop override alone does not lengthen the drain, and operations.md documents the pairing. --- src/services/docker_service.js | 11 ++++-- .../node_service/crypto_node_build.js | 4 +- src/services/node_service/node_stop.js | 14 ++++++- src/services/stop_budget_service.js | 30 ++++++++++++-- .../02_stop_container.test.js | 39 ++++++++++++------- .../blocks_dir_persistence.test.js | 11 ++++++ test/unit/stop_budget_service.test.js | 20 ++++++++++ 7 files changed, 104 insertions(+), 25 deletions(-) diff --git a/src/services/docker_service.js b/src/services/docker_service.js index 5c52b00..9d688a8 100644 --- a/src/services/docker_service.js +++ b/src/services/docker_service.js @@ -97,7 +97,7 @@ async function stopContainer(containerId) { // Graceful stop by NAME with an explicit shutdown budget, for stateful // containers (chain daemons) that must flush before they go. SIGTERM first; // docker escalates to SIGKILL only after `timeoutSeconds`. Resolves -// { stopped, seconds, killed }: `stopped` false when docker did not report +// { stopped, seconds, killed, exitCode }: `stopped` false when docker did not report // the stop (already gone, never existed, or daemon unreachable), which the // caller's subsequent force-remove/run surfaces, so a missing container is // not a failure here. `killed` is the part the caller must not stay silent @@ -105,7 +105,10 @@ async function stopContainer(containerId) { // killed at the budget, and a killed chain daemon comes back at its last // flushed state and re-validates for hours. It is read from the container's // exit code after the stop (SIGKILL reports 137), with the elapsed time as a -// second witness for a docker that does not answer the inspect. +// second witness for a docker that does not answer the inspect. `exitCode` is +// that code as read (null when unreadable): a service that ends its own +// overrun drain exits non-zero inside the budget, which is not a kill but is +// not a clean stop either. async function stopContainerByName(name, timeoutSeconds) { const startedAt = Date.now() const stopped = await new Promise((resolve) => { @@ -114,7 +117,7 @@ async function stopContainerByName(name, timeoutSeconds) { }) }) const seconds = Math.round((Date.now() - startedAt) / 1000) - if (!stopped) return { stopped: false, seconds, killed: false } + if (!stopped) return { stopped: false, seconds, killed: false, exitCode: null } const exitCode = await new Promise((resolve) => { execFile('docker', ['inspect', '--format', '{{.State.ExitCode}}', name], (error, stdout) => { const code = parseInt(String(stdout || '').trim(), 10) @@ -122,7 +125,7 @@ async function stopContainerByName(name, timeoutSeconds) { }) }) const killed = exitCode === 137 || (exitCode === null && seconds >= timeoutSeconds) - return { stopped: true, seconds, killed } + return { stopped: true, seconds, killed, exitCode } } async function startContainer(containerId) { diff --git a/src/services/node_service/crypto_node_build.js b/src/services/node_service/crypto_node_build.js index 68ce7a2..90bf844 100644 --- a/src/services/node_service/crypto_node_build.js +++ b/src/services/node_service/crypto_node_build.js @@ -39,7 +39,7 @@ const configService = require('../config_service') const peers = require('../peer_services').bindPeerServices((file) => require(path.join('..', file))) const { getLogger } = require('../../observability/logger'); const logger = getLogger(); -const { nodeStopTimeoutSeconds, describeNodeStopOutcome } = require('./node_stop.js') +const { nodeStopTimeoutSeconds, describeNodeStopOutcome, nodeStoppedUnclean } = require('./node_stop.js') // Whether the coin's pinned daemon honors `-blocksdir`. Dogecoin Core (v1.14.x) // is based on a pre-0.18 Bitcoin Core and silently ignores the flag (added @@ -231,7 +231,7 @@ async function prepareExistingContainer(containerPrefix, coin, network, storage, const stopOutcome = await stopContainerByName(containerPrefix, stopBudgetSeconds) const stopLine = describeNodeStopOutcome(coin, network, stopOutcome, stopBudgetSeconds) if (stopLine) { - if (stopOutcome.killed) logger.warn(stopLine) + if (stopOutcome.killed || nodeStoppedUnclean(stopOutcome)) logger.warn(stopLine) else logger.info(stopLine) } diff --git a/src/services/node_service/node_stop.js b/src/services/node_service/node_stop.js index a2eae28..c5a4775 100644 --- a/src/services/node_service/node_stop.js +++ b/src/services/node_service/node_stop.js @@ -56,12 +56,24 @@ function describeNodeStopOutcome(coin, network, outcome, budgetSeconds) { `It will come back at its last flushed state and re-validate from there, which can take hours on a large dbcache. ` + `Raise ${NODE_STOP_TIMEOUT_ENV} above the time this daemon needs to flush before the next update.` } + if (nodeStoppedUnclean(outcome)) { + return `WARNING: the ${coin} ${network} daemon exited with code ${outcome.exitCode} after ${outcome.seconds} s ` + + `(budget ${budgetSeconds} s), not cleanly. Check its debug.log before assuming the chainstate was flushed.` + } return `Stopped the ${coin} ${network} daemon cleanly in ${outcome.seconds} s (budget ${budgetSeconds} s).` } +// A daemon that left inside the budget with a non-zero code exited on an +// error, not a clean flush. +function nodeStoppedUnclean(outcome) { + return Boolean(outcome && outcome.stopped && !outcome.killed && + Number.isInteger(outcome.exitCode) && outcome.exitCode !== 0) +} + module.exports = { DEFAULT_NODE_STOP_TIMEOUT_SECONDS, NODE_STOP_TIMEOUT_ENV, nodeStopTimeoutSeconds, - describeNodeStopOutcome + describeNodeStopOutcome, + nodeStoppedUnclean } diff --git a/src/services/stop_budget_service.js b/src/services/stop_budget_service.js index 8b9714a..3ebba26 100644 --- a/src/services/stop_budget_service.js +++ b/src/services/stop_budget_service.js @@ -71,16 +71,37 @@ function stopTimeoutArgs(module, env = MODULE_STOP_TIMEOUT_ENV) { return ['--stop-timeout', String(moduleStopTimeoutSeconds(module, env))] } +// SIGTERM's default action (128 + 15): a process with no drain registered, +// or npm relaying its child's, which is how a drainless service always stops. +const EXIT_ON_SIGTERM_DEFAULT = 143 + +// A stop inside the budget that still ended non-zero. The drains exit 1 when +// their own hard-exit timer fires or the drain throws, so work was cut off +// even though docker never had to kill anything. +function stoppedUnclean(outcome) { + return Boolean(outcome && outcome.stopped && !outcome.killed && + Number.isInteger(outcome.exitCode) && outcome.exitCode !== 0 && + outcome.exitCode !== EXIT_ON_SIGTERM_DEFAULT) +} + // What the operator reads after a service was stopped. A stop that ran out of // budget is a kill, and a killed decoder may have been mid-rollback; that is -// worth a warning line, not silence. Returns null when there was nothing to -// stop (already gone), because the caller's remove or run surfaces that. +// worth a warning line, not silence, and so is a drain the service cut off +// itself. Returns null when there was nothing to stop (already gone), because +// the caller's remove or run surfaces that. function describeModuleStopOutcome(module, coin, network, outcome, budgetSeconds) { if (!outcome || !outcome.stopped) return null const where = coin && network ? ` (${coin} ${network})` : '' if (outcome.killed) { return `WARNING: ${module}${where} did not exit within the ${budgetSeconds} s budget and was killed. ` + - `Raise ${moduleStopTimeoutEnvName(module)} if this service needs longer to finish its block.` + `Raise ${moduleStopTimeoutEnvName(module)} if this service needs longer to finish its block, ` + + 'and keep the service\'s own SHUTDOWN_TIMEOUT_MS below it where the service reads one.' + } + if (stoppedUnclean(outcome)) { + return `WARNING: ${module}${where} exited with code ${outcome.exitCode} after ${outcome.seconds} s, inside the ` + + `${budgetSeconds} s budget, so its shutdown drain did not complete: it overran the service's own ` + + 'hard-exit timer (SHUTDOWN_TIMEOUT_MS where the service reads one) or failed. Check the service log. ' + + `Raising ${moduleStopTimeoutEnvName(module)} alone does not give that drain more time.` } return `Stopped ${module}${where} cleanly in ${outcome.seconds} s (budget ${budgetSeconds} s).` } @@ -94,7 +115,7 @@ async function stopModuleContainer(stopContainerByName, module, coin, network, c const outcome = await stopContainerByName(containerRef, budget) const line = describeModuleStopOutcome(module, coin, network, outcome, budget) if (line) { - if (outcome.killed) logger.warn(line) + if (outcome.killed || stoppedUnclean(outcome)) logger.warn(line) else logger.info(line) } return { ...outcome, budget } @@ -106,6 +127,7 @@ module.exports = { moduleStopTimeoutEnvName, moduleStopTimeoutSeconds, stopTimeoutArgs, + stoppedUnclean, describeModuleStopOutcome, stopModuleContainer } diff --git a/test/unit/docker_service.test/02_stop_container.test.js b/test/unit/docker_service.test/02_stop_container.test.js index dc52bd2..7e3ac52 100644 --- a/test/unit/docker_service.test/02_stop_container.test.js +++ b/test/unit/docker_service.test/02_stop_container.test.js @@ -82,24 +82,24 @@ describe('DockerService', function () { }) +// `docker stop` exits 0 whether the daemon left on SIGTERM or was killed +// at the budget; the container's exit code is what tells them apart. +function stopThenInspect(stubs, { stopErr = null, stopOut = 'xchain-node-bitcoin-mainnet-node\n', exitCode = '0', inspectErr = null } = {}) { + const calls = [] + stubs.execFile.callsFake((cmd, args, ...rest) => { + const cb = typeof rest[0] === 'function' ? rest[0] : rest[1] + calls.push(args) + if (args[0] === 'stop') return cb(stopErr, stopOut) + if (args[0] === 'inspect') return cb(inspectErr, exitCode + '\n') + cb(new Error('unexpected ' + args.join(' '))) + }) + return calls +} + describe('DockerService', function () { describe('stopContainerByName()', function () { - // `docker stop` exits 0 whether the daemon left on SIGTERM or was killed - // at the budget; the container's exit code is what tells them apart. - function stopThenInspect(stubs, { stopErr = null, stopOut = 'xchain-node-bitcoin-mainnet-node\n', exitCode = '0', inspectErr = null } = {}) { - const calls = [] - stubs.execFile.callsFake((cmd, args, ...rest) => { - const cb = typeof rest[0] === 'function' ? rest[0] : rest[1] - calls.push(args) - if (args[0] === 'stop') return cb(stopErr, stopOut) - if (args[0] === 'inspect') return cb(inspectErr, exitCode + '\n') - cb(new Error('unexpected ' + args.join(' '))) - }) - return calls - } - it('runs docker stop -t and reads the exit code after it', async function () { const stubs = makeStubs() const calls = stopThenInspect(stubs) @@ -120,6 +120,17 @@ describe('DockerService', function () { expect(outcome).to.include({ stopped: true, killed: true }) }) + it('carries a non-zero self exit out as exitCode without calling it a kill', async function () { + const stubs = makeStubs() + stopThenInspect(stubs, { exitCode: '1' }) + const ds = loadDockerService(stubs) + const outcome = await ds.stopContainerByName('xchain-node-bitcoin-mainnet-node', 600) + expect(outcome).to.include({ stopped: true, killed: false, exitCode: 1 }) + stopThenInspect(stubs, { inspectErr: new Error('daemon unreachable') }) + const unread = await ds.stopContainerByName('xchain-node-bitcoin-mainnet-node', 600) + expect(unread).to.include({ stopped: true, killed: false, exitCode: null }) + }) + it('reports not stopped, and does not inspect, when docker did not confirm the stop', async function () { const stubs = makeStubs() const calls = stopThenInspect(stubs, { stopErr: new Error('No such container') }) diff --git a/test/unit/node_service.test/blocks_dir_persistence.test.js b/test/unit/node_service.test/blocks_dir_persistence.test.js index 0d041fe..3ef8c7d 100644 --- a/test/unit/node_service.test/blocks_dir_persistence.test.js +++ b/test/unit/node_service.test/blocks_dir_persistence.test.js @@ -194,6 +194,17 @@ describe("NodeService: buildCryptoNode()", function () { expect(warning).to.match(/XCHAIN_NODE_STOP_TIMEOUT_SECONDS/) }) + it('warns when the daemon exited non-zero inside the budget instead of calling it clean', async function () { + const stubs = makeNodeServiceStubs() + stubs.stopContainerByName = sinon.stub().resolves({ stopped: true, seconds: 30, killed: false, exitCode: 1 }) + await build(stubs, { envBlocksDir: null }) + const warning = warnStub.args.map(a => String(a[0])).find(l => /exited with code 1/.test(l)) + expect(warning).to.match(/the bitcoin mainnet daemon exited with code 1 after 30 s/) + expect(warning).to.match(/debug\.log/) + expect(warning).to.not.match(/SHUTDOWN_TIMEOUT_MS/) + expect(logStub.args.some(a => /daemon cleanly/.test(String(a[0])))).to.be.false + }) + it('says nothing about the stop when there was no previous daemon', async function () { const stubs = makeNodeServiceStubs() stubs.stopContainerByName = sinon.stub().resolves({ stopped: false, seconds: 0, killed: false }) diff --git a/test/unit/stop_budget_service.test.js b/test/unit/stop_budget_service.test.js index 7b65462..c40c506 100644 --- a/test/unit/stop_budget_service.test.js +++ b/test/unit/stop_budget_service.test.js @@ -102,6 +102,26 @@ describe('StopBudgetService', function () { expect(warning).to.match(/XCHAIN_NODE_MODULE_STOP_TIMEOUT_SECONDS_XCHAIN_UTXO_TRACKER/) }) + it('warns when the service exited non-zero inside the budget, naming its own drain timer', async function () { + const stop = sinon.stub().resolves({ stopped: true, seconds: 100, killed: false, exitCode: 1 }) + const outcome = await sbs.stopModuleContainer(stop, 'xchain-decoder', 'bitcoin', 'mainnet', 'abc123', {}) + expect(outcome.killed).to.be.false + const warning = warnStub.args.map(a => String(a[0])).find(l => /exited with code 1/.test(l)) + expect(warning).to.match(/xchain-decoder \(bitcoin mainnet\) exited with code 1 after 100 s, inside the 120 s budget/) + expect(warning).to.match(/SHUTDOWN_TIMEOUT_MS/) + expect(warning).to.match(/XCHAIN_NODE_MODULE_STOP_TIMEOUT_SECONDS_XCHAIN_DECODER alone does not/) + expect(logStub.args.some(a => /cleanly/.test(String(a[0])))).to.be.false + }) + + it('still reports a clean stop for exit 0 and for a drainless service ended by SIGTERM (143)', async function () { + for (const exitCode of [0, 143, null]) { + const stop = sinon.stub().resolves({ stopped: true, seconds: 3, killed: false, exitCode }) + await sbs.stopModuleContainer(stop, 'xchain-sdk', 'bitcoin', 'mainnet', 'abc123', {}) + } + expect(logStub.args.filter(a => /Stopped xchain-sdk \(bitcoin mainnet\) cleanly in 3 s/.test(String(a[0])))).to.have.length(3) + expect(warnStub.called).to.be.false + }) + it('says nothing when there was nothing to stop', async function () { const stop = sinon.stub().resolves({ stopped: false, seconds: 0, killed: false }) await sbs.stopModuleContainer(stop, 'xchain-encoder', 'bitcoin', 'mainnet', 'gone', {}) From 5825221e5428c00c9c176d8da021e59d086fdc49 Mon Sep 17 00:00:00 2001 From: J-Dog Date: Tue, 22 Sep 2026 20:11:14 -0700 Subject: [PATCH 29/35] fix(node): agree on the -p port grammar, drop an unreachable return and an unused registry read --- .../database_service/database_module.js | 1 - .../database_service/user_provisioning.js | 4 +- src/services/module_service/build_and_up.js | 14 ++-- src/services/module_service/docker_args.js | 27 +++++--- .../11_port_spec_grammar.test.js | 65 +++++++++++++++++++ 5 files changed, 91 insertions(+), 20 deletions(-) create mode 100644 test/unit/module_service.test/11_port_spec_grammar.test.js diff --git a/src/services/database_service/database_module.js b/src/services/database_service/database_module.js index 765d5cf..d932659 100644 --- a/src/services/database_service/database_module.js +++ b/src/services/database_service/database_module.js @@ -61,7 +61,6 @@ async function verifyExternalDatabase() { throw new Error("Cannot reach external MariaDB at " + cfg.host + ":" + cfg.port + ": " + (err.message || err)) } return true - return true } // Cap json-file log growth so a long-running node cannot fill the host diff --git a/src/services/database_service/user_provisioning.js b/src/services/database_service/user_provisioning.js index e43f11e..e027cfb 100644 --- a/src/services/database_service/user_provisioning.js +++ b/src/services/database_service/user_provisioning.js @@ -18,7 +18,6 @@ let { execFile } = require('child_process') let { HUB_MODULE_NAME, XChainService, EXTERNAL_DB } = require('../../config') -let { db } = require('../../state') let { redactSecrets } = require('../../utils/helpers') let { assertSafeDbIdentifier, escapeSqlStringLiteral } = require('../../utils/sql_safety') let { schemaExistsSql } = require('../../db/information_schema') @@ -31,7 +30,7 @@ let { executeNativeMariaDbCommand, askMariadbRootPassword, executeDockerMariaDbC const nativeExecFile = execFile function configureDependencies(dependencies) { if (dependencies.execFile === nativeExecFile) return - ;({ execFile, HUB_MODULE_NAME, XChainService, EXTERNAL_DB, db, redactSecrets, assertSafeDbIdentifier, escapeSqlStringLiteral, schemaExistsSql, getLogger, logger } = dependencies) + ;({ execFile, HUB_MODULE_NAME, XChainService, EXTERNAL_DB, redactSecrets, assertSafeDbIdentifier, escapeSqlStringLiteral, schemaExistsSql, getLogger, logger } = dependencies) } // Fail fast when the DB container is missing or not ready, instead of @@ -204,7 +203,6 @@ async function addUserPasswordToDatabase(module, coin, network, databaseName, us assertSafeDbIdentifier(databaseName, 'database name') assertSafeDbIdentifier(user, 'database user') const mariadbRootPassword = await askMariadbRootPassword(coin, network) - const moduleContainerId = await db.getModuleContainer(module, coin, network) // Host is '%' so cross-network shared services (xchain-explorer, xchain-hub) // can authenticate against per-coin indexer/decoder DBs. Earlier code derived diff --git a/src/services/module_service/build_and_up.js b/src/services/module_service/build_and_up.js index 63b45ef..9560759 100644 --- a/src/services/module_service/build_and_up.js +++ b/src/services/module_service/build_and_up.js @@ -36,6 +36,7 @@ let rollcallWiring = require('../rollcall_wiring') let { readCheckoutIdentityFromDisk } = require('./git_checkout') let { cloneGit, resolveBundledLibRef } = require('./clone_and_refs') let { assertNoHostPortConflicts, resolveObservabilityEnv, buildHealthcheckArgs, buildModuleDockerArgs } = require('./docker_args') +const { parsePortSpec } = require('./docker_args') let { attachCrossChainNetworks, verifyContainerMemoryLimit, logDockerCreateWarnings } = require('./container_networks') const { getLogger } = require('../../observability/logger'); let logger = getLogger(); @@ -177,16 +178,15 @@ async function resolveMemoryOptions(module, coin, network, onlyExecution) { memoryLimitMb: memory.args.length > 0 ? memory.mb : null } } -// Validate all port values. +// Validate all port values, parsed with the conflict check's grammar; refuse a +// host-interface (IP-scoped) spec, which no configured module port may carry. function validatePortArgs(portArgs) { for (let i = 0; i < portArgs.length; i++) { if (portArgs[i] !== '-p') continue const pair = portArgs[i + 1] - const colonIdx = pair.indexOf(':') - if (colonIdx === -1) continue - const hostPort = pair.substring(0, colonIdx) - const containerPort = pair.substring(colonIdx + 1) - if (!validatePort(hostPort) || !validatePort(containerPort)) { + if (typeof pair === 'string' && !pair.includes(':')) continue + const spec = parsePortSpec(pair) + if (!spec || spec.ip !== '' || !validatePort(spec.hostPort) || !validatePort(spec.containerPort)) { throw "Invalid port value in configuration: " + pair } } @@ -397,4 +397,4 @@ async function buildAndUp(module, coin, network, overwriteContainerId = null, on }) } -module.exports = { configureDependencies, buildAndUp } +module.exports = { configureDependencies, buildAndUp, validatePortArgs } diff --git a/src/services/module_service/docker_args.js b/src/services/module_service/docker_args.js index 5788f5d..b2c30e9 100644 --- a/src/services/module_service/docker_args.js +++ b/src/services/module_service/docker_args.js @@ -34,6 +34,21 @@ function configureDependencies(dependencies) { } = dependencies) } +// Split a `-p` value on its last colon into [IP:]HOST:CONTAINER fields (ip is +// '' when absent); null for a non-string or colon-less value. +function parsePortSpec(pair) { + if (typeof pair !== 'string') return null + const colonIdx = pair.lastIndexOf(':') + if (colonIdx === -1) return null + const beforeContainer = pair.substring(0, colonIdx) + const hostIdx = beforeContainer.lastIndexOf(':') + return { + ip: hostIdx === -1 ? '' : beforeContainer.substring(0, hostIdx), + hostPort: beforeContainer.substring(hostIdx + 1), + containerPort: pair.substring(colonIdx + 1) + } +} + // Fail fast on host-port collisions before `docker run`. On a single-stack // host this is a no-op; on a multi-stack host (two NODE_PREFIX stacks, or a // service container hand-created outside xchain-node) two containers can request @@ -46,15 +61,8 @@ async function assertNoHostPortConflicts(portArgs, selfName) { const requested = [] for (let i = 0; i < portArgs.length; i++) { if (portArgs[i] === '-p') { - const pair = portArgs[i + 1] - if (typeof pair !== 'string') continue - const colonIdx = pair.lastIndexOf(':') - if (colonIdx === -1) continue - // "-p HOST:CONTAINER" or "-p IP:HOST:CONTAINER": the host port is the - // field before the final colon; take the last colon-separated pair's left side. - const beforeContainer = pair.substring(0, colonIdx) - const hostPort = beforeContainer.substring(beforeContainer.lastIndexOf(':') + 1) - if (/^\d+$/.test(hostPort)) requested.push(hostPort) + const spec = parsePortSpec(portArgs[i + 1]) + if (spec && /^\d+$/.test(spec.hostPort)) requested.push(spec.hostPort) } } if (requested.length === 0) return @@ -381,5 +389,6 @@ module.exports = { resolveObservabilityEnv, buildHealthcheckArgs, buildModuleDockerArgs, + parsePortSpec, assertNoHostPortConflicts } diff --git a/test/unit/module_service.test/11_port_spec_grammar.test.js b/test/unit/module_service.test/11_port_spec_grammar.test.js new file mode 100644 index 0000000..2803b1b --- /dev/null +++ b/test/unit/module_service.test/11_port_spec_grammar.test.js @@ -0,0 +1,65 @@ +'use strict' + +// Copyright © 2025–2026 Dankest, LLC +// Based on XChain Platform by Dankest, LLC – https://dankest.llc +// +// SPDX-License-Identifier: AGPL-3.0-or-later +// +// This file is part of XChain Platform. Licensed under the GNU Affero +// General Public License v3.0 or later; see LICENSE.md. A commercial +// license (without AGPL source-disclosure terms) is available - +// contact legal@dankest.llc. + +// Pins the one `-p` grammar the port validator and the host-port conflict +// check share: split on the last colon, host port is the field before it. + +const { expect, proxyquire, moduleSuite } = require('./support/helpers') +const { validatePort } = require('../../../src/services/config_service/validation') +const { parsePortSpec } = require('../../../src/services/module_service/docker_args') +// Load a fresh copy so a sibling suite's configureDependencies cannot swap validatePort. +const { validatePortArgs } = proxyquire('../../../src/services/module_service/build_and_up', { + '../config_service': { validatePort } +}) + +moduleSuite('parsePortSpec()', function () { + it('splits HOST:CONTAINER with an empty ip', function () { + expect(parsePortSpec('8080:80')).to.deep.equal({ ip: '', hostPort: '8080', containerPort: '80' }) + }) + + it('splits IP:HOST:CONTAINER on the last colon', function () { + expect(parsePortSpec('127.0.0.1:13306:3306')).to.deep.equal({ ip: '127.0.0.1', hostPort: '13306', containerPort: '3306' }) + }) + + it('keeps a bracketed IPv6 host address whole', function () { + expect(parsePortSpec('[::1]:8080:80')).to.deep.equal({ ip: '[::1]', hostPort: '8080', containerPort: '80' }) + }) + + it('returns null for a colon-less value or a non-string', function () { + expect(parsePortSpec('8080')).to.equal(null) + expect(parsePortSpec(undefined)).to.equal(null) + expect(parsePortSpec(8080)).to.equal(null) + }) +}) + +moduleSuite('validatePortArgs()', function () { + const invalid = (pair) => () => validatePortArgs(['-p', pair]) + + it('accepts in-range HOST:CONTAINER pairs and skips colon-less values', function () { + expect(() => validatePortArgs(['-p', '8080:80', '-v', 'a:b', '-p', '9000'])).not.to.throw() + }) + + it('throws on an out-of-range or non-numeric port', function () { + expect(invalid('99999:80')).to.throw('Invalid port value in configuration: 99999:80') + expect(invalid('8080:not_a_port')).to.throw('Invalid port value') + expect(invalid('8080:')).to.throw('Invalid port value') + }) + + it('refuses an IP-scoped spec even when both ports are valid', function () { + expect(invalid('127.0.0.1:13306:3306')).to.throw('Invalid port value in configuration: 127.0.0.1:13306:3306') + expect(invalid('3003:3003:3003')).to.throw('Invalid port value') + }) + + it('throws on a dangling -p with no value', function () { + expect(() => validatePortArgs(['-p'])).to.throw('Invalid port value') + }) +}) From 806bb4ed060cbb2304a17c0c1e97c3bd2e19c113 Mon Sep 17 00:00:00 2001 From: J-Dog Date: Tue, 22 Sep 2026 20:49:12 -0700 Subject: [PATCH 30/35] fix(node): derive each service's SHUTDOWN_TIMEOUT_MS from the node stop budget The node forwards SHUTDOWN_TIMEOUT_MS as its stop budget minus 20 s unless the module config sets it, so the drain always fits inside the stop; defaults keep 100000, and a container not yet recreated gets a warning. --- src/operations/module_operations.js | 4 +- .../module_operations/module_controls.js | 7 +- src/services/docker_service.js | 28 +++++ src/services/module_service/build_and_up.js | 12 +- src/services/stop_budget_service.js | 85 +++++++++++++- .../get_container_stop_settings.test.js | 55 +++++++++ .../03_module_controls.test.js | 14 +++ .../module_operations.test/helpers/harness.js | 2 + ...12_buildandup_shutdown_timeout_env.test.js | 76 +++++++++++++ test/unit/stop_budget_service.test.js | 104 +++++++++++++++++- 10 files changed, 370 insertions(+), 17 deletions(-) create mode 100644 test/unit/docker_service.test/stop_settings/get_container_stop_settings.test.js create mode 100644 test/unit/module_service.test/12_buildandup_shutdown_timeout_env.test.js diff --git a/src/operations/module_operations.js b/src/operations/module_operations.js index f864bd5..266dd20 100644 --- a/src/operations/module_operations.js +++ b/src/operations/module_operations.js @@ -28,7 +28,7 @@ const { NODE_MODULE_NAME, DB_MODULE_NAME, HUB_MODULE_NAME, EXPLORER_MODULE_NAME, const { db } = require('../state') const { sleep } = require('../utils/helpers') const { getDockerContainerImageName, getUtxoTrackerVolumeName, getDockerNetwork } = require('../services/config_service') -const { createDockerNetwork, probeContainerPresenceByName, stopContainer, stopContainerByName, startContainer, restartContainer, execContainer, shellContainer, logContainer, startDockerMonitor, waitContainer, saveContainerLogs, getContainerBindMounts, removeContainer } = require('../services/docker_service') +const { createDockerNetwork, probeContainerPresenceByName, stopContainer, stopContainerByName, startContainer, restartContainer, execContainer, shellContainer, logContainer, startDockerMonitor, waitContainer, saveContainerLogs, getContainerBindMounts, getContainerStopSettings, removeContainer } = require('../services/docker_service') const { stopModuleContainer } = require('../services/stop_budget_service') const { buildDatabaseModule, resetDatabases, clearHubPriceIngestWatermark, purgeHubCrossChainRows, manualHubCrossChainPurgeStatements, getDatabaseContainerId, pingExternalDatabase } = require('../services/database_service') const { getModuleBranch, installModule, uninstallModule } = require('../services/module_service') @@ -160,7 +160,7 @@ const dependencies = { createDockerNetwork, probeContainerPresenceByName, stopContainer, stopContainerByName, startContainer, restartContainer, execContainer, shellContainer, logContainer, startDockerMonitor, waitContainer, - saveContainerLogs, getContainerBindMounts, removeContainer, + saveContainerLogs, getContainerBindMounts, getContainerStopSettings, removeContainer, stopModuleContainer, buildDatabaseModule, resetDatabases, clearHubPriceIngestWatermark, purgeHubCrossChainRows, manualHubCrossChainPurgeStatements, getDatabaseContainerId, diff --git a/src/operations/module_operations/module_controls.js b/src/operations/module_operations/module_controls.js index abda88a..f1c9fe2 100644 --- a/src/operations/module_operations/module_controls.js +++ b/src/operations/module_operations/module_controls.js @@ -1,9 +1,9 @@ 'use strict' -let XChainService, failureReason, sleep, db, execContainer, getDockerContainerImageName, logContainer, restartContainer, shellContainer, startContainer, startDockerMonitor, statusChanged, stopContainerByName, stopModuleContainer +let XChainService, failureReason, sleep, db, execContainer, getDockerContainerImageName, logContainer, restartContainer, shellContainer, startContainer, startDockerMonitor, statusChanged, stopContainerByName, stopModuleContainer, getContainerStopSettings function configure(dependencies) { - ({ XChainService, failureReason, sleep, db, execContainer, getDockerContainerImageName, logContainer, restartContainer, shellContainer, startContainer, startDockerMonitor, statusChanged, stopContainerByName, stopModuleContainer } = dependencies) + ({ XChainService, failureReason, sleep, db, execContainer, getDockerContainerImageName, logContainer, restartContainer, shellContainer, startContainer, startDockerMonitor, statusChanged, stopContainerByName, stopModuleContainer, getContainerStopSettings } = dependencies) } async function logModules(servicesList, follow = true) { @@ -99,7 +99,8 @@ async function stopModules(servicesList) { // With the service's budget, not docker's ten seconds: a bare // `docker stop` on a container created before the budget was // stamped on it is a coin flip for a service mid-block. - await stopModuleContainer(stopContainerByName, nextModule, nextCoin, nextNetwork, containerId) + await stopModuleContainer(stopContainerByName, nextModule, nextCoin, nextNetwork, containerId, + undefined, getContainerStopSettings) } catch (err) { console.log(err) } diff --git a/src/services/docker_service.js b/src/services/docker_service.js index 9d688a8..0a792ef 100644 --- a/src/services/docker_service.js +++ b/src/services/docker_service.js @@ -248,6 +248,33 @@ async function getContainerBindMounts(name) { }) } +// A container's stamped stop budget and the SHUTDOWN_TIMEOUT_MS it was created +// with, as { stopTimeout, shutdownTimeoutMs } (each null when absent). The +// rest of its env carries secrets and never leaves this function. Resolves +// null when docker cannot answer, so a caller treats that as "unknown". +async function getContainerStopSettings(name) { + return new Promise((resolve) => { + execFile('docker', ['inspect', '--format', '{{json .Config}}', name], (error, stdout) => { + if (error) { + resolve(null) + return + } + try { + const containerConfig = JSON.parse(String(stdout).trim()) || {} + const prefix = 'SHUTDOWN_TIMEOUT_MS=' + const entry = (Array.isArray(containerConfig.Env) ? containerConfig.Env : []) + .find(e => String(e).startsWith(prefix)) + resolve({ + stopTimeout: Number.isInteger(containerConfig.StopTimeout) ? containerConfig.StopTimeout : null, + shutdownTimeoutMs: entry ? String(entry).slice(prefix.length) : null + }) + } catch { + resolve(null) + } + }) + }) +} + async function waitContainer(containerId) { return new Promise((resolve, reject) => { execFile('docker', ['wait', containerId], (error, stdout) => { @@ -280,6 +307,7 @@ module.exports = { removeContainer, killContainer, getContainerBindMounts, + getContainerStopSettings, forceRemoveContainerByName, probeContainerPresenceByName, execContainer, diff --git a/src/services/module_service/build_and_up.js b/src/services/module_service/build_and_up.js index 9560759..1756256 100644 --- a/src/services/module_service/build_and_up.js +++ b/src/services/module_service/build_and_up.js @@ -24,7 +24,7 @@ let { HUB_MODULE_NAME, LIBRARY_BUNDLES } = require('../../config') let { db } = require('../../state') let { getModuleDir, checkIfModuleExists, getDockerContainerImageName, getDockerNetwork, getDefaultConfig, validatePort } = require('../config_service') let { stopContainerByName, removeContainer, forceRemoveContainerByName, checkBuildKitAvailable } = require('../docker_service') -let { stopModuleContainer, stopTimeoutArgs } = require('../stop_budget_service') +let { stopModuleContainer, stopTimeoutArgs, shutdownTimeoutEnv } = require('../stop_budget_service') let { statusChanged } = require('../status_service') let { setHubDatabaseParameters } = require('../database_service') let { redactSecrets } = require('../../utils/helpers') @@ -236,12 +236,12 @@ function resolveSourceMetadata(reuseImage, dir) { // `docker run` error.message (which upstream logging prints, and an operator // pastes into a bug report). Mirrors DatabaseService's MYSQL_ROOT_PASSWORD // treatment. The value reaches the container identically; only argv changes. -// The observability names resolved above join the map here so they travel -// the same value-out-of-argv path. -function resolveContainerEnvironment(environmentVariables) { +// The observability names resolved above and the drain derived for drainModule +// (null on a one-shot run, which gets none) join the map here, off argv too. +function resolveContainerEnvironment(environmentVariables, drainModule) { const envArgs = [] const dockerEnv = config.childProcessEnv() - const containerEnv = { ...environmentVariables, ...resolveObservabilityEnv(environmentVariables) } + const containerEnv = { ...environmentVariables, ...resolveObservabilityEnv(environmentVariables), ...shutdownTimeoutEnv(drainModule, environmentVariables) } for (const key in containerEnv) { envArgs.push('--env', key) dockerEnv[key] = String(containerEnv[key]) @@ -389,7 +389,7 @@ async function buildAndUp(module, coin, network, overwriteContainerId = null, on await assertNoHostPortConflicts(dockerOptions.portArgs, containerPrefix) await assertReusableImage(reuseImage, containerPrefix, module, coin, network) const sourceMetadata = resolveSourceMetadata(reuseImage, dir) - const containerEnvironment = resolveContainerEnvironment(environmentVariables) + const containerEnvironment = resolveContainerEnvironment(environmentVariables, onlyExecution ? null : module) return launchContainer({ module, coin, network, overwriteContainerId, onlyExecution, dockerCmdArgs, reuseImage, environmentVariables, dir, containerPrefix, diff --git a/src/services/stop_budget_service.js b/src/services/stop_budget_service.js index 3ebba26..66f8154 100644 --- a/src/services/stop_budget_service.js +++ b/src/services/stop_budget_service.js @@ -71,6 +71,67 @@ function stopTimeoutArgs(module, env = MODULE_STOP_TIMEOUT_ENV) { return ['--stop-timeout', String(moduleStopTimeoutSeconds(module, env))] } +// Gap between a service's own hard-exit timer and docker's SIGKILL, so a drain +// that runs out of time still exits itself and logs why before the kill. +const STOP_DRAIN_MARGIN_MS = 20000 + +function isDefaultStopBudget(module, seconds) { + const fallback = MODULE_STOP_TIMEOUT_SECONDS[module] ?? DEFAULT_MODULE_STOP_TIMEOUT_SECONDS + return seconds === fallback +} + +// SHUTDOWN_TIMEOUT_MS the node hands a service, derived from its stop budget +// so the service's own hard exit always lands inside it: 120 s gives 100000, +// the decoder's and tracker's own default. Below 40 s the drain gets half the +// budget instead, since the margin would leave it nothing. Null for the coin +// node and for a service on the plain default budget, which keeps its own. +function moduleShutdownTimeoutMs(module, env = MODULE_STOP_TIMEOUT_ENV) { + if (module === 'node') return null + return shutdownTimeoutMsForBudget(module, moduleStopTimeoutSeconds(module, env)) +} + +function shutdownTimeoutMsForBudget(module, seconds) { + if (module === 'node') return null + const listed = Object.prototype.hasOwnProperty.call(MODULE_STOP_TIMEOUT_SECONDS, module) + if (!listed && isDefaultStopBudget(module, seconds)) return null + const budgetMs = seconds * 1000 + return Math.max(budgetMs - STOP_DRAIN_MARGIN_MS, Math.floor(budgetMs / 2)) +} + +// The container env entry that carries the derived drain budget. An explicit +// SHUTDOWN_TIMEOUT_MS in the module config wins and nothing is added, and so +// does a null module (a one-shot run has no stop budget to derive from). +function shutdownTimeoutEnv(module, moduleConfig, env = MODULE_STOP_TIMEOUT_ENV) { + if (!module) return {} + const configured = moduleConfig ? moduleConfig.SHUTDOWN_TIMEOUT_MS : undefined + if (configured !== undefined && configured !== null && String(configured).trim() !== '') return {} + const derived = moduleShutdownTimeoutMs(module, env) + return derived === null ? {} : { SHUTDOWN_TIMEOUT_MS: String(derived) } +} + +// Why a running service's own drain may not follow its current budget: the +// container carries a different stamped budget, or it predates the forwarded +// SHUTDOWN_TIMEOUT_MS while an override is set. Null when nothing drifted, the +// container could not be read, or neither side involves a forwarded drain. +function describeStopBudgetDrift(module, coin, network, settings, budgetSeconds) { + if (!settings || module === 'node') return null + const forwards = shutdownTimeoutMsForBudget(module, budgetSeconds) !== null + if (!forwards && settings.shutdownTimeoutMs === null) return null + const where = coin && network ? ` (${coin} ${network})` : '' + let created + if (settings.stopTimeout !== budgetSeconds) { + created = Number.isInteger(settings.stopTimeout) + ? `under a ${settings.stopTimeout} s stop budget` : 'without a stop budget' + } else if (settings.shutdownTimeoutMs === null && !isDefaultStopBudget(module, budgetSeconds)) { + created = 'before the node forwarded SHUTDOWN_TIMEOUT_MS' + } else { + return null + } + return `WARNING: ${module}${where} was created ${created}, so its own drain timer does not follow ` + + `${moduleStopTimeoutEnvName(module)} (now ${budgetSeconds} s). Recreate it (xchain-node recreate) so the ` + + 'node forwards SHUTDOWN_TIMEOUT_MS from the current budget.' +} + // SIGTERM's default action (128 + 15): a process with no drain registered, // or npm relaying its child's, which is how a drainless service always stops. const EXIT_ON_SIGTERM_DEFAULT = 143 @@ -94,14 +155,16 @@ function describeModuleStopOutcome(module, coin, network, outcome, budgetSeconds const where = coin && network ? ` (${coin} ${network})` : '' if (outcome.killed) { return `WARNING: ${module}${where} did not exit within the ${budgetSeconds} s budget and was killed. ` + - `Raise ${moduleStopTimeoutEnvName(module)} if this service needs longer to finish its block, ` + - 'and keep the service\'s own SHUTDOWN_TIMEOUT_MS below it where the service reads one.' + `Raise ${moduleStopTimeoutEnvName(module)} if this service needs longer to finish its block; the node ` + + 'derives the service\'s own SHUTDOWN_TIMEOUT_MS from it when the container is next recreated, unless ' + + 'the module config sets one.' } if (stoppedUnclean(outcome)) { return `WARNING: ${module}${where} exited with code ${outcome.exitCode} after ${outcome.seconds} s, inside the ` + `${budgetSeconds} s budget, so its shutdown drain did not complete: it overran the service's own ` + 'hard-exit timer (SHUTDOWN_TIMEOUT_MS where the service reads one) or failed. Check the service log. ' + - `Raising ${moduleStopTimeoutEnvName(module)} alone does not give that drain more time.` + `Raising ${moduleStopTimeoutEnvName(module)} gives that drain more time only once the container is ` + + 'recreated, and not while the module config sets SHUTDOWN_TIMEOUT_MS.' } return `Stopped ${module}${where} cleanly in ${outcome.seconds} s (budget ${budgetSeconds} s).` } @@ -109,9 +172,17 @@ function describeModuleStopOutcome(module, coin, network, outcome, budgetSeconds // One stop for every CLI path that takes a service container down: stop with // the budget, say what happened, return the outcome so the caller can decide // whether a kill matters to it. `stopContainerByName` accepts an id as well -// as a name (docker echoes back whatever it was given). -async function stopModuleContainer(stopContainerByName, module, coin, network, containerRef, env = MODULE_STOP_TIMEOUT_ENV) { +// as a name (docker echoes back whatever it was given). `readStopSettings`, +// when given, reads the container first so a drain that predates the current +// budget is named before the stop rather than after a kill. +async function stopModuleContainer(stopContainerByName, module, coin, network, containerRef, + env = MODULE_STOP_TIMEOUT_ENV, readStopSettings = null) { const budget = moduleStopTimeoutSeconds(module, env) + if (typeof readStopSettings === 'function' && module !== 'node') { + const settings = await Promise.resolve().then(() => readStopSettings(containerRef)).catch(() => null) + const drift = describeStopBudgetDrift(module, coin, network, settings, budget) + if (drift) logger.warn(drift) + } const outcome = await stopContainerByName(containerRef, budget) const line = describeModuleStopOutcome(module, coin, network, outcome, budget) if (line) { @@ -127,6 +198,10 @@ module.exports = { moduleStopTimeoutEnvName, moduleStopTimeoutSeconds, stopTimeoutArgs, + STOP_DRAIN_MARGIN_MS, + moduleShutdownTimeoutMs, + shutdownTimeoutEnv, + describeStopBudgetDrift, stoppedUnclean, describeModuleStopOutcome, stopModuleContainer diff --git a/test/unit/docker_service.test/stop_settings/get_container_stop_settings.test.js b/test/unit/docker_service.test/stop_settings/get_container_stop_settings.test.js new file mode 100644 index 0000000..8e0bc4c --- /dev/null +++ b/test/unit/docker_service.test/stop_settings/get_container_stop_settings.test.js @@ -0,0 +1,55 @@ +'use strict' + +// Copyright © 2025–2026 Dankest, LLC +// Based on XChain Platform by Dankest, LLC – https://dankest.llc +// +// SPDX-License-Identifier: AGPL-3.0-or-later +// +// This file is part of XChain Platform. Licensed under the GNU Affero +// General Public License v3.0 or later; see LICENSE.md. A commercial +// license (without AGPL source-disclosure terms) is available - +// contact legal@dankest.llc. + +// Pins the stop-settings readback: the stamped budget and the drain value +// come back, the rest of the env (which carries secrets) does not, and a +// docker that cannot answer reads as unknown rather than throwing. + +const sinon = require('sinon') +const { expect } = require('chai') +const { proxyquireDockerService } = require('../../../helpers/docker_service_loader') + +function loadDockerService(execFile) { + return proxyquireDockerService(require.resolve('../../../../src/services/docker_service'), { + 'child_process': { execFile, spawn: sinon.stub(), spawnSync: sinon.stub() }, + 'util': { promisify: (fn) => fn }, + 'fs': { readFileSync: sinon.stub() }, + 'blessed': { screen: sinon.stub(), text: sinon.stub(), log: sinon.stub() } + }) +} + +function answering(error, stdout) { + return sinon.stub().callsFake((cmd, args, cb) => cb(error, stdout)) +} + +describe('DockerService', function () { + describe('getContainerStopSettings()', function () { + it('returns the stamped budget and SHUTDOWN_TIMEOUT_MS and nothing else from the env', async function () { + const config = { StopTimeout: 300, Env: ['DECODER_DB_PASS=hunter2', 'SHUTDOWN_TIMEOUT_MS=280000'] } + const execFile = answering(null, JSON.stringify(config) + '\n') + const settings = await loadDockerService(execFile).getContainerStopSettings('abc123') + expect(execFile.firstCall.args[1]).to.deep.equal(['inspect', '--format', '{{json .Config}}', 'abc123']) + expect(settings).to.deep.equal({ stopTimeout: 300, shutdownTimeoutMs: '280000' }) + }) + + it('reports each value as null when the container was created without it', async function () { + const execFile = answering(null, JSON.stringify({ StopTimeout: null, Env: ['NETWORK=bitcoin-mainnet'] })) + const settings = await loadDockerService(execFile).getContainerStopSettings('abc123') + expect(settings).to.deep.equal({ stopTimeout: null, shutdownTimeoutMs: null }) + }) + + it('resolves null when docker fails or answers something unparseable', async function () { + expect(await loadDockerService(answering(new Error('no such container'), '')).getContainerStopSettings('gone')).to.equal(null) + expect(await loadDockerService(answering(null, 'not json')).getContainerStopSettings('abc123')).to.equal(null) + }) + }) +}) diff --git a/test/unit/module_operations.test/03_module_controls.test.js b/test/unit/module_operations.test/03_module_controls.test.js index af02b98..c3f33b8 100644 --- a/test/unit/module_operations.test/03_module_controls.test.js +++ b/test/unit/module_operations.test/03_module_controls.test.js @@ -138,6 +138,20 @@ describe('moduleOperations', function () { expect(stubs.stopContainerByName.calledWith('container-id-123', 30)).to.be.true expect(stubs.stopContainer.called).to.be.false }) + + it('reads the container before the stop so a drain that predates the budget is named', async function () { + const stubs = makeStubs() + stubs.getContainerStopSettings.resolves({ stopTimeout: 300, shutdownTimeoutMs: '280000' }) + const warn = sinon.stub(console, 'warn') + try { + await loadOperations(stubs).stopModules({ bitcoin: { mainnet: ['xchain-decoder'] } }) + expect(stubs.getContainerStopSettings.calledOnceWith('container-id-123')).to.be.true + expect(stubs.getContainerStopSettings.calledBefore(stubs.stopContainerByName)).to.be.true + expect(warn.args.some(a => /xchain-decoder \(bitcoin mainnet\) was created under a 300 s stop budget/.test(String(a[0])))).to.be.true + } finally { + warn.restore() + } + }) }) }) diff --git a/test/unit/module_operations.test/helpers/harness.js b/test/unit/module_operations.test/helpers/harness.js index dbbe35e..7e72255 100644 --- a/test/unit/module_operations.test/helpers/harness.js +++ b/test/unit/module_operations.test/helpers/harness.js @@ -54,6 +54,7 @@ function makeStubs() { probeContainerPresenceByName: sinon.stub().resolves('exists'), stopContainer: sinon.stub().resolves(true), stopContainerByName: sinon.stub().resolves({ stopped: true, seconds: 1, killed: false }), + getContainerStopSettings: sinon.stub().resolves(null), startContainer: sinon.stub().resolves(true), restartContainer: sinon.stub().resolves(true), execContainer: sinon.stub().resolves('exec-output'), @@ -143,6 +144,7 @@ function loadOperations(stubs, constantsOverrides = null) { probeContainerPresenceByName: stubs.probeContainerPresenceByName, stopContainer: stubs.stopContainer, stopContainerByName: stubs.stopContainerByName, + getContainerStopSettings: stubs.getContainerStopSettings, startContainer: stubs.startContainer, restartContainer: stubs.restartContainer, execContainer: stubs.execContainer, diff --git a/test/unit/module_service.test/12_buildandup_shutdown_timeout_env.test.js b/test/unit/module_service.test/12_buildandup_shutdown_timeout_env.test.js new file mode 100644 index 0000000..36d0b1e --- /dev/null +++ b/test/unit/module_service.test/12_buildandup_shutdown_timeout_env.test.js @@ -0,0 +1,76 @@ +'use strict' + +// Copyright © 2025–2026 Dankest, LLC +// Based on XChain Platform by Dankest, LLC – https://dankest.llc +// +// SPDX-License-Identifier: AGPL-3.0-or-later +// +// This file is part of XChain Platform. Licensed under the GNU Affero +// General Public License v3.0 or later; see LICENSE.md. A commercial +// license (without AGPL source-disclosure terms) is available - +// contact legal@dankest.llc. + +// Pins the drain budget the node hands a service at creation: the decoder and +// tracker get SHUTDOWN_TIMEOUT_MS derived from their stop budget, off argv like +// every other env value, and an explicit module config value is left alone. + +const { + sinon, expect, makeStubs, makeConfigServiceStub, loadModuleService, + inspectMemoryBytes, moduleSuite +} = require('./support/helpers') + +// Records every docker call with its options; the tracker's memory readback +// answers like the shared create fake does. +function captureCreate(stubs) { + const seen = [] + stubs.execFile.callsFake((cmd, args, ...rest) => { + const opts = typeof rest[0] === 'function' ? {} : (rest[0] || {}) + const cb = typeof rest[0] === 'function' ? rest[0] : rest[1] + seen.push({ cmd, args, opts }) + if (args[0] === 'run') cb(null, 'd'.repeat(64) + '\n', '') + else if (args[0] === 'inspect') cb(null, String(inspectMemoryBytes(seen, 'requested')) + '\n') + else cb(null, '') + }) + return () => seen.find(c => c.args[0] === 'run') +} + +async function createWithConfig(module, extraConfig, onlyExecution = false) { + const stubs = makeStubs() + const configService = makeConfigServiceStub() + if (extraConfig) { + const base = await configService.getDefaultConfig() + configService.getDefaultConfig = sinon.stub().resolves({ ...base, ...extraConfig }) + } + const runOf = captureCreate(stubs) + const ms = loadModuleService(stubs, null, { './config_service': configService }) + await ms.buildAndUp(module, 'bitcoin', 'mainnet', null, onlyExecution) + return runOf() +} + +moduleSuite('buildAndUp() SHUTDOWN_TIMEOUT_MS', function () { + it('hands the decoder and tracker the drain derived from their budget, by name only on argv', async function () { + for (const module of ['xchain-decoder', 'xchain-utxo-tracker']) { + const run = await createWithConfig(module) + expect(run.args, module).to.include('SHUTDOWN_TIMEOUT_MS') + expect(run.args, module).to.not.include('SHUTDOWN_TIMEOUT_MS=100000') + expect(run.opts.env.SHUTDOWN_TIMEOUT_MS, module).to.equal('100000') + } + }) + + it('leaves a service on the plain default budget to its own drain default', async function () { + const run = await createWithConfig('xchain-encoder') + expect(run.args).to.not.include('SHUTDOWN_TIMEOUT_MS') + expect(run.opts.env.SHUTDOWN_TIMEOUT_MS).to.equal(undefined) + }) + + it('keeps an explicit SHUTDOWN_TIMEOUT_MS from the module config', async function () { + const run = await createWithConfig('xchain-decoder', { SHUTDOWN_TIMEOUT_MS: '45000' }) + expect(run.opts.env.SHUTDOWN_TIMEOUT_MS).to.equal('45000') + }) + + it('gives a one-shot execution container no drain, as it gets no stop budget', async function () { + const run = await createWithConfig('xchain-decoder', null, true) + expect(run.args).to.not.include('--stop-timeout') + expect(run.args).to.not.include('SHUTDOWN_TIMEOUT_MS') + }) +}) diff --git a/test/unit/stop_budget_service.test.js b/test/unit/stop_budget_service.test.js index c40c506..a0e4075 100644 --- a/test/unit/stop_budget_service.test.js +++ b/test/unit/stop_budget_service.test.js @@ -109,7 +109,7 @@ describe('StopBudgetService', function () { const warning = warnStub.args.map(a => String(a[0])).find(l => /exited with code 1/.test(l)) expect(warning).to.match(/xchain-decoder \(bitcoin mainnet\) exited with code 1 after 100 s, inside the 120 s budget/) expect(warning).to.match(/SHUTDOWN_TIMEOUT_MS/) - expect(warning).to.match(/XCHAIN_NODE_MODULE_STOP_TIMEOUT_SECONDS_XCHAIN_DECODER alone does not/) + expect(warning).to.match(/XCHAIN_NODE_MODULE_STOP_TIMEOUT_SECONDS_XCHAIN_DECODER gives that drain more time only once the container is recreated/) expect(logStub.args.some(a => /cleanly/.test(String(a[0])))).to.be.false }) @@ -130,3 +130,105 @@ describe('StopBudgetService', function () { }) }) }) + +// The node owns the drain: the service's own hard-exit timer is derived from +// the same budget docker kills at, so it always fires first. +describe('StopBudgetService', function () { + beforeEach(prepareConsoleStubs) + afterEach(restoreConsoleStubs) + + const DECODER_300 = { XCHAIN_NODE_MODULE_STOP_TIMEOUT_SECONDS_XCHAIN_DECODER: '300' } + + describe('moduleShutdownTimeoutMs()', function () { + it('reproduces the decoder and tracker default of 100000 ms from the default 120 s budget', function () { + expect(sbs.moduleShutdownTimeoutMs('xchain-decoder', {})).to.equal(100000) + expect(sbs.moduleShutdownTimeoutMs('xchain-utxo-tracker', {})).to.equal(100000) + }) + + it('moves with an override in both directions and stays strictly under the budget', function () { + expect(sbs.moduleShutdownTimeoutMs('xchain-decoder', DECODER_300)).to.equal(280000) + const lowered = { XCHAIN_NODE_MODULE_STOP_TIMEOUT_SECONDS_XCHAIN_DECODER: '60' } + expect(sbs.moduleShutdownTimeoutMs('xchain-decoder', lowered)).to.equal(40000) + for (const seconds of ['1', '5', '20', '21', '39', '40', '41']) { + const env = { XCHAIN_NODE_MODULE_STOP_TIMEOUT_SECONDS_XCHAIN_DECODER: seconds } + const ms = sbs.moduleShutdownTimeoutMs('xchain-decoder', env) + expect(ms, seconds).to.be.greaterThan(0).and.lessThan(parseInt(seconds, 10) * 1000) + } + }) + + it('leaves a service on the plain default budget, and the coin node, to their own defaults', function () { + expect(sbs.moduleShutdownTimeoutMs('xchain-explorer', {})).to.equal(null) + expect(sbs.moduleShutdownTimeoutMs('node', {})).to.equal(null) + const raised = { XCHAIN_NODE_MODULE_STOP_TIMEOUT_SECONDS_XCHAIN_EXPLORER: '90' } + expect(sbs.moduleShutdownTimeoutMs('xchain-explorer', raised)).to.equal(70000) + }) + }) + + describe('shutdownTimeoutEnv()', function () { + it('adds the derived value unless the module config sets its own', function () { + expect(sbs.shutdownTimeoutEnv('xchain-decoder', {}, {})).to.deep.equal({ SHUTDOWN_TIMEOUT_MS: '100000' }) + expect(sbs.shutdownTimeoutEnv('xchain-decoder', { SHUTDOWN_TIMEOUT_MS: '45000' }, DECODER_300)).to.deep.equal({}) + expect(sbs.shutdownTimeoutEnv('xchain-decoder', { SHUTDOWN_TIMEOUT_MS: ' ' }, DECODER_300)) + .to.deep.equal({ SHUTDOWN_TIMEOUT_MS: '280000' }) + expect(sbs.shutdownTimeoutEnv('xchain-encoder', {}, {})).to.deep.equal({}) + expect(sbs.shutdownTimeoutEnv(null, {}, DECODER_300), 'a one-shot run').to.deep.equal({}) + }) + }) +}) + +describe('StopBudgetService', function () { + beforeEach(prepareConsoleStubs) + afterEach(restoreConsoleStubs) + + const DECODER_300 = { XCHAIN_NODE_MODULE_STOP_TIMEOUT_SECONDS_XCHAIN_DECODER: '300' } + const driftLine = () => warnStub.args.map(a => String(a[0])).find(l => /was created/.test(l)) + const cleanStop = () => sinon.stub().resolves({ stopped: true, seconds: 5, killed: false, exitCode: 0 }) + + describe('stopModuleContainer() on a container that predates its budget', function () { + it('warns before the stop when the container was stamped with another budget', async function () { + const read = sinon.stub().resolves({ stopTimeout: 120, shutdownTimeoutMs: '100000' }) + const stop = cleanStop() + await sbs.stopModuleContainer(stop, 'xchain-decoder', 'bitcoin', 'mainnet', 'abc123', DECODER_300, read) + expect(read.calledOnceWith('abc123')).to.be.true + expect(read.calledBefore(stop)).to.be.true + expect(driftLine()).to.match(/xchain-decoder \(bitcoin mainnet\) was created under a 120 s stop budget/) + expect(driftLine()).to.match(/XCHAIN_NODE_MODULE_STOP_TIMEOUT_SECONDS_XCHAIN_DECODER \(now 300 s\)/) + expect(driftLine()).to.match(/xchain-node recreate/) + }) + + it('warns when an override is set but the container never got SHUTDOWN_TIMEOUT_MS', async function () { + const read = sinon.stub().resolves({ stopTimeout: 300, shutdownTimeoutMs: null }) + await sbs.stopModuleContainer(cleanStop(), 'xchain-decoder', 'bitcoin', 'mainnet', 'abc123', DECODER_300, read) + expect(driftLine()).to.match(/was created before the node forwarded SHUTDOWN_TIMEOUT_MS/) + }) + + it('stays quiet for a container that matches its budget, at the default or recreated', async function () { + const atDefault = sinon.stub().resolves({ stopTimeout: 120, shutdownTimeoutMs: null }) + await sbs.stopModuleContainer(cleanStop(), 'xchain-decoder', 'bitcoin', 'mainnet', 'a', {}, atDefault) + const recreated = sinon.stub().resolves({ stopTimeout: 300, shutdownTimeoutMs: '280000' }) + await sbs.stopModuleContainer(cleanStop(), 'xchain-decoder', 'bitcoin', 'mainnet', 'b', DECODER_300, recreated) + const unreadable = sinon.stub().rejects(new Error('docker gone')) + await sbs.stopModuleContainer(cleanStop(), 'xchain-decoder', 'bitcoin', 'mainnet', 'c', DECODER_300, unreadable) + expect(warnStub.called).to.be.false + }) + + it('stays quiet for a default-budget service that never carried a forwarded drain', async function () { + const read = sinon.stub().resolves({ stopTimeout: null, shutdownTimeoutMs: null }) + await sbs.stopModuleContainer(cleanStop(), 'xchain-explorer', 'bitcoin', 'mainnet', 'abc123', {}, read) + expect(warnStub.called).to.be.false + }) + + it('warns when an override was dropped but the container still carries the drain it derived', async function () { + const read = sinon.stub().resolves({ stopTimeout: 90, shutdownTimeoutMs: '70000' }) + await sbs.stopModuleContainer(cleanStop(), 'xchain-explorer', 'bitcoin', 'mainnet', 'abc123', {}, read) + expect(driftLine()).to.match(/xchain-explorer \(bitcoin mainnet\) was created under a 90 s stop budget/) + expect(driftLine()).to.match(/\(now 30 s\)/) + }) + + it('never reads the coin node, whose budget is a flush and not a drain', async function () { + const read = sinon.stub().resolves({ stopTimeout: 1, shutdownTimeoutMs: '1' }) + await sbs.stopModuleContainer(cleanStop(), 'node', 'bitcoin', 'mainnet', 'abc123', {}, read) + expect(read.called).to.be.false + }) + }) +}) From 80a434f64f211b5dbcf6c5a33d2c5abe4dd35ecc Mon Sep 17 00:00:00 2001 From: J-Dog Date: Wed, 23 Sep 2026 00:17:44 -0700 Subject: [PATCH 31/35] test: allow slow bootstrap proxyquire setup Give the real-shell freshness probe a scoped timeout for cold macOS proxyquire loads while keeping the test file within its structure budget. --- test/unit/bootstrap_service.test.js | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/test/unit/bootstrap_service.test.js b/test/unit/bootstrap_service.test.js index c9fb1e0..55b9608 100644 --- a/test/unit/bootstrap_service.test.js +++ b/test/unit/bootstrap_service.test.js @@ -258,7 +258,6 @@ describe('BootstrapService', function () { }) }) - // A daemon that cannot be reached says nothing about the volume. describe('BootstrapService', function () { beforeEach(saveRequireSignedBootstrapSetting) @@ -332,7 +331,6 @@ describe('BootstrapService', function () { }) }) - // The decisive assertion for this probe is about the SHELL, and no stub // can make it: the test above stubs a REJECTED exec, which is the one // shape a failed `ls` never produced. `ls -A /data 2>/dev/null | head -1` @@ -345,6 +343,8 @@ describe('BootstrapService', function () { afterEach(restoreRequireSignedBootstrapSetting) describe('utxoTrackerVolumeFreshness()', function () { it('uses a listing command that exits non-zero when the listing fails', async function () { + // Cold macOS proxyquire setup can exceed the suite's 2-second timeout. + this.timeout(10000) const fs = require('fs') const os = require('os') const path = require('path') From eb60f94031ffc11cc20556c53cf444830259affc Mon Sep 17 00:00:00 2001 From: J-Dog Date: Wed, 23 Sep 2026 09:30:59 -0700 Subject: [PATCH 32/35] fix(node): zero the bridge and policy watermark graces on regtest The regtest grace list predates the bridge-family barriers, so an armed regtest indexer kept their 120 s frozen grace. A bitcoin-only venue never finalizes a transfer and never paid for it; the litecoin and dogecoin legs get their gas over the bridge, so after the first transfer every block past the newest effective_time was deferred eight times, about 2m10s, before it parsed. On nightly run 35829816064 every e2e step on those legs took about 130 s against about 10 s on bitcoin, and both legs hit the 360-minute job budget. HUB_SYNC_BRIDGE_GRACE_S and HUB_SYNC_POLICY_GRACE_S now join the other six in the regtest passthrough and default to 0 there; the indexer still ignores both off regtest. The nightly pins them to 2 s beside the anchor-attest grace. --- .github/workflows/nightly-e2e.yml | 10 +++++++++ src/config/env_views.js | 3 ++- src/services/config_service/services.js | 11 +++++++++- .../rollcall_and_mirror.test.js | 21 ++++++++++++------- 4 files changed, 35 insertions(+), 10 deletions(-) diff --git a/.github/workflows/nightly-e2e.yml b/.github/workflows/nightly-e2e.yml index f84bd4b..bfaed6f 100644 --- a/.github/workflows/nightly-e2e.yml +++ b/.github/workflows/nightly-e2e.yml @@ -199,6 +199,16 @@ jobs: # per-node value forks settlement - so this cannot leak onto a shared ledger even # if it is copied somewhere it does not belong. HUB_SYNC_ANCHOR_ATTEST_GRACE_S: "2" + # The same cadence mismatch on the two bridge-family barriers, which only the + # litecoin and dogecoin legs arm: their gas arrives over the bridge, and once + # one transfer is finalized every block past its effective_time waits for the + # watermark to reach blockTime + 120. Measured on run 35829816064: about 2m10s + # per affected block on both of their indexers and on the bitcoin gas rail + # beside them, so every e2e step there took about 130 s against about 10 s on + # the bitcoin leg. 2s for the reason given above; regtest-only in the indexer + # for the same reason, and needs xchain-node's passthrough of both names. + HUB_SYNC_BRIDGE_GRACE_S: "2" + HUB_SYNC_POLICY_GRACE_S: "2" # Bounds ONE mirror-barrier attempt. On expiry the block is DEFERRED and # retried, never committed uncertified - the indexer says so outright # ("purely operational: it opens no barrier and commits no block") - so a diff --git a/src/config/env_views.js b/src/config/env_views.js index 64d17c4..2581dbb 100644 --- a/src/config/env_views.js +++ b/src/config/env_views.js @@ -80,7 +80,8 @@ const LIST_VIEWS = { // The hub-sync watermark graces a regtest indexer is handed (hubSyncRegtestGraceVars). HUB_SYNC_GRACE_ENV: ['hub sync grace', [ "HUB_SYNC_PRICE_GRACE_S", "HUB_SYNC_ORACLE_GRACE_S", "HUB_SYNC_ATTEST_RESPONSE_GRACE_S", - "HUB_SYNC_MATCH_GRACE_S", "HUB_SYNC_CALL_GRACE_S", "HUB_SYNC_ANCHOR_ATTEST_GRACE_S" + "HUB_SYNC_MATCH_GRACE_S", "HUB_SYNC_CALL_GRACE_S", "HUB_SYNC_ANCHOR_ATTEST_GRACE_S", + "HUB_SYNC_BRIDGE_GRACE_S", "HUB_SYNC_POLICY_GRACE_S" ]], // The explorer's published host ports. EXPLORER_PORT_ENV: ['explorer port', ["EXPLORER_PORT_HTTP", "EXPLORER_PORT_HTTPS", "EXPLORER_PORT"]], diff --git a/src/services/config_service/services.js b/src/services/config_service/services.js index 085c607..4205d62 100644 --- a/src/services/config_service/services.js +++ b/src/services/config_service/services.js @@ -275,6 +275,14 @@ function configureIndexerBeforeHubKey(defaultValues, module, network) { // barrier, then again on the anchor-reward attestation barrier once // match was cleared (hub_db_sync.js reads all of them through // resolveWatermarkGrace, regtest-overridable only). +// The bridge and policy pair joined the indexer after this list was written +// and was missed. Their barriers stay open while no transfer is finalized, so a +// bitcoin-only venue never paid for it; the litecoin and dogecoin legs get gas +// over the bridge, so after the first transfer every block whose time passed +// the newest effective_time waited out the 120 s grace. Measured on the +// 2026-09-23 nightly (run 35829816064): 8 deferrals, about 2m10s, per affected +// block on the DOGE indexer, and every e2e step there took about 130 s where +// the bitcoin leg's took about 10 s. function configureIndexerAfterHubKey(defaultValues, module, network) { if (module === XChainService.XCHAIN_INDEXER) { if (defaultValues.HUB_API_KEY) { @@ -291,7 +299,8 @@ function configureIndexerAfterHubKey(defaultValues, module, network) { if (network === Network.REGTEST) { const hubSyncRegtestGraceVars = [ "HUB_SYNC_PRICE_GRACE_S", "HUB_SYNC_ORACLE_GRACE_S", "HUB_SYNC_ATTEST_RESPONSE_GRACE_S", - "HUB_SYNC_MATCH_GRACE_S", "HUB_SYNC_CALL_GRACE_S", "HUB_SYNC_ANCHOR_ATTEST_GRACE_S" + "HUB_SYNC_MATCH_GRACE_S", "HUB_SYNC_CALL_GRACE_S", "HUB_SYNC_ANCHOR_ATTEST_GRACE_S", + "HUB_SYNC_BRIDGE_GRACE_S", "HUB_SYNC_POLICY_GRACE_S" ] for (const varName of hubSyncRegtestGraceVars) { defaultValues[varName] = (config.HUB_SYNC_GRACE_ENV[varName] !== undefined && config.HUB_SYNC_GRACE_ENV[varName] !== "") diff --git a/test/unit/config_service.test/rollcall_and_mirror.test.js b/test/unit/config_service.test/rollcall_and_mirror.test.js index 9e7608a..5fd3aa8 100644 --- a/test/unit/config_service.test/rollcall_and_mirror.test.js +++ b/test/unit/config_service.test/rollcall_and_mirror.test.js @@ -245,15 +245,20 @@ function mirrorAdmissionPassthrough() { }) } +// Every watermark grace the indexer's hub mirror resolves, each of which must be +// zeroed on regtest or an armed venue wedges every freshly mined block (the +// price-grace failure the regtest mirror wedge records). +const GRACE_VARS = [ + 'HUB_SYNC_PRICE_GRACE_S', 'HUB_SYNC_ORACLE_GRACE_S', 'HUB_SYNC_ATTEST_RESPONSE_GRACE_S', + 'HUB_SYNC_MATCH_GRACE_S', 'HUB_SYNC_CALL_GRACE_S', 'HUB_SYNC_ANCHOR_ATTEST_GRACE_S', + // The bridge-family pair. Missing here, a venue with one finalized XBRIDGE + // transfer held every later block about 130 s at the 120 s frozen grace. + 'HUB_SYNC_BRIDGE_GRACE_S', 'HUB_SYNC_POLICY_GRACE_S' +] + // Regtest mirror arming: the regtest indexer's hub-mirror connection, unset -// before this row, and the three watermark graces that must be zeroed alongside -// it or an armed regtest venue wedges every freshly mined block (the price-grace -// failure the regtest mirror wedge records). +// before this row, and the GRACE_VARS above zeroed alongside it. function regtestMirrorArming() { - const GRACE_VARS = [ - 'HUB_SYNC_PRICE_GRACE_S', 'HUB_SYNC_ORACLE_GRACE_S', 'HUB_SYNC_ATTEST_RESPONSE_GRACE_S', - 'HUB_SYNC_MATCH_GRACE_S', 'HUB_SYNC_CALL_GRACE_S', 'HUB_SYNC_ANCHOR_ATTEST_GRACE_S' - ] let saved beforeEach(function () { saved = {} @@ -276,7 +281,7 @@ function regtestMirrorArming() { expect(config['HUB_DB_PASS']).to.equal(config['INDEXER_DB_PASS']) }) - it('defaults all three watermark graces to 0 on regtest when the host sets none of them', async function () { + it('defaults every watermark grace to 0 on regtest when the host sets none of them', async function () { const cs = makeServiceWithConfig('') const config = await cs.getDefaultConfig('xchain-indexer', 'bitcoin', 'regtest') for (const v of GRACE_VARS) expect(config[v], v).to.equal('0') From 449d62858c0a170e485b8b241bfdfe00022a3e0d Mon Sep 17 00:00:00 2001 From: J-Dog Date: Wed, 23 Sep 2026 09:46:21 -0700 Subject: [PATCH 33/35] chore(release): v0.20.1 --- CHANGELOG.md | 6 ++++++ README.md | 2 +- package-lock.json | 4 ++-- package.json | 2 +- 4 files changed, 10 insertions(+), 4 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 1ff73c7..6a68f50 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,12 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +## [0.20.1] - 2026-09-23 + +### Fixed +- Preserved branch-based installs, surfaced update and drain failures, and aligned service shutdown timeouts with the node stop budget. + + ## [0.20.0] - 2026-09-17 ### Added diff --git a/README.md b/README.md index c03688d..eeeea22 100644 --- a/README.md +++ b/README.md @@ -4,7 +4,7 @@ # XChain Platform Node

- Version + Version Tests Node License diff --git a/package-lock.json b/package-lock.json index 988e320..49036a2 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "xchain-node", - "version": "0.20.0", + "version": "0.20.1", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "xchain-node", - "version": "0.20.0", + "version": "0.20.1", "license": "AGPL-3.0-or-later", "dependencies": { "@dankest-llc/xchain-sdk": "^0.18.0", diff --git a/package.json b/package.json index d026c28..7dd05ca 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "xchain-node", - "version": "0.20.0", + "version": "0.20.1", "description": "xchain-node allows users to install, configure and run XChain platform nodes.", "license": "AGPL-3.0-or-later", "repository": { From c26dace0697b4acd97667bc47ede1e62ac4304ec Mon Sep 17 00:00:00 2001 From: J-Dog Date: Wed, 23 Sep 2026 10:55:31 -0700 Subject: [PATCH 34/35] ci(e2e): shard the action suite five ways per coin, one combined run The MT3 matrix ran each coin's full action suite on one runner: 3h14m on bitcoin (run 35605562229), and the two-stack litecoin and dogecoin legs hit the 360-minute ceiling. A plan job now splits test/actions into five weight-balanced groups (scripts/e2e_shard_plan.js, weights measured off that run), each e2e-shard (, ) job boots its own stack and runs only its groups through e2etest's existing testName argument, and shard 1 also runs security and performance. The per-coin aggregate keeps the old job name e2e () and artifact e2e-logs-, so preflight.js and judge-matrix.js read the run unchanged. It fails its coin unless every planned shard uploaded, exited 0, ran tests and failed none (scripts/e2e_shard_merge.js), then writes one bannered action log with the summed summary last and keeps the raw shard logs as .log.txt. --- .github/workflows/nightly-e2e.yml | 265 +++++++++++++++--- scripts/e2e_shard_merge.js | 201 +++++++++++++ scripts/e2e_shard_plan.js | 147 ++++++++++ scripts/e2e_shard_weights.json | 88 ++++++ .../sharding.test.js | 227 +++++++++++++++ .../two_stack_legs.test.js | 4 +- 6 files changed, 887 insertions(+), 45 deletions(-) create mode 100644 scripts/e2e_shard_merge.js create mode 100644 scripts/e2e_shard_plan.js create mode 100644 scripts/e2e_shard_weights.json create mode 100644 test/unit/nightly_e2e_workflow.test/sharding.test.js diff --git a/.github/workflows/nightly-e2e.yml b/.github/workflows/nightly-e2e.yml index bfaed6f..bd9d819 100644 --- a/.github/workflows/nightly-e2e.yml +++ b/.github/workflows/nightly-e2e.yml @@ -136,46 +136,115 @@ on: permissions: contents: read +# WHY THE ACTION SUITE IS SHARDED. One coin's full pass was a single job: boot a +# stack, then run ~110 action files, security and performance back to back. +# Measured on run 35605562229 the bitcoin action step alone took 3h14m, and the +# litecoin and dogecoin legs, which boot two stacks and wait on the bridge, ran +# into the 360-minute job ceiling, so the three-coin MT3 matrix cost about four +# hours of wall clock at every release cut. +# +# Each coin now runs as three kinds of job, still in ONE run: +# +# plan once per run: clones xchain-e2e-test at the ref +# under test and splits test/actions into +# E2E_SHARDS groups by measured weight +# (scripts/e2e_shard_plan.js). One plan, so every +# shard agrees on who owns which suite. +# e2e-shard (, ) one fresh stack per shard, running only its +# groups; shard 1 also runs security and perf. +# Uploads e2e-shard--. +# e2e () the aggregate. Same name as the old per-coin job, +# and it uploads the same e2e-logs- artifact, +# so the cut kit (preflight.js, judge-matrix.js) +# reads this run exactly as it read an unsharded one. +# It fails its coin unless every planned shard ran +# nonzero tests with zero failures +# (scripts/e2e_shard_merge.js). +# +# A shard is a fresh stack, so every suite already had to be runnable alone +# (the `suite` input has always relied on that); sharding adds no new ordering +# assumption. Five shards bring the bitcoin weights to ~42 minutes of action +# per shard, about an hour per job with setup and boot. jobs: - e2e: + plan: + runs-on: ubuntu-latest + timeout-minutes: 10 + outputs: + coins: ${{ steps.coins.outputs.coins }} + shards: ${{ steps.plan.outputs.shards }} + plan: ${{ steps.plan.outputs.plan }} + env: + STACK_REF: ${{ github.event.inputs.ref || 'develop' }} + SUITE: ${{ github.event.inputs.suite || '' }} + # Sized off run 35605562229: ~204 minutes of bitcoin action weight over 5 + # shards is ~42 each. Raise it here if the suite grows; nothing else + # hardcodes the count. Three coins at 5 shards is 15 concurrent runners. + E2E_SHARDS: '5' + steps: + # The plan and merge scripts come from THIS workflow's commit, not from + # the ref under test, so an older ref (a published vX.Y.Z) that predates + # them can still be graded. + - uses: actions/checkout@v4 + with: + sparse-checkout: scripts + + - uses: actions/setup-node@v4 + with: + node-version: '22' + + # WHY THE COIN LIST IS EVENT-DEPENDENT. A scheduled run carries NO inputs, so + # without this it would silently test bitcoin alone and the nightly would be + # two thirds blind. A dispatch keeps exactly its old behaviour, one chosen + # coin; coin 'all' reuses the identical three-coin literal. The e2e-shard + # and e2e matrices both read this one output. + - name: Resolve the coin list + id: coins + env: + COINS: ${{ github.event_name == 'schedule' && '["bitcoin","litecoin","dogecoin"]' || github.event.inputs.coin == 'all' && '["bitcoin","litecoin","dogecoin"]' || format('["{0}"]', github.event.inputs.coin || 'bitcoin') }} + run: | + echo "coins=$COINS" + echo "coins=$COINS" >> "$GITHUB_OUTPUT" + + - name: Clone the e2e suite at the ref under test + run: git clone --depth 1 --branch "$STACK_REF" https://github.com/XChain-Platform/xchain-e2e-test.git "$RUNNER_TEMP/e2e" + + - name: Plan the action-suite shards + id: plan + run: | + node scripts/e2e_shard_plan.js --e2e-dir "$RUNNER_TEMP/e2e" --shards "$E2E_SHARDS" \ + --suite "$SUITE" --weights scripts/e2e_shard_weights.json > "$RUNNER_TEMP/plan.json" + PLAN="$RUNNER_TEMP/plan.json" node -e ' + const p = JSON.parse(require("fs").readFileSync(process.env.PLAN, "utf8")) + console.log(`xchain-e2e-test ${p.version}: ${p.groups.length} groups over ${p.shard_count} shard(s)`) + if (p.unweighted && p.unweighted.length) console.log(`::warning::no measured weight for ${p.unweighted.join(", ")}; scheduled at the default`) + for (const n of p.shards) { + const s = p.plan[n] + console.log(`shard ${n}: ~${s.weight_s === null ? "?" : Math.round(s.weight_s / 60)} min, ${s.groups.length} groups${s.extras.length ? " + " + s.extras.join(" + ") : ""}: ${s.groups.join(" ")}`) + }' + echo "plan=$(cat "$RUNNER_TEMP/plan.json")" >> "$GITHUB_OUTPUT" + echo "shards=$(PLAN="$RUNNER_TEMP/plan.json" node -p 'JSON.stringify(JSON.parse(require("fs").readFileSync(process.env.PLAN, "utf8")).shards)')" >> "$GITHUB_OUTPUT" + + e2e-shard: + needs: plan runs-on: ubuntu-latest - # WHY THE COIN LIST IS EVENT-DEPENDENT. A scheduled run carries NO inputs, so - # without this it would silently test bitcoin alone and the nightly would be - # two thirds blind: the litecoin and dogecoin legs are where the chain-specific - # breakage actually lands. A dispatch keeps exactly its old behaviour, one - # chosen coin, because a subset proves a fix and must not pretend to be a train. - # The three legs are independent stacks, so fail-fast would throw away two - # answers to report one; a release needs all three verdicts, not the first. - # - # coin: 'all' is the third case: a dispatch names a ref a schedule can never - # name, and needs to cover all three coins the way a schedule does. It reuses - # the identical three-coin literal so the expanded matrix is byte-identical - # to the schedule case; anything else falls through to the single-coin shape, - # defaulting to bitcoin the same way an absent input always has. + # The three legs are independent stacks, and so are the shards within a leg, + # so fail-fast would throw away answers; a release needs every verdict. strategy: fail-fast: false matrix: - coin: ${{ fromJSON(github.event_name == 'schedule' && '["bitcoin","litecoin","dogecoin"]' || github.event.inputs.coin == 'all' && '["bitcoin","litecoin","dogecoin"]' || format('["{0}"]', github.event.inputs.coin || 'bitcoin')) }} - # A BTC full action suite alone runs ~1h50m of wall clock, and the security - # and performance suites are sequenced AFTER it, so at 120 the two of them - # shared whatever minutes the action suite happened to leave - usually none. - # They had never once executed, and no amount of fixing the action suite - # could have changed that: the budget, not the tests, was the binding - # constraint. - # - # 360 is the platform ceiling for a hosted job, i.e. this sets no limit the - # runner would not impose anyway. That is deliberate: the two suites behind - # the action pass have never run, so there is no measured duration to size - # headroom against, and guessing low would re-create the same invisible - # truncation one tier further along. The cost is that a WEDGED run (a hung - # container, a stalled indexer poll) now takes the full 6 hours to report - # instead of failing earlier - so if that starts happening, add a - # `timeout-minutes` to the individual suite steps rather than clawing this - # number back down; a per-step bound catches a hang without also capping a - # legitimately long pass. - timeout-minutes: 360 + coin: ${{ fromJSON(needs.plan.outputs.coins) }} + shard: ${{ fromJSON(needs.plan.outputs.shards) }} + # 360 was the unsharded ceiling, set because a whole coin pass had no + # measured duration to size against. A shard does: ~42 minutes of action + # plus ~15 of setup and boot on bitcoin, somewhat more on the two-stack + # legs. 240 is four times that, so a slow leg is never truncated, while a + # WEDGED shard (a hung container, a stalled indexer poll) reports in four + # hours instead of six. + timeout-minutes: 240 env: COIN: ${{ matrix.coin }} + SHARD: ${{ matrix.shard }} + PLAN_JSON: ${{ needs.plan.outputs.plan }} # The anchor-reward attestation barrier holds each block until the hub-mirror # stream watermark is 120s past that block's own timestamp. On a shared ledger # that is free, because blocks are ten minutes apart and the watermark is long @@ -521,21 +590,52 @@ jobs: node src/index.js install "$STACK_REF" all bitcoin regtest fi - - name: Run the e2e action suite - # Now exits non-zero on failure (cli.js e2etest propagates the suite's - # exit code), so this step natively gates the job. - run: node src/index.js e2etest "$COIN" $SUITE --ref "$STACK_REF" + - name: Select this shard's action suites + # The plan job owns the split; this only reads this shard's entry, so the + # pattern run here is the one the aggregate later checks it against. + run: | + PATTERN=$(node -p 'JSON.parse(process.env.PLAN_JSON).plan[process.env.SHARD].pattern') + GROUPS_LINE=$(node -p 'JSON.parse(process.env.PLAN_JSON).plan[process.env.SHARD].groups.join(" ")') + echo "shard $SHARD of $(node -p 'JSON.parse(process.env.PLAN_JSON).shard_count'): $GROUPS_LINE" + echo "SHARD_PATTERN=$PATTERN" >> "$GITHUB_ENV" + + - name: Run the e2e action suite (this shard's groups) + # The pattern is e2etest's testName argument: runE2ETest makes it + # test/actions/.test.js and mocha brace-expands it (see + # scripts/e2e_shard_plan.js). Quoted so the shell never expands it. + # The exit code and the log path are kept for the shard status file, + # which is what the aggregate reads; the step still exits non-zero on a + # failure, so a red shard is red in the run list too. + run: | + set +e + node src/index.js e2etest "$COIN" "$SHARD_PATTERN" --ref "$STACK_REF" 2>&1 | tee "$RUNNER_TEMP/action.out" + rc=${PIPESTATUS[0]} + echo "$rc" > "$RUNNER_TEMP/action.exit" + exit "$rc" - name: Run the e2e Security suite (test/security) - if: env.SUITE == '' - # Stack-dependent on-chain security suite (VM sandbox-escape / gas-bomb / - # deploy-reject). Driven via the e2etest --script option. - run: node src/index.js e2etest "$COIN" --script test:security --ref "$STACK_REF" + # Shard 1 only, after its action subset: the same "after an action pass" + # order the unsharded job ran it in, and the plan loads shard 1 lighter + # to pay for it. Stack-dependent on-chain security suite (VM + # sandbox-escape / gas-bomb / deploy-reject). Driven via the e2etest + # --script option. + if: env.SUITE == '' && matrix.shard == 1 + run: | + set +e + node src/index.js e2etest "$COIN" --script test:security --ref "$STACK_REF" 2>&1 | tee "$RUNNER_TEMP/security.out" + rc=${PIPESTATUS[0]} + echo "$rc" > "$RUNNER_TEMP/security.exit" + exit "$rc" - name: Run the e2e Performance suite (test/perf) - if: env.SUITE == '' + if: env.SUITE == '' && matrix.shard == 1 # Live-stack per-service latency budgets (regression ceilings). - run: node src/index.js e2etest "$COIN" --script test:perf:budget --ref "$STACK_REF" + run: | + set +e + node src/index.js e2etest "$COIN" --script test:perf:budget --ref "$STACK_REF" 2>&1 | tee "$RUNNER_TEMP/performance.out" + rc=${PIPESTATUS[0]} + echo "$rc" > "$RUNNER_TEMP/performance.exit" + exit "$rc" # What the stack itself said, which is the one thing a failure here has never # carried off the runner. The suite log records that a service answered @@ -614,10 +714,87 @@ jobs: echo "captured $(ls -1 "$OUT" | wc -l) diagnostic files" + # What this shard ran and how each suite exited, for the aggregate. Runs on + # every outcome: a shard that never reached a suite still records that, and + # the aggregate reads the absence as a failure rather than as a skip. + - name: Record the shard status + if: always() + run: | + OUT="$XCHAIN_NODE_DATA_DIR/e2e-logs" + mkdir -p "$OUT" + OUT="$OUT" node -e ' + const fs = require("fs"), path = require("path"), tmp = process.env.RUNNER_TEMP + const read = f => { try { return fs.readFileSync(path.join(tmp, f), "utf8") } catch { return null } } + const logOf = leg => { + const out = read(leg + ".out") + const saved = out ? [...out.matchAll(/^Logs saved to: (.+)$/gm)].pop() : null + return saved ? path.basename(saved[1].trim()) : null + } + const exitOf = leg => { const e = read(leg + ".exit"); return e === null ? null : e.trim() } + const status = { coin: process.env.COIN, shard: Number(process.env.SHARD), pattern: process.env.SHARD_PATTERN || null, stack_ref: process.env.STACK_REF } + for (const leg of ["action", "security", "performance"]) { status[leg + "_exit"] = exitOf(leg); status[leg + "_log"] = logOf(leg) } + fs.writeFileSync(path.join(process.env.OUT, "shard-status.json"), JSON.stringify(status, null, 2) + "\n") + console.log(JSON.stringify(status, null, 2))' + + - name: Upload the shard logs + if: always() + uses: actions/upload-artifact@v4 + with: + name: e2e-shard-${{ env.COIN }}-${{ matrix.shard }} + path: ${{ env.XCHAIN_NODE_DATA_DIR }}/e2e-logs/ + if-no-files-found: ignore + + # The per-coin verdict. Named `e2e ()` and uploading e2e-logs-, the + # job and artifact names the unsharded workflow had, so the cut kit's + # preflight (job names) and judge-matrix (artifact layout) need no change. + # + # `needs` cannot name one coin's slice of a matrix, so every aggregate waits + # for all shards; the wall clock is the slowest shard either way. It runs even + # when a shard failed (!cancelled()), because the verdict is per coin: it + # downloads only its own coin's shards and fails when any of them is missing, + # red, or ran no tests, so a bitcoin shard failure fails e2e (bitcoin) and + # leaves the other two coins' verdicts standing. + e2e: + needs: [plan, e2e-shard] + if: ${{ !cancelled() && needs.plan.result == 'success' }} + runs-on: ubuntu-latest + timeout-minutes: 20 + strategy: + fail-fast: false + matrix: + coin: ${{ fromJSON(needs.plan.outputs.coins) }} + env: + COIN: ${{ matrix.coin }} + PLAN_JSON: ${{ needs.plan.outputs.plan }} + steps: + - uses: actions/checkout@v4 + with: + sparse-checkout: scripts + + - uses: actions/setup-node@v4 + with: + node-version: '22' + + # A shard that never uploaded simply has no artifact here; the merge names + # it as missing, so a failed download is not allowed to stop the job first. + - name: Download this coin's shard logs + continue-on-error: true + uses: actions/download-artifact@v4 + with: + pattern: e2e-shard-${{ matrix.coin }}-* + path: ${{ runner.temp }}/shards + + - name: Verify every shard and merge the logs + run: | + printf '%s' "$PLAN_JSON" > "$RUNNER_TEMP/plan.json" + mkdir -p "$RUNNER_TEMP/shards" + node scripts/e2e_shard_merge.js --coin "$COIN" --plan "$RUNNER_TEMP/plan.json" \ + --shards-dir "$RUNNER_TEMP/shards" --out "$RUNNER_TEMP/e2e-logs-$COIN" + - name: Upload e2e logs if: always() uses: actions/upload-artifact@v4 with: name: e2e-logs-${{ env.COIN }} - path: ${{ env.XCHAIN_NODE_DATA_DIR }}/e2e-logs/ + path: ${{ runner.temp }}/e2e-logs-${{ env.COIN }}/ if-no-files-found: ignore diff --git a/scripts/e2e_shard_merge.js b/scripts/e2e_shard_merge.js new file mode 100644 index 0000000..b688c49 --- /dev/null +++ b/scripts/e2e_shard_merge.js @@ -0,0 +1,201 @@ +#!/usr/bin/env node +'use strict' + +// Copyright © 2025–2026 Dankest, LLC +// Based on XChain Platform by Dankest, LLC – https://dankest.llc +// +// SPDX-License-Identifier: AGPL-3.0-or-later +// +// This file is part of XChain Platform. Licensed under the GNU Affero +// General Public License v3.0 or later; see LICENSE.md. A commercial +// license (without AGPL source-disclosure terms) is available - +// contact legal@dankest.llc. + +// Folds one coin's e2e-shard-- artifacts back into e2e-logs-, +// refusing unless every shard in the plan ran clean. + +// The release matrix judge walks every *.log, identifies a leg by npm banner +// `> xchain-e2e-test@ test|test:security|test:perf:budget`, rejects duplicates, +// and takes the LAST passing/pending/failing summary line. + +// Artifact structure: -regtest-action-merged.log with merged summary, security +// and perf logs from shard 1, shards//* with .log renamed .log.txt, and shard-merge.json. + +// Shards run via `e2etest ` (npx mocha, no npm banner), so this +// script writes the banner on the merged log after all shards are proven clean. + +const fs = require('fs') +const path = require('path') + +const LEG_BANNER = /^>\s+xchain-e2e-test@\S+\s+(test|test:security|test:perf:budget)\s*$/m +const LEG_SCRIPT = { security: 'test:security', performance: 'test:perf:budget' } + +function stripAnsi(value) { + return value.replace(/\x1B(?:[@-Z\\-_]|\[[0-?]*[ -/]*[@-~])/g, '') +} + +// Same rule judge-matrix.js countSummary applies: last line of each kind wins. +function countSummary(text) { + const counts = { passing: null, pending: 0, failing: 0 } + for (const line of stripAnsi(text).split(/\r?\n/)) { + const match = line.match(/^\s*([\d,]+)\s+(passing|pending|failing)\b/) + if (match) counts[match[2]] = Number(match[1].replaceAll(',', '')) + } + return counts +} + +function parseArgs(argv) { + const args = {} + for (let i = 0; i < argv.length; i += 2) { + const key = argv[i] + if (!['--coin', '--plan', '--shards-dir', '--out'].includes(key)) throw new Error(`unknown argument ${key}`) + args[key.slice(2)] = argv[i + 1] + } + for (const key of ['coin', 'plan', 'shards-dir', 'out']) if (!args[key]) throw new Error(`--${key} is required`) + return args +} + +function readJson(file) { + return JSON.parse(fs.readFileSync(file, 'utf8')) +} + +function copyTree(from, to, rename) { + fs.mkdirSync(to, { recursive: true }) + for (const entry of fs.readdirSync(from, { withFileTypes: true })) { + const source = path.join(from, entry.name) + if (entry.isDirectory()) copyTree(source, path.join(to, entry.name), rename) + else if (entry.isFile()) fs.copyFileSync(source, path.join(to, rename(entry.name))) + } +} + +// One shard's verdict. Every reason names the shard, so a red aggregate says +// which runner to open. +function checkShard({ coin, n, entry, dir }) { + const reasons = [] + const tag = `shard ${n}` + if (!fs.existsSync(dir)) return { reasons: [`${tag}: no e2e-shard-${coin}-${n} artifact (the shard job did not reach its upload)`] } + const statusFile = path.join(dir, 'shard-status.json') + if (!fs.existsSync(statusFile)) return { reasons: [`${tag}: artifact has no shard-status.json`] } + const status = readJson(statusFile) + + if (status.coin !== coin) reasons.push(`${tag}: status names coin ${status.coin}`) + if (Number(status.shard) !== n) reasons.push(`${tag}: status names shard ${status.shard}`) + if (status.pattern !== entry.pattern) reasons.push(`${tag}: ran pattern ${JSON.stringify(status.pattern)}, plan says ${JSON.stringify(entry.pattern)}`) + if (String(status.action_exit) !== '0') reasons.push(`${tag}: action suite exit code ${status.action_exit || 'missing'}`) + + const legs = {} + const actionPath = status.action_log ? path.join(dir, status.action_log) : null + if (!actionPath || !fs.existsSync(actionPath)) { + reasons.push(`${tag}: action log ${status.action_log || '(none recorded)'} is missing`) + } else { + const text = fs.readFileSync(actionPath, 'utf8') + const counts = countSummary(text) + if (counts.passing === null) reasons.push(`${tag}: action log has no Mocha passing summary`) + else if (counts.passing + counts.failing === 0) reasons.push(`${tag}: action shard ran no tests`) + if (counts.failing > 0) reasons.push(`${tag}: action shard has ${counts.failing} failing test(s)`) + legs.action = { file: actionPath, text, counts } + } + + for (const leg of entry.extras || []) { + const exit = status[`${leg}_exit`] + const file = status[`${leg}_log`] + if (String(exit) !== '0') reasons.push(`${tag}: ${leg} suite exit code ${exit || 'missing'}`) + const full = file ? path.join(dir, file) : null + if (!full || !fs.existsSync(full)) { + reasons.push(`${tag}: ${leg} log ${file || '(none recorded)'} is missing`) + continue + } + const text = fs.readFileSync(full, 'utf8') + const banner = text.match(LEG_BANNER) + if (!banner || banner[1] !== LEG_SCRIPT[leg]) reasons.push(`${tag}: ${leg} log carries no ${LEG_SCRIPT[leg]} banner`) + const counts = countSummary(text) + if (counts.passing === null || counts.passing + counts.failing === 0) reasons.push(`${tag}: ${leg} suite ran no tests`) + if (counts.failing > 0) reasons.push(`${tag}: ${leg} suite has ${counts.failing} failing test(s)`) + legs[leg] = { file: full, counts } + } + return { reasons, legs } +} + +function merge({ coin, plan, shardsDir, out }) { + const reasons = [] + const shardDir = n => path.join(shardsDir, `e2e-shard-${coin}-${n}`) + const results = [] + for (const n of plan.shards) { + const entry = plan.plan[n] + if (!entry) { + reasons.push(`shard ${n}: absent from the plan`) + continue + } + const result = checkShard({ coin, n, entry, dir: shardDir(n) }) + reasons.push(...result.reasons) + results.push({ n, entry, legs: result.legs || {} }) + } + + fs.mkdirSync(out, { recursive: true }) + const summary = { coin, shard_count: plan.shard_count, suite: plan.suite, shards: [], reasons } + for (const { n, entry, legs } of results) { + summary.shards.push({ shard: n, groups: entry.groups, action: legs.action ? legs.action.counts : null }) + if (fs.existsSync(shardDir(n))) copyTree(shardDir(n), path.join(out, 'shards', String(n)), name => name.replace(/\.log$/, '.log.txt')) + } + + if (reasons.length === 0) { + const totals = { passing: 0, pending: 0, failing: 0 } + const body = [] + for (const { n, entry, legs } of results) { + for (const key of Object.keys(totals)) totals[key] += legs.action.counts[key] + body.push(`=== shard ${n}/${plan.shard_count}: ${entry.groups.join(' ')} ===`, legs.action.text) + } + // A single-suite dispatch was never a full action pass, and the old job + // gave it no npm banner at all, so the judge must not find one here. + const banner = plan.suite + ? `> xchain-e2e-test@${plan.version} test [${plan.suite} only]` + : `> xchain-e2e-test@${plan.version} test` + const lines = [ + '', + banner, + `> mocha --timeout 0 --exit --require ./test/initialCheck.test.js 'test/actions/**/*.test.js' (sharded ${plan.shard_count} ways by nightly-e2e.yml; per-shard output follows)`, + '', + ...body, + `=== merged summary over ${plan.shard_count} shard(s) ===`, + ` ${totals.passing} passing`, + ` ${totals.pending} pending`, + ` ${totals.failing} failing`, + '', + ] + fs.writeFileSync(path.join(out, `${coin}-regtest-action-merged.log`), lines.join('\n')) + summary.action = totals + for (const { legs } of results) { + for (const leg of Object.keys(LEG_SCRIPT)) { + if (legs[leg]) fs.copyFileSync(legs[leg].file, path.join(out, path.basename(legs[leg].file))) + } + } + } + fs.writeFileSync(path.join(out, 'shard-merge.json'), JSON.stringify(summary, null, 2) + '\n') + return summary +} + +function main() { + const args = parseArgs(process.argv.slice(2)) + const summary = merge({ coin: args.coin, plan: readJson(args.plan), shardsDir: args['shards-dir'], out: args.out }) + for (const s of summary.shards) { + const c = s.action + process.stdout.write(`${args.coin} shard ${s.shard}: ${c ? `${c.passing} passing, ${c.pending} pending, ${c.failing} failing` : 'no action result'} (${s.groups.length} groups)\n`) + } + if (summary.reasons.length > 0) { + for (const reason of summary.reasons) process.stdout.write(`::error::${args.coin} ${reason}\n`) + process.exit(1) + } + const t = summary.action + process.stdout.write(`${args.coin} merged action: ${t.passing} passing, ${t.pending} pending, ${t.failing} failing over ${summary.shard_count} shard(s)\n`) +} + +if (require.main === module) { + try { + main() + } catch (error) { + process.stdout.write(`::error::e2e_shard_merge: ${error.message}\n`) + process.exit(1) + } +} + +module.exports = { merge, countSummary } diff --git a/scripts/e2e_shard_plan.js b/scripts/e2e_shard_plan.js new file mode 100644 index 0000000..628c6fb --- /dev/null +++ b/scripts/e2e_shard_plan.js @@ -0,0 +1,147 @@ +#!/usr/bin/env node +'use strict' + +// Copyright © 2025–2026 Dankest, LLC +// Based on XChain Platform by Dankest, LLC – https://dankest.llc +// +// SPDX-License-Identifier: AGPL-3.0-or-later +// +// This file is part of XChain Platform. Licensed under the GNU Affero +// General Public License v3.0 or later; see LICENSE.md. A commercial +// license (without AGPL source-disclosure terms) is available - +// contact legal@dankest.llc. + +// Splits xchain-e2e-test's action suite into N shards for the nightly-e2e +// workflow, so one coin's ~3h15m action pass runs as N parallel stacks instead of one. + +// A shard's unit is a GROUP: one test/actions/.test.js plus every +// test/actions/.test/**/*.test.js beside it. Part files share state +// via modules like attestation.test/support/shared.js, so a group never splits. + +// Each shard is expressed as the testName argument `e2etest` takes: runE2ETest +// builds `test/actions/${testName}.test.js` and hands it to mocha. The mocha +// glob and locale sort match the full `test/actions/**/*.test.js` run unchanged. + +// No change needed to xchain-node's CLI or e2e repo code: both are installed at +// the ref under test, not at this workflow's. + +// Assignment is longest-processing-time-first over weights (see e2e_shard_weights.json). +// The plan is computed ONCE per run by the plan job; every shard and aggregate reads +// that one plan, so a mid-run branch change cannot create shard disagreement. + +const fs = require('fs') +const path = require('path') + +const GROUP_NAME = /^[A-Za-z0-9_-]+$/ + +function parseArgs(argv) { + const args = { e2eDir: null, shards: null, suite: '', weights: null } + for (let i = 0; i < argv.length; i += 1) { + const flag = argv[i] + const value = argv[i + 1] + if (flag === '--e2e-dir') args.e2eDir = value + else if (flag === '--shards') args.shards = Number(value) + else if (flag === '--suite') args.suite = value || '' + else if (flag === '--weights') args.weights = value + else throw new Error(`unknown argument ${flag}`) + i += 1 + } + if (!args.e2eDir) throw new Error('--e2e-dir is required') + if (!args.weights) throw new Error('--weights is required') + if (!Number.isInteger(args.shards) || args.shards < 1) throw new Error('--shards must be a positive integer') + return args +} + +function listActionFiles(actionsDir) { + const files = [] + const pending = [actionsDir] + while (pending.length > 0) { + const current = pending.pop() + for (const entry of fs.readdirSync(current, { withFileTypes: true })) { + const full = path.join(current, entry.name) + if (entry.isDirectory()) pending.push(full) + else if (entry.isFile() && entry.name.endsWith('.test.js')) files.push(path.relative(actionsDir, full).split(path.sep).join('/')) + } + } + return files.sort() +} + +function groupOf(file) { + return file.split('/')[0].replace(/\.test(\.js)?$/, '') +} + +// Both halves of a group go in the brace set, so every shard pattern carries a +// `**` and mocha takes the glob path even for a shard of one plain file. +function shardPattern(groups) { + return '{' + groups.flatMap(g => [g, `${g}.test/**/*`]).join(',') + '}' +} + +function buildPlan({ files, shards, suite, weights, version }) { + const groups = [...new Set(files.map(groupOf))].sort() + if (groups.length === 0) throw new Error('no action suites found under test/actions') + const bad = groups.filter(g => !GROUP_NAME.test(g)) + if (bad.length > 0) throw new Error(`group names outside ${GROUP_NAME}: ${bad.join(', ')}`) + + // A single-suite dispatch proves one fix; it keeps its old one-stack shape + // and its old argument, verbatim, and earns no security or performance leg. + if (suite) { + return { + version, suite, shard_count: 1, shards: [1], groups: [suite], + plan: { 1: { groups: [suite], pattern: suite, weight_s: null, extras: [] } }, + } + } + + if (shards > groups.length) throw new Error(`${shards} shards for ${groups.length} groups would leave a shard empty`) + + const table = weights.weights || {} + const weightOf = g => (Number.isFinite(table[g]) ? table[g] : weights.default_weight_s) + const unweighted = groups.filter(g => !Number.isFinite(table[g])) + + const bins = Array.from({ length: shards }, (_, i) => ({ shard: i + 1, groups: [], load: 0, extras: [] })) + // Shard 1 also runs the security and performance suites after its action + // subset, the same order the unsharded job ran them in, so it starts loaded. + bins[0].load = weights.security_performance_weight_s + bins[0].extras = ['security', 'performance'] + + const ordered = [...groups].sort((a, b) => (weightOf(b) - weightOf(a)) || a.localeCompare(b, 'en')) + for (const g of ordered) { + const bin = bins.reduce((min, b) => (b.load < min.load ? b : min), bins[0]) + bin.groups.push(g) + bin.load += weightOf(g) + } + + const plan = {} + for (const bin of bins) { + bin.groups.sort((a, b) => a.localeCompare(b, 'en')) + plan[bin.shard] = { groups: bin.groups, pattern: shardPattern(bin.groups), weight_s: bin.load, extras: bin.extras } + } + + // Coverage is asserted here rather than trusted to the loop: every group in + // exactly one shard, and no shard empty. + const seen = bins.flatMap(b => b.groups) + if (seen.length !== groups.length || new Set(seen).size !== groups.length) throw new Error('plan does not partition the groups') + if (bins.some(b => b.groups.length === 0)) throw new Error('plan left a shard empty') + + return { version, suite: '', shard_count: shards, shards: bins.map(b => b.shard), groups, unweighted, plan } +} + +function main() { + const args = parseArgs(process.argv.slice(2)) + const weights = JSON.parse(fs.readFileSync(args.weights, 'utf8')) + const actionsDir = path.join(args.e2eDir, 'test', 'actions') + const files = listActionFiles(actionsDir) + const pkg = JSON.parse(fs.readFileSync(path.join(args.e2eDir, 'package.json'), 'utf8')) + const result = buildPlan({ files, shards: args.shards, suite: args.suite, weights, version: String(pkg.version || 'unknown') }) + process.stdout.write(JSON.stringify(result) + '\n') +} + +if (require.main === module) { + try { + main() + } catch (error) { + process.stderr.write(`e2e_shard_plan: ${error.message}\n`) + process.exit(1) + } +} + +module.exports = { buildPlan, listActionFiles, groupOf, shardPattern } diff --git a/scripts/e2e_shard_weights.json b/scripts/e2e_shard_weights.json new file mode 100644 index 0000000..3ad9d07 --- /dev/null +++ b/scripts/e2e_shard_weights.json @@ -0,0 +1,88 @@ +{ + "_note": "Per-group action-suite wall time in seconds, measured on the bitcoin leg of run 35605562229 (test durations per top-level describe, scaled up to the 11626 s the action step took so hook time is spread proportionally). Groups that ran no timed test there carry a 30 s floor, or an estimate where the suite runs only on the validator-mode legs. Only the BALANCE of the shards depends on these numbers: a group missing here is still scheduled, at default_weight_s, so a new suite can never be dropped from the gate.", + "source_run": 35605562229, + "default_weight_s": 90, + "security_performance_weight_s": 420, + "weights": { + "address": 30, + "address_id_reorg_deterministic": 101, + "airdrop": 543, + "attestation": 62, + "attestation_request_cap": 99, + "attestation_widening": 30, + "batch": 41, + "batch_cost_weighting": 252, + "batch_escrowed_parent_replay": 101, + "batch_issuance_limits": 674, + "broadcast": 114, + "callback": 89, + "capability_slash": 30, + "chained_mint_reservation": 39, + "coinpay": 597, + "coinpay_reorg": 81, + "collect": 89, + "compaction_fix_live": 37, + "consensus_hash_conformance": 30, + "contract_stake_lifecycle": 190, + "contract_stake_reorg": 82, + "contract_staking": 219, + "controller_policy": 984, + "controller_reorg": 71, + "ctlseed": 30, + "delegate_rotation": 214, + "delegatedRewardSourceResolution": 51, + "destroy": 139, + "dex_reorg": 58, + "dispenser": 871, + "dividend": 181, + "envelope": 63, + "envelope_cancel": 30, + "envelope_fee_height": 38, + "envelope_mu_sig2": 30, + "envelope_reorg": 51, + "file": 30, + "gated_file": 151, + "gated_reorg": 89, + "issue": 239, + "issue_emission_fee_exempt": 30, + "link": 50, + "list": 126, + "message": 164, + "mint": 38, + "native_fee_dispenser": 51, + "native_fee_live": 38, + "negative": 150, + "nft_parity": 51, + "nft_reorg": 46, + "onboard_validator": 120, + "oracle_mirror": 30, + "order": 669, + "ownership": 382, + "price": 180, + "price_reorg": 34, + "real_url_attestation": 30, + "real_url_attestation_failures": 30, + "reorg_balances": 55, + "rollcall_eviction": 30, + "rollcall_gates": 30, + "rollcall_proof_barrier": 30, + "rollcall_publish_paths": 30, + "send": 181, + "sleep": 63, + "stake_reorg_deactivation": 36, + "staking": 271, + "swap": 324, + "sweep": 54, + "tick_id_reorg_deterministic": 101, + "vm": 38, + "vm_contract_custody": 299, + "vm_contract_reorg": 32, + "vm_contract_slash": 30, + "vm_contract_sweep": 63, + "vm_edge": 188, + "vm_emissions": 594, + "vm_execute_negative": 51, + "vm_extended": 412, + "xchainPriceDerivation": 277 + } +} diff --git a/test/unit/nightly_e2e_workflow.test/sharding.test.js b/test/unit/nightly_e2e_workflow.test/sharding.test.js new file mode 100644 index 0000000..b9b6a9e --- /dev/null +++ b/test/unit/nightly_e2e_workflow.test/sharding.test.js @@ -0,0 +1,227 @@ +'use strict' + +// Copyright © 2025–2026 Dankest, LLC +// Based on XChain Platform by Dankest, LLC – https://dankest.llc +// +// SPDX-License-Identifier: AGPL-3.0-or-later +// +// This file is part of XChain Platform. Licensed under the GNU Affero +// General Public License v3.0 or later; see LICENSE.md. A commercial +// license (without AGPL source-disclosure terms) is available - +// contact legal@dankest.llc. +// +// The action suite runs as parallel shards (scripts/e2e_shard_plan.js) that a +// per-coin aggregate folds back into the e2e-logs- artifact the release +// cut kit grades (scripts/e2e_shard_merge.js). Pinned here: the shard patterns +// select, through mocha's own glob, exactly the files the unsharded glob did; +// the aggregate keeps the job and artifact names the cut kit reads; and a +// missing, red or empty shard fails its coin. + +const { expect } = require('chai') +const fs = require('fs') +const os = require('os') +const path = require('path') +const yaml = require('js-yaml') +const lookupFiles = require('mocha/lib/cli/lookup-files') + +const { buildPlan, listActionFiles } = require('../../../scripts/e2e_shard_plan') +const { merge, countSummary } = require('../../../scripts/e2e_shard_merge') + +const WORKFLOW = path.join(__dirname, '../../../.github/workflows/nightly-e2e.yml') +const WEIGHTS = require('../../../scripts/e2e_shard_weights.json') + +const FIXTURE_FILES = [ + 'address.test.js', 'airdrop.test.js', 'airdrop.test/02_v0_balance_verification.test.js', + 'attestation.test.js', 'attestation.test/01_accepts.test.js', 'batch.test.js', 'coinpay.test.js', + 'controller_policy.test.js', 'dispenser.test.js', 'dispenser.test/08_v1_cancel.test.js', + 'order.test.js', 'order.test/02_v1_cancel.test.js', 'order.test/07_v2_edit.test.js', + 'send.test.js', 'sleep.test.js', 'brand_new_suite.test.js', +] + +function fixtureTree() { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'e2e-shard-plan-')) + for (const file of FIXTURE_FILES) { + const full = path.join(root, 'test/actions', file) + fs.mkdirSync(path.dirname(full), { recursive: true }) + fs.writeFileSync(full, '') + } + // A support module beside the part files is not a spec in either glob. + fs.mkdirSync(path.join(root, 'test/actions/attestation.test/support'), { recursive: true }) + fs.writeFileSync(path.join(root, 'test/actions/attestation.test/support/shared.js'), '') + return root +} + +function mochaFiles(root, spec) { + const cwd = process.cwd() + process.chdir(root) + try { + return [].concat(lookupFiles(spec, ['js'], false)) + } finally { + process.chdir(cwd) + } +} + +function registerPlanChecks() { + describe('the plan', function () { + let root, files, plan + + before(function () { + root = fixtureTree() + files = listActionFiles(path.join(root, 'test/actions')) + plan = buildPlan({ files, shards: 3, suite: '', weights: WEIGHTS, version: '9.9.9' }) + }) + + it('selects, through mocha lookupFiles, exactly the files the unsharded glob selects, each once', function () { + const full = mochaFiles(root, 'test/actions/**/*.test.js') + const union = plan.shards.flatMap(n => mochaFiles(root, `test/actions/${plan.plan[n].pattern}.test.js`)) + expect(union.slice().sort()).to.deep.equal(full.slice().sort()) + expect(new Set(union).size).to.equal(union.length) + }) + + it('keeps a suite and its .test/ part files on one shard', function () { + const owner = g => plan.shards.find(n => plan.plan[n].groups.includes(g)) + for (const g of ['airdrop', 'attestation', 'dispenser', 'order']) { + const shardFiles = mochaFiles(root, `test/actions/${plan.plan[owner(g)].pattern}.test.js`) + const groupFiles = FIXTURE_FILES.filter(f => f === `${g}.test.js` || f.startsWith(`${g}.test/`)) + for (const f of groupFiles) expect(shardFiles, g).to.include(`test/actions/${f}`) + } + }) + + it('schedules a suite with no measured weight instead of dropping it', function () { + expect(plan.unweighted).to.deep.equal(['brand_new_suite']) + expect(plan.shards.some(n => plan.plan[n].groups.includes('brand_new_suite'))).to.equal(true) + }) + + it('puts security and performance on shard 1 only', function () { + expect(plan.plan[1].extras).to.deep.equal(['security', 'performance']) + for (const n of plan.shards.slice(1)) expect(plan.plan[n].extras).to.deep.equal([]) + }) + + it('keeps a single-suite dispatch as one shard running its argument verbatim, with no extra legs', function () { + const single = buildPlan({ files, shards: 5, suite: 'order', weights: WEIGHTS, version: '9.9.9' }) + expect(single.shards).to.deep.equal([1]) + expect(single.plan[1]).to.include({ pattern: 'order' }) + expect(single.plan[1].extras).to.deep.equal([]) + }) + + it('refuses more shards than groups rather than leaving one empty', function () { + expect(() => buildPlan({ files, shards: 99, suite: '', weights: WEIGHTS, version: '9.9.9' })).to.throw(/empty/) + }) + }) +} + +function registerWorkflowChecks() { + describe('the workflow', function () { + let doc + + before(function () { doc = yaml.load(fs.readFileSync(WORKFLOW, 'utf8')) }) + + it('keeps the aggregate job id `e2e` with the coin as its only matrix key, so it renders as `e2e ()`', function () { + expect(Object.keys(doc.jobs.e2e.strategy.matrix)).to.deep.equal(['coin']) + expect(doc.jobs.e2e.name).to.equal(undefined) + expect(doc.jobs['e2e-shard'].strategy.matrix).to.have.keys('coin', 'shard') + }) + + it('runs the aggregate after the shards even when one failed, and fails it per coin', function () { + expect(doc.jobs.e2e.needs).to.include('e2e-shard') + expect(doc.jobs.e2e.if).to.match(/!cancelled\(\)/) + const upload = doc.jobs.e2e.steps.find(s => s.uses && s.uses.startsWith('actions/upload-artifact')) + expect(upload.with.name).to.equal('e2e-logs-${{ env.COIN }}') + const shardUpload = doc.jobs['e2e-shard'].steps.find(s => s.uses && s.uses.startsWith('actions/upload-artifact')) + expect(shardUpload.with.name).to.equal('e2e-shard-${{ env.COIN }}-${{ matrix.shard }}') + }) + + it('expands the coin list the way the unsharded matrix did: schedule and all are three coins, a dispatch is one', function () { + const expr = doc.jobs.plan.steps.find(s => s.id === 'coins').env.COINS + const evaluate = (event, coin) => { + const body = expr.replace(/^\$\{\{\s*|\s*\}\}$/g, '') + .replace(/github\.event_name/g, JSON.stringify(event)) + .replace(/github\.event\.inputs\.coin/g, JSON.stringify(coin)) + .replace(/format\('\["\{0\}"\]', ([^)]*)\)/, (_, arg) => `('["' + (${arg}) + '"]')`) + return JSON.parse(Function(`return (${body})`)()) + } + const three = ['bitcoin', 'litecoin', 'dogecoin'] + expect(evaluate('schedule', '')).to.deep.equal(three) + expect(evaluate('workflow_dispatch', 'all')).to.deep.equal(three) + expect(evaluate('workflow_dispatch', 'dogecoin')).to.deep.equal(['dogecoin']) + expect(evaluate('workflow_dispatch', '')).to.deep.equal(['bitcoin']) + }) + + it('runs security and performance on shard 1 of a full pass only', function () { + for (const prefix of ['Run the e2e Security suite', 'Run the e2e Performance suite']) { + const step = doc.jobs['e2e-shard'].steps.find(s => (s.name || '').startsWith(prefix)) + expect(step.if).to.equal("env.SUITE == '' && matrix.shard == 1") + } + }) + }) +} + +const MERGE_COIN = 'bitcoin' +const MERGE_PLAN = { + version: '9.9.9', suite: '', shard_count: 2, shards: [1, 2], + plan: { + 1: { groups: ['address'], pattern: '{address,address.test/**/*}', extras: ['security', 'performance'] }, + 2: { groups: ['send'], pattern: '{send,send.test/**/*}', extras: [] }, + }, +} + +function writeShard(dir, n, { actionExit = '0', passing = 5, failing = 0 } = {}) { + const shard = path.join(dir, `e2e-shard-${MERGE_COIN}-${n}`) + fs.mkdirSync(path.join(shard, 'diagnostics'), { recursive: true }) + fs.writeFileSync(path.join(shard, 'diagnostics', 'docker-ps.txt'), 'ps\n') + const summary = `\n ${passing} passing (1m)\n` + (failing ? ` ${failing} failing\n` : '') + fs.writeFileSync(path.join(shard, `a${n}.log`), `\u001b[0m SUITE ${n}\n ✔ works (10ms)\n${summary}`) + const status = { coin: MERGE_COIN, shard: n, pattern: MERGE_PLAN.plan[n].pattern, action_exit: actionExit, action_log: `a${n}.log` } + if (n === 1) { + fs.writeFileSync(path.join(shard, 's.log'), '\n> xchain-e2e-test@9.9.9 test:security\n> mocha\n\n 13 passing (6m)\n') + fs.writeFileSync(path.join(shard, 'p.log'), '\n> xchain-e2e-test@9.9.9 test:perf:budget\n> mocha\n\n 8 passing (20s)\n') + Object.assign(status, { security_exit: '0', security_log: 's.log', performance_exit: '0', performance_log: 'p.log' }) + } + fs.writeFileSync(path.join(shard, 'shard-status.json'), JSON.stringify(status)) +} + +function runMerge(setup) { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'e2e-shard-merge-')) + setup(dir) + const out = path.join(dir, 'out') + return { summary: merge({ coin: MERGE_COIN, plan: MERGE_PLAN, shardsDir: dir, out }), out } +} + +function registerMergeChecks() { + describe('the merge', function () { + it('writes ONE bannered action log whose last summary is the sum, and no second leg log anywhere', function () { + const { summary, out } = runMerge(dir => { writeShard(dir, 1, { passing: 3 }); writeShard(dir, 2, { passing: 4 }) }) + expect(summary.reasons).to.deep.equal([]) + const merged = fs.readFileSync(path.join(out, 'bitcoin-regtest-action-merged.log'), 'utf8') + expect(merged).to.match(/^>\s+xchain-e2e-test@9\.9\.9\s+test\s*$/m) + expect(countSummary(merged)).to.deep.equal({ passing: 7, pending: 0, failing: 0 }) + const logs = [] + const walk = d => fs.readdirSync(d, { withFileTypes: true }).forEach(e => (e.isDirectory() ? walk(path.join(d, e.name)) : e.name.endsWith('.log') && logs.push(e.name))) + walk(out) + expect(logs.sort()).to.deep.equal(['bitcoin-regtest-action-merged.log', 'p.log', 's.log']) + expect(fs.existsSync(path.join(out, 'shards', '2', 'a2.log.txt'))).to.equal(true) + }) + + it('fails the coin when a shard never uploaded', function () { + const { summary } = runMerge(dir => writeShard(dir, 1)) + expect(summary.reasons.join('; ')).to.match(/shard 2: no e2e-shard-bitcoin-2 artifact/) + }) + + it('fails the coin on a red shard, and writes no merged action log', function () { + const { summary, out } = runMerge(dir => { writeShard(dir, 1); writeShard(dir, 2, { actionExit: '1', failing: 2 }) }) + expect(summary.reasons.join('; ')).to.match(/shard 2: action suite exit code 1/).and.match(/2 failing/) + expect(fs.existsSync(path.join(out, 'bitcoin-regtest-action-merged.log'))).to.equal(false) + }) + + it('fails the coin on a shard that ran no tests', function () { + const { summary } = runMerge(dir => { writeShard(dir, 1); writeShard(dir, 2, { passing: 0 }) }) + expect(summary.reasons.join('; ')).to.match(/shard 2: action shard ran no tests/) + }) + }) +} + +describe('nightly-e2e.yml action-suite sharding', function () { + registerPlanChecks() + registerWorkflowChecks() + registerMergeChecks() +}) diff --git a/test/unit/nightly_e2e_workflow.test/two_stack_legs.test.js b/test/unit/nightly_e2e_workflow.test/two_stack_legs.test.js index 93ca875..0eb502e 100644 --- a/test/unit/nightly_e2e_workflow.test/two_stack_legs.test.js +++ b/test/unit/nightly_e2e_workflow.test/two_stack_legs.test.js @@ -37,7 +37,9 @@ const CHAIN_RAIL_DEFAULT_PORTS = { function loadSteps() { const doc = yaml.load(fs.readFileSync(WORKFLOW, 'utf8')) - const steps = doc.jobs.e2e.steps + // The stack steps live on the shard job since the action suite was sharded; + // `e2e` is now the per-coin aggregate, which boots nothing. + const steps = doc.jobs['e2e-shard'].steps const find = (prefix) => { const step = steps.find(s => typeof s.name === 'string' && s.name.startsWith(prefix)) if (!step) throw new Error('workflow step not found: ' + prefix) From c281a531ae9b3e0e6670ff65e5d90a0e76936953 Mon Sep 17 00:00:00 2001 From: J-Dog Date: Wed, 23 Sep 2026 21:17:01 -0700 Subject: [PATCH 35/35] release: pin the v0.20.1 manifest to the signed sibling tags --- src/release-manifest.json | 54 +++++++++++++++++++-------------------- 1 file changed, 27 insertions(+), 27 deletions(-) diff --git a/src/release-manifest.json b/src/release-manifest.json index bb23033..7ca8d0f 100644 --- a/src/release-manifest.json +++ b/src/release-manifest.json @@ -1,6 +1,6 @@ { "_comment": [ - "Pinned component set for XChain Platform v0.20.0.", + "Pinned component set for XChain Platform v0.20.1.", "Generated by bin/write-release-manifest.js from the ACTUAL tagged master merge commits.", "xchain-node is the carrier and is not listed: checking out its tag IS this manifest.", "A train tags only the repos it touches. This one moves xchain-vm, xchain-decoder, xchain-indexer, xchain-hub, xchain-sync, xchain-encoder, xchain-utxo-tracker, xchain-explorer, xchain-sdk, xchain-e2e-test and xchain-regtest-miner.", @@ -12,56 +12,56 @@ "repo's origin/master, and is a GPG-signed annotated tag verifying against the", "platform release key (fingerprint 1DA7C4896F56EA22CF491EDF4361611A82F90B70)." ], - "platform_version": "0.20.0", - "released": "2026-09-18", + "platform_version": "0.20.1", + "released": "2026-09-24", "components": { "xchain-vm": { - "tag": "v0.20.0", - "commit": "c28d3338318ba0494eebe4203432dfe78b127f2a" + "tag": "v0.20.1", + "commit": "e7dbc3e3a00a26ebdfb22941bac4e284cca180d7" }, "xchain-decoder": { - "tag": "v0.20.0", - "commit": "0b07c59464e3631db5d0d775f5aa9c550bbe7b0c" + "tag": "v0.20.1", + "commit": "6d49ed2c583b444ad127e8bb5800cc589631590f" }, "xchain-indexer": { - "tag": "v0.20.0", - "commit": "777ef8604b400f398a05bdb5b0da4e2242896b98" + "tag": "v0.20.1", + "commit": "10ae153b0bb197bc05903cd7e504cb559104c20d" }, "xchain-hub": { - "tag": "v0.20.0", - "commit": "0f7c78ac6fb9c139155d8df018e9668e70ca63f3" + "tag": "v0.20.1", + "commit": "933fba792e1ab71d22219301c3b43bc1c85f3433" }, "xchain-sync": { - "tag": "v0.20.0", - "commit": "5c03236b2b77d6b49b5282cc1b4d6ebbe1913efe" + "tag": "v0.20.1", + "commit": "d2f37e168aee2fa122ad3ae950965dee17901105" }, "xchain-encoder": { - "tag": "v0.20.0", - "commit": "830d937c0c2cbbbe4d5cb6ce2b11d308e30a16ca" + "tag": "v0.20.1", + "commit": "0f5b08f4ceb2f8c37e2f3f13c7cc30dd98b82d82" }, "xchain-utxo-tracker": { - "tag": "v0.20.0", - "commit": "a14eea102792338ee6d02f3afeea4719981936f4" + "tag": "v0.20.1", + "commit": "dee5717e4b03176bb9b432391a39c762b50c3c18" }, "xchain-explorer": { - "tag": "v0.20.0", - "commit": "6daa89b0ef67315d52328b8b1bc4416139f8c3d5" + "tag": "v0.20.1", + "commit": "1f1eb1b50af6815c8a795326f88910f892da359f" }, "xchain-sdk": { - "tag": "v0.20.0", - "commit": "340ff29550d8b24f2360b0eec97eb3c5a64b3701" + "tag": "v0.20.1", + "commit": "fc192852f0b86a489b45689eb940f3e710ae4e98" }, "xchain-e2e-test": { - "tag": "v0.20.0", - "commit": "469bd3e652ae9c715ac58d4a28bb9e19cf3abd8d" + "tag": "v0.20.1", + "commit": "2da7650c0098e34962f48cf663dd8ae491bc504d" }, "xchain-contracts": { "tag": "v0.17.0", "commit": "684a6311be7232542346c984bd9f2d62f3251810" }, "xchain-regtest-miner": { - "tag": "v0.20.0", - "commit": "9426829225b19844f2d5de434dd2bed3605fcb20" + "tag": "v0.20.1", + "commit": "8e818394d854d796951f8959c840405c4d5ccaf6" } }, "trainActivation": { @@ -70,12 +70,12 @@ "heights": { "mainnet": 9999999999, "regtest": 0, - "testnet": 153116 + "testnet": 154074 }, "computedFromBtcTip": { "mainnet": 967625, "regtest": 0, - "testnet": 153018 + "testnet": 153698 } } }