From 0c8ea4b96329365230141d9823a3594455358576 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 15 Sep 2026 12:23:03 +0000 Subject: [PATCH 01/61] Split an indexer across processes when the connection budget affords it An isolated schema could already be driven by several processes, but only by hand: one `envio start --chain` per chain, each with its own port, its own logs and its own metrics endpoint, and nothing tying them together. A plain `envio start` now does it. When every entity is per-chain and ENVIO_PG_MAX_CONNECTIONS affords two connections per process, the run forks a worker per group of chains and supervises them, so the operator still sees one indexer: one metrics endpoint over the merged snapshot, one console, one progress display, one process to stop. BREAKING: ENVIO_PG_MAX_CONNECTIONS is now the budget for the whole run rather than the cap on one process's pool. A run that set it to 10 for throughput used to get a single process with ten connections; it now gets five workers with two each, which also gives each its own event loop and heap. Chains are placed round-robin over config order: what would balance them is how much work each has left, and that isn't known until they report their heights. The supervisor creates the schema for every chain and closes its own connection before forking, so the budget belongs entirely to the workers. Workers talk over the fork's own channel under structured-clone serialization, which a metrics snapshot's timestamps need. A worker exits when the channel closes, so a supervisor that dies can't leave chains running unwatched, and one worker ending in a way the supervisor didn't ask for takes the group down rather than leaving the run half indexed. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- packages/cli/CommandLineHelp.md | 2 +- packages/cli/src/cli_args/clap_definitions.rs | 3 + .../envio-tests/test/SupervisorFork_test.res | 80 +++++ .../envio-tests/test/helpers/fakeWorker.mjs | 28 ++ .../test/lib_tests/Metrics_test.res | 131 ++++++++ .../test/lib_tests/Supervisor_test.res | 154 +++++++++ packages/envio/src/Bin.res | 46 ++- packages/envio/src/Main.res | 77 +++-- packages/envio/src/Metrics.res | 100 ++++++ packages/envio/src/Supervisor.res | 292 ++++++++++++++++++ packages/envio/src/Worker.res | 53 ++++ packages/envio/src/bindings/NodeJs.res | 37 +++ 12 files changed, 965 insertions(+), 38 deletions(-) create mode 100644 packages/envio-tests/test/SupervisorFork_test.res create mode 100644 packages/envio-tests/test/helpers/fakeWorker.mjs create mode 100644 packages/envio-tests/test/lib_tests/Supervisor_test.res create mode 100644 packages/envio/src/Supervisor.res create mode 100644 packages/envio/src/Worker.res diff --git a/packages/cli/CommandLineHelp.md b/packages/cli/CommandLineHelp.md index 286a9ed22..959a7aba2 100644 --- a/packages/cli/CommandLineHelp.md +++ b/packages/cli/CommandLineHelp.md @@ -378,7 +378,7 @@ Start the indexer. Runs codegen automatically before launching so the on-disk ty ###### **Options:** * `-r`, `--restart` — Clear your database and restart indexing from scratch -* `--chain ` — Index only this chain, leaving the others to their own `envio start --chain` processes. Repeat the flag for several chains. Requires a schema whose entities are all per-chain, created for every chain by `envio local db-migrate up` before any process starts. Assign each configured chain to exactly one process, and give each its own `ENVIO_INDEXER_PORT`. Each process builds the indexes for its own chains and reports them ready as they catch up, independently of the others +* `--chain ` — Index only this chain, leaving the others to their own `envio start --chain` processes. Only needed to place the chains yourself: a plain `envio start` already splits them across processes when `ENVIO_PG_MAX_CONNECTIONS` affords two connections per process, and manages those processes for you. Repeat the flag for several chains. Requires a schema whose entities are all per-chain, created for every chain by `envio local db-migrate up` before any process starts. Assign each configured chain to exactly one process, and give each its own `ENVIO_INDEXER_PORT`. Each process builds the indexes for its own chains and reports them ready as they catch up, independently of the others diff --git a/packages/cli/src/cli_args/clap_definitions.rs b/packages/cli/src/cli_args/clap_definitions.rs index 260571f92..0f1e9c135 100644 --- a/packages/cli/src/cli_args/clap_definitions.rs +++ b/packages/cli/src/cli_args/clap_definitions.rs @@ -145,6 +145,9 @@ pub struct StartArgs { pub restart: bool, ///Index only this chain, leaving the others to their own `envio start --chain` processes. + ///Only needed to place the chains yourself: a plain `envio start` already splits them across + ///processes when `ENVIO_PG_MAX_CONNECTIONS` affords two connections per process, and manages + ///those processes for you. ///Repeat the flag for several chains. Requires a schema whose entities are all per-chain, ///created for every chain by `envio local db-migrate up` before any process starts. ///Assign each configured chain to exactly one process, and give each its own diff --git a/packages/envio-tests/test/SupervisorFork_test.res b/packages/envio-tests/test/SupervisorFork_test.res new file mode 100644 index 000000000..f44e51e3e --- /dev/null +++ b/packages/envio-tests/test/SupervisorFork_test.res @@ -0,0 +1,80 @@ +open Vitest + +// What the fixture worker reports back in place of a metrics snapshot: the +// narrowing and the environment its supervisor handed it. +type fixtureReport = { + isolatedChains: array, + maxConnections: string, + logFile: string, + startTime: Date.t, +} + +let fixturePath = `${NodeJs.Process.cwd()}/test/helpers/fakeWorker.mjs` + +let forkFixture = (~chainIds, ~maxConnections=2, ~workerIndex=0) => + Supervisor.fork( + {chainIds: chainIds->Array.map(ChainId.fromInt), maxConnections}, + ~workerIndex, + ~configJson=JSON.Object(Dict.fromArray([("name", JSON.String("indexer"))])), + ~entryPath=fixturePath, + ) + +describe("Supervisor.fork", () => { + Async.it("Hands a worker its chains, its budget share, and its own log file", async t => { + let running = forkFixture(~chainIds=[1, 137], ~maxConnections=3, ~workerIndex=1) + + let report = await Promise.make( + (resolve, _) => + running.child->NodeJs.ChildProcess.onMessage( + message => + switch message { + | Worker.Snapshot({metrics}) => + resolve(metrics->(Utils.magic: Metrics.t => fixtureReport)) + }, + ), + ) + running.child->NodeJs.ChildProcess.kill("SIGTERM")->ignore + + t.expect(report).toStrictEqual({ + isolatedChains: [1., 137.], + maxConnections: "3", + logFile: Supervisor.logFilePath(~workerIndex=1), + // Proof the channel clones rather than stringifies: a JSON round trip + // would have turned this into a string. + startTime: Date.fromTime(1700000000000.), + }) + }) +}) + +describe("Supervisor.awaitExit", () => { + Async.it("Returns once every worker has finished on its own", async t => { + NodeJs.Process.process.env->Dict.set("FAKE_WORKER", "succeed") + let group: Supervisor.group = { + running: [forkFixture(~chainIds=[1]), forkFixture(~chainIds=[137])], + stopping: false, + } + + let outcome = switch await group->Supervisor.awaitExit { + | () => "returned" + | exception _ => "threw" + } + + t.expect(outcome).toBe("returned") + }) + + Async.it("Stops the group and fails the run when one worker dies", async t => { + NodeJs.Process.process.env->Dict.set("FAKE_WORKER", "fail") + let failing = forkFixture(~chainIds=[1]) + NodeJs.Process.process.env->Dict.set("FAKE_WORKER", "linger") + let lingering = forkFixture(~chainIds=[137]) + let group: Supervisor.group = {running: [failing, lingering], stopping: false} + + let outcome = switch await group->Supervisor.awaitExit { + | () => "returned" + | exception _ => "threw" + } + + // The survivor was taken down rather than left indexing half a schema. + t.expect((outcome, group.stopping, lingering.settled)).toStrictEqual(("threw", true, true)) + }) +}) diff --git a/packages/envio-tests/test/helpers/fakeWorker.mjs b/packages/envio-tests/test/helpers/fakeWorker.mjs new file mode 100644 index 000000000..e6f7342bf --- /dev/null +++ b/packages/envio-tests/test/helpers/fakeWorker.mjs @@ -0,0 +1,28 @@ +// Stands in for a forked indexer process. `FAKE_WORKER` picks how it ends, so a +// supervisor's handling of a clean finish and of a failure can both be driven +// with real processes. +const mode = process.env.FAKE_WORKER ?? "report"; + +process.on("message", (message) => { + if (message.kind === "init") { + process.send({ + kind: "snapshot", + metrics: { + isolatedChains: message.config.isolatedChains, + maxConnections: process.env.ENVIO_PG_MAX_CONNECTIONS, + logFile: process.env.LOG_FILE, + // A Date survives only under structured-clone serialization, which is + // what a metrics snapshot's timestamps need. + startTime: new Date(1700000000000), + }, + }); + if (mode === "succeed") process.exit(0); + if (mode === "fail") process.exit(1); + } + if (message.kind === "syncCache") { + process.exit(0); + } +}); + +// Nothing else keeps a "linger" worker alive; it waits to be stopped. +if (mode === "linger") setInterval(() => {}, 1000); diff --git a/packages/envio-tests/test/lib_tests/Metrics_test.res b/packages/envio-tests/test/lib_tests/Metrics_test.res index 1b42babf7..f3c3abfdb 100644 --- a/packages/envio-tests/test/lib_tests/Metrics_test.res +++ b/packages/envio-tests/test/lib_tests/Metrics_test.res @@ -663,3 +663,134 @@ envio_indexing_contract_addresses{chainId="1",contract="NftFactory"} 2 ) }) }) + +describe("Metrics.merge", () => { + let startTime = Date.fromTime(1000.) + let metricTime = Date.fromTime(5000.) + + let handler = (~event, ~processingCount): Metrics.handlerMetrics => { + contract: "Token", + event, + processingSeconds: 1., + processingCount, + preloadSeconds: 0.5, + preloadCount: 2., + preloadSecondsTotal: 3., + } + + let effect = (~cacheCount): Metrics.effectMetrics => { + effect: "getMetadata", + scope: "crossChain", + callSeconds: 1., + callSecondsTotal: 2., + callCount: 3., + activeCallsCount: 1, + queueCount: 2, + queueWaitSeconds: 0.25, + invalidationsCount: 1., + cacheCount, + } + + it("Returns one worker's snapshot unchanged, taking the clock from the caller", t => { + let only: Metrics.t = { + ...baseMetrics, + startTime: Date.fromTime(777.), + metricTime: Date.fromTime(888.), + elapsedSeconds: 42., + processingSeconds: 1.5, + maxBatchSize: 5000, + chains: [TestChainMetrics.make(~progressBlockNumber=400, ~firstEventBlockNumber=Some(150))], + handlers: [handler(~event="Transfer", ~processingCount=4.)], + effects: [effect(~cacheCount=Some(7))], + } + + t.expect(Metrics.merge([only], ~startTime, ~metricTime, ~elapsedSeconds=9.)).toStrictEqual({ + ...only, + startTime, + metricTime, + elapsedSeconds: 9., + }) + }) + + it("Concatenates chain series, sums what shares a key, and folds the scalars", t => { + let chainOne = TestChainMetrics.make(~progressBlockNumber=400, ~firstEventBlockNumber=None) + let chainTwo = {...chainOne, Metrics.chainId: 137->ChainId.fromInt} + + let first: Metrics.t = { + ...baseMetrics, + targetBufferSize: 100, + maxBatchSize: 5000, + isInReorgThreshold: false, + rollbackEnabled: true, + processingSeconds: 1.5, + rollbackCount: 1, + chains: [chainOne], + handlers: [handler(~event="Transfer", ~processingCount=4.)], + effects: [effect(~cacheCount=Some(7))], + storageWrites: [{storage: "Postgres", seconds: 2., count: 3}], + } + let second: Metrics.t = { + ...baseMetrics, + targetBufferSize: 50, + maxBatchSize: 1000, + isInReorgThreshold: true, + rollbackEnabled: true, + processingSeconds: 0.5, + rollbackCount: 2, + chains: [chainTwo], + handlers: [ + handler(~event="Transfer", ~processingCount=6.), + handler(~event="Approval", ~processingCount=1.), + ], + effects: [effect(~cacheCount=None)], + storageWrites: [{storage: "Postgres", seconds: 1., count: 4}], + } + + t.expect( + Metrics.merge([first, second], ~startTime, ~metricTime, ~elapsedSeconds=9.), + ).toStrictEqual({ + ...baseMetrics, + startTime, + metricTime, + elapsedSeconds: 9., + targetBufferSize: 150, + maxBatchSize: 5000, + isInReorgThreshold: true, + rollbackEnabled: true, + processingSeconds: 2., + rollbackCount: 3, + chains: [chainOne, chainTwo], + handlers: [ + { + ...handler(~event="Transfer", ~processingCount=10.), + processingSeconds: 2., + preloadSeconds: 1., + preloadCount: 4., + preloadSecondsTotal: 6., + }, + handler(~event="Approval", ~processingCount=1.), + ], + effects: [ + { + ...effect(~cacheCount=Some(7)), + callSeconds: 2., + callSecondsTotal: 4., + callCount: 6., + activeCallsCount: 2, + queueCount: 4, + queueWaitSeconds: 0.5, + invalidationsCount: 2., + }, + ], + storageWrites: [{storage: "Postgres", seconds: 3., count: 7}], + }) + }) + + it("Renders an empty group as an indexer that has reported nothing yet", t => { + t.expect(Metrics.merge([], ~startTime, ~metricTime, ~elapsedSeconds=0.)).toStrictEqual({ + ...baseMetrics, + startTime, + metricTime, + }) + }) +}) diff --git a/packages/envio-tests/test/lib_tests/Supervisor_test.res b/packages/envio-tests/test/lib_tests/Supervisor_test.res new file mode 100644 index 000000000..5617040f8 --- /dev/null +++ b/packages/envio-tests/test/lib_tests/Supervisor_test.res @@ -0,0 +1,154 @@ +open Vitest + +let chains = n => Array.make(~length=n, 0)->Array.mapWithIndex((_, i) => (i + 1)->ChainId.fromInt) + +describe("Supervisor.plan", () => { + it("Splits only when the budget affords two workers, and spends all of it", t => { + let plan = (~chainCount, ~maxConnections) => + Supervisor.plan(~chainIds=chains(chainCount), ~maxConnections)->Option.map( + workers => + workers->Array.map( + ({chainIds, maxConnections}: Supervisor.worker) => ( + chainIds->Array.map(ChainId.toString)->Array.joinUnsafe(","), + maxConnections, + ), + ), + ) + + t.expect([ + // The default budget is what one process uses today, so nothing splits. + plan(~chainCount=4, ~maxConnections=2), + plan(~chainCount=4, ~maxConnections=3), + plan(~chainCount=4, ~maxConnections=4), + // More budget than chains: the surplus widens every worker's pool + // instead of going unused. + plan(~chainCount=4, ~maxConnections=10), + plan(~chainCount=3, ~maxConnections=12), + // A single chain has nothing to split against, whatever the budget. + plan(~chainCount=1, ~maxConnections=100), + ]).toStrictEqual([ + None, + None, + Some([("1,3", 2), ("2,4", 2)]), + Some([("1", 3), ("2", 3), ("3", 2), ("4", 2)]), + Some([("1", 4), ("2", 4), ("3", 4)]), + None, + ]) + }) +}) + +describe("Supervisor.planForRun", () => { + let configYaml = ` +name: supervised-run +disable_default_cross_chain: true +contracts: + - name: Counters + events: + - event: Bumped(uint256 amount) +chains: + - id: 1 + start_block: 0 + contracts: + - name: Counters + address: "0x1111111111111111111111111111111111111111" + - id: 137 + start_block: 0 + contracts: + - name: Counters + address: "0x2222222222222222222222222222222222222222" +` + + let config = (~schema, ~isolatedChains=?) => { + let json = Core.fromUserApi(~schema, configYaml).config->JSON.parseOrThrow + switch (json, isolatedChains) { + | (Object(obj), Some(chainIds)) => obj->Dict.set("isolatedChains", JSON.Encode.array(chainIds)) + | _ => () + } + Config.fromPublic(json) + } + + let perChain = ` +type Counter { + id: ID! + count: BigInt! +} +` + let crossChain = ` +type Counter { + id: ID! + count: BigInt! +} +type GlobalCounter @crossChain { + id: ID! + count: BigInt! +} +` + + it("Splits a per-chain schema, and leaves everything else in one process", t => { + let workerChains = (~schema, ~maxConnections, ~isolatedChains=?) => + Supervisor.planForRun(~config=config(~schema, ~isolatedChains?), ~maxConnections)->Option.map( + workers => + workers->Array.map( + worker => worker.chainIds->Array.map(ChainId.toString)->Array.joinUnsafe(","), + ), + ) + + t.expect([ + workerChains(~schema=perChain, ~maxConnections=4), + // An entity shared across chains can't be split: workers would each + // advance their own checkpoint over rows the others reach. + workerChains(~schema=crossChain, ~maxConnections=4), + // The budget one process uses today buys nothing to split with. + workerChains(~schema=perChain, ~maxConnections=2), + // Already one chain's process: whoever started it owns the layout. + workerChains(~schema=perChain, ~maxConnections=100, ~isolatedChains=[JSON.Number(1.)]), + ]).toStrictEqual([Some(["1", "137"]), None, None, None]) + }) +}) + +describe("Supervisor worker plumbing", () => { + it("Holds back a chunk's partial tail until the rest of the line arrives", t => { + let split = Supervisor.makeLineSplitter() + + t.expect([ + split("one\ntw"), + split("o\nthree\n"), + split(""), + split("four\nfive\n"), + ]).toStrictEqual([["one\n"], ["two\n", "three\n"], [], ["four\n", "five\n"]]) + }) + + it("Narrows the config it hands a worker to that worker's chains", t => { + let configJson = JSON.Object( + Dict.fromArray([("name", JSON.String("indexer")), ("isolatedChains", JSON.Null)]), + ) + + t.expect( + configJson->Supervisor.configForWorker( + ~worker={chainIds: [1, 137]->Array.map(ChainId.fromInt), maxConnections: 2}, + ), + ).toStrictEqual( + JSON.Object( + Dict.fromArray([ + ("name", JSON.String("indexer")), + ("isolatedChains", JSON.Array([JSON.Number(1.), JSON.Number(137.)])), + ]), + ), + ) + }) + + it("Gives every worker a log file of its own", t => { + t.expect([ + Supervisor.logFilePath(~workerIndex=0, ~path="logs/envio.log"), + Supervisor.logFilePath(~workerIndex=1, ~path="logs/envio.log"), + // A dot in a directory name is not an extension. + Supervisor.logFilePath(~workerIndex=1, ~path="./logs/envio"), + Supervisor.logFilePath(~workerIndex=2, ~path="envio"), + ]).toStrictEqual([ + "logs/envio.worker-0.log", + "logs/envio.worker-1.log", + "./logs/envio.worker-1", + "envio.worker-2", + ]) + }) +}) diff --git a/packages/envio/src/Bin.res b/packages/envio/src/Bin.res index b0b3439c4..af572c2d9 100644 --- a/packages/envio/src/Bin.res +++ b/packages/envio/src/Bin.res @@ -51,23 +51,35 @@ let applyEnv = (env: dict) => let run = async args => { try { - switch (await Core.runCli(args))->Null.toOption { - // Rust-only command (codegen / init / stop / docker / metrics / help / - // version / scripts) — nothing for JS to do, exit cleanly. - | None => () - | Some(json) => - switch decodeCommand(json->JSON.parseOrThrow) { - | Start({reset, cwd, env, config}) => - Config.prime(config) - processChdir(cwd) - applyEnv(env) - await Main.start(~reset) - | Migrate({reset, config}) => - Config.prime(config) - await Main.migrate(~reset) - | DropSchema({config}) => - Config.prime(config) - await Main.dropSchema() + if Worker.isEnabled { + Worker.exitWithSupervisor() + // A worker is handed the config its supervisor already parsed, narrowed to + // the chains it drives, so the two can't disagree about what is indexed. + // Its working directory and environment came with the fork. + Config.prime(await Worker.awaitInit()) + await Main.start() + } else { + switch (await Core.runCli(args))->Null.toOption { + // Rust-only command (codegen / init / stop / docker / metrics / help / + // version / scripts) — nothing for JS to do, exit cleanly. + | None => () + | Some(json) => + switch decodeCommand(json->JSON.parseOrThrow) { + | Start({reset, cwd, env, config}) => + Config.prime(config) + processChdir(cwd) + applyEnv(env) + switch Supervisor.planForRun(~config=Config.load()) { + | Some(workers) => await Supervisor.run(~workers, ~configJson=config, ~reset) + | None => await Main.start(~reset) + } + | Migrate({reset, config}) => + Config.prime(config) + await Main.migrate(~reset) + | DropSchema({config}) => + Config.prime(config) + await Main.dropSchema() + } } } } catch { diff --git a/packages/envio/src/Main.res b/packages/envio/src/Main.res index a1562d83f..2758a44c0 100644 --- a/packages/envio/src/Main.res +++ b/packages/envio/src/Main.res @@ -491,7 +491,7 @@ let getGlobalIndexer = (): 'indexer => { let startServer = ( ~getMetrics: unit => option, ~envioVersion: string, - ~persistence: Persistence.t, + ~onSyncCache: unit => promise, ~isDevelopmentMode: bool, ) => { open Express @@ -546,8 +546,8 @@ let startServer = ( app->post("/console/syncCache", (_req, res) => { if isDevelopmentMode { - (persistence->Persistence.getInitializedStorageOrThrow).dumpEffectCache() - ->Promise.thenResolve(_ => res->json(Boolean(true))) + onSyncCache() + ->Promise.thenResolve(() => res->json(Boolean(true))) ->Promise.ignore } else { res->json(Boolean(false)) @@ -593,7 +593,17 @@ type mainArgs = Yargs.parsedArgs // `envio_info` (on initialize) and validates against (on resume). let getEnvioInfo = () => Config.getPublicConfigJson()->Config.stripSensitiveData -let migrate = async (~reset) => { +let migrate = async ( + ~reset, + // A supervisor creating the schema for a run it is about to start names that + // run's commands, not the migration's, in what a config change prints. + ~resetCommand="envio local db-migrate setup", + ~runCommand=None, + // A migration command runs once and exits, with nobody watching it recover: + // an unreachable chain should say so now rather than hold the command open. + // A run that is about to start wants the opposite. + ~startBlockRetry=StartBlockResolver.Once, +) => { let config = Config.load() let persistence = PgStorage.makePersistenceFromConfig(~config) await persistence->Persistence.init( @@ -601,12 +611,10 @@ let migrate = async (~reset) => { ~chainConfigs=config.chainMap->ChainMap.values, ~contractMapping=config.contractMapping, ~envioInfo=getEnvioInfo(), - ~resetCommand="envio local db-migrate setup", - ~runCommand=None, + ~resetCommand, + ~runCommand, ~lowercaseAddresses=config.lowercaseAddresses, - // A migration command runs once and exits, with nobody watching it recover: - // an unreachable chain should say so now rather than hold the command open. - ~startBlockRetry=StartBlockResolver.Once, + ~startBlockRetry, ) await persistence.storage.close() } @@ -622,23 +630,31 @@ let dropSchema = async () => { // context, so callers should act on it (exit / re-throw) without logging again. exception FatalError(exn) -let start = async ( - ~persistence: option=?, - ~reset=false, - ~isTest=false, - ~exitAfterFirstEventBlock=false, - ~patchConfig: option<(Config.t, HandlerRegister.registrationsByChainId) => Config.t>=?, -) => { +// Whether this process draws the progress display: `--tui-off` first, then +// `ENVIO_TUI`, then whether anything is watching. A supervisor asks the same +// question its workers would have, since it is the one drawing for the run. +let shouldUseTui = (~suppressed=false) => { let mainArgs: mainArgs = process->argv->Yargs.hideBin->Yargs.yargs->Yargs.argv let explicitTui = switch mainArgs.tuiOff { | Some(off) => Some(!off) | None => Env.tuiEnvVar } - let shouldUseTui = switch (isTest, explicitTui) { + switch (suppressed, explicitTui) { | (true, _) => false | (_, Some(tui)) => tui | (_, None) => !Envio.isNonInteractive() } +} + +let start = async ( + ~persistence: option=?, + ~reset=false, + ~isTest=false, + ~exitAfterFirstEventBlock=false, + ~patchConfig: option<(Config.t, HandlerRegister.registrationsByChainId) => Config.t>=?, +) => { + // A worker reports to its supervisor, which draws for the whole run. + let shouldUseTui = shouldUseTui(~suppressed=isTest || Worker.isEnabled) // Initialize persistence first so the exported indexer value contains state from the database // when handler files are loaded (they may access the indexer at module top level). let config = Config.load() @@ -697,9 +713,18 @@ let start = async ( let envioVersion = Utils.EnvioPackage.value.version let getMetrics = () => getIndexerState()->Option.map(IndexerState.toMetrics) - - if !isTest { - startServer(~persistence, ~isDevelopmentMode, ~envioVersion, ~getMetrics) + let dumpEffectCache = () => + (persistence->Persistence.getInitializedStorageOrThrow).dumpEffectCache() + + // A worker reports through its supervisor, which owns the one server and the + // one display the run has. + if !isTest && !Worker.isEnabled { + startServer( + ~onSyncCache=() => dumpEffectCache()->Promise.thenResolve(ignore), + ~isDevelopmentMode, + ~envioVersion, + ~getMetrics, + ) } let state = IndexerState.makeFromDbState( @@ -715,6 +740,18 @@ let start = async ( if shouldUseTui { let _rerender = Tui.start(~config, ~getMetrics=() => state->IndexerState.toMetrics) } + if Worker.isEnabled { + Worker.onParentMessage(message => + switch message { + | SyncCache(_) => dumpEffectCache()->Promise.ignore + | Init(_) => () + } + ) + let _intervalId = setInterval( + () => Worker.send(Snapshot({metrics: state->IndexerState.toMetrics})), + Worker.snapshotIntervalMillis, + ) + } setIndexerState(state) state->IndexerLoop.start await runUntilFatalError diff --git a/packages/envio/src/Metrics.res b/packages/envio/src/Metrics.res index cee91045c..8f3cd7bbd 100644 --- a/packages/envio/src/Metrics.res +++ b/packages/envio/src/Metrics.res @@ -148,6 +148,106 @@ type t = { sourceHeightStreams: array, } +// Folds items that share a key into one, keeping first-seen order so the +// rendered series doesn't reshuffle between scrapes. +let sumByKey = (items: array<'item>, ~key: 'item => string, ~add: ('item, 'item) => 'item) => { + let byKey = Dict.make() + let order = [] + items->Array.forEach(item => { + let k = item->key + switch byKey->Utils.Dict.dangerouslyGetNonOption(k) { + | Some(existing) => byKey->Dict.set(k, add(existing, item)) + | None => { + byKey->Dict.set(k, item) + order->Array.push(k) + } + } + }) + order->Array.map(k => byKey->Dict.getUnsafe(k)) +} + +// Combines the snapshots a supervised run's workers reported into the one an +// unsplit run would have produced. Series keyed by chain concatenate, since a +// chain belongs to exactly one worker; series keyed by name are summed, since +// every worker runs the same handlers and effects over its own chains. The +// clock is the caller's: it belongs to the group, not to any worker. +let merge = (snapshots: array, ~startTime, ~metricTime, ~elapsedSeconds) => { + let concat = select => snapshots->Array.flatMap(select) + let sumInt = select => snapshots->Array.reduce(0, (acc, snapshot) => acc + snapshot->select) + let sumFloat = select => snapshots->Array.reduce(0., (acc, snapshot) => acc +. snapshot->select) + + { + startTime, + metricTime, + elapsedSeconds, + targetBufferSize: sumInt(s => s.targetBufferSize), + isInReorgThreshold: snapshots->Array.some(s => s.isInReorgThreshold), + rollbackEnabled: snapshots->Array.some(s => s.rollbackEnabled), + maxBatchSize: snapshots->Array.reduce(0, (acc, s) => Pervasives.max(acc, s.maxBatchSize)), + preloadSeconds: sumFloat(s => s.preloadSeconds), + processingSeconds: sumFloat(s => s.processingSeconds), + processingStalledOnFetchSeconds: sumFloat(s => s.processingStalledOnFetchSeconds), + processingStalledOnStorageWriteSeconds: sumFloat(s => s.processingStalledOnStorageWriteSeconds), + rollbackSeconds: sumFloat(s => s.rollbackSeconds), + rollbackCount: sumInt(s => s.rollbackCount), + rollbackEventsCount: sumFloat(s => s.rollbackEventsCount), + chains: concat(s => s.chains), + sourceRequests: concat(s => s.sourceRequests), + sourceHeights: concat(s => s.sourceHeights), + sourceHeightStreams: concat(s => s.sourceHeightStreams), + handlers: concat(s => s.handlers)->sumByKey( + ~key=h => `${h.contract}.${h.event}`, + ~add=(a, b) => { + ...a, + processingSeconds: a.processingSeconds +. b.processingSeconds, + processingCount: a.processingCount +. b.processingCount, + preloadSeconds: a.preloadSeconds +. b.preloadSeconds, + preloadCount: a.preloadCount +. b.preloadCount, + preloadSecondsTotal: a.preloadSecondsTotal +. b.preloadSecondsTotal, + }, + ), + effects: concat(s => s.effects)->sumByKey( + ~key=e => `${e.effect}.${e.scope}`, + ~add=(a, b) => { + ...a, + callSeconds: a.callSeconds +. b.callSeconds, + callSecondsTotal: a.callSecondsTotal +. b.callSecondsTotal, + callCount: a.callCount +. b.callCount, + activeCallsCount: a.activeCallsCount + b.activeCallsCount, + queueCount: a.queueCount + b.queueCount, + queueWaitSeconds: a.queueWaitSeconds +. b.queueWaitSeconds, + invalidationsCount: a.invalidationsCount +. b.invalidationsCount, + // An effect's cache rows are per chain, so worker counts are disjoint. + // Absent unless some worker persists the cache at all. + cacheCount: switch (a.cacheCount, b.cacheCount) { + | (Some(x), Some(y)) => Some(x + y) + | (Some(x), None) => Some(x) + | (None, y) => y + }, + }, + ), + storageLoads: concat(s => s.storageLoads)->sumByKey( + ~key=l => `${l.storage}.${l.operation}`, + ~add=(a, b) => { + ...a, + seconds: a.seconds +. b.seconds, + secondsTotal: a.secondsTotal +. b.secondsTotal, + count: a.count +. b.count, + whereSize: a.whereSize +. b.whereSize, + size: a.size +. b.size, + }, + ), + storageWrites: concat(s => s.storageWrites)->sumByKey( + ~key=w => w.storage, + ~add=(a, b) => {...a, seconds: a.seconds +. b.seconds, count: a.count + b.count}, + ), + historyPrunes: concat(s => s.historyPrunes)->sumByKey( + ~key=p => p.entity, + ~add=(a, b) => {...a, seconds: a.seconds +. b.seconds, count: a.count + b.count}, + ), + } +} + // Prometheus floats keep at most 3 decimals; integral values render without a // fractional part. @inline diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res new file mode 100644 index 000000000..5c3e1ff4e --- /dev/null +++ b/packages/envio/src/Supervisor.res @@ -0,0 +1,292 @@ +// One worker process and the chains it drives. `maxConnections` is its slice of +// the run's connection budget, which the pool it opens is capped to. +type worker = {chainIds: array, maxConnections: int} + +// Every worker needs enough connections to read and write without serializing +// on a single one, so the budget buys workers two at a time. +let minConnectionsPerWorker = 2 + +// How to spend a connection budget on the chains a run indexes. `None` keeps +// the run in one process, which is what a budget too small to afford two +// workers, or a config with nothing to split, has to do. +// +// Chains go round-robin over the config's own order rather than by size: what +// balances a run is knowing how much work each chain has left, and that isn't +// known until the chains report their heights. +let plan = (~chainIds: array, ~maxConnections: int): option> => { + let workerCount = Pervasives.min(chainIds->Array.length, maxConnections / minConnectionsPerWorker) + if workerCount < 2 { + None + } else { + // The remainder is handed out one connection at a time rather than left + // unspent, so a budget with slack widens the earliest workers' pools. + let evenShare = maxConnections / workerCount + let remainder = mod(maxConnections, workerCount) + Some( + Array.fromInitializer(~length=workerCount, workerIndex => { + chainIds: chainIds->Array.filterWithIndex((_, chainIndex) => + mod(chainIndex, workerCount) === workerIndex + ), + maxConnections: evenShare + (workerIndex < remainder ? 1 : 0), + }), + ) + } +} + +// Whether this run splits, and how. A schema that shares entities across chains +// can't be split: workers each advance their own checkpoint sequence, which only +// holds while no entity has rows another chain can reach. A run that is already +// one chain's process doesn't split again — whoever started it owns the layout. +let planForRun = (~config: Config.t, ~maxConnections=Env.Db.maxConnections) => + if config.isolated || config.userEntities->Array.some(entity => entity.crossChain) { + None + } else { + plan(~chainIds=config.chainMap->ChainMap.values->Array.map(chain => chain.id), ~maxConnections) + } + +// One forked worker: the process, the chains it drives, and the last snapshot +// it reported. `None` until it reports, which is what makes a run that hasn't +// heard from anyone yet render as initializing rather than as empty. +type running = { + worker: worker, + child: NodeJs.ChildProcess.child, + mutable snapshot: option, + // A spawn failure can raise `error` and `exit` both, and a worker counted + // twice would end the run while its siblings are still indexing. + mutable settled: bool, +} + +let label = (worker: worker) => + `[chain ${worker.chainIds->Array.map(ChainId.toString)->Array.joinUnsafe(",")}]` + +// Workers append to files of their own. Pino writes a line per call, and +// several processes appending to one file can still tear a long line apart. +let logFilePath = (~workerIndex, ~path=Env.logFilePath) => { + let suffix = `.worker-${workerIndex->Int.toString}` + // A dot in a directory name isn't an extension: `./logs/envio` keeps its + // whole path and takes the suffix at the end. + let dot = path->String.lastIndexOf(".") + if dot > path->String.lastIndexOf("/") && dot !== -1 { + `${path->String.slice(~start=0, ~end=dot)}${suffix}${path->String.slice( + ~start=dot, + ~end=path->String.length, + )}` + } else { + `${path}${suffix}` + } +} + +// Only the pretty strategy is written for a person to read, so only it takes a +// prefix. The structured strategies pass through untouched, since a line a log +// shipper has to parse must stay exactly what the worker emitted. +let shouldPrefixLogs = Env.logStrategy === Logging.ConsolePretty + +// A chunk off a worker's pipe ends mid-line as often as not, so the tail is +// held back until the rest of it arrives. Returns the whole lines a chunk +// completed, each still newline-terminated so it writes through unchanged. +let makeLineSplitter = () => { + let pending = ref("") + chunk => { + let lines = (pending.contents ++ chunk)->String.split("\n") + pending := lines->Array.pop->Option.getOr("") + lines->Array.map(line => `${line}\n`) + } +} + +let forward = (stream, ~prefix, ~write) => { + let split = makeLineSplitter() + stream->NodeJs.ChildProcess.setEncoding("utf8") + stream->NodeJs.ChildProcess.onData(chunk => + split(chunk)->Array.forEach(line => write(`${prefix}${line}`)) + ) +} + +let configForWorker = (configJson: JSON.t, ~worker) => + switch configJson->JSON.Decode.object { + | Some(fields) => { + let narrowed = fields->Dict.copy + narrowed->Dict.set( + "isolatedChains", + worker.chainIds->S.reverseConvertToJsonOrThrow(S.array(ChainId.schema)), + ) + JSON.Object(narrowed) + } + | None => JsError.throwWithMessage("Invalid indexer config: not an object") + } + +let fork = ( + worker: worker, + ~workerIndex, + ~configJson, + // The entry this process was itself started from, so a worker is the same + // program as its supervisor however the package was installed. + ~entryPath=NodeJs.Process.argv->Array.getUnsafe(1), +) => { + let env = NodeJs.Process.process.env->Dict.copy + env->Dict.set("ENVIO_WORKER", "true") + // The worker's slice of the budget. Read when the worker's own Env module + // loads, which is why it rides in the spawn environment rather than a message. + env->Dict.set("ENVIO_PG_MAX_CONNECTIONS", worker.maxConnections->Int.toString) + env->Dict.set("LOG_FILE", logFilePath(~workerIndex)) + + let child = NodeJs.ChildProcess.fork( + entryPath, + [], + { + env, + serialization: "advanced", + stdio: ["pipe", "pipe", "pipe", "ipc"], + }, + ) + child + ->NodeJs.ChildProcess.send(Worker.Init({config: configJson->configForWorker(~worker)})) + ->ignore + + let prefix = shouldPrefixLogs ? `${worker->label} ` : "" + child + ->NodeJs.ChildProcess.stdout + ->Null.toOption + ->Option.forEach(stream => stream->forward(~prefix, ~write=NodeJs.Process.writeStdout)) + child + ->NodeJs.ChildProcess.stderr + ->Null.toOption + ->Option.forEach(stream => stream->forward(~prefix, ~write=NodeJs.Process.writeStderr)) + + {worker, child, snapshot: None, settled: false} +} + +// The forked workers of one run, and whether their supervisor is the one +// taking them down. A stop it asked for is expected; every other way a worker +// can end is a failure. +type group = {running: array, mutable stopping: bool} + +let stop = group => { + group.stopping = true + group.running->Array.forEach(r => r.child->NodeJs.ChildProcess.kill("SIGTERM")->ignore) +} + +// Resolves once every worker has ended. Throws if any of them ended in a way +// the supervisor didn't ask for, having first taken the rest down: one worker +// short leaves its chains unindexed, and a run that kept the others going would +// look healthy while falling behind. +let awaitExit = async group => { + let failed = ref(false) + let alive = ref(group.running->Array.length) + + await Promise.make((resolve, _) => { + let onGone = (r, ~failure) => + if !r.settled { + r.settled = true + if failure { + failed := true + if !group.stopping { + group->stop + } + } + alive := alive.contents - 1 + if alive.contents === 0 { + resolve() + } + } + + group.running->Array.forEach(r => { + r.child->NodeJs.ChildProcess.onExit( + (code, _signal) => + // Only an exit the supervisor asked for is expected. Anything else — a + // non-zero code, or a signal like the kernel's out-of-memory kill. + r->onGone(~failure=!group.stopping && code->Null.toOption !== Some(0)), + ) + r.child->NodeJs.ChildProcess.onChildError( + exn => { + Logging.errorWithExn(exn, `${r.worker->label} failed to start`) + r->onGone(~failure=true) + }, + ) + }) + }) + + if failed.contents { + JsError.throwWithMessage("An indexer process exited with a failure. Stopped the others.") + } +} + +// Runs the group: creates the schema for every chain, forks a worker per plan +// entry, and serves the run's metrics, console and display from what they +// report. Returns once every worker has exited; throws if any of them failed. +let run = async (~workers: array, ~configJson: JSON.t, ~reset) => { + // Every chain's state has to exist before a worker resumes it: an isolated + // run refuses to initialize, precisely so it can't create rows for its own + // chains and leave the chains it skipped with nothing to resume. + await Main.migrate( + ~reset, + ~resetCommand="envio start -r", + ~runCommand=Some("envio start"), + ~startBlockRetry=StartBlockResolver.UntilItAnswers, + ) + + let config = Config.load() + let startTime = Date.make() + let startTimeRef = Performance.now() + + Logging.info( + `Splitting ${config.chainMap + ->ChainMap.values + ->Array.length + ->Int.toString} chains across ${workers + ->Array.length + ->Int.toString} processes, from a budget of ${Env.Db.maxConnections->Int.toString} database connections.`, + ) + + let group = { + running: workers->Array.mapWithIndex((worker, workerIndex) => + worker->fork(~workerIndex, ~configJson) + ), + stopping: false, + } + + let reported = () => group.running->Array.filterMap(r => r.snapshot) + let merge = snapshots => + Metrics.merge( + snapshots, + ~startTime, + ~metricTime=Date.make(), + ~elapsedSeconds=startTimeRef->Performance.secondsSince, + ) + + group.running->Array.forEach(r => + r.child->NodeJs.ChildProcess.onMessage(message => + switch message { + | Worker.Snapshot({metrics}) => r.snapshot = Some(metrics) + } + ) + ) + + Main.startServer( + // Nothing to report until a worker has: the run reads as initializing + // rather than as an indexer with no chains. + ~getMetrics=() => + switch reported() { + | [] => None + | snapshots => Some(snapshots->merge) + }, + ~envioVersion=Utils.EnvioPackage.value.version, + ~isDevelopmentMode=config.isDev, + ~onSyncCache=() => { + group.running->Array.forEach(r => + r.child->NodeJs.ChildProcess.send(Worker.SyncCache({}))->ignore + ) + Promise.resolve() + }, + ) + + if Main.shouldUseTui() { + let _rerender = Tui.start(~config, ~getMetrics=() => reported()->merge) + } + + // Only the supervisor is signalled when the run is asked to stop, so it + // passes that on. A terminal's own interrupt already reaches the whole group. + NodeJs.Process.onSignal("SIGTERM", () => group->stop) + NodeJs.Process.onSignal("SIGINT", () => group->stop) + + await group->awaitExit +} diff --git a/packages/envio/src/Worker.res b/packages/envio/src/Worker.res new file mode 100644 index 000000000..6f3dd47bb --- /dev/null +++ b/packages/envio/src/Worker.res @@ -0,0 +1,53 @@ +// The worker side of a supervised run: a process the supervisor forked to drive +// a subset of the chains. It has no server and no TUI of its own — it reports +// through the IPC channel, and the supervisor is the one operational surface. + +// Set by the supervisor on the processes it forks. An indexer a user started +// themselves never has it, and takes every path it takes today. +let isEnabled = + NodeJs.Process.process.env->Utils.Dict.dangerouslyGetNonOption("ENVIO_WORKER")->Option.isSome + +@tag("kind") +type parentMessage = + // The config the supervisor parsed, narrowed to this worker's chains. Sent + // instead of re-derived so a worker and its supervisor can never disagree + // about what is being indexed. + | @as("init") Init({config: JSON.t}) + | @as("syncCache") SyncCache({}) + +@tag("kind") +type workerMessage = | @as("snapshot") Snapshot({metrics: Metrics.t}) + +// How often a worker reports. Matches the TUI's own refresh, so the supervised +// display moves at the same rate an unsplit run's does. +let snapshotIntervalMillis = 500 + +// A worker with no supervisor has nobody reading its metrics and nobody to stop +// it: the supervisor could have died before it ever got to tear the group down. +// Losing the channel is that signal. +let exitWithSupervisor = () => + if isEnabled { + NodeJs.Process.onDisconnect(() => { + Logging.error("The indexer supervisor is gone. Stopping this chain's process.") + NodeJs.process->NodeJs.exitWithCode(Failure) + }) + } + +let send = (message: workerMessage) => + if isEnabled { + NodeJs.Process.sendToParent(message)->ignore + } + +let onParentMessage = (handle: parentMessage => unit) => NodeJs.Process.onMessage(handle) + +// Resolves with the init payload the supervisor sends immediately after the +// fork. Nothing else can run first: the worker has no config until it lands. +let awaitInit = (): promise => + Promise.make((resolve, _) => + onParentMessage(message => + switch message { + | Init({config}) => resolve(config) + | SyncCache(_) => () + } + ) + ) diff --git a/packages/envio/src/bindings/NodeJs.res b/packages/envio/src/bindings/NodeJs.res index 1b62ca6cf..821abeb50 100644 --- a/packages/envio/src/bindings/NodeJs.res +++ b/packages/envio/src/bindings/NodeJs.res @@ -64,6 +64,19 @@ module Process = { @module("process") external version: string = "version" @module("process") external getActiveResourcesInfo: unit => array = "getActiveResourcesInfo" + + // Only a process forked with an IPC channel has these. Called through + // `process` rather than off a namespace import, which would drop the + // receiver Node's own implementations read. + @val @scope("process") external sendToParent: 'msg => bool = "send" + @val @scope("process") + external onMessage: (@as("message") _, 'msg => unit) => unit = "on" + @val @scope("process") external onSignal: (string, unit => unit) => unit = "on" + @val @scope("process") + external onDisconnect: (@as("disconnect") _, unit => unit) => unit = "on" + @val @scope("process") external argv: array = "argv" + @val @scope(("process", "stdout")) external writeStdout: string => unit = "write" + @val @scope(("process", "stderr")) external writeStderr: string => unit = "write" } module Buffer = { @@ -141,6 +154,30 @@ module ChildProcess = { @module("child_process") external execWithOptions: (string, execOptions, callback) => unit = "exec" + + type child + type readable + type forkOptions = { + cwd?: string, + env?: dict, + // "advanced" uses the structured clone algorithm, so a message keeps the + // Date values a metrics snapshot carries instead of stringifying them. + serialization?: string, + stdio?: array, + } + @module("child_process") + external fork: (string, array, forkOptions) => child = "fork" + @send external send: (child, 'msg) => bool = "send" + @send external onMessage: (child, @as("message") _, 'msg => unit) => unit = "on" + @send + external onExit: (child, @as("exit") _, (Null.t, Null.t) => unit) => unit = "on" + @send external onChildError: (child, @as("error") _, exn => unit) => unit = "on" + @send external kill: (child, string) => bool = "kill" + @get external pid: child => Null.t = "pid" + @get external stdout: child => Null.t = "stdout" + @get external stderr: child => Null.t = "stderr" + @send external onData: (readable, @as("data") _, string => unit) => unit = "on" + @send external setEncoding: (readable, string) => unit = "setEncoding" } module Url = { From 6464a9ca6a7b8706288e8284a775c971a169b6e6 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 15 Sep 2026 19:32:05 +0000 Subject: [PATCH 02/61] Stamp a worker's chains on its logs, and spread the busiest chains apart Two changes to how a split run reads and how it is laid out. A worker's logger now carries the chains it drives, so every line it writes says where it came from even where the call site had no chain in hand. That replaces the prefix the supervisor used to paste onto each line it read back: workers write straight to the run's output now, which also drops the piping and the line-reassembly it needed. One chain reports `chainId`, matching the field chain-scoped logs already use, so queries over both unify; several report `chainIds`. Chains are dealt busiest-first rather than in config order, reversing direction each pass, so the heaviest chains lead different workers and the worker that took the heaviest picks up the lightest. The ranking is an explicit list of chain ids and nothing more: it decides only which worker a chain lands on, so a chain ranked wrong, or missing from the list, costs balance and nothing else. Chains it doesn't name sort behind the ones it does, in config order. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- .../envio-tests/test/SupervisorFork_test.res | 3 + .../envio-tests/test/helpers/fakeWorker.mjs | 1 + .../test/lib_tests/Supervisor_test.res | 56 ++++++++--- packages/envio/src/Env.res | 10 ++ packages/envio/src/Logging.res | 20 +++- packages/envio/src/Supervisor.res | 96 ++++++++++--------- packages/envio/src/Worker.res | 3 +- packages/envio/src/bindings/NodeJs.res | 7 -- 8 files changed, 128 insertions(+), 68 deletions(-) diff --git a/packages/envio-tests/test/SupervisorFork_test.res b/packages/envio-tests/test/SupervisorFork_test.res index f44e51e3e..9b7fcc682 100644 --- a/packages/envio-tests/test/SupervisorFork_test.res +++ b/packages/envio-tests/test/SupervisorFork_test.res @@ -6,6 +6,7 @@ type fixtureReport = { isolatedChains: array, maxConnections: string, logFile: string, + workerChains: string, startTime: Date.t, } @@ -39,6 +40,8 @@ describe("Supervisor.fork", () => { isolatedChains: [1., 137.], maxConnections: "3", logFile: Supervisor.logFilePath(~workerIndex=1), + // What the worker's logger stamps onto every line it writes. + workerChains: "1,137", // Proof the channel clones rather than stringifies: a JSON round trip // would have turned this into a string. startTime: Date.fromTime(1700000000000.), diff --git a/packages/envio-tests/test/helpers/fakeWorker.mjs b/packages/envio-tests/test/helpers/fakeWorker.mjs index e6f7342bf..acf4a5cc5 100644 --- a/packages/envio-tests/test/helpers/fakeWorker.mjs +++ b/packages/envio-tests/test/helpers/fakeWorker.mjs @@ -11,6 +11,7 @@ process.on("message", (message) => { isolatedChains: message.config.isolatedChains, maxConnections: process.env.ENVIO_PG_MAX_CONNECTIONS, logFile: process.env.LOG_FILE, + workerChains: process.env.ENVIO_WORKER, // A Date survives only under structured-clone serialization, which is // what a metrics snapshot's timestamps need. startTime: new Date(1700000000000), diff --git a/packages/envio-tests/test/lib_tests/Supervisor_test.res b/packages/envio-tests/test/lib_tests/Supervisor_test.res index 5617040f8..0d58d15ae 100644 --- a/packages/envio-tests/test/lib_tests/Supervisor_test.res +++ b/packages/envio-tests/test/lib_tests/Supervisor_test.res @@ -29,7 +29,7 @@ describe("Supervisor.plan", () => { ]).toStrictEqual([ None, None, - Some([("1,3", 2), ("2,4", 2)]), + Some([("1,4", 2), ("2,3", 2)]), Some([("1", 3), ("2", 3), ("3", 2), ("4", 2)]), Some([("1", 4), ("2", 4), ("3", 4)]), None, @@ -37,6 +37,32 @@ describe("Supervisor.plan", () => { }) }) +describe("Supervisor.plan volume spreading", () => { + let assignment = (~chainIds, ~maxConnections) => + Supervisor.plan(~chainIds=chainIds->Array.map(ChainId.fromInt), ~maxConnections) + ->Option.getOrThrow + ->Array.map(worker => + worker.chainIds->Array.map(ChainId.toString)->Array.joinUnsafe(",") + ) + + it("Keeps the busiest chains apart, and pairs them with the quietest", t => { + t.expect([ + // Ethereum and BNB are the two busiest of these, so they lead different + // workers, and each is paired with the lighter of the remaining two. + assignment(~chainIds=[8453, 56, 42161, 1], ~maxConnections=4), + // Unranked chains sort behind the ranked ones, keeping config order. + assignment(~chainIds=[999, 137, 888, 1], ~maxConnections=4), + // Three workers take the top three, then fold back so the heaviest + // worker picks up the lightest chain. + assignment(~chainIds=[1, 56, 137, 8453, 42161, 10], ~maxConnections=6), + ]).toStrictEqual([ + ["1,42161", "56,8453"], + ["1,888", "137,999"], + ["1,10", "56,42161", "137,8453"], + ]) + }) +}) + describe("Supervisor.planForRun", () => { let configYaml = ` name: supervised-run @@ -107,17 +133,6 @@ type GlobalCounter @crossChain { }) describe("Supervisor worker plumbing", () => { - it("Holds back a chunk's partial tail until the rest of the line arrives", t => { - let split = Supervisor.makeLineSplitter() - - t.expect([ - split("one\ntw"), - split("o\nthree\n"), - split(""), - split("four\nfive\n"), - ]).toStrictEqual([["one\n"], ["two\n", "three\n"], [], ["four\n", "five\n"]]) - }) - it("Narrows the config it hands a worker to that worker's chains", t => { let configJson = JSON.Object( Dict.fromArray([("name", JSON.String("indexer")), ("isolatedChains", JSON.Null)]), @@ -152,3 +167,20 @@ describe("Supervisor worker plumbing", () => { ]) }) }) + +describe("Logging.makeBase", () => { + it("Stamps a worker's chains onto every line it logs", t => { + t.expect([ + // Not a worker: nothing extra, which is what keeps pid and hostname out. + Logging.makeBase(~workerChainIds=None), + Logging.makeBase(~workerChainIds=Some([137.])), + Logging.makeBase(~workerChainIds=Some([1., 137.])), + ]).toStrictEqual([ + JSON.Object(Dict.make()), + JSON.Object(Dict.fromArray([("chainId", JSON.Number(137.))])), + JSON.Object( + Dict.fromArray([("chainIds", JSON.Array([JSON.Number(1.), JSON.Number(137.)]))]), + ), + ]) + }) +}) diff --git a/packages/envio/src/Env.res b/packages/envio/src/Env.res index ab23da20d..b5315cef6 100644 --- a/packages/envio/src/Env.res +++ b/packages/envio/src/Env.res @@ -77,6 +77,15 @@ let hypersyncClientEnableQueryCaching = let hypersyncLogLevel = envSafe->EnvSafe.get("ENVIO_HYPERSYNC_LOG_LEVEL", HyperSyncClient.logLevelSchema, ~fallback=#info) +// The chains a supervisor forked this process to drive. Present only on a +// worker, and the chain ids it carries are what every line that worker logs is +// stamped with, so a split run's output says which chain it came from without +// anyone having to rewrite it downstream. +let workerChainIds = + envSafe + ->EnvSafe.get("ENVIO_WORKER", S.option(S.string)) + ->Option.map(value => value->String.split(",")->Array.filterMap(Float.fromString)) + let logStrategy = envSafe->EnvSafe.get( "LOG_STRATEGY", @@ -94,6 +103,7 @@ let logStrategy = Logging.setLogger( Logging.makeLogger( + ~base=Logging.makeBase(~workerChainIds), ~logStrategy, ~logFilePath, ~defaultFileLogLevel, diff --git a/packages/envio/src/Logging.res b/packages/envio/src/Logging.res index e9c897c69..358784b83 100644 --- a/packages/envio/src/Logging.res +++ b/packages/envio/src/Logging.res @@ -26,7 +26,22 @@ let logLevels = [ %%private(let logger = ref(None)) -let makeLogger = (~logStrategy, ~logFilePath, ~defaultFileLogLevel, ~userLogLevel) => { +// The fields every line a process logs carries. Empty is what keeps pid and +// hostname out. A worker names the chains it drives, so a split run's output +// says which chain a line came from wherever the call site had no chain in hand. +let makeBase = (~workerChainIds: option>): JSON.t => + switch workerChainIds { + | Some([chainId]) => JSON.Object(Dict.fromArray([("chainId", JSON.Number(chainId))])) + | Some([]) | None => JSON.Object(Dict.make()) + | Some(chainIds) => + JSON.Object( + Dict.fromArray([ + ("chainIds", JSON.Array(chainIds->Array.map(chainId => JSON.Number(chainId)))), + ]), + ) + } + +let makeLogger = (~logStrategy, ~logFilePath, ~defaultFileLogLevel, ~userLogLevel, ~base) => { // Currently unused - useful if using multiple transports. // let pinoRaw = {"target": "pino/file", "level": Config.userLogLevel} let pinoFile: Transport.transportTarget = { @@ -46,9 +61,6 @@ let makeLogger = (~logStrategy, ~logFilePath, ~defaultFileLogLevel, ~userLogLeve ... ) - // Empty base disables pid and hostname in logs - let base: JSON.t = %raw("{}") - switch logStrategy { | EcsFile => makeWithOptionsAndTransport( diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index 5c3e1ff4e..e53a9dcc6 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -6,27 +6,67 @@ type worker = {chainIds: array, maxConnections: int} // on a single one, so the budget buys workers two at a time. let minConnectionsPerWorker = 2 +// Roughly how much a chain costs to index, busiest first. An estimate only: +// it decides nothing but which worker a chain lands on, so a chain in the wrong +// place, or missing from the list entirely, costs balance and nothing else. +// Chains it doesn't name sort behind the ones it does, in config order. +let byDescendingVolume = [ + 1, // Ethereum + 56, // BNB Smart Chain + 137, // Polygon + 8453, // Base + 42161, // Arbitrum One + 10, // Optimism + 43114, // Avalanche + 81457, // Blast + 59144, // Linea + 534352, // Scroll + 324, // zkSync Era + 5000, // Mantle + 204, // opBNB + 100, // Gnosis + 42220, // Celo +] + +let volumeRank = (chainId: ChainId.t) => + switch byDescendingVolume->Array.indexOf(chainId->ChainId.toInt) { + | -1 => byDescendingVolume->Array.length + | rank => rank + } + // How to spend a connection budget on the chains a run indexes. `None` keeps // the run in one process, which is what a budget too small to afford two // workers, or a config with nothing to split, has to do. // -// Chains go round-robin over the config's own order rather than by size: what -// balances a run is knowing how much work each chain has left, and that isn't -// known until the chains report their heights. +// Chains are dealt busiest-first and the direction reverses each pass, so the +// heaviest chains lead different workers and the worker that took the heaviest +// picks up the lightest. Volume is only ever an estimate, which is why the +// layout it produces is a starting balance rather than a guarantee. let plan = (~chainIds: array, ~maxConnections: int): option> => { let workerCount = Pervasives.min(chainIds->Array.length, maxConnections / minConnectionsPerWorker) if workerCount < 2 { None } else { + let byVolume = + chainIds + ->Array.mapWithIndex((chainId, configIndex) => (chainId, chainId->volumeRank, configIndex)) + ->Array.toSorted(((_, aRank, aIndex), (_, bRank, bIndex)) => + // Two chains the list doesn't rank keep the order config gave them. + aRank === bRank ? Int.compare(aIndex, bIndex) : Int.compare(aRank, bRank) + ) + ->Array.map(((chainId, _, _)) => chainId) + // The remainder is handed out one connection at a time rather than left // unspent, so a budget with slack widens the earliest workers' pools. let evenShare = maxConnections / workerCount let remainder = mod(maxConnections, workerCount) Some( Array.fromInitializer(~length=workerCount, workerIndex => { - chainIds: chainIds->Array.filterWithIndex((_, chainIndex) => - mod(chainIndex, workerCount) === workerIndex - ), + chainIds: byVolume->Array.filterWithIndex((_, dealIndex) => { + let position = mod(dealIndex, workerCount) + let isReversePass = mod(dealIndex / workerCount, 2) === 1 + (isReversePass ? workerCount - 1 - position : position) === workerIndex + }), maxConnections: evenShare + (workerIndex < remainder ? 1 : 0), }), ) @@ -76,31 +116,6 @@ let logFilePath = (~workerIndex, ~path=Env.logFilePath) => { } } -// Only the pretty strategy is written for a person to read, so only it takes a -// prefix. The structured strategies pass through untouched, since a line a log -// shipper has to parse must stay exactly what the worker emitted. -let shouldPrefixLogs = Env.logStrategy === Logging.ConsolePretty - -// A chunk off a worker's pipe ends mid-line as often as not, so the tail is -// held back until the rest of it arrives. Returns the whole lines a chunk -// completed, each still newline-terminated so it writes through unchanged. -let makeLineSplitter = () => { - let pending = ref("") - chunk => { - let lines = (pending.contents ++ chunk)->String.split("\n") - pending := lines->Array.pop->Option.getOr("") - lines->Array.map(line => `${line}\n`) - } -} - -let forward = (stream, ~prefix, ~write) => { - let split = makeLineSplitter() - stream->NodeJs.ChildProcess.setEncoding("utf8") - stream->NodeJs.ChildProcess.onData(chunk => - split(chunk)->Array.forEach(line => write(`${prefix}${line}`)) - ) -} - let configForWorker = (configJson: JSON.t, ~worker) => switch configJson->JSON.Decode.object { | Some(fields) => { @@ -123,7 +138,9 @@ let fork = ( ~entryPath=NodeJs.Process.argv->Array.getUnsafe(1), ) => { let env = NodeJs.Process.process.env->Dict.copy - env->Dict.set("ENVIO_WORKER", "true") + // Marks the process a worker, and names the chains it drives: its logger + // stamps them onto every line it writes. + env->Dict.set("ENVIO_WORKER", worker.chainIds->Array.map(ChainId.toString)->Array.joinUnsafe(",")) // The worker's slice of the budget. Read when the worker's own Env module // loads, which is why it rides in the spawn environment rather than a message. env->Dict.set("ENVIO_PG_MAX_CONNECTIONS", worker.maxConnections->Int.toString) @@ -135,23 +152,16 @@ let fork = ( { env, serialization: "advanced", - stdio: ["pipe", "pipe", "pipe", "ipc"], + // Workers write straight to the run's own output. Their lines already say + // which chain they came from, so there is nothing for the supervisor to + // add by reading them first. + stdio: ["inherit", "inherit", "inherit", "ipc"], }, ) child ->NodeJs.ChildProcess.send(Worker.Init({config: configJson->configForWorker(~worker)})) ->ignore - let prefix = shouldPrefixLogs ? `${worker->label} ` : "" - child - ->NodeJs.ChildProcess.stdout - ->Null.toOption - ->Option.forEach(stream => stream->forward(~prefix, ~write=NodeJs.Process.writeStdout)) - child - ->NodeJs.ChildProcess.stderr - ->Null.toOption - ->Option.forEach(stream => stream->forward(~prefix, ~write=NodeJs.Process.writeStderr)) - {worker, child, snapshot: None, settled: false} } diff --git a/packages/envio/src/Worker.res b/packages/envio/src/Worker.res index 6f3dd47bb..27b7e3387 100644 --- a/packages/envio/src/Worker.res +++ b/packages/envio/src/Worker.res @@ -4,8 +4,7 @@ // Set by the supervisor on the processes it forks. An indexer a user started // themselves never has it, and takes every path it takes today. -let isEnabled = - NodeJs.Process.process.env->Utils.Dict.dangerouslyGetNonOption("ENVIO_WORKER")->Option.isSome +let isEnabled = Env.workerChainIds->Option.isSome @tag("kind") type parentMessage = diff --git a/packages/envio/src/bindings/NodeJs.res b/packages/envio/src/bindings/NodeJs.res index 821abeb50..15e130487 100644 --- a/packages/envio/src/bindings/NodeJs.res +++ b/packages/envio/src/bindings/NodeJs.res @@ -75,8 +75,6 @@ module Process = { @val @scope("process") external onDisconnect: (@as("disconnect") _, unit => unit) => unit = "on" @val @scope("process") external argv: array = "argv" - @val @scope(("process", "stdout")) external writeStdout: string => unit = "write" - @val @scope(("process", "stderr")) external writeStderr: string => unit = "write" } module Buffer = { @@ -156,7 +154,6 @@ module ChildProcess = { external execWithOptions: (string, execOptions, callback) => unit = "exec" type child - type readable type forkOptions = { cwd?: string, env?: dict, @@ -174,10 +171,6 @@ module ChildProcess = { @send external onChildError: (child, @as("error") _, exn => unit) => unit = "on" @send external kill: (child, string) => bool = "kill" @get external pid: child => Null.t = "pid" - @get external stdout: child => Null.t = "stdout" - @get external stderr: child => Null.t = "stderr" - @send external onData: (readable, @as("data") _, string => unit) => unit = "on" - @send external setEncoding: (readable, string) => unit = "setEncoding" } module Url = { From 2aa6f1b7bef108917f3a33b12158b5e5187cf511 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 15 Sep 2026 19:49:45 +0000 Subject: [PATCH 03/61] Attribute a per-chain run's logs to its chains, from config rather than the logger MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The chains belong to the run, not to the logger, and they apply whenever the schema is per-chain — a single process driving every chain included, not only a process a supervisor forked. So the config says what a line is attributable to and the indexer sets it once at startup. The logger only carries what it is handed: it keeps the root it was built with, so setting context twice in one process replaces it rather than stacking. A run driving one chain reports `chainId`, matching the field chain-scoped logs already use; one driving several reports `chainIds`; a schema with an entity shared across chains reports neither, since that work is no single chain's. `Config.isPerChain` now names the condition the split already tested for, so the two read the same predicate. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- .../envio-tests/test/SupervisorFork_test.res | 3 -- .../envio-tests/test/helpers/fakeWorker.mjs | 1 - .../test/lib_tests/Supervisor_test.res | 50 ++++++++++--------- packages/envio/src/Config.res | 28 +++++++++++ packages/envio/src/Env.res | 11 +--- packages/envio/src/Logging.res | 32 ++++++------ packages/envio/src/Main.res | 3 ++ packages/envio/src/Supervisor.res | 6 +-- packages/envio/src/Worker.res | 2 +- 9 files changed, 78 insertions(+), 58 deletions(-) diff --git a/packages/envio-tests/test/SupervisorFork_test.res b/packages/envio-tests/test/SupervisorFork_test.res index 9b7fcc682..f44e51e3e 100644 --- a/packages/envio-tests/test/SupervisorFork_test.res +++ b/packages/envio-tests/test/SupervisorFork_test.res @@ -6,7 +6,6 @@ type fixtureReport = { isolatedChains: array, maxConnections: string, logFile: string, - workerChains: string, startTime: Date.t, } @@ -40,8 +39,6 @@ describe("Supervisor.fork", () => { isolatedChains: [1., 137.], maxConnections: "3", logFile: Supervisor.logFilePath(~workerIndex=1), - // What the worker's logger stamps onto every line it writes. - workerChains: "1,137", // Proof the channel clones rather than stringifies: a JSON round trip // would have turned this into a string. startTime: Date.fromTime(1700000000000.), diff --git a/packages/envio-tests/test/helpers/fakeWorker.mjs b/packages/envio-tests/test/helpers/fakeWorker.mjs index acf4a5cc5..e6f7342bf 100644 --- a/packages/envio-tests/test/helpers/fakeWorker.mjs +++ b/packages/envio-tests/test/helpers/fakeWorker.mjs @@ -11,7 +11,6 @@ process.on("message", (message) => { isolatedChains: message.config.isolatedChains, maxConnections: process.env.ENVIO_PG_MAX_CONNECTIONS, logFile: process.env.LOG_FILE, - workerChains: process.env.ENVIO_WORKER, // A Date survives only under structured-clone serialization, which is // what a metrics snapshot's timestamps need. startTime: new Date(1700000000000), diff --git a/packages/envio-tests/test/lib_tests/Supervisor_test.res b/packages/envio-tests/test/lib_tests/Supervisor_test.res index 0d58d15ae..f374c2c9e 100644 --- a/packages/envio-tests/test/lib_tests/Supervisor_test.res +++ b/packages/envio-tests/test/lib_tests/Supervisor_test.res @@ -41,9 +41,7 @@ describe("Supervisor.plan volume spreading", () => { let assignment = (~chainIds, ~maxConnections) => Supervisor.plan(~chainIds=chainIds->Array.map(ChainId.fromInt), ~maxConnections) ->Option.getOrThrow - ->Array.map(worker => - worker.chainIds->Array.map(ChainId.toString)->Array.joinUnsafe(",") - ) + ->Array.map(worker => worker.chainIds->Array.map(ChainId.toString)->Array.joinUnsafe(",")) it("Keeps the busiest chains apart, and pairs them with the quietest", t => { t.expect([ @@ -63,8 +61,7 @@ describe("Supervisor.plan volume spreading", () => { }) }) -describe("Supervisor.planForRun", () => { - let configYaml = ` +let configYaml = ` name: supervised-run disable_default_cross_chain: true contracts: @@ -84,22 +81,22 @@ chains: address: "0x2222222222222222222222222222222222222222" ` - let config = (~schema, ~isolatedChains=?) => { - let json = Core.fromUserApi(~schema, configYaml).config->JSON.parseOrThrow - switch (json, isolatedChains) { - | (Object(obj), Some(chainIds)) => obj->Dict.set("isolatedChains", JSON.Encode.array(chainIds)) - | _ => () - } - Config.fromPublic(json) +let config = (~schema, ~isolatedChains=?) => { + let json = Core.fromUserApi(~schema, configYaml).config->JSON.parseOrThrow + switch (json, isolatedChains) { + | (Object(obj), Some(chainIds)) => obj->Dict.set("isolatedChains", JSON.Encode.array(chainIds)) + | _ => () } + Config.fromPublic(json) +} - let perChain = ` +let perChain = ` type Counter { id: ID! count: BigInt! } ` - let crossChain = ` +let crossChain = ` type Counter { id: ID! count: BigInt! @@ -110,6 +107,7 @@ type GlobalCounter @crossChain { } ` +describe("Supervisor.planForRun", () => { it("Splits a per-chain schema, and leaves everything else in one process", t => { let workerChains = (~schema, ~maxConnections, ~isolatedChains=?) => Supervisor.planForRun(~config=config(~schema, ~isolatedChains?), ~maxConnections)->Option.map( @@ -168,19 +166,23 @@ describe("Supervisor worker plumbing", () => { }) }) -describe("Logging.makeBase", () => { - it("Stamps a worker's chains onto every line it logs", t => { +describe("Config.logContext", () => { + it("Attributes a per-chain run's logs to its chains, and a shared one's to none", t => { t.expect([ - // Not a worker: nothing extra, which is what keeps pid and hostname out. - Logging.makeBase(~workerChainIds=None), - Logging.makeBase(~workerChainIds=Some([137.])), - Logging.makeBase(~workerChainIds=Some([1., 137.])), + // Every entity is per-chain, and this process drives one of them. + config(~schema=perChain, ~isolatedChains=[JSON.Number(137.)])->Config.logContext, + // Still per-chain, but this process drives both: no single chain owns a line. + config(~schema=perChain)->Config.logContext, + // An entity shared across chains: the work isn't any one chain's. + config(~schema=crossChain)->Config.logContext, ]).toStrictEqual([ - JSON.Object(Dict.make()), - JSON.Object(Dict.fromArray([("chainId", JSON.Number(137.))])), - JSON.Object( - Dict.fromArray([("chainIds", JSON.Array([JSON.Number(1.), JSON.Number(137.)]))]), + Some(JSON.Object(Dict.fromArray([("chainId", JSON.Number(137.))]))), + Some( + JSON.Object( + Dict.fromArray([("chainIds", JSON.Array([JSON.Number(1.), JSON.Number(137.)]))]), + ), ), + None, ]) }) }) diff --git a/packages/envio/src/Config.res b/packages/envio/src/Config.res index 5cfb57912..ea404a1c7 100644 --- a/packages/envio/src/Config.res +++ b/packages/envio/src/Config.res @@ -625,6 +625,34 @@ let getChain = (config, ~chainId) => "No chain with id " ++ chainId->ChainId.toString ++ " found in config.yaml", ) +// Whether every entity belongs to exactly one chain. Only then is a unit of +// this indexer's work attributable to a chain at all, which is what lets a run +// be split across processes and what lets its logs name a chain. +let isPerChain = (config: t) => !(config.userEntities->Array.some(entity => entity.crossChain)) + +// What every line this indexer logs is attributed to: the chains it drives. +// A schema shared across chains has none, since its work is no single chain's. +let logContext = (config: t): option => + if config->isPerChain { + let chainIds = config.chainMap->ChainMap.keys + Some( + switch chainIds { + | [chainId] => + JSON.Object( + Dict.fromArray([("chainId", chainId->S.reverseConvertToJsonOrThrow(ChainId.schema))]), + ) + | chainIds => + JSON.Object( + Dict.fromArray([ + ("chainIds", chainIds->S.reverseConvertToJsonOrThrow(S.array(ChainId.schema))), + ]), + ) + }, + ) + } else { + None + } + // Narrows a config to the chains one `envio start --chain` process drives. // `contractMapping` is deliberately left whole: its ids are what the migration // that created the schema stored, and one rebuilt from a subset would hand the diff --git a/packages/envio/src/Env.res b/packages/envio/src/Env.res index b5315cef6..ff719d835 100644 --- a/packages/envio/src/Env.res +++ b/packages/envio/src/Env.res @@ -77,14 +77,8 @@ let hypersyncClientEnableQueryCaching = let hypersyncLogLevel = envSafe->EnvSafe.get("ENVIO_HYPERSYNC_LOG_LEVEL", HyperSyncClient.logLevelSchema, ~fallback=#info) -// The chains a supervisor forked this process to drive. Present only on a -// worker, and the chain ids it carries are what every line that worker logs is -// stamped with, so a split run's output says which chain it came from without -// anyone having to rewrite it downstream. -let workerChainIds = - envSafe - ->EnvSafe.get("ENVIO_WORKER", S.option(S.string)) - ->Option.map(value => value->String.split(",")->Array.filterMap(Float.fromString)) +// Set by a supervisor on the processes it forks, and by nothing else. +let isWorker = envSafe->EnvSafe.get("ENVIO_WORKER", S.option(S.bool))->Option.getOr(false) let logStrategy = envSafe->EnvSafe.get( @@ -103,7 +97,6 @@ let logStrategy = Logging.setLogger( Logging.makeLogger( - ~base=Logging.makeBase(~workerChainIds), ~logStrategy, ~logFilePath, ~defaultFileLogLevel, diff --git a/packages/envio/src/Logging.res b/packages/envio/src/Logging.res index 358784b83..8b902e032 100644 --- a/packages/envio/src/Logging.res +++ b/packages/envio/src/Logging.res @@ -25,23 +25,11 @@ let logLevels = [ ]->Dict.fromArray %%private(let logger = ref(None)) +// The logger as configured, before any context a run added to it. Kept so +// setting context twice in one process replaces it rather than stacking it. +%%private(let rootLogger = ref(None)) -// The fields every line a process logs carries. Empty is what keeps pid and -// hostname out. A worker names the chains it drives, so a split run's output -// says which chain a line came from wherever the call site had no chain in hand. -let makeBase = (~workerChainIds: option>): JSON.t => - switch workerChainIds { - | Some([chainId]) => JSON.Object(Dict.fromArray([("chainId", JSON.Number(chainId))])) - | Some([]) | None => JSON.Object(Dict.make()) - | Some(chainIds) => - JSON.Object( - Dict.fromArray([ - ("chainIds", JSON.Array(chainIds->Array.map(chainId => JSON.Number(chainId)))), - ]), - ) - } - -let makeLogger = (~logStrategy, ~logFilePath, ~defaultFileLogLevel, ~userLogLevel, ~base) => { +let makeLogger = (~logStrategy, ~logFilePath, ~defaultFileLogLevel, ~userLogLevel) => { // Currently unused - useful if using multiple transports. // let pinoRaw = {"target": "pino/file", "level": Config.userLogLevel} let pinoFile: Transport.transportTarget = { @@ -54,6 +42,9 @@ let makeLogger = (~logStrategy, ~logFilePath, ~defaultFileLogLevel, ~userLogLeve level: defaultFileLogLevel, } + // Empty base disables pid and hostname in logs + let base: JSON.t = %raw("{}") + let makeMultiStreamLogger = MultiStreamLogger.make( ~userLogLevel, ~defaultFileLogLevel, @@ -96,6 +87,7 @@ let makeLogger = (~logStrategy, ~logFilePath, ~defaultFileLogLevel, ~userLogLeve } let setLogger = l => { + rootLogger := Some(l) logger := Some(l) } @@ -163,6 +155,14 @@ let childFatal = (logger, params: 'a) => { let createChild = (~params: 'a) => { getLogger()->child(params->createChildParams) } +// Fields every line this process logs from here on carries. What belongs on +// them is the run's to decide; the logger only carries what it is handed. +let setContext = (params: 'a) => + switch rootLogger.contents { + | Some(root) => logger := Some(root->child(params->createChildParams)) + | None => () + } + let createChildFrom = (~logger: t, ~params: 'a) => { logger->child(params->createChildParams) } diff --git a/packages/envio/src/Main.res b/packages/envio/src/Main.res index 2758a44c0..2ab0932ca 100644 --- a/packages/envio/src/Main.res +++ b/packages/envio/src/Main.res @@ -658,6 +658,9 @@ let start = async ( // Initialize persistence first so the exported indexer value contains state from the database // when handler files are loaded (they may access the indexer at module top level). let config = Config.load() + // In per-chain mode every line this process writes belongs to the chains it + // drives, whether or not a supervisor split the run across processes. + config->Config.logContext->Option.forEach(Logging.setContext) // isDevelopmentMode controls whether the indexer stays alive after all // chains finish (keepProcessAlive) and whether the console API is exposed. // Set by `envio dev` via the public config's `isDev` field; `envio start` diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index e53a9dcc6..447f9263c 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -78,7 +78,7 @@ let plan = (~chainIds: array, ~maxConnections: int): option - if config.isolated || config.userEntities->Array.some(entity => entity.crossChain) { + if config.isolated || !(config->Config.isPerChain) { None } else { plan(~chainIds=config.chainMap->ChainMap.values->Array.map(chain => chain.id), ~maxConnections) @@ -138,9 +138,7 @@ let fork = ( ~entryPath=NodeJs.Process.argv->Array.getUnsafe(1), ) => { let env = NodeJs.Process.process.env->Dict.copy - // Marks the process a worker, and names the chains it drives: its logger - // stamps them onto every line it writes. - env->Dict.set("ENVIO_WORKER", worker.chainIds->Array.map(ChainId.toString)->Array.joinUnsafe(",")) + env->Dict.set("ENVIO_WORKER", "true") // The worker's slice of the budget. Read when the worker's own Env module // loads, which is why it rides in the spawn environment rather than a message. env->Dict.set("ENVIO_PG_MAX_CONNECTIONS", worker.maxConnections->Int.toString) diff --git a/packages/envio/src/Worker.res b/packages/envio/src/Worker.res index 27b7e3387..687bbea53 100644 --- a/packages/envio/src/Worker.res +++ b/packages/envio/src/Worker.res @@ -4,7 +4,7 @@ // Set by the supervisor on the processes it forks. An indexer a user started // themselves never has it, and takes every path it takes today. -let isEnabled = Env.workerChainIds->Option.isSome +let isEnabled = Env.isWorker @tag("kind") type parentMessage = From 6dd37b6f4d7863088689cb069b71fe51bfad64be Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 09:54:18 +0000 Subject: [PATCH 04/61] Name both conditions automatic splitting needs in the CLI help "affords two connections per process" reads as satisfied by the default of 2, which buys one process and no split, and the per-chain schema requirement sat in a sentence about `--chain`'s own prerequisites. Say both outright, and say what a shared-entity schema or a smaller budget does instead. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- packages/cli/CommandLineHelp.md | 2 +- packages/cli/src/cli_args/clap_definitions.rs | 6 ++++-- 2 files changed, 5 insertions(+), 3 deletions(-) diff --git a/packages/cli/CommandLineHelp.md b/packages/cli/CommandLineHelp.md index 959a7aba2..d9dcbdf0a 100644 --- a/packages/cli/CommandLineHelp.md +++ b/packages/cli/CommandLineHelp.md @@ -378,7 +378,7 @@ Start the indexer. Runs codegen automatically before launching so the on-disk ty ###### **Options:** * `-r`, `--restart` — Clear your database and restart indexing from scratch -* `--chain ` — Index only this chain, leaving the others to their own `envio start --chain` processes. Only needed to place the chains yourself: a plain `envio start` already splits them across processes when `ENVIO_PG_MAX_CONNECTIONS` affords two connections per process, and manages those processes for you. Repeat the flag for several chains. Requires a schema whose entities are all per-chain, created for every chain by `envio local db-migrate up` before any process starts. Assign each configured chain to exactly one process, and give each its own `ENVIO_INDEXER_PORT`. Each process builds the indexes for its own chains and reports them ready as they catch up, independently of the others +* `--chain ` — Index only this chain, leaving the others to their own `envio start --chain` processes. Only needed to place the chains yourself: a plain `envio start` already splits them across processes, and manages those processes for you, whenever the schema's entities are all per-chain and `ENVIO_PG_MAX_CONNECTIONS` is at least 4 — two per process, for two processes. A schema with an entity shared across chains, or a smaller budget, runs in one process as it always has. Repeat the flag for several chains. Requires a schema whose entities are all per-chain, created for every chain by `envio local db-migrate up` before any process starts. Assign each configured chain to exactly one process, and give each its own `ENVIO_INDEXER_PORT`. Each process builds the indexes for its own chains and reports them ready as they catch up, independently of the others diff --git a/packages/cli/src/cli_args/clap_definitions.rs b/packages/cli/src/cli_args/clap_definitions.rs index 0f1e9c135..f58968583 100644 --- a/packages/cli/src/cli_args/clap_definitions.rs +++ b/packages/cli/src/cli_args/clap_definitions.rs @@ -146,8 +146,10 @@ pub struct StartArgs { ///Index only this chain, leaving the others to their own `envio start --chain` processes. ///Only needed to place the chains yourself: a plain `envio start` already splits them across - ///processes when `ENVIO_PG_MAX_CONNECTIONS` affords two connections per process, and manages - ///those processes for you. + ///processes, and manages those processes for you, whenever the schema's entities are all + ///per-chain and `ENVIO_PG_MAX_CONNECTIONS` is at least 4 — two per process, for two processes. + ///A schema with an entity shared across chains, or a smaller budget, runs in one process as + ///it always has. ///Repeat the flag for several chains. Requires a schema whose entities are all per-chain, ///created for every chain by `envio local db-migrate up` before any process starts. ///Assign each configured chain to exactly one process, and give each its own From 996e103f98b4db209605040259fa88a0f88c5e81 Mon Sep 17 00:00:00 2001 From: dzakh Date: Wed, 16 Sep 2026 10:23:47 +0000 Subject: [PATCH 05/61] End the supervisor with its group, deal chains in config order, and scope log context to isolated runs A supervisor whose workers had all exited stayed up: its server and signal handlers kept the event loop alive, so `envio start` never returned after a SIGTERM, a Ctrl-C, or every chain reaching its end block, where a single process exits. `awaitExit` now reports whether the group finished on its own or was stopped, and the run exits on either, keeping the process only for a display to hold the final state, the way a single process does. The hard-coded chain volume ranking is gone. How much work a chain has is the contracts' to decide, not the chain's, and a fixed table gave the operator no way to correct a layout it got wrong. Chains are dealt in config order, still reversing direction each pass, so ordering chains busiest-first in config.yaml is what balances the split. Log context now applies only to an isolated run. One process driving every chain has nothing to tell its lines apart from, and its chain-scoped lines already carry `chainId`; stamping `chainIds` on every line of every per-chain indexer only added bytes. The worker reads its init payload with a one-shot listener rather than a handler that stayed registered for the process's life. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_017CceLz4mPquMmpWrU74P3j --- .../envio-tests/test/SupervisorFork_test.res | 40 +++++++---- .../test/lib_tests/Supervisor_test.res | 32 +++++---- packages/envio/src/Config.res | 9 +-- packages/envio/src/Supervisor.res | 70 +++++++------------ packages/envio/src/Worker.res | 20 +++--- packages/envio/src/bindings/NodeJs.res | 3 +- 6 files changed, 86 insertions(+), 88 deletions(-) diff --git a/packages/envio-tests/test/SupervisorFork_test.res b/packages/envio-tests/test/SupervisorFork_test.res index f44e51e3e..fbace3981 100644 --- a/packages/envio-tests/test/SupervisorFork_test.res +++ b/packages/envio-tests/test/SupervisorFork_test.res @@ -47,21 +47,36 @@ describe("Supervisor.fork", () => { }) describe("Supervisor.awaitExit", () => { - Async.it("Returns once every worker has finished on its own", async t => { + let outcome = async group => + switch await group->Supervisor.awaitExit { + | outcome => Ok(outcome) + | exception _ => Error() + } + + Async.it("Reports a group whose every worker finished on its own", async t => { NodeJs.Process.process.env->Dict.set("FAKE_WORKER", "succeed") let group: Supervisor.group = { running: [forkFixture(~chainIds=[1]), forkFixture(~chainIds=[137])], stopping: false, } - let outcome = switch await group->Supervisor.awaitExit { - | () => "returned" - | exception _ => "threw" - } - - t.expect(outcome).toBe("returned") + t.expect(await outcome(group)).toStrictEqual(Ok(Supervisor.Finished)) }) + Async.it( + "Reports a group its supervisor took down as stopped, whatever the exit codes", + async t => { + NodeJs.Process.process.env->Dict.set("FAKE_WORKER", "linger") + let group: Supervisor.group = { + running: [forkFixture(~chainIds=[1]), forkFixture(~chainIds=[137])], + stopping: false, + } + group->Supervisor.stop + + t.expect(await outcome(group)).toStrictEqual(Ok(Supervisor.Stopped)) + }, + ) + Async.it("Stops the group and fails the run when one worker dies", async t => { NodeJs.Process.process.env->Dict.set("FAKE_WORKER", "fail") let failing = forkFixture(~chainIds=[1]) @@ -69,12 +84,11 @@ describe("Supervisor.awaitExit", () => { let lingering = forkFixture(~chainIds=[137]) let group: Supervisor.group = {running: [failing, lingering], stopping: false} - let outcome = switch await group->Supervisor.awaitExit { - | () => "returned" - | exception _ => "threw" - } - // The survivor was taken down rather than left indexing half a schema. - t.expect((outcome, group.stopping, lingering.settled)).toStrictEqual(("threw", true, true)) + t.expect((await outcome(group), group.stopping, lingering.settled)).toStrictEqual(( + Error(), + true, + true, + )) }) }) diff --git a/packages/envio-tests/test/lib_tests/Supervisor_test.res b/packages/envio-tests/test/lib_tests/Supervisor_test.res index f374c2c9e..792cd1827 100644 --- a/packages/envio-tests/test/lib_tests/Supervisor_test.res +++ b/packages/envio-tests/test/lib_tests/Supervisor_test.res @@ -37,26 +37,25 @@ describe("Supervisor.plan", () => { }) }) -describe("Supervisor.plan volume spreading", () => { +describe("Supervisor.plan dealing order", () => { let assignment = (~chainIds, ~maxConnections) => Supervisor.plan(~chainIds=chainIds->Array.map(ChainId.fromInt), ~maxConnections) ->Option.getOrThrow ->Array.map(worker => worker.chainIds->Array.map(ChainId.toString)->Array.joinUnsafe(",")) - it("Keeps the busiest chains apart, and pairs them with the quietest", t => { + it("Deals chains in config order, reversing direction each pass", t => { t.expect([ - // Ethereum and BNB are the two busiest of these, so they lead different - // workers, and each is paired with the lighter of the remaining two. + // The first two lead different workers; the worker that took the first + // picks up the last. Config order is the ranking, not the chain ids. assignment(~chainIds=[8453, 56, 42161, 1], ~maxConnections=4), - // Unranked chains sort behind the ranked ones, keeping config order. - assignment(~chainIds=[999, 137, 888, 1], ~maxConnections=4), - // Three workers take the top three, then fold back so the heaviest - // worker picks up the lightest chain. + // Three workers take the first three, then fold back. assignment(~chainIds=[1, 56, 137, 8453, 42161, 10], ~maxConnections=6), + // An odd count leaves the fold short: the last chain lands mid-pass. + assignment(~chainIds=[1, 56, 137, 8453, 42161], ~maxConnections=4), ]).toStrictEqual([ - ["1,42161", "56,8453"], - ["1,888", "137,999"], + ["8453,1", "56,42161"], ["1,10", "56,42161", "137,8453"], + ["1,8453,42161", "56,137"], ]) }) }) @@ -167,13 +166,17 @@ describe("Supervisor worker plumbing", () => { }) describe("Config.logContext", () => { - it("Attributes a per-chain run's logs to its chains, and a shared one's to none", t => { + it("Attributes an isolated run's logs to its chains, and any other run's to none", t => { t.expect([ - // Every entity is per-chain, and this process drives one of them. + // This process drives one of the schema's chains while siblings drive the rest. config(~schema=perChain, ~isolatedChains=[JSON.Number(137.)])->Config.logContext, - // Still per-chain, but this process drives both: no single chain owns a line. + config( + ~schema=perChain, + ~isolatedChains=[JSON.Number(1.), JSON.Number(137.)], + )->Config.logContext, + // One process driving every chain: chain-scoped lines already name theirs, + // and the rest belong to the run as a whole. config(~schema=perChain)->Config.logContext, - // An entity shared across chains: the work isn't any one chain's. config(~schema=crossChain)->Config.logContext, ]).toStrictEqual([ Some(JSON.Object(Dict.fromArray([("chainId", JSON.Number(137.))]))), @@ -183,6 +186,7 @@ describe("Config.logContext", () => { ), ), None, + None, ]) }) }) diff --git a/packages/envio/src/Config.res b/packages/envio/src/Config.res index ea404a1c7..82eab5603 100644 --- a/packages/envio/src/Config.res +++ b/packages/envio/src/Config.res @@ -627,13 +627,14 @@ let getChain = (config, ~chainId) => // Whether every entity belongs to exactly one chain. Only then is a unit of // this indexer's work attributable to a chain at all, which is what lets a run -// be split across processes and what lets its logs name a chain. +// be split across processes. let isPerChain = (config: t) => !(config.userEntities->Array.some(entity => entity.crossChain)) -// What every line this indexer logs is attributed to: the chains it drives. -// A schema shared across chains has none, since its work is no single chain's. +// What every line this process logs is attributed to: the chains it drives, +// when sibling processes drive the rest. A process driving every chain has +// nothing to tell apart from, and its chain-scoped lines already name theirs. let logContext = (config: t): option => - if config->isPerChain { + if config.isolated { let chainIds = config.chainMap->ChainMap.keys Some( switch chainIds { diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index 447f9263c..879f6e49e 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -6,63 +6,27 @@ type worker = {chainIds: array, maxConnections: int} // on a single one, so the budget buys workers two at a time. let minConnectionsPerWorker = 2 -// Roughly how much a chain costs to index, busiest first. An estimate only: -// it decides nothing but which worker a chain lands on, so a chain in the wrong -// place, or missing from the list entirely, costs balance and nothing else. -// Chains it doesn't name sort behind the ones it does, in config order. -let byDescendingVolume = [ - 1, // Ethereum - 56, // BNB Smart Chain - 137, // Polygon - 8453, // Base - 42161, // Arbitrum One - 10, // Optimism - 43114, // Avalanche - 81457, // Blast - 59144, // Linea - 534352, // Scroll - 324, // zkSync Era - 5000, // Mantle - 204, // opBNB - 100, // Gnosis - 42220, // Celo -] - -let volumeRank = (chainId: ChainId.t) => - switch byDescendingVolume->Array.indexOf(chainId->ChainId.toInt) { - | -1 => byDescendingVolume->Array.length - | rank => rank - } - // How to spend a connection budget on the chains a run indexes. `None` keeps // the run in one process, which is what a budget too small to afford two // workers, or a config with nothing to split, has to do. // -// Chains are dealt busiest-first and the direction reverses each pass, so the -// heaviest chains lead different workers and the worker that took the heaviest -// picks up the lightest. Volume is only ever an estimate, which is why the -// layout it produces is a starting balance rather than a guarantee. +// Chains are dealt in config order and the direction reverses each pass, so +// the first chains lead different workers and the worker that took the first +// picks up the last. How much work a chain has is the contracts' to decide, +// not the chain's, so config order is the one ranking the run can be given: +// listing chains busiest-first in config.yaml is what balances the layout. let plan = (~chainIds: array, ~maxConnections: int): option> => { let workerCount = Pervasives.min(chainIds->Array.length, maxConnections / minConnectionsPerWorker) if workerCount < 2 { None } else { - let byVolume = - chainIds - ->Array.mapWithIndex((chainId, configIndex) => (chainId, chainId->volumeRank, configIndex)) - ->Array.toSorted(((_, aRank, aIndex), (_, bRank, bIndex)) => - // Two chains the list doesn't rank keep the order config gave them. - aRank === bRank ? Int.compare(aIndex, bIndex) : Int.compare(aRank, bRank) - ) - ->Array.map(((chainId, _, _)) => chainId) - // The remainder is handed out one connection at a time rather than left // unspent, so a budget with slack widens the earliest workers' pools. let evenShare = maxConnections / workerCount let remainder = mod(maxConnections, workerCount) Some( Array.fromInitializer(~length=workerCount, workerIndex => { - chainIds: byVolume->Array.filterWithIndex((_, dealIndex) => { + chainIds: chainIds->Array.filterWithIndex((_, dealIndex) => { let position = mod(dealIndex, workerCount) let isReversePass = mod(dealIndex / workerCount, 2) === 1 (isReversePass ? workerCount - 1 - position : position) === workerIndex @@ -173,11 +137,15 @@ let stop = group => { group.running->Array.forEach(r => r.child->NodeJs.ChildProcess.kill("SIGTERM")->ignore) } +// How a group ended. `Finished` is every worker exiting cleanly on its own, +// which is what indexing to every end block looks like. +type outcome = Finished | Stopped + // Resolves once every worker has ended. Throws if any of them ended in a way // the supervisor didn't ask for, having first taken the rest down: one worker // short leaves its chains unindexed, and a run that kept the others going would // look healthy while falling behind. -let awaitExit = async group => { +let awaitExit = async (group): outcome => { let failed = ref(false) let alive = ref(group.running->Array.length) @@ -216,6 +184,7 @@ let awaitExit = async group => { if failed.contents { JsError.throwWithMessage("An indexer process exited with a failure. Stopped the others.") } + group.stopping ? Stopped : Finished } // Runs the group: creates the schema for every chain, forks a worker per plan @@ -287,7 +256,8 @@ let run = async (~workers: array, ~configJson: JSON.t, ~reset) => { }, ) - if Main.shouldUseTui() { + let shouldUseTui = Main.shouldUseTui() + if shouldUseTui { let _rerender = Tui.start(~config, ~getMetrics=() => reported()->merge) } @@ -296,5 +266,15 @@ let run = async (~workers: array, ~configJson: JSON.t, ~reset) => { NodeJs.Process.onSignal("SIGTERM", () => group->stop) NodeJs.Process.onSignal("SIGINT", () => group->stop) - await group->awaitExit + // The server and the signal handlers would keep this process up after its + // last worker is gone, so the group's end has to end the process. A display + // is the exception, as it is for a single process: it keeps the final state + // on screen until the terminal closes it. + switch await group->awaitExit { + | Stopped => NodeJs.process->NodeJs.exitWithCode(Success) + | Finished if !shouldUseTui => + Logging.info("Exiting with success") + NodeJs.process->NodeJs.exitWithCode(Success) + | Finished => () + } } diff --git a/packages/envio/src/Worker.res b/packages/envio/src/Worker.res index 687bbea53..0a47fd20c 100644 --- a/packages/envio/src/Worker.res +++ b/packages/envio/src/Worker.res @@ -25,12 +25,10 @@ let snapshotIntervalMillis = 500 // it: the supervisor could have died before it ever got to tear the group down. // Losing the channel is that signal. let exitWithSupervisor = () => - if isEnabled { - NodeJs.Process.onDisconnect(() => { - Logging.error("The indexer supervisor is gone. Stopping this chain's process.") - NodeJs.process->NodeJs.exitWithCode(Failure) - }) - } + NodeJs.Process.onDisconnect(() => { + Logging.error("The indexer supervisor is gone. Stopping this chain's process.") + NodeJs.process->NodeJs.exitWithCode(Failure) + }) let send = (message: workerMessage) => if isEnabled { @@ -39,14 +37,14 @@ let send = (message: workerMessage) => let onParentMessage = (handle: parentMessage => unit) => NodeJs.Process.onMessage(handle) -// Resolves with the init payload the supervisor sends immediately after the -// fork. Nothing else can run first: the worker has no config until it lands. +// Resolves with the init payload. The supervisor sends it right after the +// fork, before anything else, so the first message is the only one to read. let awaitInit = (): promise => - Promise.make((resolve, _) => - onParentMessage(message => + Promise.make((resolve, reject) => + NodeJs.Process.onceMessage((message: parentMessage) => switch message { | Init({config}) => resolve(config) - | SyncCache(_) => () + | SyncCache(_) => reject(Utils.Error.make("Expected the supervisor's init message first")) } ) ) diff --git a/packages/envio/src/bindings/NodeJs.res b/packages/envio/src/bindings/NodeJs.res index 15e130487..b45f00637 100644 --- a/packages/envio/src/bindings/NodeJs.res +++ b/packages/envio/src/bindings/NodeJs.res @@ -71,6 +71,8 @@ module Process = { @val @scope("process") external sendToParent: 'msg => bool = "send" @val @scope("process") external onMessage: (@as("message") _, 'msg => unit) => unit = "on" + @val @scope("process") + external onceMessage: (@as("message") _, 'msg => unit) => unit = "once" @val @scope("process") external onSignal: (string, unit => unit) => unit = "on" @val @scope("process") external onDisconnect: (@as("disconnect") _, unit => unit) => unit = "on" @@ -170,7 +172,6 @@ module ChildProcess = { external onExit: (child, @as("exit") _, (Null.t, Null.t) => unit) => unit = "on" @send external onChildError: (child, @as("error") _, exn => unit) => unit = "on" @send external kill: (child, string) => bool = "kill" - @get external pid: child => Null.t = "pid" } module Url = { From 549f611c3d0ed078c32a65890f2102970cdb4519 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 10:33:24 +0000 Subject: [PATCH 06/61] Answer a supervised cache sync once its workers have dumped The console's /console/syncCache resolved as soon as the supervisor had sent the request, so a split run reported a dump that was still being written. Workers now acknowledge the dump, and the supervisor waits for every one of them. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- .../envio-tests/test/SupervisorFork_test.res | 23 +++++++++ .../envio-tests/test/helpers/fakeWorker.mjs | 7 ++- packages/envio/src/Main.res | 5 +- packages/envio/src/Supervisor.res | 48 +++++++++++++------ packages/envio/src/Worker.res | 6 ++- 5 files changed, 71 insertions(+), 18 deletions(-) diff --git a/packages/envio-tests/test/SupervisorFork_test.res b/packages/envio-tests/test/SupervisorFork_test.res index fbace3981..14872d382 100644 --- a/packages/envio-tests/test/SupervisorFork_test.res +++ b/packages/envio-tests/test/SupervisorFork_test.res @@ -9,6 +9,9 @@ type fixtureReport = { startTime: Date.t, } +// What it reports once its dump is done. +type fixtureDump = {synced: bool} + let fixturePath = `${NodeJs.Process.cwd()}/test/helpers/fakeWorker.mjs` let forkFixture = (~chainIds, ~maxConnections=2, ~workerIndex=0) => @@ -30,6 +33,7 @@ describe("Supervisor.fork", () => { switch message { | Worker.Snapshot({metrics}) => resolve(metrics->(Utils.magic: Metrics.t => fixtureReport)) + | Worker.CacheSynced(_) => () }, ), ) @@ -46,6 +50,25 @@ describe("Supervisor.fork", () => { }) }) +describe("Supervisor.syncCache", () => { + Async.it("Answers only once every worker has dumped its cache", async t => { + NodeJs.Process.process.env->Dict.set("FAKE_WORKER", "linger") + let group: Supervisor.group = { + running: [forkFixture(~chainIds=[1]), forkFixture(~chainIds=[137])], + stopping: false, + } + + await group->Supervisor.syncCache + let dumped = + group.running->Array.map(r => + r.snapshot->Option.map(metrics => (metrics->(Utils.magic: Metrics.t => fixtureDump)).synced) + ) + group->Supervisor.stop + + t.expect(dumped).toStrictEqual([Some(true), Some(true)]) + }) +}) + describe("Supervisor.awaitExit", () => { let outcome = async group => switch await group->Supervisor.awaitExit { diff --git a/packages/envio-tests/test/helpers/fakeWorker.mjs b/packages/envio-tests/test/helpers/fakeWorker.mjs index e6f7342bf..47251b576 100644 --- a/packages/envio-tests/test/helpers/fakeWorker.mjs +++ b/packages/envio-tests/test/helpers/fakeWorker.mjs @@ -20,7 +20,12 @@ process.on("message", (message) => { if (mode === "fail") process.exit(1); } if (message.kind === "syncCache") { - process.exit(0); + // Reports the dump through a snapshot before acknowledging it, so a + // supervisor that answers early can be caught having answered before it. + setTimeout(() => { + process.send({ kind: "snapshot", metrics: { synced: true } }); + process.send({ kind: "cacheSynced" }); + }, 50); } }); diff --git a/packages/envio/src/Main.res b/packages/envio/src/Main.res index 2ab0932ca..ba3346711 100644 --- a/packages/envio/src/Main.res +++ b/packages/envio/src/Main.res @@ -746,7 +746,10 @@ let start = async ( if Worker.isEnabled { Worker.onParentMessage(message => switch message { - | SyncCache(_) => dumpEffectCache()->Promise.ignore + | SyncCache(_) => + dumpEffectCache() + ->Promise.thenResolve(() => Worker.send(CacheSynced({}))) + ->Promise.ignore | Init(_) => () } ) diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index 879f6e49e..a75b3e93f 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -58,6 +58,8 @@ type running = { // A spawn failure can raise `error` and `exit` both, and a worker counted // twice would end the run while its siblings are still indexing. mutable settled: bool, + // Waiting for this worker's cache dump, when a console asked for one. + mutable onCacheSynced: option unit>, } let label = (worker: worker) => @@ -124,7 +126,20 @@ let fork = ( ->NodeJs.ChildProcess.send(Worker.Init({config: configJson->configForWorker(~worker)})) ->ignore - {worker, child, snapshot: None, settled: false} + let running = {worker, child, snapshot: None, settled: false, onCacheSynced: None} + child->NodeJs.ChildProcess.onMessage(message => + switch message { + | Worker.Snapshot({metrics}) => running.snapshot = Some(metrics) + | Worker.CacheSynced(_) => + switch running.onCacheSynced { + | Some(resolve) => + running.onCacheSynced = None + resolve() + | None => () + } + } + ) + running } // The forked workers of one run, and whether their supervisor is the one @@ -137,6 +152,22 @@ let stop = group => { group.running->Array.forEach(r => r.child->NodeJs.ChildProcess.kill("SIGTERM")->ignore) } +// Dumps every worker's effect cache. Resolves once they have all reported the +// dump done, so the console the supervisor serves can't answer for writes that +// are still in flight. +let syncCache = async group => { + let _: array = + await group.running + ->Array.filter(r => !r.settled) + ->Array.map(r => + Promise.make((resolve, _) => { + r.onCacheSynced = Some(() => resolve()) + r.child->NodeJs.ChildProcess.send(Worker.SyncCache({}))->ignore + }) + ) + ->Promise.all +} + // How a group ended. `Finished` is every worker exiting cleanly on its own, // which is what indexing to every end block looks like. type outcome = Finished | Stopped @@ -230,14 +261,6 @@ let run = async (~workers: array, ~configJson: JSON.t, ~reset) => { ~elapsedSeconds=startTimeRef->Performance.secondsSince, ) - group.running->Array.forEach(r => - r.child->NodeJs.ChildProcess.onMessage(message => - switch message { - | Worker.Snapshot({metrics}) => r.snapshot = Some(metrics) - } - ) - ) - Main.startServer( // Nothing to report until a worker has: the run reads as initializing // rather than as an indexer with no chains. @@ -248,12 +271,7 @@ let run = async (~workers: array, ~configJson: JSON.t, ~reset) => { }, ~envioVersion=Utils.EnvioPackage.value.version, ~isDevelopmentMode=config.isDev, - ~onSyncCache=() => { - group.running->Array.forEach(r => - r.child->NodeJs.ChildProcess.send(Worker.SyncCache({}))->ignore - ) - Promise.resolve() - }, + ~onSyncCache=() => group->syncCache, ) let shouldUseTui = Main.shouldUseTui() diff --git a/packages/envio/src/Worker.res b/packages/envio/src/Worker.res index 0a47fd20c..72afda61d 100644 --- a/packages/envio/src/Worker.res +++ b/packages/envio/src/Worker.res @@ -15,7 +15,11 @@ type parentMessage = | @as("syncCache") SyncCache({}) @tag("kind") -type workerMessage = | @as("snapshot") Snapshot({metrics: Metrics.t}) +type workerMessage = + | @as("snapshot") Snapshot({metrics: Metrics.t}) + // Sent once the worker's effect cache is on disk, so the supervisor's + // console can answer for a dump that has actually happened. + | @as("cacheSynced") CacheSynced({}) // How often a worker reports. Matches the TUI's own refresh, so the supervised // display moves at the same rate an unsplit run's does. From 0f12b89f442edb6579562214594251534a596121 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 10:45:16 +0000 Subject: [PATCH 07/61] Keep a cache sync from hanging on a busy or starting worker Two ways the supervised /console/syncCache could wait forever: a second request overwrote the first one's waiter, and a request that reached a worker before its handler was installed was dropped. Overlapping requests now join the dump already in flight, and a worker holds what arrives while it is still coming up. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- .../envio-tests/test/SupervisorFork_test.res | 26 ++++++++++++- .../test/lib_tests/Worker_test.res | 14 +++++++ packages/envio/src/Bin.res | 1 + packages/envio/src/Supervisor.res | 39 ++++++++++++------- packages/envio/src/Worker.res | 21 +++++++++- packages/envio/src/bindings/NodeJs.res | 2 + 6 files changed, 88 insertions(+), 15 deletions(-) create mode 100644 packages/envio-tests/test/lib_tests/Worker_test.res diff --git a/packages/envio-tests/test/SupervisorFork_test.res b/packages/envio-tests/test/SupervisorFork_test.res index 14872d382..9798d1c08 100644 --- a/packages/envio-tests/test/SupervisorFork_test.res +++ b/packages/envio-tests/test/SupervisorFork_test.res @@ -56,6 +56,7 @@ describe("Supervisor.syncCache", () => { let group: Supervisor.group = { running: [forkFixture(~chainIds=[1]), forkFixture(~chainIds=[137])], stopping: false, + syncing: None, } await group->Supervisor.syncCache @@ -67,6 +68,27 @@ describe("Supervisor.syncCache", () => { t.expect(dumped).toStrictEqual([Some(true), Some(true)]) }) + + Async.it("Answers requests that overlap rather than leaving the first hanging", async t => { + NodeJs.Process.process.env->Dict.set("FAKE_WORKER", "linger") + let group: Supervisor.group = { + running: [forkFixture(~chainIds=[1])], + stopping: false, + syncing: None, + } + + let answered = async request => + await Promise.race([ + request->Promise.thenResolve(() => "answered"), + Utils.delay(2000)->Promise.thenResolve(() => "still waiting"), + ]) + let first = group->Supervisor.syncCache + let second = group->Supervisor.syncCache + let outcomes = (await first->answered, await second->answered) + group->Supervisor.stop + + t.expect(outcomes).toStrictEqual(("answered", "answered")) + }) }) describe("Supervisor.awaitExit", () => { @@ -81,6 +103,7 @@ describe("Supervisor.awaitExit", () => { let group: Supervisor.group = { running: [forkFixture(~chainIds=[1]), forkFixture(~chainIds=[137])], stopping: false, + syncing: None, } t.expect(await outcome(group)).toStrictEqual(Ok(Supervisor.Finished)) @@ -93,6 +116,7 @@ describe("Supervisor.awaitExit", () => { let group: Supervisor.group = { running: [forkFixture(~chainIds=[1]), forkFixture(~chainIds=[137])], stopping: false, + syncing: None, } group->Supervisor.stop @@ -105,7 +129,7 @@ describe("Supervisor.awaitExit", () => { let failing = forkFixture(~chainIds=[1]) NodeJs.Process.process.env->Dict.set("FAKE_WORKER", "linger") let lingering = forkFixture(~chainIds=[137]) - let group: Supervisor.group = {running: [failing, lingering], stopping: false} + let group: Supervisor.group = {running: [failing, lingering], stopping: false, syncing: None} // The survivor was taken down rather than left indexing half a schema. t.expect((await outcome(group), group.stopping, lingering.settled)).toStrictEqual(( diff --git a/packages/envio-tests/test/lib_tests/Worker_test.res b/packages/envio-tests/test/lib_tests/Worker_test.res new file mode 100644 index 000000000..e18ebfc4c --- /dev/null +++ b/packages/envio-tests/test/lib_tests/Worker_test.res @@ -0,0 +1,14 @@ +open Vitest + +describe("Worker.onParentMessage", () => { + Async.it("Holds a request that arrived before the indexer could handle it", async t => { + Worker.listen() + NodeJs.Process.emitMessage(Worker.SyncCache({}))->ignore + + let handled = [] + Worker.onParentMessage(message => handled->Array.push(message)) + NodeJs.Process.emitMessage(Worker.SyncCache({}))->ignore + + t.expect(handled).toStrictEqual([Worker.SyncCache({}), Worker.SyncCache({})]) + }) +}) diff --git a/packages/envio/src/Bin.res b/packages/envio/src/Bin.res index af572c2d9..2b2b4b142 100644 --- a/packages/envio/src/Bin.res +++ b/packages/envio/src/Bin.res @@ -53,6 +53,7 @@ let run = async args => { try { if Worker.isEnabled { Worker.exitWithSupervisor() + Worker.listen() // A worker is handed the config its supervisor already parsed, narrowed to // the chains it drives, so the two can't disagree about what is indexed. // Its working directory and environment came with the fork. diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index a75b3e93f..cec2fff5c 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -145,7 +145,13 @@ let fork = ( // The forked workers of one run, and whether their supervisor is the one // taking them down. A stop it asked for is expected; every other way a worker // can end is a failure. -type group = {running: array, mutable stopping: bool} +type group = { + running: array, + mutable stopping: bool, + // The dump in flight, if any. A second request joins it rather than asking + // for a dump of its own, so overlapping requests both get an answer. + mutable syncing: option>, +} let stop = group => { group.stopping = true @@ -155,18 +161,24 @@ let stop = group => { // Dumps every worker's effect cache. Resolves once they have all reported the // dump done, so the console the supervisor serves can't answer for writes that // are still in flight. -let syncCache = async group => { - let _: array = - await group.running - ->Array.filter(r => !r.settled) - ->Array.map(r => - Promise.make((resolve, _) => { - r.onCacheSynced = Some(() => resolve()) - r.child->NodeJs.ChildProcess.send(Worker.SyncCache({}))->ignore - }) - ) - ->Promise.all -} +let syncCache = group => + switch group.syncing { + | Some(inFlight) => inFlight + | None => + let inFlight = + group.running + ->Array.filter(r => !r.settled) + ->Array.map(r => + Promise.make((resolve, _) => { + r.onCacheSynced = Some(() => resolve()) + r.child->NodeJs.ChildProcess.send(Worker.SyncCache({}))->ignore + }) + ) + ->Promise.all + ->Promise.thenResolve(_ => group.syncing = None) + group.syncing = Some(inFlight) + inFlight + } // How a group ended. `Finished` is every worker exiting cleanly on its own, // which is what indexing to every end block looks like. @@ -250,6 +262,7 @@ let run = async (~workers: array, ~configJson: JSON.t, ~reset) => { worker->fork(~workerIndex, ~configJson) ), stopping: false, + syncing: None, } let reported = () => group.running->Array.filterMap(r => r.snapshot) diff --git a/packages/envio/src/Worker.res b/packages/envio/src/Worker.res index 72afda61d..62a3e0d78 100644 --- a/packages/envio/src/Worker.res +++ b/packages/envio/src/Worker.res @@ -39,7 +39,26 @@ let send = (message: workerMessage) => NodeJs.Process.sendToParent(message)->ignore } -let onParentMessage = (handle: parentMessage => unit) => NodeJs.Process.onMessage(handle) +%%private(let pending: ref> = ref([])) +%%private(let handler: ref unit>> = ref(None)) + +// Installed before the indexer starts. A request can reach a worker while it is +// still coming up, and a dropped one leaves the supervisor waiting for an answer +// that will never be sent, so it waits for its handler instead. +let listen = () => + NodeJs.Process.onMessage((message: parentMessage) => + switch handler.contents { + | Some(handle) => handle(message) + | None => pending := pending.contents->Array.concat([message]) + } + ) + +let onParentMessage = (handle: parentMessage => unit) => { + handler := Some(handle) + let held = pending.contents + pending := [] + held->Array.forEach(handle) +} // Resolves with the init payload. The supervisor sends it right after the // fork, before anything else, so the first message is the only one to read. diff --git a/packages/envio/src/bindings/NodeJs.res b/packages/envio/src/bindings/NodeJs.res index b45f00637..754651069 100644 --- a/packages/envio/src/bindings/NodeJs.res +++ b/packages/envio/src/bindings/NodeJs.res @@ -77,6 +77,8 @@ module Process = { @val @scope("process") external onDisconnect: (@as("disconnect") _, unit => unit) => unit = "on" @val @scope("process") external argv: array = "argv" + @val @scope("process") + external emitMessage: (@as("message") _, 'msg) => bool = "emit" } module Buffer = { From b944c5a5c075c72f2297e0f6933105d2c320e237 Mon Sep 17 00:00:00 2001 From: dzakh Date: Wed, 16 Sep 2026 12:59:37 +0000 Subject: [PATCH 08/61] Mark workers by their fork, stop them only through the supervisor, and report every process's runtime metrics A worker is now the process the supervisor forked: told by the argument it was forked with together with the fork's own channel, rather than by an environment variable of its own. A user who types the argument starts nothing, and a process manager that forks with a channel changes nothing. A terminal's interrupt reaches the workers as well as the supervisor. Workers now leave it alone: the supervisor is the one that stops them, so a worker gone on its own can no longer read as a failure to a supervisor still deciding what the interrupt meant. A finished run held up by its display exits on the next stop signal instead of holding it, since the supervisor's own handler had taken over from Node's default exit. `/metrics/runtime` reported the supervisor's heap and event loop alone. Every process of the run now reports under a `worker` label, the supervisor included, from the runtime sample each worker sends with its snapshot. Log context names the one chain an isolated process drives and nothing else; a process driving several has no single owner to name. The split_test scenario and its e2e suite drive a real split run: two chains over a per-chain schema, a budget of four connections, every process on one metrics endpoint, one exit once both chains are done, and one interrupt to the supervisor ending them all. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_017CceLz4mPquMmpWrU74P3j --- packages/e2e-tests/src/e2e/split-run.test.ts | 129 +++++++++ .../test/lib_tests/Metrics_test.res | 113 ++++++++ .../test/lib_tests/Supervisor_test.res | 21 +- packages/envio/src/Bin.res | 2 +- packages/envio/src/Config.res | 29 +- packages/envio/src/Env.res | 3 - packages/envio/src/Main.res | 10 +- packages/envio/src/Metrics.res | 260 +++++++++++------- packages/envio/src/Supervisor.res | 39 ++- packages/envio/src/Worker.res | 30 +- packages/envio/src/bindings/NodeJs.res | 2 + pnpm-lock.yaml | 13 + scenarios/split_test/config.head.yaml | 22 ++ scenarios/split_test/config.yaml | 24 ++ scenarios/split_test/envio-env.d.ts | 7 + scenarios/split_test/package.json | 20 ++ scenarios/split_test/schema.graphql | 7 + scenarios/split_test/src/handlers/ERC20.ts | 11 + scenarios/split_test/tsconfig.json | 21 ++ 19 files changed, 614 insertions(+), 149 deletions(-) create mode 100644 packages/e2e-tests/src/e2e/split-run.test.ts create mode 100644 scenarios/split_test/config.head.yaml create mode 100644 scenarios/split_test/config.yaml create mode 100644 scenarios/split_test/envio-env.d.ts create mode 100644 scenarios/split_test/package.json create mode 100644 scenarios/split_test/schema.graphql create mode 100644 scenarios/split_test/src/handlers/ERC20.ts create mode 100644 scenarios/split_test/tsconfig.json diff --git a/packages/e2e-tests/src/e2e/split-run.test.ts b/packages/e2e-tests/src/e2e/split-run.test.ts new file mode 100644 index 000000000..681e142ed --- /dev/null +++ b/packages/e2e-tests/src/e2e/split-run.test.ts @@ -0,0 +1,129 @@ +/** + * A plain `envio start` over a per-chain schema, with a connection budget that + * affords two processes, splits the chains across forked workers and stays + * one indexer to the operator: one metrics endpoint over every process, one + * exit once every chain reaches its end block, one signal to stop it all. + * + * Needs Postgres but no Docker: it drives `envio start` with Hasura disabled. + */ + +import { describe, it, expect, beforeAll, afterAll } from "vitest"; +import { ChildProcess } from "child_process"; +import path from "path"; +import { config } from "../config.js"; +import { runCommand, startBackground, waitForOutput } from "../utils/process.js"; +import { pgRows, closePg, isPgReachable } from "../utils/pg-direct.js"; + +const PROJECT_DIR = path.join(config.scenariosDir, "split_test"); +const PG_SCHEMA = "e2e_split_run"; +const PORT = 9897; + +const indexerEnv = { + ENVIO_PG_SCHEMA: PG_SCHEMA, + // Two processes' worth: the split's own condition. + ENVIO_PG_MAX_CONNECTIONS: "4", + ENVIO_HASURA: "false", + ENVIO_TUI: "false", + ENVIO_INDEXER_PORT: String(PORT), + ENVIO_API_TOKEN: process.env.ENVIO_API_TOKEN ?? "", +}; + +const reachable = await isPgReachable(); + +if (!reachable && process.env.CI) { + throw new Error( + "Postgres is unreachable, so the split-run suite cannot run. Refusing to skip it in CI." + ); +} + +const exitCode = (child: ChildProcess) => + new Promise((resolve) => child.on("close", resolve)); + +/** Polls an endpoint of the supervisor until its body satisfies `ready`. */ +const scrapeUntil = async (route: string, ready: (body: string) => boolean) => { + const deadline = Date.now() + config.timeouts.indexerStartup; + let body = ""; + while (Date.now() < deadline) { + try { + body = await (await fetch(`http://localhost:${PORT}${route}`)).text(); + if (ready(body)) return body; + } catch {} + await new Promise((r) => setTimeout(r, 500)); + } + throw new Error(`Timed out scraping ${route}\n--- last body ---\n${body}`); +}; + +const start = (args: string[]) => + startBackground(config.envioCommand, [...config.envioArgs, "start", ...args], { + cwd: PROJECT_DIR, + env: indexerEnv, + }); + +describe.skipIf(!reachable)("E2E: a split run is one indexer", () => { + beforeAll(async () => { + const codegen = await runCommand( + config.envioCommand, + [...config.envioArgs, "codegen"], + { cwd: PROJECT_DIR, env: indexerEnv, timeout: config.timeouts.codegen } + ); + expect(codegen.exitCode, `codegen failed: ${codegen.stderr}`).toBe(0); + }, config.timeouts.codegen); + + afterAll(async () => { + await closePg(); + }); + + it("Exits once every chain is done, with both chains' rows written", async () => { + const indexer = start(["-r"]); + const exit = exitCode(indexer); + await waitForOutput(indexer, "Splitting 2 chains across 2 processes", config.timeouts.indexerStartup); + + expect({ + exitCode: await exit, + rowsPerChain: await pgRows( + `SELECT "chain_id", COUNT(*) > 0 FROM "${PG_SCHEMA}"."Transfer" GROUP BY "chain_id" ORDER BY "chain_id"` + ), + }).toEqual({ + exitCode: 0, + rowsPerChain: [ + [1, true], + [8453, true], + ], + }); + }); + + // The same chains with no end block: a run that nothing but a stop ends, so + // there is time to read what it serves. + it("Serves every process's metrics, and stops them all on one interrupt", async () => { + const indexer = start(["-r", "--config", "config.head.yaml"]); + const exit = exitCode(indexer); + await waitForOutput(indexer, "Splitting 2 chains across 2 processes", config.timeouts.indexerStartup); + + const [runtime, metrics] = await Promise.all([ + // The supervisor's own readings alongside each worker's, told apart by label. + scrapeUntil("/metrics/runtime", (body) => body.includes('worker="1"')), + // Both chains on one endpoint, whichever process drives each. + scrapeUntil( + "/metrics", + (body) => body.includes('chainId="1"') && body.includes('chainId="8453"') + ), + ]); + + // Only the supervisor is signalled, the way a process manager would. + indexer.kill("SIGINT"); + + expect({ + exitCode: await exit, + runtimeWorkers: ["supervisor", "0", "1"].map((worker) => + runtime.includes(`nodejs_heap_size_used_bytes{worker="${worker}"}`) + ), + metricsChains: [1, 8453].map((chainId) => + metrics.includes(`envio_progress_block{chainId="${chainId}"}`) + ), + }).toEqual({ + exitCode: 0, + runtimeWorkers: [true, true, true], + metricsChains: [true, true], + }); + }); +}); diff --git a/packages/envio-tests/test/lib_tests/Metrics_test.res b/packages/envio-tests/test/lib_tests/Metrics_test.res index f3c3abfdb..0cd7cc798 100644 --- a/packages/envio-tests/test/lib_tests/Metrics_test.res +++ b/packages/envio-tests/test/lib_tests/Metrics_test.res @@ -794,3 +794,116 @@ describe("Metrics.merge", () => { }) }) }) + +describe("Metrics.renderRuntime", () => { + let sample = (~heapUsed, ~gc): Metrics.runtimeSample => { + cpuUserSeconds: 1.5, + cpuSystemSeconds: 0.5, + processStartTimeSeconds: 1700000000., + residentMemoryBytes: 300., + heapTotalBytes: 200., + heapUsedBytes: heapUsed, + externalMemoryBytes: 10., + eventLoopUtilization: 0.25, + eventLoopLagMeanSeconds: 0.001, + eventLoopLagMinSeconds: 0., + eventLoopLagMaxSeconds: 0.002, + eventLoopLagStddevSeconds: 0.0005, + eventLoopLagP50Seconds: 0.001, + eventLoopLagP90Seconds: 0.0015, + eventLoopLagP99Seconds: 0.002, + heapSpaces: [{space: "new", size: 100., used: 40., available: 60.}], + activeResources: [("TCPSocketWrap", 2.)], + gc, + nodeVersion: "v24.1.2", + } + + // Comment lines are the same in every layout, so only the samples are compared. + let samples = rendered => + rendered + ->String.split("\n") + ->Array.filter(line => line !== "" && !(line->String.startsWith("#"))) + + it("Renders one process without labels, and a run's processes under a worker label", t => { + t.expect(( + Metrics.renderRuntime([("", sample(~heapUsed=150., ~gc=[]))])->samples, + Metrics.renderRuntime([ + (`worker="supervisor"`, sample(~heapUsed=50., ~gc=[])), + (`worker="0"`, sample(~heapUsed=150., ~gc=[{kind: "minor", count: 3., seconds: 0.03}])), + ])->samples, + )).toStrictEqual(( + [ + "process_cpu_user_seconds_total 1.5", + "process_cpu_system_seconds_total 0.5", + "process_cpu_seconds_total 2", + "process_start_time_seconds 1700000000", + "process_resident_memory_bytes 300", + "nodejs_heap_size_total_bytes 200", + "nodejs_heap_size_used_bytes 150", + "nodejs_external_memory_bytes 10", + "nodejs_eventloop_utilization 0.25", + "nodejs_eventloop_lag_mean_seconds 0.001", + "nodejs_eventloop_lag_min_seconds 0", + "nodejs_eventloop_lag_max_seconds 0.002", + "nodejs_eventloop_lag_stddev_seconds 0.001", + "nodejs_eventloop_lag_p50_seconds 0.001", + "nodejs_eventloop_lag_p90_seconds 0.002", + "nodejs_eventloop_lag_p99_seconds 0.002", + `nodejs_heap_space_size_total_bytes{space="new"} 100`, + `nodejs_heap_space_size_used_bytes{space="new"} 40`, + `nodejs_heap_space_size_available_bytes{space="new"} 60`, + `nodejs_active_resources{type="TCPSocketWrap"} 2`, + "nodejs_active_resources_total 2", + `nodejs_version_info{version="v24.1.2",major="24",minor="1",patch="2"} 1`, + ], + [ + `process_cpu_user_seconds_total{worker="supervisor"} 1.5`, + `process_cpu_user_seconds_total{worker="0"} 1.5`, + `process_cpu_system_seconds_total{worker="supervisor"} 0.5`, + `process_cpu_system_seconds_total{worker="0"} 0.5`, + `process_cpu_seconds_total{worker="supervisor"} 2`, + `process_cpu_seconds_total{worker="0"} 2`, + `process_start_time_seconds{worker="supervisor"} 1700000000`, + `process_start_time_seconds{worker="0"} 1700000000`, + `process_resident_memory_bytes{worker="supervisor"} 300`, + `process_resident_memory_bytes{worker="0"} 300`, + `nodejs_heap_size_total_bytes{worker="supervisor"} 200`, + `nodejs_heap_size_total_bytes{worker="0"} 200`, + `nodejs_heap_size_used_bytes{worker="supervisor"} 50`, + `nodejs_heap_size_used_bytes{worker="0"} 150`, + `nodejs_external_memory_bytes{worker="supervisor"} 10`, + `nodejs_external_memory_bytes{worker="0"} 10`, + `nodejs_eventloop_utilization{worker="supervisor"} 0.25`, + `nodejs_eventloop_utilization{worker="0"} 0.25`, + `nodejs_eventloop_lag_mean_seconds{worker="supervisor"} 0.001`, + `nodejs_eventloop_lag_mean_seconds{worker="0"} 0.001`, + `nodejs_eventloop_lag_min_seconds{worker="supervisor"} 0`, + `nodejs_eventloop_lag_min_seconds{worker="0"} 0`, + `nodejs_eventloop_lag_max_seconds{worker="supervisor"} 0.002`, + `nodejs_eventloop_lag_max_seconds{worker="0"} 0.002`, + `nodejs_eventloop_lag_stddev_seconds{worker="supervisor"} 0.001`, + `nodejs_eventloop_lag_stddev_seconds{worker="0"} 0.001`, + `nodejs_eventloop_lag_p50_seconds{worker="supervisor"} 0.001`, + `nodejs_eventloop_lag_p50_seconds{worker="0"} 0.001`, + `nodejs_eventloop_lag_p90_seconds{worker="supervisor"} 0.002`, + `nodejs_eventloop_lag_p90_seconds{worker="0"} 0.002`, + `nodejs_eventloop_lag_p99_seconds{worker="supervisor"} 0.002`, + `nodejs_eventloop_lag_p99_seconds{worker="0"} 0.002`, + `nodejs_heap_space_size_total_bytes{worker="supervisor",space="new"} 100`, + `nodejs_heap_space_size_total_bytes{worker="0",space="new"} 100`, + `nodejs_heap_space_size_used_bytes{worker="supervisor",space="new"} 40`, + `nodejs_heap_space_size_used_bytes{worker="0",space="new"} 40`, + `nodejs_heap_space_size_available_bytes{worker="supervisor",space="new"} 60`, + `nodejs_heap_space_size_available_bytes{worker="0",space="new"} 60`, + `nodejs_active_resources{worker="supervisor",type="TCPSocketWrap"} 2`, + `nodejs_active_resources{worker="0",type="TCPSocketWrap"} 2`, + `nodejs_active_resources_total{worker="supervisor"} 2`, + `nodejs_active_resources_total{worker="0"} 2`, + `nodejs_gc_duration_seconds_sum{worker="0",kind="minor"} 0.03`, + `nodejs_gc_duration_seconds_count{worker="0",kind="minor"} 3`, + `nodejs_version_info{worker="supervisor",version="v24.1.2",major="24",minor="1",patch="2"} 1`, + `nodejs_version_info{worker="0",version="v24.1.2",major="24",minor="1",patch="2"} 1`, + ], + )) + }) +}) diff --git a/packages/envio-tests/test/lib_tests/Supervisor_test.res b/packages/envio-tests/test/lib_tests/Supervisor_test.res index 792cd1827..d16ec90ad 100644 --- a/packages/envio-tests/test/lib_tests/Supervisor_test.res +++ b/packages/envio-tests/test/lib_tests/Supervisor_test.res @@ -166,10 +166,11 @@ describe("Supervisor worker plumbing", () => { }) describe("Config.logContext", () => { - it("Attributes an isolated run's logs to its chains, and any other run's to none", t => { + it("Names the one chain an isolated process drives, and nothing otherwise", t => { t.expect([ // This process drives one of the schema's chains while siblings drive the rest. config(~schema=perChain, ~isolatedChains=[JSON.Number(137.)])->Config.logContext, + // Several chains have no single owner to name. config( ~schema=perChain, ~isolatedChains=[JSON.Number(1.), JSON.Number(137.)], @@ -180,13 +181,21 @@ describe("Config.logContext", () => { config(~schema=crossChain)->Config.logContext, ]).toStrictEqual([ Some(JSON.Object(Dict.fromArray([("chainId", JSON.Number(137.))]))), - Some( - JSON.Object( - Dict.fromArray([("chainIds", JSON.Array([JSON.Number(1.), JSON.Number(137.)]))]), - ), - ), + None, None, None, ]) }) }) + +describe("Worker.detect", () => { + it("Counts as a worker only when forked with the argument and a channel", t => { + let forked = ["node", "envio", Worker.forkArg] + t.expect([ + Worker.detect(~argv=forked, ~hasChannel=true), + // A user typing the argument, or a process manager forking with a channel. + Worker.detect(~argv=forked, ~hasChannel=false), + Worker.detect(~argv=["node", "envio", "start"], ~hasChannel=true), + ]).toStrictEqual([true, false, false]) + }) +}) diff --git a/packages/envio/src/Bin.res b/packages/envio/src/Bin.res index 2b2b4b142..346a1e32d 100644 --- a/packages/envio/src/Bin.res +++ b/packages/envio/src/Bin.res @@ -52,7 +52,7 @@ let applyEnv = (env: dict) => let run = async args => { try { if Worker.isEnabled { - Worker.exitWithSupervisor() + Worker.bindToSupervisor() Worker.listen() // A worker is handed the config its supervisor already parsed, narrowed to // the chains it drives, so the two can't disagree about what is indexed. diff --git a/packages/envio/src/Config.res b/packages/envio/src/Config.res index 82eab5603..1b1254396 100644 --- a/packages/envio/src/Config.res +++ b/packages/envio/src/Config.res @@ -630,28 +630,19 @@ let getChain = (config, ~chainId) => // be split across processes. let isPerChain = (config: t) => !(config.userEntities->Array.some(entity => entity.crossChain)) -// What every line this process logs is attributed to: the chains it drives, -// when sibling processes drive the rest. A process driving every chain has -// nothing to tell apart from, and its chain-scoped lines already name theirs. +// What every line this process logs is attributed to. A process driving one +// of the schema's chains while siblings drive the rest names it, on the lines +// that had no chain in hand. One driving several has no single owner to name, +// and its chain-scoped lines already carry theirs. let logContext = (config: t): option => - if config.isolated { - let chainIds = config.chainMap->ChainMap.keys + switch (config.isolated, config.chainMap->ChainMap.keys) { + | (true, [chainId]) => Some( - switch chainIds { - | [chainId] => - JSON.Object( - Dict.fromArray([("chainId", chainId->S.reverseConvertToJsonOrThrow(ChainId.schema))]), - ) - | chainIds => - JSON.Object( - Dict.fromArray([ - ("chainIds", chainIds->S.reverseConvertToJsonOrThrow(S.array(ChainId.schema))), - ]), - ) - }, + JSON.Object( + Dict.fromArray([("chainId", chainId->S.reverseConvertToJsonOrThrow(ChainId.schema))]), + ), ) - } else { - None + | _ => None } // Narrows a config to the chains one `envio start --chain` process drives. diff --git a/packages/envio/src/Env.res b/packages/envio/src/Env.res index ff719d835..ab23da20d 100644 --- a/packages/envio/src/Env.res +++ b/packages/envio/src/Env.res @@ -77,9 +77,6 @@ let hypersyncClientEnableQueryCaching = let hypersyncLogLevel = envSafe->EnvSafe.get("ENVIO_HYPERSYNC_LOG_LEVEL", HyperSyncClient.logLevelSchema, ~fallback=#info) -// Set by a supervisor on the processes it forks, and by nothing else. -let isWorker = envSafe->EnvSafe.get("ENVIO_WORKER", S.option(S.bool))->Option.getOr(false) - let logStrategy = envSafe->EnvSafe.get( "LOG_STRATEGY", diff --git a/packages/envio/src/Main.res b/packages/envio/src/Main.res index ba3346711..9353be315 100644 --- a/packages/envio/src/Main.res +++ b/packages/envio/src/Main.res @@ -492,6 +492,7 @@ let startServer = ( ~getMetrics: unit => option, ~envioVersion: string, ~onSyncCache: unit => promise, + ~collectRuntime: unit => string, ~isDevelopmentMode: bool, ) => { open Express @@ -563,7 +564,7 @@ let startServer = ( app->get("/metrics/runtime", (_req, res) => { res->set("Content-Type", Metrics.contentType) - let _ = res->endWithData(Metrics.collectRuntime()) + let _ = res->endWithData(collectRuntime()) }) let server = app->listen(Env.serverPort) @@ -724,6 +725,7 @@ let start = async ( if !isTest && !Worker.isEnabled { startServer( ~onSyncCache=() => dumpEffectCache()->Promise.thenResolve(ignore), + ~collectRuntime=Metrics.collectRuntime, ~isDevelopmentMode, ~envioVersion, ~getMetrics, @@ -753,8 +755,12 @@ let start = async ( | Init(_) => () } ) + Metrics.startRuntimeCollectors() let _intervalId = setInterval( - () => Worker.send(Snapshot({metrics: state->IndexerState.toMetrics})), + () => + Worker.send( + Snapshot({metrics: state->IndexerState.toMetrics, runtime: Metrics.sampleRuntime()}), + ), Worker.snapshotIntervalMillis, ) } diff --git a/packages/envio/src/Metrics.res b/packages/envio/src/Metrics.res index 8f3cd7bbd..59993ef1c 100644 --- a/packages/envio/src/Metrics.res +++ b/packages/envio/src/Metrics.res @@ -1004,199 +1004,253 @@ let getRuntimeCollectors = () => // from the beginning of the run, not from the first scrape. let startRuntimeCollectors = () => getRuntimeCollectors()->ignore -let collectRuntime = () => { - let b = {out: ""} +type runtimeGc = {kind: string, count: float, seconds: float} +type runtimeHeapSpace = {space: string, size: float, used: float, available: float} + +// One process's runtime readings at one moment. Plain numbers, so a worker can +// hand its sample to the supervisor over the fork's channel and the supervisor +// can render every process's under one label set. +type runtimeSample = { + cpuUserSeconds: float, + cpuSystemSeconds: float, + processStartTimeSeconds: float, + residentMemoryBytes: float, + heapTotalBytes: float, + heapUsedBytes: float, + externalMemoryBytes: float, + eventLoopUtilization: float, + eventLoopLagMeanSeconds: float, + eventLoopLagMinSeconds: float, + eventLoopLagMaxSeconds: float, + eventLoopLagStddevSeconds: float, + eventLoopLagP50Seconds: float, + eventLoopLagP90Seconds: float, + eventLoopLagP99Seconds: float, + heapSpaces: array, + activeResources: array<(string, float)>, + gc: array, + nodeVersion: string, +} + +let sampleRuntime = (): runtimeSample => { let memory = NodeJs.Process.memoryUsage() let cpu = NodeJs.Process.cpuUsage() let elu = NodeJs.PerfHooks.performance->NodeJs.PerfHooks.eventLoopUtilization let {eventLoopDelay, gcStats, processStartTimeSeconds} = getRuntimeCollectors() - b->single( + // Nanoseconds in the histogram; reset after sampling so each scrape reports + // the delay distribution since the previous one, matching prom-client. With + // no samples yet (e.g. the first scrape, which starts the sampler) the + // histogram reports NaN means and a sentinel min — report 0 instead. + let hasLagSamples = eventLoopDelay.max > 0. + let nsToSeconds = ns => hasLagSamples && !(ns->Float.isNaN) ? ns /. 1_000_000_000. : 0. + let sample = { + cpuUserSeconds: cpu.user /. 1_000_000., + cpuSystemSeconds: cpu.system /. 1_000_000., + processStartTimeSeconds, + residentMemoryBytes: memory.rss, + heapTotalBytes: memory.heapTotal, + heapUsedBytes: memory.heapUsed, + externalMemoryBytes: memory.external_, + eventLoopUtilization: elu.utilization, + eventLoopLagMeanSeconds: eventLoopDelay.mean->nsToSeconds, + eventLoopLagMinSeconds: eventLoopDelay.min->nsToSeconds, + eventLoopLagMaxSeconds: eventLoopDelay.max->nsToSeconds, + eventLoopLagStddevSeconds: eventLoopDelay.stddev->nsToSeconds, + eventLoopLagP50Seconds: eventLoopDelay->NodeJs.PerfHooks.percentile(50)->nsToSeconds, + eventLoopLagP90Seconds: eventLoopDelay->NodeJs.PerfHooks.percentile(90)->nsToSeconds, + eventLoopLagP99Seconds: eventLoopDelay->NodeJs.PerfHooks.percentile(99)->nsToSeconds, + heapSpaces: NodeJs.V8.getHeapSpaceStatistics()->Array.map(s => { + space: s.spaceName->String.replace("_space", ""), + size: s.spaceSize, + used: s.spaceUsedSize, + available: s.spaceAvailableSize, + }), + activeResources: { + let byType = Dict.make() + NodeJs.Process.getActiveResourcesInfo()->Array.forEach(resource => + byType->Dict.set( + resource, + byType->Utils.Dict.dangerouslyGetNonOption(resource)->Option.getOr(0.) +. 1., + ) + ) + byType->Dict.toArray + }, + gc: gcStats + ->Dict.toArray + ->Array.map(((kind, stat)) => {kind, count: stat.count, seconds: stat.seconds}), + nodeVersion: NodeJs.Process.version, + } + eventLoopDelay->NodeJs.PerfHooks.reset + sample +} + +// Renders every sample under `scope`, the labels telling one process's readings +// from another's (`worker="0"`), or "" for a run that is one process. +let renderRuntime = (samples: array<(string, runtimeSample)>) => { + let b = {out: ""} + let labels = (scope, own) => + switch (scope, own) { + | ("", "") => "" + | ("", own) | (own, "") => `{${own}}` + | (scope, own) => `{${scope},${own}}` + } + let scoped = samples->Array.map(((scope, sample)) => (labels(scope, ""), sample)) + let each = select => + samples->Array.flatMap(((scope, sample)) => + select(sample)->Array.map(((own, value)) => (labels(scope, own), value)) + ) + let gauge = (~name, ~help, ~value) => + b->series(~name, ~help, ~kind="gauge", ~entries=scoped, ~value) + let counter = (~name, ~help, ~value) => + b->series(~name, ~help, ~kind="counter", ~entries=scoped, ~value) + + counter( ~name="process_cpu_user_seconds_total", ~help="Total user CPU time spent in seconds.", - ~kind="counter", - ~value=cpu.user /. 1_000_000., + ~value=s => s.cpuUserSeconds, ) - b->single( + counter( ~name="process_cpu_system_seconds_total", ~help="Total system CPU time spent in seconds.", - ~kind="counter", - ~value=cpu.system /. 1_000_000., + ~value=s => s.cpuSystemSeconds, ) - b->single( + counter( ~name="process_cpu_seconds_total", ~help="Total user and system CPU time spent in seconds.", - ~kind="counter", - ~value=(cpu.user +. cpu.system) /. 1_000_000., + ~value=s => s.cpuUserSeconds +. s.cpuSystemSeconds, ) - b->single( + gauge( ~name="process_start_time_seconds", ~help="Start time of the process since unix epoch in seconds.", - ~kind="gauge", - ~value=processStartTimeSeconds, + ~value=s => s.processStartTimeSeconds, ) - b->single( - ~name="process_resident_memory_bytes", - ~help="Resident memory size in bytes.", - ~kind="gauge", - ~value=memory.rss, + gauge(~name="process_resident_memory_bytes", ~help="Resident memory size in bytes.", ~value=s => + s.residentMemoryBytes ) - b->single( + gauge( ~name="nodejs_heap_size_total_bytes", ~help="Process heap size from Node.js in bytes.", - ~kind="gauge", - ~value=memory.heapTotal, + ~value=s => s.heapTotalBytes, ) - b->single( + gauge( ~name="nodejs_heap_size_used_bytes", ~help="Process heap size used from Node.js in bytes.", - ~kind="gauge", - ~value=memory.heapUsed, + ~value=s => s.heapUsedBytes, ) - b->single( + gauge( ~name="nodejs_external_memory_bytes", ~help="Node.js external memory size in bytes.", - ~kind="gauge", - ~value=memory.external_, + ~value=s => s.externalMemoryBytes, ) - b->single( + gauge( ~name="nodejs_eventloop_utilization", ~help="Ratio of time the event loop is active, since process start.", - ~kind="gauge", - ~value=elu.utilization, + ~value=s => s.eventLoopUtilization, ) - // Nanoseconds in the histogram; reset after rendering so each scrape reports - // the delay distribution since the previous one, matching prom-client. With - // no samples yet (e.g. the first scrape, which starts the sampler) the - // histogram reports NaN means and a sentinel min — render 0 instead. - let hasLagSamples = eventLoopDelay.max > 0. - let nsToSeconds = ns => hasLagSamples && !(ns->Float.isNaN) ? ns /. 1_000_000_000. : 0. - b->single( + gauge( ~name="nodejs_eventloop_lag_mean_seconds", ~help="The mean of the recorded event loop delays.", - ~kind="gauge", - ~value=eventLoopDelay.mean->nsToSeconds, + ~value=s => s.eventLoopLagMeanSeconds, ) - b->single( + gauge( ~name="nodejs_eventloop_lag_min_seconds", ~help="The minimum recorded event loop delay.", - ~kind="gauge", - ~value=eventLoopDelay.min->nsToSeconds, + ~value=s => s.eventLoopLagMinSeconds, ) - b->single( + gauge( ~name="nodejs_eventloop_lag_max_seconds", ~help="The maximum recorded event loop delay.", - ~kind="gauge", - ~value=eventLoopDelay.max->nsToSeconds, + ~value=s => s.eventLoopLagMaxSeconds, ) - b->single( + gauge( ~name="nodejs_eventloop_lag_stddev_seconds", ~help="The standard deviation of the recorded event loop delays.", - ~kind="gauge", - ~value=eventLoopDelay.stddev->nsToSeconds, + ~value=s => s.eventLoopLagStddevSeconds, ) - b->single( + gauge( ~name="nodejs_eventloop_lag_p50_seconds", ~help="The 50th percentile of the recorded event loop delays.", - ~kind="gauge", - ~value=eventLoopDelay->NodeJs.PerfHooks.percentile(50)->nsToSeconds, + ~value=s => s.eventLoopLagP50Seconds, ) - b->single( + gauge( ~name="nodejs_eventloop_lag_p90_seconds", ~help="The 90th percentile of the recorded event loop delays.", - ~kind="gauge", - ~value=eventLoopDelay->NodeJs.PerfHooks.percentile(90)->nsToSeconds, + ~value=s => s.eventLoopLagP90Seconds, ) - b->single( + gauge( ~name="nodejs_eventloop_lag_p99_seconds", ~help="The 99th percentile of the recorded event loop delays.", - ~kind="gauge", - ~value=eventLoopDelay->NodeJs.PerfHooks.percentile(99)->nsToSeconds, + ~value=s => s.eventLoopLagP99Seconds, ) - eventLoopDelay->NodeJs.PerfHooks.reset - let heapSpaces = - NodeJs.V8.getHeapSpaceStatistics()->Array.map(s => ( - `{space="${s.spaceName->String.replace("_space", "")}"}`, - s, - )) + + let heapSpaces = each(s => s.heapSpaces->Array.map(h => (`space="${h.space}"`, h))) b->series( ~name="nodejs_heap_space_size_total_bytes", ~help="Process heap space size total from Node.js in bytes.", ~kind="gauge", ~entries=heapSpaces, - ~value=s => s.spaceSize, + ~value=h => h.size, ) b->series( ~name="nodejs_heap_space_size_used_bytes", ~help="Process heap space size used from Node.js in bytes.", ~kind="gauge", ~entries=heapSpaces, - ~value=s => s.spaceUsedSize, + ~value=h => h.used, ) b->series( ~name="nodejs_heap_space_size_available_bytes", ~help="Process heap space size available from Node.js in bytes.", ~kind="gauge", ~entries=heapSpaces, - ~value=s => s.spaceAvailableSize, - ) - let activeResources = { - let byType = Dict.make() - NodeJs.Process.getActiveResourcesInfo()->Array.forEach(resource => { - let label = `{type="${resource->escapeLabelValue}"}` - byType->Dict.set( - label, - byType->Utils.Dict.dangerouslyGetNonOption(label)->Option.getOr(0.) +. 1., - ) - }) - byType->Dict.toArray - } + ~value=h => h.available, + ) + b->series( ~name="nodejs_active_resources", ~help="Number of active resources that are currently keeping the event loop alive, grouped by async resource type.", ~kind="gauge", - ~entries=activeResources, + ~entries=each(s => + s.activeResources->Array.map( + ((resource, count)) => (`type="${resource->escapeLabelValue}"`, count), + ) + ), ~value=count => count, ) - b->single( + gauge( ~name="nodejs_active_resources_total", ~help="Total number of active resources.", - ~kind="gauge", - ~value=activeResources->Array.reduce(0., (acc, (_, count)) => acc +. count), - ) - let gcEntries = [] - gcStats->Utils.Dict.forEachWithKey((stat, kind) => - gcEntries->Array.push((`{kind="${kind}"}`, stat)) + ~value=s => s.activeResources->Array.reduce(0., (acc, (_, count)) => acc +. count), ) + + let gc = each(s => s.gc->Array.map(g => (`kind="${g.kind}"`, g))) b->series( ~name="nodejs_gc_duration_seconds_sum", ~help="Cumulative garbage collection pause time by kind, one of major, minor, incremental or weakcb.", ~kind="counter", - ~entries=gcEntries, - ~value=s => s.seconds, + ~entries=gc, + ~value=g => g.seconds, ) b->series( ~name="nodejs_gc_duration_seconds_count", ~help="Number of garbage collection pauses by kind, one of major, minor, incremental or weakcb.", ~kind="counter", - ~entries=gcEntries, - ~value=s => s.count, + ~entries=gc, + ~value=g => g.count, ) - let version = NodeJs.Process.version - let versionParts = version->String.replace("v", "")->String.split(".") - let versionPart = i => versionParts->Array.get(i)->Option.getOr("0") + b->series( ~name="nodejs_version_info", ~help="Node.js version info.", ~kind="gauge", - ~entries=[ - ( - `{version="${version}",major="${versionPart(0)}",minor="${versionPart( - 1, - )}",patch="${versionPart(2)}"}`, - (), - ), - ], + ~entries=each(s => { + let parts = s.nodeVersion->String.replace("v", "")->String.split(".") + let part = i => parts->Array.get(i)->Option.getOr("0") + [(`version="${s.nodeVersion}",major="${part(0)}",minor="${part(1)}",patch="${part(2)}"`, ())] + }), ~value=() => 1., ) - b.out ++ "\n" + b.out } + +let collectRuntime = () => renderRuntime([("", sampleRuntime())]) diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index cec2fff5c..9850b96ae 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -55,6 +55,7 @@ type running = { worker: worker, child: NodeJs.ChildProcess.child, mutable snapshot: option, + mutable runtime: option, // A spawn failure can raise `error` and `exit` both, and a worker counted // twice would end the run while its siblings are still indexing. mutable settled: bool, @@ -104,7 +105,6 @@ let fork = ( ~entryPath=NodeJs.Process.argv->Array.getUnsafe(1), ) => { let env = NodeJs.Process.process.env->Dict.copy - env->Dict.set("ENVIO_WORKER", "true") // The worker's slice of the budget. Read when the worker's own Env module // loads, which is why it rides in the spawn environment rather than a message. env->Dict.set("ENVIO_PG_MAX_CONNECTIONS", worker.maxConnections->Int.toString) @@ -112,7 +112,7 @@ let fork = ( let child = NodeJs.ChildProcess.fork( entryPath, - [], + [Worker.forkArg], { env, serialization: "advanced", @@ -126,10 +126,20 @@ let fork = ( ->NodeJs.ChildProcess.send(Worker.Init({config: configJson->configForWorker(~worker)})) ->ignore - let running = {worker, child, snapshot: None, settled: false, onCacheSynced: None} + let running = { + worker, + child, + snapshot: None, + runtime: None, + settled: false, + onCacheSynced: None, + } child->NodeJs.ChildProcess.onMessage(message => switch message { - | Worker.Snapshot({metrics}) => running.snapshot = Some(metrics) + | Worker.Snapshot({metrics, runtime}) => { + running.snapshot = Some(metrics) + running.runtime = Some(runtime) + } | Worker.CacheSynced(_) => switch running.onCacheSynced { | Some(resolve) => @@ -283,6 +293,16 @@ let run = async (~workers: array, ~configJson: JSON.t, ~reset) => { | snapshots => Some(snapshots->merge) }, ~envioVersion=Utils.EnvioPackage.value.version, + // Every process of the run, the supervisor included, under a `worker` + // label: the memory and the event loop that matter are the workers' own. + ~collectRuntime=() => + Metrics.renderRuntime( + [(`worker="supervisor"`, Metrics.sampleRuntime())]->Array.concat( + group.running->Array.filterMapWithIndex((r, workerIndex) => + r.runtime->Option.map(runtime => (`worker="${workerIndex->Int.toString}"`, runtime)) + ), + ), + ), ~isDevelopmentMode=config.isDev, ~onSyncCache=() => group->syncCache, ) @@ -292,8 +312,9 @@ let run = async (~workers: array, ~configJson: JSON.t, ~reset) => { let _rerender = Tui.start(~config, ~getMetrics=() => reported()->merge) } - // Only the supervisor is signalled when the run is asked to stop, so it - // passes that on. A terminal's own interrupt already reaches the whole group. + // Whichever signal asks the run to stop, the supervisor is the one that + // stops the workers: an interrupt from the terminal reaches them too, but + // they leave it to the supervisor. NodeJs.Process.onSignal("SIGTERM", () => group->stop) NodeJs.Process.onSignal("SIGINT", () => group->stop) @@ -306,6 +327,10 @@ let run = async (~workers: array, ~configJson: JSON.t, ~reset) => { | Finished if !shouldUseTui => Logging.info("Exiting with success") NodeJs.process->NodeJs.exitWithCode(Success) - | Finished => () + | Finished => + // With nothing left to stop, the stop signals end the display instead. + // Registering a handler above took over from Node's default exit. + NodeJs.Process.onSignal("SIGTERM", () => NodeJs.process->NodeJs.exitWithCode(Success)) + NodeJs.Process.onSignal("SIGINT", () => NodeJs.process->NodeJs.exitWithCode(Success)) } } diff --git a/packages/envio/src/Worker.res b/packages/envio/src/Worker.res index 62a3e0d78..3175daca9 100644 --- a/packages/envio/src/Worker.res +++ b/packages/envio/src/Worker.res @@ -2,9 +2,18 @@ // a subset of the chains. It has no server and no TUI of its own — it reports // through the IPC channel, and the supervisor is the one operational surface. -// Set by the supervisor on the processes it forks. An indexer a user started -// themselves never has it, and takes every path it takes today. -let isEnabled = Env.isWorker +// The argument a supervisor forks its workers with. It means nothing to the +// CLI, and it counts only together with the fork's own channel, so a user who +// types it starts nothing: an indexer they start themselves takes every path +// it takes today. +let forkArg = "--supervised-worker" + +let detect = (~argv: array, ~hasChannel) => hasChannel && argv->Array.includes(forkArg) + +let isEnabled = detect( + ~argv=NodeJs.Process.argv, + ~hasChannel=NodeJs.Process.channel->Nullable.toOption->Option.isSome, +) @tag("kind") type parentMessage = @@ -16,7 +25,7 @@ type parentMessage = @tag("kind") type workerMessage = - | @as("snapshot") Snapshot({metrics: Metrics.t}) + | @as("snapshot") Snapshot({metrics: Metrics.t, runtime: Metrics.runtimeSample}) // Sent once the worker's effect cache is on disk, so the supervisor's // console can answer for a dump that has actually happened. | @as("cacheSynced") CacheSynced({}) @@ -25,14 +34,19 @@ type workerMessage = // display moves at the same rate an unsplit run's does. let snapshotIntervalMillis = 500 -// A worker with no supervisor has nobody reading its metrics and nobody to stop -// it: the supervisor could have died before it ever got to tear the group down. -// Losing the channel is that signal. -let exitWithSupervisor = () => +// The supervisor is the one that stops a worker, and the one whose absence +// ends it. A terminal's interrupt reaches the whole group at once, so the +// worker leaves it to the supervisor, which stops every worker in turn; without +// that, a worker gone on its own would read as a failure to the supervisor +// still deciding what the interrupt meant. A supervisor that dies can't tear +// the group down, so losing the channel is what ends the worker then. +let bindToSupervisor = () => { + NodeJs.Process.onSignal("SIGINT", () => ()) NodeJs.Process.onDisconnect(() => { Logging.error("The indexer supervisor is gone. Stopping this chain's process.") NodeJs.process->NodeJs.exitWithCode(Failure) }) +} let send = (message: workerMessage) => if isEnabled { diff --git a/packages/envio/src/bindings/NodeJs.res b/packages/envio/src/bindings/NodeJs.res index 754651069..8ad8ef9fc 100644 --- a/packages/envio/src/bindings/NodeJs.res +++ b/packages/envio/src/bindings/NodeJs.res @@ -74,6 +74,8 @@ module Process = { @val @scope("process") external onceMessage: (@as("message") _, 'msg => unit) => unit = "once" @val @scope("process") external onSignal: (string, unit => unit) => unit = "on" + // Present only in a process forked with an IPC channel. + @val @scope("process") external channel: Nullable.t = "channel" @val @scope("process") external onDisconnect: (@as("disconnect") _, unit => unit) => unit = "on" @val @scope("process") external argv: array = "argv" diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index d033f8ed5..48ea8d39d 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -231,6 +231,19 @@ importers: specifier: 4.1.0 version: 4.1.0(@opentelemetry/api@1.9.0)(@types/node@24.12.2)(jsdom@16.7.0)(vite@7.3.1(@types/node@24.12.2)(tsx@4.21.0)) + scenarios/split_test: + dependencies: + envio: + specifier: file:../../packages/envio + version: link:../../packages/envio + devDependencies: + '@types/node': + specifier: 24.12.2 + version: 24.12.2 + typescript: + specifier: 6.0.3 + version: 6.0.3 + scenarios/svm_flow_xray: dependencies: envio: diff --git a/scenarios/split_test/config.head.yaml b/scenarios/split_test/config.head.yaml new file mode 100644 index 000000000..cd78e7489 --- /dev/null +++ b/scenarios/split_test/config.head.yaml @@ -0,0 +1,22 @@ +# yaml-language-server: $schema=../../packages/envio/evm.schema.json +name: split_test +description: The same chains with no end block, for a run that only a stop signal ends +disable_default_cross_chain: true +storage: + postgres: + column_name_format: snake_case +contracts: + - name: ERC20 + events: + - event: "Transfer(address indexed from, address indexed to, uint256 value)" +chains: + - id: 1 + start_block: 10861674 + contracts: + - name: ERC20 + address: "0x1f9840a85d5aF5bf1D1762F925BDADdC4201F984" + - id: 8453 + start_block: 10000000 + contracts: + - name: ERC20 + address: "0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913" diff --git a/scenarios/split_test/config.yaml b/scenarios/split_test/config.yaml new file mode 100644 index 000000000..f9c29e391 --- /dev/null +++ b/scenarios/split_test/config.yaml @@ -0,0 +1,24 @@ +# yaml-language-server: $schema=../../packages/envio/evm.schema.json +name: split_test +description: Two chains over a per-chain schema, so a plain `envio start` splits them across processes +disable_default_cross_chain: true +storage: + postgres: + column_name_format: snake_case +contracts: + - name: ERC20 + events: + - event: "Transfer(address indexed from, address indexed to, uint256 value)" +chains: + - id: 1 + start_block: 10861674 + end_block: 10861774 + contracts: + - name: ERC20 + address: "0x1f9840a85d5aF5bf1D1762F925BDADdC4201F984" + - id: 8453 + start_block: 10000000 + end_block: 10000050 + contracts: + - name: ERC20 + address: "0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913" diff --git a/scenarios/split_test/envio-env.d.ts b/scenarios/split_test/envio-env.d.ts new file mode 100644 index 000000000..c8458812c --- /dev/null +++ b/scenarios/split_test/envio-env.d.ts @@ -0,0 +1,7 @@ +/** + * This file is generated by HyperIndex codegen. Do not edit manually. + * It wires project-specific types from `.envio/types.d.ts` into the `envio` module. + * If your project's types look out of date, run `envio codegen` + * (or your package manager's `codegen` script, e.g. `pnpm codegen`). + */ +/// diff --git a/scenarios/split_test/package.json b/scenarios/split_test/package.json new file mode 100644 index 000000000..0eb506550 --- /dev/null +++ b/scenarios/split_test/package.json @@ -0,0 +1,20 @@ +{ + "name": "split_test", + "version": "0.1.0", + "type": "module", + "scripts": { + "codegen": "envio codegen", + "dev": "envio dev", + "start": "envio start" + }, + "devDependencies": { + "@types/node": "24.12.2", + "typescript": "6.0.3" + }, + "dependencies": { + "envio": "file:../../packages/envio" + }, + "engines": { + "node": ">=22.0.0" + } +} diff --git a/scenarios/split_test/schema.graphql b/scenarios/split_test/schema.graphql new file mode 100644 index 000000000..efc094d75 --- /dev/null +++ b/scenarios/split_test/schema.graphql @@ -0,0 +1,7 @@ +type Transfer { + id: ID! + from: String! + to: String! + value: BigInt! + blockNumber: Int! +} diff --git a/scenarios/split_test/src/handlers/ERC20.ts b/scenarios/split_test/src/handlers/ERC20.ts new file mode 100644 index 000000000..4c7a4f75e --- /dev/null +++ b/scenarios/split_test/src/handlers/ERC20.ts @@ -0,0 +1,11 @@ +import { indexer } from "envio"; + +indexer.onEvent({ contract: "ERC20", event: "Transfer" }, async ({ event, context }) => { + context.Transfer.set({ + id: `${event.chainId}-${event.block.number}-${event.logIndex}`, + from: event.params.from, + to: event.params.to, + value: event.params.value, + blockNumber: event.block.number, + }); +}); diff --git a/scenarios/split_test/tsconfig.json b/scenarios/split_test/tsconfig.json new file mode 100644 index 000000000..cf725ad13 --- /dev/null +++ b/scenarios/split_test/tsconfig.json @@ -0,0 +1,21 @@ +{ + "compilerOptions": { + "esModuleInterop": true, + "skipLibCheck": true, + "target": "es2023", + "allowJs": true, + "resolveJsonModule": true, + "moduleDetection": "force", + "isolatedModules": true, + "verbatimModuleSyntax": true, + "strict": true, + "noUncheckedIndexedAccess": true, + "noImplicitOverride": true, + "module": "ESNext", + "moduleResolution": "bundler", + "noEmit": true, + "lib": ["es2023"], + "types": ["node"] + }, + "include": ["src", "envio-env.d.ts"] +} From 09d29fe16c02810fc74b8cd72e60b48a5142fbc0 Mon Sep 17 00:00:00 2001 From: dzakh Date: Wed, 16 Sep 2026 13:04:34 +0000 Subject: [PATCH 09/61] Ignore the split_test scenario's codegen output Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_017CceLz4mPquMmpWrU74P3j --- scenarios/split_test/.envio/.gitignore | 2 ++ 1 file changed, 2 insertions(+) create mode 100644 scenarios/split_test/.envio/.gitignore diff --git a/scenarios/split_test/.envio/.gitignore b/scenarios/split_test/.envio/.gitignore new file mode 100644 index 000000000..007e7e171 --- /dev/null +++ b/scenarios/split_test/.envio/.gitignore @@ -0,0 +1,2 @@ +# Ephemeral codegen output. Add other .envio entries here as needed. +types.d.ts From 84907a272fb4ee8525b6f4b53f44b3bf8b8a98f3 Mon Sep 17 00:00:00 2001 From: dzakh Date: Wed, 16 Sep 2026 13:08:44 +0000 Subject: [PATCH 10/61] Mark a worker with an internal environment variable rather than a fork argument ENVIO_INTERNAL_WORKER replaces the argument the supervisor forked its workers with: nothing to show in a process listing, nothing to read off argv before the CLI sees it. It still counts only together with the fork's own channel, so a copy left in a shell starts nothing. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_017CceLz4mPquMmpWrU74P3j --- .../test/lib_tests/Supervisor_test.res | 13 +++++++------ packages/envio/src/Supervisor.res | 3 ++- packages/envio/src/Worker.res | 15 ++++++++------- 3 files changed, 17 insertions(+), 14 deletions(-) diff --git a/packages/envio-tests/test/lib_tests/Supervisor_test.res b/packages/envio-tests/test/lib_tests/Supervisor_test.res index d16ec90ad..c32082f2e 100644 --- a/packages/envio-tests/test/lib_tests/Supervisor_test.res +++ b/packages/envio-tests/test/lib_tests/Supervisor_test.res @@ -189,13 +189,14 @@ describe("Config.logContext", () => { }) describe("Worker.detect", () => { - it("Counts as a worker only when forked with the argument and a channel", t => { - let forked = ["node", "envio", Worker.forkArg] + it("Counts as a worker only when forked with the variable and a channel", t => { + let forked = Dict.fromArray([(Worker.envVar, "true")]) t.expect([ - Worker.detect(~argv=forked, ~hasChannel=true), - // A user typing the argument, or a process manager forking with a channel. - Worker.detect(~argv=forked, ~hasChannel=false), - Worker.detect(~argv=["node", "envio", "start"], ~hasChannel=true), + Worker.detect(~env=forked, ~hasChannel=true), + // A copy of the variable left in a shell, or a process manager forking + // with a channel. + Worker.detect(~env=forked, ~hasChannel=false), + Worker.detect(~env=Dict.make(), ~hasChannel=true), ]).toStrictEqual([true, false, false]) }) }) diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index 9850b96ae..e54121023 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -105,6 +105,7 @@ let fork = ( ~entryPath=NodeJs.Process.argv->Array.getUnsafe(1), ) => { let env = NodeJs.Process.process.env->Dict.copy + env->Dict.set(Worker.envVar, "true") // The worker's slice of the budget. Read when the worker's own Env module // loads, which is why it rides in the spawn environment rather than a message. env->Dict.set("ENVIO_PG_MAX_CONNECTIONS", worker.maxConnections->Int.toString) @@ -112,7 +113,7 @@ let fork = ( let child = NodeJs.ChildProcess.fork( entryPath, - [Worker.forkArg], + [], { env, serialization: "advanced", diff --git a/packages/envio/src/Worker.res b/packages/envio/src/Worker.res index 3175daca9..76e109b46 100644 --- a/packages/envio/src/Worker.res +++ b/packages/envio/src/Worker.res @@ -2,16 +2,17 @@ // a subset of the chains. It has no server and no TUI of its own — it reports // through the IPC channel, and the supervisor is the one operational surface. -// The argument a supervisor forks its workers with. It means nothing to the -// CLI, and it counts only together with the fork's own channel, so a user who -// types it starts nothing: an indexer they start themselves takes every path -// it takes today. -let forkArg = "--supervised-worker" +// Set by a supervisor in the environment of the workers it forks. Internal: +// it counts only together with the fork's own channel, so a copy left in a +// shell starts nothing, and an indexer a user starts themselves takes every +// path it takes today. +let envVar = "ENVIO_INTERNAL_WORKER" -let detect = (~argv: array, ~hasChannel) => hasChannel && argv->Array.includes(forkArg) +let detect = (~env: dict, ~hasChannel) => + hasChannel && env->Dict.get(envVar) === Some("true") let isEnabled = detect( - ~argv=NodeJs.Process.argv, + ~env=NodeJs.Process.process.env, ~hasChannel=NodeJs.Process.channel->Nullable.toOption->Option.isSome, ) From d54392fb27f34c967ce910297aebd1999b91f713 Mon Sep 17 00:00:00 2001 From: dzakh Date: Wed, 16 Sep 2026 13:09:28 +0000 Subject: [PATCH 11/61] Name a worker's runtime metrics by the chains it drives `worker="1,137"` in place of `worker="0"`: an index says nothing about whose heap or event loop a reading is, and the chains are what an operator wants to know. The supervisor's own readings keep `worker="supervisor"`. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_017CceLz4mPquMmpWrU74P3j --- packages/e2e-tests/src/e2e/split-run.test.ts | 5 +++-- packages/envio/src/Supervisor.res | 11 +++++++---- 2 files changed, 10 insertions(+), 6 deletions(-) diff --git a/packages/e2e-tests/src/e2e/split-run.test.ts b/packages/e2e-tests/src/e2e/split-run.test.ts index 681e142ed..8fb09700a 100644 --- a/packages/e2e-tests/src/e2e/split-run.test.ts +++ b/packages/e2e-tests/src/e2e/split-run.test.ts @@ -101,7 +101,7 @@ describe.skipIf(!reachable)("E2E: a split run is one indexer", () => { const [runtime, metrics] = await Promise.all([ // The supervisor's own readings alongside each worker's, told apart by label. - scrapeUntil("/metrics/runtime", (body) => body.includes('worker="1"')), + scrapeUntil("/metrics/runtime", (body) => body.includes('worker="8453"')), // Both chains on one endpoint, whichever process drives each. scrapeUntil( "/metrics", @@ -114,7 +114,8 @@ describe.skipIf(!reachable)("E2E: a split run is one indexer", () => { expect({ exitCode: await exit, - runtimeWorkers: ["supervisor", "0", "1"].map((worker) => + // Workers are named by the chains they drive. + runtimeWorkers: ["supervisor", "1", "8453"].map((worker) => runtime.includes(`nodejs_heap_size_used_bytes{worker="${worker}"}`) ), metricsChains: [1, 8453].map((chainId) => diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index e54121023..1b43f442e 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -63,8 +63,11 @@ type running = { mutable onCacheSynced: option unit>, } -let label = (worker: worker) => - `[chain ${worker.chainIds->Array.map(ChainId.toString)->Array.joinUnsafe(",")}]` +// A worker is named by the chains it drives, which is what an operator reading +// its memory or its event loop wants to know. +let name = (worker: worker) => worker.chainIds->Array.map(ChainId.toString)->Array.joinUnsafe(",") + +let label = (worker: worker) => `[chain ${worker->name}]` // Workers append to files of their own. Pino writes a line per call, and // several processes appending to one file can still tear a long line apart. @@ -299,8 +302,8 @@ let run = async (~workers: array, ~configJson: JSON.t, ~reset) => { ~collectRuntime=() => Metrics.renderRuntime( [(`worker="supervisor"`, Metrics.sampleRuntime())]->Array.concat( - group.running->Array.filterMapWithIndex((r, workerIndex) => - r.runtime->Option.map(runtime => (`worker="${workerIndex->Int.toString}"`, runtime)) + group.running->Array.filterMap(r => + r.runtime->Option.map(runtime => (`worker="${r.worker->name}"`, runtime)) ), ), ), From 662deb26afb1e8c28da3b7c481c3f5aba9249c8f Mon Sep 17 00:00:00 2001 From: dzakh Date: Wed, 16 Sep 2026 13:11:57 +0000 Subject: [PATCH 12/61] Join a worker's chain ids with a semicolon in its metrics name Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_017CceLz4mPquMmpWrU74P3j --- packages/envio/src/Supervisor.res | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index 1b43f442e..3cda45417 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -65,7 +65,7 @@ type running = { // A worker is named by the chains it drives, which is what an operator reading // its memory or its event loop wants to know. -let name = (worker: worker) => worker.chainIds->Array.map(ChainId.toString)->Array.joinUnsafe(",") +let name = (worker: worker) => worker.chainIds->Array.map(ChainId.toString)->Array.joinUnsafe(";") let label = (worker: worker) => `[chain ${worker->name}]` From 2bd63afa96d2bb513663ee12008f32d9aaa8ab74 Mon Sep 17 00:00:00 2001 From: dzakh Date: Wed, 16 Sep 2026 13:21:54 +0000 Subject: [PATCH 13/61] Report only the workers' runtime metrics on a split run The supervisor's own heap and event loop carry nothing the indexing runs on, so its readings no longer sit beside the workers' under a label of their own. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_017CceLz4mPquMmpWrU74P3j --- packages/e2e-tests/src/e2e/split-run.test.ts | 6 +- .../test/lib_tests/Metrics_test.res | 96 +++++++++---------- packages/envio/src/Supervisor.res | 10 +- 3 files changed, 55 insertions(+), 57 deletions(-) diff --git a/packages/e2e-tests/src/e2e/split-run.test.ts b/packages/e2e-tests/src/e2e/split-run.test.ts index 8fb09700a..1ede3416d 100644 --- a/packages/e2e-tests/src/e2e/split-run.test.ts +++ b/packages/e2e-tests/src/e2e/split-run.test.ts @@ -100,7 +100,7 @@ describe.skipIf(!reachable)("E2E: a split run is one indexer", () => { await waitForOutput(indexer, "Splitting 2 chains across 2 processes", config.timeouts.indexerStartup); const [runtime, metrics] = await Promise.all([ - // The supervisor's own readings alongside each worker's, told apart by label. + // Each worker's readings, told apart by label. scrapeUntil("/metrics/runtime", (body) => body.includes('worker="8453"')), // Both chains on one endpoint, whichever process drives each. scrapeUntil( @@ -115,7 +115,7 @@ describe.skipIf(!reachable)("E2E: a split run is one indexer", () => { expect({ exitCode: await exit, // Workers are named by the chains they drive. - runtimeWorkers: ["supervisor", "1", "8453"].map((worker) => + runtimeWorkers: ["1", "8453"].map((worker) => runtime.includes(`nodejs_heap_size_used_bytes{worker="${worker}"}`) ), metricsChains: [1, 8453].map((chainId) => @@ -123,7 +123,7 @@ describe.skipIf(!reachable)("E2E: a split run is one indexer", () => { ), }).toEqual({ exitCode: 0, - runtimeWorkers: [true, true, true], + runtimeWorkers: [true, true], metricsChains: [true, true], }); }); diff --git a/packages/envio-tests/test/lib_tests/Metrics_test.res b/packages/envio-tests/test/lib_tests/Metrics_test.res index 0cd7cc798..15d0277dc 100644 --- a/packages/envio-tests/test/lib_tests/Metrics_test.res +++ b/packages/envio-tests/test/lib_tests/Metrics_test.res @@ -828,8 +828,8 @@ describe("Metrics.renderRuntime", () => { t.expect(( Metrics.renderRuntime([("", sample(~heapUsed=150., ~gc=[]))])->samples, Metrics.renderRuntime([ - (`worker="supervisor"`, sample(~heapUsed=50., ~gc=[])), - (`worker="0"`, sample(~heapUsed=150., ~gc=[{kind: "minor", count: 3., seconds: 0.03}])), + (`worker="1"`, sample(~heapUsed=50., ~gc=[])), + (`worker="137"`, sample(~heapUsed=150., ~gc=[{kind: "minor", count: 3., seconds: 0.03}])), ])->samples, )).toStrictEqual(( [ @@ -857,52 +857,52 @@ describe("Metrics.renderRuntime", () => { `nodejs_version_info{version="v24.1.2",major="24",minor="1",patch="2"} 1`, ], [ - `process_cpu_user_seconds_total{worker="supervisor"} 1.5`, - `process_cpu_user_seconds_total{worker="0"} 1.5`, - `process_cpu_system_seconds_total{worker="supervisor"} 0.5`, - `process_cpu_system_seconds_total{worker="0"} 0.5`, - `process_cpu_seconds_total{worker="supervisor"} 2`, - `process_cpu_seconds_total{worker="0"} 2`, - `process_start_time_seconds{worker="supervisor"} 1700000000`, - `process_start_time_seconds{worker="0"} 1700000000`, - `process_resident_memory_bytes{worker="supervisor"} 300`, - `process_resident_memory_bytes{worker="0"} 300`, - `nodejs_heap_size_total_bytes{worker="supervisor"} 200`, - `nodejs_heap_size_total_bytes{worker="0"} 200`, - `nodejs_heap_size_used_bytes{worker="supervisor"} 50`, - `nodejs_heap_size_used_bytes{worker="0"} 150`, - `nodejs_external_memory_bytes{worker="supervisor"} 10`, - `nodejs_external_memory_bytes{worker="0"} 10`, - `nodejs_eventloop_utilization{worker="supervisor"} 0.25`, - `nodejs_eventloop_utilization{worker="0"} 0.25`, - `nodejs_eventloop_lag_mean_seconds{worker="supervisor"} 0.001`, - `nodejs_eventloop_lag_mean_seconds{worker="0"} 0.001`, - `nodejs_eventloop_lag_min_seconds{worker="supervisor"} 0`, - `nodejs_eventloop_lag_min_seconds{worker="0"} 0`, - `nodejs_eventloop_lag_max_seconds{worker="supervisor"} 0.002`, - `nodejs_eventloop_lag_max_seconds{worker="0"} 0.002`, - `nodejs_eventloop_lag_stddev_seconds{worker="supervisor"} 0.001`, - `nodejs_eventloop_lag_stddev_seconds{worker="0"} 0.001`, - `nodejs_eventloop_lag_p50_seconds{worker="supervisor"} 0.001`, - `nodejs_eventloop_lag_p50_seconds{worker="0"} 0.001`, - `nodejs_eventloop_lag_p90_seconds{worker="supervisor"} 0.002`, - `nodejs_eventloop_lag_p90_seconds{worker="0"} 0.002`, - `nodejs_eventloop_lag_p99_seconds{worker="supervisor"} 0.002`, - `nodejs_eventloop_lag_p99_seconds{worker="0"} 0.002`, - `nodejs_heap_space_size_total_bytes{worker="supervisor",space="new"} 100`, - `nodejs_heap_space_size_total_bytes{worker="0",space="new"} 100`, - `nodejs_heap_space_size_used_bytes{worker="supervisor",space="new"} 40`, - `nodejs_heap_space_size_used_bytes{worker="0",space="new"} 40`, - `nodejs_heap_space_size_available_bytes{worker="supervisor",space="new"} 60`, - `nodejs_heap_space_size_available_bytes{worker="0",space="new"} 60`, - `nodejs_active_resources{worker="supervisor",type="TCPSocketWrap"} 2`, - `nodejs_active_resources{worker="0",type="TCPSocketWrap"} 2`, - `nodejs_active_resources_total{worker="supervisor"} 2`, - `nodejs_active_resources_total{worker="0"} 2`, - `nodejs_gc_duration_seconds_sum{worker="0",kind="minor"} 0.03`, - `nodejs_gc_duration_seconds_count{worker="0",kind="minor"} 3`, - `nodejs_version_info{worker="supervisor",version="v24.1.2",major="24",minor="1",patch="2"} 1`, - `nodejs_version_info{worker="0",version="v24.1.2",major="24",minor="1",patch="2"} 1`, + `process_cpu_user_seconds_total{worker="1"} 1.5`, + `process_cpu_user_seconds_total{worker="137"} 1.5`, + `process_cpu_system_seconds_total{worker="1"} 0.5`, + `process_cpu_system_seconds_total{worker="137"} 0.5`, + `process_cpu_seconds_total{worker="1"} 2`, + `process_cpu_seconds_total{worker="137"} 2`, + `process_start_time_seconds{worker="1"} 1700000000`, + `process_start_time_seconds{worker="137"} 1700000000`, + `process_resident_memory_bytes{worker="1"} 300`, + `process_resident_memory_bytes{worker="137"} 300`, + `nodejs_heap_size_total_bytes{worker="1"} 200`, + `nodejs_heap_size_total_bytes{worker="137"} 200`, + `nodejs_heap_size_used_bytes{worker="1"} 50`, + `nodejs_heap_size_used_bytes{worker="137"} 150`, + `nodejs_external_memory_bytes{worker="1"} 10`, + `nodejs_external_memory_bytes{worker="137"} 10`, + `nodejs_eventloop_utilization{worker="1"} 0.25`, + `nodejs_eventloop_utilization{worker="137"} 0.25`, + `nodejs_eventloop_lag_mean_seconds{worker="1"} 0.001`, + `nodejs_eventloop_lag_mean_seconds{worker="137"} 0.001`, + `nodejs_eventloop_lag_min_seconds{worker="1"} 0`, + `nodejs_eventloop_lag_min_seconds{worker="137"} 0`, + `nodejs_eventloop_lag_max_seconds{worker="1"} 0.002`, + `nodejs_eventloop_lag_max_seconds{worker="137"} 0.002`, + `nodejs_eventloop_lag_stddev_seconds{worker="1"} 0.001`, + `nodejs_eventloop_lag_stddev_seconds{worker="137"} 0.001`, + `nodejs_eventloop_lag_p50_seconds{worker="1"} 0.001`, + `nodejs_eventloop_lag_p50_seconds{worker="137"} 0.001`, + `nodejs_eventloop_lag_p90_seconds{worker="1"} 0.002`, + `nodejs_eventloop_lag_p90_seconds{worker="137"} 0.002`, + `nodejs_eventloop_lag_p99_seconds{worker="1"} 0.002`, + `nodejs_eventloop_lag_p99_seconds{worker="137"} 0.002`, + `nodejs_heap_space_size_total_bytes{worker="1",space="new"} 100`, + `nodejs_heap_space_size_total_bytes{worker="137",space="new"} 100`, + `nodejs_heap_space_size_used_bytes{worker="1",space="new"} 40`, + `nodejs_heap_space_size_used_bytes{worker="137",space="new"} 40`, + `nodejs_heap_space_size_available_bytes{worker="1",space="new"} 60`, + `nodejs_heap_space_size_available_bytes{worker="137",space="new"} 60`, + `nodejs_active_resources{worker="1",type="TCPSocketWrap"} 2`, + `nodejs_active_resources{worker="137",type="TCPSocketWrap"} 2`, + `nodejs_active_resources_total{worker="1"} 2`, + `nodejs_active_resources_total{worker="137"} 2`, + `nodejs_gc_duration_seconds_sum{worker="137",kind="minor"} 0.03`, + `nodejs_gc_duration_seconds_count{worker="137",kind="minor"} 3`, + `nodejs_version_info{worker="1",version="v24.1.2",major="24",minor="1",patch="2"} 1`, + `nodejs_version_info{worker="137",version="v24.1.2",major="24",minor="1",patch="2"} 1`, ], )) }) diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index 3cda45417..f4765f03b 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -297,14 +297,12 @@ let run = async (~workers: array, ~configJson: JSON.t, ~reset) => { | snapshots => Some(snapshots->merge) }, ~envioVersion=Utils.EnvioPackage.value.version, - // Every process of the run, the supervisor included, under a `worker` - // label: the memory and the event loop that matter are the workers' own. + // The workers' readings, each under a `worker` label: theirs are the memory + // and the event loop the indexing runs on. ~collectRuntime=() => Metrics.renderRuntime( - [(`worker="supervisor"`, Metrics.sampleRuntime())]->Array.concat( - group.running->Array.filterMap(r => - r.runtime->Option.map(runtime => (`worker="${r.worker->name}"`, runtime)) - ), + group.running->Array.filterMap(r => + r.runtime->Option.map(runtime => (`worker="${r.worker->name}"`, runtime)) ), ), ~isDevelopmentMode=config.isDev, From 8887a4817173509840e5693c98add0abc3c4bf86 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 13:36:48 +0000 Subject: [PATCH 14/61] End the runtime metrics body with a line feed, and leave no indexer behind /metrics/runtime served a body whose last line had no line feed, which a strict Prometheus consumer reads as a truncated stream and drops whole; `Metrics.collect` already appended one, `renderRuntime` did not. The split-run e2e tests now stop their indexer in a `finally`, so a failure before the test's own shutdown can't leave a process holding the port. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- packages/e2e-tests/src/e2e/split-run.test.ts | 111 +++++++++++------- .../test/lib_tests/Metrics_test.res | 8 ++ packages/envio/src/Metrics.res | 2 +- 3 files changed, 75 insertions(+), 46 deletions(-) diff --git a/packages/e2e-tests/src/e2e/split-run.test.ts b/packages/e2e-tests/src/e2e/split-run.test.ts index 1ede3416d..9ef4c6aaa 100644 --- a/packages/e2e-tests/src/e2e/split-run.test.ts +++ b/packages/e2e-tests/src/e2e/split-run.test.ts @@ -39,6 +39,19 @@ if (!reachable && process.env.CI) { const exitCode = (child: ChildProcess) => new Promise((resolve) => child.on("close", resolve)); +/** + * Leaves no indexer behind when a test fails before its own shutdown: one that + * survived would hold the port and the schema against everything after it. + */ +const stopIfRunning = async (indexer: ChildProcess) => { + if (indexer.exitCode !== null || indexer.signalCode !== null) return; + const exited = exitCode(indexer); + indexer.kill("SIGINT"); + const abandon = setTimeout(() => indexer.kill("SIGKILL"), 10_000); + await exited; + clearTimeout(abandon); +}; + /** Polls an endpoint of the supervisor until its body satisfies `ready`. */ const scrapeUntil = async (route: string, ready: (body: string) => boolean) => { const deadline = Date.now() + config.timeouts.indexerStartup; @@ -75,56 +88,64 @@ describe.skipIf(!reachable)("E2E: a split run is one indexer", () => { it("Exits once every chain is done, with both chains' rows written", async () => { const indexer = start(["-r"]); - const exit = exitCode(indexer); - await waitForOutput(indexer, "Splitting 2 chains across 2 processes", config.timeouts.indexerStartup); - - expect({ - exitCode: await exit, - rowsPerChain: await pgRows( - `SELECT "chain_id", COUNT(*) > 0 FROM "${PG_SCHEMA}"."Transfer" GROUP BY "chain_id" ORDER BY "chain_id"` - ), - }).toEqual({ - exitCode: 0, - rowsPerChain: [ - [1, true], - [8453, true], - ], - }); + try { + const exit = exitCode(indexer); + await waitForOutput(indexer, "Splitting 2 chains across 2 processes", config.timeouts.indexerStartup); + + expect({ + exitCode: await exit, + rowsPerChain: await pgRows( + `SELECT "chain_id", COUNT(*) > 0 FROM "${PG_SCHEMA}"."Transfer" GROUP BY "chain_id" ORDER BY "chain_id"` + ), + }).toEqual({ + exitCode: 0, + rowsPerChain: [ + [1, true], + [8453, true], + ], + }); + } finally { + await stopIfRunning(indexer); + } }); // The same chains with no end block: a run that nothing but a stop ends, so // there is time to read what it serves. it("Serves every process's metrics, and stops them all on one interrupt", async () => { const indexer = start(["-r", "--config", "config.head.yaml"]); - const exit = exitCode(indexer); - await waitForOutput(indexer, "Splitting 2 chains across 2 processes", config.timeouts.indexerStartup); - - const [runtime, metrics] = await Promise.all([ - // Each worker's readings, told apart by label. - scrapeUntil("/metrics/runtime", (body) => body.includes('worker="8453"')), - // Both chains on one endpoint, whichever process drives each. - scrapeUntil( - "/metrics", - (body) => body.includes('chainId="1"') && body.includes('chainId="8453"') - ), - ]); - - // Only the supervisor is signalled, the way a process manager would. - indexer.kill("SIGINT"); - - expect({ - exitCode: await exit, - // Workers are named by the chains they drive. - runtimeWorkers: ["1", "8453"].map((worker) => - runtime.includes(`nodejs_heap_size_used_bytes{worker="${worker}"}`) - ), - metricsChains: [1, 8453].map((chainId) => - metrics.includes(`envio_progress_block{chainId="${chainId}"}`) - ), - }).toEqual({ - exitCode: 0, - runtimeWorkers: [true, true], - metricsChains: [true, true], - }); + try { + const exit = exitCode(indexer); + await waitForOutput(indexer, "Splitting 2 chains across 2 processes", config.timeouts.indexerStartup); + + const [runtime, metrics] = await Promise.all([ + // Each worker's readings, told apart by label. + scrapeUntil("/metrics/runtime", (body) => body.includes('worker="8453"')), + // Both chains on one endpoint, whichever process drives each. + scrapeUntil( + "/metrics", + (body) => body.includes('chainId="1"') && body.includes('chainId="8453"') + ), + ]); + + // Only the supervisor is signalled, the way a process manager would. + indexer.kill("SIGINT"); + + expect({ + exitCode: await exit, + // Workers are named by the chains they drive. + runtimeWorkers: ["1", "8453"].map((worker) => + runtime.includes(`nodejs_heap_size_used_bytes{worker="${worker}"}`) + ), + metricsChains: [1, 8453].map((chainId) => + metrics.includes(`envio_progress_block{chainId="${chainId}"}`) + ), + }).toEqual({ + exitCode: 0, + runtimeWorkers: [true, true], + metricsChains: [true, true], + }); + } finally { + await stopIfRunning(indexer); + } }); }); diff --git a/packages/envio-tests/test/lib_tests/Metrics_test.res b/packages/envio-tests/test/lib_tests/Metrics_test.res index 15d0277dc..cc1aff025 100644 --- a/packages/envio-tests/test/lib_tests/Metrics_test.res +++ b/packages/envio-tests/test/lib_tests/Metrics_test.res @@ -906,4 +906,12 @@ describe("Metrics.renderRuntime", () => { ], )) }) + + // A scrape whose last line has no line feed is a parse error to a strict + // consumer, which drops the whole body rather than its last sample. + it("Ends its body with a line feed, the way the text format requires", t => { + t.expect( + Metrics.renderRuntime([("", sample(~heapUsed=150., ~gc=[]))])->String.endsWith("\n"), + ).toBe(true) + }) }) diff --git a/packages/envio/src/Metrics.res b/packages/envio/src/Metrics.res index 59993ef1c..77bfcceea 100644 --- a/packages/envio/src/Metrics.res +++ b/packages/envio/src/Metrics.res @@ -1250,7 +1250,7 @@ let renderRuntime = (samples: array<(string, runtimeSample)>) => { }), ~value=() => 1., ) - b.out + b.out ++ "\n" } let collectRuntime = () => renderRuntime([("", sampleRuntime())]) From 000c26eb06e45ecdd8149762c4378b33260e3649 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 17 Sep 2026 08:28:57 +0000 Subject: [PATCH 15/61] Sample the runtime where it is reported, and tidy the review leftovers A supervisor served `/metrics/runtime` from its workers' samples but still started collectors of its own, whose readings nothing rendered. Starting them is now the job of a process that reports its own. Also: say what `ENVIO_PG_MAX_CONNECTIONS` now means where it is read, drop a redundant clause from the worker log path, and put `base` back where it was in the logger. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- packages/envio/src/Env.res | 2 ++ packages/envio/src/Logging.res | 7 ++++--- packages/envio/src/Main.res | 3 +-- packages/envio/src/Supervisor.res | 6 +++--- 4 files changed, 10 insertions(+), 8 deletions(-) diff --git a/packages/envio/src/Env.res b/packages/envio/src/Env.res index ab23da20d..5df6e03b6 100644 --- a/packages/envio/src/Env.res +++ b/packages/envio/src/Env.res @@ -127,6 +127,8 @@ module Db = { //the SSL modes should be provided as string otherwise as 'require' | 'allow' | 'prefer' | 'verify-full' ~devFallback=Bool(false), ) + // The budget for the whole run, not for one process: a run that splits across + // workers divides it among them, and each caps its own pool to its share. let maxConnections = envSafe->EnvSafe.get("ENVIO_PG_MAX_CONNECTIONS", S.int, ~fallback=2) } diff --git a/packages/envio/src/Logging.res b/packages/envio/src/Logging.res index 8b902e032..abea6133d 100644 --- a/packages/envio/src/Logging.res +++ b/packages/envio/src/Logging.res @@ -42,9 +42,6 @@ let makeLogger = (~logStrategy, ~logFilePath, ~defaultFileLogLevel, ~userLogLeve level: defaultFileLogLevel, } - // Empty base disables pid and hostname in logs - let base: JSON.t = %raw("{}") - let makeMultiStreamLogger = MultiStreamLogger.make( ~userLogLevel, ~defaultFileLogLevel, @@ -52,6 +49,9 @@ let makeLogger = (~logStrategy, ~logFilePath, ~defaultFileLogLevel, ~userLogLeve ... ) + // Empty base disables pid and hostname in logs + let base: JSON.t = %raw("{}") + switch logStrategy { | EcsFile => makeWithOptionsAndTransport( @@ -155,6 +155,7 @@ let childFatal = (logger, params: 'a) => { let createChild = (~params: 'a) => { getLogger()->child(params->createChildParams) } + // Fields every line this process logs from here on carries. What belongs on // them is the run's to decide; the logger only carries what it is handed. let setContext = (params: 'a) => diff --git a/packages/envio/src/Main.res b/packages/envio/src/Main.res index 9353be315..97f96f8b1 100644 --- a/packages/envio/src/Main.res +++ b/packages/envio/src/Main.res @@ -555,8 +555,6 @@ let startServer = ( } }) - Metrics.startRuntimeCollectors() - app->get("/metrics", (_req, res) => { res->set("Content-Type", Metrics.contentType) let _ = res->endWithData(Metrics.collect(~metrics=getMetrics())) @@ -723,6 +721,7 @@ let start = async ( // A worker reports through its supervisor, which owns the one server and the // one display the run has. if !isTest && !Worker.isEnabled { + Metrics.startRuntimeCollectors() startServer( ~onSyncCache=() => dumpEffectCache()->Promise.thenResolve(ignore), ~collectRuntime=Metrics.collectRuntime, diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index f4765f03b..ff3738e4c 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -73,10 +73,10 @@ let label = (worker: worker) => `[chain ${worker->name}]` // several processes appending to one file can still tear a long line apart. let logFilePath = (~workerIndex, ~path=Env.logFilePath) => { let suffix = `.worker-${workerIndex->Int.toString}` - // A dot in a directory name isn't an extension: `./logs/envio` keeps its - // whole path and takes the suffix at the end. + // A dot in a directory name isn't an extension, and a path with no dot at + // all has none either: `./logs/envio` takes the suffix at the end. let dot = path->String.lastIndexOf(".") - if dot > path->String.lastIndexOf("/") && dot !== -1 { + if dot > path->String.lastIndexOf("/") { `${path->String.slice(~start=0, ~end=dot)}${suffix}${path->String.slice( ~start=dot, ~end=path->String.length, From 342f1c2e39bd4c99acbf28a6e4490c5feb841536 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 17 Sep 2026 08:47:31 +0000 Subject: [PATCH 16/61] Give up a cache sync when the worker it waits on is gone A worker that died mid-dump left its request waiting for an acknowledgement that could never come, and `/console/syncCache` held the connection open for as long as the supervisor lived. A worker's end now settles whatever waits on it, a request that arrives once it is already gone is settled the same way, and the endpoint answers the `false` it already has for a dump it cannot confirm rather than never answering at all. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- .../envio-tests/test/SupervisorFork_test.res | 29 +++++++++ .../envio-tests/test/helpers/fakeWorker.mjs | 8 ++- packages/envio/src/Main.res | 8 +++ packages/envio/src/Supervisor.res | 62 ++++++++++++++----- 4 files changed, 88 insertions(+), 19 deletions(-) diff --git a/packages/envio-tests/test/SupervisorFork_test.res b/packages/envio-tests/test/SupervisorFork_test.res index 9798d1c08..9452f4e08 100644 --- a/packages/envio-tests/test/SupervisorFork_test.res +++ b/packages/envio-tests/test/SupervisorFork_test.res @@ -89,6 +89,35 @@ describe("Supervisor.syncCache", () => { t.expect(outcomes).toStrictEqual(("answered", "answered")) }) + + Async.it("Gives up on a worker that is gone before it could dump", async t => { + NodeJs.Process.process.env->Dict.set("FAKE_WORKER", "mute") + let running = forkFixture(~chainIds=[1]) + let group: Supervisor.group = {running: [running], stopping: false, syncing: None} + + let settled = async request => + switch await request { + | () => "answered" + | exception _ => "gave up" + } + // Raced, so a request that never settles reads as the hang it is rather + // than as the suite timing out. + let answered = request => + Promise.race([ + request->settled, + Utils.delay(2000)->Promise.thenResolve(() => "still waiting"), + ]) + + let duringDump = group->Supervisor.syncCache + running.child->NodeJs.ChildProcess.kill("SIGTERM")->ignore + let outcomes = ( + await duringDump->answered, + // And a request that arrives once it is already gone. + await group->Supervisor.syncCache->answered, + ) + + t.expect(outcomes).toStrictEqual(("gave up", "gave up")) + }) }) describe("Supervisor.awaitExit", () => { diff --git a/packages/envio-tests/test/helpers/fakeWorker.mjs b/packages/envio-tests/test/helpers/fakeWorker.mjs index 47251b576..d9de89982 100644 --- a/packages/envio-tests/test/helpers/fakeWorker.mjs +++ b/packages/envio-tests/test/helpers/fakeWorker.mjs @@ -19,7 +19,9 @@ process.on("message", (message) => { if (mode === "succeed") process.exit(0); if (mode === "fail") process.exit(1); } - if (message.kind === "syncCache") { + // A "mute" worker takes the request and never answers, the way one that dies + // mid-dump leaves it. + if (message.kind === "syncCache" && mode !== "mute") { // Reports the dump through a snapshot before acknowledging it, so a // supervisor that answers early can be caught having answered before it. setTimeout(() => { @@ -29,5 +31,5 @@ process.on("message", (message) => { } }); -// Nothing else keeps a "linger" worker alive; it waits to be stopped. -if (mode === "linger") setInterval(() => {}, 1000); +// Nothing else keeps these alive; they wait to be stopped. +if (mode === "linger" || mode === "mute") setInterval(() => {}, 1000); diff --git a/packages/envio/src/Main.res b/packages/envio/src/Main.res index 97f96f8b1..97bb649e9 100644 --- a/packages/envio/src/Main.res +++ b/packages/envio/src/Main.res @@ -549,6 +549,14 @@ let startServer = ( if isDevelopmentMode { onSyncCache() ->Promise.thenResolve(() => res->json(Boolean(true))) + // A dump that couldn't be made, or couldn't be confirmed, answers the + // same `false` a disabled console does. Leaving it unanswered would hold + // the request open for as long as the indexer runs. + ->Promise.catch(exn => { + Logging.errorWithExn(exn, "Failed to sync the effect cache") + res->json(Boolean(false)) + Promise.resolve() + }) ->Promise.ignore } else { res->json(Boolean(false)) diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index ff3738e4c..7d410aa9c 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -48,6 +48,11 @@ let planForRun = (~config: Config.t, ~maxConnections=Env.Db.maxConnections) => plan(~chainIds=config.chainMap->ChainMap.values->Array.map(chain => chain.id), ~maxConnections) } +// A console request waiting on one worker's cache dump. Settled by the worker's +// acknowledgement, or given up on when the worker is gone before it could send +// one. +type cacheSync = {done: unit => unit, giveUp: exn => unit} + // One forked worker: the process, the chains it drives, and the last snapshot // it reported. `None` until it reports, which is what makes a run that hasn't // heard from anyone yet render as initializing rather than as empty. @@ -59,8 +64,11 @@ type running = { // A spawn failure can raise `error` and `exit` both, and a worker counted // twice would end the run while its siblings are still indexing. mutable settled: bool, - // Waiting for this worker's cache dump, when a console asked for one. - mutable onCacheSynced: option unit>, + // Whether the process is gone. Told apart from `settled`, which is the run's + // count of the workers it has accounted for. + mutable ended: bool, + // The console request waiting on this worker's cache dump, if one is. + mutable cacheSync: option, } // A worker is named by the chains it drives, which is what an operator reading @@ -69,6 +77,9 @@ let name = (worker: worker) => worker.chainIds->Array.map(ChainId.toString)->Arr let label = (worker: worker) => `[chain ${worker->name}]` +let endedBeforeDump = worker => + Utils.Error.make(`${worker->label} ended before it could dump its cache`) + // Workers append to files of their own. Pino writes a line per call, and // several processes appending to one file can still tear a long line apart. let logFilePath = (~workerIndex, ~path=Env.logFilePath) => { @@ -136,23 +147,34 @@ let fork = ( snapshot: None, runtime: None, settled: false, - onCacheSynced: None, + ended: false, + cacheSync: None, } + let settleCacheSync = settle => + switch running.cacheSync { + | Some(cacheSync) => { + running.cacheSync = None + settle(cacheSync) + } + | None => () + } child->NodeJs.ChildProcess.onMessage(message => switch message { | Worker.Snapshot({metrics, runtime}) => { running.snapshot = Some(metrics) running.runtime = Some(runtime) } - | Worker.CacheSynced(_) => - switch running.onCacheSynced { - | Some(resolve) => - running.onCacheSynced = None - resolve() - | None => () - } + | Worker.CacheSynced(_) => settleCacheSync(cacheSync => cacheSync.done()) } ) + // A worker that ends mid-dump answers nothing ever again, and a request left + // waiting on it would hang for as long as the supervisor lives. + let end = () => { + running.ended = true + settleCacheSync(cacheSync => cacheSync.giveUp(worker->endedBeforeDump)) + } + child->NodeJs.ChildProcess.onExit((_code, _signal) => end()) + child->NodeJs.ChildProcess.onChildError(_ => end()) running } @@ -181,15 +203,23 @@ let syncCache = group => | None => let inFlight = group.running - ->Array.filter(r => !r.settled) ->Array.map(r => - Promise.make((resolve, _) => { - r.onCacheSynced = Some(() => resolve()) - r.child->NodeJs.ChildProcess.send(Worker.SyncCache({}))->ignore - }) + Promise.make((resolve, reject) => + // A worker already gone has no dump left to ask for, and the request + // is answered by the same give-up as one that dies mid-dump. + if r.ended { + reject(r.worker->endedBeforeDump) + } else { + r.cacheSync = Some({done: () => resolve(), giveUp: reject}) + r.child->NodeJs.ChildProcess.send(Worker.SyncCache({}))->ignore + } + ) ) ->Promise.all - ->Promise.thenResolve(_ => group.syncing = None) + ->Promise.thenResolve(ignore) + // However it ended, the next request asks for a dump of its own rather + // than joining one that is over. + ->Promise.finally(() => group.syncing = None) group.syncing = Some(inFlight) inFlight } From 06f1c8cc3003627f45955d4f47092013667f36bb Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 17 Sep 2026 08:58:24 +0000 Subject: [PATCH 17/61] Dump the cache from the supervisor rather than from every worker A dump copies every effect cache table in the schema to a file named after the effect, so asking the workers for one had each of them copy its siblings' chains as well, all of them writing the same files at the same time. Nothing about it is a worker's to know: the rows it copies are the ones already committed, which any process with a connection can read. The supervisor does it alone now, on a connection it opens for the dump and closes after, and requests that overlap join the dump in flight. That takes the `syncCache`/`cacheSynced` message pair, the per-worker waiters and the queue that held a request until a worker could serve it back out again. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- .../envio-tests/test/SupervisorFork_test.res | 78 +------------ .../envio-tests/test/helpers/fakeWorker.mjs | 14 +-- .../test/lib_tests/Supervisor_test.res | 18 +++ .../test/lib_tests/Worker_test.res | 14 --- packages/envio/src/Bin.res | 1 - packages/envio/src/Main.res | 9 -- packages/envio/src/PgStorage.res | 4 +- packages/envio/src/Supervisor.res | 106 +++++------------- packages/envio/src/Worker.res | 31 +---- 9 files changed, 54 insertions(+), 221 deletions(-) delete mode 100644 packages/envio-tests/test/lib_tests/Worker_test.res diff --git a/packages/envio-tests/test/SupervisorFork_test.res b/packages/envio-tests/test/SupervisorFork_test.res index 9452f4e08..fbace3981 100644 --- a/packages/envio-tests/test/SupervisorFork_test.res +++ b/packages/envio-tests/test/SupervisorFork_test.res @@ -9,9 +9,6 @@ type fixtureReport = { startTime: Date.t, } -// What it reports once its dump is done. -type fixtureDump = {synced: bool} - let fixturePath = `${NodeJs.Process.cwd()}/test/helpers/fakeWorker.mjs` let forkFixture = (~chainIds, ~maxConnections=2, ~workerIndex=0) => @@ -33,7 +30,6 @@ describe("Supervisor.fork", () => { switch message { | Worker.Snapshot({metrics}) => resolve(metrics->(Utils.magic: Metrics.t => fixtureReport)) - | Worker.CacheSynced(_) => () }, ), ) @@ -50,76 +46,6 @@ describe("Supervisor.fork", () => { }) }) -describe("Supervisor.syncCache", () => { - Async.it("Answers only once every worker has dumped its cache", async t => { - NodeJs.Process.process.env->Dict.set("FAKE_WORKER", "linger") - let group: Supervisor.group = { - running: [forkFixture(~chainIds=[1]), forkFixture(~chainIds=[137])], - stopping: false, - syncing: None, - } - - await group->Supervisor.syncCache - let dumped = - group.running->Array.map(r => - r.snapshot->Option.map(metrics => (metrics->(Utils.magic: Metrics.t => fixtureDump)).synced) - ) - group->Supervisor.stop - - t.expect(dumped).toStrictEqual([Some(true), Some(true)]) - }) - - Async.it("Answers requests that overlap rather than leaving the first hanging", async t => { - NodeJs.Process.process.env->Dict.set("FAKE_WORKER", "linger") - let group: Supervisor.group = { - running: [forkFixture(~chainIds=[1])], - stopping: false, - syncing: None, - } - - let answered = async request => - await Promise.race([ - request->Promise.thenResolve(() => "answered"), - Utils.delay(2000)->Promise.thenResolve(() => "still waiting"), - ]) - let first = group->Supervisor.syncCache - let second = group->Supervisor.syncCache - let outcomes = (await first->answered, await second->answered) - group->Supervisor.stop - - t.expect(outcomes).toStrictEqual(("answered", "answered")) - }) - - Async.it("Gives up on a worker that is gone before it could dump", async t => { - NodeJs.Process.process.env->Dict.set("FAKE_WORKER", "mute") - let running = forkFixture(~chainIds=[1]) - let group: Supervisor.group = {running: [running], stopping: false, syncing: None} - - let settled = async request => - switch await request { - | () => "answered" - | exception _ => "gave up" - } - // Raced, so a request that never settles reads as the hang it is rather - // than as the suite timing out. - let answered = request => - Promise.race([ - request->settled, - Utils.delay(2000)->Promise.thenResolve(() => "still waiting"), - ]) - - let duringDump = group->Supervisor.syncCache - running.child->NodeJs.ChildProcess.kill("SIGTERM")->ignore - let outcomes = ( - await duringDump->answered, - // And a request that arrives once it is already gone. - await group->Supervisor.syncCache->answered, - ) - - t.expect(outcomes).toStrictEqual(("gave up", "gave up")) - }) -}) - describe("Supervisor.awaitExit", () => { let outcome = async group => switch await group->Supervisor.awaitExit { @@ -132,7 +58,6 @@ describe("Supervisor.awaitExit", () => { let group: Supervisor.group = { running: [forkFixture(~chainIds=[1]), forkFixture(~chainIds=[137])], stopping: false, - syncing: None, } t.expect(await outcome(group)).toStrictEqual(Ok(Supervisor.Finished)) @@ -145,7 +70,6 @@ describe("Supervisor.awaitExit", () => { let group: Supervisor.group = { running: [forkFixture(~chainIds=[1]), forkFixture(~chainIds=[137])], stopping: false, - syncing: None, } group->Supervisor.stop @@ -158,7 +82,7 @@ describe("Supervisor.awaitExit", () => { let failing = forkFixture(~chainIds=[1]) NodeJs.Process.process.env->Dict.set("FAKE_WORKER", "linger") let lingering = forkFixture(~chainIds=[137]) - let group: Supervisor.group = {running: [failing, lingering], stopping: false, syncing: None} + let group: Supervisor.group = {running: [failing, lingering], stopping: false} // The survivor was taken down rather than left indexing half a schema. t.expect((await outcome(group), group.stopping, lingering.settled)).toStrictEqual(( diff --git a/packages/envio-tests/test/helpers/fakeWorker.mjs b/packages/envio-tests/test/helpers/fakeWorker.mjs index d9de89982..18c7b8572 100644 --- a/packages/envio-tests/test/helpers/fakeWorker.mjs +++ b/packages/envio-tests/test/helpers/fakeWorker.mjs @@ -19,17 +19,7 @@ process.on("message", (message) => { if (mode === "succeed") process.exit(0); if (mode === "fail") process.exit(1); } - // A "mute" worker takes the request and never answers, the way one that dies - // mid-dump leaves it. - if (message.kind === "syncCache" && mode !== "mute") { - // Reports the dump through a snapshot before acknowledging it, so a - // supervisor that answers early can be caught having answered before it. - setTimeout(() => { - process.send({ kind: "snapshot", metrics: { synced: true } }); - process.send({ kind: "cacheSynced" }); - }, 50); - } }); -// Nothing else keeps these alive; they wait to be stopped. -if (mode === "linger" || mode === "mute") setInterval(() => {}, 1000); +// Nothing else keeps a "linger" worker alive; it waits to be stopped. +if (mode === "linger") setInterval(() => {}, 1000); diff --git a/packages/envio-tests/test/lib_tests/Supervisor_test.res b/packages/envio-tests/test/lib_tests/Supervisor_test.res index c32082f2e..2471e504c 100644 --- a/packages/envio-tests/test/lib_tests/Supervisor_test.res +++ b/packages/envio-tests/test/lib_tests/Supervisor_test.res @@ -200,3 +200,21 @@ describe("Worker.detect", () => { ]).toStrictEqual([true, false, false]) }) }) + +describe("Supervisor.syncCache", () => { + Async.it("Dumps once for requests that overlap, and again for a later one", async t => { + let dumps = ref(0) + let dump = () => { + dumps := dumps.contents + 1 + Utils.delay(20) + } + + let first = Supervisor.syncCache(~dump) + let second = Supervisor.syncCache(~dump) + await first + await second + await Supervisor.syncCache(~dump) + + t.expect(dumps.contents).toBe(2) + }) +}) diff --git a/packages/envio-tests/test/lib_tests/Worker_test.res b/packages/envio-tests/test/lib_tests/Worker_test.res deleted file mode 100644 index e18ebfc4c..000000000 --- a/packages/envio-tests/test/lib_tests/Worker_test.res +++ /dev/null @@ -1,14 +0,0 @@ -open Vitest - -describe("Worker.onParentMessage", () => { - Async.it("Holds a request that arrived before the indexer could handle it", async t => { - Worker.listen() - NodeJs.Process.emitMessage(Worker.SyncCache({}))->ignore - - let handled = [] - Worker.onParentMessage(message => handled->Array.push(message)) - NodeJs.Process.emitMessage(Worker.SyncCache({}))->ignore - - t.expect(handled).toStrictEqual([Worker.SyncCache({}), Worker.SyncCache({})]) - }) -}) diff --git a/packages/envio/src/Bin.res b/packages/envio/src/Bin.res index 346a1e32d..f24088388 100644 --- a/packages/envio/src/Bin.res +++ b/packages/envio/src/Bin.res @@ -53,7 +53,6 @@ let run = async args => { try { if Worker.isEnabled { Worker.bindToSupervisor() - Worker.listen() // A worker is handed the config its supervisor already parsed, narrowed to // the chains it drives, so the two can't disagree about what is indexed. // Its working directory and environment came with the fork. diff --git a/packages/envio/src/Main.res b/packages/envio/src/Main.res index 97bb649e9..5424850c9 100644 --- a/packages/envio/src/Main.res +++ b/packages/envio/src/Main.res @@ -753,15 +753,6 @@ let start = async ( let _rerender = Tui.start(~config, ~getMetrics=() => state->IndexerState.toMetrics) } if Worker.isEnabled { - Worker.onParentMessage(message => - switch message { - | SyncCache(_) => - dumpEffectCache() - ->Promise.thenResolve(() => Worker.send(CacheSynced({}))) - ->Promise.ignore - | Init(_) => () - } - ) Metrics.startRuntimeCollectors() let _intervalId = setInterval( () => diff --git a/packages/envio/src/PgStorage.res b/packages/envio/src/PgStorage.res index e2dadb3cd..5c1508765 100644 --- a/packages/envio/src/PgStorage.res +++ b/packages/envio/src/PgStorage.res @@ -1,4 +1,4 @@ -let makeClient = () => { +let makeClient = (~maxConnections=Env.Db.maxConnections) => { Postgres.makeSql( ~config={ host: Env.Db.host, @@ -14,7 +14,7 @@ let makeClient = () => { : Some(_str => ()) ), transform: {undefined: Null}, - max: Env.Db.maxConnections, + max: maxConnections, // debug: (~connection, ~query, ~params as _, ~types as _) => Js.log2(connection, query), }, ) diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index 7d410aa9c..09f569c77 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -48,11 +48,6 @@ let planForRun = (~config: Config.t, ~maxConnections=Env.Db.maxConnections) => plan(~chainIds=config.chainMap->ChainMap.values->Array.map(chain => chain.id), ~maxConnections) } -// A console request waiting on one worker's cache dump. Settled by the worker's -// acknowledgement, or given up on when the worker is gone before it could send -// one. -type cacheSync = {done: unit => unit, giveUp: exn => unit} - // One forked worker: the process, the chains it drives, and the last snapshot // it reported. `None` until it reports, which is what makes a run that hasn't // heard from anyone yet render as initializing rather than as empty. @@ -64,11 +59,6 @@ type running = { // A spawn failure can raise `error` and `exit` both, and a worker counted // twice would end the run while its siblings are still indexing. mutable settled: bool, - // Whether the process is gone. Told apart from `settled`, which is the run's - // count of the workers it has accounted for. - mutable ended: bool, - // The console request waiting on this worker's cache dump, if one is. - mutable cacheSync: option, } // A worker is named by the chains it drives, which is what an operator reading @@ -77,9 +67,6 @@ let name = (worker: worker) => worker.chainIds->Array.map(ChainId.toString)->Arr let label = (worker: worker) => `[chain ${worker->name}]` -let endedBeforeDump = worker => - Utils.Error.make(`${worker->label} ended before it could dump its cache`) - // Workers append to files of their own. Pino writes a line per call, and // several processes appending to one file can still tear a long line apart. let logFilePath = (~workerIndex, ~path=Env.logFilePath) => { @@ -141,88 +128,54 @@ let fork = ( ->NodeJs.ChildProcess.send(Worker.Init({config: configJson->configForWorker(~worker)})) ->ignore - let running = { - worker, - child, - snapshot: None, - runtime: None, - settled: false, - ended: false, - cacheSync: None, - } - let settleCacheSync = settle => - switch running.cacheSync { - | Some(cacheSync) => { - running.cacheSync = None - settle(cacheSync) - } - | None => () - } + let running = {worker, child, snapshot: None, runtime: None, settled: false} child->NodeJs.ChildProcess.onMessage(message => switch message { | Worker.Snapshot({metrics, runtime}) => { running.snapshot = Some(metrics) running.runtime = Some(runtime) } - | Worker.CacheSynced(_) => settleCacheSync(cacheSync => cacheSync.done()) } ) - // A worker that ends mid-dump answers nothing ever again, and a request left - // waiting on it would hang for as long as the supervisor lives. - let end = () => { - running.ended = true - settleCacheSync(cacheSync => cacheSync.giveUp(worker->endedBeforeDump)) - } - child->NodeJs.ChildProcess.onExit((_code, _signal) => end()) - child->NodeJs.ChildProcess.onChildError(_ => end()) running } // The forked workers of one run, and whether their supervisor is the one // taking them down. A stop it asked for is expected; every other way a worker // can end is a failure. -type group = { - running: array, - mutable stopping: bool, - // The dump in flight, if any. A second request joins it rather than asking - // for a dump of its own, so overlapping requests both get an answer. - mutable syncing: option>, -} +type group = {running: array, mutable stopping: bool} let stop = group => { group.stopping = true group.running->Array.forEach(r => r.child->NodeJs.ChildProcess.kill("SIGTERM")->ignore) } -// Dumps every worker's effect cache. Resolves once they have all reported the -// dump done, so the console the supervisor serves can't answer for writes that -// are still in flight. -let syncCache = group => - switch group.syncing { - | Some(inFlight) => inFlight - | None => - let inFlight = - group.running - ->Array.map(r => - Promise.make((resolve, reject) => - // A worker already gone has no dump left to ask for, and the request - // is answered by the same give-up as one that dies mid-dump. - if r.ended { - reject(r.worker->endedBeforeDump) - } else { - r.cacheSync = Some({done: () => resolve(), giveUp: reject}) - r.child->NodeJs.ChildProcess.send(Worker.SyncCache({}))->ignore - } - ) - ) - ->Promise.all - ->Promise.thenResolve(ignore) - // However it ended, the next request asks for a dump of its own rather - // than joining one that is over. - ->Promise.finally(() => group.syncing = None) - group.syncing = Some(inFlight) - inFlight - } +// The dev console's cache dump, which belongs to the supervisor rather than to +// its workers: a dump copies every effect cache table in the schema to a file +// named after the effect, so a worker asked to do it would copy its siblings' +// chains too, and several asked at once would write the same files at the same +// time. Nothing in it is a worker's to know — the rows it copies are the ones +// already committed. +// +// Requests that overlap join the dump in flight, for the same reason. +let syncCache = { + let inFlight = ref(None) + (~dump) => + switch inFlight.contents { + | Some(dumping) => dumping + | None => + let dumping = dump()->Promise.finally(() => inFlight := None) + inFlight := Some(dumping) + dumping + } +} + +// The supervisor handed its connections to the workers, so a dump opens one of +// its own for as long as it takes. +let dumpCache = (~config) => { + let storage = PgStorage.makeStorageFromEnv(~config, ~sql=PgStorage.makeClient(~maxConnections=1)) + storage.dumpEffectCache()->Promise.finally(() => storage.close()->Promise.ignore) +} // How a group ended. `Finished` is every worker exiting cleanly on its own, // which is what indexing to every end block looks like. @@ -306,7 +259,6 @@ let run = async (~workers: array, ~configJson: JSON.t, ~reset) => { worker->fork(~workerIndex, ~configJson) ), stopping: false, - syncing: None, } let reported = () => group.running->Array.filterMap(r => r.snapshot) @@ -336,7 +288,7 @@ let run = async (~workers: array, ~configJson: JSON.t, ~reset) => { ), ), ~isDevelopmentMode=config.isDev, - ~onSyncCache=() => group->syncCache, + ~onSyncCache=() => syncCache(~dump=() => dumpCache(~config)), ) let shouldUseTui = Main.shouldUseTui() diff --git a/packages/envio/src/Worker.res b/packages/envio/src/Worker.res index 76e109b46..888a15a50 100644 --- a/packages/envio/src/Worker.res +++ b/packages/envio/src/Worker.res @@ -22,14 +22,10 @@ type parentMessage = // instead of re-derived so a worker and its supervisor can never disagree // about what is being indexed. | @as("init") Init({config: JSON.t}) - | @as("syncCache") SyncCache({}) @tag("kind") type workerMessage = | @as("snapshot") Snapshot({metrics: Metrics.t, runtime: Metrics.runtimeSample}) - // Sent once the worker's effect cache is on disk, so the supervisor's - // console can answer for a dump that has actually happened. - | @as("cacheSynced") CacheSynced({}) // How often a worker reports. Matches the TUI's own refresh, so the supervised // display moves at the same rate an unsplit run's does. @@ -54,35 +50,12 @@ let send = (message: workerMessage) => NodeJs.Process.sendToParent(message)->ignore } -%%private(let pending: ref> = ref([])) -%%private(let handler: ref unit>> = ref(None)) - -// Installed before the indexer starts. A request can reach a worker while it is -// still coming up, and a dropped one leaves the supervisor waiting for an answer -// that will never be sent, so it waits for its handler instead. -let listen = () => - NodeJs.Process.onMessage((message: parentMessage) => - switch handler.contents { - | Some(handle) => handle(message) - | None => pending := pending.contents->Array.concat([message]) - } - ) - -let onParentMessage = (handle: parentMessage => unit) => { - handler := Some(handle) - let held = pending.contents - pending := [] - held->Array.forEach(handle) -} - -// Resolves with the init payload. The supervisor sends it right after the -// fork, before anything else, so the first message is the only one to read. +// Resolves with the init payload, the one message a supervisor sends its worker. let awaitInit = (): promise => - Promise.make((resolve, reject) => + Promise.make((resolve, _) => NodeJs.Process.onceMessage((message: parentMessage) => switch message { | Init({config}) => resolve(config) - | SyncCache(_) => reject(Utils.Error.make("Expected the supervisor's init message first")) } ) ) From 0deb561846f8a457251f435132e850ddfeb901e8 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 17 Sep 2026 09:03:01 +0000 Subject: [PATCH 18/61] Say why a dump may take the run one connection over its budget Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- packages/envio/src/Supervisor.res | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index 09f569c77..fa32bcfd3 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -171,7 +171,10 @@ let syncCache = { } // The supervisor handed its connections to the workers, so a dump opens one of -// its own for as long as it takes. +// its own for as long as it takes. That puts the run one connection over its +// budget, deliberately: the console that asks for a dump is `envio dev` only, +// one connection is a cheaper price than pausing the indexing to free one, and +// the pool is capped at that one. let dumpCache = (~config) => { let storage = PgStorage.makeStorageFromEnv(~config, ~sql=PgStorage.makeClient(~maxConnections=1)) storage.dumpEffectCache()->Promise.finally(() => storage.close()->Promise.ignore) From 5941925d3d19c8fe3ed064b0a6dd61c2a5dab079 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 17 Sep 2026 09:25:31 +0000 Subject: [PATCH 19/61] Split by default, initialize like a run, and name the chain once MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `ENVIO_PG_MAX_CONNECTIONS` now falls back to 10 rather than 2, which affords five workers: a per-chain schema of several chains splits without being asked to, and a run that stays in one process gets a pool of ten. The supervisor was creating the schema through `Main.migrate`, whose messages name the migration command — a split `envio dev` told the operator to run `envio start -r` after a config change. Both it and `Main.start` now go through one initializer that names the run's own commands, leaving `migrate` to the `db-migrate` command it belongs to. A process that carries a chain on every line had pino write `chainId` twice on the lines that name it themselves: a child's bindings and the context are concatenated into the line, not merged. The context is now the one place a line names it. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- packages/cli/CommandLineHelp.md | 2 +- packages/cli/src/cli_args/clap_definitions.rs | 6 +- .../test/lib_tests/LoggingContext_test.res | 55 +++++++++++++++++++ .../test/lib_tests/Supervisor_test.res | 2 +- packages/envio/src/Config.res | 8 +-- packages/envio/src/Env.res | 4 +- packages/envio/src/Logging.res | 53 ++++++++++++------ packages/envio/src/Main.res | 46 +++++++++------- packages/envio/src/Supervisor.res | 17 ++++-- 9 files changed, 137 insertions(+), 56 deletions(-) create mode 100644 packages/envio-tests/test/lib_tests/LoggingContext_test.res diff --git a/packages/cli/CommandLineHelp.md b/packages/cli/CommandLineHelp.md index d9dcbdf0a..58dec8ccc 100644 --- a/packages/cli/CommandLineHelp.md +++ b/packages/cli/CommandLineHelp.md @@ -378,7 +378,7 @@ Start the indexer. Runs codegen automatically before launching so the on-disk ty ###### **Options:** * `-r`, `--restart` — Clear your database and restart indexing from scratch -* `--chain ` — Index only this chain, leaving the others to their own `envio start --chain` processes. Only needed to place the chains yourself: a plain `envio start` already splits them across processes, and manages those processes for you, whenever the schema's entities are all per-chain and `ENVIO_PG_MAX_CONNECTIONS` is at least 4 — two per process, for two processes. A schema with an entity shared across chains, or a smaller budget, runs in one process as it always has. Repeat the flag for several chains. Requires a schema whose entities are all per-chain, created for every chain by `envio local db-migrate up` before any process starts. Assign each configured chain to exactly one process, and give each its own `ENVIO_INDEXER_PORT`. Each process builds the indexes for its own chains and reports them ready as they catch up, independently of the others +* `--chain ` — Index only this chain, leaving the others to their own `envio start --chain` processes. Only needed to place the chains yourself: a plain `envio start` already splits them across processes, and manages those processes for you, whenever the schema's entities are all per-chain. `ENVIO_PG_MAX_CONNECTIONS` is the budget for the whole run and buys one process per two connections, so its default of 10 affords five. A schema with an entity shared across chains, or a budget under 4, runs in one process as it always has. Repeat the flag for several chains. Requires a schema whose entities are all per-chain, created for every chain by `envio local db-migrate up` before any process starts. Assign each configured chain to exactly one process, and give each its own `ENVIO_INDEXER_PORT`. Each process builds the indexes for its own chains and reports them ready as they catch up, independently of the others diff --git a/packages/cli/src/cli_args/clap_definitions.rs b/packages/cli/src/cli_args/clap_definitions.rs index f58968583..699e2eb41 100644 --- a/packages/cli/src/cli_args/clap_definitions.rs +++ b/packages/cli/src/cli_args/clap_definitions.rs @@ -147,9 +147,9 @@ pub struct StartArgs { ///Index only this chain, leaving the others to their own `envio start --chain` processes. ///Only needed to place the chains yourself: a plain `envio start` already splits them across ///processes, and manages those processes for you, whenever the schema's entities are all - ///per-chain and `ENVIO_PG_MAX_CONNECTIONS` is at least 4 — two per process, for two processes. - ///A schema with an entity shared across chains, or a smaller budget, runs in one process as - ///it always has. + ///per-chain. `ENVIO_PG_MAX_CONNECTIONS` is the budget for the whole run and buys one process + ///per two connections, so its default of 10 affords five. A schema with an entity shared + ///across chains, or a budget under 4, runs in one process as it always has. ///Repeat the flag for several chains. Requires a schema whose entities are all per-chain, ///created for every chain by `envio local db-migrate up` before any process starts. ///Assign each configured chain to exactly one process, and give each its own diff --git a/packages/envio-tests/test/lib_tests/LoggingContext_test.res b/packages/envio-tests/test/lib_tests/LoggingContext_test.res new file mode 100644 index 000000000..8756d7a49 --- /dev/null +++ b/packages/envio-tests/test/lib_tests/LoggingContext_test.res @@ -0,0 +1,55 @@ +open Vitest + +// Reads what pino actually wrote: a duplicated field survives JSON parsing +// (the last one wins), so the raw line is the only place it shows. +let occurrences = (line, ~field) => + line->String.split(`"${field}"`)->Array.length - 1 + +// The file is written by pino's transport worker, so it lands a moment later. +let readWhenWritten = async path => { + let deadline = Date.now() +. 3000. + let read = async () => + switch await NodeJs.Fs.Promises.readFile( + ~filepath=NodeJs.Path.resolve([path]), + ~encoding=Utf8, + ) { + | contents => contents + | exception _ => "" + } + let rec until = async () => + switch await read() { + | "" if Date.now() < deadline => + await Utils.delay(50) + await until() + | contents => contents + } + await until() +} + +describe("Logging.setContext", () => { + Async.it("Names the chain once, whoever else on the line names it", async t => { + let path = `${NodeJs.Process.cwd()}/lib/envio-logging-context-${Date.now() + ->Float.toString}.log` + Logging.setLogger( + Logging.makeLogger( + ~logStrategy=FileOnly, + ~logFilePath=path, + ~defaultFileLogLevel=#info, + ~userLogLevel=#info, + ), + ) + Logging.setContext(Dict.fromArray([("chainId", JSON.Number(137.))])) + + Logging.info("a line with no chain in hand") + Logging.createChild(~params={"chainId": 137, "source": "hypersync"})->Logging.childInfo({ + "msg": "a line from a chain-scoped logger", + }) + + let lines = + (await readWhenWritten(path)) + ->String.trim + ->String.split("\n") + + t.expect(lines->Array.map(line => occurrences(line, ~field="chainId"))).toStrictEqual([1, 1]) + }) +}) diff --git a/packages/envio-tests/test/lib_tests/Supervisor_test.res b/packages/envio-tests/test/lib_tests/Supervisor_test.res index 2471e504c..9525cf32c 100644 --- a/packages/envio-tests/test/lib_tests/Supervisor_test.res +++ b/packages/envio-tests/test/lib_tests/Supervisor_test.res @@ -180,7 +180,7 @@ describe("Config.logContext", () => { config(~schema=perChain)->Config.logContext, config(~schema=crossChain)->Config.logContext, ]).toStrictEqual([ - Some(JSON.Object(Dict.fromArray([("chainId", JSON.Number(137.))]))), + Some(Dict.fromArray([("chainId", JSON.Number(137.))])), None, None, None, diff --git a/packages/envio/src/Config.res b/packages/envio/src/Config.res index 1b1254396..82aa6baf6 100644 --- a/packages/envio/src/Config.res +++ b/packages/envio/src/Config.res @@ -634,14 +634,10 @@ let isPerChain = (config: t) => !(config.userEntities->Array.some(entity => enti // of the schema's chains while siblings drive the rest names it, on the lines // that had no chain in hand. One driving several has no single owner to name, // and its chain-scoped lines already carry theirs. -let logContext = (config: t): option => +let logContext = (config: t): option> => switch (config.isolated, config.chainMap->ChainMap.keys) { | (true, [chainId]) => - Some( - JSON.Object( - Dict.fromArray([("chainId", chainId->S.reverseConvertToJsonOrThrow(ChainId.schema))]), - ), - ) + Some(Dict.fromArray([("chainId", chainId->S.reverseConvertToJsonOrThrow(ChainId.schema))])) | _ => None } diff --git a/packages/envio/src/Env.res b/packages/envio/src/Env.res index 5df6e03b6..b3b77b184 100644 --- a/packages/envio/src/Env.res +++ b/packages/envio/src/Env.res @@ -129,7 +129,9 @@ module Db = { ) // The budget for the whole run, not for one process: a run that splits across // workers divides it among them, and each caps its own pool to its share. - let maxConnections = envSafe->EnvSafe.get("ENVIO_PG_MAX_CONNECTIONS", S.int, ~fallback=2) + // The default affords five workers, so a per-chain schema of several chains + // splits without being asked to. + let maxConnections = envSafe->EnvSafe.get("ENVIO_PG_MAX_CONNECTIONS", S.int, ~fallback=10) } // Required env vars are validated lazily in PgStorage when the user diff --git a/packages/envio/src/Logging.res b/packages/envio/src/Logging.res index abea6133d..2bcb4136c 100644 --- a/packages/envio/src/Logging.res +++ b/packages/envio/src/Logging.res @@ -25,9 +25,30 @@ let logLevels = [ ]->Dict.fromArray %%private(let logger = ref(None)) -// The logger as configured, before any context a run added to it. Kept so -// setting context twice in one process replaces it rather than stacking it. -%%private(let rootLogger = ref(None)) + +// Fields every line this process logs carries. Merged into each line rather +// than bound to a child logger: pino writes a child's bindings and the line's +// own fields side by side, so a line that names the same key would carry it +// twice. A fresh object per line, since pino merges the line's fields into +// whatever this returns. +%%private(let context: ref> = ref(Dict.make())) +%%private(let mixin = () => JSON.Object(context.contents->Dict.copy)) + +// A child logger that binds a field the process already carries would have pino +// write it twice: a child's bindings and the context are concatenated into the +// line, not merged. The context is the one place a line names it, which it can +// be because a process only takes one when its every line is about that chain. +%%private( + let withoutContext = (params: 'a) => + switch context.contents->Dict.keysToArray { + | [] => params + | keys => { + let narrowed = params->(Utils.magic: 'a => dict)->Dict.copy + keys->Array.forEach(key => narrowed->Dict.delete(key)) + narrowed->(Utils.magic: dict => 'a) + } + } +) let makeLogger = (~logStrategy, ~logFilePath, ~defaultFileLogLevel, ~userLogLevel) => { // Currently unused - useful if using multiple transports. @@ -59,17 +80,19 @@ let makeLogger = (~logStrategy, ~logFilePath, ~defaultFileLogLevel, ~userLogLeve ...Pino.ECS.make(), customLevels: logLevels, base, + mixin, }, Transport.make(pinoFile), ) | EcsConsoleMultistream => - makeMultiStreamLogger(~logFile=None, ~options=Some({...Pino.ECS.make(), base})) + makeMultiStreamLogger(~logFile=None, ~options=Some({...Pino.ECS.make(), base, mixin})) | EcsConsole => make({ ...Pino.ECS.make(), level: userLogLevel, customLevels: logLevels, base, + mixin, }) | FileOnly => makeWithOptionsAndTransport( @@ -77,17 +100,17 @@ let makeLogger = (~logStrategy, ~logFilePath, ~defaultFileLogLevel, ~userLogLeve customLevels: logLevels, level: defaultFileLogLevel, base, + mixin, }, Transport.make(pinoFile), ) - | ConsoleRaw => makeMultiStreamLogger(~logFile=None, ~options=Some({base: base})) - | ConsolePretty => makeMultiStreamLogger(~logFile=None, ~options=Some({base: base})) - | Both => makeMultiStreamLogger(~logFile=Some(logFilePath), ~options=Some({base: base})) + | ConsoleRaw => makeMultiStreamLogger(~logFile=None, ~options=Some({base, mixin})) + | ConsolePretty => makeMultiStreamLogger(~logFile=None, ~options=Some({base, mixin})) + | Both => makeMultiStreamLogger(~logFile=Some(logFilePath), ~options=Some({base, mixin})) } } let setLogger = l => { - rootLogger := Some(l) logger := Some(l) } @@ -153,19 +176,15 @@ let childFatal = (logger, params: 'a) => { } let createChild = (~params: 'a) => { - getLogger()->child(params->createChildParams) + getLogger()->child(params->withoutContext->createChildParams) } -// Fields every line this process logs from here on carries. What belongs on -// them is the run's to decide; the logger only carries what it is handed. -let setContext = (params: 'a) => - switch rootLogger.contents { - | Some(root) => logger := Some(root->child(params->createChildParams)) - | None => () - } +// What belongs on every line is the run's to decide; the logger only carries +// what it is handed. A line that names one of these fields itself wins. +let setContext = (fields: dict) => context := fields let createChildFrom = (~logger: t, ~params: 'a) => { - logger->child(params->createChildParams) + logger->child(params->withoutContext->createChildParams) } @inline diff --git a/packages/envio/src/Main.res b/packages/envio/src/Main.res index 5424850c9..79c3ba215 100644 --- a/packages/envio/src/Main.res +++ b/packages/envio/src/Main.res @@ -600,17 +600,23 @@ type mainArgs = Yargs.parsedArgs // `envio_info` (on initialize) and validates against (on resume). let getEnvioInfo = () => Config.getPublicConfigJson()->Config.stripSensitiveData -let migrate = async ( - ~reset, - // A supervisor creating the schema for a run it is about to start names that - // run's commands, not the migration's, in what a config change prints. - ~resetCommand="envio local db-migrate setup", - ~runCommand=None, - // A migration command runs once and exits, with nobody watching it recover: - // an unreachable chain should say so now rather than hold the command open. - // A run that is about to start wants the opposite. - ~startBlockRetry=StartBlockResolver.Once, -) => { +// Brings the schema up to date for a run that is about to start, as opposed to +// a migration command: what a config change prints names the command the +// operator ran, and an unreachable chain is waited on rather than reported, +// since somebody is watching the run come up. +let initForRun = (persistence, ~config: Config.t, ~reset, ~isDevelopmentMode, ~requireInitialized) => + persistence->Persistence.init( + ~reset, + ~chainConfigs=config.chainMap->ChainMap.values, + ~contractMapping=config.contractMapping, + ~envioInfo=getEnvioInfo(), + ~resetCommand=isDevelopmentMode ? "envio dev -r" : "envio start -r", + ~runCommand=Some(isDevelopmentMode ? "envio dev" : "envio start"), + ~lowercaseAddresses=config.lowercaseAddresses, + ~requireInitialized, + ) + +let migrate = async (~reset) => { let config = Config.load() let persistence = PgStorage.makePersistenceFromConfig(~config) await persistence->Persistence.init( @@ -618,10 +624,12 @@ let migrate = async ( ~chainConfigs=config.chainMap->ChainMap.values, ~contractMapping=config.contractMapping, ~envioInfo=getEnvioInfo(), - ~resetCommand, - ~runCommand, + ~resetCommand="envio local db-migrate setup", + ~runCommand=None, ~lowercaseAddresses=config.lowercaseAddresses, - ~startBlockRetry, + // A migration command runs once and exits, with nobody watching it recover: + // an unreachable chain should say so now rather than hold the command open. + ~startBlockRetry=StartBlockResolver.Once, ) await persistence.storage.close() } @@ -678,14 +686,10 @@ let start = async ( | None => PgStorage.makePersistenceFromConfig(~config) } setGlobalPersistence(persistence) - await persistence->Persistence.init( + await persistence->initForRun( + ~config, ~reset, - ~chainConfigs=config.chainMap->ChainMap.values, - ~contractMapping=config.contractMapping, - ~envioInfo=getEnvioInfo(), - ~resetCommand=isDevelopmentMode ? "envio dev -r" : "envio start -r", - ~runCommand=Some(isDevelopmentMode ? "envio dev" : "envio start"), - ~lowercaseAddresses=config.lowercaseAddresses, + ~isDevelopmentMode, ~requireInitialized=config.isolated, ) diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index fa32bcfd3..5526a6700 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -234,17 +234,22 @@ let awaitExit = async (group): outcome => { // entry, and serves the run's metrics, console and display from what they // report. Returns once every worker has exited; throws if any of them failed. let run = async (~workers: array, ~configJson: JSON.t, ~reset) => { + let config = Config.load() + // Every chain's state has to exist before a worker resumes it: an isolated // run refuses to initialize, precisely so it can't create rows for its own - // chains and leave the chains it skipped with nothing to resume. - await Main.migrate( + // chains and leave the chains it skipped with nothing to resume. It is the + // same initialization an unsplit run does, and the supervisor hands the + // connections it used to its workers. + let persistence = PgStorage.makePersistenceFromConfig(~config) + await persistence->Main.initForRun( + ~config, ~reset, - ~resetCommand="envio start -r", - ~runCommand=Some("envio start"), - ~startBlockRetry=StartBlockResolver.UntilItAnswers, + ~isDevelopmentMode=config.isDev, + ~requireInitialized=false, ) + await persistence.storage.close() - let config = Config.load() let startTime = Date.make() let startTimeRef = Performance.now() From 643c5f208fb6d15359be26a4e8b5ff66bd8658e0 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 17 Sep 2026 09:37:29 +0000 Subject: [PATCH 20/61] Decide in Main whether a run supervises or indexes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `Bin` chose between supervising a group and being the indexer, because `Supervisor` reached into `Main` for the server, the display question and the schema initialization, and a module can't be reached into by what it reaches into. Each of those has a home below both now — `Server` for the HTTP surface and the console payloads it serves, `Tui.shouldUse` for the display question, `Persistence.initForRun` for the initialization a run does — so `Main.start` makes the decision and `Bin` just starts. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- packages/envio/src/Bin.res | 5 +- packages/envio/src/Config.res | 4 + packages/envio/src/Main.res | 442 ++++++++--------------------- packages/envio/src/Persistence.res | 22 ++ packages/envio/src/Server.res | 181 ++++++++++++ packages/envio/src/Supervisor.res | 10 +- packages/envio/src/tui/Tui.res | 24 ++ 7 files changed, 364 insertions(+), 324 deletions(-) create mode 100644 packages/envio/src/Server.res diff --git a/packages/envio/src/Bin.res b/packages/envio/src/Bin.res index f24088388..27d76279e 100644 --- a/packages/envio/src/Bin.res +++ b/packages/envio/src/Bin.res @@ -69,10 +69,7 @@ let run = async args => { Config.prime(config) processChdir(cwd) applyEnv(env) - switch Supervisor.planForRun(~config=Config.load()) { - | Some(workers) => await Supervisor.run(~workers, ~configJson=config, ~reset) - | None => await Main.start(~reset) - } + await Main.start(~reset) | Migrate({reset, config}) => Config.prime(config) await Main.migrate(~reset) diff --git a/packages/envio/src/Config.res b/packages/envio/src/Config.res index 82aa6baf6..185963836 100644 --- a/packages/envio/src/Config.res +++ b/packages/envio/src/Config.res @@ -1282,6 +1282,10 @@ let stripSensitiveData = (json: JSON.t): JSON.t => { cloned } +// What the storage layer records as the config this schema was built from, +// and checks a resuming run against. +let envioInfo = () => getPublicConfigJson()->stripSensitiveData + // Postgres jsonb doesn't preserve key order, so canonicalize with sorted // keys before string-comparing. let rec canonicalJson = (json: JSON.t): JSON.t => diff --git a/packages/envio/src/Main.res b/packages/envio/src/Main.res index 79c3ba215..1873c1144 100644 --- a/packages/envio/src/Main.res +++ b/packages/envio/src/Main.res @@ -1,80 +1,3 @@ -// The public console/state chain shape. Kept to exactly this field set for -// backward compatibility with consumers like RACE — new metric fields stay off -// the HTTP response. -type chainData = { - chainId: ChainId.t, - poweredByHyperSync: bool, - firstEventBlockNumber: option, - latestProcessedBlock: option, - timestampCaughtUpToHeadOrEndblock: option, - numEventsProcessed: float, - latestFetchedBlockNumber: int, - // Need this for API backwards compatibility - @as("currentBlockHeight") - knownHeight: int, - numBatchesFetched: int, - startBlock: int, - endBlock: option, - numAddresses: int, -} -@tag("status") -type state = - | @as("disabled") Disabled({}) - | @as("initializing") Initializing({}) - | @as("active") - Active({ - envioVersion: string, - chains: array, - indexerStartTime: Date.t, - isPreRegisteringDynamicContracts: bool, - rollbackOnReorg: bool, - }) - -let toChainData = (m: Metrics.chainMetrics): chainData => { - chainId: m.chainId, - poweredByHyperSync: m.poweredByHyperSync, - firstEventBlockNumber: m.firstEventBlockNumber, - latestProcessedBlock: m.latestProcessedBlock, - timestampCaughtUpToHeadOrEndblock: m.timestampCaughtUpToHeadOrEndblock, - numEventsProcessed: m.numEventsProcessed, - latestFetchedBlockNumber: m.latestFetchedBlockNumber, - knownHeight: m.knownHeight, - numBatchesFetched: m.numBatchesFetched, - startBlock: m.startBlock, - endBlock: m.endBlock, - numAddresses: m.numAddresses, -} - -let chainDataSchema = S.schema((s): chainData => { - chainId: s.matches(ChainId.schema), - poweredByHyperSync: s.matches(S.bool), - firstEventBlockNumber: s.matches(S.option(S.int)), - latestProcessedBlock: s.matches(S.option(S.int)), - timestampCaughtUpToHeadOrEndblock: s.matches(S.option(S.datetime(S.string))), - numEventsProcessed: s.matches(S.float), - latestFetchedBlockNumber: s.matches(S.int), - knownHeight: s.matches(S.int), - numBatchesFetched: s.matches(S.int), - startBlock: s.matches(S.int), - endBlock: s.matches(S.option(S.int)), - numAddresses: s.matches(S.int), -}) -let stateSchema = S.union([ - S.literal(Disabled({})), - S.literal(Initializing({})), - S.schema(s => Active({ - envioVersion: s.matches(S.string), - chains: s.matches(S.array(chainDataSchema)), - indexerStartTime: s.matches(S.datetime(S.string)), - // Keep the field, since Dev Console expects it to be present - isPreRegisteringDynamicContracts: false, - rollbackOnReorg: s.matches(S.bool), - })), -]) - -// Runtime state lives in the process-wide `EnvioGlobal` record (shared -// across duplicate envio module instances); the slots are opaque there, so -// cast them to the real types here. let getIndexerState = () => EnvioGlobal.value.indexerState->(Utils.magic: option => option) let setIndexerState = (state: IndexerState.t) => @@ -488,134 +411,8 @@ let getGlobalIndexer = (): 'indexer => { Utils.Proxy.make(Utils.Object.createNullObject(), traps)->(Utils.magic: {..} => 'indexer) } -let startServer = ( - ~getMetrics: unit => option, - ~envioVersion: string, - ~onSyncCache: unit => promise, - ~collectRuntime: unit => string, - ~isDevelopmentMode: bool, -) => { - open Express - - let app = make() - - let consoleCorsMiddleware = (req, res, next) => { - switch req.headers->Dict.get("origin") { - | Some(origin) if origin === Env.prodEnvioAppUrl || origin === Env.envioAppUrl => - res->setHeader("Access-Control-Allow-Origin", origin) - | _ => () - } - - res->setHeader("Access-Control-Allow-Methods", "GET, POST, PUT, DELETE, OPTIONS") - res->setHeader("Access-Control-Allow-Headers", "Origin, X-Requested-With, Content-Type, Accept") - - if req.method === Rest.Options { - res->sendStatus(200) - } else { - next() - } - } - app->useFor("/console", consoleCorsMiddleware) - app->useFor("/metrics", consoleCorsMiddleware) - app->useFor("/metrics/runtime", consoleCorsMiddleware) - - app->get("/healthz", (_req, res) => { - // this is the machine readable port used in kubernetes to check the health of this service. - // aditional health information could be added in the future (info about errors, back-offs, etc). - res->sendStatus(200) - }) - - app->get("/console/state", (_req, res) => { - let state = if !isDevelopmentMode { - Disabled({}) - } else { - switch getMetrics() { - | None => Initializing({}) - | Some(metrics) => - Active({ - envioVersion, - chains: metrics.chains->Array.map(toChainData), - indexerStartTime: metrics.startTime, - isPreRegisteringDynamicContracts: false, - rollbackOnReorg: metrics.rollbackEnabled, - }) - } - } - - res->json(state->S.reverseConvertToJsonOrThrow(stateSchema)) - }) - - app->post("/console/syncCache", (_req, res) => { - if isDevelopmentMode { - onSyncCache() - ->Promise.thenResolve(() => res->json(Boolean(true))) - // A dump that couldn't be made, or couldn't be confirmed, answers the - // same `false` a disabled console does. Leaving it unanswered would hold - // the request open for as long as the indexer runs. - ->Promise.catch(exn => { - Logging.errorWithExn(exn, "Failed to sync the effect cache") - res->json(Boolean(false)) - Promise.resolve() - }) - ->Promise.ignore - } else { - res->json(Boolean(false)) - } - }) - - app->get("/metrics", (_req, res) => { - res->set("Content-Type", Metrics.contentType) - let _ = res->endWithData(Metrics.collect(~metrics=getMetrics())) - }) - - app->get("/metrics/runtime", (_req, res) => { - res->set("Content-Type", Metrics.contentType) - let _ = res->endWithData(collectRuntime()) - }) - - let server = app->listen(Env.serverPort) - server->Express.onError(err => { - let code = (err->(Utils.magic: JsExn.t => {..}))["code"] - if code === "EADDRINUSE" { - Logging.error( - `Port ${Env.serverPort->Int.toString} is already in use. To fix this either:` ++ - `\n 1. Kill the process using the port: lsof -ti :${Env.serverPort->Int.toString} | xargs kill -9` ++ `\n 2. Use a different port by setting the ENVIO_INDEXER_PORT environment variable: ENVIO_INDEXER_PORT=9899 envio start`, - ) - } else { - Logging.errorWithExn(err, "Failed to start indexer server") - } - NodeJs.process->NodeJs.exitWithCode(Failure) - }) -} - -type args = {@as("tui-off") tuiOff?: bool} - -type process -@val external process: process = "process" -@get external argv: process => 'a = "argv" - -type mainArgs = Yargs.parsedArgs - // The RPC-stripped public config that the storage layer persists in // `envio_info` (on initialize) and validates against (on resume). -let getEnvioInfo = () => Config.getPublicConfigJson()->Config.stripSensitiveData - -// Brings the schema up to date for a run that is about to start, as opposed to -// a migration command: what a config change prints names the command the -// operator ran, and an unreachable chain is waited on rather than reported, -// since somebody is watching the run come up. -let initForRun = (persistence, ~config: Config.t, ~reset, ~isDevelopmentMode, ~requireInitialized) => - persistence->Persistence.init( - ~reset, - ~chainConfigs=config.chainMap->ChainMap.values, - ~contractMapping=config.contractMapping, - ~envioInfo=getEnvioInfo(), - ~resetCommand=isDevelopmentMode ? "envio dev -r" : "envio start -r", - ~runCommand=Some(isDevelopmentMode ? "envio dev" : "envio start"), - ~lowercaseAddresses=config.lowercaseAddresses, - ~requireInitialized, - ) - let migrate = async (~reset) => { let config = Config.load() let persistence = PgStorage.makePersistenceFromConfig(~config) @@ -623,7 +420,7 @@ let migrate = async (~reset) => { ~reset, ~chainConfigs=config.chainMap->ChainMap.values, ~contractMapping=config.contractMapping, - ~envioInfo=getEnvioInfo(), + ~envioInfo=Config.envioInfo(), ~resetCommand="envio local db-migrate setup", ~runCommand=None, ~lowercaseAddresses=config.lowercaseAddresses, @@ -645,22 +442,123 @@ let dropSchema = async () => { // context, so callers should act on it (exit / re-throw) without logging again. exception FatalError(exn) -// Whether this process draws the progress display: `--tui-off` first, then -// `ENVIO_TUI`, then whether anything is watching. A supervisor asks the same -// question its workers would have, since it is the one drawing for the run. -let shouldUseTui = (~suppressed=false) => { - let mainArgs: mainArgs = process->argv->Yargs.hideBin->Yargs.yargs->Yargs.argv - let explicitTui = switch mainArgs.tuiOff { - | Some(off) => Some(!off) - | None => Env.tuiEnvVar - } - switch (suppressed, explicitTui) { - | (true, _) => false - | (_, Some(tui)) => tui - | (_, None) => !Envio.isNonInteractive() +%%private( + let startIndexer = async ( + ~config: Config.t, + ~persistence: option=?, + ~reset=false, + ~isTest=false, + ~exitAfterFirstEventBlock=false, + ~patchConfig: option<(Config.t, HandlerRegister.registrationsByChainId) => Config.t>=?, + ) => { + // A worker reports to its supervisor, which draws for the whole run. + let shouldUseTui = Tui.shouldUse(~suppressed=isTest || Worker.isEnabled) + // In per-chain mode every line this process writes belongs to the chains it + // drives, whether or not a supervisor split the run across processes. + config->Config.logContext->Option.forEach(Logging.setContext) + // isDevelopmentMode controls whether the indexer stays alive after all + // chains finish (keepProcessAlive) and whether the console API is exposed. + // Set by `envio dev` via the public config's `isDev` field; `envio start` + // leaves it false so the process exits cleanly when indexing completes. + let isDevelopmentMode = !isTest && config.isDev + // Initialized first so the exported indexer value contains state from the + // database when handler files are loaded (they may access the indexer at + // module top level). + let persistence = switch persistence { + | Some(p) => p + | None => PgStorage.makePersistenceFromConfig(~config) + } + setGlobalPersistence(persistence) + await persistence->Persistence.initForRun( + ~config, + ~reset, + ~isDevelopmentMode, + ~requireInitialized=config.isolated, + ) + + // Loads user handler files, which register handler/contractRegister/where + // state into the global `HandlerRegister` registry as a side effect; this + // returns that state resolved into per-chain registrations. `config` itself + // is never mutated by registration — it holds only event definitions. + let registrationsByChainId = await HandlerLoader.registerAllHandlers(~config) + let config = if isTest { + {...config, shouldRollbackOnReorg: false} + } else { + config + } + + let config = switch patchConfig { + | Some(patchConfig) => patchConfig(config, registrationsByChainId) + | None => config + } + // The single fatal-error handler, invoked once via IndexerState.errorExit. + // It logs the failure once (with chain context) and rejects the run wrapped in + // `FatalError` so callers know it's already logged — `Bin.res` just exits, the + // test worker unwraps and re-throws it to the parent thread. `runUntilFatalError` + // only ever rejects: on a clean run it stays pending and the process exits via + // ExitOnCaughtUp / when the indexer loop drains. + let onErrorReject = ref(None) + let runUntilFatalError: promise = Promise.make((_resolve, reject) => + onErrorReject := Some(reject) + ) + // `onErrorReject` is filled synchronously by `Promise.make` above, before the + // indexer can run and call `onError`, so it's always present here. + let onError = (errHandler: ErrorHandling.t) => { + errHandler->ErrorHandling.log + (onErrorReject.contents->Option.getUnsafe)(FatalError(errHandler.exn->Utils.prettifyExn)) + } + let envioVersion = Utils.EnvioPackage.value.version + + let getMetrics = () => getIndexerState()->Option.map(IndexerState.toMetrics) + let dumpEffectCache = () => + (persistence->Persistence.getInitializedStorageOrThrow).dumpEffectCache() + + // A worker reports through its supervisor, which owns the one server and the + // one display the run has. + if !isTest && !Worker.isEnabled { + Metrics.startRuntimeCollectors() + Server.startServer( + ~onSyncCache=() => dumpEffectCache()->Promise.thenResolve(ignore), + ~collectRuntime=Metrics.collectRuntime, + ~isDevelopmentMode, + ~envioVersion, + ~getMetrics, + ) + } + + let state = IndexerState.makeFromDbState( + ~config, + ~persistence, + ~initialState=persistence->Persistence.getInitializedState, + ~registrationsByChainId, + ~isDevelopmentMode, + ~shouldUseTui, + ~exitAfterFirstEventBlock, + ~onError, + ) + if shouldUseTui { + let _rerender = Tui.start(~config, ~getMetrics=() => state->IndexerState.toMetrics) + } + if Worker.isEnabled { + Metrics.startRuntimeCollectors() + let _intervalId = setInterval( + () => + Worker.send( + Snapshot({metrics: state->IndexerState.toMetrics, runtime: Metrics.sampleRuntime()}), + ), + Worker.snapshotIntervalMillis, + ) + } + setIndexerState(state) + state->IndexerLoop.start + await runUntilFatalError } -} +) +// Starts this process's part of a run: the group's supervisor when the budget +// and the schema afford splitting the chains across processes, and the indexer +// itself otherwise. A worker is already one process's part, so it never splits +// again — `planForRun` refuses an isolated config. let start = async ( ~persistence: option=?, ~reset=false, @@ -668,105 +566,17 @@ let start = async ( ~exitAfterFirstEventBlock=false, ~patchConfig: option<(Config.t, HandlerRegister.registrationsByChainId) => Config.t>=?, ) => { - // A worker reports to its supervisor, which draws for the whole run. - let shouldUseTui = shouldUseTui(~suppressed=isTest || Worker.isEnabled) - // Initialize persistence first so the exported indexer value contains state from the database - // when handler files are loaded (they may access the indexer at module top level). let config = Config.load() - // In per-chain mode every line this process writes belongs to the chains it - // drives, whether or not a supervisor split the run across processes. - config->Config.logContext->Option.forEach(Logging.setContext) - // isDevelopmentMode controls whether the indexer stays alive after all - // chains finish (keepProcessAlive) and whether the console API is exposed. - // Set by `envio dev` via the public config's `isDev` field; `envio start` - // leaves it false so the process exits cleanly when indexing completes. - let isDevelopmentMode = !isTest && config.isDev - let persistence = switch persistence { - | Some(p) => p - | None => PgStorage.makePersistenceFromConfig(~config) - } - setGlobalPersistence(persistence) - await persistence->initForRun( - ~config, - ~reset, - ~isDevelopmentMode, - ~requireInitialized=config.isolated, - ) - - // Loads user handler files, which register handler/contractRegister/where - // state into the global `HandlerRegister` registry as a side effect; this - // returns that state resolved into per-chain registrations. `config` itself - // is never mutated by registration — it holds only event definitions. - let registrationsByChainId = await HandlerLoader.registerAllHandlers(~config) - let config = if isTest { - {...config, shouldRollbackOnReorg: false} - } else { - config - } - - let config = switch patchConfig { - | Some(patchConfig) => patchConfig(config, registrationsByChainId) - | None => config - } - // The single fatal-error handler, invoked once via IndexerState.errorExit. - // It logs the failure once (with chain context) and rejects the run wrapped in - // `FatalError` so callers know it's already logged — `Bin.res` just exits, the - // test worker unwraps and re-throws it to the parent thread. `runUntilFatalError` - // only ever rejects: on a clean run it stays pending and the process exits via - // ExitOnCaughtUp / when the indexer loop drains. - let onErrorReject = ref(None) - let runUntilFatalError: promise = Promise.make((_resolve, reject) => - onErrorReject := Some(reject) - ) - // `onErrorReject` is filled synchronously by `Promise.make` above, before the - // indexer can run and call `onError`, so it's always present here. - let onError = (errHandler: ErrorHandling.t) => { - errHandler->ErrorHandling.log - (onErrorReject.contents->Option.getUnsafe)(FatalError(errHandler.exn->Utils.prettifyExn)) - } - let envioVersion = Utils.EnvioPackage.value.version - - let getMetrics = () => getIndexerState()->Option.map(IndexerState.toMetrics) - let dumpEffectCache = () => - (persistence->Persistence.getInitializedStorageOrThrow).dumpEffectCache() - - // A worker reports through its supervisor, which owns the one server and the - // one display the run has. - if !isTest && !Worker.isEnabled { - Metrics.startRuntimeCollectors() - startServer( - ~onSyncCache=() => dumpEffectCache()->Promise.thenResolve(ignore), - ~collectRuntime=Metrics.collectRuntime, - ~isDevelopmentMode, - ~envioVersion, - ~getMetrics, - ) - } - - let state = IndexerState.makeFromDbState( - ~config, - ~persistence, - ~initialState=persistence->Persistence.getInitializedState, - ~registrationsByChainId, - ~isDevelopmentMode, - ~shouldUseTui, - ~exitAfterFirstEventBlock, - ~onError, - ) - if shouldUseTui { - let _rerender = Tui.start(~config, ~getMetrics=() => state->IndexerState.toMetrics) - } - if Worker.isEnabled { - Metrics.startRuntimeCollectors() - let _intervalId = setInterval( - () => - Worker.send( - Snapshot({metrics: state->IndexerState.toMetrics, runtime: Metrics.sampleRuntime()}), - ), - Worker.snapshotIntervalMillis, + switch isTest ? None : Supervisor.planForRun(~config) { + | Some(workers) => await Supervisor.run(~workers, ~reset) + | None => + await startIndexer( + ~config, + ~persistence?, + ~reset, + ~isTest, + ~exitAfterFirstEventBlock, + ~patchConfig?, ) } - setIndexerState(state) - state->IndexerLoop.start - await runUntilFatalError } diff --git a/packages/envio/src/Persistence.res b/packages/envio/src/Persistence.res index 27ad2742a..a6539be4e 100644 --- a/packages/envio/src/Persistence.res +++ b/packages/envio/src/Persistence.res @@ -356,6 +356,28 @@ let init = { } } +// Brings the schema up to date for a run that is about to start, as opposed to +// a migration command: what a config change prints names the command the +// operator ran, and an unreachable chain is waited on rather than reported, +// since somebody is watching the run come up. +let initForRun = ( + persistence, + ~config: Config.t, + ~reset, + ~isDevelopmentMode, + ~requireInitialized, +) => + persistence->init( + ~reset, + ~chainConfigs=config.chainMap->ChainMap.values, + ~contractMapping=config.contractMapping, + ~envioInfo=Config.envioInfo(), + ~resetCommand=isDevelopmentMode ? "envio dev -r" : "envio start -r", + ~runCommand=Some(isDevelopmentMode ? "envio dev" : "envio start"), + ~lowercaseAddresses=config.lowercaseAddresses, + ~requireInitialized, + ) + let getInitializedStorageOrThrow = persistence => { switch persistence.storageStatus { | Unknown diff --git a/packages/envio/src/Server.res b/packages/envio/src/Server.res new file mode 100644 index 000000000..39641c9c6 --- /dev/null +++ b/packages/envio/src/Server.res @@ -0,0 +1,181 @@ +// The indexer's own HTTP surface: metrics for a scraper, health for an +// orchestrator, and the console's view of the run. What it serves is handed to +// it, so one process's readings and a supervised group's merged ones render the +// same way. + +// The public console/state chain shape. Kept to exactly this field set for +// backward compatibility with consumers like RACE — new metric fields stay off +// the HTTP response. +type chainData = { + chainId: ChainId.t, + poweredByHyperSync: bool, + firstEventBlockNumber: option, + latestProcessedBlock: option, + timestampCaughtUpToHeadOrEndblock: option, + numEventsProcessed: float, + latestFetchedBlockNumber: int, + // Need this for API backwards compatibility + @as("currentBlockHeight") + knownHeight: int, + numBatchesFetched: int, + startBlock: int, + endBlock: option, + numAddresses: int, +} +@tag("status") +type state = + | @as("disabled") Disabled({}) + | @as("initializing") Initializing({}) + | @as("active") + Active({ + envioVersion: string, + chains: array, + indexerStartTime: Date.t, + isPreRegisteringDynamicContracts: bool, + rollbackOnReorg: bool, + }) + +let toChainData = (m: Metrics.chainMetrics): chainData => { + chainId: m.chainId, + poweredByHyperSync: m.poweredByHyperSync, + firstEventBlockNumber: m.firstEventBlockNumber, + latestProcessedBlock: m.latestProcessedBlock, + timestampCaughtUpToHeadOrEndblock: m.timestampCaughtUpToHeadOrEndblock, + numEventsProcessed: m.numEventsProcessed, + latestFetchedBlockNumber: m.latestFetchedBlockNumber, + knownHeight: m.knownHeight, + numBatchesFetched: m.numBatchesFetched, + startBlock: m.startBlock, + endBlock: m.endBlock, + numAddresses: m.numAddresses, +} + +let chainDataSchema = S.schema((s): chainData => { + chainId: s.matches(ChainId.schema), + poweredByHyperSync: s.matches(S.bool), + firstEventBlockNumber: s.matches(S.option(S.int)), + latestProcessedBlock: s.matches(S.option(S.int)), + timestampCaughtUpToHeadOrEndblock: s.matches(S.option(S.datetime(S.string))), + numEventsProcessed: s.matches(S.float), + latestFetchedBlockNumber: s.matches(S.int), + knownHeight: s.matches(S.int), + numBatchesFetched: s.matches(S.int), + startBlock: s.matches(S.int), + endBlock: s.matches(S.option(S.int)), + numAddresses: s.matches(S.int), +}) +let stateSchema = S.union([ + S.literal(Disabled({})), + S.literal(Initializing({})), + S.schema(s => Active({ + envioVersion: s.matches(S.string), + chains: s.matches(S.array(chainDataSchema)), + indexerStartTime: s.matches(S.datetime(S.string)), + // Keep the field, since Dev Console expects it to be present + isPreRegisteringDynamicContracts: false, + rollbackOnReorg: s.matches(S.bool), + })), +]) + +// Runtime state lives in the process-wide `EnvioGlobal` record (shared +// across duplicate envio module instances); the slots are opaque there, so +// cast them to the real types here. +let startServer = ( + ~getMetrics: unit => option, + ~envioVersion: string, + ~onSyncCache: unit => promise, + ~collectRuntime: unit => string, + ~isDevelopmentMode: bool, +) => { + open Express + + let app = make() + + let consoleCorsMiddleware = (req, res, next) => { + switch req.headers->Dict.get("origin") { + | Some(origin) if origin === Env.prodEnvioAppUrl || origin === Env.envioAppUrl => + res->setHeader("Access-Control-Allow-Origin", origin) + | _ => () + } + + res->setHeader("Access-Control-Allow-Methods", "GET, POST, PUT, DELETE, OPTIONS") + res->setHeader("Access-Control-Allow-Headers", "Origin, X-Requested-With, Content-Type, Accept") + + if req.method === Rest.Options { + res->sendStatus(200) + } else { + next() + } + } + app->useFor("/console", consoleCorsMiddleware) + app->useFor("/metrics", consoleCorsMiddleware) + app->useFor("/metrics/runtime", consoleCorsMiddleware) + + app->get("/healthz", (_req, res) => { + // this is the machine readable port used in kubernetes to check the health of this service. + // aditional health information could be added in the future (info about errors, back-offs, etc). + res->sendStatus(200) + }) + + app->get("/console/state", (_req, res) => { + let state = if !isDevelopmentMode { + Disabled({}) + } else { + switch getMetrics() { + | None => Initializing({}) + | Some(metrics) => + Active({ + envioVersion, + chains: metrics.chains->Array.map(toChainData), + indexerStartTime: metrics.startTime, + isPreRegisteringDynamicContracts: false, + rollbackOnReorg: metrics.rollbackEnabled, + }) + } + } + + res->json(state->S.reverseConvertToJsonOrThrow(stateSchema)) + }) + + app->post("/console/syncCache", (_req, res) => { + if isDevelopmentMode { + onSyncCache() + ->Promise.thenResolve(() => res->json(Boolean(true))) + // A dump that couldn't be made, or couldn't be confirmed, answers the + // same `false` a disabled console does. Leaving it unanswered would hold + // the request open for as long as the indexer runs. + ->Promise.catch(exn => { + Logging.errorWithExn(exn, "Failed to sync the effect cache") + res->json(Boolean(false)) + Promise.resolve() + }) + ->Promise.ignore + } else { + res->json(Boolean(false)) + } + }) + + app->get("/metrics", (_req, res) => { + res->set("Content-Type", Metrics.contentType) + let _ = res->endWithData(Metrics.collect(~metrics=getMetrics())) + }) + + app->get("/metrics/runtime", (_req, res) => { + res->set("Content-Type", Metrics.contentType) + let _ = res->endWithData(collectRuntime()) + }) + + let server = app->listen(Env.serverPort) + server->Express.onError(err => { + let code = (err->(Utils.magic: JsExn.t => {..}))["code"] + if code === "EADDRINUSE" { + Logging.error( + `Port ${Env.serverPort->Int.toString} is already in use. To fix this either:` ++ + `\n 1. Kill the process using the port: lsof -ti :${Env.serverPort->Int.toString} | xargs kill -9` ++ `\n 2. Use a different port by setting the ENVIO_INDEXER_PORT environment variable: ENVIO_INDEXER_PORT=9899 envio start`, + ) + } else { + Logging.errorWithExn(err, "Failed to start indexer server") + } + NodeJs.process->NodeJs.exitWithCode(Failure) + }) +} diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index 5526a6700..cf913d0a9 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -233,7 +233,7 @@ let awaitExit = async (group): outcome => { // Runs the group: creates the schema for every chain, forks a worker per plan // entry, and serves the run's metrics, console and display from what they // report. Returns once every worker has exited; throws if any of them failed. -let run = async (~workers: array, ~configJson: JSON.t, ~reset) => { +let run = async (~workers: array, ~reset) => { let config = Config.load() // Every chain's state has to exist before a worker resumes it: an isolated @@ -242,7 +242,7 @@ let run = async (~workers: array, ~configJson: JSON.t, ~reset) => { // same initialization an unsplit run does, and the supervisor hands the // connections it used to its workers. let persistence = PgStorage.makePersistenceFromConfig(~config) - await persistence->Main.initForRun( + await persistence->Persistence.initForRun( ~config, ~reset, ~isDevelopmentMode=config.isDev, @@ -262,6 +262,8 @@ let run = async (~workers: array, ~configJson: JSON.t, ~reset) => { ->Int.toString} processes, from a budget of ${Env.Db.maxConnections->Int.toString} database connections.`, ) + // The config as the CLI handed it over, narrowed per worker on the way out. + let configJson = Config.getPublicConfigJson() let group = { running: workers->Array.mapWithIndex((worker, workerIndex) => worker->fork(~workerIndex, ~configJson) @@ -278,7 +280,7 @@ let run = async (~workers: array, ~configJson: JSON.t, ~reset) => { ~elapsedSeconds=startTimeRef->Performance.secondsSince, ) - Main.startServer( + Server.startServer( // Nothing to report until a worker has: the run reads as initializing // rather than as an indexer with no chains. ~getMetrics=() => @@ -299,7 +301,7 @@ let run = async (~workers: array, ~configJson: JSON.t, ~reset) => { ~onSyncCache=() => syncCache(~dump=() => dumpCache(~config)), ) - let shouldUseTui = Main.shouldUseTui() + let shouldUseTui = Tui.shouldUse() if shouldUseTui { let _rerender = Tui.start(~config, ~getMetrics=() => reported()->merge) } diff --git a/packages/envio/src/tui/Tui.res b/packages/envio/src/tui/Tui.res index a01bc422e..245462056 100644 --- a/packages/envio/src/tui/Tui.res +++ b/packages/envio/src/tui/Tui.res @@ -248,6 +248,30 @@ module App = { } } +type args = {@as("tui-off") tuiOff?: bool} + +type process +@val external process: process = "process" +@get external argv: process => 'a = "argv" + +type mainArgs = Yargs.parsedArgs + +// Whether this process draws the progress display: `--tui-off` first, then +// `ENVIO_TUI`, then whether anything is watching. A supervisor asks the same +// question its workers would have, since it is the one drawing for the run. +let shouldUse = (~suppressed=false) => { + let mainArgs: mainArgs = process->argv->Yargs.hideBin->Yargs.yargs->Yargs.argv + let explicitTui = switch mainArgs.tuiOff { + | Some(off) => Some(!off) + | None => Env.tuiEnvVar + } + switch (suppressed, explicitTui) { + | (true, _) => false + | (_, Some(tui)) => tui + | (_, None) => !Envio.isNonInteractive() + } +} + let start = (~config, ~getMetrics) => { let {rerender} = render() () => { From a9d84091a076c31c9bcd1d45ee8c70b4323020a8 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 17 Sep 2026 09:50:39 +0000 Subject: [PATCH 21/61] Let a worker report itself, and supervise the config that was planned The snapshot interval, the runtime sampling and the message they build sat in the middle of the indexer's startup, where a reader of `Main` met the fork's protocol. A worker reports itself now, and `send` and the interval are its own. `Supervisor.run` re-loaded the config that `planForRun` had just been given, so a plan and the run it supervised came from two reads; it takes the one that was planned. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- packages/envio/src/Main.res | 13 ++----------- packages/envio/src/Supervisor.res | 4 +--- packages/envio/src/Worker.res | 15 ++++++++++++--- 3 files changed, 15 insertions(+), 17 deletions(-) diff --git a/packages/envio/src/Main.res b/packages/envio/src/Main.res index 1873c1144..5f775311e 100644 --- a/packages/envio/src/Main.res +++ b/packages/envio/src/Main.res @@ -539,16 +539,7 @@ exception FatalError(exn) if shouldUseTui { let _rerender = Tui.start(~config, ~getMetrics=() => state->IndexerState.toMetrics) } - if Worker.isEnabled { - Metrics.startRuntimeCollectors() - let _intervalId = setInterval( - () => - Worker.send( - Snapshot({metrics: state->IndexerState.toMetrics, runtime: Metrics.sampleRuntime()}), - ), - Worker.snapshotIntervalMillis, - ) - } + Worker.startReporting(~getMetrics=() => state->IndexerState.toMetrics) setIndexerState(state) state->IndexerLoop.start await runUntilFatalError @@ -568,7 +559,7 @@ let start = async ( ) => { let config = Config.load() switch isTest ? None : Supervisor.planForRun(~config) { - | Some(workers) => await Supervisor.run(~workers, ~reset) + | Some(workers) => await Supervisor.run(~config, ~workers, ~reset) | None => await startIndexer( ~config, diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index cf913d0a9..d0d26b82c 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -233,9 +233,7 @@ let awaitExit = async (group): outcome => { // Runs the group: creates the schema for every chain, forks a worker per plan // entry, and serves the run's metrics, console and display from what they // report. Returns once every worker has exited; throws if any of them failed. -let run = async (~workers: array, ~reset) => { - let config = Config.load() - +let run = async (~config: Config.t, ~workers: array, ~reset) => { // Every chain's state has to exist before a worker resumes it: an isolated // run refuses to initialize, precisely so it can't create rows for its own // chains and leave the chains it skipped with nothing to resume. It is the diff --git a/packages/envio/src/Worker.res b/packages/envio/src/Worker.res index 888a15a50..bcfb4c377 100644 --- a/packages/envio/src/Worker.res +++ b/packages/envio/src/Worker.res @@ -29,7 +29,7 @@ type workerMessage = // How often a worker reports. Matches the TUI's own refresh, so the supervised // display moves at the same rate an unsplit run's does. -let snapshotIntervalMillis = 500 +%%private(let snapshotIntervalMillis = 500) // The supervisor is the one that stops a worker, and the one whose absence // ends it. A terminal's interrupt reaches the whole group at once, so the @@ -45,9 +45,18 @@ let bindToSupervisor = () => { }) } -let send = (message: workerMessage) => +%%private(let send = (message: workerMessage) => NodeJs.Process.sendToParent(message)->ignore) + +// Reports this process's chains and its own runtime for as long as it runs, so +// the supervisor can merge every worker's into the one snapshot the run serves. +// Does nothing in a process nobody forked. +let startReporting = (~getMetrics: unit => Metrics.t) => if isEnabled { - NodeJs.Process.sendToParent(message)->ignore + Metrics.startRuntimeCollectors() + let _intervalId = setInterval( + () => send(Snapshot({metrics: getMetrics(), runtime: Metrics.sampleRuntime()})), + snapshotIntervalMillis, + ) } // Resolves with the init payload, the one message a supervisor sends its worker. From 5b0baf7aaf6854ba983cbc3831fe0467d6459987 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 17 Sep 2026 10:21:28 +0000 Subject: [PATCH 22/61] Say the run's storage once, and don't call an empty run synced MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A supervised run drew its first frame before any worker had reported, and an empty chain list reduced to "fully synced" — the frame claimed the run had been synced for 56 years. Every worker also repeated the resume the supervisor had just announced, once per chain, and every forked worker re-ran `cargo build` in a dev run, contending over the cargo lock. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- .../test/lib_tests/SyncETA_test.res | 9 ++ .../lib_tests/WorkerResumeLogging_test.res | 84 +++++++++++++++++++ packages/envio/src/Core.res | 4 + packages/envio/src/Persistence.res | 7 +- packages/envio/src/tui/components/SyncETA.res | 18 ++-- 5 files changed, 114 insertions(+), 8 deletions(-) create mode 100644 packages/envio-tests/test/lib_tests/SyncETA_test.res create mode 100644 packages/envio-tests/test/lib_tests/WorkerResumeLogging_test.res diff --git a/packages/envio-tests/test/lib_tests/SyncETA_test.res b/packages/envio-tests/test/lib_tests/SyncETA_test.res new file mode 100644 index 000000000..121fc8ef5 --- /dev/null +++ b/packages/envio-tests/test/lib_tests/SyncETA_test.res @@ -0,0 +1,9 @@ +open Vitest + +describe("SyncETA.isIndexerFullySynced", () => { + // The supervisor renders a frame before any worker has reported, and a run + // with nothing to report is not a run that finished syncing. + it("Doesn't call an indexer with no reported chains synced", t => { + t.expect(SyncETA.isIndexerFullySynced([])).toBe(false) + }) +}) diff --git a/packages/envio-tests/test/lib_tests/WorkerResumeLogging_test.res b/packages/envio-tests/test/lib_tests/WorkerResumeLogging_test.res new file mode 100644 index 000000000..569d78567 --- /dev/null +++ b/packages/envio-tests/test/lib_tests/WorkerResumeLogging_test.res @@ -0,0 +1,84 @@ +open Vitest + +let sql = PgStorage.makeClient() +let pgSchema = TestPgSchema.make() +let config = TestConfig.make() + +Async.afterAll(async () => { + await sql->TestPgSchema.drop(~pgSchema) + await sql->Postgres.endSql +}) + +let makePersistence = () => + Persistence.make( + ~userEntities=config.userEntities, + ~allEnums=config.allEnums, + ~storage=PgStorage.make( + ~sql, + ~pgHost=Env.Db.host, + ~pgSchema, + ~pgPort=Env.Db.port, + ~pgUser=Env.Db.user, + ~pgDatabase=Env.Db.database, + ~pgPassword=Env.Db.password, + ~isHasuraEnabled=false, + ~ecosystem=Evm, + ), + ) + +let initRun = (~requireInitialized) => + makePersistence()->Persistence.init( + ~chainConfigs=config.chainMap->ChainMap.values, + ~contractMapping=config.contractMapping, + ~envioInfo=JSON.Encode.object(Dict.make()), + ~resetCommand="envio dev -r", + ~runCommand=Some("envio dev"), + ~lowercaseAddresses=config.lowercaseAddresses, + ~requireInitialized, + ) + +let logLines = async path => + switch await NodeJs.Fs.Promises.readFile(~filepath=NodeJs.Path.resolve([path]), ~encoding=Utf8) { + | contents => + contents + ->String.trim + ->String.split("\n") + ->Array.filterMap(line => + switch line->JSON.parseOrThrow->JSON.Decode.object { + | Some(fields) => fields->Dict.get("msg")->Option.flatMap(JSON.Decode.string) + | None => None + } + ) + | exception _ => [] + } + +describe("Resuming an isolated worker", () => { + // The supervisor announces the run's storage once, for every chain. A worker + // resuming the state it was handed has nothing to add to that. + Async.it("Stays quiet about storage the supervisor already announced", async t => { + await initRun(~requireInitialized=false) + + let path = `${NodeJs.Process.cwd()}/lib/envio-worker-resume-${Date.now()->Float.toString}.log` + Logging.setLogger( + Logging.makeLogger( + ~logStrategy=FileOnly, + ~logFilePath=path, + ~defaultFileLogLevel=#info, + ~userLogLevel=#info, + ), + ) + + await initRun(~requireInitialized=true) + Logging.info("done") + + let rec until = async deadline => + switch await logLines(path) { + | lines if lines->Array.includes("done") || Date.now() > deadline => lines + | _ => + await Utils.delay(50) + await until(deadline) + } + + t.expect(await until(Date.now() +. 3000.)).toStrictEqual(["done"]) + }) +}) diff --git a/packages/envio/src/Core.res b/packages/envio/src/Core.res index 6f1a5f90e..33bc9e342 100644 --- a/packages/envio/src/Core.res +++ b/packages/envio/src/Core.res @@ -166,6 +166,10 @@ let loadDevAddon: ({..}, string) => addon = %raw(`function(req, envioDir) { fs.copyFileSync(srcPath, nodePath); } + // Forked workers inherit this, so only the first process in a run pays for + // the cargo build (and they don't contend over the cargo lock). + process.env.ENVIO_DEV_ADDON = nodePath; + return req(nodePath); }`) diff --git a/packages/envio/src/Persistence.res b/packages/envio/src/Persistence.res index a6539be4e..7b8354b08 100644 --- a/packages/envio/src/Persistence.res +++ b/packages/envio/src/Persistence.res @@ -324,7 +324,10 @@ let init = { | _ => false } ) { - Logging.info(`Found existing indexer storage. Resuming indexing state...`) + // An isolated process resumes state its supervisor already announced + // for the whole run, so it says so only to its own log file. + let logResume = requireInitialized ? Logging.debug : Logging.info + logResume(`Found existing indexer storage. Resuming indexing state...`) let initialState = await persistence.storage.resumeInitialState( ~entities=persistence.allEntities, ~chainIds=chainConfigs->Array.map(chain => chain.id), @@ -343,7 +346,7 @@ let init = { initialState.chains->Array.forEach(c => { progress->ChainId.Dict.set(c.id, c.progressBlockNumber) }) - Logging.info({ + logResume({ "msg": `Successfully resumed indexing state! Continuing from the last checkpoint.`, "progress": progress, }) diff --git a/packages/envio/src/tui/components/SyncETA.res b/packages/envio/src/tui/components/SyncETA.res index fcb0e1358..197e9be51 100644 --- a/packages/envio/src/tui/components/SyncETA.res +++ b/packages/envio/src/tui/components/SyncETA.res @@ -1,12 +1,18 @@ open Ink let isIndexerFullySynced = (chains: array) => { - chains->Array.reduce(true, (accum, current) => { - switch current.progress { - | Synced(_) => accum - | _ => false - } - }) + switch chains { + // A supervised run draws its first frame before any worker has reported, and + // a run with nothing to report hasn't finished syncing. + | [] => false + | chains => + chains->Array.every(chain => + switch chain.progress { + | Synced(_) => true + | _ => false + } + ) + } } let getTotalRemainingBlocks = (chains: array) => { From b1664c0826341646752aa57cd55ea10a0f203d0c Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 18 Sep 2026 10:40:45 +0000 Subject: [PATCH 23/61] Keep the default connection budget at two Splitting a run spends connections the operator never agreed to, so the budget they didn't set stays what one process has always used: raising ENVIO_PG_MAX_CONNECTIONS is what opts a run into being split. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- packages/cli/CommandLineHelp.md | 2 +- packages/cli/src/cli_args/clap_definitions.rs | 5 +++-- packages/envio-tests/test/lib_tests/Supervisor_test.res | 8 ++++++++ packages/envio/src/Env.res | 6 +++--- 4 files changed, 15 insertions(+), 6 deletions(-) diff --git a/packages/cli/CommandLineHelp.md b/packages/cli/CommandLineHelp.md index 58dec8ccc..2839157d3 100644 --- a/packages/cli/CommandLineHelp.md +++ b/packages/cli/CommandLineHelp.md @@ -378,7 +378,7 @@ Start the indexer. Runs codegen automatically before launching so the on-disk ty ###### **Options:** * `-r`, `--restart` — Clear your database and restart indexing from scratch -* `--chain ` — Index only this chain, leaving the others to their own `envio start --chain` processes. Only needed to place the chains yourself: a plain `envio start` already splits them across processes, and manages those processes for you, whenever the schema's entities are all per-chain. `ENVIO_PG_MAX_CONNECTIONS` is the budget for the whole run and buys one process per two connections, so its default of 10 affords five. A schema with an entity shared across chains, or a budget under 4, runs in one process as it always has. Repeat the flag for several chains. Requires a schema whose entities are all per-chain, created for every chain by `envio local db-migrate up` before any process starts. Assign each configured chain to exactly one process, and give each its own `ENVIO_INDEXER_PORT`. Each process builds the indexes for its own chains and reports them ready as they catch up, independently of the others +* `--chain ` — Index only this chain, leaving the others to their own `envio start --chain` processes. Only needed to place the chains yourself: a plain `envio start` already splits them across processes, and manages those processes for you, whenever the schema's entities are all per-chain. `ENVIO_PG_MAX_CONNECTIONS` is the budget for the whole run and buys one process per two connections, so raising it from its default of 2 is what splits a run. A schema with an entity shared across chains, or a budget under 4, runs in one process as it always has. Repeat the flag for several chains. Requires a schema whose entities are all per-chain, created for every chain by `envio local db-migrate up` before any process starts. Assign each configured chain to exactly one process, and give each its own `ENVIO_INDEXER_PORT`. Each process builds the indexes for its own chains and reports them ready as they catch up, independently of the others diff --git a/packages/cli/src/cli_args/clap_definitions.rs b/packages/cli/src/cli_args/clap_definitions.rs index 699e2eb41..583d32bb7 100644 --- a/packages/cli/src/cli_args/clap_definitions.rs +++ b/packages/cli/src/cli_args/clap_definitions.rs @@ -148,8 +148,9 @@ pub struct StartArgs { ///Only needed to place the chains yourself: a plain `envio start` already splits them across ///processes, and manages those processes for you, whenever the schema's entities are all ///per-chain. `ENVIO_PG_MAX_CONNECTIONS` is the budget for the whole run and buys one process - ///per two connections, so its default of 10 affords five. A schema with an entity shared - ///across chains, or a budget under 4, runs in one process as it always has. + ///per two connections, so raising it from its default of 2 is what splits a run. A schema + ///with an entity shared across chains, or a budget under 4, runs in one process as it always + ///has. ///Repeat the flag for several chains. Requires a schema whose entities are all per-chain, ///created for every chain by `envio local db-migrate up` before any process starts. ///Assign each configured chain to exactly one process, and give each its own diff --git a/packages/envio-tests/test/lib_tests/Supervisor_test.res b/packages/envio-tests/test/lib_tests/Supervisor_test.res index 9525cf32c..671ee0a1e 100644 --- a/packages/envio-tests/test/lib_tests/Supervisor_test.res +++ b/packages/envio-tests/test/lib_tests/Supervisor_test.res @@ -37,6 +37,14 @@ describe("Supervisor.plan", () => { }) }) +describe("Supervisor.plan on the default budget", () => { + // Splitting a run costs connections the operator didn't ask to spend, so the + // budget they didn't set is the one a single process has always used. + it("Keeps a run in one process until the budget is raised", t => { + t.expect(Supervisor.plan(~chainIds=chains(4), ~maxConnections=Env.Db.maxConnections)).toBe(None) + }) +}) + describe("Supervisor.plan dealing order", () => { let assignment = (~chainIds, ~maxConnections) => Supervisor.plan(~chainIds=chainIds->Array.map(ChainId.fromInt), ~maxConnections) diff --git a/packages/envio/src/Env.res b/packages/envio/src/Env.res index b3b77b184..b60490926 100644 --- a/packages/envio/src/Env.res +++ b/packages/envio/src/Env.res @@ -129,9 +129,9 @@ module Db = { ) // The budget for the whole run, not for one process: a run that splits across // workers divides it among them, and each caps its own pool to its share. - // The default affords five workers, so a per-chain schema of several chains - // splits without being asked to. - let maxConnections = envSafe->EnvSafe.get("ENVIO_PG_MAX_CONNECTIONS", S.int, ~fallback=10) + // The default buys a single worker, so a run splits only once the operator + // raises the budget it may spend. + let maxConnections = envSafe->EnvSafe.get("ENVIO_PG_MAX_CONNECTIONS", S.int, ~fallback=2) } // Required env vars are validated lazily in PgStorage when the user From 22c71c050dab068932985d025e1c1f79f4af1f8b Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 18 Sep 2026 11:32:29 +0000 Subject: [PATCH 24/61] Read a worker's output rather than let it write behind the frame Logging goes through console.log precisely so ink can keep it out of its frame, but that interception is per-process: a worker's line reached the terminal as a raw write the supervisor's display knew nothing about, so every one of them tore the frame and redrew it. A supervisor that draws now reads its workers' pipes and logs the lines as its own. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- .../envio-tests/test/SupervisorFork_test.res | 49 ++++++++++++++- .../envio-tests/test/helpers/fakeWorker.mjs | 9 +++ packages/envio/src/Supervisor.res | 62 +++++++++++++++++-- packages/envio/src/bindings/NodeJs.res | 8 +++ 4 files changed, 121 insertions(+), 7 deletions(-) diff --git a/packages/envio-tests/test/SupervisorFork_test.res b/packages/envio-tests/test/SupervisorFork_test.res index fbace3981..0095ecd75 100644 --- a/packages/envio-tests/test/SupervisorFork_test.res +++ b/packages/envio-tests/test/SupervisorFork_test.res @@ -11,12 +11,14 @@ type fixtureReport = { let fixturePath = `${NodeJs.Process.cwd()}/test/helpers/fakeWorker.mjs` -let forkFixture = (~chainIds, ~maxConnections=2, ~workerIndex=0) => +let forkFixture = (~chainIds, ~maxConnections=2, ~workerIndex=0, ~pipeOutput=false, ~onOutput=?) => Supervisor.fork( {chainIds: chainIds->Array.map(ChainId.fromInt), maxConnections}, ~workerIndex, ~configJson=JSON.Object(Dict.fromArray([("name", JSON.String("indexer"))])), ~entryPath=fixturePath, + ~pipeOutput, + ~onOutput?, ) describe("Supervisor.fork", () => { @@ -92,3 +94,48 @@ describe("Supervisor.awaitExit", () => { )) }) }) + +describe("Supervisor.readLines", () => { + it("Holds a half line until the chunk that finishes it, or the stream ends", t => { + let lines = [] + let (read, flush) = Supervisor.readLines(~onLine=line => lines->Array.push(line)->ignore) + ["a line\nand ", "half of ", "another\nlast\n", "no newline here"]->Array.forEach(read) + flush() + // Nothing is left to flush twice. + flush() + + t.expect(lines).toStrictEqual([ + "a line", + "and half of another", + "last", + "no newline here", + ]) + }) +}) + +describe("Supervisor.fork output", () => { + // A worker writing straight to the terminal tears the frame its supervisor + // draws: ink only knows about the lines its own process logs. Sorted, since + // stdout and stderr are two pipes and neither waits for the other. + Async.it("Hands the supervisor every line a worker writes, whole", async t => { + NodeJs.Process.process.env->Dict.set("FAKE_WORKER", "print") + let lines = [] + let group: Supervisor.group = { + running: [ + forkFixture( + ~chainIds=[1], + ~pipeOutput=true, + ~onOutput=line => lines->Array.push(line)->ignore, + ), + ], + stopping: false, + } + let _ = await group->Supervisor.awaitExit + + t.expect(lines->Array.toSorted(String.compare)).toStrictEqual([ + "first line", + "from stderr", + "second line", + ]) + }) +}) diff --git a/packages/envio-tests/test/helpers/fakeWorker.mjs b/packages/envio-tests/test/helpers/fakeWorker.mjs index 18c7b8572..37cdb799b 100644 --- a/packages/envio-tests/test/helpers/fakeWorker.mjs +++ b/packages/envio-tests/test/helpers/fakeWorker.mjs @@ -21,5 +21,14 @@ process.on("message", (message) => { } }); +// Writes across chunk boundaries the way a real process does: a pipe hands the +// supervisor whatever has been flushed, not whole lines. +if (mode === "print") { + process.stdout.write("first line\nsecond "); + process.stderr.write("from stderr\n"); + process.stdout.write("line\n"); + setTimeout(() => process.exit(0), 50); +} + // Nothing else keeps a "linger" worker alive; it waits to be stopped. if (mode === "linger") setInterval(() => {}, 1000); diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index d0d26b82c..cb6115e22 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -97,6 +97,33 @@ let configForWorker = (configJson: JSON.t, ~worker) => | None => JsError.throwWithMessage("Invalid indexer config: not an object") } +// A pipe hands over whatever has been flushed, so a chunk boundary falls +// wherever the OS put it: the tail of a chunk is a line only once the chunk +// that ends it arrives. Reading pairs with a flush, since a process that dies +// mid-line still wrote what it managed to — which is when it matters most. +let readLines = (~onLine) => { + let pending = ref("") + let read = chunk => { + let parts = (pending.contents ++ chunk)->String.split("\n") + pending := parts->Array.pop->Option.getOr("") + parts->Array.forEach(onLine) + } + let flush = () => + switch pending.contents { + | "" => () + | line => { + pending := "" + onLine(line) + } + } + (read, flush) +} + +// Whether this process's own output is a terminal. `pino-pretty` colorizes on +// that test, and a piped worker would fail it for a run the operator is +// watching in colour. +@val external stdoutIsTty: Nullable.t = "process.stdout.isTTY" + let fork = ( worker: worker, ~workerIndex, @@ -104,6 +131,10 @@ let fork = ( // The entry this process was itself started from, so a worker is the same // program as its supervisor however the package was installed. ~entryPath=NodeJs.Process.argv->Array.getUnsafe(1), + // A run that draws a display reads its workers' output instead of letting + // them write to the terminal behind the frame's back. + ~pipeOutput=false, + ~onOutput=Console.log, ) => { let env = NodeJs.Process.process.env->Dict.copy env->Dict.set(Worker.envVar, "true") @@ -111,6 +142,9 @@ let fork = ( // loads, which is why it rides in the spawn environment rather than a message. env->Dict.set("ENVIO_PG_MAX_CONNECTIONS", worker.maxConnections->Int.toString) env->Dict.set("LOG_FILE", logFilePath(~workerIndex)) + if pipeOutput && stdoutIsTty->Nullable.toOption->Option.getOr(false) { + env->Dict.set("FORCE_COLOR", "1") + } let child = NodeJs.ChildProcess.fork( entryPath, @@ -118,12 +152,26 @@ let fork = ( { env, serialization: "advanced", - // Workers write straight to the run's own output. Their lines already say - // which chain they came from, so there is nothing for the supervisor to - // add by reading them first. - stdio: ["inherit", "inherit", "inherit", "ipc"], + stdio: pipeOutput + ? ["inherit", "pipe", "pipe", "ipc"] + : ["inherit", "inherit", "inherit", "ipc"], }, ) + if pipeOutput { + // Both streams become one stream of lines: the supervisor logs them the way + // it logs its own, which is the only way ink can keep them out of its frame. + [child->NodeJs.ChildProcess.stdout, child->NodeJs.ChildProcess.stderr]->Array.forEach(stream => + switch stream->Null.toOption { + | Some(stream) => { + let (read, flush) = readLines(~onLine=onOutput) + stream->NodeJs.ChildProcess.setEncoding("utf8") + stream->NodeJs.ChildProcess.onData(read) + stream->NodeJs.ChildProcess.onEnd(flush) + } + | None => () + } + ) + } child ->NodeJs.ChildProcess.send(Worker.Init({config: configJson->configForWorker(~worker)})) ->ignore @@ -262,9 +310,12 @@ let run = async (~config: Config.t, ~workers: array, ~reset) => { // The config as the CLI handed it over, narrowed per worker on the way out. let configJson = Config.getPublicConfigJson() + // Decided before the first fork: it is what makes a worker's output the + // supervisor's to print. + let shouldUseTui = Tui.shouldUse() let group = { running: workers->Array.mapWithIndex((worker, workerIndex) => - worker->fork(~workerIndex, ~configJson) + worker->fork(~workerIndex, ~configJson, ~pipeOutput=shouldUseTui) ), stopping: false, } @@ -299,7 +350,6 @@ let run = async (~config: Config.t, ~workers: array, ~reset) => { ~onSyncCache=() => syncCache(~dump=() => dumpCache(~config)), ) - let shouldUseTui = Tui.shouldUse() if shouldUseTui { let _rerender = Tui.start(~config, ~getMetrics=() => reported()->merge) } diff --git a/packages/envio/src/bindings/NodeJs.res b/packages/envio/src/bindings/NodeJs.res index 8ad8ef9fc..3390a8dcb 100644 --- a/packages/envio/src/bindings/NodeJs.res +++ b/packages/envio/src/bindings/NodeJs.res @@ -176,6 +176,14 @@ module ChildProcess = { external onExit: (child, @as("exit") _, (Null.t, Null.t) => unit) => unit = "on" @send external onChildError: (child, @as("error") _, exn => unit) => unit = "on" @send external kill: (child, string) => bool = "kill" + + // Present only for a stdio slot the parent asked to pipe. + type stdioStream + @get external stdout: child => Null.t = "stdout" + @get external stderr: child => Null.t = "stderr" + @send external setEncoding: (stdioStream, string) => unit = "setEncoding" + @send external onData: (stdioStream, @as("data") _, string => unit) => unit = "on" + @send external onEnd: (stdioStream, @as("end") _, unit => unit) => unit = "on" } module Url = { From e82cca3da0f168d6390c5f5db8398eabe2e0c42e Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 18 Sep 2026 12:32:41 +0000 Subject: [PATCH 25/61] Switch a split run to realtime as one indexer MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit An indexer enters the reorg threshold and stamps every chain ready as a whole, which a process driving part of a run can't decide for itself: its own chains reach the head while another process is still backfilling. A worker now holds both transitions until its supervisor, the only one who sees every chain, says the run has arrived — so a split run switches over exactly where an unsplit one does. A chain with no reorg threshold is held by the same gate rather than racing ahead of the chains that have one. The supervisor holds nobody when every chain resumed already caught up, and counts a chain that resumed realtime or reached its end block as arrived, so the barrier can always be opened. A held worker also stays for the run at its end block instead of exiting with the indexes it still owes. What a worker is told now rides entirely in the fork's environment, as a JSON value beside the budget and the log file it already carried: a worker parses the same config its supervisor did and narrows it to the chains it was handed. The storage it resumes already refuses a config that disagrees with the one the run was created from, which is a stronger guarantee than the handover message it replaces. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- .../test/SupervisedRealtime_test.res | 141 ++++++++++++++++++ .../envio-tests/test/SupervisorFork_test.res | 27 +++- .../test/helpers/IndexerRunner.res | 8 + .../envio-tests/test/helpers/Scenario.res | 4 + .../test/helpers/TestChainMetrics.res | 28 ++++ .../envio-tests/test/helpers/fakeWorker.mjs | 31 ++-- .../test/lib_tests/Metrics_test.res | 27 +--- .../test/lib_tests/Supervisor_test.res | 57 ++++++- packages/envio/src/BatchProcessing.res | 14 +- packages/envio/src/Bin.res | 6 +- packages/envio/src/ChainState.res | 3 + packages/envio/src/Config.res | 15 ++ packages/envio/src/CrossChainState.res | 43 +++++- packages/envio/src/CrossChainState.resi | 10 +- packages/envio/src/IndexerLoop.res | 2 + packages/envio/src/IndexerState.res | 34 ++++- packages/envio/src/IndexerState.resi | 8 + packages/envio/src/Main.res | 13 +- packages/envio/src/Metrics.res | 4 + packages/envio/src/Supervisor.res | 88 ++++++++--- packages/envio/src/Worker.res | 48 ++++-- packages/envio/src/bindings/NodeJs.res | 2 - 22 files changed, 501 insertions(+), 112 deletions(-) create mode 100644 packages/envio-tests/test/SupervisedRealtime_test.res diff --git a/packages/envio-tests/test/SupervisedRealtime_test.res b/packages/envio-tests/test/SupervisedRealtime_test.res new file mode 100644 index 000000000..775a953b4 --- /dev/null +++ b/packages/envio-tests/test/SupervisedRealtime_test.res @@ -0,0 +1,141 @@ +open Vitest + +// An indexer switches to realtime as a whole: every chain enters the reorg +// threshold together and every chain is stamped ready at one instant. A split +// run's processes each see only their own chains, so a supervised worker holds +// those transitions until the supervisor says every chain in the run has +// arrived. + +let schema = ` +type A { + id: ID! +} +` + +let chainYaml = (chainId, address) => + ` + - id: ${chainId->Int.toString} + rpc: + url: https://rpc${chainId->Int.toString}.example.test + for: sync + start_block: 1 + contracts: + - name: Gravatar + address: "${address}" +` + +let endBlockScenario = Scenario.make( + ~configYaml=` +name: supervised-realtime-end-block +disable_default_cross_chain: true +contracts: + - name: Gravatar + events: + - event: "TestEvent()" +chains: + - id: 1 + rpc: + url: https://rpc1.example.test + for: sync + start_block: 1 + end_block: 100 + contracts: + - name: Gravatar + address: "0x2B2f78c5BF6D9C12Ee1225D5F374aa91204580c3"`, + ~schema, +) + +let scenario = Scenario.make( + ~configYaml=` +name: supervised-realtime +disable_default_cross_chain: true +contracts: + - name: Gravatar + events: + - event: "TestEvent()" +chains:${chainYaml(1, "0x2B2f78c5BF6D9C12Ee1225D5F374aa91204580c3")}${chainYaml( + 137, + "0x3B2f78c5BF6D9C12Ee1225D5F374aa91204580c3", + )}`, + ~schema, +) + +let readyAtByChainId = async (~sql, ~pgSchema) => { + let rows: array<{ + "id": ChainId.t, + "ready_at": Null.t, + }> = await sql->Postgres.unsafe( + `SELECT "id", "ready_at" FROM "${pgSchema}"."envio_chains" ORDER BY "id";`, + ) + rows->Array.map(row => (row["id"]->ChainId.toString, row["ready_at"]->Null.toOption)) +} + +let catchUp = (~source: MockSource.t) => { + source.resolveGetHeightOrThrow(100) + source.resolveGetItemsOrThrow([], ~latestFetchedBlockNumber=100) +} + +describe("A supervised worker", () => { + scenario->Scenario.it( + "Waits for the run before going realtime, then stamps every chain at one instant", + ~sources=[{chain: 1}, {chain: 137}], + ~holdRealtime=true, + async (~t, ~indexer, ~source) => { + let {sql, pgSchema} = indexer.pg + catchUp(~source=source(1)) + catchUp(~source=source(137)) + await indexer.waitUntilIdle() + + t.expect( + (await readyAtByChainId(~sql, ~pgSchema), await indexer.metric("envio_progress_ready")), + ~message="Both chains are at the head, but the run has not said so", + ).toEqual(( + [("1", None), ("137", None)], + [{value: "0", labels: dict{"chainId": "1"}}, {value: "0", labels: dict{"chainId": "137"}}], + )) + + indexer.releaseRealtime() + await indexer.waitUntilReady() + + let readyAt = await readyAtByChainId(~sql, ~pgSchema) + let stamps = readyAt->Array.filterMap(((_, at)) => at->Option.map(Date.getTime)) + t.expect( + (readyAt->Array.map(((chainId, _)) => chainId), stamps->Array.length, stamps->Set.fromArray->Set.size), + ~message="The release stamps every chain, and one caught-up indexer is one instant", + ).toEqual((["1", "137"], 2, 1)) + }, + ) +}) + +describe("A supervised worker at its end block", () => { + let exited = ref(false) + + endBlockScenario->Scenario.it( + "Stays for the run rather than exiting with the indexes it still owes", + ~sources=[{chain: 1}], + ~holdRealtime=true, + async (~t, ~indexer, ~source) => { + let {sql, pgSchema} = indexer.pg + catchUp(~source=source(1)) + await indexer.getBatchWritePromise() + await indexer.waitUntilIdle() + + t.expect( + (exited.contents, await readyAtByChainId(~sql, ~pgSchema)), + ~message="Its chain is done, but it still owes the schema the indexes it deferred", + ).toEqual((false, [("1", None)])) + + indexer.releaseRealtime() + await indexer.waitUntilReady() + + t.expect( + (await readyAtByChainId(~sql, ~pgSchema))->Array.map(((chainId, readyAt)) => ( + chainId, + readyAt->Option.isSome, + )), + ~message="Released, it finalizes and stamps its chain", + ).toEqual([("1", true)]) + }, + ~onExit=() => exited := true, + ) +}) diff --git a/packages/envio-tests/test/SupervisorFork_test.res b/packages/envio-tests/test/SupervisorFork_test.res index 0095ecd75..542482f63 100644 --- a/packages/envio-tests/test/SupervisorFork_test.res +++ b/packages/envio-tests/test/SupervisorFork_test.res @@ -1,9 +1,10 @@ open Vitest // What the fixture worker reports back in place of a metrics snapshot: the -// narrowing and the environment its supervisor handed it. +// environment its supervisor handed it, which is the whole of what a worker is +// told before it starts. type fixtureReport = { - isolatedChains: array, + workerConfig: string, maxConnections: string, logFile: string, startTime: Date.t, @@ -11,11 +12,18 @@ type fixtureReport = { let fixturePath = `${NodeJs.Process.cwd()}/test/helpers/fakeWorker.mjs` -let forkFixture = (~chainIds, ~maxConnections=2, ~workerIndex=0, ~pipeOutput=false, ~onOutput=?) => +let forkFixture = ( + ~chainIds, + ~maxConnections=2, + ~workerIndex=0, + ~holdRealtime=false, + ~pipeOutput=false, + ~onOutput=?, +) => Supervisor.fork( {chainIds: chainIds->Array.map(ChainId.fromInt), maxConnections}, ~workerIndex, - ~configJson=JSON.Object(Dict.fromArray([("name", JSON.String("indexer"))])), + ~holdRealtime, ~entryPath=fixturePath, ~pipeOutput, ~onOutput?, @@ -23,7 +31,12 @@ let forkFixture = (~chainIds, ~maxConnections=2, ~workerIndex=0, ~pipeOutput=fal describe("Supervisor.fork", () => { Async.it("Hands a worker its chains, its budget share, and its own log file", async t => { - let running = forkFixture(~chainIds=[1, 137], ~maxConnections=3, ~workerIndex=1) + let running = forkFixture( + ~chainIds=[1, 137], + ~maxConnections=3, + ~workerIndex=1, + ~holdRealtime=true, + ) let report = await Promise.make( (resolve, _) => @@ -38,7 +51,9 @@ describe("Supervisor.fork", () => { running.child->NodeJs.ChildProcess.kill("SIGTERM")->ignore t.expect(report).toStrictEqual({ - isolatedChains: [1., 137.], + // Everything the supervisor decided, in the environment: a worker needs it + // before it can load its own config, so it can't arrive as a message. + workerConfig: `{"chainIds":[1,137],"holdRealtime":true}`, maxConnections: "3", logFile: Supervisor.logFilePath(~workerIndex=1), // Proof the channel clones rather than stringifies: a JSON round trip diff --git a/packages/envio-tests/test/helpers/IndexerRunner.res b/packages/envio-tests/test/helpers/IndexerRunner.res index 77ec72d50..093be6c74 100644 --- a/packages/envio-tests/test/helpers/IndexerRunner.res +++ b/packages/envio-tests/test/helpers/IndexerRunner.res @@ -59,6 +59,9 @@ type rec t = { // `~chains` resumes the same schema driving only those chains, the way // `envio start --chain` does. The chains left out keep their stored state. restart: (~chains: array=?, unit) => promise, + // Stands in for the supervisor's go-ahead in a run started with + // `~holdRealtime`. + releaseRealtime: unit => unit, } let entityConfigByName = (config: Config.t, name): Internal.entityConfig => @@ -77,6 +80,9 @@ let run = async ( ~backend: backend=selectedBackend, ~reducedPollingInterval=?, ~targetBufferSize=?, + // Runs the indexer the way a supervised worker runs: it waits to be released + // before entering the reorg threshold or switching to realtime. + ~holdRealtime=false, ~onError=?, ~onExit=?, ~mapStorage: Persistence.storage => Persistence.storage=storage => storage, @@ -163,6 +169,7 @@ let run = async ( ~targetBufferSize?, ~isDevelopmentMode=false, ~shouldUseTui=false, + ~holdRealtime, ~onError, ~onExit?, ) @@ -333,6 +340,7 @@ let run = async ( JsError.throwWithMessage("Timed out waiting for the indexer to go idle") } }, + releaseRealtime: () => state->IndexerState.releaseRealtime, waitUntilReady: async () => { let isReady = () => state diff --git a/packages/envio-tests/test/helpers/Scenario.res b/packages/envio-tests/test/helpers/Scenario.res index c1c849e11..aaecddea5 100644 --- a/packages/envio-tests/test/helpers/Scenario.res +++ b/packages/envio-tests/test/helpers/Scenario.res @@ -168,6 +168,7 @@ let run = async ( ~maxAddrInPartition=?, ~clientFilterAddressThreshold=?, ~reorgThresholdReadyTolerance=?, + ~holdRealtime=?, ~onError=?, ~onExit=?, ~mapStorage=?, @@ -237,6 +238,7 @@ let run = async ( }), ~reducedPollingInterval?, ~targetBufferSize?, + ~holdRealtime?, ~onError?, ~onExit?, ~mapStorage?, @@ -267,6 +269,7 @@ let it = ( ~maxAddrInPartition=?, ~clientFilterAddressThreshold=?, ~reorgThresholdReadyTolerance=?, + ~holdRealtime=?, ~onError=?, ~onExit=?, ~mapStorage=?, @@ -293,6 +296,7 @@ let it = ( ~maxAddrInPartition?, ~clientFilterAddressThreshold?, ~reorgThresholdReadyTolerance?, + ~holdRealtime?, ~onError?, ~onExit?, ~mapStorage?, diff --git a/packages/envio-tests/test/helpers/TestChainMetrics.res b/packages/envio-tests/test/helpers/TestChainMetrics.res index cae0a1793..394a0e9c1 100644 --- a/packages/envio-tests/test/helpers/TestChainMetrics.res +++ b/packages/envio-tests/test/helpers/TestChainMetrics.res @@ -63,3 +63,31 @@ let make = ( ~contractMapping=TestConfig.default.contractMapping, ~registrationsByChainId, )->ChainState.toMetrics + +// The snapshot a run reports when nothing has happened yet. Tests spread this +// and name only the chains they assert on. +let emptySnapshot: Metrics.t = { + startTime: Date.fromTime(0.), + metricTime: Date.fromTime(0.), + elapsedSeconds: 0., + targetBufferSize: 0, + isInReorgThreshold: false, + rollbackEnabled: false, + maxBatchSize: 0, + preloadSeconds: 0., + processingSeconds: 0., + processingStalledOnFetchSeconds: 0., + processingStalledOnStorageWriteSeconds: 0., + rollbackSeconds: 0., + rollbackCount: 0, + rollbackEventsCount: 0., + chains: [], + handlers: [], + effects: [], + storageLoads: [], + storageWrites: [], + historyPrunes: [], + sourceRequests: [], + sourceHeights: [], + sourceHeightStreams: [], +} diff --git a/packages/envio-tests/test/helpers/fakeWorker.mjs b/packages/envio-tests/test/helpers/fakeWorker.mjs index 37cdb799b..eae6b60cb 100644 --- a/packages/envio-tests/test/helpers/fakeWorker.mjs +++ b/packages/envio-tests/test/helpers/fakeWorker.mjs @@ -3,24 +3,23 @@ // with real processes. const mode = process.env.FAKE_WORKER ?? "report"; -process.on("message", (message) => { - if (message.kind === "init") { - process.send({ - kind: "snapshot", - metrics: { - isolatedChains: message.config.isolatedChains, - maxConnections: process.env.ENVIO_PG_MAX_CONNECTIONS, - logFile: process.env.LOG_FILE, - // A Date survives only under structured-clone serialization, which is - // what a metrics snapshot's timestamps need. - startTime: new Date(1700000000000), - }, - }); - if (mode === "succeed") process.exit(0); - if (mode === "fail") process.exit(1); - } +// A real worker reports on a timer; the fixture reports once, as soon as it is +// started, since nothing tells it when its supervisor is listening. +process.send({ + kind: "snapshot", + metrics: { + workerConfig: process.env.ENVIO_INTERNAL_WORKER, + maxConnections: process.env.ENVIO_PG_MAX_CONNECTIONS, + logFile: process.env.LOG_FILE, + // A Date survives only under structured-clone serialization, which is + // what a metrics snapshot's timestamps need. + startTime: new Date(1700000000000), + }, }); +if (mode === "succeed") process.exit(0); +if (mode === "fail") process.exit(1); + // Writes across chunk boundaries the way a real process does: a pipe hands the // supervisor whatever has been flushed, not whole lines. if (mode === "print") { diff --git a/packages/envio-tests/test/lib_tests/Metrics_test.res b/packages/envio-tests/test/lib_tests/Metrics_test.res index ddafd1e4f..360bc6c94 100644 --- a/packages/envio-tests/test/lib_tests/Metrics_test.res +++ b/packages/envio-tests/test/lib_tests/Metrics_test.res @@ -62,31 +62,7 @@ envio_source_request_seconds_total{method="getLogs"} 1.5`) // The state a Metrics.t carries when a test says nothing about it. Each test // below spreads this and names only the fields it asserts on. -let baseMetrics: Metrics.t = { - startTime: Date.fromTime(0.), - metricTime: Date.fromTime(0.), - elapsedSeconds: 0., - targetBufferSize: 0, - isInReorgThreshold: false, - rollbackEnabled: false, - maxBatchSize: 0, - preloadSeconds: 0., - processingSeconds: 0., - processingStalledOnFetchSeconds: 0., - processingStalledOnStorageWriteSeconds: 0., - rollbackSeconds: 0., - rollbackCount: 0, - rollbackEventsCount: 0., - chains: [], - handlers: [], - effects: [], - storageLoads: [], - storageWrites: [], - historyPrunes: [], - sourceRequests: [], - sourceHeights: [], - sourceHeightStreams: [], -} +let baseMetrics = TestChainMetrics.emptySnapshot describe("Metrics.collect", () => { it("Renders only the indexer info when there is no state", t => { @@ -266,6 +242,7 @@ envio_info{version="${Utils.EnvioPackage.value.version}"} 1 numAddresses: 7, addressesByContract: [("Gravatar", 5), ("NftFactory", 2)], isReady: true, + isReadyForReorgThreshold: true, sourceBlockNumber: 305, progressBlockNumber: 200, progressLatencyMs: Some(1500), diff --git a/packages/envio-tests/test/lib_tests/Supervisor_test.res b/packages/envio-tests/test/lib_tests/Supervisor_test.res index 671ee0a1e..7e2308ef4 100644 --- a/packages/envio-tests/test/lib_tests/Supervisor_test.res +++ b/packages/envio-tests/test/lib_tests/Supervisor_test.res @@ -138,15 +138,13 @@ describe("Supervisor.planForRun", () => { }) describe("Supervisor worker plumbing", () => { - it("Narrows the config it hands a worker to that worker's chains", t => { + it("Narrows a config it parsed itself to the chains it was given", t => { let configJson = JSON.Object( Dict.fromArray([("name", JSON.String("indexer")), ("isolatedChains", JSON.Null)]), ) t.expect( - configJson->Supervisor.configForWorker( - ~worker={chainIds: [1, 137]->Array.map(ChainId.fromInt), maxConnections: 2}, - ), + configJson->Config.withIsolatedChains(~chainIds=[1, 137]->Array.map(ChainId.fromInt)), ).toStrictEqual( JSON.Object( Dict.fromArray([ @@ -198,14 +196,29 @@ describe("Config.logContext", () => { describe("Worker.detect", () => { it("Counts as a worker only when forked with the variable and a channel", t => { - let forked = Dict.fromArray([(Worker.envVar, "true")]) + let forked = Dict.fromArray([(Worker.envVar, `{"chainIds":[137],"holdRealtime":true}`)]) t.expect([ Worker.detect(~env=forked, ~hasChannel=true), // A copy of the variable left in a shell, or a process manager forking // with a channel. Worker.detect(~env=forked, ~hasChannel=false), Worker.detect(~env=Dict.make(), ~hasChannel=true), - ]).toStrictEqual([true, false, false]) + ]).toStrictEqual([ + Some({Worker.chainIds: [137->ChainId.fromInt], holdRealtime: true}), + None, + None, + ]) + }) + + // The supervisor decides whether a run waits; a worker forked before that + // decision existed reads as one that doesn't. + it("Takes a config without the hold as one that doesn't wait", t => { + t.expect( + Worker.detect( + ~env=Dict.fromArray([(Worker.envVar, `{"chainIds":[1]}`)]), + ~hasChannel=true, + ), + ).toStrictEqual(Some({Worker.chainIds: [1->ChainId.fromInt], holdRealtime: false})) }) }) @@ -226,3 +239,35 @@ describe("Supervisor.syncCache", () => { t.expect(dumps.contents).toBe(2) }) }) + +describe("Supervisor.isRunAtHead", () => { + let chain = (~isReadyForReorgThreshold=false, ~isReady=false, ~endBlock=None, ~progressBlockNumber=50) => { + ...TestChainMetrics.make(~progressBlockNumber, ~firstEventBlockNumber=None, ~endBlock), + Metrics.isReadyForReorgThreshold, + isReady, + } + let snapshot = (chains): Metrics.t => {...TestChainMetrics.emptySnapshot, chains} + + it("Holds the run until every chain of every worker has arrived", t => { + t.expect([ + // A worker that hasn't reported yet drives chains nobody can see. Reading + // the run as arrived here would release it on a partial view. + [snapshot([chain(~isReadyForReorgThreshold=true)])]->Supervisor.isRunAtHead(~workerCount=2), + [ + snapshot([chain(~isReadyForReorgThreshold=true)]), + snapshot([chain(~isReadyForReorgThreshold=true), chain()]), + ]->Supervisor.isRunAtHead(~workerCount=2), + [ + snapshot([chain(~isReadyForReorgThreshold=true)]), + snapshot([chain(~isReadyForReorgThreshold=true)]), + ]->Supervisor.isRunAtHead(~workerCount=2), + // A chain resumed already realtime never reaches the head again, and one + // that processed to its end block never will: both have arrived as far as + // the run is concerned, and waiting on either would never end. + [snapshot([chain(~isReady=true)])]->Supervisor.isRunAtHead(~workerCount=1), + [ + snapshot([chain(~endBlock=Some(200), ~progressBlockNumber=200)]), + ]->Supervisor.isRunAtHead(~workerCount=1), + ]).toStrictEqual([false, false, true, true, true]) + }) +}) diff --git a/packages/envio/src/BatchProcessing.res b/packages/envio/src/BatchProcessing.res index 75bccf9ae..4290a8e74 100644 --- a/packages/envio/src/BatchProcessing.res +++ b/packages/envio/src/BatchProcessing.res @@ -76,11 +76,7 @@ and processNextBatch = async (state: IndexerState.t, ~scheduleFetch): unit => { let isBelowReorgThreshold = !isInReorgThresholdBeforeUpdate && (state->IndexerState.config).shouldRollbackOnReorg let shouldEnterReorgThreshold = - isBelowReorgThreshold && - state - ->IndexerState.chainStates - ->Dict.valuesToArray - ->Array.every(cs => cs->ChainState.isReadyToEnterReorgThresholdAfterBatch(~batch)) + isBelowReorgThreshold && state->IndexerState.isReadyToEnterReorgThreshold(~batch) if shouldEnterReorgThreshold { IndexerState.enterReorgThreshold(state) @@ -111,7 +107,7 @@ and processNextBatch = async (state: IndexerState.t, ~scheduleFetch): unit => { // When resuming from persisted state, all events may already be processed. if EventProcessing.allChainsEventsProcessedToEndblock(state->IndexerState.chainStates) { Logging.info("All chains are caught up to end blocks.") - if !(state->IndexerState.keepProcessAlive) { + if !(state->IndexerState.keepProcessAlive) && !(state->IndexerState.isHoldingRealtime) { await ExitOnCaughtUp.run(state) } } @@ -186,7 +182,11 @@ and processNextBatch = async (state: IndexerState.t, ~scheduleFetch): unit => { Logging.info("All chains are caught up to end blocks.") } - if allCaughtUp && !(state->IndexerState.keepProcessAlive) { + if ( + allCaughtUp && + !(state->IndexerState.keepProcessAlive) && + !(state->IndexerState.isHoldingRealtime) + ) { await ExitOnCaughtUp.run(state) } else if ( // In auto-exit mode, error if all chains reached head with no events found diff --git a/packages/envio/src/Bin.res b/packages/envio/src/Bin.res index 27d76279e..06ff54de0 100644 --- a/packages/envio/src/Bin.res +++ b/packages/envio/src/Bin.res @@ -53,10 +53,8 @@ let run = async args => { try { if Worker.isEnabled { Worker.bindToSupervisor() - // A worker is handed the config its supervisor already parsed, narrowed to - // the chains it drives, so the two can't disagree about what is indexed. - // Its working directory and environment came with the fork. - Config.prime(await Worker.awaitInit()) + // Its working directory, its environment and the chains it drives all came + // with the fork, so a worker starts the same way every other process does. await Main.start() } else { switch (await Core.runCli(args))->Null.toOption { diff --git a/packages/envio/src/ChainState.res b/packages/envio/src/ChainState.res index 11f3b6e2d..0768b4923 100644 --- a/packages/envio/src/ChainState.res +++ b/packages/envio/src/ChainState.res @@ -994,6 +994,9 @@ let toMetrics = (cs: t): Metrics.chainMetrics => { ->AddressStore.contractCounts ->Array.map(({contractName, count}) => (contractName, count)), isReady: cs->isReady, + isReadyForReorgThreshold: cs.fetchState->FetchState.isReadyToEnterReorgThreshold( + ~tolerance=cs.reorgThresholdReadyTolerance, + ), sourceBlockNumber: cs.fetchState.knownHeight, progressBlockNumber: cs.committedProgressBlockNumber, progressLatencyMs: cs.progressLatencyMs, diff --git a/packages/envio/src/Config.res b/packages/envio/src/Config.res index 185963836..67071ce17 100644 --- a/packages/envio/src/Config.res +++ b/packages/envio/src/Config.res @@ -1237,6 +1237,21 @@ let prime = (json: JSON.t): unit => { cached := None } +// Narrows a public config to the chains one process drives. The supervisor +// plans the split; each worker applies the plan to the config it parsed itself. +let withIsolatedChains = (json: JSON.t, ~chainIds) => + switch json->JSON.Decode.object { + | Some(fields) => { + let narrowed = fields->Dict.copy + narrowed->Dict.set( + "isolatedChains", + chainIds->S.reverseConvertToJsonOrThrow(S.array(ChainId.schema)), + ) + JSON.Object(narrowed) + } + | None => JsError.throwWithMessage("Invalid indexer config: not an object") + } + let getPublicConfigJson = () => switch primedJson.contents { | Some(json) => json diff --git a/packages/envio/src/CrossChainState.res b/packages/envio/src/CrossChainState.res index 3d82f0f6d..0eaad2b82 100644 --- a/packages/envio/src/CrossChainState.res +++ b/packages/envio/src/CrossChainState.res @@ -17,6 +17,11 @@ type t = { mutable isCaughtUp: bool, // Indexer-wide fetch buffer pool (item count), shared across all chains. targetBufferSize: int, + // Set on a process driving part of a split run: the chains it drives may be + // at the head while chains in another process are still backfilling, and an + // indexer switches to realtime as a whole or not at all. Cleared by the + // supervisor once every chain in the run has arrived. + mutable holdRealtime: bool, } // The whole-indexer fetch buffer pool, independent of chain count. @@ -26,16 +31,28 @@ let calculateTargetBufferSize = () => | None => 100_000 } -let make = (~chainStates, ~isRealtime, ~targetBufferSize=calculateTargetBufferSize()): t => { +let make = ( + ~chainStates, + ~isRealtime, + ~targetBufferSize=calculateTargetBufferSize(), + ~holdRealtime=false, +): t => { { chainStates, chainIds: chainStates->Dict.valuesToArray->Array.map(cs => (cs->ChainState.chainConfig).id), isRealtime, isCaughtUp: isRealtime, targetBufferSize, + holdRealtime, } } +// The supervisor's go-ahead: every chain in the run has reached the head, so +// this process may make the transitions it has been holding back. +let releaseRealtime = (crossChainState: t) => crossChainState.holdRealtime = false + +let isHoldingRealtime = (crossChainState: t) => crossChainState.holdRealtime + // Resolve a chain's state by id. The id always comes from `chainIds`, which is // derived from `chainStates`, so the entry is guaranteed present. let getChainState = (crossChainState: t, chainId) => @@ -122,6 +139,16 @@ let createBatch = ( // Enter the reorg threshold: shrink each chain's buffer by its configured // blockLag and flip the flag. +// Whether every chain this process drives has buffered close enough to the head +// to enter the threshold together — and, in a split run, whether the rest of the +// run has too. Chains enter it as one indexer, so one chain still backfilling +// holds the others back whatever process it runs in. +let isReadyToEnterReorgThreshold = (crossChainState: t, ~batch) => + !crossChainState.holdRealtime && + crossChainState.chainStates + ->Dict.valuesToArray + ->Array.every(cs => cs->ChainState.isReadyToEnterReorgThresholdAfterBatch(~batch)) + let enterReorgThreshold = (crossChainState: t) => { Logging.info("Reorg threshold reached") @@ -150,7 +177,10 @@ let applyBatchProgress = (crossChainState: t, ~batch: Batch.t, ~blockTimestampNa } crossChainState.isCaughtUp = - crossChainState.isCaughtUp || (crossChainState->nextItemIsNone && everyChainCaughtUp.contents) + crossChainState.isCaughtUp || + (!crossChainState.holdRealtime && + crossChainState->nextItemIsNone && + everyChainCaughtUp.contents) } // Every chain has buffered up to its head (or endblock) with nothing @@ -175,7 +205,7 @@ let isSettledAtHead = (crossChainState: t) => { // Enter the FinalizingIndexes phase without a batch, for the resume above. let markCaughtUpIfSettled = (crossChainState: t) => - if crossChainState->isSettledAtHead { + if !crossChainState.holdRealtime && crossChainState->isSettledAtHead { crossChainState.isCaughtUp = true } @@ -196,7 +226,12 @@ let markCaughtUpOnResume = (crossChainState: t) => { } } - if everyChainCaughtUp.contents && !crossChainState.isRealtime && crossChainState->nextItemIsNone { + if ( + !crossChainState.holdRealtime && + everyChainCaughtUp.contents && + !crossChainState.isRealtime && + crossChainState->nextItemIsNone + ) { crossChainState.isCaughtUp = true } } diff --git a/packages/envio/src/CrossChainState.resi b/packages/envio/src/CrossChainState.resi index 3f4b8365c..db2e40756 100644 --- a/packages/envio/src/CrossChainState.resi +++ b/packages/envio/src/CrossChainState.resi @@ -5,7 +5,12 @@ type t let calculateTargetBufferSize: unit => int -let make: (~chainStates: dict, ~isRealtime: bool, ~targetBufferSize: int=?) => t +let make: ( + ~chainStates: dict, + ~isRealtime: bool, + ~targetBufferSize: int=?, + ~holdRealtime: bool=?, +) => t // Accessors. let chainStates: t => dict @@ -25,6 +30,9 @@ let getSafeCheckpointIdByChain: ( ) => array<(ChainId.t, option)> // Cross-chain transitions. +let releaseRealtime: t => unit +let isHoldingRealtime: t => bool +let isReadyToEnterReorgThreshold: (t, ~batch: Batch.t) => bool let createBatch: (t, ~config: Config.t, ~frontier: Frontier.t, ~batchSizeTarget: int) => Batch.t let enterReorgThreshold: t => unit let applyBatchProgress: (t, ~batch: Batch.t, ~blockTimestampName: string) => unit diff --git a/packages/envio/src/IndexerLoop.res b/packages/envio/src/IndexerLoop.res index 9e979fa6c..a2e56d489 100644 --- a/packages/envio/src/IndexerLoop.res +++ b/packages/envio/src/IndexerLoop.res @@ -45,6 +45,8 @@ let start = (state: IndexerState.t) => { launch(state, () => FinalizeBackfill.repairSchemaIndexes(state)) } + state->IndexerState.bindScheduleProcessing(scheduleProcessing) + scheduleFetch() scheduleProcessing() } diff --git a/packages/envio/src/IndexerState.res b/packages/envio/src/IndexerState.res index b43c47320..22c49b931 100644 --- a/packages/envio/src/IndexerState.res +++ b/packages/envio/src/IndexerState.res @@ -115,6 +115,10 @@ type t = { // waitForNewBlock waiter is bound to the old, pre-realtime source). A fetch // response or waiter carrying an older epoch than this is discarded. mutable epoch: int, + // The loop's one door in from outside it: IndexerLoop owns scheduling and + // wires this when it starts, so an event the loop can't see for itself can + // still make it re-evaluate. A no-op before then. + mutable scheduleProcessing: unit => unit, // None off the simulate path. simulateDeadInputTracker: option, // --- Metric counters, rendered by Metrics at scrape time. --- @@ -141,6 +145,7 @@ let make = ( ~chainStates: dict, ~isRealtime: bool, ~targetBufferSize=CrossChainState.calculateTargetBufferSize(), + ~holdRealtime=false, ~committedFrontier=Frontier.empty(), ~isDevelopmentMode=false, ~shouldUseTui=false, @@ -180,11 +185,17 @@ let make = ( chainMetaDirty: false, chainMetaThrottler, isProcessing: false, - crossChainState: CrossChainState.make(~chainStates, ~isRealtime, ~targetBufferSize), + crossChainState: CrossChainState.make( + ~chainStates, + ~isRealtime, + ~targetBufferSize, + ~holdRealtime, + ), indexerStartTime: Date.make(), indexerStartTimeRef: Performance.now(), rollbackState: NoRollback, lastPrunedAtMillis: Dict.make(), + scheduleProcessing: () => (), loadManager: LoadManager.make(), keepProcessAlive: isDevelopmentMode || shouldUseTui, exitAfterFirstEventBlock, @@ -227,6 +238,9 @@ let makeFromDbState = ( ~exitAfterFirstEventBlock=false, ~reducedPollingInterval=?, ~targetBufferSize=CrossChainState.calculateTargetBufferSize(), + // A process driving part of a split run waits for its supervisor before + // entering the reorg threshold or switching to realtime. + ~holdRealtime=false, ~onError, ~onExit=?, ) => { @@ -274,6 +288,7 @@ let makeFromDbState = ( ~chainStates, ~isRealtime, ~targetBufferSize, + ~holdRealtime, ~committedFrontier=initialState.checkpointFrontier, ~isDevelopmentMode, ~shouldUseTui, @@ -503,6 +518,23 @@ let isFinalizingIndexes = (state: t) => let markCaughtUpIfSettled = (state: t) => state.crossChainState->CrossChainState.markCaughtUpIfSettled +let isReadyToEnterReorgThreshold = (state: t, ~batch) => + state.crossChainState->CrossChainState.isReadyToEnterReorgThreshold(~batch) + +let bindScheduleProcessing = (state: t, scheduleProcessing) => + state.scheduleProcessing = scheduleProcessing + +// A process still waiting on its supervisor owes the schema the indexes its +// chains deferred, so reaching every end block doesn't make it done. +let isHoldingRealtime = (state: t) => state.crossChainState->CrossChainState.isHoldingRealtime + +let releaseRealtime = (state: t) => { + state.crossChainState->CrossChainState.releaseRealtime + // Every chain is parked at the head with no batch coming, so nothing would + // notice the hold is gone without a pass through processing. + state.scheduleProcessing() +} + let markReady = (state: t, ~readyAt) => state.crossChainState->CrossChainState.markReady(~readyAt) let rollbackState = (state: t) => state.rollbackState diff --git a/packages/envio/src/IndexerState.resi b/packages/envio/src/IndexerState.resi index 99e4a3598..75cd69338 100644 --- a/packages/envio/src/IndexerState.resi +++ b/packages/envio/src/IndexerState.resi @@ -16,6 +16,7 @@ let make: ( ~chainStates: dict, ~isRealtime: bool, ~targetBufferSize: int=?, + ~holdRealtime: bool=?, ~committedFrontier: Frontier.t=?, ~isDevelopmentMode: bool=?, ~shouldUseTui: bool=?, @@ -34,6 +35,7 @@ let makeFromDbState: ( ~exitAfterFirstEventBlock: bool=?, ~reducedPollingInterval: int=?, ~targetBufferSize: int=?, + ~holdRealtime: bool=?, ~onError: ErrorHandling.t => unit, ~onExit: unit => unit=?, ) => t @@ -93,6 +95,12 @@ let shouldSaveHistory: t => dict let isRealtime: t => bool let isFinalizingIndexes: t => bool let markCaughtUpIfSettled: t => unit +let isReadyToEnterReorgThreshold: (t, ~batch: Batch.t) => bool +// Wires the loop's way back in. IndexerLoop calls this as it starts. +let bindScheduleProcessing: (t, unit => unit) => unit +let isHoldingRealtime: t => bool +// The supervisor's go-ahead for a process driving part of a split run. +let releaseRealtime: t => unit let markReady: (t, ~readyAt: Date.t) => unit let rollbackState: t => rollbackState let indexerStartTime: t => Date.t diff --git a/packages/envio/src/Main.res b/packages/envio/src/Main.res index 5f775311e..9b353ee3c 100644 --- a/packages/envio/src/Main.res +++ b/packages/envio/src/Main.res @@ -534,12 +534,16 @@ exception FatalError(exn) ~isDevelopmentMode, ~shouldUseTui, ~exitAfterFirstEventBlock, + ~holdRealtime=Worker.config->Option.mapOr(false, worker => worker.holdRealtime), ~onError, ) if shouldUseTui { let _rerender = Tui.start(~config, ~getMetrics=() => state->IndexerState.toMetrics) } - Worker.startReporting(~getMetrics=() => state->IndexerState.toMetrics) + Worker.bindRun( + ~getMetrics=() => state->IndexerState.toMetrics, + ~onReleaseRealtime=() => state->IndexerState.releaseRealtime, + ) setIndexerState(state) state->IndexerLoop.start await runUntilFatalError @@ -557,6 +561,13 @@ let start = async ( ~exitAfterFirstEventBlock=false, ~patchConfig: option<(Config.t, HandlerRegister.registrationsByChainId) => Config.t>=?, ) => { + // A worker parses the same config its supervisor did and narrows it to the + // chains it was handed, rather than being told what to index: the storage it + // resumes refuses a config that disagrees with the one the run was created + // from, which is a stronger guarantee than a handover could give. + Worker.config->Option.forEach(({chainIds}) => + Config.prime(Config.getPublicConfigJson()->Config.withIsolatedChains(~chainIds)) + ) let config = Config.load() switch isTest ? None : Supervisor.planForRun(~config) { | Some(workers) => await Supervisor.run(~config, ~workers, ~reset) diff --git a/packages/envio/src/Metrics.res b/packages/envio/src/Metrics.res index b12ac8294..ab0f62c2a 100644 --- a/packages/envio/src/Metrics.res +++ b/packages/envio/src/Metrics.res @@ -15,6 +15,10 @@ type chainMetrics = { // Per-contract registration counts, in the chain's contract-id order. addressesByContract: array<(string, int)>, isReady: bool, + // The chain has buffered to the (lagged) head or its end block with nothing + // processable left — what a supervisor reads to decide that every chain in a + // split run has arrived and the run may go realtime as one. + isReadyForReorgThreshold: bool, // Raw source height, unlike knownHeight which is clamped to endBlock. sourceBlockNumber: int, // Raw committed progress (may be -1), unlike the optional latestProcessedBlock. diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index cb6115e22..2848bd25f 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -84,19 +84,6 @@ let logFilePath = (~workerIndex, ~path=Env.logFilePath) => { } } -let configForWorker = (configJson: JSON.t, ~worker) => - switch configJson->JSON.Decode.object { - | Some(fields) => { - let narrowed = fields->Dict.copy - narrowed->Dict.set( - "isolatedChains", - worker.chainIds->S.reverseConvertToJsonOrThrow(S.array(ChainId.schema)), - ) - JSON.Object(narrowed) - } - | None => JsError.throwWithMessage("Invalid indexer config: not an object") - } - // A pipe hands over whatever has been flushed, so a chunk boundary falls // wherever the OS put it: the tail of a chunk is a line only once the chunk // that ends it arrives. Reading pairs with a flush, since a process that dies @@ -127,7 +114,10 @@ let readLines = (~onLine) => { let fork = ( worker: worker, ~workerIndex, - ~configJson, + // Whether this worker waits for the run before going realtime. False when + // every chain resumed already caught up: there is nothing left to wait for, + // and a barrier nobody can open would hold the run forever. + ~holdRealtime, // The entry this process was itself started from, so a worker is the same // program as its supervisor however the package was installed. ~entryPath=NodeJs.Process.argv->Array.getUnsafe(1), @@ -137,7 +127,12 @@ let fork = ( ~onOutput=Console.log, ) => { let env = NodeJs.Process.process.env->Dict.copy - env->Dict.set(Worker.envVar, "true") + env->Dict.set( + Worker.envVar, + {Worker.chainIds: worker.chainIds, holdRealtime}->S.reverseConvertToJsonStringOrThrow( + Worker.configSchema, + ), + ) // The worker's slice of the budget. Read when the worker's own Env module // loads, which is why it rides in the spawn environment rather than a message. env->Dict.set("ENVIO_PG_MAX_CONNECTIONS", worker.maxConnections->Int.toString) @@ -172,10 +167,6 @@ let fork = ( } ) } - child - ->NodeJs.ChildProcess.send(Worker.Init({config: configJson->configForWorker(~worker)})) - ->ignore - let running = {worker, child, snapshot: None, runtime: None, settled: false} child->NodeJs.ChildProcess.onMessage(message => switch message { @@ -278,6 +269,24 @@ let awaitExit = async (group): outcome => { group.stopping ? Stopped : Finished } +// How often the supervisor asks whether the run may go realtime. Matches the +// rate its workers report at: nothing changes in between. +%%private(let releaseCheckIntervalMillis = 500) + +// Whether a run holding its workers back may let them go: every worker has +// reported, and every chain any of them drives has reached the head. A chain +// resumed already realtime, or one that processed to its end block, has arrived +// as far as the run is concerned — waiting on either would never end. +let isRunAtHead = (snapshots: array, ~workerCount) => + snapshots->Array.length === workerCount && + snapshots->Array.every(snapshot => + snapshot.chains->Utils.Array.notEmpty && + snapshot.chains->Array.every( + chain => + chain.isReadyForReorgThreshold || chain.isReady || chain->Metrics.hasProcessedToEndblock, + ) + ) + // Runs the group: creates the schema for every chain, forks a worker per plan // entry, and serves the run's metrics, console and display from what they // report. Returns once every worker has exited; throws if any of them failed. @@ -308,14 +317,18 @@ let run = async (~config: Config.t, ~workers: array, ~reset) => { ->Int.toString} processes, from a budget of ${Env.Db.maxConnections->Int.toString} database connections.`, ) - // The config as the CLI handed it over, narrowed per worker on the way out. - let configJson = Config.getPublicConfigJson() // Decided before the first fork: it is what makes a worker's output the // supervisor's to print. let shouldUseTui = Tui.shouldUse() + // A run that resumed with every chain already caught up owes nobody a wait: + // its workers start realtime and there is no barrier to open. + let holdRealtime = + (persistence->Persistence.getInitializedState).chains->Array.some(chain => + chain.timestampCaughtUpToHeadOrEndblock->Option.isNone + ) let group = { running: workers->Array.mapWithIndex((worker, workerIndex) => - worker->fork(~workerIndex, ~configJson, ~pipeOutput=shouldUseTui) + worker->fork(~workerIndex, ~holdRealtime, ~pipeOutput=shouldUseTui) ), stopping: false, } @@ -350,6 +363,30 @@ let run = async (~config: Config.t, ~workers: array, ~reset) => { ~onSyncCache=() => syncCache(~dump=() => dumpCache(~config)), ) + // Chains enter the reorg threshold and go realtime as one indexer, which in a + // split run only the supervisor can see. Every worker is held until the last + // one arrives, then released together, so the run switches over exactly as an + // unsplit one does. + let releaseCheck = ref(None) + let stopReleaseCheck = () => { + releaseCheck.contents->Option.forEach(clearInterval) + releaseCheck := None + } + if holdRealtime { + releaseCheck := + Some( + setInterval(() => + if reported()->isRunAtHead(~workerCount=group.running->Array.length) { + stopReleaseCheck() + group.running->Array.forEach(r => + r.child->NodeJs.ChildProcess.send(Worker.ReleaseRealtime)->ignore + ) + Logging.info("Every chain has reached the head. Switching the run to realtime.") + } + , releaseCheckIntervalMillis), + ) + } + if shouldUseTui { let _rerender = Tui.start(~config, ~getMetrics=() => reported()->merge) } @@ -364,7 +401,12 @@ let run = async (~config: Config.t, ~workers: array, ~reset) => { // last worker is gone, so the group's end has to end the process. A display // is the exception, as it is for a single process: it keeps the final state // on screen until the terminal closes it. - switch await group->awaitExit { + let outcome = await group->awaitExit + // Nothing left to release, and a display keeps this process alive long enough + // for the check to reach children that are gone. + stopReleaseCheck() + + switch outcome { | Stopped => NodeJs.process->NodeJs.exitWithCode(Success) | Finished if !shouldUseTui => Logging.info("Exiting with success") diff --git a/packages/envio/src/Worker.res b/packages/envio/src/Worker.res index bcfb4c377..1d28ad5d9 100644 --- a/packages/envio/src/Worker.res +++ b/packages/envio/src/Worker.res @@ -8,20 +8,40 @@ // path it takes today. let envVar = "ENVIO_INTERNAL_WORKER" +// What the supervisor decided about this worker, handed over in the spawn +// environment rather than over the channel: it is settled before the process +// starts, and the worker needs it before it can load its own config. +type config = { + chainIds: array, + // The chains this worker drives may reach the head while chains in another + // process are still backfilling, and an indexer goes realtime as a whole or + // not at all. Cleared by the supervisor's `ReleaseRealtime`. + holdRealtime: bool, +} + +let configSchema = S.object((s): config => { + chainIds: s.field("chainIds", S.array(ChainId.schema)), + holdRealtime: s.fieldOr("holdRealtime", S.bool, false), +}) + let detect = (~env: dict, ~hasChannel) => - hasChannel && env->Dict.get(envVar) === Some("true") + switch (hasChannel, env->Dict.get(envVar)) { + | (true, Some(json)) => Some(json->S.parseJsonStringOrThrow(configSchema)) + | _ => None + } -let isEnabled = detect( +let config = detect( ~env=NodeJs.Process.process.env, ~hasChannel=NodeJs.Process.channel->Nullable.toOption->Option.isSome, ) +let isEnabled = config->Option.isSome + @tag("kind") type parentMessage = - // The config the supervisor parsed, narrowed to this worker's chains. Sent - // instead of re-derived so a worker and its supervisor can never disagree - // about what is being indexed. - | @as("init") Init({config: JSON.t}) + // Every chain in the run has reached the head, so this worker may enter the + // reorg threshold and switch to realtime with the rest of them. + | @as("release-realtime") ReleaseRealtime @tag("kind") type workerMessage = @@ -48,23 +68,19 @@ let bindToSupervisor = () => { %%private(let send = (message: workerMessage) => NodeJs.Process.sendToParent(message)->ignore) // Reports this process's chains and its own runtime for as long as it runs, so -// the supervisor can merge every worker's into the one snapshot the run serves. +// the supervisor can merge every worker's into the one snapshot the run serves, +// and listens for the one decision the supervisor makes on the run's behalf. // Does nothing in a process nobody forked. -let startReporting = (~getMetrics: unit => Metrics.t) => +let bindRun = (~getMetrics: unit => Metrics.t, ~onReleaseRealtime: unit => unit) => if isEnabled { Metrics.startRuntimeCollectors() let _intervalId = setInterval( () => send(Snapshot({metrics: getMetrics(), runtime: Metrics.sampleRuntime()})), snapshotIntervalMillis, ) - } - -// Resolves with the init payload, the one message a supervisor sends its worker. -let awaitInit = (): promise => - Promise.make((resolve, _) => - NodeJs.Process.onceMessage((message: parentMessage) => + NodeJs.Process.onMessage((message: parentMessage) => switch message { - | Init({config}) => resolve(config) + | ReleaseRealtime => onReleaseRealtime() } ) - ) + } diff --git a/packages/envio/src/bindings/NodeJs.res b/packages/envio/src/bindings/NodeJs.res index 3390a8dcb..e644b52b8 100644 --- a/packages/envio/src/bindings/NodeJs.res +++ b/packages/envio/src/bindings/NodeJs.res @@ -71,8 +71,6 @@ module Process = { @val @scope("process") external sendToParent: 'msg => bool = "send" @val @scope("process") external onMessage: (@as("message") _, 'msg => unit) => unit = "on" - @val @scope("process") - external onceMessage: (@as("message") _, 'msg => unit) => unit = "once" @val @scope("process") external onSignal: (string, unit => unit) => unit = "on" // Present only in a process forked with an IPC channel. @val @scope("process") external channel: Nullable.t = "channel" From 00e5b7efec0f305ce09b67be322fff9eec7b28a6 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 18 Sep 2026 12:35:42 +0000 Subject: [PATCH 26/61] Say which variable a worker couldn't be started from `ENVIO_INTERNAL_WORKER` is read as the module loads, before anything that could catch a bare schema error and say where it came from. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- .../envio-tests/test/lib_tests/Supervisor_test.res | 9 +++++++++ packages/envio/src/Worker.res | 11 ++++++++++- 2 files changed, 19 insertions(+), 1 deletion(-) diff --git a/packages/envio-tests/test/lib_tests/Supervisor_test.res b/packages/envio-tests/test/lib_tests/Supervisor_test.res index 7e2308ef4..0fcdc593d 100644 --- a/packages/envio-tests/test/lib_tests/Supervisor_test.res +++ b/packages/envio-tests/test/lib_tests/Supervisor_test.res @@ -210,6 +210,15 @@ describe("Worker.detect", () => { ]) }) + // Thrown as the module loads, before anything that could say where a bare + // schema error came from. + it("Names the variable when its value isn't a worker config", t => { + t->Vitest.toThrowErrorEqual( + () => Worker.detect(~env=Dict.fromArray([(Worker.envVar, "137")]), ~hasChannel=true), + `Invalid ENVIO_INTERNAL_WORKER: Failed parsing at root. Reason: Expected { chainIds: array; holdRealtime: boolean | undefined; }, received 137. It is set by an indexer supervisor for the processes it forks, and isn't meant to be set by hand.`, + ) + }) + // The supervisor decides whether a run waits; a worker forked before that // decision existed reads as one that doesn't. it("Takes a config without the hold as one that doesn't wait", t => { diff --git a/packages/envio/src/Worker.res b/packages/envio/src/Worker.res index 1d28ad5d9..d04504545 100644 --- a/packages/envio/src/Worker.res +++ b/packages/envio/src/Worker.res @@ -24,9 +24,18 @@ let configSchema = S.object((s): config => { holdRealtime: s.fieldOr("holdRealtime", S.bool, false), }) +// Read as this module loads, which is before anything that could catch a bare +// schema error and say where it came from. let detect = (~env: dict, ~hasChannel) => switch (hasChannel, env->Dict.get(envVar)) { - | (true, Some(json)) => Some(json->S.parseJsonStringOrThrow(configSchema)) + | (true, Some(json)) => + switch json->S.parseJsonStringOrThrow(configSchema) { + | config => Some(config) + | exception S.Raised(error) => + JsError.throwWithMessage( + `Invalid ${envVar}: ${error->S.Error.message}. It is set by an indexer supervisor for the processes it forks, and isn't meant to be set by hand.`, + ) + } | _ => None } From 7365d0c035de5e09fcf931a85e6290b3b31d8298 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 18 Sep 2026 13:06:39 +0000 Subject: [PATCH 27/61] Check a split run stamps its chains, not just that it exits The suite asserted the run's exit code and its rows, both of which a worker that exited at its end block owing the schema its deferred indexes would still satisfy. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- packages/e2e-tests/src/e2e/split-run.test.ts | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/packages/e2e-tests/src/e2e/split-run.test.ts b/packages/e2e-tests/src/e2e/split-run.test.ts index 9ef4c6aaa..48ddcf2df 100644 --- a/packages/e2e-tests/src/e2e/split-run.test.ts +++ b/packages/e2e-tests/src/e2e/split-run.test.ts @@ -97,12 +97,22 @@ describe.skipIf(!reachable)("E2E: a split run is one indexer", () => { rowsPerChain: await pgRows( `SELECT "chain_id", COUNT(*) > 0 FROM "${PG_SCHEMA}"."Transfer" GROUP BY "chain_id" ORDER BY "chain_id"` ), + // Reaching an end block doesn't make a worker done: it owes the schema + // the indexes its chains deferred, and until the run goes realtime it + // has no leave to commit them. + readyPerChain: await pgRows( + `SELECT "id"::text, "ready_at" IS NOT NULL FROM "${PG_SCHEMA}"."envio_chains" ORDER BY "id"` + ), }).toEqual({ exitCode: 0, rowsPerChain: [ [1, true], [8453, true], ], + readyPerChain: [ + ["1", true], + ["8453", true], + ], }); } finally { await stopIfRunning(indexer); From b06b7f332122980a016ae7f1845c541a48895b60 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 18 Sep 2026 13:28:11 +0000 Subject: [PATCH 28/61] Hold the transition, not the record that it is owed MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Gating `isCaughtUp` stopped a resumed process ever recording that it owed the schema its deferred indexes. `markCaughtUpOnResume` decides that from persisted state before any source request precisely because no later reading can: once the head has moved on, a chain committed at what was the head looks behind again. A supervised worker resuming such a chain could therefore never report itself arrived, and its run would wait for good. The hold belongs on the transition instead — the finalization that commits `ready_at` and switches the indexer to realtime — leaving the record of what is owed to be made as it always was. What a worker reports is then its own conclusion rather than a reading its supervisor reassembles, which is also the only form that carries the resumed case. A chain that caught up never un-catches up, so a metadata write staged before the stamp can no longer clear it on its way to the database. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- .../test/helpers/IndexerRunner.res | 5 +- .../test/helpers/TestChainMetrics.res | 1 + .../test/lib_tests/ChainMetaReadyAt_test.res | 72 +++++++++++++++++++ .../test/lib_tests/Metrics_test.res | 5 +- .../test/lib_tests/Supervisor_test.res | 29 +++----- packages/envio/src/BatchProcessing.res | 4 +- packages/envio/src/ChainState.res | 3 - packages/envio/src/CrossChainState.res | 25 ++++--- packages/envio/src/CrossChainState.resi | 1 + packages/envio/src/IndexerState.res | 8 +++ packages/envio/src/IndexerState.resi | 2 + packages/envio/src/Metrics.res | 11 +-- packages/envio/src/Supervisor.res | 15 ++-- packages/envio/src/db/InternalTable.res | 9 ++- 14 files changed, 137 insertions(+), 53 deletions(-) create mode 100644 packages/envio-tests/test/lib_tests/ChainMetaReadyAt_test.res diff --git a/packages/envio-tests/test/helpers/IndexerRunner.res b/packages/envio-tests/test/helpers/IndexerRunner.res index 093be6c74..2503618e1 100644 --- a/packages/envio-tests/test/helpers/IndexerRunner.res +++ b/packages/envio-tests/test/helpers/IndexerRunner.res @@ -175,6 +175,7 @@ let run = async ( ) state->IndexerLoop.start + // Persist before stopping, else a resumed indexer loses uncommitted state, // then let any in-flight batch or write settle so nothing from this run // lands on the database afterwards. @@ -289,7 +290,7 @@ let run = async ( // phase is over. The idle fallback below still bounds the wait. if ( before < state->IndexerState.processedBatchesCount && - !(state->IndexerState.isFinalizingIndexes) + !(state->IndexerState.shouldFinalizeIndexes) ) { () } else if isIdle && idleChecks.contents >= 5 { @@ -327,7 +328,7 @@ let run = async ( settled := if ( !(state->IndexerState.isProcessing) && state->IndexerState.writeFiber->Option.isNone && - !(state->IndexerState.isFinalizingIndexes) && + !(state->IndexerState.shouldFinalizeIndexes) && Frontier.equals(state->IndexerState.committedFrontier, state->IndexerState.processedFrontier) ) { settled.contents + 1 diff --git a/packages/envio-tests/test/helpers/TestChainMetrics.res b/packages/envio-tests/test/helpers/TestChainMetrics.res index 394a0e9c1..b58b19ce1 100644 --- a/packages/envio-tests/test/helpers/TestChainMetrics.res +++ b/packages/envio-tests/test/helpers/TestChainMetrics.res @@ -72,6 +72,7 @@ let emptySnapshot: Metrics.t = { elapsedSeconds: 0., targetBufferSize: 0, isInReorgThreshold: false, + hasArrivedAtHead: false, rollbackEnabled: false, maxBatchSize: 0, preloadSeconds: 0., diff --git a/packages/envio-tests/test/lib_tests/ChainMetaReadyAt_test.res b/packages/envio-tests/test/lib_tests/ChainMetaReadyAt_test.res new file mode 100644 index 000000000..464316993 --- /dev/null +++ b/packages/envio-tests/test/lib_tests/ChainMetaReadyAt_test.res @@ -0,0 +1,72 @@ +open Vitest + +// `ready_at` is committed by the finalization and carried by every chain +// metadata write after it. The two race: metadata is written on a throttle of +// its own, outside the batch the finalization flushes, so a snapshot taken +// before the stamp can reach the database after it. +let sql = PgStorage.makeClient() +let pgSchema = TestPgSchema.make() +let config = TestConfig.make() + +Async.afterAll(async () => { + await sql->TestPgSchema.drop(~pgSchema) + await sql->Postgres.endSql +}) + +let storage = PgStorage.make( + ~sql, + ~pgHost=Env.Db.host, + ~pgSchema, + ~pgPort=Env.Db.port, + ~pgUser=Env.Db.user, + ~pgDatabase=Env.Db.database, + ~pgPassword=Env.Db.password, + ~isHasuraEnabled=false, + ~ecosystem=Evm, +) + +let readyAt = async () => { + let rows: array<{"ready_at": Null.t}> = await sql->Postgres.unsafe( + `SELECT "ready_at" FROM "${pgSchema}"."envio_chains" ORDER BY "id";`, + ) + rows->Array.map(row => row["ready_at"]->Null.toOption->Option.isSome) +} + +describe("A chain metadata write", () => { + Async.it("Can't clear the ready timestamp the finalization committed", async t => { + let _ = await storage.initialize( + ~chainConfigs=config.chainMap->ChainMap.values, + ~contractMapping=config.contractMapping, + ~entities=config.userEntities, + ~enums=config.allEnums->Array.concat([ + EntityHistory.RowAction.config->Table.fromGenericEnumConfig, + ]), + ~envioInfo=JSON.Encode.object(Dict.make()), + ) + await storage.finalizeBackfill( + ~entities=config.userEntities, + ~chainIds=config.chainMap->ChainMap.keys, + ~readyAt=Date.make(), + ) + let stamped = await readyAt() + + // What a process staged before it was ready, landing after the stamp. + let stale = Dict.make() + config.chainMap + ->ChainMap.keys + ->Array.forEach(chainId => + stale->Dict.set( + chainId->ChainId.toString, + { + InternalTable.Chains.firstEventBlockNumber: Null.null, + latestFetchedBlockNumber: 10, + timestampCaughtUpToHeadOrEndblock: Null.null, + isHyperSync: false, + }, + ) + ) + let _ = await storage.setChainMeta(stale) + + t.expect((stamped, await readyAt())).toStrictEqual(([true], [true])) + }) +}) diff --git a/packages/envio-tests/test/lib_tests/Metrics_test.res b/packages/envio-tests/test/lib_tests/Metrics_test.res index 360bc6c94..a2cdde394 100644 --- a/packages/envio-tests/test/lib_tests/Metrics_test.res +++ b/packages/envio-tests/test/lib_tests/Metrics_test.res @@ -217,6 +217,7 @@ envio_info{version="${Utils.EnvioPackage.value.version}"} 1 elapsedSeconds: 123.456, targetBufferSize: 5000, isInReorgThreshold: true, + hasArrivedAtHead: true, rollbackEnabled: true, maxBatchSize: 5000, preloadSeconds: 12.3456, @@ -242,7 +243,6 @@ envio_info{version="${Utils.EnvioPackage.value.version}"} 1 numAddresses: 7, addressesByContract: [("Gravatar", 5), ("NftFactory", 2)], isReady: true, - isReadyForReorgThreshold: true, sourceBlockNumber: 305, progressBlockNumber: 200, progressLatencyMs: Some(1500), @@ -716,6 +716,7 @@ describe("Metrics.merge", () => { targetBufferSize: 50, maxBatchSize: 1000, isInReorgThreshold: true, + hasArrivedAtHead: true, rollbackEnabled: true, processingSeconds: 0.5, rollbackCount: 2, @@ -738,6 +739,8 @@ describe("Metrics.merge", () => { targetBufferSize: 150, maxBatchSize: 5000, isInReorgThreshold: true, + // One worker still backfilling speaks for the whole indexer. + hasArrivedAtHead: false, rollbackEnabled: true, processingSeconds: 2., rollbackCount: 3, diff --git a/packages/envio-tests/test/lib_tests/Supervisor_test.res b/packages/envio-tests/test/lib_tests/Supervisor_test.res index 0fcdc593d..ecd2af3ac 100644 --- a/packages/envio-tests/test/lib_tests/Supervisor_test.res +++ b/packages/envio-tests/test/lib_tests/Supervisor_test.res @@ -250,33 +250,24 @@ describe("Supervisor.syncCache", () => { }) describe("Supervisor.isRunAtHead", () => { - let chain = (~isReadyForReorgThreshold=false, ~isReady=false, ~endBlock=None, ~progressBlockNumber=50) => { - ...TestChainMetrics.make(~progressBlockNumber, ~firstEventBlockNumber=None, ~endBlock), - Metrics.isReadyForReorgThreshold, - isReady, + let snapshot = (~hasArrivedAtHead): Metrics.t => { + ...TestChainMetrics.emptySnapshot, + hasArrivedAtHead, } - let snapshot = (chains): Metrics.t => {...TestChainMetrics.emptySnapshot, chains} - it("Holds the run until every chain of every worker has arrived", t => { + it("Holds the run until every worker has arrived", t => { t.expect([ // A worker that hasn't reported yet drives chains nobody can see. Reading // the run as arrived here would release it on a partial view. - [snapshot([chain(~isReadyForReorgThreshold=true)])]->Supervisor.isRunAtHead(~workerCount=2), + [snapshot(~hasArrivedAtHead=true)]->Supervisor.isRunAtHead(~workerCount=2), [ - snapshot([chain(~isReadyForReorgThreshold=true)]), - snapshot([chain(~isReadyForReorgThreshold=true), chain()]), + snapshot(~hasArrivedAtHead=true), + snapshot(~hasArrivedAtHead=false), ]->Supervisor.isRunAtHead(~workerCount=2), [ - snapshot([chain(~isReadyForReorgThreshold=true)]), - snapshot([chain(~isReadyForReorgThreshold=true)]), + snapshot(~hasArrivedAtHead=true), + snapshot(~hasArrivedAtHead=true), ]->Supervisor.isRunAtHead(~workerCount=2), - // A chain resumed already realtime never reaches the head again, and one - // that processed to its end block never will: both have arrived as far as - // the run is concerned, and waiting on either would never end. - [snapshot([chain(~isReady=true)])]->Supervisor.isRunAtHead(~workerCount=1), - [ - snapshot([chain(~endBlock=Some(200), ~progressBlockNumber=200)]), - ]->Supervisor.isRunAtHead(~workerCount=1), - ]).toStrictEqual([false, false, true, true, true]) + ]).toStrictEqual([false, false, true]) }) }) diff --git a/packages/envio/src/BatchProcessing.res b/packages/envio/src/BatchProcessing.res index 4290a8e74..c77ba4156 100644 --- a/packages/envio/src/BatchProcessing.res +++ b/packages/envio/src/BatchProcessing.res @@ -91,7 +91,7 @@ and processNextBatch = async (state: IndexerState.t, ~scheduleFetch): unit => { // finalizing resumes exactly here: it still owes the schema its deferred // indexes, and no batch will ever come along to notice. state->IndexerState.markCaughtUpIfSettled - if state->IndexerState.isFinalizingIndexes { + if state->IndexerState.shouldFinalizeIndexes { await FinalizeBackfill.run(state) } @@ -162,7 +162,7 @@ and processNextBatch = async (state: IndexerState.t, ~scheduleFetch): unit => { // Backfilling → FinalizingIndexes → Ready. Awaiting here holds the // processing loop for the whole finalize, which is what pauses // processing while the indexes are built. - if state->IndexerState.isFinalizingIndexes { + if state->IndexerState.shouldFinalizeIndexes { await FinalizeBackfill.run(state) } diff --git a/packages/envio/src/ChainState.res b/packages/envio/src/ChainState.res index 6c1f29280..ca8317fea 100644 --- a/packages/envio/src/ChainState.res +++ b/packages/envio/src/ChainState.res @@ -994,9 +994,6 @@ let toMetrics = (cs: t): Metrics.chainMetrics => { ->AddressStore.contractCounts ->Array.map(({contractName, count}) => (contractName, count)), isReady: cs->isReady, - isReadyForReorgThreshold: cs.fetchState->FetchState.isReadyToEnterReorgThreshold( - ~tolerance=cs.reorgThresholdReadyTolerance, - ), sourceBlockNumber: cs.fetchState.knownHeight, progressBlockNumber: cs.committedProgressBlockNumber, progressLatencyMs: cs.progressLatencyMs, diff --git a/packages/envio/src/CrossChainState.res b/packages/envio/src/CrossChainState.res index 45b9d5a96..1ee37e326 100644 --- a/packages/envio/src/CrossChainState.res +++ b/packages/envio/src/CrossChainState.res @@ -53,6 +53,17 @@ let releaseRealtime = (crossChainState: t) => crossChainState.holdRealtime = fal let isHoldingRealtime = (crossChainState: t) => crossChainState.holdRealtime +// Whether this process has got as far as it can without the run's leave: its +// chains have caught up, or it resumed already realtime. What a supervisor +// reads to decide that a split run may go realtime as one. +// +// The process's own conclusion rather than a reading a supervisor reassembles: +// a chain committed at what was the head and resumed once the head had moved on +// has arrived, and no live reading of it can say so — which is the same reason +// `markCaughtUpOnResume` decides before any source request. +let hasArrivedAtHead = (crossChainState: t) => + crossChainState.isCaughtUp || crossChainState.isRealtime + // Resolve a chain's state by id. The id always comes from `chainIds`, which is // derived from `chainStates`, so the entry is guaranteed present. let getChainState = (crossChainState: t, chainId) => @@ -177,10 +188,7 @@ let applyBatchProgress = (crossChainState: t, ~batch: Batch.t, ~blockTimestampNa } crossChainState.isCaughtUp = - crossChainState.isCaughtUp || - (!crossChainState.holdRealtime && - crossChainState->nextItemIsNone && - everyChainCaughtUp.contents) + crossChainState.isCaughtUp || (crossChainState->nextItemIsNone && everyChainCaughtUp.contents) } // Every chain has buffered up to its head (or endblock) with nothing @@ -205,7 +213,7 @@ let isSettledAtHead = (crossChainState: t) => { // Enter the FinalizingIndexes phase without a batch, for the resume above. let markCaughtUpIfSettled = (crossChainState: t) => - if !crossChainState.holdRealtime && crossChainState->isSettledAtHead { + if crossChainState->isSettledAtHead { crossChainState.isCaughtUp = true } @@ -226,12 +234,7 @@ let markCaughtUpOnResume = (crossChainState: t) => { } } - if ( - !crossChainState.holdRealtime && - everyChainCaughtUp.contents && - !crossChainState.isRealtime && - crossChainState->nextItemIsNone - ) { + if everyChainCaughtUp.contents && !crossChainState.isRealtime && crossChainState->nextItemIsNone { crossChainState.isCaughtUp = true } } diff --git a/packages/envio/src/CrossChainState.resi b/packages/envio/src/CrossChainState.resi index db2e40756..a047f0bb5 100644 --- a/packages/envio/src/CrossChainState.resi +++ b/packages/envio/src/CrossChainState.resi @@ -32,6 +32,7 @@ let getSafeCheckpointIdByChain: ( // Cross-chain transitions. let releaseRealtime: t => unit let isHoldingRealtime: t => bool +let hasArrivedAtHead: t => bool let isReadyToEnterReorgThreshold: (t, ~batch: Batch.t) => bool let createBatch: (t, ~config: Config.t, ~frontier: Frontier.t, ~batchSizeTarget: int) => Batch.t let enterReorgThreshold: t => unit diff --git a/packages/envio/src/IndexerState.res b/packages/envio/src/IndexerState.res index 22c49b931..ca376ffbb 100644 --- a/packages/envio/src/IndexerState.res +++ b/packages/envio/src/IndexerState.res @@ -515,6 +515,11 @@ let isFinalizingIndexes = (state: t) => state.crossChainState->CrossChainState.isCaughtUp && !(state.crossChainState->CrossChainState.isRealtime) +// The FinalizingIndexes phase is the transition a held process waits on: it +// ends with `ready_at` committed and the indexer realtime. +let shouldFinalizeIndexes = (state: t) => + state->isFinalizingIndexes && !(state.crossChainState->CrossChainState.isHoldingRealtime) + let markCaughtUpIfSettled = (state: t) => state.crossChainState->CrossChainState.markCaughtUpIfSettled @@ -528,6 +533,8 @@ let bindScheduleProcessing = (state: t, scheduleProcessing) => // chains deferred, so reaching every end block doesn't make it done. let isHoldingRealtime = (state: t) => state.crossChainState->CrossChainState.isHoldingRealtime +let hasArrivedAtHead = (state: t) => state.crossChainState->CrossChainState.hasArrivedAtHead + let releaseRealtime = (state: t) => { state.crossChainState->CrossChainState.releaseRealtime // Every chain is parked at the head with no batch coming, so nothing would @@ -604,6 +611,7 @@ let toMetrics = (state: t): Metrics.t => { elapsedSeconds: state.indexerStartTimeRef->Performance.secondsSince, targetBufferSize: state.crossChainState->CrossChainState.targetBufferSize, isInReorgThreshold: state.crossChainState->CrossChainState.isInReorgThreshold, + hasArrivedAtHead: state.crossChainState->CrossChainState.hasArrivedAtHead, rollbackEnabled: state.config.shouldRollbackOnReorg, maxBatchSize: state.config.batchSize, preloadSeconds: state.preloadSeconds, diff --git a/packages/envio/src/IndexerState.resi b/packages/envio/src/IndexerState.resi index 75cd69338..e73ffafda 100644 --- a/packages/envio/src/IndexerState.resi +++ b/packages/envio/src/IndexerState.resi @@ -96,9 +96,11 @@ let isRealtime: t => bool let isFinalizingIndexes: t => bool let markCaughtUpIfSettled: t => unit let isReadyToEnterReorgThreshold: (t, ~batch: Batch.t) => bool +let shouldFinalizeIndexes: t => bool // Wires the loop's way back in. IndexerLoop calls this as it starts. let bindScheduleProcessing: (t, unit => unit) => unit let isHoldingRealtime: t => bool +let hasArrivedAtHead: t => bool // The supervisor's go-ahead for a process driving part of a split run. let releaseRealtime: t => unit let markReady: (t, ~readyAt: Date.t) => unit diff --git a/packages/envio/src/Metrics.res b/packages/envio/src/Metrics.res index ab0f62c2a..9baab8e22 100644 --- a/packages/envio/src/Metrics.res +++ b/packages/envio/src/Metrics.res @@ -15,10 +15,6 @@ type chainMetrics = { // Per-contract registration counts, in the chain's contract-id order. addressesByContract: array<(string, int)>, isReady: bool, - // The chain has buffered to the (lagged) head or its end block with nothing - // processable left — what a supervisor reads to decide that every chain in a - // split run has arrived and the run may go realtime as one. - isReadyForReorgThreshold: bool, // Raw source height, unlike knownHeight which is clamped to endBlock. sourceBlockNumber: int, // Raw committed progress (may be -1), unlike the optional latestProcessedBlock. @@ -133,6 +129,9 @@ type t = { elapsedSeconds: float, targetBufferSize: int, isInReorgThreshold: bool, + // This process has got as far as it can without the run's leave. What a + // supervisor reads to decide that a split run may go realtime as one. + hasArrivedAtHead: bool, rollbackEnabled: bool, maxBatchSize: int, preloadSeconds: float, @@ -187,6 +186,10 @@ let merge = (snapshots: array, ~startTime, ~metricTime, ~elapsedSeconds) => { elapsedSeconds, targetBufferSize: sumInt(s => s.targetBufferSize), isInReorgThreshold: snapshots->Array.some(s => s.isInReorgThreshold), + // The run has arrived only once every process has: one still backfilling + // speaks for the whole indexer. + hasArrivedAtHead: snapshots->Utils.Array.notEmpty && + snapshots->Array.every(s => s.hasArrivedAtHead), rollbackEnabled: snapshots->Array.some(s => s.rollbackEnabled), maxBatchSize: snapshots->Array.reduce(0, (acc, s) => Pervasives.max(acc, s.maxBatchSize)), preloadSeconds: sumFloat(s => s.preloadSeconds), diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index 2848bd25f..e93566760 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -274,17 +274,12 @@ let awaitExit = async (group): outcome => { %%private(let releaseCheckIntervalMillis = 500) // Whether a run holding its workers back may let them go: every worker has -// reported, and every chain any of them drives has reached the head. A chain -// resumed already realtime, or one that processed to its end block, has arrived -// as far as the run is concerned — waiting on either would never end. +// reported, and every one of them has got as far as it can on its own. What +// counts as arrived is the worker's own conclusion — the supervisor only asks +// each of them the question an unsplit run asks itself. let isRunAtHead = (snapshots: array, ~workerCount) => - snapshots->Array.length === workerCount && - snapshots->Array.every(snapshot => - snapshot.chains->Utils.Array.notEmpty && - snapshot.chains->Array.every( - chain => - chain.isReadyForReorgThreshold || chain.isReady || chain->Metrics.hasProcessedToEndblock, - ) + snapshots->Array.length === workerCount && snapshots->Array.every(snapshot => + snapshot.hasArrivedAtHead ) // Runs the group: creates the schema for every chain, forks a worker per plan diff --git a/packages/envio/src/db/InternalTable.res b/packages/envio/src/db/InternalTable.res index 234f193c2..05c28ab63 100644 --- a/packages/envio/src/db/InternalTable.res +++ b/packages/envio/src/db/InternalTable.res @@ -344,7 +344,14 @@ VALUES ${valuesRows->Array.joinUnsafe(",\n ")};`, let setClauses = Array.mapWithIndex(metaFields, (field, index) => { let fieldName = (field :> string) let paramIndex = index + 2 // +2 because $1 is for id in WHERE clause - `"${fieldName}" = $${Int.toString(paramIndex)}` + switch field { + // A chain that caught up never un-catches up, so a metadata write staged + // before `markReady` and flushed after the stamp must not clear it. The + // writes race: metadata is written on a throttle of its own, outside the + // batch the finalization flushes. + | #ready_at => `"${fieldName}" = COALESCE($${Int.toString(paramIndex)}, "${fieldName}")` + | _ => `"${fieldName}" = $${Int.toString(paramIndex)}` + } }) `UPDATE "${pgSchema}"."${table.tableName}" From dda300f49885c58651545b4fdb9a1a84cf3a55ec Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 18 Sep 2026 13:51:20 +0000 Subject: [PATCH 29/61] Let a multichain scenario run again behind the barrier A scenario with more than one chain can run a second time the way a supervised worker runs: held, and released by the real predicate over its own metrics. A split run can't be simulated here, since the sources a test drives are objects in this process that a forked worker wouldn't have, but the hold, the predicate and the release are the production ones. Off unless a scenario asks for it. A describe block's counters are incremented by both passes and most bodies assert on them absolutely, so turning it on everywhere means making those counts independent of how many times a body runs first. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- .../test/ChainFilterFinalize_test.res | 1 + .../envio-tests/test/SchemaIndexes_test.res | 9 ++++-- .../test/helpers/IndexerRunner.res | 31 ++++++++++++++++++- .../envio-tests/test/helpers/Scenario.res | 26 +++++++++++++--- 4 files changed, 59 insertions(+), 8 deletions(-) diff --git a/packages/envio-tests/test/ChainFilterFinalize_test.res b/packages/envio-tests/test/ChainFilterFinalize_test.res index 7ffc98958..866a88fd6 100644 --- a/packages/envio-tests/test/ChainFilterFinalize_test.res +++ b/packages/envio-tests/test/ChainFilterFinalize_test.res @@ -80,6 +80,7 @@ let catchUp = async (~indexer: IndexerRunner.t, ~source: MockSource.t) => { describe("envio start --chain", () => { scenario->Scenario.it( "Indexes and stamps each chain as it catches up, with the other chain never started", + ~supervised=true, ~sources=[{chain: 1}, {chain: 137}], async (~t, ~indexer, ~source) => { let first = await indexer.restart(~chains=[ChainId.fromInt(1)], ()) diff --git a/packages/envio-tests/test/SchemaIndexes_test.res b/packages/envio-tests/test/SchemaIndexes_test.res index 00a9ec830..9da6940ec 100644 --- a/packages/envio-tests/test/SchemaIndexes_test.res +++ b/packages/envio-tests/test/SchemaIndexes_test.res @@ -394,7 +394,10 @@ describe("Deferred schema indexes", () => { }, }, async (~t, ~indexer, ~source) => { - let finalizeCalls = multichainFinalizeCalls + // Counted from where this pass began: a multichain scenario runs a second + // time behind the barrier, over the same counter. + let before = multichainFinalizeCalls.contents + let finalizeCalls = () => multichainFinalizeCalls.contents - before let chainA = source(100) let chainB = source(1337) let {sql, pgSchema} = indexer.pg @@ -409,7 +412,7 @@ describe("Deferred schema indexes", () => { await indexer.getBatchWritePromise() t.expect( - (finalizeCalls.contents, await readyAtByChainId(~sql, ~pgSchema)), + (finalizeCalls(), await readyAtByChainId(~sql, ~pgSchema)), ~message="Chain A is at its head, but chain B is still backfilling", ).toEqual((0, [(ChainId.fromInt(100), false), (ChainId.fromInt(1337), false)])) @@ -421,7 +424,7 @@ describe("Deferred schema indexes", () => { let times = readyAtTimes->Array.map(((_, readyAt)) => readyAt) t.expect( ( - finalizeCalls.contents, + finalizeCalls(), readyAtTimes->Array.map(((id, _)) => id), times->Array.every(Option.isSome), times->Array.get(0) == times->Array.get(1), diff --git a/packages/envio-tests/test/helpers/IndexerRunner.res b/packages/envio-tests/test/helpers/IndexerRunner.res index 2503618e1..281996d94 100644 --- a/packages/envio-tests/test/helpers/IndexerRunner.res +++ b/packages/envio-tests/test/helpers/IndexerRunner.res @@ -64,6 +64,11 @@ type rec t = { releaseRealtime: unit => unit, } +// How often the stand-in supervisor of a supervised pass asks whether the run +// may go realtime. Short enough that the run gets there in the same tick a test +// would otherwise see it. +%%private(let releaseCheckIntervalMillis = 1) + let entityConfigByName = (config: Config.t, name): Internal.entityConfig => config.userEntitiesByName->Dict.get(name)->Option.getOrThrow @@ -83,6 +88,12 @@ let run = async ( // Runs the indexer the way a supervised worker runs: it waits to be released // before entering the reorg threshold or switching to realtime. ~holdRealtime=false, + // Runs it behind the same barrier with a stand-in supervisor releasing it on + // the real predicate, so a scenario exercises the held path without a test + // having to drive it. A split run can't be simulated here — the sources a + // test drives are objects in this process, which a forked worker wouldn't + // have — but the hold, the predicate and the release are the production ones. + ~superviseRun=false, ~onError=?, ~onExit=?, ~mapStorage: Persistence.storage => Persistence.storage=storage => storage, @@ -169,12 +180,28 @@ let run = async ( ~targetBufferSize?, ~isDevelopmentMode=false, ~shouldUseTui=false, - ~holdRealtime, + ~holdRealtime={holdRealtime || superviseRun}, ~onError, ~onExit?, ) state->IndexerLoop.start + // Only when the test didn't ask for the hold itself: one that did is + // testing the barrier and owns its own release. + let releaseCheck = ref(None) + if superviseRun && !holdRealtime { + releaseCheck := + Some( + setInterval(() => + if [state->IndexerState.toMetrics]->Supervisor.isRunAtHead(~workerCount=1) { + releaseCheck.contents->Option.forEach(clearInterval) + releaseCheck := None + state->IndexerState.releaseRealtime + } + , releaseCheckIntervalMillis), + ) + } + // Persist before stopping, else a resumed indexer loses uncommitted state, // then let any in-flight batch or write settle so nothing from this run @@ -188,6 +215,8 @@ let run = async ( | None => let promise = ( async () => { + releaseCheck.contents->Option.forEach(clearInterval) + releaseCheck := None await state->Writing.flush state->IndexerState.stop // Tests deliberately leave handlers that never resolve, which pins diff --git a/packages/envio-tests/test/helpers/Scenario.res b/packages/envio-tests/test/helpers/Scenario.res index aaecddea5..84c9f8353 100644 --- a/packages/envio-tests/test/helpers/Scenario.res +++ b/packages/envio-tests/test/helpers/Scenario.res @@ -169,6 +169,7 @@ let run = async ( ~clientFilterAddressThreshold=?, ~reorgThresholdReadyTolerance=?, ~holdRealtime=?, + ~superviseRun=?, ~onError=?, ~onExit=?, ~mapStorage=?, @@ -239,6 +240,7 @@ let run = async ( ~reducedPollingInterval?, ~targetBufferSize?, ~holdRealtime?, + ~superviseRun?, ~onError?, ~onExit?, ~mapStorage?, @@ -270,6 +272,12 @@ let it = ( ~clientFilterAddressThreshold=?, ~reorgThresholdReadyTolerance=?, ~holdRealtime=?, + // Runs a multichain scenario a second time behind the barrier a supervised + // worker runs behind, so the scenario covers the held path as well as the + // plain one. Off until the suite's own shared counters are made independent + // of how many times a body runs — a ref a describe block owns is incremented + // by both passes, which is what most of them assert on. + ~supervised=false, ~onError=?, ~onExit=?, ~mapStorage=?, @@ -288,7 +296,7 @@ let it = ( async _ => (), ) | None => - let runBody = async (t: Vitest.testContext) => + let runBody = (~superviseRun) => async (t: Vitest.testContext) => await scenario->run( ~sources, ~reducedPollingInterval?, @@ -297,14 +305,24 @@ let it = ( ~clientFilterAddressThreshold?, ~reorgThresholdReadyTolerance?, ~holdRealtime?, + ~superviseRun, ~onError?, ~onExit?, ~mapStorage?, (~indexer, ~source) => body(~t, ~indexer, ~source), ) - switch retry { - | Some(retry) => Vitest.Async.itWithOptions(name, {retry, ?timeout}, runBody) - | None => Vitest.Async.it(name, runBody, ~timeout?) + let register = (name, ~superviseRun) => + switch retry { + | Some(retry) => + Vitest.Async.itWithOptions(name, {retry, ?timeout}, runBody(~superviseRun)) + | None => Vitest.Async.it(name, runBody(~superviseRun), ~timeout?) + } + + register(name, ~superviseRun=false) + // One chain is a run whose every chain is its own process's already, so the + // barrier has nothing to hold: only a multichain scenario says anything new. + if supervised && scenario.config.chainMap->ChainMap.keys->Array.length > 1 { + register(`${name} [supervised]`, ~superviseRun=true) } } } From 77622363a00b4d86a522b76320f93cbf8eb50295 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 18 Sep 2026 13:53:11 +0000 Subject: [PATCH 30/61] Pin the ready_at clause the metadata update now generates Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- packages/envio-tests/test/lib_tests/PgStorage_test.res | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/packages/envio-tests/test/lib_tests/PgStorage_test.res b/packages/envio-tests/test/lib_tests/PgStorage_test.res index f0cc99f5f..9cbcf5554 100644 --- a/packages/envio-tests/test/lib_tests/PgStorage_test.res +++ b/packages/envio-tests/test/lib_tests/PgStorage_test.res @@ -891,10 +891,12 @@ VALUES($1,$2)ON CONFLICT("id") DO UPDATE SET "c_id" = EXCLUDED."c_id";` async t => { let query = InternalTable.Chains.makeMetaFieldsUpdateQuery(~pgSchema="test_schema") + // `ready_at` keeps what is committed: a write staged before the + // finalization stamped it must not clear it on its way to the database. let expectedQuery = `UPDATE "test_schema"."envio_chains" SET "buffer_block" = $2, "first_event_block" = $3, - "ready_at" = $4, + "ready_at" = COALESCE($4, "ready_at"), "_is_hyper_sync" = $5 WHERE "id" = $1;` From ecf1d13a875b60188f53b3835d919408890e426d Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 18 Sep 2026 13:58:27 +0000 Subject: [PATCH 31/61] Run a multichain scenario behind the barrier by default MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Thirty-seven of them now run twice: once plainly, once the way a supervised worker runs, released by the real predicate over the run's own metrics. Opted out are the suites that stage their chains at different heights and drive the reorg-threshold transition themselves. The hold defers exactly that transition until every chain is at the head, which is the premise those bodies set up — a scenario there would be re-running against a precondition it had just been denied, not covering more ground. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- .../test/BelowHeadPollingPin_test.res | 2 +- .../test/ChainFilterFinalize_test.res | 1 - packages/envio-tests/test/E2E_test.res | 4 +-- .../test/EnterReorgThreshold_test.res | 4 +-- .../test/HandlerChainInfo_test.res | 2 +- .../test/IsolatedRollback_test.res | 12 ++++----- .../envio-tests/test/PerChainEntity_test.res | 8 +++--- .../test/PerChainHistoryPrune_test.res | 12 ++++----- .../envio-tests/test/ResumeFinalize_test.res | 6 ++--- .../test/RollbackDiffCheckpointIds_test.res | 2 +- packages/envio-tests/test/Rollback_test.res | 2 +- .../envio-tests/test/SchemaIndexes_test.res | 8 +++--- .../test/ZeroReorgDepthHistory_test.res | 4 +-- .../envio-tests/test/helpers/Scenario.res | 26 ++++++++++++++----- 14 files changed, 53 insertions(+), 40 deletions(-) diff --git a/packages/envio-tests/test/BelowHeadPollingPin_test.res b/packages/envio-tests/test/BelowHeadPollingPin_test.res index a17c255bc..d41ce3914 100644 --- a/packages/envio-tests/test/BelowHeadPollingPin_test.res +++ b/packages/envio-tests/test/BelowHeadPollingPin_test.res @@ -1,6 +1,6 @@ open Vitest -let scenario = Scenario.make( +let scenario = Scenario.make(~supervised=false, ~configYaml=` name: below-head-polling contracts: diff --git a/packages/envio-tests/test/ChainFilterFinalize_test.res b/packages/envio-tests/test/ChainFilterFinalize_test.res index 866a88fd6..7ffc98958 100644 --- a/packages/envio-tests/test/ChainFilterFinalize_test.res +++ b/packages/envio-tests/test/ChainFilterFinalize_test.res @@ -80,7 +80,6 @@ let catchUp = async (~indexer: IndexerRunner.t, ~source: MockSource.t) => { describe("envio start --chain", () => { scenario->Scenario.it( "Indexes and stamps each chain as it catches up, with the other chain never started", - ~supervised=true, ~sources=[{chain: 1}, {chain: 137}], async (~t, ~indexer, ~source) => { let first = await indexer.restart(~chains=[ChainId.fromInt(1)], ()) diff --git a/packages/envio-tests/test/E2E_test.res b/packages/envio-tests/test/E2E_test.res index 1749c1ce3..02b12ddfe 100644 --- a/packages/envio-tests/test/E2E_test.res +++ b/packages/envio-tests/test/E2E_test.res @@ -28,7 +28,7 @@ let chainYaml = (chainId, ~startBlock=1) => ` let makeScenario = (~name, ~rollback=true, ~chains) => - Scenario.make( + Scenario.make(~supervised=false, ~configYaml=` name: ${name} rollback_on_reorg: ${rollback ? "true" : "false"}${contractsYaml}chains:${chains}`, @@ -44,7 +44,7 @@ let scenario = makeScenario(~name="e2e", ~chains=chainYaml(1337)) // Partition ids and the chain's range-cost budget follow the contract set, so // this scenario keeps the address-less contracts alongside the addressed ones. -let partitionScenario = Scenario.make( +let partitionScenario = Scenario.make(~supervised=false, ~configYaml=` name: e2e-partitions rollback_on_reorg: true${contractsYaml} - name: SimpleNft diff --git a/packages/envio-tests/test/EnterReorgThreshold_test.res b/packages/envio-tests/test/EnterReorgThreshold_test.res index 4744c30ef..e40f8e12c 100644 --- a/packages/envio-tests/test/EnterReorgThreshold_test.res +++ b/packages/envio-tests/test/EnterReorgThreshold_test.res @@ -20,7 +20,7 @@ type Gravatar { // Two chains, each lagging maxReorgDepth (200) below head before the // threshold. Head starts at 1000, so the pre-threshold head is 800. -let multichain = Scenario.make( +let multichain = Scenario.make(~supervised=false, ~configYaml=` name: enter-reorg-threshold-multichain contracts: @@ -52,7 +52,7 @@ chains: ~schema, ) -let singleChain = Scenario.make( +let singleChain = Scenario.make(~supervised=false, ~configYaml=` name: enter-reorg-threshold-single chains: diff --git a/packages/envio-tests/test/HandlerChainInfo_test.res b/packages/envio-tests/test/HandlerChainInfo_test.res index b412e18e7..be732e0a7 100644 --- a/packages/envio-tests/test/HandlerChainInfo_test.res +++ b/packages/envio-tests/test/HandlerChainInfo_test.res @@ -4,7 +4,7 @@ open Vitest // item came from, not whichever chain the batch happens to start on, and // `isRealtime` only flips once every chain in the indexer is at its head. -let scenario = Scenario.make( +let scenario = Scenario.make(~supervised=false, ~configYaml=` name: handler-chain-info contracts: diff --git a/packages/envio-tests/test/IsolatedRollback_test.res b/packages/envio-tests/test/IsolatedRollback_test.res index d7c1fdc5d..4541425e1 100644 --- a/packages/envio-tests/test/IsolatedRollback_test.res +++ b/packages/envio-tests/test/IsolatedRollback_test.res @@ -52,7 +52,7 @@ type Counter { } ` -let scenario = Scenario.make( +let scenario = Scenario.make(~supervised=false, ~schema=perChainSchema, ~configYaml=makeConfigYaml(~name="isolated-rollback"), ) @@ -60,7 +60,7 @@ let scenario = Scenario.make( // A single chain with no cross-chain entity is a per-chain sequence too: nothing // about the mode needs a sibling, and the chain-id column every per-chain entity // carries is what its bounds join against. -let singleChainScenario = Scenario.make( +let singleChainScenario = Scenario.make(~supervised=false, ~schema=perChainSchema, ~configYaml=` name: single-chain-per-chain @@ -77,7 +77,7 @@ chains:${chainYaml(100)} // One cross-chain entity is enough to couple the chains: a value chain 1337 // wrote can be what chain 100 read and overwrote, so its reorg has to take // every chain back with it. -let crossChainScenario = Scenario.make( +let crossChainScenario = Scenario.make(~supervised=false, ~schema=perChainSchema ++ ` type Total @crossChain { id: ID! @@ -90,13 +90,13 @@ type Total @crossChain { // The sink is append-only: an isolated rollback reaches it as the diff rows the // next batch carries, and its current-state view has to resolve to the same // thing Postgres holds. -let clickHouseScenario = Scenario.make( +let clickHouseScenario = Scenario.make(~supervised=false, ~schema=perChainSchema, ~configYaml=makeConfigYaml(~name="isolated-rollback-clickhouse"), ~unsupported=[{backend: #postgres, reason: "asserts against a ClickHouse server"}], ) -let fullHistoryScenario = Scenario.make( +let fullHistoryScenario = Scenario.make(~supervised=false, ~schema=perChainSchema, ~configYaml=makeConfigYaml( ~name="isolated-rollback-full-history", @@ -107,7 +107,7 @@ let fullHistoryScenario = Scenario.make( // A per-chain entity's chain column is named `chain_id` under snake_case, the // same name the per-chain bounds relation gives its own, so the rollback // queries have to keep the two apart. -let snakeCaseScenario = Scenario.make( +let snakeCaseScenario = Scenario.make(~supervised=false, ~schema=perChainSchema, ~configYaml=makeConfigYaml( ~name="isolated-rollback-snake-case", diff --git a/packages/envio-tests/test/PerChainEntity_test.res b/packages/envio-tests/test/PerChainEntity_test.res index 8a2bce09b..6a29518dc 100644 --- a/packages/envio-tests/test/PerChainEntity_test.res +++ b/packages/envio-tests/test/PerChainEntity_test.res @@ -66,11 +66,11 @@ type GlobalCounter @crossChain { } ` -let scenario = Scenario.make(~schema, ~configYaml=makeConfigYaml()) +let scenario = Scenario.make(~supervised=false, ~schema, ~configYaml=makeConfigYaml()) // The two chains need a reorg threshold to roll back within, so this variant // sets one — `max_reorg_depth` is per chain, so it goes in the chain blocks. -let rollbackScenario = Scenario.make( +let rollbackScenario = Scenario.make(~supervised=false, ~schema, ~configYaml=makeConfigYaml(~rollback="\nrollback_on_reorg: true")->String.replaceAll( " start_block: 1\n", @@ -80,7 +80,7 @@ let rollbackScenario = Scenario.make( // The entity object and the getWhere filter key the chain by `chainId` while // the column is `chain_id`. -let snakeCaseScenario = Scenario.make( +let snakeCaseScenario = Scenario.make(~supervised=false, ~schema, ~configYaml=makeConfigYaml( ~storage=`storage: @@ -91,7 +91,7 @@ let snakeCaseScenario = Scenario.make( ) // The history prune is asserted through raw SQL against the history tables. -let pruneScenario = Scenario.make(~schema, ~configYaml=makeConfigYaml()) +let pruneScenario = Scenario.make(~supervised=false, ~schema, ~configYaml=makeConfigYaml()) let methods: array = [#getHeightOrThrow, #getItemsOrThrow] let reorgMethods: array = [#getHeightOrThrow, #getItemsOrThrow, #getBlockHashes] diff --git a/packages/envio-tests/test/PerChainHistoryPrune_test.res b/packages/envio-tests/test/PerChainHistoryPrune_test.res index ccca66e7d..a5b34bb86 100644 --- a/packages/envio-tests/test/PerChainHistoryPrune_test.res +++ b/packages/envio-tests/test/PerChainHistoryPrune_test.res @@ -56,7 +56,7 @@ chains:${chainYaml(100, ~startBlock=110, ~maxReorgDepth=15)}${chainYaml( )} ` -let scenario = Scenario.make( +let scenario = Scenario.make(~supervised=false, ~schema, ~configYaml=makeConfigYaml("per-chain-prune", ~laggingChainId=1337), ) @@ -71,13 +71,13 @@ type Total @crossChain { // One cross-chain entity couples the chains: a reorg on the chain furthest // behind can reach a row any chain wrote, so none may prune past its safe point. -let crossChainScenario = Scenario.make( +let crossChainScenario = Scenario.make(~supervised=false, ~schema=crossChainSchema, ~configYaml=makeConfigYaml("per-chain-prune-cross-chain", ~laggingChainId=1337), ) // The same, with the lagging chain visited first. -let crossChainLaggingFirstScenario = Scenario.make( +let crossChainLaggingFirstScenario = Scenario.make(~supervised=false, ~schema=crossChainSchema, ~configYaml=makeConfigYaml("per-chain-prune-cross-chain-lagging-first", ~laggingChainId=5), ) @@ -86,7 +86,7 @@ let crossChainLaggingFirstScenario = Scenario.make( // is safe. Under the shared sequence a cross-chain entity brings, its own last // id is not the bound — an idle chain's would hold every other chain's prune // back for as long as it stays idle. -let crossChainZeroDepthScenario = Scenario.make( +let crossChainZeroDepthScenario = Scenario.make(~supervised=false, ~schema=crossChainSchema, ~configYaml=` name: per-chain-prune-cross-chain-zero-depth @@ -107,7 +107,7 @@ chains:${chainYaml(100, ~startBlock=110, ~maxReorgDepth=15)}${chainYaml( // Every chain reaches a safe checkpoint of its own, so the prune carries a bound // per chain. Three of them rather than two: a pair of bounds can be crossed and // still look right, while three cannot. -let manyBoundsScenario = Scenario.make( +let manyBoundsScenario = Scenario.make(~supervised=false, ~schema, ~configYaml=` name: per-chain-prune-many-bounds @@ -128,7 +128,7 @@ chains:${chainYaml(100, ~startBlock=110, ~maxReorgDepth=15)}${chainYaml( // A chain with no reorg depth can't be rolled back, so with no cross-chain // entity to let a sibling's rollback reach its rows, it has no history to keep // and everything it has committed is safe to prune. -let zeroDepthScenario = Scenario.make( +let zeroDepthScenario = Scenario.make(~supervised=false, ~schema, ~configYaml=` name: per-chain-prune-zero-depth diff --git a/packages/envio-tests/test/ResumeFinalize_test.res b/packages/envio-tests/test/ResumeFinalize_test.res index 358e9c05c..73e935258 100644 --- a/packages/envio-tests/test/ResumeFinalize_test.res +++ b/packages/envio-tests/test/ResumeFinalize_test.res @@ -34,13 +34,13 @@ let chainYaml = (chainId, address, extra) => let gravatar1337 = "0x2B2f78c5BF6D9C12Ee1225D5F374aa91204580c3" let gravatar1 = "0x3B2f78c5BF6D9C12Ee1225D5F374aa91204580c3" -let scenario = Scenario.make( +let scenario = Scenario.make(~supervised=false, ~configYaml=` name: resume-finalize${contractsYaml}chains:${chainYaml(1337, gravatar1337, "")}`, ~schema, ) -let endBlockScenario = Scenario.make( +let endBlockScenario = Scenario.make(~supervised=false, ~configYaml=` name: resume-finalize-end-block${contractsYaml}chains:${chainYaml( 1337, @@ -50,7 +50,7 @@ name: resume-finalize-end-block${contractsYaml}chains:${chainYaml( ~schema, ) -let multichainScenario = Scenario.make( +let multichainScenario = Scenario.make(~supervised=false, ~configYaml=` name: resume-finalize-multichain${contractsYaml}chains:${chainYaml(1, gravatar1, "")}${chainYaml( 1337, diff --git a/packages/envio-tests/test/RollbackDiffCheckpointIds_test.res b/packages/envio-tests/test/RollbackDiffCheckpointIds_test.res index fb64c8da4..0f9b5fde1 100644 --- a/packages/envio-tests/test/RollbackDiffCheckpointIds_test.res +++ b/packages/envio-tests/test/RollbackDiffCheckpointIds_test.res @@ -35,7 +35,7 @@ let chainYaml = chainId => // One cross-chain entity is what makes the checkpoint sequence shared, and what // makes a reorg on either chain roll both of them back. -let scenario = Scenario.make( +let scenario = Scenario.make(~supervised=false, ~schema=` type Counter { id: ID! diff --git a/packages/envio-tests/test/Rollback_test.res b/packages/envio-tests/test/Rollback_test.res index a174d7bd2..c0096dc25 100644 --- a/packages/envio-tests/test/Rollback_test.res +++ b/packages/envio-tests/test/Rollback_test.res @@ -50,7 +50,7 @@ indexer.onEvent({ contract: "SimpleNft", event: "Transfer" }, async () => {}); ` let makeScenario = (~name, ~chains, ~extra="") => - Scenario.make( + Scenario.make(~supervised=false, ~configYaml=` name: ${name} rollback_on_reorg: true${extra}${contractsYaml}chains:${chains}`, diff --git a/packages/envio-tests/test/SchemaIndexes_test.res b/packages/envio-tests/test/SchemaIndexes_test.res index 9da6940ec..b480f059f 100644 --- a/packages/envio-tests/test/SchemaIndexes_test.res +++ b/packages/envio-tests/test/SchemaIndexes_test.res @@ -38,7 +38,7 @@ contracts: - event: "TestEvent()" ` -let scenario = Scenario.make( +let scenario = Scenario.make(~supervised=false, ~configYaml=` name: schema-indexes${contractsYaml}chains:${chainYaml( 1337, @@ -50,7 +50,7 @@ name: schema-indexes${contractsYaml}chains:${chainYaml( // An `end_block` the chain never reaches: the indexer still counts itself caught // up once progress sits at the head, so the deferred indexes are owed then, not // at the unreachable end block. -let unreachableEndBlockScenario = Scenario.make( +let unreachableEndBlockScenario = Scenario.make(~supervised=false, ~configYaml=` name: schema-indexes-unreachable-end${contractsYaml}chains: - id: 1337 @@ -68,7 +68,7 @@ name: schema-indexes-unreachable-end${contractsYaml}chains: // A `start_block` past the head: the chain is at its head from the first moment // and never has a batch to process, so nothing ever writes its progress row. -let aheadOfHeadScenario = Scenario.make( +let aheadOfHeadScenario = Scenario.make(~supervised=false, ~configYaml=` name: schema-indexes-ahead-of-head${contractsYaml}chains: - id: 1337 @@ -83,7 +83,7 @@ name: schema-indexes-ahead-of-head${contractsYaml}chains: ~schema, ) -let multichainScenario = Scenario.make( +let multichainScenario = Scenario.make(~supervised=false, ~configYaml=` name: schema-indexes-multichain${contractsYaml}chains:${chainYaml( 100, diff --git a/packages/envio-tests/test/ZeroReorgDepthHistory_test.res b/packages/envio-tests/test/ZeroReorgDepthHistory_test.res index 0efc3cb7e..f3c7b6bf2 100644 --- a/packages/envio-tests/test/ZeroReorgDepthHistory_test.res +++ b/packages/envio-tests/test/ZeroReorgDepthHistory_test.res @@ -37,7 +37,7 @@ let chainYaml = (chainId, ~maxReorgDepth) => // One chain, and the default cross-chain entities that make its checkpoint // sequence a shared one. -let singleChainScenario = Scenario.make( +let singleChainScenario = Scenario.make(~supervised=false, ~schema, ~configYaml=` name: zero-reorg-depth-history @@ -52,7 +52,7 @@ chains:${chainYaml(100, ~maxReorgDepth=0)} // No cross-chain entity, so each chain counts its own checkpoints and only the // chain that can be rolled back keeps any. -let perChainScenario = Scenario.make( +let perChainScenario = Scenario.make(~supervised=false, ~schema, ~configYaml=` name: zero-reorg-depth-history-per-chain diff --git a/packages/envio-tests/test/helpers/Scenario.res b/packages/envio-tests/test/helpers/Scenario.res index 84c9f8353..d420704b0 100644 --- a/packages/envio-tests/test/helpers/Scenario.res +++ b/packages/envio-tests/test/helpers/Scenario.res @@ -15,6 +15,11 @@ type t = { handlers: option, unsupported: array, site: string, + // Whether a multichain scenario also runs behind the barrier. Off for a + // scenario that stages its chains at different heights and drives the + // reorg-threshold transition itself: the hold deferring that transition is + // the premise such a body sets up being taken away. + supervised: bool, } type sourceMock = { @@ -46,7 +51,15 @@ let withClickHouseStorage = configYaml => configYaml ++ "\nstorage:\n postgres:\n default: true\n clickhouse:\n default: true\n" } -let make = (~configYaml, ~schema=?, ~env=?, ~files=?, ~handlers=?, ~unsupported=[]): t => { +let make = ( + ~configYaml, + ~schema=?, + ~env=?, + ~files=?, + ~handlers=?, + ~unsupported=[], + ~supervised=true, +): t => { let isUnsupported = unsupported->Array.some(({backend}) => backend === IndexerRunner.selectedBackend) @@ -90,6 +103,7 @@ let make = (~configYaml, ~schema=?, ~env=?, ~files=?, ~handlers=?, ~unsupported= handlers, unsupported, site, + supervised, } } @@ -274,10 +288,8 @@ let it = ( ~holdRealtime=?, // Runs a multichain scenario a second time behind the barrier a supervised // worker runs behind, so the scenario covers the held path as well as the - // plain one. Off until the suite's own shared counters are made independent - // of how many times a body runs — a ref a describe block owns is incremented - // by both passes, which is what most of them assert on. - ~supervised=false, + // plain one. `Scenario.make(~supervised=false)` opts a whole scenario out. + ~supervised=true, ~onError=?, ~onExit=?, ~mapStorage=?, @@ -321,7 +333,9 @@ let it = ( register(name, ~superviseRun=false) // One chain is a run whose every chain is its own process's already, so the // barrier has nothing to hold: only a multichain scenario says anything new. - if supervised && scenario.config.chainMap->ChainMap.keys->Array.length > 1 { + if ( + supervised && scenario.supervised && scenario.config.chainMap->ChainMap.keys->Array.length > 1 + ) { register(`${name} [supervised]`, ~superviseRun=true) } } From d8675b74f4542dd2885d00b0f1fa8d96828bd7ae Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 18 Sep 2026 14:34:58 +0000 Subject: [PATCH 32/61] Say what a chain fetched to, and what it is waiting on MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit "All events have been fetched" claimed three things it didn't mean. It is one chain's line, not the run's, so a five-chain run printed it five times, each saying "all". Fetched is not processed and not ready: the buffer still holds events, the deferred indexes are still to build, and the line opened the longest silence in a run rather than closing it. And below the reorg threshold a chain may only fetch the finalized range, so what it had reached was the safe block, not the head. It now names the block it reached and what it reached — the safe block, the chain head past the threshold, or an end block — carries what is left to process, and says whether anything else has to catch up first: the chains this process drives, or, for a supervised worker, the ones its siblings do. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- .../test/SupervisedRealtime_test.res | 79 +++++++++++++++++++ .../test/helpers/TestChainMetrics.res | 22 +++++- .../test/lib_tests/FetchedTo_test.res | 24 ++++++ packages/envio/src/ChainFetching.res | 17 +++- packages/envio/src/ChainState.res | 16 ++++ packages/envio/src/ChainState.resi | 1 + 6 files changed, 155 insertions(+), 4 deletions(-) create mode 100644 packages/envio-tests/test/lib_tests/FetchedTo_test.res diff --git a/packages/envio-tests/test/SupervisedRealtime_test.res b/packages/envio-tests/test/SupervisedRealtime_test.res index 775a953b4..e6ebb2b06 100644 --- a/packages/envio-tests/test/SupervisedRealtime_test.res +++ b/packages/envio-tests/test/SupervisedRealtime_test.res @@ -45,6 +45,25 @@ chains: ~schema, ) +// Rollback on, which is what gives a chain a reorg depth: pre-threshold its +// fetch frontier is capped at the safe block, so progress can never reach the +// head until the chain enters the threshold. +let rollbackScenario = Scenario.make( + ~configYaml=` +name: supervised-realtime-rollback +rollback_on_reorg: true +disable_default_cross_chain: true +contracts: + - name: Gravatar + events: + - event: "TestEvent()" +chains:${chainYaml(1, "0x2B2f78c5BF6D9C12Ee1225D5F374aa91204580c3")}${chainYaml( + 137, + "0x3B2f78c5BF6D9C12Ee1225D5F374aa91204580c3", + )}`, + ~schema, +) + let scenario = Scenario.make( ~configYaml=` name: supervised-realtime @@ -139,3 +158,63 @@ describe("A supervised worker at its end block", () => { ~onExit=() => exited := true, ) }) + +describe("A run whose chains have a reorg depth", () => { + // The transition the barrier holds is the one that lifts the pre-threshold + // lag. Held until its chains reach the head, a run would be waiting on + // progress only that transition makes reachable — so what a worker reports + // as arrived has to be the safe block, as far as it can fetch until then. + rollbackScenario->Scenario.it( + "Enters the reorg threshold once every chain is at its safe block", + ~sources=[{chain: 1}, {chain: 137}], + async (~t, ~indexer, ~source) => { + await Scenario.enterReorgThreshold(~t, ~indexer, ~source=source(1)) + await Scenario.enterReorgThreshold(~t, ~indexer, ~source=source(137)) + + t.expect( + await indexer.metric("envio_reorg_threshold"), + ~message="Both chains have fetched the whole finalized range", + ).toEqual([{value: "1", labels: Dict.make()}]) + }, + ) +}) + +describe("A supervised worker on a chain with a reorg depth", () => { + // The transition being held is the one that lifts the pre-threshold lag, so a + // run held until its chains reach the head would be waiting on progress that + // only the transition itself makes reachable. + rollbackScenario->Scenario.it( + "Arrives at the safe block, which is as far as it can fetch before the threshold", + ~sources=[{chain: 1}, {chain: 137}], + ~holdRealtime=true, + async (~t, ~indexer, ~source) => { + let {sql, pgSchema} = indexer.pg + let head = 300 + // A response short of the head by the reorg depth: the whole finalized + // range, and all this chain may fetch until it enters the threshold. + let catchUpToSafeBlock = (~source: MockSource.t) => { + source.resolveGetHeightOrThrow(head) + source.resolveGetItemsOrThrow([], ~latestFetchedBlockNumber=head - 200) + } + catchUpToSafeBlock(~source=source(1)) + catchUpToSafeBlock(~source=source(137)) + await indexer.waitUntilIdle() + + t.expect( + await readyAtByChainId(~sql, ~pgSchema), + ~message="Both chains are as far as they can fetch, and the run has not said so", + ).toEqual([("1", None), ("137", None)]) + + indexer.releaseRealtime() + await indexer.waitUntilReady() + + t.expect( + (await readyAtByChainId(~sql, ~pgSchema))->Array.map(((chainId, readyAt)) => ( + chainId, + readyAt->Option.isSome, + )), + ~message="Released, the run enters the threshold and stamps its chains", + ).toEqual([("1", true), ("137", true)]) + }, + ) +}) diff --git a/packages/envio-tests/test/helpers/TestChainMetrics.res b/packages/envio-tests/test/helpers/TestChainMetrics.res index b58b19ce1..9cdc6c5d6 100644 --- a/packages/envio-tests/test/helpers/TestChainMetrics.res +++ b/packages/envio-tests/test/helpers/TestChainMetrics.res @@ -34,13 +34,14 @@ let registrationsByChainId: HandlerRegister.registrationsByChainId = { let caughtUpAt = Date.fromTime(1700000000000.) -let make = ( +let makeChainState = ( ~endBlock=None, ~progressBlockNumber, ~firstEventBlockNumber, ~timestampCaughtUpToHeadOrEndblock=None, ~sourceBlockNumber=1000, -): Metrics.chainMetrics => + ~isInReorgThreshold=false, +): ChainState.t => ChainState.makeFromDbState( chainConfig, ~resumedChainState={ @@ -57,11 +58,26 @@ let make = ( sourceBlockNumber, }, ~reorgCheckpoints=[], - ~isInReorgThreshold=false, + ~isInReorgThreshold, ~isRealtime=false, ~config=TestConfig.default, ~contractMapping=TestConfig.default.contractMapping, ~registrationsByChainId, + ) + +let make = ( + ~endBlock=None, + ~progressBlockNumber, + ~firstEventBlockNumber, + ~timestampCaughtUpToHeadOrEndblock=None, + ~sourceBlockNumber=1000, +): Metrics.chainMetrics => + makeChainState( + ~endBlock, + ~progressBlockNumber, + ~firstEventBlockNumber, + ~timestampCaughtUpToHeadOrEndblock, + ~sourceBlockNumber, )->ChainState.toMetrics // The snapshot a run reports when nothing has happened yet. Tests spread this diff --git a/packages/envio-tests/test/lib_tests/FetchedTo_test.res b/packages/envio-tests/test/lib_tests/FetchedTo_test.res new file mode 100644 index 000000000..0cde48136 --- /dev/null +++ b/packages/envio-tests/test/lib_tests/FetchedTo_test.res @@ -0,0 +1,24 @@ +open Vitest + +open TestChainMetrics + +// What a chain has actually fetched up to, which is what the line reporting it +// has to say. Below the reorg threshold a chain may only fetch the finalized +// range, so reaching the end of that is not reaching the head. +describe("ChainState.fetchedTo", () => { + it("Names the block a chain has fetched to, not the one it hasn't", t => { + t.expect([ + makeChainState(~progressBlockNumber=500, ~firstEventBlockNumber=None)->ChainState.fetchedTo, + makeChainState( + ~progressBlockNumber=500, + ~firstEventBlockNumber=None, + ~isInReorgThreshold=true, + )->ChainState.fetchedTo, + makeChainState( + ~progressBlockNumber=500, + ~firstEventBlockNumber=None, + ~endBlock=Some(600), + )->ChainState.fetchedTo, + ]).toStrictEqual([("the safe block", 800), ("the chain head", 1000), ("the end block", 600)]) + }) +}) diff --git a/packages/envio/src/ChainFetching.res b/packages/envio/src/ChainFetching.res index a323b141a..652a3e8fd 100644 --- a/packages/envio/src/ChainFetching.res +++ b/packages/envio/src/ChainFetching.res @@ -288,7 +288,22 @@ and applyQueryResponse = ( !(chainState->ChainState.isReady) && chainState->ChainState.isFetchingAtHead ) { - chainState->ChainState.logger->Logging.childInfo("All events have been fetched") + let (target, block) = chainState->ChainState.fetchedTo + // What the chain is waiting on, which is not itself. A supervised worker's + // siblings are other processes, so nothing it can report says how they are + // doing; one process driving several chains at least names its own. + let waitingOn = if state->IndexerState.isHoldingRealtime { + " Waiting for the chains the run's other processes drive." + } else if state->IndexerState.chainStates->Dict.keysToArray->Array.length > 1 { + " Waiting for the other chains." + } else { + "" + } + chainState->ChainState.logger->Logging.childInfo({ + "msg": `Fetched to ${target}.${waitingOn}`, + "block": block, + "eventsToProcess": chainState->ChainState.bufferReadyCount, + }) } } diff --git a/packages/envio/src/ChainState.res b/packages/envio/src/ChainState.res index ca8317fea..560816d0a 100644 --- a/packages/envio/src/ChainState.res +++ b/packages/envio/src/ChainState.res @@ -642,6 +642,22 @@ let dispatch = ( // --- Derived (pure). --- +// Where the fetch frontier has just landed, for the line that reports it. Below +// the reorg threshold a chain can only fetch the finalized range, so reaching +// the end of it is not reaching the head — the rest opens up once the indexer +// enters the threshold. +let fetchedTo = (cs: t) => + switch cs.fetchState.endBlock { + | Some(endBlock) if endBlock <= cs.fetchState.knownHeight - cs.fetchState.blockLag => ( + "the end block", + endBlock, + ) + | _ => + cs.isInReorgThreshold + ? ("the chain head", cs.fetchState.knownHeight) + : ("the safe block", cs.fetchState.knownHeight - cs.fetchState.blockLag) + } + let hasProcessedToEndblock = (cs: t) => { let {committedProgressBlockNumber, fetchState} = cs switch fetchState.endBlock { diff --git a/packages/envio/src/ChainState.resi b/packages/envio/src/ChainState.resi index a96be0315..f8148641f 100644 --- a/packages/envio/src/ChainState.resi +++ b/packages/envio/src/ChainState.resi @@ -123,6 +123,7 @@ let toChainBeforeBatch: (t, ~isRealtime: bool) => Batch.chainBeforeBatch let isReadyToEnterReorgThresholdAfterBatch: (t, ~batch: Batch.t) => bool // Derived (pure). +let fetchedTo: t => (string, int) let hasProcessedToEndblock: t => bool let isDurablyCaughtUp: t => bool let getHighestBlockBelowThreshold: t => int From 0930bd8c5628687b6578fd96ff31a6d567bc760d Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 18 Sep 2026 14:36:51 +0000 Subject: [PATCH 33/61] Say where the rest of a split run is without the possessives Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- packages/envio/src/ChainFetching.res | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/packages/envio/src/ChainFetching.res b/packages/envio/src/ChainFetching.res index 652a3e8fd..3256fe766 100644 --- a/packages/envio/src/ChainFetching.res +++ b/packages/envio/src/ChainFetching.res @@ -289,11 +289,11 @@ and applyQueryResponse = ( chainState->ChainState.isFetchingAtHead ) { let (target, block) = chainState->ChainState.fetchedTo - // What the chain is waiting on, which is not itself. A supervised worker's - // siblings are other processes, so nothing it can report says how they are - // doing; one process driving several chains at least names its own. + // What the chain is waiting on, which is not itself. A supervised worker + // says where the rest of the run is, since its siblings are processes of + // their own and this one can report nothing about how they are doing. let waitingOn = if state->IndexerState.isHoldingRealtime { - " Waiting for the chains the run's other processes drive." + " Waiting for the other chains, indexed by other processes." } else if state->IndexerState.chainStates->Dict.keysToArray->Array.length > 1 { " Waiting for the other chains." } else { From 283f69b68ee761327f733e6e23117bf95ca3d4ab Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 18 Sep 2026 14:40:31 +0000 Subject: [PATCH 34/61] Leave how a run is split out of what a chain reports A held process waits on chains it doesn't drive, so it says so even when it drives only one. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- packages/envio/src/ChainFetching.res | 13 +++++++------ 1 file changed, 7 insertions(+), 6 deletions(-) diff --git a/packages/envio/src/ChainFetching.res b/packages/envio/src/ChainFetching.res index 3256fe766..98c2deeab 100644 --- a/packages/envio/src/ChainFetching.res +++ b/packages/envio/src/ChainFetching.res @@ -289,12 +289,13 @@ and applyQueryResponse = ( chainState->ChainState.isFetchingAtHead ) { let (target, block) = chainState->ChainState.fetchedTo - // What the chain is waiting on, which is not itself. A supervised worker - // says where the rest of the run is, since its siblings are processes of - // their own and this one can report nothing about how they are doing. - let waitingOn = if state->IndexerState.isHoldingRealtime { - " Waiting for the other chains, indexed by other processes." - } else if state->IndexerState.chainStates->Dict.keysToArray->Array.length > 1 { + // What the chain is waiting on, which is not itself. A held process is + // waiting on chains it doesn't drive, so it says so even when it drives + // only one — how the run is split is not the reader's problem. + let waitingOn = if ( + state->IndexerState.isHoldingRealtime || + state->IndexerState.chainStates->Dict.keysToArray->Array.length > 1 + ) { " Waiting for the other chains." } else { "" From ff3cf0a8160429bed61a25c6ed72f30332bd0e92 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 18 Sep 2026 14:47:47 +0000 Subject: [PATCH 35/61] Drop the buffered count from the line a chain reports Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- packages/envio/src/ChainFetching.res | 1 - 1 file changed, 1 deletion(-) diff --git a/packages/envio/src/ChainFetching.res b/packages/envio/src/ChainFetching.res index 98c2deeab..4cf1de22c 100644 --- a/packages/envio/src/ChainFetching.res +++ b/packages/envio/src/ChainFetching.res @@ -303,7 +303,6 @@ and applyQueryResponse = ( chainState->ChainState.logger->Logging.childInfo({ "msg": `Fetched to ${target}.${waitingOn}`, "block": block, - "eventsToProcess": chainState->ChainState.bufferReadyCount, }) } } From 64bbd0fc5f47fd858981c1e7269ae30a78595ca2 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 18 Sep 2026 14:54:58 +0000 Subject: [PATCH 36/61] Report a milestone once, not on every catch-up to a moving head MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The line was gated on the backfill→head transition, which re-arms every time the head advances: the frontier falls behind it, catches up, and the transition happens again. `!isReady` held that back only once a chain was ready, so the line repeated for as long as it wasn't — which a supervised worker now is for as long as the run holds it. A chain reports what it reached instead, and only the first time it reaches it. The safe block, the head past the reorg threshold and an end block are different milestones, so each still gets its line. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- .../test/lib_tests/FetchedTo_test.res | 32 ++++++++++++++----- packages/envio/src/ChainFetching.res | 21 ++++++------ packages/envio/src/ChainState.res | 29 ++++++++++++++--- packages/envio/src/ChainState.resi | 2 +- 4 files changed, 59 insertions(+), 25 deletions(-) diff --git a/packages/envio-tests/test/lib_tests/FetchedTo_test.res b/packages/envio-tests/test/lib_tests/FetchedTo_test.res index 0cde48136..cab18ac9a 100644 --- a/packages/envio-tests/test/lib_tests/FetchedTo_test.res +++ b/packages/envio-tests/test/lib_tests/FetchedTo_test.res @@ -2,23 +2,39 @@ open Vitest open TestChainMetrics -// What a chain has actually fetched up to, which is what the line reporting it -// has to say. Below the reorg threshold a chain may only fetch the finalized -// range, so reaching the end of that is not reaching the head. -describe("ChainState.fetchedTo", () => { +// What a chain reports reaching, and how often. Below the reorg threshold a +// chain may only fetch the finalized range, so reaching the end of that is not +// reaching the head — and since the head moves, it reaches whatever it is +// fetching to over and over. +describe("ChainState.takeFetchedTo", () => { it("Names the block a chain has fetched to, not the one it hasn't", t => { + let takeFrom = cs => cs->ChainState.takeFetchedTo + t.expect([ - makeChainState(~progressBlockNumber=500, ~firstEventBlockNumber=None)->ChainState.fetchedTo, + makeChainState(~progressBlockNumber=500, ~firstEventBlockNumber=None)->takeFrom, makeChainState( ~progressBlockNumber=500, ~firstEventBlockNumber=None, ~isInReorgThreshold=true, - )->ChainState.fetchedTo, + )->takeFrom, makeChainState( ~progressBlockNumber=500, ~firstEventBlockNumber=None, ~endBlock=Some(600), - )->ChainState.fetchedTo, - ]).toStrictEqual([("the safe block", 800), ("the chain head", 1000), ("the end block", 600)]) + )->takeFrom, + ]).toStrictEqual([ + Some(("the safe block", 800)), + Some(("the chain head", 1000)), + Some(("the end block", 600)), + ]) + }) + + it("Reports a milestone once, however many times the chain reaches it", t => { + let chainState = makeChainState(~progressBlockNumber=500, ~firstEventBlockNumber=None) + + t.expect(( + chainState->ChainState.takeFetchedTo, + chainState->ChainState.takeFetchedTo, + )).toStrictEqual((Some(("the safe block", 800)), None)) }) }) diff --git a/packages/envio/src/ChainFetching.res b/packages/envio/src/ChainFetching.res index 4cf1de22c..9bd643b5b 100644 --- a/packages/envio/src/ChainFetching.res +++ b/packages/envio/src/ChainFetching.res @@ -262,7 +262,6 @@ and applyQueryResponse = ( ~transactionStore, ) => { let chainState = state->IndexerState.getChainState(~chainId) - let wasFetchingAtHead = chainState->ChainState.isFetchingAtHead chainState->ChainState.handleQueryResult( ~query, @@ -280,15 +279,15 @@ and applyQueryResponse = ( ) } - // Log the backfill→head transition once: this response brought the fetch - // frontier to the head. Gated on !isReady so realtime re-catch-ups (a new - // block arrives, gets fetched) don't spam the log after the chain is synced. - if ( - !wasFetchingAtHead && - !(chainState->ChainState.isReady) && - chainState->ChainState.isFetchingAtHead - ) { - let (target, block) = chainState->ChainState.fetchedTo + // Report the milestone this response brought the fetch frontier to, once. + // The chain reaches it again every time the head moves and it catches up, so + // what keeps the line off the log is the chain having already reported it, + // not the transition — which re-arms on every advance. + switch chainState->ChainState.isFetchingAtHead + ? chainState->ChainState.takeFetchedTo + : None { + | None => () + | Some((target, block)) => // What the chain is waiting on, which is not itself. A held process is // waiting on chains it doesn't drive, so it says so even when it drives // only one — how the run is split is not the reader's problem. @@ -296,7 +295,7 @@ and applyQueryResponse = ( state->IndexerState.isHoldingRealtime || state->IndexerState.chainStates->Dict.keysToArray->Array.length > 1 ) { - " Waiting for the other chains." + " Waiting for other chains." } else { "" } diff --git a/packages/envio/src/ChainState.res b/packages/envio/src/ChainState.res index 560816d0a..65765e243 100644 --- a/packages/envio/src/ChainState.res +++ b/packages/envio/src/ChainState.res @@ -58,6 +58,9 @@ type t = { mutable blockRangeFetchCount: float, mutable blockRangeFetchedEvents: float, mutable blockRangeFetchedBlocks: float, + // What this chain last reported reaching. The head moves, so a chain reaches + // it again on every catch-up; only a new milestone is worth a line. + mutable reportedFetchedTo: option, mutable reorgCount: int, mutable reorgDetectedBlock: option, mutable rollbackTargetBlock: option, @@ -171,6 +174,7 @@ let make = ( blockRangeFetchCount: 0., blockRangeFetchedEvents: 0., blockRangeFetchedBlocks: 0., + reportedFetchedTo: None, reorgCount: 0, reorgDetectedBlock: None, rollbackTargetBlock: None, @@ -642,11 +646,11 @@ let dispatch = ( // --- Derived (pure). --- -// Where the fetch frontier has just landed, for the line that reports it. Below -// the reorg threshold a chain can only fetch the finalized range, so reaching -// the end of it is not reaching the head — the rest opens up once the indexer -// enters the threshold. -let fetchedTo = (cs: t) => +%%private( + // Where the fetch frontier has just landed. Below the reorg threshold a chain + // can only fetch the finalized range, so reaching the end of it is not + // reaching the head — the rest opens up once the indexer enters the threshold. + let fetchedTo = (cs: t) => switch cs.fetchState.endBlock { | Some(endBlock) if endBlock <= cs.fetchState.knownHeight - cs.fetchState.blockLag => ( "the end block", @@ -657,6 +661,21 @@ let fetchedTo = (cs: t) => ? ("the chain head", cs.fetchState.knownHeight) : ("the safe block", cs.fetchState.knownHeight - cs.fetchState.blockLag) } +) + +// What this chain has just reached, the first time it reaches it. `None` once +// it has been reported: a chain catches up to a moving head over and over, and +// the milestone is the same one every time. A new one — the head past the +// threshold, an end block — is a line of its own. +let takeFetchedTo = (cs: t) => { + let (target, block) = cs->fetchedTo + if cs.reportedFetchedTo == Some(target) { + None + } else { + cs.reportedFetchedTo = Some(target) + Some((target, block)) + } +} let hasProcessedToEndblock = (cs: t) => { let {committedProgressBlockNumber, fetchState} = cs diff --git a/packages/envio/src/ChainState.resi b/packages/envio/src/ChainState.resi index f8148641f..fc5361bb8 100644 --- a/packages/envio/src/ChainState.resi +++ b/packages/envio/src/ChainState.resi @@ -123,7 +123,7 @@ let toChainBeforeBatch: (t, ~isRealtime: bool) => Batch.chainBeforeBatch let isReadyToEnterReorgThresholdAfterBatch: (t, ~batch: Batch.t) => bool // Derived (pure). -let fetchedTo: t => (string, int) +let takeFetchedTo: t => option<(string, int)> let hasProcessedToEndblock: t => bool let isDurablyCaughtUp: t => bool let getHighestBlockBelowThreshold: t => int From b0369c9700630e112f6d79522f5a343ba4929336 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 18 Sep 2026 14:58:22 +0000 Subject: [PATCH 37/61] Say nothing about indexes a schema never declared Finalizing a schema with no indexes logged that all zero of them were in place and then that zero of them had been committed. The line that matters there is the indexer reporting itself ready, which finalization already logs; these two only have something to say when there are indexes to report on. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- .../lib_tests/FinalizeIndexLogging_test.res | 80 +++++++++++++++++++ packages/envio/src/PgStorage.res | 22 +++-- 2 files changed, 95 insertions(+), 7 deletions(-) create mode 100644 packages/envio-tests/test/lib_tests/FinalizeIndexLogging_test.res diff --git a/packages/envio-tests/test/lib_tests/FinalizeIndexLogging_test.res b/packages/envio-tests/test/lib_tests/FinalizeIndexLogging_test.res new file mode 100644 index 000000000..107edb7cf --- /dev/null +++ b/packages/envio-tests/test/lib_tests/FinalizeIndexLogging_test.res @@ -0,0 +1,80 @@ +open Vitest + +// Finalizing a schema that declares no indexes has nothing to say about them. +// The indexer reporting itself ready is `FinalizeBackfill`'s line, not this +// one's, so a schema with no indexes should leave no trace here. +let sql = PgStorage.makeClient() +let pgSchema = TestPgSchema.make() +let config = TestConfig.make() + +Async.afterAll(async () => { + await sql->TestPgSchema.drop(~pgSchema) + await sql->Postgres.endSql +}) + +let loggedMessages = async path => + switch await NodeJs.Fs.Promises.readFile(~filepath=NodeJs.Path.resolve([path]), ~encoding=Utf8) { + | contents => + contents + ->String.trim + ->String.split("\n") + ->Array.filterMap(line => + switch line->JSON.parseOrThrow->JSON.Decode.object { + | Some(fields) => fields->Dict.get("msg")->Option.flatMap(JSON.Decode.string) + | None => None + } + ) + | exception _ => [] + } + +describe("Finalizing a schema with no indexes", () => { + Async.it("Says nothing about the indexes it didn't have to build", async t => { + let storage = PgStorage.make( + ~sql, + ~pgHost=Env.Db.host, + ~pgSchema, + ~pgPort=Env.Db.port, + ~pgUser=Env.Db.user, + ~pgDatabase=Env.Db.database, + ~pgPassword=Env.Db.password, + ~isHasuraEnabled=false, + ~ecosystem=Evm, + ) + let _ = await storage.initialize( + ~chainConfigs=config.chainMap->ChainMap.values, + ~contractMapping=config.contractMapping, + ~entities=config.userEntities, + ~enums=config.allEnums->Array.concat([ + EntityHistory.RowAction.config->Table.fromGenericEnumConfig, + ]), + ~envioInfo=JSON.Encode.object(Dict.make()), + ) + + let path = `${NodeJs.Process.cwd()}/lib/envio-finalize-indexes-${Date.now()->Float.toString}.log` + Logging.setLogger( + Logging.makeLogger( + ~logStrategy=FileOnly, + ~logFilePath=path, + ~defaultFileLogLevel=#info, + ~userLogLevel=#info, + ), + ) + + await storage.finalizeBackfill( + ~entities=config.userEntities, + ~chainIds=config.chainMap->ChainMap.keys, + ~readyAt=Date.make(), + ) + Logging.info("done") + + let rec until = async deadline => + switch await loggedMessages(path) { + | messages if messages->Array.includes("done") || Date.now() > deadline => messages + | _ => + await Utils.delay(50) + await until(deadline) + } + + t.expect(await until(Date.now() +. 3000.)).toStrictEqual(["done"]) + }) +}) diff --git a/packages/envio/src/PgStorage.res b/packages/envio/src/PgStorage.res index 5c1508765..026a3a5e9 100644 --- a/packages/envio/src/PgStorage.res +++ b/packages/envio/src/PgStorage.res @@ -2264,13 +2264,17 @@ let make = ( } switch missing { - | [] => + // A schema that declares no indexes has nothing to say about them, and one + // whose indexes are all in place says it once. Either way the line that + // matters is the indexer reporting itself ready, which finalization logs. + | [] if schemaIndexes->Utils.Array.notEmpty => Logging.info({ "storage": storageName, "msg": `All ${schemaIndexes ->Array.length ->Int.toString} schema indexes are already in place. Marking the indexer ready.`, }) + | [] => () | _ => Logging.info({ "storage": storageName, @@ -2318,12 +2322,16 @@ let make = ( } }) - Logging.info({ - "storage": storageName, - "msg": `Committed ${missing - ->Array.length - ->Int.toString} schema indexes and the ready timestamp in ${timeRef->formatSeconds}s.`, - }) + // Only when something was built: the wait this closes is the index build, + // and the stamp on its own is not one anybody waited through. + if missing->Utils.Array.notEmpty { + Logging.info({ + "storage": storageName, + "msg": `Committed ${missing + ->Array.length + ->Int.toString} schema indexes and the ready timestamp in ${timeRef->formatSeconds}s.`, + }) + } } let setOrThrow = ( From 6bf52a86ea485cace7a27de9b39496e0c0b4f64e Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 18 Sep 2026 15:20:43 +0000 Subject: [PATCH 38/61] Don't let a held process conclude it has caught up MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A chain held below the reorg threshold keeps the lag that pins its fetch frontier to the safe block. Where the head sits within a reorg depth of the start block, that lagged head is below anything the chain would ever fetch, so it reads as settled at a head it never reached — and the run, released on that, finalized and reported itself ready having indexed nothing. What a held process may conclude from a live reading is only that it has fetched as far as the lag allows, which is what it now reports as having arrived. Catching up is a conclusion it draws once released, except on a resume, where it is read from what was persisted and the hold distorts nothing. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KrhL53BgXb1sHUeDBJx5PH --- .../test/SupervisedRealtime_test.res | 29 +++++++++-------- packages/envio/src/ChainState.res | 8 +++++ packages/envio/src/ChainState.resi | 1 + packages/envio/src/CrossChainState.res | 31 +++++++++++++++---- 4 files changed, 48 insertions(+), 21 deletions(-) diff --git a/packages/envio-tests/test/SupervisedRealtime_test.res b/packages/envio-tests/test/SupervisedRealtime_test.res index e6ebb2b06..22587048b 100644 --- a/packages/envio-tests/test/SupervisedRealtime_test.res +++ b/packages/envio-tests/test/SupervisedRealtime_test.res @@ -184,37 +184,36 @@ describe("A supervised worker on a chain with a reorg depth", () => { // run held until its chains reach the head would be waiting on progress that // only the transition itself makes reachable. rollbackScenario->Scenario.it( - "Arrives at the safe block, which is as far as it can fetch before the threshold", + "Holds the transition that would let it fetch past the safe block", ~sources=[{chain: 1}, {chain: 137}], ~holdRealtime=true, async (~t, ~indexer, ~source) => { let {sql, pgSchema} = indexer.pg - let head = 300 - // A response short of the head by the reorg depth: the whole finalized - // range, and all this chain may fetch until it enters the threshold. + // Each chain fetches the whole finalized range, which is all it may fetch + // until the indexer enters the reorg threshold. let catchUpToSafeBlock = (~source: MockSource.t) => { - source.resolveGetHeightOrThrow(head) - source.resolveGetItemsOrThrow([], ~latestFetchedBlockNumber=head - 200) + source.resolveGetHeightOrThrow(300) + source.resolveGetItemsOrThrow([], ~latestFetchedBlockNumber=100) } catchUpToSafeBlock(~source=source(1)) catchUpToSafeBlock(~source=source(137)) await indexer.waitUntilIdle() t.expect( - await readyAtByChainId(~sql, ~pgSchema), + ( + await indexer.metric("envio_reorg_threshold"), + await readyAtByChainId(~sql, ~pgSchema), + ), ~message="Both chains are as far as they can fetch, and the run has not said so", - ).toEqual([("1", None), ("137", None)]) + ).toEqual(([{value: "0", labels: Dict.make()}], [("1", None), ("137", None)])) indexer.releaseRealtime() - await indexer.waitUntilReady() + await indexer.waitUntilIdle() t.expect( - (await readyAtByChainId(~sql, ~pgSchema))->Array.map(((chainId, readyAt)) => ( - chainId, - readyAt->Option.isSome, - )), - ~message="Released, the run enters the threshold and stamps its chains", - ).toEqual([("1", true), ("137", true)]) + await indexer.metric("envio_reorg_threshold"), + ~message="Released, the run enters the threshold and the rest opens up", + ).toEqual([{value: "1", labels: Dict.make()}]) }, ) }) diff --git a/packages/envio/src/ChainState.res b/packages/envio/src/ChainState.res index 65765e243..1c69d63ec 100644 --- a/packages/envio/src/ChainState.res +++ b/packages/envio/src/ChainState.res @@ -1086,6 +1086,14 @@ let toChainBeforeBatch = (cs: t, ~isRealtime): Batch.chainBeforeBatch => { // Whether the chain's post-batch fetch frontier is ready to cross into the reorg // threshold, using the batch's progressed frontier when this chain advanced. +// The same question asked of where the chain stands now rather than of where a +// batch would leave it. Entering the threshold is what lifts the pre-threshold +// lag, so a chain waiting to enter it has fetched as far as it can. +let isReadyToEnterReorgThreshold = (cs: t) => + cs.fetchState->FetchState.isReadyToEnterReorgThreshold( + ~tolerance=cs.reorgThresholdReadyTolerance, + ) + let isReadyToEnterReorgThresholdAfterBatch = (cs: t, ~batch: Batch.t) => { let fetchState = switch batch.progressedChainsById->ChainId.Dict.dangerouslyGetNonOption( cs.fetchState.chainId, diff --git a/packages/envio/src/ChainState.resi b/packages/envio/src/ChainState.resi index fc5361bb8..a3212ffe6 100644 --- a/packages/envio/src/ChainState.resi +++ b/packages/envio/src/ChainState.resi @@ -120,6 +120,7 @@ let dispatch: ( let toMetrics: t => Metrics.chainMetrics let toChainMetadata: t => InternalTable.Chains.metaFields let toChainBeforeBatch: (t, ~isRealtime: bool) => Batch.chainBeforeBatch +let isReadyToEnterReorgThreshold: t => bool let isReadyToEnterReorgThresholdAfterBatch: (t, ~batch: Batch.t) => bool // Derived (pure). diff --git a/packages/envio/src/CrossChainState.res b/packages/envio/src/CrossChainState.res index 1ee37e326..88a2eaff3 100644 --- a/packages/envio/src/CrossChainState.res +++ b/packages/envio/src/CrossChainState.res @@ -53,16 +53,27 @@ let releaseRealtime = (crossChainState: t) => crossChainState.holdRealtime = fal let isHoldingRealtime = (crossChainState: t) => crossChainState.holdRealtime -// Whether this process has got as far as it can without the run's leave: its -// chains have caught up, or it resumed already realtime. What a supervisor -// reads to decide that a split run may go realtime as one. +// Whether this process has got as far as it can without the run's leave. What a +// supervisor reads to decide that a split run may go realtime as one. +// +// Three ways to have arrived, because a chain can be as far along as it can get +// in three different states. Its chains have caught up; or it resumed already +// realtime; or every chain is waiting to enter the reorg threshold, which is as +// far as one can fetch while the pre-threshold lag holds it at the safe block — +// entering the threshold is what lifts that lag, so a run held until its chains +// reached the head would be holding back the transition that gets them there. // // The process's own conclusion rather than a reading a supervisor reassembles: // a chain committed at what was the head and resumed once the head had moved on // has arrived, and no live reading of it can say so — which is the same reason // `markCaughtUpOnResume` decides before any source request. let hasArrivedAtHead = (crossChainState: t) => - crossChainState.isCaughtUp || crossChainState.isRealtime + crossChainState.isCaughtUp || + crossChainState.isRealtime || { + let chainStates = crossChainState.chainStates->Dict.valuesToArray + chainStates->Utils.Array.notEmpty && + chainStates->Array.every(ChainState.isReadyToEnterReorgThreshold) + } // Resolve a chain's state by id. The id always comes from `chainIds`, which is // derived from `chainStates`, so the entry is guaranteed present. @@ -188,7 +199,10 @@ let applyBatchProgress = (crossChainState: t, ~batch: Batch.t, ~blockTimestampNa } crossChainState.isCaughtUp = - crossChainState.isCaughtUp || (crossChainState->nextItemIsNone && everyChainCaughtUp.contents) + crossChainState.isCaughtUp || + (!crossChainState.holdRealtime && + crossChainState->nextItemIsNone && + everyChainCaughtUp.contents) } // Every chain has buffered up to its head (or endblock) with nothing @@ -212,8 +226,13 @@ let isSettledAtHead = (crossChainState: t) => { } // Enter the FinalizingIndexes phase without a batch, for the resume above. +// Not while the run holds this process back: the hold keeps the pre-threshold +// lag in place, and a chain that has fetched to a lagged head it was never +// going to get past reads as settled without having indexed anything. What a +// held process may conclude about where it stands, it concludes from what was +// persisted — see `markCaughtUpOnResume`, which the hold leaves alone. let markCaughtUpIfSettled = (crossChainState: t) => - if crossChainState->isSettledAtHead { + if !crossChainState.holdRealtime && crossChainState->isSettledAtHead { crossChainState.isCaughtUp = true } From 4f6d76e1eb7da724952ee995bff5bfe6e638ca0b Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 21 Sep 2026 09:09:18 +0000 Subject: [PATCH 39/61] Tell a worker which command started the run A worker parses the project's files itself, and they say nothing about whether `envio dev` or `envio start` was run. Its config came back with `isDev` false however the run began, so a split dev run's workers dropped `keepProcessAlive` and exited at their end blocks, leaving the console the supervisor was still serving short of their chains, and pointing at `envio start -r` when they had something to say about resetting. What the command decided is now the one thing that rides along with the fork, `isDev` next to the chain selection that was already there, and re-applying both is what makes a worker's config its supervisor's. Handing the whole config over instead would mean carrying every contract's ABI in the spawn environment. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_013saxnEKXJJXrZCkodmDMzE --- .../envio-tests/test/SupervisorFork_test.res | 9 +++- .../test/lib_tests/Supervisor_test.res | 49 +++++++++++++++---- packages/envio/src/Config.res | 10 ++-- packages/envio/src/Main.res | 12 ++--- packages/envio/src/Supervisor.res | 13 +++-- packages/envio/src/Worker.res | 6 +++ 6 files changed, 75 insertions(+), 24 deletions(-) diff --git a/packages/envio-tests/test/SupervisorFork_test.res b/packages/envio-tests/test/SupervisorFork_test.res index 542482f63..65f9a9e17 100644 --- a/packages/envio-tests/test/SupervisorFork_test.res +++ b/packages/envio-tests/test/SupervisorFork_test.res @@ -17,6 +17,7 @@ let forkFixture = ( ~maxConnections=2, ~workerIndex=0, ~holdRealtime=false, + ~isDev=false, ~pipeOutput=false, ~onOutput=?, ) => @@ -24,6 +25,7 @@ let forkFixture = ( {chainIds: chainIds->Array.map(ChainId.fromInt), maxConnections}, ~workerIndex, ~holdRealtime, + ~isDev, ~entryPath=fixturePath, ~pipeOutput, ~onOutput?, @@ -36,6 +38,7 @@ describe("Supervisor.fork", () => { ~maxConnections=3, ~workerIndex=1, ~holdRealtime=true, + ~isDev=true, ) let report = await Promise.make( @@ -52,8 +55,10 @@ describe("Supervisor.fork", () => { t.expect(report).toStrictEqual({ // Everything the supervisor decided, in the environment: a worker needs it - // before it can load its own config, so it can't arrive as a message. - workerConfig: `{"chainIds":[1,137],"holdRealtime":true}`, + // before it can load its own config, so it can't arrive as a message. The + // project's files say nothing about which command started the run, which + // is why `isDev` is among them. + workerConfig: `{"chainIds":[1,137],"holdRealtime":true,"isDev":true}`, maxConnections: "3", logFile: Supervisor.logFilePath(~workerIndex=1), // Proof the channel clones rather than stringifies: a JSON round trip diff --git a/packages/envio-tests/test/lib_tests/Supervisor_test.res b/packages/envio-tests/test/lib_tests/Supervisor_test.res index ecd2af3ac..9816a7345 100644 --- a/packages/envio-tests/test/lib_tests/Supervisor_test.res +++ b/packages/envio-tests/test/lib_tests/Supervisor_test.res @@ -138,23 +138,50 @@ describe("Supervisor.planForRun", () => { }) describe("Supervisor worker plumbing", () => { - it("Narrows a config it parsed itself to the chains it was given", t => { - let configJson = JSON.Object( - Dict.fromArray([("name", JSON.String("indexer")), ("isolatedChains", JSON.Null)]), + // A worker's own parse of the project's files is what `envio start` would + // produce: no chain selection, and no dev run. Both are the command's to say. + it("Restores what the command decided onto a config it parsed itself", t => { + let parsedItself = JSON.Object( + Dict.fromArray([ + ("name", JSON.String("indexer")), + ("isolatedChains", JSON.Null), + ("isDev", JSON.Boolean(false)), + ]), ) t.expect( - configJson->Config.withIsolatedChains(~chainIds=[1, 137]->Array.map(ChainId.fromInt)), + parsedItself->Config.withCommandFields( + ~chainIds=[1, 137]->Array.map(ChainId.fromInt), + ~isDev=true, + ), ).toStrictEqual( JSON.Object( Dict.fromArray([ ("name", JSON.String("indexer")), ("isolatedChains", JSON.Array([JSON.Number(1.), JSON.Number(137.)])), + ("isDev", JSON.Boolean(true)), ]), ), ) }) + // `envio dev` keeps the run up once every chain has reached its end block, so + // the console it serves stays whole. A worker that read the run as a plain + // `envio start` would exit there and take its chains out of that console. + it("Keeps a dev run a dev run in the process that drives part of it", t => { + let devRun = Core.fromUserApi(~schema=perChain, configYaml).config->JSON.parseOrThrow + let workerConfig = + devRun + ->Config.withCommandFields(~chainIds=[1->ChainId.fromInt], ~isDev=true) + ->Config.fromPublic + + t.expect(( + workerConfig.isDev, + workerConfig.isolated, + workerConfig.chainMap->ChainMap.keys, + )).toStrictEqual((true, true, [1->ChainId.fromInt])) + }) + it("Gives every worker a log file of its own", t => { t.expect([ Supervisor.logFilePath(~workerIndex=0, ~path="logs/envio.log"), @@ -196,7 +223,9 @@ describe("Config.logContext", () => { describe("Worker.detect", () => { it("Counts as a worker only when forked with the variable and a channel", t => { - let forked = Dict.fromArray([(Worker.envVar, `{"chainIds":[137],"holdRealtime":true}`)]) + let forked = Dict.fromArray([ + (Worker.envVar, `{"chainIds":[137],"holdRealtime":true,"isDev":true}`), + ]) t.expect([ Worker.detect(~env=forked, ~hasChannel=true), // A copy of the variable left in a shell, or a process manager forking @@ -204,7 +233,7 @@ describe("Worker.detect", () => { Worker.detect(~env=forked, ~hasChannel=false), Worker.detect(~env=Dict.make(), ~hasChannel=true), ]).toStrictEqual([ - Some({Worker.chainIds: [137->ChainId.fromInt], holdRealtime: true}), + Some({Worker.chainIds: [137->ChainId.fromInt], holdRealtime: true, isDev: true}), None, None, ]) @@ -215,7 +244,7 @@ describe("Worker.detect", () => { it("Names the variable when its value isn't a worker config", t => { t->Vitest.toThrowErrorEqual( () => Worker.detect(~env=Dict.fromArray([(Worker.envVar, "137")]), ~hasChannel=true), - `Invalid ENVIO_INTERNAL_WORKER: Failed parsing at root. Reason: Expected { chainIds: array; holdRealtime: boolean | undefined; }, received 137. It is set by an indexer supervisor for the processes it forks, and isn't meant to be set by hand.`, + `Invalid ENVIO_INTERNAL_WORKER: Failed parsing at root. Reason: Expected { chainIds: array; holdRealtime: boolean | undefined; isDev: boolean; }, received 137. It is set by an indexer supervisor for the processes it forks, and isn't meant to be set by hand.`, ) }) @@ -224,10 +253,12 @@ describe("Worker.detect", () => { it("Takes a config without the hold as one that doesn't wait", t => { t.expect( Worker.detect( - ~env=Dict.fromArray([(Worker.envVar, `{"chainIds":[1]}`)]), + ~env=Dict.fromArray([(Worker.envVar, `{"chainIds":[1],"isDev":false}`)]), ~hasChannel=true, ), - ).toStrictEqual(Some({Worker.chainIds: [1->ChainId.fromInt], holdRealtime: false})) + ).toStrictEqual( + Some({Worker.chainIds: [1->ChainId.fromInt], holdRealtime: false, isDev: false}), + ) }) }) diff --git a/packages/envio/src/Config.res b/packages/envio/src/Config.res index 67071ce17..512aab0b4 100644 --- a/packages/envio/src/Config.res +++ b/packages/envio/src/Config.res @@ -1237,9 +1237,12 @@ let prime = (json: JSON.t): unit => { cached := None } -// Narrows a public config to the chains one process drives. The supervisor -// plans the split; each worker applies the plan to the config it parsed itself. -let withIsolatedChains = (json: JSON.t, ~chainIds) => +// The fields `envio start` and `envio dev` set on a public config from the +// command they were given rather than from the project's files: which chains +// the process drives, and whether the run is a dev run. A worker parses the +// same files its supervisor did, so these are the only two it cannot arrive at +// on its own, and re-applying them is what makes its config the supervisor's. +let withCommandFields = (json: JSON.t, ~chainIds, ~isDev) => switch json->JSON.Decode.object { | Some(fields) => { let narrowed = fields->Dict.copy @@ -1247,6 +1250,7 @@ let withIsolatedChains = (json: JSON.t, ~chainIds) => "isolatedChains", chainIds->S.reverseConvertToJsonOrThrow(S.array(ChainId.schema)), ) + narrowed->Dict.set("isDev", JSON.Encode.bool(isDev)) JSON.Object(narrowed) } | None => JsError.throwWithMessage("Invalid indexer config: not an object") diff --git a/packages/envio/src/Main.res b/packages/envio/src/Main.res index 9b353ee3c..1468908d3 100644 --- a/packages/envio/src/Main.res +++ b/packages/envio/src/Main.res @@ -561,12 +561,12 @@ let start = async ( ~exitAfterFirstEventBlock=false, ~patchConfig: option<(Config.t, HandlerRegister.registrationsByChainId) => Config.t>=?, ) => { - // A worker parses the same config its supervisor did and narrows it to the - // chains it was handed, rather than being told what to index: the storage it - // resumes refuses a config that disagrees with the one the run was created - // from, which is a stronger guarantee than a handover could give. - Worker.config->Option.forEach(({chainIds}) => - Config.prime(Config.getPublicConfigJson()->Config.withIsolatedChains(~chainIds)) + // A worker parses the project's files itself rather than being handed the + // config: a public config carries every contract's ABI, which is far more + // than a spawn environment should. What the command decided rides along + // instead, and re-applying it is what makes the two configs the same one. + Worker.config->Option.forEach(({chainIds, isDev}) => + Config.prime(Config.getPublicConfigJson()->Config.withCommandFields(~chainIds, ~isDev)) ) let config = Config.load() switch isTest ? None : Supervisor.planForRun(~config) { diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index e93566760..e36a60002 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -118,6 +118,9 @@ let fork = ( // every chain resumed already caught up: there is nothing left to wait for, // and a barrier nobody can open would hold the run forever. ~holdRealtime, + // Which command the run was started by, which a worker's own parse of the + // project's files can't tell it. + ~isDev, // The entry this process was itself started from, so a worker is the same // program as its supervisor however the package was installed. ~entryPath=NodeJs.Process.argv->Array.getUnsafe(1), @@ -129,9 +132,11 @@ let fork = ( let env = NodeJs.Process.process.env->Dict.copy env->Dict.set( Worker.envVar, - {Worker.chainIds: worker.chainIds, holdRealtime}->S.reverseConvertToJsonStringOrThrow( - Worker.configSchema, - ), + { + Worker.chainIds: worker.chainIds, + holdRealtime, + isDev, + }->S.reverseConvertToJsonStringOrThrow(Worker.configSchema), ) // The worker's slice of the budget. Read when the worker's own Env module // loads, which is why it rides in the spawn environment rather than a message. @@ -323,7 +328,7 @@ let run = async (~config: Config.t, ~workers: array, ~reset) => { ) let group = { running: workers->Array.mapWithIndex((worker, workerIndex) => - worker->fork(~workerIndex, ~holdRealtime, ~pipeOutput=shouldUseTui) + worker->fork(~workerIndex, ~holdRealtime, ~isDev=config.isDev, ~pipeOutput=shouldUseTui) ), stopping: false, } diff --git a/packages/envio/src/Worker.res b/packages/envio/src/Worker.res index d04504545..7fb3baa0d 100644 --- a/packages/envio/src/Worker.res +++ b/packages/envio/src/Worker.res @@ -17,11 +17,17 @@ type config = { // process are still backfilling, and an indexer goes realtime as a whole or // not at all. Cleared by the supervisor's `ReleaseRealtime`. holdRealtime: bool, + // Which command the run was started by. A worker re-parses the project's + // files, which say nothing about that, so it can only be told — and a worker + // that took a dev run for a plain one would exit at its end block and leave + // the console it was still serving with a chain missing. + isDev: bool, } let configSchema = S.object((s): config => { chainIds: s.field("chainIds", S.array(ChainId.schema)), holdRealtime: s.fieldOr("holdRealtime", S.bool, false), + isDev: s.field("isDev", S.bool), }) // Read as this module loads, which is before anything that could catch a bare From 7230c20db48b566f949421b466c1ac76580f4132 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 21 Sep 2026 09:09:27 +0000 Subject: [PATCH 40/61] Keep a worker's stderr on stderr Both of a worker's streams were read into one sink and written back out with `Console.log`, so anything it reported on stderr arrived on the supervisor's stdout. A run whose stderr is redirected somewhere of its own saw nothing from the processes doing the indexing. Each stream now keeps the one it was written to. Ink patches both, so the lines still stay out of the frame a run draws. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_013saxnEKXJJXrZCkodmDMzE --- .../envio-tests/test/SupervisorFork_test.res | 23 ++++++++++++------- packages/envio/src/Supervisor.res | 14 +++++++---- 2 files changed, 25 insertions(+), 12 deletions(-) diff --git a/packages/envio-tests/test/SupervisorFork_test.res b/packages/envio-tests/test/SupervisorFork_test.res index 65f9a9e17..5b887dd9b 100644 --- a/packages/envio-tests/test/SupervisorFork_test.res +++ b/packages/envio-tests/test/SupervisorFork_test.res @@ -20,6 +20,7 @@ let forkFixture = ( ~isDev=false, ~pipeOutput=false, ~onOutput=?, + ~onErrorOutput=?, ) => Supervisor.fork( {chainIds: chainIds->Array.map(ChainId.fromInt), maxConnections}, @@ -29,6 +30,7 @@ let forkFixture = ( ~entryPath=fixturePath, ~pipeOutput, ~onOutput?, + ~onErrorOutput?, ) describe("Supervisor.fork", () => { @@ -135,9 +137,11 @@ describe("Supervisor.readLines", () => { describe("Supervisor.fork output", () => { // A worker writing straight to the terminal tears the frame its supervisor - // draws: ink only knows about the lines its own process logs. Sorted, since - // stdout and stderr are two pipes and neither waits for the other. - Async.it("Hands the supervisor every line a worker writes, whole", async t => { + // draws: ink only knows about the lines its own process logs. Each line keeps + // the stream it was written to, so redirecting the run's stderr still catches + // what its workers wrote there. Sorted, since stdout and stderr are two pipes + // and neither waits for the other. + Async.it("Hands the supervisor every line a worker writes, on its own stream", async t => { NodeJs.Process.process.env->Dict.set("FAKE_WORKER", "print") let lines = [] let group: Supervisor.group = { @@ -145,17 +149,20 @@ describe("Supervisor.fork output", () => { forkFixture( ~chainIds=[1], ~pipeOutput=true, - ~onOutput=line => lines->Array.push(line)->ignore, + ~onOutput=line => lines->Array.push(("stdout", line))->ignore, + ~onErrorOutput=line => lines->Array.push(("stderr", line))->ignore, ), ], stopping: false, } let _ = await group->Supervisor.awaitExit - t.expect(lines->Array.toSorted(String.compare)).toStrictEqual([ - "first line", - "from stderr", - "second line", + t.expect( + lines->Array.toSorted(((_, a), (_, b)) => String.compare(a, b)), + ).toStrictEqual([ + ("stdout", "first line"), + ("stderr", "from stderr"), + ("stdout", "second line"), ]) }) }) diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index e36a60002..d0571b893 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -128,6 +128,7 @@ let fork = ( // them write to the terminal behind the frame's back. ~pipeOutput=false, ~onOutput=Console.log, + ~onErrorOutput=Console.error, ) => { let env = NodeJs.Process.process.env->Dict.copy env->Dict.set( @@ -158,12 +159,17 @@ let fork = ( }, ) if pipeOutput { - // Both streams become one stream of lines: the supervisor logs them the way - // it logs its own, which is the only way ink can keep them out of its frame. - [child->NodeJs.ChildProcess.stdout, child->NodeJs.ChildProcess.stderr]->Array.forEach(stream => + // The supervisor writes a worker's lines the way it writes its own, which is + // the only way ink can keep them out of its frame. Each stream keeps the one + // it was written to, so a worker's errors stay on stderr for whoever is + // redirecting it. + [ + (child->NodeJs.ChildProcess.stdout, onOutput), + (child->NodeJs.ChildProcess.stderr, onErrorOutput), + ]->Array.forEach(((stream, onLine)) => switch stream->Null.toOption { | Some(stream) => { - let (read, flush) = readLines(~onLine=onOutput) + let (read, flush) = readLines(~onLine) stream->NodeJs.ChildProcess.setEncoding("utf8") stream->NodeJs.ChildProcess.onData(read) stream->NodeJs.ChildProcess.onEnd(flush) From 1d442a2deb08a6099ac9905ee2d197760b36ff80 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 21 Sep 2026 09:09:38 +0000 Subject: [PATCH 41/61] Drop a comment about state the server doesn't hold It described the `EnvioGlobal` slots and came along with the server when it moved out of Main, where the accessors it was about stayed. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_013saxnEKXJJXrZCkodmDMzE --- packages/envio/src/Server.res | 3 --- 1 file changed, 3 deletions(-) diff --git a/packages/envio/src/Server.res b/packages/envio/src/Server.res index 39641c9c6..53137e622 100644 --- a/packages/envio/src/Server.res +++ b/packages/envio/src/Server.res @@ -77,9 +77,6 @@ let stateSchema = S.union([ })), ]) -// Runtime state lives in the process-wide `EnvioGlobal` record (shared -// across duplicate envio module instances); the slots are opaque there, so -// cast them to the real types here. let startServer = ( ~getMetrics: unit => option, ~envioVersion: string, From 4f6fb66147401252cf97d6b72dd9d955c7a5a026 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 21 Sep 2026 09:09:38 +0000 Subject: [PATCH 42/61] Say what --chain is for without explaining the split MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Its help had grown to explain how automatic splitting decides what it decides — the connection arithmetic, the budget a run needs before it splits at all — in the middle of the option for placing chains by hand. The option now says what it is for and what it requires. That a per-chain schema is indexed across several processes belongs to `envio start` itself, which is where it now is, in a paragraph of its own so the command list stays a list. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_013saxnEKXJJXrZCkodmDMzE --- packages/cli/CommandLineHelp.md | 8 +++++-- packages/cli/src/cli_args/clap_definitions.rs | 22 +++++++++---------- 2 files changed, 17 insertions(+), 13 deletions(-) diff --git a/packages/cli/CommandLineHelp.md b/packages/cli/CommandLineHelp.md index 2839157d3..bc578bb68 100644 --- a/packages/cli/CommandLineHelp.md +++ b/packages/cli/CommandLineHelp.md @@ -371,14 +371,18 @@ Setup database by dropping schema and then running migrations ## `envio start` -Start the indexer. Runs codegen automatically before launching so the on-disk types stay in sync with `config.yaml` and `schema.graphql` +Start the indexer. Runs codegen automatically before launching so the on-disk types stay in sync with `config.yaml` and `schema.graphql`. + +A schema whose entities are all per-chain is indexed across several processes, as many as `ENVIO_PG_MAX_CONNECTIONS` affords. List the busiest chains first in `config.yaml` to balance them. **Usage:** `envio start [OPTIONS]` ###### **Options:** * `-r`, `--restart` — Clear your database and restart indexing from scratch -* `--chain ` — Index only this chain, leaving the others to their own `envio start --chain` processes. Only needed to place the chains yourself: a plain `envio start` already splits them across processes, and manages those processes for you, whenever the schema's entities are all per-chain. `ENVIO_PG_MAX_CONNECTIONS` is the budget for the whole run and buys one process per two connections, so raising it from its default of 2 is what splits a run. A schema with an entity shared across chains, or a budget under 4, runs in one process as it always has. Repeat the flag for several chains. Requires a schema whose entities are all per-chain, created for every chain by `envio local db-migrate up` before any process starts. Assign each configured chain to exactly one process, and give each its own `ENVIO_INDEXER_PORT`. Each process builds the indexes for its own chains and reports them ready as they catch up, independently of the others +* `--chain ` — Index only this chain, leaving the others to their own `envio start --chain` processes. Repeat the flag for several chains. + + Only needed to place the chains yourself, across machines or under your own process manager. A plain `envio start` already splits a per-chain schema across processes and manages them for you. Requires a schema whose entities are all per-chain, created for every chain by `envio local db-migrate up` before any process starts. Assign each configured chain to exactly one process, and give each its own `ENVIO_INDEXER_PORT`. Each process reports its own chains ready as they catch up, independently of the others. diff --git a/packages/cli/src/cli_args/clap_definitions.rs b/packages/cli/src/cli_args/clap_definitions.rs index 583d32bb7..4c16e9988 100644 --- a/packages/cli/src/cli_args/clap_definitions.rs +++ b/packages/cli/src/cli_args/clap_definitions.rs @@ -57,6 +57,8 @@ pub enum CommandType { Local(LocalCommandTypes), ///Start the indexer. Runs codegen automatically before launching so the on-disk types stay in sync with `config.yaml` and `schema.graphql`. + /// + ///A schema whose entities are all per-chain is indexed across several processes, as many as `ENVIO_PG_MAX_CONNECTIONS` affords. List the busiest chains first in `config.yaml` to balance them. Start(StartArgs), ///Fetch raw Prometheus metrics from the running indexer's /metrics endpoint @@ -145,17 +147,15 @@ pub struct StartArgs { pub restart: bool, ///Index only this chain, leaving the others to their own `envio start --chain` processes. - ///Only needed to place the chains yourself: a plain `envio start` already splits them across - ///processes, and manages those processes for you, whenever the schema's entities are all - ///per-chain. `ENVIO_PG_MAX_CONNECTIONS` is the budget for the whole run and buys one process - ///per two connections, so raising it from its default of 2 is what splits a run. A schema - ///with an entity shared across chains, or a budget under 4, runs in one process as it always - ///has. - ///Repeat the flag for several chains. Requires a schema whose entities are all per-chain, - ///created for every chain by `envio local db-migrate up` before any process starts. - ///Assign each configured chain to exactly one process, and give each its own - ///`ENVIO_INDEXER_PORT`. Each process builds the indexes for its own chains and reports them - ///ready as they catch up, independently of the others. + ///Repeat the flag for several chains. + /// + ///Only needed to place the chains yourself, across machines or under your own process + ///manager. A plain `envio start` already splits a per-chain schema across processes and + ///manages them for you. + ///Requires a schema whose entities are all per-chain, created for every chain by + ///`envio local db-migrate up` before any process starts. Assign each configured chain to + ///exactly one process, and give each its own `ENVIO_INDEXER_PORT`. Each process reports its + ///own chains ready as they catch up, independently of the others. #[arg(long = "chain", value_name = "CHAIN_ID")] pub chains: Vec, } From 3ee5fd2a15a2fb004ab665d8f8e4542c95664d68 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 21 Sep 2026 11:10:55 +0000 Subject: [PATCH 43/61] Hold the barrier open only for workers still there to release The barrier read the run from the snapshots its workers had reported, and a snapshot outlives the process that sent it. A worker that died while its siblings were still held left its last reading behind, so the run could read as whole when it was a process short, and the release went into a channel Node had already closed. That comes back on the child's `error` event, which the supervisor reports as a worker failing to start, and which fails the run outright when it beats the exit it is really about. The channel is what says a worker is still there, since Node closes it before it reports the exit and a process on its way out reads as running everywhere else. Stopping the group now stops the barrier too, rather than leaving it asking about children that are being killed. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_013saxnEKXJJXrZCkodmDMzE --- .../envio-tests/test/SupervisorFork_test.res | 88 ++++++++++++++++++- .../test/helpers/IndexerRunner.res | 2 +- .../envio-tests/test/helpers/fakeWorker.mjs | 6 ++ .../test/lib_tests/Supervisor_test.res | 23 ----- packages/envio/src/Supervisor.res | 71 +++++++++------ packages/envio/src/bindings/NodeJs.res | 3 + 6 files changed, 140 insertions(+), 53 deletions(-) diff --git a/packages/envio-tests/test/SupervisorFork_test.res b/packages/envio-tests/test/SupervisorFork_test.res index 5b887dd9b..7f468073b 100644 --- a/packages/envio-tests/test/SupervisorFork_test.res +++ b/packages/envio-tests/test/SupervisorFork_test.res @@ -8,6 +8,7 @@ type fixtureReport = { maxConnections: string, logFile: string, startTime: Date.t, + hasArrivedAtHead: bool, } let fixturePath = `${NodeJs.Process.cwd()}/test/helpers/fakeWorker.mjs` @@ -66,6 +67,7 @@ describe("Supervisor.fork", () => { // Proof the channel clones rather than stringifies: a JSON round trip // would have turned this into a string. startTime: Date.fromTime(1700000000000.), + hasArrivedAtHead: false, }) }) }) @@ -82,6 +84,7 @@ describe("Supervisor.awaitExit", () => { let group: Supervisor.group = { running: [forkFixture(~chainIds=[1]), forkFixture(~chainIds=[137])], stopping: false, + releaseCheck: None, } t.expect(await outcome(group)).toStrictEqual(Ok(Supervisor.Finished)) @@ -94,6 +97,7 @@ describe("Supervisor.awaitExit", () => { let group: Supervisor.group = { running: [forkFixture(~chainIds=[1]), forkFixture(~chainIds=[137])], stopping: false, + releaseCheck: None, } group->Supervisor.stop @@ -106,7 +110,11 @@ describe("Supervisor.awaitExit", () => { let failing = forkFixture(~chainIds=[1]) NodeJs.Process.process.env->Dict.set("FAKE_WORKER", "linger") let lingering = forkFixture(~chainIds=[137]) - let group: Supervisor.group = {running: [failing, lingering], stopping: false} + let group: Supervisor.group = { + running: [failing, lingering], + stopping: false, + releaseCheck: None, + } // The survivor was taken down rather than left indexing half a schema. t.expect((await outcome(group), group.stopping, lingering.settled)).toStrictEqual(( @@ -154,6 +162,7 @@ describe("Supervisor.fork output", () => { ), ], stopping: false, + releaseCheck: None, } let _ = await group->Supervisor.awaitExit @@ -166,3 +175,80 @@ describe("Supervisor.fork output", () => { ]) }) }) + +describe("Supervisor.isRunAtHead", () => { + let untilReported = async (running: array) => { + let rec until = async deadline => + if !(running->Array.every(r => r.snapshot->Option.isSome)) && Date.now() < deadline { + await Utils.delay(10) + await until(deadline) + } + await until(Date.now() +. 3000.) + } + + let untilGone = async (r: Supervisor.running) => { + let rec until = async deadline => + if r.child->NodeJs.ChildProcess.connected && Date.now() < deadline { + await Utils.delay(10) + await until(deadline) + } + await until(Date.now() +. 3000.) + } + + let arrivingFixture = (~chainIds, ~mode) => { + NodeJs.Process.process.env->Dict.set("FAKE_WORKER", mode) + NodeJs.Process.process.env->Dict.set("FAKE_WORKER_ARRIVED", "1") + forkFixture(~chainIds) + } + + Async.it("Holds the run until every worker has arrived", async t => { + let arrived = arrivingFixture(~chainIds=[1], ~mode="linger") + NodeJs.Process.process.env->Dict.set("FAKE_WORKER_ARRIVED", "0") + let backfilling = forkFixture(~chainIds=[137]) + let group: Supervisor.group = { + running: [arrived, backfilling], + stopping: false, + releaseCheck: None, + } + await untilReported(group.running) + + // One worker still backfilling speaks for the whole run, and a worker that + // has yet to report drives chains nobody can see. + let readings = ( + group.running->Supervisor.isRunAtHead, + [arrived]->Supervisor.isRunAtHead, + []->Supervisor.isRunAtHead, + ) + group->Supervisor.stop + let _ = await group->Supervisor.awaitExit + + t.expect(readings).toStrictEqual((false, true, false)) + }) + + // Every worker said it had arrived, and then one of them was gone. Its + // snapshot outlives it, so a run read from the snapshots alone still looks + // whole, and the release would be sent into a channel Node had already + // closed — which comes back as the error a supervisor reports as a worker + // failing to start. + Async.it("Never releases a run a worker has left", async t => { + let lingering = arrivingFixture(~chainIds=[1], ~mode="linger") + let leaving = arrivingFixture(~chainIds=[137], ~mode="succeed-later") + let group: Supervisor.group = { + running: [lingering, leaving], + stopping: false, + releaseCheck: None, + } + await untilReported(group.running) + await untilGone(leaving) + + let readings = ( + group.running->Supervisor.isRunAtHead, + // Both snapshots are still there, and both of them still say arrived. + group.running->Array.filterMap(r => r.snapshot)->Array.length, + ) + group->Supervisor.stop + await untilGone(lingering) + + t.expect(readings).toStrictEqual((false, 2)) + }) +}) diff --git a/packages/envio-tests/test/helpers/IndexerRunner.res b/packages/envio-tests/test/helpers/IndexerRunner.res index 281996d94..9af7abef9 100644 --- a/packages/envio-tests/test/helpers/IndexerRunner.res +++ b/packages/envio-tests/test/helpers/IndexerRunner.res @@ -193,7 +193,7 @@ let run = async ( releaseCheck := Some( setInterval(() => - if [state->IndexerState.toMetrics]->Supervisor.isRunAtHead(~workerCount=1) { + if state->IndexerState.hasArrivedAtHead { releaseCheck.contents->Option.forEach(clearInterval) releaseCheck := None state->IndexerState.releaseRealtime diff --git a/packages/envio-tests/test/helpers/fakeWorker.mjs b/packages/envio-tests/test/helpers/fakeWorker.mjs index eae6b60cb..a7319481b 100644 --- a/packages/envio-tests/test/helpers/fakeWorker.mjs +++ b/packages/envio-tests/test/helpers/fakeWorker.mjs @@ -14,12 +14,18 @@ process.send({ // A Date survives only under structured-clone serialization, which is // what a metrics snapshot's timestamps need. startTime: new Date(1700000000000), + // The one reading the supervisor's barrier asks each worker for. + hasArrivedAtHead: process.env.FAKE_WORKER_ARRIVED === "1", }, }); if (mode === "succeed") process.exit(0); if (mode === "fail") process.exit(1); +// Ends the same way "succeed" does, but late enough for a supervisor to have +// taken its report and registered whatever it listens with. +if (mode === "succeed-later") setTimeout(() => process.exit(0), 60); + // Writes across chunk boundaries the way a real process does: a pipe hands the // supervisor whatever has been flushed, not whole lines. if (mode === "print") { diff --git a/packages/envio-tests/test/lib_tests/Supervisor_test.res b/packages/envio-tests/test/lib_tests/Supervisor_test.res index 9816a7345..acdc931ed 100644 --- a/packages/envio-tests/test/lib_tests/Supervisor_test.res +++ b/packages/envio-tests/test/lib_tests/Supervisor_test.res @@ -279,26 +279,3 @@ describe("Supervisor.syncCache", () => { t.expect(dumps.contents).toBe(2) }) }) - -describe("Supervisor.isRunAtHead", () => { - let snapshot = (~hasArrivedAtHead): Metrics.t => { - ...TestChainMetrics.emptySnapshot, - hasArrivedAtHead, - } - - it("Holds the run until every worker has arrived", t => { - t.expect([ - // A worker that hasn't reported yet drives chains nobody can see. Reading - // the run as arrived here would release it on a partial view. - [snapshot(~hasArrivedAtHead=true)]->Supervisor.isRunAtHead(~workerCount=2), - [ - snapshot(~hasArrivedAtHead=true), - snapshot(~hasArrivedAtHead=false), - ]->Supervisor.isRunAtHead(~workerCount=2), - [ - snapshot(~hasArrivedAtHead=true), - snapshot(~hasArrivedAtHead=true), - ]->Supervisor.isRunAtHead(~workerCount=2), - ]).toStrictEqual([false, false, true]) - }) -}) diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index d0571b893..4a8003643 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -193,10 +193,22 @@ let fork = ( // The forked workers of one run, and whether their supervisor is the one // taking them down. A stop it asked for is expected; every other way a worker // can end is a failure. -type group = {running: array, mutable stopping: bool} +type group = { + running: array, + mutable stopping: bool, + // The poll that asks whether the run may go realtime, while it is still + // asking. A group being taken down has nothing left to release. + mutable releaseCheck: option, +} + +let stopReleaseCheck = group => { + group.releaseCheck->Option.forEach(clearInterval) + group.releaseCheck = None +} let stop = group => { group.stopping = true + group->stopReleaseCheck group.running->Array.forEach(r => r.child->NodeJs.ChildProcess.kill("SIGTERM")->ignore) } @@ -284,13 +296,21 @@ let awaitExit = async (group): outcome => { // rate its workers report at: nothing changes in between. %%private(let releaseCheckIntervalMillis = 500) -// Whether a run holding its workers back may let them go: every worker has -// reported, and every one of them has got as far as it can on its own. What -// counts as arrived is the worker's own conclusion — the supervisor only asks -// each of them the question an unsplit run asks itself. -let isRunAtHead = (snapshots: array, ~workerCount) => - snapshots->Array.length === workerCount && snapshots->Array.every(snapshot => - snapshot.hasArrivedAtHead +// Whether a run holding its workers back may let them go: every worker is still +// there to be released, has reported, and has got as far as it can on its own. +// What counts as arrived is the worker's own conclusion, the supervisor only +// asking each of them the question an unsplit run asks itself. +// +// A worker that is gone leaves the run a process short, so there is nothing to +// release it into, and its last snapshot outlives it. The channel is what says +// so: Node closes it before it reports the exit, so a process on its way out +// still reads as running everywhere else, and the release sent to it comes back +// as the error a supervisor reports as a worker failing to start. +let isRunAtHead = (running: array) => + running->Utils.Array.notEmpty && + running->Array.every(r => + r.child->NodeJs.ChildProcess.connected && + r.snapshot->Option.mapOr(false, snapshot => snapshot.hasArrivedAtHead) ) // Runs the group: creates the schema for every chain, forks a worker per plan @@ -337,6 +357,7 @@ let run = async (~config: Config.t, ~workers: array, ~reset) => { worker->fork(~workerIndex, ~holdRealtime, ~isDev=config.isDev, ~pipeOutput=shouldUseTui) ), stopping: false, + releaseCheck: None, } let reported = () => group.running->Array.filterMap(r => r.snapshot) @@ -373,24 +394,18 @@ let run = async (~config: Config.t, ~workers: array, ~reset) => { // split run only the supervisor can see. Every worker is held until the last // one arrives, then released together, so the run switches over exactly as an // unsplit one does. - let releaseCheck = ref(None) - let stopReleaseCheck = () => { - releaseCheck.contents->Option.forEach(clearInterval) - releaseCheck := None - } if holdRealtime { - releaseCheck := - Some( - setInterval(() => - if reported()->isRunAtHead(~workerCount=group.running->Array.length) { - stopReleaseCheck() - group.running->Array.forEach(r => - r.child->NodeJs.ChildProcess.send(Worker.ReleaseRealtime)->ignore - ) - Logging.info("Every chain has reached the head. Switching the run to realtime.") - } - , releaseCheckIntervalMillis), - ) + group.releaseCheck = Some( + setInterval(() => + if group.running->isRunAtHead { + group->stopReleaseCheck + group.running->Array.forEach(r => + r.child->NodeJs.ChildProcess.send(Worker.ReleaseRealtime)->ignore + ) + Logging.info("Every chain has reached the head. Switching the run to realtime.") + } + , releaseCheckIntervalMillis), + ) } if shouldUseTui { @@ -408,9 +423,9 @@ let run = async (~config: Config.t, ~workers: array, ~reset) => { // is the exception, as it is for a single process: it keeps the final state // on screen until the terminal closes it. let outcome = await group->awaitExit - // Nothing left to release, and a display keeps this process alive long enough - // for the check to reach children that are gone. - stopReleaseCheck() + // A group that ended on its own was never stopped, and a display keeps this + // process alive long past the last worker the check was asking about. + group->stopReleaseCheck switch outcome { | Stopped => NodeJs.process->NodeJs.exitWithCode(Success) diff --git a/packages/envio/src/bindings/NodeJs.res b/packages/envio/src/bindings/NodeJs.res index e644b52b8..c58fd4a99 100644 --- a/packages/envio/src/bindings/NodeJs.res +++ b/packages/envio/src/bindings/NodeJs.res @@ -174,6 +174,9 @@ module ChildProcess = { external onExit: (child, @as("exit") _, (Null.t, Null.t) => unit) => unit = "on" @send external onChildError: (child, @as("error") _, exn => unit) => unit = "on" @send external kill: (child, string) => bool = "kill" + // Whether the IPC channel is still open. Node closes it before it reports the + // exit, so this goes false while the child is still running. + @get external connected: child => bool = "connected" // Present only for a stdio slot the parent asked to pipe. type stdioStream From d3d7848d0fd16d4aec0c350b226d5eff709abe65 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 21 Sep 2026 11:12:38 +0000 Subject: [PATCH 44/61] Keep a hand-placed process's resume on the record Whether a process says it resumed was keyed on whether it required the storage to exist already, which is true of every process driving a subset of the chains. A supervisor's workers are one of those, and a person running `envio start --chain 1` is another. That one is nobody's worker: its terminal is the only window onto it, and it had gone quiet about both the resume and the checkpoints it resumed from. What a process requires of the storage now decides nothing about what it says. Being a forked worker does, since that is exactly the case where a supervisor has already announced the run. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_013saxnEKXJJXrZCkodmDMzE --- .../lib_tests/WorkerResumeLogging_test.res | 69 ++++++++++++------- packages/envio/src/Persistence.res | 9 ++- 2 files changed, 50 insertions(+), 28 deletions(-) diff --git a/packages/envio-tests/test/lib_tests/WorkerResumeLogging_test.res b/packages/envio-tests/test/lib_tests/WorkerResumeLogging_test.res index 569d78567..bfea0c43d 100644 --- a/packages/envio-tests/test/lib_tests/WorkerResumeLogging_test.res +++ b/packages/envio-tests/test/lib_tests/WorkerResumeLogging_test.res @@ -26,7 +26,7 @@ let makePersistence = () => ), ) -let initRun = (~requireInitialized) => +let initRun = (~requireInitialized, ~announceResume=true) => makePersistence()->Persistence.init( ~chainConfigs=config.chainMap->ChainMap.values, ~contractMapping=config.contractMapping, @@ -35,6 +35,7 @@ let initRun = (~requireInitialized) => ~runCommand=Some("envio dev"), ~lowercaseAddresses=config.lowercaseAddresses, ~requireInitialized, + ~announceResume, ) let logLines = async path => @@ -52,33 +53,51 @@ let logLines = async path => | exception _ => [] } -describe("Resuming an isolated worker", () => { - // The supervisor announces the run's storage once, for every chain. A worker - // resuming the state it was handed has nothing to add to that. - Async.it("Stays quiet about storage the supervisor already announced", async t => { - await initRun(~requireInitialized=false) +let resumeLines = async (~announceResume) => { + let path = `${NodeJs.Process.cwd()}/lib/envio-worker-resume-${Date.now()->Float.toString}-${announceResume + ? "announced" + : "quiet"}.log` + Logging.setLogger( + Logging.makeLogger( + ~logStrategy=FileOnly, + ~logFilePath=path, + ~defaultFileLogLevel=#info, + ~userLogLevel=#info, + ), + ) - let path = `${NodeJs.Process.cwd()}/lib/envio-worker-resume-${Date.now()->Float.toString}.log` - Logging.setLogger( - Logging.makeLogger( - ~logStrategy=FileOnly, - ~logFilePath=path, - ~defaultFileLogLevel=#info, - ~userLogLevel=#info, - ), - ) + await initRun(~requireInitialized=true, ~announceResume) + Logging.info("done") - await initRun(~requireInitialized=true) - Logging.info("done") + let rec until = async deadline => + switch await logLines(path) { + | lines if lines->Array.includes("done") || Date.now() > deadline => lines + | _ => + await Utils.delay(50) + await until(deadline) + } + await until(Date.now() +. 3000.) +} - let rec until = async deadline => - switch await logLines(path) { - | lines if lines->Array.includes("done") || Date.now() > deadline => lines - | _ => - await Utils.delay(50) - await until(deadline) - } +describe("Announcing a resume", () => { + // The supervisor says it once for the whole run, so a worker it forked has + // nothing to add. Every other process resuming a subset of the chains is + // somebody's only window onto it, `envio start --chain` included, and both + // of them need the schema to exist already — so what a process requires of + // the storage can't be what decides whether it speaks. + Async.it("Quiet for a forked worker, and not for anyone else", async t => { + await initRun(~requireInitialized=false) + + let quiet = await resumeLines(~announceResume=false) + let announced = await resumeLines(~announceResume=true) - t.expect(await until(Date.now() +. 3000.)).toStrictEqual(["done"]) + t.expect((quiet, announced)).toStrictEqual(( + ["done"], + [ + "Found existing indexer storage. Resuming indexing state...", + "Successfully resumed indexing state! Continuing from the last checkpoint.", + "done", + ], + )) }) }) diff --git a/packages/envio/src/Persistence.res b/packages/envio/src/Persistence.res index 7b8354b08..19223a1ca 100644 --- a/packages/envio/src/Persistence.res +++ b/packages/envio/src/Persistence.res @@ -275,6 +275,10 @@ let init = { // would create rows for this process's chains only, leaving the ones it // skipped with no state for their own processes to resume. ~requireInitialized=false, + // Whether this process is the one that tells the operator the run resumed. + // A supervisor says it once for the whole run, so the workers it forked + // keep it to their own log files. + ~announceResume=true, ~startBlockRetry=StartBlockResolver.UntilItAnswers, ) => { try { @@ -324,9 +328,7 @@ let init = { | _ => false } ) { - // An isolated process resumes state its supervisor already announced - // for the whole run, so it says so only to its own log file. - let logResume = requireInitialized ? Logging.debug : Logging.info + let logResume = announceResume ? Logging.info : Logging.debug logResume(`Found existing indexer storage. Resuming indexing state...`) let initialState = await persistence.storage.resumeInitialState( ~entities=persistence.allEntities, @@ -371,6 +373,7 @@ let initForRun = ( ~requireInitialized, ) => persistence->init( + ~announceResume=!Worker.isEnabled, ~reset, ~chainConfigs=config.chainMap->ChainMap.values, ~contractMapping=config.contractMapping, From 953279e5c105f501204f229dd6225b119277815a Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 21 Sep 2026 11:13:27 +0000 Subject: [PATCH 45/61] Report the run's buffer target and threshold as the run's Two of the merged readings were arrived at by treating a run's processes as parts of a sum, when neither is a quantity a worker holds a share of. Every worker reads the same buffer target and opens a pool of that size, so adding them up reported a target nobody set, growing with the number of processes the run happened to be split into. It now comes from the supervisor, which is where the run's configuration is. A run was in the reorg threshold as soon as any worker was. Chains cross into it as one indexer, held there until the last of them arrives, so one process still below it speaks for the whole run, exactly as one still backfilling does for arriving at the head. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_013saxnEKXJJXrZCkodmDMzE --- .../test/lib_tests/Metrics_test.res | 27 ++++++++++++++----- packages/envio/src/Metrics.res | 13 ++++++--- packages/envio/src/Supervisor.res | 4 +++ 3 files changed, 34 insertions(+), 10 deletions(-) diff --git a/packages/envio-tests/test/lib_tests/Metrics_test.res b/packages/envio-tests/test/lib_tests/Metrics_test.res index a2cdde394..5b43121ea 100644 --- a/packages/envio-tests/test/lib_tests/Metrics_test.res +++ b/packages/envio-tests/test/lib_tests/Metrics_test.res @@ -686,11 +686,14 @@ describe("Metrics.merge", () => { effects: [effect(~cacheCount=Some(7))], } - t.expect(Metrics.merge([only], ~startTime, ~metricTime, ~elapsedSeconds=9.)).toStrictEqual({ + t.expect( + Metrics.merge([only], ~startTime, ~metricTime, ~elapsedSeconds=9., ~targetBufferSize=100), + ).toStrictEqual({ ...only, startTime, metricTime, elapsedSeconds: 9., + targetBufferSize: 100, }) }) @@ -730,16 +733,25 @@ describe("Metrics.merge", () => { } t.expect( - Metrics.merge([first, second], ~startTime, ~metricTime, ~elapsedSeconds=9.), + Metrics.merge( + [first, second], + ~startTime, + ~metricTime, + ~elapsedSeconds=9., + ~targetBufferSize=100, + ), ).toStrictEqual({ ...baseMetrics, startTime, metricTime, elapsedSeconds: 9., - targetBufferSize: 150, + // Every worker holds a pool of the run's target, so the targets are one + // number the run was configured with, not a total to add up. + targetBufferSize: 100, maxBatchSize: 5000, - isInReorgThreshold: true, - // One worker still backfilling speaks for the whole indexer. + // Chains cross into the threshold as one indexer, so one worker still + // below it speaks for the whole run, exactly as for arriving at the head. + isInReorgThreshold: false, hasArrivedAtHead: false, rollbackEnabled: true, processingSeconds: 2., @@ -772,10 +784,13 @@ describe("Metrics.merge", () => { }) it("Renders an empty group as an indexer that has reported nothing yet", t => { - t.expect(Metrics.merge([], ~startTime, ~metricTime, ~elapsedSeconds=0.)).toStrictEqual({ + t.expect( + Metrics.merge([], ~startTime, ~metricTime, ~elapsedSeconds=0., ~targetBufferSize=100), + ).toStrictEqual({ ...baseMetrics, startTime, metricTime, + targetBufferSize: 100, }) }) }) diff --git a/packages/envio/src/Metrics.res b/packages/envio/src/Metrics.res index 9baab8e22..edee9761a 100644 --- a/packages/envio/src/Metrics.res +++ b/packages/envio/src/Metrics.res @@ -174,8 +174,10 @@ let sumByKey = (items: array<'item>, ~key: 'item => string, ~add: ('item, 'item) // unsplit run would have produced. Series keyed by chain concatenate, since a // chain belongs to exactly one worker; series keyed by name are summed, since // every worker runs the same handlers and effects over its own chains. The -// clock is the caller's: it belongs to the group, not to any worker. -let merge = (snapshots: array, ~startTime, ~metricTime, ~elapsedSeconds) => { +// clock and the buffer target are the caller's: both belong to the group, and +// the target is what the run was configured with rather than anything a worker +// could add up to. +let merge = (snapshots: array, ~startTime, ~metricTime, ~elapsedSeconds, ~targetBufferSize) => { let concat = select => snapshots->Array.flatMap(select) let sumInt = select => snapshots->Array.reduce(0, (acc, snapshot) => acc + snapshot->select) let sumFloat = select => snapshots->Array.reduce(0., (acc, snapshot) => acc +. snapshot->select) @@ -184,8 +186,11 @@ let merge = (snapshots: array, ~startTime, ~metricTime, ~elapsedSeconds) => { startTime, metricTime, elapsedSeconds, - targetBufferSize: sumInt(s => s.targetBufferSize), - isInReorgThreshold: snapshots->Array.some(s => s.isInReorgThreshold), + targetBufferSize, + // Chains cross into the threshold as one indexer, so a run is in it once + // every process is, the same reading as the arrival below. + isInReorgThreshold: snapshots->Utils.Array.notEmpty && + snapshots->Array.every(s => s.isInReorgThreshold), // The run has arrived only once every process has: one still backfilling // speaks for the whole indexer. hasArrivedAtHead: snapshots->Utils.Array.notEmpty && diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index 4a8003643..7e82d69c2 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -367,6 +367,10 @@ let run = async (~config: Config.t, ~workers: array, ~reset) => { ~startTime, ~metricTime=Date.make(), ~elapsedSeconds=startTimeRef->Performance.secondsSince, + // Every worker reads the same buffer target this process does, and each + // holds a pool of that size. What the run was asked for is the one number + // that means anything across them. + ~targetBufferSize=CrossChainState.calculateTargetBufferSize(), ) Server.startServer( From 7844302896e6dc33840f652775f8a4cb236bdd3d Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 21 Sep 2026 11:15:48 +0000 Subject: [PATCH 46/61] Let a line name its own chain, and drop the process-wide one Every process took a pino mixin that stamped a chain onto each line it logged, but only ever when the process drove exactly one chain. A run split across workers deals chains two at a time, so the common split got nothing, and the lines that did get it were the process-wide ones: reaching the reorg threshold, catching up to every end block, finalizing. None of those is one chain's to claim. A line that is about a chain already says so where it is written, from the chain-scoped loggers in the fetching and source paths to the calls that pass the id themselves. With nothing stamped from outside, a child logger can bind what it likes without the line carrying the field twice, so the stripping that existed to prevent that goes too, and with it a dict copied per line in every process. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_013saxnEKXJJXrZCkodmDMzE --- .../test/lib_tests/LoggingContext_test.res | 55 ------------------- .../test/lib_tests/Supervisor_test.res | 22 -------- packages/envio/src/Config.res | 11 ---- packages/envio/src/Logging.res | 43 ++------------- packages/envio/src/Main.res | 3 - 5 files changed, 6 insertions(+), 128 deletions(-) delete mode 100644 packages/envio-tests/test/lib_tests/LoggingContext_test.res diff --git a/packages/envio-tests/test/lib_tests/LoggingContext_test.res b/packages/envio-tests/test/lib_tests/LoggingContext_test.res deleted file mode 100644 index 8756d7a49..000000000 --- a/packages/envio-tests/test/lib_tests/LoggingContext_test.res +++ /dev/null @@ -1,55 +0,0 @@ -open Vitest - -// Reads what pino actually wrote: a duplicated field survives JSON parsing -// (the last one wins), so the raw line is the only place it shows. -let occurrences = (line, ~field) => - line->String.split(`"${field}"`)->Array.length - 1 - -// The file is written by pino's transport worker, so it lands a moment later. -let readWhenWritten = async path => { - let deadline = Date.now() +. 3000. - let read = async () => - switch await NodeJs.Fs.Promises.readFile( - ~filepath=NodeJs.Path.resolve([path]), - ~encoding=Utf8, - ) { - | contents => contents - | exception _ => "" - } - let rec until = async () => - switch await read() { - | "" if Date.now() < deadline => - await Utils.delay(50) - await until() - | contents => contents - } - await until() -} - -describe("Logging.setContext", () => { - Async.it("Names the chain once, whoever else on the line names it", async t => { - let path = `${NodeJs.Process.cwd()}/lib/envio-logging-context-${Date.now() - ->Float.toString}.log` - Logging.setLogger( - Logging.makeLogger( - ~logStrategy=FileOnly, - ~logFilePath=path, - ~defaultFileLogLevel=#info, - ~userLogLevel=#info, - ), - ) - Logging.setContext(Dict.fromArray([("chainId", JSON.Number(137.))])) - - Logging.info("a line with no chain in hand") - Logging.createChild(~params={"chainId": 137, "source": "hypersync"})->Logging.childInfo({ - "msg": "a line from a chain-scoped logger", - }) - - let lines = - (await readWhenWritten(path)) - ->String.trim - ->String.split("\n") - - t.expect(lines->Array.map(line => occurrences(line, ~field="chainId"))).toStrictEqual([1, 1]) - }) -}) diff --git a/packages/envio-tests/test/lib_tests/Supervisor_test.res b/packages/envio-tests/test/lib_tests/Supervisor_test.res index acdc931ed..773348565 100644 --- a/packages/envio-tests/test/lib_tests/Supervisor_test.res +++ b/packages/envio-tests/test/lib_tests/Supervisor_test.res @@ -198,28 +198,6 @@ describe("Supervisor worker plumbing", () => { }) }) -describe("Config.logContext", () => { - it("Names the one chain an isolated process drives, and nothing otherwise", t => { - t.expect([ - // This process drives one of the schema's chains while siblings drive the rest. - config(~schema=perChain, ~isolatedChains=[JSON.Number(137.)])->Config.logContext, - // Several chains have no single owner to name. - config( - ~schema=perChain, - ~isolatedChains=[JSON.Number(1.), JSON.Number(137.)], - )->Config.logContext, - // One process driving every chain: chain-scoped lines already name theirs, - // and the rest belong to the run as a whole. - config(~schema=perChain)->Config.logContext, - config(~schema=crossChain)->Config.logContext, - ]).toStrictEqual([ - Some(Dict.fromArray([("chainId", JSON.Number(137.))])), - None, - None, - None, - ]) - }) -}) describe("Worker.detect", () => { it("Counts as a worker only when forked with the variable and a channel", t => { diff --git a/packages/envio/src/Config.res b/packages/envio/src/Config.res index 512aab0b4..df76456e5 100644 --- a/packages/envio/src/Config.res +++ b/packages/envio/src/Config.res @@ -630,17 +630,6 @@ let getChain = (config, ~chainId) => // be split across processes. let isPerChain = (config: t) => !(config.userEntities->Array.some(entity => entity.crossChain)) -// What every line this process logs is attributed to. A process driving one -// of the schema's chains while siblings drive the rest names it, on the lines -// that had no chain in hand. One driving several has no single owner to name, -// and its chain-scoped lines already carry theirs. -let logContext = (config: t): option> => - switch (config.isolated, config.chainMap->ChainMap.keys) { - | (true, [chainId]) => - Some(Dict.fromArray([("chainId", chainId->S.reverseConvertToJsonOrThrow(ChainId.schema))])) - | _ => None - } - // Narrows a config to the chains one `envio start --chain` process drives. // `contractMapping` is deliberately left whole: its ids are what the migration // that created the schema stored, and one rebuilt from a subset would hand the diff --git a/packages/envio/src/Logging.res b/packages/envio/src/Logging.res index 2bcb4136c..76a760487 100644 --- a/packages/envio/src/Logging.res +++ b/packages/envio/src/Logging.res @@ -26,30 +26,6 @@ let logLevels = [ %%private(let logger = ref(None)) -// Fields every line this process logs carries. Merged into each line rather -// than bound to a child logger: pino writes a child's bindings and the line's -// own fields side by side, so a line that names the same key would carry it -// twice. A fresh object per line, since pino merges the line's fields into -// whatever this returns. -%%private(let context: ref> = ref(Dict.make())) -%%private(let mixin = () => JSON.Object(context.contents->Dict.copy)) - -// A child logger that binds a field the process already carries would have pino -// write it twice: a child's bindings and the context are concatenated into the -// line, not merged. The context is the one place a line names it, which it can -// be because a process only takes one when its every line is about that chain. -%%private( - let withoutContext = (params: 'a) => - switch context.contents->Dict.keysToArray { - | [] => params - | keys => { - let narrowed = params->(Utils.magic: 'a => dict)->Dict.copy - keys->Array.forEach(key => narrowed->Dict.delete(key)) - narrowed->(Utils.magic: dict => 'a) - } - } -) - let makeLogger = (~logStrategy, ~logFilePath, ~defaultFileLogLevel, ~userLogLevel) => { // Currently unused - useful if using multiple transports. // let pinoRaw = {"target": "pino/file", "level": Config.userLogLevel} @@ -80,19 +56,17 @@ let makeLogger = (~logStrategy, ~logFilePath, ~defaultFileLogLevel, ~userLogLeve ...Pino.ECS.make(), customLevels: logLevels, base, - mixin, }, Transport.make(pinoFile), ) | EcsConsoleMultistream => - makeMultiStreamLogger(~logFile=None, ~options=Some({...Pino.ECS.make(), base, mixin})) + makeMultiStreamLogger(~logFile=None, ~options=Some({...Pino.ECS.make(), base})) | EcsConsole => make({ ...Pino.ECS.make(), level: userLogLevel, customLevels: logLevels, base, - mixin, }) | FileOnly => makeWithOptionsAndTransport( @@ -100,13 +74,12 @@ let makeLogger = (~logStrategy, ~logFilePath, ~defaultFileLogLevel, ~userLogLeve customLevels: logLevels, level: defaultFileLogLevel, base, - mixin, }, Transport.make(pinoFile), ) - | ConsoleRaw => makeMultiStreamLogger(~logFile=None, ~options=Some({base, mixin})) - | ConsolePretty => makeMultiStreamLogger(~logFile=None, ~options=Some({base, mixin})) - | Both => makeMultiStreamLogger(~logFile=Some(logFilePath), ~options=Some({base, mixin})) + | ConsoleRaw => makeMultiStreamLogger(~logFile=None, ~options=Some({base: base})) + | ConsolePretty => makeMultiStreamLogger(~logFile=None, ~options=Some({base: base})) + | Both => makeMultiStreamLogger(~logFile=Some(logFilePath), ~options=Some({base: base})) } } @@ -176,15 +149,11 @@ let childFatal = (logger, params: 'a) => { } let createChild = (~params: 'a) => { - getLogger()->child(params->withoutContext->createChildParams) + getLogger()->child(params->createChildParams) } -// What belongs on every line is the run's to decide; the logger only carries -// what it is handed. A line that names one of these fields itself wins. -let setContext = (fields: dict) => context := fields - let createChildFrom = (~logger: t, ~params: 'a) => { - logger->child(params->withoutContext->createChildParams) + logger->child(params->createChildParams) } @inline diff --git a/packages/envio/src/Main.res b/packages/envio/src/Main.res index 1468908d3..e08dab486 100644 --- a/packages/envio/src/Main.res +++ b/packages/envio/src/Main.res @@ -453,9 +453,6 @@ exception FatalError(exn) ) => { // A worker reports to its supervisor, which draws for the whole run. let shouldUseTui = Tui.shouldUse(~suppressed=isTest || Worker.isEnabled) - // In per-chain mode every line this process writes belongs to the chains it - // drives, whether or not a supervisor split the run across processes. - config->Config.logContext->Option.forEach(Logging.setContext) // isDevelopmentMode controls whether the indexer stays alive after all // chains finish (keepProcessAlive) and whether the console API is exposed. // Set by `envio dev` via the public config's `isDev` field; `envio start` From 5196f25b772bce01f736ea0dac2cea3ed277cd16 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 21 Sep 2026 12:16:32 +0000 Subject: [PATCH 47/61] Read a signalled worker as a run being stopped `systemctl stop` on a unit with the default `KillMode=control-group` sends SIGTERM to every process in the unit, so a run's workers are told directly, before their supervisor has passed anything on. A worker that exits on a signal carries no exit code at all, which the supervisor read as a worker dying: every clean shutdown under systemd logged a failure and left the run exiting non-zero. Being signalled now reads as what it is. The rest of the group still goes down with it, since one worker short leaves its chains unindexed, but the run reports itself stopped rather than failed. SIGKILL stays a failure, which is what the kernel's out-of-memory killer sends. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_013saxnEKXJJXrZCkodmDMzE --- .../envio-tests/test/SupervisorFork_test.res | 21 +++++++ .../test/lib_tests/Supervisor_test.res | 27 +++++++++ packages/envio/src/Supervisor.res | 57 +++++++++++++++---- 3 files changed, 95 insertions(+), 10 deletions(-) diff --git a/packages/envio-tests/test/SupervisorFork_test.res b/packages/envio-tests/test/SupervisorFork_test.res index 7f468073b..23d674738 100644 --- a/packages/envio-tests/test/SupervisorFork_test.res +++ b/packages/envio-tests/test/SupervisorFork_test.res @@ -105,6 +105,27 @@ describe("Supervisor.awaitExit", () => { }, ) + // A process manager that signals the whole group reaches the workers itself, + // so they exit on a SIGTERM the supervisor has not passed on and may not even + // have handled yet. Reading that as a worker dying would fail every clean + // shutdown under systemd's default kill mode. + Async.it("Takes a worker signalled from outside as the run being stopped", async t => { + NodeJs.Process.process.env->Dict.set("FAKE_WORKER", "linger") + let signalled = forkFixture(~chainIds=[1]) + let sibling = forkFixture(~chainIds=[137]) + let group: Supervisor.group = { + running: [signalled, sibling], + stopping: false, + releaseCheck: None, + } + let ended = outcome(group) + signalled.child->NodeJs.ChildProcess.kill("SIGTERM")->ignore + + // The sibling still went down with it: one worker short leaves its chains + // unindexed. + t.expect((await ended, group.stopping)).toStrictEqual((Ok(Supervisor.Stopped), true)) + }) + Async.it("Stops the group and fails the run when one worker dies", async t => { NodeJs.Process.process.env->Dict.set("FAKE_WORKER", "fail") let failing = forkFixture(~chainIds=[1]) diff --git a/packages/envio-tests/test/lib_tests/Supervisor_test.res b/packages/envio-tests/test/lib_tests/Supervisor_test.res index 773348565..3f52066e1 100644 --- a/packages/envio-tests/test/lib_tests/Supervisor_test.res +++ b/packages/envio-tests/test/lib_tests/Supervisor_test.res @@ -240,6 +240,33 @@ describe("Worker.detect", () => { }) }) +describe("Supervisor.classifyExit", () => { + let classify = (~code=Null.null, ~signal=Null.null, ~stopping=false) => + Supervisor.classifyExit(~code, ~signal, ~stopping) + + it("Reads a signalled worker as a run being stopped, not as one failing", t => { + t.expect([ + // `systemctl stop` on a unit with the default kill mode signals every + // process in it, so a worker is told before its supervisor has passed it + // on. Its exit carries no code at all. + classify(~signal=Null.make("SIGTERM")), + // The supervisor's own stop, once it has decided. + classify(~code=Null.null, ~signal=Null.make("SIGTERM"), ~stopping=true), + // Indexing to every end block. + classify(~code=Null.make(0)), + // The kernel's out-of-memory killer, and a worker that threw. + classify(~signal=Null.make("SIGKILL")), + classify(~code=Null.make(1)), + ]).toStrictEqual([ + Supervisor.Stopping, + Supervisor.Expected, + Supervisor.Expected, + Supervisor.Failed, + Supervisor.Failed, + ]) + }) +}) + describe("Supervisor.syncCache", () => { Async.it("Dumps once for requests that overlap, and again for a later one", async t => { let dumps = ref(0) diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index 7e82d69c2..dfe839888 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -212,6 +212,14 @@ let stop = group => { group.running->Array.forEach(r => r.child->NodeJs.ChildProcess.kill("SIGTERM")->ignore) } +// Takes the group down unless it is already going. Signalling a worker that has +// been signalled changes nothing, but `stopping` is what tells an exit from an +// expected one, so the first stop is the one that counts. +let stopOnce = group => + if !group.stopping { + group->stop + } + // The dev console's cache dump, which belongs to the supervisor rather than to // its workers: a dump copies every effect cache table in the schema to a file // named after the effect, so a worker asked to do it would copy its siblings' @@ -246,6 +254,31 @@ let dumpCache = (~config) => { // which is what indexing to every end block looks like. type outcome = Finished | Stopped +// How one worker's ending reads. +type ending = + // On its own terms, or because the supervisor asked. + | Expected + // Asked to stop by someone other than the supervisor. A process manager that + // signals a whole group reaches the workers itself, so a worker can be told + // before the supervisor has decided what the signal meant. + | Stopping + | Failed + +// A worker that stops on a signal is being stopped, not failing: `systemctl +// stop` on a unit with the default `KillMode=control-group` sends SIGTERM to +// every process in it, so the workers get it directly and exit on it. Reading +// that as a failure would fail every clean shutdown under systemd. +// +// The kernel's out-of-memory killer sends SIGKILL, which stays a failure — as +// does every non-zero exit of a worker the supervisor didn't ask to stop. +let classifyExit = (~code: Null.t, ~signal: Null.t, ~stopping) => + switch (stopping, code->Null.toOption, signal->Null.toOption) { + | (true, _, _) + | (_, Some(0), _) => Expected + | (_, _, Some("SIGTERM")) => Stopping + | _ => Failed + } + // Resolves once every worker has ended. Throws if any of them ended in a way // the supervisor didn't ask for, having first taken the rest down: one worker // short leaves its chains unindexed, and a run that kept the others going would @@ -255,13 +288,19 @@ let awaitExit = async (group): outcome => { let alive = ref(group.running->Array.length) await Promise.make((resolve, _) => { - let onGone = (r, ~failure) => + let onGone = (r, ~ending) => if !r.settled { r.settled = true - if failure { - failed := true - if !group.stopping { - group->stop + // The rest of the run goes down with it either way: one worker short + // leaves its chains unindexed, and a run that kept the others going + // would look healthy while falling behind. What differs is whether the + // run reports itself as having failed. + switch ending { + | Expected => () + | Stopping => group->stopOnce + | Failed => { + failed := true + group->stopOnce } } alive := alive.contents - 1 @@ -272,15 +311,13 @@ let awaitExit = async (group): outcome => { group.running->Array.forEach(r => { r.child->NodeJs.ChildProcess.onExit( - (code, _signal) => - // Only an exit the supervisor asked for is expected. Anything else — a - // non-zero code, or a signal like the kernel's out-of-memory kill. - r->onGone(~failure=!group.stopping && code->Null.toOption !== Some(0)), + (code, signal) => + r->onGone(~ending=classifyExit(~code, ~signal, ~stopping=group.stopping)), ) r.child->NodeJs.ChildProcess.onChildError( exn => { Logging.errorWithExn(exn, `${r.worker->label} failed to start`) - r->onGone(~failure=true) + r->onGone(~ending=Failed) }, ) }) From d4e6be31cca7a74a7ab5d5dce8942cc6d30bbc59 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 21 Sep 2026 12:16:32 +0000 Subject: [PATCH 48/61] Take the split's precondition from the counter it depends on Whether a run may split was worked out from the entities a second time, next to the checkpoint sequence that is worked out from exactly the same fact. Two readings of one thing, free to drift apart, and the sequence is the one that makes splitting safe: a chain gets a counter of its own only when no other chain can reach its rows, which is what lets a process advance one chain without saying anything about the others. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_013saxnEKXJJXrZCkodmDMzE --- packages/envio/src/Config.res | 14 ++++++++++---- 1 file changed, 10 insertions(+), 4 deletions(-) diff --git a/packages/envio/src/Config.res b/packages/envio/src/Config.res index df76456e5..fe2b6d90a 100644 --- a/packages/envio/src/Config.res +++ b/packages/envio/src/Config.res @@ -625,10 +625,16 @@ let getChain = (config, ~chainId) => "No chain with id " ++ chainId->ChainId.toString ++ " found in config.yaml", ) -// Whether every entity belongs to exactly one chain. Only then is a unit of -// this indexer's work attributable to a chain at all, which is what lets a run -// be split across processes. -let isPerChain = (config: t) => !(config.userEntities->Array.some(entity => entity.crossChain)) +// Whether every entity belongs to exactly one chain. Read off the checkpoint +// sequence rather than the entities again, because that is the same fact and +// the one that makes splitting safe: a chain only gets a counter of its own +// when no other chain can reach its rows, and a counter of its own is what lets +// a process advance one chain without saying anything about the others. +let isPerChain = (config: t) => + switch config.checkpointSequence { + | PerChain => true + | SharedAcrossChains => false + } // Narrows a config to the chains one `envio start --chain` process drives. // `contractMapping` is deliberately left whole: its ids are what the migration From 9adc49f8bb351f05174d6470143153cee2198d61 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 21 Sep 2026 12:19:47 +0000 Subject: [PATCH 49/61] Split the run's memory budgets the way its connections are split The fetch buffer pool is the whole indexer's and deliberately independent of how many chains it has, so that adding a chain doesn't add a pool. Splitting the chains across processes did exactly that by the back door: every worker read the budget whole and held all of it, and the same for the in-memory object target, so a run held as many times its configured memory as the connection budget happened to buy it workers. Each worker now gets a share, handed over in the spawn environment beside its share of the connections. That also makes the target the run reports the one it holds, rather than the one it was asked for. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_013saxnEKXJJXrZCkodmDMzE --- .../envio-tests/test/SupervisorFork_test.res | 9 +++++ .../envio-tests/test/helpers/fakeWorker.mjs | 2 + packages/envio/src/Supervisor.res | 40 ++++++++++++++++--- 3 files changed, 45 insertions(+), 6 deletions(-) diff --git a/packages/envio-tests/test/SupervisorFork_test.res b/packages/envio-tests/test/SupervisorFork_test.res index 23d674738..05b80f405 100644 --- a/packages/envio-tests/test/SupervisorFork_test.res +++ b/packages/envio-tests/test/SupervisorFork_test.res @@ -6,6 +6,8 @@ open Vitest type fixtureReport = { workerConfig: string, maxConnections: string, + bufferSize: string, + objectsTarget: string, logFile: string, startTime: Date.t, hasArrivedAtHead: bool, @@ -17,6 +19,7 @@ let forkFixture = ( ~chainIds, ~maxConnections=2, ~workerIndex=0, + ~workerCount=2, ~holdRealtime=false, ~isDev=false, ~pipeOutput=false, @@ -26,6 +29,7 @@ let forkFixture = ( Supervisor.fork( {chainIds: chainIds->Array.map(ChainId.fromInt), maxConnections}, ~workerIndex, + ~workerCount, ~holdRealtime, ~isDev, ~entryPath=fixturePath, @@ -40,6 +44,7 @@ describe("Supervisor.fork", () => { ~chainIds=[1, 137], ~maxConnections=3, ~workerIndex=1, + ~workerCount=4, ~holdRealtime=true, ~isDev=true, ) @@ -63,6 +68,10 @@ describe("Supervisor.fork", () => { // is why `isDev` is among them. workerConfig: `{"chainIds":[1,137],"holdRealtime":true,"isDev":true}`, maxConnections: "3", + // The run's memory budgets are the whole indexer's, so a worker gets a + // share rather than the whole of each. + bufferSize: (CrossChainState.calculateTargetBufferSize() / 4)->Int.toString, + objectsTarget: (Env.inMemoryObjectsTarget->Float.toInt / 4)->Int.toString, logFile: Supervisor.logFilePath(~workerIndex=1), // Proof the channel clones rather than stringifies: a JSON round trip // would have turned this into a string. diff --git a/packages/envio-tests/test/helpers/fakeWorker.mjs b/packages/envio-tests/test/helpers/fakeWorker.mjs index a7319481b..0adc69c36 100644 --- a/packages/envio-tests/test/helpers/fakeWorker.mjs +++ b/packages/envio-tests/test/helpers/fakeWorker.mjs @@ -10,6 +10,8 @@ process.send({ metrics: { workerConfig: process.env.ENVIO_INTERNAL_WORKER, maxConnections: process.env.ENVIO_PG_MAX_CONNECTIONS, + bufferSize: process.env.ENVIO_INDEXING_MAX_BUFFER_SIZE, + objectsTarget: process.env.ENVIO_IN_MEMORY_OBJECTS_TARGET, logFile: process.env.LOG_FILE, // A Date survives only under structured-clone serialization, which is // what a metrics snapshot's timestamps need. diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index dfe839888..ca5415b02 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -106,6 +106,25 @@ let readLines = (~onLine) => { (read, flush) } +// The run's memory budgets, and each worker's share of them. Both are the whole +// indexer's rather than one chain's or one process's: the fetch buffer pool is +// deliberately independent of how many chains a run has, and an indexer split +// across processes that took each budget whole in every one of them would hold +// as many times the memory as it happened to have workers. +// +// Read here and handed over in the spawn environment for the same reason the +// connection share is: a worker's `Env` reads them as it loads. +let memoryBudgets = (~workerCount) => + [ + ("ENVIO_INDEXING_MAX_BUFFER_SIZE", CrossChainState.calculateTargetBufferSize()), + ("ENVIO_IN_MEMORY_OBJECTS_TARGET", Env.inMemoryObjectsTarget->Float.toInt), + ]->Array.map(((name, budget)) => ( + name, + // A budget smaller than the run has workers still leaves each one something + // to hold, rather than a pool it can never put anything in. + Pervasives.max(1, budget / workerCount)->Int.toString, + )) + // Whether this process's own output is a terminal. `pino-pretty` colorizes on // that test, and a piped worker would fail it for a run the operator is // watching in colour. @@ -114,6 +133,8 @@ let readLines = (~onLine) => { let fork = ( worker: worker, ~workerIndex, + // How many processes the run's budgets are being split between. + ~workerCount, // Whether this worker waits for the run before going realtime. False when // every chain resumed already caught up: there is nothing left to wait for, // and a barrier nobody can open would hold the run forever. @@ -139,9 +160,10 @@ let fork = ( isDev, }->S.reverseConvertToJsonStringOrThrow(Worker.configSchema), ) - // The worker's slice of the budget. Read when the worker's own Env module - // loads, which is why it rides in the spawn environment rather than a message. + // The worker's slice of the budgets. Read when the worker's own Env module + // loads, which is why they ride in the spawn environment rather than a message. env->Dict.set("ENVIO_PG_MAX_CONNECTIONS", worker.maxConnections->Int.toString) + memoryBudgets(~workerCount)->Array.forEach(((name, share)) => env->Dict.set(name, share)) env->Dict.set("LOG_FILE", logFilePath(~workerIndex)) if pipeOutput && stdoutIsTty->Nullable.toOption->Option.getOr(false) { env->Dict.set("FORCE_COLOR", "1") @@ -391,7 +413,13 @@ let run = async (~config: Config.t, ~workers: array, ~reset) => { ) let group = { running: workers->Array.mapWithIndex((worker, workerIndex) => - worker->fork(~workerIndex, ~holdRealtime, ~isDev=config.isDev, ~pipeOutput=shouldUseTui) + worker->fork( + ~workerIndex, + ~workerCount=workers->Array.length, + ~holdRealtime, + ~isDev=config.isDev, + ~pipeOutput=shouldUseTui, + ) ), stopping: false, releaseCheck: None, @@ -404,9 +432,9 @@ let run = async (~config: Config.t, ~workers: array, ~reset) => { ~startTime, ~metricTime=Date.make(), ~elapsedSeconds=startTimeRef->Performance.secondsSince, - // Every worker reads the same buffer target this process does, and each - // holds a pool of that size. What the run was asked for is the one number - // that means anything across them. + // The run's pool, which its workers hold a share of each. Reporting the + // shares added back up would say the same thing less directly, and say + // nothing at all before every worker has reported. ~targetBufferSize=CrossChainState.calculateTargetBufferSize(), ) From 56f65d8fa007485936a2dcc8a75119bb09de451f Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 21 Sep 2026 12:21:18 +0000 Subject: [PATCH 50/61] Give the barrier a name of its own It was fifteen lines in the middle of starting a run, between the server it serves and the display it draws, and it is the one piece of that function with a lifetime: it pairs with the stop that ends it, which was already a function. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_013saxnEKXJJXrZCkodmDMzE --- packages/envio/src/Supervisor.res | 33 +++++++++++++++++-------------- 1 file changed, 18 insertions(+), 15 deletions(-) diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index ca5415b02..cc9be8ea7 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -372,6 +372,23 @@ let isRunAtHead = (running: array) => r.snapshot->Option.mapOr(false, snapshot => snapshot.hasArrivedAtHead) ) +// Holds every worker at the head until the last of them arrives, then releases +// them together. Chains enter the reorg threshold and go realtime as one +// indexer, and in a split run only the supervisor can see when that is, so the +// run switches over exactly as an unsplit one does. +let startReleaseCheck = group => + group.releaseCheck = Some( + setInterval(() => + if group.running->isRunAtHead { + group->stopReleaseCheck + group.running->Array.forEach(r => + r.child->NodeJs.ChildProcess.send(Worker.ReleaseRealtime)->ignore + ) + Logging.info("Every chain has reached the head. Switching the run to realtime.") + } + , releaseCheckIntervalMillis), + ) + // Runs the group: creates the schema for every chain, forks a worker per plan // entry, and serves the run's metrics, console and display from what they // report. Returns once every worker has exited; throws if any of them failed. @@ -459,22 +476,8 @@ let run = async (~config: Config.t, ~workers: array, ~reset) => { ~onSyncCache=() => syncCache(~dump=() => dumpCache(~config)), ) - // Chains enter the reorg threshold and go realtime as one indexer, which in a - // split run only the supervisor can see. Every worker is held until the last - // one arrives, then released together, so the run switches over exactly as an - // unsplit one does. if holdRealtime { - group.releaseCheck = Some( - setInterval(() => - if group.running->isRunAtHead { - group->stopReleaseCheck - group.running->Array.forEach(r => - r.child->NodeJs.ChildProcess.send(Worker.ReleaseRealtime)->ignore - ) - Logging.info("Every chain has reached the head. Switching the run to realtime.") - } - , releaseCheckIntervalMillis), - ) + group->startReleaseCheck } if shouldUseTui { From 1dee7d6c94f24ceeeaa0fbe524a9bb285fbf2458 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 21 Sep 2026 12:55:25 +0000 Subject: [PATCH 51/61] Let each chain report its own milestones A run's milestones were announced for the indexer as a whole by whichever process happened to reach them, which in a split run is every process. The same five lines appeared two and three times over with nothing to tell them apart, and one of them repeated on every pass while the release was held, six times for two chains. Each is now said by the chain it is about, through the logger that already carries its id. Entering the reorg threshold, reaching an end block and becoming ready are per-chain facts: `ready_at` is a column per chain, the threshold lifts one chain's lag and starts one chain's history, and an end block belongs to the chain that declared it. Said once each, by whoever drives that chain, a split run and an unsplit one say exactly the same things about exactly the same chains. The lines are also worded for what the reader is about to see rather than for the state being left: that crossing the threshold is what starts writing history, that a chain at the safe block is holding for the others rather than stalled, and that indexing is paused while the indexes build. What is left unattributed is what belongs to a process rather than a chain: setting up the storage, the run's shape, and exiting. Finalizing names the chains it pauses, since a split run has one process saying it for each part. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_013saxnEKXJJXrZCkodmDMzE --- .../test/lib_tests/ChainMilestones_test.res | 50 +++++++++++++++++++ packages/envio/src/BatchProcessing.res | 8 +-- packages/envio/src/ChainFetching.res | 6 ++- packages/envio/src/ChainState.res | 26 ++++++++++ packages/envio/src/ChainState.resi | 2 + packages/envio/src/CrossChainState.res | 43 +++++++++++++--- packages/envio/src/CrossChainState.resi | 1 + packages/envio/src/FinalizeBackfill.res | 14 ++++-- packages/envio/src/IndexerState.res | 3 ++ packages/envio/src/IndexerState.resi | 1 + 10 files changed, 137 insertions(+), 17 deletions(-) create mode 100644 packages/envio-tests/test/lib_tests/ChainMilestones_test.res diff --git a/packages/envio-tests/test/lib_tests/ChainMilestones_test.res b/packages/envio-tests/test/lib_tests/ChainMilestones_test.res new file mode 100644 index 000000000..28da76f6a --- /dev/null +++ b/packages/envio-tests/test/lib_tests/ChainMilestones_test.res @@ -0,0 +1,50 @@ +open Vitest + +open TestChainMetrics + +// The milestones a chain reports for itself. Each is the chain's own to say: +// what it indexed to, what changed when it crossed the reorg threshold, and +// when it became ready. A run split across processes has no process that can +// speak for chains it doesn't drive, and an unsplit run says exactly the same +// things about exactly the same chains. + +describe("ChainState.takeProcessedToEndBlock", () => { + it("Names the end block a chain finished, once, and only once it has", t => { + let unfinished = makeChainState( + ~progressBlockNumber=500, + ~firstEventBlockNumber=None, + ~endBlock=Some(600), + ) + let finished = makeChainState( + ~progressBlockNumber=600, + ~firstEventBlockNumber=None, + ~endBlock=Some(600), + ) + // A chain that runs to the head has no end block to finish. + let endless = makeChainState(~progressBlockNumber=600, ~firstEventBlockNumber=None) + + t.expect(( + unfinished->ChainState.takeProcessedToEndBlock, + finished->ChainState.takeProcessedToEndBlock, + // Reaching it stays true, and every later pass would say so again. + finished->ChainState.takeProcessedToEndBlock, + endless->ChainState.takeProcessedToEndBlock, + )).toStrictEqual((None, Some(600), None, None)) + }) +}) + +describe("ChainState.reorgThresholdEntryMessage", () => { + // Crossing lifts the lag that held the chain short of the head, and starts + // the history a rollback replays. A reader watching writes grow wants the + // second half of that. + it("Says what crossing changed, and mentions history only when it is kept", t => { + let chainState = makeChainState(~progressBlockNumber=500, ~firstEventBlockNumber=None) + let beforeCrossing = chainState->ChainState.reorgThresholdEntryMessage + chainState->ChainState.enterReorgThreshold + + t.expect((beforeCrossing, chainState->ChainState.reorgThresholdEntryMessage)).toStrictEqual(( + "Entered the reorg threshold. Indexing to the chain head from here.", + "Entered the reorg threshold. Indexing to the chain head from here, and keeping the entity history a reorg would be rolled back through.", + )) + }) +}) diff --git a/packages/envio/src/BatchProcessing.res b/packages/envio/src/BatchProcessing.res index c77ba4156..51330e780 100644 --- a/packages/envio/src/BatchProcessing.res +++ b/packages/envio/src/BatchProcessing.res @@ -105,8 +105,8 @@ and processNextBatch = async (state: IndexerState.t, ~scheduleFetch): unit => { } // When resuming from persisted state, all events may already be processed. + state->IndexerState.reportProcessedToEndBlock if EventProcessing.allChainsEventsProcessedToEndblock(state->IndexerState.chainStates) { - Logging.info("All chains are caught up to end blocks.") if !(state->IndexerState.keepProcessAlive) && !(state->IndexerState.isHoldingRealtime) { await ExitOnCaughtUp.run(state) } @@ -158,6 +158,9 @@ and processNextBatch = async (state: IndexerState.t, ~scheduleFetch): unit => { // Can safely reset rollback state, since overwrite is not possible. state->IndexerState.clearRollback state->IndexerState.applyBatchProgress(~batch) + // Before the finalize below, so a chain says it reached its end block + // ahead of the run saying what it does about that. + state->IndexerState.reportProcessedToEndBlock // Backfilling → FinalizingIndexes → Ready. Awaiting here holds the // processing loop for the whole finalize, which is what pauses @@ -178,9 +181,6 @@ and processNextBatch = async (state: IndexerState.t, ~scheduleFetch): unit => { let allCaughtUp = EventProcessing.allChainsEventsProcessedToEndblock( state->IndexerState.chainStates, ) - if allCaughtUp { - Logging.info("All chains are caught up to end blocks.") - } if ( allCaughtUp && diff --git a/packages/envio/src/ChainFetching.res b/packages/envio/src/ChainFetching.res index 9bd643b5b..d991775ae 100644 --- a/packages/envio/src/ChainFetching.res +++ b/packages/envio/src/ChainFetching.res @@ -291,11 +291,15 @@ and applyQueryResponse = ( // What the chain is waiting on, which is not itself. A held process is // waiting on chains it doesn't drive, so it says so even when it drives // only one — how the run is split is not the reader's problem. + // + // Naming the wait matters most at the safe block, which is as far as a + // chain can fetch until the indexer enters the reorg threshold: it looks + // stalled short of the head, and the reason is the chains it is waiting on. let waitingOn = if ( state->IndexerState.isHoldingRealtime || state->IndexerState.chainStates->Dict.keysToArray->Array.length > 1 ) { - " Waiting for other chains." + " Holding here until every chain has caught up." } else { "" } diff --git a/packages/envio/src/ChainState.res b/packages/envio/src/ChainState.res index 1c69d63ec..de0195fed 100644 --- a/packages/envio/src/ChainState.res +++ b/packages/envio/src/ChainState.res @@ -61,6 +61,10 @@ type t = { // What this chain last reported reaching. The head moves, so a chain reaches // it again on every catch-up; only a new milestone is worth a line. mutable reportedFetchedTo: option, + // Whether this chain has said it finished its end block. Reaching it stays + // true for the rest of the run, and every pass over the chains would say so + // again. + mutable reportedEndBlock: bool, mutable reorgCount: int, mutable reorgDetectedBlock: option, mutable rollbackTargetBlock: option, @@ -175,6 +179,7 @@ let make = ( blockRangeFetchedEvents: 0., blockRangeFetchedBlocks: 0., reportedFetchedTo: None, + reportedEndBlock: false, reorgCount: 0, reorgDetectedBlock: None, rollbackTargetBlock: None, @@ -685,6 +690,18 @@ let hasProcessedToEndblock = (cs: t) => { } } +// The end block this chain has just finished indexing to, the first time it +// has. `None` for a chain still working, one with no end block at all, and +// every pass after the one that reported it. +let takeProcessedToEndBlock = (cs: t) => + switch cs.fetchState.endBlock { + | Some(endBlock) if !cs.reportedEndBlock && cs->hasProcessedToEndblock => { + cs.reportedEndBlock = true + Some(endBlock) + } + | _ => None + } + // Caught up as judged by persisted values alone: progress reached the endBlock, // or the head the previous run had already observed (less the lag that holds the // tip back). Unlike `isFetchingAtHead` this doesn't move when a fresh height @@ -992,6 +1009,7 @@ let enterReorgThreshold = (cs: t) => { cs.fetchState = cs.fetchState->FetchState.updateInternal(~blockLag=cs.chainConfig.blockLag) } + let isInReorgThreshold = (cs: t) => cs.isInReorgThreshold // Whether the chain's writes need history: only what a rollback could still @@ -999,6 +1017,14 @@ let isInReorgThreshold = (cs: t) => cs.isInReorgThreshold // progress has run. let shouldSaveHistory = (cs: t) => cs.shouldRollbackOnReorg && cs.maxReorgDepth > 0 && cs.isInReorgThreshold +// What entering the reorg threshold changed for this chain. Below it a chain +// stops short of the head by its reorg depth, because it keeps no history to +// roll back with; crossing lifts both at once, and the second half is the +// answer to why the indexer starts writing more than it was. +let reorgThresholdEntryMessage = (cs: t) => + cs->shouldSaveHistory + ? "Entered the reorg threshold. Indexing to the chain head from here, and keeping the entity history a reorg would be rolled back through." + : "Entered the reorg threshold. Indexing to the chain head from here." // Snapshot the chain's metadata fields for staging into the chains table. let toChainMetadata = (cs: t): InternalTable.Chains.metaFields => { diff --git a/packages/envio/src/ChainState.resi b/packages/envio/src/ChainState.resi index a3212ffe6..41a16953f 100644 --- a/packages/envio/src/ChainState.resi +++ b/packages/envio/src/ChainState.resi @@ -125,6 +125,8 @@ let isReadyToEnterReorgThresholdAfterBatch: (t, ~batch: Batch.t) => bool // Derived (pure). let takeFetchedTo: t => option<(string, int)> +let takeProcessedToEndBlock: t => option +let reorgThresholdEntryMessage: t => string let hasProcessedToEndblock: t => bool let isDurablyCaughtUp: t => bool let getHighestBlockBelowThreshold: t => int diff --git a/packages/envio/src/CrossChainState.res b/packages/envio/src/CrossChainState.res index 88a2eaff3..c7e261a79 100644 --- a/packages/envio/src/CrossChainState.res +++ b/packages/envio/src/CrossChainState.res @@ -171,13 +171,14 @@ let isReadyToEnterReorgThreshold = (crossChainState: t, ~batch) => ->Dict.valuesToArray ->Array.every(cs => cs->ChainState.isReadyToEnterReorgThresholdAfterBatch(~batch)) +// Said by each chain rather than once for the indexer: what crossing changes +// is a chain's own, and the chains of a split run cross in processes that can +// only speak for the ones they drive. let enterReorgThreshold = (crossChainState: t) => { - Logging.info("Reorg threshold reached") - for i in 0 to crossChainState.chainIds->Array.length - 1 { - crossChainState - ->getChainState(crossChainState.chainIds->Array.getUnsafe(i)) - ->ChainState.enterReorgThreshold + let cs = crossChainState->getChainState(crossChainState.chainIds->Array.getUnsafe(i)) + cs->ChainState.enterReorgThreshold + cs->ChainState.logger->Logging.childInfo(cs->ChainState.reorgThresholdEntryMessage) } } @@ -263,13 +264,39 @@ let markCaughtUpOnResume = (crossChainState: t) => { // and switches the indexer to realtime. let markReady = (crossChainState: t, ~readyAt) => { for i in 0 to crossChainState.chainIds->Array.length - 1 { - crossChainState - ->getChainState(crossChainState.chainIds->Array.getUnsafe(i)) - ->ChainState.markReady(~readyAt) + let cs = crossChainState->getChainState(crossChainState.chainIds->Array.getUnsafe(i)) + let wasReady = cs->ChainState.isReady + cs->ChainState.markReady(~readyAt) + // One line per chain, because `ready_at` is one column per chain: what the + // log says and what a reader finds in the row are the same fact. + if !wasReady { + cs + ->ChainState.logger + ->Logging.childInfo("Ready. Caught up, with every index the schema promises it.") + } } crossChainState.isRealtime = true } +// Each chain that has just finished indexing to its end block, said once, by +// the chain it is about. A chain that finishes early says so then, rather than +// when the last chain in its process catches up. +let reportProcessedToEndBlock = (crossChainState: t) => + crossChainState.chainStates + ->Dict.valuesToArray + ->Array.forEach(cs => + switch cs->ChainState.takeProcessedToEndBlock { + | Some(endBlock) => + cs + ->ChainState.logger + ->Logging.childInfo({ + "msg": "Indexed to the end block. Nothing further to index on this chain.", + "block": endBlock, + }) + | None => () + } + ) + // --- Fetch control. --- // Chains ordered furthest-behind first by fetch-frontier progress, so the diff --git a/packages/envio/src/CrossChainState.resi b/packages/envio/src/CrossChainState.resi index a047f0bb5..f11f805b0 100644 --- a/packages/envio/src/CrossChainState.resi +++ b/packages/envio/src/CrossChainState.resi @@ -40,6 +40,7 @@ let applyBatchProgress: (t, ~batch: Batch.t, ~blockTimestampName: string) => uni let markCaughtUpIfSettled: t => unit let markCaughtUpOnResume: t => unit let markReady: (t, ~readyAt: Date.t) => unit +let reportProcessedToEndBlock: t => unit // Fetch control. let priorityOrder: t => array diff --git a/packages/envio/src/FinalizeBackfill.res b/packages/envio/src/FinalizeBackfill.res index 96836f027..b6e5eb92d 100644 --- a/packages/envio/src/FinalizeBackfill.res +++ b/packages/envio/src/FinalizeBackfill.res @@ -11,9 +11,15 @@ // `envio start --chain` process is indexing and never waits on one. let runOnce = async (state: IndexerState.t) => { - Logging.info( - "All chains are caught up. Finalizing the indexer before switching to realtime: flushing pending writes, then creating the indexes the schema promises.", - ) + // Said by the process rather than by each of its chains: the indexes are one + // build over the tables, and the pause is the whole process's. A chain has + // already said it caught up, and says it is ready once this commits. The + // chains are named because the pause is theirs, and a split run has a process + // saying this for each part of it. + Logging.info({ + "msg": "Backfill finished. Flushing pending writes, then building the indexes the schema promises. Indexing is paused until they are built, which on a large database can take a while.", + "chainIds": state->IndexerState.crossChainState->CrossChainState.chainIds, + }) await Writing.flush(state) @@ -32,8 +38,8 @@ let runOnce = async (state: IndexerState.t) => { // Only after the commit: in-memory readiness must never run ahead of the // `ready_at` a restart would read back. + // Says so per chain, which is the grain `ready_at` is committed at. state->IndexerState.markReady(~readyAt) - Logging.info("The indexer is ready. Switching to realtime indexing.") } } diff --git a/packages/envio/src/IndexerState.res b/packages/envio/src/IndexerState.res index ca376ffbb..7bbebf4f9 100644 --- a/packages/envio/src/IndexerState.res +++ b/packages/envio/src/IndexerState.res @@ -544,6 +544,9 @@ let releaseRealtime = (state: t) => { let markReady = (state: t, ~readyAt) => state.crossChainState->CrossChainState.markReady(~readyAt) +let reportProcessedToEndBlock = (state: t) => + state.crossChainState->CrossChainState.reportProcessedToEndBlock + let rollbackState = (state: t) => state.rollbackState let indexerStartTime = (state: t) => state.indexerStartTime let loadManager = (state: t) => state.loadManager diff --git a/packages/envio/src/IndexerState.resi b/packages/envio/src/IndexerState.resi index e73ffafda..7ff6e6f92 100644 --- a/packages/envio/src/IndexerState.resi +++ b/packages/envio/src/IndexerState.resi @@ -104,6 +104,7 @@ let hasArrivedAtHead: t => bool // The supervisor's go-ahead for a process driving part of a split run. let releaseRealtime: t => unit let markReady: (t, ~readyAt: Date.t) => unit +let reportProcessedToEndBlock: t => unit let rollbackState: t => rollbackState let indexerStartTime: t => Date.t let loadManager: t => LoadManager.t From 9ea5909856f913cdfdf8ac1470f064f0e5671365 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 21 Sep 2026 13:04:27 +0000 Subject: [PATCH 52/61] Say what a run is doing in words its user already has The lifecycle lines named the machinery rather than what was happening to somebody's indexer. A reader was told it had entered the reorg threshold, fetched to the safe block, finished a backfill and built the indexes the schema promises them, none of which is a phrase that appears anywhere a user reads: config.yaml has `max_reorg_depth`, `rollback_on_reorg` and `end_block`, and the display says synced. Each now says what changed and what follows from it. Crossing into the recent blocks is the indexer starting to index up to the latest block and starting to keep what a rollback would need, which is the answer to why the writes grow. Finalizing is the pause it is, and says the indexes are what make queries fast. The block a chain stops short of is named by why it stops there rather than by what the code calls it. Internal comments keep the precise terms, and so does the metric named after one. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_013saxnEKXJJXrZCkodmDMzE --- packages/e2e-tests/src/e2e/split-run.test.ts | 4 ++-- .../test/lib_tests/ChainMilestones_test.res | 4 ++-- .../envio-tests/test/lib_tests/FetchedTo_test.res | 6 +++--- packages/envio/src/ChainState.res | 14 +++++++------- packages/envio/src/CrossChainState.res | 4 ++-- packages/envio/src/FinalizeBackfill.res | 2 +- packages/envio/src/PgStorage.res | 2 +- packages/envio/src/Supervisor.res | 6 +++--- 8 files changed, 21 insertions(+), 21 deletions(-) diff --git a/packages/e2e-tests/src/e2e/split-run.test.ts b/packages/e2e-tests/src/e2e/split-run.test.ts index 48ddcf2df..34316a563 100644 --- a/packages/e2e-tests/src/e2e/split-run.test.ts +++ b/packages/e2e-tests/src/e2e/split-run.test.ts @@ -90,7 +90,7 @@ describe.skipIf(!reachable)("E2E: a split run is one indexer", () => { const indexer = start(["-r"]); try { const exit = exitCode(indexer); - await waitForOutput(indexer, "Splitting 2 chains across 2 processes", config.timeouts.indexerStartup); + await waitForOutput(indexer, "Indexing 2 chains across 2 processes", config.timeouts.indexerStartup); expect({ exitCode: await exit, @@ -125,7 +125,7 @@ describe.skipIf(!reachable)("E2E: a split run is one indexer", () => { const indexer = start(["-r", "--config", "config.head.yaml"]); try { const exit = exitCode(indexer); - await waitForOutput(indexer, "Splitting 2 chains across 2 processes", config.timeouts.indexerStartup); + await waitForOutput(indexer, "Indexing 2 chains across 2 processes", config.timeouts.indexerStartup); const [runtime, metrics] = await Promise.all([ // Each worker's readings, told apart by label. diff --git a/packages/envio-tests/test/lib_tests/ChainMilestones_test.res b/packages/envio-tests/test/lib_tests/ChainMilestones_test.res index 28da76f6a..43cde7fcb 100644 --- a/packages/envio-tests/test/lib_tests/ChainMilestones_test.res +++ b/packages/envio-tests/test/lib_tests/ChainMilestones_test.res @@ -43,8 +43,8 @@ describe("ChainState.reorgThresholdEntryMessage", () => { chainState->ChainState.enterReorgThreshold t.expect((beforeCrossing, chainState->ChainState.reorgThresholdEntryMessage)).toStrictEqual(( - "Entered the reorg threshold. Indexing to the chain head from here.", - "Entered the reorg threshold. Indexing to the chain head from here, and keeping the entity history a reorg would be rolled back through.", + "Now indexing up to the latest block.", + "Now indexing up to the latest block. These can still be reorged, so changes are kept ready to roll back.", )) }) }) diff --git a/packages/envio-tests/test/lib_tests/FetchedTo_test.res b/packages/envio-tests/test/lib_tests/FetchedTo_test.res index cab18ac9a..f31e29b89 100644 --- a/packages/envio-tests/test/lib_tests/FetchedTo_test.res +++ b/packages/envio-tests/test/lib_tests/FetchedTo_test.res @@ -23,8 +23,8 @@ describe("ChainState.takeFetchedTo", () => { ~endBlock=Some(600), )->takeFrom, ]).toStrictEqual([ - Some(("the safe block", 800)), - Some(("the chain head", 1000)), + Some(("the last block that can't be reorged", 800)), + Some(("the head", 1000)), Some(("the end block", 600)), ]) }) @@ -35,6 +35,6 @@ describe("ChainState.takeFetchedTo", () => { t.expect(( chainState->ChainState.takeFetchedTo, chainState->ChainState.takeFetchedTo, - )).toStrictEqual((Some(("the safe block", 800)), None)) + )).toStrictEqual((Some(("the last block that can't be reorged", 800)), None)) }) }) diff --git a/packages/envio/src/ChainState.res b/packages/envio/src/ChainState.res index de0195fed..b896ac216 100644 --- a/packages/envio/src/ChainState.res +++ b/packages/envio/src/ChainState.res @@ -663,8 +663,8 @@ let dispatch = ( ) | _ => cs.isInReorgThreshold - ? ("the chain head", cs.fetchState.knownHeight) - : ("the safe block", cs.fetchState.knownHeight - cs.fetchState.blockLag) + ? ("the head", cs.fetchState.knownHeight) + : ("the last block that can't be reorged", cs.fetchState.knownHeight - cs.fetchState.blockLag) } ) @@ -1017,14 +1017,14 @@ let isInReorgThreshold = (cs: t) => cs.isInReorgThreshold // progress has run. let shouldSaveHistory = (cs: t) => cs.shouldRollbackOnReorg && cs.maxReorgDepth > 0 && cs.isInReorgThreshold -// What entering the reorg threshold changed for this chain. Below it a chain -// stops short of the head by its reorg depth, because it keeps no history to -// roll back with; crossing lifts both at once, and the second half is the +// What crossing into the recent blocks changed for this chain. Until now it +// stopped short of the head by its reorg depth, because it kept nothing it +// could roll back with; crossing lifts both at once. The second half is the // answer to why the indexer starts writing more than it was. let reorgThresholdEntryMessage = (cs: t) => cs->shouldSaveHistory - ? "Entered the reorg threshold. Indexing to the chain head from here, and keeping the entity history a reorg would be rolled back through." - : "Entered the reorg threshold. Indexing to the chain head from here." + ? "Now indexing up to the latest block. These can still be reorged, so changes are kept ready to roll back." + : "Now indexing up to the latest block." // Snapshot the chain's metadata fields for staging into the chains table. let toChainMetadata = (cs: t): InternalTable.Chains.metaFields => { diff --git a/packages/envio/src/CrossChainState.res b/packages/envio/src/CrossChainState.res index c7e261a79..f88bb5483 100644 --- a/packages/envio/src/CrossChainState.res +++ b/packages/envio/src/CrossChainState.res @@ -272,7 +272,7 @@ let markReady = (crossChainState: t, ~readyAt) => { if !wasReady { cs ->ChainState.logger - ->Logging.childInfo("Ready. Caught up, with every index the schema promises it.") + ->Logging.childInfo("Ready. Synced to the head and fully indexed for queries.") } } crossChainState.isRealtime = true @@ -290,7 +290,7 @@ let reportProcessedToEndBlock = (crossChainState: t) => cs ->ChainState.logger ->Logging.childInfo({ - "msg": "Indexed to the end block. Nothing further to index on this chain.", + "msg": "Indexed to the end block. This chain is done.", "block": endBlock, }) | None => () diff --git a/packages/envio/src/FinalizeBackfill.res b/packages/envio/src/FinalizeBackfill.res index b6e5eb92d..061bf52d9 100644 --- a/packages/envio/src/FinalizeBackfill.res +++ b/packages/envio/src/FinalizeBackfill.res @@ -17,7 +17,7 @@ let runOnce = async (state: IndexerState.t) => { // chains are named because the pause is theirs, and a split run has a process // saying this for each part of it. Logging.info({ - "msg": "Backfill finished. Flushing pending writes, then building the indexes the schema promises. Indexing is paused until they are built, which on a large database can take a while.", + "msg": "Synced. Saving the last of the data, then building database indexes. Indexing is paused until that finishes, which can take a while on a large database.", "chainIds": state->IndexerState.crossChainState->CrossChainState.chainIds, }) diff --git a/packages/envio/src/PgStorage.res b/packages/envio/src/PgStorage.res index 026a3a5e9..b9f4a3fa4 100644 --- a/packages/envio/src/PgStorage.res +++ b/packages/envio/src/PgStorage.res @@ -1857,7 +1857,7 @@ let make = ( if withUpload { // Try to restore cache tables from the .envio/cache TSV files switch await scanCacheDir() { - | [] => Logging.info("No cache found to upload.") + | [] => Logging.info("No saved effect cache to load from .envio/cache.") | entries => switch await getConnectedPsqlExec(~pgUser, ~pgHost, ~pgDatabase, ~pgPort, ~containerName) { | Ok(psqlExec) => diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index cc9be8ea7..90898cc7a 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -384,7 +384,7 @@ let startReleaseCheck = group => group.running->Array.forEach(r => r.child->NodeJs.ChildProcess.send(Worker.ReleaseRealtime)->ignore ) - Logging.info("Every chain has reached the head. Switching the run to realtime.") + Logging.info("Every chain has caught up. Switching to realtime indexing.") } , releaseCheckIntervalMillis), ) @@ -411,12 +411,12 @@ let run = async (~config: Config.t, ~workers: array, ~reset) => { let startTimeRef = Performance.now() Logging.info( - `Splitting ${config.chainMap + `Indexing ${config.chainMap ->ChainMap.values ->Array.length ->Int.toString} chains across ${workers ->Array.length - ->Int.toString} processes, from a budget of ${Env.Db.maxConnections->Int.toString} database connections.`, + ->Int.toString} processes, from a limit of ${Env.Db.maxConnections->Int.toString} database connections.`, ) // Decided before the first fork: it is what makes a worker's output the From 5c9c44bf81558313aeff9a26a7f8e3078b817d59 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 21 Sep 2026 13:40:52 +0000 Subject: [PATCH 53/61] Log what a run does, not the steps it takes to do it The transition out of backfill was told three times over: a chain said what it had fetched, then the supervisor said the run was switching to realtime, then each process said it had synced. None of the three was quite true when it was said. Fetching to a block is not indexing it, realtime does not begin until the indexes are built two steps later, and the display already uses synced for that last step, not the first. A chain now reports where it finished indexing, once: its end block if it has one, otherwise its history, saying what it waits on when it is waiting. Crossing into the blocks that can still be reorged is still worth a line, since it is the only account of why the writes grow, but only for a chain the crossing gives something more to index. A chain whose end block sits below those blocks was never held back and says nothing. What is left says what it does: the indexes are being built and indexing is paused for it, and a chain is ready once queries are fully indexed. The two remaining debug lines are traces now, so nothing between trace and info. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_013saxnEKXJJXrZCkodmDMzE --- packages/e2e-tests/src/e2e/split-run.test.ts | 4 +- .../test/lib_tests/ChainMilestones_test.res | 59 +++++++++--- .../test/lib_tests/FetchedTo_test.res | 40 --------- packages/envio/src/BatchProcessing.res | 8 +- packages/envio/src/ChainFetching.res | 29 ------ packages/envio/src/ChainState.res | 90 ++++++++----------- packages/envio/src/ChainState.resi | 5 +- packages/envio/src/CrossChainState.res | 38 +++++--- packages/envio/src/CrossChainState.resi | 2 +- packages/envio/src/FinalizeBackfill.res | 2 +- packages/envio/src/IndexerState.res | 3 +- packages/envio/src/IndexerState.resi | 2 +- packages/envio/src/Persistence.res | 2 +- packages/envio/src/PgStorage.res | 2 +- packages/envio/src/Supervisor.res | 18 ++-- 15 files changed, 136 insertions(+), 168 deletions(-) delete mode 100644 packages/envio-tests/test/lib_tests/FetchedTo_test.res diff --git a/packages/e2e-tests/src/e2e/split-run.test.ts b/packages/e2e-tests/src/e2e/split-run.test.ts index 34316a563..3d53450d0 100644 --- a/packages/e2e-tests/src/e2e/split-run.test.ts +++ b/packages/e2e-tests/src/e2e/split-run.test.ts @@ -90,7 +90,7 @@ describe.skipIf(!reachable)("E2E: a split run is one indexer", () => { const indexer = start(["-r"]); try { const exit = exitCode(indexer); - await waitForOutput(indexer, "Indexing 2 chains across 2 processes", config.timeouts.indexerStartup); + await waitForOutput(indexer, "Indexing will be split across multiple processes", config.timeouts.indexerStartup); expect({ exitCode: await exit, @@ -125,7 +125,7 @@ describe.skipIf(!reachable)("E2E: a split run is one indexer", () => { const indexer = start(["-r", "--config", "config.head.yaml"]); try { const exit = exitCode(indexer); - await waitForOutput(indexer, "Indexing 2 chains across 2 processes", config.timeouts.indexerStartup); + await waitForOutput(indexer, "Indexing will be split across multiple processes", config.timeouts.indexerStartup); const [runtime, metrics] = await Promise.all([ // Each worker's readings, told apart by label. diff --git a/packages/envio-tests/test/lib_tests/ChainMilestones_test.res b/packages/envio-tests/test/lib_tests/ChainMilestones_test.res index 43cde7fcb..a7d547051 100644 --- a/packages/envio-tests/test/lib_tests/ChainMilestones_test.res +++ b/packages/envio-tests/test/lib_tests/ChainMilestones_test.res @@ -8,28 +8,59 @@ open TestChainMetrics // speak for chains it doesn't drive, and an unsplit run says exactly the same // things about exactly the same chains. -describe("ChainState.takeProcessedToEndBlock", () => { - it("Names the end block a chain finished, once, and only once it has", t => { - let unfinished = makeChainState( +describe("ChainState.takeFinished", () => { + it("Names where a chain finished indexing, once, and only once it has", t => { + let stillWorking = makeChainState( ~progressBlockNumber=500, ~firstEventBlockNumber=None, ~endBlock=Some(600), ) - let finished = makeChainState( + let atEndBlock = makeChainState( ~progressBlockNumber=600, ~firstEventBlockNumber=None, ~endBlock=Some(600), ) - // A chain that runs to the head has no end block to finish. - let endless = makeChainState(~progressBlockNumber=600, ~firstEventBlockNumber=None) t.expect(( - unfinished->ChainState.takeProcessedToEndBlock, - finished->ChainState.takeProcessedToEndBlock, - // Reaching it stays true, and every later pass would say so again. - finished->ChainState.takeProcessedToEndBlock, - endless->ChainState.takeProcessedToEndBlock, - )).toStrictEqual((None, Some(600), None, None)) + stillWorking->ChainState.takeFinished, + atEndBlock->ChainState.takeFinished, + // Being finished stays true, and every later pass would say so again. + atEndBlock->ChainState.takeFinished, + )).toStrictEqual((None, Some(ChainState.EndBlock(600)), None)) + }) +}) + +// `makeChainState` puts the head at 1000 with a reorg depth of 200, so a chain +// is held at block 800 until the indexer crosses into the blocks above it. +describe("ChainState.reorgThresholdLiftsCeiling", () => { + it("Is nothing to a chain whose end block sits below the blocks it opens up", t => { + t.expect([ + // Runs to the head, so crossing is what lets it get there. + makeChainState(~progressBlockNumber=500, ~firstEventBlockNumber=None), + // An end block above the held frontier: crossing lets it reach the rest. + makeChainState( + ~progressBlockNumber=500, + ~firstEventBlockNumber=None, + ~endBlock=Some(900), + ), + // An end block below it was never held back, whether or not the chain + // has got there yet. + makeChainState( + ~progressBlockNumber=500, + ~firstEventBlockNumber=None, + ~endBlock=Some(600), + ), + makeChainState( + ~progressBlockNumber=600, + ~firstEventBlockNumber=None, + ~endBlock=Some(600), + ), + ]->Array.map(ChainState.reorgThresholdLiftsCeiling)).toStrictEqual([ + true, + true, + false, + false, + ]) }) }) @@ -43,8 +74,8 @@ describe("ChainState.reorgThresholdEntryMessage", () => { chainState->ChainState.enterReorgThreshold t.expect((beforeCrossing, chainState->ChainState.reorgThresholdEntryMessage)).toStrictEqual(( - "Now indexing up to the latest block.", - "Now indexing up to the latest block. These can still be reorged, so changes are kept ready to roll back.", + "Indexing the latest blocks now.", + "Indexing the latest blocks now. They can still be reorged, so changes are saved in a way that can be rolled back.", )) }) }) diff --git a/packages/envio-tests/test/lib_tests/FetchedTo_test.res b/packages/envio-tests/test/lib_tests/FetchedTo_test.res deleted file mode 100644 index f31e29b89..000000000 --- a/packages/envio-tests/test/lib_tests/FetchedTo_test.res +++ /dev/null @@ -1,40 +0,0 @@ -open Vitest - -open TestChainMetrics - -// What a chain reports reaching, and how often. Below the reorg threshold a -// chain may only fetch the finalized range, so reaching the end of that is not -// reaching the head — and since the head moves, it reaches whatever it is -// fetching to over and over. -describe("ChainState.takeFetchedTo", () => { - it("Names the block a chain has fetched to, not the one it hasn't", t => { - let takeFrom = cs => cs->ChainState.takeFetchedTo - - t.expect([ - makeChainState(~progressBlockNumber=500, ~firstEventBlockNumber=None)->takeFrom, - makeChainState( - ~progressBlockNumber=500, - ~firstEventBlockNumber=None, - ~isInReorgThreshold=true, - )->takeFrom, - makeChainState( - ~progressBlockNumber=500, - ~firstEventBlockNumber=None, - ~endBlock=Some(600), - )->takeFrom, - ]).toStrictEqual([ - Some(("the last block that can't be reorged", 800)), - Some(("the head", 1000)), - Some(("the end block", 600)), - ]) - }) - - it("Reports a milestone once, however many times the chain reaches it", t => { - let chainState = makeChainState(~progressBlockNumber=500, ~firstEventBlockNumber=None) - - t.expect(( - chainState->ChainState.takeFetchedTo, - chainState->ChainState.takeFetchedTo, - )).toStrictEqual((Some(("the last block that can't be reorged", 800)), None)) - }) -}) diff --git a/packages/envio/src/BatchProcessing.res b/packages/envio/src/BatchProcessing.res index 51330e780..7804f0487 100644 --- a/packages/envio/src/BatchProcessing.res +++ b/packages/envio/src/BatchProcessing.res @@ -105,7 +105,7 @@ and processNextBatch = async (state: IndexerState.t, ~scheduleFetch): unit => { } // When resuming from persisted state, all events may already be processed. - state->IndexerState.reportProcessedToEndBlock + state->IndexerState.reportFinished if EventProcessing.allChainsEventsProcessedToEndblock(state->IndexerState.chainStates) { if !(state->IndexerState.keepProcessAlive) && !(state->IndexerState.isHoldingRealtime) { await ExitOnCaughtUp.run(state) @@ -158,9 +158,9 @@ and processNextBatch = async (state: IndexerState.t, ~scheduleFetch): unit => { // Can safely reset rollback state, since overwrite is not possible. state->IndexerState.clearRollback state->IndexerState.applyBatchProgress(~batch) - // Before the finalize below, so a chain says it reached its end block - // ahead of the run saying what it does about that. - state->IndexerState.reportProcessedToEndBlock + // Before the finalize below, so a chain says where it finished ahead of + // the process saying what it does about that. + state->IndexerState.reportFinished // Backfilling → FinalizingIndexes → Ready. Awaiting here holds the // processing loop for the whole finalize, which is what pauses diff --git a/packages/envio/src/ChainFetching.res b/packages/envio/src/ChainFetching.res index d991775ae..3f7fbefd3 100644 --- a/packages/envio/src/ChainFetching.res +++ b/packages/envio/src/ChainFetching.res @@ -279,35 +279,6 @@ and applyQueryResponse = ( ) } - // Report the milestone this response brought the fetch frontier to, once. - // The chain reaches it again every time the head moves and it catches up, so - // what keeps the line off the log is the chain having already reported it, - // not the transition — which re-arms on every advance. - switch chainState->ChainState.isFetchingAtHead - ? chainState->ChainState.takeFetchedTo - : None { - | None => () - | Some((target, block)) => - // What the chain is waiting on, which is not itself. A held process is - // waiting on chains it doesn't drive, so it says so even when it drives - // only one — how the run is split is not the reader's problem. - // - // Naming the wait matters most at the safe block, which is as far as a - // chain can fetch until the indexer enters the reorg threshold: it looks - // stalled short of the head, and the reason is the chains it is waiting on. - let waitingOn = if ( - state->IndexerState.isHoldingRealtime || - state->IndexerState.chainStates->Dict.keysToArray->Array.length > 1 - ) { - " Holding here until every chain has caught up." - } else { - "" - } - chainState->ChainState.logger->Logging.childInfo({ - "msg": `Fetched to ${target}.${waitingOn}`, - "block": block, - }) - } } let finishWaitingForNewBlock = ( diff --git a/packages/envio/src/ChainState.res b/packages/envio/src/ChainState.res index b896ac216..e0c0ac385 100644 --- a/packages/envio/src/ChainState.res +++ b/packages/envio/src/ChainState.res @@ -58,13 +58,10 @@ type t = { mutable blockRangeFetchCount: float, mutable blockRangeFetchedEvents: float, mutable blockRangeFetchedBlocks: float, - // What this chain last reported reaching. The head moves, so a chain reaches - // it again on every catch-up; only a new milestone is worth a line. - mutable reportedFetchedTo: option, - // Whether this chain has said it finished its end block. Reaching it stays - // true for the rest of the run, and every pass over the chains would say so - // again. - mutable reportedEndBlock: bool, + // Whether this chain has said where it finished indexing. Being finished + // stays true for the rest of the run, and every pass over the chains would + // say so again. + mutable reportedFinished: bool, mutable reorgCount: int, mutable reorgDetectedBlock: option, mutable rollbackTargetBlock: option, @@ -178,8 +175,7 @@ let make = ( blockRangeFetchCount: 0., blockRangeFetchedEvents: 0., blockRangeFetchedBlocks: 0., - reportedFetchedTo: None, - reportedEndBlock: false, + reportedFinished: false, reorgCount: 0, reorgDetectedBlock: None, rollbackTargetBlock: None, @@ -651,37 +647,6 @@ let dispatch = ( // --- Derived (pure). --- -%%private( - // Where the fetch frontier has just landed. Below the reorg threshold a chain - // can only fetch the finalized range, so reaching the end of it is not - // reaching the head — the rest opens up once the indexer enters the threshold. - let fetchedTo = (cs: t) => - switch cs.fetchState.endBlock { - | Some(endBlock) if endBlock <= cs.fetchState.knownHeight - cs.fetchState.blockLag => ( - "the end block", - endBlock, - ) - | _ => - cs.isInReorgThreshold - ? ("the head", cs.fetchState.knownHeight) - : ("the last block that can't be reorged", cs.fetchState.knownHeight - cs.fetchState.blockLag) - } -) - -// What this chain has just reached, the first time it reaches it. `None` once -// it has been reported: a chain catches up to a moving head over and over, and -// the milestone is the same one every time. A new one — the head past the -// threshold, an end block — is a line of its own. -let takeFetchedTo = (cs: t) => { - let (target, block) = cs->fetchedTo - if cs.reportedFetchedTo == Some(target) { - None - } else { - cs.reportedFetchedTo = Some(target) - Some((target, block)) - } -} - let hasProcessedToEndblock = (cs: t) => { let {committedProgressBlockNumber, fetchState} = cs switch fetchState.endBlock { @@ -690,16 +655,28 @@ let hasProcessedToEndblock = (cs: t) => { } } -// The end block this chain has just finished indexing to, the first time it -// has. `None` for a chain still working, one with no end block at all, and -// every pass after the one that reported it. -let takeProcessedToEndBlock = (cs: t) => - switch cs.fetchState.endBlock { - | Some(endBlock) if !cs.reportedEndBlock && cs->hasProcessedToEndblock => { - cs.reportedEndBlock = true - Some(endBlock) +// Where this chain has finished indexing, the first time it gets there. +// `EndBlock` is terminal: the chain indexed everything it was configured to. +// `Backfill` is the rest of the history, up to the point where blocks can +// still be reorged, which is as far as a chain indexes before the indexer +// crosses into them. +type finished = EndBlock(int) | Backfill(int) + +let takeFinished = (cs: t) => + if cs.reportedFinished { + None + } else { + switch (cs.fetchState.endBlock, cs->hasProcessedToEndblock, cs.isProgressAtHead) { + | (Some(endBlock), true, _) => { + cs.reportedFinished = true + Some(EndBlock(endBlock)) + } + | (_, _, true) => { + cs.reportedFinished = true + Some(Backfill(cs.committedProgressBlockNumber)) + } + | _ => None } - | _ => None } // Caught up as judged by persisted values alone: progress reached the endBlock, @@ -1021,10 +998,21 @@ let shouldSaveHistory = (cs: t) => // stopped short of the head by its reorg depth, because it kept nothing it // could roll back with; crossing lifts both at once. The second half is the // answer to why the indexer starts writing more than it was. +// Whether crossing into the recent blocks gives this chain anything more to +// index. A chain whose end block already sits below the lagged head was never +// held back by the lag, so crossing changes nothing about it and it has nothing +// to say. Read before the crossing, while the lag it was holding back is still +// the one in place. +let reorgThresholdLiftsCeiling = (cs: t) => + switch cs.fetchState.endBlock { + | Some(endBlock) => endBlock > cs.fetchState.knownHeight - cs.fetchState.blockLag + | None => true + } + let reorgThresholdEntryMessage = (cs: t) => cs->shouldSaveHistory - ? "Now indexing up to the latest block. These can still be reorged, so changes are kept ready to roll back." - : "Now indexing up to the latest block." + ? "Indexing the latest blocks now. They can still be reorged, so changes are saved in a way that can be rolled back." + : "Indexing the latest blocks now." // Snapshot the chain's metadata fields for staging into the chains table. let toChainMetadata = (cs: t): InternalTable.Chains.metaFields => { diff --git a/packages/envio/src/ChainState.resi b/packages/envio/src/ChainState.resi index 41a16953f..a99879e95 100644 --- a/packages/envio/src/ChainState.resi +++ b/packages/envio/src/ChainState.resi @@ -124,8 +124,9 @@ let isReadyToEnterReorgThreshold: t => bool let isReadyToEnterReorgThresholdAfterBatch: (t, ~batch: Batch.t) => bool // Derived (pure). -let takeFetchedTo: t => option<(string, int)> -let takeProcessedToEndBlock: t => option +type finished = EndBlock(int) | Backfill(int) +let takeFinished: t => option +let reorgThresholdLiftsCeiling: t => bool let reorgThresholdEntryMessage: t => string let hasProcessedToEndblock: t => bool let isDurablyCaughtUp: t => bool diff --git a/packages/envio/src/CrossChainState.res b/packages/envio/src/CrossChainState.res index f88bb5483..e79aaa759 100644 --- a/packages/envio/src/CrossChainState.res +++ b/packages/envio/src/CrossChainState.res @@ -177,8 +177,13 @@ let isReadyToEnterReorgThreshold = (crossChainState: t, ~batch) => let enterReorgThreshold = (crossChainState: t) => { for i in 0 to crossChainState.chainIds->Array.length - 1 { let cs = crossChainState->getChainState(crossChainState.chainIds->Array.getUnsafe(i)) + // A chain whose end block sits below these blocks says nothing: crossing + // gives it nothing more to index, and it will never reach one of them. + let liftsCeiling = cs->ChainState.reorgThresholdLiftsCeiling cs->ChainState.enterReorgThreshold - cs->ChainState.logger->Logging.childInfo(cs->ChainState.reorgThresholdEntryMessage) + if liftsCeiling { + cs->ChainState.logger->Logging.childInfo(cs->ChainState.reorgThresholdEntryMessage) + } } } @@ -272,30 +277,43 @@ let markReady = (crossChainState: t, ~readyAt) => { if !wasReady { cs ->ChainState.logger - ->Logging.childInfo("Ready. Synced to the head and fully indexed for queries.") + ->Logging.childInfo("Ready. Fully indexed for queries.") } } crossChainState.isRealtime = true } -// Each chain that has just finished indexing to its end block, said once, by -// the chain it is about. A chain that finishes early says so then, rather than -// when the last chain in its process catches up. -let reportProcessedToEndBlock = (crossChainState: t) => +// Each chain that has just finished indexing, said once, by the chain it is +// about. A chain that finishes early says so then, rather than when the last +// chain in its process catches up. +// +// A chain with no end block has only finished its history: the blocks that can +// still be reorged are indexed after every chain gets this far, so it says what +// it is waiting on when there is anything to wait for. +let reportFinished = (crossChainState: t) => { + let waitingOnOthers = + crossChainState.holdRealtime || crossChainState.chainIds->Array.length > 1 crossChainState.chainStates ->Dict.valuesToArray ->Array.forEach(cs => - switch cs->ChainState.takeProcessedToEndBlock { - | Some(endBlock) => + switch cs->ChainState.takeFinished { + | Some(EndBlock(block)) => + cs + ->ChainState.logger + ->Logging.childInfo({"msg": "Indexed to the end block. This chain is done.", "block": block}) + | Some(Backfill(block)) => cs ->ChainState.logger ->Logging.childInfo({ - "msg": "Indexed to the end block. This chain is done.", - "block": endBlock, + "msg": waitingOnOthers + ? "Finished backfill. Waiting for the other chains." + : "Finished backfill.", + "block": block, }) | None => () } ) +} // --- Fetch control. --- diff --git a/packages/envio/src/CrossChainState.resi b/packages/envio/src/CrossChainState.resi index f11f805b0..320e2bdd5 100644 --- a/packages/envio/src/CrossChainState.resi +++ b/packages/envio/src/CrossChainState.resi @@ -40,7 +40,7 @@ let applyBatchProgress: (t, ~batch: Batch.t, ~blockTimestampName: string) => uni let markCaughtUpIfSettled: t => unit let markCaughtUpOnResume: t => unit let markReady: (t, ~readyAt: Date.t) => unit -let reportProcessedToEndBlock: t => unit +let reportFinished: t => unit // Fetch control. let priorityOrder: t => array diff --git a/packages/envio/src/FinalizeBackfill.res b/packages/envio/src/FinalizeBackfill.res index 061bf52d9..c4a7383ee 100644 --- a/packages/envio/src/FinalizeBackfill.res +++ b/packages/envio/src/FinalizeBackfill.res @@ -17,7 +17,7 @@ let runOnce = async (state: IndexerState.t) => { // chains are named because the pause is theirs, and a split run has a process // saying this for each part of it. Logging.info({ - "msg": "Synced. Saving the last of the data, then building database indexes. Indexing is paused until that finishes, which can take a while on a large database.", + "msg": "Building database indexes. Indexing is paused until they are ready, which can take a while on a large database.", "chainIds": state->IndexerState.crossChainState->CrossChainState.chainIds, }) diff --git a/packages/envio/src/IndexerState.res b/packages/envio/src/IndexerState.res index 7bbebf4f9..13d55e410 100644 --- a/packages/envio/src/IndexerState.res +++ b/packages/envio/src/IndexerState.res @@ -544,8 +544,7 @@ let releaseRealtime = (state: t) => { let markReady = (state: t, ~readyAt) => state.crossChainState->CrossChainState.markReady(~readyAt) -let reportProcessedToEndBlock = (state: t) => - state.crossChainState->CrossChainState.reportProcessedToEndBlock +let reportFinished = (state: t) => state.crossChainState->CrossChainState.reportFinished let rollbackState = (state: t) => state.rollbackState let indexerStartTime = (state: t) => state.indexerStartTime diff --git a/packages/envio/src/IndexerState.resi b/packages/envio/src/IndexerState.resi index 7ff6e6f92..83748bbd6 100644 --- a/packages/envio/src/IndexerState.resi +++ b/packages/envio/src/IndexerState.resi @@ -104,7 +104,7 @@ let hasArrivedAtHead: t => bool // The supervisor's go-ahead for a process driving part of a split run. let releaseRealtime: t => unit let markReady: (t, ~readyAt: Date.t) => unit -let reportProcessedToEndBlock: t => unit +let reportFinished: t => unit let rollbackState: t => rollbackState let indexerStartTime: t => Date.t let loadManager: t => LoadManager.t diff --git a/packages/envio/src/Persistence.res b/packages/envio/src/Persistence.res index 19223a1ca..5fd922f65 100644 --- a/packages/envio/src/Persistence.res +++ b/packages/envio/src/Persistence.res @@ -328,7 +328,7 @@ let init = { | _ => false } ) { - let logResume = announceResume ? Logging.info : Logging.debug + let logResume = announceResume ? Logging.info : Logging.trace logResume(`Found existing indexer storage. Resuming indexing state...`) let initialState = await persistence.storage.resumeInitialState( ~entities=persistence.allEntities, diff --git a/packages/envio/src/PgStorage.res b/packages/envio/src/PgStorage.res index b9f4a3fa4..0f63fd35d 100644 --- a/packages/envio/src/PgStorage.res +++ b/packages/envio/src/PgStorage.res @@ -2104,7 +2104,7 @@ let make = ( switch await sql->loadCatalogRows(~indexName=name) { | rows => indexManager->IndexManager.resync(~name, ~rows) | exception exn => - Logging.debug({ + Logging.trace({ "storage": storageName, "msg": `Could not re-read the index "${name}" after a failed build. The next attempt reads it again.`, "err": exn->Utils.prettifyExn, diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index 90898cc7a..89d20fb92 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -384,7 +384,6 @@ let startReleaseCheck = group => group.running->Array.forEach(r => r.child->NodeJs.ChildProcess.send(Worker.ReleaseRealtime)->ignore ) - Logging.info("Every chain has caught up. Switching to realtime indexing.") } , releaseCheckIntervalMillis), ) @@ -410,14 +409,15 @@ let run = async (~config: Config.t, ~workers: array, ~reset) => { let startTime = Date.make() let startTimeRef = Performance.now() - Logging.info( - `Indexing ${config.chainMap - ->ChainMap.values - ->Array.length - ->Int.toString} chains across ${workers - ->Array.length - ->Int.toString} processes, from a limit of ${Env.Db.maxConnections->Int.toString} database connections.`, - ) + // The counts ride as fields rather than in the sentence: the connection limit + // is the only setting that decides any of this, and a reader who wants to + // change it has nothing else to go on. + Logging.info({ + "msg": "Indexing will be split across multiple processes for faster and more reliable indexing.", + "chains": config.chainMap->ChainMap.values->Array.length, + "processes": workers->Array.length, + "maxConnections": Env.Db.maxConnections, + }) // Decided before the first fork: it is what makes a worker's output the // supervisor's to print. From cc3f36a8b7c5d55d81f45540e6aae6a0abf5a83b Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 21 Sep 2026 13:52:06 +0000 Subject: [PATCH 54/61] Say which history the latest blocks start, and say it only where it starts MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The line said changes were "saved in a way that can be rolled back", which names no mechanism a reader could go and look at. It is a history of every change, written from the crossing on, and saying so is what accounts for the writes growing at that moment. It also had a second form for a chain that keeps no history, and that form could never be reached. A chain is only told about the crossing when the crossing gives it more to index, which takes a reorg depth that was holding it back, which is the same thing that makes the history worth keeping. Every configuration without one — no reorg depth, or nothing rolled back on a reorg — already fetched as far as it ever would, so nothing about it changes and it has nothing to say. That was not what the code tested, though: it compared the end block against the held frontier and let the rest through. It now compares the last block the chain may fetch after the crossing against the last it may fetch now, which is the question being asked. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_013saxnEKXJJXrZCkodmDMzE --- .../test/helpers/TestChainMetrics.res | 6 +- .../test/lib_tests/ChainMilestones_test.res | 68 ++++++++----------- packages/envio/src/ChainState.res | 35 ++++++---- packages/envio/src/ChainState.resi | 2 +- packages/envio/src/CrossChainState.res | 2 +- 5 files changed, 58 insertions(+), 55 deletions(-) diff --git a/packages/envio-tests/test/helpers/TestChainMetrics.res b/packages/envio-tests/test/helpers/TestChainMetrics.res index 9cdc6c5d6..3fe19a784 100644 --- a/packages/envio-tests/test/helpers/TestChainMetrics.res +++ b/packages/envio-tests/test/helpers/TestChainMetrics.res @@ -41,6 +41,8 @@ let makeChainState = ( ~timestampCaughtUpToHeadOrEndblock=None, ~sourceBlockNumber=1000, ~isInReorgThreshold=false, + ~maxReorgDepth=200, + ~config=TestConfig.default, ): ChainState.t => ChainState.makeFromDbState( chainConfig, @@ -48,7 +50,7 @@ let makeChainState = ( id: chainId, startBlock: 100, endBlock, - maxReorgDepth: 200, + maxReorgDepth, progressBlockNumber, progressBlockTime: None, numEventsProcessed: 7., @@ -60,7 +62,7 @@ let makeChainState = ( ~reorgCheckpoints=[], ~isInReorgThreshold, ~isRealtime=false, - ~config=TestConfig.default, + ~config, ~contractMapping=TestConfig.default.contractMapping, ~registrationsByChainId, ) diff --git a/packages/envio-tests/test/lib_tests/ChainMilestones_test.res b/packages/envio-tests/test/lib_tests/ChainMilestones_test.res index a7d547051..54bd3b29e 100644 --- a/packages/envio-tests/test/lib_tests/ChainMilestones_test.res +++ b/packages/envio-tests/test/lib_tests/ChainMilestones_test.res @@ -30,52 +30,44 @@ describe("ChainState.takeFinished", () => { }) }) -// `makeChainState` puts the head at 1000 with a reorg depth of 200, so a chain -// is held at block 800 until the indexer crosses into the blocks above it. +// `makeChainState` puts the head at 1000 with a reorg depth of 200 and no +// block lag, so a chain fetches no further than block 800 until the indexer +// crosses into the blocks above it. describe("ChainState.reorgThresholdLiftsCeiling", () => { - it("Is nothing to a chain whose end block sits below the blocks it opens up", t => { - t.expect([ - // Runs to the head, so crossing is what lets it get there. - makeChainState(~progressBlockNumber=500, ~firstEventBlockNumber=None), - // An end block above the held frontier: crossing lets it reach the rest. - makeChainState( - ~progressBlockNumber=500, - ~firstEventBlockNumber=None, - ~endBlock=Some(900), - ), - // An end block below it was never held back, whether or not the chain - // has got there yet. + it("Is true only where the lag was holding the chain back", t => { + let chain = (~endBlock=None, ~maxReorgDepth=200, ~config=TestConfig.default) => makeChainState( ~progressBlockNumber=500, ~firstEventBlockNumber=None, - ~endBlock=Some(600), - ), - makeChainState( - ~progressBlockNumber=600, - ~firstEventBlockNumber=None, - ~endBlock=Some(600), - ), - ]->Array.map(ChainState.reorgThresholdLiftsCeiling)).toStrictEqual([ - true, - true, - false, - false, - ]) + ~endBlock, + ~maxReorgDepth, + ~config, + )->ChainState.reorgThresholdLiftsCeiling + + t.expect([ + // Runs to the head, so crossing is what lets it get there. + chain(), + // An end block above the held frontier: crossing opens up the rest of it. + chain(~endBlock=Some(900)), + // An end block below it was never held back by the lag. + chain(~endBlock=Some(600)), + // No reorg depth to be held back by, so crossing changes nothing. This is + // also every configuration that keeps no history, which is why the line + // has no form that leaves the history out. + chain(~maxReorgDepth=0), + // Nothing is rolled back, so the chain already fetches to the head. + chain(~config={...TestConfig.default, shouldRollbackOnReorg: false}), + ]).toStrictEqual([true, true, false, false, false]) }) }) describe("ChainState.reorgThresholdEntryMessage", () => { - // Crossing lifts the lag that held the chain short of the head, and starts - // the history a rollback replays. A reader watching writes grow wants the + // Crossing lifts the lag that held the chain short of the head and starts the + // history a rollback replays. A reader watching the writes grow wants the // second half of that. - it("Says what crossing changed, and mentions history only when it is kept", t => { - let chainState = makeChainState(~progressBlockNumber=500, ~firstEventBlockNumber=None) - let beforeCrossing = chainState->ChainState.reorgThresholdEntryMessage - chainState->ChainState.enterReorgThreshold - - t.expect((beforeCrossing, chainState->ChainState.reorgThresholdEntryMessage)).toStrictEqual(( - "Indexing the latest blocks now.", - "Indexing the latest blocks now. They can still be reorged, so changes are saved in a way that can be rolled back.", - )) + it("Says what crossing changed, and what it starts writing", t => { + t.expect(ChainState.reorgThresholdEntryMessage).toBe( + "Indexing the latest blocks now. These can be reorged, so the indexer starts storing a history of every change to roll back with.", + ) }) }) diff --git a/packages/envio/src/ChainState.res b/packages/envio/src/ChainState.res index e0c0ac385..cfd7f11e9 100644 --- a/packages/envio/src/ChainState.res +++ b/packages/envio/src/ChainState.res @@ -998,21 +998,30 @@ let shouldSaveHistory = (cs: t) => // stopped short of the head by its reorg depth, because it kept nothing it // could roll back with; crossing lifts both at once. The second half is the // answer to why the indexer starts writing more than it was. -// Whether crossing into the recent blocks gives this chain anything more to -// index. A chain whose end block already sits below the lagged head was never -// held back by the lag, so crossing changes nothing about it and it has nothing -// to say. Read before the crossing, while the lag it was holding back is still -// the one in place. -let reorgThresholdLiftsCeiling = (cs: t) => - switch cs.fetchState.endBlock { - | Some(endBlock) => endBlock > cs.fetchState.knownHeight - cs.fetchState.blockLag - | None => true +// Whether crossing gives this chain anything more to index: the last block it +// may fetch afterwards against the last it may fetch now. Read before the +// crossing, while the lag being lifted is still the one in place. +// +// False wherever the lag was never holding this chain back, which is every +// configuration that also keeps no history: a chain that isn't rolled back on a +// reorg, or has no reorg depth, already fetches as far as it ever will. It is +// false too for a chain whose end block sits below the blocks being opened up, +// which will never reach one of them. +let reorgThresholdLiftsCeiling = (cs: t) => { + let ceiling = (~blockLag) => { + let head = Pervasives.max(0, cs.fetchState.knownHeight - blockLag) + switch cs.fetchState.endBlock { + | Some(endBlock) => Pervasives.min(endBlock, head) + | None => head + } } + ceiling(~blockLag=cs.chainConfig.blockLag) > ceiling(~blockLag=cs.fetchState.blockLag) +} -let reorgThresholdEntryMessage = (cs: t) => - cs->shouldSaveHistory - ? "Indexing the latest blocks now. They can still be reorged, so changes are saved in a way that can be rolled back." - : "Indexing the latest blocks now." +// Only ever said by a chain the crossing lifts, and lifting it takes a reorg +// depth to have been held back by, which is the same thing that makes the +// history below worth keeping. So there is no second form without it. +let reorgThresholdEntryMessage = "Indexing the latest blocks now. These can be reorged, so the indexer starts storing a history of every change to roll back with." // Snapshot the chain's metadata fields for staging into the chains table. let toChainMetadata = (cs: t): InternalTable.Chains.metaFields => { diff --git a/packages/envio/src/ChainState.resi b/packages/envio/src/ChainState.resi index a99879e95..e540700e9 100644 --- a/packages/envio/src/ChainState.resi +++ b/packages/envio/src/ChainState.resi @@ -127,7 +127,7 @@ let isReadyToEnterReorgThresholdAfterBatch: (t, ~batch: Batch.t) => bool type finished = EndBlock(int) | Backfill(int) let takeFinished: t => option let reorgThresholdLiftsCeiling: t => bool -let reorgThresholdEntryMessage: t => string +let reorgThresholdEntryMessage: string let hasProcessedToEndblock: t => bool let isDurablyCaughtUp: t => bool let getHighestBlockBelowThreshold: t => int diff --git a/packages/envio/src/CrossChainState.res b/packages/envio/src/CrossChainState.res index e79aaa759..145f256c5 100644 --- a/packages/envio/src/CrossChainState.res +++ b/packages/envio/src/CrossChainState.res @@ -182,7 +182,7 @@ let enterReorgThreshold = (crossChainState: t) => { let liftsCeiling = cs->ChainState.reorgThresholdLiftsCeiling cs->ChainState.enterReorgThreshold if liftsCeiling { - cs->ChainState.logger->Logging.childInfo(cs->ChainState.reorgThresholdEntryMessage) + cs->ChainState.logger->Logging.childInfo(ChainState.reorgThresholdEntryMessage) } } } From 7f212999cb439990b4325b99b397d3d4bccfcb3a Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 22 Sep 2026 08:44:14 +0000 Subject: [PATCH 55/61] Wait on a line the run still prints Three end-to-end tests gated on "All chains are caught up to end blocks", which no longer exists: a chain says it reached its end block itself now, and says it once. The tests waited two minutes for a line that was never coming and took the whole job down with them. They wait on what the chain says instead. Each of the three drives the same single-chain project with an end block, so the line they were waiting for and the one they wait for now mean the same thing for that run. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_013saxnEKXJJXrZCkodmDMzE --- packages/e2e-tests/src/dependency-tests/install.test.ts | 2 +- packages/e2e-tests/src/e2e/e2e.test.ts | 6 +++--- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/packages/e2e-tests/src/dependency-tests/install.test.ts b/packages/e2e-tests/src/dependency-tests/install.test.ts index 213016281..0e6fcc424 100644 --- a/packages/e2e-tests/src/dependency-tests/install.test.ts +++ b/packages/e2e-tests/src/dependency-tests/install.test.ts @@ -183,7 +183,7 @@ describe("Isolated dependency e2e", () => { await waitForOutput( indexerProcess, - "All chains are caught up to end blocks", + "Indexed to the end block", 120_000 ); diff --git a/packages/e2e-tests/src/e2e/e2e.test.ts b/packages/e2e-tests/src/e2e/e2e.test.ts index 929a4c502..f68784bfc 100644 --- a/packages/e2e-tests/src/e2e/e2e.test.ts +++ b/packages/e2e-tests/src/e2e/e2e.test.ts @@ -4,7 +4,7 @@ * Tests the full indexer flow with database and ClickHouse sink: * 1. Ensure ClickHouse is running (CI service or local container) * 2. Start `envio dev` in background with ClickHouse sink enabled - * 3. Wait for "All chains are caught up to end blocks" in stdout + * 3. Wait for "Indexed to the end block" in stdout * 4. Verify GraphQL queries return expected data * 5. Verify ClickHouse sink received the indexed data */ @@ -96,7 +96,7 @@ describe.skipIf(!dockerAvailable)("E2E: Indexer with GraphQL and ClickHouse sink await waitForOutput( indexerProcess, - "All chains are caught up to end blocks", + "Indexed to the end block", 120_000 ); @@ -842,7 +842,7 @@ describe.skipIf(!dockerAvailable)("E2E: Indexer with GraphQL and ClickHouse sink // waitForOutput rejects. Success means DB state was used. await waitForOutput( secondProcess, - "All chains are caught up to end blocks", + "Indexed to the end block", 120_000 ); } finally { From bcc6c572f19c6de881041ce9b9002eaa1d2df13d Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 22 Sep 2026 09:20:38 +0000 Subject: [PATCH 56/61] Stop spending a budget past four processes A budget bought a worker per two connections with nothing to stop it: ten chains and twenty connections became ten Node processes, each with its own heap, its own copy of the handler modules and its own source clients, all sharing one machine. Four is the ceiling now. Past it a raised budget widens the workers' pools instead of adding workers, so the connections are still spent and the remainder still goes to the earliest of them. A conservative number while the split is new, and the only thing that has to change to raise it. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_013saxnEKXJJXrZCkodmDMzE --- packages/cli/CommandLineHelp.md | 2 +- packages/cli/src/cli_args/clap_definitions.rs | 2 +- .../test/lib_tests/Supervisor_test.res | 23 +++++++++++++++++++ packages/envio/src/Supervisor.res | 13 ++++++++++- 4 files changed, 37 insertions(+), 3 deletions(-) diff --git a/packages/cli/CommandLineHelp.md b/packages/cli/CommandLineHelp.md index bc578bb68..f52a1ceef 100644 --- a/packages/cli/CommandLineHelp.md +++ b/packages/cli/CommandLineHelp.md @@ -373,7 +373,7 @@ Setup database by dropping schema and then running migrations Start the indexer. Runs codegen automatically before launching so the on-disk types stay in sync with `config.yaml` and `schema.graphql`. -A schema whose entities are all per-chain is indexed across several processes, as many as `ENVIO_PG_MAX_CONNECTIONS` affords. List the busiest chains first in `config.yaml` to balance them. +A schema whose entities are all per-chain is indexed across several processes, as many as `ENVIO_PG_MAX_CONNECTIONS` affords, up to four. List the busiest chains first in `config.yaml` to balance them. **Usage:** `envio start [OPTIONS]` diff --git a/packages/cli/src/cli_args/clap_definitions.rs b/packages/cli/src/cli_args/clap_definitions.rs index 4c16e9988..424289683 100644 --- a/packages/cli/src/cli_args/clap_definitions.rs +++ b/packages/cli/src/cli_args/clap_definitions.rs @@ -58,7 +58,7 @@ pub enum CommandType { ///Start the indexer. Runs codegen automatically before launching so the on-disk types stay in sync with `config.yaml` and `schema.graphql`. /// - ///A schema whose entities are all per-chain is indexed across several processes, as many as `ENVIO_PG_MAX_CONNECTIONS` affords. List the busiest chains first in `config.yaml` to balance them. + ///A schema whose entities are all per-chain is indexed across several processes, as many as `ENVIO_PG_MAX_CONNECTIONS` affords, up to four. List the busiest chains first in `config.yaml` to balance them. Start(StartArgs), ///Fetch raw Prometheus metrics from the running indexer's /metrics endpoint diff --git a/packages/envio-tests/test/lib_tests/Supervisor_test.res b/packages/envio-tests/test/lib_tests/Supervisor_test.res index 3f52066e1..449def58b 100644 --- a/packages/envio-tests/test/lib_tests/Supervisor_test.res +++ b/packages/envio-tests/test/lib_tests/Supervisor_test.res @@ -37,6 +37,29 @@ describe("Supervisor.plan", () => { }) }) +describe("Supervisor.plan at the ceiling", () => { + // A worker is a whole Node process: its own heap, its own copy of the + // handler modules, its own source clients. A budget that would buy more of + // them than this spends the surplus widening their pools instead. + it("Never spends a budget on more than four processes", t => { + let plan = (~chainCount, ~maxConnections) => + Supervisor.plan(~chainIds=chains(chainCount), ~maxConnections) + ->Option.getOrThrow + ->Array.map(({maxConnections}: Supervisor.worker) => maxConnections) + + t.expect([ + // Ten chains and the connections for ten workers: four, with the budget + // spread across them rather than two connections each and the rest + // unspent. + plan(~chainCount=10, ~maxConnections=20), + // The remainder still goes to the earliest workers. + plan(~chainCount=10, ~maxConnections=22), + // Below the ceiling nothing changes. + plan(~chainCount=10, ~maxConnections=6), + ]).toStrictEqual([[5, 5, 5, 5], [6, 6, 5, 5], [2, 2, 2]]) + }) +}) + describe("Supervisor.plan on the default budget", () => { // Splitting a run costs connections the operator didn't ask to spend, so the // budget they didn't set is the one a single process has always used. diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index 89d20fb92..b8b3fb5dc 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -6,6 +6,13 @@ type worker = {chainIds: array, maxConnections: int} // on a single one, so the budget buys workers two at a time. let minConnectionsPerWorker = 2 +// Most processes a run is split into, however much budget it is given. A worker +// is a whole Node process, with its own heap, its own copy of the handler +// modules and its own source clients, and a run holding more of them shares one +// machine between them. A conservative ceiling while the split is new: past it +// a raised budget widens the workers' pools rather than adding workers. +let maxWorkers = 4 + // How to spend a connection budget on the chains a run indexes. `None` keeps // the run in one process, which is what a budget too small to afford two // workers, or a config with nothing to split, has to do. @@ -16,7 +23,11 @@ let minConnectionsPerWorker = 2 // not the chain's, so config order is the one ranking the run can be given: // listing chains busiest-first in config.yaml is what balances the layout. let plan = (~chainIds: array, ~maxConnections: int): option> => { - let workerCount = Pervasives.min(chainIds->Array.length, maxConnections / minConnectionsPerWorker) + let workerCount = + [chainIds->Array.length, maxConnections / minConnectionsPerWorker, maxWorkers]->Array.reduce( + maxWorkers, + Pervasives.min, + ) if workerCount < 2 { None } else { From 92afd400171df012ea2fc21ee874a7d0e9e35efe Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 22 Sep 2026 09:35:13 +0000 Subject: [PATCH 57/61] Drop a stop nobody had to ask twice, and a line nobody added `stopOnce` guarded a stop that was already idempotent: signalling a worker that has been signalled changes nothing, clearing a cleared poll changes nothing, and the flag it set was read before the call, never after. The group's own comment still said every ending but a stop it asked for was a failure, which stopped being true when a signalled worker started reading as the run being stopped. Logging.res was left a blank line apart from main. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_013saxnEKXJJXrZCkodmDMzE --- packages/envio/src/Logging.res | 1 - packages/envio/src/Supervisor.res | 15 +++------------ 2 files changed, 3 insertions(+), 13 deletions(-) diff --git a/packages/envio/src/Logging.res b/packages/envio/src/Logging.res index 76a760487..e9c897c69 100644 --- a/packages/envio/src/Logging.res +++ b/packages/envio/src/Logging.res @@ -151,7 +151,6 @@ let childFatal = (logger, params: 'a) => { let createChild = (~params: 'a) => { getLogger()->child(params->createChildParams) } - let createChildFrom = (~logger: t, ~params: 'a) => { logger->child(params->createChildParams) } diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index b8b3fb5dc..54c7f8dff 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -224,8 +224,7 @@ let fork = ( } // The forked workers of one run, and whether their supervisor is the one -// taking them down. A stop it asked for is expected; every other way a worker -// can end is a failure. +// taking them down, which is what tells an expected exit from the rest. type group = { running: array, mutable stopping: bool, @@ -245,14 +244,6 @@ let stop = group => { group.running->Array.forEach(r => r.child->NodeJs.ChildProcess.kill("SIGTERM")->ignore) } -// Takes the group down unless it is already going. Signalling a worker that has -// been signalled changes nothing, but `stopping` is what tells an exit from an -// expected one, so the first stop is the one that counts. -let stopOnce = group => - if !group.stopping { - group->stop - } - // The dev console's cache dump, which belongs to the supervisor rather than to // its workers: a dump copies every effect cache table in the schema to a file // named after the effect, so a worker asked to do it would copy its siblings' @@ -330,10 +321,10 @@ let awaitExit = async (group): outcome => { // run reports itself as having failed. switch ending { | Expected => () - | Stopping => group->stopOnce + | Stopping => group->stop | Failed => { failed := true - group->stopOnce + group->stop } } alive := alive.contents - 1 From 7bf6a3a33e2a31c40f440443a68272a8c2bcf419 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 22 Sep 2026 11:07:30 +0000 Subject: [PATCH 58/61] Address review: trim help text, namespace child_process bindings, format MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Drop the auto-split paragraph from `envio start` help and shorten the `--chain` long help to what the flag requires of the user. - Split `NodeJs.ChildProcess` into `Stream` and `Child` namespaces. - Run the ReScript formatter over the files this branch touches, and add a PreToolUse hook that blocks `git push` while any of them is unformatted — Bash-driven edits bypass the per-file PostToolUse formatter. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_013saxnEKXJJXrZCkodmDMzE --- .claude/hooks/check-format.sh | 44 ++++++++++++++++++ .claude/settings.json | 13 ++++++ packages/cli/CommandLineHelp.md | 6 +-- packages/cli/src/cli_args/clap_definitions.rs | 11 +---- .../test/BelowHeadPollingPin_test.res | 15 +++---- packages/envio-tests/test/E2E_test.res | 6 ++- .../test/EnterReorgThreshold_test.res | 12 +++-- .../test/HandlerChainInfo_test.res | 6 +-- .../test/IsolatedRollback_test.res | 18 +++++--- .../envio-tests/test/PerChainEntity_test.res | 6 ++- .../test/PerChainHistoryPrune_test.res | 18 +++++--- .../envio-tests/test/ResumeFinalize_test.res | 9 ++-- .../test/RollbackDiffCheckpointIds_test.res | 18 +++++--- packages/envio-tests/test/Rollback_test.res | 14 +++--- .../envio-tests/test/SchemaIndexes_test.res | 40 ++++++++++------- .../test/SupervisedRealtime_test.res | 11 ++--- .../envio-tests/test/SupervisorFork_test.res | 19 +++----- .../test/ZeroReorgDepthHistory_test.res | 6 ++- .../test/helpers/IndexerRunner.res | 18 ++++---- .../envio-tests/test/helpers/Scenario.res | 35 ++++++++------- .../test/lib_tests/ChainMetaReadyAt_test.res | 25 ++++++----- .../test/lib_tests/PgStorage_test.res | 45 ++++++++++++++----- .../test/lib_tests/Supervisor_test.res | 1 - packages/envio/src/BatchProcessing.res | 3 +- packages/envio/src/ChainFetching.res | 1 - packages/envio/src/ChainState.res | 6 +-- packages/envio/src/CrossChainState.res | 24 +++++----- packages/envio/src/Supervisor.res | 39 ++++++++-------- packages/envio/src/bindings/NodeJs.res | 44 ++++++++++-------- 29 files changed, 306 insertions(+), 207 deletions(-) create mode 100755 .claude/hooks/check-format.sh diff --git a/.claude/hooks/check-format.sh b/.claude/hooks/check-format.sh new file mode 100755 index 000000000..0a180dc35 --- /dev/null +++ b/.claude/hooks/check-format.sh @@ -0,0 +1,44 @@ +#!/usr/bin/env bash +# Blocks `git push` while a file this branch touches is unformatted. Edits made +# through Bash bypass the PostToolUse formatter, so without this the first +# anyone hears of the formatting is a CI failure. +set -uo pipefail + +cmd=$(jq -r '.tool_input.command // ""') +case "$cmd" in +*"git push"*) ;; +*) exit 0 ;; +esac + +cd "${CLAUDE_PROJECT_DIR:-.}" || exit 0 + +base=$(git merge-base HEAD origin/main 2>/dev/null || true) +changed=$( + { + git diff --name-only HEAD + [ -n "$base" ] && git diff --name-only "$base"...HEAD + git ls-files -o --exclude-standard + } | sort -u | while read -r f; do [ -f "$f" ] && echo "$f"; done +) + +problems="" + +res=$(printf '%s\n' "$changed" | grep -E '\.resi?$' || true) +if [ -n "$res" ]; then + out=$(printf '%s\n' "$res" | xargs pnpx rescript@12.2.0 format --check 2>&1 | grep '^\[format check\]' || true) + if [ -n "$out" ]; then + problems="$problems"$'\n'"Unformatted ReScript (fix: pnpx rescript@12.2.0 format ):"$'\n'"$out" + fi +fi + +if printf '%s\n' "$changed" | grep -q '^packages/cli/'; then + if ! out=$(cd packages/cli && cargo fmt --check 2>&1); then + problems="$problems"$'\n'"Unformatted Rust (fix: cd packages/cli && cargo fmt):"$'\n'"$out" + fi +fi + +if [ -n "$problems" ]; then + jq -n --arg r "Formatting check failed, so the push was not run.$problems" \ + '{hookSpecificOutput:{hookEventName:"PreToolUse",permissionDecision:"deny",permissionDecisionReason:$r}}' +fi +exit 0 diff --git a/.claude/settings.json b/.claude/settings.json index 9c3611018..e95aefc8f 100644 --- a/.claude/settings.json +++ b/.claude/settings.json @@ -34,6 +34,19 @@ } ] } + ], + "PreToolUse": [ + { + "matcher": "Bash", + "hooks": [ + { + "type": "command", + "command": "$CLAUDE_PROJECT_DIR/.claude/hooks/check-format.sh", + "timeout": 120, + "statusMessage": "Checking formatting before push..." + } + ] + } ] } } diff --git a/packages/cli/CommandLineHelp.md b/packages/cli/CommandLineHelp.md index f52a1ceef..9215959c2 100644 --- a/packages/cli/CommandLineHelp.md +++ b/packages/cli/CommandLineHelp.md @@ -371,9 +371,7 @@ Setup database by dropping schema and then running migrations ## `envio start` -Start the indexer. Runs codegen automatically before launching so the on-disk types stay in sync with `config.yaml` and `schema.graphql`. - -A schema whose entities are all per-chain is indexed across several processes, as many as `ENVIO_PG_MAX_CONNECTIONS` affords, up to four. List the busiest chains first in `config.yaml` to balance them. +Start the indexer. Runs codegen automatically before launching so the on-disk types stay in sync with `config.yaml` and `schema.graphql` **Usage:** `envio start [OPTIONS]` @@ -382,7 +380,7 @@ A schema whose entities are all per-chain is indexed across several processes, a * `-r`, `--restart` — Clear your database and restart indexing from scratch * `--chain ` — Index only this chain, leaving the others to their own `envio start --chain` processes. Repeat the flag for several chains. - Only needed to place the chains yourself, across machines or under your own process manager. A plain `envio start` already splits a per-chain schema across processes and manages them for you. Requires a schema whose entities are all per-chain, created for every chain by `envio local db-migrate up` before any process starts. Assign each configured chain to exactly one process, and give each its own `ENVIO_INDEXER_PORT`. Each process reports its own chains ready as they catch up, independently of the others. + Only needed to place chains yourself. Requires a per-chain schema, migrated for every chain before any process starts, and a separate `ENVIO_INDEXER_PORT` per process. diff --git a/packages/cli/src/cli_args/clap_definitions.rs b/packages/cli/src/cli_args/clap_definitions.rs index 424289683..1d64172da 100644 --- a/packages/cli/src/cli_args/clap_definitions.rs +++ b/packages/cli/src/cli_args/clap_definitions.rs @@ -57,8 +57,6 @@ pub enum CommandType { Local(LocalCommandTypes), ///Start the indexer. Runs codegen automatically before launching so the on-disk types stay in sync with `config.yaml` and `schema.graphql`. - /// - ///A schema whose entities are all per-chain is indexed across several processes, as many as `ENVIO_PG_MAX_CONNECTIONS` affords, up to four. List the busiest chains first in `config.yaml` to balance them. Start(StartArgs), ///Fetch raw Prometheus metrics from the running indexer's /metrics endpoint @@ -149,13 +147,8 @@ pub struct StartArgs { ///Index only this chain, leaving the others to their own `envio start --chain` processes. ///Repeat the flag for several chains. /// - ///Only needed to place the chains yourself, across machines or under your own process - ///manager. A plain `envio start` already splits a per-chain schema across processes and - ///manages them for you. - ///Requires a schema whose entities are all per-chain, created for every chain by - ///`envio local db-migrate up` before any process starts. Assign each configured chain to - ///exactly one process, and give each its own `ENVIO_INDEXER_PORT`. Each process reports its - ///own chains ready as they catch up, independently of the others. + ///Only needed to place chains yourself. Requires a per-chain schema, migrated for every + ///chain before any process starts, and a separate `ENVIO_INDEXER_PORT` per process. #[arg(long = "chain", value_name = "CHAIN_ID")] pub chains: Vec, } diff --git a/packages/envio-tests/test/BelowHeadPollingPin_test.res b/packages/envio-tests/test/BelowHeadPollingPin_test.res index d41ce3914..b1e1d8cfd 100644 --- a/packages/envio-tests/test/BelowHeadPollingPin_test.res +++ b/packages/envio-tests/test/BelowHeadPollingPin_test.res @@ -1,6 +1,7 @@ open Vitest -let scenario = Scenario.make(~supervised=false, +let scenario = Scenario.make( + ~supervised=false, ~configYaml=` name: below-head-polling contracts: @@ -84,9 +85,7 @@ describe("PIN: chains keep indexing after entering the reorg threshold", () => { await MockSource.waitItemsQuery(chainWithThresholdWork) t.expect( - chainWithThresholdWork.getItemsOrThrowCalls->Array.map( - call => call.payload["fromBlock"], - ), + chainWithThresholdWork.getItemsOrThrowCalls->Array.map(call => call.payload["fromBlock"]), ~message="the zero-lag chain first fetches to its pre-threshold head", ).toEqual([1]) chainWithThresholdWork.resolveGetItemsOrThrow( @@ -101,9 +100,7 @@ describe("PIN: chains keep indexing after entering the reorg threshold", () => { // progress and lets it lead, which sidesteps the production ordering. await MockSource.waitItemsQuery(chainWithThresholdWork) t.expect( - chainWithThresholdWork.getItemsOrThrowCalls->Array.map( - call => call.payload["fromBlock"], - ), + chainWithThresholdWork.getItemsOrThrowCalls->Array.map(call => call.payload["fromBlock"]), ~message="the second response reaches the zero-lag chain's pre-threshold head", ).toEqual([401]) chainWithThresholdWork.resolveGetItemsOrThrow( @@ -154,9 +151,7 @@ describe("PIN: chains keep indexing after entering the reorg threshold", () => { // lets chain 100 claim the progress-alignment line before discovering that // it is WaitingForNewBlock, which clamps chain 1337 behind block 800. t.expect( - chainWithThresholdWork.getItemsOrThrowCalls->Array.map( - call => call.payload["fromBlock"], - ), + chainWithThresholdWork.getItemsOrThrowCalls->Array.map(call => call.payload["fromBlock"]), ~message="the below-head chain is not blocked by an unchanged source", ).toEqual([801]) diff --git a/packages/envio-tests/test/E2E_test.res b/packages/envio-tests/test/E2E_test.res index 02b12ddfe..8742aeb12 100644 --- a/packages/envio-tests/test/E2E_test.res +++ b/packages/envio-tests/test/E2E_test.res @@ -28,7 +28,8 @@ let chainYaml = (chainId, ~startBlock=1) => ` let makeScenario = (~name, ~rollback=true, ~chains) => - Scenario.make(~supervised=false, + Scenario.make( + ~supervised=false, ~configYaml=` name: ${name} rollback_on_reorg: ${rollback ? "true" : "false"}${contractsYaml}chains:${chains}`, @@ -44,7 +45,8 @@ let scenario = makeScenario(~name="e2e", ~chains=chainYaml(1337)) // Partition ids and the chain's range-cost budget follow the contract set, so // this scenario keeps the address-less contracts alongside the addressed ones. -let partitionScenario = Scenario.make(~supervised=false, +let partitionScenario = Scenario.make( + ~supervised=false, ~configYaml=` name: e2e-partitions rollback_on_reorg: true${contractsYaml} - name: SimpleNft diff --git a/packages/envio-tests/test/EnterReorgThreshold_test.res b/packages/envio-tests/test/EnterReorgThreshold_test.res index e40f8e12c..1dad106fb 100644 --- a/packages/envio-tests/test/EnterReorgThreshold_test.res +++ b/packages/envio-tests/test/EnterReorgThreshold_test.res @@ -20,7 +20,8 @@ type Gravatar { // Two chains, each lagging maxReorgDepth (200) below head before the // threshold. Head starts at 1000, so the pre-threshold head is 800. -let multichain = Scenario.make(~supervised=false, +let multichain = Scenario.make( + ~supervised=false, ~configYaml=` name: enter-reorg-threshold-multichain contracts: @@ -52,7 +53,8 @@ chains: ~schema, ) -let singleChain = Scenario.make(~supervised=false, +let singleChain = Scenario.make( + ~supervised=false, ~configYaml=` name: enter-reorg-threshold-single chains: @@ -119,11 +121,7 @@ describe("PIN: multichain indexer enters the reorg threshold", () => { logIndex: 0, }, ) - chainA.resolveGetItemsOrThrow( - densitySeed, - ~latestFetchedBlockNumber=800, - ~knownHeight=1000, - ) + chainA.resolveGetItemsOrThrow(densitySeed, ~latestFetchedBlockNumber=800, ~knownHeight=1000) await indexer.getBatchWritePromise() // Chain A is now at its lagged head with an empty buffer — momentarily diff --git a/packages/envio-tests/test/HandlerChainInfo_test.res b/packages/envio-tests/test/HandlerChainInfo_test.res index be732e0a7..58763b0c0 100644 --- a/packages/envio-tests/test/HandlerChainInfo_test.res +++ b/packages/envio-tests/test/HandlerChainInfo_test.res @@ -4,7 +4,8 @@ open Vitest // item came from, not whichever chain the batch happens to start on, and // `isRealtime` only flips once every chain in the indexer is at its head. -let scenario = Scenario.make(~supervised=false, +let scenario = Scenario.make( + ~supervised=false, ~configYaml=` name: handler-chain-info contracts: @@ -56,8 +57,7 @@ let recordChain = (~block, ~label): MockSource.itemMock => { }, } -let sortById = (rows: array) => - rows->Array.toSorted((a, b) => String.compare(a.id, b.id)) +let sortById = (rows: array) => rows->Array.toSorted((a, b) => String.compare(a.id, b.id)) describe("context.chain inside a handler", () => { scenario->Scenario.it( diff --git a/packages/envio-tests/test/IsolatedRollback_test.res b/packages/envio-tests/test/IsolatedRollback_test.res index 4541425e1..5b17bbff5 100644 --- a/packages/envio-tests/test/IsolatedRollback_test.res +++ b/packages/envio-tests/test/IsolatedRollback_test.res @@ -52,7 +52,8 @@ type Counter { } ` -let scenario = Scenario.make(~supervised=false, +let scenario = Scenario.make( + ~supervised=false, ~schema=perChainSchema, ~configYaml=makeConfigYaml(~name="isolated-rollback"), ) @@ -60,7 +61,8 @@ let scenario = Scenario.make(~supervised=false, // A single chain with no cross-chain entity is a per-chain sequence too: nothing // about the mode needs a sibling, and the chain-id column every per-chain entity // carries is what its bounds join against. -let singleChainScenario = Scenario.make(~supervised=false, +let singleChainScenario = Scenario.make( + ~supervised=false, ~schema=perChainSchema, ~configYaml=` name: single-chain-per-chain @@ -77,7 +79,8 @@ chains:${chainYaml(100)} // One cross-chain entity is enough to couple the chains: a value chain 1337 // wrote can be what chain 100 read and overwrote, so its reorg has to take // every chain back with it. -let crossChainScenario = Scenario.make(~supervised=false, +let crossChainScenario = Scenario.make( + ~supervised=false, ~schema=perChainSchema ++ ` type Total @crossChain { id: ID! @@ -90,13 +93,15 @@ type Total @crossChain { // The sink is append-only: an isolated rollback reaches it as the diff rows the // next batch carries, and its current-state view has to resolve to the same // thing Postgres holds. -let clickHouseScenario = Scenario.make(~supervised=false, +let clickHouseScenario = Scenario.make( + ~supervised=false, ~schema=perChainSchema, ~configYaml=makeConfigYaml(~name="isolated-rollback-clickhouse"), ~unsupported=[{backend: #postgres, reason: "asserts against a ClickHouse server"}], ) -let fullHistoryScenario = Scenario.make(~supervised=false, +let fullHistoryScenario = Scenario.make( + ~supervised=false, ~schema=perChainSchema, ~configYaml=makeConfigYaml( ~name="isolated-rollback-full-history", @@ -107,7 +112,8 @@ let fullHistoryScenario = Scenario.make(~supervised=false, // A per-chain entity's chain column is named `chain_id` under snake_case, the // same name the per-chain bounds relation gives its own, so the rollback // queries have to keep the two apart. -let snakeCaseScenario = Scenario.make(~supervised=false, +let snakeCaseScenario = Scenario.make( + ~supervised=false, ~schema=perChainSchema, ~configYaml=makeConfigYaml( ~name="isolated-rollback-snake-case", diff --git a/packages/envio-tests/test/PerChainEntity_test.res b/packages/envio-tests/test/PerChainEntity_test.res index 6a29518dc..55ed15524 100644 --- a/packages/envio-tests/test/PerChainEntity_test.res +++ b/packages/envio-tests/test/PerChainEntity_test.res @@ -70,7 +70,8 @@ let scenario = Scenario.make(~supervised=false, ~schema, ~configYaml=makeConfigY // The two chains need a reorg threshold to roll back within, so this variant // sets one — `max_reorg_depth` is per chain, so it goes in the chain blocks. -let rollbackScenario = Scenario.make(~supervised=false, +let rollbackScenario = Scenario.make( + ~supervised=false, ~schema, ~configYaml=makeConfigYaml(~rollback="\nrollback_on_reorg: true")->String.replaceAll( " start_block: 1\n", @@ -80,7 +81,8 @@ let rollbackScenario = Scenario.make(~supervised=false, // The entity object and the getWhere filter key the chain by `chainId` while // the column is `chain_id`. -let snakeCaseScenario = Scenario.make(~supervised=false, +let snakeCaseScenario = Scenario.make( + ~supervised=false, ~schema, ~configYaml=makeConfigYaml( ~storage=`storage: diff --git a/packages/envio-tests/test/PerChainHistoryPrune_test.res b/packages/envio-tests/test/PerChainHistoryPrune_test.res index a5b34bb86..ae987d770 100644 --- a/packages/envio-tests/test/PerChainHistoryPrune_test.res +++ b/packages/envio-tests/test/PerChainHistoryPrune_test.res @@ -56,7 +56,8 @@ chains:${chainYaml(100, ~startBlock=110, ~maxReorgDepth=15)}${chainYaml( )} ` -let scenario = Scenario.make(~supervised=false, +let scenario = Scenario.make( + ~supervised=false, ~schema, ~configYaml=makeConfigYaml("per-chain-prune", ~laggingChainId=1337), ) @@ -71,13 +72,15 @@ type Total @crossChain { // One cross-chain entity couples the chains: a reorg on the chain furthest // behind can reach a row any chain wrote, so none may prune past its safe point. -let crossChainScenario = Scenario.make(~supervised=false, +let crossChainScenario = Scenario.make( + ~supervised=false, ~schema=crossChainSchema, ~configYaml=makeConfigYaml("per-chain-prune-cross-chain", ~laggingChainId=1337), ) // The same, with the lagging chain visited first. -let crossChainLaggingFirstScenario = Scenario.make(~supervised=false, +let crossChainLaggingFirstScenario = Scenario.make( + ~supervised=false, ~schema=crossChainSchema, ~configYaml=makeConfigYaml("per-chain-prune-cross-chain-lagging-first", ~laggingChainId=5), ) @@ -86,7 +89,8 @@ let crossChainLaggingFirstScenario = Scenario.make(~supervised=false, // is safe. Under the shared sequence a cross-chain entity brings, its own last // id is not the bound — an idle chain's would hold every other chain's prune // back for as long as it stays idle. -let crossChainZeroDepthScenario = Scenario.make(~supervised=false, +let crossChainZeroDepthScenario = Scenario.make( + ~supervised=false, ~schema=crossChainSchema, ~configYaml=` name: per-chain-prune-cross-chain-zero-depth @@ -107,7 +111,8 @@ chains:${chainYaml(100, ~startBlock=110, ~maxReorgDepth=15)}${chainYaml( // Every chain reaches a safe checkpoint of its own, so the prune carries a bound // per chain. Three of them rather than two: a pair of bounds can be crossed and // still look right, while three cannot. -let manyBoundsScenario = Scenario.make(~supervised=false, +let manyBoundsScenario = Scenario.make( + ~supervised=false, ~schema, ~configYaml=` name: per-chain-prune-many-bounds @@ -128,7 +133,8 @@ chains:${chainYaml(100, ~startBlock=110, ~maxReorgDepth=15)}${chainYaml( // A chain with no reorg depth can't be rolled back, so with no cross-chain // entity to let a sibling's rollback reach its rows, it has no history to keep // and everything it has committed is safe to prune. -let zeroDepthScenario = Scenario.make(~supervised=false, +let zeroDepthScenario = Scenario.make( + ~supervised=false, ~schema, ~configYaml=` name: per-chain-prune-zero-depth diff --git a/packages/envio-tests/test/ResumeFinalize_test.res b/packages/envio-tests/test/ResumeFinalize_test.res index 73e935258..53425517b 100644 --- a/packages/envio-tests/test/ResumeFinalize_test.res +++ b/packages/envio-tests/test/ResumeFinalize_test.res @@ -34,13 +34,15 @@ let chainYaml = (chainId, address, extra) => let gravatar1337 = "0x2B2f78c5BF6D9C12Ee1225D5F374aa91204580c3" let gravatar1 = "0x3B2f78c5BF6D9C12Ee1225D5F374aa91204580c3" -let scenario = Scenario.make(~supervised=false, +let scenario = Scenario.make( + ~supervised=false, ~configYaml=` name: resume-finalize${contractsYaml}chains:${chainYaml(1337, gravatar1337, "")}`, ~schema, ) -let endBlockScenario = Scenario.make(~supervised=false, +let endBlockScenario = Scenario.make( + ~supervised=false, ~configYaml=` name: resume-finalize-end-block${contractsYaml}chains:${chainYaml( 1337, @@ -50,7 +52,8 @@ name: resume-finalize-end-block${contractsYaml}chains:${chainYaml( ~schema, ) -let multichainScenario = Scenario.make(~supervised=false, +let multichainScenario = Scenario.make( + ~supervised=false, ~configYaml=` name: resume-finalize-multichain${contractsYaml}chains:${chainYaml(1, gravatar1, "")}${chainYaml( 1337, diff --git a/packages/envio-tests/test/RollbackDiffCheckpointIds_test.res b/packages/envio-tests/test/RollbackDiffCheckpointIds_test.res index 0f9b5fde1..5628a18ba 100644 --- a/packages/envio-tests/test/RollbackDiffCheckpointIds_test.res +++ b/packages/envio-tests/test/RollbackDiffCheckpointIds_test.res @@ -35,7 +35,8 @@ let chainYaml = chainId => // One cross-chain entity is what makes the checkpoint sequence shared, and what // makes a reorg on either chain roll both of them back. -let scenario = Scenario.make(~supervised=false, +let scenario = Scenario.make( + ~supervised=false, ~schema=` type Counter { id: ID! @@ -146,8 +147,7 @@ describe("Rollback diff checkpoint ids", () => { await indexer.getBatchWritePromise() let highestCommitted = - (await indexer.queryCheckpoints()) - ->Array.reduce(0n, (highest, checkpoint) => + (await indexer.queryCheckpoints())->Array.reduce(0n, (highest, checkpoint) => checkpoint.id > highest ? checkpoint.id : highest ) @@ -169,7 +169,10 @@ describe("Rollback diff checkpoint ids", () => { ~message="the rollback's depth search to re-fetch the scanned block hashes", ) source1337.resolveGetBlockHashes( - [(100, "0x100"), (101, "0x101")]->Array.map(((blockNumber, blockHash)): BlockStore.inputBlock => { + [(100, "0x100"), (101, "0x101")]->Array.map((( + blockNumber, + blockHash, + )): BlockStore.inputBlock => { blockNumber, blockHash, blockTimestamp: blockNumber, @@ -186,9 +189,10 @@ describe("Rollback diff checkpoint ids", () => { ) await indexer.getBatchWritePromise() - let sorted = stagedDiffs->Array.toSorted((a, b) => - a.checkpointId < b.checkpointId ? -1. : a.checkpointId > b.checkpointId ? 1. : 0. - ) + let sorted = + stagedDiffs->Array.toSorted((a, b) => + a.checkpointId < b.checkpointId ? -1. : a.checkpointId > b.checkpointId ? 1. : 0. + ) t.expect( ( stagedDiffs->Array.length, diff --git a/packages/envio-tests/test/Rollback_test.res b/packages/envio-tests/test/Rollback_test.res index c0096dc25..d843191ad 100644 --- a/packages/envio-tests/test/Rollback_test.res +++ b/packages/envio-tests/test/Rollback_test.res @@ -50,7 +50,8 @@ indexer.onEvent({ contract: "SimpleNft", event: "Transfer" }, async () => {}); ` let makeScenario = (~name, ~chains, ~extra="") => - Scenario.make(~supervised=false, + Scenario.make( + ~supervised=false, ~configYaml=` name: ${name} rollback_on_reorg: true${extra}${contractsYaml}chains:${chains}`, @@ -776,9 +777,11 @@ describe("E2E rollback tests", () => { // registration at suite scope would also collect the rollbacks every // other case in this file fires. let rollbackCommitCalls = [] - let unregister = RollbackCommit.register(async (args: RollbackCommit.args) => { - rollbackCommitCalls->Array.push(args) - }) + let unregister = RollbackCommit.register( + async (args: RollbackCommit.args) => { + rollbackCommitCalls->Array.push(args) + }, + ) let sourceMock = source(1337) await Utils.delay(0) @@ -2369,7 +2372,6 @@ describe("E2E rollback tests", () => { // so getHighestBlockBelowThreshold = 300 - 200 = 100 is used directly. // Wait for the SetRollbackState tasks (NextQuery, ProcessEventBatch) to be scheduled - sourceMock1337.resolveGetItemsOrThrow( [], ~prevRangeLastBlock={ @@ -3026,7 +3028,7 @@ describe("E2E rollback tests", () => { await storage.writeBatch( ~batch, ~rollback, - ~config, + ~config, ~allEntities, ~updatedEffectsCache, ~updatedEntities, diff --git a/packages/envio-tests/test/SchemaIndexes_test.res b/packages/envio-tests/test/SchemaIndexes_test.res index b480f059f..1d3137df3 100644 --- a/packages/envio-tests/test/SchemaIndexes_test.res +++ b/packages/envio-tests/test/SchemaIndexes_test.res @@ -20,7 +20,8 @@ type C { } ` -let chainYaml = (chainId, address) => ` +let chainYaml = (chainId, address) => + ` - id: ${chainId->Int.toString} rpc: url: https://rpc${chainId->Int.toString}.example.test @@ -38,7 +39,8 @@ contracts: - event: "TestEvent()" ` -let scenario = Scenario.make(~supervised=false, +let scenario = Scenario.make( + ~supervised=false, ~configYaml=` name: schema-indexes${contractsYaml}chains:${chainYaml( 1337, @@ -50,7 +52,8 @@ name: schema-indexes${contractsYaml}chains:${chainYaml( // An `end_block` the chain never reaches: the indexer still counts itself caught // up once progress sits at the head, so the deferred indexes are owed then, not // at the unreachable end block. -let unreachableEndBlockScenario = Scenario.make(~supervised=false, +let unreachableEndBlockScenario = Scenario.make( + ~supervised=false, ~configYaml=` name: schema-indexes-unreachable-end${contractsYaml}chains: - id: 1337 @@ -68,7 +71,8 @@ name: schema-indexes-unreachable-end${contractsYaml}chains: // A `start_block` past the head: the chain is at its head from the first moment // and never has a batch to process, so nothing ever writes its progress row. -let aheadOfHeadScenario = Scenario.make(~supervised=false, +let aheadOfHeadScenario = Scenario.make( + ~supervised=false, ~configYaml=` name: schema-indexes-ahead-of-head${contractsYaml}chains: - id: 1337 @@ -83,7 +87,8 @@ name: schema-indexes-ahead-of-head${contractsYaml}chains: ~schema, ) -let multichainScenario = Scenario.make(~supervised=false, +let multichainScenario = Scenario.make( + ~supervised=false, ~configYaml=` name: schema-indexes-multichain${contractsYaml}chains:${chainYaml( 100, @@ -310,8 +315,8 @@ describe("Deferred schema indexes", () => { await indexer.waitUntilReady() t.expect(( - (await findIndexes(~sql, ~tableName="A", ~columns=["b_id"], ~pgSchema))->Array.map( - entry => entry.name, + (await findIndexes(~sql, ~tableName="A", ~columns=["b_id"], ~pgSchema))->Array.map(entry => + entry.name ), await readyAtByChainId(~sql, ~pgSchema), )).toEqual(([aBIdIndexName], [(ChainId.fromInt(1337), true)])) @@ -474,7 +479,11 @@ describe("Deferred schema indexes", () => { aIndexes->Array.map(entry => (entry->isValid, entry->isPartial, entry->predicate)), ), ~message="The conflicting index is left alone and A(b_id) still gets a usable index of its own", - ).toEqual(([{value: "1", labels: dict{"chainId": "1337"}}], ["A_b_id"], [(true, false, None)])) + ).toEqual(( + [{value: "1", labels: dict{"chainId": "1337"}}], + ["A_b_id"], + [(true, false, None)], + )) t.expect( aIndexes->Array.map(entry => entry.name), @@ -555,9 +564,9 @@ describe("Automatic getWhere indexes", () => { t.expect( ( matched.contents, - (await findIndexes(~sql, ~tableName="A", ~columns=[optionalColumn], ~pgSchema))->Array.map( - entry => (entry.name, entry->isValid), - ), + ( + await findIndexes(~sql, ~tableName="A", ~columns=[optionalColumn], ~pgSchema) + )->Array.map(entry => (entry.name, entry->isValid)), await findIndexes(~sql, ~tableName="A", ~columns=["b_id"], ~pgSchema), await readyAtByChainId(~sql, ~pgSchema), ), @@ -583,12 +592,9 @@ describe("Automatic getWhere indexes", () => { t.expect( ( - (await findIndexes( - ~sql, - ~tableName="A", - ~columns=[optionalColumn], - ~pgSchema, - ))->Array.length, + ( + await findIndexes(~sql, ~tableName="A", ~columns=[optionalColumn], ~pgSchema) + )->Array.length, (await findIndexes(~sql, ~tableName="A", ~columns=["b_id"], ~pgSchema))->Array.map( entry => entry.name, ), diff --git a/packages/envio-tests/test/SupervisedRealtime_test.res b/packages/envio-tests/test/SupervisedRealtime_test.res index 22587048b..204c92e2b 100644 --- a/packages/envio-tests/test/SupervisedRealtime_test.res +++ b/packages/envio-tests/test/SupervisedRealtime_test.res @@ -119,7 +119,11 @@ describe("A supervised worker", () => { let readyAt = await readyAtByChainId(~sql, ~pgSchema) let stamps = readyAt->Array.filterMap(((_, at)) => at->Option.map(Date.getTime)) t.expect( - (readyAt->Array.map(((chainId, _)) => chainId), stamps->Array.length, stamps->Set.fromArray->Set.size), + ( + readyAt->Array.map(((chainId, _)) => chainId), + stamps->Array.length, + stamps->Set.fromArray->Set.size, + ), ~message="The release stamps every chain, and one caught-up indexer is one instant", ).toEqual((["1", "137"], 2, 1)) }, @@ -200,10 +204,7 @@ describe("A supervised worker on a chain with a reorg depth", () => { await indexer.waitUntilIdle() t.expect( - ( - await indexer.metric("envio_reorg_threshold"), - await readyAtByChainId(~sql, ~pgSchema), - ), + (await indexer.metric("envio_reorg_threshold"), await readyAtByChainId(~sql, ~pgSchema)), ~message="Both chains are as far as they can fetch, and the run has not said so", ).toEqual(([{value: "0", labels: Dict.make()}], [("1", None), ("137", None)])) diff --git a/packages/envio-tests/test/SupervisorFork_test.res b/packages/envio-tests/test/SupervisorFork_test.res index 05b80f405..286f139aa 100644 --- a/packages/envio-tests/test/SupervisorFork_test.res +++ b/packages/envio-tests/test/SupervisorFork_test.res @@ -51,7 +51,7 @@ describe("Supervisor.fork", () => { let report = await Promise.make( (resolve, _) => - running.child->NodeJs.ChildProcess.onMessage( + running.child->NodeJs.ChildProcess.Child.onMessage( message => switch message { | Worker.Snapshot({metrics}) => @@ -59,7 +59,7 @@ describe("Supervisor.fork", () => { }, ), ) - running.child->NodeJs.ChildProcess.kill("SIGTERM")->ignore + running.child->NodeJs.ChildProcess.Child.kill("SIGTERM")->ignore t.expect(report).toStrictEqual({ // Everything the supervisor decided, in the environment: a worker needs it @@ -128,7 +128,7 @@ describe("Supervisor.awaitExit", () => { releaseCheck: None, } let ended = outcome(group) - signalled.child->NodeJs.ChildProcess.kill("SIGTERM")->ignore + signalled.child->NodeJs.ChildProcess.Child.kill("SIGTERM")->ignore // The sibling still went down with it: one worker short leaves its chains // unindexed. @@ -164,12 +164,7 @@ describe("Supervisor.readLines", () => { // Nothing is left to flush twice. flush() - t.expect(lines).toStrictEqual([ - "a line", - "and half of another", - "last", - "no newline here", - ]) + t.expect(lines).toStrictEqual(["a line", "and half of another", "last", "no newline here"]) }) }) @@ -196,9 +191,7 @@ describe("Supervisor.fork output", () => { } let _ = await group->Supervisor.awaitExit - t.expect( - lines->Array.toSorted(((_, a), (_, b)) => String.compare(a, b)), - ).toStrictEqual([ + t.expect(lines->Array.toSorted(((_, a), (_, b)) => String.compare(a, b))).toStrictEqual([ ("stdout", "first line"), ("stderr", "from stderr"), ("stdout", "second line"), @@ -218,7 +211,7 @@ describe("Supervisor.isRunAtHead", () => { let untilGone = async (r: Supervisor.running) => { let rec until = async deadline => - if r.child->NodeJs.ChildProcess.connected && Date.now() < deadline { + if r.child->NodeJs.ChildProcess.Child.connected && Date.now() < deadline { await Utils.delay(10) await until(deadline) } diff --git a/packages/envio-tests/test/ZeroReorgDepthHistory_test.res b/packages/envio-tests/test/ZeroReorgDepthHistory_test.res index f3c7b6bf2..3a9a927af 100644 --- a/packages/envio-tests/test/ZeroReorgDepthHistory_test.res +++ b/packages/envio-tests/test/ZeroReorgDepthHistory_test.res @@ -37,7 +37,8 @@ let chainYaml = (chainId, ~maxReorgDepth) => // One chain, and the default cross-chain entities that make its checkpoint // sequence a shared one. -let singleChainScenario = Scenario.make(~supervised=false, +let singleChainScenario = Scenario.make( + ~supervised=false, ~schema, ~configYaml=` name: zero-reorg-depth-history @@ -52,7 +53,8 @@ chains:${chainYaml(100, ~maxReorgDepth=0)} // No cross-chain entity, so each chain counts its own checkpoints and only the // chain that can be rolled back keeps any. -let perChainScenario = Scenario.make(~supervised=false, +let perChainScenario = Scenario.make( + ~supervised=false, ~schema, ~configYaml=` name: zero-reorg-depth-history-per-chain diff --git a/packages/envio-tests/test/helpers/IndexerRunner.res b/packages/envio-tests/test/helpers/IndexerRunner.res index 9af7abef9..5b6e5ab76 100644 --- a/packages/envio-tests/test/helpers/IndexerRunner.res +++ b/packages/envio-tests/test/helpers/IndexerRunner.res @@ -190,19 +190,15 @@ let run = async ( // testing the barrier and owns its own release. let releaseCheck = ref(None) if superviseRun && !holdRealtime { - releaseCheck := - Some( - setInterval(() => + releaseCheck := Some(setInterval(() => if state->IndexerState.hasArrivedAtHead { releaseCheck.contents->Option.forEach(clearInterval) releaseCheck := None state->IndexerState.releaseRealtime } - , releaseCheckIntervalMillis), - ) + , releaseCheckIntervalMillis)) } - // Persist before stopping, else a resumed indexer loses uncommitted state, // then let any in-flight batch or write settle so nothing from this run // lands on the database afterwards. @@ -312,7 +308,10 @@ let run = async ( let isIdle = !(state->IndexerState.isProcessing) && state->IndexerState.writeFiber->Option.isNone && - Frontier.equals(state->IndexerState.committedFrontier, state->IndexerState.processedFrontier) + Frontier.equals( + state->IndexerState.committedFrontier, + state->IndexerState.processedFrontier, + ) // Catching up hands off to the FinalizingIndexes phase, which is // where readiness is decided — so a batch isn't settled until that @@ -358,7 +357,10 @@ let run = async ( !(state->IndexerState.isProcessing) && state->IndexerState.writeFiber->Option.isNone && !(state->IndexerState.shouldFinalizeIndexes) && - Frontier.equals(state->IndexerState.committedFrontier, state->IndexerState.processedFrontier) + Frontier.equals( + state->IndexerState.committedFrontier, + state->IndexerState.processedFrontier, + ) ) { settled.contents + 1 } else { diff --git a/packages/envio-tests/test/helpers/Scenario.res b/packages/envio-tests/test/helpers/Scenario.res index d420704b0..d7c220fdf 100644 --- a/packages/envio-tests/test/helpers/Scenario.res +++ b/packages/envio-tests/test/helpers/Scenario.res @@ -308,29 +308,30 @@ let it = ( async _ => (), ) | None => - let runBody = (~superviseRun) => async (t: Vitest.testContext) => - await scenario->run( - ~sources, - ~reducedPollingInterval?, - ~targetBufferSize?, - ~maxAddrInPartition?, - ~clientFilterAddressThreshold?, - ~reorgThresholdReadyTolerance?, - ~holdRealtime?, - ~superviseRun, - ~onError?, - ~onExit?, - ~mapStorage?, - (~indexer, ~source) => body(~t, ~indexer, ~source), - ) + let runBody = (~superviseRun) => + async (t: Vitest.testContext) => + await scenario->run( + ~sources, + ~reducedPollingInterval?, + ~targetBufferSize?, + ~maxAddrInPartition?, + ~clientFilterAddressThreshold?, + ~reorgThresholdReadyTolerance?, + ~holdRealtime?, + ~superviseRun, + ~onError?, + ~onExit?, + ~mapStorage?, + (~indexer, ~source) => body(~t, ~indexer, ~source), + ) let register = (name, ~superviseRun) => switch retry { - | Some(retry) => - Vitest.Async.itWithOptions(name, {retry, ?timeout}, runBody(~superviseRun)) + | Some(retry) => Vitest.Async.itWithOptions(name, {retry, ?timeout}, runBody(~superviseRun)) | None => Vitest.Async.it(name, runBody(~superviseRun), ~timeout?) } register(name, ~superviseRun=false) + // One chain is a run whose every chain is its own process's already, so the // barrier has nothing to hold: only a multichain scenario says anything new. if ( diff --git a/packages/envio-tests/test/lib_tests/ChainMetaReadyAt_test.res b/packages/envio-tests/test/lib_tests/ChainMetaReadyAt_test.res index 464316993..75139fe3e 100644 --- a/packages/envio-tests/test/lib_tests/ChainMetaReadyAt_test.res +++ b/packages/envio-tests/test/lib_tests/ChainMetaReadyAt_test.res @@ -26,7 +26,9 @@ let storage = PgStorage.make( ) let readyAt = async () => { - let rows: array<{"ready_at": Null.t}> = await sql->Postgres.unsafe( + let rows: array<{ + "ready_at": Null.t, + }> = await sql->Postgres.unsafe( `SELECT "ready_at" FROM "${pgSchema}"."envio_chains" ORDER BY "id";`, ) rows->Array.map(row => row["ready_at"]->Null.toOption->Option.isSome) @@ -54,16 +56,17 @@ describe("A chain metadata write", () => { let stale = Dict.make() config.chainMap ->ChainMap.keys - ->Array.forEach(chainId => - stale->Dict.set( - chainId->ChainId.toString, - { - InternalTable.Chains.firstEventBlockNumber: Null.null, - latestFetchedBlockNumber: 10, - timestampCaughtUpToHeadOrEndblock: Null.null, - isHyperSync: false, - }, - ) + ->Array.forEach( + chainId => + stale->Dict.set( + chainId->ChainId.toString, + { + InternalTable.Chains.firstEventBlockNumber: Null.null, + latestFetchedBlockNumber: 10, + timestampCaughtUpToHeadOrEndblock: Null.null, + isHyperSync: false, + }, + ), ) let _ = await storage.setChainMeta(stale) diff --git a/packages/envio-tests/test/lib_tests/PgStorage_test.res b/packages/envio-tests/test/lib_tests/PgStorage_test.res index 9cbcf5554..b8d025dbe 100644 --- a/packages/envio-tests/test/lib_tests/PgStorage_test.res +++ b/packages/envio-tests/test/lib_tests/PgStorage_test.res @@ -626,11 +626,20 @@ FROM "public"."envio_chains";` async t => { let params = [] let condition = PgStorage.makeFilterCondition( - ~filter=dict{"tag": dict{"_eq": Uint8Array.fromArray([0xaa])->(Utils.magic: Uint8Array.t => unknown), "_in": [Uint8Array.fromArray([1, 2]), Uint8Array.fromLength(0)]->( - Utils.magic: array => unknown - )}, "chunks": dict{"_eq": [Uint8Array.fromArray([3])]->(Utils.magic: array => unknown), "_in": [[Uint8Array.fromArray([4])], [Uint8Array.fromArray([5])]]->( - Utils.magic: array> => unknown - )}}->parse(~table=bytesTable), + ~filter=dict{ + "tag": dict{ + "_eq": Uint8Array.fromArray([0xaa])->(Utils.magic: Uint8Array.t => unknown), + "_in": [Uint8Array.fromArray([1, 2]), Uint8Array.fromLength(0)]->( + Utils.magic: array => unknown + ), + }, + "chunks": dict{ + "_eq": [Uint8Array.fromArray([3])]->(Utils.magic: array => unknown), + "_in": [[Uint8Array.fromArray([4])], [Uint8Array.fromArray([5])]]->( + Utils.magic: array> => unknown + ), + }, + }->parse(~table=bytesTable), ~table=bytesTable, ~pgSchema="test_schema", ~params, @@ -654,7 +663,9 @@ FROM "public"."envio_chains";` async t => { let params = [] let condition = PgStorage.makeFilterCondition( - ~filter=dict{"id": dict{"_in": ["1", "2"]->(Utils.magic: array => unknown)}}->parse(~table), + ~filter=dict{ + "id": dict{"_in": ["1", "2"]->(Utils.magic: array => unknown)}, + }->parse(~table), ~table, ~pgSchema="test_schema", ~params, @@ -690,7 +701,10 @@ FROM "public"."envio_chains";` async t => { let params = [] let condition = PgStorage.makeFilterCondition( - ~filter=dict{"score": dict{"_gte": 5->(Utils.magic: int => unknown)}, "id": dict{"_lte": "9"->(Utils.magic: string => unknown)}}->parse(~table), + ~filter=dict{ + "score": dict{"_gte": 5->(Utils.magic: int => unknown)}, + "id": dict{"_lte": "9"->(Utils.magic: string => unknown)}, + }->parse(~table), ~table, ~pgSchema="test_schema", ~params, @@ -708,7 +722,13 @@ FROM "public"."envio_chains";` async t => { let params = [] let condition = PgStorage.makeFilterCondition( - ~filter=dict{"id": dict{"_eq": "1"->(Utils.magic: string => unknown)}, "score": dict{"_gt": 5->(Utils.magic: int => unknown), "_lt": 10->(Utils.magic: int => unknown)}}->parse(~table), + ~filter=dict{ + "id": dict{"_eq": "1"->(Utils.magic: string => unknown)}, + "score": dict{ + "_gt": 5->(Utils.magic: int => unknown), + "_lt": 10->(Utils.magic: int => unknown), + }, + }->parse(~table), ~table, ~pgSchema="test_schema", ~params, @@ -756,11 +776,16 @@ FROM "public"."envio_chains";` t.expect(( condition(tagsIn([["a"], ["a", "b"]])), condition(tagsIn([])), - condition(dict{"flag": dict{"_in": [true, false]->(Utils.magic: array => unknown)}}), + condition( + dict{"flag": dict{"_in": [true, false]->(Utils.magic: array => unknown)}}, + ), )).toEqual(( ( `("tags" = $1 OR "tags" = $2)`, - [["a"]->(Utils.magic: array => unknown), ["a", "b"]->(Utils.magic: array => unknown)], + [ + ["a"]->(Utils.magic: array => unknown), + ["a", "b"]->(Utils.magic: array => unknown), + ], ), ("FALSE", []), ( diff --git a/packages/envio-tests/test/lib_tests/Supervisor_test.res b/packages/envio-tests/test/lib_tests/Supervisor_test.res index 449def58b..5cd33ea7a 100644 --- a/packages/envio-tests/test/lib_tests/Supervisor_test.res +++ b/packages/envio-tests/test/lib_tests/Supervisor_test.res @@ -221,7 +221,6 @@ describe("Supervisor worker plumbing", () => { }) }) - describe("Worker.detect", () => { it("Counts as a worker only when forked with the variable and a channel", t => { let forked = Dict.fromArray([ diff --git a/packages/envio/src/BatchProcessing.res b/packages/envio/src/BatchProcessing.res index 7804f0487..2af155349 100644 --- a/packages/envio/src/BatchProcessing.res +++ b/packages/envio/src/BatchProcessing.res @@ -69,7 +69,8 @@ and processNextBatch = async (state: IndexerState.t, ~scheduleFetch): unit => { let isInReorgThresholdBeforeUpdate = state->IndexerState.isInReorgThreshold let isRealtimeBeforeUpdate = state->IndexerState.isRealtime - let batch = state->IndexerState.createBatch(~batchSizeTarget=(state->IndexerState.config).batchSize) + let batch = + state->IndexerState.createBatch(~batchSizeTarget=(state->IndexerState.config).batchSize) let progressedChainsById = batch.progressedChainsById diff --git a/packages/envio/src/ChainFetching.res b/packages/envio/src/ChainFetching.res index 2bedf4137..d68ec8a6e 100644 --- a/packages/envio/src/ChainFetching.res +++ b/packages/envio/src/ChainFetching.res @@ -279,7 +279,6 @@ and applyQueryResponse = ( ~blockNumber=newItems->Array.getUnsafe(0)->Internal.getItemBlockNumber, ) } - } let finishWaitingForNewBlock = ( diff --git a/packages/envio/src/ChainState.res b/packages/envio/src/ChainState.res index 5902d588d..24122808c 100644 --- a/packages/envio/src/ChainState.res +++ b/packages/envio/src/ChainState.res @@ -993,7 +993,6 @@ let enterReorgThreshold = (cs: t) => { cs.fetchState = cs.fetchState->FetchState.updateInternal(~blockLag=cs.chainConfig.blockLag) } - let isInReorgThreshold = (cs: t) => cs.isInReorgThreshold // Whether the chain's writes need history: only what a rollback could still @@ -1120,9 +1119,7 @@ let toChainBeforeBatch = (cs: t, ~isRealtime): Batch.chainBeforeBatch => { // batch would leave it. Entering the threshold is what lifts the pre-threshold // lag, so a chain waiting to enter it has fetched as far as it can. let isReadyToEnterReorgThreshold = (cs: t) => - cs.fetchState->FetchState.isReadyToEnterReorgThreshold( - ~tolerance=cs.reorgThresholdReadyTolerance, - ) + cs.fetchState->FetchState.isReadyToEnterReorgThreshold(~tolerance=cs.reorgThresholdReadyTolerance) let isReadyToEnterReorgThresholdAfterBatch = (cs: t, ~batch: Batch.t) => { let fetchState = switch batch.progressedChainsById->ChainId.Dict.dangerouslyGetNonOption( @@ -1272,6 +1269,7 @@ let markReady = (cs: t, ~readyAt) => let rollbackCommittedProgress = (cs: t, blockNumber) => if blockNumber !== cs.committedProgressBlockNumber { cs.committedProgressBlockNumber = blockNumber + // Exact block only: the rolled-back region is about to be refetched, and // the next batch re-establishes the time either way. cs.committedProgressBlockTime = diff --git a/packages/envio/src/CrossChainState.res b/packages/envio/src/CrossChainState.res index 145f256c5..c86b65736 100644 --- a/packages/envio/src/CrossChainState.res +++ b/packages/envio/src/CrossChainState.res @@ -70,10 +70,10 @@ let isHoldingRealtime = (crossChainState: t) => crossChainState.holdRealtime let hasArrivedAtHead = (crossChainState: t) => crossChainState.isCaughtUp || crossChainState.isRealtime || { - let chainStates = crossChainState.chainStates->Dict.valuesToArray - chainStates->Utils.Array.notEmpty && - chainStates->Array.every(ChainState.isReadyToEnterReorgThreshold) - } + let chainStates = crossChainState.chainStates->Dict.valuesToArray + chainStates->Utils.Array.notEmpty && + chainStates->Array.every(ChainState.isReadyToEnterReorgThreshold) + } // Resolve a chain's state by id. The id always comes from `chainIds`, which is // derived from `chainStates`, so the entry is guaranteed present. @@ -167,9 +167,9 @@ let createBatch = ( // holds the others back whatever process it runs in. let isReadyToEnterReorgThreshold = (crossChainState: t, ~batch) => !crossChainState.holdRealtime && - crossChainState.chainStates - ->Dict.valuesToArray - ->Array.every(cs => cs->ChainState.isReadyToEnterReorgThresholdAfterBatch(~batch)) + crossChainState.chainStates + ->Dict.valuesToArray + ->Array.every(cs => cs->ChainState.isReadyToEnterReorgThresholdAfterBatch(~batch)) // Said by each chain rather than once for the indexer: what crossing changes // is a chain's own, and the chains of a split run cross in processes that can @@ -206,9 +206,9 @@ let applyBatchProgress = (crossChainState: t, ~batch: Batch.t, ~blockTimestampNa crossChainState.isCaughtUp = crossChainState.isCaughtUp || - (!crossChainState.holdRealtime && - crossChainState->nextItemIsNone && - everyChainCaughtUp.contents) + (!crossChainState.holdRealtime && + crossChainState->nextItemIsNone && + everyChainCaughtUp.contents) } // Every chain has buffered up to its head (or endblock) with nothing @@ -272,6 +272,7 @@ let markReady = (crossChainState: t, ~readyAt) => { let cs = crossChainState->getChainState(crossChainState.chainIds->Array.getUnsafe(i)) let wasReady = cs->ChainState.isReady cs->ChainState.markReady(~readyAt) + // One line per chain, because `ready_at` is one column per chain: what the // log says and what a reader finds in the row are the same fact. if !wasReady { @@ -291,8 +292,7 @@ let markReady = (crossChainState: t, ~readyAt) => { // still be reorged are indexed after every chain gets this far, so it says what // it is waiting on when there is anything to wait for. let reportFinished = (crossChainState: t) => { - let waitingOnOthers = - crossChainState.holdRealtime || crossChainState.chainIds->Array.length > 1 + let waitingOnOthers = crossChainState.holdRealtime || crossChainState.chainIds->Array.length > 1 crossChainState.chainStates ->Dict.valuesToArray ->Array.forEach(cs => diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index 54c7f8dff..7bdf4f09f 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -64,7 +64,7 @@ let planForRun = (~config: Config.t, ~maxConnections=Env.Db.maxConnections) => // heard from anyone yet render as initializing rather than as empty. type running = { worker: worker, - child: NodeJs.ChildProcess.child, + child: NodeJs.ChildProcess.Child.t, mutable snapshot: option, mutable runtime: option, // A spawn failure can raise `error` and `exit` both, and a worker counted @@ -197,22 +197,22 @@ let fork = ( // it was written to, so a worker's errors stay on stderr for whoever is // redirecting it. [ - (child->NodeJs.ChildProcess.stdout, onOutput), - (child->NodeJs.ChildProcess.stderr, onErrorOutput), + (child->NodeJs.ChildProcess.Child.stdout, onOutput), + (child->NodeJs.ChildProcess.Child.stderr, onErrorOutput), ]->Array.forEach(((stream, onLine)) => switch stream->Null.toOption { | Some(stream) => { let (read, flush) = readLines(~onLine) - stream->NodeJs.ChildProcess.setEncoding("utf8") - stream->NodeJs.ChildProcess.onData(read) - stream->NodeJs.ChildProcess.onEnd(flush) + stream->NodeJs.ChildProcess.Stream.setEncoding("utf8") + stream->NodeJs.ChildProcess.Stream.onData(read) + stream->NodeJs.ChildProcess.Stream.onEnd(flush) } | None => () } ) } let running = {worker, child, snapshot: None, runtime: None, settled: false} - child->NodeJs.ChildProcess.onMessage(message => + child->NodeJs.ChildProcess.Child.onMessage(message => switch message { | Worker.Snapshot({metrics, runtime}) => { running.snapshot = Some(metrics) @@ -241,7 +241,7 @@ let stopReleaseCheck = group => { let stop = group => { group.stopping = true group->stopReleaseCheck - group.running->Array.forEach(r => r.child->NodeJs.ChildProcess.kill("SIGTERM")->ignore) + group.running->Array.forEach(r => r.child->NodeJs.ChildProcess.Child.kill("SIGTERM")->ignore) } // The dev console's cache dump, which belongs to the supervisor rather than to @@ -298,7 +298,8 @@ type ending = let classifyExit = (~code: Null.t, ~signal: Null.t, ~stopping) => switch (stopping, code->Null.toOption, signal->Null.toOption) { | (true, _, _) - | (_, Some(0), _) => Expected + | (_, Some(0), _) => + Expected | (_, _, Some("SIGTERM")) => Stopping | _ => Failed } @@ -334,11 +335,10 @@ let awaitExit = async (group): outcome => { } group.running->Array.forEach(r => { - r.child->NodeJs.ChildProcess.onExit( - (code, signal) => - r->onGone(~ending=classifyExit(~code, ~signal, ~stopping=group.stopping)), + r.child->NodeJs.ChildProcess.Child.onExit( + (code, signal) => r->onGone(~ending=classifyExit(~code, ~signal, ~stopping=group.stopping)), ) - r.child->NodeJs.ChildProcess.onChildError( + r.child->NodeJs.ChildProcess.Child.onError( exn => { Logging.errorWithExn(exn, `${r.worker->label} failed to start`) r->onGone(~ending=Failed) @@ -370,25 +370,22 @@ let awaitExit = async (group): outcome => { let isRunAtHead = (running: array) => running->Utils.Array.notEmpty && running->Array.every(r => - r.child->NodeJs.ChildProcess.connected && - r.snapshot->Option.mapOr(false, snapshot => snapshot.hasArrivedAtHead) + r.child->NodeJs.ChildProcess.Child.connected && + r.snapshot->Option.mapOr(false, snapshot => snapshot.hasArrivedAtHead) ) // Holds every worker at the head until the last of them arrives, then releases // them together. Chains enter the reorg threshold and go realtime as one // indexer, and in a split run only the supervisor can see when that is, so the // run switches over exactly as an unsplit one does. -let startReleaseCheck = group => - group.releaseCheck = Some( - setInterval(() => +let startReleaseCheck = group => group.releaseCheck = Some(setInterval(() => if group.running->isRunAtHead { group->stopReleaseCheck group.running->Array.forEach(r => - r.child->NodeJs.ChildProcess.send(Worker.ReleaseRealtime)->ignore + r.child->NodeJs.ChildProcess.Child.send(Worker.ReleaseRealtime)->ignore ) } - , releaseCheckIntervalMillis), - ) + , releaseCheckIntervalMillis)) // Runs the group: creates the schema for every chain, forks a worker per plan // entry, and serves the run's metrics, console and display from what they diff --git a/packages/envio/src/bindings/NodeJs.res b/packages/envio/src/bindings/NodeJs.res index c58fd4a99..6fd8c1ac0 100644 --- a/packages/envio/src/bindings/NodeJs.res +++ b/packages/envio/src/bindings/NodeJs.res @@ -157,7 +157,30 @@ module ChildProcess = { @module("child_process") external execWithOptions: (string, execOptions, callback) => unit = "exec" - type child + // One of a child's stdio slots, present only for a slot the parent asked to + // pipe rather than inherit. + module Stream = { + type t + @send external setEncoding: (t, string) => unit = "setEncoding" + @send external onData: (t, @as("data") _, string => unit) => unit = "on" + @send external onEnd: (t, @as("end") _, unit => unit) => unit = "on" + } + + module Child = { + type t + @send external send: (t, 'msg) => bool = "send" + @send external onMessage: (t, @as("message") _, 'msg => unit) => unit = "on" + @send + external onExit: (t, @as("exit") _, (Null.t, Null.t) => unit) => unit = "on" + @send external onError: (t, @as("error") _, exn => unit) => unit = "on" + @send external kill: (t, string) => bool = "kill" + // Whether the IPC channel is still open. Node closes it before it reports + // the exit, so this goes false while the child is still running. + @get external connected: t => bool = "connected" + @get external stdout: t => Null.t = "stdout" + @get external stderr: t => Null.t = "stderr" + } + type forkOptions = { cwd?: string, env?: dict, @@ -167,24 +190,7 @@ module ChildProcess = { stdio?: array, } @module("child_process") - external fork: (string, array, forkOptions) => child = "fork" - @send external send: (child, 'msg) => bool = "send" - @send external onMessage: (child, @as("message") _, 'msg => unit) => unit = "on" - @send - external onExit: (child, @as("exit") _, (Null.t, Null.t) => unit) => unit = "on" - @send external onChildError: (child, @as("error") _, exn => unit) => unit = "on" - @send external kill: (child, string) => bool = "kill" - // Whether the IPC channel is still open. Node closes it before it reports the - // exit, so this goes false while the child is still running. - @get external connected: child => bool = "connected" - - // Present only for a stdio slot the parent asked to pipe. - type stdioStream - @get external stdout: child => Null.t = "stdout" - @get external stderr: child => Null.t = "stderr" - @send external setEncoding: (stdioStream, string) => unit = "setEncoding" - @send external onData: (stdioStream, @as("data") _, string => unit) => unit = "on" - @send external onEnd: (stdioStream, @as("end") _, unit => unit) => unit = "on" + external fork: (string, array, forkOptions) => Child.t = "fork" } module Url = { From 2c319e4013fdb5220f190d4cbd2ee191d20ddbd6 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 22 Sep 2026 11:40:04 +0000 Subject: [PATCH 59/61] Drop the dead `--tui-off` flag and the Yargs parser with it MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Clap parses the whole argv and rejects unknown arguments, and `--tui-off` was never defined there, so `node bin.mjs start --tui-off` errors before any JS runs. `Tui.shouldUse` was re-parsing an argv that could not carry the flag it looked for. The display question is now `ENVIO_TUI`, then whether anything is watching — which is what it already resolved to. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_013saxnEKXJJXrZCkodmDMzE --- CONTRIBUTING.md | 1 - .../test/lib_tests/TuiShouldUse_test.res | 15 +++++++++ packages/envio/package.json | 1 - packages/envio/src/bindings/Yargs.res | 8 ----- packages/envio/src/tui/Tui.res | 22 +++---------- pnpm-lock.yaml | 33 ------------------- 6 files changed, 19 insertions(+), 61 deletions(-) create mode 100644 packages/envio-tests/test/lib_tests/TuiShouldUse_test.res delete mode 100644 packages/envio/src/bindings/Yargs.res diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index d4c273857..d9099d3cf 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -118,7 +118,6 @@ Entry point: - For `start`: primes the config JSON (`Config.prime`), sets `cwd` + env vars, then calls `Main.start(~migrate?)`. - For `migrate` / `drop-schema`: primes config and calls `Main.migrate` / `Main.dropSchema`. - `Main.start` (in `packages/envio`) is the indexer entry proper. Responsibilities: - - Parses CLI flags (`--tui-off`, etc.). - Loads runtime configuration (`Config.res`). - Starts an Express server that serves `/metrics`, `/health`, and the Development Console endpoints. - Initializes the Persistence layer (Postgres + Hasura) — a single `init()` call that also handles `~reset` + `upsertPersistedState` when `~migrate` is provided. diff --git a/packages/envio-tests/test/lib_tests/TuiShouldUse_test.res b/packages/envio-tests/test/lib_tests/TuiShouldUse_test.res new file mode 100644 index 000000000..a813c0c03 --- /dev/null +++ b/packages/envio-tests/test/lib_tests/TuiShouldUse_test.res @@ -0,0 +1,15 @@ +open Vitest + +describe("Tui.shouldUse", () => { + it( + "prefers ENVIO_TUI over what the terminal looks like, and never draws where it is suppressed", + t => { + t.expect(( + Tui.shouldUse(~suppressed=true, ~explicitTui=Some(true)), + Tui.shouldUse(~explicitTui=Some(true)), + Tui.shouldUse(~explicitTui=Some(false)), + Tui.shouldUse(~explicitTui=None), + )).toEqual((false, true, false, !Envio.isNonInteractive())) + }, + ) +}) diff --git a/packages/envio/package.json b/packages/envio/package.json index 04c7c7e7d..28e3bd82d 100644 --- a/packages/envio/package.json +++ b/packages/envio/package.json @@ -58,7 +58,6 @@ "express": "4.19.2", "pino": "10.3.1", "pino-pretty": "13.1.3", - "yargs": "17.7.2", "@rescript/runtime": "12.2.0", "rescript-schema": "9.5.1", "viem": "2.54.0", diff --git a/packages/envio/src/bindings/Yargs.res b/packages/envio/src/bindings/Yargs.res deleted file mode 100644 index 7805cab50..000000000 --- a/packages/envio/src/bindings/Yargs.res +++ /dev/null @@ -1,8 +0,0 @@ -type arg = string - -type parsedArgs<'a> = 'a - -@module("yargs/yargs") external yargs: array => parsedArgs<'a> = "default" -@module("yargs/helpers") external hideBin: array => array = "hideBin" - -@get external argv: parsedArgs<'a> => 'a = "argv" diff --git a/packages/envio/src/tui/Tui.res b/packages/envio/src/tui/Tui.res index 245462056..d42b54f5c 100644 --- a/packages/envio/src/tui/Tui.res +++ b/packages/envio/src/tui/Tui.res @@ -248,29 +248,15 @@ module App = { } } -type args = {@as("tui-off") tuiOff?: bool} - -type process -@val external process: process = "process" -@get external argv: process => 'a = "argv" - -type mainArgs = Yargs.parsedArgs - -// Whether this process draws the progress display: `--tui-off` first, then -// `ENVIO_TUI`, then whether anything is watching. A supervisor asks the same -// question its workers would have, since it is the one drawing for the run. -let shouldUse = (~suppressed=false) => { - let mainArgs: mainArgs = process->argv->Yargs.hideBin->Yargs.yargs->Yargs.argv - let explicitTui = switch mainArgs.tuiOff { - | Some(off) => Some(!off) - | None => Env.tuiEnvVar - } +// Whether this process draws the progress display: `ENVIO_TUI` first, then +// whether anything is watching. A supervisor asks the same question its +// workers would have, since it is the one drawing for the run. +let shouldUse = (~suppressed=false, ~explicitTui=Env.tuiEnvVar) => switch (suppressed, explicitTui) { | (true, _) => false | (_, Some(tui)) => tui | (_, None) => !Envio.isNonInteractive() } -} let start = (~config, ~getMetrics) => { let {rerender} = render() diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 48ea8d39d..24310080d 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -117,9 +117,6 @@ importers: viem: specifier: 2.54.0 version: 2.54.0(typescript@6.0.3) - yargs: - specifier: 17.7.2 - version: 17.7.2 devDependencies: rescript: specifier: 12.2.0 @@ -1297,10 +1294,6 @@ packages: cliui@7.0.4: resolution: {integrity: sha512-OcRE68cOsVMXp1Yvonl/fzkQOyjLSu/8bhPDfQt0e0/Eb283TKP20Fs2MqoPsr9SwA595rRCA+QMzYc9nBP+JQ==} - cliui@8.0.1: - resolution: {integrity: sha512-BSeNnyus75C4//NQ9gQt1/csTXyo/8Sb+afLAkzAptFuMsod9HFokGNudZpi/oQV73hnVK+sR+5PVRMd+Dr7YQ==} - engines: {node: '>=12'} - co@4.6.0: resolution: {integrity: sha512-QVb0dM5HvG+uaxitm8wONl7jltx8dqhfU33DcqtOZcLSVIKSDDLDi7+0LbAKiyI8hD9u42m2YxXSkMGWThaecQ==} engines: {iojs: '>= 1.0.0', node: '>= 0.12.0'} @@ -2933,18 +2926,10 @@ packages: resolution: {integrity: sha512-WOkpgNhPTlE73h4VFAFsOnomJVaovO8VqLDzy5saChRBFQFBoMYirowyW+Q9HB4HFF4Z7VZTiG3iSzJJA29yRA==} engines: {node: '>=10'} - yargs-parser@21.1.1: - resolution: {integrity: sha512-tVpsJW7DdjecAiFpbIB1e3qxIQsE6NoPc5/eTdrbbIC4h0LVsWhnoa3g+m2HclBIujHzsxZ4VJVA+GUuc2/LBw==} - engines: {node: '>=12'} - yargs@16.2.0: resolution: {integrity: sha512-D1mvvtDG0L5ft/jGWkLpG1+m0eQxOfaBvTNELraWj22wSVUMWxZUvYgJYcKh6jGGIkJFhH4IZPQhR4TKpc8mBw==} engines: {node: '>=10'} - yargs@17.7.2: - resolution: {integrity: sha512-7dSzzRQ++CKnNI/krKnYRV7JKKPUXMEh61soaHKg9mrWEhzFWhFnxPxGl+69cD1Ou63C13NUPCnmIcrvqCuM6w==} - engines: {node: '>=12'} - yoga-layout@3.2.1: resolution: {integrity: sha512-0LPOt3AxKqMdFBZA3HBAt/t/8vIKq7VaQYbuA8WxCgung+p9TVyKRYdpvCb80HcdTN2NkbIKbhNwKUfm3tQywQ==} @@ -3943,12 +3928,6 @@ snapshots: strip-ansi: 6.0.1 wrap-ansi: 7.0.0 - cliui@8.0.1: - dependencies: - string-width: 4.2.3 - strip-ansi: 6.0.1 - wrap-ansi: 7.0.0 - co@4.6.0: {} code-excerpt@4.0.0: @@ -5730,8 +5709,6 @@ snapshots: yargs-parser@20.2.4: {} - yargs-parser@21.1.1: {} - yargs@16.2.0: dependencies: cliui: 7.0.4 @@ -5742,14 +5719,4 @@ snapshots: y18n: 5.0.8 yargs-parser: 20.2.4 - yargs@17.7.2: - dependencies: - cliui: 8.0.1 - escalade: 3.2.0 - get-caller-file: 2.0.5 - require-directory: 2.1.1 - string-width: 4.2.3 - y18n: 5.0.8 - yargs-parser: 21.1.1 - yoga-layout@3.2.1: {} From 0a4b0a014ced198db83de780ed15b7fb4ebb4643 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 22 Sep 2026 12:06:27 +0000 Subject: [PATCH 60/61] Route source edits through Write/Edit so the formatter sees them Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_013saxnEKXJJXrZCkodmDMzE --- CLAUDE.md | 1 + 1 file changed, 1 insertion(+) diff --git a/CLAUDE.md b/CLAUDE.md index e6328afdd..fdb8bcb13 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -1,4 +1,5 @@ - Use `pnpm` over `npm`/`npx`. +- Edit `.res`/`.resi` and Rust files with Write/Edit, never a Bash heredoc or `sed`. The formatter runs on what those tools write; a push carrying an unformatted file is refused. - Always use single assert to check the whole value instead of multiple asserts for every field. ## Comments From 6b05de79c2c17685f084ade0b1f85288a3da3b3e Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 22 Sep 2026 12:23:31 +0000 Subject: [PATCH 61/61] Draw configured chains until workers report, release on the report itself - The supervised display drew an indexer with no chains at all until the first snapshot arrived; it now draws the run's chains as config.yaml has them, which is what an unsplit indexer shows in its first half second. - The realtime barrier rode a 500ms poll. A worker's report is the only thing that can change the answer, so the check now hangs off the report and the interval is gone. - Trim the comments to what the code can't say, and note that every exit during a stop reads as expected, so the run still exits 0. - Env: the connection budget that splits a run is 4, not the default 2. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_013saxnEKXJJXrZCkodmDMzE --- .../envio-tests/test/SupervisorFork_test.res | 38 +++- .../envio-tests/test/helpers/fakeWorker.mjs | 7 + .../test/lib_tests/Supervisor_test.res | 44 ++++ packages/envio/src/Config.res | 16 +- packages/envio/src/CrossChainState.res | 8 +- packages/envio/src/Env.res | 4 +- packages/envio/src/Supervisor.res | 192 ++++++++++-------- 7 files changed, 204 insertions(+), 105 deletions(-) diff --git a/packages/envio-tests/test/SupervisorFork_test.res b/packages/envio-tests/test/SupervisorFork_test.res index 286f139aa..e7b296c0f 100644 --- a/packages/envio-tests/test/SupervisorFork_test.res +++ b/packages/envio-tests/test/SupervisorFork_test.res @@ -25,6 +25,7 @@ let forkFixture = ( ~pipeOutput=false, ~onOutput=?, ~onErrorOutput=?, + ~onSnapshot=?, ) => Supervisor.fork( {chainIds: chainIds->Array.map(ChainId.fromInt), maxConnections}, @@ -36,6 +37,7 @@ let forkFixture = ( ~pipeOutput, ~onOutput?, ~onErrorOutput?, + ~onSnapshot?, ) describe("Supervisor.fork", () => { @@ -93,7 +95,7 @@ describe("Supervisor.awaitExit", () => { let group: Supervisor.group = { running: [forkFixture(~chainIds=[1]), forkFixture(~chainIds=[137])], stopping: false, - releaseCheck: None, + holdingRealtime: false, } t.expect(await outcome(group)).toStrictEqual(Ok(Supervisor.Finished)) @@ -106,7 +108,7 @@ describe("Supervisor.awaitExit", () => { let group: Supervisor.group = { running: [forkFixture(~chainIds=[1]), forkFixture(~chainIds=[137])], stopping: false, - releaseCheck: None, + holdingRealtime: false, } group->Supervisor.stop @@ -125,7 +127,7 @@ describe("Supervisor.awaitExit", () => { let group: Supervisor.group = { running: [signalled, sibling], stopping: false, - releaseCheck: None, + holdingRealtime: false, } let ended = outcome(group) signalled.child->NodeJs.ChildProcess.Child.kill("SIGTERM")->ignore @@ -143,7 +145,7 @@ describe("Supervisor.awaitExit", () => { let group: Supervisor.group = { running: [failing, lingering], stopping: false, - releaseCheck: None, + holdingRealtime: false, } // The survivor was taken down rather than left indexing half a schema. @@ -187,7 +189,7 @@ describe("Supervisor.fork output", () => { ), ], stopping: false, - releaseCheck: None, + holdingRealtime: false, } let _ = await group->Supervisor.awaitExit @@ -231,7 +233,7 @@ describe("Supervisor.isRunAtHead", () => { let group: Supervisor.group = { running: [arrived, backfilling], stopping: false, - releaseCheck: None, + holdingRealtime: false, } await untilReported(group.running) @@ -259,7 +261,7 @@ describe("Supervisor.isRunAtHead", () => { let group: Supervisor.group = { running: [lingering, leaving], stopping: false, - releaseCheck: None, + holdingRealtime: false, } await untilReported(group.running) await untilGone(leaving) @@ -274,4 +276,26 @@ describe("Supervisor.isRunAtHead", () => { t.expect(readings).toStrictEqual((false, 2)) }) + + // The release rides the workers' reports rather than a clock: the run opens on + // the report that completes it, and these workers exit only once released. + Async.it("Releases every worker on the report that completes the run", async t => { + NodeJs.Process.process.env->Dict.set("FAKE_WORKER", "await-release") + NodeJs.Process.process.env->Dict.set("FAKE_WORKER_ARRIVED", "1") + let group: Supervisor.group = {running: [], stopping: false, holdingRealtime: true} + group.running = + [[1], [137]]->Array.map( + chainIds => + forkFixture( + ~chainIds, + ~holdRealtime=true, + ~onSnapshot=() => group->Supervisor.releaseIfAtHead, + ), + ) + + t.expect((await group->Supervisor.awaitExit, group.holdingRealtime)).toStrictEqual(( + Supervisor.Finished, + false, + )) + }) }) diff --git a/packages/envio-tests/test/helpers/fakeWorker.mjs b/packages/envio-tests/test/helpers/fakeWorker.mjs index 0adc69c36..edc00abe1 100644 --- a/packages/envio-tests/test/helpers/fakeWorker.mjs +++ b/packages/envio-tests/test/helpers/fakeWorker.mjs @@ -39,3 +39,10 @@ if (mode === "print") { // Nothing else keeps a "linger" worker alive; it waits to be stopped. if (mode === "linger") setInterval(() => {}, 1000); + +// Ends only once the supervisor releases it, so a run that reaches its exit is +// a run whose barrier opened. +if (mode === "await-release") { + setInterval(() => {}, 1000); + process.on("message", () => process.exit(0)); +} diff --git a/packages/envio-tests/test/lib_tests/Supervisor_test.res b/packages/envio-tests/test/lib_tests/Supervisor_test.res index 5cd33ea7a..c6b9c0b88 100644 --- a/packages/envio-tests/test/lib_tests/Supervisor_test.res +++ b/packages/envio-tests/test/lib_tests/Supervisor_test.res @@ -306,3 +306,47 @@ describe("Supervisor.syncCache", () => { t.expect(dumps.contents).toBe(2) }) }) + +describe("Supervisor.configuredChains", () => { + it("Draws the run's chains at their configured blocks, with nothing indexed", t => { + let config = TestConfig.fromUserApi(` +name: test-config +chains: + - id: 1 + start_block: 100 + end_block: 500 + contracts: + - name: Gravatar + address: "0x2B2f78c5BF6D9C12Ee1225D5F374aa91204580c3" + events: + - event: "TestEvent()" + - id: 137 + rpc: + url: https://rpc.example.test + for: sync + start_block: 0 + contracts: + - name: Poap + address: "0x2B2f78c5BF6D9C12Ee1225D5F374aa91204580c3" + events: + - event: "TestEvent()" +`) + + t.expect( + Supervisor.configuredChains(config)->Array.map( + chain => ( + chain.chainId->ChainId.toString, + chain.startBlock, + chain.endBlock, + chain.poweredByHyperSync, + chain.progressBlockNumber, + chain.numEventsProcessed, + chain.isReady, + ), + ), + ).toStrictEqual([ + ("1", 100, Some(500), true, -1, 0., false), + ("137", 0, None, false, -1, 0., false), + ]) + }) +}) diff --git a/packages/envio/src/Config.res b/packages/envio/src/Config.res index fe2b6d90a..9bf0d3a3c 100644 --- a/packages/envio/src/Config.res +++ b/packages/envio/src/Config.res @@ -626,10 +626,9 @@ let getChain = (config, ~chainId) => ) // Whether every entity belongs to exactly one chain. Read off the checkpoint -// sequence rather than the entities again, because that is the same fact and -// the one that makes splitting safe: a chain only gets a counter of its own -// when no other chain can reach its rows, and a counter of its own is what lets -// a process advance one chain without saying anything about the others. +// sequence rather than the entities again: a chain gets a counter of its own +// only when no other chain can reach its rows, which is the same fact and the +// one that makes splitting a run across processes safe. let isPerChain = (config: t) => switch config.checkpointSequence { | PerChain => true @@ -1232,11 +1231,10 @@ let prime = (json: JSON.t): unit => { cached := None } -// The fields `envio start` and `envio dev` set on a public config from the -// command they were given rather than from the project's files: which chains -// the process drives, and whether the run is a dev run. A worker parses the -// same files its supervisor did, so these are the only two it cannot arrive at -// on its own, and re-applying them is what makes its config the supervisor's. +// What the command decided rather than the project's files: which chains this +// process drives, and whether the run is a dev run. A worker parses the same +// files its supervisor did, so these are the only two it cannot arrive at on +// its own. let withCommandFields = (json: JSON.t, ~chainIds, ~isDev) => switch json->JSON.Decode.object { | Some(fields) => { diff --git a/packages/envio/src/CrossChainState.res b/packages/envio/src/CrossChainState.res index c86b65736..837f0f23e 100644 --- a/packages/envio/src/CrossChainState.res +++ b/packages/envio/src/CrossChainState.res @@ -285,12 +285,8 @@ let markReady = (crossChainState: t, ~readyAt) => { } // Each chain that has just finished indexing, said once, by the chain it is -// about. A chain that finishes early says so then, rather than when the last -// chain in its process catches up. -// -// A chain with no end block has only finished its history: the blocks that can -// still be reorged are indexed after every chain gets this far, so it says what -// it is waiting on when there is anything to wait for. +// about — so a chain that finishes early says so then, rather than when the +// last chain in its process catches up. let reportFinished = (crossChainState: t) => { let waitingOnOthers = crossChainState.holdRealtime || crossChainState.chainIds->Array.length > 1 crossChainState.chainStates diff --git a/packages/envio/src/Env.res b/packages/envio/src/Env.res index b60490926..c38b1a581 100644 --- a/packages/envio/src/Env.res +++ b/packages/envio/src/Env.res @@ -129,8 +129,8 @@ module Db = { ) // The budget for the whole run, not for one process: a run that splits across // workers divides it among them, and each caps its own pool to its share. - // The default buys a single worker, so a run splits only once the operator - // raises the budget it may spend. + // Splitting takes two workers and a worker takes two connections, so a budget + // under 4 — the default among them — keeps the run in one process. let maxConnections = envSafe->EnvSafe.get("ENVIO_PG_MAX_CONNECTIONS", S.int, ~fallback=2) } diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res index 7bdf4f09f..f59cee3c9 100644 --- a/packages/envio/src/Supervisor.res +++ b/packages/envio/src/Supervisor.res @@ -7,21 +7,19 @@ type worker = {chainIds: array, maxConnections: int} let minConnectionsPerWorker = 2 // Most processes a run is split into, however much budget it is given. A worker -// is a whole Node process, with its own heap, its own copy of the handler -// modules and its own source clients, and a run holding more of them shares one -// machine between them. A conservative ceiling while the split is new: past it -// a raised budget widens the workers' pools rather than adding workers. +// is a whole Node process with its own heap, handler modules and source +// clients, and however many of them a run has, they share one machine. Past +// this a raised budget widens the workers' pools rather than adding workers. let maxWorkers = 4 // How to spend a connection budget on the chains a run indexes. `None` keeps -// the run in one process, which is what a budget too small to afford two -// workers, or a config with nothing to split, has to do. +// the run in one process. // -// Chains are dealt in config order and the direction reverses each pass, so -// the first chains lead different workers and the worker that took the first -// picks up the last. How much work a chain has is the contracts' to decide, -// not the chain's, so config order is the one ranking the run can be given: -// listing chains busiest-first in config.yaml is what balances the layout. +// Chains are dealt in config order and the direction reverses each pass, so the +// first chains lead different workers and the worker that took the first picks +// up the last. How much work a chain has is the contracts' to decide, so config +// order is the only ranking the run can be given: listing chains busiest-first +// in config.yaml is what balances the layout. let plan = (~chainIds: array, ~maxConnections: int): option> => { let workerCount = [chainIds->Array.length, maxConnections / minConnectionsPerWorker, maxWorkers]->Array.reduce( @@ -60,8 +58,7 @@ let planForRun = (~config: Config.t, ~maxConnections=Env.Db.maxConnections) => } // One forked worker: the process, the chains it drives, and the last snapshot -// it reported. `None` until it reports, which is what makes a run that hasn't -// heard from anyone yet render as initializing rather than as empty. +// it reported. type running = { worker: worker, child: NodeJs.ChildProcess.Child.t, @@ -72,8 +69,6 @@ type running = { mutable settled: bool, } -// A worker is named by the chains it drives, which is what an operator reading -// its memory or its event loop wants to know. let name = (worker: worker) => worker.chainIds->Array.map(ChainId.toString)->Array.joinUnsafe(";") let label = (worker: worker) => `[chain ${worker->name}]` @@ -117,14 +112,9 @@ let readLines = (~onLine) => { (read, flush) } -// The run's memory budgets, and each worker's share of them. Both are the whole -// indexer's rather than one chain's or one process's: the fetch buffer pool is -// deliberately independent of how many chains a run has, and an indexer split -// across processes that took each budget whole in every one of them would hold -// as many times the memory as it happened to have workers. -// -// Read here and handed over in the spawn environment for the same reason the -// connection share is: a worker's `Env` reads them as it loads. +// Each worker's share of the run's memory budgets. Both are the whole indexer's +// rather than one process's, so workers that each took the whole of one would +// hold as many times the memory as the run happened to have workers. let memoryBudgets = (~workerCount) => [ ("ENVIO_INDEXING_MAX_BUFFER_SIZE", CrossChainState.calculateTargetBufferSize()), @@ -136,9 +126,8 @@ let memoryBudgets = (~workerCount) => Pervasives.max(1, budget / workerCount)->Int.toString, )) -// Whether this process's own output is a terminal. `pino-pretty` colorizes on -// that test, and a piped worker would fail it for a run the operator is -// watching in colour. +// `pino-pretty` colorizes on this test, and a piped worker would fail it for a +// run the operator is watching in colour. @val external stdoutIsTty: Nullable.t = "process.stdout.isTTY" let fork = ( @@ -161,6 +150,7 @@ let fork = ( ~pipeOutput=false, ~onOutput=Console.log, ~onErrorOutput=Console.error, + ~onSnapshot=() => (), ) => { let env = NodeJs.Process.process.env->Dict.copy env->Dict.set( @@ -217,6 +207,7 @@ let fork = ( | Worker.Snapshot({metrics, runtime}) => { running.snapshot = Some(metrics) running.runtime = Some(runtime) + onSnapshot() } } ) @@ -226,32 +217,23 @@ let fork = ( // The forked workers of one run, and whether their supervisor is the one // taking them down, which is what tells an expected exit from the rest. type group = { - running: array, + // Assigned once the forks are made, which is after the group exists: a + // worker's report asks the group whether the run may go realtime. + mutable running: array, mutable stopping: bool, - // The poll that asks whether the run may go realtime, while it is still - // asking. A group being taken down has nothing left to release. - mutable releaseCheck: option, -} - -let stopReleaseCheck = group => { - group.releaseCheck->Option.forEach(clearInterval) - group.releaseCheck = None + // Whether the workers are still waiting for the run's leave to go realtime. + mutable holdingRealtime: bool, } let stop = group => { group.stopping = true - group->stopReleaseCheck group.running->Array.forEach(r => r.child->NodeJs.ChildProcess.Child.kill("SIGTERM")->ignore) } -// The dev console's cache dump, which belongs to the supervisor rather than to -// its workers: a dump copies every effect cache table in the schema to a file -// named after the effect, so a worker asked to do it would copy its siblings' -// chains too, and several asked at once would write the same files at the same -// time. Nothing in it is a worker's to know — the rows it copies are the ones -// already committed. -// -// Requests that overlap join the dump in flight, for the same reason. +// The dev console's cache dump belongs to the supervisor: a dump copies every +// effect cache table in the schema, so a worker asked to do it would copy its +// siblings' chains too, and several asked at once would write the same files at +// the same time. Overlapping requests join the dump in flight for that reason. let syncCache = { let inFlight = ref(None) (~dump) => @@ -265,10 +247,9 @@ let syncCache = { } // The supervisor handed its connections to the workers, so a dump opens one of -// its own for as long as it takes. That puts the run one connection over its -// budget, deliberately: the console that asks for a dump is `envio dev` only, -// one connection is a cheaper price than pausing the indexing to free one, and -// the pool is capped at that one. +// its own and puts the run one connection over its budget — deliberately: only +// `envio dev` asks for a dump, and the alternative is pausing the indexing to +// free one. let dumpCache = (~config) => { let storage = PgStorage.makeStorageFromEnv(~config, ~sql=PgStorage.makeClient(~maxConnections=1)) storage.dumpEffectCache()->Promise.finally(() => storage.close()->Promise.ignore) @@ -295,6 +276,11 @@ type ending = // // The kernel's out-of-memory killer sends SIGKILL, which stays a failure — as // does every non-zero exit of a worker the supervisor didn't ask to stop. +// +// Once the supervisor is stopping, though, every exit reads as expected and the +// run exits 0: a worker that crashes on its way down is indistinguishable from +// one that took the SIGTERM, and a stop that reported a failure would fail +// every restart the crash happened to race. let classifyExit = (~code: Null.t, ~signal: Null.t, ~stopping) => switch (stopping, code->Null.toOption, signal->Null.toOption) { | (true, _, _) @@ -316,9 +302,7 @@ let awaitExit = async (group): outcome => { let onGone = (r, ~ending) => if !r.settled { r.settled = true - // The rest of the run goes down with it either way: one worker short - // leaves its chains unindexed, and a run that kept the others going - // would look healthy while falling behind. What differs is whether the + // The rest of the run goes down either way; what differs is whether the // run reports itself as having failed. switch ending { | Expected => () @@ -353,14 +337,8 @@ let awaitExit = async (group): outcome => { group.stopping ? Stopped : Finished } -// How often the supervisor asks whether the run may go realtime. Matches the -// rate its workers report at: nothing changes in between. -%%private(let releaseCheckIntervalMillis = 500) - // Whether a run holding its workers back may let them go: every worker is still // there to be released, has reported, and has got as far as it can on its own. -// What counts as arrived is the worker's own conclusion, the supervisor only -// asking each of them the question an unsplit run asks itself. // // A worker that is gone leaves the run a process short, so there is nothing to // release it into, and its last snapshot outlives it. The channel is what says @@ -376,16 +354,71 @@ let isRunAtHead = (running: array) => // Holds every worker at the head until the last of them arrives, then releases // them together. Chains enter the reorg threshold and go realtime as one -// indexer, and in a split run only the supervisor can see when that is, so the -// run switches over exactly as an unsplit one does. -let startReleaseCheck = group => group.releaseCheck = Some(setInterval(() => - if group.running->isRunAtHead { - group->stopReleaseCheck - group.running->Array.forEach(r => - r.child->NodeJs.ChildProcess.Child.send(Worker.ReleaseRealtime)->ignore - ) - } - , releaseCheckIntervalMillis)) +// indexer, and in a split run only the supervisor can see when that is. +// +// Asked on every report rather than on a clock of its own: a report is the only +// thing that can change the answer. +let releaseIfAtHead = group => + if group.holdingRealtime && !group.stopping && group.running->isRunAtHead { + group.holdingRealtime = false + group.running->Array.forEach(r => + r.child->NodeJs.ChildProcess.Child.send(Worker.ReleaseRealtime)->ignore + ) + } + +// The run's chains as an unsplit indexer reports them before it has fetched +// anything: at their configured blocks, with nothing indexed. A display that +// hasn't heard from a worker yet draws these, rather than the indexer with no +// chains at all that an empty merge would render. +let configuredChains = (config: Config.t): array => + config.chainMap + ->ChainMap.values + ->Array.map((chain): Metrics.chainMetrics => { + chainId: chain.id, + poweredByHyperSync: switch chain.sourceConfig { + | EvmSourceConfig({hypersync}) => hypersync->Option.isSome + | FuelSourceConfig(_) | SvmSourceConfig(_) => true + | SimulateSourceConfig(_) | CustomSources(_) => false + }, + firstEventBlockNumber: None, + latestProcessedBlock: None, + timestampCaughtUpToHeadOrEndblock: None, + numEventsProcessed: 0., + latestFetchedBlockNumber: 0, + knownHeight: 0, + numBatchesFetched: 0, + // A chain resolves `start_block: latest` against its own head as it starts, + // which is a worker's to do and no supervisor's to guess. + startBlock: switch chain.startBlock { + | Block(block) => block + | Latest => 0 + }, + endBlock: chain.endBlock, + numAddresses: 0, + addressesByContract: [], + isReady: false, + sourceBlockNumber: 0, + progressBlockNumber: -1, + progressLatencyMs: None, + progressBlockTime: None, + concurrency: 0, + partitionsCount: 0, + bufferSize: 0, + bufferBlockNumber: -1, + idleSeconds: 0., + waitingForNewBlockSeconds: 0., + queryingSeconds: 0., + blockRangeFetchSeconds: 0., + blockRangeParseSeconds: 0., + blockRangeFetchCount: 0., + blockRangeFetchedEvents: 0., + blockRangeFetchedBlocks: 0., + reorgCount: 0, + reorgDetectedBlock: None, + rollbackTargetBlock: None, + rateLimitTimeMs: 0., + rateLimitResetInMs: None, + }) // Runs the group: creates the schema for every chain, forks a worker per plan // entry, and serves the run's metrics, console and display from what they @@ -427,19 +460,18 @@ let run = async (~config: Config.t, ~workers: array, ~reset) => { (persistence->Persistence.getInitializedState).chains->Array.some(chain => chain.timestampCaughtUpToHeadOrEndblock->Option.isNone ) - let group = { - running: workers->Array.mapWithIndex((worker, workerIndex) => + let group = {running: [], stopping: false, holdingRealtime: holdRealtime} + group.running = + workers->Array.mapWithIndex((worker, workerIndex) => worker->fork( ~workerIndex, ~workerCount=workers->Array.length, ~holdRealtime, ~isDev=config.isDev, ~pipeOutput=shouldUseTui, + ~onSnapshot=() => group->releaseIfAtHead, ) - ), - stopping: false, - releaseCheck: None, - } + ) let reported = () => group.running->Array.filterMap(r => r.snapshot) let merge = snapshots => @@ -475,12 +507,13 @@ let run = async (~config: Config.t, ~workers: array, ~reset) => { ~onSyncCache=() => syncCache(~dump=() => dumpCache(~config)), ) - if holdRealtime { - group->startReleaseCheck - } - if shouldUseTui { - let _rerender = Tui.start(~config, ~getMetrics=() => reported()->merge) + let _rerender = Tui.start(~config, ~getMetrics=() => + switch reported() { + | [] => {...[]->merge, chains: configuredChains(config)} + | snapshots => snapshots->merge + } + ) } // Whichever signal asks the run to stop, the supervisor is the one that @@ -494,9 +527,6 @@ let run = async (~config: Config.t, ~workers: array, ~reset) => { // is the exception, as it is for a single process: it keeps the final state // on screen until the terminal closes it. let outcome = await group->awaitExit - // A group that ended on its own was never stopped, and a display keeps this - // process alive long past the last worker the check was asking about. - group->stopReleaseCheck switch outcome { | Stopped => NodeJs.process->NodeJs.exitWithCode(Success)