diff --git a/.claude/hooks/check-format.sh b/.claude/hooks/check-format.sh new file mode 100755 index 0000000000..0a180dc35f --- /dev/null +++ b/.claude/hooks/check-format.sh @@ -0,0 +1,44 @@ +#!/usr/bin/env bash +# Blocks `git push` while a file this branch touches is unformatted. Edits made +# through Bash bypass the PostToolUse formatter, so without this the first +# anyone hears of the formatting is a CI failure. +set -uo pipefail + +cmd=$(jq -r '.tool_input.command // ""') +case "$cmd" in +*"git push"*) ;; +*) exit 0 ;; +esac + +cd "${CLAUDE_PROJECT_DIR:-.}" || exit 0 + +base=$(git merge-base HEAD origin/main 2>/dev/null || true) +changed=$( + { + git diff --name-only HEAD + [ -n "$base" ] && git diff --name-only "$base"...HEAD + git ls-files -o --exclude-standard + } | sort -u | while read -r f; do [ -f "$f" ] && echo "$f"; done +) + +problems="" + +res=$(printf '%s\n' "$changed" | grep -E '\.resi?$' || true) +if [ -n "$res" ]; then + out=$(printf '%s\n' "$res" | xargs pnpx rescript@12.2.0 format --check 2>&1 | grep '^\[format check\]' || true) + if [ -n "$out" ]; then + problems="$problems"$'\n'"Unformatted ReScript (fix: pnpx rescript@12.2.0 format ):"$'\n'"$out" + fi +fi + +if printf '%s\n' "$changed" | grep -q '^packages/cli/'; then + if ! out=$(cd packages/cli && cargo fmt --check 2>&1); then + problems="$problems"$'\n'"Unformatted Rust (fix: cd packages/cli && cargo fmt):"$'\n'"$out" + fi +fi + +if [ -n "$problems" ]; then + jq -n --arg r "Formatting check failed, so the push was not run.$problems" \ + '{hookSpecificOutput:{hookEventName:"PreToolUse",permissionDecision:"deny",permissionDecisionReason:$r}}' +fi +exit 0 diff --git a/.claude/settings.json b/.claude/settings.json index 9c36110182..e95aefc8f9 100644 --- a/.claude/settings.json +++ b/.claude/settings.json @@ -34,6 +34,19 @@ } ] } + ], + "PreToolUse": [ + { + "matcher": "Bash", + "hooks": [ + { + "type": "command", + "command": "$CLAUDE_PROJECT_DIR/.claude/hooks/check-format.sh", + "timeout": 120, + "statusMessage": "Checking formatting before push..." + } + ] + } ] } } diff --git a/CLAUDE.md b/CLAUDE.md index e6328afdd5..fdb8bcb135 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -1,4 +1,5 @@ - Use `pnpm` over `npm`/`npx`. +- Edit `.res`/`.resi` and Rust files with Write/Edit, never a Bash heredoc or `sed`. The formatter runs on what those tools write; a push carrying an unformatted file is refused. - Always use single assert to check the whole value instead of multiple asserts for every field. ## Comments diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index d4c2738574..d9099d3cfe 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -118,7 +118,6 @@ Entry point: - For `start`: primes the config JSON (`Config.prime`), sets `cwd` + env vars, then calls `Main.start(~migrate?)`. - For `migrate` / `drop-schema`: primes config and calls `Main.migrate` / `Main.dropSchema`. - `Main.start` (in `packages/envio`) is the indexer entry proper. Responsibilities: - - Parses CLI flags (`--tui-off`, etc.). - Loads runtime configuration (`Config.res`). - Starts an Express server that serves `/metrics`, `/health`, and the Development Console endpoints. - Initializes the Persistence layer (Postgres + Hasura) — a single `init()` call that also handles `~reset` + `upsertPersistedState` when `~migrate` is provided. diff --git a/packages/cli/CommandLineHelp.md b/packages/cli/CommandLineHelp.md index 286a9ed221..9215959c27 100644 --- a/packages/cli/CommandLineHelp.md +++ b/packages/cli/CommandLineHelp.md @@ -378,7 +378,9 @@ Start the indexer. Runs codegen automatically before launching so the on-disk ty ###### **Options:** * `-r`, `--restart` — Clear your database and restart indexing from scratch -* `--chain ` — Index only this chain, leaving the others to their own `envio start --chain` processes. Repeat the flag for several chains. Requires a schema whose entities are all per-chain, created for every chain by `envio local db-migrate up` before any process starts. Assign each configured chain to exactly one process, and give each its own `ENVIO_INDEXER_PORT`. Each process builds the indexes for its own chains and reports them ready as they catch up, independently of the others +* `--chain ` — Index only this chain, leaving the others to their own `envio start --chain` processes. Repeat the flag for several chains. + + Only needed to place chains yourself. Requires a per-chain schema, migrated for every chain before any process starts, and a separate `ENVIO_INDEXER_PORT` per process. diff --git a/packages/cli/src/cli_args/clap_definitions.rs b/packages/cli/src/cli_args/clap_definitions.rs index 260571f924..1d64172dac 100644 --- a/packages/cli/src/cli_args/clap_definitions.rs +++ b/packages/cli/src/cli_args/clap_definitions.rs @@ -145,11 +145,10 @@ pub struct StartArgs { pub restart: bool, ///Index only this chain, leaving the others to their own `envio start --chain` processes. - ///Repeat the flag for several chains. Requires a schema whose entities are all per-chain, - ///created for every chain by `envio local db-migrate up` before any process starts. - ///Assign each configured chain to exactly one process, and give each its own - ///`ENVIO_INDEXER_PORT`. Each process builds the indexes for its own chains and reports them - ///ready as they catch up, independently of the others. + ///Repeat the flag for several chains. + /// + ///Only needed to place chains yourself. Requires a per-chain schema, migrated for every + ///chain before any process starts, and a separate `ENVIO_INDEXER_PORT` per process. #[arg(long = "chain", value_name = "CHAIN_ID")] pub chains: Vec, } diff --git a/packages/e2e-tests/src/dependency-tests/install.test.ts b/packages/e2e-tests/src/dependency-tests/install.test.ts index 2130162812..0e6fcc4242 100644 --- a/packages/e2e-tests/src/dependency-tests/install.test.ts +++ b/packages/e2e-tests/src/dependency-tests/install.test.ts @@ -183,7 +183,7 @@ describe("Isolated dependency e2e", () => { await waitForOutput( indexerProcess, - "All chains are caught up to end blocks", + "Indexed to the end block", 120_000 ); diff --git a/packages/e2e-tests/src/e2e/e2e.test.ts b/packages/e2e-tests/src/e2e/e2e.test.ts index 929a4c502a..f68784bfca 100644 --- a/packages/e2e-tests/src/e2e/e2e.test.ts +++ b/packages/e2e-tests/src/e2e/e2e.test.ts @@ -4,7 +4,7 @@ * Tests the full indexer flow with database and ClickHouse sink: * 1. Ensure ClickHouse is running (CI service or local container) * 2. Start `envio dev` in background with ClickHouse sink enabled - * 3. Wait for "All chains are caught up to end blocks" in stdout + * 3. Wait for "Indexed to the end block" in stdout * 4. Verify GraphQL queries return expected data * 5. Verify ClickHouse sink received the indexed data */ @@ -96,7 +96,7 @@ describe.skipIf(!dockerAvailable)("E2E: Indexer with GraphQL and ClickHouse sink await waitForOutput( indexerProcess, - "All chains are caught up to end blocks", + "Indexed to the end block", 120_000 ); @@ -842,7 +842,7 @@ describe.skipIf(!dockerAvailable)("E2E: Indexer with GraphQL and ClickHouse sink // waitForOutput rejects. Success means DB state was used. await waitForOutput( secondProcess, - "All chains are caught up to end blocks", + "Indexed to the end block", 120_000 ); } finally { diff --git a/packages/e2e-tests/src/e2e/split-run.test.ts b/packages/e2e-tests/src/e2e/split-run.test.ts new file mode 100644 index 0000000000..3d53450d00 --- /dev/null +++ b/packages/e2e-tests/src/e2e/split-run.test.ts @@ -0,0 +1,161 @@ +/** + * A plain `envio start` over a per-chain schema, with a connection budget that + * affords two processes, splits the chains across forked workers and stays + * one indexer to the operator: one metrics endpoint over every process, one + * exit once every chain reaches its end block, one signal to stop it all. + * + * Needs Postgres but no Docker: it drives `envio start` with Hasura disabled. + */ + +import { describe, it, expect, beforeAll, afterAll } from "vitest"; +import { ChildProcess } from "child_process"; +import path from "path"; +import { config } from "../config.js"; +import { runCommand, startBackground, waitForOutput } from "../utils/process.js"; +import { pgRows, closePg, isPgReachable } from "../utils/pg-direct.js"; + +const PROJECT_DIR = path.join(config.scenariosDir, "split_test"); +const PG_SCHEMA = "e2e_split_run"; +const PORT = 9897; + +const indexerEnv = { + ENVIO_PG_SCHEMA: PG_SCHEMA, + // Two processes' worth: the split's own condition. + ENVIO_PG_MAX_CONNECTIONS: "4", + ENVIO_HASURA: "false", + ENVIO_TUI: "false", + ENVIO_INDEXER_PORT: String(PORT), + ENVIO_API_TOKEN: process.env.ENVIO_API_TOKEN ?? "", +}; + +const reachable = await isPgReachable(); + +if (!reachable && process.env.CI) { + throw new Error( + "Postgres is unreachable, so the split-run suite cannot run. Refusing to skip it in CI." + ); +} + +const exitCode = (child: ChildProcess) => + new Promise((resolve) => child.on("close", resolve)); + +/** + * Leaves no indexer behind when a test fails before its own shutdown: one that + * survived would hold the port and the schema against everything after it. + */ +const stopIfRunning = async (indexer: ChildProcess) => { + if (indexer.exitCode !== null || indexer.signalCode !== null) return; + const exited = exitCode(indexer); + indexer.kill("SIGINT"); + const abandon = setTimeout(() => indexer.kill("SIGKILL"), 10_000); + await exited; + clearTimeout(abandon); +}; + +/** Polls an endpoint of the supervisor until its body satisfies `ready`. */ +const scrapeUntil = async (route: string, ready: (body: string) => boolean) => { + const deadline = Date.now() + config.timeouts.indexerStartup; + let body = ""; + while (Date.now() < deadline) { + try { + body = await (await fetch(`http://localhost:${PORT}${route}`)).text(); + if (ready(body)) return body; + } catch {} + await new Promise((r) => setTimeout(r, 500)); + } + throw new Error(`Timed out scraping ${route}\n--- last body ---\n${body}`); +}; + +const start = (args: string[]) => + startBackground(config.envioCommand, [...config.envioArgs, "start", ...args], { + cwd: PROJECT_DIR, + env: indexerEnv, + }); + +describe.skipIf(!reachable)("E2E: a split run is one indexer", () => { + beforeAll(async () => { + const codegen = await runCommand( + config.envioCommand, + [...config.envioArgs, "codegen"], + { cwd: PROJECT_DIR, env: indexerEnv, timeout: config.timeouts.codegen } + ); + expect(codegen.exitCode, `codegen failed: ${codegen.stderr}`).toBe(0); + }, config.timeouts.codegen); + + afterAll(async () => { + await closePg(); + }); + + it("Exits once every chain is done, with both chains' rows written", async () => { + const indexer = start(["-r"]); + try { + const exit = exitCode(indexer); + await waitForOutput(indexer, "Indexing will be split across multiple processes", config.timeouts.indexerStartup); + + expect({ + exitCode: await exit, + rowsPerChain: await pgRows( + `SELECT "chain_id", COUNT(*) > 0 FROM "${PG_SCHEMA}"."Transfer" GROUP BY "chain_id" ORDER BY "chain_id"` + ), + // Reaching an end block doesn't make a worker done: it owes the schema + // the indexes its chains deferred, and until the run goes realtime it + // has no leave to commit them. + readyPerChain: await pgRows( + `SELECT "id"::text, "ready_at" IS NOT NULL FROM "${PG_SCHEMA}"."envio_chains" ORDER BY "id"` + ), + }).toEqual({ + exitCode: 0, + rowsPerChain: [ + [1, true], + [8453, true], + ], + readyPerChain: [ + ["1", true], + ["8453", true], + ], + }); + } finally { + await stopIfRunning(indexer); + } + }); + + // The same chains with no end block: a run that nothing but a stop ends, so + // there is time to read what it serves. + it("Serves every process's metrics, and stops them all on one interrupt", async () => { + const indexer = start(["-r", "--config", "config.head.yaml"]); + try { + const exit = exitCode(indexer); + await waitForOutput(indexer, "Indexing will be split across multiple processes", config.timeouts.indexerStartup); + + const [runtime, metrics] = await Promise.all([ + // Each worker's readings, told apart by label. + scrapeUntil("/metrics/runtime", (body) => body.includes('worker="8453"')), + // Both chains on one endpoint, whichever process drives each. + scrapeUntil( + "/metrics", + (body) => body.includes('chainId="1"') && body.includes('chainId="8453"') + ), + ]); + + // Only the supervisor is signalled, the way a process manager would. + indexer.kill("SIGINT"); + + expect({ + exitCode: await exit, + // Workers are named by the chains they drive. + runtimeWorkers: ["1", "8453"].map((worker) => + runtime.includes(`nodejs_heap_size_used_bytes{worker="${worker}"}`) + ), + metricsChains: [1, 8453].map((chainId) => + metrics.includes(`envio_progress_block{chainId="${chainId}"}`) + ), + }).toEqual({ + exitCode: 0, + runtimeWorkers: [true, true], + metricsChains: [true, true], + }); + } finally { + await stopIfRunning(indexer); + } + }); +}); diff --git a/packages/envio-tests/test/BelowHeadPollingPin_test.res b/packages/envio-tests/test/BelowHeadPollingPin_test.res index a17c255bce..b1e1d8cfd9 100644 --- a/packages/envio-tests/test/BelowHeadPollingPin_test.res +++ b/packages/envio-tests/test/BelowHeadPollingPin_test.res @@ -1,6 +1,7 @@ open Vitest let scenario = Scenario.make( + ~supervised=false, ~configYaml=` name: below-head-polling contracts: @@ -84,9 +85,7 @@ describe("PIN: chains keep indexing after entering the reorg threshold", () => { await MockSource.waitItemsQuery(chainWithThresholdWork) t.expect( - chainWithThresholdWork.getItemsOrThrowCalls->Array.map( - call => call.payload["fromBlock"], - ), + chainWithThresholdWork.getItemsOrThrowCalls->Array.map(call => call.payload["fromBlock"]), ~message="the zero-lag chain first fetches to its pre-threshold head", ).toEqual([1]) chainWithThresholdWork.resolveGetItemsOrThrow( @@ -101,9 +100,7 @@ describe("PIN: chains keep indexing after entering the reorg threshold", () => { // progress and lets it lead, which sidesteps the production ordering. await MockSource.waitItemsQuery(chainWithThresholdWork) t.expect( - chainWithThresholdWork.getItemsOrThrowCalls->Array.map( - call => call.payload["fromBlock"], - ), + chainWithThresholdWork.getItemsOrThrowCalls->Array.map(call => call.payload["fromBlock"]), ~message="the second response reaches the zero-lag chain's pre-threshold head", ).toEqual([401]) chainWithThresholdWork.resolveGetItemsOrThrow( @@ -154,9 +151,7 @@ describe("PIN: chains keep indexing after entering the reorg threshold", () => { // lets chain 100 claim the progress-alignment line before discovering that // it is WaitingForNewBlock, which clamps chain 1337 behind block 800. t.expect( - chainWithThresholdWork.getItemsOrThrowCalls->Array.map( - call => call.payload["fromBlock"], - ), + chainWithThresholdWork.getItemsOrThrowCalls->Array.map(call => call.payload["fromBlock"]), ~message="the below-head chain is not blocked by an unchanged source", ).toEqual([801]) diff --git a/packages/envio-tests/test/E2E_test.res b/packages/envio-tests/test/E2E_test.res index 1749c1ce3f..8742aeb12f 100644 --- a/packages/envio-tests/test/E2E_test.res +++ b/packages/envio-tests/test/E2E_test.res @@ -29,6 +29,7 @@ let chainYaml = (chainId, ~startBlock=1) => let makeScenario = (~name, ~rollback=true, ~chains) => Scenario.make( + ~supervised=false, ~configYaml=` name: ${name} rollback_on_reorg: ${rollback ? "true" : "false"}${contractsYaml}chains:${chains}`, @@ -45,6 +46,7 @@ let scenario = makeScenario(~name="e2e", ~chains=chainYaml(1337)) // Partition ids and the chain's range-cost budget follow the contract set, so // this scenario keeps the address-less contracts alongside the addressed ones. let partitionScenario = Scenario.make( + ~supervised=false, ~configYaml=` name: e2e-partitions rollback_on_reorg: true${contractsYaml} - name: SimpleNft diff --git a/packages/envio-tests/test/EnterReorgThreshold_test.res b/packages/envio-tests/test/EnterReorgThreshold_test.res index 4744c30efe..1dad106fb5 100644 --- a/packages/envio-tests/test/EnterReorgThreshold_test.res +++ b/packages/envio-tests/test/EnterReorgThreshold_test.res @@ -21,6 +21,7 @@ type Gravatar { // Two chains, each lagging maxReorgDepth (200) below head before the // threshold. Head starts at 1000, so the pre-threshold head is 800. let multichain = Scenario.make( + ~supervised=false, ~configYaml=` name: enter-reorg-threshold-multichain contracts: @@ -53,6 +54,7 @@ chains: ) let singleChain = Scenario.make( + ~supervised=false, ~configYaml=` name: enter-reorg-threshold-single chains: @@ -119,11 +121,7 @@ describe("PIN: multichain indexer enters the reorg threshold", () => { logIndex: 0, }, ) - chainA.resolveGetItemsOrThrow( - densitySeed, - ~latestFetchedBlockNumber=800, - ~knownHeight=1000, - ) + chainA.resolveGetItemsOrThrow(densitySeed, ~latestFetchedBlockNumber=800, ~knownHeight=1000) await indexer.getBatchWritePromise() // Chain A is now at its lagged head with an empty buffer — momentarily diff --git a/packages/envio-tests/test/HandlerChainInfo_test.res b/packages/envio-tests/test/HandlerChainInfo_test.res index b412e18e7f..58763b0c03 100644 --- a/packages/envio-tests/test/HandlerChainInfo_test.res +++ b/packages/envio-tests/test/HandlerChainInfo_test.res @@ -5,6 +5,7 @@ open Vitest // `isRealtime` only flips once every chain in the indexer is at its head. let scenario = Scenario.make( + ~supervised=false, ~configYaml=` name: handler-chain-info contracts: @@ -56,8 +57,7 @@ let recordChain = (~block, ~label): MockSource.itemMock => { }, } -let sortById = (rows: array) => - rows->Array.toSorted((a, b) => String.compare(a.id, b.id)) +let sortById = (rows: array) => rows->Array.toSorted((a, b) => String.compare(a.id, b.id)) describe("context.chain inside a handler", () => { scenario->Scenario.it( diff --git a/packages/envio-tests/test/IsolatedRollback_test.res b/packages/envio-tests/test/IsolatedRollback_test.res index d7c1fdc5d9..5b17bbff52 100644 --- a/packages/envio-tests/test/IsolatedRollback_test.res +++ b/packages/envio-tests/test/IsolatedRollback_test.res @@ -53,6 +53,7 @@ type Counter { ` let scenario = Scenario.make( + ~supervised=false, ~schema=perChainSchema, ~configYaml=makeConfigYaml(~name="isolated-rollback"), ) @@ -61,6 +62,7 @@ let scenario = Scenario.make( // about the mode needs a sibling, and the chain-id column every per-chain entity // carries is what its bounds join against. let singleChainScenario = Scenario.make( + ~supervised=false, ~schema=perChainSchema, ~configYaml=` name: single-chain-per-chain @@ -78,6 +80,7 @@ chains:${chainYaml(100)} // wrote can be what chain 100 read and overwrote, so its reorg has to take // every chain back with it. let crossChainScenario = Scenario.make( + ~supervised=false, ~schema=perChainSchema ++ ` type Total @crossChain { id: ID! @@ -91,12 +94,14 @@ type Total @crossChain { // next batch carries, and its current-state view has to resolve to the same // thing Postgres holds. let clickHouseScenario = Scenario.make( + ~supervised=false, ~schema=perChainSchema, ~configYaml=makeConfigYaml(~name="isolated-rollback-clickhouse"), ~unsupported=[{backend: #postgres, reason: "asserts against a ClickHouse server"}], ) let fullHistoryScenario = Scenario.make( + ~supervised=false, ~schema=perChainSchema, ~configYaml=makeConfigYaml( ~name="isolated-rollback-full-history", @@ -108,6 +113,7 @@ let fullHistoryScenario = Scenario.make( // same name the per-chain bounds relation gives its own, so the rollback // queries have to keep the two apart. let snakeCaseScenario = Scenario.make( + ~supervised=false, ~schema=perChainSchema, ~configYaml=makeConfigYaml( ~name="isolated-rollback-snake-case", diff --git a/packages/envio-tests/test/PerChainEntity_test.res b/packages/envio-tests/test/PerChainEntity_test.res index 8a2bce09bf..55ed155245 100644 --- a/packages/envio-tests/test/PerChainEntity_test.res +++ b/packages/envio-tests/test/PerChainEntity_test.res @@ -66,11 +66,12 @@ type GlobalCounter @crossChain { } ` -let scenario = Scenario.make(~schema, ~configYaml=makeConfigYaml()) +let scenario = Scenario.make(~supervised=false, ~schema, ~configYaml=makeConfigYaml()) // The two chains need a reorg threshold to roll back within, so this variant // sets one — `max_reorg_depth` is per chain, so it goes in the chain blocks. let rollbackScenario = Scenario.make( + ~supervised=false, ~schema, ~configYaml=makeConfigYaml(~rollback="\nrollback_on_reorg: true")->String.replaceAll( " start_block: 1\n", @@ -81,6 +82,7 @@ let rollbackScenario = Scenario.make( // The entity object and the getWhere filter key the chain by `chainId` while // the column is `chain_id`. let snakeCaseScenario = Scenario.make( + ~supervised=false, ~schema, ~configYaml=makeConfigYaml( ~storage=`storage: @@ -91,7 +93,7 @@ let snakeCaseScenario = Scenario.make( ) // The history prune is asserted through raw SQL against the history tables. -let pruneScenario = Scenario.make(~schema, ~configYaml=makeConfigYaml()) +let pruneScenario = Scenario.make(~supervised=false, ~schema, ~configYaml=makeConfigYaml()) let methods: array = [#getHeightOrThrow, #getItemsOrThrow] let reorgMethods: array = [#getHeightOrThrow, #getItemsOrThrow, #getBlockHashes] diff --git a/packages/envio-tests/test/PerChainHistoryPrune_test.res b/packages/envio-tests/test/PerChainHistoryPrune_test.res index ccca66e7d1..ae987d7700 100644 --- a/packages/envio-tests/test/PerChainHistoryPrune_test.res +++ b/packages/envio-tests/test/PerChainHistoryPrune_test.res @@ -57,6 +57,7 @@ chains:${chainYaml(100, ~startBlock=110, ~maxReorgDepth=15)}${chainYaml( ` let scenario = Scenario.make( + ~supervised=false, ~schema, ~configYaml=makeConfigYaml("per-chain-prune", ~laggingChainId=1337), ) @@ -72,12 +73,14 @@ type Total @crossChain { // One cross-chain entity couples the chains: a reorg on the chain furthest // behind can reach a row any chain wrote, so none may prune past its safe point. let crossChainScenario = Scenario.make( + ~supervised=false, ~schema=crossChainSchema, ~configYaml=makeConfigYaml("per-chain-prune-cross-chain", ~laggingChainId=1337), ) // The same, with the lagging chain visited first. let crossChainLaggingFirstScenario = Scenario.make( + ~supervised=false, ~schema=crossChainSchema, ~configYaml=makeConfigYaml("per-chain-prune-cross-chain-lagging-first", ~laggingChainId=5), ) @@ -87,6 +90,7 @@ let crossChainLaggingFirstScenario = Scenario.make( // id is not the bound — an idle chain's would hold every other chain's prune // back for as long as it stays idle. let crossChainZeroDepthScenario = Scenario.make( + ~supervised=false, ~schema=crossChainSchema, ~configYaml=` name: per-chain-prune-cross-chain-zero-depth @@ -108,6 +112,7 @@ chains:${chainYaml(100, ~startBlock=110, ~maxReorgDepth=15)}${chainYaml( // per chain. Three of them rather than two: a pair of bounds can be crossed and // still look right, while three cannot. let manyBoundsScenario = Scenario.make( + ~supervised=false, ~schema, ~configYaml=` name: per-chain-prune-many-bounds @@ -129,6 +134,7 @@ chains:${chainYaml(100, ~startBlock=110, ~maxReorgDepth=15)}${chainYaml( // entity to let a sibling's rollback reach its rows, it has no history to keep // and everything it has committed is safe to prune. let zeroDepthScenario = Scenario.make( + ~supervised=false, ~schema, ~configYaml=` name: per-chain-prune-zero-depth diff --git a/packages/envio-tests/test/ResumeFinalize_test.res b/packages/envio-tests/test/ResumeFinalize_test.res index 358e9c05cf..53425517bb 100644 --- a/packages/envio-tests/test/ResumeFinalize_test.res +++ b/packages/envio-tests/test/ResumeFinalize_test.res @@ -35,12 +35,14 @@ let gravatar1337 = "0x2B2f78c5BF6D9C12Ee1225D5F374aa91204580c3" let gravatar1 = "0x3B2f78c5BF6D9C12Ee1225D5F374aa91204580c3" let scenario = Scenario.make( + ~supervised=false, ~configYaml=` name: resume-finalize${contractsYaml}chains:${chainYaml(1337, gravatar1337, "")}`, ~schema, ) let endBlockScenario = Scenario.make( + ~supervised=false, ~configYaml=` name: resume-finalize-end-block${contractsYaml}chains:${chainYaml( 1337, @@ -51,6 +53,7 @@ name: resume-finalize-end-block${contractsYaml}chains:${chainYaml( ) let multichainScenario = Scenario.make( + ~supervised=false, ~configYaml=` name: resume-finalize-multichain${contractsYaml}chains:${chainYaml(1, gravatar1, "")}${chainYaml( 1337, diff --git a/packages/envio-tests/test/RollbackDiffCheckpointIds_test.res b/packages/envio-tests/test/RollbackDiffCheckpointIds_test.res index fb64c8da40..5628a18ba4 100644 --- a/packages/envio-tests/test/RollbackDiffCheckpointIds_test.res +++ b/packages/envio-tests/test/RollbackDiffCheckpointIds_test.res @@ -36,6 +36,7 @@ let chainYaml = chainId => // One cross-chain entity is what makes the checkpoint sequence shared, and what // makes a reorg on either chain roll both of them back. let scenario = Scenario.make( + ~supervised=false, ~schema=` type Counter { id: ID! @@ -146,8 +147,7 @@ describe("Rollback diff checkpoint ids", () => { await indexer.getBatchWritePromise() let highestCommitted = - (await indexer.queryCheckpoints()) - ->Array.reduce(0n, (highest, checkpoint) => + (await indexer.queryCheckpoints())->Array.reduce(0n, (highest, checkpoint) => checkpoint.id > highest ? checkpoint.id : highest ) @@ -169,7 +169,10 @@ describe("Rollback diff checkpoint ids", () => { ~message="the rollback's depth search to re-fetch the scanned block hashes", ) source1337.resolveGetBlockHashes( - [(100, "0x100"), (101, "0x101")]->Array.map(((blockNumber, blockHash)): BlockStore.inputBlock => { + [(100, "0x100"), (101, "0x101")]->Array.map((( + blockNumber, + blockHash, + )): BlockStore.inputBlock => { blockNumber, blockHash, blockTimestamp: blockNumber, @@ -186,9 +189,10 @@ describe("Rollback diff checkpoint ids", () => { ) await indexer.getBatchWritePromise() - let sorted = stagedDiffs->Array.toSorted((a, b) => - a.checkpointId < b.checkpointId ? -1. : a.checkpointId > b.checkpointId ? 1. : 0. - ) + let sorted = + stagedDiffs->Array.toSorted((a, b) => + a.checkpointId < b.checkpointId ? -1. : a.checkpointId > b.checkpointId ? 1. : 0. + ) t.expect( ( stagedDiffs->Array.length, diff --git a/packages/envio-tests/test/Rollback_test.res b/packages/envio-tests/test/Rollback_test.res index a174d7bd23..d843191ad7 100644 --- a/packages/envio-tests/test/Rollback_test.res +++ b/packages/envio-tests/test/Rollback_test.res @@ -51,6 +51,7 @@ indexer.onEvent({ contract: "SimpleNft", event: "Transfer" }, async () => {}); let makeScenario = (~name, ~chains, ~extra="") => Scenario.make( + ~supervised=false, ~configYaml=` name: ${name} rollback_on_reorg: true${extra}${contractsYaml}chains:${chains}`, @@ -776,9 +777,11 @@ describe("E2E rollback tests", () => { // registration at suite scope would also collect the rollbacks every // other case in this file fires. let rollbackCommitCalls = [] - let unregister = RollbackCommit.register(async (args: RollbackCommit.args) => { - rollbackCommitCalls->Array.push(args) - }) + let unregister = RollbackCommit.register( + async (args: RollbackCommit.args) => { + rollbackCommitCalls->Array.push(args) + }, + ) let sourceMock = source(1337) await Utils.delay(0) @@ -2369,7 +2372,6 @@ describe("E2E rollback tests", () => { // so getHighestBlockBelowThreshold = 300 - 200 = 100 is used directly. // Wait for the SetRollbackState tasks (NextQuery, ProcessEventBatch) to be scheduled - sourceMock1337.resolveGetItemsOrThrow( [], ~prevRangeLastBlock={ @@ -3026,7 +3028,7 @@ describe("E2E rollback tests", () => { await storage.writeBatch( ~batch, ~rollback, - ~config, + ~config, ~allEntities, ~updatedEffectsCache, ~updatedEntities, diff --git a/packages/envio-tests/test/SchemaIndexes_test.res b/packages/envio-tests/test/SchemaIndexes_test.res index 00a9ec8309..1d3137df3a 100644 --- a/packages/envio-tests/test/SchemaIndexes_test.res +++ b/packages/envio-tests/test/SchemaIndexes_test.res @@ -20,7 +20,8 @@ type C { } ` -let chainYaml = (chainId, address) => ` +let chainYaml = (chainId, address) => + ` - id: ${chainId->Int.toString} rpc: url: https://rpc${chainId->Int.toString}.example.test @@ -39,6 +40,7 @@ contracts: ` let scenario = Scenario.make( + ~supervised=false, ~configYaml=` name: schema-indexes${contractsYaml}chains:${chainYaml( 1337, @@ -51,6 +53,7 @@ name: schema-indexes${contractsYaml}chains:${chainYaml( // up once progress sits at the head, so the deferred indexes are owed then, not // at the unreachable end block. let unreachableEndBlockScenario = Scenario.make( + ~supervised=false, ~configYaml=` name: schema-indexes-unreachable-end${contractsYaml}chains: - id: 1337 @@ -69,6 +72,7 @@ name: schema-indexes-unreachable-end${contractsYaml}chains: // A `start_block` past the head: the chain is at its head from the first moment // and never has a batch to process, so nothing ever writes its progress row. let aheadOfHeadScenario = Scenario.make( + ~supervised=false, ~configYaml=` name: schema-indexes-ahead-of-head${contractsYaml}chains: - id: 1337 @@ -84,6 +88,7 @@ name: schema-indexes-ahead-of-head${contractsYaml}chains: ) let multichainScenario = Scenario.make( + ~supervised=false, ~configYaml=` name: schema-indexes-multichain${contractsYaml}chains:${chainYaml( 100, @@ -310,8 +315,8 @@ describe("Deferred schema indexes", () => { await indexer.waitUntilReady() t.expect(( - (await findIndexes(~sql, ~tableName="A", ~columns=["b_id"], ~pgSchema))->Array.map( - entry => entry.name, + (await findIndexes(~sql, ~tableName="A", ~columns=["b_id"], ~pgSchema))->Array.map(entry => + entry.name ), await readyAtByChainId(~sql, ~pgSchema), )).toEqual(([aBIdIndexName], [(ChainId.fromInt(1337), true)])) @@ -394,7 +399,10 @@ describe("Deferred schema indexes", () => { }, }, async (~t, ~indexer, ~source) => { - let finalizeCalls = multichainFinalizeCalls + // Counted from where this pass began: a multichain scenario runs a second + // time behind the barrier, over the same counter. + let before = multichainFinalizeCalls.contents + let finalizeCalls = () => multichainFinalizeCalls.contents - before let chainA = source(100) let chainB = source(1337) let {sql, pgSchema} = indexer.pg @@ -409,7 +417,7 @@ describe("Deferred schema indexes", () => { await indexer.getBatchWritePromise() t.expect( - (finalizeCalls.contents, await readyAtByChainId(~sql, ~pgSchema)), + (finalizeCalls(), await readyAtByChainId(~sql, ~pgSchema)), ~message="Chain A is at its head, but chain B is still backfilling", ).toEqual((0, [(ChainId.fromInt(100), false), (ChainId.fromInt(1337), false)])) @@ -421,7 +429,7 @@ describe("Deferred schema indexes", () => { let times = readyAtTimes->Array.map(((_, readyAt)) => readyAt) t.expect( ( - finalizeCalls.contents, + finalizeCalls(), readyAtTimes->Array.map(((id, _)) => id), times->Array.every(Option.isSome), times->Array.get(0) == times->Array.get(1), @@ -471,7 +479,11 @@ describe("Deferred schema indexes", () => { aIndexes->Array.map(entry => (entry->isValid, entry->isPartial, entry->predicate)), ), ~message="The conflicting index is left alone and A(b_id) still gets a usable index of its own", - ).toEqual(([{value: "1", labels: dict{"chainId": "1337"}}], ["A_b_id"], [(true, false, None)])) + ).toEqual(( + [{value: "1", labels: dict{"chainId": "1337"}}], + ["A_b_id"], + [(true, false, None)], + )) t.expect( aIndexes->Array.map(entry => entry.name), @@ -552,9 +564,9 @@ describe("Automatic getWhere indexes", () => { t.expect( ( matched.contents, - (await findIndexes(~sql, ~tableName="A", ~columns=[optionalColumn], ~pgSchema))->Array.map( - entry => (entry.name, entry->isValid), - ), + ( + await findIndexes(~sql, ~tableName="A", ~columns=[optionalColumn], ~pgSchema) + )->Array.map(entry => (entry.name, entry->isValid)), await findIndexes(~sql, ~tableName="A", ~columns=["b_id"], ~pgSchema), await readyAtByChainId(~sql, ~pgSchema), ), @@ -580,12 +592,9 @@ describe("Automatic getWhere indexes", () => { t.expect( ( - (await findIndexes( - ~sql, - ~tableName="A", - ~columns=[optionalColumn], - ~pgSchema, - ))->Array.length, + ( + await findIndexes(~sql, ~tableName="A", ~columns=[optionalColumn], ~pgSchema) + )->Array.length, (await findIndexes(~sql, ~tableName="A", ~columns=["b_id"], ~pgSchema))->Array.map( entry => entry.name, ), diff --git a/packages/envio-tests/test/SupervisedRealtime_test.res b/packages/envio-tests/test/SupervisedRealtime_test.res new file mode 100644 index 0000000000..204c92e2be --- /dev/null +++ b/packages/envio-tests/test/SupervisedRealtime_test.res @@ -0,0 +1,220 @@ +open Vitest + +// An indexer switches to realtime as a whole: every chain enters the reorg +// threshold together and every chain is stamped ready at one instant. A split +// run's processes each see only their own chains, so a supervised worker holds +// those transitions until the supervisor says every chain in the run has +// arrived. + +let schema = ` +type A { + id: ID! +} +` + +let chainYaml = (chainId, address) => + ` + - id: ${chainId->Int.toString} + rpc: + url: https://rpc${chainId->Int.toString}.example.test + for: sync + start_block: 1 + contracts: + - name: Gravatar + address: "${address}" +` + +let endBlockScenario = Scenario.make( + ~configYaml=` +name: supervised-realtime-end-block +disable_default_cross_chain: true +contracts: + - name: Gravatar + events: + - event: "TestEvent()" +chains: + - id: 1 + rpc: + url: https://rpc1.example.test + for: sync + start_block: 1 + end_block: 100 + contracts: + - name: Gravatar + address: "0x2B2f78c5BF6D9C12Ee1225D5F374aa91204580c3"`, + ~schema, +) + +// Rollback on, which is what gives a chain a reorg depth: pre-threshold its +// fetch frontier is capped at the safe block, so progress can never reach the +// head until the chain enters the threshold. +let rollbackScenario = Scenario.make( + ~configYaml=` +name: supervised-realtime-rollback +rollback_on_reorg: true +disable_default_cross_chain: true +contracts: + - name: Gravatar + events: + - event: "TestEvent()" +chains:${chainYaml(1, "0x2B2f78c5BF6D9C12Ee1225D5F374aa91204580c3")}${chainYaml( + 137, + "0x3B2f78c5BF6D9C12Ee1225D5F374aa91204580c3", + )}`, + ~schema, +) + +let scenario = Scenario.make( + ~configYaml=` +name: supervised-realtime +disable_default_cross_chain: true +contracts: + - name: Gravatar + events: + - event: "TestEvent()" +chains:${chainYaml(1, "0x2B2f78c5BF6D9C12Ee1225D5F374aa91204580c3")}${chainYaml( + 137, + "0x3B2f78c5BF6D9C12Ee1225D5F374aa91204580c3", + )}`, + ~schema, +) + +let readyAtByChainId = async (~sql, ~pgSchema) => { + let rows: array<{ + "id": ChainId.t, + "ready_at": Null.t, + }> = await sql->Postgres.unsafe( + `SELECT "id", "ready_at" FROM "${pgSchema}"."envio_chains" ORDER BY "id";`, + ) + rows->Array.map(row => (row["id"]->ChainId.toString, row["ready_at"]->Null.toOption)) +} + +let catchUp = (~source: MockSource.t) => { + source.resolveGetHeightOrThrow(100) + source.resolveGetItemsOrThrow([], ~latestFetchedBlockNumber=100) +} + +describe("A supervised worker", () => { + scenario->Scenario.it( + "Waits for the run before going realtime, then stamps every chain at one instant", + ~sources=[{chain: 1}, {chain: 137}], + ~holdRealtime=true, + async (~t, ~indexer, ~source) => { + let {sql, pgSchema} = indexer.pg + catchUp(~source=source(1)) + catchUp(~source=source(137)) + await indexer.waitUntilIdle() + + t.expect( + (await readyAtByChainId(~sql, ~pgSchema), await indexer.metric("envio_progress_ready")), + ~message="Both chains are at the head, but the run has not said so", + ).toEqual(( + [("1", None), ("137", None)], + [{value: "0", labels: dict{"chainId": "1"}}, {value: "0", labels: dict{"chainId": "137"}}], + )) + + indexer.releaseRealtime() + await indexer.waitUntilReady() + + let readyAt = await readyAtByChainId(~sql, ~pgSchema) + let stamps = readyAt->Array.filterMap(((_, at)) => at->Option.map(Date.getTime)) + t.expect( + ( + readyAt->Array.map(((chainId, _)) => chainId), + stamps->Array.length, + stamps->Set.fromArray->Set.size, + ), + ~message="The release stamps every chain, and one caught-up indexer is one instant", + ).toEqual((["1", "137"], 2, 1)) + }, + ) +}) + +describe("A supervised worker at its end block", () => { + let exited = ref(false) + + endBlockScenario->Scenario.it( + "Stays for the run rather than exiting with the indexes it still owes", + ~sources=[{chain: 1}], + ~holdRealtime=true, + async (~t, ~indexer, ~source) => { + let {sql, pgSchema} = indexer.pg + catchUp(~source=source(1)) + await indexer.getBatchWritePromise() + await indexer.waitUntilIdle() + + t.expect( + (exited.contents, await readyAtByChainId(~sql, ~pgSchema)), + ~message="Its chain is done, but it still owes the schema the indexes it deferred", + ).toEqual((false, [("1", None)])) + + indexer.releaseRealtime() + await indexer.waitUntilReady() + + t.expect( + (await readyAtByChainId(~sql, ~pgSchema))->Array.map(((chainId, readyAt)) => ( + chainId, + readyAt->Option.isSome, + )), + ~message="Released, it finalizes and stamps its chain", + ).toEqual([("1", true)]) + }, + ~onExit=() => exited := true, + ) +}) + +describe("A run whose chains have a reorg depth", () => { + // The transition the barrier holds is the one that lifts the pre-threshold + // lag. Held until its chains reach the head, a run would be waiting on + // progress only that transition makes reachable — so what a worker reports + // as arrived has to be the safe block, as far as it can fetch until then. + rollbackScenario->Scenario.it( + "Enters the reorg threshold once every chain is at its safe block", + ~sources=[{chain: 1}, {chain: 137}], + async (~t, ~indexer, ~source) => { + await Scenario.enterReorgThreshold(~t, ~indexer, ~source=source(1)) + await Scenario.enterReorgThreshold(~t, ~indexer, ~source=source(137)) + + t.expect( + await indexer.metric("envio_reorg_threshold"), + ~message="Both chains have fetched the whole finalized range", + ).toEqual([{value: "1", labels: Dict.make()}]) + }, + ) +}) + +describe("A supervised worker on a chain with a reorg depth", () => { + // The transition being held is the one that lifts the pre-threshold lag, so a + // run held until its chains reach the head would be waiting on progress that + // only the transition itself makes reachable. + rollbackScenario->Scenario.it( + "Holds the transition that would let it fetch past the safe block", + ~sources=[{chain: 1}, {chain: 137}], + ~holdRealtime=true, + async (~t, ~indexer, ~source) => { + let {sql, pgSchema} = indexer.pg + // Each chain fetches the whole finalized range, which is all it may fetch + // until the indexer enters the reorg threshold. + let catchUpToSafeBlock = (~source: MockSource.t) => { + source.resolveGetHeightOrThrow(300) + source.resolveGetItemsOrThrow([], ~latestFetchedBlockNumber=100) + } + catchUpToSafeBlock(~source=source(1)) + catchUpToSafeBlock(~source=source(137)) + await indexer.waitUntilIdle() + + t.expect( + (await indexer.metric("envio_reorg_threshold"), await readyAtByChainId(~sql, ~pgSchema)), + ~message="Both chains are as far as they can fetch, and the run has not said so", + ).toEqual(([{value: "0", labels: Dict.make()}], [("1", None), ("137", None)])) + + indexer.releaseRealtime() + await indexer.waitUntilIdle() + + t.expect( + await indexer.metric("envio_reorg_threshold"), + ~message="Released, the run enters the threshold and the rest opens up", + ).toEqual([{value: "1", labels: Dict.make()}]) + }, + ) +}) diff --git a/packages/envio-tests/test/SupervisorFork_test.res b/packages/envio-tests/test/SupervisorFork_test.res new file mode 100644 index 0000000000..e7b296c0f5 --- /dev/null +++ b/packages/envio-tests/test/SupervisorFork_test.res @@ -0,0 +1,301 @@ +open Vitest + +// What the fixture worker reports back in place of a metrics snapshot: the +// environment its supervisor handed it, which is the whole of what a worker is +// told before it starts. +type fixtureReport = { + workerConfig: string, + maxConnections: string, + bufferSize: string, + objectsTarget: string, + logFile: string, + startTime: Date.t, + hasArrivedAtHead: bool, +} + +let fixturePath = `${NodeJs.Process.cwd()}/test/helpers/fakeWorker.mjs` + +let forkFixture = ( + ~chainIds, + ~maxConnections=2, + ~workerIndex=0, + ~workerCount=2, + ~holdRealtime=false, + ~isDev=false, + ~pipeOutput=false, + ~onOutput=?, + ~onErrorOutput=?, + ~onSnapshot=?, +) => + Supervisor.fork( + {chainIds: chainIds->Array.map(ChainId.fromInt), maxConnections}, + ~workerIndex, + ~workerCount, + ~holdRealtime, + ~isDev, + ~entryPath=fixturePath, + ~pipeOutput, + ~onOutput?, + ~onErrorOutput?, + ~onSnapshot?, + ) + +describe("Supervisor.fork", () => { + Async.it("Hands a worker its chains, its budget share, and its own log file", async t => { + let running = forkFixture( + ~chainIds=[1, 137], + ~maxConnections=3, + ~workerIndex=1, + ~workerCount=4, + ~holdRealtime=true, + ~isDev=true, + ) + + let report = await Promise.make( + (resolve, _) => + running.child->NodeJs.ChildProcess.Child.onMessage( + message => + switch message { + | Worker.Snapshot({metrics}) => + resolve(metrics->(Utils.magic: Metrics.t => fixtureReport)) + }, + ), + ) + running.child->NodeJs.ChildProcess.Child.kill("SIGTERM")->ignore + + t.expect(report).toStrictEqual({ + // Everything the supervisor decided, in the environment: a worker needs it + // before it can load its own config, so it can't arrive as a message. The + // project's files say nothing about which command started the run, which + // is why `isDev` is among them. + workerConfig: `{"chainIds":[1,137],"holdRealtime":true,"isDev":true}`, + maxConnections: "3", + // The run's memory budgets are the whole indexer's, so a worker gets a + // share rather than the whole of each. + bufferSize: (CrossChainState.calculateTargetBufferSize() / 4)->Int.toString, + objectsTarget: (Env.inMemoryObjectsTarget->Float.toInt / 4)->Int.toString, + logFile: Supervisor.logFilePath(~workerIndex=1), + // Proof the channel clones rather than stringifies: a JSON round trip + // would have turned this into a string. + startTime: Date.fromTime(1700000000000.), + hasArrivedAtHead: false, + }) + }) +}) + +describe("Supervisor.awaitExit", () => { + let outcome = async group => + switch await group->Supervisor.awaitExit { + | outcome => Ok(outcome) + | exception _ => Error() + } + + Async.it("Reports a group whose every worker finished on its own", async t => { + NodeJs.Process.process.env->Dict.set("FAKE_WORKER", "succeed") + let group: Supervisor.group = { + running: [forkFixture(~chainIds=[1]), forkFixture(~chainIds=[137])], + stopping: false, + holdingRealtime: false, + } + + t.expect(await outcome(group)).toStrictEqual(Ok(Supervisor.Finished)) + }) + + Async.it( + "Reports a group its supervisor took down as stopped, whatever the exit codes", + async t => { + NodeJs.Process.process.env->Dict.set("FAKE_WORKER", "linger") + let group: Supervisor.group = { + running: [forkFixture(~chainIds=[1]), forkFixture(~chainIds=[137])], + stopping: false, + holdingRealtime: false, + } + group->Supervisor.stop + + t.expect(await outcome(group)).toStrictEqual(Ok(Supervisor.Stopped)) + }, + ) + + // A process manager that signals the whole group reaches the workers itself, + // so they exit on a SIGTERM the supervisor has not passed on and may not even + // have handled yet. Reading that as a worker dying would fail every clean + // shutdown under systemd's default kill mode. + Async.it("Takes a worker signalled from outside as the run being stopped", async t => { + NodeJs.Process.process.env->Dict.set("FAKE_WORKER", "linger") + let signalled = forkFixture(~chainIds=[1]) + let sibling = forkFixture(~chainIds=[137]) + let group: Supervisor.group = { + running: [signalled, sibling], + stopping: false, + holdingRealtime: false, + } + let ended = outcome(group) + signalled.child->NodeJs.ChildProcess.Child.kill("SIGTERM")->ignore + + // The sibling still went down with it: one worker short leaves its chains + // unindexed. + t.expect((await ended, group.stopping)).toStrictEqual((Ok(Supervisor.Stopped), true)) + }) + + Async.it("Stops the group and fails the run when one worker dies", async t => { + NodeJs.Process.process.env->Dict.set("FAKE_WORKER", "fail") + let failing = forkFixture(~chainIds=[1]) + NodeJs.Process.process.env->Dict.set("FAKE_WORKER", "linger") + let lingering = forkFixture(~chainIds=[137]) + let group: Supervisor.group = { + running: [failing, lingering], + stopping: false, + holdingRealtime: false, + } + + // The survivor was taken down rather than left indexing half a schema. + t.expect((await outcome(group), group.stopping, lingering.settled)).toStrictEqual(( + Error(), + true, + true, + )) + }) +}) + +describe("Supervisor.readLines", () => { + it("Holds a half line until the chunk that finishes it, or the stream ends", t => { + let lines = [] + let (read, flush) = Supervisor.readLines(~onLine=line => lines->Array.push(line)->ignore) + ["a line\nand ", "half of ", "another\nlast\n", "no newline here"]->Array.forEach(read) + flush() + // Nothing is left to flush twice. + flush() + + t.expect(lines).toStrictEqual(["a line", "and half of another", "last", "no newline here"]) + }) +}) + +describe("Supervisor.fork output", () => { + // A worker writing straight to the terminal tears the frame its supervisor + // draws: ink only knows about the lines its own process logs. Each line keeps + // the stream it was written to, so redirecting the run's stderr still catches + // what its workers wrote there. Sorted, since stdout and stderr are two pipes + // and neither waits for the other. + Async.it("Hands the supervisor every line a worker writes, on its own stream", async t => { + NodeJs.Process.process.env->Dict.set("FAKE_WORKER", "print") + let lines = [] + let group: Supervisor.group = { + running: [ + forkFixture( + ~chainIds=[1], + ~pipeOutput=true, + ~onOutput=line => lines->Array.push(("stdout", line))->ignore, + ~onErrorOutput=line => lines->Array.push(("stderr", line))->ignore, + ), + ], + stopping: false, + holdingRealtime: false, + } + let _ = await group->Supervisor.awaitExit + + t.expect(lines->Array.toSorted(((_, a), (_, b)) => String.compare(a, b))).toStrictEqual([ + ("stdout", "first line"), + ("stderr", "from stderr"), + ("stdout", "second line"), + ]) + }) +}) + +describe("Supervisor.isRunAtHead", () => { + let untilReported = async (running: array) => { + let rec until = async deadline => + if !(running->Array.every(r => r.snapshot->Option.isSome)) && Date.now() < deadline { + await Utils.delay(10) + await until(deadline) + } + await until(Date.now() +. 3000.) + } + + let untilGone = async (r: Supervisor.running) => { + let rec until = async deadline => + if r.child->NodeJs.ChildProcess.Child.connected && Date.now() < deadline { + await Utils.delay(10) + await until(deadline) + } + await until(Date.now() +. 3000.) + } + + let arrivingFixture = (~chainIds, ~mode) => { + NodeJs.Process.process.env->Dict.set("FAKE_WORKER", mode) + NodeJs.Process.process.env->Dict.set("FAKE_WORKER_ARRIVED", "1") + forkFixture(~chainIds) + } + + Async.it("Holds the run until every worker has arrived", async t => { + let arrived = arrivingFixture(~chainIds=[1], ~mode="linger") + NodeJs.Process.process.env->Dict.set("FAKE_WORKER_ARRIVED", "0") + let backfilling = forkFixture(~chainIds=[137]) + let group: Supervisor.group = { + running: [arrived, backfilling], + stopping: false, + holdingRealtime: false, + } + await untilReported(group.running) + + // One worker still backfilling speaks for the whole run, and a worker that + // has yet to report drives chains nobody can see. + let readings = ( + group.running->Supervisor.isRunAtHead, + [arrived]->Supervisor.isRunAtHead, + []->Supervisor.isRunAtHead, + ) + group->Supervisor.stop + let _ = await group->Supervisor.awaitExit + + t.expect(readings).toStrictEqual((false, true, false)) + }) + + // Every worker said it had arrived, and then one of them was gone. Its + // snapshot outlives it, so a run read from the snapshots alone still looks + // whole, and the release would be sent into a channel Node had already + // closed — which comes back as the error a supervisor reports as a worker + // failing to start. + Async.it("Never releases a run a worker has left", async t => { + let lingering = arrivingFixture(~chainIds=[1], ~mode="linger") + let leaving = arrivingFixture(~chainIds=[137], ~mode="succeed-later") + let group: Supervisor.group = { + running: [lingering, leaving], + stopping: false, + holdingRealtime: false, + } + await untilReported(group.running) + await untilGone(leaving) + + let readings = ( + group.running->Supervisor.isRunAtHead, + // Both snapshots are still there, and both of them still say arrived. + group.running->Array.filterMap(r => r.snapshot)->Array.length, + ) + group->Supervisor.stop + await untilGone(lingering) + + t.expect(readings).toStrictEqual((false, 2)) + }) + + // The release rides the workers' reports rather than a clock: the run opens on + // the report that completes it, and these workers exit only once released. + Async.it("Releases every worker on the report that completes the run", async t => { + NodeJs.Process.process.env->Dict.set("FAKE_WORKER", "await-release") + NodeJs.Process.process.env->Dict.set("FAKE_WORKER_ARRIVED", "1") + let group: Supervisor.group = {running: [], stopping: false, holdingRealtime: true} + group.running = + [[1], [137]]->Array.map( + chainIds => + forkFixture( + ~chainIds, + ~holdRealtime=true, + ~onSnapshot=() => group->Supervisor.releaseIfAtHead, + ), + ) + + t.expect((await group->Supervisor.awaitExit, group.holdingRealtime)).toStrictEqual(( + Supervisor.Finished, + false, + )) + }) +}) diff --git a/packages/envio-tests/test/ZeroReorgDepthHistory_test.res b/packages/envio-tests/test/ZeroReorgDepthHistory_test.res index 0efc3cb7ed..3a9a927afc 100644 --- a/packages/envio-tests/test/ZeroReorgDepthHistory_test.res +++ b/packages/envio-tests/test/ZeroReorgDepthHistory_test.res @@ -38,6 +38,7 @@ let chainYaml = (chainId, ~maxReorgDepth) => // One chain, and the default cross-chain entities that make its checkpoint // sequence a shared one. let singleChainScenario = Scenario.make( + ~supervised=false, ~schema, ~configYaml=` name: zero-reorg-depth-history @@ -53,6 +54,7 @@ chains:${chainYaml(100, ~maxReorgDepth=0)} // No cross-chain entity, so each chain counts its own checkpoints and only the // chain that can be rolled back keeps any. let perChainScenario = Scenario.make( + ~supervised=false, ~schema, ~configYaml=` name: zero-reorg-depth-history-per-chain diff --git a/packages/envio-tests/test/helpers/IndexerRunner.res b/packages/envio-tests/test/helpers/IndexerRunner.res index 77ec72d50a..5b6e5ab76a 100644 --- a/packages/envio-tests/test/helpers/IndexerRunner.res +++ b/packages/envio-tests/test/helpers/IndexerRunner.res @@ -59,8 +59,16 @@ type rec t = { // `~chains` resumes the same schema driving only those chains, the way // `envio start --chain` does. The chains left out keep their stored state. restart: (~chains: array=?, unit) => promise, + // Stands in for the supervisor's go-ahead in a run started with + // `~holdRealtime`. + releaseRealtime: unit => unit, } +// How often the stand-in supervisor of a supervised pass asks whether the run +// may go realtime. Short enough that the run gets there in the same tick a test +// would otherwise see it. +%%private(let releaseCheckIntervalMillis = 1) + let entityConfigByName = (config: Config.t, name): Internal.entityConfig => config.userEntitiesByName->Dict.get(name)->Option.getOrThrow @@ -77,6 +85,15 @@ let run = async ( ~backend: backend=selectedBackend, ~reducedPollingInterval=?, ~targetBufferSize=?, + // Runs the indexer the way a supervised worker runs: it waits to be released + // before entering the reorg threshold or switching to realtime. + ~holdRealtime=false, + // Runs it behind the same barrier with a stand-in supervisor releasing it on + // the real predicate, so a scenario exercises the held path without a test + // having to drive it. A split run can't be simulated here — the sources a + // test drives are objects in this process, which a forked worker wouldn't + // have — but the hold, the predicate and the release are the production ones. + ~superviseRun=false, ~onError=?, ~onExit=?, ~mapStorage: Persistence.storage => Persistence.storage=storage => storage, @@ -163,11 +180,25 @@ let run = async ( ~targetBufferSize?, ~isDevelopmentMode=false, ~shouldUseTui=false, + ~holdRealtime={holdRealtime || superviseRun}, ~onError, ~onExit?, ) state->IndexerLoop.start + // Only when the test didn't ask for the hold itself: one that did is + // testing the barrier and owns its own release. + let releaseCheck = ref(None) + if superviseRun && !holdRealtime { + releaseCheck := Some(setInterval(() => + if state->IndexerState.hasArrivedAtHead { + releaseCheck.contents->Option.forEach(clearInterval) + releaseCheck := None + state->IndexerState.releaseRealtime + } + , releaseCheckIntervalMillis)) + } + // Persist before stopping, else a resumed indexer loses uncommitted state, // then let any in-flight batch or write settle so nothing from this run // lands on the database afterwards. @@ -180,6 +211,8 @@ let run = async ( | None => let promise = ( async () => { + releaseCheck.contents->Option.forEach(clearInterval) + releaseCheck := None await state->Writing.flush state->IndexerState.stop // Tests deliberately leave handlers that never resolve, which pins @@ -275,14 +308,17 @@ let run = async ( let isIdle = !(state->IndexerState.isProcessing) && state->IndexerState.writeFiber->Option.isNone && - Frontier.equals(state->IndexerState.committedFrontier, state->IndexerState.processedFrontier) + Frontier.equals( + state->IndexerState.committedFrontier, + state->IndexerState.processedFrontier, + ) // Catching up hands off to the FinalizingIndexes phase, which is // where readiness is decided — so a batch isn't settled until that // phase is over. The idle fallback below still bounds the wait. if ( before < state->IndexerState.processedBatchesCount && - !(state->IndexerState.isFinalizingIndexes) + !(state->IndexerState.shouldFinalizeIndexes) ) { () } else if isIdle && idleChecks.contents >= 5 { @@ -320,8 +356,11 @@ let run = async ( settled := if ( !(state->IndexerState.isProcessing) && state->IndexerState.writeFiber->Option.isNone && - !(state->IndexerState.isFinalizingIndexes) && - Frontier.equals(state->IndexerState.committedFrontier, state->IndexerState.processedFrontier) + !(state->IndexerState.shouldFinalizeIndexes) && + Frontier.equals( + state->IndexerState.committedFrontier, + state->IndexerState.processedFrontier, + ) ) { settled.contents + 1 } else { @@ -333,6 +372,7 @@ let run = async ( JsError.throwWithMessage("Timed out waiting for the indexer to go idle") } }, + releaseRealtime: () => state->IndexerState.releaseRealtime, waitUntilReady: async () => { let isReady = () => state diff --git a/packages/envio-tests/test/helpers/Scenario.res b/packages/envio-tests/test/helpers/Scenario.res index c1c849e11f..d7c220fdfe 100644 --- a/packages/envio-tests/test/helpers/Scenario.res +++ b/packages/envio-tests/test/helpers/Scenario.res @@ -15,6 +15,11 @@ type t = { handlers: option, unsupported: array, site: string, + // Whether a multichain scenario also runs behind the barrier. Off for a + // scenario that stages its chains at different heights and drives the + // reorg-threshold transition itself: the hold deferring that transition is + // the premise such a body sets up being taken away. + supervised: bool, } type sourceMock = { @@ -46,7 +51,15 @@ let withClickHouseStorage = configYaml => configYaml ++ "\nstorage:\n postgres:\n default: true\n clickhouse:\n default: true\n" } -let make = (~configYaml, ~schema=?, ~env=?, ~files=?, ~handlers=?, ~unsupported=[]): t => { +let make = ( + ~configYaml, + ~schema=?, + ~env=?, + ~files=?, + ~handlers=?, + ~unsupported=[], + ~supervised=true, +): t => { let isUnsupported = unsupported->Array.some(({backend}) => backend === IndexerRunner.selectedBackend) @@ -90,6 +103,7 @@ let make = (~configYaml, ~schema=?, ~env=?, ~files=?, ~handlers=?, ~unsupported= handlers, unsupported, site, + supervised, } } @@ -168,6 +182,8 @@ let run = async ( ~maxAddrInPartition=?, ~clientFilterAddressThreshold=?, ~reorgThresholdReadyTolerance=?, + ~holdRealtime=?, + ~superviseRun=?, ~onError=?, ~onExit=?, ~mapStorage=?, @@ -237,6 +253,8 @@ let run = async ( }), ~reducedPollingInterval?, ~targetBufferSize?, + ~holdRealtime?, + ~superviseRun?, ~onError?, ~onExit?, ~mapStorage?, @@ -267,6 +285,11 @@ let it = ( ~maxAddrInPartition=?, ~clientFilterAddressThreshold=?, ~reorgThresholdReadyTolerance=?, + ~holdRealtime=?, + // Runs a multichain scenario a second time behind the barrier a supervised + // worker runs behind, so the scenario covers the held path as well as the + // plain one. `Scenario.make(~supervised=false)` opts a whole scenario out. + ~supervised=true, ~onError=?, ~onExit=?, ~mapStorage=?, @@ -285,22 +308,36 @@ let it = ( async _ => (), ) | None => - let runBody = async (t: Vitest.testContext) => - await scenario->run( - ~sources, - ~reducedPollingInterval?, - ~targetBufferSize?, - ~maxAddrInPartition?, - ~clientFilterAddressThreshold?, - ~reorgThresholdReadyTolerance?, - ~onError?, - ~onExit?, - ~mapStorage?, - (~indexer, ~source) => body(~t, ~indexer, ~source), - ) - switch retry { - | Some(retry) => Vitest.Async.itWithOptions(name, {retry, ?timeout}, runBody) - | None => Vitest.Async.it(name, runBody, ~timeout?) + let runBody = (~superviseRun) => + async (t: Vitest.testContext) => + await scenario->run( + ~sources, + ~reducedPollingInterval?, + ~targetBufferSize?, + ~maxAddrInPartition?, + ~clientFilterAddressThreshold?, + ~reorgThresholdReadyTolerance?, + ~holdRealtime?, + ~superviseRun, + ~onError?, + ~onExit?, + ~mapStorage?, + (~indexer, ~source) => body(~t, ~indexer, ~source), + ) + let register = (name, ~superviseRun) => + switch retry { + | Some(retry) => Vitest.Async.itWithOptions(name, {retry, ?timeout}, runBody(~superviseRun)) + | None => Vitest.Async.it(name, runBody(~superviseRun), ~timeout?) + } + + register(name, ~superviseRun=false) + + // One chain is a run whose every chain is its own process's already, so the + // barrier has nothing to hold: only a multichain scenario says anything new. + if ( + supervised && scenario.supervised && scenario.config.chainMap->ChainMap.keys->Array.length > 1 + ) { + register(`${name} [supervised]`, ~superviseRun=true) } } } diff --git a/packages/envio-tests/test/helpers/TestChainMetrics.res b/packages/envio-tests/test/helpers/TestChainMetrics.res index cae0a17930..3fe19a784d 100644 --- a/packages/envio-tests/test/helpers/TestChainMetrics.res +++ b/packages/envio-tests/test/helpers/TestChainMetrics.res @@ -34,20 +34,23 @@ let registrationsByChainId: HandlerRegister.registrationsByChainId = { let caughtUpAt = Date.fromTime(1700000000000.) -let make = ( +let makeChainState = ( ~endBlock=None, ~progressBlockNumber, ~firstEventBlockNumber, ~timestampCaughtUpToHeadOrEndblock=None, ~sourceBlockNumber=1000, -): Metrics.chainMetrics => + ~isInReorgThreshold=false, + ~maxReorgDepth=200, + ~config=TestConfig.default, +): ChainState.t => ChainState.makeFromDbState( chainConfig, ~resumedChainState={ id: chainId, startBlock: 100, endBlock, - maxReorgDepth: 200, + maxReorgDepth, progressBlockNumber, progressBlockTime: None, numEventsProcessed: 7., @@ -57,9 +60,53 @@ let make = ( sourceBlockNumber, }, ~reorgCheckpoints=[], - ~isInReorgThreshold=false, + ~isInReorgThreshold, ~isRealtime=false, - ~config=TestConfig.default, + ~config, ~contractMapping=TestConfig.default.contractMapping, ~registrationsByChainId, + ) + +let make = ( + ~endBlock=None, + ~progressBlockNumber, + ~firstEventBlockNumber, + ~timestampCaughtUpToHeadOrEndblock=None, + ~sourceBlockNumber=1000, +): Metrics.chainMetrics => + makeChainState( + ~endBlock, + ~progressBlockNumber, + ~firstEventBlockNumber, + ~timestampCaughtUpToHeadOrEndblock, + ~sourceBlockNumber, )->ChainState.toMetrics + +// The snapshot a run reports when nothing has happened yet. Tests spread this +// and name only the chains they assert on. +let emptySnapshot: Metrics.t = { + startTime: Date.fromTime(0.), + metricTime: Date.fromTime(0.), + elapsedSeconds: 0., + targetBufferSize: 0, + isInReorgThreshold: false, + hasArrivedAtHead: false, + rollbackEnabled: false, + maxBatchSize: 0, + preloadSeconds: 0., + processingSeconds: 0., + processingStalledOnFetchSeconds: 0., + processingStalledOnStorageWriteSeconds: 0., + rollbackSeconds: 0., + rollbackCount: 0, + rollbackEventsCount: 0., + chains: [], + handlers: [], + effects: [], + storageLoads: [], + storageWrites: [], + historyPrunes: [], + sourceRequests: [], + sourceHeights: [], + sourceHeightStreams: [], +} diff --git a/packages/envio-tests/test/helpers/fakeWorker.mjs b/packages/envio-tests/test/helpers/fakeWorker.mjs new file mode 100644 index 0000000000..edc00abe13 --- /dev/null +++ b/packages/envio-tests/test/helpers/fakeWorker.mjs @@ -0,0 +1,48 @@ +// Stands in for a forked indexer process. `FAKE_WORKER` picks how it ends, so a +// supervisor's handling of a clean finish and of a failure can both be driven +// with real processes. +const mode = process.env.FAKE_WORKER ?? "report"; + +// A real worker reports on a timer; the fixture reports once, as soon as it is +// started, since nothing tells it when its supervisor is listening. +process.send({ + kind: "snapshot", + metrics: { + workerConfig: process.env.ENVIO_INTERNAL_WORKER, + maxConnections: process.env.ENVIO_PG_MAX_CONNECTIONS, + bufferSize: process.env.ENVIO_INDEXING_MAX_BUFFER_SIZE, + objectsTarget: process.env.ENVIO_IN_MEMORY_OBJECTS_TARGET, + logFile: process.env.LOG_FILE, + // A Date survives only under structured-clone serialization, which is + // what a metrics snapshot's timestamps need. + startTime: new Date(1700000000000), + // The one reading the supervisor's barrier asks each worker for. + hasArrivedAtHead: process.env.FAKE_WORKER_ARRIVED === "1", + }, +}); + +if (mode === "succeed") process.exit(0); +if (mode === "fail") process.exit(1); + +// Ends the same way "succeed" does, but late enough for a supervisor to have +// taken its report and registered whatever it listens with. +if (mode === "succeed-later") setTimeout(() => process.exit(0), 60); + +// Writes across chunk boundaries the way a real process does: a pipe hands the +// supervisor whatever has been flushed, not whole lines. +if (mode === "print") { + process.stdout.write("first line\nsecond "); + process.stderr.write("from stderr\n"); + process.stdout.write("line\n"); + setTimeout(() => process.exit(0), 50); +} + +// Nothing else keeps a "linger" worker alive; it waits to be stopped. +if (mode === "linger") setInterval(() => {}, 1000); + +// Ends only once the supervisor releases it, so a run that reaches its exit is +// a run whose barrier opened. +if (mode === "await-release") { + setInterval(() => {}, 1000); + process.on("message", () => process.exit(0)); +} diff --git a/packages/envio-tests/test/lib_tests/ChainMetaReadyAt_test.res b/packages/envio-tests/test/lib_tests/ChainMetaReadyAt_test.res new file mode 100644 index 0000000000..75139fe3e3 --- /dev/null +++ b/packages/envio-tests/test/lib_tests/ChainMetaReadyAt_test.res @@ -0,0 +1,75 @@ +open Vitest + +// `ready_at` is committed by the finalization and carried by every chain +// metadata write after it. The two race: metadata is written on a throttle of +// its own, outside the batch the finalization flushes, so a snapshot taken +// before the stamp can reach the database after it. +let sql = PgStorage.makeClient() +let pgSchema = TestPgSchema.make() +let config = TestConfig.make() + +Async.afterAll(async () => { + await sql->TestPgSchema.drop(~pgSchema) + await sql->Postgres.endSql +}) + +let storage = PgStorage.make( + ~sql, + ~pgHost=Env.Db.host, + ~pgSchema, + ~pgPort=Env.Db.port, + ~pgUser=Env.Db.user, + ~pgDatabase=Env.Db.database, + ~pgPassword=Env.Db.password, + ~isHasuraEnabled=false, + ~ecosystem=Evm, +) + +let readyAt = async () => { + let rows: array<{ + "ready_at": Null.t, + }> = await sql->Postgres.unsafe( + `SELECT "ready_at" FROM "${pgSchema}"."envio_chains" ORDER BY "id";`, + ) + rows->Array.map(row => row["ready_at"]->Null.toOption->Option.isSome) +} + +describe("A chain metadata write", () => { + Async.it("Can't clear the ready timestamp the finalization committed", async t => { + let _ = await storage.initialize( + ~chainConfigs=config.chainMap->ChainMap.values, + ~contractMapping=config.contractMapping, + ~entities=config.userEntities, + ~enums=config.allEnums->Array.concat([ + EntityHistory.RowAction.config->Table.fromGenericEnumConfig, + ]), + ~envioInfo=JSON.Encode.object(Dict.make()), + ) + await storage.finalizeBackfill( + ~entities=config.userEntities, + ~chainIds=config.chainMap->ChainMap.keys, + ~readyAt=Date.make(), + ) + let stamped = await readyAt() + + // What a process staged before it was ready, landing after the stamp. + let stale = Dict.make() + config.chainMap + ->ChainMap.keys + ->Array.forEach( + chainId => + stale->Dict.set( + chainId->ChainId.toString, + { + InternalTable.Chains.firstEventBlockNumber: Null.null, + latestFetchedBlockNumber: 10, + timestampCaughtUpToHeadOrEndblock: Null.null, + isHyperSync: false, + }, + ), + ) + let _ = await storage.setChainMeta(stale) + + t.expect((stamped, await readyAt())).toStrictEqual(([true], [true])) + }) +}) diff --git a/packages/envio-tests/test/lib_tests/ChainMilestones_test.res b/packages/envio-tests/test/lib_tests/ChainMilestones_test.res new file mode 100644 index 0000000000..54bd3b29e8 --- /dev/null +++ b/packages/envio-tests/test/lib_tests/ChainMilestones_test.res @@ -0,0 +1,73 @@ +open Vitest + +open TestChainMetrics + +// The milestones a chain reports for itself. Each is the chain's own to say: +// what it indexed to, what changed when it crossed the reorg threshold, and +// when it became ready. A run split across processes has no process that can +// speak for chains it doesn't drive, and an unsplit run says exactly the same +// things about exactly the same chains. + +describe("ChainState.takeFinished", () => { + it("Names where a chain finished indexing, once, and only once it has", t => { + let stillWorking = makeChainState( + ~progressBlockNumber=500, + ~firstEventBlockNumber=None, + ~endBlock=Some(600), + ) + let atEndBlock = makeChainState( + ~progressBlockNumber=600, + ~firstEventBlockNumber=None, + ~endBlock=Some(600), + ) + + t.expect(( + stillWorking->ChainState.takeFinished, + atEndBlock->ChainState.takeFinished, + // Being finished stays true, and every later pass would say so again. + atEndBlock->ChainState.takeFinished, + )).toStrictEqual((None, Some(ChainState.EndBlock(600)), None)) + }) +}) + +// `makeChainState` puts the head at 1000 with a reorg depth of 200 and no +// block lag, so a chain fetches no further than block 800 until the indexer +// crosses into the blocks above it. +describe("ChainState.reorgThresholdLiftsCeiling", () => { + it("Is true only where the lag was holding the chain back", t => { + let chain = (~endBlock=None, ~maxReorgDepth=200, ~config=TestConfig.default) => + makeChainState( + ~progressBlockNumber=500, + ~firstEventBlockNumber=None, + ~endBlock, + ~maxReorgDepth, + ~config, + )->ChainState.reorgThresholdLiftsCeiling + + t.expect([ + // Runs to the head, so crossing is what lets it get there. + chain(), + // An end block above the held frontier: crossing opens up the rest of it. + chain(~endBlock=Some(900)), + // An end block below it was never held back by the lag. + chain(~endBlock=Some(600)), + // No reorg depth to be held back by, so crossing changes nothing. This is + // also every configuration that keeps no history, which is why the line + // has no form that leaves the history out. + chain(~maxReorgDepth=0), + // Nothing is rolled back, so the chain already fetches to the head. + chain(~config={...TestConfig.default, shouldRollbackOnReorg: false}), + ]).toStrictEqual([true, true, false, false, false]) + }) +}) + +describe("ChainState.reorgThresholdEntryMessage", () => { + // Crossing lifts the lag that held the chain short of the head and starts the + // history a rollback replays. A reader watching the writes grow wants the + // second half of that. + it("Says what crossing changed, and what it starts writing", t => { + t.expect(ChainState.reorgThresholdEntryMessage).toBe( + "Indexing the latest blocks now. These can be reorged, so the indexer starts storing a history of every change to roll back with.", + ) + }) +}) diff --git a/packages/envio-tests/test/lib_tests/FinalizeIndexLogging_test.res b/packages/envio-tests/test/lib_tests/FinalizeIndexLogging_test.res new file mode 100644 index 0000000000..107edb7cf2 --- /dev/null +++ b/packages/envio-tests/test/lib_tests/FinalizeIndexLogging_test.res @@ -0,0 +1,80 @@ +open Vitest + +// Finalizing a schema that declares no indexes has nothing to say about them. +// The indexer reporting itself ready is `FinalizeBackfill`'s line, not this +// one's, so a schema with no indexes should leave no trace here. +let sql = PgStorage.makeClient() +let pgSchema = TestPgSchema.make() +let config = TestConfig.make() + +Async.afterAll(async () => { + await sql->TestPgSchema.drop(~pgSchema) + await sql->Postgres.endSql +}) + +let loggedMessages = async path => + switch await NodeJs.Fs.Promises.readFile(~filepath=NodeJs.Path.resolve([path]), ~encoding=Utf8) { + | contents => + contents + ->String.trim + ->String.split("\n") + ->Array.filterMap(line => + switch line->JSON.parseOrThrow->JSON.Decode.object { + | Some(fields) => fields->Dict.get("msg")->Option.flatMap(JSON.Decode.string) + | None => None + } + ) + | exception _ => [] + } + +describe("Finalizing a schema with no indexes", () => { + Async.it("Says nothing about the indexes it didn't have to build", async t => { + let storage = PgStorage.make( + ~sql, + ~pgHost=Env.Db.host, + ~pgSchema, + ~pgPort=Env.Db.port, + ~pgUser=Env.Db.user, + ~pgDatabase=Env.Db.database, + ~pgPassword=Env.Db.password, + ~isHasuraEnabled=false, + ~ecosystem=Evm, + ) + let _ = await storage.initialize( + ~chainConfigs=config.chainMap->ChainMap.values, + ~contractMapping=config.contractMapping, + ~entities=config.userEntities, + ~enums=config.allEnums->Array.concat([ + EntityHistory.RowAction.config->Table.fromGenericEnumConfig, + ]), + ~envioInfo=JSON.Encode.object(Dict.make()), + ) + + let path = `${NodeJs.Process.cwd()}/lib/envio-finalize-indexes-${Date.now()->Float.toString}.log` + Logging.setLogger( + Logging.makeLogger( + ~logStrategy=FileOnly, + ~logFilePath=path, + ~defaultFileLogLevel=#info, + ~userLogLevel=#info, + ), + ) + + await storage.finalizeBackfill( + ~entities=config.userEntities, + ~chainIds=config.chainMap->ChainMap.keys, + ~readyAt=Date.make(), + ) + Logging.info("done") + + let rec until = async deadline => + switch await loggedMessages(path) { + | messages if messages->Array.includes("done") || Date.now() > deadline => messages + | _ => + await Utils.delay(50) + await until(deadline) + } + + t.expect(await until(Date.now() +. 3000.)).toStrictEqual(["done"]) + }) +}) diff --git a/packages/envio-tests/test/lib_tests/Metrics_test.res b/packages/envio-tests/test/lib_tests/Metrics_test.res index fc6ca98276..5b43121ea1 100644 --- a/packages/envio-tests/test/lib_tests/Metrics_test.res +++ b/packages/envio-tests/test/lib_tests/Metrics_test.res @@ -62,31 +62,7 @@ envio_source_request_seconds_total{method="getLogs"} 1.5`) // The state a Metrics.t carries when a test says nothing about it. Each test // below spreads this and names only the fields it asserts on. -let baseMetrics: Metrics.t = { - startTime: Date.fromTime(0.), - metricTime: Date.fromTime(0.), - elapsedSeconds: 0., - targetBufferSize: 0, - isInReorgThreshold: false, - rollbackEnabled: false, - maxBatchSize: 0, - preloadSeconds: 0., - processingSeconds: 0., - processingStalledOnFetchSeconds: 0., - processingStalledOnStorageWriteSeconds: 0., - rollbackSeconds: 0., - rollbackCount: 0, - rollbackEventsCount: 0., - chains: [], - handlers: [], - effects: [], - storageLoads: [], - storageWrites: [], - historyPrunes: [], - sourceRequests: [], - sourceHeights: [], - sourceHeightStreams: [], -} +let baseMetrics = TestChainMetrics.emptySnapshot describe("Metrics.collect", () => { it("Renders only the indexer info when there is no state", t => { @@ -241,6 +217,7 @@ envio_info{version="${Utils.EnvioPackage.value.version}"} 1 elapsedSeconds: 123.456, targetBufferSize: 5000, isInReorgThreshold: true, + hasArrivedAtHead: true, rollbackEnabled: true, maxBatchSize: 5000, preloadSeconds: 12.3456, @@ -668,3 +645,273 @@ envio_indexing_contract_addresses{chainId="1",contract="NftFactory"} 2 ) }) }) + +describe("Metrics.merge", () => { + let startTime = Date.fromTime(1000.) + let metricTime = Date.fromTime(5000.) + + let handler = (~event, ~processingCount): Metrics.handlerMetrics => { + contract: "Token", + event, + processingSeconds: 1., + processingCount, + preloadSeconds: 0.5, + preloadCount: 2., + preloadSecondsTotal: 3., + } + + let effect = (~cacheCount): Metrics.effectMetrics => { + effect: "getMetadata", + scope: "crossChain", + callSeconds: 1., + callSecondsTotal: 2., + callCount: 3., + activeCallsCount: 1, + queueCount: 2, + queueWaitSeconds: 0.25, + invalidationsCount: 1., + cacheCount, + } + + it("Returns one worker's snapshot unchanged, taking the clock from the caller", t => { + let only: Metrics.t = { + ...baseMetrics, + startTime: Date.fromTime(777.), + metricTime: Date.fromTime(888.), + elapsedSeconds: 42., + processingSeconds: 1.5, + maxBatchSize: 5000, + chains: [TestChainMetrics.make(~progressBlockNumber=400, ~firstEventBlockNumber=Some(150))], + handlers: [handler(~event="Transfer", ~processingCount=4.)], + effects: [effect(~cacheCount=Some(7))], + } + + t.expect( + Metrics.merge([only], ~startTime, ~metricTime, ~elapsedSeconds=9., ~targetBufferSize=100), + ).toStrictEqual({ + ...only, + startTime, + metricTime, + elapsedSeconds: 9., + targetBufferSize: 100, + }) + }) + + it("Concatenates chain series, sums what shares a key, and folds the scalars", t => { + let chainOne = TestChainMetrics.make(~progressBlockNumber=400, ~firstEventBlockNumber=None) + let chainTwo = {...chainOne, Metrics.chainId: 137->ChainId.fromInt} + + let first: Metrics.t = { + ...baseMetrics, + targetBufferSize: 100, + maxBatchSize: 5000, + isInReorgThreshold: false, + rollbackEnabled: true, + processingSeconds: 1.5, + rollbackCount: 1, + chains: [chainOne], + handlers: [handler(~event="Transfer", ~processingCount=4.)], + effects: [effect(~cacheCount=Some(7))], + storageWrites: [{storage: "Postgres", seconds: 2., count: 3}], + } + let second: Metrics.t = { + ...baseMetrics, + targetBufferSize: 50, + maxBatchSize: 1000, + isInReorgThreshold: true, + hasArrivedAtHead: true, + rollbackEnabled: true, + processingSeconds: 0.5, + rollbackCount: 2, + chains: [chainTwo], + handlers: [ + handler(~event="Transfer", ~processingCount=6.), + handler(~event="Approval", ~processingCount=1.), + ], + effects: [effect(~cacheCount=None)], + storageWrites: [{storage: "Postgres", seconds: 1., count: 4}], + } + + t.expect( + Metrics.merge( + [first, second], + ~startTime, + ~metricTime, + ~elapsedSeconds=9., + ~targetBufferSize=100, + ), + ).toStrictEqual({ + ...baseMetrics, + startTime, + metricTime, + elapsedSeconds: 9., + // Every worker holds a pool of the run's target, so the targets are one + // number the run was configured with, not a total to add up. + targetBufferSize: 100, + maxBatchSize: 5000, + // Chains cross into the threshold as one indexer, so one worker still + // below it speaks for the whole run, exactly as for arriving at the head. + isInReorgThreshold: false, + hasArrivedAtHead: false, + rollbackEnabled: true, + processingSeconds: 2., + rollbackCount: 3, + chains: [chainOne, chainTwo], + handlers: [ + { + ...handler(~event="Transfer", ~processingCount=10.), + processingSeconds: 2., + preloadSeconds: 1., + preloadCount: 4., + preloadSecondsTotal: 6., + }, + handler(~event="Approval", ~processingCount=1.), + ], + effects: [ + { + ...effect(~cacheCount=Some(7)), + callSeconds: 2., + callSecondsTotal: 4., + callCount: 6., + activeCallsCount: 2, + queueCount: 4, + queueWaitSeconds: 0.5, + invalidationsCount: 2., + }, + ], + storageWrites: [{storage: "Postgres", seconds: 3., count: 7}], + }) + }) + + it("Renders an empty group as an indexer that has reported nothing yet", t => { + t.expect( + Metrics.merge([], ~startTime, ~metricTime, ~elapsedSeconds=0., ~targetBufferSize=100), + ).toStrictEqual({ + ...baseMetrics, + startTime, + metricTime, + targetBufferSize: 100, + }) + }) +}) + +describe("Metrics.renderRuntime", () => { + let sample = (~heapUsed, ~gc): Metrics.runtimeSample => { + cpuUserSeconds: 1.5, + cpuSystemSeconds: 0.5, + processStartTimeSeconds: 1700000000., + residentMemoryBytes: 300., + heapTotalBytes: 200., + heapUsedBytes: heapUsed, + externalMemoryBytes: 10., + eventLoopUtilization: 0.25, + eventLoopLagMeanSeconds: 0.001, + eventLoopLagMinSeconds: 0., + eventLoopLagMaxSeconds: 0.002, + eventLoopLagStddevSeconds: 0.0005, + eventLoopLagP50Seconds: 0.001, + eventLoopLagP90Seconds: 0.0015, + eventLoopLagP99Seconds: 0.002, + heapSpaces: [{space: "new", size: 100., used: 40., available: 60.}], + activeResources: [("TCPSocketWrap", 2.)], + gc, + nodeVersion: "v24.1.2", + } + + // Comment lines are the same in every layout, so only the samples are compared. + let samples = rendered => + rendered + ->String.split("\n") + ->Array.filter(line => line !== "" && !(line->String.startsWith("#"))) + + it("Renders one process without labels, and a run's processes under a worker label", t => { + t.expect(( + Metrics.renderRuntime([("", sample(~heapUsed=150., ~gc=[]))])->samples, + Metrics.renderRuntime([ + (`worker="1"`, sample(~heapUsed=50., ~gc=[])), + (`worker="137"`, sample(~heapUsed=150., ~gc=[{kind: "minor", count: 3., seconds: 0.03}])), + ])->samples, + )).toStrictEqual(( + [ + "process_cpu_user_seconds_total 1.5", + "process_cpu_system_seconds_total 0.5", + "process_cpu_seconds_total 2", + "process_start_time_seconds 1700000000", + "process_resident_memory_bytes 300", + "nodejs_heap_size_total_bytes 200", + "nodejs_heap_size_used_bytes 150", + "nodejs_external_memory_bytes 10", + "nodejs_eventloop_utilization 0.25", + "nodejs_eventloop_lag_mean_seconds 0.001", + "nodejs_eventloop_lag_min_seconds 0", + "nodejs_eventloop_lag_max_seconds 0.002", + "nodejs_eventloop_lag_stddev_seconds 0.001", + "nodejs_eventloop_lag_p50_seconds 0.001", + "nodejs_eventloop_lag_p90_seconds 0.002", + "nodejs_eventloop_lag_p99_seconds 0.002", + `nodejs_heap_space_size_total_bytes{space="new"} 100`, + `nodejs_heap_space_size_used_bytes{space="new"} 40`, + `nodejs_heap_space_size_available_bytes{space="new"} 60`, + `nodejs_active_resources{type="TCPSocketWrap"} 2`, + "nodejs_active_resources_total 2", + `nodejs_version_info{version="v24.1.2",major="24",minor="1",patch="2"} 1`, + ], + [ + `process_cpu_user_seconds_total{worker="1"} 1.5`, + `process_cpu_user_seconds_total{worker="137"} 1.5`, + `process_cpu_system_seconds_total{worker="1"} 0.5`, + `process_cpu_system_seconds_total{worker="137"} 0.5`, + `process_cpu_seconds_total{worker="1"} 2`, + `process_cpu_seconds_total{worker="137"} 2`, + `process_start_time_seconds{worker="1"} 1700000000`, + `process_start_time_seconds{worker="137"} 1700000000`, + `process_resident_memory_bytes{worker="1"} 300`, + `process_resident_memory_bytes{worker="137"} 300`, + `nodejs_heap_size_total_bytes{worker="1"} 200`, + `nodejs_heap_size_total_bytes{worker="137"} 200`, + `nodejs_heap_size_used_bytes{worker="1"} 50`, + `nodejs_heap_size_used_bytes{worker="137"} 150`, + `nodejs_external_memory_bytes{worker="1"} 10`, + `nodejs_external_memory_bytes{worker="137"} 10`, + `nodejs_eventloop_utilization{worker="1"} 0.25`, + `nodejs_eventloop_utilization{worker="137"} 0.25`, + `nodejs_eventloop_lag_mean_seconds{worker="1"} 0.001`, + `nodejs_eventloop_lag_mean_seconds{worker="137"} 0.001`, + `nodejs_eventloop_lag_min_seconds{worker="1"} 0`, + `nodejs_eventloop_lag_min_seconds{worker="137"} 0`, + `nodejs_eventloop_lag_max_seconds{worker="1"} 0.002`, + `nodejs_eventloop_lag_max_seconds{worker="137"} 0.002`, + `nodejs_eventloop_lag_stddev_seconds{worker="1"} 0.001`, + `nodejs_eventloop_lag_stddev_seconds{worker="137"} 0.001`, + `nodejs_eventloop_lag_p50_seconds{worker="1"} 0.001`, + `nodejs_eventloop_lag_p50_seconds{worker="137"} 0.001`, + `nodejs_eventloop_lag_p90_seconds{worker="1"} 0.002`, + `nodejs_eventloop_lag_p90_seconds{worker="137"} 0.002`, + `nodejs_eventloop_lag_p99_seconds{worker="1"} 0.002`, + `nodejs_eventloop_lag_p99_seconds{worker="137"} 0.002`, + `nodejs_heap_space_size_total_bytes{worker="1",space="new"} 100`, + `nodejs_heap_space_size_total_bytes{worker="137",space="new"} 100`, + `nodejs_heap_space_size_used_bytes{worker="1",space="new"} 40`, + `nodejs_heap_space_size_used_bytes{worker="137",space="new"} 40`, + `nodejs_heap_space_size_available_bytes{worker="1",space="new"} 60`, + `nodejs_heap_space_size_available_bytes{worker="137",space="new"} 60`, + `nodejs_active_resources{worker="1",type="TCPSocketWrap"} 2`, + `nodejs_active_resources{worker="137",type="TCPSocketWrap"} 2`, + `nodejs_active_resources_total{worker="1"} 2`, + `nodejs_active_resources_total{worker="137"} 2`, + `nodejs_gc_duration_seconds_sum{worker="137",kind="minor"} 0.03`, + `nodejs_gc_duration_seconds_count{worker="137",kind="minor"} 3`, + `nodejs_version_info{worker="1",version="v24.1.2",major="24",minor="1",patch="2"} 1`, + `nodejs_version_info{worker="137",version="v24.1.2",major="24",minor="1",patch="2"} 1`, + ], + )) + }) + + // A scrape whose last line has no line feed is a parse error to a strict + // consumer, which drops the whole body rather than its last sample. + it("Ends its body with a line feed, the way the text format requires", t => { + t.expect( + Metrics.renderRuntime([("", sample(~heapUsed=150., ~gc=[]))])->String.endsWith("\n"), + ).toBe(true) + }) +}) diff --git a/packages/envio-tests/test/lib_tests/PgStorage_test.res b/packages/envio-tests/test/lib_tests/PgStorage_test.res index f0cc99f5f6..b8d025dbec 100644 --- a/packages/envio-tests/test/lib_tests/PgStorage_test.res +++ b/packages/envio-tests/test/lib_tests/PgStorage_test.res @@ -626,11 +626,20 @@ FROM "public"."envio_chains";` async t => { let params = [] let condition = PgStorage.makeFilterCondition( - ~filter=dict{"tag": dict{"_eq": Uint8Array.fromArray([0xaa])->(Utils.magic: Uint8Array.t => unknown), "_in": [Uint8Array.fromArray([1, 2]), Uint8Array.fromLength(0)]->( - Utils.magic: array => unknown - )}, "chunks": dict{"_eq": [Uint8Array.fromArray([3])]->(Utils.magic: array => unknown), "_in": [[Uint8Array.fromArray([4])], [Uint8Array.fromArray([5])]]->( - Utils.magic: array> => unknown - )}}->parse(~table=bytesTable), + ~filter=dict{ + "tag": dict{ + "_eq": Uint8Array.fromArray([0xaa])->(Utils.magic: Uint8Array.t => unknown), + "_in": [Uint8Array.fromArray([1, 2]), Uint8Array.fromLength(0)]->( + Utils.magic: array => unknown + ), + }, + "chunks": dict{ + "_eq": [Uint8Array.fromArray([3])]->(Utils.magic: array => unknown), + "_in": [[Uint8Array.fromArray([4])], [Uint8Array.fromArray([5])]]->( + Utils.magic: array> => unknown + ), + }, + }->parse(~table=bytesTable), ~table=bytesTable, ~pgSchema="test_schema", ~params, @@ -654,7 +663,9 @@ FROM "public"."envio_chains";` async t => { let params = [] let condition = PgStorage.makeFilterCondition( - ~filter=dict{"id": dict{"_in": ["1", "2"]->(Utils.magic: array => unknown)}}->parse(~table), + ~filter=dict{ + "id": dict{"_in": ["1", "2"]->(Utils.magic: array => unknown)}, + }->parse(~table), ~table, ~pgSchema="test_schema", ~params, @@ -690,7 +701,10 @@ FROM "public"."envio_chains";` async t => { let params = [] let condition = PgStorage.makeFilterCondition( - ~filter=dict{"score": dict{"_gte": 5->(Utils.magic: int => unknown)}, "id": dict{"_lte": "9"->(Utils.magic: string => unknown)}}->parse(~table), + ~filter=dict{ + "score": dict{"_gte": 5->(Utils.magic: int => unknown)}, + "id": dict{"_lte": "9"->(Utils.magic: string => unknown)}, + }->parse(~table), ~table, ~pgSchema="test_schema", ~params, @@ -708,7 +722,13 @@ FROM "public"."envio_chains";` async t => { let params = [] let condition = PgStorage.makeFilterCondition( - ~filter=dict{"id": dict{"_eq": "1"->(Utils.magic: string => unknown)}, "score": dict{"_gt": 5->(Utils.magic: int => unknown), "_lt": 10->(Utils.magic: int => unknown)}}->parse(~table), + ~filter=dict{ + "id": dict{"_eq": "1"->(Utils.magic: string => unknown)}, + "score": dict{ + "_gt": 5->(Utils.magic: int => unknown), + "_lt": 10->(Utils.magic: int => unknown), + }, + }->parse(~table), ~table, ~pgSchema="test_schema", ~params, @@ -756,11 +776,16 @@ FROM "public"."envio_chains";` t.expect(( condition(tagsIn([["a"], ["a", "b"]])), condition(tagsIn([])), - condition(dict{"flag": dict{"_in": [true, false]->(Utils.magic: array => unknown)}}), + condition( + dict{"flag": dict{"_in": [true, false]->(Utils.magic: array => unknown)}}, + ), )).toEqual(( ( `("tags" = $1 OR "tags" = $2)`, - [["a"]->(Utils.magic: array => unknown), ["a", "b"]->(Utils.magic: array => unknown)], + [ + ["a"]->(Utils.magic: array => unknown), + ["a", "b"]->(Utils.magic: array => unknown), + ], ), ("FALSE", []), ( @@ -891,10 +916,12 @@ VALUES($1,$2)ON CONFLICT("id") DO UPDATE SET "c_id" = EXCLUDED."c_id";` async t => { let query = InternalTable.Chains.makeMetaFieldsUpdateQuery(~pgSchema="test_schema") + // `ready_at` keeps what is committed: a write staged before the + // finalization stamped it must not clear it on its way to the database. let expectedQuery = `UPDATE "test_schema"."envio_chains" SET "buffer_block" = $2, "first_event_block" = $3, - "ready_at" = $4, + "ready_at" = COALESCE($4, "ready_at"), "_is_hyper_sync" = $5 WHERE "id" = $1;` diff --git a/packages/envio-tests/test/lib_tests/Supervisor_test.res b/packages/envio-tests/test/lib_tests/Supervisor_test.res new file mode 100644 index 0000000000..c6b9c0b883 --- /dev/null +++ b/packages/envio-tests/test/lib_tests/Supervisor_test.res @@ -0,0 +1,352 @@ +open Vitest + +let chains = n => Array.make(~length=n, 0)->Array.mapWithIndex((_, i) => (i + 1)->ChainId.fromInt) + +describe("Supervisor.plan", () => { + it("Splits only when the budget affords two workers, and spends all of it", t => { + let plan = (~chainCount, ~maxConnections) => + Supervisor.plan(~chainIds=chains(chainCount), ~maxConnections)->Option.map( + workers => + workers->Array.map( + ({chainIds, maxConnections}: Supervisor.worker) => ( + chainIds->Array.map(ChainId.toString)->Array.joinUnsafe(","), + maxConnections, + ), + ), + ) + + t.expect([ + // The default budget is what one process uses today, so nothing splits. + plan(~chainCount=4, ~maxConnections=2), + plan(~chainCount=4, ~maxConnections=3), + plan(~chainCount=4, ~maxConnections=4), + // More budget than chains: the surplus widens every worker's pool + // instead of going unused. + plan(~chainCount=4, ~maxConnections=10), + plan(~chainCount=3, ~maxConnections=12), + // A single chain has nothing to split against, whatever the budget. + plan(~chainCount=1, ~maxConnections=100), + ]).toStrictEqual([ + None, + None, + Some([("1,4", 2), ("2,3", 2)]), + Some([("1", 3), ("2", 3), ("3", 2), ("4", 2)]), + Some([("1", 4), ("2", 4), ("3", 4)]), + None, + ]) + }) +}) + +describe("Supervisor.plan at the ceiling", () => { + // A worker is a whole Node process: its own heap, its own copy of the + // handler modules, its own source clients. A budget that would buy more of + // them than this spends the surplus widening their pools instead. + it("Never spends a budget on more than four processes", t => { + let plan = (~chainCount, ~maxConnections) => + Supervisor.plan(~chainIds=chains(chainCount), ~maxConnections) + ->Option.getOrThrow + ->Array.map(({maxConnections}: Supervisor.worker) => maxConnections) + + t.expect([ + // Ten chains and the connections for ten workers: four, with the budget + // spread across them rather than two connections each and the rest + // unspent. + plan(~chainCount=10, ~maxConnections=20), + // The remainder still goes to the earliest workers. + plan(~chainCount=10, ~maxConnections=22), + // Below the ceiling nothing changes. + plan(~chainCount=10, ~maxConnections=6), + ]).toStrictEqual([[5, 5, 5, 5], [6, 6, 5, 5], [2, 2, 2]]) + }) +}) + +describe("Supervisor.plan on the default budget", () => { + // Splitting a run costs connections the operator didn't ask to spend, so the + // budget they didn't set is the one a single process has always used. + it("Keeps a run in one process until the budget is raised", t => { + t.expect(Supervisor.plan(~chainIds=chains(4), ~maxConnections=Env.Db.maxConnections)).toBe(None) + }) +}) + +describe("Supervisor.plan dealing order", () => { + let assignment = (~chainIds, ~maxConnections) => + Supervisor.plan(~chainIds=chainIds->Array.map(ChainId.fromInt), ~maxConnections) + ->Option.getOrThrow + ->Array.map(worker => worker.chainIds->Array.map(ChainId.toString)->Array.joinUnsafe(",")) + + it("Deals chains in config order, reversing direction each pass", t => { + t.expect([ + // The first two lead different workers; the worker that took the first + // picks up the last. Config order is the ranking, not the chain ids. + assignment(~chainIds=[8453, 56, 42161, 1], ~maxConnections=4), + // Three workers take the first three, then fold back. + assignment(~chainIds=[1, 56, 137, 8453, 42161, 10], ~maxConnections=6), + // An odd count leaves the fold short: the last chain lands mid-pass. + assignment(~chainIds=[1, 56, 137, 8453, 42161], ~maxConnections=4), + ]).toStrictEqual([ + ["8453,1", "56,42161"], + ["1,10", "56,42161", "137,8453"], + ["1,8453,42161", "56,137"], + ]) + }) +}) + +let configYaml = ` +name: supervised-run +disable_default_cross_chain: true +contracts: + - name: Counters + events: + - event: Bumped(uint256 amount) +chains: + - id: 1 + start_block: 0 + contracts: + - name: Counters + address: "0x1111111111111111111111111111111111111111" + - id: 137 + start_block: 0 + contracts: + - name: Counters + address: "0x2222222222222222222222222222222222222222" +` + +let config = (~schema, ~isolatedChains=?) => { + let json = Core.fromUserApi(~schema, configYaml).config->JSON.parseOrThrow + switch (json, isolatedChains) { + | (Object(obj), Some(chainIds)) => obj->Dict.set("isolatedChains", JSON.Encode.array(chainIds)) + | _ => () + } + Config.fromPublic(json) +} + +let perChain = ` +type Counter { + id: ID! + count: BigInt! +} +` +let crossChain = ` +type Counter { + id: ID! + count: BigInt! +} +type GlobalCounter @crossChain { + id: ID! + count: BigInt! +} +` + +describe("Supervisor.planForRun", () => { + it("Splits a per-chain schema, and leaves everything else in one process", t => { + let workerChains = (~schema, ~maxConnections, ~isolatedChains=?) => + Supervisor.planForRun(~config=config(~schema, ~isolatedChains?), ~maxConnections)->Option.map( + workers => + workers->Array.map( + worker => worker.chainIds->Array.map(ChainId.toString)->Array.joinUnsafe(","), + ), + ) + + t.expect([ + workerChains(~schema=perChain, ~maxConnections=4), + // An entity shared across chains can't be split: workers would each + // advance their own checkpoint over rows the others reach. + workerChains(~schema=crossChain, ~maxConnections=4), + // The budget one process uses today buys nothing to split with. + workerChains(~schema=perChain, ~maxConnections=2), + // Already one chain's process: whoever started it owns the layout. + workerChains(~schema=perChain, ~maxConnections=100, ~isolatedChains=[JSON.Number(1.)]), + ]).toStrictEqual([Some(["1", "137"]), None, None, None]) + }) +}) + +describe("Supervisor worker plumbing", () => { + // A worker's own parse of the project's files is what `envio start` would + // produce: no chain selection, and no dev run. Both are the command's to say. + it("Restores what the command decided onto a config it parsed itself", t => { + let parsedItself = JSON.Object( + Dict.fromArray([ + ("name", JSON.String("indexer")), + ("isolatedChains", JSON.Null), + ("isDev", JSON.Boolean(false)), + ]), + ) + + t.expect( + parsedItself->Config.withCommandFields( + ~chainIds=[1, 137]->Array.map(ChainId.fromInt), + ~isDev=true, + ), + ).toStrictEqual( + JSON.Object( + Dict.fromArray([ + ("name", JSON.String("indexer")), + ("isolatedChains", JSON.Array([JSON.Number(1.), JSON.Number(137.)])), + ("isDev", JSON.Boolean(true)), + ]), + ), + ) + }) + + // `envio dev` keeps the run up once every chain has reached its end block, so + // the console it serves stays whole. A worker that read the run as a plain + // `envio start` would exit there and take its chains out of that console. + it("Keeps a dev run a dev run in the process that drives part of it", t => { + let devRun = Core.fromUserApi(~schema=perChain, configYaml).config->JSON.parseOrThrow + let workerConfig = + devRun + ->Config.withCommandFields(~chainIds=[1->ChainId.fromInt], ~isDev=true) + ->Config.fromPublic + + t.expect(( + workerConfig.isDev, + workerConfig.isolated, + workerConfig.chainMap->ChainMap.keys, + )).toStrictEqual((true, true, [1->ChainId.fromInt])) + }) + + it("Gives every worker a log file of its own", t => { + t.expect([ + Supervisor.logFilePath(~workerIndex=0, ~path="logs/envio.log"), + Supervisor.logFilePath(~workerIndex=1, ~path="logs/envio.log"), + // A dot in a directory name is not an extension. + Supervisor.logFilePath(~workerIndex=1, ~path="./logs/envio"), + Supervisor.logFilePath(~workerIndex=2, ~path="envio"), + ]).toStrictEqual([ + "logs/envio.worker-0.log", + "logs/envio.worker-1.log", + "./logs/envio.worker-1", + "envio.worker-2", + ]) + }) +}) + +describe("Worker.detect", () => { + it("Counts as a worker only when forked with the variable and a channel", t => { + let forked = Dict.fromArray([ + (Worker.envVar, `{"chainIds":[137],"holdRealtime":true,"isDev":true}`), + ]) + t.expect([ + Worker.detect(~env=forked, ~hasChannel=true), + // A copy of the variable left in a shell, or a process manager forking + // with a channel. + Worker.detect(~env=forked, ~hasChannel=false), + Worker.detect(~env=Dict.make(), ~hasChannel=true), + ]).toStrictEqual([ + Some({Worker.chainIds: [137->ChainId.fromInt], holdRealtime: true, isDev: true}), + None, + None, + ]) + }) + + // Thrown as the module loads, before anything that could say where a bare + // schema error came from. + it("Names the variable when its value isn't a worker config", t => { + t->Vitest.toThrowErrorEqual( + () => Worker.detect(~env=Dict.fromArray([(Worker.envVar, "137")]), ~hasChannel=true), + `Invalid ENVIO_INTERNAL_WORKER: Failed parsing at root. Reason: Expected { chainIds: array; holdRealtime: boolean | undefined; isDev: boolean; }, received 137. It is set by an indexer supervisor for the processes it forks, and isn't meant to be set by hand.`, + ) + }) + + // The supervisor decides whether a run waits; a worker forked before that + // decision existed reads as one that doesn't. + it("Takes a config without the hold as one that doesn't wait", t => { + t.expect( + Worker.detect( + ~env=Dict.fromArray([(Worker.envVar, `{"chainIds":[1],"isDev":false}`)]), + ~hasChannel=true, + ), + ).toStrictEqual( + Some({Worker.chainIds: [1->ChainId.fromInt], holdRealtime: false, isDev: false}), + ) + }) +}) + +describe("Supervisor.classifyExit", () => { + let classify = (~code=Null.null, ~signal=Null.null, ~stopping=false) => + Supervisor.classifyExit(~code, ~signal, ~stopping) + + it("Reads a signalled worker as a run being stopped, not as one failing", t => { + t.expect([ + // `systemctl stop` on a unit with the default kill mode signals every + // process in it, so a worker is told before its supervisor has passed it + // on. Its exit carries no code at all. + classify(~signal=Null.make("SIGTERM")), + // The supervisor's own stop, once it has decided. + classify(~code=Null.null, ~signal=Null.make("SIGTERM"), ~stopping=true), + // Indexing to every end block. + classify(~code=Null.make(0)), + // The kernel's out-of-memory killer, and a worker that threw. + classify(~signal=Null.make("SIGKILL")), + classify(~code=Null.make(1)), + ]).toStrictEqual([ + Supervisor.Stopping, + Supervisor.Expected, + Supervisor.Expected, + Supervisor.Failed, + Supervisor.Failed, + ]) + }) +}) + +describe("Supervisor.syncCache", () => { + Async.it("Dumps once for requests that overlap, and again for a later one", async t => { + let dumps = ref(0) + let dump = () => { + dumps := dumps.contents + 1 + Utils.delay(20) + } + + let first = Supervisor.syncCache(~dump) + let second = Supervisor.syncCache(~dump) + await first + await second + await Supervisor.syncCache(~dump) + + t.expect(dumps.contents).toBe(2) + }) +}) + +describe("Supervisor.configuredChains", () => { + it("Draws the run's chains at their configured blocks, with nothing indexed", t => { + let config = TestConfig.fromUserApi(` +name: test-config +chains: + - id: 1 + start_block: 100 + end_block: 500 + contracts: + - name: Gravatar + address: "0x2B2f78c5BF6D9C12Ee1225D5F374aa91204580c3" + events: + - event: "TestEvent()" + - id: 137 + rpc: + url: https://rpc.example.test + for: sync + start_block: 0 + contracts: + - name: Poap + address: "0x2B2f78c5BF6D9C12Ee1225D5F374aa91204580c3" + events: + - event: "TestEvent()" +`) + + t.expect( + Supervisor.configuredChains(config)->Array.map( + chain => ( + chain.chainId->ChainId.toString, + chain.startBlock, + chain.endBlock, + chain.poweredByHyperSync, + chain.progressBlockNumber, + chain.numEventsProcessed, + chain.isReady, + ), + ), + ).toStrictEqual([ + ("1", 100, Some(500), true, -1, 0., false), + ("137", 0, None, false, -1, 0., false), + ]) + }) +}) diff --git a/packages/envio-tests/test/lib_tests/SyncETA_test.res b/packages/envio-tests/test/lib_tests/SyncETA_test.res new file mode 100644 index 0000000000..121fc8ef56 --- /dev/null +++ b/packages/envio-tests/test/lib_tests/SyncETA_test.res @@ -0,0 +1,9 @@ +open Vitest + +describe("SyncETA.isIndexerFullySynced", () => { + // The supervisor renders a frame before any worker has reported, and a run + // with nothing to report is not a run that finished syncing. + it("Doesn't call an indexer with no reported chains synced", t => { + t.expect(SyncETA.isIndexerFullySynced([])).toBe(false) + }) +}) diff --git a/packages/envio-tests/test/lib_tests/TuiShouldUse_test.res b/packages/envio-tests/test/lib_tests/TuiShouldUse_test.res new file mode 100644 index 0000000000..a813c0c037 --- /dev/null +++ b/packages/envio-tests/test/lib_tests/TuiShouldUse_test.res @@ -0,0 +1,15 @@ +open Vitest + +describe("Tui.shouldUse", () => { + it( + "prefers ENVIO_TUI over what the terminal looks like, and never draws where it is suppressed", + t => { + t.expect(( + Tui.shouldUse(~suppressed=true, ~explicitTui=Some(true)), + Tui.shouldUse(~explicitTui=Some(true)), + Tui.shouldUse(~explicitTui=Some(false)), + Tui.shouldUse(~explicitTui=None), + )).toEqual((false, true, false, !Envio.isNonInteractive())) + }, + ) +}) diff --git a/packages/envio-tests/test/lib_tests/WorkerResumeLogging_test.res b/packages/envio-tests/test/lib_tests/WorkerResumeLogging_test.res new file mode 100644 index 0000000000..bfea0c43d7 --- /dev/null +++ b/packages/envio-tests/test/lib_tests/WorkerResumeLogging_test.res @@ -0,0 +1,103 @@ +open Vitest + +let sql = PgStorage.makeClient() +let pgSchema = TestPgSchema.make() +let config = TestConfig.make() + +Async.afterAll(async () => { + await sql->TestPgSchema.drop(~pgSchema) + await sql->Postgres.endSql +}) + +let makePersistence = () => + Persistence.make( + ~userEntities=config.userEntities, + ~allEnums=config.allEnums, + ~storage=PgStorage.make( + ~sql, + ~pgHost=Env.Db.host, + ~pgSchema, + ~pgPort=Env.Db.port, + ~pgUser=Env.Db.user, + ~pgDatabase=Env.Db.database, + ~pgPassword=Env.Db.password, + ~isHasuraEnabled=false, + ~ecosystem=Evm, + ), + ) + +let initRun = (~requireInitialized, ~announceResume=true) => + makePersistence()->Persistence.init( + ~chainConfigs=config.chainMap->ChainMap.values, + ~contractMapping=config.contractMapping, + ~envioInfo=JSON.Encode.object(Dict.make()), + ~resetCommand="envio dev -r", + ~runCommand=Some("envio dev"), + ~lowercaseAddresses=config.lowercaseAddresses, + ~requireInitialized, + ~announceResume, + ) + +let logLines = async path => + switch await NodeJs.Fs.Promises.readFile(~filepath=NodeJs.Path.resolve([path]), ~encoding=Utf8) { + | contents => + contents + ->String.trim + ->String.split("\n") + ->Array.filterMap(line => + switch line->JSON.parseOrThrow->JSON.Decode.object { + | Some(fields) => fields->Dict.get("msg")->Option.flatMap(JSON.Decode.string) + | None => None + } + ) + | exception _ => [] + } + +let resumeLines = async (~announceResume) => { + let path = `${NodeJs.Process.cwd()}/lib/envio-worker-resume-${Date.now()->Float.toString}-${announceResume + ? "announced" + : "quiet"}.log` + Logging.setLogger( + Logging.makeLogger( + ~logStrategy=FileOnly, + ~logFilePath=path, + ~defaultFileLogLevel=#info, + ~userLogLevel=#info, + ), + ) + + await initRun(~requireInitialized=true, ~announceResume) + Logging.info("done") + + let rec until = async deadline => + switch await logLines(path) { + | lines if lines->Array.includes("done") || Date.now() > deadline => lines + | _ => + await Utils.delay(50) + await until(deadline) + } + await until(Date.now() +. 3000.) +} + +describe("Announcing a resume", () => { + // The supervisor says it once for the whole run, so a worker it forked has + // nothing to add. Every other process resuming a subset of the chains is + // somebody's only window onto it, `envio start --chain` included, and both + // of them need the schema to exist already — so what a process requires of + // the storage can't be what decides whether it speaks. + Async.it("Quiet for a forked worker, and not for anyone else", async t => { + await initRun(~requireInitialized=false) + + let quiet = await resumeLines(~announceResume=false) + let announced = await resumeLines(~announceResume=true) + + t.expect((quiet, announced)).toStrictEqual(( + ["done"], + [ + "Found existing indexer storage. Resuming indexing state...", + "Successfully resumed indexing state! Continuing from the last checkpoint.", + "done", + ], + )) + }) +}) diff --git a/packages/envio/package.json b/packages/envio/package.json index 04c7c7e7d1..28e3bd82db 100644 --- a/packages/envio/package.json +++ b/packages/envio/package.json @@ -58,7 +58,6 @@ "express": "4.19.2", "pino": "10.3.1", "pino-pretty": "13.1.3", - "yargs": "17.7.2", "@rescript/runtime": "12.2.0", "rescript-schema": "9.5.1", "viem": "2.54.0", diff --git a/packages/envio/src/BatchProcessing.res b/packages/envio/src/BatchProcessing.res index 75bccf9ae8..2af1553499 100644 --- a/packages/envio/src/BatchProcessing.res +++ b/packages/envio/src/BatchProcessing.res @@ -69,18 +69,15 @@ and processNextBatch = async (state: IndexerState.t, ~scheduleFetch): unit => { let isInReorgThresholdBeforeUpdate = state->IndexerState.isInReorgThreshold let isRealtimeBeforeUpdate = state->IndexerState.isRealtime - let batch = state->IndexerState.createBatch(~batchSizeTarget=(state->IndexerState.config).batchSize) + let batch = + state->IndexerState.createBatch(~batchSizeTarget=(state->IndexerState.config).batchSize) let progressedChainsById = batch.progressedChainsById let isBelowReorgThreshold = !isInReorgThresholdBeforeUpdate && (state->IndexerState.config).shouldRollbackOnReorg let shouldEnterReorgThreshold = - isBelowReorgThreshold && - state - ->IndexerState.chainStates - ->Dict.valuesToArray - ->Array.every(cs => cs->ChainState.isReadyToEnterReorgThresholdAfterBatch(~batch)) + isBelowReorgThreshold && state->IndexerState.isReadyToEnterReorgThreshold(~batch) if shouldEnterReorgThreshold { IndexerState.enterReorgThreshold(state) @@ -95,7 +92,7 @@ and processNextBatch = async (state: IndexerState.t, ~scheduleFetch): unit => { // finalizing resumes exactly here: it still owes the schema its deferred // indexes, and no batch will ever come along to notice. state->IndexerState.markCaughtUpIfSettled - if state->IndexerState.isFinalizingIndexes { + if state->IndexerState.shouldFinalizeIndexes { await FinalizeBackfill.run(state) } @@ -109,9 +106,9 @@ and processNextBatch = async (state: IndexerState.t, ~scheduleFetch): unit => { } // When resuming from persisted state, all events may already be processed. + state->IndexerState.reportFinished if EventProcessing.allChainsEventsProcessedToEndblock(state->IndexerState.chainStates) { - Logging.info("All chains are caught up to end blocks.") - if !(state->IndexerState.keepProcessAlive) { + if !(state->IndexerState.keepProcessAlive) && !(state->IndexerState.isHoldingRealtime) { await ExitOnCaughtUp.run(state) } } @@ -162,11 +159,14 @@ and processNextBatch = async (state: IndexerState.t, ~scheduleFetch): unit => { // Can safely reset rollback state, since overwrite is not possible. state->IndexerState.clearRollback state->IndexerState.applyBatchProgress(~batch) + // Before the finalize below, so a chain says where it finished ahead of + // the process saying what it does about that. + state->IndexerState.reportFinished // Backfilling → FinalizingIndexes → Ready. Awaiting here holds the // processing loop for the whole finalize, which is what pauses // processing while the indexes are built. - if state->IndexerState.isFinalizingIndexes { + if state->IndexerState.shouldFinalizeIndexes { await FinalizeBackfill.run(state) } @@ -182,11 +182,12 @@ and processNextBatch = async (state: IndexerState.t, ~scheduleFetch): unit => { let allCaughtUp = EventProcessing.allChainsEventsProcessedToEndblock( state->IndexerState.chainStates, ) - if allCaughtUp { - Logging.info("All chains are caught up to end blocks.") - } - if allCaughtUp && !(state->IndexerState.keepProcessAlive) { + if ( + allCaughtUp && + !(state->IndexerState.keepProcessAlive) && + !(state->IndexerState.isHoldingRealtime) + ) { await ExitOnCaughtUp.run(state) } else if ( // In auto-exit mode, error if all chains reached head with no events found diff --git a/packages/envio/src/Bin.res b/packages/envio/src/Bin.res index b0b3439c4a..06ff54de00 100644 --- a/packages/envio/src/Bin.res +++ b/packages/envio/src/Bin.res @@ -51,23 +51,30 @@ let applyEnv = (env: dict) => let run = async args => { try { - switch (await Core.runCli(args))->Null.toOption { - // Rust-only command (codegen / init / stop / docker / metrics / help / - // version / scripts) — nothing for JS to do, exit cleanly. - | None => () - | Some(json) => - switch decodeCommand(json->JSON.parseOrThrow) { - | Start({reset, cwd, env, config}) => - Config.prime(config) - processChdir(cwd) - applyEnv(env) - await Main.start(~reset) - | Migrate({reset, config}) => - Config.prime(config) - await Main.migrate(~reset) - | DropSchema({config}) => - Config.prime(config) - await Main.dropSchema() + if Worker.isEnabled { + Worker.bindToSupervisor() + // Its working directory, its environment and the chains it drives all came + // with the fork, so a worker starts the same way every other process does. + await Main.start() + } else { + switch (await Core.runCli(args))->Null.toOption { + // Rust-only command (codegen / init / stop / docker / metrics / help / + // version / scripts) — nothing for JS to do, exit cleanly. + | None => () + | Some(json) => + switch decodeCommand(json->JSON.parseOrThrow) { + | Start({reset, cwd, env, config}) => + Config.prime(config) + processChdir(cwd) + applyEnv(env) + await Main.start(~reset) + | Migrate({reset, config}) => + Config.prime(config) + await Main.migrate(~reset) + | DropSchema({config}) => + Config.prime(config) + await Main.dropSchema() + } } } } catch { diff --git a/packages/envio/src/ChainFetching.res b/packages/envio/src/ChainFetching.res index 287eca89c8..d68ec8a6e9 100644 --- a/packages/envio/src/ChainFetching.res +++ b/packages/envio/src/ChainFetching.res @@ -263,7 +263,6 @@ and applyQueryResponse = ( ~transactionStore, ) => { let chainState = state->IndexerState.getChainState(~chainId) - let wasFetchingAtHead = chainState->ChainState.isFetchingAtHead chainState->ChainState.handleQueryResult( ~query, @@ -280,17 +279,6 @@ and applyQueryResponse = ( ~blockNumber=newItems->Array.getUnsafe(0)->Internal.getItemBlockNumber, ) } - - // Log the backfill→head transition once: this response brought the fetch - // frontier to the head. Gated on !isReady so realtime re-catch-ups (a new - // block arrives, gets fetched) don't spam the log after the chain is synced. - if ( - !wasFetchingAtHead && - !(chainState->ChainState.isReady) && - chainState->ChainState.isFetchingAtHead - ) { - chainState->ChainState.logger->Logging.childInfo("All events have been fetched") - } } let finishWaitingForNewBlock = ( diff --git a/packages/envio/src/ChainState.res b/packages/envio/src/ChainState.res index 1dee8ac13b..24122808c8 100644 --- a/packages/envio/src/ChainState.res +++ b/packages/envio/src/ChainState.res @@ -62,6 +62,10 @@ type t = { mutable blockRangeFetchCount: float, mutable blockRangeFetchedEvents: float, mutable blockRangeFetchedBlocks: float, + // Whether this chain has said where it finished indexing. Being finished + // stays true for the rest of the run, and every pass over the chains would + // say so again. + mutable reportedFinished: bool, mutable reorgCount: int, mutable reorgDetectedBlock: option, mutable rollbackTargetBlock: option, @@ -177,6 +181,7 @@ let make = ( blockRangeFetchCount: 0., blockRangeFetchedEvents: 0., blockRangeFetchedBlocks: 0., + reportedFinished: false, reorgCount: 0, reorgDetectedBlock: None, rollbackTargetBlock: None, @@ -650,6 +655,30 @@ let hasProcessedToEndblock = (cs: t) => { } } +// Where this chain has finished indexing, the first time it gets there. +// `EndBlock` is terminal: the chain indexed everything it was configured to. +// `Backfill` is the rest of the history, up to the point where blocks can +// still be reorged, which is as far as a chain indexes before the indexer +// crosses into them. +type finished = EndBlock(int) | Backfill(int) + +let takeFinished = (cs: t) => + if cs.reportedFinished { + None + } else { + switch (cs.fetchState.endBlock, cs->hasProcessedToEndblock, cs.isProgressAtHead) { + | (Some(endBlock), true, _) => { + cs.reportedFinished = true + Some(EndBlock(endBlock)) + } + | (_, _, true) => { + cs.reportedFinished = true + Some(Backfill(cs.committedProgressBlockNumber)) + } + | _ => None + } + } + // Caught up as judged by persisted values alone: progress reached the endBlock, // or the head the previous run had already observed (less the lag that holds the // tip back). Unlike `isFetchingAtHead` this doesn't move when a fresh height @@ -971,6 +1000,34 @@ let isInReorgThreshold = (cs: t) => cs.isInReorgThreshold // progress has run. let shouldSaveHistory = (cs: t) => cs.shouldRollbackOnReorg && cs.maxReorgDepth > 0 && cs.isInReorgThreshold +// What crossing into the recent blocks changed for this chain. Until now it +// stopped short of the head by its reorg depth, because it kept nothing it +// could roll back with; crossing lifts both at once. The second half is the +// answer to why the indexer starts writing more than it was. +// Whether crossing gives this chain anything more to index: the last block it +// may fetch afterwards against the last it may fetch now. Read before the +// crossing, while the lag being lifted is still the one in place. +// +// False wherever the lag was never holding this chain back, which is every +// configuration that also keeps no history: a chain that isn't rolled back on a +// reorg, or has no reorg depth, already fetches as far as it ever will. It is +// false too for a chain whose end block sits below the blocks being opened up, +// which will never reach one of them. +let reorgThresholdLiftsCeiling = (cs: t) => { + let ceiling = (~blockLag) => { + let head = Pervasives.max(0, cs.fetchState.knownHeight - blockLag) + switch cs.fetchState.endBlock { + | Some(endBlock) => Pervasives.min(endBlock, head) + | None => head + } + } + ceiling(~blockLag=cs.chainConfig.blockLag) > ceiling(~blockLag=cs.fetchState.blockLag) +} + +// Only ever said by a chain the crossing lifts, and lifting it takes a reorg +// depth to have been held back by, which is the same thing that makes the +// history below worth keeping. So there is no second form without it. +let reorgThresholdEntryMessage = "Indexing the latest blocks now. These can be reorged, so the indexer starts storing a history of every change to roll back with." // Snapshot the chain's metadata fields for staging into the chains table. let toChainMetadata = (cs: t): InternalTable.Chains.metaFields => { @@ -1058,6 +1115,12 @@ let toChainBeforeBatch = (cs: t, ~isRealtime): Batch.chainBeforeBatch => { // Whether the chain's post-batch fetch frontier is ready to cross into the reorg // threshold, using the batch's progressed frontier when this chain advanced. +// The same question asked of where the chain stands now rather than of where a +// batch would leave it. Entering the threshold is what lifts the pre-threshold +// lag, so a chain waiting to enter it has fetched as far as it can. +let isReadyToEnterReorgThreshold = (cs: t) => + cs.fetchState->FetchState.isReadyToEnterReorgThreshold(~tolerance=cs.reorgThresholdReadyTolerance) + let isReadyToEnterReorgThresholdAfterBatch = (cs: t, ~batch: Batch.t) => { let fetchState = switch batch.progressedChainsById->ChainId.Dict.dangerouslyGetNonOption( cs.fetchState.chainId, @@ -1206,6 +1269,7 @@ let markReady = (cs: t, ~readyAt) => let rollbackCommittedProgress = (cs: t, blockNumber) => if blockNumber !== cs.committedProgressBlockNumber { cs.committedProgressBlockNumber = blockNumber + // Exact block only: the rolled-back region is about to be refetched, and // the next batch re-establishes the time either way. cs.committedProgressBlockTime = diff --git a/packages/envio/src/ChainState.resi b/packages/envio/src/ChainState.resi index 93ad6a4893..101645c9d2 100644 --- a/packages/envio/src/ChainState.resi +++ b/packages/envio/src/ChainState.resi @@ -122,9 +122,14 @@ let dispatch: ( let toMetrics: t => Metrics.chainMetrics let toChainMetadata: t => InternalTable.Chains.metaFields let toChainBeforeBatch: (t, ~isRealtime: bool) => Batch.chainBeforeBatch +let isReadyToEnterReorgThreshold: t => bool let isReadyToEnterReorgThresholdAfterBatch: (t, ~batch: Batch.t) => bool // Derived (pure). +type finished = EndBlock(int) | Backfill(int) +let takeFinished: t => option +let reorgThresholdLiftsCeiling: t => bool +let reorgThresholdEntryMessage: string let hasProcessedToEndblock: t => bool let isDurablyCaughtUp: t => bool let getHighestBlockBelowThreshold: t => int diff --git a/packages/envio/src/Config.res b/packages/envio/src/Config.res index 5cfb579120..9bf0d3a3c5 100644 --- a/packages/envio/src/Config.res +++ b/packages/envio/src/Config.res @@ -625,6 +625,16 @@ let getChain = (config, ~chainId) => "No chain with id " ++ chainId->ChainId.toString ++ " found in config.yaml", ) +// Whether every entity belongs to exactly one chain. Read off the checkpoint +// sequence rather than the entities again: a chain gets a counter of its own +// only when no other chain can reach its rows, which is the same fact and the +// one that makes splitting a run across processes safe. +let isPerChain = (config: t) => + switch config.checkpointSequence { + | PerChain => true + | SharedAcrossChains => false + } + // Narrows a config to the chains one `envio start --chain` process drives. // `contractMapping` is deliberately left whole: its ids are what the migration // that created the schema stored, and one rebuilt from a subset would hand the @@ -1221,6 +1231,24 @@ let prime = (json: JSON.t): unit => { cached := None } +// What the command decided rather than the project's files: which chains this +// process drives, and whether the run is a dev run. A worker parses the same +// files its supervisor did, so these are the only two it cannot arrive at on +// its own. +let withCommandFields = (json: JSON.t, ~chainIds, ~isDev) => + switch json->JSON.Decode.object { + | Some(fields) => { + let narrowed = fields->Dict.copy + narrowed->Dict.set( + "isolatedChains", + chainIds->S.reverseConvertToJsonOrThrow(S.array(ChainId.schema)), + ) + narrowed->Dict.set("isDev", JSON.Encode.bool(isDev)) + JSON.Object(narrowed) + } + | None => JsError.throwWithMessage("Invalid indexer config: not an object") + } + let getPublicConfigJson = () => switch primedJson.contents { | Some(json) => json @@ -1266,6 +1294,10 @@ let stripSensitiveData = (json: JSON.t): JSON.t => { cloned } +// What the storage layer records as the config this schema was built from, +// and checks a resuming run against. +let envioInfo = () => getPublicConfigJson()->stripSensitiveData + // Postgres jsonb doesn't preserve key order, so canonicalize with sorted // keys before string-comparing. let rec canonicalJson = (json: JSON.t): JSON.t => diff --git a/packages/envio/src/Core.res b/packages/envio/src/Core.res index 6f1a5f90ed..33bc9e3420 100644 --- a/packages/envio/src/Core.res +++ b/packages/envio/src/Core.res @@ -166,6 +166,10 @@ let loadDevAddon: ({..}, string) => addon = %raw(`function(req, envioDir) { fs.copyFileSync(srcPath, nodePath); } + // Forked workers inherit this, so only the first process in a run pays for + // the cargo build (and they don't contend over the cargo lock). + process.env.ENVIO_DEV_ADDON = nodePath; + return req(nodePath); }`) diff --git a/packages/envio/src/CrossChainState.res b/packages/envio/src/CrossChainState.res index b7cb3d1d83..837f0f23ed 100644 --- a/packages/envio/src/CrossChainState.res +++ b/packages/envio/src/CrossChainState.res @@ -17,6 +17,11 @@ type t = { mutable isCaughtUp: bool, // Indexer-wide fetch buffer pool (item count), shared across all chains. targetBufferSize: int, + // Set on a process driving part of a split run: the chains it drives may be + // at the head while chains in another process are still backfilling, and an + // indexer switches to realtime as a whole or not at all. Cleared by the + // supervisor once every chain in the run has arrived. + mutable holdRealtime: bool, } // The whole-indexer fetch buffer pool, independent of chain count. @@ -26,16 +31,50 @@ let calculateTargetBufferSize = () => | None => 100_000 } -let make = (~chainStates, ~isRealtime, ~targetBufferSize=calculateTargetBufferSize()): t => { +let make = ( + ~chainStates, + ~isRealtime, + ~targetBufferSize=calculateTargetBufferSize(), + ~holdRealtime=false, +): t => { { chainStates, chainIds: chainStates->Dict.valuesToArray->Array.map(cs => (cs->ChainState.chainConfig).id), isRealtime, isCaughtUp: isRealtime, targetBufferSize, + holdRealtime, } } +// The supervisor's go-ahead: every chain in the run has reached the head, so +// this process may make the transitions it has been holding back. +let releaseRealtime = (crossChainState: t) => crossChainState.holdRealtime = false + +let isHoldingRealtime = (crossChainState: t) => crossChainState.holdRealtime + +// Whether this process has got as far as it can without the run's leave. What a +// supervisor reads to decide that a split run may go realtime as one. +// +// Three ways to have arrived, because a chain can be as far along as it can get +// in three different states. Its chains have caught up; or it resumed already +// realtime; or every chain is waiting to enter the reorg threshold, which is as +// far as one can fetch while the pre-threshold lag holds it at the safe block — +// entering the threshold is what lifts that lag, so a run held until its chains +// reached the head would be holding back the transition that gets them there. +// +// The process's own conclusion rather than a reading a supervisor reassembles: +// a chain committed at what was the head and resumed once the head had moved on +// has arrived, and no live reading of it can say so — which is the same reason +// `markCaughtUpOnResume` decides before any source request. +let hasArrivedAtHead = (crossChainState: t) => + crossChainState.isCaughtUp || + crossChainState.isRealtime || { + let chainStates = crossChainState.chainStates->Dict.valuesToArray + chainStates->Utils.Array.notEmpty && + chainStates->Array.every(ChainState.isReadyToEnterReorgThreshold) + } + // Resolve a chain's state by id. The id always comes from `chainIds`, which is // derived from `chainStates`, so the entry is guaranteed present. let getChainState = (crossChainState: t, chainId) => @@ -122,13 +161,29 @@ let createBatch = ( // Enter the reorg threshold: shrink each chain's buffer by its configured // blockLag and flip the flag. -let enterReorgThreshold = (crossChainState: t) => { - Logging.info("Reorg threshold reached") +// Whether every chain this process drives has buffered close enough to the head +// to enter the threshold together — and, in a split run, whether the rest of the +// run has too. Chains enter it as one indexer, so one chain still backfilling +// holds the others back whatever process it runs in. +let isReadyToEnterReorgThreshold = (crossChainState: t, ~batch) => + !crossChainState.holdRealtime && + crossChainState.chainStates + ->Dict.valuesToArray + ->Array.every(cs => cs->ChainState.isReadyToEnterReorgThresholdAfterBatch(~batch)) +// Said by each chain rather than once for the indexer: what crossing changes +// is a chain's own, and the chains of a split run cross in processes that can +// only speak for the ones they drive. +let enterReorgThreshold = (crossChainState: t) => { for i in 0 to crossChainState.chainIds->Array.length - 1 { - crossChainState - ->getChainState(crossChainState.chainIds->Array.getUnsafe(i)) - ->ChainState.enterReorgThreshold + let cs = crossChainState->getChainState(crossChainState.chainIds->Array.getUnsafe(i)) + // A chain whose end block sits below these blocks says nothing: crossing + // gives it nothing more to index, and it will never reach one of them. + let liftsCeiling = cs->ChainState.reorgThresholdLiftsCeiling + cs->ChainState.enterReorgThreshold + if liftsCeiling { + cs->ChainState.logger->Logging.childInfo(ChainState.reorgThresholdEntryMessage) + } } } @@ -150,7 +205,10 @@ let applyBatchProgress = (crossChainState: t, ~batch: Batch.t, ~blockTimestampNa } crossChainState.isCaughtUp = - crossChainState.isCaughtUp || (crossChainState->nextItemIsNone && everyChainCaughtUp.contents) + crossChainState.isCaughtUp || + (!crossChainState.holdRealtime && + crossChainState->nextItemIsNone && + everyChainCaughtUp.contents) } // Every chain has buffered up to its head (or endblock) with nothing @@ -174,8 +232,13 @@ let isSettledAtHead = (crossChainState: t) => { } // Enter the FinalizingIndexes phase without a batch, for the resume above. +// Not while the run holds this process back: the hold keeps the pre-threshold +// lag in place, and a chain that has fetched to a lagged head it was never +// going to get past reads as settled without having indexed anything. What a +// held process may conclude about where it stands, it concludes from what was +// persisted — see `markCaughtUpOnResume`, which the hold leaves alone. let markCaughtUpIfSettled = (crossChainState: t) => - if crossChainState->isSettledAtHead { + if !crossChainState.holdRealtime && crossChainState->isSettledAtHead { crossChainState.isCaughtUp = true } @@ -206,13 +269,48 @@ let markCaughtUpOnResume = (crossChainState: t) => { // and switches the indexer to realtime. let markReady = (crossChainState: t, ~readyAt) => { for i in 0 to crossChainState.chainIds->Array.length - 1 { - crossChainState - ->getChainState(crossChainState.chainIds->Array.getUnsafe(i)) - ->ChainState.markReady(~readyAt) + let cs = crossChainState->getChainState(crossChainState.chainIds->Array.getUnsafe(i)) + let wasReady = cs->ChainState.isReady + cs->ChainState.markReady(~readyAt) + + // One line per chain, because `ready_at` is one column per chain: what the + // log says and what a reader finds in the row are the same fact. + if !wasReady { + cs + ->ChainState.logger + ->Logging.childInfo("Ready. Fully indexed for queries.") + } } crossChainState.isRealtime = true } +// Each chain that has just finished indexing, said once, by the chain it is +// about — so a chain that finishes early says so then, rather than when the +// last chain in its process catches up. +let reportFinished = (crossChainState: t) => { + let waitingOnOthers = crossChainState.holdRealtime || crossChainState.chainIds->Array.length > 1 + crossChainState.chainStates + ->Dict.valuesToArray + ->Array.forEach(cs => + switch cs->ChainState.takeFinished { + | Some(EndBlock(block)) => + cs + ->ChainState.logger + ->Logging.childInfo({"msg": "Indexed to the end block. This chain is done.", "block": block}) + | Some(Backfill(block)) => + cs + ->ChainState.logger + ->Logging.childInfo({ + "msg": waitingOnOthers + ? "Finished backfill. Waiting for the other chains." + : "Finished backfill.", + "block": block, + }) + | None => () + } + ) +} + // --- Fetch control. --- // Chains ordered furthest-behind first by fetch-frontier progress, so the diff --git a/packages/envio/src/CrossChainState.resi b/packages/envio/src/CrossChainState.resi index 3f4b8365ce..320e2bdd53 100644 --- a/packages/envio/src/CrossChainState.resi +++ b/packages/envio/src/CrossChainState.resi @@ -5,7 +5,12 @@ type t let calculateTargetBufferSize: unit => int -let make: (~chainStates: dict, ~isRealtime: bool, ~targetBufferSize: int=?) => t +let make: ( + ~chainStates: dict, + ~isRealtime: bool, + ~targetBufferSize: int=?, + ~holdRealtime: bool=?, +) => t // Accessors. let chainStates: t => dict @@ -25,12 +30,17 @@ let getSafeCheckpointIdByChain: ( ) => array<(ChainId.t, option)> // Cross-chain transitions. +let releaseRealtime: t => unit +let isHoldingRealtime: t => bool +let hasArrivedAtHead: t => bool +let isReadyToEnterReorgThreshold: (t, ~batch: Batch.t) => bool let createBatch: (t, ~config: Config.t, ~frontier: Frontier.t, ~batchSizeTarget: int) => Batch.t let enterReorgThreshold: t => unit let applyBatchProgress: (t, ~batch: Batch.t, ~blockTimestampName: string) => unit let markCaughtUpIfSettled: t => unit let markCaughtUpOnResume: t => unit let markReady: (t, ~readyAt: Date.t) => unit +let reportFinished: t => unit // Fetch control. let priorityOrder: t => array diff --git a/packages/envio/src/Env.res b/packages/envio/src/Env.res index ab23da20d3..c38b1a5810 100644 --- a/packages/envio/src/Env.res +++ b/packages/envio/src/Env.res @@ -127,6 +127,10 @@ module Db = { //the SSL modes should be provided as string otherwise as 'require' | 'allow' | 'prefer' | 'verify-full' ~devFallback=Bool(false), ) + // The budget for the whole run, not for one process: a run that splits across + // workers divides it among them, and each caps its own pool to its share. + // Splitting takes two workers and a worker takes two connections, so a budget + // under 4 — the default among them — keeps the run in one process. let maxConnections = envSafe->EnvSafe.get("ENVIO_PG_MAX_CONNECTIONS", S.int, ~fallback=2) } diff --git a/packages/envio/src/FinalizeBackfill.res b/packages/envio/src/FinalizeBackfill.res index 96836f027e..c4a7383eec 100644 --- a/packages/envio/src/FinalizeBackfill.res +++ b/packages/envio/src/FinalizeBackfill.res @@ -11,9 +11,15 @@ // `envio start --chain` process is indexing and never waits on one. let runOnce = async (state: IndexerState.t) => { - Logging.info( - "All chains are caught up. Finalizing the indexer before switching to realtime: flushing pending writes, then creating the indexes the schema promises.", - ) + // Said by the process rather than by each of its chains: the indexes are one + // build over the tables, and the pause is the whole process's. A chain has + // already said it caught up, and says it is ready once this commits. The + // chains are named because the pause is theirs, and a split run has a process + // saying this for each part of it. + Logging.info({ + "msg": "Building database indexes. Indexing is paused until they are ready, which can take a while on a large database.", + "chainIds": state->IndexerState.crossChainState->CrossChainState.chainIds, + }) await Writing.flush(state) @@ -32,8 +38,8 @@ let runOnce = async (state: IndexerState.t) => { // Only after the commit: in-memory readiness must never run ahead of the // `ready_at` a restart would read back. + // Says so per chain, which is the grain `ready_at` is committed at. state->IndexerState.markReady(~readyAt) - Logging.info("The indexer is ready. Switching to realtime indexing.") } } diff --git a/packages/envio/src/IndexerLoop.res b/packages/envio/src/IndexerLoop.res index 9e979fa6c9..a2e56d4892 100644 --- a/packages/envio/src/IndexerLoop.res +++ b/packages/envio/src/IndexerLoop.res @@ -45,6 +45,8 @@ let start = (state: IndexerState.t) => { launch(state, () => FinalizeBackfill.repairSchemaIndexes(state)) } + state->IndexerState.bindScheduleProcessing(scheduleProcessing) + scheduleFetch() scheduleProcessing() } diff --git a/packages/envio/src/IndexerState.res b/packages/envio/src/IndexerState.res index b43c47320b..13d55e4108 100644 --- a/packages/envio/src/IndexerState.res +++ b/packages/envio/src/IndexerState.res @@ -115,6 +115,10 @@ type t = { // waitForNewBlock waiter is bound to the old, pre-realtime source). A fetch // response or waiter carrying an older epoch than this is discarded. mutable epoch: int, + // The loop's one door in from outside it: IndexerLoop owns scheduling and + // wires this when it starts, so an event the loop can't see for itself can + // still make it re-evaluate. A no-op before then. + mutable scheduleProcessing: unit => unit, // None off the simulate path. simulateDeadInputTracker: option, // --- Metric counters, rendered by Metrics at scrape time. --- @@ -141,6 +145,7 @@ let make = ( ~chainStates: dict, ~isRealtime: bool, ~targetBufferSize=CrossChainState.calculateTargetBufferSize(), + ~holdRealtime=false, ~committedFrontier=Frontier.empty(), ~isDevelopmentMode=false, ~shouldUseTui=false, @@ -180,11 +185,17 @@ let make = ( chainMetaDirty: false, chainMetaThrottler, isProcessing: false, - crossChainState: CrossChainState.make(~chainStates, ~isRealtime, ~targetBufferSize), + crossChainState: CrossChainState.make( + ~chainStates, + ~isRealtime, + ~targetBufferSize, + ~holdRealtime, + ), indexerStartTime: Date.make(), indexerStartTimeRef: Performance.now(), rollbackState: NoRollback, lastPrunedAtMillis: Dict.make(), + scheduleProcessing: () => (), loadManager: LoadManager.make(), keepProcessAlive: isDevelopmentMode || shouldUseTui, exitAfterFirstEventBlock, @@ -227,6 +238,9 @@ let makeFromDbState = ( ~exitAfterFirstEventBlock=false, ~reducedPollingInterval=?, ~targetBufferSize=CrossChainState.calculateTargetBufferSize(), + // A process driving part of a split run waits for its supervisor before + // entering the reorg threshold or switching to realtime. + ~holdRealtime=false, ~onError, ~onExit=?, ) => { @@ -274,6 +288,7 @@ let makeFromDbState = ( ~chainStates, ~isRealtime, ~targetBufferSize, + ~holdRealtime, ~committedFrontier=initialState.checkpointFrontier, ~isDevelopmentMode, ~shouldUseTui, @@ -500,11 +515,37 @@ let isFinalizingIndexes = (state: t) => state.crossChainState->CrossChainState.isCaughtUp && !(state.crossChainState->CrossChainState.isRealtime) +// The FinalizingIndexes phase is the transition a held process waits on: it +// ends with `ready_at` committed and the indexer realtime. +let shouldFinalizeIndexes = (state: t) => + state->isFinalizingIndexes && !(state.crossChainState->CrossChainState.isHoldingRealtime) + let markCaughtUpIfSettled = (state: t) => state.crossChainState->CrossChainState.markCaughtUpIfSettled +let isReadyToEnterReorgThreshold = (state: t, ~batch) => + state.crossChainState->CrossChainState.isReadyToEnterReorgThreshold(~batch) + +let bindScheduleProcessing = (state: t, scheduleProcessing) => + state.scheduleProcessing = scheduleProcessing + +// A process still waiting on its supervisor owes the schema the indexes its +// chains deferred, so reaching every end block doesn't make it done. +let isHoldingRealtime = (state: t) => state.crossChainState->CrossChainState.isHoldingRealtime + +let hasArrivedAtHead = (state: t) => state.crossChainState->CrossChainState.hasArrivedAtHead + +let releaseRealtime = (state: t) => { + state.crossChainState->CrossChainState.releaseRealtime + // Every chain is parked at the head with no batch coming, so nothing would + // notice the hold is gone without a pass through processing. + state.scheduleProcessing() +} + let markReady = (state: t, ~readyAt) => state.crossChainState->CrossChainState.markReady(~readyAt) +let reportFinished = (state: t) => state.crossChainState->CrossChainState.reportFinished + let rollbackState = (state: t) => state.rollbackState let indexerStartTime = (state: t) => state.indexerStartTime let loadManager = (state: t) => state.loadManager @@ -572,6 +613,7 @@ let toMetrics = (state: t): Metrics.t => { elapsedSeconds: state.indexerStartTimeRef->Performance.secondsSince, targetBufferSize: state.crossChainState->CrossChainState.targetBufferSize, isInReorgThreshold: state.crossChainState->CrossChainState.isInReorgThreshold, + hasArrivedAtHead: state.crossChainState->CrossChainState.hasArrivedAtHead, rollbackEnabled: state.config.shouldRollbackOnReorg, maxBatchSize: state.config.batchSize, preloadSeconds: state.preloadSeconds, diff --git a/packages/envio/src/IndexerState.resi b/packages/envio/src/IndexerState.resi index 99e4a35980..83748bbd64 100644 --- a/packages/envio/src/IndexerState.resi +++ b/packages/envio/src/IndexerState.resi @@ -16,6 +16,7 @@ let make: ( ~chainStates: dict, ~isRealtime: bool, ~targetBufferSize: int=?, + ~holdRealtime: bool=?, ~committedFrontier: Frontier.t=?, ~isDevelopmentMode: bool=?, ~shouldUseTui: bool=?, @@ -34,6 +35,7 @@ let makeFromDbState: ( ~exitAfterFirstEventBlock: bool=?, ~reducedPollingInterval: int=?, ~targetBufferSize: int=?, + ~holdRealtime: bool=?, ~onError: ErrorHandling.t => unit, ~onExit: unit => unit=?, ) => t @@ -93,7 +95,16 @@ let shouldSaveHistory: t => dict let isRealtime: t => bool let isFinalizingIndexes: t => bool let markCaughtUpIfSettled: t => unit +let isReadyToEnterReorgThreshold: (t, ~batch: Batch.t) => bool +let shouldFinalizeIndexes: t => bool +// Wires the loop's way back in. IndexerLoop calls this as it starts. +let bindScheduleProcessing: (t, unit => unit) => unit +let isHoldingRealtime: t => bool +let hasArrivedAtHead: t => bool +// The supervisor's go-ahead for a process driving part of a split run. +let releaseRealtime: t => unit let markReady: (t, ~readyAt: Date.t) => unit +let reportFinished: t => unit let rollbackState: t => rollbackState let indexerStartTime: t => Date.t let loadManager: t => LoadManager.t diff --git a/packages/envio/src/Main.res b/packages/envio/src/Main.res index a1562d83fc..e08dab4866 100644 --- a/packages/envio/src/Main.res +++ b/packages/envio/src/Main.res @@ -1,80 +1,3 @@ -// The public console/state chain shape. Kept to exactly this field set for -// backward compatibility with consumers like RACE — new metric fields stay off -// the HTTP response. -type chainData = { - chainId: ChainId.t, - poweredByHyperSync: bool, - firstEventBlockNumber: option, - latestProcessedBlock: option, - timestampCaughtUpToHeadOrEndblock: option, - numEventsProcessed: float, - latestFetchedBlockNumber: int, - // Need this for API backwards compatibility - @as("currentBlockHeight") - knownHeight: int, - numBatchesFetched: int, - startBlock: int, - endBlock: option, - numAddresses: int, -} -@tag("status") -type state = - | @as("disabled") Disabled({}) - | @as("initializing") Initializing({}) - | @as("active") - Active({ - envioVersion: string, - chains: array, - indexerStartTime: Date.t, - isPreRegisteringDynamicContracts: bool, - rollbackOnReorg: bool, - }) - -let toChainData = (m: Metrics.chainMetrics): chainData => { - chainId: m.chainId, - poweredByHyperSync: m.poweredByHyperSync, - firstEventBlockNumber: m.firstEventBlockNumber, - latestProcessedBlock: m.latestProcessedBlock, - timestampCaughtUpToHeadOrEndblock: m.timestampCaughtUpToHeadOrEndblock, - numEventsProcessed: m.numEventsProcessed, - latestFetchedBlockNumber: m.latestFetchedBlockNumber, - knownHeight: m.knownHeight, - numBatchesFetched: m.numBatchesFetched, - startBlock: m.startBlock, - endBlock: m.endBlock, - numAddresses: m.numAddresses, -} - -let chainDataSchema = S.schema((s): chainData => { - chainId: s.matches(ChainId.schema), - poweredByHyperSync: s.matches(S.bool), - firstEventBlockNumber: s.matches(S.option(S.int)), - latestProcessedBlock: s.matches(S.option(S.int)), - timestampCaughtUpToHeadOrEndblock: s.matches(S.option(S.datetime(S.string))), - numEventsProcessed: s.matches(S.float), - latestFetchedBlockNumber: s.matches(S.int), - knownHeight: s.matches(S.int), - numBatchesFetched: s.matches(S.int), - startBlock: s.matches(S.int), - endBlock: s.matches(S.option(S.int)), - numAddresses: s.matches(S.int), -}) -let stateSchema = S.union([ - S.literal(Disabled({})), - S.literal(Initializing({})), - S.schema(s => Active({ - envioVersion: s.matches(S.string), - chains: s.matches(S.array(chainDataSchema)), - indexerStartTime: s.matches(S.datetime(S.string)), - // Keep the field, since Dev Console expects it to be present - isPreRegisteringDynamicContracts: false, - rollbackOnReorg: s.matches(S.bool), - })), -]) - -// Runtime state lives in the process-wide `EnvioGlobal` record (shared -// across duplicate envio module instances); the slots are opaque there, so -// cast them to the real types here. let getIndexerState = () => EnvioGlobal.value.indexerState->(Utils.magic: option => option) let setIndexerState = (state: IndexerState.t) => @@ -488,111 +411,8 @@ let getGlobalIndexer = (): 'indexer => { Utils.Proxy.make(Utils.Object.createNullObject(), traps)->(Utils.magic: {..} => 'indexer) } -let startServer = ( - ~getMetrics: unit => option, - ~envioVersion: string, - ~persistence: Persistence.t, - ~isDevelopmentMode: bool, -) => { - open Express - - let app = make() - - let consoleCorsMiddleware = (req, res, next) => { - switch req.headers->Dict.get("origin") { - | Some(origin) if origin === Env.prodEnvioAppUrl || origin === Env.envioAppUrl => - res->setHeader("Access-Control-Allow-Origin", origin) - | _ => () - } - - res->setHeader("Access-Control-Allow-Methods", "GET, POST, PUT, DELETE, OPTIONS") - res->setHeader("Access-Control-Allow-Headers", "Origin, X-Requested-With, Content-Type, Accept") - - if req.method === Rest.Options { - res->sendStatus(200) - } else { - next() - } - } - app->useFor("/console", consoleCorsMiddleware) - app->useFor("/metrics", consoleCorsMiddleware) - app->useFor("/metrics/runtime", consoleCorsMiddleware) - - app->get("/healthz", (_req, res) => { - // this is the machine readable port used in kubernetes to check the health of this service. - // aditional health information could be added in the future (info about errors, back-offs, etc). - res->sendStatus(200) - }) - - app->get("/console/state", (_req, res) => { - let state = if !isDevelopmentMode { - Disabled({}) - } else { - switch getMetrics() { - | None => Initializing({}) - | Some(metrics) => - Active({ - envioVersion, - chains: metrics.chains->Array.map(toChainData), - indexerStartTime: metrics.startTime, - isPreRegisteringDynamicContracts: false, - rollbackOnReorg: metrics.rollbackEnabled, - }) - } - } - - res->json(state->S.reverseConvertToJsonOrThrow(stateSchema)) - }) - - app->post("/console/syncCache", (_req, res) => { - if isDevelopmentMode { - (persistence->Persistence.getInitializedStorageOrThrow).dumpEffectCache() - ->Promise.thenResolve(_ => res->json(Boolean(true))) - ->Promise.ignore - } else { - res->json(Boolean(false)) - } - }) - - Metrics.startRuntimeCollectors() - - app->get("/metrics", (_req, res) => { - res->set("Content-Type", Metrics.contentType) - let _ = res->endWithData(Metrics.collect(~metrics=getMetrics())) - }) - - app->get("/metrics/runtime", (_req, res) => { - res->set("Content-Type", Metrics.contentType) - let _ = res->endWithData(Metrics.collectRuntime()) - }) - - let server = app->listen(Env.serverPort) - server->Express.onError(err => { - let code = (err->(Utils.magic: JsExn.t => {..}))["code"] - if code === "EADDRINUSE" { - Logging.error( - `Port ${Env.serverPort->Int.toString} is already in use. To fix this either:` ++ - `\n 1. Kill the process using the port: lsof -ti :${Env.serverPort->Int.toString} | xargs kill -9` ++ `\n 2. Use a different port by setting the ENVIO_INDEXER_PORT environment variable: ENVIO_INDEXER_PORT=9899 envio start`, - ) - } else { - Logging.errorWithExn(err, "Failed to start indexer server") - } - NodeJs.process->NodeJs.exitWithCode(Failure) - }) -} - -type args = {@as("tui-off") tuiOff?: bool} - -type process -@val external process: process = "process" -@get external argv: process => 'a = "argv" - -type mainArgs = Yargs.parsedArgs - // The RPC-stripped public config that the storage layer persists in // `envio_info` (on initialize) and validates against (on resume). -let getEnvioInfo = () => Config.getPublicConfigJson()->Config.stripSensitiveData - let migrate = async (~reset) => { let config = Config.load() let persistence = PgStorage.makePersistenceFromConfig(~config) @@ -600,7 +420,7 @@ let migrate = async (~reset) => { ~reset, ~chainConfigs=config.chainMap->ChainMap.values, ~contractMapping=config.contractMapping, - ~envioInfo=getEnvioInfo(), + ~envioInfo=Config.envioInfo(), ~resetCommand="envio local db-migrate setup", ~runCommand=None, ~lowercaseAddresses=config.lowercaseAddresses, @@ -622,6 +442,115 @@ let dropSchema = async () => { // context, so callers should act on it (exit / re-throw) without logging again. exception FatalError(exn) +%%private( + let startIndexer = async ( + ~config: Config.t, + ~persistence: option=?, + ~reset=false, + ~isTest=false, + ~exitAfterFirstEventBlock=false, + ~patchConfig: option<(Config.t, HandlerRegister.registrationsByChainId) => Config.t>=?, + ) => { + // A worker reports to its supervisor, which draws for the whole run. + let shouldUseTui = Tui.shouldUse(~suppressed=isTest || Worker.isEnabled) + // isDevelopmentMode controls whether the indexer stays alive after all + // chains finish (keepProcessAlive) and whether the console API is exposed. + // Set by `envio dev` via the public config's `isDev` field; `envio start` + // leaves it false so the process exits cleanly when indexing completes. + let isDevelopmentMode = !isTest && config.isDev + // Initialized first so the exported indexer value contains state from the + // database when handler files are loaded (they may access the indexer at + // module top level). + let persistence = switch persistence { + | Some(p) => p + | None => PgStorage.makePersistenceFromConfig(~config) + } + setGlobalPersistence(persistence) + await persistence->Persistence.initForRun( + ~config, + ~reset, + ~isDevelopmentMode, + ~requireInitialized=config.isolated, + ) + + // Loads user handler files, which register handler/contractRegister/where + // state into the global `HandlerRegister` registry as a side effect; this + // returns that state resolved into per-chain registrations. `config` itself + // is never mutated by registration — it holds only event definitions. + let registrationsByChainId = await HandlerLoader.registerAllHandlers(~config) + let config = if isTest { + {...config, shouldRollbackOnReorg: false} + } else { + config + } + + let config = switch patchConfig { + | Some(patchConfig) => patchConfig(config, registrationsByChainId) + | None => config + } + // The single fatal-error handler, invoked once via IndexerState.errorExit. + // It logs the failure once (with chain context) and rejects the run wrapped in + // `FatalError` so callers know it's already logged — `Bin.res` just exits, the + // test worker unwraps and re-throws it to the parent thread. `runUntilFatalError` + // only ever rejects: on a clean run it stays pending and the process exits via + // ExitOnCaughtUp / when the indexer loop drains. + let onErrorReject = ref(None) + let runUntilFatalError: promise = Promise.make((_resolve, reject) => + onErrorReject := Some(reject) + ) + // `onErrorReject` is filled synchronously by `Promise.make` above, before the + // indexer can run and call `onError`, so it's always present here. + let onError = (errHandler: ErrorHandling.t) => { + errHandler->ErrorHandling.log + (onErrorReject.contents->Option.getUnsafe)(FatalError(errHandler.exn->Utils.prettifyExn)) + } + let envioVersion = Utils.EnvioPackage.value.version + + let getMetrics = () => getIndexerState()->Option.map(IndexerState.toMetrics) + let dumpEffectCache = () => + (persistence->Persistence.getInitializedStorageOrThrow).dumpEffectCache() + + // A worker reports through its supervisor, which owns the one server and the + // one display the run has. + if !isTest && !Worker.isEnabled { + Metrics.startRuntimeCollectors() + Server.startServer( + ~onSyncCache=() => dumpEffectCache()->Promise.thenResolve(ignore), + ~collectRuntime=Metrics.collectRuntime, + ~isDevelopmentMode, + ~envioVersion, + ~getMetrics, + ) + } + + let state = IndexerState.makeFromDbState( + ~config, + ~persistence, + ~initialState=persistence->Persistence.getInitializedState, + ~registrationsByChainId, + ~isDevelopmentMode, + ~shouldUseTui, + ~exitAfterFirstEventBlock, + ~holdRealtime=Worker.config->Option.mapOr(false, worker => worker.holdRealtime), + ~onError, + ) + if shouldUseTui { + let _rerender = Tui.start(~config, ~getMetrics=() => state->IndexerState.toMetrics) + } + Worker.bindRun( + ~getMetrics=() => state->IndexerState.toMetrics, + ~onReleaseRealtime=() => state->IndexerState.releaseRealtime, + ) + setIndexerState(state) + state->IndexerLoop.start + await runUntilFatalError + } +) + +// Starts this process's part of a run: the group's supervisor when the budget +// and the schema afford splitting the chains across processes, and the indexer +// itself otherwise. A worker is already one process's part, so it never splits +// again — `planForRun` refuses an isolated config. let start = async ( ~persistence: option=?, ~reset=false, @@ -629,93 +558,24 @@ let start = async ( ~exitAfterFirstEventBlock=false, ~patchConfig: option<(Config.t, HandlerRegister.registrationsByChainId) => Config.t>=?, ) => { - let mainArgs: mainArgs = process->argv->Yargs.hideBin->Yargs.yargs->Yargs.argv - let explicitTui = switch mainArgs.tuiOff { - | Some(off) => Some(!off) - | None => Env.tuiEnvVar - } - let shouldUseTui = switch (isTest, explicitTui) { - | (true, _) => false - | (_, Some(tui)) => tui - | (_, None) => !Envio.isNonInteractive() - } - // Initialize persistence first so the exported indexer value contains state from the database - // when handler files are loaded (they may access the indexer at module top level). - let config = Config.load() - // isDevelopmentMode controls whether the indexer stays alive after all - // chains finish (keepProcessAlive) and whether the console API is exposed. - // Set by `envio dev` via the public config's `isDev` field; `envio start` - // leaves it false so the process exits cleanly when indexing completes. - let isDevelopmentMode = !isTest && config.isDev - let persistence = switch persistence { - | Some(p) => p - | None => PgStorage.makePersistenceFromConfig(~config) - } - setGlobalPersistence(persistence) - await persistence->Persistence.init( - ~reset, - ~chainConfigs=config.chainMap->ChainMap.values, - ~contractMapping=config.contractMapping, - ~envioInfo=getEnvioInfo(), - ~resetCommand=isDevelopmentMode ? "envio dev -r" : "envio start -r", - ~runCommand=Some(isDevelopmentMode ? "envio dev" : "envio start"), - ~lowercaseAddresses=config.lowercaseAddresses, - ~requireInitialized=config.isolated, + // A worker parses the project's files itself rather than being handed the + // config: a public config carries every contract's ABI, which is far more + // than a spawn environment should. What the command decided rides along + // instead, and re-applying it is what makes the two configs the same one. + Worker.config->Option.forEach(({chainIds, isDev}) => + Config.prime(Config.getPublicConfigJson()->Config.withCommandFields(~chainIds, ~isDev)) ) - - // Loads user handler files, which register handler/contractRegister/where - // state into the global `HandlerRegister` registry as a side effect; this - // returns that state resolved into per-chain registrations. `config` itself - // is never mutated by registration — it holds only event definitions. - let registrationsByChainId = await HandlerLoader.registerAllHandlers(~config) - let config = if isTest { - {...config, shouldRollbackOnReorg: false} - } else { - config - } - - let config = switch patchConfig { - | Some(patchConfig) => patchConfig(config, registrationsByChainId) - | None => config - } - // The single fatal-error handler, invoked once via IndexerState.errorExit. - // It logs the failure once (with chain context) and rejects the run wrapped in - // `FatalError` so callers know it's already logged — `Bin.res` just exits, the - // test worker unwraps and re-throws it to the parent thread. `runUntilFatalError` - // only ever rejects: on a clean run it stays pending and the process exits via - // ExitOnCaughtUp / when the indexer loop drains. - let onErrorReject = ref(None) - let runUntilFatalError: promise = Promise.make((_resolve, reject) => - onErrorReject := Some(reject) - ) - // `onErrorReject` is filled synchronously by `Promise.make` above, before the - // indexer can run and call `onError`, so it's always present here. - let onError = (errHandler: ErrorHandling.t) => { - errHandler->ErrorHandling.log - (onErrorReject.contents->Option.getUnsafe)(FatalError(errHandler.exn->Utils.prettifyExn)) - } - let envioVersion = Utils.EnvioPackage.value.version - - let getMetrics = () => getIndexerState()->Option.map(IndexerState.toMetrics) - - if !isTest { - startServer(~persistence, ~isDevelopmentMode, ~envioVersion, ~getMetrics) - } - - let state = IndexerState.makeFromDbState( - ~config, - ~persistence, - ~initialState=persistence->Persistence.getInitializedState, - ~registrationsByChainId, - ~isDevelopmentMode, - ~shouldUseTui, - ~exitAfterFirstEventBlock, - ~onError, - ) - if shouldUseTui { - let _rerender = Tui.start(~config, ~getMetrics=() => state->IndexerState.toMetrics) + let config = Config.load() + switch isTest ? None : Supervisor.planForRun(~config) { + | Some(workers) => await Supervisor.run(~config, ~workers, ~reset) + | None => + await startIndexer( + ~config, + ~persistence?, + ~reset, + ~isTest, + ~exitAfterFirstEventBlock, + ~patchConfig?, + ) } - setIndexerState(state) - state->IndexerLoop.start - await runUntilFatalError } diff --git a/packages/envio/src/Metrics.res b/packages/envio/src/Metrics.res index e01e6d6cce..edee9761aa 100644 --- a/packages/envio/src/Metrics.res +++ b/packages/envio/src/Metrics.res @@ -129,6 +129,9 @@ type t = { elapsedSeconds: float, targetBufferSize: int, isInReorgThreshold: bool, + // This process has got as far as it can without the run's leave. What a + // supervisor reads to decide that a split run may go realtime as one. + hasArrivedAtHead: bool, rollbackEnabled: bool, maxBatchSize: int, preloadSeconds: float, @@ -149,6 +152,115 @@ type t = { sourceHeightStreams: array, } +// Folds items that share a key into one, keeping first-seen order so the +// rendered series doesn't reshuffle between scrapes. +let sumByKey = (items: array<'item>, ~key: 'item => string, ~add: ('item, 'item) => 'item) => { + let byKey = Dict.make() + let order = [] + items->Array.forEach(item => { + let k = item->key + switch byKey->Utils.Dict.dangerouslyGetNonOption(k) { + | Some(existing) => byKey->Dict.set(k, add(existing, item)) + | None => { + byKey->Dict.set(k, item) + order->Array.push(k) + } + } + }) + order->Array.map(k => byKey->Dict.getUnsafe(k)) +} + +// Combines the snapshots a supervised run's workers reported into the one an +// unsplit run would have produced. Series keyed by chain concatenate, since a +// chain belongs to exactly one worker; series keyed by name are summed, since +// every worker runs the same handlers and effects over its own chains. The +// clock and the buffer target are the caller's: both belong to the group, and +// the target is what the run was configured with rather than anything a worker +// could add up to. +let merge = (snapshots: array, ~startTime, ~metricTime, ~elapsedSeconds, ~targetBufferSize) => { + let concat = select => snapshots->Array.flatMap(select) + let sumInt = select => snapshots->Array.reduce(0, (acc, snapshot) => acc + snapshot->select) + let sumFloat = select => snapshots->Array.reduce(0., (acc, snapshot) => acc +. snapshot->select) + + { + startTime, + metricTime, + elapsedSeconds, + targetBufferSize, + // Chains cross into the threshold as one indexer, so a run is in it once + // every process is, the same reading as the arrival below. + isInReorgThreshold: snapshots->Utils.Array.notEmpty && + snapshots->Array.every(s => s.isInReorgThreshold), + // The run has arrived only once every process has: one still backfilling + // speaks for the whole indexer. + hasArrivedAtHead: snapshots->Utils.Array.notEmpty && + snapshots->Array.every(s => s.hasArrivedAtHead), + rollbackEnabled: snapshots->Array.some(s => s.rollbackEnabled), + maxBatchSize: snapshots->Array.reduce(0, (acc, s) => Pervasives.max(acc, s.maxBatchSize)), + preloadSeconds: sumFloat(s => s.preloadSeconds), + processingSeconds: sumFloat(s => s.processingSeconds), + processingStalledOnFetchSeconds: sumFloat(s => s.processingStalledOnFetchSeconds), + processingStalledOnStorageWriteSeconds: sumFloat(s => s.processingStalledOnStorageWriteSeconds), + rollbackSeconds: sumFloat(s => s.rollbackSeconds), + rollbackCount: sumInt(s => s.rollbackCount), + rollbackEventsCount: sumFloat(s => s.rollbackEventsCount), + chains: concat(s => s.chains), + sourceRequests: concat(s => s.sourceRequests), + sourceHeights: concat(s => s.sourceHeights), + sourceHeightStreams: concat(s => s.sourceHeightStreams), + handlers: concat(s => s.handlers)->sumByKey( + ~key=h => `${h.contract}.${h.event}`, + ~add=(a, b) => { + ...a, + processingSeconds: a.processingSeconds +. b.processingSeconds, + processingCount: a.processingCount +. b.processingCount, + preloadSeconds: a.preloadSeconds +. b.preloadSeconds, + preloadCount: a.preloadCount +. b.preloadCount, + preloadSecondsTotal: a.preloadSecondsTotal +. b.preloadSecondsTotal, + }, + ), + effects: concat(s => s.effects)->sumByKey( + ~key=e => `${e.effect}.${e.scope}`, + ~add=(a, b) => { + ...a, + callSeconds: a.callSeconds +. b.callSeconds, + callSecondsTotal: a.callSecondsTotal +. b.callSecondsTotal, + callCount: a.callCount +. b.callCount, + activeCallsCount: a.activeCallsCount + b.activeCallsCount, + queueCount: a.queueCount + b.queueCount, + queueWaitSeconds: a.queueWaitSeconds +. b.queueWaitSeconds, + invalidationsCount: a.invalidationsCount +. b.invalidationsCount, + // An effect's cache rows are per chain, so worker counts are disjoint. + // Absent unless some worker persists the cache at all. + cacheCount: switch (a.cacheCount, b.cacheCount) { + | (Some(x), Some(y)) => Some(x + y) + | (Some(x), None) => Some(x) + | (None, y) => y + }, + }, + ), + storageLoads: concat(s => s.storageLoads)->sumByKey( + ~key=l => `${l.storage}.${l.operation}`, + ~add=(a, b) => { + ...a, + seconds: a.seconds +. b.seconds, + secondsTotal: a.secondsTotal +. b.secondsTotal, + count: a.count +. b.count, + whereSize: a.whereSize +. b.whereSize, + size: a.size +. b.size, + }, + ), + storageWrites: concat(s => s.storageWrites)->sumByKey( + ~key=w => w.storage, + ~add=(a, b) => {...a, seconds: a.seconds +. b.seconds, count: a.count + b.count}, + ), + historyPrunes: concat(s => s.historyPrunes)->sumByKey( + ~key=p => p.entity, + ~add=(a, b) => {...a, seconds: a.seconds +. b.seconds, count: a.count + b.count}, + ), + } +} + // Prometheus floats keep at most 3 decimals; integral values render without a // fractional part. @inline @@ -912,199 +1024,253 @@ let getRuntimeCollectors = () => // from the beginning of the run, not from the first scrape. let startRuntimeCollectors = () => getRuntimeCollectors()->ignore -let collectRuntime = () => { - let b = {out: ""} +type runtimeGc = {kind: string, count: float, seconds: float} +type runtimeHeapSpace = {space: string, size: float, used: float, available: float} + +// One process's runtime readings at one moment. Plain numbers, so a worker can +// hand its sample to the supervisor over the fork's channel and the supervisor +// can render every process's under one label set. +type runtimeSample = { + cpuUserSeconds: float, + cpuSystemSeconds: float, + processStartTimeSeconds: float, + residentMemoryBytes: float, + heapTotalBytes: float, + heapUsedBytes: float, + externalMemoryBytes: float, + eventLoopUtilization: float, + eventLoopLagMeanSeconds: float, + eventLoopLagMinSeconds: float, + eventLoopLagMaxSeconds: float, + eventLoopLagStddevSeconds: float, + eventLoopLagP50Seconds: float, + eventLoopLagP90Seconds: float, + eventLoopLagP99Seconds: float, + heapSpaces: array, + activeResources: array<(string, float)>, + gc: array, + nodeVersion: string, +} + +let sampleRuntime = (): runtimeSample => { let memory = NodeJs.Process.memoryUsage() let cpu = NodeJs.Process.cpuUsage() let elu = NodeJs.PerfHooks.performance->NodeJs.PerfHooks.eventLoopUtilization let {eventLoopDelay, gcStats, processStartTimeSeconds} = getRuntimeCollectors() - b->single( + // Nanoseconds in the histogram; reset after sampling so each scrape reports + // the delay distribution since the previous one, matching prom-client. With + // no samples yet (e.g. the first scrape, which starts the sampler) the + // histogram reports NaN means and a sentinel min — report 0 instead. + let hasLagSamples = eventLoopDelay.max > 0. + let nsToSeconds = ns => hasLagSamples && !(ns->Float.isNaN) ? ns /. 1_000_000_000. : 0. + let sample = { + cpuUserSeconds: cpu.user /. 1_000_000., + cpuSystemSeconds: cpu.system /. 1_000_000., + processStartTimeSeconds, + residentMemoryBytes: memory.rss, + heapTotalBytes: memory.heapTotal, + heapUsedBytes: memory.heapUsed, + externalMemoryBytes: memory.external_, + eventLoopUtilization: elu.utilization, + eventLoopLagMeanSeconds: eventLoopDelay.mean->nsToSeconds, + eventLoopLagMinSeconds: eventLoopDelay.min->nsToSeconds, + eventLoopLagMaxSeconds: eventLoopDelay.max->nsToSeconds, + eventLoopLagStddevSeconds: eventLoopDelay.stddev->nsToSeconds, + eventLoopLagP50Seconds: eventLoopDelay->NodeJs.PerfHooks.percentile(50)->nsToSeconds, + eventLoopLagP90Seconds: eventLoopDelay->NodeJs.PerfHooks.percentile(90)->nsToSeconds, + eventLoopLagP99Seconds: eventLoopDelay->NodeJs.PerfHooks.percentile(99)->nsToSeconds, + heapSpaces: NodeJs.V8.getHeapSpaceStatistics()->Array.map(s => { + space: s.spaceName->String.replace("_space", ""), + size: s.spaceSize, + used: s.spaceUsedSize, + available: s.spaceAvailableSize, + }), + activeResources: { + let byType = Dict.make() + NodeJs.Process.getActiveResourcesInfo()->Array.forEach(resource => + byType->Dict.set( + resource, + byType->Utils.Dict.dangerouslyGetNonOption(resource)->Option.getOr(0.) +. 1., + ) + ) + byType->Dict.toArray + }, + gc: gcStats + ->Dict.toArray + ->Array.map(((kind, stat)) => {kind, count: stat.count, seconds: stat.seconds}), + nodeVersion: NodeJs.Process.version, + } + eventLoopDelay->NodeJs.PerfHooks.reset + sample +} + +// Renders every sample under `scope`, the labels telling one process's readings +// from another's (`worker="0"`), or "" for a run that is one process. +let renderRuntime = (samples: array<(string, runtimeSample)>) => { + let b = {out: ""} + let labels = (scope, own) => + switch (scope, own) { + | ("", "") => "" + | ("", own) | (own, "") => `{${own}}` + | (scope, own) => `{${scope},${own}}` + } + let scoped = samples->Array.map(((scope, sample)) => (labels(scope, ""), sample)) + let each = select => + samples->Array.flatMap(((scope, sample)) => + select(sample)->Array.map(((own, value)) => (labels(scope, own), value)) + ) + let gauge = (~name, ~help, ~value) => + b->series(~name, ~help, ~kind="gauge", ~entries=scoped, ~value) + let counter = (~name, ~help, ~value) => + b->series(~name, ~help, ~kind="counter", ~entries=scoped, ~value) + + counter( ~name="process_cpu_user_seconds_total", ~help="Total user CPU time spent in seconds.", - ~kind="counter", - ~value=cpu.user /. 1_000_000., + ~value=s => s.cpuUserSeconds, ) - b->single( + counter( ~name="process_cpu_system_seconds_total", ~help="Total system CPU time spent in seconds.", - ~kind="counter", - ~value=cpu.system /. 1_000_000., + ~value=s => s.cpuSystemSeconds, ) - b->single( + counter( ~name="process_cpu_seconds_total", ~help="Total user and system CPU time spent in seconds.", - ~kind="counter", - ~value=(cpu.user +. cpu.system) /. 1_000_000., + ~value=s => s.cpuUserSeconds +. s.cpuSystemSeconds, ) - b->single( + gauge( ~name="process_start_time_seconds", ~help="Start time of the process since unix epoch in seconds.", - ~kind="gauge", - ~value=processStartTimeSeconds, + ~value=s => s.processStartTimeSeconds, ) - b->single( - ~name="process_resident_memory_bytes", - ~help="Resident memory size in bytes.", - ~kind="gauge", - ~value=memory.rss, + gauge(~name="process_resident_memory_bytes", ~help="Resident memory size in bytes.", ~value=s => + s.residentMemoryBytes ) - b->single( + gauge( ~name="nodejs_heap_size_total_bytes", ~help="Process heap size from Node.js in bytes.", - ~kind="gauge", - ~value=memory.heapTotal, + ~value=s => s.heapTotalBytes, ) - b->single( + gauge( ~name="nodejs_heap_size_used_bytes", ~help="Process heap size used from Node.js in bytes.", - ~kind="gauge", - ~value=memory.heapUsed, + ~value=s => s.heapUsedBytes, ) - b->single( + gauge( ~name="nodejs_external_memory_bytes", ~help="Node.js external memory size in bytes.", - ~kind="gauge", - ~value=memory.external_, + ~value=s => s.externalMemoryBytes, ) - b->single( + gauge( ~name="nodejs_eventloop_utilization", ~help="Ratio of time the event loop is active, since process start.", - ~kind="gauge", - ~value=elu.utilization, + ~value=s => s.eventLoopUtilization, ) - // Nanoseconds in the histogram; reset after rendering so each scrape reports - // the delay distribution since the previous one, matching prom-client. With - // no samples yet (e.g. the first scrape, which starts the sampler) the - // histogram reports NaN means and a sentinel min — render 0 instead. - let hasLagSamples = eventLoopDelay.max > 0. - let nsToSeconds = ns => hasLagSamples && !(ns->Float.isNaN) ? ns /. 1_000_000_000. : 0. - b->single( + gauge( ~name="nodejs_eventloop_lag_mean_seconds", ~help="The mean of the recorded event loop delays.", - ~kind="gauge", - ~value=eventLoopDelay.mean->nsToSeconds, + ~value=s => s.eventLoopLagMeanSeconds, ) - b->single( + gauge( ~name="nodejs_eventloop_lag_min_seconds", ~help="The minimum recorded event loop delay.", - ~kind="gauge", - ~value=eventLoopDelay.min->nsToSeconds, + ~value=s => s.eventLoopLagMinSeconds, ) - b->single( + gauge( ~name="nodejs_eventloop_lag_max_seconds", ~help="The maximum recorded event loop delay.", - ~kind="gauge", - ~value=eventLoopDelay.max->nsToSeconds, + ~value=s => s.eventLoopLagMaxSeconds, ) - b->single( + gauge( ~name="nodejs_eventloop_lag_stddev_seconds", ~help="The standard deviation of the recorded event loop delays.", - ~kind="gauge", - ~value=eventLoopDelay.stddev->nsToSeconds, + ~value=s => s.eventLoopLagStddevSeconds, ) - b->single( + gauge( ~name="nodejs_eventloop_lag_p50_seconds", ~help="The 50th percentile of the recorded event loop delays.", - ~kind="gauge", - ~value=eventLoopDelay->NodeJs.PerfHooks.percentile(50)->nsToSeconds, + ~value=s => s.eventLoopLagP50Seconds, ) - b->single( + gauge( ~name="nodejs_eventloop_lag_p90_seconds", ~help="The 90th percentile of the recorded event loop delays.", - ~kind="gauge", - ~value=eventLoopDelay->NodeJs.PerfHooks.percentile(90)->nsToSeconds, + ~value=s => s.eventLoopLagP90Seconds, ) - b->single( + gauge( ~name="nodejs_eventloop_lag_p99_seconds", ~help="The 99th percentile of the recorded event loop delays.", - ~kind="gauge", - ~value=eventLoopDelay->NodeJs.PerfHooks.percentile(99)->nsToSeconds, + ~value=s => s.eventLoopLagP99Seconds, ) - eventLoopDelay->NodeJs.PerfHooks.reset - let heapSpaces = - NodeJs.V8.getHeapSpaceStatistics()->Array.map(s => ( - `{space="${s.spaceName->String.replace("_space", "")}"}`, - s, - )) + + let heapSpaces = each(s => s.heapSpaces->Array.map(h => (`space="${h.space}"`, h))) b->series( ~name="nodejs_heap_space_size_total_bytes", ~help="Process heap space size total from Node.js in bytes.", ~kind="gauge", ~entries=heapSpaces, - ~value=s => s.spaceSize, + ~value=h => h.size, ) b->series( ~name="nodejs_heap_space_size_used_bytes", ~help="Process heap space size used from Node.js in bytes.", ~kind="gauge", ~entries=heapSpaces, - ~value=s => s.spaceUsedSize, + ~value=h => h.used, ) b->series( ~name="nodejs_heap_space_size_available_bytes", ~help="Process heap space size available from Node.js in bytes.", ~kind="gauge", ~entries=heapSpaces, - ~value=s => s.spaceAvailableSize, - ) - let activeResources = { - let byType = Dict.make() - NodeJs.Process.getActiveResourcesInfo()->Array.forEach(resource => { - let label = `{type="${resource->escapeLabelValue}"}` - byType->Dict.set( - label, - byType->Utils.Dict.dangerouslyGetNonOption(label)->Option.getOr(0.) +. 1., - ) - }) - byType->Dict.toArray - } + ~value=h => h.available, + ) + b->series( ~name="nodejs_active_resources", ~help="Number of active resources that are currently keeping the event loop alive, grouped by async resource type.", ~kind="gauge", - ~entries=activeResources, + ~entries=each(s => + s.activeResources->Array.map( + ((resource, count)) => (`type="${resource->escapeLabelValue}"`, count), + ) + ), ~value=count => count, ) - b->single( + gauge( ~name="nodejs_active_resources_total", ~help="Total number of active resources.", - ~kind="gauge", - ~value=activeResources->Array.reduce(0., (acc, (_, count)) => acc +. count), - ) - let gcEntries = [] - gcStats->Utils.Dict.forEachWithKey((stat, kind) => - gcEntries->Array.push((`{kind="${kind}"}`, stat)) + ~value=s => s.activeResources->Array.reduce(0., (acc, (_, count)) => acc +. count), ) + + let gc = each(s => s.gc->Array.map(g => (`kind="${g.kind}"`, g))) b->series( ~name="nodejs_gc_duration_seconds_sum", ~help="Cumulative garbage collection pause time by kind, one of major, minor, incremental or weakcb.", ~kind="counter", - ~entries=gcEntries, - ~value=s => s.seconds, + ~entries=gc, + ~value=g => g.seconds, ) b->series( ~name="nodejs_gc_duration_seconds_count", ~help="Number of garbage collection pauses by kind, one of major, minor, incremental or weakcb.", ~kind="counter", - ~entries=gcEntries, - ~value=s => s.count, + ~entries=gc, + ~value=g => g.count, ) - let version = NodeJs.Process.version - let versionParts = version->String.replace("v", "")->String.split(".") - let versionPart = i => versionParts->Array.get(i)->Option.getOr("0") + b->series( ~name="nodejs_version_info", ~help="Node.js version info.", ~kind="gauge", - ~entries=[ - ( - `{version="${version}",major="${versionPart(0)}",minor="${versionPart( - 1, - )}",patch="${versionPart(2)}"}`, - (), - ), - ], + ~entries=each(s => { + let parts = s.nodeVersion->String.replace("v", "")->String.split(".") + let part = i => parts->Array.get(i)->Option.getOr("0") + [(`version="${s.nodeVersion}",major="${part(0)}",minor="${part(1)}",patch="${part(2)}"`, ())] + }), ~value=() => 1., ) b.out ++ "\n" } + +let collectRuntime = () => renderRuntime([("", sampleRuntime())]) diff --git a/packages/envio/src/Persistence.res b/packages/envio/src/Persistence.res index 27ad2742ab..5fd922f653 100644 --- a/packages/envio/src/Persistence.res +++ b/packages/envio/src/Persistence.res @@ -275,6 +275,10 @@ let init = { // would create rows for this process's chains only, leaving the ones it // skipped with no state for their own processes to resume. ~requireInitialized=false, + // Whether this process is the one that tells the operator the run resumed. + // A supervisor says it once for the whole run, so the workers it forked + // keep it to their own log files. + ~announceResume=true, ~startBlockRetry=StartBlockResolver.UntilItAnswers, ) => { try { @@ -324,7 +328,8 @@ let init = { | _ => false } ) { - Logging.info(`Found existing indexer storage. Resuming indexing state...`) + let logResume = announceResume ? Logging.info : Logging.trace + logResume(`Found existing indexer storage. Resuming indexing state...`) let initialState = await persistence.storage.resumeInitialState( ~entities=persistence.allEntities, ~chainIds=chainConfigs->Array.map(chain => chain.id), @@ -343,7 +348,7 @@ let init = { initialState.chains->Array.forEach(c => { progress->ChainId.Dict.set(c.id, c.progressBlockNumber) }) - Logging.info({ + logResume({ "msg": `Successfully resumed indexing state! Continuing from the last checkpoint.`, "progress": progress, }) @@ -356,6 +361,29 @@ let init = { } } +// Brings the schema up to date for a run that is about to start, as opposed to +// a migration command: what a config change prints names the command the +// operator ran, and an unreachable chain is waited on rather than reported, +// since somebody is watching the run come up. +let initForRun = ( + persistence, + ~config: Config.t, + ~reset, + ~isDevelopmentMode, + ~requireInitialized, +) => + persistence->init( + ~announceResume=!Worker.isEnabled, + ~reset, + ~chainConfigs=config.chainMap->ChainMap.values, + ~contractMapping=config.contractMapping, + ~envioInfo=Config.envioInfo(), + ~resetCommand=isDevelopmentMode ? "envio dev -r" : "envio start -r", + ~runCommand=Some(isDevelopmentMode ? "envio dev" : "envio start"), + ~lowercaseAddresses=config.lowercaseAddresses, + ~requireInitialized, + ) + let getInitializedStorageOrThrow = persistence => { switch persistence.storageStatus { | Unknown diff --git a/packages/envio/src/PgStorage.res b/packages/envio/src/PgStorage.res index e2dadb3cd7..0f63fd35da 100644 --- a/packages/envio/src/PgStorage.res +++ b/packages/envio/src/PgStorage.res @@ -1,4 +1,4 @@ -let makeClient = () => { +let makeClient = (~maxConnections=Env.Db.maxConnections) => { Postgres.makeSql( ~config={ host: Env.Db.host, @@ -14,7 +14,7 @@ let makeClient = () => { : Some(_str => ()) ), transform: {undefined: Null}, - max: Env.Db.maxConnections, + max: maxConnections, // debug: (~connection, ~query, ~params as _, ~types as _) => Js.log2(connection, query), }, ) @@ -1857,7 +1857,7 @@ let make = ( if withUpload { // Try to restore cache tables from the .envio/cache TSV files switch await scanCacheDir() { - | [] => Logging.info("No cache found to upload.") + | [] => Logging.info("No saved effect cache to load from .envio/cache.") | entries => switch await getConnectedPsqlExec(~pgUser, ~pgHost, ~pgDatabase, ~pgPort, ~containerName) { | Ok(psqlExec) => @@ -2104,7 +2104,7 @@ let make = ( switch await sql->loadCatalogRows(~indexName=name) { | rows => indexManager->IndexManager.resync(~name, ~rows) | exception exn => - Logging.debug({ + Logging.trace({ "storage": storageName, "msg": `Could not re-read the index "${name}" after a failed build. The next attempt reads it again.`, "err": exn->Utils.prettifyExn, @@ -2264,13 +2264,17 @@ let make = ( } switch missing { - | [] => + // A schema that declares no indexes has nothing to say about them, and one + // whose indexes are all in place says it once. Either way the line that + // matters is the indexer reporting itself ready, which finalization logs. + | [] if schemaIndexes->Utils.Array.notEmpty => Logging.info({ "storage": storageName, "msg": `All ${schemaIndexes ->Array.length ->Int.toString} schema indexes are already in place. Marking the indexer ready.`, }) + | [] => () | _ => Logging.info({ "storage": storageName, @@ -2318,12 +2322,16 @@ let make = ( } }) - Logging.info({ - "storage": storageName, - "msg": `Committed ${missing - ->Array.length - ->Int.toString} schema indexes and the ready timestamp in ${timeRef->formatSeconds}s.`, - }) + // Only when something was built: the wait this closes is the index build, + // and the stamp on its own is not one anybody waited through. + if missing->Utils.Array.notEmpty { + Logging.info({ + "storage": storageName, + "msg": `Committed ${missing + ->Array.length + ->Int.toString} schema indexes and the ready timestamp in ${timeRef->formatSeconds}s.`, + }) + } } let setOrThrow = ( diff --git a/packages/envio/src/Server.res b/packages/envio/src/Server.res new file mode 100644 index 0000000000..53137e622a --- /dev/null +++ b/packages/envio/src/Server.res @@ -0,0 +1,178 @@ +// The indexer's own HTTP surface: metrics for a scraper, health for an +// orchestrator, and the console's view of the run. What it serves is handed to +// it, so one process's readings and a supervised group's merged ones render the +// same way. + +// The public console/state chain shape. Kept to exactly this field set for +// backward compatibility with consumers like RACE — new metric fields stay off +// the HTTP response. +type chainData = { + chainId: ChainId.t, + poweredByHyperSync: bool, + firstEventBlockNumber: option, + latestProcessedBlock: option, + timestampCaughtUpToHeadOrEndblock: option, + numEventsProcessed: float, + latestFetchedBlockNumber: int, + // Need this for API backwards compatibility + @as("currentBlockHeight") + knownHeight: int, + numBatchesFetched: int, + startBlock: int, + endBlock: option, + numAddresses: int, +} +@tag("status") +type state = + | @as("disabled") Disabled({}) + | @as("initializing") Initializing({}) + | @as("active") + Active({ + envioVersion: string, + chains: array, + indexerStartTime: Date.t, + isPreRegisteringDynamicContracts: bool, + rollbackOnReorg: bool, + }) + +let toChainData = (m: Metrics.chainMetrics): chainData => { + chainId: m.chainId, + poweredByHyperSync: m.poweredByHyperSync, + firstEventBlockNumber: m.firstEventBlockNumber, + latestProcessedBlock: m.latestProcessedBlock, + timestampCaughtUpToHeadOrEndblock: m.timestampCaughtUpToHeadOrEndblock, + numEventsProcessed: m.numEventsProcessed, + latestFetchedBlockNumber: m.latestFetchedBlockNumber, + knownHeight: m.knownHeight, + numBatchesFetched: m.numBatchesFetched, + startBlock: m.startBlock, + endBlock: m.endBlock, + numAddresses: m.numAddresses, +} + +let chainDataSchema = S.schema((s): chainData => { + chainId: s.matches(ChainId.schema), + poweredByHyperSync: s.matches(S.bool), + firstEventBlockNumber: s.matches(S.option(S.int)), + latestProcessedBlock: s.matches(S.option(S.int)), + timestampCaughtUpToHeadOrEndblock: s.matches(S.option(S.datetime(S.string))), + numEventsProcessed: s.matches(S.float), + latestFetchedBlockNumber: s.matches(S.int), + knownHeight: s.matches(S.int), + numBatchesFetched: s.matches(S.int), + startBlock: s.matches(S.int), + endBlock: s.matches(S.option(S.int)), + numAddresses: s.matches(S.int), +}) +let stateSchema = S.union([ + S.literal(Disabled({})), + S.literal(Initializing({})), + S.schema(s => Active({ + envioVersion: s.matches(S.string), + chains: s.matches(S.array(chainDataSchema)), + indexerStartTime: s.matches(S.datetime(S.string)), + // Keep the field, since Dev Console expects it to be present + isPreRegisteringDynamicContracts: false, + rollbackOnReorg: s.matches(S.bool), + })), +]) + +let startServer = ( + ~getMetrics: unit => option, + ~envioVersion: string, + ~onSyncCache: unit => promise, + ~collectRuntime: unit => string, + ~isDevelopmentMode: bool, +) => { + open Express + + let app = make() + + let consoleCorsMiddleware = (req, res, next) => { + switch req.headers->Dict.get("origin") { + | Some(origin) if origin === Env.prodEnvioAppUrl || origin === Env.envioAppUrl => + res->setHeader("Access-Control-Allow-Origin", origin) + | _ => () + } + + res->setHeader("Access-Control-Allow-Methods", "GET, POST, PUT, DELETE, OPTIONS") + res->setHeader("Access-Control-Allow-Headers", "Origin, X-Requested-With, Content-Type, Accept") + + if req.method === Rest.Options { + res->sendStatus(200) + } else { + next() + } + } + app->useFor("/console", consoleCorsMiddleware) + app->useFor("/metrics", consoleCorsMiddleware) + app->useFor("/metrics/runtime", consoleCorsMiddleware) + + app->get("/healthz", (_req, res) => { + // this is the machine readable port used in kubernetes to check the health of this service. + // aditional health information could be added in the future (info about errors, back-offs, etc). + res->sendStatus(200) + }) + + app->get("/console/state", (_req, res) => { + let state = if !isDevelopmentMode { + Disabled({}) + } else { + switch getMetrics() { + | None => Initializing({}) + | Some(metrics) => + Active({ + envioVersion, + chains: metrics.chains->Array.map(toChainData), + indexerStartTime: metrics.startTime, + isPreRegisteringDynamicContracts: false, + rollbackOnReorg: metrics.rollbackEnabled, + }) + } + } + + res->json(state->S.reverseConvertToJsonOrThrow(stateSchema)) + }) + + app->post("/console/syncCache", (_req, res) => { + if isDevelopmentMode { + onSyncCache() + ->Promise.thenResolve(() => res->json(Boolean(true))) + // A dump that couldn't be made, or couldn't be confirmed, answers the + // same `false` a disabled console does. Leaving it unanswered would hold + // the request open for as long as the indexer runs. + ->Promise.catch(exn => { + Logging.errorWithExn(exn, "Failed to sync the effect cache") + res->json(Boolean(false)) + Promise.resolve() + }) + ->Promise.ignore + } else { + res->json(Boolean(false)) + } + }) + + app->get("/metrics", (_req, res) => { + res->set("Content-Type", Metrics.contentType) + let _ = res->endWithData(Metrics.collect(~metrics=getMetrics())) + }) + + app->get("/metrics/runtime", (_req, res) => { + res->set("Content-Type", Metrics.contentType) + let _ = res->endWithData(collectRuntime()) + }) + + let server = app->listen(Env.serverPort) + server->Express.onError(err => { + let code = (err->(Utils.magic: JsExn.t => {..}))["code"] + if code === "EADDRINUSE" { + Logging.error( + `Port ${Env.serverPort->Int.toString} is already in use. To fix this either:` ++ + `\n 1. Kill the process using the port: lsof -ti :${Env.serverPort->Int.toString} | xargs kill -9` ++ `\n 2. Use a different port by setting the ENVIO_INDEXER_PORT environment variable: ENVIO_INDEXER_PORT=9899 envio start`, + ) + } else { + Logging.errorWithExn(err, "Failed to start indexer server") + } + NodeJs.process->NodeJs.exitWithCode(Failure) + }) +} diff --git a/packages/envio/src/Supervisor.res b/packages/envio/src/Supervisor.res new file mode 100644 index 0000000000..f59cee3c96 --- /dev/null +++ b/packages/envio/src/Supervisor.res @@ -0,0 +1,542 @@ +// One worker process and the chains it drives. `maxConnections` is its slice of +// the run's connection budget, which the pool it opens is capped to. +type worker = {chainIds: array, maxConnections: int} + +// Every worker needs enough connections to read and write without serializing +// on a single one, so the budget buys workers two at a time. +let minConnectionsPerWorker = 2 + +// Most processes a run is split into, however much budget it is given. A worker +// is a whole Node process with its own heap, handler modules and source +// clients, and however many of them a run has, they share one machine. Past +// this a raised budget widens the workers' pools rather than adding workers. +let maxWorkers = 4 + +// How to spend a connection budget on the chains a run indexes. `None` keeps +// the run in one process. +// +// Chains are dealt in config order and the direction reverses each pass, so the +// first chains lead different workers and the worker that took the first picks +// up the last. How much work a chain has is the contracts' to decide, so config +// order is the only ranking the run can be given: listing chains busiest-first +// in config.yaml is what balances the layout. +let plan = (~chainIds: array, ~maxConnections: int): option> => { + let workerCount = + [chainIds->Array.length, maxConnections / minConnectionsPerWorker, maxWorkers]->Array.reduce( + maxWorkers, + Pervasives.min, + ) + if workerCount < 2 { + None + } else { + // The remainder is handed out one connection at a time rather than left + // unspent, so a budget with slack widens the earliest workers' pools. + let evenShare = maxConnections / workerCount + let remainder = mod(maxConnections, workerCount) + Some( + Array.fromInitializer(~length=workerCount, workerIndex => { + chainIds: chainIds->Array.filterWithIndex((_, dealIndex) => { + let position = mod(dealIndex, workerCount) + let isReversePass = mod(dealIndex / workerCount, 2) === 1 + (isReversePass ? workerCount - 1 - position : position) === workerIndex + }), + maxConnections: evenShare + (workerIndex < remainder ? 1 : 0), + }), + ) + } +} + +// Whether this run splits, and how. A schema that shares entities across chains +// can't be split: workers each advance their own checkpoint sequence, which only +// holds while no entity has rows another chain can reach. A run that is already +// one chain's process doesn't split again — whoever started it owns the layout. +let planForRun = (~config: Config.t, ~maxConnections=Env.Db.maxConnections) => + if config.isolated || !(config->Config.isPerChain) { + None + } else { + plan(~chainIds=config.chainMap->ChainMap.values->Array.map(chain => chain.id), ~maxConnections) + } + +// One forked worker: the process, the chains it drives, and the last snapshot +// it reported. +type running = { + worker: worker, + child: NodeJs.ChildProcess.Child.t, + mutable snapshot: option, + mutable runtime: option, + // A spawn failure can raise `error` and `exit` both, and a worker counted + // twice would end the run while its siblings are still indexing. + mutable settled: bool, +} + +let name = (worker: worker) => worker.chainIds->Array.map(ChainId.toString)->Array.joinUnsafe(";") + +let label = (worker: worker) => `[chain ${worker->name}]` + +// Workers append to files of their own. Pino writes a line per call, and +// several processes appending to one file can still tear a long line apart. +let logFilePath = (~workerIndex, ~path=Env.logFilePath) => { + let suffix = `.worker-${workerIndex->Int.toString}` + // A dot in a directory name isn't an extension, and a path with no dot at + // all has none either: `./logs/envio` takes the suffix at the end. + let dot = path->String.lastIndexOf(".") + if dot > path->String.lastIndexOf("/") { + `${path->String.slice(~start=0, ~end=dot)}${suffix}${path->String.slice( + ~start=dot, + ~end=path->String.length, + )}` + } else { + `${path}${suffix}` + } +} + +// A pipe hands over whatever has been flushed, so a chunk boundary falls +// wherever the OS put it: the tail of a chunk is a line only once the chunk +// that ends it arrives. Reading pairs with a flush, since a process that dies +// mid-line still wrote what it managed to — which is when it matters most. +let readLines = (~onLine) => { + let pending = ref("") + let read = chunk => { + let parts = (pending.contents ++ chunk)->String.split("\n") + pending := parts->Array.pop->Option.getOr("") + parts->Array.forEach(onLine) + } + let flush = () => + switch pending.contents { + | "" => () + | line => { + pending := "" + onLine(line) + } + } + (read, flush) +} + +// Each worker's share of the run's memory budgets. Both are the whole indexer's +// rather than one process's, so workers that each took the whole of one would +// hold as many times the memory as the run happened to have workers. +let memoryBudgets = (~workerCount) => + [ + ("ENVIO_INDEXING_MAX_BUFFER_SIZE", CrossChainState.calculateTargetBufferSize()), + ("ENVIO_IN_MEMORY_OBJECTS_TARGET", Env.inMemoryObjectsTarget->Float.toInt), + ]->Array.map(((name, budget)) => ( + name, + // A budget smaller than the run has workers still leaves each one something + // to hold, rather than a pool it can never put anything in. + Pervasives.max(1, budget / workerCount)->Int.toString, + )) + +// `pino-pretty` colorizes on this test, and a piped worker would fail it for a +// run the operator is watching in colour. +@val external stdoutIsTty: Nullable.t = "process.stdout.isTTY" + +let fork = ( + worker: worker, + ~workerIndex, + // How many processes the run's budgets are being split between. + ~workerCount, + // Whether this worker waits for the run before going realtime. False when + // every chain resumed already caught up: there is nothing left to wait for, + // and a barrier nobody can open would hold the run forever. + ~holdRealtime, + // Which command the run was started by, which a worker's own parse of the + // project's files can't tell it. + ~isDev, + // The entry this process was itself started from, so a worker is the same + // program as its supervisor however the package was installed. + ~entryPath=NodeJs.Process.argv->Array.getUnsafe(1), + // A run that draws a display reads its workers' output instead of letting + // them write to the terminal behind the frame's back. + ~pipeOutput=false, + ~onOutput=Console.log, + ~onErrorOutput=Console.error, + ~onSnapshot=() => (), +) => { + let env = NodeJs.Process.process.env->Dict.copy + env->Dict.set( + Worker.envVar, + { + Worker.chainIds: worker.chainIds, + holdRealtime, + isDev, + }->S.reverseConvertToJsonStringOrThrow(Worker.configSchema), + ) + // The worker's slice of the budgets. Read when the worker's own Env module + // loads, which is why they ride in the spawn environment rather than a message. + env->Dict.set("ENVIO_PG_MAX_CONNECTIONS", worker.maxConnections->Int.toString) + memoryBudgets(~workerCount)->Array.forEach(((name, share)) => env->Dict.set(name, share)) + env->Dict.set("LOG_FILE", logFilePath(~workerIndex)) + if pipeOutput && stdoutIsTty->Nullable.toOption->Option.getOr(false) { + env->Dict.set("FORCE_COLOR", "1") + } + + let child = NodeJs.ChildProcess.fork( + entryPath, + [], + { + env, + serialization: "advanced", + stdio: pipeOutput + ? ["inherit", "pipe", "pipe", "ipc"] + : ["inherit", "inherit", "inherit", "ipc"], + }, + ) + if pipeOutput { + // The supervisor writes a worker's lines the way it writes its own, which is + // the only way ink can keep them out of its frame. Each stream keeps the one + // it was written to, so a worker's errors stay on stderr for whoever is + // redirecting it. + [ + (child->NodeJs.ChildProcess.Child.stdout, onOutput), + (child->NodeJs.ChildProcess.Child.stderr, onErrorOutput), + ]->Array.forEach(((stream, onLine)) => + switch stream->Null.toOption { + | Some(stream) => { + let (read, flush) = readLines(~onLine) + stream->NodeJs.ChildProcess.Stream.setEncoding("utf8") + stream->NodeJs.ChildProcess.Stream.onData(read) + stream->NodeJs.ChildProcess.Stream.onEnd(flush) + } + | None => () + } + ) + } + let running = {worker, child, snapshot: None, runtime: None, settled: false} + child->NodeJs.ChildProcess.Child.onMessage(message => + switch message { + | Worker.Snapshot({metrics, runtime}) => { + running.snapshot = Some(metrics) + running.runtime = Some(runtime) + onSnapshot() + } + } + ) + running +} + +// The forked workers of one run, and whether their supervisor is the one +// taking them down, which is what tells an expected exit from the rest. +type group = { + // Assigned once the forks are made, which is after the group exists: a + // worker's report asks the group whether the run may go realtime. + mutable running: array, + mutable stopping: bool, + // Whether the workers are still waiting for the run's leave to go realtime. + mutable holdingRealtime: bool, +} + +let stop = group => { + group.stopping = true + group.running->Array.forEach(r => r.child->NodeJs.ChildProcess.Child.kill("SIGTERM")->ignore) +} + +// The dev console's cache dump belongs to the supervisor: a dump copies every +// effect cache table in the schema, so a worker asked to do it would copy its +// siblings' chains too, and several asked at once would write the same files at +// the same time. Overlapping requests join the dump in flight for that reason. +let syncCache = { + let inFlight = ref(None) + (~dump) => + switch inFlight.contents { + | Some(dumping) => dumping + | None => + let dumping = dump()->Promise.finally(() => inFlight := None) + inFlight := Some(dumping) + dumping + } +} + +// The supervisor handed its connections to the workers, so a dump opens one of +// its own and puts the run one connection over its budget — deliberately: only +// `envio dev` asks for a dump, and the alternative is pausing the indexing to +// free one. +let dumpCache = (~config) => { + let storage = PgStorage.makeStorageFromEnv(~config, ~sql=PgStorage.makeClient(~maxConnections=1)) + storage.dumpEffectCache()->Promise.finally(() => storage.close()->Promise.ignore) +} + +// How a group ended. `Finished` is every worker exiting cleanly on its own, +// which is what indexing to every end block looks like. +type outcome = Finished | Stopped + +// How one worker's ending reads. +type ending = + // On its own terms, or because the supervisor asked. + | Expected + // Asked to stop by someone other than the supervisor. A process manager that + // signals a whole group reaches the workers itself, so a worker can be told + // before the supervisor has decided what the signal meant. + | Stopping + | Failed + +// A worker that stops on a signal is being stopped, not failing: `systemctl +// stop` on a unit with the default `KillMode=control-group` sends SIGTERM to +// every process in it, so the workers get it directly and exit on it. Reading +// that as a failure would fail every clean shutdown under systemd. +// +// The kernel's out-of-memory killer sends SIGKILL, which stays a failure — as +// does every non-zero exit of a worker the supervisor didn't ask to stop. +// +// Once the supervisor is stopping, though, every exit reads as expected and the +// run exits 0: a worker that crashes on its way down is indistinguishable from +// one that took the SIGTERM, and a stop that reported a failure would fail +// every restart the crash happened to race. +let classifyExit = (~code: Null.t, ~signal: Null.t, ~stopping) => + switch (stopping, code->Null.toOption, signal->Null.toOption) { + | (true, _, _) + | (_, Some(0), _) => + Expected + | (_, _, Some("SIGTERM")) => Stopping + | _ => Failed + } + +// Resolves once every worker has ended. Throws if any of them ended in a way +// the supervisor didn't ask for, having first taken the rest down: one worker +// short leaves its chains unindexed, and a run that kept the others going would +// look healthy while falling behind. +let awaitExit = async (group): outcome => { + let failed = ref(false) + let alive = ref(group.running->Array.length) + + await Promise.make((resolve, _) => { + let onGone = (r, ~ending) => + if !r.settled { + r.settled = true + // The rest of the run goes down either way; what differs is whether the + // run reports itself as having failed. + switch ending { + | Expected => () + | Stopping => group->stop + | Failed => { + failed := true + group->stop + } + } + alive := alive.contents - 1 + if alive.contents === 0 { + resolve() + } + } + + group.running->Array.forEach(r => { + r.child->NodeJs.ChildProcess.Child.onExit( + (code, signal) => r->onGone(~ending=classifyExit(~code, ~signal, ~stopping=group.stopping)), + ) + r.child->NodeJs.ChildProcess.Child.onError( + exn => { + Logging.errorWithExn(exn, `${r.worker->label} failed to start`) + r->onGone(~ending=Failed) + }, + ) + }) + }) + + if failed.contents { + JsError.throwWithMessage("An indexer process exited with a failure. Stopped the others.") + } + group.stopping ? Stopped : Finished +} + +// Whether a run holding its workers back may let them go: every worker is still +// there to be released, has reported, and has got as far as it can on its own. +// +// A worker that is gone leaves the run a process short, so there is nothing to +// release it into, and its last snapshot outlives it. The channel is what says +// so: Node closes it before it reports the exit, so a process on its way out +// still reads as running everywhere else, and the release sent to it comes back +// as the error a supervisor reports as a worker failing to start. +let isRunAtHead = (running: array) => + running->Utils.Array.notEmpty && + running->Array.every(r => + r.child->NodeJs.ChildProcess.Child.connected && + r.snapshot->Option.mapOr(false, snapshot => snapshot.hasArrivedAtHead) + ) + +// Holds every worker at the head until the last of them arrives, then releases +// them together. Chains enter the reorg threshold and go realtime as one +// indexer, and in a split run only the supervisor can see when that is. +// +// Asked on every report rather than on a clock of its own: a report is the only +// thing that can change the answer. +let releaseIfAtHead = group => + if group.holdingRealtime && !group.stopping && group.running->isRunAtHead { + group.holdingRealtime = false + group.running->Array.forEach(r => + r.child->NodeJs.ChildProcess.Child.send(Worker.ReleaseRealtime)->ignore + ) + } + +// The run's chains as an unsplit indexer reports them before it has fetched +// anything: at their configured blocks, with nothing indexed. A display that +// hasn't heard from a worker yet draws these, rather than the indexer with no +// chains at all that an empty merge would render. +let configuredChains = (config: Config.t): array => + config.chainMap + ->ChainMap.values + ->Array.map((chain): Metrics.chainMetrics => { + chainId: chain.id, + poweredByHyperSync: switch chain.sourceConfig { + | EvmSourceConfig({hypersync}) => hypersync->Option.isSome + | FuelSourceConfig(_) | SvmSourceConfig(_) => true + | SimulateSourceConfig(_) | CustomSources(_) => false + }, + firstEventBlockNumber: None, + latestProcessedBlock: None, + timestampCaughtUpToHeadOrEndblock: None, + numEventsProcessed: 0., + latestFetchedBlockNumber: 0, + knownHeight: 0, + numBatchesFetched: 0, + // A chain resolves `start_block: latest` against its own head as it starts, + // which is a worker's to do and no supervisor's to guess. + startBlock: switch chain.startBlock { + | Block(block) => block + | Latest => 0 + }, + endBlock: chain.endBlock, + numAddresses: 0, + addressesByContract: [], + isReady: false, + sourceBlockNumber: 0, + progressBlockNumber: -1, + progressLatencyMs: None, + progressBlockTime: None, + concurrency: 0, + partitionsCount: 0, + bufferSize: 0, + bufferBlockNumber: -1, + idleSeconds: 0., + waitingForNewBlockSeconds: 0., + queryingSeconds: 0., + blockRangeFetchSeconds: 0., + blockRangeParseSeconds: 0., + blockRangeFetchCount: 0., + blockRangeFetchedEvents: 0., + blockRangeFetchedBlocks: 0., + reorgCount: 0, + reorgDetectedBlock: None, + rollbackTargetBlock: None, + rateLimitTimeMs: 0., + rateLimitResetInMs: None, + }) + +// Runs the group: creates the schema for every chain, forks a worker per plan +// entry, and serves the run's metrics, console and display from what they +// report. Returns once every worker has exited; throws if any of them failed. +let run = async (~config: Config.t, ~workers: array, ~reset) => { + // Every chain's state has to exist before a worker resumes it: an isolated + // run refuses to initialize, precisely so it can't create rows for its own + // chains and leave the chains it skipped with nothing to resume. It is the + // same initialization an unsplit run does, and the supervisor hands the + // connections it used to its workers. + let persistence = PgStorage.makePersistenceFromConfig(~config) + await persistence->Persistence.initForRun( + ~config, + ~reset, + ~isDevelopmentMode=config.isDev, + ~requireInitialized=false, + ) + await persistence.storage.close() + + let startTime = Date.make() + let startTimeRef = Performance.now() + + // The counts ride as fields rather than in the sentence: the connection limit + // is the only setting that decides any of this, and a reader who wants to + // change it has nothing else to go on. + Logging.info({ + "msg": "Indexing will be split across multiple processes for faster and more reliable indexing.", + "chains": config.chainMap->ChainMap.values->Array.length, + "processes": workers->Array.length, + "maxConnections": Env.Db.maxConnections, + }) + + // Decided before the first fork: it is what makes a worker's output the + // supervisor's to print. + let shouldUseTui = Tui.shouldUse() + // A run that resumed with every chain already caught up owes nobody a wait: + // its workers start realtime and there is no barrier to open. + let holdRealtime = + (persistence->Persistence.getInitializedState).chains->Array.some(chain => + chain.timestampCaughtUpToHeadOrEndblock->Option.isNone + ) + let group = {running: [], stopping: false, holdingRealtime: holdRealtime} + group.running = + workers->Array.mapWithIndex((worker, workerIndex) => + worker->fork( + ~workerIndex, + ~workerCount=workers->Array.length, + ~holdRealtime, + ~isDev=config.isDev, + ~pipeOutput=shouldUseTui, + ~onSnapshot=() => group->releaseIfAtHead, + ) + ) + + let reported = () => group.running->Array.filterMap(r => r.snapshot) + let merge = snapshots => + Metrics.merge( + snapshots, + ~startTime, + ~metricTime=Date.make(), + ~elapsedSeconds=startTimeRef->Performance.secondsSince, + // The run's pool, which its workers hold a share of each. Reporting the + // shares added back up would say the same thing less directly, and say + // nothing at all before every worker has reported. + ~targetBufferSize=CrossChainState.calculateTargetBufferSize(), + ) + + Server.startServer( + // Nothing to report until a worker has: the run reads as initializing + // rather than as an indexer with no chains. + ~getMetrics=() => + switch reported() { + | [] => None + | snapshots => Some(snapshots->merge) + }, + ~envioVersion=Utils.EnvioPackage.value.version, + // The workers' readings, each under a `worker` label: theirs are the memory + // and the event loop the indexing runs on. + ~collectRuntime=() => + Metrics.renderRuntime( + group.running->Array.filterMap(r => + r.runtime->Option.map(runtime => (`worker="${r.worker->name}"`, runtime)) + ), + ), + ~isDevelopmentMode=config.isDev, + ~onSyncCache=() => syncCache(~dump=() => dumpCache(~config)), + ) + + if shouldUseTui { + let _rerender = Tui.start(~config, ~getMetrics=() => + switch reported() { + | [] => {...[]->merge, chains: configuredChains(config)} + | snapshots => snapshots->merge + } + ) + } + + // Whichever signal asks the run to stop, the supervisor is the one that + // stops the workers: an interrupt from the terminal reaches them too, but + // they leave it to the supervisor. + NodeJs.Process.onSignal("SIGTERM", () => group->stop) + NodeJs.Process.onSignal("SIGINT", () => group->stop) + + // The server and the signal handlers would keep this process up after its + // last worker is gone, so the group's end has to end the process. A display + // is the exception, as it is for a single process: it keeps the final state + // on screen until the terminal closes it. + let outcome = await group->awaitExit + + switch outcome { + | Stopped => NodeJs.process->NodeJs.exitWithCode(Success) + | Finished if !shouldUseTui => + Logging.info("Exiting with success") + NodeJs.process->NodeJs.exitWithCode(Success) + | Finished => + // With nothing left to stop, the stop signals end the display instead. + // Registering a handler above took over from Node's default exit. + NodeJs.Process.onSignal("SIGTERM", () => NodeJs.process->NodeJs.exitWithCode(Success)) + NodeJs.Process.onSignal("SIGINT", () => NodeJs.process->NodeJs.exitWithCode(Success)) + } +} diff --git a/packages/envio/src/Worker.res b/packages/envio/src/Worker.res new file mode 100644 index 0000000000..7fb3baa0d6 --- /dev/null +++ b/packages/envio/src/Worker.res @@ -0,0 +1,101 @@ +// The worker side of a supervised run: a process the supervisor forked to drive +// a subset of the chains. It has no server and no TUI of its own — it reports +// through the IPC channel, and the supervisor is the one operational surface. + +// Set by a supervisor in the environment of the workers it forks. Internal: +// it counts only together with the fork's own channel, so a copy left in a +// shell starts nothing, and an indexer a user starts themselves takes every +// path it takes today. +let envVar = "ENVIO_INTERNAL_WORKER" + +// What the supervisor decided about this worker, handed over in the spawn +// environment rather than over the channel: it is settled before the process +// starts, and the worker needs it before it can load its own config. +type config = { + chainIds: array, + // The chains this worker drives may reach the head while chains in another + // process are still backfilling, and an indexer goes realtime as a whole or + // not at all. Cleared by the supervisor's `ReleaseRealtime`. + holdRealtime: bool, + // Which command the run was started by. A worker re-parses the project's + // files, which say nothing about that, so it can only be told — and a worker + // that took a dev run for a plain one would exit at its end block and leave + // the console it was still serving with a chain missing. + isDev: bool, +} + +let configSchema = S.object((s): config => { + chainIds: s.field("chainIds", S.array(ChainId.schema)), + holdRealtime: s.fieldOr("holdRealtime", S.bool, false), + isDev: s.field("isDev", S.bool), +}) + +// Read as this module loads, which is before anything that could catch a bare +// schema error and say where it came from. +let detect = (~env: dict, ~hasChannel) => + switch (hasChannel, env->Dict.get(envVar)) { + | (true, Some(json)) => + switch json->S.parseJsonStringOrThrow(configSchema) { + | config => Some(config) + | exception S.Raised(error) => + JsError.throwWithMessage( + `Invalid ${envVar}: ${error->S.Error.message}. It is set by an indexer supervisor for the processes it forks, and isn't meant to be set by hand.`, + ) + } + | _ => None + } + +let config = detect( + ~env=NodeJs.Process.process.env, + ~hasChannel=NodeJs.Process.channel->Nullable.toOption->Option.isSome, +) + +let isEnabled = config->Option.isSome + +@tag("kind") +type parentMessage = + // Every chain in the run has reached the head, so this worker may enter the + // reorg threshold and switch to realtime with the rest of them. + | @as("release-realtime") ReleaseRealtime + +@tag("kind") +type workerMessage = + | @as("snapshot") Snapshot({metrics: Metrics.t, runtime: Metrics.runtimeSample}) + +// How often a worker reports. Matches the TUI's own refresh, so the supervised +// display moves at the same rate an unsplit run's does. +%%private(let snapshotIntervalMillis = 500) + +// The supervisor is the one that stops a worker, and the one whose absence +// ends it. A terminal's interrupt reaches the whole group at once, so the +// worker leaves it to the supervisor, which stops every worker in turn; without +// that, a worker gone on its own would read as a failure to the supervisor +// still deciding what the interrupt meant. A supervisor that dies can't tear +// the group down, so losing the channel is what ends the worker then. +let bindToSupervisor = () => { + NodeJs.Process.onSignal("SIGINT", () => ()) + NodeJs.Process.onDisconnect(() => { + Logging.error("The indexer supervisor is gone. Stopping this chain's process.") + NodeJs.process->NodeJs.exitWithCode(Failure) + }) +} + +%%private(let send = (message: workerMessage) => NodeJs.Process.sendToParent(message)->ignore) + +// Reports this process's chains and its own runtime for as long as it runs, so +// the supervisor can merge every worker's into the one snapshot the run serves, +// and listens for the one decision the supervisor makes on the run's behalf. +// Does nothing in a process nobody forked. +let bindRun = (~getMetrics: unit => Metrics.t, ~onReleaseRealtime: unit => unit) => + if isEnabled { + Metrics.startRuntimeCollectors() + let _intervalId = setInterval( + () => send(Snapshot({metrics: getMetrics(), runtime: Metrics.sampleRuntime()})), + snapshotIntervalMillis, + ) + NodeJs.Process.onMessage((message: parentMessage) => + switch message { + | ReleaseRealtime => onReleaseRealtime() + } + ) + } diff --git a/packages/envio/src/bindings/NodeJs.res b/packages/envio/src/bindings/NodeJs.res index 1b62ca6cfd..6fd8c1ac0e 100644 --- a/packages/envio/src/bindings/NodeJs.res +++ b/packages/envio/src/bindings/NodeJs.res @@ -64,6 +64,21 @@ module Process = { @module("process") external version: string = "version" @module("process") external getActiveResourcesInfo: unit => array = "getActiveResourcesInfo" + + // Only a process forked with an IPC channel has these. Called through + // `process` rather than off a namespace import, which would drop the + // receiver Node's own implementations read. + @val @scope("process") external sendToParent: 'msg => bool = "send" + @val @scope("process") + external onMessage: (@as("message") _, 'msg => unit) => unit = "on" + @val @scope("process") external onSignal: (string, unit => unit) => unit = "on" + // Present only in a process forked with an IPC channel. + @val @scope("process") external channel: Nullable.t = "channel" + @val @scope("process") + external onDisconnect: (@as("disconnect") _, unit => unit) => unit = "on" + @val @scope("process") external argv: array = "argv" + @val @scope("process") + external emitMessage: (@as("message") _, 'msg) => bool = "emit" } module Buffer = { @@ -141,6 +156,41 @@ module ChildProcess = { @module("child_process") external execWithOptions: (string, execOptions, callback) => unit = "exec" + + // One of a child's stdio slots, present only for a slot the parent asked to + // pipe rather than inherit. + module Stream = { + type t + @send external setEncoding: (t, string) => unit = "setEncoding" + @send external onData: (t, @as("data") _, string => unit) => unit = "on" + @send external onEnd: (t, @as("end") _, unit => unit) => unit = "on" + } + + module Child = { + type t + @send external send: (t, 'msg) => bool = "send" + @send external onMessage: (t, @as("message") _, 'msg => unit) => unit = "on" + @send + external onExit: (t, @as("exit") _, (Null.t, Null.t) => unit) => unit = "on" + @send external onError: (t, @as("error") _, exn => unit) => unit = "on" + @send external kill: (t, string) => bool = "kill" + // Whether the IPC channel is still open. Node closes it before it reports + // the exit, so this goes false while the child is still running. + @get external connected: t => bool = "connected" + @get external stdout: t => Null.t = "stdout" + @get external stderr: t => Null.t = "stderr" + } + + type forkOptions = { + cwd?: string, + env?: dict, + // "advanced" uses the structured clone algorithm, so a message keeps the + // Date values a metrics snapshot carries instead of stringifying them. + serialization?: string, + stdio?: array, + } + @module("child_process") + external fork: (string, array, forkOptions) => Child.t = "fork" } module Url = { diff --git a/packages/envio/src/bindings/Yargs.res b/packages/envio/src/bindings/Yargs.res deleted file mode 100644 index 7805cab501..0000000000 --- a/packages/envio/src/bindings/Yargs.res +++ /dev/null @@ -1,8 +0,0 @@ -type arg = string - -type parsedArgs<'a> = 'a - -@module("yargs/yargs") external yargs: array => parsedArgs<'a> = "default" -@module("yargs/helpers") external hideBin: array => array = "hideBin" - -@get external argv: parsedArgs<'a> => 'a = "argv" diff --git a/packages/envio/src/db/InternalTable.res b/packages/envio/src/db/InternalTable.res index 234f193c2a..05c28ab638 100644 --- a/packages/envio/src/db/InternalTable.res +++ b/packages/envio/src/db/InternalTable.res @@ -344,7 +344,14 @@ VALUES ${valuesRows->Array.joinUnsafe(",\n ")};`, let setClauses = Array.mapWithIndex(metaFields, (field, index) => { let fieldName = (field :> string) let paramIndex = index + 2 // +2 because $1 is for id in WHERE clause - `"${fieldName}" = $${Int.toString(paramIndex)}` + switch field { + // A chain that caught up never un-catches up, so a metadata write staged + // before `markReady` and flushed after the stamp must not clear it. The + // writes race: metadata is written on a throttle of its own, outside the + // batch the finalization flushes. + | #ready_at => `"${fieldName}" = COALESCE($${Int.toString(paramIndex)}, "${fieldName}")` + | _ => `"${fieldName}" = $${Int.toString(paramIndex)}` + } }) `UPDATE "${pgSchema}"."${table.tableName}" diff --git a/packages/envio/src/tui/Tui.res b/packages/envio/src/tui/Tui.res index a01bc422ee..d42b54f5cd 100644 --- a/packages/envio/src/tui/Tui.res +++ b/packages/envio/src/tui/Tui.res @@ -248,6 +248,16 @@ module App = { } } +// Whether this process draws the progress display: `ENVIO_TUI` first, then +// whether anything is watching. A supervisor asks the same question its +// workers would have, since it is the one drawing for the run. +let shouldUse = (~suppressed=false, ~explicitTui=Env.tuiEnvVar) => + switch (suppressed, explicitTui) { + | (true, _) => false + | (_, Some(tui)) => tui + | (_, None) => !Envio.isNonInteractive() + } + let start = (~config, ~getMetrics) => { let {rerender} = render() () => { diff --git a/packages/envio/src/tui/components/SyncETA.res b/packages/envio/src/tui/components/SyncETA.res index fcb0e13586..197e9be515 100644 --- a/packages/envio/src/tui/components/SyncETA.res +++ b/packages/envio/src/tui/components/SyncETA.res @@ -1,12 +1,18 @@ open Ink let isIndexerFullySynced = (chains: array) => { - chains->Array.reduce(true, (accum, current) => { - switch current.progress { - | Synced(_) => accum - | _ => false - } - }) + switch chains { + // A supervised run draws its first frame before any worker has reported, and + // a run with nothing to report hasn't finished syncing. + | [] => false + | chains => + chains->Array.every(chain => + switch chain.progress { + | Synced(_) => true + | _ => false + } + ) + } } let getTotalRemainingBlocks = (chains: array) => { diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index d033f8ed5f..24310080d9 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -117,9 +117,6 @@ importers: viem: specifier: 2.54.0 version: 2.54.0(typescript@6.0.3) - yargs: - specifier: 17.7.2 - version: 17.7.2 devDependencies: rescript: specifier: 12.2.0 @@ -231,6 +228,19 @@ importers: specifier: 4.1.0 version: 4.1.0(@opentelemetry/api@1.9.0)(@types/node@24.12.2)(jsdom@16.7.0)(vite@7.3.1(@types/node@24.12.2)(tsx@4.21.0)) + scenarios/split_test: + dependencies: + envio: + specifier: file:../../packages/envio + version: link:../../packages/envio + devDependencies: + '@types/node': + specifier: 24.12.2 + version: 24.12.2 + typescript: + specifier: 6.0.3 + version: 6.0.3 + scenarios/svm_flow_xray: dependencies: envio: @@ -1284,10 +1294,6 @@ packages: cliui@7.0.4: resolution: {integrity: sha512-OcRE68cOsVMXp1Yvonl/fzkQOyjLSu/8bhPDfQt0e0/Eb283TKP20Fs2MqoPsr9SwA595rRCA+QMzYc9nBP+JQ==} - cliui@8.0.1: - resolution: {integrity: sha512-BSeNnyus75C4//NQ9gQt1/csTXyo/8Sb+afLAkzAptFuMsod9HFokGNudZpi/oQV73hnVK+sR+5PVRMd+Dr7YQ==} - engines: {node: '>=12'} - co@4.6.0: resolution: {integrity: sha512-QVb0dM5HvG+uaxitm8wONl7jltx8dqhfU33DcqtOZcLSVIKSDDLDi7+0LbAKiyI8hD9u42m2YxXSkMGWThaecQ==} engines: {iojs: '>= 1.0.0', node: '>= 0.12.0'} @@ -2920,18 +2926,10 @@ packages: resolution: {integrity: sha512-WOkpgNhPTlE73h4VFAFsOnomJVaovO8VqLDzy5saChRBFQFBoMYirowyW+Q9HB4HFF4Z7VZTiG3iSzJJA29yRA==} engines: {node: '>=10'} - yargs-parser@21.1.1: - resolution: {integrity: sha512-tVpsJW7DdjecAiFpbIB1e3qxIQsE6NoPc5/eTdrbbIC4h0LVsWhnoa3g+m2HclBIujHzsxZ4VJVA+GUuc2/LBw==} - engines: {node: '>=12'} - yargs@16.2.0: resolution: {integrity: sha512-D1mvvtDG0L5ft/jGWkLpG1+m0eQxOfaBvTNELraWj22wSVUMWxZUvYgJYcKh6jGGIkJFhH4IZPQhR4TKpc8mBw==} engines: {node: '>=10'} - yargs@17.7.2: - resolution: {integrity: sha512-7dSzzRQ++CKnNI/krKnYRV7JKKPUXMEh61soaHKg9mrWEhzFWhFnxPxGl+69cD1Ou63C13NUPCnmIcrvqCuM6w==} - engines: {node: '>=12'} - yoga-layout@3.2.1: resolution: {integrity: sha512-0LPOt3AxKqMdFBZA3HBAt/t/8vIKq7VaQYbuA8WxCgung+p9TVyKRYdpvCb80HcdTN2NkbIKbhNwKUfm3tQywQ==} @@ -3930,12 +3928,6 @@ snapshots: strip-ansi: 6.0.1 wrap-ansi: 7.0.0 - cliui@8.0.1: - dependencies: - string-width: 4.2.3 - strip-ansi: 6.0.1 - wrap-ansi: 7.0.0 - co@4.6.0: {} code-excerpt@4.0.0: @@ -5717,8 +5709,6 @@ snapshots: yargs-parser@20.2.4: {} - yargs-parser@21.1.1: {} - yargs@16.2.0: dependencies: cliui: 7.0.4 @@ -5729,14 +5719,4 @@ snapshots: y18n: 5.0.8 yargs-parser: 20.2.4 - yargs@17.7.2: - dependencies: - cliui: 8.0.1 - escalade: 3.2.0 - get-caller-file: 2.0.5 - require-directory: 2.1.1 - string-width: 4.2.3 - y18n: 5.0.8 - yargs-parser: 21.1.1 - yoga-layout@3.2.1: {} diff --git a/scenarios/split_test/.envio/.gitignore b/scenarios/split_test/.envio/.gitignore new file mode 100644 index 0000000000..007e7e1713 --- /dev/null +++ b/scenarios/split_test/.envio/.gitignore @@ -0,0 +1,2 @@ +# Ephemeral codegen output. Add other .envio entries here as needed. +types.d.ts diff --git a/scenarios/split_test/config.head.yaml b/scenarios/split_test/config.head.yaml new file mode 100644 index 0000000000..cd78e74893 --- /dev/null +++ b/scenarios/split_test/config.head.yaml @@ -0,0 +1,22 @@ +# yaml-language-server: $schema=../../packages/envio/evm.schema.json +name: split_test +description: The same chains with no end block, for a run that only a stop signal ends +disable_default_cross_chain: true +storage: + postgres: + column_name_format: snake_case +contracts: + - name: ERC20 + events: + - event: "Transfer(address indexed from, address indexed to, uint256 value)" +chains: + - id: 1 + start_block: 10861674 + contracts: + - name: ERC20 + address: "0x1f9840a85d5aF5bf1D1762F925BDADdC4201F984" + - id: 8453 + start_block: 10000000 + contracts: + - name: ERC20 + address: "0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913" diff --git a/scenarios/split_test/config.yaml b/scenarios/split_test/config.yaml new file mode 100644 index 0000000000..f9c29e391e --- /dev/null +++ b/scenarios/split_test/config.yaml @@ -0,0 +1,24 @@ +# yaml-language-server: $schema=../../packages/envio/evm.schema.json +name: split_test +description: Two chains over a per-chain schema, so a plain `envio start` splits them across processes +disable_default_cross_chain: true +storage: + postgres: + column_name_format: snake_case +contracts: + - name: ERC20 + events: + - event: "Transfer(address indexed from, address indexed to, uint256 value)" +chains: + - id: 1 + start_block: 10861674 + end_block: 10861774 + contracts: + - name: ERC20 + address: "0x1f9840a85d5aF5bf1D1762F925BDADdC4201F984" + - id: 8453 + start_block: 10000000 + end_block: 10000050 + contracts: + - name: ERC20 + address: "0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913" diff --git a/scenarios/split_test/envio-env.d.ts b/scenarios/split_test/envio-env.d.ts new file mode 100644 index 0000000000..c8458812c0 --- /dev/null +++ b/scenarios/split_test/envio-env.d.ts @@ -0,0 +1,7 @@ +/** + * This file is generated by HyperIndex codegen. Do not edit manually. + * It wires project-specific types from `.envio/types.d.ts` into the `envio` module. + * If your project's types look out of date, run `envio codegen` + * (or your package manager's `codegen` script, e.g. `pnpm codegen`). + */ +/// diff --git a/scenarios/split_test/package.json b/scenarios/split_test/package.json new file mode 100644 index 0000000000..0eb506550a --- /dev/null +++ b/scenarios/split_test/package.json @@ -0,0 +1,20 @@ +{ + "name": "split_test", + "version": "0.1.0", + "type": "module", + "scripts": { + "codegen": "envio codegen", + "dev": "envio dev", + "start": "envio start" + }, + "devDependencies": { + "@types/node": "24.12.2", + "typescript": "6.0.3" + }, + "dependencies": { + "envio": "file:../../packages/envio" + }, + "engines": { + "node": ">=22.0.0" + } +} diff --git a/scenarios/split_test/schema.graphql b/scenarios/split_test/schema.graphql new file mode 100644 index 0000000000..efc094d755 --- /dev/null +++ b/scenarios/split_test/schema.graphql @@ -0,0 +1,7 @@ +type Transfer { + id: ID! + from: String! + to: String! + value: BigInt! + blockNumber: Int! +} diff --git a/scenarios/split_test/src/handlers/ERC20.ts b/scenarios/split_test/src/handlers/ERC20.ts new file mode 100644 index 0000000000..4c7a4f75ee --- /dev/null +++ b/scenarios/split_test/src/handlers/ERC20.ts @@ -0,0 +1,11 @@ +import { indexer } from "envio"; + +indexer.onEvent({ contract: "ERC20", event: "Transfer" }, async ({ event, context }) => { + context.Transfer.set({ + id: `${event.chainId}-${event.block.number}-${event.logIndex}`, + from: event.params.from, + to: event.params.to, + value: event.params.value, + blockNumber: event.block.number, + }); +}); diff --git a/scenarios/split_test/tsconfig.json b/scenarios/split_test/tsconfig.json new file mode 100644 index 0000000000..cf725ad133 --- /dev/null +++ b/scenarios/split_test/tsconfig.json @@ -0,0 +1,21 @@ +{ + "compilerOptions": { + "esModuleInterop": true, + "skipLibCheck": true, + "target": "es2023", + "allowJs": true, + "resolveJsonModule": true, + "moduleDetection": "force", + "isolatedModules": true, + "verbatimModuleSyntax": true, + "strict": true, + "noUncheckedIndexedAccess": true, + "noImplicitOverride": true, + "module": "ESNext", + "moduleResolution": "bundler", + "noEmit": true, + "lib": ["es2023"], + "types": ["node"] + }, + "include": ["src", "envio-env.d.ts"] +}