Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
@@ -0,0 +1,10 @@
{
"changes": [
{
"packageName": "@rushstack/rush-client-core",
"comment": "Retry daemon restarts with jittered backoff inside the admission deadline and return a `restartRetriesExhausted` fallback outcome instead of failing when successors keep restarting for other environments.",
"type": "patch"
}
],
"packageName": "@rushstack/rush-client-core"
}
Original file line number Diff line number Diff line change
@@ -0,0 +1,10 @@
{
"changes": [
{
"packageName": "@rushstack/rush-daemon-protocol",
"comment": "Document that clients may retry `retryAfterRestart` a bounded number of times within the admission deadline.",
"type": "none"
}
],
"packageName": "@rushstack/rush-daemon-protocol"
}
Original file line number Diff line number Diff line change
@@ -0,0 +1,10 @@
{
"changes": [
{
"packageName": "@rushstack/rush-daemon",
"comment": "Queue a request whose environment requires a process restart until every queued or in-flight request that matches the running process has drained, so concurrent clients with different environments no longer preempt each other.",
"type": "patch"
}
],
"packageName": "@rushstack/rush-daemon"
}
2 changes: 1 addition & 1 deletion common/reviews/api/rush-client-core.api.md
Original file line number Diff line number Diff line change
Expand Up @@ -49,7 +49,7 @@ export type DaemonClientOutcome = {
readonly result: IDaemonCommandResult;
} | {
readonly kind: 'fallback';
readonly reason: 'unsupported' | 'controllingTerminalRequired' | 'stdinEndUnsupported';
readonly reason: 'unsupported' | 'controllingTerminalRequired' | 'stdinEndUnsupported' | 'restartRetriesExhausted';
readonly message?: string;
} | {
readonly kind: 'rejected';
Expand Down
17 changes: 10 additions & 7 deletions libraries/rush-client-core/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -24,19 +24,22 @@ or admitting stdin: that is a protocol error, not permission to replay the comma
Abort signals send `requestCancel`, then wait for the result; cancellation has a
bounded grace period. Disconnects, protocol errors and sink failures are errors,
never reasons to replay possibly executed work. Only pre-execution `unsupported`,
`controllingTerminalRequired`, and `stdinEndUnsupported` outcomes permit fallback. Raw-mode changes are
`controllingTerminalRequired`, `stdinEndUnsupported`, and `restartRetriesExhausted` outcomes permit fallback. Raw-mode changes are
acknowledged only after applying them. Input listeners and raw state are restored
on success, cancellation, disconnect and failure. No resize messages are sent.

`executeWithDaemonRestartAsync(readyClient, connectionOptions, executionOptions)`
adds one bounded retry for an explicit `retryAfterRestart: true` result. It captures
the endpoint's PID/start identity before sending, requires protocol 0.10, waits for
that ownership to be released, and reconnects through the same startup mutex.
retries an explicit `retryAfterRestart: true` result a bounded number of times. Before
each hand-off it captures the endpoint's PID/start identity, requires protocol 0.10,
waits for that ownership to be released, and reconnects through the same startup mutex.
Retries after the first use jittered backoff, and the backoff, the successor hand-off
and the resubmitted request all share the request's admission deadline.
The original immutable request and unread input are preserved. Output, events,
terminal control, or stdin admission forbid retry, as do connection loss and plain
error messages. A second restart result fails explicitly. Cancellation stops waiting
without killing a daemon. Disabling auto-start still permits waiting for a
host-started successor, but never lets the client spawn one.
error messages. When the retry bound or the admission deadline is exhausted, it
returns a `restartRetriesExhausted` fallback outcome so the caller can run in-process.
Cancellation stops waiting without killing a daemon. Disabling auto-start still
permits waiting for a host-started successor, but never lets the client spawn one.

`connectOrStartDaemonAsync()` accepts an **explicit, version-selected** executable,
arguments, environment and cwd. It does not discover or install a Rush version.
Expand Down
6 changes: 5 additions & 1 deletion libraries/rush-client-core/src/DaemonClient.ts
Original file line number Diff line number Diff line change
Expand Up @@ -67,7 +67,11 @@ export type DaemonClientOutcome =
| { readonly kind: 'result'; readonly result: IDaemonCommandResult }
| {
readonly kind: 'fallback';
readonly reason: 'unsupported' | 'controllingTerminalRequired' | 'stdinEndUnsupported';
readonly reason:
| 'unsupported'
| 'controllingTerminalRequired'
| 'stdinEndUnsupported'
| 'restartRetriesExhausted';
readonly message?: string;
}
| { readonly kind: 'rejected'; readonly rejection: IDaemonRequestRejectedMessage['payload'] };
Expand Down
173 changes: 120 additions & 53 deletions libraries/rush-client-core/src/executeWithDaemonRestart.ts
Original file line number Diff line number Diff line change
@@ -1,6 +1,8 @@
// Copyright (c) Microsoft Corporation. All rights reserved. Licensed under the MIT license.
// See LICENSE in the project root for license information.

import { setTimeout as delayAsync } from 'node:timers/promises';

import { DAEMON_WORKSPACE_RESTART_PROTOCOL_MINOR } from '@rushstack/rush-daemon-protocol';
import { readDaemonLockfile, type IDaemonLockfile } from '@rushstack/rush-daemon-transport';

Expand All @@ -10,8 +12,20 @@ import type { DaemonClient, DaemonClientOutcome, IDaemonClientExecuteOptions } f
import { DaemonClientError } from './DaemonClientError';

/**
* Executes on a ready client, retrying once only for a typed pre-execution restart.
* The maximum number of successors a single request follows. Each restart serves at least one other
* environment first, so this bounds the wait when several environments share one workspace daemon.
*/
const MAX_RESTART_RETRIES: number = 6;
const RETRY_JITTER_BASE_MS: number = 50;
const RETRY_JITTER_MAX_MS: number = 1000;
/** Matches the default of {@link IConnectOrStartDaemonOptions.startupTimeoutMs}. */
const DEFAULT_STARTUP_TIMEOUT_MS: number = 15000;

/**
* Executes on a ready client, retrying only for a typed pre-execution restart.
* Preserves the original request, callbacks and unread input; never retries connection loss.
* Restarts are retried with jittered backoff inside the request's admission deadline; once the
* retries or the deadline are exhausted, a `fallback` outcome lets the caller run in-process instead.
* The connection options must select the request's expected daemon and startup environment.
* @beta
*/
Expand All @@ -25,69 +39,122 @@ export async function executeWithDaemonRestartAsync(
execution.abortSignal && connection.abortSignal
? AbortSignal.any([execution.abortSignal, connection.abortSignal])
: (execution.abortSignal ?? connection.abortSignal);
const waitTimeoutMs: number | undefined = execution.request.admission?.waitTimeoutMs;
let owner: IDaemonLockfile | undefined = await attestOwnerAsync(client, connection);
let outcome: DaemonClientOutcome = await client.executeAsync({ ...execution, abortSignal });
let previous: DaemonClient | undefined;
try {
for (let retry: number = 1; outcome.kind === 'result' && outcome.result.retryAfterRestart; retry++) {
if (!owner) {
throw new DaemonClientError(
'startupFailed',
'Cannot attest the restarting daemon ownership; the request was not retried.'
);
}
if (abortSignal?.aborted) return abortedOutcome(execution);
const getRemainingMs = (): number | undefined =>
waitTimeoutMs === undefined ? undefined : waitTimeoutMs - (Date.now() - startedAt);
if (retry > MAX_RESTART_RETRIES || isExpired(getRemainingMs())) {
return restartExhaustedOutcome(retry - 1);
}
let successor: DaemonClient;
let boundedByAdmission: boolean = false;
try {
// The first retry follows the planned successor immediately; later ones back off with jitter so
// clients whose environments differ do not reach each new successor in lockstep.
if (retry > 1) {
await delayAsync(getRetryDelayMs(retry, getRemainingMs()), undefined, { signal: abortSignal });
}
const remainingMs: number | undefined = getRemainingMs();
if (isExpired(remainingMs)) return restartExhaustedOutcome(retry - 1);
const startupTimeoutMs: number = connection.startupTimeoutMs ?? DEFAULT_STARTUP_TIMEOUT_MS;
// The successor handoff shares the request's admission deadline rather than starting a fresh one.
boundedByAdmission = remainingMs !== undefined && remainingMs < startupTimeoutMs;
successor = await connectOrStartDaemonAsync({
...connection,
startupTimeoutMs: boundedByAdmission ? Math.max(1, Math.ceil(remainingMs!)) : startupTimeoutMs,
previousDaemon: { pid: owner.pid, startedAt: owner.startedAt },
abortSignal
});
} catch (error) {
if (
abortSignal?.aborted &&
(error === abortSignal.reason ||
(typeof error === 'object' && error !== null && 'code' in error && error.code === 'ABORT_ERR'))
) {
return abortedOutcome(execution);
}
// A startup error after the admission deadline expired is the deadline, not a new failure mode.
if (boundedByAdmission && error instanceof DaemonClientError && isExpired(getRemainingMs())) {
return restartExhaustedOutcome(retry);
}
throw error;
} finally {
await previous?.closeAsync().catch(() => undefined);
previous = undefined;
}
previous = successor;
const remainingMs: number | undefined = getRemainingMs();
if (isExpired(remainingMs)) return restartExhaustedOutcome(retry);
owner = await attestOwnerAsync(successor, connection);
outcome = await successor.executeAsync({
...execution,
abortSignal,
request:
remainingMs === undefined
? execution.request
: captureDaemonRequest({
...execution.request,
admission: { ...execution.request.admission, waitTimeoutMs: Math.floor(remainingMs) }
})
});
}
return outcome;
} finally {
await previous?.closeAsync().catch(() => undefined);
}
}

function isExpired(remainingMs: number | undefined): boolean {
return remainingMs !== undefined && remainingMs <= 0;
}
/** Returns the published ownership record only when it names the connected, restart-capable process. */
async function attestOwnerAsync(
client: DaemonClient,
connection: IConnectOrStartDaemonOptions
): Promise<IDaemonLockfile | undefined> {
const owner: IDaemonLockfile | undefined = readDaemonLockfile(connection.paths.lockfilePath);
const { pid } = await client.status;
const attested: boolean =
client.protocolVersion.minor >= DAEMON_WORKSPACE_RESTART_PROTOCOL_MINOR &&
return client.protocolVersion.minor >= DAEMON_WORKSPACE_RESTART_PROTOCOL_MINOR &&
owner !== undefined &&
owner.pid === pid &&
owner.socketPath === connection.paths.socketPath &&
Number.isSafeInteger(owner.pid) &&
owner.pid > 0 &&
Number.isFinite(Date.parse(owner.startedAt));
const outcome: DaemonClientOutcome = await client.executeAsync({ ...execution, abortSignal });
if (outcome.kind !== 'result' || !outcome.result.retryAfterRestart) return outcome;
if (!attested || !owner) {
throw new DaemonClientError(
'startupFailed',
'Cannot attest the restarting daemon ownership; the request was not retried.'
);
}
if (abortSignal?.aborted) return abortedOutcome(execution);
let successor: DaemonClient;
try {
successor = await connectOrStartDaemonAsync({
...connection,
previousDaemon: { pid: owner.pid, startedAt: owner.startedAt },
abortSignal
});
} catch (error) {
if (
abortSignal?.aborted &&
(error === abortSignal.reason ||
(typeof error === 'object' && error !== null && 'code' in error && error.code === 'ABORT_ERR'))
) {
return abortedOutcome(execution);
}
throw error;
}
const waitTimeoutMs: number | undefined = execution.request.admission?.waitTimeoutMs;
const result: DaemonClientOutcome = await successor.executeAsync({
...execution,
abortSignal,
request:
waitTimeoutMs === undefined
? execution.request
: captureDaemonRequest({
...execution.request,
admission: {
...execution.request.admission,
waitTimeoutMs: Math.max(0, waitTimeoutMs - (Date.now() - startedAt))
}
})
});
if (result.kind === 'result' && result.result.retryAfterRestart) {
throw new DaemonClientError(
'startupFailed',
'The successor requested another restart; the single safe retry was exhausted.'
);
}
return result;
Number.isFinite(Date.parse(owner.startedAt))
? owner
: undefined;
}

function getRetryDelayMs(retry: number, remainingMs: number | undefined): number {
const ceilingMs: number = Math.min(RETRY_JITTER_MAX_MS, RETRY_JITTER_BASE_MS * 2 ** (retry - 1));
const delayMs: number = Math.floor(ceilingMs / 2 + (Math.random() * ceilingMs) / 2);
return remainingMs === undefined ? delayMs : Math.max(0, Math.min(delayMs, remainingMs - 1));
}

function restartExhaustedOutcome(restarts: number): DaemonClientOutcome {
return {
kind: 'fallback',
reason: 'restartRetriesExhausted',
message: `The daemon was still restarting for other environments after ${restarts} ${
restarts === 1 ? 'restart' : 'restarts'
}; no operation was started by the daemon`
};
}

function abortedOutcome(execution: IDaemonClientExecuteOptions): DaemonClientOutcome {
return {
kind: 'result',
result: { requestId: execution.request.requestId, exitCode: 130, outcome: 'aborted', aborted: true }
};
}
}
Loading
Loading