diff --git a/README.md b/README.md index 1f46bad3013..981421ea761 100644 --- a/README.md +++ b/README.md @@ -28,6 +28,8 @@ for large scale TypeScript monorepos. - [API Documenter](https://api-extractor.com/pages/setup/generating_docs/) - use TSDoc comments to publish an API documentation website - [Lockfile Explorer](https://lfx.rushstack.io/) - investigate and solve version conflicts for PNPM lockfiles - [TSDoc](https://tsdoc.org/) - the standard for doc comments in TypeScript code +- [Dogfooding the Rush daemon](./docs/rush/dogfooding-rush-daemon.md) - contributor guide for building this repo + with the opt-in `rush-client` daemon, built from source ## Related Repos diff --git a/apps/heft/src/cli/HeftCommandLineParser.test.ts b/apps/heft/src/cli/HeftCommandLineParser.test.ts new file mode 100644 index 00000000000..7be596af5ee --- /dev/null +++ b/apps/heft/src/cli/HeftCommandLineParser.test.ts @@ -0,0 +1,140 @@ +// Copyright (c) Microsoft Corporation. All rights reserved. Licensed under the MIT license. +// See LICENSE in the project root for license information. + +import * as childProcess from 'node:child_process'; +import * as fs from 'node:fs'; +import * as os from 'node:os'; +import * as path from 'node:path'; + +jest.setTimeout(30_000); + +const HEFT_START_PATH: string = path.resolve(__dirname, '../start.js'); + +// A project with one task. The task sets process.exitCode to FIXTURE_EXIT_CODE in the Heft process, the +// way a test that Jest runs in band can, and then fails if FIXTURE_FAIL is set. +const FIXTURE_FILES: Record = { + 'package.json': JSON.stringify({ name: 'heft-exit-code-fixture', version: '1.0.0', private: true }), + 'heft-plugin.json': JSON.stringify({ + taskPlugins: [{ pluginName: 'exit-code-fixture-plugin', entryPoint: './plugin.js' }] + }), + 'config/heft.json': JSON.stringify({ + phasesByName: { + test: { tasksByName: { fixture: { taskPlugin: { pluginPackage: 'heft-exit-code-fixture' } } } } + } + }), + 'plugin.js': `module.exports = class { + apply(taskSession) { + taskSession.hooks.run.tapPromise('exit-code-fixture-plugin', async () => { + process.exitCode = Number(process.env.FIXTURE_EXIT_CODE); + if (process.env.FIXTURE_FAIL) { + throw new Error('The fixture task failed on purpose'); + } + }); + } +}; +` +}; + +// The same project, but its heft.json declares an alias with the name of the "test" phase's action, so Heft fails +// before any task runs, with an error that nothing has reported yet. +const ALIAS_CLASH_FIXTURE_FILES: Record = { + ...FIXTURE_FILES, + 'config/heft.json': JSON.stringify({ + ...JSON.parse(FIXTURE_FILES['config/heft.json']), + aliasesByName: { test: { actionName: 'test' } } + }) +}; + +function createFixtureFolder(files: Record): string { + const fixtureFolderPath: string = fs.mkdtempSync(path.join(os.tmpdir(), 'heft-exit-code-')); + for (const [relativePath, contents] of Object.entries(files)) { + const filePath: string = path.join(fixtureFolderPath, relativePath); + fs.mkdirSync(path.dirname(filePath), { recursive: true }); + fs.writeFileSync(filePath, contents); + } + return fixtureFolderPath; +} + +interface IHeftResult { + exitCode: number | undefined; + output: string; +} + +describe('HeftCommandLineParser', () => { + let fixtureFolderPath: string; + let aliasClashFixtureFolderPath: string; + + beforeAll(() => { + fixtureFolderPath = createFixtureFolder(FIXTURE_FILES); + aliasClashFixtureFolderPath = createFixtureFolder(ALIAS_CLASH_FIXTURE_FILES); + }); + + afterAll(() => { + fs.rmSync(fixtureFolderPath, { recursive: true, force: true }); + fs.rmSync(aliasClashFixtureFolderPath, { recursive: true, force: true }); + }); + + async function runHeftTestAsync( + taskExitCode: number, + taskFails: boolean, + folderPath: string = fixtureFolderPath + ): Promise { + const env: NodeJS.ProcessEnv = { ...process.env, FIXTURE_EXIT_CODE: String(taskExitCode) }; + delete env.FIXTURE_FAIL; + delete env._RUSH_REPORTER_CHILD_FD; + delete env._RUSH_REPORTER_CHILD_ACK_FD; + if (taskFails) { + env.FIXTURE_FAIL = '1'; + } + + const child: childProcess.ChildProcess = childProcess.spawn(process.execPath, [HEFT_START_PATH, 'test'], { + cwd: folderPath, + env, + stdio: ['ignore', 'pipe', 'pipe'] + }); + let output: string = ''; + const appendOutput: (chunk: string) => void = (chunk: string) => { + output += chunk; + }; + child.stdout?.setEncoding('utf8').on('data', appendOutput); + child.stderr?.setEncoding('utf8').on('data', appendOutput); + const exitCode: number | null = await new Promise((resolve, reject) => { + child.once('error', reject); + child.once('close', resolve); + }); + + return { exitCode: exitCode ?? undefined, output }; + } + + it('exits with 1 when a task fails after code in the Heft process set process.exitCode to 0', async () => { + const { exitCode, output }: IHeftResult = await runHeftTestAsync(0, true); + expect(output).toContain('The fixture task failed on purpose'); + expect(exitCode).toBe(1); + }); + + it('exits with 1 when a task fails after code in the Heft process set a negative process.exitCode', async () => { + const { exitCode, output }: IHeftResult = await runHeftTestAsync(-1, true); + expect(output).toContain('The fixture task failed on purpose'); + expect(exitCode).toBe(1); + }); + + it('keeps a positive process.exitCode when a task fails', async () => { + const { exitCode, output }: IHeftResult = await runHeftTestAsync(3, true); + expect(output).toContain('The fixture task failed on purpose'); + expect(exitCode).toBe(3); + }); + + it('exits with 0 when the task succeeds after code in the Heft process set process.exitCode to 0', async () => { + const { exitCode, output }: IHeftResult = await runHeftTestAsync(0, false); + expect(output).not.toContain('The fixture task failed on purpose'); + expect(exitCode).toBe(0); + }); + + it('writes an error that nothing reported before it exits, and exits with 1', async () => { + const { exitCode, output }: IHeftResult = await runHeftTestAsync(0, false, aliasClashFixtureFolderPath); + expect(output).toContain( + 'The alias "test" specified in heft.json cannot be used because an action with that name already exists.' + ); + expect(exitCode).toBe(1); + }); +}); diff --git a/apps/heft/src/cli/HeftCommandLineParser.ts b/apps/heft/src/cli/HeftCommandLineParser.ts index 6b7ada39707..7aec0c77f7c 100644 --- a/apps/heft/src/cli/HeftCommandLineParser.ts +++ b/apps/heft/src/cli/HeftCommandLineParser.ts @@ -26,7 +26,7 @@ import { PhaseAction } from './actions/PhaseAction'; import { RunAction } from './actions/RunAction'; import type { IHeftActionOptions } from './actions/IHeftAction'; import { AliasAction } from './actions/AliasAction'; -import { getToolParameterNamesFromArgs } from '../utilities/CliUtilities'; +import { getErrorExitCode, getToolParameterNamesFromArgs } from '../utilities/CliUtilities'; import { Constants } from '../utilities/Constants'; import { HeftChildReporter } from '../pluginFramework/logging/HeftChildReporter'; @@ -271,11 +271,6 @@ export class HeftCommandLineParser extends CommandLineParser { this.globalTerminal.writeErrorLine(error.stack!); } - const exitCode: string | number | undefined = process.exitCode; - if (!exitCode || typeof exitCode !== 'number' || exitCode > 0) { - process.exit(exitCode); - } else { - process.exit(1); - } + process.exit(getErrorExitCode(process.exitCode)); } } diff --git a/apps/heft/src/index.ts b/apps/heft/src/index.ts index 0c6006dfc8c..7ccd112e96b 100644 --- a/apps/heft/src/index.ts +++ b/apps/heft/src/index.ts @@ -63,6 +63,13 @@ export type { IReaddirOptions } from './utilities/WatchFileSystemAdapter'; +export { + type IWatchpackPendingEventState as _IWatchpackPendingEventState, + type IWatchpackPendingFileEvent as _IWatchpackPendingFileEvent, + _tryGetWatchpackPendingEventState, + _waitForWatchpackPendingEventsAsync +} from './utilities/WatchpackUtilities'; + export { type IHeftRecordMetricsHookOptions, type IMetricsData, diff --git a/apps/heft/src/utilities/CliUtilities.ts b/apps/heft/src/utilities/CliUtilities.ts index d322dc45e1f..d2a0b1e4fb1 100644 --- a/apps/heft/src/utilities/CliUtilities.ts +++ b/apps/heft/src/utilities/CliUtilities.ts @@ -20,3 +20,15 @@ export function getToolParameterNamesFromArgs(argv: string[] = process.argv): Se } return toolParameters; } + +/** + * Get the exit code for a Heft process that is exiting because of an error. This is `exitCode` if it + * is a positive integer, and 1 otherwise, so it is never 0. + * + * @param exitCode - The current value of `process.exitCode`. Code that ran in the Heft process, such as + * a test that Jest ran in band, may have set it to 0. + */ +export function getErrorExitCode(exitCode: string | number | undefined): number { + const numericExitCode: number = Number(exitCode); + return Number.isInteger(numericExitCode) && numericExitCode > 0 ? numericExitCode : 1; +} diff --git a/apps/heft/src/utilities/WatchFileSystemAdapter.ts b/apps/heft/src/utilities/WatchFileSystemAdapter.ts index cc886e24a27..25ef14692ab 100644 --- a/apps/heft/src/utilities/WatchFileSystemAdapter.ts +++ b/apps/heft/src/utilities/WatchFileSystemAdapter.ts @@ -6,6 +6,8 @@ import * as path from 'node:path'; import Watchpack from 'watchpack'; +import { _tryGetWatchpackPendingEventState } from './WatchpackUtilities'; + /** * Options for `fs.readdir` * @public @@ -129,6 +131,96 @@ interface ITimeEntry { safeTime: number; } +/** + * What a run read. The run's watcher checks the changes that it reports against this. + */ +interface IRunInputs { + /** + * When the run started. The watcher reports files whose times are close to this time, or later. + */ + baseline: number | undefined; + /** + * The time of each file that the run read, as the run recorded it + */ + files: ReadonlyMap; + /** + * The folders that the run read + */ + contexts: ReadonlyMap; + /** + * The times that the previous watcher had when the run started + */ + times: ReadonlyMap | undefined; +} + +const OUTDATED_ON_ATTACH_EXPLANATION: string = 'watch (outdated on attach)'; +// A new watcher's first scan reports each file whose mtime is later than the watcher's start time, less the file +// system's accuracy. A watcher that attaches to the watcher of a folder that has already scanned reports the +// folder in the same way. Watchpack's other explanations are for events from the file system. +const SCAN_EXPLANATIONS: ReadonlySet = new Set(['scan (file)', OUTDATED_ON_ATTACH_EXPLANATION]); + +function tryLstatSync(filePath: string): fs.Stats | undefined { + try { + return fs.lstatSync(filePath, { throwIfNoEntry: false }); + } catch { + return undefined; + } +} + +function getTimestamp(stats: fs.Stats): number { + return stats.mtime.getTime() || stats.ctime.getTime() || Date.now(); +} + +/** + * Watchpack records a file's new time only when its `fs.lstat()` after the file system's event finishes. If a + * run starts before then, the previous watcher's time for the file is from before the change. Read the time + * again, so that the run sees the change, and so that the run's watcher knows that the run saw it. + */ +function refreshPendingTimes(watcher: Watchpack, times: Map): void { + const pendingEventState: ReturnType = + _tryGetWatchpackPendingEventState(watcher); + if (!pendingEventState) { + return; + } + + for (const { filePath } of pendingEventState.pendingFileEvents) { + const stats: fs.Stats | undefined = tryLstatSync(filePath); + if (!stats) { + times.delete(filePath); + } else if (!stats.isDirectory()) { + const timestamp: number = getTimestamp(stats); + times.set(filePath, { timestamp, safeTime: timestamp }); + } + } +} + +/** + * Returns false if the watcher reported the path only because the path's time is close to the run's start, and + * the run has already read the path as it is now. + */ +function isReportedChangeNew(filePath: string, explanation: string, inputs: IRunInputs): boolean { + if (!SCAN_EXPLANATIONS.has(explanation) || inputs.baseline === undefined) { + return true; + } + + const stats: fs.Stats | undefined = tryLstatSync(filePath); + if (!stats) { + return true; + } + + if (stats.isDirectory()) { + // The watcher of a folder that the run read reports the folder's files one by one + return explanation !== OUTDATED_ON_ATTACH_EXPLANATION || !inputs.contexts.has(filePath); + } + + if (stats.ctimeMs >= inputs.baseline) { + return true; + } + + const readTime: number | undefined = inputs.files.get(filePath) ?? inputs.times?.get(filePath)?.timestamp; + return readTime !== getTimestamp(stats); +} + /** * A filesystem adapter for use with the "fast-glob" package. This adapter tracks file system accesses * to initialize `watchpack`. @@ -261,11 +353,21 @@ export class WatchFileSystemAdapter implements IWatchFileSystemAdapter { public setBaseline(): void { this.#lastQueryTime = Date.now(); - if (this.#watcher) { - const times: Map = new Map(); - this.#watcher.pause(); - this.#watcher.collectTimeInfoEntries(times, times); + const watcher: Watchpack | undefined = this.#watcher; + if (watcher) { + this.#watcher = undefined; + const times: Map = new Map(); + watcher.collectTimeInfoEntries(times, times); + refreshPendingTimes(watcher, times); + // Close the previous watcher instead of only pausing it. A paused watcher keeps its directory watchers, + // so every run would leak another set of them. A kept watcher also keeps its OS watch on a folder that + // was deleted and recreated, and later watchers on that folder would share the dead watch, so edits in + // the recreated folder would be reported one run late. + watcher.close(); this.#times = times; + } else { + // Nothing is watching, so times collected for an earlier run may be out of date. + this.#times = undefined; } } @@ -277,11 +379,39 @@ export class WatchFileSystemAdapter implements IWatchFileSystemAdapter { return; } + const inputs: IRunInputs = { + baseline: this.#lastQueryTime, + files: this.#files, + contexts: this.#contexts, + times: this.#times + }; + const watcher: Watchpack = new Watchpack({ aggregateTimeout: 0, followSymlinks: false }); + // A file that changed just before the run started is reported again when the watcher starts, although the + // run has read the change. Call onChange only for a change that the run may not have seen. + let hasNewChange: boolean = false; + const onReportedChange = (filePath: string, modifiedTime: number, explanation: string): void => { + hasNewChange ||= isReportedChangeNew(filePath, explanation, inputs); + }; + const onReportedRemove = (): void => { + hasNewChange = true; + }; + const onAggregated = (): void => { + if (hasNewChange) { + watcher.off('change', onReportedChange); + watcher.off('remove', onReportedRemove); + watcher.off('aggregated', onAggregated); + onChange(); + } + }; + watcher.on('change', onReportedChange); + watcher.on('remove', onReportedRemove); + watcher.on('aggregated', onAggregated); + this.#watcher = watcher; watcher.watch({ files: this.#files.keys(), @@ -290,12 +420,11 @@ export class WatchFileSystemAdapter implements IWatchFileSystemAdapter { startTime: this.#lastQueryTime }); + // The watcher checks its reports against `inputs`, so the next run records into new maps this.#lastFiles = this.#files; this.#files = new Map(); - this.#contexts.clear(); + this.#contexts = new Map(); this.#missing.clear(); - - watcher.once('aggregated', onChange); } /** diff --git a/apps/heft/src/utilities/WatchpackUtilities.ts b/apps/heft/src/utilities/WatchpackUtilities.ts new file mode 100644 index 00000000000..cdebaeab639 --- /dev/null +++ b/apps/heft/src/utilities/WatchpackUtilities.ts @@ -0,0 +1,168 @@ +// Copyright (c) Microsoft Corporation. All rights reserved. Licensed under the MIT license. +// See LICENSE in the project root for license information. + +import * as path from 'node:path'; +import { performance } from 'node:perf_hooks'; +import { setImmediate as setImmediateAsync, setTimeout as setTimeoutAsync } from 'node:timers/promises'; + +/** + * The longest time that {@link _waitForWatchpackPendingEventsAsync} waits for watchpack to finish recording + * the file system events that it has received. + */ +const MAX_PENDING_EVENTS_WAIT_MS: number = 1000; +const PENDING_EVENTS_POLL_INTERVAL_MS: number = 1; + +/** + * A file system event that watchpack has received but has not finished recording. + * + * @internal + */ +export interface IWatchpackPendingFileEvent { + /** + * The full path of the file whose event is still being recorded. + */ + filePath: string; +} + +/** + * The state of watchpack's pending file system events. + * + * @internal + */ +export interface IWatchpackPendingEventState { + /** + * True if any directory watcher is scanning, or if any file system event is still being recorded. + */ + hasPendingEvents: boolean; + + /** + * The file system events that watchpack has received but has not finished recording. + */ + pendingFileEvents: ReadonlyArray; +} + +/** + * The fields of watchpack's internal `DirectoryWatcher` that show whether it is still recording a change. + * They aren't part of watchpack's public API, so they are all optional. + */ +interface IDirectoryWatcherInternals { + /** + * The watched folder. + */ + path?: string; + + /** + * True while the watcher reads the directory. Changes that the scan finds are recorded as it goes. + */ + scanning?: boolean; + + /** + * The names of the files that have an OS event whose `fs.lstat()` hasn't finished yet. The change is + * recorded only when the `fs.lstat()` finishes. + */ + _activeEvents?: Map; +} + +/** + * A watchpack instance with the internal state that this module inspects. + */ +interface IWatchpackWithInternals { + watcherManager?: { + directoryWatchers?: Map; + }; +} + +/** + * Gets the file system events that watchpack has received but has not finished recording. + * + * @remarks + * This helper centralizes the dependency on watchpack's internal `directoryWatchers` map and the `path`, + * `scanning` and `_activeEvents` fields of each directory watcher. It returns `undefined` if any of them cannot + * be recognized, so callers that must not miss a change can choose a conservative fallback. + * + * @internal + */ +export function _tryGetWatchpackPendingEventState( + watcher: object | undefined +): IWatchpackPendingEventState | undefined { + if (!watcher) { + return { + hasPendingEvents: false, + pendingFileEvents: [] + }; + } + + const directoryWatchers: Map | undefined = ( + watcher as IWatchpackWithInternals + ).watcherManager?.directoryWatchers; + if (!(directoryWatchers instanceof Map)) { + return undefined; + } + + let hasAnyPendingEvents: boolean = false; + const pendingFileEvents: IWatchpackPendingFileEvent[] = []; + for (const { path: folderPath, scanning, _activeEvents: pendingNames } of directoryWatchers.values()) { + if (typeof folderPath !== 'string' || typeof scanning !== 'boolean' || !(pendingNames instanceof Map)) { + return undefined; + } + + if (scanning || pendingNames.size > 0) { + hasAnyPendingEvents = true; + } + + for (const name of pendingNames.keys()) { + pendingFileEvents.push({ + filePath: path.join(folderPath, name) + }); + } + } + + return { + hasPendingEvents: hasAnyPendingEvents, + pendingFileEvents + }; +} + +/** + * Waits for watchpack to finish recording the file system events that it has already received. + * + * @remarks + * Watchpack records a changed file only after an asynchronous `fs.lstat()` of it, so a caller can miss a + * change if it reads aggregated changes before that `fs.lstat()` finishes. This method waits until the + * directory watchers have no events or scans in progress, for up to 1 second. If watchpack's internals have an + * unrecognized shape, this method waits for the same maximum instead of assuming that no events are pending. + * + * @internal + */ +export async function _waitForWatchpackPendingEventsAsync( + getWatcher: () => object | undefined, + hasChanges: () => boolean +): Promise { + // Let the event loop reach its poll phase, which delivers the OS events that were already queued when this + // method was called. If this method was called during a poll phase, the first check phase comes before the + // next poll phase, so it takes two turns. + await setImmediateAsync(); + await setImmediateAsync(); + + const deadline: number = performance.now() + MAX_PENDING_EVENTS_WAIT_MS; + while (hasPendingEvents(getWatcher()) && performance.now() < deadline) { + await setTimeoutAsync(PENDING_EVENTS_POLL_INTERVAL_MS); + } + + if (hasChanges()) { + const recordedTime: number = Date.now(); + while (Date.now() <= recordedTime) { + await setTimeoutAsync(PENDING_EVENTS_POLL_INTERVAL_MS); + } + } +} + +function hasPendingEvents(watcher: object | undefined): boolean { + if (!watcher) { + return false; + } + + const pendingEventState: IWatchpackPendingEventState | undefined = + _tryGetWatchpackPendingEventState(watcher); + return pendingEventState ? pendingEventState.hasPendingEvents : true; +} diff --git a/apps/heft/src/utilities/test/CliUtilities.test.ts b/apps/heft/src/utilities/test/CliUtilities.test.ts new file mode 100644 index 00000000000..ef2fcf633dc --- /dev/null +++ b/apps/heft/src/utilities/test/CliUtilities.test.ts @@ -0,0 +1,26 @@ +// Copyright (c) Microsoft Corporation. All rights reserved. Licensed under the MIT license. +// See LICENSE in the project root for license information. + +import { getErrorExitCode } from '../CliUtilities'; + +describe(getErrorExitCode.name, () => { + const cases: [string | number | undefined, number][] = [ + [undefined, 1], + [0, 1], + ['0', 1], + ['', 1], + [-1, 1], + [1.5, 1], + ['abc', 1], + [1, 1], + [2, 2], + ['3', 3] + ]; + + it.each(cases)( + 'maps an exit code of %p to %p', + (exitCode: string | number | undefined, expected: number) => { + expect(getErrorExitCode(exitCode)).toBe(expected); + } + ); +}); diff --git a/apps/heft/src/utilities/test/WatchFileSystemAdapter.test.ts b/apps/heft/src/utilities/test/WatchFileSystemAdapter.test.ts new file mode 100644 index 00000000000..19b869c4118 --- /dev/null +++ b/apps/heft/src/utilities/test/WatchFileSystemAdapter.test.ts @@ -0,0 +1,588 @@ +// Copyright (c) Microsoft Corporation. All rights reserved. Licensed under the MIT license. +// See LICENSE in the project root for license information. + +import * as fs from 'node:fs'; +import * as os from 'node:os'; +import * as path from 'node:path'; + +import Watchpack from 'watchpack'; + +import { Async } from '@rushstack/node-core-library'; + +import { watchGlobAsync } from '../../plugins/FileGlobSpecifier'; +import { type IWatchedFileState, type StatCallback, WatchFileSystemAdapter } from '../WatchFileSystemAdapter'; + +const RUN_REQUEST_TIMEOUT_MS: number = 5000; +// Long enough for a watcher to finish its first scan, and for events to arrive +const SETTLE_MS: number = 250; +const TEST_TIMEOUT_MS: number = 30000; +// An hour ago, plus some milliseconds so that watchpack sees the file system's timestamps are precise. Files get +// old mtimes so that a new watcher doesn't report a file as changed just because it was written right before +// the watcher started. +const BASE_MTIME_MS: number = Math.floor(Date.now() / 1000) * 1000 - 3600 * 1000 + 123; +// A new watcher reports a file whose mtime is later than the watcher's start time. A file that gets an mtime in +// the future is reported by the next run's watcher for certain. +const FUTURE_MTIME_OFFSET_MS: number = 60 * 1000; +// Long enough for file timestamps to show which of two events came first, even where they are coarse +const TIMESTAMP_GAP_MS: number = 20; +// A file that the glob doesn't match, in a folder that the glob reads +const NOTES_PATH: string = 'src/sub/notes.txt'; + +// Watchpack calls fs.lstat() through this graceful-fs instance +const watcherFs: typeof fs = jest.requireActual( + require.resolve('graceful-fs', { paths: [path.dirname(require.resolve('watchpack/package.json'))] }) +); +// The adapter's calls to fs.lstatSync() look up the function on this module when they are made +const nodeFs: typeof fs = jest.requireActual('node:fs'); + +interface IRunOptions { + watch?: boolean; + // The glob that the run reads. The default reads every folder under src. + pattern?: string; + beforeWatch?: () => Promise | void; +} + +interface IRun { + changed: string[]; + isRunRequested(): boolean; + waitForRunRequestAsync(): Promise; + /** + * Waits until the run's watcher has reported the path with the explanation, and has passed the report on + * in an 'aggregated' event. Then waits SETTLE_MS more, so that later events can arrive. Returns false if the + * watcher doesn't report the path. + */ + waitForReportAsync(relativePath: string, explanation: string): Promise; +} + +describe(WatchFileSystemAdapter.name, () => { + let rootFolder: string; + let writeCount: number; + let adapter: WatchFileSystemAdapter; + let watchSpy: jest.SpyInstance; + let collectSpy: jest.SpyInstance; + let closeSpy: jest.SpyInstance; + + function writeFile(relativePath: string, mtimeMs?: number): void { + const filePath: string = path.join(rootFolder, relativePath); + fs.mkdirSync(path.dirname(filePath), { recursive: true }); + writeCount++; + fs.writeFileSync(filePath, `export const value: number = ${writeCount};\n`); + const mtime: Date = new Date(mtimeMs ?? BASE_MTIME_MS + writeCount * 1000); + fs.utimesSync(filePath, mtime, mtime); + } + + // Drives the adapter the way TaskOperationRunner does in watch mode + async function runAsync(options: IRunOptions = {}): Promise { + const { watch = true, pattern = 'src/**/*.ts', beforeWatch } = options; + adapter.setBaseline(); + const states: Map = await watchGlobAsync(pattern, { + cwd: rootFolder, + fs: adapter + }); + const changed: string[] = []; + for (const [file, state] of states) { + if (state.changed) { + changed.push(file); + } + } + changed.sort(); + + await beforeWatch?.(); + + let requested: boolean = false; + let onRequest: (() => void) | undefined; + const watcherCount: number = watchSpy.mock.contexts.length; + if (watch) { + adapter.watch(() => { + requested = true; + onRequest?.(); + }); + } + + // The reports of the run's watcher, as of its last 'aggregated' event + const reports: Set = new Set(); + let onReports: (() => void) | undefined; + const watcher: Watchpack | undefined = watchSpy.mock.contexts[watcherCount]; + if (watcher) { + let newReports: string[] = []; + watcher.on('change', (filePath: string, modifiedTime: number, explanation: string) => { + newReports.push(`${explanation}: ${filePath}`); + }); + watcher.on('remove', (filePath: string, explanation: string) => { + newReports.push(`${explanation}: ${filePath}`); + }); + // Runs after the adapter's listener, which the adapter added first + watcher.on('aggregated', () => { + for (const report of newReports) { + reports.add(report); + } + newReports = []; + onReports?.(); + }); + } + + return { + changed, + isRunRequested: () => requested, + waitForRunRequestAsync: () => + new Promise((resolve: (value: boolean) => void) => { + if (requested) { + resolve(true); + return; + } + const timeout: NodeJS.Timeout = setTimeout(() => resolve(false), RUN_REQUEST_TIMEOUT_MS); + onRequest = () => { + clearTimeout(timeout); + resolve(true); + }; + }), + waitForReportAsync: async (relativePath: string, explanation: string): Promise => { + const report: string = `${explanation}: ${path.join(rootFolder, relativePath)}`; + const isReported: boolean = await new Promise((resolve: (value: boolean) => void) => { + const timeout: NodeJS.Timeout = setTimeout(() => resolve(false), RUN_REQUEST_TIMEOUT_MS); + onReports = () => { + if (reports.has(report)) { + clearTimeout(timeout); + resolve(true); + } + }; + onReports(); + }); + await Async.sleepAsync(SETTLE_MS); + return isReported; + } + }; + } + + // The first run's watcher may request a run just because the folders are new, so tests only use it to start + async function startAsync(): Promise { + writeFile('src/a.ts'); + writeFile('src/sub/b.ts'); + const run: IRun = await runAsync(); + expect(run.changed).toEqual(['src/a.ts', 'src/sub/b.ts']); + await Async.sleepAsync(SETTLE_MS); + } + + // Also writes the file at NOTES_PATH. Returns the second run, whose watcher has scanned the files without + // requesting a run. The next run collects that watcher's times. + async function startWatchingAsync(): Promise { + writeFile(NOTES_PATH); + await startAsync(); + const run2: IRun = await runAsync(); + expect(run2.changed).toEqual([]); + await Async.sleepAsync(SETTLE_MS); + expect(run2.isRunRequested()).toBe(false); + return run2; + } + + // Gives the file a future mtime, and waits until the run's watcher has recorded the change. Returns the mtime. + async function changeBeforeNextRunAsync(run: IRun, relativePath: string): Promise { + const mtimeMs: number = Date.now() + FUTURE_MTIME_OFFSET_MS; + writeFile(relativePath, mtimeMs); + expect(await run.waitForRunRequestAsync()).toBe(true); + await Async.sleepAsync(SETTLE_MS); + return mtimeMs; + } + + // From now on, the adapter sees a ctime for the file that is earlier than the run's start. A file that changed + // after the run started can have such a ctime where timestamps are coarse, or where the clock that sets them + // lags Date.now(). + function backdateCtime(relativePath: string): void { + const backdatedPath: string = path.join(rootFolder, relativePath); + const lstatSync: typeof fs.lstatSync = nodeFs.lstatSync; + jest.spyOn(nodeFs, 'lstatSync').mockImplementation((( + filePath: fs.PathLike, + options?: fs.StatSyncOptions + ) => { + const stats: fs.Stats | undefined = lstatSync(filePath, options) as fs.Stats | undefined; + if (stats && filePath === backdatedPath) { + stats.ctimeMs = BASE_MTIME_MS; + } + return stats; + }) as unknown as typeof fs.lstatSync); + } + + // Replaces watchpack's fs.lstat(). The handler calls lstatAsync() to do the fs.lstat() and give watchpack the + // result. + function interceptWatcherLstat(handler: (filePath: string, lstatAsync: () => Promise) => void): void { + const lstat: (filePath: string, callback: StatCallback) => void = watcherFs.lstat; + jest.spyOn(watcherFs, 'lstat').mockImplementation(((filePath: string, callback: StatCallback) => { + handler(filePath, async () => { + await new Promise((resolve: () => void) => { + lstat(filePath, (error, stats) => { + callback(error, stats); + resolve(); + }); + }); + }); + }) as unknown as typeof watcherFs.lstat); + } + + // Holds watchpack's next fs.lstat() of the file, as a slow file system can. Resolves when watchpack calls it, + // with a function that finishes it. + function holdNextWatcherLstatAsync(relativePath: string): Promise<(() => void) | undefined> { + const heldPath: string = path.join(rootFolder, relativePath); + return new Promise((resolve: (finish: (() => void) | undefined) => void) => { + const timeout: NodeJS.Timeout = setTimeout(() => resolve(undefined), RUN_REQUEST_TIMEOUT_MS); + let isHolding: boolean = false; + interceptWatcherLstat((filePath: string, lstatAsync: () => Promise) => { + if (filePath === heldPath && !isHolding) { + isHolding = true; + clearTimeout(timeout); + resolve(() => { + void lstatAsync(); + }); + } else { + void lstatAsync(); + } + }); + }); + } + + // Holds watchpack's fs.lstat() of the folder until watchpack has finished its fs.lstat() of each of the files + function holdWatcherFolderLstat(relativeFolderPath: string, fileNames: string[]): void { + const folderPath: string = path.join(rootFolder, relativeFolderPath); + const unreadFiles: Set = new Set(fileNames.map((name: string) => path.join(folderPath, name))); + let finishFolderLstat: (() => Promise) | undefined; + const finishWhenFilesAreRead = (): void => { + if (finishFolderLstat && unreadFiles.size === 0) { + void finishFolderLstat(); + finishFolderLstat = undefined; + } + }; + interceptWatcherLstat((filePath: string, lstatAsync: () => Promise) => { + if (filePath === folderPath) { + finishFolderLstat = lstatAsync; + finishWhenFilesAreRead(); + } else { + void lstatAsync().then(() => { + unreadFiles.delete(filePath); + finishWhenFilesAreRead(); + }); + } + }); + } + + beforeEach(() => { + rootFolder = fs.mkdtempSync(path.join(fs.realpathSync.native(os.tmpdir()), 'heft-watch-fs-')); + writeCount = 0; + watchSpy = jest.spyOn(Watchpack.prototype, 'watch'); + collectSpy = jest.spyOn(Watchpack.prototype, 'collectTimeInfoEntries'); + closeSpy = jest.spyOn(Watchpack.prototype, 'close'); + adapter = new WatchFileSystemAdapter(); + }); + + afterEach(() => { + for (const watcher of watchSpy.mock.contexts) { + (watcher as Watchpack).close(); + } + jest.restoreAllMocks(); + fs.rmSync(rootFolder, { recursive: true, force: true }); + }); + + it( + 'requests a run when a watched file changes, and reports the file in that run', + async () => { + await startAsync(); + + const run2: IRun = await runAsync(); + expect(run2.changed).toEqual([]); + await Async.sleepAsync(SETTLE_MS); + expect(run2.isRunRequested()).toBe(false); + + writeFile('src/sub/b.ts'); + expect(await run2.waitForRunRequestAsync()).toBe(true); + const run3: IRun = await runAsync(); + expect(run3.changed).toEqual(['src/sub/b.ts']); + }, + TEST_TIMEOUT_MS + ); + + it( + 'requests another run for a file that changes while a run is in progress', + async () => { + await startAsync(); + + // The file changes after the glob has read it, and before the run starts watching + const run2: IRun = await runAsync({ + beforeWatch: () => writeFile('src/sub/b.ts', Date.now()) + }); + expect(run2.changed).toEqual([]); + expect(await run2.waitForRunRequestAsync()).toBe(true); + const run3: IRun = await runAsync(); + expect(run3.changed).toEqual(['src/sub/b.ts']); + }, + TEST_TIMEOUT_MS + ); + + it( + 'closes the previous watcher, after collecting its times, when the next run starts', + async () => { + writeFile('src/a.ts'); + await runAsync(); + expect(watchSpy).toHaveBeenCalledTimes(1); + const watcher1: Watchpack = watchSpy.mock.contexts[0]; + expect(closeSpy).not.toHaveBeenCalled(); + + await runAsync(); + expect(watchSpy).toHaveBeenCalledTimes(2); + const watcher2: Watchpack = watchSpy.mock.contexts[1]; + expect(watcher2).not.toBe(watcher1); + expect(collectSpy).toHaveBeenCalledTimes(1); + expect(collectSpy.mock.contexts[0]).toBe(watcher1); + expect(closeSpy).toHaveBeenCalledTimes(1); + expect(closeSpy.mock.contexts[0]).toBe(watcher1); + expect(collectSpy.mock.invocationCallOrder[0]).toBeLessThan(closeSpy.mock.invocationCallOrder[0]); + + await runAsync(); + expect(closeSpy).toHaveBeenCalledTimes(2); + expect(closeSpy.mock.contexts[1]).toBe(watcher2); + }, + TEST_TIMEOUT_MS + ); + + it( + 'reports a file that changed after a run that did not start watching', + async () => { + await startAsync(); + + // Like a run in which copying files fails, so that TaskOperationRunner never calls watch() + const run2: IRun = await runAsync({ watch: false }); + expect(run2.changed).toEqual([]); + + writeFile('src/sub/b.ts'); + await Async.sleepAsync(SETTLE_MS); + const run3: IRun = await runAsync(); + expect(run3.changed).toEqual(['src/sub/b.ts']); + }, + TEST_TIMEOUT_MS + ); + + // On Linux, a watch stays on the deleted folder. Windows doesn't let a watched folder be recreated. + (process.platform === 'linux' ? it : it.skip)( + 'reports a change in a folder that was deleted and recreated, in the run after the change', + async () => { + await startAsync(); + + const run2: IRun = await runAsync(); + expect(run2.changed).toEqual([]); + await Async.sleepAsync(SETTLE_MS); + + // Delete and recreate the folder, as switching branches can + fs.rmSync(path.join(rootFolder, 'src/sub'), { recursive: true }); + writeFile('src/sub/b.ts'); + expect(await run2.waitForRunRequestAsync()).toBe(true); + await Async.sleepAsync(SETTLE_MS); + + const run3: IRun = await runAsync(); + expect(run3.changed).toEqual(['src/sub/b.ts']); + await Async.sleepAsync(SETTLE_MS); + expect(run3.isRunRequested()).toBe(false); + + writeFile('src/sub/b.ts'); + expect(await run3.waitForRunRequestAsync()).toBe(true); + const run4: IRun = await runAsync(); + expect(run4.changed).toEqual(['src/sub/b.ts']); + await Async.sleepAsync(SETTLE_MS); + + const run5: IRun = await runAsync(); + expect(run5.changed).toEqual([]); + }, + TEST_TIMEOUT_MS + ); + + it( + 'does not request a run for a file that changed before the run, but does for a later change', + async () => { + const run2: IRun = await startWatchingAsync(); + await changeBeforeNextRunAsync(run2, NOTES_PATH); + + // The new watcher reports the file, because of its mtime. The glob doesn't match it, but the previous + // watcher's times show that it hasn't changed since the run started. + const run3: IRun = await runAsync(); + expect(run3.changed).toEqual([]); + expect(await run3.waitForReportAsync(NOTES_PATH, 'scan (file)')).toBe(true); + expect(run3.isRunRequested()).toBe(false); + + writeFile('src/sub/b.ts'); + expect(await run3.waitForRunRequestAsync()).toBe(true); + const run4: IRun = await runAsync(); + expect(run4.changed).toEqual(['src/sub/b.ts']); + }, + TEST_TIMEOUT_MS + ); + + it( + 'does not request another run for a watched file that changed before the run started', + async () => { + const run2: IRun = await startWatchingAsync(); + await changeBeforeNextRunAsync(run2, 'src/sub/b.ts'); + + const run3: IRun = await runAsync(); + expect(run3.changed).toEqual(['src/sub/b.ts']); + expect(await run3.waitForReportAsync('src/sub/b.ts', 'scan (file)')).toBe(true); + expect(run3.isRunRequested()).toBe(false); + }, + TEST_TIMEOUT_MS + ); + + it( + 'does not request another run for a file that the first run read', + async () => { + writeFile('src/b.ts', Date.now() + FUTURE_MTIME_OFFSET_MS); + await Async.sleepAsync(TIMESTAMP_GAP_MS); + + // There is no previous watcher, so only the time that the run read is known + const run1: IRun = await runAsync(); + expect(run1.changed).toEqual(['src/b.ts']); + expect(await run1.waitForReportAsync('src/b.ts', 'scan (file)')).toBe(true); + expect(run1.isRunRequested()).toBe(false); + }, + TEST_TIMEOUT_MS + ); + + it( + 'requests a run for a file that is written again during the run, with the same mtime', + async () => { + const run2: IRun = await startWatchingAsync(); + const mtimeMs: number = await changeBeforeNextRunAsync(run2, NOTES_PATH); + + const run3: IRun = await runAsync({ + // The wait makes the file's ctime later than the run's start + beforeWatch: () => Async.sleepAsync(TIMESTAMP_GAP_MS).then(() => writeFile(NOTES_PATH, mtimeMs)) + }); + expect(run3.changed).toEqual([]); + expect(await run3.waitForRunRequestAsync()).toBe(true); + }, + TEST_TIMEOUT_MS + ); + + it( + 'requests a run for a watched file whose mtime moves back after the run read it', + async () => { + const run2: IRun = await startWatchingAsync(); + const mtimeMs: number = await changeBeforeNextRunAsync(run2, 'src/sub/b.ts'); + + const run3: IRun = await runAsync({ + beforeWatch: () => { + // Earlier than the mtime that the run read, but later than the watcher's start, so that the watcher + // reports the file + writeFile('src/sub/b.ts', mtimeMs - FUTURE_MTIME_OFFSET_MS / 2); + backdateCtime('src/sub/b.ts'); + } + }); + expect(run3.changed).toEqual(['src/sub/b.ts']); + expect(await run3.waitForReportAsync('src/sub/b.ts', 'scan (file)')).toBe(true); + expect(run3.isRunRequested()).toBe(true); + }, + TEST_TIMEOUT_MS + ); + + it( + 'does not request a run for a file whose change the previous watcher had not finished reading', + async () => { + await startWatchingAsync(); + const lstatHeld: Promise<(() => void) | undefined> = holdNextWatcherLstatAsync(NOTES_PATH); + writeFile(NOTES_PATH, Date.now() + FUTURE_MTIME_OFFSET_MS); + const finishLstat: (() => void) | undefined = await lstatHeld; + expect(finishLstat).toBeDefined(); + await Async.sleepAsync(TIMESTAMP_GAP_MS); + + // The run starts while the previous watcher still has the file's old time + const run3: IRun = await runAsync(); + finishLstat?.(); + expect(run3.changed).toEqual([]); + expect(await run3.waitForReportAsync(NOTES_PATH, 'scan (file)')).toBe(true); + expect(run3.isRunRequested()).toBe(false); + }, + TEST_TIMEOUT_MS + ); + + it( + 'reports a watched file whose change the previous watcher had not finished reading, in that run', + async () => { + await startWatchingAsync(); + const lstatHeld: Promise<(() => void) | undefined> = holdNextWatcherLstatAsync('src/sub/b.ts'); + writeFile('src/sub/b.ts', Date.now() + FUTURE_MTIME_OFFSET_MS); + const finishLstat: (() => void) | undefined = await lstatHeld; + expect(finishLstat).toBeDefined(); + await Async.sleepAsync(TIMESTAMP_GAP_MS); + + const run3: IRun = await runAsync(); + finishLstat?.(); + expect(run3.changed).toEqual(['src/sub/b.ts']); + expect(await run3.waitForReportAsync('src/sub/b.ts', 'scan (file)')).toBe(true); + expect(run3.isRunRequested()).toBe(false); + }, + TEST_TIMEOUT_MS + ); + + it( + 'does not request a run when a folder that the run read is reported as its watcher is attached', + async () => { + const run2: IRun = await startWatchingAsync(); + await changeBeforeNextRunAsync(run2, NOTES_PATH); + + // The watcher of src attaches to the watcher of src/sub when its fs.lstat() of src/sub finishes. If + // src/sub's watcher has read a file with a new time by then, the watcher reports src/sub. + const run3: IRun = await runAsync({ + beforeWatch: () => holdWatcherFolderLstat('src/sub', ['b.ts', 'notes.txt']) + }); + expect(run3.changed).toEqual([]); + expect(await run3.waitForReportAsync('src/sub', 'watch (outdated on attach)')).toBe(true); + expect(run3.isRunRequested()).toBe(false); + }, + TEST_TIMEOUT_MS + ); + + it( + 'requests a run when a folder that the run did not read is reported as its watcher is attached', + async () => { + const run2: IRun = await startWatchingAsync(); + await changeBeforeNextRunAsync(run2, NOTES_PATH); + + // The run reads src and src/sub/b.ts, but not src/sub, so the watcher doesn't report the files in src/sub + // one by one. The watcher of src attaches to the watcher that src/sub has for b.ts, and reports src/sub. + const run3: IRun = await runAsync({ + pattern: 'src/*.ts', + beforeWatch: () => { + adapter.getStateAndTrack(path.join(rootFolder, 'src/sub/b.ts')); + holdWatcherFolderLstat('src/sub', ['b.ts', 'notes.txt']); + } + }); + expect(run3.changed).toEqual([]); + expect(await run3.waitForReportAsync('src/sub', 'watch (outdated on attach)')).toBe(true); + expect(run3.isRunRequested()).toBe(true); + }, + TEST_TIMEOUT_MS + ); + + it( + 'requests a run for a file system event, even if the file looks unchanged', + async () => { + const run2: IRun = await startWatchingAsync(); + const mtimeMs: number = await changeBeforeNextRunAsync(run2, NOTES_PATH); + const run3: IRun = await runAsync(); + expect(await run3.waitForReportAsync(NOTES_PATH, 'scan (file)')).toBe(true); + + backdateCtime(NOTES_PATH); + writeFile(NOTES_PATH, mtimeMs); + expect(await run3.waitForRunRequestAsync()).toBe(true); + }, + TEST_TIMEOUT_MS + ); + + it( + 'requests a run for a watched file that is deleted during the run', + async () => { + await startWatchingAsync(); + + const run3: IRun = await runAsync({ + beforeWatch: () => fs.unlinkSync(path.join(rootFolder, 'src/sub/b.ts')) + }); + expect(run3.changed).toEqual([]); + expect(await run3.waitForRunRequestAsync()).toBe(true); + }, + TEST_TIMEOUT_MS + ); +}); diff --git a/apps/heft/src/utilities/test/WatchpackUtilities.test.ts b/apps/heft/src/utilities/test/WatchpackUtilities.test.ts new file mode 100644 index 00000000000..37690cc77f4 --- /dev/null +++ b/apps/heft/src/utilities/test/WatchpackUtilities.test.ts @@ -0,0 +1,87 @@ +// Copyright (c) Microsoft Corporation. All rights reserved. Licensed under the MIT license. +// See LICENSE in the project root for license information. + +import * as fs from 'node:fs'; +import * as os from 'node:os'; +import * as path from 'node:path'; +import { performance } from 'node:perf_hooks'; + +import Watchpack from 'watchpack'; + +import { + type IWatchpackPendingEventState, + _tryGetWatchpackPendingEventState, + _waitForWatchpackPendingEventsAsync +} from '../WatchpackUtilities'; + +const FOLDER_PATH: string = path.resolve('temp/test/WatchpackUtilities/src'); + +describe('WatchpackUtilities', () => { + it('reports active file events and scans from watchpack internals', () => { + const pendingEventState: IWatchpackPendingEventState | undefined = _tryGetWatchpackPendingEventState({ + watcherManager: { + directoryWatchers: new Map([ + [ + FOLDER_PATH, + { + path: FOLDER_PATH, + scanning: true, + _activeEvents: new Map([['index.ts', true]]) + } + ] + ]) + } + }); + + expect(pendingEventState).toEqual({ + hasPendingEvents: true, + pendingFileEvents: [ + { + filePath: path.join(FOLDER_PATH, 'index.ts') + } + ] + }); + }); + + it.each([ + { field: 'path', directoryWatcher: { scanning: false, _activeEvents: new Map() } }, + { field: 'scanning', directoryWatcher: { path: FOLDER_PATH, _activeEvents: new Map() } }, + { field: '_activeEvents', directoryWatcher: { path: FOLDER_PATH, scanning: false } } + ])('does not recognize a directory watcher without $field', ({ directoryWatcher }) => { + expect( + _tryGetWatchpackPendingEventState({ + watcherManager: { directoryWatchers: new Map([[FOLDER_PATH, directoryWatcher]]) } + }) + ).toBeUndefined(); + }); + + it('recognizes the directory watchers of a real watchpack watcher', () => { + const folderPath: string = fs.mkdtempSync(path.join(os.tmpdir(), 'heft-watchpack-utilities-')); + const watcher: Watchpack = new Watchpack({}); + try { + watcher.watch({ directories: [folderPath], startTime: Date.now() }); + // watch() creates the folder's directory watcher, so the probe checks that watcher's fields. + const { directoryWatchers } = ( + watcher as unknown as { watcherManager: { directoryWatchers: Map } } + ).watcherManager; + expect(directoryWatchers.size).toBeGreaterThan(0); + expect(_tryGetWatchpackPendingEventState(watcher)?.pendingFileEvents).toEqual([]); + } finally { + watcher.close(); + fs.rmSync(folderPath, { recursive: true, force: true }); + } + }); + + it('waits for the full pending-event window when watchpack internals are not recognized', async () => { + const startTime: number = performance.now(); + + await _waitForWatchpackPendingEventsAsync( + () => ({}), + () => false + ); + + const elapsedMs: number = performance.now() - startTime; + expect(elapsedMs).toBeGreaterThanOrEqual(999); + expect(elapsedMs).toBeLessThan(3000); + }); +}); diff --git a/apps/rush-cli-client/README.md b/apps/rush-cli-client/README.md index 66d54e7f3b8..712ccb9fdbd 100644 --- a/apps/rush-cli-client/README.md +++ b/apps/rush-cli-client/README.md @@ -3,6 +3,9 @@ Separate `rush-client` and `rushx-client` binaries, opt-in until cutover. Existing `rush`, `rushx`, and their reporter entrypoints are unchanged. +To try the daemon on the rushstack repository itself before a release contains it, follow the +[contributor dogfooding guide](../../docs/rush/dogfooding-rush-daemon.md). + ## Native frontend dependency The dependency on `@microsoft/rush` is intentional: it is the version-selecting @@ -34,11 +37,73 @@ reporter behavior without duplicating or relocating bootstrap code. Routing precedence: -1. `--no-daemon` before `--`, help, never-daemonize commands, and Rushx with any TTY stdio stay in-process. +1. `--no-daemon` before `--`, help, never-daemonize commands, an option before the command other than + `--quiet`/`-q` (such as `--debug`), and Rushx with any TTY stdio stay in-process. `--quiet` and `-q` only hide + native Rush's startup banner, which a daemon request never prints, so `rush -q build` is sent to the daemon + without them. 2. CI stays in-process unless `RUSH_DAEMON=1` explicitly opts in, even if config enables the daemon. 3. `RUSH_DAEMON` overrides `rush.json`'s `daemon.enabled`; the default is false. 4. Auto-start is considered only after selecting daemon execution. +A command that routing keeps in-process shows no progress line. It says why in one stderr line before native +Rush starts, in agent mode (see [Output modes](#output-modes)) and, in legacy mode, when `RUSH_DAEMON=1` asked +for the daemon. Rushx uses the `rushx-client:` prefix. `--no-daemon` and help print nothing. The lines are: + +- `rush-client: RUSH_LOG_LEVEL selects the native reporter; using in-process Rush.` The same line names + `RUSH_REPORTER=`, `--reporter`, `--output`, `--log-level` or `useRushReporter in experiments.json` + (see below). +- `rush-client: the daemon does not support "--debug"; using in-process Rush.` +- `rush-client: the daemon does not run "check"; using in-process Rush.` +- `rushx-client: the daemon does not run scripts in a terminal; using in-process Rush.` +- `rush-client: RUSH_DAEMON=0 turns the daemon off; using in-process Rush.` +- `rush-client: the daemon is not enabled for this repo; using in-process Rush.` To enable it, set `daemon.enabled` + to `true` in `rush.json`, or set `RUSH_DAEMON=1` in the environment. +- `rush-client: CI is set, so the daemon is off unless RUSH_DAEMON=1; using in-process Rush.` The line names + the first CI marker that is set: `CI`, `TF_BUILD`, `GITHUB_ACTIONS`, `JENKINS_URL` or `TEAMCITY_VERSION`. + +When the selected daemon cannot be reached or started, an ordinary invocation prints the reason and runs +in-process (`rush-client: ; using in-process Rush.`). A startup failure while a live process can +still make the daemon ready is the exception: a process that listens at the endpoint but does not +complete hello/ping in time, a startup helper that still waits for its daemon, or another client that +holds the start mutex, as in a burst of clients that all find no daemon. In-process Rush would take the +repository lock, and the requests that the daemon serves would then wait for it, or fail when their wait +timeout ends (see below). Instead, the client keeps trying for one more startup deadline +(15 seconds, so about 30 seconds in all) and uses the daemon once it is ready. It says so when it starts +waiting (`rush-client: The daemon is not ready yet. , so this command waits up to 15 s more +for it instead of running Rush in-process.`; agent output shows "rushd is still starting; waiting for it" +as the progress phase, and on a pipe writes it as a progress line that ends with `because `). If the daemon is still not ready, the command exits with code 1. The message +gives the startup error with its `--no-daemon` hint, then the process that is still live, "so Rush was +not run in-process", and a pointer to `rush-client daemon status`. When the process that the daemon's +ownership record names still runs but does not answer (on Linux, for example because a signal stopped +it), the message instead says what that process is doing, then on a line of its own that Rush was not +run in-process, and its last line says what to do, for example `Resume it with "kill -CONT "; it +then serves the next command.` On Linux, when that process has this workspace's ownership record open, as the daemon that +wrote it does, and stays stopped (state T or t) while the client samples it for 1.5 s, the command does +not wait for either deadline: it exits with code 1 and that message once the 1.5 s have passed. It names +a signal to send only to a Rush daemon that has that record open; for any other process it says to end +that process if it is this workspace's daemon, and else to delete the ownership record, and that until +then each command that uses the daemon first waits 15 s for a daemon to answer. Such a Rush +daemon that still runs after its socket file was deleted also fails the command this way, with code 1 +and without running Rush in-process, once the client has waited 15 seconds for it to exit; its last line +says that it may exit once its running requests finish. When the process that the ownership record names +has exited but is not reaped yet (on Linux, state Z), nothing live can make the daemon ready, so the +command runs Rush in-process once its startup deadline (15 s) has passed. The last line of its message +names the parent that has not reaped that process, and says that until then each command that uses the +daemon first waits 15 s for a daemon to answer. + +When the client runs Rush in-process after it tried the daemon, because it could not reach one or because the +daemon handed the request back, and another Rush process holds the repository's lock, such as the daemon while it +builds for another request, Rush waits for the lock instead of failing at once with "Another Rush command is +already running in this repository." It waits only for what is left of the request's wait timeout (see below), +counted from when the client sent the request, or from when it gave up on the daemon if it could not reach one; +the built-in 30-second default applies. It writes one stderr line when it starts to wait (`Waiting up to 28 s for +the Rush daemon (PID 4242) to release this repository's lock.`). If the lock is still held at the deadline, the +command fails as before, and the error names the holder (`The Rush daemon (PID 4242) still holds this repository's +lock.`). With `--no-wait` or a zero timeout, Rush tries once and names the holder. On Windows the lock file does not +name a process, so Rush says "another Rush process". Rush that routing keeps in-process (such as `--no-daemon`, +`RUSH_DAEMON=0` or CI), `rushx-client`, and a workspace that selects another Rush release than the one the client +bundles fail at once, as native Rush does. + `--no-wait` fails immediately when daemon admission is unavailable. `--wait-timeout SECONDS` (or `--wait-timeout=SECONDS`) overrides the configured queue timeout; finite nonnegative decimal seconds up to 2147483.647 are accepted and @@ -46,20 +111,62 @@ rounded down to milliseconds. These controls are mutually exclusive and are consumed before forwarding, never appended to a project script. Arguments after `--` remain literal script arguments. -The queue timeout is measured from when the daemon receives the request. An -explicit `--no-wait`, `--wait-timeout`, `RUSH_DAEMON_QUEUE_TIMEOUT_SECONDS`, or -`daemon.queueTimeoutSeconds` in `rush.json` bounds the entire wait: waiting for -workspace admission and waiting for a running build that the request could not -join. The built-in 30-second default bounds only workspace admission (for example, -waiting for a command that needs exclusive access). With the default, a build that -arrives while a compatible build is already running waits for it to finish and then -runs, instead of failing after 30 seconds. On a timeout, the client exits with -code 1 and says how to wait longer. +Only time that the request spends waiting for other requests counts against the +queue timeout, whether it comes from `--wait-timeout`, +`RUSH_DAEMON_QUEUE_TIMEOUT_SECONDS`, `daemon.queueTimeoutSeconds` in `rush.json`, +or the built-in 30-second default. Waiting while another request +loads or reloads the workspace graph does not count, so every build that arrives +while the first build after startup loads the graph runs once the load finishes. +That wait fails after 10 times the timeout (5 minutes with the default), so a load +that never finishes does not hold other requests forever. The request's own work, +such as checking its inputs, loading the graph, routing and execution, does not +count either. A configured or per-invocation timeout also +limits waiting for a running build to start so that the request can join it (with +`joinRunningBatch`), waiting for a running build that the request could not join, and waiting for +the requests that the daemon is serving to finish before it restarts for the +request's environment. The built-in default does not: with it, a build that arrives +while a compatible build is already running waits for it to finish and then runs, +instead of failing after 30 seconds, and a request that needs a restart waits for +the requests that were running when it arrived to finish and then runs on the +restarted daemon. The default still limits a restart wait while the daemon runs a +`rushx` script, such as a dev server, which may not exit until it is stopped, and +while it serves requests that arrived later. A `rushx-client` script that arrives +while another request waits for the daemon to restart does not start on the old +daemon, where the restart would wait for it to exit: it waits for the restart and +then runs on the restarted daemon, and its timeout applies to that wait as it does +to the restart wait. `--no-wait` fails wherever the request would wait. On a +timeout, the client exits with code 1 and suggests `--wait-timeout`. It does not +suggest exporting `RUSH_DAEMON_QUEUE_TIMEOUT_SECONDS`, because Rush versions that do +not recognize a `RUSH_` environment variable fail every command while it is set. +In legacy output and in `rushx-client`, the admission failure line +(`rush-client: daemon admission failed (wait-timeout): …`, or `(no-wait)`; in +`rushx-client` it begins with `rushx-client:`) gives the daemon's reason, as agent +mode's summary line does, so it names what the request waited for, such as a daemon +restart, and why the daemon restarts. + +A request also waits while a Rush process that the daemon does not run, such as +`rush install` or a `--no-daemon` build, holds the repository's lock. Every timeout, +the built-in default included, limits that wait, since that process can run for any +length of time. Stderr, on a terminal and on a pipe, names the process at once +(`rush-client: waiting for another Rush process (PID 12345: rush install) to release +this repository's lock.`), and again with the time waited every 10 seconds +(`still waiting after 10s for …`); agent output shows it as the progress phase, and if the +daemon then restarts, `request resubmitted to the new daemon; preparing the workspace graph` +until the new daemon reports a queue position or starts the command. Only +Linux tells the PID and command; elsewhere the line says `another Rush process`. On +Linux, the name also says when that process is stopped, since it cannot release the +lock until something resumes it (`another Rush process (PID 12345: rush install; it is +stopped (state T), for example by SIGSTOP)`). +`--no-wait` and `--wait-timeout 0` fail at once, and the admission failure line of +these and of a timeout names the process. A request never waits for a lock that the +daemon itself holds for another request: it fails at once, as before. Admission controls also apply to experimental graph requests, but not -`start|stop|restart|status|logs`. They affect daemon admission only; native fallback +`start|stop|restart|status|logs`. They affect daemon admission, and how long Rush that runs in-process after +the client tried the daemon waits for the repository's lock (see above); otherwise native fallback retains native command behavior. Waiting positions are shown on interactive stderr, -and admission failures report their typed reason and a nonzero exit code. +and a wait for a daemon restart (see below) or for another Rush process on a pipe +too. Admission failures report their typed reason and a nonzero exit code. Explicit reporter/output/log-level controls (`--reporter`, `--output`, `--log-level`, `RUSH_REPORTER` other than `legacy`, or `RUSH_LOG_LEVEL`) retain the native frontend @@ -84,14 +191,97 @@ agent mode writes nothing ahead of it. Otherwise, selection precedence is: 3. Otherwise `legacy`: the unchanged collated operation stream. Agent mode is plain text for humans and agents, not the AI reporter's JSON record format; -use `--reporter=ai` for machine-parsed records. It writes a first status line before -`@microsoft/rush-lib` is loaded, then at most three live rows on a TTY (append-only lines -throttled to one per 2 seconds on a pipe), the queue position when waiting for admission, -and always one final summary line (`rush build: SUCCESS 12/12 operations (...) in 3.1s`, or -`up to date (no operations needed)`). On failure, it lists failed operations and a -bounded tail (10 lines) of their stderr, or of their stdout when they wrote no stderr. -Operation logs are otherwise not printed; use `RUSHD_OUTPUT=legacy` for full logs. When -a request falls back to in-process Rush, agent mode stops and native output follows. +use `--reporter=ai` for machine-parsed records. On a TTY it paints at most three live rows, the +first before `@microsoft/rush-lib` is loaded. On a pipe it writes one progress line when the +daemon has the request (`rush build · 0.1s · sent to rushd; preparing the workspace graph +(status at least every 25s)`), however many operations run. A longer request also gets status +lines, so that it does not look hung: one whenever nothing was written for 25 s, with the counts +and the running operations, and one when connecting to the daemon takes more than 10 s. A +wait for a daemon that is still starting gets a line of its own +(`rushd is still starting; waiting for it (up to 15s more) because its startup helper (PID 4242) is +still waiting for the daemon`). A +request that waited for admission says so at the end of its summary line +(`· queued behind another request (position 1 at 0.2s)`), unless it failed: a failure's summary +line gives the failure, not the wait. It always ends with one summary +line, for example +`rush build: SUCCESS 772/772 operations (12 success, 760 from cache) in 3.1s`, or +`up to date (no operations needed)`, or, when the selection parameters matched no projects, +`rush build: SUCCESS 0 operations in 0.5s · the selection parameters did not match any projects`. +The counts follow the native summary: silent operations +(such as phases a project does not define) are not counted unless they fail, and operations +that did not need to run (`SKIPPED` or `NO OP` in native output) are counted as `up to date`. The verdict is +`SUCCESS`, `FAILURE` or `CANCELLED` (Ctrl+C or a termination signal); a request that a daemon +shutdown aborted is a `FAILURE` whose summary line gives the reason, with exit code 1. When the +daemon did not admit the request in time, or at once with `--no-wait`, the reason on the summary +line starts with `daemon admission failed (wait-timeout)` or `daemon admission failed (no-wait)`, +as in legacy output, followed by the daemon's reason in full. Warnings and errors that Rush or a +Rush plugin writes outside any operation (for example a plugin that continues without the cloud +build cache) are written at the end, at most three lines of them, before the summary line and any +operations reported with it. + +A failed operation is reported as soon as it fails, while the rest of the request runs on: a +`failed: · full log: ` line and a short excerpt of its output, error lines +with the line that follows them first, then the last lines. Stack frames, `Require stack:` lists +and progress noise are left out, and so are a message that a tool repeats in its summary, an +error count that the shown errors account for, and, when the first error shown names a source +location, the lines before it. Up to three operations are reported. Two kinds are reported just +before the summary line instead: a failed operation that wrote no output, with the error from the +daemon's result, and, when no operation failed, the operations whose warnings failed the request +(`warnings: …`). An operation that succeeded and then got warnings, because its build cache entry +could not be written, is shown with the output that it wrote after it succeeded. The error of a +reported operation that wrote output is printed too, unless its excerpt shows it or it only gives +the exit code (`Returned error code: 1`): for example an error +thrown while the operation's build cache entry was restored. Only the daemon's result carries it, +so for an operation reported as it failed it comes just before the summary line, as an +`error: ` line followed by the error. On a pipe, a status line names a failed +operation that wrote no output 1 s after it failed, unless the result came first. The summary +line names up to five failed (or warning) operations. Every operation's full output is in its +project's `rush-logs/` folder, whether or not it was printed. +When the daemon ran an operation's incremental command (its `:incremental` script; see +`incrementalBuilds` below) and that command failed, the line reads +`failed: · incremental command; its next run uses the initial command · full log: `. +That command can fail where the initial command, which `--no-daemon` runs, would not. Warnings +that an incremental command reported get `· incremental command` in the same place. +When a request falls back to in-process Rush, agent mode stops and native output follows. + +In agent mode a failed `rush build` doesn't wait for all of its work. Its result comes once an +operation failed and none of the selected projects that no other selected project depends on (for +example, the projects named by `--to`) is still waiting or running. The daemon keeps running the +operations that the failure didn't block, so that the next build finds them done, and the summary +line counts them and names up to three, in name order +(`· 2 independent operations continue in rushd: lib-b (build), lib-c (build)`). A later `rush build` +waits for them. While it waits only for them, its output says so and names up to three of them. +In agent mode the phase reads +`queued behind 2 operations left running by an earlier failed command (position 1): lib-b (build), lib-c (build)`, +a status line on a pipe reads +`waiting for 2 operations left running by an earlier failed command (queue position 1 at 0.1s): lib-b (build), lib-c (build)`, +and the summary line ends with +`· queued behind 2 operations left running by an earlier failed command (position 1 at 0.1s): lib-b (build), lib-c (build)`. +Legacy output on a terminal prints +`rush-client: waiting for daemon admission (position 1) behind 2 operations left running by an earlier failed command: lib-b (build), lib-c (build).` +The daemon reports the position again each time one of them ends, so the phase and the status +line name only the ones that still run; the summary line keeps the first position that named them. +`rush rebuild`, `rush install` and `rush update`, a restart of the daemon for another +environment, and `rush-client daemon stop` stop them instead. So does a command that the daemon +doesn't run (such as a custom command that it can't serve), before the client runs it in-process; +the client then prints, indented under its fallback line, +`rushd stopped 2 operations left running by an earlier failed command (lib-b (build), lib-c (build)), so that this command can run in-process.` +A served command that makes the daemon reload its graph, such as `rush test` after `rush build`, +stops them as well, and so does a served phased command that isn't incremental, such as a custom +`rush retest`. These commands and a served `rush rebuild` name them while the daemon stops them: +the agent phase reads +`stopping 2 operations left running by an earlier failed command (position 1): lib-b (build), lib-c (build)`, +a status line on a pipe reads +`waiting while rushd stops 2 operations left running by an earlier failed command (queue position 1 at 0.1s): lib-b (build), lib-c (build)`, +and if the command succeeds or is cancelled, its summary line ends with +`· stopped 2 operations left running by an earlier failed command (position 1 at 0.1s): lib-b (build), lib-c (build)`. +Legacy output on a terminal prints +`rush-client: waiting for daemon admission (position 1) while rushd stops 2 operations left running by an earlier failed command: lib-b (build), lib-c (build).` +Older clients say that such a command is queued behind them. Older daemons don't name them, so the +summary line ends with `· queued behind another request (position 1 at 0.1s)`. +Rushx scripts, and built-in commands that only read the workspace (such as `rush list`), run +in-process alongside them, like two Rush commands at once in one checkout. +Older daemons report the failure when all of the work has ended. Positively identified built-in `install` and `update` follow the same opt-in routing precedence as workspace builds and require protocol **0.10** @@ -104,8 +294,9 @@ Request cwd, environment, argv, width and color are captured before connecting. The protocol currently expresses request color as a boolean; subscriptions carry the corresponding color level. There is no SIGWINCH forwarding. -The standalone host now binds native `build`/`rebuild` requests to a reusable -all-project graph. Native Rush parsing, project selection, graph plugins, and +The standalone host now binds native `build`/`rebuild` requests, and the phased +commands of command-line.json (such as `test`), to a reusable all-project graph. +Global commands still run in-process. Native Rush parsing, project selection, graph plugins, and incremental/cache semantics are reused rather than spawning another Rush CLI. The client renders operation headers, collated text, and activity events; global command byte streams remain byte-preserving. A `rushx build` script never claims @@ -154,17 +345,20 @@ changed configuration or command shape replaces the session and graph in the sam process. Environment, installed dependencies, implementation content, or selected Rush version changes require a process restart rather than patching the existing engine. Direct, inherited, and rig-based project configuration uses private native -loaders and is rechecked before execution. External plugins, `.env`, phased +loaders and is rechecked before execution. External plugins that participate in the +requested command (unassociated plugins, plugins associated with it, or plugin command-line +files that define it, its phases or parameters for either), `.env`, phased watch/install options, and unsupported event-hook scripts still use typed -pre-execution fallback; this does not exclude the built-in `install` and `update` +pre-execution fallback; plugins scoped only to other commands are permitted. This does not exclude the built-in `install` and `update` commands described above. The native Rush lock is held for preparation and each coalesced iteration, not while idle; native commands and `--no-daemon` can run after a completed request without stopping the daemon. Native workspace dispatch copies the request envelope and normalizes only the -engine-owned `_RUSH_LIB_PATH` to this daemon's real engine. Foreign client SDK -paths therefore neither select the wrong SDK nor cause a false restart. All other -environment inputs remain unchanged and participate in normal lifecycle checks. +engine-owned `_RUSH_LIB_PATH` to this daemon's own engine, keeping the spelling +that the engine chose when it loaded. Foreign client SDK paths therefore neither +select the wrong SDK nor cause a false restart. All other environment inputs +remain unchanged and participate in normal lifecycle checks. Protocol 0.10 permits a bounded retry only when a pre-execution command result explicitly carries `retryAfterRestart: true`. `executeWithDaemonRestartAsync` @@ -176,6 +370,136 @@ only an unstarted request can receive the typed retry authorization. Accepted queued requests drain their typed restart results before the old connection closes. +When the daemon's own installation was removed or replaced (for example a deleted +snapshot folder or a reinstalled Rush release), the daemon lets its running requests +finish, answers each other request with that typed restart once they have, and then +exits. The timeout rules of a restart for the request's environment apply (see +above): the built-in default does not limit waiting for the requests that were running +when the command arrived, but still limits it while the daemon runs a `rushx` script, +and `--no-wait` and an explicit `--wait-timeout` limit the whole wait. A command that +times out exits with code 1, names the changed folder and, if a script runs, suggests +stopping it. Otherwise the client starts a daemon from its own launcher once the wait +ends, resubmits the request, and prints one line on stderr (or above the agent +progress rows): `rush-client: The daemon's installation at was removed; +restarted the daemon (PID ).` + +While a command waits for a daemon restart, for its installation, the command's +environment or the workspace's inputs, the agent progress status (or stderr, on a +terminal and on a pipe) says what it waits for and why the daemon restarts, as soon as +the daemon reports the wait: `rush-client: waiting for 2 running requests to finish, +including 1 rushx script; the daemon (PID ) then restarts, because +common/config/rush/pnpm-lock.yaml changed.` After `because`, the cause is `its +installation at was removed` (or `replaced`), `this request's environment +differs from the daemon's in NODE_OPTIONS` (variable names, never their values), +` changed` for the workspace's installation, `the code of Rush or a Rush plugin +changed ()`, or `this request selects Rush `. The count names `rushx` +scripts, because a script such as a dev server may run until it is stopped. A +`rushx-client` script that waits for another request's restart prints `rushx-client: +waiting for the daemon (PID ) to restart for another request (2 requests ahead), +because .` The line goes to the stderr that the script writes to, which is a +pipe, because a `rushx-client` with a terminal runs the script in-process. Its first line +comes once `rushx-client` itself has started and sent the request, which takes about half a +second on a busy machine. A native `install` or `update` restarts the daemon once it ends, +unless it fails before it changes the installation (for example on the Rush lock), and +that restart would end the `rushx` scripts that the daemon runs. The daemon can't tell +beforehand whether the command will fail, so it first waits for them and says so the +same way: `rush-client: waiting for 1 running rushx script to finish, since this command +restarts the daemon (PID ), which would end it.` A terminal +gets a line whenever the wait changes, and a pipe when the wait begins or its cause +changes. Both get the line again with the time waited (`still waiting after 25s for +…`) whenever 25 seconds pass without one, until the command follows the restart, +starts, or ends. Once the client asks rushd to cancel the command (Ctrl+C) and says so, +it writes no more wait lines. In agent mode, the progress phase says it, and on a pipe +a status line is written at once when the wait begins or its cause changes. + +When the daemon restarts for a command's environment, the client prints a line of the +same kind that names the variables that differed, never their values: +`rush-client: A command's environment differed from the daemon's in NODE_OPTIONS; +restarted the daemon (PID ).` It names at most four variables and then says how +many more differed (`A, B, C, D and 2 more`). A command that was waiting when another +command's environment restarted the daemon prints that command's variables, because the +successor starts with that command's environment. A daemon that does not name the +variables gets no line. A variable name that holds a control character, such as a newline +or ESC, is printed with that character written as a `\xHH` escape, so the line stays one +line. + +If the daemon that replaces it does not start, after either kind of restart, the command +fails with exit code 1 and one line that gives the reason for the restart before the +startup error: `rush-client: A command's environment differed from the daemon's in +NODE_OPTIONS; the restarted daemon did not start: `. A variable that keeps +the daemon from starting is then among the names in that line. + +The client leaves out the `…; restarted the daemon (PID ).` line of either kind when +the command already wrote a wait line (see above) for the same cause since it last +restarted, because that line said why the daemon restarts: the same variables, or the same +change to the same installation. The wait line counts when it was written as a line, on +stderr or as an agent status line on a pipe. With agent output on a terminal, the wait is +only in the live rows, so the line is printed above them. + +When the connection is lost before a command's result, the command fails with exit code 1 +and is not retried. The diagnostic keeps "Daemon disconnected before delivering a result; the +command was not retried." and says what happened to rushd. If its process exited (a crash, an +out-of-memory kill or a signal), it names the PID, points to `rush-client daemon logs` and, if +the daemon exits again, to `--no-daemon` (`rushx-client --no-daemon` for Rushx), and quotes on a +second line the fatal error that the launcher log recorded after the command was sent. Before it +prints that, the client removes the exited daemon's ownership record and socket, as the next daemon +start would. On Linux, it first stops the operations that the daemon left running, so a rerun, with +or without `--no-daemon`, does not race them. +Rush run in-process, with `--no-daemon` or as a fallback, first does the same when the ownership +record names a daemon that no longer runs, for example when the client that ran the command was +killed along with the daemon. On Linux that includes a daemon that has exited but is not reaped yet: +the client waits up to 1 second for it to be reaped, and else runs Rush without the reclaim, which a +later command does once the daemon's parent reaps it. Each of these reclaims, and the ones that `daemon start`, an automatic +start and `daemon stop --force` do, prints one line that says what it stopped, for example: + +``` +rush-client: Stopped the operations that the exited daemon (PID 4242) left running (process group 4242). +``` + +The line begins "Killed" instead, and ends "they did not exit after SIGTERM", when an operation +needed SIGKILL. While agent mode shows its progress lines, the line is written among them. +If rushd still runs, it says that only the connection closed. Ctrl+C and an orderly `daemon stop` +or `daemon restart` still end a command as cancelled (exit code 130). + +A command that was still waiting in rushd's queue when rushd exited has not run, if rushd says +when it starts a command (protocol 0.14) and had not said so. The client then sends it to a new +daemon once, within its `--wait-timeout`, and before that daemon starts it prints one line on +stderr (or above the agent progress rows): `rush-client: rushd (PID ) exited while the +command was queued; sending the command to a new daemon.` In agent mode the phase, and the status +lines on a pipe, then read `request resubmitted to the new daemon; preparing the workspace graph` +rather than what the command waited for in the exited daemon's queue, until the new daemon reports +a queue position or starts the command. If the new daemon does not start, +the command fails after that line with the startup error. If the connection to the new daemon +is lost too, the diagnostic begins "Daemon disconnected before delivering a result; the command +was already sent to a new daemon once." Commands that waited together reach the new daemon in +the order in which their clients noticed that rushd exited, not in their order in the queue. + +While a command runs, the client checks that rushd still responds. Once rushd has sent nothing +for 10 s, the client pings it. Once it has sent nothing for 30 s, not even the reply, for example +because its process was stopped, the client says so at once and says what that means: +`rush-client: rushd (PID ) has not responded for 30s; its process may be stopped or +overloaded; on Linux, "rush-client daemon status" says which. This command goes on if rushd +responds; interrupt it (Ctrl+C) to stop waiting.` When +rushd sends anything again, a second line says so: `rush-client: rushd (PID ) responded again +after 70s.` In agent mode, the progress phase says it; on a pipe both lines are written at once, +and until rushd responds, the status lines say how long it has not responded instead of what runs. +Time in which the client itself was stopped or busy writing output does not count. An interrupt +asks rushd to cancel the command, which a stopped rushd cannot confirm, so the client stops waiting +when its 5 s cancellation wait ends, and writes no more of these lines once it asked. Daemons older +than protocol 0.13 are not checked. + +When the process that reads the client's output exits first, for example `head` in +`rush-client build | head -5`, the client's next write to that stream fails with EPIPE. The client +does not report that as a lost connection. It asks rushd to cancel the command, waits for the stop +as it does after Ctrl+C, and exits with code 141 (128 + SIGPIPE), which a shell reports for a writer +that SIGPIPE ended. Instead of the cancelling and cancelled lines it prints one line, which names +the stream: `rush-client: build cancelled, because the process reading its stdout exited (EPIPE).` +In `rushx-client` it begins with `rushx-client:`. +Agent output prints the same line on stderr. The client only learns of the exit at its next write, +which in agent output on a pipe can be the next status line, up to 25 s later. A command whose +result arrived before a write failed keeps the result's exit code. Rush run in-process and +`rush-client daemon logs` still report a failed write as an error. + Piped input uses protocol 0.7's negotiated stdin admission and EOF. The client does not read input until the command attaches an input destination, and sends bounded chunks only as the daemon grants write credits. EOF follows all preceding writes; @@ -207,9 +531,14 @@ keys and unknown `RUSH_DAEMON*` variables fail validation. | `enabled` | `RUSH_DAEMON` | false | Client routing | | `autoStart` | `RUSH_DAEMON_AUTO_START` | true | Only after opt-in | | `idleTimeoutSeconds` | `RUSH_DAEMON_IDLE_TIMEOUT_SECONDS` | 900 | Host idle shutdown after request/output/cleanup drain | -| `queueTimeoutSeconds` | `RUSH_DAEMON_QUEUE_TIMEOUT_SECONDS` | 30 | Admission wait limit. The default does not bound waiting behind a running compatible build; an explicit value does | +| `queueTimeoutSeconds` | `RUSH_DAEMON_QUEUE_TIMEOUT_SECONDS` | 30 | Admission wait limit. Time behind another request's graph load (up to 10 times the limit) and the request's own work do not count. The default does not limit waiting behind a running compatible build, or, for a daemon restart, behind requests that were already running (while no `rushx` script is running); an explicit value does | | `watch` | `RUSH_DAEMON_WATCH` | false | Persistent host observation of requested warm projects; false keeps root/config guards only. Never schedules builds | | `usePersistentIpcRunners` | `RUSH_DAEMON_USE_PERSISTENT_IPC_RUNNERS` | false | Enables explicit per-operation `daemonIpc` Node launchers for unsharded incremental daemon builds | +| `incrementalBuilds` | `RUSH_DAEMON_INCREMENTAL_BUILDS` | true | Runs an operation's `:incremental` script instead of its initial script when only files it builds were edited since its last successful run in the daemon and its output folders are unchanged. Additions, deletions, renames, configuration, tool, environment and command-line changes, bundled outputs, cache restores and native Rush commands run the initial script. Incremental results are never written to the build cache | +| `warmWorkers` | `RUSH_DAEMON_WARM_WORKERS` | false | With `incrementalBuilds`, keeps a watch-mode worker (the `:incremental:ipc` script) alive between builds for each operation whose `rush-project.json` operation settings set `allowDaemonWarmWorker`, and sends it the next incremental run. Opt in only if that script runs every task and check that the initial script runs, for Heft including lint and API Extractor. When an incremental run is not allowed, the worker is closed and the initial script runs. Workers count toward `warmMemoryBudgetMB` and `warmSetMaxProjects`, so raise both to keep them alive | +| `joinRunningBatch` | `RUSH_DAEMON_JOIN_RUNNING_BATCH` | false | Experimental. A build request that arrives while the daemon executes an incremental batch with the same request settings adds its operations to the executing iteration and gets its result once they complete, instead of waiting for the iteration to end. When the iteration can't take its work, the request waits as before | +| `deferCacheWrites` | `RUSH_DAEMON_DEFER_CACHE_WRITES` | false | Lets an operation complete once its output files are cloned, and writes its build cache entry from the clones in the background. Needs a file system that can clone files (such as Btrfs, XFS or APFS); otherwise, and in cobuilds, the entry is written before the operation completes. A failed background write doesn't change the operation's status; the next command reports it in a warning (on stderr in legacy output), which agent output also prints. A restore of the entry's key misses until the entry is written, and `daemon stop` drops the entries that aren't written yet | +| `backgroundPrepare` | `RUSH_DAEMON_BACKGROUND_PREPARE` | false | Experimental. When a watched workspace input changes so that the next request would reload the workspace graph (`lastReloadTier` 1), for example after an edit of `rush.json` or `common/config/rush/command-line.json`, an idle daemon reloads it and creates the engine for the command line of the last phased command that it served, 2 seconds after the last change. It never runs an operation or restarts the daemon, and it doesn't start while another Rush process holds the repository lock. A request with the same command line waits for it; any other request stops it and runs as before | | `warmIdleTimeoutSeconds` | `RUSH_DAEMON_WARM_IDLE_TIMEOUT_SECONDS` | 300 | Idle runner, project-watcher and retained-result eviction | | `warmMemoryBudgetMB` | `RUSH_DAEMON_WARM_MEMORY_BUDGET_MB` | 512 | Best-effort sampled RSS budget in MiB, not a hard ceiling. Compared against whole-daemon RSS plus measured child RSS, so keep it above the daemon baseline (~130-190 MiB) | | `warmSetMaxProjects` | `RUSH_DAEMON_WARM_SET_MAX_PROJECTS` | 20 | Best-effort limit on projects holding warm resources (active runners, watchers); retained results of resource-free projects do not count. Never trims requested execution | @@ -254,13 +583,80 @@ It does not replace a peer lacking safe shutdown support. Foreign package instal is a client preparation step; host self-restart selects only bundled or already cached compatible installations, never installing while the old workspace is being cleaned up. +Every client of a checkout finds its daemon in one per-user runtime folder: on Linux and +macOS, `/tmp/rushd-/`, whatever `TMPDIR` or `XDG_RUNTIME_DIR` a shell, job, service or +sandbox sets. It holds the socket (`.sock`), the ownership record +(`.pid.json`) and the launcher log. To move it, set `RUSHD_RUNTIME_DIR` to an absolute +path for every client of that checkout; the folder becomes `$RUSHD_RUNTIME_DIR/rushd-/`, +and its file system must support hard links. A relative `RUSHD_RUNTIME_DIR` is ignored. Clients +that disagree about `RUSHD_RUNTIME_DIR` use different folders, so each folder gets its own daemon +for the checkout. A client passes the folder to the daemon it starts. +The socket path must fit in a socket address: at most 108 bytes on Linux and 104 on macOS. +`RUSHD_RUNTIME_DIR` can therefore be at most 57 bytes on Linux and 53 on macOS, minus the +number of digits in your uid (50 bytes on Linux for uid 1234567). +Windows uses the named pipe `\\.\pipe\rushd-` and is unchanged. +The client refuses a runtime folder that is a symbolic link, is not a directory or belongs to +another user, and a `RUSHD_RUNTIME_DIR` too long for the socket path: commands run in-process +with that reason, and `daemon` commands exit 1. Remove the folder or change `RUSHD_RUNTIME_DIR`. +Auto-start likewise refuses a launcher log that is not a regular file of yours with one link, or +that it cannot open for writing, such as a symlink, a directory or a FIFO: commands run in-process +with that reason (for example `Launcher log cannot be opened for writing (ELOOP): `), and +`daemon start` exits 1. Remove the file. +A folder that others can open is made owner-only (`0700`). +Within one runtime folder, `TMPDIR`, `TMP`, `TEMP`, `XDG_RUNTIME_DIR` and `RUSHD_RUNTIME_DIR` +never select a different daemon; each operation receives the requesting client's values. +Clients and daemons before protocol 0.12 used `$XDG_RUNTIME_DIR/rushd-/` or the +temporary folder instead. A daemon started there stays there, where current clients do not +look, until it idles out or is stopped with that older client (`rush-client daemon stop`). +An older client that starts a current engine while `XDG_RUNTIME_DIR` or `TMPDIR` is set does +not find it and runs in-process, so upgrade `rush-cli-client` with the engine. +When a daemon before protocol 0.12 serves a current client, for example one listening in +`/tmp/rushd-/`, the client leaves `XDG_RUNTIME_DIR`, `TMPDIR`, `TMP` and `TEMP` out of +its requests on Linux and macOS, because that daemon would restart into the folder they name. +Its operations see the daemon's own values; stop it (`rush-client daemon stop`) to use yours. + `rush-client daemon status` only connects and checks hello/pong. It never starts a process, reclaims files, or treats a PID file as evidence of readiness. Both commands print one JSON object with `state: "ready"`, `socketPath`, and the actual pong fields (`uptimeMs`, available versions, optional `pid` and `residentMemoryBytes`, and an optional `workspace` snapshot). Exit code 0 means protocol readiness, not build support. An unreachable/incompatible endpoint, invalid -arguments, or startup failure returns exit code 1 with a diagnostic. +arguments, or startup failure returns exit code 1 with a diagnostic. When no daemon runs +at all, status also exits 1, and its diagnostic says `No daemon is running for +(Rush )` and what the next command does: with `enabled` and `autoStart`, the next +rush-client command that uses the daemon starts one; otherwise rush-client commands run Rush +in-process. This is the normal state after `daemon stop`, the idle timeout or SIGTERM. Status +reports it only when the endpoint refuses connections and neither an ownership record, a startup +reservation nor (outside Windows) a socket file remains; with any of these, the diagnostic +still says that it could not connect. When the endpoint +refuses connections and its ownership record (`.pid.json`) names a PID that no longer +exists, the diagnostic adds that rushd exited without shutting down (an orderly shutdown +removes the record) and that `daemon logs` may show why. When that PID still exists but the endpoint +refuses connections or does not complete hello/ping, the diagnostic adds what that process is doing, +on Linux for example `rushd (PID ) still owns .pid.json: it is stopped (state T), for example +by SIGSTOP, and it started 5 min ago.`, and on the next line what to do about it. +A client that lost its connection to that daemon, or that ran Rush in-process, removes the record +when it reclaims the daemon, and appends a line that names the daemon to the launcher log. Until a +daemon becomes ready again or `daemon stop --force` resets the workspace, the `No daemon is running` +diagnostic then adds `The last daemon, rushd (PID ), exited without shutting down; "rush-client +daemon logs" may show why.` +A daemon whose installation was removed or replaced still answers, but it restarts on +the next command: status then prints `state: "installationChanged"` with the pong's +`installationChange` (`change` and `folder`), a hint on stderr, and exits with code 1. + +A startup reservation (`.pid.json.starting`) refuses another daemon launch until +the daemon it reserved becomes ready. Status reports one that remains as +`startupReservation` with its `path`, the startup helper's `helperPid` when recorded, and +`helperState`: `running` (the helper still waits for readiness), `exited` (the helper will not +release it), or `unknown` (written by an older client). Status never removes it. Next to a +ready daemon, the next command that uses, stops or restarts that daemon removes it; when status +cannot connect, its diagnostic explains the reservation. After an `exited` helper, status also +reports `relaunchAfter`, 15 seconds after that helper was launched. Until then every automatic +start is refused at once (the command runs in-process), so that a daemon that fails the same way +each time, for example because of a configuration error, is not launched by every command. The +first command after it that finds nothing listening at the endpoint takes the reservation over +and starts the daemon again, and `daemon logs` shows a line saying so. `daemon logs` may also show +why the daemon did not become ready. The optional workspace snapshot reports the provider generation/token, graph existence, and available warm accounting without initializing a graph. Missing fields are unknown, @@ -286,9 +682,25 @@ attests a restart request, not completion of successor startup or success of a c `rush-client daemon stop` requires protocol >= 0.6 and waits for `shutdownAck` followed by EOF. It reports `state: "shutdownAccepted"` with exit code 0; this does not assert successful workspace disposal. Stop is idempotent: when nothing -listens at the endpoint it reports `state: "notRunning"` with exit code 0. An +listens at the endpoint and no daemon is starting, it reports `state: "notRunning"` with exit code 0. +When nothing listens but the ownership record names a process that still runs, for example a daemon +that removed its socket while it shuts down, stop first waits up to 15 seconds for that process to exit, +and says so on stderr once it has waited a second. If it still runs then, stop exits with code 1 and +says what that process is doing and what to do about it; a daemon that accepts the connection but does +not complete hello/ping gets the same diagnostic. On Linux, stop does not wait for a Rush daemon that +has this workspace's ownership record open and stays stopped (state T or t) while stop samples it for +1.5 s, because it cannot exit before something resumes it: stop exits with code 1 and that diagnostic +once the 1.5 s have passed. +While a daemon is still starting (its startup helper still runs, or another client holds the start +mutex), stop says so on stderr and waits up to 15 seconds for that daemon to become ready, then stops +it as below; reporting `notRunning` would leave it running afterwards. If it is still not ready by then, +stop exits with code 1 and leaves it running; run stop again once `daemon status` reports it ready. An unsupported protocol, missing acknowledgement, handshake failure, or timeout -returns exit code 1. It does not auto-start anything. +returns exit code 1. It does not auto-start anything. Before shutdown, it removes a startup +reservation that remains next to that daemon, as restart does, so that the reservation cannot +refuse the next start once the daemon is gone. It does so only for the live owner in the +ownership record, under the start mutex (waiting up to 15 seconds for it); a reservation that +it cannot resolve stays in place and is reported as `startupReservation`. `rush-client daemon stop --force` stops a running daemon the same way, then waits (up to 15 seconds) for it to release its listener and ownership record and removes @@ -298,14 +710,38 @@ any remaining artifacts, such as an abandoned startup reservation, reporting the `state: "reset"` and the `removedPaths` (or `state: "notRunning"` if nothing was left behind). It holds the start mutex, proves that no listener is bound, and refuses (exit 1) while the recorded owner PID still exists and cannot be shown to -be a reused PID. It never kills a process. Automatic startup already reclaims +be a reused PID, saying what that process is doing, that no process was killed, and what to do about it. +It fails the same way, without killing that process, when that process accepts the connection but does +not complete hello/ping, for example because a signal stopped it. When the recorded owner PID no longer exists, the daemon exited without shutting +down and may have left operations running that only its records name, so the reset first stops them +as the next daemon start would: SIGTERM, then SIGKILL 2 seconds later, to the daemon's own process +group and to each operation process group that it recorded whose leader still has the recorded start +time (or has exited, while every live member of the group is in the group's own session and one of +them still has the `RUSHD_OPERATION_GROUPS` variable that the daemon gives the processes it starts). When a +process that started after the record was written has the recorded PID now, the daemon exited the same +way, so the reset stops the operation process groups that it recorded the same way, but never the +process group whose ID is that PID, which the later process may lead. It prints the +line shown above for a lost connection and reports what it stopped in `orphansReaped` (`daemonPid`, +`processGroupIds`, `outcome`). A recorded group that it cannot prove, such as a PID that a later +process now has, or a group whose leader has exited under a daemon from a release that did not set +`RUSHD_OPERATION_GROUPS`, is not signalled; its record is removed with the others. When such a group still +has a live process, the launcher log that `daemon logs` prints gets a line that names the group and the +daemon and says which check the group failed; nothing about it is printed. The reclaims after a lost +connection, before Rush runs in-process and before a daemon start do the same. If the operations cannot be +stopped, it exits with code 1 and removes nothing. While another process reclaims the same files it +also exits with code 1, except that after a shutdown it re-checks for up to 15 seconds. Otherwise it +never signals a process. Automatic startup already reclaims the common leftovers on its own (see below); this is the documented escape hatch -that every fail-closed startup message points to. +that every fail-closed startup message points to. A reset also appends a line to the launcher log +that clears the report of a daemon that a client reclaimed, so status no longer names it. `rush-client daemon restart` first verifies that the selected Rush version has a launcher and captures the original lock's PID/start timestamp, checking that it matches pong's positive PID and the selected endpoint, then performs acknowledged -shutdown. It waits for original ownership release or a demonstrably dead owner +shutdown. Before shutdown, it removes a startup reservation that remains next to that +daemon (under the start mutex), so that the reservation cannot refuse the successor; if +another client holds the mutex for 15 seconds, restart fails without stopping the daemon. +It waits for original ownership release or a demonstrably dead owner before calling the existing locked starter. A live owner fails closed at the startup deadline; no PID is killed and no live ownership record is deleted. A newly @@ -313,11 +749,15 @@ started/reused successor must pass hello/ping before reporting `state: "ready"`. When nothing listens at the endpoint, restart starts a daemon exactly like `daemon start`. Automatic and explicit startup reclaim stale artifacts only when that is provably -safe: while holding the start mutex with no `.starting` reservation, a socket +safe: while holding the start mutex with no `.starting` reservation (or after taking over +one whose helper exited, as described above), a socket without an ownership record, or an unreadable/corrupt record, is removed only after a connection attempt is refused (so no listener exists). On Linux, a record whose PID now belongs to a process that started after the record's `startedAt` (PID reuse) -is treated as dead; other platforms fail closed and point to `daemon stop --force`. +is treated as dead; other platforms fail closed and point to `daemon stop --force`. Before such a +record is removed, the operation process groups that the daemon recorded are stopped as +`daemon stop --force` stops them, and the line shown above for a lost connection is printed; if they +cannot be stopped, the command fails and the record stays. Restart is explicit even when automatic startup or CI execution routing is disabled, but conflicts with `--no-daemon`. The two-phase host retains ownership @@ -334,6 +774,14 @@ parent closes its descriptor after spawning. On POSIX the launcher enforces mode `0600` and rejects linked destinations; Windows uses the existing per-user transport directory permissions. +Each daemon writes `