From f0dab80e6f33d6a6d089ab2e56d0367cac4343a9 Mon Sep 17 00:00:00 2001 From: Sean Larkin <3408176+TheLarkInn@users.noreply.github.com> Date: Sun, 27 Sep 2026 16:25:35 +0000 Subject: [PATCH] [heft-native] Round 4: SIMD, SHA-NI, syscall + static-link squeeze Stacked on the Rust rewrite. Pure apps/heft-native changes. - SIMD JSON byte-scans (SSE2/AVX2, src/simd) behind runtime CPU detection with a scalar fallback and a kill switch. Isolated 8.6x at 256 B, up to 26x on 4 KB runs, 10.3x on CJK text. - SHA-NI file hashing and forked workers for large trees: 64 MiB asset up-to-date 6.1x, 10k-file clean 3.9x vs the Rust base. - Syscall cuts on the Tier 0 path: heft --help 211 -> 158 syscalls, kernel time -26%. Static glibc link. - Tier 0 wall 1.40-1.55x, peak RSS -48..-50% vs the Rust base; heft --help 2.0 ms (203x vs heft 1.3.1). Node-plugin tiers unchanged (Node-bound). Output byte-identical, exit codes too. std only, no crates, no Node bindings. unsafe only in src/sys and src/simd. ~285M SIMD-vs-scalar fuzz cases, 0 mismatches. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- apps/heft-native/.cargo/config.toml | 2 + apps/heft-native/README.md | 42 +++- apps/heft-native/build.rs | 35 +++ apps/heft-native/pgo-release-build.sh | 102 ++++++++ apps/heft-native/src/builtin/build_info.rs | 58 ++--- .../src/builtin/build_info_json.rs | 48 ++-- .../src/builtin/build_info_json_tests.rs | 26 ++- apps/heft-native/src/builtin/builtin_task.rs | 68 ++++-- .../src/builtin/copy_descriptors.rs | 137 +++++++++++ .../src/builtin/copy_descriptors_tests.rs | 70 ++++++ apps/heft-native/src/builtin/copy_files.rs | 218 +++++++++--------- apps/heft-native/src/builtin/delete_files.rs | 62 +++-- .../src/builtin/deletion_permissions.rs | 99 +++++--- .../src/builtin/file_operations.rs | 73 +++--- .../heft-native/src/builtin/file_selection.rs | 12 +- apps/heft-native/src/builtin/mod.rs | 21 +- .../heft-native/src/builtin/modified_paths.rs | 72 ++++++ .../src/builtin/node_file_system_error.rs | 17 ++ apps/heft-native/src/builtin/node_rimraf.rs | 78 +++++-- .../heft-native/src/builtin/open_file_copy.rs | 88 +++++++ .../heft-native/src/builtin/parallel_items.rs | 184 +++++++++++++++ .../src/builtin/parallel_items_tests.rs | 84 +++++++ apps/heft-native/src/builtin/path_hash.rs | 76 ++++++ apps/heft-native/src/builtin/posix_path.rs | 50 ++++ apps/heft-native/src/builtin/sha256.rs | 149 ++++++------ apps/heft-native/src/builtin/simple_glob.rs | 19 +- .../src/builtin/worker_transfer.rs | 89 +++++++ apps/heft-native/src/cli/action_invocation.rs | 4 +- apps/heft-native/src/cli/action_text.rs | 5 +- apps/heft-native/src/cli/actions.rs | 76 ++++-- apps/heft-native/src/cli/defined_parameter.rs | 21 +- apps/heft-native/src/cli/entry.rs | 13 +- apps/heft-native/src/cli/help_args.rs | 116 +++++----- apps/heft-native/src/cli/help_builders.rs | 22 +- apps/heft-native/src/cli/help_format.rs | 110 +++++---- apps/heft-native/src/cli/help_lines.rs | 31 ++- apps/heft-native/src/cli/help_model.rs | 22 +- apps/heft-native/src/cli/help_usage.rs | 7 +- apps/heft-native/src/cli/invocation.rs | 16 +- apps/heft-native/src/cli/mod.rs | 3 + apps/heft-native/src/cli/parameters.rs | 32 ++- .../heft-native/src/cli/phase_action_check.rs | 29 +++ apps/heft-native/src/cli/registration.rs | 30 ++- apps/heft-native/src/cli/render.rs | 9 +- apps/heft-native/src/cli/run_invocation.rs | 6 +- .../heft-native/src/cli/tests_registration.rs | 69 ++++++ .../src/config/embedded_schemas.rs | 4 +- apps/heft-native/src/config/fs_probe.rs | 162 ++++++------- .../heft-native/src/config/heft_json_chain.rs | 4 +- ...eft_plugin_schema_without_annotations.json | 1 + .../heft_schema_without_annotations.json | 1 + .../src/config/javascript_order.rs | 22 +- apps/heft-native/src/config/loader.rs | 27 ++- apps/heft-native/src/config/mod.rs | 7 +- apps/heft-native/src/config/node_path.rs | 60 ++++- apps/heft-native/src/config/node_resolve.rs | 21 +- apps/heft-native/src/config/package_json.rs | 19 +- .../src/config/path_component_cache.rs | 160 +++++++++++++ apps/heft-native/src/config/path_probes.rs | 132 +++++++++++ .../heft-native/src/config/plugin_manifest.rs | 11 +- apps/heft-native/src/config/plugin_options.rs | 75 ++++-- .../src/config/real_path_resolver.rs | 104 --------- apps/heft-native/src/config/rig.rs | 11 +- .../src/config/tests_embedded_schemas.rs | 82 +++++++ .../src/config/tests_node_path_scan.rs | 40 ++++ .../src/config/tests_node_paths.rs | 52 ++++- .../src/config/tests_real_path_resolver.rs | 43 ++-- apps/heft-native/src/host_link/plan_file.rs | 9 +- apps/heft-native/src/host_link/warm_client.rs | 18 +- apps/heft-native/src/json/lexer_string.rs | 10 +- apps/heft-native/src/json/lexer_trivia.rs | 32 ++- apps/heft-native/src/json/writer.rs | 63 +++-- apps/heft-native/src/main.rs | 8 +- apps/heft-native/src/run/mod.rs | 53 +++-- .../src/run/tier_zero_execution.rs | 34 ++- apps/heft-native/src/run/tier_zero_plan.rs | 20 +- apps/heft-native/src/simd/mod.rs | 68 ++++++ apps/heft-native/src/simd/scalar_scans.rs | 43 ++++ apps/heft-native/src/simd/tests_microbench.rs | 80 +++++++ .../src/simd/tests_scan_equivalence.rs | 113 +++++++++ apps/heft-native/src/simd/tests_scan_fuzz.rs | 87 +++++++ .../src/simd/tests_sha256_equivalence.rs | 61 +++++ .../heft-native/src/simd/x86_64_avx2_scans.rs | 108 +++++++++ apps/heft-native/src/simd/x86_64_dispatch.rs | 82 +++++++ apps/heft-native/src/simd/x86_64_sha256.rs | 124 ++++++++++ .../heft-native/src/simd/x86_64_sse2_scans.rs | 198 ++++++++++++++++ apps/heft-native/src/sys/mod.rs | 23 ++ .../src/sys/no_worker_processes.rs | 17 ++ apps/heft-native/src/sys/process_entry.rs | 8 + .../src/sys/signal_dispositions.rs | 5 +- apps/heft-native/src/sys/standard_streams.rs | 76 ++++++ apps/heft-native/src/sys/worker_processes.rs | 45 ++++ apps/heft-native/src/terminal/heft_console.rs | 87 ++++--- .../src/terminal/heft_console_tests.rs | 33 +++ apps/heft-native/src/terminal/mod.rs | 2 + 95 files changed, 4189 insertions(+), 956 deletions(-) create mode 100644 apps/heft-native/.cargo/config.toml create mode 100644 apps/heft-native/build.rs create mode 100644 apps/heft-native/pgo-release-build.sh create mode 100644 apps/heft-native/src/builtin/copy_descriptors.rs create mode 100644 apps/heft-native/src/builtin/copy_descriptors_tests.rs create mode 100644 apps/heft-native/src/builtin/modified_paths.rs create mode 100644 apps/heft-native/src/builtin/open_file_copy.rs create mode 100644 apps/heft-native/src/builtin/parallel_items.rs create mode 100644 apps/heft-native/src/builtin/parallel_items_tests.rs create mode 100644 apps/heft-native/src/builtin/path_hash.rs create mode 100644 apps/heft-native/src/builtin/worker_transfer.rs create mode 100644 apps/heft-native/src/cli/phase_action_check.rs create mode 100644 apps/heft-native/src/cli/tests_registration.rs create mode 100644 apps/heft-native/src/config/heft_plugin_schema_without_annotations.json create mode 100644 apps/heft-native/src/config/heft_schema_without_annotations.json create mode 100644 apps/heft-native/src/config/path_component_cache.rs create mode 100644 apps/heft-native/src/config/path_probes.rs delete mode 100644 apps/heft-native/src/config/real_path_resolver.rs create mode 100644 apps/heft-native/src/config/tests_embedded_schemas.rs create mode 100644 apps/heft-native/src/config/tests_node_path_scan.rs create mode 100644 apps/heft-native/src/simd/mod.rs create mode 100644 apps/heft-native/src/simd/scalar_scans.rs create mode 100644 apps/heft-native/src/simd/tests_microbench.rs create mode 100644 apps/heft-native/src/simd/tests_scan_equivalence.rs create mode 100644 apps/heft-native/src/simd/tests_scan_fuzz.rs create mode 100644 apps/heft-native/src/simd/tests_sha256_equivalence.rs create mode 100644 apps/heft-native/src/simd/x86_64_avx2_scans.rs create mode 100644 apps/heft-native/src/simd/x86_64_dispatch.rs create mode 100644 apps/heft-native/src/simd/x86_64_sha256.rs create mode 100644 apps/heft-native/src/simd/x86_64_sse2_scans.rs create mode 100644 apps/heft-native/src/sys/no_worker_processes.rs create mode 100644 apps/heft-native/src/sys/process_entry.rs create mode 100644 apps/heft-native/src/sys/standard_streams.rs create mode 100644 apps/heft-native/src/sys/worker_processes.rs create mode 100644 apps/heft-native/src/terminal/heft_console_tests.rs diff --git a/apps/heft-native/.cargo/config.toml b/apps/heft-native/.cargo/config.toml new file mode 100644 index 0000000000..aafd54379f --- /dev/null +++ b/apps/heft-native/.cargo/config.toml @@ -0,0 +1,2 @@ +[target.x86_64-unknown-linux-gnu] +rustflags = ["-C", "target-feature=+crt-static"] diff --git a/apps/heft-native/README.md b/apps/heft-native/README.md index 7366050229..87b967dfac 100644 --- a/apps/heft-native/README.md +++ b/apps/heft-native/README.md @@ -14,6 +14,28 @@ cargo build --release The binary is written to `apps/heft-native/target/release/heft` (`target/` is ignored by Git). The crate has no dependencies; `Cargo.lock` only lists `heft-native` itself. +On x86_64 Linux the release binary is linked statically against glibc (`.cargo/config.toml` adds +`-C target-feature=+crt-static`), as a position-independent executable whose relative relocations are packed +(`-z pack-relative-relocs`, added by `build.rs` only when the build machine has glibc 2.36 or newer, whose static +startup code applies them). A static binary starts without the dynamic loader, so it starts faster and uses less +memory. The binary provides its own C `main` (`src/sys/process_entry.rs`) instead of the Rust runtime's, which +skips work the binary does not need at startup, such as reading `/proc/self/maps` to locate the main thread's +stack. + +To build a dynamically linked binary instead, override the flags: + +```bash +RUSTFLAGS="-C target-feature=-crt-static" cargo build --release +``` + +### Profile-guided build (optional) + +`bash pgo-release-build.sh` builds an instrumented binary, runs it on a generated project that only uses Heft's +built-in tasks and on the `--help` pages of the build tests, merges the profile with `llvm-profdata` (from +`rustup component add llvm-tools`) and rebuilds `target/release/heft` with it. The profile is regenerated from the +current code on every run, so it never goes stale. The result is a few percent faster on native runs but about +60 KB larger, so it is not the default build. + ## Using it Run the binary instead of `heft` from a project folder: @@ -84,13 +106,6 @@ file descriptors) makes the host refuse and the binary run cold. When no host ac cold as usual and starts a host (`lib-commonjs/host/WarmHostEntry.js`) in the background for the next run, unless four hosts are already running for the user. -## Static build (optional) - -`cargo build --release --target x86_64-unknown-linux-musl` (after `rustup target add x86_64-unknown-linux-musl`) -produces a statically linked binary in `target/x86_64-unknown-linux-musl/release/heft`. It is about as fast as the -default build and uses less memory (about 1 MB instead of 2.8 MB peak for a native build), but it is about 90 KB -larger. - ## Layout Each folder under `src/` is a module with a single owner: @@ -101,8 +116,15 @@ Each folder under `src/` is a module with a single owner: | `cli` | command line model, parsing, help and error rendering | | `config` | `heft.json`, rigs, `heft-plugin.json`, plugin options, package resolution | | `graph`, `run`, `builtin`, `terminal` | phase and task graph, execution, native built-in plugins, terminal output | -| `sys` | the only place with `unsafe` code (minimal operating system calls) | +| `simd` | SSE2/AVX2 byte scans used by the JSON lexer and writer, each with a scalar twin that gives identical results | +| `sys` | minimal operating system calls; with `simd` the only places with `unsafe` code | | `process`, `version`, `host_link` | running Node.js, selecting the Heft version, the connection to the JavaScript plugin host | -The code follows these rules: only the Rust standard library, `#![deny(unsafe_code)]` outside `src/sys`, no -comments in the code, source files of at most 200 lines, and a stripped release binary of at most 1 MB. +The code follows these rules: only the Rust standard library, `#![deny(unsafe_code)]` outside `src/sys` and +`src/simd`, no comments in the code, source files of at most 200 lines, and a stripped release binary of at most +1 MB. + +SIMD code runs only on x86_64. AVX2 is used after `is_x86_feature_detected!("avx2")` confirms it, and SSE2 +otherwise; every other architecture uses the scalar twins. Setting `HEFT_NATIVE_NO_SIMD=1` (any non-empty value +except `0`, or `HEFT_NATIVE_SIMD=0`) makes the whole process use the scalar twins, which is how the SIMD paths are +checked against them. diff --git a/apps/heft-native/build.rs b/apps/heft-native/build.rs new file mode 100644 index 0000000000..9b33d6d6c4 --- /dev/null +++ b/apps/heft-native/build.rs @@ -0,0 +1,35 @@ +const GLIBC_FEATURES_HEADER: &str = "/usr/include/features.h"; +const FIRST_GLIBC_MINOR_VERSION_WITH_STATIC_RELR: u32 = 36; + +fn main() { + println!("cargo:rerun-if-changed=build.rs"); + println!("cargo:rerun-if-changed={GLIBC_FEATURES_HEADER}"); + if static_glibc_startup_applies_packed_relative_relocations() { + println!("cargo:rustc-link-arg-bins=-Wl,-z,pack-relative-relocs"); + } +} + +fn static_glibc_startup_applies_packed_relative_relocations() -> bool { + let target_is_static_glibc = std::env::var("CARGO_CFG_TARGET_OS").as_deref() == Ok("linux") + && std::env::var("CARGO_CFG_TARGET_ENV").as_deref() == Ok("gnu") + && std::env::var("CARGO_CFG_TARGET_FEATURE") + .unwrap_or_default() + .split(',') + .any(|target_feature| target_feature == "crt-static"); + target_is_static_glibc && glibc_version_of_build_host().is_some_and(|(major, minor)| { + major > 2 || (major == 2 && minor >= FIRST_GLIBC_MINOR_VERSION_WITH_STATIC_RELR) + }) +} + +fn glibc_version_of_build_host() -> Option<(u32, u32)> { + let features_header = std::fs::read_to_string(GLIBC_FEATURES_HEADER).ok()?; + let defined_number = |macro_name: &str| { + features_header.lines().find_map(|header_line| { + let mut words = header_line.split_whitespace(); + (words.next() == Some("#define") && words.next() == Some(macro_name)) + .then(|| words.next()?.parse::().ok()) + .flatten() + }) + }; + Some((defined_number("__GLIBC__")?, defined_number("__GLIBC_MINOR__")?)) +} diff --git a/apps/heft-native/pgo-release-build.sh b/apps/heft-native/pgo-release-build.sh new file mode 100644 index 0000000000..5eaef411a7 --- /dev/null +++ b/apps/heft-native/pgo-release-build.sh @@ -0,0 +1,102 @@ +#!/usr/bin/env bash +set -euo pipefail + +crate_folder="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +heft_package_folder="$(cd "$crate_folder/../heft" && pwd)" +build_tests_folder="$(cd "$crate_folder/../.." && pwd)/build-tests" +host_triple="$(rustc -vV | sed -n 's/^host: //p')" +llvm_profdata="$(rustc --print sysroot)/lib/rustlib/$host_triple/bin/llvm-profdata" +profile_folder="$crate_folder/target/pgo-profiles" +instrumented_target_folder="$crate_folder/target/pgo-instrumented" +flag_separator=$'\x1f' + +if [ ! -x "$llvm_profdata" ]; then + echo "llvm-profdata was not found; install it with: rustup component add llvm-tools" >&2 + exit 1 +fi + +rm -rf "$profile_folder" +CARGO_ENCODED_RUSTFLAGS="-Ctarget-feature=+crt-static${flag_separator}-Cprofile-generate=$profile_folder" \ + cargo build --release --manifest-path "$crate_folder/Cargo.toml" --target-dir "$instrumented_target_folder" +instrumented_heft="$instrumented_target_folder/release/heft" + +workload_folder="$(mktemp -d)" +trap 'rm -rf "$workload_folder"' EXIT +mkdir -p "$workload_folder/config" "$workload_folder/src/assets" "$workload_folder/temp/scratch" "$workload_folder/node_modules/@rushstack" +ln -s "$heft_package_folder" "$workload_folder/node_modules/@rushstack/heft" +cat > "$workload_folder/package.json" <<'PACKAGE_JSON' +{ + "name": "heft-native-pgo-workload", + "version": "1.0.0", + "private": true, + "devDependencies": { + "@rushstack/heft": "*" + } +} +PACKAGE_JSON +cat > "$workload_folder/config/heft.json" <<'HEFT_JSON' +{ + "$schema": "https://developer.microsoft.com/json-schemas/heft/v0/heft.schema.json", + "phasesByName": { + "build": { + "cleanFiles": [{ "includeGlobs": ["lib"] }], + "tasksByName": { + "set-env": { + "taskPlugin": { + "pluginPackage": "@rushstack/heft", + "pluginName": "set-environment-variables-plugin", + "options": { "environmentVariablesToSet": { "HEFT_NATIVE_PGO_WORKLOAD": "1" } } + } + }, + "copy-assets": { + "taskDependencies": ["set-env"], + "taskPlugin": { + "pluginPackage": "@rushstack/heft", + "pluginName": "copy-files-plugin", + "options": { + "copyOperations": [ + { "sourcePath": "src/assets", "destinationFolders": ["lib/assets"], "fileExtensions": [".txt", ".json"] } + ] + } + } + }, + "delete-scratch": { + "taskDependencies": ["copy-assets"], + "taskPlugin": { + "pluginPackage": "@rushstack/heft", + "pluginName": "delete-files-plugin", + "options": { "deleteOperations": [{ "sourcePath": "temp/scratch", "includeGlobs": ["**/*"] }] } + } + } + } + } + } +} +HEFT_JSON +for asset_number in $(seq 1 40); do + echo "heft native pgo asset $asset_number" > "$workload_folder/src/assets/asset-$asset_number.txt" +done +echo '{"workload":"pgo"}' > "$workload_folder/src/assets/data.json" + +run_native_workload() { + for repetition in 1 2 3 4 5; do + "$instrumented_heft" --help || true + "$instrumented_heft" --version || true + "$instrumented_heft" build --help || true + "$instrumented_heft" build --clean || true + "$instrumented_heft" build || true + "$instrumented_heft" nosuch-action || true + "$instrumented_heft" clean || true + done +} +(cd "$workload_folder" && run_native_workload) < /dev/null > /dev/null 2>&1 + +for project_folder in "$build_tests_folder"/*/; do + if [ -e "$project_folder/node_modules/@rushstack/heft/package.json" ]; then + (cd "$project_folder" && { "$instrumented_heft" --help || true; "$instrumented_heft" build --help || true; }) < /dev/null > /dev/null 2>&1 + fi +done + +"$llvm_profdata" merge -o "$profile_folder/merged.profdata" "$profile_folder"/*.profraw +CARGO_ENCODED_RUSTFLAGS="-Ctarget-feature=+crt-static${flag_separator}-Cprofile-use=$profile_folder/merged.profdata" \ + cargo build --release --manifest-path "$crate_folder/Cargo.toml" diff --git a/apps/heft-native/src/builtin/build_info.rs b/apps/heft-native/src/builtin/build_info.rs index a6ec3d44be..00a155e741 100644 --- a/apps/heft-native/src/builtin/build_info.rs +++ b/apps/heft-native/src/builtin/build_info.rs @@ -1,11 +1,13 @@ -use std::collections::HashMap; use std::fs::{self, File}; use std::io::{BufWriter, ErrorKind, Write}; -use super::build_info_json::{is_array_index_key, parse_build_info_json}; +use super::build_info_json::{is_array_index_key, parse_build_info_json, BuildInfoJson}; use super::javascript_json::append_json_string; use super::node_file_system_error::{is_node_not_exist_error, NodeFileSystemError}; -use super::posix_path::{directory_name, relative_path, resolve_path}; +use super::path_hash::path_hash_set_with_capacity; +use super::posix_path::{ + directory_name, is_normalized_absolute_folder_path, relative_path, resolve_path, resolve_relative_path_against_normalized_folder, +}; pub struct IncrementalBuildInfo { pub configuration_hash: String, @@ -24,42 +26,30 @@ pub fn try_read_build_info(build_info_path: &str) -> BuildInfoReadResult { Err(error) if is_node_not_exist_error(&error) => return BuildInfoReadResult::Missing, Err(_) => return BuildInfoReadResult::NeedsJavaScript, }; - let Some(parsed) = std::str::from_utf8(&bytes).ok().and_then(parse_build_info_json) else { + let Some(BuildInfoJson { configuration_hash, input_file_versions: relative_file_versions }) = + std::str::from_utf8(&bytes).ok().and_then(parse_build_info_json) + else { return BuildInfoReadResult::NeedsJavaScript; }; - drop(bytes); let base_folder_path = directory_name(build_info_path); - let mut input_file_versions: Vec<(String, String)> = parsed - .input_file_versions + let base_folder_is_normalized = is_normalized_absolute_folder_path(base_folder_path); + let input_file_versions: Vec<(String, String)> = relative_file_versions .into_iter() - .map(|(relative_file_path, version)| (resolve_path(base_folder_path, &relative_file_path), version)) + .map(|(relative_file_path, version)| { + let absolute_file_path = base_folder_is_normalized + .then(|| resolve_relative_path_against_normalized_folder(base_folder_path, &relative_file_path)) + .flatten() + .unwrap_or_else(|| resolve_path(base_folder_path, &relative_file_path)); + (absolute_file_path, version) + }) .collect(); - let mut duplicates: Vec<(usize, usize)> = Vec::new(); - { - let mut first_index_by_path: HashMap<&str, usize> = HashMap::with_capacity(input_file_versions.len()); - for (index, (absolute_file_path, _)) in input_file_versions.iter().enumerate() { - if let Some(&first_index) = first_index_by_path.get(absolute_file_path.as_str()) { - duplicates.push((first_index, index)); - } else { - first_index_by_path.insert(absolute_file_path, index); - } - } - } - if !duplicates.is_empty() { - for &(first_index, duplicate_index) in &duplicates { - input_file_versions[first_index].1 = std::mem::take(&mut input_file_versions[duplicate_index].1); - } - let mut is_duplicate = vec![false; input_file_versions.len()]; - for &(_, duplicate_index) in &duplicates { - is_duplicate[duplicate_index] = true; - } - let mut flags = is_duplicate.into_iter(); - input_file_versions.retain(|_| !flags.next().unwrap_or(false)); + drop(bytes); + let mut seen_paths = path_hash_set_with_capacity::<&str>(input_file_versions.len()); + if !input_file_versions.iter().all(|(absolute_file_path, _)| seen_paths.insert(absolute_file_path)) { + return BuildInfoReadResult::NeedsJavaScript; } - BuildInfoReadResult::Found(IncrementalBuildInfo { - configuration_hash: parsed.configuration_hash, - input_file_versions, - }) + drop(seen_paths); + BuildInfoReadResult::Found(IncrementalBuildInfo { configuration_hash, input_file_versions }) } pub fn write_build_info<'entries>( @@ -71,7 +61,7 @@ pub fn write_build_info<'entries>( let relative_entries = input_file_versions.map(|(absolute_file_path, version)| (relative_path(base_folder_path, absolute_file_path), version)); let mut array_index_entries: Vec<(String, &str)> = relative_entries.clone().filter(|(key, _)| is_array_index_key(key)).collect(); - array_index_entries.sort_by_key(|(key, _)| key.parse::().unwrap_or(0)); + array_index_entries.sort_unstable_by_key(|(key, _)| key.parse::().unwrap_or(0)); let named_entries = relative_entries.filter(|(key, _)| !is_array_index_key(key)); let file = create_file_ensuring_folder_exists(build_info_path)?; let mut writer = BufWriter::with_capacity(16384, file); diff --git a/apps/heft-native/src/builtin/build_info_json.rs b/apps/heft-native/src/builtin/build_info_json.rs index 6ccb087723..b27cf3c8eb 100644 --- a/apps/heft-native/src/builtin/build_info_json.rs +++ b/apps/heft-native/src/builtin/build_info_json.rs @@ -1,9 +1,11 @@ -pub struct BuildInfoJson { +use std::borrow::Cow; + +pub struct BuildInfoJson<'text> { pub configuration_hash: String, - pub input_file_versions: Vec<(String, String)>, + pub input_file_versions: Vec<(Cow<'text, str>, String)>, } -pub fn parse_build_info_json(text: &str) -> Option { +pub fn parse_build_info_json(text: &str) -> Option> { let mut reader = StrictJsonReader { bytes: text.as_bytes(), position: 0 }; let mut configuration_hash = None; let mut input_file_versions = None; @@ -11,8 +13,8 @@ pub fn parse_build_info_json(text: &str) -> Option { loop { let key = reader.read_string()?; reader.expect_byte(b':')?; - match key.as_str() { - "configHash" if configuration_hash.is_none() => configuration_hash = Some(reader.read_string()?), + match &*key { + "configHash" if configuration_hash.is_none() => configuration_hash = Some(reader.read_string()?.into_owned()), "inputFileVersions" if input_file_versions.is_none() => input_file_versions = Some(reader.read_string_map()?), _ => return None, } @@ -42,7 +44,7 @@ struct StrictJsonReader<'text> { position: usize, } -impl StrictJsonReader<'_> { +impl<'text> StrictJsonReader<'text> { fn skip_whitespace(&mut self) { while let Some(b' ' | b'\t' | b'\n' | b'\r') = self.bytes.get(self.position) { self.position += 1; @@ -65,10 +67,9 @@ impl StrictJsonReader<'_> { } } - fn read_string_map(&mut self) -> Option> { + fn read_string_map(&mut self) -> Option, String)>> { self.expect_byte(b'{')?; - let mut entries: Vec<(String, String)> = Vec::new(); - let mut seen_keys: std::collections::HashSet = std::collections::HashSet::new(); + let mut entries: Vec<(Cow<'text, str>, String)> = Vec::new(); self.skip_whitespace(); if self.bytes.get(self.position) == Some(&b'}') { self.position += 1; @@ -77,8 +78,8 @@ impl StrictJsonReader<'_> { loop { let key = self.read_string()?; self.expect_byte(b':')?; - let value = self.read_string()?; - if is_array_index_key(&key) || !seen_keys.insert(key.clone()) { + let value = self.read_string()?.into_owned(); + if is_array_index_key(&key) { return None; } entries.push((key, value)); @@ -88,23 +89,32 @@ impl StrictJsonReader<'_> { } } - fn read_string(&mut self) -> Option { + fn read_string(&mut self) -> Option> { self.expect_byte(b'"')?; - let mut value = String::new(); + let bytes: &'text [u8] = self.bytes; + let mut value: Option = None; loop { let start = self.position; - while let Some(&byte) = self.bytes.get(self.position) { + while let Some(&byte) = bytes.get(self.position) { if byte == b'"' || byte == b'\\' || byte < 0x20 { break; } self.position += 1; } - value.push_str(std::str::from_utf8(&self.bytes[start..self.position]).ok()?); - let byte = *self.bytes.get(self.position)?; + let unescaped_run = std::str::from_utf8(&bytes[start..self.position]).ok()?; + let byte = *bytes.get(self.position)?; self.position += 1; - match byte { - b'"' => return Some(value), - b'\\' => value.push(self.read_escape()?), + match (byte, value.as_mut()) { + (b'"', None) => return Some(Cow::Borrowed(unescaped_run)), + (b'"', Some(escaped_value)) => { + escaped_value.push_str(unescaped_run); + return value.map(Cow::Owned); + } + (b'\\', _) => { + let escaped_value = value.get_or_insert_with(String::new); + escaped_value.push_str(unescaped_run); + escaped_value.push(self.read_escape()?); + } _ => return None, } } diff --git a/apps/heft-native/src/builtin/build_info_json_tests.rs b/apps/heft-native/src/builtin/build_info_json_tests.rs index dbf0efa6c6..488ca64d71 100644 --- a/apps/heft-native/src/builtin/build_info_json_tests.rs +++ b/apps/heft-native/src/builtin/build_info_json_tests.rs @@ -1,7 +1,11 @@ +use super::build_info::{try_read_build_info, BuildInfoReadResult}; use super::build_info_json::{is_array_index_key, parse_build_info_json}; fn entries(text: &str) -> Option<(String, Vec<(String, String)>)> { - parse_build_info_json(text).map(|parsed| (parsed.configuration_hash, parsed.input_file_versions)) + parse_build_info_json(text).map(|parsed| { + let versions = parsed.input_file_versions.into_iter().map(|(key, version)| (key.into_owned(), version)).collect(); + (parsed.configuration_hash, versions) + }) } fn owned(pairs: &[(&str, &str)]) -> Vec<(String, String)> { @@ -29,7 +33,6 @@ fn refuses_everything_outside_the_shape_heft_writes() { r#"{"configHash":"h"}"#, r#"{"configHash":1,"inputFileVersions":{}}"#, r#"{"configHash":"h","inputFileVersions":{"a":1}}"#, - r#"{"configHash":"h","inputFileVersions":{"a":"1","a":"2"}}"#, r#"{"configHash":"h","inputFileVersions":{"7":"1"}}"#, r#"{"configHash":"h","inputFileVersions":{},"fileDependencies":{}}"#, r#"{"configHash":"h","configHash":"h","inputFileVersions":{}}"#, @@ -49,3 +52,22 @@ fn array_index_keys_follow_ecmascript() { assert!(!is_array_index_key("") && !is_array_index_key("01") && !is_array_index_key("4294967295")); assert!(!is_array_index_key("-1") && !is_array_index_key("1.5") && !is_array_index_key("../1")); } + +#[test] +fn duplicate_raw_or_resolved_paths_are_left_to_the_javascript_heft() { + let folder = std::env::temp_dir().join(format!("heft-native-build-info-{}", std::process::id())); + std::fs::create_dir_all(&folder).unwrap(); + let state_path = folder.join("file-copy.json"); + let state_path_text = state_path.to_str().unwrap(); + for (text, is_found) in [ + (r#"{"configHash":"h","inputFileVersions":{"../s/a":"1","../s/b":"2"}}"#, true), + (r#"{"configHash":"h","inputFileVersions":{"a":"1","a":"2"}}"#, false), + (r#"{"configHash":"h","inputFileVersions":{"../s/a":"1","../s/x/../a":"2"}}"#, false), + ] { + std::fs::write(&state_path, text).unwrap(); + let result = try_read_build_info(state_path_text); + assert_eq!(matches!(result, BuildInfoReadResult::Found(_)), is_found, "{text}"); + assert_eq!(matches!(result, BuildInfoReadResult::NeedsJavaScript), !is_found, "{text}"); + } + std::fs::remove_dir_all(&folder).unwrap(); +} diff --git a/apps/heft-native/src/builtin/builtin_task.rs b/apps/heft-native/src/builtin/builtin_task.rs index b1cf88dbb3..89e7162c09 100644 --- a/apps/heft-native/src/builtin/builtin_task.rs +++ b/apps/heft-native/src/builtin/builtin_task.rs @@ -1,8 +1,10 @@ -use super::copy_files::{preflight_copy_files, run_copy_files_task, CopyFilesTaskPlan}; +use super::build_info::{try_read_build_info, BuildInfoReadResult}; +use super::copy_files::{run_copy_files_task, CopyFilesPreflight, CopyFilesTaskPlan}; use super::copy_operation::{plan_copy_operations, CopyOperation}; -use super::delete_files::run_delete_operations; -use super::deletion_permissions::selections_are_deletable_without_permission_errors; +use super::delete_files::{run_deletion_plan, DeletionPlan}; +use super::deletion_permissions::preflight_deletions; use super::file_selection::{AbsoluteFileSelection, FileSelectionSpecifier}; +use super::modified_paths::ModifiedPaths; use super::node_file_system_error::NodeFileSystemError; use super::set_environment_variables::{order_like_javascript_object_entries, run_set_environment_variables}; use crate::terminal::ScopedLoggerOutput; @@ -15,7 +17,7 @@ pub enum BuiltinTaskOptions { pub enum PlannedBuiltinTask { CopyFiles(CopyFilesTaskPlan), - DeleteFiles(Vec), + DeleteFiles(DeletionPlan), SetEnvironmentVariables(Vec<(String, String)>), } @@ -38,7 +40,7 @@ pub fn plan_builtin_task( .iter() .map(|operation| operation.to_absolute_selection(build_folder_path)) .collect::>>() - .map(PlannedBuiltinTask::DeleteFiles), + .map(|selections| PlannedBuiltinTask::DeleteFiles(DeletionPlan { selections, preflight_entries: None })), BuiltinTaskOptions::SetEnvironmentVariables(variables) => Some(PlannedBuiltinTask::SetEnvironmentVariables( order_like_javascript_object_entries(variables.clone()), )), @@ -63,14 +65,26 @@ pub fn plan_phase_clean( .collect() } -pub fn builtin_task_touches_files(planned_task: &PlannedBuiltinTask) -> bool { - !matches!(planned_task, PlannedBuiltinTask::SetEnvironmentVariables(_)) +pub fn plan_phase_clean_deletion(selections: Vec, modified_paths: &mut ModifiedPaths) -> Option { + let entries_per_selection = preflight_deletions(&selections)?; + let entries_are_current = selections.iter().all(|selection| modified_paths.selection_is_unchanged(selection)); + let plan = DeletionPlan { selections, preflight_entries: entries_are_current.then_some(entries_per_selection) }; + modified_paths.add_deletion_plan(&plan); + Some(plan) +} + +pub fn builtin_task_does_file_system_work(planned_task: &PlannedBuiltinTask) -> bool { + match planned_task { + PlannedBuiltinTask::CopyFiles(_) => true, + PlannedBuiltinTask::DeleteFiles(plan) => plan.does_file_system_work(), + PlannedBuiltinTask::SetEnvironmentVariables(_) => false, + } } pub fn builtin_task_passes_preflight( planned_task: &mut PlannedBuiltinTask, temp_folder_path: &str, - keep_preflight_results: bool, + modified_paths: &mut ModifiedPaths, ) -> bool { match planned_task { PlannedBuiltinTask::CopyFiles(plan) => { @@ -81,26 +95,54 @@ pub fn builtin_task_passes_preflight( .iter() .all(|folder| folder != temp_folder_path && !folder.starts_with(&temp_folder_prefix)) }); - let Some(preflight) = preflight_copy_files(plan).filter(|_| destinations_are_outside_temp) else { + if !destinations_are_outside_temp || !preflight_copy_task(plan, modified_paths) { + return false; + } + for operation in &plan.operations { + operation.destination_folder_paths.iter().for_each(|folder| modified_paths.add_modified_path(folder)); + } + modified_paths.add_modified_path(&plan.build_info_path); + true + } + PlannedBuiltinTask::DeleteFiles(plan) => { + let Some(entries_per_selection) = preflight_deletions(&plan.selections) else { return false; }; - if keep_preflight_results { - plan.preflight = Some(preflight); + if plan.selections.iter().all(|selection| modified_paths.selection_is_unchanged(selection)) { + plan.preflight_entries = Some(entries_per_selection); } + modified_paths.add_deletion_plan(plan); true } - PlannedBuiltinTask::DeleteFiles(selections) => selections_are_deletable_without_permission_errors(selections), PlannedBuiltinTask::SetEnvironmentVariables(_) => true, } } +fn preflight_copy_task(plan: &mut CopyFilesTaskPlan, modified_paths: &ModifiedPaths) -> bool { + let Some(source_files) = plan.operations.iter().map(|operation| operation.selection.select(false)).collect::>>() else { + return false; + }; + let build_info = if modified_paths.deletes(&plan.build_info_path) { + None + } else { + match try_read_build_info(&plan.build_info_path) { + BuildInfoReadResult::NeedsJavaScript => return false, + build_info => (!modified_paths.overlaps(&plan.build_info_path)).then_some(build_info), + } + }; + if plan.operations.iter().all(|operation| modified_paths.selection_is_unchanged(&operation.selection)) { + plan.preflight = Some(CopyFilesPreflight { source_files, build_info }); + } + true +} + pub fn run_planned_builtin_task( planned_task: PlannedBuiltinTask, output: &ScopedLoggerOutput<'_>, ) -> Result<(), NodeFileSystemError> { match planned_task { PlannedBuiltinTask::CopyFiles(plan) => run_copy_files_task(plan, output), - PlannedBuiltinTask::DeleteFiles(selections) => run_delete_operations(&selections, output), + PlannedBuiltinTask::DeleteFiles(plan) => run_deletion_plan(plan, output), PlannedBuiltinTask::SetEnvironmentVariables(variables) => { run_set_environment_variables(&variables, output); Ok(()) diff --git a/apps/heft-native/src/builtin/copy_descriptors.rs b/apps/heft-native/src/builtin/copy_descriptors.rs new file mode 100644 index 0000000000..655e74f75b --- /dev/null +++ b/apps/heft-native/src/builtin/copy_descriptors.rs @@ -0,0 +1,137 @@ +use std::borrow::Cow; +use std::collections::hash_map::Entry; + +use super::copy_operation::AbsoluteCopyOperation; +use super::node_file_system_error::NodeFileSystemError; +use super::path_hash::{path_hash_map_with_capacity, path_hash_set_with_capacity, PathHashSet}; +use super::posix_path::{base_name, directory_name, path_contains, relative_path}; +use super::simple_glob::GlobbedEntry; + +#[derive(Clone, Copy)] +pub struct CopyDescriptor { + pub operation_index: usize, + pub destination_folder_index: usize, + pub source_index: usize, +} + +pub struct CopyDescriptors<'plan> { + operations: &'plan [AbsoluteCopyOperation], + source_files: &'plan [Vec], + pub descriptors: Vec, + pub sources_are_distinct_and_in_order: bool, +} + +fn path_relative_to_source_folder<'path>(source_folder_path: &str, file_path: &'path str) -> Cow<'path, str> { + let prefix_length = if source_folder_path.ends_with('/') { source_folder_path.len() } else { source_folder_path.len() + 1 }; + match file_path.get(prefix_length..) { + Some(relative) if file_path.starts_with(source_folder_path) && file_path.as_bytes()[prefix_length - 1] == b'/' => { + Cow::Borrowed(relative) + } + _ => Cow::Owned(relative_path(source_folder_path, file_path)), + } +} + +impl<'plan> CopyDescriptors<'plan> { + pub fn source_path_of(&self, descriptor: &CopyDescriptor) -> &'plan str { + self.source_files[descriptor.operation_index][descriptor.source_index].absolute_path.as_str() + } + + pub fn is_hard_link(&self, descriptor: &CopyDescriptor) -> bool { + self.operations[descriptor.operation_index].hardlink + } + + pub fn destination_path_of(&self, descriptor: &CopyDescriptor) -> String { + let operation = &self.operations[descriptor.operation_index]; + let destination_folder_path = &operation.destination_folder_paths[descriptor.destination_folder_index]; + let source_path = self.source_path_of(descriptor); + let destination_relative_path = if operation.flatten { + Cow::Borrowed(base_name(source_path)) + } else { + path_relative_to_source_folder(&operation.selection.source_folder_path, source_path) + }; + let mut destination_path = String::with_capacity(destination_folder_path.len() + destination_relative_path.len() + 1); + destination_path.push_str(destination_folder_path); + if !destination_folder_path.ends_with('/') { + destination_path.push('/'); + } + destination_path.push_str(&destination_relative_path); + destination_path + } + + pub fn copies_may_depend_on_their_order(&self, descriptors_with_work: &[&CopyDescriptor]) -> bool { + if self.sources_are_distinct_and_in_order { + let operation = &self.operations[0]; + let (source_folder_path, destination_folder_path) = + (&operation.selection.source_folder_path, &operation.destination_folder_paths[0]); + return path_contains(source_folder_path, destination_folder_path) + || path_contains(destination_folder_path, source_folder_path); + } + let destination_paths: Vec = + descriptors_with_work.iter().map(|descriptor| self.destination_path_of(descriptor)).collect(); + let destination_path_set: PathHashSet<&str> = destination_paths.iter().map(String::as_str).collect(); + let mut destination_folder_paths = path_hash_set_with_capacity::<&str>(16); + for destination_path in &destination_paths { + let mut folder_path = directory_name(destination_path); + while destination_folder_paths.insert(folder_path) { + folder_path = directory_name(folder_path); + } + } + descriptors_with_work.iter().zip(&destination_paths).any(|(descriptor, destination_path)| { + destination_path_set.contains(self.source_path_of(descriptor)) + || destination_folder_paths.contains(destination_path.as_str()) + }) + } + + fn remove_duplicate_destinations(&mut self) -> Result<(), NodeFileSystemError> { + let mut keep = vec![true; self.descriptors.len()]; + let mut first_index_by_destination = path_hash_map_with_capacity::(self.descriptors.len()); + for (index, candidate) in self.descriptors.iter().enumerate() { + match first_index_by_destination.entry(self.destination_path_of(candidate)) { + Entry::Occupied(occupied) => { + let existing = &self.descriptors[*occupied.get()]; + if self.source_path_of(existing) != self.source_path_of(candidate) + || self.is_hard_link(existing) != self.is_hard_link(candidate) + { + return Err(NodeFileSystemError::from_message(&format!( + "Cannot copy multiple files to the same destination \"{}\".", + occupied.key() + ))); + } + keep[index] = false; + } + Entry::Vacant(vacant) => { + vacant.insert(index); + } + } + } + drop(first_index_by_destination); + let mut keep_flags = keep.into_iter(); + self.descriptors.retain(|_| keep_flags.next().unwrap_or(false)); + Ok(()) + } +} + +pub fn collect_copy_descriptors<'plan>( + operations: &'plan [AbsoluteCopyOperation], + source_files: &'plan [Vec], +) -> Result, NodeFileSystemError> { + let sources_are_distinct_and_in_order = matches!( + operations, + [operation] if operation.destination_folder_paths.len() == 1 + && !operation.flatten + && operation.selection.selects_each_path_once_inside_its_folder() + ); + let mut copy_descriptors = CopyDescriptors { operations, source_files, descriptors: Vec::new(), sources_are_distinct_and_in_order }; + for (operation_index, operation) in operations.iter().enumerate() { + for destination_folder_index in 0..operation.destination_folder_paths.len() { + copy_descriptors.descriptors.extend( + (0..source_files[operation_index].len()) + .map(|source_index| CopyDescriptor { operation_index, destination_folder_index, source_index }), + ); + } + } + if !sources_are_distinct_and_in_order { + copy_descriptors.remove_duplicate_destinations()?; + } + Ok(copy_descriptors) +} diff --git a/apps/heft-native/src/builtin/copy_descriptors_tests.rs b/apps/heft-native/src/builtin/copy_descriptors_tests.rs new file mode 100644 index 0000000000..d4584d054b --- /dev/null +++ b/apps/heft-native/src/builtin/copy_descriptors_tests.rs @@ -0,0 +1,70 @@ +use super::copy_descriptors::{collect_copy_descriptors, CopyDescriptor, CopyDescriptors}; +use super::copy_operation::AbsoluteCopyOperation; +use super::file_selection::AbsoluteFileSelection; +use super::simple_glob::GlobbedEntry; + +fn operation(source_folder_path: &str, include_globs: &[&str], destinations: &[&str], flatten: bool) -> AbsoluteCopyOperation { + AbsoluteCopyOperation { + selection: AbsoluteFileSelection { + source_folder_path: source_folder_path.to_owned(), + include_globs: include_globs.iter().map(|glob| (*glob).to_owned()).collect(), + }, + destination_folder_paths: destinations.iter().map(|folder| (*folder).to_owned()).collect(), + flatten, + hardlink: false, + } +} + +fn files(paths: &[&str]) -> Vec { + paths.iter().map(|path| GlobbedEntry { absolute_path: (*path).to_owned(), is_directory: false }).collect() +} + +fn order_matters(copy_descriptors: &CopyDescriptors<'_>) -> bool { + let all_descriptors: Vec<&CopyDescriptor> = copy_descriptors.descriptors.iter().collect(); + copy_descriptors.copies_may_depend_on_their_order(&all_descriptors) +} + +#[test] +fn a_single_dynamic_operation_keeps_every_source_in_order() { + let operations = [operation("/p/src", &["**/*"], &["/p/lib"], false)]; + let sources = [files(&["/p/src/a", "/p/src/d/b"])]; + let copy_descriptors = collect_copy_descriptors(&operations, &sources).unwrap(); + assert!(copy_descriptors.sources_are_distinct_and_in_order); + let destinations: Vec = + copy_descriptors.descriptors.iter().map(|descriptor| copy_descriptors.destination_path_of(descriptor)).collect(); + assert_eq!(destinations, ["/p/lib/a", "/p/lib/d/b"]); + assert!(!order_matters(©_descriptors)); + for (source_folder, destination_folder) in [("/p/lib", "/p/lib/copy"), ("/p/lib/copy", "/p/lib"), ("/p", "/p"), ("/", "/q")] { + let overlapping = [operation(source_folder, &["**/*"], &[destination_folder], false)]; + let overlapping_sources = [files(&["/p/lib/copy/x"])]; + assert!(order_matters(&collect_copy_descriptors(&overlapping, &overlapping_sources).unwrap())); + } + let siblings = [operation("/p/lib", &["**/*"], &["/p/lib-copy"], false)]; + assert!(!order_matters(&collect_copy_descriptors(&siblings, &[files(&["/p/lib/x"])]).unwrap())); +} + +#[test] +fn duplicate_destinations_are_removed_or_rejected_like_heft() { + let operations = [operation("/p/src", &["**/*"], &["/p/lib", "/p/lib"], false)]; + let sources = [files(&["/p/src/a"])]; + let copy_descriptors = collect_copy_descriptors(&operations, &sources).unwrap(); + assert!(!copy_descriptors.sources_are_distinct_and_in_order); + assert_eq!(copy_descriptors.descriptors.len(), 1); + let flattened = [operation("/p/src", &["**/*"], &["/p/lib"], true)]; + let flattened_sources = [files(&["/p/src/a/x", "/p/src/b/x"])]; + let error = collect_copy_descriptors(&flattened, &flattened_sources).err().unwrap(); + assert_eq!(error.message, "Cannot copy multiple files to the same destination \"/p/lib/x\"."); + let literal = [operation("/p/src", &["a.txt", "./a.txt"], &["/p/lib"], false)]; + let literal_sources = [files(&["/p/src/a.txt", "/p/src/a.txt"])]; + assert_eq!(collect_copy_descriptors(&literal, &literal_sources).unwrap().descriptors.len(), 1); +} + +#[test] +fn copies_reading_a_destination_or_nested_under_one_depend_on_their_order() { + let chained = [operation("/p/src", &["**/*"], &["/p/lib"], false), operation("/p/lib", &["**/*"], &["/p/out"], false)]; + assert!(order_matters(&collect_copy_descriptors(&chained, &[files(&["/p/src/a"]), files(&["/p/lib/a"])]).unwrap())); + let nested = [operation("/p/src", &["a", "a/b"], &["/p/lib"], false)]; + assert!(order_matters(&collect_copy_descriptors(&nested, &[files(&["/p/src/a", "/p/src/a/b"])]).unwrap())); + let independent = [operation("/p/src", &["x.txt"], &["/p/lib"], false), operation("/p/src2", &["**/*"], &["/p/out"], false)]; + assert!(!order_matters(&collect_copy_descriptors(&independent, &[files(&["/p/src/x.txt"]), files(&["/p/src2/y"])]).unwrap())); +} diff --git a/apps/heft-native/src/builtin/copy_files.rs b/apps/heft-native/src/builtin/copy_files.rs index 9f80b4464b..d975882659 100644 --- a/apps/heft-native/src/builtin/copy_files.rs +++ b/apps/heft-native/src/builtin/copy_files.rs @@ -1,19 +1,24 @@ use std::collections::hash_map::Entry; -use std::collections::HashMap; use std::fs::File; use std::io::Read; use super::base64::sha256_digest_as_base64; use super::build_info::{try_read_build_info, write_build_info, BuildInfoReadResult}; +use super::copy_descriptors::{collect_copy_descriptors, CopyDescriptor, CopyDescriptors}; use super::copy_operation::AbsoluteCopyOperation; use super::delete_files::glob_changed_during_run; use super::file_operations::{copy_file_overwriting, hard_link_overwriting}; use super::node_file_system_error::NodeFileSystemError; -use super::posix_path::{base_name, relative_path}; +use super::parallel_items::{process_items_in_order, SEQUENTIAL_ONLY}; +use super::path_hash::{path_hash_map_with_capacity, PathHashMap}; use super::sha256::Sha256; use super::simple_glob::GlobbedEntry; use crate::terminal::ScopedLoggerOutput; +const HASHED_FILES_PER_WORKER: usize = 64; +const COPIED_FILES_PER_WORKER: usize = 64; +const NOT_PREVIOUSLY_HASHED: usize = usize::MAX; + pub struct CopyFilesTaskPlan { pub operations: Vec, pub configuration_hash: String, @@ -23,68 +28,70 @@ pub struct CopyFilesTaskPlan { pub struct CopyFilesPreflight { pub source_files: Vec>, - pub build_info: BuildInfoReadResult, + pub build_info: Option, } -struct CopyDescriptor { - operation_index: usize, - source_index: usize, - destination_path: String, +struct HashedSources<'plan> { + source_paths: Vec<&'plan str>, + versions: Vec<[u8; 44]>, + version_index_of_descriptor: Option>, } -pub fn preflight_copy_files(plan: &CopyFilesTaskPlan) -> Option { - let source_files = plan.operations.iter().map(|operation| operation.selection.select(false)).collect::>>()?; - let build_info = try_read_build_info(&plan.build_info_path); - (!matches!(build_info, BuildInfoReadResult::NeedsJavaScript)).then_some(CopyFilesPreflight { source_files, build_info }) +impl HashedSources<'_> { + fn version_index_of(&self, descriptor_index: usize) -> usize { + self.version_index_of_descriptor.as_ref().map_or(descriptor_index, |indices| indices[descriptor_index]) + } + + fn version_text(&self, version_index: usize) -> &str { + std::str::from_utf8(&self.versions[version_index]).unwrap_or_default() + } } pub fn run_copy_files_task(plan: CopyFilesTaskPlan, output: &ScopedLoggerOutput<'_>) -> Result<(), NodeFileSystemError> { - let CopyFilesPreflight { source_files, build_info } = match plan.preflight { - Some(preflight) => preflight, - None => preflight_copy_files(&plan).ok_or_else(glob_changed_during_run)?, + let (source_files, build_info) = match plan.preflight { + Some(CopyFilesPreflight { source_files, build_info }) => (source_files, build_info), + None => { + let source_files = plan.operations.iter().map(|operation| operation.selection.select(false)).collect::>>(); + (source_files.ok_or_else(glob_changed_during_run)?, None) + } + }; + let build_info = match build_info.unwrap_or_else(|| try_read_build_info(&plan.build_info_path)) { + BuildInfoReadResult::NeedsJavaScript => return Err(glob_changed_during_run()), + build_info => build_info, }; let copy_descriptors = collect_copy_descriptors(&plan.operations, &source_files)?; - if copy_descriptors.is_empty() { + if copy_descriptors.descriptors.is_empty() { return Ok(()); } - let source_path_of = |descriptor: &CopyDescriptor| source_files[descriptor.operation_index][descriptor.source_index].absolute_path.as_str(); let old_entries: Vec<(String, String)> = match build_info { BuildInfoReadResult::Found(old) if old.configuration_hash == plan.configuration_hash => old.input_file_versions, _ => Vec::new(), }; - let old_versions: HashMap<&str, &str> = old_entries.iter().map(|(path, version)| (path.as_str(), version.as_str())).collect(); - let mut new_versions: HashMap<&str, [u8; 44]> = HashMap::with_capacity(copy_descriptors.len()); - let mut added_input_files: Vec<&str> = Vec::new(); - for descriptor in ©_descriptors { - let source_path = source_path_of(descriptor); - if let Entry::Vacant(vacant) = new_versions.entry(source_path) { - vacant.insert(hash_file_contents(source_path)?); - if !old_versions.contains_key(source_path) { - added_input_files.push(source_path); - } - } - } - let version_text = |path: &str| new_versions.get(path).and_then(|version| std::str::from_utf8(version).ok()); - let mut copied_file_count = 0; - let mut linked_file_count = 0; - let mut last_existing_folder: Option = None; - for descriptor in ©_descriptors { - let source_path = source_path_of(descriptor); - if old_versions.get(source_path).copied() == version_text(source_path) { - continue; - } - if plan.operations[descriptor.operation_index].hardlink { - linked_file_count += 1; - hard_link_overwriting(source_path, &descriptor.destination_path)?; - } else { - copied_file_count += 1; - copy_file_overwriting(source_path, &descriptor.destination_path, &mut last_existing_folder)?; - } - } - if copied_file_count == 0 && linked_file_count == 0 { + let hashed_sources = hash_sources(©_descriptors)?; + let old_entry_index_of_source: Vec = { + let old_entry_index_by_path: PathHashMap<&str, usize> = + old_entries.iter().enumerate().map(|(index, (path, _))| (path.as_str(), index)).collect(); + let old_entry_index_of = |path: &&str| old_entry_index_by_path.get(path).copied().unwrap_or(NOT_PREVIOUSLY_HASHED); + hashed_sources.source_paths.iter().map(old_entry_index_of).collect() + }; + let source_is_up_to_date = |version_index: usize| { + let old_entry_index = old_entry_index_of_source[version_index]; + old_entry_index != NOT_PREVIOUSLY_HASHED && old_entries[old_entry_index].1 == hashed_sources.version_text(version_index) + }; + let descriptors_with_work: Vec<&CopyDescriptor> = copy_descriptors + .descriptors + .iter() + .enumerate() + .filter(|(descriptor_index, _)| !source_is_up_to_date(hashed_sources.version_index_of(*descriptor_index))) + .map(|(_, descriptor)| descriptor) + .collect(); + if descriptors_with_work.is_empty() { output.write_line("All requested file copy operations are up to date. Nothing to do."); return Ok(()); } + let linked_file_count = descriptors_with_work.iter().filter(|descriptor| copy_descriptors.is_hard_link(descriptor)).count(); + let copied_file_count = descriptors_with_work.len() - linked_file_count; + copy_or_link_files(©_descriptors, &descriptors_with_work)?; output.write_line(&format!( "Copied {copied_file_count} file{} and linked {linked_file_count} file{}", if copied_file_count == 1 { "" } else { "s" }, @@ -93,71 +100,70 @@ pub fn run_copy_files_task(plan: CopyFilesTaskPlan, output: &ScopedLoggerOutput< if output.output_is_closed() { return Ok(()); } - let input_file_versions = old_entries - .iter() - .map(move |(path, version)| (path.as_str(), version_text(path).unwrap_or(version.as_str()))) - .chain(added_input_files.iter().map(move |path| (*path, version_text(path).unwrap_or_default()))); - write_build_info(&plan.configuration_hash, input_file_versions, &plan.build_info_path) -} - -fn path_relative_to_source_folder<'path>(source_folder_path: &str, file_path: &'path str) -> std::borrow::Cow<'path, str> { - let prefix_length = if source_folder_path.ends_with('/') { source_folder_path.len() } else { source_folder_path.len() + 1 }; - match file_path.get(prefix_length..) { - Some(relative) if file_path.starts_with(source_folder_path) && file_path.as_bytes()[prefix_length - 1] == b'/' => { - std::borrow::Cow::Borrowed(relative) + let mut new_version_index_of_old_entry = vec![NOT_PREVIOUSLY_HASHED; old_entries.len()]; + for (version_index, old_entry_index) in old_entry_index_of_source.iter().enumerate() { + if *old_entry_index != NOT_PREVIOUSLY_HASHED { + new_version_index_of_old_entry[*old_entry_index] = version_index; } - _ => std::borrow::Cow::Owned(relative_path(source_folder_path, file_path)), } + let updated_old_entries = old_entries.iter().zip(&new_version_index_of_old_entry).map(|((path, old_version), version_index)| { + let version = if *version_index == NOT_PREVIOUSLY_HASHED { old_version } else { hashed_sources.version_text(*version_index) }; + (path.as_str(), version) + }); + let added_entries = hashed_sources + .source_paths + .iter() + .enumerate() + .filter(|(version_index, _)| old_entry_index_of_source[*version_index] == NOT_PREVIOUSLY_HASHED) + .map(|(version_index, path)| (*path, hashed_sources.version_text(version_index))); + write_build_info(&plan.configuration_hash, updated_old_entries.chain(added_entries), &plan.build_info_path) } -fn collect_copy_descriptors( - operations: &[AbsoluteCopyOperation], - source_files: &[Vec], -) -> Result, NodeFileSystemError> { - let mut candidates: Vec = Vec::new(); - for (operation_index, operation) in operations.iter().enumerate() { - for destination_folder_path in &operation.destination_folder_paths { - for (source_index, source_file) in source_files[operation_index].iter().enumerate() { - let destination_relative_path = if operation.flatten { - std::borrow::Cow::Borrowed(base_name(&source_file.absolute_path)) - } else { - path_relative_to_source_folder(&operation.selection.source_folder_path, &source_file.absolute_path) - }; - let mut destination_path = String::with_capacity(destination_folder_path.len() + destination_relative_path.len() + 1); - destination_path.push_str(destination_folder_path); - if !destination_folder_path.ends_with('/') { - destination_path.push('/'); +fn hash_sources<'plan>(copy_descriptors: &CopyDescriptors<'plan>) -> Result, NodeFileSystemError> { + let (source_paths, version_index_of_descriptor) = if copy_descriptors.sources_are_distinct_and_in_order { + (copy_descriptors.descriptors.iter().map(|descriptor| copy_descriptors.source_path_of(descriptor)).collect(), None) + } else { + let mut source_paths: Vec<&str> = Vec::new(); + let mut version_index_by_source_path = path_hash_map_with_capacity::<&str, usize>(copy_descriptors.descriptors.len()); + let mut version_index_of_descriptor = Vec::with_capacity(copy_descriptors.descriptors.len()); + for descriptor in ©_descriptors.descriptors { + let version_index = match version_index_by_source_path.entry(copy_descriptors.source_path_of(descriptor)) { + Entry::Occupied(occupied) => *occupied.get(), + Entry::Vacant(vacant) => { + source_paths.push(vacant.key()); + *vacant.insert(source_paths.len() - 1) } - destination_path.push_str(&destination_relative_path); - candidates.push(CopyDescriptor { operation_index, source_index, destination_path }); - } + }; + version_index_of_descriptor.push(version_index); } - } - let mut keep = vec![true; candidates.len()]; - let mut first_index_by_destination: HashMap<&str, usize> = HashMap::with_capacity(candidates.len()); - for (index, candidate) in candidates.iter().enumerate() { - match first_index_by_destination.entry(candidate.destination_path.as_str()) { - Entry::Occupied(occupied) => { - let existing = &candidates[*occupied.get()]; - let same_source = source_files[existing.operation_index][existing.source_index].absolute_path - == source_files[candidate.operation_index][candidate.source_index].absolute_path; - if !same_source || operations[existing.operation_index].hardlink != operations[candidate.operation_index].hardlink { - return Err(NodeFileSystemError::from_message(&format!( - "Cannot copy multiple files to the same destination \"{}\".", - candidate.destination_path - ))); - } - keep[index] = false; - } - Entry::Vacant(vacant) => { - vacant.insert(index); - } + (source_paths, Some(version_index_of_descriptor)) + }; + let mut versions = vec![[0u8; 44]; source_paths.len()]; + process_items_in_order(&source_paths, &mut versions, HASHED_FILES_PER_WORKER, &|source_path, version| { + *version = hash_file_contents(source_path)?; + Ok(()) + })?; + Ok(HashedSources { source_paths, versions, version_index_of_descriptor }) +} + +fn copy_or_link_files(copy_descriptors: &CopyDescriptors<'_>, descriptors_with_work: &[&CopyDescriptor]) -> Result<(), NodeFileSystemError> { + let copies_per_worker = if descriptors_with_work.len() < 2 * COPIED_FILES_PER_WORKER + || copy_descriptors.copies_may_depend_on_their_order(descriptors_with_work) + { + SEQUENTIAL_ONLY + } else { + COPIED_FILES_PER_WORKER + }; + let mut copy_results = vec![(); descriptors_with_work.len()]; + process_items_in_order(descriptors_with_work, &mut copy_results, copies_per_worker, &|descriptor, ()| { + let source_path = copy_descriptors.source_path_of(descriptor); + let destination_path = copy_descriptors.destination_path_of(descriptor); + if copy_descriptors.is_hard_link(descriptor) { + hard_link_overwriting(source_path, &destination_path) + } else { + copy_file_overwriting(source_path, &destination_path) } - } - drop(first_index_by_destination); - let mut keep_flags = keep.into_iter(); - candidates.retain(|_| keep_flags.next().unwrap_or(false)); - Ok(candidates) + }) } fn hash_file_contents(file_path: &str) -> Result<[u8; 44], NodeFileSystemError> { @@ -166,10 +172,10 @@ fn hash_file_contents(file_path: &str) -> Result<[u8; 44], NodeFileSystemError> let mut buffer = [0u8; 16384]; loop { let read_length = file.read(&mut buffer).map_err(|error| NodeFileSystemError::new(error, "read", file_path, None))?; - if read_length == 0 { + hasher.update(&buffer[..read_length]); + if read_length < buffer.len() { break; } - hasher.update(&buffer[..read_length]); } Ok(sha256_digest_as_base64(&hasher.finalize())) } diff --git a/apps/heft-native/src/builtin/delete_files.rs b/apps/heft-native/src/builtin/delete_files.rs index f9786bba4c..0feeb61753 100644 --- a/apps/heft-native/src/builtin/delete_files.rs +++ b/apps/heft-native/src/builtin/delete_files.rs @@ -1,29 +1,44 @@ -use std::collections::HashSet; - -use super::file_operations::{delete_file_if_it_exists, delete_folder_recursively}; +use super::file_operations::{delete_file_if_it_exists, delete_folder_found_by_preflight, delete_folder_recursively}; use super::file_selection::AbsoluteFileSelection; use super::node_file_system_error::NodeFileSystemError; +use super::path_hash::path_hash_set_with_capacity; +use super::simple_glob::GlobbedEntry; use crate::terminal::ScopedLoggerOutput; -pub fn run_delete_operations( - selections: &[AbsoluteFileSelection], - output: &ScopedLoggerOutput<'_>, -) -> Result<(), NodeFileSystemError> { +pub struct DeletionPlan { + pub selections: Vec, + pub preflight_entries: Option>>, +} + +impl DeletionPlan { + pub fn does_file_system_work(&self) -> bool { + self.preflight_entries.as_ref().is_none_or(|entries_per_selection| entries_per_selection.iter().any(|entries| !entries.is_empty())) + } +} + +pub fn run_deletion_plan(plan: DeletionPlan, output: &ScopedLoggerOutput<'_>) -> Result<(), NodeFileSystemError> { + let entries_were_found_by_preflight = plan.preflight_entries.is_some(); + let entries_per_selection = match plan.preflight_entries { + Some(entries_per_selection) => entries_per_selection, + None => plan + .selections + .iter() + .map(|selection| selection.select(true)) + .collect::>>() + .ok_or_else(glob_changed_during_run)?, + }; let mut files_to_delete: Vec = Vec::new(); let mut folders_to_delete: Vec = Vec::new(); - let mut seen_files: HashSet = HashSet::new(); - let mut seen_folders: HashSet = HashSet::new(); - for selection in selections { - let entries = selection.select(true).ok_or_else(glob_changed_during_run)?; - for entry in entries { - let (paths, seen) = if entry.is_directory { - (&mut folders_to_delete, &mut seen_folders) - } else { - (&mut files_to_delete, &mut seen_files) - }; - if seen.insert(entry.absolute_path.clone()) { - paths.push(entry.absolute_path); - } + let mut seen_files = path_hash_set_with_capacity::(0); + let mut seen_folders = path_hash_set_with_capacity::(0); + for entry in entries_per_selection.into_iter().flatten() { + let (paths, seen) = if entry.is_directory { + (&mut folders_to_delete, &mut seen_folders) + } else { + (&mut files_to_delete, &mut seen_files) + }; + if seen.insert(entry.absolute_path.clone()) { + paths.push(entry.absolute_path); } } let mut deleted_file_count = 0; @@ -34,7 +49,12 @@ pub fn run_delete_operations( } let mut deleted_folder_count = 0; for folder_path in folders_to_delete.iter().rev() { - if delete_folder_recursively(folder_path)? { + let was_deleted = if entries_were_found_by_preflight { + delete_folder_found_by_preflight(folder_path)? + } else { + delete_folder_recursively(folder_path)? + }; + if was_deleted { deleted_folder_count += 1; } } diff --git a/apps/heft-native/src/builtin/deletion_permissions.rs b/apps/heft-native/src/builtin/deletion_permissions.rs index 6948945418..442b79d079 100644 --- a/apps/heft-native/src/builtin/deletion_permissions.rs +++ b/apps/heft-native/src/builtin/deletion_permissions.rs @@ -1,56 +1,85 @@ use super::file_selection::AbsoluteFileSelection; +use super::simple_glob::GlobbedEntry; #[cfg(unix)] -pub fn selections_are_deletable_without_permission_errors(selections: &[AbsoluteFileSelection]) -> bool { - use std::fs; - use std::os::unix::fs::MetadataExt; - +pub fn preflight_deletions(selections: &[AbsoluteFileSelection]) -> Option>> { use super::posix_path::directory_name; - let user_id = crate::sys::effective_user_id(); - let folder_is_modifiable = |path: &str| { - user_id == 0 - || fs::symlink_metadata(path).is_ok_and(|metadata| { - metadata.is_dir() && metadata.uid() == user_id && metadata.mode() & 0o1300 == 0o300 - }) - }; + let mut checked_folders = CheckedFolders::new(crate::sys::effective_user_id()); + let mut entries_per_selection = Vec::with_capacity(selections.len()); for selection in selections { - let Some(entries) = selection.select(true) else { - return false; - }; - for entry in entries { - if !folder_is_modifiable(directory_name(&entry.absolute_path)) { - return false; + let entries = selection.select(true)?; + for entry in &entries { + if !checked_folders.folder_is_modifiable(directory_name(&entry.absolute_path)) { + return None; + } + if entry.is_directory && !checked_folders.every_folder_in_tree_is_modifiable(entry.absolute_path.clone()) { + return None; } - if !entry.is_directory { + } + entries_per_selection.push(entries); + } + Some(entries_per_selection) +} + +#[cfg(unix)] +struct CheckedFolders { + user_id: u32, + modifiable_folder_paths: super::path_hash::PathHashSet, + walked_folder_paths: super::path_hash::PathHashSet, +} + +#[cfg(unix)] +impl CheckedFolders { + fn new(user_id: u32) -> CheckedFolders { + CheckedFolders { user_id, modifiable_folder_paths: Default::default(), walked_folder_paths: Default::default() } + } + + fn folder_is_modifiable(&mut self, folder_path: &str) -> bool { + use std::os::unix::fs::MetadataExt; + if self.modifiable_folder_paths.contains(folder_path) { + return true; + } + let is_modifiable = self.user_id == 0 + || std::fs::symlink_metadata(folder_path).is_ok_and(|metadata| { + metadata.is_dir() && metadata.uid() == self.user_id && metadata.mode() & 0o1300 == 0o300 + }); + if is_modifiable { + self.modifiable_folder_paths.insert(folder_path.to_owned()); + } + is_modifiable + } + + fn every_folder_in_tree_is_modifiable(&mut self, root_folder_path: String) -> bool { + let mut folders_to_check = vec![root_folder_path]; + while let Some(folder) = folders_to_check.pop() { + if self.walked_folder_paths.contains(&folder) { continue; } - let mut folders_to_check = vec![entry.absolute_path]; - while let Some(folder) = folders_to_check.pop() { - if !folder_is_modifiable(&folder) { - return false; - } - let Ok(reader) = fs::read_dir(&folder) else { + if !self.folder_is_modifiable(&folder) { + return false; + } + let Ok(reader) = std::fs::read_dir(&folder) else { + return false; + }; + for child in reader { + let Ok(child) = child else { return false; }; - for child in reader { - let Ok(child) = child else { + if child.file_type().is_ok_and(|file_type| file_type.is_dir()) { + let Ok(child_path) = child.path().into_os_string().into_string() else { return false; }; - if child.file_type().is_ok_and(|file_type| file_type.is_dir()) { - let Ok(child_path) = child.path().into_os_string().into_string() else { - return false; - }; - folders_to_check.push(child_path); - } + folders_to_check.push(child_path); } } + self.walked_folder_paths.insert(folder); } + true } - true } #[cfg(not(unix))] -pub fn selections_are_deletable_without_permission_errors(_selections: &[AbsoluteFileSelection]) -> bool { - false +pub fn preflight_deletions(_selections: &[AbsoluteFileSelection]) -> Option>> { + None } diff --git a/apps/heft-native/src/builtin/file_operations.rs b/apps/heft-native/src/builtin/file_operations.rs index a8842ab9e3..8fc74fbd81 100644 --- a/apps/heft-native/src/builtin/file_operations.rs +++ b/apps/heft-native/src/builtin/file_operations.rs @@ -3,16 +3,19 @@ use std::io::ErrorKind; use std::path::Path; use super::node_file_system_error::{is_node_not_exist_error, NodeFileSystemError}; -use super::node_rimraf::remove_like_node_rimraf; +use super::node_rimraf::{remove_entry_like_node_rimraf, RimrafEntryKind}; use super::posix_path::directory_name; const EEXIST: i32 = 17; -pub fn copy_file_overwriting( - source_path: &str, - destination_path: &str, - last_existing_folder: &mut Option, -) -> Result<(), NodeFileSystemError> { +pub fn copy_file_overwriting(source_path: &str, destination_path: &str) -> Result<(), NodeFileSystemError> { + match super::open_file_copy::copy_through_open_source_file(source_path, destination_path) { + Some(result) => result, + None => copy_file_overwriting_step_by_step(source_path, destination_path), + } +} + +fn copy_file_overwriting_step_by_step(source_path: &str, destination_path: &str) -> Result<(), NodeFileSystemError> { let source_metadata = fs::symlink_metadata(source_path).map_err(|error| NodeFileSystemError::new(error, "lstat", source_path, None))?; let destination_metadata = match fs::symlink_metadata(destination_path) { @@ -20,30 +23,36 @@ pub fn copy_file_overwriting( Err(error) if error.kind() == ErrorKind::NotFound => None, Err(error) => return Err(NodeFileSystemError::new(error, "lstat", destination_path, None)), }; - if let Some(destination_metadata) = &destination_metadata { - if are_the_same_file(&source_metadata, destination_metadata) { - return Err(NodeFileSystemError::from_message("Source and destination must not be the same.")); + match &destination_metadata { + Some(destination_metadata) => { + check_destination_can_be_replaced(source_path, &source_metadata, destination_path, destination_metadata)?; + fs::remove_file(destination_path) + .map_err(|error| NodeFileSystemError::new(error, "unlink", destination_path, None))?; } - if !source_metadata.is_dir() && destination_metadata.is_dir() { - return Err(NodeFileSystemError::from_message(&format!( - "Cannot overwrite directory '{destination_path}' with non-directory '{source_path}'." - ))); - } - } - let destination_folder = directory_name(destination_path); - if destination_metadata.is_none() && last_existing_folder.as_deref() != Some(destination_folder) { - ensure_folder_exists(destination_folder)?; - *last_existing_folder = Some(destination_folder.to_owned()); - } - if destination_metadata.is_some() { - fs::remove_file(destination_path) - .map_err(|error| NodeFileSystemError::new(error, "unlink", destination_path, None))?; + None => ensure_folder_exists(directory_name(destination_path))?, } fs::copy(source_path, destination_path) .map(|_| ()) .map_err(|error| NodeFileSystemError::new(error, "copyfile", source_path, Some(destination_path))) } +pub fn check_destination_can_be_replaced( + source_path: &str, + source_metadata: &fs::Metadata, + destination_path: &str, + destination_metadata: &fs::Metadata, +) -> Result<(), NodeFileSystemError> { + if are_the_same_file(source_metadata, destination_metadata) { + return Err(NodeFileSystemError::from_message("Source and destination must not be the same.")); + } + if !source_metadata.is_dir() && destination_metadata.is_dir() { + return Err(NodeFileSystemError::from_message(&format!( + "Cannot overwrite directory '{destination_path}' with non-directory '{source_path}'." + ))); + } + Ok(()) +} + pub fn hard_link_overwriting(link_target_path: &str, new_link_path: &str) -> Result<(), NodeFileSystemError> { let link = || { fs::hard_link(link_target_path, new_link_path) @@ -72,13 +81,21 @@ pub fn delete_file_if_it_exists(file_path: &str) -> Result Result { - match fs::symlink_metadata(folder_path) { + let is_directory = match fs::symlink_metadata(folder_path) { Err(error) if error.kind() == ErrorKind::NotFound => return Ok(true), Err(error) if is_node_not_exist_error(&error) => return Ok(false), Err(error) => return Err(NodeFileSystemError::new(error, "lstat", folder_path, None)), - Ok(_) => {} - } - match remove_like_node_rimraf(Path::new(folder_path)) { + Ok(metadata) => metadata.is_dir(), + }; + remove_folder_like_file_system_extra(folder_path, if is_directory { RimrafEntryKind::Folder } else { RimrafEntryKind::Other }) +} + +pub fn delete_folder_found_by_preflight(folder_path: &str) -> Result { + remove_folder_like_file_system_extra(folder_path, RimrafEntryKind::Folder) +} + +fn remove_folder_like_file_system_extra(folder_path: &str, entry_kind: RimrafEntryKind) -> Result { + match remove_entry_like_node_rimraf(Path::new(folder_path), entry_kind, true) { Ok(()) => Ok(true), Err(failure) if is_node_not_exist_error(&failure.error) => Ok(false), Err(failure) => Err(NodeFileSystemError::new(failure.error, failure.syscall, &failure.path, None)), @@ -99,7 +116,7 @@ fn are_the_same_file(_source_metadata: &fs::Metadata, _destination_metadata: &fs false } -fn ensure_folder_exists(folder_path: &str) -> Result<(), NodeFileSystemError> { +pub fn ensure_folder_exists(folder_path: &str) -> Result<(), NodeFileSystemError> { if Path::new(folder_path).exists() { return Ok(()); } diff --git a/apps/heft-native/src/builtin/file_selection.rs b/apps/heft-native/src/builtin/file_selection.rs index bffee4753e..284742c132 100644 --- a/apps/heft-native/src/builtin/file_selection.rs +++ b/apps/heft-native/src/builtin/file_selection.rs @@ -1,5 +1,7 @@ use super::posix_path::resolve_path; -use super::simple_glob::{patterns_are_simple, try_simple_glob, GlobbedEntry}; +use super::simple_glob::{ + patterns_are_simple, patterns_only_read_inside, patterns_select_each_path_once, try_simple_glob, GlobbedEntry, +}; use super::simple_glob_pattern::is_extension; #[derive(Clone, Debug, Default)] @@ -68,6 +70,14 @@ impl AbsoluteFileSelection { pub fn select(&self, include_folders: bool) -> Option> { try_simple_glob(&self.include_globs, &self.source_folder_path, !include_folders) } + + pub fn selects_each_path_once_inside_its_folder(&self) -> bool { + patterns_select_each_path_once(&self.include_globs) + } + + pub fn only_reads_inside_its_folder(&self) -> bool { + patterns_only_read_inside(&self.include_globs, &self.source_folder_path) + } } #[cfg(test)] diff --git a/apps/heft-native/src/builtin/mod.rs b/apps/heft-native/src/builtin/mod.rs index 50adf41e77..154c0dcdce 100644 --- a/apps/heft-native/src/builtin/mod.rs +++ b/apps/heft-native/src/builtin/mod.rs @@ -4,6 +4,9 @@ mod build_info_json; #[cfg(test)] mod build_info_json_tests; mod builtin_task; +mod copy_descriptors; +#[cfg(test)] +mod copy_descriptors_tests; mod copy_files; mod copy_operation; mod delete_files; @@ -11,8 +14,14 @@ mod deletion_permissions; mod file_operations; mod file_selection; mod javascript_json; +mod modified_paths; mod node_file_system_error; mod node_rimraf; +mod open_file_copy; +mod parallel_items; +mod path_hash; +#[cfg(test)] +mod parallel_items_tests; mod posix_path; mod set_environment_variables; mod sha256; @@ -20,12 +29,16 @@ mod sha256; mod sha256_tests; mod simple_glob; mod simple_glob_pattern; +mod worker_transfer; pub use builtin_task::{ - builtin_task_passes_preflight, builtin_task_touches_files, plan_builtin_task, plan_phase_clean, - run_planned_builtin_task, BuiltinTaskOptions, PlannedBuiltinTask, + builtin_task_does_file_system_work, builtin_task_passes_preflight, plan_builtin_task, plan_phase_clean, + plan_phase_clean_deletion, run_planned_builtin_task, BuiltinTaskOptions, PlannedBuiltinTask, }; pub use copy_operation::{CopyOperation, CopyOperationField}; -pub use deletion_permissions::selections_are_deletable_without_permission_errors; -pub use delete_files::run_delete_operations; +pub use delete_files::{run_deletion_plan, DeletionPlan}; +pub use deletion_permissions::preflight_deletions; pub use file_selection::{AbsoluteFileSelection, FileSelectionSpecifier}; +pub use modified_paths::ModifiedPaths; +#[cfg(all(test, target_arch = "x86_64"))] +pub use sha256::compress_blocks_without_simd as compress_sha256_blocks_without_simd; diff --git a/apps/heft-native/src/builtin/modified_paths.rs b/apps/heft-native/src/builtin/modified_paths.rs new file mode 100644 index 0000000000..7b553a2e77 --- /dev/null +++ b/apps/heft-native/src/builtin/modified_paths.rs @@ -0,0 +1,72 @@ +use super::delete_files::DeletionPlan; +use super::file_selection::AbsoluteFileSelection; +use super::posix_path::path_contains; + +const MAXIMUM_EXACT_DELETED_PATHS_PER_STEP: usize = 32; + +#[derive(Default)] +pub struct ModifiedPaths { + paths_in_step_order: Vec<(String, bool)>, +} + +impl ModifiedPaths { + pub fn overlaps(&self, path: &str) -> bool { + self.paths_in_step_order + .iter() + .any(|(modified_path, _)| path_contains(modified_path, path) || path_contains(path, modified_path)) + } + + pub fn selection_is_unchanged(&self, selection: &AbsoluteFileSelection) -> bool { + selection.only_reads_inside_its_folder() && !self.overlaps(&selection.source_folder_path) + } + + pub fn deletes(&self, path: &str) -> bool { + let last_overlapping_path = self + .paths_in_step_order + .iter() + .rev() + .find(|(modified_path, _)| path_contains(modified_path, path) || path_contains(path, modified_path)); + matches!(last_overlapping_path, Some((deleted_path, true)) if path_contains(deleted_path, path)) + } + + pub fn add_modified_path(&mut self, path: &str) { + self.paths_in_step_order.push((path.to_owned(), false)); + } + + pub fn add_deletion_plan(&mut self, plan: &DeletionPlan) { + match &plan.preflight_entries { + Some(entries_per_selection) + if entries_per_selection.iter().map(Vec::len).sum::() <= MAXIMUM_EXACT_DELETED_PATHS_PER_STEP => + { + for entry in entries_per_selection.iter().flatten() { + self.paths_in_step_order.push((entry.absolute_path.clone(), true)); + } + } + _ => { + for selection in &plan.selections { + self.add_modified_path(&selection.source_folder_path); + } + } + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn deletions_and_modifications_are_tracked_in_step_order() { + let mut modified_paths = ModifiedPaths::default(); + assert!(!modified_paths.overlaps("/p/src") && !modified_paths.deletes("/p/temp/build/x")); + modified_paths.paths_in_step_order.push(("/p/temp/build".to_owned(), true)); + modified_paths.add_modified_path("/p/lib/assets"); + assert!(modified_paths.deletes("/p/temp/build/copy/file-copy.json")); + assert!(!modified_paths.deletes("/p/temp") && modified_paths.overlaps("/p/temp")); + assert!(!modified_paths.overlaps("/p/src/assets") && !modified_paths.overlaps("/p/temp/scratch")); + assert!(modified_paths.overlaps("/p/lib") && modified_paths.overlaps("/p/lib/assets/a")); + modified_paths.add_modified_path("/p/temp/build/copy/file-copy.json"); + assert!(!modified_paths.deletes("/p/temp/build/copy/file-copy.json")); + assert!(modified_paths.deletes("/p/temp/build/other")); + } +} diff --git a/apps/heft-native/src/builtin/node_file_system_error.rs b/apps/heft-native/src/builtin/node_file_system_error.rs index 8122223e33..d664ecb50f 100644 --- a/apps/heft-native/src/builtin/node_file_system_error.rs +++ b/apps/heft-native/src/builtin/node_file_system_error.rs @@ -1,5 +1,7 @@ use std::io; +use super::worker_transfer::{append_text, take_array, take_text}; + const LIBUV_ERROR_NAMES_AND_DESCRIPTIONS: &[(i32, &str, &str)] = &[ (1, "EPERM", "operation not permitted"), (2, "ENOENT", "no such file or directory"), @@ -72,6 +74,21 @@ impl NodeFileSystemError { } } +impl super::parallel_items::WorkerTransfer for NodeFileSystemError { + fn append_to(&self, bytes: &mut Vec) { + append_text(&self.message, bytes); + bytes.push(u8::from(self.node_core_library_prefix.is_some())); + append_text(self.node_core_library_prefix.as_deref().unwrap_or_default(), bytes); + } + + fn take_from(bytes: &mut &[u8]) -> Option { + let message = take_text(bytes)?; + let [has_prefix] = take_array::<1>(bytes)?; + let prefix = take_text(bytes)?; + Some(NodeFileSystemError { message, node_core_library_prefix: (has_prefix != 0).then_some(prefix) }) + } +} + pub fn is_node_not_exist_error(error: &io::Error) -> bool { matches!(error.raw_os_error(), Some(ENOENT | ENOTDIR)) } diff --git a/apps/heft-native/src/builtin/node_rimraf.rs b/apps/heft-native/src/builtin/node_rimraf.rs index a7e084a3a0..90b63b97c1 100644 --- a/apps/heft-native/src/builtin/node_rimraf.rs +++ b/apps/heft-native/src/builtin/node_rimraf.rs @@ -2,11 +2,15 @@ use std::fs; use std::io::{self, ErrorKind}; use std::path::{Path, PathBuf}; +use super::parallel_items::{attempt_every_item_in_order, WorkerTransfer, SEQUENTIAL_ONLY}; +use super::worker_transfer::{append_text, take_array, take_text}; + const EPERM: i32 = 1; const EEXIST: i32 = 17; const ENOTDIR: i32 = 20; const EISDIR: i32 = 21; const ENOTEMPTY: i32 = 39; +const REMOVED_CHILDREN_PER_WORKER: usize = 16; pub struct RimrafFailure { pub error: io::Error, @@ -14,31 +18,69 @@ pub struct RimrafFailure { pub path: String, } +#[derive(Clone, Copy)] +pub enum RimrafEntryKind { + Folder, + Missing, + Other, +} + +pub fn rimraf_entry_kind(is_directory: io::Result) -> RimrafEntryKind { + match is_directory { + Ok(true) => RimrafEntryKind::Folder, + Err(error) if error.kind() == ErrorKind::NotFound => RimrafEntryKind::Missing, + _ => RimrafEntryKind::Other, + } +} + +const RIMRAF_SYSCALL_NAMES: [&str; 3] = ["unlink", "rmdir", "scandir"]; + +impl WorkerTransfer for RimrafFailure { + fn append_to(&self, bytes: &mut Vec) { + bytes.extend_from_slice(&self.error.raw_os_error().unwrap_or_default().to_le_bytes()); + let syscall_index = RIMRAF_SYSCALL_NAMES.iter().position(|name| *name == self.syscall).unwrap_or(usize::MAX); + bytes.extend_from_slice(&(syscall_index as u64).to_le_bytes()); + append_text(&self.path, bytes); + } + + fn take_from(bytes: &mut &[u8]) -> Option { + let error = io::Error::from_raw_os_error(i32::from_le_bytes(take_array(bytes)?)); + let syscall = RIMRAF_SYSCALL_NAMES.get(usize::try_from(u64::from_le_bytes(take_array(bytes)?)).ok()?)?; + Some(RimrafFailure { error, syscall, path: take_text(bytes)? }) + } +} + fn failure(error: io::Error, syscall: &'static str, path: &Path) -> RimrafFailure { RimrafFailure { error, syscall, path: path.to_string_lossy().into_owned() } } -pub fn remove_like_node_rimraf(path: &Path) -> Result<(), RimrafFailure> { - match fs::symlink_metadata(path) { - Ok(metadata) if metadata.is_dir() => return remove_folder_like_node_rimraf(path, None), - Err(error) if error.kind() == ErrorKind::NotFound => return Ok(()), - _ => {} +pub fn remove_entry_like_node_rimraf(path: &Path, entry_kind: RimrafEntryKind, may_use_workers: bool) -> Result<(), RimrafFailure> { + match entry_kind { + RimrafEntryKind::Folder => return remove_folder_like_node_rimraf(path, None, may_use_workers), + RimrafEntryKind::Missing => return Ok(()), + RimrafEntryKind::Other => {} } match fs::remove_file(path) { Ok(()) => Ok(()), Err(error) if error.kind() == ErrorKind::NotFound => Ok(()), Err(error) if matches!(error.raw_os_error(), Some(EISDIR | EPERM)) => { - remove_folder_like_node_rimraf(path, Some(failure(error, "unlink", path))) + remove_folder_like_node_rimraf(path, Some(failure(error, "unlink", path)), may_use_workers) } Err(error) => Err(failure(error, "unlink", path)), } } -fn remove_folder_like_node_rimraf(path: &Path, original_failure: Option) -> Result<(), RimrafFailure> { +fn remove_folder_like_node_rimraf( + path: &Path, + original_failure: Option, + may_use_workers: bool, +) -> Result<(), RimrafFailure> { match fs::remove_dir(path) { Ok(()) => Ok(()), Err(error) if error.kind() == ErrorKind::NotFound => Ok(()), - Err(error) if matches!(error.raw_os_error(), Some(ENOTEMPTY | EEXIST | EPERM)) => remove_children_then_folder(path), + Err(error) if matches!(error.raw_os_error(), Some(ENOTEMPTY | EEXIST | EPERM)) => { + remove_children_then_folder(path, may_use_workers) + } Err(error) if error.raw_os_error() == Some(ENOTDIR) => match original_failure { Some(original_failure) => Err(original_failure), None => Err(failure(error, "rmdir", path)), @@ -47,19 +89,19 @@ fn remove_folder_like_node_rimraf(path: &Path, original_failure: Option Result<(), RimrafFailure> { +fn remove_children_then_folder(path: &Path, may_use_workers: bool) -> Result<(), RimrafFailure> { let reader = fs::read_dir(path).map_err(|error| failure(error, "scandir", path))?; - let mut child_paths: Vec = Vec::new(); + let mut children: Vec<(PathBuf, RimrafEntryKind)> = Vec::new(); for child in reader { - child_paths.push(child.map_err(|error| failure(error, "scandir", path))?.path()); - } - child_paths.sort(); - let mut first_failure: Option = None; - for child_path in child_paths { - if let Err(child_failure) = remove_like_node_rimraf(&child_path) { - first_failure.get_or_insert(child_failure); - } + let child = child.map_err(|error| failure(error, "scandir", path))?; + children.push((child.path(), rimraf_entry_kind(child.file_type().map(|file_type| file_type.is_dir())))); } + children.sort_unstable_by(|left, right| left.0.cmp(&right.0)); + let children_per_worker = if may_use_workers { REMOVED_CHILDREN_PER_WORKER } else { SEQUENTIAL_ONLY }; + let children_may_use_workers = may_use_workers && children.len() < 2 * REMOVED_CHILDREN_PER_WORKER; + let first_failure = attempt_every_item_in_order(&children, children_per_worker, &|(child_path, child_kind)| { + remove_entry_like_node_rimraf(child_path, *child_kind, children_may_use_workers) + }); if let Some(first_failure) = first_failure { return Err(first_failure); } diff --git a/apps/heft-native/src/builtin/open_file_copy.rs b/apps/heft-native/src/builtin/open_file_copy.rs new file mode 100644 index 0000000000..5e36a06400 --- /dev/null +++ b/apps/heft-native/src/builtin/open_file_copy.rs @@ -0,0 +1,88 @@ +use super::node_file_system_error::NodeFileSystemError; + +#[cfg(unix)] +pub fn copy_through_open_source_file(source_path: &str, destination_path: &str) -> Option> { + use std::fs::{self, File}; + use std::io::{self, ErrorKind}; + + use super::file_operations::{check_destination_can_be_replaced, ensure_folder_exists}; + use super::posix_path::directory_name; + + let source_file = File::open(source_path).ok()?; + let source_metadata = source_file.metadata().ok().filter(fs::Metadata::is_file)?; + let copy_failure = |error: io::Error| NodeFileSystemError::new(error, "copyfile", source_path, Some(destination_path)); + let destination_file = match create_new_file_like(destination_path, &source_metadata) { + Ok(destination_file) => destination_file, + Err(error) if error.kind() == ErrorKind::NotFound => { + if let Err(failure) = ensure_folder_exists(directory_name(destination_path)) { + return Some(Err(failure)); + } + create_new_file_like(destination_path, &source_metadata).ok()? + } + Err(error) if error.kind() == ErrorKind::AlreadyExists => { + let destination_metadata = fs::symlink_metadata(destination_path).ok()?; + if let Err(failure) = + check_destination_can_be_replaced(source_path, &source_metadata, destination_path, &destination_metadata) + { + return Some(Err(failure)); + } + if let Err(error) = fs::remove_file(destination_path) { + return Some(Err(NodeFileSystemError::new(error, "unlink", destination_path, None))); + } + match open_destination_like_standard_library_copy(destination_path, &source_metadata) { + Ok(destination_file) => destination_file, + Err(error) => return Some(Err(copy_failure(error))), + } + } + Err(_) => return None, + }; + Some(copy_file_contents(&source_file, &source_metadata, destination_file).map_err(copy_failure)) +} + +#[cfg(unix)] +fn create_new_file_like(destination_path: &str, source_metadata: &std::fs::Metadata) -> std::io::Result { + use std::os::unix::fs::{OpenOptionsExt, PermissionsExt}; + let destination_file = std::fs::OpenOptions::new() + .mode(source_metadata.permissions().mode()) + .write(true) + .create_new(true) + .open(destination_path)?; + destination_file.set_permissions(source_metadata.permissions())?; + Ok(destination_file) +} + +#[cfg(unix)] +fn open_destination_like_standard_library_copy( + destination_path: &str, + source_metadata: &std::fs::Metadata, +) -> std::io::Result { + use std::os::unix::fs::{OpenOptionsExt, PermissionsExt}; + let destination_file = std::fs::OpenOptions::new() + .mode(source_metadata.permissions().mode()) + .write(true) + .create(true) + .truncate(true) + .open(destination_path)?; + if destination_file.metadata()?.is_file() { + destination_file.set_permissions(source_metadata.permissions())?; + } + Ok(destination_file) +} + +#[cfg(unix)] +fn copy_file_contents( + source_file: &std::fs::File, + source_metadata: &std::fs::Metadata, + mut destination_file: std::fs::File, +) -> std::io::Result<()> { + use std::io::Read; + if source_metadata.len() > 0 { + std::io::copy(&mut source_file.take(source_metadata.len()), &mut destination_file)?; + } + Ok(()) +} + +#[cfg(not(unix))] +pub fn copy_through_open_source_file(_source_path: &str, _destination_path: &str) -> Option> { + None +} diff --git a/apps/heft-native/src/builtin/parallel_items.rs b/apps/heft-native/src/builtin/parallel_items.rs new file mode 100644 index 0000000000..87d9bb6b74 --- /dev/null +++ b/apps/heft-native/src/builtin/parallel_items.rs @@ -0,0 +1,184 @@ +pub use super::worker_transfer::WorkerTransfer; +use super::worker_transfer::{ + finish_worker_process, start_worker_process, CHUNK_COMPLETE_RECORD, ITEM_DONE_RECORD, ITEM_FAILED_RECORD, +}; + +pub const MAXIMUM_WORKER_COUNT: usize = 4; +pub const SEQUENTIAL_ONLY: usize = usize::MAX; + +pub fn worker_count_for(item_count: usize, minimum_items_per_worker: usize) -> usize { + let wanted_worker_count = (item_count / minimum_items_per_worker.max(1)).clamp(1, MAXIMUM_WORKER_COUNT); + if wanted_worker_count == 1 || !cfg!(target_os = "linux") || !this_process_may_fork_workers() { + return 1; + } + wanted_worker_count +} + +fn this_process_may_fork_workers() -> bool { + cfg!(test) || std::fs::read_dir("/proc/self/task").is_ok_and(|threads| threads.count() == 1) +} + +trait ItemRunner { + fn run_in_worker(&mut self, item_index: usize, results: &mut Vec) -> bool; + fn run_here(&mut self, item_index: usize) -> bool; + fn accept_output(&mut self, item_index: usize, results: &mut &[u8]) -> bool; + fn accept_failure(&mut self, item_index: usize, results: &mut &[u8]); +} + +struct TypedItemRunner<'run, Item, Output, Failure, ProcessItem> { + items: &'run [Item], + outputs: &'run mut [Output], + failures: Vec<(usize, Failure)>, + process_item: &'run ProcessItem, +} + +impl ItemRunner for TypedItemRunner<'_, Item, Output, Failure, ProcessItem> +where + Output: WorkerTransfer, + Failure: WorkerTransfer, + ProcessItem: Fn(&Item, &mut Output) -> Result<(), Failure>, +{ + fn run_in_worker(&mut self, item_index: usize, results: &mut Vec) -> bool { + match (self.process_item)(&self.items[item_index], &mut self.outputs[item_index]) { + Ok(()) => { + results.push(ITEM_DONE_RECORD); + self.outputs[item_index].append_to(results); + true + } + Err(failure) => { + results.push(ITEM_FAILED_RECORD); + failure.append_to(results); + false + } + } + } + + fn run_here(&mut self, item_index: usize) -> bool { + let result = (self.process_item)(&self.items[item_index], &mut self.outputs[item_index]); + result.map_err(|failure| self.failures.push((item_index, failure))).is_ok() + } + + fn accept_output(&mut self, item_index: usize, results: &mut &[u8]) -> bool { + Output::take_from(results).map(|output| self.outputs[item_index] = output).is_some() + } + + fn accept_failure(&mut self, item_index: usize, results: &mut &[u8]) { + self.failures.extend(Failure::take_from(results).map(|failure| (item_index, failure))); + } +} + +pub fn process_items_in_order( + items: &[Item], + outputs: &mut [Output], + minimum_items_per_worker: usize, + process_item: &impl Fn(&Item, &mut Output) -> Result<(), Failure>, +) -> Result<(), Failure> { + let worker_count = worker_count_for(items.len(), minimum_items_per_worker); + let mut item_is_done = vec![false; items.len()]; + let mut runner = TypedItemRunner { items, outputs, failures: Vec::new(), process_item }; + if worker_count > 1 { + run_items_with_worker_processes(&mut runner, items.len(), worker_count, &mut item_is_done); + } + let TypedItemRunner { outputs, failures, .. } = runner; + let mut failures = failures.into_iter().peekable(); + for (item_index, ((item, output), is_done)) in items.iter().zip(outputs.iter_mut()).zip(item_is_done).enumerate() { + if is_done { + continue; + } + if let Some((_, failure)) = failures.next_if(|(failure_index, _)| *failure_index == item_index) { + return Err(failure); + } + process_item(item, output)?; + } + Ok(()) +} + +fn run_items_with_worker_processes(runner: &mut dyn ItemRunner, item_count: usize, worker_count: usize, item_is_done: &mut [bool]) { + let chunk_length = item_count.div_ceil(worker_count); + let mut running_workers = Vec::new(); + for first_item_index in (chunk_length..item_count).step_by(chunk_length) { + let item_range = first_item_index..(first_item_index + chunk_length).min(item_count); + let run_chunk = &mut |results: &mut Vec| { + for item_index in item_range.clone() { + if !runner.run_in_worker(item_index, results) { + return; + } + } + }; + if let Some(running_worker) = start_worker_process(run_chunk) { + running_workers.push((item_range, running_worker)); + } + } + for (item_index, is_done) in item_is_done[..chunk_length.min(item_count)].iter_mut().enumerate() { + if !runner.run_here(item_index) { + break; + } + *is_done = true; + } + for (item_range, running_worker) in running_workers { + let results = finish_worker_process(running_worker).unwrap_or_default(); + let mut remaining_results = results.as_slice(); + let mut item_index = item_range.start; + while let Some((&record, record_body)) = remaining_results.split_first() { + remaining_results = record_body; + if record == ITEM_FAILED_RECORD { + runner.accept_failure(item_index, &mut remaining_results); + break; + } + if record != ITEM_DONE_RECORD || item_index >= item_range.end || !runner.accept_output(item_index, &mut remaining_results) { + break; + } + item_is_done[item_index] = true; + item_index += 1; + } + } +} + +pub fn attempt_every_item_in_order( + items: &[Item], + minimum_items_per_worker: usize, + attempt_item: &impl Fn(&Item) -> Result<(), Failure>, +) -> Option { + let worker_count = worker_count_for(items.len(), minimum_items_per_worker); + if worker_count == 1 { + return first_failure_attempting_every_item(items, attempt_item); + } + let chunk_length = items.len().div_ceil(worker_count); + let mut running_workers = Vec::new(); + for chunk_items in items.chunks(chunk_length).skip(1) { + let run_chunk = &mut |results: &mut Vec| { + if let Some(failure) = first_failure_attempting_every_item(chunk_items, attempt_item) { + results.push(ITEM_FAILED_RECORD); + failure.append_to(results); + } + }; + running_workers.push((chunk_items, start_worker_process(run_chunk))); + } + let mut first_failure = first_failure_attempting_every_item(&items[..chunk_length], attempt_item); + for (chunk_items, running_worker) in running_workers { + let results = running_worker.and_then(finish_worker_process); + let reported_failure = match results.as_deref().and_then(<[u8]>::split_last) { + Some((&CHUNK_COMPLETE_RECORD, [])) => Some(None), + Some((&CHUNK_COMPLETE_RECORD, [ITEM_FAILED_RECORD, failure_bytes @ ..])) => { + Failure::take_from(&mut &failure_bytes[..]).map(Some) + } + _ => None, + }; + let worker_failure = reported_failure.unwrap_or_else(|| first_failure_attempting_every_item(chunk_items, attempt_item)); + first_failure = first_failure.or(worker_failure); + } + first_failure +} + +fn first_failure_attempting_every_item( + items: &[Item], + attempt_item: &impl Fn(&Item) -> Result<(), Failure>, +) -> Option { + let mut first_failure = None; + for item in items { + if let Err(failure) = attempt_item(item) { + first_failure.get_or_insert(failure); + } + } + first_failure +} diff --git a/apps/heft-native/src/builtin/parallel_items_tests.rs b/apps/heft-native/src/builtin/parallel_items_tests.rs new file mode 100644 index 0000000000..b3972aed92 --- /dev/null +++ b/apps/heft-native/src/builtin/parallel_items_tests.rs @@ -0,0 +1,84 @@ +use super::parallel_items::{attempt_every_item_in_order, process_items_in_order, worker_count_for, MAXIMUM_WORKER_COUNT, SEQUENTIAL_ONLY}; + +fn scratch_folder(name: &str) -> std::path::PathBuf { + let folder = std::env::temp_dir().join(format!("heft-native-parallel-{name}-{}", std::process::id())); + let _ = std::fs::remove_dir_all(&folder); + std::fs::create_dir_all(&folder).unwrap(); + folder +} + +#[test] +fn outputs_are_in_item_order_with_and_without_workers() { + let items: Vec = (0..1000).collect(); + for minimum_items_per_worker in [1, 64, SEQUENTIAL_ONLY] { + let mut outputs = vec![0usize; items.len()]; + let result: Result<(), ()> = process_items_in_order(&items, &mut outputs, minimum_items_per_worker, &|item, output| { + *output = item * 3; + Ok(()) + }); + assert!(result.is_ok()); + assert!(outputs.iter().enumerate().all(|(index, output)| *output == index * 3)); + } +} + +#[test] +fn the_first_failing_item_in_order_is_reported_and_earlier_items_are_all_processed() { + let items: Vec = (0..997).collect(); + for failing_items in [vec![990], vec![5, 600], vec![700, 260, 998], vec![0]] { + let mut outputs = vec![false; items.len()]; + let result = process_items_in_order(&items, &mut outputs, 8, &|item, output| { + if failing_items.contains(item) { + return Err(*item); + } + *output = true; + Ok(()) + }); + let first_failing_item = failing_items.iter().copied().filter(|item| *item < items.len()).min(); + assert_eq!(result.err(), first_failing_item); + assert!(outputs[..first_failing_item.unwrap_or(items.len())].iter().all(|output| *output)); + } +} + +#[test] +fn items_of_a_worker_that_dies_are_processed_by_the_parent() { + let parent_process_id = std::process::id(); + let items: Vec = (0..400).collect(); + let mut outputs = vec![0usize; items.len()]; + let result: Result<(), usize> = process_items_in_order(&items, &mut outputs, 8, &|item, output| { + if *item == 250 && std::process::id() != parent_process_id { + std::process::abort(); + } + *output = item + 1; + Ok(()) + }); + assert!(result.is_ok()); + assert!(outputs.iter().enumerate().all(|(index, output)| *output == index + 1)); +} + +#[test] +fn every_item_is_attempted_and_the_first_failure_in_order_is_reported() { + let items: Vec = (0..1000).collect(); + for minimum_items_per_worker in [1, 16, SEQUENTIAL_ONLY] { + let folder = scratch_folder(&format!("attempt-{minimum_items_per_worker}")); + let first_failure = attempt_every_item_in_order(&items, minimum_items_per_worker, &|item| { + std::fs::write(folder.join(format!("{item}.txt")), b"x").map_err(|_| usize::MAX)?; + if *item == 999 || *item == 420 || *item == 421 { + return Err(*item); + } + Ok(()) + }); + assert_eq!(first_failure, Some(420)); + assert_eq!(std::fs::read_dir(&folder).unwrap().count(), items.len()); + std::fs::remove_dir_all(&folder).unwrap(); + } + assert_eq!(attempt_every_item_in_order(&items, 1, &|_| Ok::<(), usize>(())), None); +} + +#[test] +fn empty_and_tiny_inputs_never_start_workers() { + assert_eq!(worker_count_for(0, 64), 1); + assert_eq!(worker_count_for(6, 64), 1); + assert_eq!(worker_count_for(127, 64), 1); + assert_eq!(worker_count_for(100_000, SEQUENTIAL_ONLY), 1); + assert!(worker_count_for(100_000, 64) <= MAXIMUM_WORKER_COUNT); +} diff --git a/apps/heft-native/src/builtin/path_hash.rs b/apps/heft-native/src/builtin/path_hash.rs new file mode 100644 index 0000000000..883fdbe854 --- /dev/null +++ b/apps/heft-native/src/builtin/path_hash.rs @@ -0,0 +1,76 @@ +use std::collections::{HashMap, HashSet}; +use std::hash::{BuildHasherDefault, Hasher}; + +const WORD_MULTIPLIER: u64 = 0x517c_c1b7_2722_0a95; +const WORD_ROTATION: u32 = 5; +const FINAL_ROTATION: u32 = 26; + +#[derive(Default)] +pub struct PathHasher { + hash: u64, +} + +impl PathHasher { + fn add_word(&mut self, word: u64) { + self.hash = (self.hash.rotate_left(WORD_ROTATION) ^ word).wrapping_mul(WORD_MULTIPLIER); + } +} + +impl Hasher for PathHasher { + fn write(&mut self, bytes: &[u8]) { + let (words, tail) = bytes.as_chunks::<8>(); + for word in words { + self.add_word(u64::from_le_bytes(*word)); + } + let mut tail_word = [0u8; 8]; + tail_word[..tail.len()].copy_from_slice(tail); + self.add_word(u64::from_le_bytes(tail_word) ^ ((bytes.len() as u64) << 56)); + } + + fn write_u8(&mut self, byte: u8) { + self.add_word(u64::from(byte)); + } + + fn write_usize(&mut self, value: usize) { + self.add_word(value as u64); + } + + fn finish(&self) -> u64 { + self.hash.rotate_left(FINAL_ROTATION) + } +} + +pub type PathHashMap = HashMap>; +pub type PathHashSet = HashSet>; + +pub fn path_hash_map_with_capacity(capacity: usize) -> PathHashMap { + PathHashMap::with_capacity_and_hasher(capacity, BuildHasherDefault::default()) +} + +pub fn path_hash_set_with_capacity(capacity: usize) -> PathHashSet { + PathHashSet::with_capacity_and_hasher(capacity, BuildHasherDefault::default()) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn similar_paths_spread_over_buckets_and_tables_behave_like_std() { + let paths: Vec = (0..20_000).map(|index| format!("/p/src/assets/d{:03}/file-{:03}.txt", index / 100, index % 100)).collect(); + let mut index_by_path = path_hash_map_with_capacity::<&str, usize>(paths.len()); + for (index, path) in paths.iter().enumerate() { + assert!(index_by_path.insert(path, index).is_none()); + } + assert!(paths.iter().enumerate().all(|(index, path)| index_by_path.get(path.as_str()) == Some(&index))); + let low_bits: PathHashSet = paths + .iter() + .map(|path| { + let mut hasher = PathHasher::default(); + hasher.write(path.as_bytes()); + hasher.finish() & 0xffff + }) + .collect(); + assert!(low_bits.len() > 15_000); + } +} diff --git a/apps/heft-native/src/builtin/posix_path.rs b/apps/heft-native/src/builtin/posix_path.rs index c584509fc9..620c5c729e 100644 --- a/apps/heft-native/src/builtin/posix_path.rs +++ b/apps/heft-native/src/builtin/posix_path.rs @@ -30,6 +30,27 @@ pub fn normalize_absolute_path(path: &str) -> String { normalized } +pub fn is_normalized_absolute_folder_path(path: &str) -> bool { + path.strip_prefix('/').is_some_and(|rest| rest.split('/').all(|segment| !matches!(segment, "" | "." | ".."))) +} + +pub fn resolve_relative_path_against_normalized_folder(normalized_folder_path: &str, relative_path: &str) -> Option { + let mut folder_path = normalized_folder_path; + let mut remaining_path = relative_path; + while let Some(rest) = remaining_path.strip_prefix("../") { + folder_path = &folder_path[..folder_path.rfind('/').filter(|separator_index| *separator_index > 0)?]; + remaining_path = rest; + } + if remaining_path.split('/').any(|segment| matches!(segment, "" | "." | "..")) { + return None; + } + let mut resolved_path = String::with_capacity(folder_path.len() + 1 + remaining_path.len()); + resolved_path.push_str(folder_path); + resolved_path.push('/'); + resolved_path.push_str(remaining_path); + Some(resolved_path) +} + pub fn relative_path(from_folder: &str, to_path: &str) -> String { let from_segments: Vec<&str> = from_folder.split('/').filter(|s| !s.is_empty()).collect(); let to_segments: Vec<&str> = to_path.split('/').filter(|s| !s.is_empty()).collect(); @@ -62,6 +83,12 @@ pub fn base_name(path: &str) -> &str { } } +pub fn path_contains(outer_path: &str, inner_path: &str) -> bool { + outer_path == "/" + || inner_path == outer_path + || (inner_path.starts_with(outer_path) && inner_path.as_bytes().get(outer_path.len()) == Some(&b'/')) +} + pub fn directory_name(path: &str) -> &str { match path.rfind('/') { Some(0) => "/", @@ -74,6 +101,27 @@ pub fn directory_name(path: &str) -> &str { mod tests { use super::*; + #[test] + fn the_relative_path_fast_path_agrees_with_resolve_path() { + let folders = ["/p/temp/build/copy-assets", "/p", "/a/b", "/p/x y/z"]; + let relatives = [ + "../../../src/a.txt", "src/a.txt", "a", "../a", "../../a", "../../../../../../a", "./a", "a/./b", "a//b", "a/", + "..", "../", "a/../b", "../..", "", ".", "../src/../a", "d000/file-000.txt", + ]; + for folder in folders { + assert!(is_normalized_absolute_folder_path(folder)); + for relative in relatives { + if let Some(resolved) = resolve_relative_path_against_normalized_folder(folder, relative) { + assert_eq!(resolved, resolve_path(folder, relative), "{folder} + {relative}"); + } + } + } + assert_eq!(resolve_relative_path_against_normalized_folder("/p/t/b", "../../s/a.txt").as_deref(), Some("/p/s/a.txt")); + for not_normalized in ["/", "p", "/p/", "/p//q", "/p/./q", "/p/../q", ""] { + assert!(!is_normalized_absolute_folder_path(not_normalized), "{not_normalized}"); + } + } + #[test] fn paths_resolve_and_relativize_like_node_posix_path() { assert_eq!(resolve_path("/p", "src"), "/p/src"); @@ -89,5 +137,7 @@ mod tests { assert_eq!(relative_path("/foo/bar", "/"), "../.."); assert_eq!(base_name("/p/src/x.txt"), "x.txt"); assert_eq!(directory_name("/p/temp/file-copy.json"), "/p/temp"); + assert!(path_contains("/p/lib", "/p/lib") && path_contains("/p/lib", "/p/lib/a") && path_contains("/", "/p")); + assert!(!path_contains("/p/lib", "/p/lib2") && !path_contains("/p/lib/a", "/p/lib")); } } diff --git a/apps/heft-native/src/builtin/sha256.rs b/apps/heft-native/src/builtin/sha256.rs index c5fc5e073b..0c9f9bd8e3 100644 --- a/apps/heft-native/src/builtin/sha256.rs +++ b/apps/heft-native/src/builtin/sha256.rs @@ -40,16 +40,14 @@ impl Sha256 { self.block_buffer_len += copied_len; data = &data[copied_len..]; if self.block_buffer_len == 64 { - let full_block = self.block_buffer; - self.process_block(&full_block); + compress_blocks(&mut self.state, &self.block_buffer); self.block_buffer_len = 0; } } - while data.len() >= 64 { - let mut full_block = [0u8; 64]; - full_block.copy_from_slice(&data[..64]); - self.process_block(&full_block); - data = &data[64..]; + let whole_blocks_length = data.len() / 64 * 64; + if whole_blocks_length > 0 { + compress_blocks(&mut self.state, &data[..whole_blocks_length]); + data = &data[whole_blocks_length..]; } if !data.is_empty() { self.block_buffer[..data.len()].copy_from_slice(data); @@ -65,8 +63,7 @@ impl Sha256 { for byte in &mut self.block_buffer[self.block_buffer_len..] { *byte = 0; } - let full_block = self.block_buffer; - self.process_block(&full_block); + compress_blocks(&mut self.state, &self.block_buffer); self.block_buffer = [0; 64]; self.block_buffer_len = 0; } @@ -74,75 +71,87 @@ impl Sha256 { *byte = 0; } self.block_buffer[56..64].copy_from_slice(&length_bits.to_be_bytes()); - let final_block = self.block_buffer; - self.process_block(&final_block); + compress_blocks(&mut self.state, &self.block_buffer); let mut digest = [0u8; 32]; for (word_index, state_word) in self.state.iter().enumerate() { digest[word_index * 4..word_index * 4 + 4].copy_from_slice(&state_word.to_be_bytes()); } digest } +} - fn process_block(&mut self, block: &[u8; 64]) { - let mut message_schedule = [0u32; 64]; - for (word_index, message_word) in message_schedule.iter_mut().enumerate().take(16) { - let byte_index = word_index * 4; - *message_word = u32::from_be_bytes([ - block[byte_index], - block[byte_index + 1], - block[byte_index + 2], - block[byte_index + 3], - ]); - } - for word_index in 16..64 { - let small_sigma_zero = message_schedule[word_index - 15].rotate_right(7) - ^ message_schedule[word_index - 15].rotate_right(18) - ^ (message_schedule[word_index - 15] >> 3); - let small_sigma_one = message_schedule[word_index - 2].rotate_right(17) - ^ message_schedule[word_index - 2].rotate_right(19) - ^ (message_schedule[word_index - 2] >> 10); - message_schedule[word_index] = message_schedule[word_index - 16] - .wrapping_add(small_sigma_zero) - .wrapping_add(message_schedule[word_index - 7]) - .wrapping_add(small_sigma_one); - } - let mut a = self.state[0]; - let mut b = self.state[1]; - let mut c = self.state[2]; - let mut d = self.state[3]; - let mut e = self.state[4]; - let mut f = self.state[5]; - let mut g = self.state[6]; - let mut h = self.state[7]; - for round_index in 0..64 { - let big_sigma_one = e.rotate_right(6) ^ e.rotate_right(11) ^ e.rotate_right(25); - let choice = (e & f) ^ (!e & g); - let temporary_one = h - .wrapping_add(big_sigma_one) - .wrapping_add(choice) - .wrapping_add(ROUND_CONSTANTS[round_index]) - .wrapping_add(message_schedule[round_index]); - let big_sigma_zero = a.rotate_right(2) ^ a.rotate_right(13) ^ a.rotate_right(22); - let majority = (a & b) ^ (a & c) ^ (b & c); - let temporary_two = big_sigma_zero.wrapping_add(majority); - h = g; - g = f; - f = e; - e = d.wrapping_add(temporary_one); - d = c; - c = b; - b = a; - a = temporary_one.wrapping_add(temporary_two); - } - self.state[0] = self.state[0].wrapping_add(a); - self.state[1] = self.state[1].wrapping_add(b); - self.state[2] = self.state[2].wrapping_add(c); - self.state[3] = self.state[3].wrapping_add(d); - self.state[4] = self.state[4].wrapping_add(e); - self.state[5] = self.state[5].wrapping_add(f); - self.state[6] = self.state[6].wrapping_add(g); - self.state[7] = self.state[7].wrapping_add(h); +fn compress_blocks(state: &mut [u32; 8], blocks: &[u8]) { + if !crate::simd::compress_sha256_blocks_if_the_cpu_can(state, blocks) { + compress_blocks_without_simd(state, blocks); + } +} + +pub fn compress_blocks_without_simd(state: &mut [u32; 8], blocks: &[u8]) { + let (whole_blocks, _) = blocks.as_chunks::<64>(); + for block in whole_blocks { + compress_block_without_simd(state, block); + } +} + +fn compress_block_without_simd(state: &mut [u32; 8], block: &[u8; 64]) { + let mut message_schedule = [0u32; 64]; + for (word_index, message_word) in message_schedule.iter_mut().enumerate().take(16) { + let byte_index = word_index * 4; + *message_word = u32::from_be_bytes([ + block[byte_index], + block[byte_index + 1], + block[byte_index + 2], + block[byte_index + 3], + ]); + } + for word_index in 16..64 { + let small_sigma_zero = message_schedule[word_index - 15].rotate_right(7) + ^ message_schedule[word_index - 15].rotate_right(18) + ^ (message_schedule[word_index - 15] >> 3); + let small_sigma_one = message_schedule[word_index - 2].rotate_right(17) + ^ message_schedule[word_index - 2].rotate_right(19) + ^ (message_schedule[word_index - 2] >> 10); + message_schedule[word_index] = message_schedule[word_index - 16] + .wrapping_add(small_sigma_zero) + .wrapping_add(message_schedule[word_index - 7]) + .wrapping_add(small_sigma_one); + } + let mut a = state[0]; + let mut b = state[1]; + let mut c = state[2]; + let mut d = state[3]; + let mut e = state[4]; + let mut f = state[5]; + let mut g = state[6]; + let mut h = state[7]; + for round_index in 0..64 { + let big_sigma_one = e.rotate_right(6) ^ e.rotate_right(11) ^ e.rotate_right(25); + let choice = (e & f) ^ (!e & g); + let temporary_one = h + .wrapping_add(big_sigma_one) + .wrapping_add(choice) + .wrapping_add(ROUND_CONSTANTS[round_index]) + .wrapping_add(message_schedule[round_index]); + let big_sigma_zero = a.rotate_right(2) ^ a.rotate_right(13) ^ a.rotate_right(22); + let majority = (a & b) ^ (a & c) ^ (b & c); + let temporary_two = big_sigma_zero.wrapping_add(majority); + h = g; + g = f; + f = e; + e = d.wrapping_add(temporary_one); + d = c; + c = b; + b = a; + a = temporary_one.wrapping_add(temporary_two); } + state[0] = state[0].wrapping_add(a); + state[1] = state[1].wrapping_add(b); + state[2] = state[2].wrapping_add(c); + state[3] = state[3].wrapping_add(d); + state[4] = state[4].wrapping_add(e); + state[5] = state[5].wrapping_add(f); + state[6] = state[6].wrapping_add(g); + state[7] = state[7].wrapping_add(h); } impl Default for Sha256 { diff --git a/apps/heft-native/src/builtin/simple_glob.rs b/apps/heft-native/src/builtin/simple_glob.rs index 492d4d06d3..6ca1dffe97 100644 --- a/apps/heft-native/src/builtin/simple_glob.rs +++ b/apps/heft-native/src/builtin/simple_glob.rs @@ -1,8 +1,8 @@ -use std::collections::HashSet; use std::fs; use std::io::ErrorKind; -use super::posix_path::resolve_path; +use super::path_hash::path_hash_set_with_capacity; +use super::posix_path::{path_contains, resolve_path}; use super::simple_glob_pattern::{parse_simple_glob_pattern, SimpleGlobPattern}; pub struct GlobbedEntry { @@ -28,6 +28,17 @@ pub fn patterns_are_simple(patterns: &[String]) -> bool { summarize_patterns(patterns).is_some() } +pub fn patterns_select_each_path_once(patterns: &[String]) -> bool { + summarize_patterns(patterns).is_some_and(|summary| summary.literal_paths.is_empty()) +} + +pub fn patterns_only_read_inside(patterns: &[String], cwd: &str) -> bool { + summarize_patterns(patterns).is_some_and(|summary| { + let folder_path = resolve_path(cwd, ""); + summary.literal_paths.iter().all(|literal_path| path_contains(&folder_path, &resolve_path(cwd, literal_path))) + }) +} + fn summarize_patterns(patterns: &[String]) -> Option { if patterns.is_empty() { return None; @@ -55,7 +66,7 @@ pub fn try_simple_glob(patterns: &[String], cwd: &str, only_files: bool) -> Opti let is_recursive = summary.match_any_recursive || !summary.suffixes.is_empty(); let has_dynamic_patterns = is_recursive || summary.match_any_top_level || !summary.prefixes.is_empty(); let mut entries: Vec = Vec::new(); - let mut literal_relative_paths: HashSet<&str> = HashSet::new(); + let mut literal_relative_paths = path_hash_set_with_capacity::<&str>(summary.literal_paths.len()); for literal_path in &summary.literal_paths { let absolute_path = resolve_path(cwd, literal_path); match fs::symlink_metadata(&absolute_path) { @@ -144,7 +155,7 @@ fn read_sorted_folder(folder_path: &str) -> FolderReadResult { } children.push((name, file_type.is_dir(), file_type.is_file())); } - children.sort_by(|left, right| left.0.as_bytes().cmp(right.0.as_bytes())); + children.sort_unstable_by(|left, right| left.0.as_bytes().cmp(right.0.as_bytes())); FolderReadResult::Entries(children) } diff --git a/apps/heft-native/src/builtin/worker_transfer.rs b/apps/heft-native/src/builtin/worker_transfer.rs new file mode 100644 index 0000000000..8d1863f851 --- /dev/null +++ b/apps/heft-native/src/builtin/worker_transfer.rs @@ -0,0 +1,89 @@ +use std::io::{Read, Write}; + +pub const ITEM_DONE_RECORD: u8 = 1; +pub const ITEM_FAILED_RECORD: u8 = 2; +pub const CHUNK_COMPLETE_RECORD: u8 = 3; + +pub trait WorkerTransfer: Sized { + fn append_to(&self, bytes: &mut Vec); + fn take_from(bytes: &mut &[u8]) -> Option; +} + +pub fn append_text(text: &str, bytes: &mut Vec) { + bytes.extend_from_slice(&(text.len() as u64).to_le_bytes()); + bytes.extend_from_slice(text.as_bytes()); +} + +pub fn take_text(bytes: &mut &[u8]) -> Option { + let text_length = usize::try_from(u64::from_le_bytes(take_array(bytes)?)).ok()?; + let (text_bytes, remaining_bytes) = bytes.split_at_checked(text_length)?; + *bytes = remaining_bytes; + String::from_utf8(text_bytes.to_vec()).ok() +} + +pub fn take_array(bytes: &mut &[u8]) -> Option<[u8; LENGTH]> { + let (array, remaining_bytes) = bytes.split_first_chunk::()?; + *bytes = remaining_bytes; + Some(*array) +} + +impl WorkerTransfer for () { + fn append_to(&self, _bytes: &mut Vec) {} + + fn take_from(_bytes: &mut &[u8]) -> Option<()> { + Some(()) + } +} + +impl WorkerTransfer for bool { + fn append_to(&self, bytes: &mut Vec) { + bytes.push(u8::from(*self)); + } + + fn take_from(bytes: &mut &[u8]) -> Option { + take_array::<1>(bytes).map(|[byte]| byte != 0) + } +} + +impl WorkerTransfer for usize { + fn append_to(&self, bytes: &mut Vec) { + bytes.extend_from_slice(&(*self as u64).to_le_bytes()); + } + + fn take_from(bytes: &mut &[u8]) -> Option { + usize::try_from(u64::from_le_bytes(take_array(bytes)?)).ok() + } +} + +impl WorkerTransfer for [u8; LENGTH] { + fn append_to(&self, bytes: &mut Vec) { + bytes.extend_from_slice(self); + } + + fn take_from(bytes: &mut &[u8]) -> Option<[u8; LENGTH]> { + take_array(bytes) + } +} + +pub fn finish_worker_process((worker_process_id, mut results_reader): (i32, std::io::PipeReader)) -> Option> { + let mut results = Vec::new(); + let results_were_read = results_reader.read_to_end(&mut results).is_ok(); + drop(results_reader); + crate::sys::wait_for_worker_process(worker_process_id); + results_were_read.then_some(results) +} + +pub fn start_worker_process(run_chunk_in_worker: &mut dyn FnMut(&mut Vec)) -> Option<(i32, std::io::PipeReader)> { + let (results_reader, mut results_writer) = std::io::pipe().ok()?; + match crate::sys::fork_worker_process_of_this_single_threaded_process().ok()? { + crate::sys::ForkedWorker::InParent { worker_process_id } => Some((worker_process_id, results_reader)), + crate::sys::ForkedWorker::InWorker => { + drop(results_reader); + let mut results = Vec::new(); + run_chunk_in_worker(&mut results); + results.push(CHUNK_COMPLETE_RECORD); + let exit_status = if results_writer.write_all(&results).is_ok() { 0 } else { 1 }; + crate::sys::exit_worker_process_immediately(exit_status) + } + } +} diff --git a/apps/heft-native/src/cli/action_invocation.rs b/apps/heft-native/src/cli/action_invocation.rs index cb400132f2..6915e07b7d 100644 --- a/apps/heft-native/src/cli/action_invocation.rs +++ b/apps/heft-native/src/cli/action_invocation.rs @@ -72,7 +72,7 @@ pub fn print_error<'a>(request: &ActionRequest<'a, '_>, registration: &Registrat let mut text: ActionHelpText<'a> = help_text(request, None); let error_prog: String = text.prog.to_string(); if matches!(error, ArgumentError::Ambiguous(_)) { - text.prog = Cow::Owned(format!("heft {}", request.table.actions[request.action_index].name)); + text.prog = Cow::Owned(format!("heft {}", request.table.actions[request.action_index].name())); } let parser = action_help_parser(registration, parameters, text, has_remainder)?; let output = match error { @@ -92,7 +92,7 @@ pub fn execute<'a>(request: &ActionRequest<'a, '_>, selected_phases: Vec, let uses_explicit_values: bool = defaults.iter().chain(request.rest).any(|arg| arg.starts_with('-') && arg.contains('=')); Some(CliOutcome::Execute(Box::new(ParsedCommand { command_name: request.command_name, - unaliased_command_name: action.name.to_string(), + unaliased_command_name: action.name().into_owned(), action_kind: action.kind, watch: action.watch, debug: request.tool_args.contains(&"--debug"), diff --git a/apps/heft-native/src/cli/action_text.rs b/apps/heft-native/src/cli/action_text.rs index 90ad3ed920..c7a799a97d 100644 --- a/apps/heft-native/src/cli/action_text.rs +++ b/apps/heft-native/src/cli/action_text.rs @@ -42,7 +42,8 @@ pub fn action_documentation<'x>(model: &'x CliModel<'x>, action: &ActionEntry<'_ } pub fn alias_expanded_command(table: &ActionTable<'_>, alias: &AliasEntry<'_>) -> String { - let mut expanded: String = format!("heft {}", table.actions[alias.target_index].name); + let mut expanded: String = String::from("heft "); + table.actions[alias.target_index].push_name(&mut expanded); let defaults: String = alias.default_parameters.join(" "); if !defaults.is_empty() { expanded.push(' '); @@ -59,7 +60,7 @@ pub fn alias_documentation(table: &ActionTable<'_>, alias: &AliasEntry<'_>) -> S format!( "{} For more information on the aliased command, use \"heft {} --help\".", alias_summary(table, alias), - table.actions[alias.target_index].name + table.actions[alias.target_index].name() ) } diff --git a/apps/heft-native/src/cli/actions.rs b/apps/heft-native/src/cli/actions.rs index 1cb1b39114..db86334069 100644 --- a/apps/heft-native/src/cli/actions.rs +++ b/apps/heft-native/src/cli/actions.rs @@ -10,13 +10,48 @@ pub enum ActionKind { Phase(usize), } +const WATCH_SUFFIX: &str = "-watch"; + #[derive(Debug)] pub struct ActionEntry<'a> { - pub name: Cow<'a, str>, + pub base_name: &'a str, pub kind: ActionKind, pub watch: bool, } +impl<'a> ActionEntry<'a> { + pub fn name(&self) -> Cow<'a, str> { + if self.watch { + Cow::Owned(format!("{}{WATCH_SUFFIX}", self.base_name)) + } else { + Cow::Borrowed(self.base_name) + } + } + + pub fn has_name(&self, text: &str) -> bool { + if self.watch { + text.strip_suffix(WATCH_SUFFIX) == Some(self.base_name) + } else { + text == self.base_name + } + } + + pub fn push_name(&self, output: &mut String) { + output.push_str(self.base_name); + if self.watch { + output.push_str(WATCH_SUFFIX); + } + } + + fn has_same_name_as(&self, other: &ActionEntry<'_>) -> bool { + match (self.watch, other.watch) { + (true, false) => self.has_name(other.base_name), + (false, true) => other.has_name(self.base_name), + _ => self.base_name == other.base_name, + } + } +} + #[derive(Debug)] pub struct AliasEntry<'a> { pub name: &'a str, @@ -33,22 +68,32 @@ pub struct ActionTable<'a> { impl<'a> ActionTable<'a> { pub fn find_action(&self, name: &str) -> Option { - self.actions.iter().position(|action| action.name == name) + self.actions.iter().position(|action| action.has_name(name)) } pub fn find_alias(&self, name: &str) -> Option<&AliasEntry<'a>> { self.aliases.iter().find(|alias| alias.name == name) } - pub fn command_names(&self) -> impl Iterator { - self.actions.iter().map(|action| action.name.as_ref()).chain(self.aliases.iter().map(|alias| alias.name)) + pub fn push_command_names(&self, output: &mut String) { + for (index, action) in self.actions.iter().enumerate() { + if index > 0 { + output.push_str(", "); + } + action.push_name(output); + } + for alias in &self.aliases { + output.push_str(", "); + output.push_str(alias.name); + } } - fn try_add(&mut self, name: Cow<'a, str>, kind: ActionKind, watch: bool) -> bool { - if !is_valid_action_name(&name) || self.find_action(&name).is_some() { + fn try_add(&mut self, base_name: &'a str, kind: ActionKind, watch: bool) -> bool { + let entry: ActionEntry<'a> = ActionEntry { base_name, kind, watch }; + if !is_valid_action_name(base_name) || self.actions.iter().any(|existing| existing.has_same_name_as(&entry)) { return false; } - self.actions.push(ActionEntry { name, kind, watch }); + self.actions.push(entry); true } } @@ -74,18 +119,18 @@ pub fn build_action_table<'a>(model: &'a CliModel<'a>) -> Option aliases: Vec::with_capacity(model.aliases.len()), phase_dependencies: resolve_phase_dependencies(model)?, }; - table.try_add(Cow::Borrowed("clean"), ActionKind::Clean, false); - table.try_add(Cow::Borrowed("run"), ActionKind::Run, false); + table.try_add("clean", ActionKind::Clean, false); + table.try_add("run", ActionKind::Run, false); for (phase_index, phase) in model.phases.iter().enumerate() { - if !table.try_add(Cow::Borrowed(phase.name), ActionKind::Phase(phase_index), false) { + if !table.try_add(phase.name, ActionKind::Phase(phase_index), false) { return None; } } - if !table.try_add(Cow::Borrowed("run-watch"), ActionKind::Run, true) { + if !table.try_add("run", ActionKind::Run, true) { return None; } for (phase_index, phase) in model.phases.iter().enumerate() { - if !table.try_add(Cow::Owned(format!("{}-watch", phase.name)), ActionKind::Phase(phase_index), true) { + if !table.try_add(phase.name, ActionKind::Phase(phase_index), true) { return None; } } @@ -102,6 +147,12 @@ pub fn build_action_table<'a>(model: &'a CliModel<'a>) -> Option pub fn selected_phases_from(table: &ActionTable<'_>, seeds: impl IntoIterator) -> Vec { let mut selected: Vec = Vec::new(); + select_phases_into(table, seeds, &mut selected); + selected +} + +pub fn select_phases_into(table: &ActionTable<'_>, seeds: impl IntoIterator, selected: &mut Vec) { + selected.clear(); for seed in seeds { if !selected.contains(&seed) { selected.push(seed); @@ -116,5 +167,4 @@ pub fn selected_phases_from(table: &ActionTable<'_>, seeds: impl IntoIterator { + Declared(&'a [&'a str]), + Deduplicated(Vec<&'a str>), +} + +impl<'a> std::ops::Deref for ChoiceAlternatives<'a> { + type Target = [&'a str]; + + fn deref(&self) -> &[&'a str] { + match self { + ChoiceAlternatives::Declared(alternatives) => alternatives, + ChoiceAlternatives::Deduplicated(alternatives) => alternatives, + } + } +} + #[derive(Clone, Debug)] pub struct DefinedParameter<'a> { pub kind: ParameterKind, @@ -12,7 +29,7 @@ pub struct DefinedParameter<'a> { pub scoping_group: bool, pub required: bool, pub argument_name: Option<&'a str>, - pub alternatives: Vec<&'a str>, + pub alternatives: ChoiceAlternatives<'a>, pub default_value: Option>, pub description: Cow<'a, str>, } @@ -27,7 +44,7 @@ impl<'a> DefinedParameter<'a> { scoping_group: false, required: false, argument_name: None, - alternatives: Vec::new(), + alternatives: ChoiceAlternatives::Declared(&[]), default_value: None, description, } diff --git a/apps/heft-native/src/cli/entry.rs b/apps/heft-native/src/cli/entry.rs index 12d08e705b..d8d969341d 100644 --- a/apps/heft-native/src/cli/entry.rs +++ b/apps/heft-native/src/cli/entry.rs @@ -26,10 +26,17 @@ pub fn interpret_command_line_with_color<'a>( interpret_with_output(args, model, read_help_width(), Some(supports_color)) } -pub fn write_printed_output(output: &PrintedOutput) -> i32 { - if !output.stdout.is_empty() { +pub fn write_printed_output_after(standard_output_prefix: &str, output: &PrintedOutput) -> i32 { + let joined: String; + let standard_output: &str = if standard_output_prefix.is_empty() { + &output.stdout + } else { + joined = [standard_output_prefix, output.stdout.as_str()].concat(); + &joined + }; + if !standard_output.is_empty() { let mut stdout = std::io::stdout().lock(); - let _ = stdout.write_all(output.stdout.as_bytes()); + let _ = stdout.write_all(standard_output.as_bytes()); let _ = stdout.flush(); } if !output.stderr.is_empty() { diff --git a/apps/heft-native/src/cli/help_args.rs b/apps/heft-native/src/cli/help_args.rs index 6f520f815f..18ef0e8a91 100644 --- a/apps/heft-native/src/cli/help_args.rs +++ b/apps/heft-native/src/cli/help_args.rs @@ -1,57 +1,61 @@ use super::help_model::{HelpAction, HelpNargs}; -pub fn build_metavar(action: &HelpAction<'_>, default_metavar: &str) -> String { +fn push_metavar(output: &mut String, action: &HelpAction<'_>, uppercase_default: bool) { if let Some(metavar) = action.metavar { - return metavar.to_string(); + output.push_str(metavar); + return; } - if let Some(choices) = &action.choices { - let mut result: String = String::from("{"); + if let Some(choices) = action.choices { + output.push('{'); for (choice_index, choice) in choices.iter().enumerate() { if choice_index > 0 { - result.push(','); + output.push(','); } - result.push_str(choice); + output.push_str(choice); } - result.push('}'); - return result; + output.push('}'); + return; + } + if uppercase_default { + output.extend(action.dest.chars().map(|character| character.to_ascii_uppercase())); + } else { + output.push_str(&action.dest); } - default_metavar.to_string() } -pub fn format_args(action: &HelpAction<'_>, default_metavar: &str) -> String { - let metavar: String = build_metavar(action, default_metavar); +fn push_args(output: &mut String, action: &HelpAction<'_>, uppercase_default: bool) { match action.nargs { - HelpNargs::Single | HelpNargs::Zero => metavar, - HelpNargs::ZeroOrMore => format!("[{} [{} ...]]", metavar, metavar), - HelpNargs::Remainder => String::from("..."), - HelpNargs::Parser => format!("{} ...", metavar), + HelpNargs::Single | HelpNargs::Zero => push_metavar(output, action, uppercase_default), + HelpNargs::ZeroOrMore => { + output.push('['); + push_metavar(output, action, uppercase_default); + output.push_str(" ["); + push_metavar(output, action, uppercase_default); + output.push_str(" ...]]"); + } + HelpNargs::Remainder => output.push_str("..."), + HelpNargs::Parser => { + push_metavar(output, action, uppercase_default); + output.push_str(" ..."); + } } } -pub fn format_action_invocation(action: &HelpAction<'_>) -> String { +pub fn push_action_invocation(output: &mut String, action: &HelpAction<'_>) { if !action.is_optional() { - return build_metavar(action, &action.dest); + push_metavar(output, action, false); + return; } - let mut result: String = String::new(); - if action.nargs == HelpNargs::Zero { - for (index, option_string) in action.option_strings.iter().enumerate() { - if index > 0 { - result.push_str(", "); - } - result.push_str(option_string); - } - return result; - } - let args_string: String = format_args(action, &action.dest.to_ascii_uppercase()); for (index, option_string) in action.option_strings.iter().enumerate() { if index > 0 { - result.push_str(", "); + output.push_str(", "); + } + output.push_str(option_string); + if action.nargs != HelpNargs::Zero { + output.push(' '); + push_args(output, action, true); } - result.push_str(option_string); - result.push(' '); - result.push_str(&args_string); } - result } pub fn format_actions_usage(actions: &[&HelpAction<'_>]) -> String { @@ -60,33 +64,41 @@ pub fn format_actions_usage(actions: &[&HelpAction<'_>]) -> String { if action.help.is_suppressed() { continue; } - let part: String = if !action.is_optional() { - format_args(action, &action.dest) + let separator_start: usize = text.len(); + if !text.is_empty() { + text.push(' '); + } + let part_start: usize = text.len(); + if !action.is_optional() { + push_args(&mut text, action, false); } else { - let option_string: &str = &action.option_strings[0]; - let unbracketed: String = if action.nargs == HelpNargs::Zero { - option_string.to_string() - } else { - format!("{} {}", option_string, format_args(action, &action.dest.to_ascii_uppercase())) - }; - if action.required { - unbracketed - } else { - format!("[{}]", unbracketed) + if !action.required { + text.push('['); + } + text.push_str(&action.option_strings[0]); + if action.nargs != HelpNargs::Zero { + text.push(' '); + push_args(&mut text, action, true); + } + if !action.required { + text.push(']'); } - }; - if part.is_empty() { - continue; } - if !text.is_empty() { - text.push(' '); + if text.len() == part_start { + text.truncate(separator_start); } - text.push_str(&part); } clean_usage_separators(&text) } -fn clean_usage_separators(text: &str) -> String { +pub fn clean_usage_separators(text: &str) -> String { + if !["(", "[ ", " ]", " )", "[]"].iter().any(|pattern| text.contains(pattern)) { + return super::text::trim_javascript_whitespace(text).to_string(); + } + clean_usage_separators_in_every_pass(text) +} + +pub fn clean_usage_separators_in_every_pass(text: &str) -> String { let without_space_after_open: String = remove_space_after_open_bracket(text); let without_space_before_close: String = remove_space_before_close_bracket(&without_space_after_open); let without_empty_brackets: String = remove_empty_pair(&without_space_before_close, b'[', b']'); diff --git a/apps/heft-native/src/cli/help_builders.rs b/apps/heft-native/src/cli/help_builders.rs index aefdc9b3da..7e528b12ca 100644 --- a/apps/heft-native/src/cli/help_builders.rs +++ b/apps/heft-native/src/cli/help_builders.rs @@ -1,7 +1,7 @@ use std::borrow::Cow; use super::defined_parameter::DefinedParameter; -use super::help_model::{bold, HelpAction, HelpGroup, HelpNargs, HelpParser, HelpText}; +use super::help_model::{bold, HelpAction, HelpGroup, HelpNargs, HelpParser, HelpText, DEBUG_OPTION_STRINGS, UNMANAGED_OPTION_STRINGS}; use super::model::ParameterKind; use super::registration::{Registration, RegistrationStep}; @@ -14,7 +14,7 @@ pub fn root_help_parser<'a>(summaries: Vec<(Cow<'a, str>, Cow<'a, str>)>) -> Hel let subactions: Vec> = summaries .into_iter() .map(|(name, summary)| HelpAction { - option_strings: Vec::new(), + option_strings: &[], dest: name, nargs: HelpNargs::Single, metavar: None, @@ -25,7 +25,7 @@ pub fn root_help_parser<'a>(summaries: Vec<(Cow<'a, str>, Cow<'a, str>)>) -> Hel }) .collect(); let subparsers: HelpAction<'a> = HelpAction { - option_strings: Vec::new(), + option_strings: &[], dest: Cow::Borrowed("action"), nargs: HelpNargs::Parser, metavar: Some(""), @@ -41,8 +41,8 @@ pub fn root_help_parser<'a>(summaries: Vec<(Cow<'a, str>, Cow<'a, str>)>) -> Hel actions: vec![ HelpAction::help_option(), subparsers, - HelpAction::flag_option("--debug", DEBUG_DESCRIPTION), - HelpAction::flag_option("--unmanaged", UNMANAGED_DESCRIPTION), + HelpAction::flag_option(&DEBUG_OPTION_STRINGS, DEBUG_DESCRIPTION), + HelpAction::flag_option(&UNMANAGED_OPTION_STRINGS, UNMANAGED_DESCRIPTION), ], groups: vec![ HelpGroup { title: Cow::Borrowed("Positional arguments"), action_indices: vec![1] }, @@ -51,15 +51,15 @@ pub fn root_help_parser<'a>(summaries: Vec<(Cow<'a, str>, Cow<'a, str>)>) -> Hel } } -fn parameter_action<'a>(parameter: &'a DefinedParameter<'a>, option_strings: &[Cow<'a, str>]) -> Option> { +fn parameter_action<'a>(parameter: &'a DefinedParameter<'a>, option_strings: &'a [Cow<'a, str>]) -> Option> { let help_text: Cow<'a, str> = parameter.help_text()?; Some(HelpAction { - option_strings: option_strings.to_vec(), + option_strings, dest: Cow::Borrowed(parameter.long_name), nargs: if parameter.kind == ParameterKind::Flag { HelpNargs::Zero } else { HelpNargs::Single }, metavar: parameter.argument_name, help: HelpText::Text(help_text), - choices: if parameter.kind.has_alternatives() { Some(parameter.alternatives.clone()) } else { None }, + choices: if parameter.kind.has_alternatives() { Some(¶meter.alternatives) } else { None }, required: parameter.required, subactions: Vec::new(), }) @@ -72,7 +72,7 @@ pub struct ActionHelpText<'a> { } pub fn action_help_parser<'a>( - registration: &Registration<'a>, + registration: &'a Registration<'a>, parameters: &'a [DefinedParameter<'a>], text: ActionHelpText<'a>, has_remainder: bool, @@ -85,7 +85,7 @@ pub fn action_help_parser<'a>( let action_index: usize = actions.len(); match step { RegistrationStep::Ambiguous(name) => { - actions.push(HelpAction::hidden_option(name.clone())); + actions.push(HelpAction::hidden_option(std::slice::from_ref(name))); optionals.push(action_index); } RegistrationStep::Parameter { parameter_index, option_strings } => { @@ -98,7 +98,7 @@ pub fn action_help_parser<'a>( if has_remainder { positionals.push(actions.len()); actions.push(HelpAction { - option_strings: Vec::new(), + option_strings: &[], dest: Cow::Borrowed("..."), nargs: HelpNargs::Remainder, metavar: Some("\"...\""), diff --git a/apps/heft-native/src/cli/help_format.rs b/apps/heft-native/src/cli/help_format.rs index 2d6e3756bd..c5f7ffb3fc 100644 --- a/apps/heft-native/src/cli/help_format.rs +++ b/apps/heft-native/src/cli/help_format.rs @@ -1,23 +1,21 @@ -use super::help_args::format_action_invocation; +use super::help_args::push_action_invocation; use super::help_lines::{fill_help_text, for_each_help_line}; use super::help_model::{HelpAction, HelpParser}; -use super::help_usage::format_usage; +use super::help_usage::push_usage; use super::text::push_spaces; const MAX_HELP_POSITION: f64 = 24.0; -fn action_invocations(action: &HelpAction<'_>) -> Vec { - if action.help.is_suppressed() { - return Vec::new(); +fn compute_action_max_length(actions: &[HelpAction<'_>], scratch: &mut String) -> f64 { + let mut longest: Option = None; + for action in actions.iter().filter(|action| !action.help.is_suppressed()) { + for measured_action in std::iter::once(action).chain(action.subactions.iter()) { + scratch.clear(); + push_action_invocation(scratch, measured_action); + longest = Some(longest.map_or(scratch.len(), |length| length.max(scratch.len()))); + } } - let mut invocations: Vec = Vec::with_capacity(1 + action.subactions.len()); - invocations.push(format_action_invocation(action)); - invocations.extend(action.subactions.iter().map(format_action_invocation)); - invocations -} - -fn compute_action_max_length(invocations: &[Vec]) -> f64 { - let longest: Option = invocations.iter().flatten().map(String::len).max(); + scratch.clear(); longest.map_or(0.0, |length| length as f64 + 2.0) } @@ -26,20 +24,21 @@ struct ActionLayout { action_max_length: f64, } -fn format_action(output: &mut String, action: &HelpAction<'_>, invocations: &[String], current_indent: f64, layout: &ActionLayout) { +fn format_action(output: &mut String, action: &HelpAction<'_>, current_indent: f64, layout: &ActionLayout) { let help_position: f64 = (layout.action_max_length + 2.0).min(MAX_HELP_POSITION); let help_width: f64 = layout.width - help_position; let action_width: f64 = help_position - current_indent - 2.0; - let header: &str = &invocations[0]; let help_text: Option<&str> = action.help.visible_text(); let mut indent_first: f64 = 0.0; push_spaces(output, current_indent); - output.push_str(header); + let header_start: usize = output.len(); + push_action_invocation(output, action); + let header_length: f64 = (output.len() - header_start) as f64; match help_text { None => output.push('\n'), - Some(_) if header.len() as f64 <= action_width => { + Some(_) if header_length <= action_width => { output.push_str(" "); - push_spaces(output, action_width - header.len() as f64); + push_spaces(output, action_width - header_length); } Some(_) => { output.push('\n'); @@ -53,11 +52,18 @@ fn format_action(output: &mut String, action: &HelpAction<'_>, invocations: &[St output.push('\n'); }); } - for (subaction_index, subaction) in action.subactions.iter().enumerate() { - format_action(output, subaction, &invocations[1 + subaction_index..], current_indent + 2.0, layout); + for subaction in &action.subactions { + format_action(output, subaction, current_indent + 2.0, layout); } } +fn estimated_help_length(parser: &HelpParser<'_>) -> usize { + let text_length = |action: &HelpAction<'_>| action.help.visible_text().map_or(0, str::len) * 3 / 2 + 64; + let actions: usize = parser.actions.iter().map(|action| text_length(action) + action.subactions.iter().map(text_length).sum::()).sum(); + let texts: usize = parser.description.as_deref().map_or(0, str::len) + parser.epilog.as_deref().map_or(0, str::len); + 512 + parser.prog.len() * 8 + texts * 3 / 2 + actions +} + fn format_text_block(output: &mut String, text: &str, width: f64) { fill_help_text(output, text, width, ""); output.push_str("\n\n"); @@ -87,64 +93,56 @@ pub fn format_help(parser: &HelpParser<'_>, width: f64) -> Option { if !is_parser_text_ascii(parser) { return None; } - let invocations: Vec> = parser.actions.iter().map(action_invocations).collect(); - let layout: ActionLayout = ActionLayout { width, action_max_length: compute_action_max_length(&invocations) }; - let mut help: String = format_usage(&parser.prog, &parser.actions, width)?; + let mut help: String = String::with_capacity(estimated_help_length(parser)); + let layout: ActionLayout = ActionLayout { width, action_max_length: compute_action_max_length(&parser.actions, &mut help) }; + push_usage(&mut help, &parser.prog, &parser.actions, width)?; if let Some(description) = parser.description.as_deref().filter(|text| !text.is_empty()) { format_text_block(&mut help, description, width); } for group in &parser.groups { - let mut section: String = String::new(); + let section_start: usize = help.len(); + help.push('\n'); + help.push_str(&group.title); + help.push_str(":\n"); + let content_start: usize = help.len(); for action_index in &group.action_indices { let action: &HelpAction<'_> = &parser.actions[*action_index]; if !action.help.is_suppressed() { - format_action(&mut section, action, &invocations[*action_index], 2.0, &layout); + format_action(&mut help, action, 2.0, &layout); } } - if !section.is_empty() { - help.push('\n'); - help.push_str(&group.title); - help.push_str(":\n"); - help.push_str(§ion); + if help.len() == content_start { + help.truncate(section_start); + } else { help.push('\n'); } } if let Some(epilog) = parser.epilog.as_deref().filter(|text| !text.is_empty()) { format_text_block(&mut help, epilog, width); } - Some(finish_help(help)) + finish_help(&mut help); + Some(help) } -fn finish_help(help: String) -> String { - let mut collapsed: String = String::with_capacity(help.len()); +fn finish_help(help: &mut String) { let mut newline_run: usize = 0; - for character in help.chars() { - if character == '\n' { - newline_run += 1; - continue; - } - push_newlines(&mut collapsed, newline_run); - newline_run = 0; - collapsed.push(character); - } - push_newlines(&mut collapsed, newline_run); - let trimmed: &str = collapsed.trim_matches('\n'); - let mut result: String = String::with_capacity(trimmed.len() + 1); - result.push_str(trimmed); - result.push('\n'); - result -} - -fn push_newlines(output: &mut String, count: usize) { - for _ in 0..count.min(2) { - output.push('\n'); - } + help.retain(|character| { + newline_run = if character == '\n' { newline_run + 1 } else { 0 }; + newline_run <= 2 + }); + let leading_newlines: usize = help.len() - help.trim_start_matches('\n').len(); + help.drain(..leading_newlines); + let trimmed_length: usize = help.trim_end_matches('\n').len(); + help.truncate(trimmed_length); + help.push('\n'); } pub fn format_usage_only(parser: &HelpParser<'_>, width: f64) -> Option { if !parser.prog.is_ascii() || !parser.actions.iter().all(is_usage_text_ascii) { return None; } - let usage: String = format_usage(&parser.prog, &parser.actions, width)?; - Some(finish_help(usage)) + let mut usage: String = String::with_capacity(256 + parser.prog.len() * 8 + parser.actions.len() * 48); + push_usage(&mut usage, &parser.prog, &parser.actions, width)?; + finish_help(&mut usage); + Some(usage) } diff --git a/apps/heft-native/src/cli/help_lines.rs b/apps/heft-native/src/cli/help_lines.rs index b99af5703c..53ab181ea7 100644 --- a/apps/heft-native/src/cli/help_lines.rs +++ b/apps/heft-native/src/cli/help_lines.rs @@ -1,3 +1,5 @@ +use std::borrow::Cow; + use super::text::{is_javascript_whitespace, javascript_substring}; fn is_wrap_delimiter(byte: u8) -> bool { @@ -29,11 +31,30 @@ fn normalize_help_whitespace(text: &str) -> String { normalized } +fn is_normalized_help_text(text: &str) -> bool { + let bytes: &[u8] = text.as_bytes(); + if bytes.first() == Some(&b' ') || bytes.last() == Some(&b' ') { + return false; + } + let mut previous_was_space: bool = false; + for byte in bytes { + match *byte { + b' ' if previous_was_space => return false, + b' ' => previous_was_space = true, + b'|' => return false, + other if is_javascript_whitespace(other) => return false, + _ => previous_was_space = false, + } + } + true +} + pub fn for_each_help_line(text: &str, width: f64, mut emit: impl FnMut(usize, &str)) { - let line: String = normalize_help_whitespace(text); + let normalized: Cow<'_, str> = if is_normalized_help_text(text) { Cow::Borrowed(text) } else { Cow::Owned(normalize_help_whitespace(text)) }; + let line: &str = &normalized; let length: f64 = line.len() as f64; if width >= length { - emit(0, &line); + emit(0, line); return; } let mut line_index: usize = 0; @@ -41,16 +62,16 @@ pub fn for_each_help_line(text: &str, width: f64, mut emit: impl FnMut(usize, &s let mut wrap_end: f64 = width; while wrap_end <= length { if wrap_end != length { - let segment: &str = javascript_substring(&line, wrap_start, wrap_end); + let segment: &str = javascript_substring(line, wrap_start, wrap_end); wrap_end = wrap_start + find_last_delimiter_index(segment) + 1.0; } - emit(line_index, javascript_substring(&line, wrap_start, wrap_end)); + emit(line_index, javascript_substring(line, wrap_start, wrap_end)); line_index += 1; wrap_start = wrap_end; wrap_end += width; } if wrap_start < length { - emit(line_index, javascript_substring(&line, wrap_start, wrap_end)); + emit(line_index, javascript_substring(line, wrap_start, wrap_end)); } } diff --git a/apps/heft-native/src/cli/help_model.rs b/apps/heft-native/src/cli/help_model.rs index 52dc8aa3e3..144a4cce85 100644 --- a/apps/heft-native/src/cli/help_model.rs +++ b/apps/heft-native/src/cli/help_model.rs @@ -29,14 +29,18 @@ impl<'a> HelpText<'a> { } } +const HELP_OPTION_STRINGS: [Cow<'static, str>; 2] = [Cow::Borrowed("-h"), Cow::Borrowed("--help")]; +pub const DEBUG_OPTION_STRINGS: [Cow<'static, str>; 1] = [Cow::Borrowed("--debug")]; +pub const UNMANAGED_OPTION_STRINGS: [Cow<'static, str>; 1] = [Cow::Borrowed("--unmanaged")]; + #[derive(Clone, Debug)] pub struct HelpAction<'a> { - pub option_strings: Vec>, + pub option_strings: &'a [Cow<'a, str>], pub dest: Cow<'a, str>, pub nargs: HelpNargs, pub metavar: Option<&'a str>, pub help: HelpText<'a>, - pub choices: Option>, + pub choices: Option<&'a [&'a str]>, pub required: bool, pub subactions: Vec>, } @@ -48,7 +52,7 @@ impl<'a> HelpAction<'a> { pub fn help_option() -> HelpAction<'static> { HelpAction { - option_strings: vec![Cow::Borrowed("-h"), Cow::Borrowed("--help")], + option_strings: &HELP_OPTION_STRINGS, dest: Cow::Borrowed("==SUPPRESS=="), nargs: HelpNargs::Zero, metavar: None, @@ -59,10 +63,10 @@ impl<'a> HelpAction<'a> { } } - pub fn flag_option(option_string: &'a str, help: &'a str) -> HelpAction<'a> { + pub fn flag_option(option_strings: &'a [Cow<'a, str>], help: &'a str) -> HelpAction<'a> { HelpAction { - option_strings: vec![Cow::Borrowed(option_string)], - dest: Cow::Borrowed(option_string), + option_strings, + dest: Cow::Borrowed(option_strings[0].as_ref()), nargs: HelpNargs::Zero, metavar: None, help: HelpText::Text(Cow::Borrowed(help)), @@ -72,10 +76,10 @@ impl<'a> HelpAction<'a> { } } - pub fn hidden_option(option_string: Cow<'a, str>) -> HelpAction<'a> { + pub fn hidden_option(option_strings: &'a [Cow<'a, str>]) -> HelpAction<'a> { HelpAction { - dest: option_string.clone(), - option_strings: vec![option_string], + dest: Cow::Borrowed(option_strings[0].as_ref()), + option_strings, nargs: HelpNargs::ZeroOrMore, metavar: None, help: HelpText::Suppressed, diff --git a/apps/heft-native/src/cli/help_usage.rs b/apps/heft-native/src/cli/help_usage.rs index 1af232ee37..ba5b21ec56 100644 --- a/apps/heft-native/src/cli/help_usage.rs +++ b/apps/heft-native/src/cli/help_usage.rs @@ -64,7 +64,7 @@ fn collect_wrapped_lines(parts: &[&str], indent: &str, prefix: Option<&str>, tex lines } -pub fn format_usage(prog: &str, actions: &[HelpAction<'_>], width: f64) -> Option { +pub fn push_usage(output: &mut String, prog: &str, actions: &[HelpAction<'_>], width: f64) -> Option<()> { let prefix: &str = "usage: "; let optionals: Vec<&HelpAction<'_>> = actions.iter().filter(|action| action.is_optional()).collect(); let positionals: Vec<&HelpAction<'_>> = actions.iter().filter(|action| !action.is_optional()).collect(); @@ -106,5 +106,8 @@ pub fn format_usage(prog: &str, actions: &[HelpAction<'_>], width: f64) -> Optio }; usage = lines.join("\n"); } - Some(format!("{}{}\n\n", prefix, usage)) + output.push_str(prefix); + output.push_str(&usage); + output.push_str("\n\n"); + Some(()) } diff --git a/apps/heft-native/src/cli/invocation.rs b/apps/heft-native/src/cli/invocation.rs index 8d8b100999..616ad9234c 100644 --- a/apps/heft-native/src/cli/invocation.rs +++ b/apps/heft-native/src/cli/invocation.rs @@ -1,13 +1,12 @@ use std::borrow::Cow; -use super::action_invocation::{invoke_action, phase_action_parameters, ActionRequest}; +use super::action_invocation::{invoke_action, ActionRequest}; use super::action_text::{action_summary, alias_summary}; use super::actions::{build_action_table, ActionTable, AliasEntry}; use super::help_builders::root_help_parser; use super::model::CliModel; use super::outcome::CliOutcome; -use super::parameters::ROOT_PARAMETER_NAMES; -use super::registration::try_register_parameters; +use super::phase_action_check::PhaseActionCheck; use super::render::{help_output, invalid_command_message, usage_error_output}; use super::validate::is_valid_long_name; @@ -48,7 +47,7 @@ fn is_debug_enabled(args: &[&str]) -> bool { fn root_summaries<'x>(model: &'x CliModel<'x>, table: &'x ActionTable<'x>) -> Vec<(Cow<'x, str>, Cow<'x, str>)> { let mut summaries: Vec<(Cow<'x, str>, Cow<'x, str>)> = Vec::with_capacity(table.actions.len() + table.aliases.len()); for action in &table.actions { - summaries.push((Cow::Borrowed(action.name.as_ref()), action_summary(model, action))); + summaries.push((action.name(), action_summary(model, action))); } for alias in &table.aliases { summaries.push((Cow::Borrowed(alias.name), Cow::Owned(alias_summary(table, alias)))); @@ -72,9 +71,8 @@ fn is_unknown_tool_option(arg: &str) -> bool { fn interpret<'a>(args: &'a [&'a str], model: &'a CliModel<'a>, width: Option, supports_color: Option<&dyn Fn() -> bool>) -> Option> { let table: ActionTable<'a> = build_action_table(model)?; - for phase_index in 0..model.phases.len() { - let parameters = phase_action_parameters(model, &table, phase_index, false)?; - try_register_parameters(¶meters, &ROOT_PARAMETER_NAMES)?; + if !PhaseActionCheck::can_define_every_phase_action(model, &table) { + return None; } let action_position: Option = args.iter().position(|arg| !arg.starts_with('-')); let mut has_unknown_tool_option: bool = false; @@ -82,7 +80,7 @@ fn interpret<'a>(args: &'a [&'a str], model: &'a CliModel<'a>, width: Option return print_root_help(model, &table, width), "--debug" | "--unmanaged" => {} - "--" => return root_usage_error(&invalid_command_message("--", table.command_names()), width), + "--" => return root_usage_error(&invalid_command_message("--", &table), width), _ if is_unknown_tool_option(arg) => has_unknown_tool_option = true, _ => return None, } @@ -97,7 +95,7 @@ fn interpret<'a>(args: &'a [&'a str], model: &'a CliModel<'a>, width: Option> = table.find_alias(command_name); let action_index: usize = match alias.map(|alias| alias.target_index).or_else(|| table.find_action(command_name)) { Some(action_index) => action_index, - None => return root_usage_error(&invalid_command_message(command_name, table.command_names()), width), + None => return root_usage_error(&invalid_command_message(command_name, &table), width), }; invoke_action(ActionRequest { model, diff --git a/apps/heft-native/src/cli/mod.rs b/apps/heft-native/src/cli/mod.rs index 07cc4fb0de..aa05c96578 100644 --- a/apps/heft-native/src/cli/mod.rs +++ b/apps/heft-native/src/cli/mod.rs @@ -15,6 +15,7 @@ pub mod outcome; mod parameters; mod plan_command; pub mod parse; +mod phase_action_check; mod phase_selection; mod registration; mod render; @@ -31,4 +32,6 @@ mod test_harness; #[cfg(test)] mod tests_cli; #[cfg(test)] +mod tests_registration; +#[cfg(test)] mod test_model; diff --git a/apps/heft-native/src/cli/parameters.rs b/apps/heft-native/src/cli/parameters.rs index 986317a7b2..2142f6881b 100644 --- a/apps/heft-native/src/cli/parameters.rs +++ b/apps/heft-native/src/cli/parameters.rs @@ -1,6 +1,6 @@ use std::borrow::Cow; -use super::defined_parameter::DefinedParameter; +use super::defined_parameter::{ChoiceAlternatives, DefinedParameter}; use super::model::{CliModel, ParameterKind, PluginParameterDefinition}; use super::validate::is_valid_definition; @@ -41,14 +41,18 @@ fn push_unique(indices: &mut Vec, index: usize) { } } -fn unique_alternatives<'a>(alternatives: &[&'a str]) -> Vec<&'a str> { +fn unique_alternatives<'a>(alternatives: &'a [&'a str]) -> ChoiceAlternatives<'a> { + let has_duplicates: bool = alternatives.iter().enumerate().any(|(index, alternative)| alternatives[..index].contains(alternative)); + if !has_duplicates { + return ChoiceAlternatives::Declared(alternatives); + } let mut unique: Vec<&'a str> = Vec::with_capacity(alternatives.len()); for alternative in alternatives { if !unique.contains(alternative) { unique.push(alternative); } } - unique + ChoiceAlternatives::Deduplicated(unique) } fn define_plugin_parameter<'a>(definition: &'a PluginParameterDefinition<'a>, scope: &'a str) -> DefinedParameter<'a> { @@ -63,7 +67,7 @@ fn define_plugin_parameter<'a>(definition: &'a PluginParameterDefinition<'a>, sc scoping_group: false, required: definition.required, argument_name: if takes_argument_name { definition.argument_name } else { None }, - alternatives: if kind.has_alternatives() { unique_alternatives(&definition.alternatives) } else { Vec::new() }, + alternatives: if kind.has_alternatives() { unique_alternatives(&definition.alternatives) } else { ChoiceAlternatives::Declared(&[]) }, default_value: if takes_default { definition.default_value } else { None }, description: Cow::Borrowed(definition.description), } @@ -74,17 +78,27 @@ pub fn push_plugin_parameters<'a>( model: &'a CliModel<'a>, selected_phases: &[usize], ) -> Option<()> { - let mut plugin_indices: Vec = Vec::new(); + collect_plugin_parameters(parameters, model, selected_phases, &mut Vec::new(), &mut Vec::new()) +} + +pub fn collect_plugin_parameters<'a>( + parameters: &mut Vec>, + model: &'a CliModel<'a>, + selected_phases: &[usize], + plugin_indices: &mut Vec, + plugins_by_scope: &mut Vec<(&'a str, usize)>, +) -> Option<()> { + plugin_indices.clear(); + plugins_by_scope.clear(); for plugin_index in &model.lifecycle_plugin_indices { - push_unique(&mut plugin_indices, *plugin_index); + push_unique(plugin_indices, *plugin_index); } for phase_index in selected_phases { for plugin_index in &model.phases[*phase_index].task_plugin_indices { - push_unique(&mut plugin_indices, *plugin_index); + push_unique(plugin_indices, *plugin_index); } } - let mut plugins_by_scope: Vec<(&str, usize)> = Vec::with_capacity(plugin_indices.len()); - for plugin_index in plugin_indices { + for plugin_index in plugin_indices.iter().copied() { let plugin = model.plugins.get(plugin_index)?; match plugins_by_scope.iter().find(|(scope, _)| *scope == plugin.parameter_scope) { Some((_, existing_index)) if *existing_index != plugin_index => return None, diff --git a/apps/heft-native/src/cli/phase_action_check.rs b/apps/heft-native/src/cli/phase_action_check.rs new file mode 100644 index 0000000000..7c626aa9f7 --- /dev/null +++ b/apps/heft-native/src/cli/phase_action_check.rs @@ -0,0 +1,29 @@ +use super::actions::{select_phases_into, ActionTable}; +use super::defined_parameter::DefinedParameter; +use super::model::CliModel; +use super::parameters::{collect_plugin_parameters, push_builtin_parameters, ROOT_PARAMETER_NAMES}; +use super::registration::is_registration_possible; + +#[derive(Default)] +pub struct PhaseActionCheck<'a> { + parameters: Vec>, + selected_phases: Vec, + plugin_indices: Vec, + plugins_by_scope: Vec<(&'a str, usize)>, +} + +impl<'a> PhaseActionCheck<'a> { + pub fn can_define_every_phase_action(model: &'a CliModel<'a>, table: &ActionTable<'a>) -> bool { + let mut check: PhaseActionCheck<'a> = PhaseActionCheck::default(); + (0..model.phases.len()).all(|phase_index| check.can_define_phase_action(model, table, phase_index)) + } + + fn can_define_phase_action(&mut self, model: &'a CliModel<'a>, table: &ActionTable<'a>, phase_index: usize) -> bool { + self.parameters.clear(); + push_builtin_parameters(&mut self.parameters, false); + select_phases_into(table, [phase_index], &mut self.selected_phases); + let parameters = &mut self.parameters; + collect_plugin_parameters(parameters, model, &self.selected_phases, &mut self.plugin_indices, &mut self.plugins_by_scope).is_some() + && is_registration_possible(parameters, &ROOT_PARAMETER_NAMES) + } +} diff --git a/apps/heft-native/src/cli/registration.rs b/apps/heft-native/src/cli/registration.rs index bafe2a4255..6a510ccedd 100644 --- a/apps/heft-native/src/cli/registration.rs +++ b/apps/heft-native/src/cli/registration.rs @@ -88,13 +88,6 @@ fn count_long_name(parameters: &[DefinedParameter<'_>], long_name: &str) -> usiz parameters.iter().filter(|parameter| parameter.long_name == long_name).count() } -fn long_name_order<'p>(parameters: &'p [DefinedParameter<'_>]) -> impl Iterator + 'p { - let first_index_of = |long_name: &str| parameters.iter().position(|parameter| parameter.long_name == long_name); - (0..parameters.len()) - .filter(move |index| first_index_of(parameters[*index].long_name) == Some(*index)) - .flat_map(move |first| (first..parameters.len()).filter(move |index| parameters[*index].long_name == parameters[first].long_name)) -} - pub fn try_register_parameters<'a>(parameters: &[DefinedParameter<'a>], parent_names: &[Cow<'a, str>]) -> Option> { let mut registration: Registration<'a> = Registration::default(); let mut ambiguous_names: Vec> = Vec::new(); @@ -103,7 +96,9 @@ pub fn try_register_parameters<'a>(parameters: &[DefinedParameter<'a>], parent_n push_unique(&mut ambiguous_names, Cow::Borrowed(short_name)); } } - for parameter_index in long_name_order(parameters) { + let groups = (0..parameters.len()).filter(|first| parameters.iter().position(|other| other.long_name == parameters[*first].long_name) == Some(*first)); + let long_name_order = groups.flat_map(|first| (first..parameters.len()).filter(move |index| parameters[*index].long_name == parameters[first].long_name)); + for parameter_index in long_name_order { let parameter: &DefinedParameter<'a> = ¶meters[parameter_index]; let use_scoped_long_name: bool = count_long_name(parameters, parameter.long_name) > 1; if use_scoped_long_name { @@ -139,3 +134,22 @@ pub fn try_register_parameters<'a>(parameters: &[DefinedParameter<'a>], parent_n } Some(registration) } + +pub fn is_registration_possible(parameters: &[DefinedParameter<'_>], parent_names: &[Cow<'_, str>]) -> bool { + let is_help_option = |name: &str| HELP_OPTION_STRINGS.contains(&name); + for (index, parameter) in parameters.iter().enumerate() { + if parameter.scope.is_none() && count_long_name(parameters, parameter.long_name) > 1 { + return false; + } + if parameter.short_name.is_some_and(is_help_option) || is_help_option(parameter.long_name) { + return false; + } + if let Some(scope) = parameter.scope { + let earlier = ¶meters[..index]; + if earlier.iter().any(|other| other.scope == Some(scope) && other.long_name == parameter.long_name) { + return false; + } + } + } + !parent_names.iter().any(|name| is_help_option(name)) +} diff --git a/apps/heft-native/src/cli/render.rs b/apps/heft-native/src/cli/render.rs index 0727aae3f2..552fd12386 100644 --- a/apps/heft-native/src/cli/render.rs +++ b/apps/heft-native/src/cli/render.rs @@ -1,3 +1,4 @@ +use super::actions::ActionTable; use super::defined_parameter::DefinedParameter; use super::help_format::{format_help, format_usage_only}; use super::help_model::HelpParser; @@ -18,9 +19,11 @@ pub fn usage_error_output(parser: &HelpParser<'_>, width: Option, message: Some(PrintedOutput { stdout, stderr, exit_code: 1 }) } -pub fn invalid_command_message<'x>(value: &str, command_names: impl Iterator) -> String { - let choices: Vec<&str> = command_names.collect(); - format!("argument \"\": Invalid choice: {} (choose from [{}])", value, choices.join(", ")) +pub fn invalid_command_message(value: &str, table: &ActionTable<'_>) -> String { + let mut message: String = format!("argument \"\": Invalid choice: {value} (choose from ["); + table.push_command_names(&mut message); + message.push_str("])"); + message } pub fn argument_error_message(error: &ArgumentError<'_>, registration: &Registration, parameters: &[DefinedParameter<'_>]) -> Option { diff --git a/apps/heft-native/src/cli/run_invocation.rs b/apps/heft-native/src/cli/run_invocation.rs index 84e6e5ac9d..1d6c00ecbe 100644 --- a/apps/heft-native/src/cli/run_invocation.rs +++ b/apps/heft-native/src/cli/run_invocation.rs @@ -15,7 +15,7 @@ use super::render::{argument_error_message, help_output, remainder_error_output, pub fn invoke_run<'a>(request: &ActionRequest<'a, '_>, action_args: &[&'a str]) -> Option> { let action = &request.table.actions[request.action_index]; let mut definitions: Vec> = Vec::new(); - push_scoping_parameters(&mut definitions, &action.name); + push_scoping_parameters(&mut definitions, &action.name()); let registration: Registration = try_register_parameters(&definitions, &ROOT_PARAMETER_NAMES)?; let (values, remainder_start) = match parse_arguments(®istration, &definitions, action_args, true) { ParseOutcome::Help => return print_help(request, ®istration, &definitions, true), @@ -41,7 +41,7 @@ pub fn invoke_run<'a>(request: &ActionRequest<'a, '_>, action_args: &[&'a str]) } let supports_color: bool = (request.supports_color?)(); let parser = action_help_parser(®istration, &definitions, help_text(request, None), true)?; - let output = remainder_error_output(&parser, request.width, &action.name, action_args[start], supports_color)?; + let output = remainder_error_output(&parser, request.width, &action.name(), action_args[start], supports_color)?; return Some(CliOutcome::Print(output)); } let mut scoped: Vec> = Vec::new(); @@ -90,7 +90,7 @@ fn scoped_help_text<'a>(request: &ActionRequest<'a, '_>, definitions: &[DefinedP scope.push(name); } } - let epilog: String = bold(&format!("For more information on available unscoped parameters, use \"heft {} --help\"", action.name)); + let epilog: String = bold(&format!("For more information on available unscoped parameters, use \"heft {} --help\"", action.name())); let (prog, description, banner) = match request.alias { Some(alias) => ( format!("heft {}", request.command_name), diff --git a/apps/heft-native/src/cli/tests_registration.rs b/apps/heft-native/src/cli/tests_registration.rs new file mode 100644 index 0000000000..634c63d83e --- /dev/null +++ b/apps/heft-native/src/cli/tests_registration.rs @@ -0,0 +1,69 @@ +use std::borrow::Cow; + +use super::defined_parameter::DefinedParameter; +use super::help_args::{clean_usage_separators, clean_usage_separators_in_every_pass}; +use super::model::ParameterKind; +use super::registration::{is_registration_possible, try_register_parameters}; + +struct PseudoRandom(u64); + +impl PseudoRandom { + fn next(&mut self, bound: usize) -> usize { + self.0 ^= self.0 << 13; + self.0 ^= self.0 >> 7; + self.0 ^= self.0 << 17; + (self.0 % bound as u64) as usize + } +} + +const LONG_NAMES: [&str; 5] = ["--alpha", "--beta", "--help", "--verbose", "--gamma-delta"]; +const SHORT_NAMES: [&str; 4] = ["-a", "-b", "-h", "-v"]; +const SCOPES: [&str; 3] = ["one", "two", "lint"]; +const PARENT_NAMES: [&str; 4] = ["--debug", "--unmanaged", "--alpha", "-h"]; + +fn random_parameter(random: &mut PseudoRandom) -> DefinedParameter<'static> { + let short_name: Option<&'static str> = match random.next(3) { + 0 => Some(SHORT_NAMES[random.next(SHORT_NAMES.len())]), + _ => None, + }; + let mut parameter: DefinedParameter<'static> = DefinedParameter::flag(LONG_NAMES[random.next(LONG_NAMES.len())], short_name, Cow::Borrowed("d")); + if random.next(4) > 0 { + parameter.scope = Some(SCOPES[random.next(SCOPES.len())]); + } + if random.next(2) == 0 { + parameter.kind = ParameterKind::StringList; + parameter.argument_name = Some("VALUE"); + } + parameter +} + +#[test] +fn registration_check_agrees_with_full_registration() { + let mut random: PseudoRandom = PseudoRandom(0x9e37_79b9_7f4a_7c15); + let mut disagreements: usize = 0; + let mut possible: usize = 0; + for _ in 0..200_000 { + let count: usize = random.next(7); + let parameters: Vec> = (0..count).map(|_| random_parameter(&mut random)).collect(); + let parents: Vec> = (0..random.next(3)).map(|_| Cow::Borrowed(PARENT_NAMES[random.next(PARENT_NAMES.len())])).collect(); + let expected: bool = try_register_parameters(¶meters, &parents).is_some(); + if expected { + possible += 1; + } + if is_registration_possible(¶meters, &parents) != expected { + disagreements += 1; + } + } + assert_eq!(disagreements, 0); + assert!(possible > 20_000 && possible < 180_000, "{possible}"); +} + +#[test] +fn usage_cleanup_fast_path_agrees_with_every_pass() { + let mut random: PseudoRandom = PseudoRandom(0x2545_f491_4f6c_dd1d); + let alphabet: [char; 9] = ['[', ']', '(', ')', ' ', ' ', 'a', '|', '-']; + for _ in 0..300_000 { + let text: String = (0..random.next(14)).map(|_| alphabet[random.next(alphabet.len())]).collect(); + assert_eq!(clean_usage_separators(&text), clean_usage_separators_in_every_pass(&text), "{text:?}"); + } +} diff --git a/apps/heft-native/src/config/embedded_schemas.rs b/apps/heft-native/src/config/embedded_schemas.rs index 0ec5fa541a..9527bd0d91 100644 --- a/apps/heft-native/src/config/embedded_schemas.rs +++ b/apps/heft-native/src/config/embedded_schemas.rs @@ -1,9 +1,9 @@ use super::fallback::{fallback, ConfigResult}; use crate::json::{parse_json_with_comments_exactly_like_jju, JsonValue}; -pub const HEFT_JSON_SCHEMA_TEXT: &str = include_str!("../../../heft/src/schemas/heft.schema.json"); +pub const HEFT_JSON_SCHEMA_TEXT: &str = include_str!("heft_schema_without_annotations.json"); pub const HEFT_PLUGIN_JSON_SCHEMA_TEXT: &str = - include_str!("../../../heft/src/schemas/heft-plugin.schema.json"); + include_str!("heft_plugin_schema_without_annotations.json"); pub fn parse_embedded_schema(schema_text: &'static str) -> ConfigResult> { match parse_json_with_comments_exactly_like_jju(schema_text) { diff --git a/apps/heft-native/src/config/fs_probe.rs b/apps/heft-native/src/config/fs_probe.rs index 319776ca90..2650ac23e6 100644 --- a/apps/heft-native/src/config/fs_probe.rs +++ b/apps/heft-native/src/config/fs_probe.rs @@ -1,90 +1,65 @@ -use std::collections::HashMap; -use std::fs; -use std::io; +use std::fs::File; +use std::io::{self, Read}; use super::fallback::{fallback, ConfigResult}; -use super::real_path_resolver::RealPathResolver; +use super::path_component_cache::PathComponentCache; +use super::path_probes::is_missing_entry_error; -#[derive(Clone, Copy, PartialEq, Eq)] -pub enum EntryKind { - File, - Directory, - Other, -} - -fn is_missing_entry_error(error: &io::Error) -> bool { - matches!( - error.kind(), - io::ErrorKind::NotFound | io::ErrorKind::NotADirectory - ) || matches!(error.raw_os_error(), Some(2) | Some(20)) -} - -pub fn stat_entry_kind(path: &str) -> ConfigResult> { - match fs::metadata(path) { - Ok(metadata) => { - let file_type: fs::FileType = metadata.file_type(); - if file_type.is_dir() { - Ok(Some(EntryKind::Directory)) - } else if file_type.is_file() || is_fifo(&file_type) { - Ok(Some(EntryKind::File)) - } else { - Ok(Some(EntryKind::Other)) - } - } - Err(error) if is_missing_entry_error(&error) => Ok(None), - Err(_) => fallback("stat failed with an unexpected error"), - } -} - -#[cfg(unix)] -fn is_fifo(file_type: &fs::FileType) -> bool { - use std::os::unix::fs::FileTypeExt; - file_type.is_fifo() -} +pub use super::path_probes::{EntryKind, ResolvedEntry, StatEntry}; -#[cfg(not(unix))] -fn is_fifo(_file_type: &fs::FileType) -> bool { - false -} +const LARGEST_SIZE_HINT_TRUSTED_FOR_ONE_READ: u64 = 1 << 26; +const INITIAL_BUFFER_LENGTH_FOR_UNKNOWN_SIZES: usize = 4096; -pub fn is_file_like_resolve(path: &str) -> ConfigResult { - Ok(stat_entry_kind(path)? == Some(EntryKind::File)) +#[derive(Default)] +pub struct FileSystemProbeCache { + paths: PathComponentCache, } -pub fn is_directory_like_resolve(path: &str) -> ConfigResult { - Ok(stat_entry_kind(path)? == Some(EntryKind::Directory)) +fn length_to_expect(size: u64) -> usize { + if size > LARGEST_SIZE_HINT_TRUSTED_FOR_ONE_READ { + 0 + } else { + size as usize + } } -pub fn exists_like_exists_sync(path: &str) -> bool { - fs::metadata(path).is_ok() +fn read_all_bytes_expecting_length(file: &mut File, expected_length: usize) -> io::Result> { + let initial_length: usize = if expected_length == 0 { + INITIAL_BUFFER_LENGTH_FOR_UNKNOWN_SIZES + } else { + expected_length + 1 + }; + let mut buffer: Vec = vec![0; initial_length]; + let mut filled: usize = 0; + loop { + if filled == buffer.len() { + buffer.resize(buffer.len() * 2, 0); + } + let read: usize = match file.read(&mut buffer[filled..]) { + Ok(read) => read, + Err(error) if error.kind() == io::ErrorKind::Interrupted => continue, + Err(error) => return Err(error), + }; + filled += read; + if read == 0 || (filled == expected_length && expected_length > 0) { + break; + } + } + buffer.truncate(filled); + Ok(buffer) } -pub fn read_text_or_missing(path: &str) -> ConfigResult> { - match fs::read(path) { - Ok(bytes) => match String::from_utf8(bytes) { - Ok(text) => Ok(Some(text)), - Err(_) => fallback("a file is not valid UTF-8"), - }, - Err(error) if is_missing_entry_error(&error) => Ok(None), - Err(_) => fallback("reading a file failed with an unexpected error"), +impl FileSystemProbeCache { + pub fn remember_physical_directory_path(&mut self, directory_path: &str) { + self.paths.remember_physical_directory_path(directory_path); } -} -#[derive(Default)] -pub struct FileSystemProbeCache { - real_paths: HashMap>, - entry_kinds: HashMap>, - resolver: RealPathResolver, -} + pub fn resolve(&mut self, path: &str) -> ConfigResult> { + self.paths.resolve(path) + } -impl FileSystemProbeCache { pub fn real_path_or_missing(&mut self, path: &str) -> ConfigResult> { - if let Some(cached) = self.real_paths.get(path) { - return Ok(cached.clone()); - } - let real_path: Option = self.resolver.resolve_real_path(path)?; - self.real_paths.insert(path.to_string(), real_path.clone()); - Ok(real_path) + Ok(self.resolve(path)?.map(|entry| entry.real_path)) } pub fn real_path(&mut self, path: &str) -> ConfigResult { @@ -95,12 +70,7 @@ impl FileSystemProbeCache { } fn entry_kind(&mut self, path: &str) -> ConfigResult> { - if let Some(cached) = self.entry_kinds.get(path) { - return Ok(*cached); - } - let entry_kind: Option = stat_entry_kind(path)?; - self.entry_kinds.insert(path.to_string(), entry_kind); - Ok(entry_kind) + Ok(self.resolve(path)?.map(|entry| entry.kind)) } pub fn is_file_like_resolve(&mut self, path: &str) -> ConfigResult { @@ -110,4 +80,40 @@ impl FileSystemProbeCache { pub fn is_directory_like_resolve(&mut self, path: &str) -> ConfigResult { Ok(self.entry_kind(path)? == Some(EntryKind::Directory)) } + + pub fn exists_like_exists_sync(&mut self, path: &str) -> ConfigResult { + Ok(self.paths.stat(path)?.is_some()) + } + + pub fn read_text_or_missing(&mut self, path: &str) -> ConfigResult> { + let known_length: Option = match self.paths.cached_stat(path) { + Some(None) => return Ok(None), + Some(Some(entry)) if entry.kind != EntryKind::File => { + return fallback("a configuration path is not a file") + } + Some(Some(entry)) => Some(length_to_expect(entry.size)), + None => None, + }; + let mut file: File = match File::open(path) { + Ok(file) => file, + Err(error) if is_missing_entry_error(&error) => return Ok(None), + Err(_) => return fallback("opening a file failed with an unexpected error"), + }; + let expected_length: usize = match known_length { + Some(expected_length) => expected_length, + None => match file.metadata() { + Ok(metadata) if metadata.is_file() => length_to_expect(metadata.len()), + Ok(_) => return fallback("a configuration path is not a regular file"), + Err(_) => return fallback("fstat failed with an unexpected error"), + }, + }; + let bytes: Vec = match read_all_bytes_expecting_length(&mut file, expected_length) { + Ok(bytes) => bytes, + Err(_) => return fallback("reading a file failed with an unexpected error"), + }; + match String::from_utf8(bytes) { + Ok(text) => Ok(Some(text)), + Err(_) => fallback("a file is not valid UTF-8"), + } + } } diff --git a/apps/heft-native/src/config/heft_json_chain.rs b/apps/heft-native/src/config/heft_json_chain.rs index ae1f5ff6eb..e9a4885a9c 100644 --- a/apps/heft-native/src/config/heft_json_chain.rs +++ b/apps/heft-native/src/config/heft_json_chain.rs @@ -1,5 +1,5 @@ use super::fallback::{fallback, ConfigResult}; -use super::fs_probe::{read_text_or_missing, FileSystemProbeCache}; +use super::fs_probe::FileSystemProbeCache; use super::node_path::{dirname, resolve}; use super::node_resolve::resolve_module; use super::package_json::PackageJsonLookup; @@ -79,7 +79,7 @@ impl HeftJsonChain { return fallback("a loop in the extends chain"); } visited.push(path.to_string()); - let text: String = match read_text_or_missing(path)? { + let text: String = match file_system.read_text_or_missing(path)? { Some(text) => text, None => return Ok(None), }; diff --git a/apps/heft-native/src/config/heft_plugin_schema_without_annotations.json b/apps/heft-native/src/config/heft_plugin_schema_without_annotations.json new file mode 100644 index 0000000000..d8a8c1f4bc --- /dev/null +++ b/apps/heft-native/src/config/heft_plugin_schema_without_annotations.json @@ -0,0 +1 @@ +{"$schema":"http://json-schema.org/draft-04/schema#","type":"object","definitions":{"anything":{"type":["array","boolean","integer","number","object","string"],"items":{"$ref":"#/definitions/anything"}},"baseParameter":{"type":"object","additionalProperties":true,"required":["parameterKind","longName","description"],"properties":{"parameterKind":{"type":"string","enum":["choice","choiceList","flag","integer","integerList","string","stringList"]},"longName":{"type":"string","pattern":"^-(-[a-z0-9]+)+$"},"shortName":{"type":"string","pattern":"^-[a-zA-Z]$"},"description":{"type":"string"},"required":{"type":"boolean"}}},"choiceParameter":{"type":"object","allOf":[{"$ref":"#/definitions/baseParameter"},{"type":"object","additionalProperties":true,"required":["alternatives"],"properties":{"parameterKind":{"enum":["choice"]},"alternatives":{"type":"array","minItems":1,"items":{"type":"object","additionalProperties":false,"required":["name","description"],"properties":{"name":{"type":"string"},"description":{"type":"string"}}}},"defaultValue":{"type":"string"}}},{"type":"object","additionalProperties":false,"properties":{"parameterKind":{"$ref":"#/definitions/anything"},"longName":{"$ref":"#/definitions/anything"},"shortName":{"$ref":"#/definitions/anything"},"description":{"$ref":"#/definitions/anything"},"required":{"$ref":"#/definitions/anything"},"alternatives":{"$ref":"#/definitions/anything"},"defaultValue":{"$ref":"#/definitions/anything"}}}]},"choiceListParameter":{"type":"object","allOf":[{"$ref":"#/definitions/baseParameter"},{"type":"object","additionalProperties":true,"required":["alternatives"],"properties":{"parameterKind":{"enum":["choiceList"]},"alternatives":{"type":"array","minItems":1,"items":{"type":"object","additionalProperties":false,"required":["name","description"],"properties":{"name":{"type":"string"},"description":{"type":"string"}}}}}},{"type":"object","additionalProperties":false,"properties":{"parameterKind":{"$ref":"#/definitions/anything"},"longName":{"$ref":"#/definitions/anything"},"shortName":{"$ref":"#/definitions/anything"},"description":{"$ref":"#/definitions/anything"},"required":{"$ref":"#/definitions/anything"},"alternatives":{"$ref":"#/definitions/anything"}}}]},"flagParameter":{"type":"object","allOf":[{"$ref":"#/definitions/baseParameter"},{"type":"object","additionalProperties":true,"properties":{"parameterKind":{"enum":["flag"]}}},{"type":"object","additionalProperties":false,"properties":{"parameterKind":{"$ref":"#/definitions/anything"},"longName":{"$ref":"#/definitions/anything"},"shortName":{"$ref":"#/definitions/anything"},"description":{"$ref":"#/definitions/anything"},"required":{"$ref":"#/definitions/anything"}}}]},"integerParameter":{"type":"object","allOf":[{"$ref":"#/definitions/baseParameter"},{"type":"object","additionalProperties":true,"required":["argumentName"],"properties":{"parameterKind":{"enum":["integer"]},"argumentName":{"type":"string"},"defaultValue":{"type":"integer"}}},{"type":"object","additionalProperties":false,"properties":{"parameterKind":{"$ref":"#/definitions/anything"},"longName":{"$ref":"#/definitions/anything"},"shortName":{"$ref":"#/definitions/anything"},"description":{"$ref":"#/definitions/anything"},"required":{"$ref":"#/definitions/anything"},"argumentName":{"$ref":"#/definitions/anything"},"defaultValue":{"$ref":"#/definitions/anything"}}}]},"integerListParameter":{"type":"object","allOf":[{"$ref":"#/definitions/baseParameter"},{"type":"object","additionalProperties":true,"required":["argumentName"],"properties":{"parameterKind":{"enum":["integerList"]},"argumentName":{"type":"string"}}},{"type":"object","additionalProperties":false,"properties":{"parameterKind":{"$ref":"#/definitions/anything"},"longName":{"$ref":"#/definitions/anything"},"shortName":{"$ref":"#/definitions/anything"},"description":{"$ref":"#/definitions/anything"},"required":{"$ref":"#/definitions/anything"},"argumentName":{"$ref":"#/definitions/anything"}}}]},"stringParameter":{"type":"object","allOf":[{"$ref":"#/definitions/baseParameter"},{"type":"object","additionalProperties":true,"required":["argumentName"],"properties":{"parameterKind":{"enum":["string"]},"argumentName":{"type":"string"},"defaultValue":{"type":"string"}}},{"type":"object","additionalProperties":false,"properties":{"parameterKind":{"$ref":"#/definitions/anything"},"longName":{"$ref":"#/definitions/anything"},"shortName":{"$ref":"#/definitions/anything"},"description":{"$ref":"#/definitions/anything"},"required":{"$ref":"#/definitions/anything"},"argumentName":{"$ref":"#/definitions/anything"},"defaultValue":{"$ref":"#/definitions/anything"}}}]},"stringListParameter":{"type":"object","allOf":[{"$ref":"#/definitions/baseParameter"},{"type":"object","additionalProperties":true,"required":["argumentName"],"properties":{"parameterKind":{"enum":["stringList"]},"argumentName":{"type":"string"}}},{"type":"object","additionalProperties":false,"properties":{"parameterKind":{"$ref":"#/definitions/anything"},"longName":{"$ref":"#/definitions/anything"},"shortName":{"$ref":"#/definitions/anything"},"description":{"$ref":"#/definitions/anything"},"required":{"$ref":"#/definitions/anything"},"argumentName":{"$ref":"#/definitions/anything"}}}]},"heft-plugin-base":{"type":"object","additionalProperties":false,"required":["pluginName","entryPoint"],"properties":{"pluginName":{"type":"string","pattern":"^[a-z][a-z0-9]*([-][a-z0-9]+)*$"},"entryPoint":{"type":"string","pattern":"[^\\\\]"},"optionsSchema":{"type":"string","pattern":"[^\\\\]"},"parameterScope":{"type":"string","pattern":"^[a-z][a-z0-9]*([-][a-z0-9]+)*$"},"parameters":{"type":"array","items":{"type":"object","oneOf":[{"$ref":"#/definitions/flagParameter"},{"$ref":"#/definitions/integerParameter"},{"$ref":"#/definitions/integerListParameter"},{"$ref":"#/definitions/choiceParameter"},{"$ref":"#/definitions/choiceListParameter"},{"$ref":"#/definitions/stringParameter"},{"$ref":"#/definitions/stringListParameter"}]}}}},"heft-lifecycle-plugin":{"$ref":"#/definitions/heft-plugin-base"},"heft-task-plugin":{"$ref":"#/definitions/heft-plugin-base"}},"additionalProperties":false,"properties":{"$schema":{"type":"string"},"lifecyclePlugins":{"type":"array","items":{"$ref":"#/definitions/heft-lifecycle-plugin"}},"taskPlugins":{"type":"array","items":{"$ref":"#/definitions/heft-task-plugin"}}}} \ No newline at end of file diff --git a/apps/heft-native/src/config/heft_schema_without_annotations.json b/apps/heft-native/src/config/heft_schema_without_annotations.json new file mode 100644 index 0000000000..70c8e3c559 --- /dev/null +++ b/apps/heft-native/src/config/heft_schema_without_annotations.json @@ -0,0 +1 @@ +{"$schema":"http://json-schema.org/draft-04/schema#","type":"object","definitions":{"anything":{"type":["array","boolean","integer","number","object","string"],"items":{"$ref":"#/definitions/anything"}},"heft-plugin":{"type":"object","required":["pluginPackage"],"additionalProperties":false,"properties":{"pluginPackage":{"type":"string","pattern":"[^\\\\]"},"pluginName":{"type":"string","pattern":"^[a-z][a-z0-9]*([-][a-z0-9]+)*$"},"options":{"type":"object"}}},"heft-event":{"type":"object","required":["eventKind"],"additionalProperties":false,"properties":{"eventKind":{"type":"string","enum":["copyFiles","deleteFiles","runScript","nodeService"]},"options":{"type":"object"}}},"delete-operation":{"type":"object","additionalProperties":false,"anyOf":[{"required":["sourcePath"]},{"required":["fileExtensions"]},{"required":["includeGlobs"]},{"required":["excludeGlobs"]}],"properties":{"sourcePath":{"type":"string","pattern":"[^\\\\]"},"fileExtensions":{"type":"array","items":{"type":"string","pattern":"^\\.[A-z0-9-_.]*[A-z0-9-_]+$"}},"excludeGlobs":{"type":"array","items":{"type":"string","pattern":"[^\\\\]"}},"includeGlobs":{"type":"array","items":{"type":"string","pattern":"[^\\\\]"}}}}},"additionalProperties":false,"properties":{"$schema":{"type":"string"},"extends":{"type":"string"},"heftPlugins":{"type":"array","items":{"$ref":"#/definitions/heft-plugin"}},"aliasesByName":{"type":"object","additionalProperties":false,"patternProperties":{"^[a-z][a-z0-9]*([-][a-z0-9]+)*$":{"type":"object","additionalProperties":false,"required":["actionName"],"properties":{"actionName":{"type":"string","pattern":"^[a-z][a-z0-9]*([-][a-z0-9]+)*$"},"defaultParameters":{"type":"array","items":{"type":"string"}}}}}},"phasesByName":{"type":"object","additionalProperties":false,"patternProperties":{"^[a-z][a-z0-9]*([-][a-z0-9]+)*$":{"type":"object","additionalProperties":false,"properties":{"phaseDescription":{"type":"string"},"phaseDependencies":{"type":"array","items":{"type":"string","pattern":"^[a-z][a-z0-9]*([-][a-z0-9]+)*$"}},"cleanFiles":{"type":"array","items":{"$ref":"#/definitions/delete-operation"}},"tasksByName":{"type":"object","additionalProperties":false,"patternProperties":{"^[a-z][a-z0-9]*([-][a-z0-9]+)*$":{"type":"object","additionalProperties":false,"required":["taskPlugin"],"properties":{"taskPlugin":{"$ref":"#/definitions/heft-plugin"},"taskDependencies":{"type":"array","items":{"type":"string","pattern":"^[a-z][a-z0-9]*([-][a-z0-9]+)*$"}}}}}}}}}}}} \ No newline at end of file diff --git a/apps/heft-native/src/config/javascript_order.rs b/apps/heft-native/src/config/javascript_order.rs index 5337d93ea6..e1e33643eb 100644 --- a/apps/heft-native/src/config/javascript_order.rs +++ b/apps/heft-native/src/config/javascript_order.rs @@ -33,14 +33,24 @@ pub fn array_index_of_key(key: &str) -> Option { } } -pub fn order_entries_like_javascript(entries: &mut [(Cow<'_, str>, T)]) { - if entries +pub fn order_entries_like_javascript(entries: &mut Vec<(Cow<'_, str>, T)>) { + if !entries .iter() .any(|(key, _)| array_index_of_key(key).is_some()) { - entries.sort_by_key(|(key, _)| match array_index_of_key(key) { - Some(index) => (0u8, index), - None => (1u8, 0), - }); + return; } + let mut index_entries: Vec<(u32, (Cow<'_, str>, T))> = Vec::new(); + let mut named_entries: Vec<(Cow<'_, str>, T)> = Vec::with_capacity(entries.len()); + for entry in entries.drain(..) { + match array_index_of_key(&entry.0) { + Some(index) => { + let position: usize = index_entries.partition_point(|(other, _)| *other <= index); + index_entries.insert(position, (index, entry)); + } + None => named_entries.push(entry), + } + } + entries.extend(index_entries.into_iter().map(|(_, entry)| entry)); + entries.extend(named_entries); } diff --git a/apps/heft-native/src/config/loader.rs b/apps/heft-native/src/config/loader.rs index dedef80b77..56072d14f2 100644 --- a/apps/heft-native/src/config/loader.rs +++ b/apps/heft-native/src/config/loader.rs @@ -2,6 +2,7 @@ use super::embedded_schemas::{ parse_embedded_schema, HEFT_JSON_SCHEMA_TEXT, HEFT_PLUGIN_JSON_SCHEMA_TEXT, }; use super::fallback::{fallback, ConfigResult}; +use super::fs_probe::FileSystemProbeCache; use super::heft_json_chain::{discover_heft_json_chain, HeftJsonChain}; use super::heft_json_merge::{merge_heft_json_chain, PluginPackageResolver}; use super::normalize::normalize_heft_configuration; @@ -57,6 +58,7 @@ fn validate_merged_heft_json(tree: &ConfigTree, merged: NodeId) -> ConfigResult< } fn read_plugin_package_manifests( + file_system: &mut FileSystemProbeCache, references: &PluginReferences, ) -> ConfigResult> { let mut manifests: Vec = Vec::new(); @@ -66,6 +68,7 @@ fn read_plugin_package_manifests( .any(|manifest| manifest.package_root == reference.package_root) { manifests.push(read_plugin_package_manifest( + file_system, reference.package_root, reference.package_name, )?); @@ -93,12 +96,13 @@ pub fn load_heft_configuration_and_then( if cfg!(not(unix)) { return fallback("native configuration loading implements POSIX paths only"); } - let rig: RigConfigData = load_rig_config_data(request.build_folder_path)?; + let rig: RigConfigData = + load_rig_config_data(&mut lookup.file_system, request.build_folder_path)?; let heft_json_chain: HeftJsonChain = discover_heft_json_chain(lookup, request.build_folder_path, &rig)?; let mut tree: ConfigTree = ConfigTree::default(); let mut resolver: PluginPackageResolver = PluginPackageResolver { - lookup, + lookup: &mut *lookup, heft_module_folder: request.heft_module_folder, heft_package_folder: None, }; @@ -112,11 +116,18 @@ pub fn load_heft_configuration_and_then( let heft_json: NodeId = normalize_heft_configuration(&mut tree, merged)?; let tree: ConfigTree = tree; let references: PluginReferences = collect_plugin_references(&tree, heft_json)?; - let manifests: Vec = read_plugin_package_manifests(&references)?; + let manifests: Vec = + read_plugin_package_manifests(&mut lookup.file_system, &references)?; let parsed_manifests: Vec = parse_plugin_package_manifests(&manifests)?; let mut definitions: Vec = Vec::new(); for (package, (manifest, parsed)) in manifests.iter().zip(&parsed_manifests).enumerate() { - load_plugin_definitions(package, manifest, parsed, &mut definitions)?; + load_plugin_definitions( + &mut lookup.file_system, + package, + manifest, + parsed, + &mut definitions, + )?; } let package_roots: Vec<&str> = manifests .iter() @@ -124,7 +135,13 @@ pub fn load_heft_configuration_and_then( .collect(); let selected_definitions: Vec = select_plugin_definitions(&references, &package_roots, &definitions)?; - validate_plugin_options(&tree, &references, &selected_definitions, &definitions)?; + validate_plugin_options( + &mut lookup.file_system, + &tree, + &references, + &selected_definitions, + &definitions, + )?; Ok(consume(&LoadedHeftConfiguration { build_folder_path: request.build_folder_path, rig: &rig, diff --git a/apps/heft-native/src/config/mod.rs b/apps/heft-native/src/config/mod.rs index 44c3e860f5..586617d0df 100644 --- a/apps/heft-native/src/config/mod.rs +++ b/apps/heft-native/src/config/mod.rs @@ -16,6 +16,8 @@ pub mod node_path; pub mod node_resolve; pub mod normalize; pub mod package_json; +pub mod path_component_cache; +pub mod path_probes; pub mod plan_graph_numbering; pub mod plan_graph_writer; pub mod plan_members; @@ -23,11 +25,14 @@ pub mod plugin_manifest; pub mod plugin_options; pub mod plugin_references; pub mod plugin_selection; -pub mod real_path_resolver; pub mod rig; #[cfg(test)] +mod tests_embedded_schemas; +#[cfg(test)] mod tests_merge_semantics; #[cfg(test)] +mod tests_node_path_scan; +#[cfg(test)] mod tests_node_paths; #[cfg(all(test, unix))] mod tests_real_path_resolver; diff --git a/apps/heft-native/src/config/node_path.rs b/apps/heft-native/src/config/node_path.rs index 66caa44ade..e4bd5bf48b 100644 --- a/apps/heft-native/src/config/node_path.rs +++ b/apps/heft-native/src/config/node_path.rs @@ -2,6 +2,40 @@ pub fn is_absolute(path: &str) -> bool { path.starts_with('/') } +const LOW_SEVEN_BITS: u64 = 0x7f7f_7f7f_7f7f_7f7f; +const SEPARATORS: u64 = 0x2f2f_2f2f_2f2f_2f2f; + +fn zero_byte_mask(word: u64) -> u64 { + !(((word & LOW_SEVEN_BITS).wrapping_add(LOW_SEVEN_BITS)) | word) & !LOW_SEVEN_BITS +} + +fn has_separator_before_dot_or_separator(bytes: &[u8]) -> bool { + let (words, rest) = bytes.as_chunks::<8>(); + let mut tail: [u8; 8] = [0; 8]; + tail[..rest.len()].copy_from_slice(rest); + let mut previous_ends_with_separator: bool = false; + for word in words.iter().chain(std::iter::once(&tail)) { + let word: u64 = u64::from_le_bytes(*word); + let separators: u64 = zero_byte_mask(word ^ SEPARATORS); + let dots_or_separators: u64 = zero_byte_mask((word | 0x0101_0101_0101_0101) ^ SEPARATORS); + if separators & (dots_or_separators >> 8) != 0 + || (previous_ends_with_separator && dots_or_separators & 0x80 != 0) + { + return true; + } + previous_ends_with_separator = separators >> 63 != 0; + } + false +} + +fn is_normalized_absolute_path(path: &str) -> bool { + let bytes: &[u8] = path.as_bytes(); + bytes == b"/" + || (bytes.first() == Some(&b'/') + && bytes.last() != Some(&b'/') + && !has_separator_before_dot_or_separator(bytes)) +} + fn normalize_segments(path: &str, allow_above_root: bool) -> String { let mut segments: Vec<&str> = Vec::with_capacity(16); let mut leading_parent_count: usize = 0; @@ -33,6 +67,13 @@ fn normalize_segments(path: &str, allow_above_root: bool) -> String { } pub fn normalize(path: &str) -> String { + if is_normalized_absolute_path(path) { + return path.to_string(); + } + normalize_slowly(path) +} + +fn normalize_slowly(path: &str) -> String { if path.is_empty() { return ".".to_string(); } @@ -62,14 +103,25 @@ pub fn resolve(base: &str, relative: &str) -> String { if is_absolute(relative) { return resolve_absolute(relative); } + let relative: &str = relative.trim_start_matches("./"); let mut combined: String = String::with_capacity(base.len() + relative.len() + 1); combined.push_str(base); combined.push('/'); combined.push_str(relative); - resolve_absolute(&combined) + if is_normalized_absolute_path(&combined) { + return combined; + } + resolve_absolute_slowly(&combined) } pub fn resolve_absolute(absolute_path: &str) -> String { + if is_normalized_absolute_path(absolute_path) { + return absolute_path.to_string(); + } + resolve_absolute_slowly(absolute_path) +} + +fn resolve_absolute_slowly(absolute_path: &str) -> String { let normalized: String = normalize_segments(absolute_path, false); let mut result: String = String::with_capacity(normalized.len() + 1); result.push('/'); @@ -84,11 +136,15 @@ pub fn join(base: &str, relative: &str) -> String { if base.is_empty() { return normalize(relative); } + let relative: &str = relative.trim_start_matches("./"); let mut combined: String = String::with_capacity(base.len() + relative.len() + 1); combined.push_str(base); combined.push('/'); combined.push_str(relative); - normalize(&combined) + if is_normalized_absolute_path(&combined) { + return combined; + } + normalize_slowly(&combined) } pub fn dirname(path: &str) -> &str { diff --git a/apps/heft-native/src/config/node_resolve.rs b/apps/heft-native/src/config/node_resolve.rs index 369ec169bb..6d65ee0283 100644 --- a/apps/heft-native/src/config/node_resolve.rs +++ b/apps/heft-native/src/config/node_resolve.rs @@ -31,19 +31,16 @@ pub fn is_definitely_valid_package_name(package_name: &str) -> bool { unscoped_name.chars().all(is_name_char) && unscoped_name != "." && unscoped_name != ".." } -pub fn node_modules_folders(start: &str) -> Vec { - let absolute_start: String = resolve_absolute(start); - let mut folders: Vec = Vec::with_capacity(16); - let mut current: String = absolute_start; - loop { - folders.push(resolve(¤t, "node_modules")); - let parent: &str = dirname(¤t); - if parent == current { - break; +pub fn node_modules_folders(start: &str) -> impl Iterator { + let mut next_folder: Option = Some(resolve_absolute(start)); + std::iter::from_fn(move || { + let folder: String = next_folder.take()?; + let parent: &str = dirname(&folder); + if parent != folder { + next_folder = Some(parent.to_string()); } - current = parent.to_string(); - } - folders + Some(resolve(&folder, "node_modules")) + }) } fn realpath_like_resolve( diff --git a/apps/heft-native/src/config/package_json.rs b/apps/heft-native/src/config/package_json.rs index 01802cd0e0..94daf23ea4 100644 --- a/apps/heft-native/src/config/package_json.rs +++ b/apps/heft-native/src/config/package_json.rs @@ -1,8 +1,7 @@ -use std::collections::HashMap; - use super::fallback::{fallback, ConfigResult}; -use super::fs_probe::{read_text_or_missing, FileSystemProbeCache}; +use super::fs_probe::FileSystemProbeCache; use super::node_path::{dirname, join, resolve_absolute}; +use super::path_probes::PathKeyedMap; use crate::json::{parse_json_with_comments_exactly_like_jju, JsonValue}; #[derive(Clone)] @@ -13,8 +12,8 @@ pub struct PackageJsonIdentity { #[derive(Default)] pub struct PackageJsonLookup { - package_folder_by_path: HashMap>, - identity_by_real_path: HashMap, + package_folder_by_path: PathKeyedMap>, + identity_by_real_path: PathKeyedMap, pub file_system: FileSystemProbeCache, } @@ -29,6 +28,14 @@ fn optional_string_field(value: &JsonValue, key: &str) -> ConfigResult PackageJsonLookup { + let mut lookup: PackageJsonLookup = PackageJsonLookup::default(); + lookup + .file_system + .remember_physical_directory_path(current_folder); + lookup + } + fn try_load_identity( &mut self, package_json_path: &str, @@ -40,7 +47,7 @@ impl PackageJsonLookup { if let Some(identity) = self.identity_by_real_path.get(&real_path) { return Ok(Some(identity.clone())); } - let text: String = match read_text_or_missing(&real_path)? { + let text: String = match self.file_system.read_text_or_missing(&real_path)? { Some(text) => text, None => return fallback("a package.json disappeared while it was read"), }; diff --git a/apps/heft-native/src/config/path_component_cache.rs b/apps/heft-native/src/config/path_component_cache.rs new file mode 100644 index 0000000000..a86dfde64e --- /dev/null +++ b/apps/heft-native/src/config/path_component_cache.rs @@ -0,0 +1,160 @@ +use super::fallback::{fallback, ConfigResult}; +use super::path_probes::{ + next_component_range, probe_path_component, stat_following_symbolic_links, EntryKind, + PathComponentState, PathKeyedMap, ResolvedEntry, StatEntry, +}; + +const MAXIMUM_FOLLOWED_SYMBOLIC_LINKS: u32 = 40; + +#[derive(Default)] +pub struct PathComponentCache { + component_states: PathKeyedMap, + resolved_entries: PathKeyedMap>, + stat_entries: PathKeyedMap>, +} + +impl PathComponentCache { + pub fn remember_physical_directory_path(&mut self, directory_path: &str) { + if !directory_path.starts_with('/') { + return; + } + let mut prefix: String = String::with_capacity(directory_path.len()); + for component in directory_path + .split('/') + .filter(|component| !component.is_empty()) + { + prefix.push('/'); + prefix.push_str(component); + if component == "." || component == ".." { + return; + } + self.component_states + .entry(prefix.clone()) + .or_insert(PathComponentState::Present { + kind: EntryKind::Directory, + size: 0, + }); + } + } + + fn path_component_state(&mut self, path: &str) -> ConfigResult { + if let Some(state) = self.component_states.get(path) { + return Ok(state.clone()); + } + let state: PathComponentState = probe_path_component(path)?; + self.component_states + .insert(path.to_string(), state.clone()); + Ok(state) + } + + fn parent_is_a_known_physical_directory(&self, path: &str) -> bool { + let parent: &str = &path[..path.rfind('/').unwrap_or(0)]; + parent.is_empty() + || matches!( + self.component_states.get(parent), + Some(PathComponentState::Present { + kind: EntryKind::Directory, + .. + }) + ) + } + + pub fn cached_stat(&self, path: &str) -> Option> { + if let Some(resolved) = self.resolved_entries.get(path) { + return Some(resolved.as_ref().map(|entry| StatEntry { + kind: entry.kind, + size: entry.size, + })); + } + self.stat_entries.get(path).copied() + } + + pub fn stat(&mut self, path: &str) -> ConfigResult> { + if let Some(resolved) = self.resolved_entries.get(path) { + return Ok(resolved.as_ref().map(|entry| StatEntry { + kind: entry.kind, + size: entry.size, + })); + } + if let Some(stat_entry) = self.stat_entries.get(path) { + return Ok(*stat_entry); + } + if path.starts_with('/') && self.parent_is_a_known_physical_directory(path) { + let resolved: Option = self.resolve(path)?; + return Ok(resolved.map(|entry| StatEntry { + kind: entry.kind, + size: entry.size, + })); + } + let stat_entry: Option = stat_following_symbolic_links(path)?; + self.stat_entries.insert(path.to_string(), stat_entry); + Ok(stat_entry) + } + + pub fn resolve(&mut self, path: &str) -> ConfigResult> { + if let Some(resolved) = self.resolved_entries.get(path) { + return Ok(resolved.clone()); + } + let resolved: Option = self.resolve_uncached(path)?; + self.resolved_entries + .insert(path.to_string(), resolved.clone()); + Ok(resolved) + } + + fn resolve_uncached(&mut self, path: &str) -> ConfigResult> { + if !path.starts_with('/') { + return fallback("a path to resolve is not absolute"); + } + let mut remaining: String = path.to_string(); + let mut position: usize = 0; + let mut resolved: String = String::with_capacity(path.len()); + let (mut kind, mut size): (EntryKind, u64) = (EntryKind::Directory, 0); + let mut followed_symbolic_links: u32 = 0; + while let Some((start, end)) = next_component_range(&remaining, position) { + position = end; + if kind != EntryKind::Directory { + return Ok(None); + } + let component: &str = &remaining[start..end]; + if component == "." { + continue; + } + if component == ".." { + resolved.truncate(resolved.rfind('/').unwrap_or(0)); + continue; + } + let parent_length: usize = resolved.len(); + resolved.push('/'); + resolved.push_str(component); + match self.path_component_state(&resolved)? { + PathComponentState::Missing => return Ok(None), + PathComponentState::Present { + kind: component_kind, + size: component_size, + } => (kind, size) = (component_kind, component_size), + PathComponentState::SymbolicLink(target) => { + followed_symbolic_links += 1; + if followed_symbolic_links > MAXIMUM_FOLLOWED_SYMBOLIC_LINKS { + return fallback("too many levels of symbolic links"); + } + resolved.truncate(if target.starts_with('/') { + 0 + } else { + parent_length + }); + (kind, size) = (EntryKind::Directory, 0); + remaining = format!("{target}/{}", &remaining[position..]); + position = 0; + } + } + } + if resolved.is_empty() { + resolved.push('/'); + } + Ok(Some(ResolvedEntry { + real_path: resolved, + kind, + size, + })) + } +} diff --git a/apps/heft-native/src/config/path_probes.rs b/apps/heft-native/src/config/path_probes.rs new file mode 100644 index 0000000000..eca0e63d00 --- /dev/null +++ b/apps/heft-native/src/config/path_probes.rs @@ -0,0 +1,132 @@ +use std::collections::HashMap; +use std::fs; +use std::hash::{BuildHasherDefault, Hasher}; +use std::io; + +use super::fallback::{fallback, ConfigResult}; + +pub type PathKeyedMap = HashMap>; + +#[derive(Default)] +pub struct PathHasher(u64); + +impl PathHasher { + fn mix(&mut self, word: u64) { + self.0 = (self.0.rotate_left(5) ^ word).wrapping_mul(0x517c_c1b7_2722_0a95); + } +} + +impl Hasher for PathHasher { + fn write(&mut self, bytes: &[u8]) { + let (words, rest) = bytes.as_chunks::<8>(); + for word in words { + self.mix(u64::from_le_bytes(*word)); + } + let mut tail: [u8; 8] = [0; 8]; + tail[..rest.len()].copy_from_slice(rest); + self.mix(u64::from_le_bytes(tail) ^ ((rest.len() as u64) << 59)); + } + + fn write_u8(&mut self, byte: u8) { + self.mix(u64::from(byte)); + } + + fn finish(&self) -> u64 { + self.0 + } +} + +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub enum EntryKind { + File, + Directory, + Other, +} + +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct ResolvedEntry { + pub real_path: String, + pub kind: EntryKind, + pub size: u64, +} + +#[derive(Clone)] +pub enum PathComponentState { + Missing, + Present { kind: EntryKind, size: u64 }, + SymbolicLink(String), +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct StatEntry { + pub kind: EntryKind, + pub size: u64, +} + +pub fn is_missing_entry_error(error: &io::Error) -> bool { + matches!( + error.kind(), + io::ErrorKind::NotFound | io::ErrorKind::NotADirectory + ) || matches!(error.raw_os_error(), Some(2) | Some(20)) +} + +pub fn entry_kind_of(file_type: fs::FileType) -> EntryKind { + if file_type.is_dir() { + EntryKind::Directory + } else if file_type.is_file() || is_fifo(&file_type) { + EntryKind::File + } else { + EntryKind::Other + } +} + +#[cfg(unix)] +fn is_fifo(file_type: &fs::FileType) -> bool { + use std::os::unix::fs::FileTypeExt; + file_type.is_fifo() +} + +#[cfg(not(unix))] +fn is_fifo(_file_type: &fs::FileType) -> bool { + false +} + +pub fn probe_path_component(path: &str) -> ConfigResult { + match fs::symlink_metadata(path) { + Ok(metadata) if metadata.file_type().is_symlink() => match fs::read_link(path) { + Ok(target) => match target.into_os_string().into_string() { + Ok(target) => Ok(PathComponentState::SymbolicLink(target)), + Err(_) => fallback("a symbolic link target is not valid UTF-8"), + }, + Err(error) if is_missing_entry_error(&error) => Ok(PathComponentState::Missing), + Err(_) => fallback("readlink failed with an unexpected error"), + }, + Ok(metadata) => Ok(PathComponentState::Present { + kind: entry_kind_of(metadata.file_type()), + size: metadata.len(), + }), + Err(error) if is_missing_entry_error(&error) => Ok(PathComponentState::Missing), + Err(_) => fallback("lstat failed with an unexpected error"), + } +} + +pub fn stat_following_symbolic_links(path: &str) -> ConfigResult> { + match fs::metadata(path) { + Ok(metadata) => Ok(Some(StatEntry { + kind: entry_kind_of(metadata.file_type()), + size: metadata.len(), + })), + Err(error) if is_missing_entry_error(&error) => Ok(None), + Err(_) => fallback("stat failed with an unexpected error"), + } +} + +pub fn next_component_range(path: &str, from: usize) -> Option<(usize, usize)> { + let bytes: &[u8] = path.as_bytes(); + let start: usize = from + bytes[from..].iter().position(|byte| *byte != b'/')?; + let end: usize = bytes[start..] + .iter() + .position(|byte| *byte == b'/') + .map_or(bytes.len(), |length| start + length); + Some((start, end)) +} diff --git a/apps/heft-native/src/config/plugin_manifest.rs b/apps/heft-native/src/config/plugin_manifest.rs index 182b0b5be9..dd4209c824 100644 --- a/apps/heft-native/src/config/plugin_manifest.rs +++ b/apps/heft-native/src/config/plugin_manifest.rs @@ -1,5 +1,5 @@ use super::fallback::{fallback, ConfigResult}; -use super::fs_probe::{exists_like_exists_sync, read_text_or_missing}; +use super::fs_probe::FileSystemProbeCache; use super::node_path::resolve; use crate::json::{parse_json_with_comments_exactly_like_jju, JsonValue}; use crate::schema::CompiledJsonSchema; @@ -28,13 +28,14 @@ pub struct PluginDefinition<'manifest> { } pub fn read_plugin_package_manifest( + file_system: &mut FileSystemProbeCache, package_root: &str, package_name: &str, ) -> ConfigResult { let mut manifest_file_path: String = String::with_capacity(package_root.len() + 17); manifest_file_path.push_str(package_root); manifest_file_path.push_str("/heft-plugin.json"); - match read_text_or_missing(&manifest_file_path)? { + match file_system.read_text_or_missing(&manifest_file_path)? { Some(text) => Ok(PluginPackageManifest { package_root: package_root.to_string(), package_name: package_name.to_string(), @@ -79,6 +80,7 @@ fn truthy_string<'manifest>( } fn load_plugin_definition<'manifest>( + file_system: &mut FileSystemProbeCache, kind: PluginKind, package: usize, manifest: &PluginPackageManifest, @@ -105,7 +107,7 @@ fn load_plugin_definition<'manifest>( let options_schema_path: Option = match truthy_string(definition, "optionsSchema")? { Some(options_schema) => { let resolved_schema_path: String = resolve(&manifest.package_root, options_schema); - if !exists_like_exists_sync(&resolved_schema_path) { + if !file_system.exists_like_exists_sync(&resolved_schema_path)? { return fallback("a plugin options schema file does not exist"); } Some(resolved_schema_path) @@ -124,6 +126,7 @@ fn load_plugin_definition<'manifest>( } pub fn load_plugin_definitions<'manifest>( + file_system: &mut FileSystemProbeCache, package: usize, manifest: &PluginPackageManifest, parsed: &'manifest JsonValue<'manifest>, @@ -141,7 +144,7 @@ pub fn load_plugin_definitions<'manifest>( ] { for definition in list { let loaded: PluginDefinition = - load_plugin_definition(kind, package, manifest, definition)?; + load_plugin_definition(file_system, kind, package, manifest, definition)?; if definitions[first_definition..] .iter() .any(|other| other.plugin_name == loaded.plugin_name) diff --git a/apps/heft-native/src/config/plugin_options.rs b/apps/heft-native/src/config/plugin_options.rs index 4cb9a924cc..df7397c5dd 100644 --- a/apps/heft-native/src/config/plugin_options.rs +++ b/apps/heft-native/src/config/plugin_options.rs @@ -1,5 +1,5 @@ use super::fallback::{fallback, ConfigResult}; -use super::fs_probe::read_text_or_missing; +use super::fs_probe::FileSystemProbeCache; use super::plugin_manifest::PluginDefinition; use super::plugin_references::PluginReferences; use super::tree::{ConfigTree, NodeId}; @@ -7,25 +7,6 @@ use super::tree_json::{empty_json_object, tree_to_json_value}; use crate::json::{parse_json_with_comments_exactly_like_jju, JsonValue}; use crate::schema::{compile_json_schema_for_fast_validation, CompiledJsonSchema}; -fn options_are_definitely_valid(schema_path: &str, options: &JsonValue) -> ConfigResult { - let schema_text: String = match read_text_or_missing(schema_path)? { - Some(schema_text) => schema_text, - None => return fallback("a plugin options schema file disappeared"), - }; - let schema_document: JsonValue = match parse_json_with_comments_exactly_like_jju(&schema_text) { - Ok(schema_document) => schema_document, - Err(_) => return fallback("a plugin options schema can't be parsed exactly"), - }; - let compiled_schema: CompiledJsonSchema = - match compile_json_schema_for_fast_validation(&schema_document) { - Some(compiled_schema) => compiled_schema, - None => { - return fallback("a plugin options schema is outside the fast validation subset") - } - }; - Ok(compiled_schema.is_definitely_valid(options)) -} - fn options_or_empty_object<'text>( tree: &ConfigTree<'text>, options: Option, @@ -36,16 +17,68 @@ fn options_or_empty_object<'text>( } } +fn read_options_schema_texts( + file_system: &mut FileSystemProbeCache, + schema_paths: &[&str], +) -> ConfigResult> { + let mut schema_texts: Vec = Vec::with_capacity(schema_paths.len()); + for schema_path in schema_paths { + match file_system.read_text_or_missing(schema_path)? { + Some(schema_text) => schema_texts.push(schema_text), + None => return fallback("a plugin options schema file disappeared"), + } + } + Ok(schema_texts) +} + +fn parse_options_schema(schema_text: &str) -> ConfigResult> { + match parse_json_with_comments_exactly_like_jju(schema_text) { + Ok(schema_document) => Ok(schema_document), + Err(_) => fallback("a plugin options schema can't be parsed exactly"), + } +} + +fn compile_options_schema<'schema>( + schema_document: &'schema JsonValue<'schema>, +) -> ConfigResult> { + match compile_json_schema_for_fast_validation(schema_document) { + Some(compiled_schema) => Ok(compiled_schema), + None => fallback("a plugin options schema is outside the fast validation subset"), + } +} + pub fn validate_plugin_options( + file_system: &mut FileSystemProbeCache, tree: &ConfigTree, references: &PluginReferences, selected: &[usize], definitions: &[PluginDefinition], ) -> ConfigResult<()> { + let mut schema_paths: Vec<&str> = Vec::new(); + for definition in selected { + if let Some(schema_path) = &definitions[*definition].options_schema_path { + if !schema_paths.contains(&schema_path.as_str()) { + schema_paths.push(schema_path); + } + } + } + let schema_texts: Vec = read_options_schema_texts(file_system, &schema_paths)?; + let schema_documents: Vec = schema_texts + .iter() + .map(|schema_text| parse_options_schema(schema_text)) + .collect::>>()?; + let compiled_schemas: Vec = schema_documents + .iter() + .map(compile_options_schema) + .collect::>>()?; for (reference, definition) in references.all_plugin_references().zip(selected) { if let Some(schema_path) = &definitions[*definition].options_schema_path { + let schema_index: Option = + schema_paths.iter().position(|path| path == schema_path); let options: JsonValue = options_or_empty_object(tree, reference.options); - if !options_are_definitely_valid(schema_path, &options)? { + let valid: bool = schema_index + .is_some_and(|index| compiled_schemas[index].is_definitely_valid(&options)); + if !valid { return fallback("plugin options are not definitely valid"); } } diff --git a/apps/heft-native/src/config/real_path_resolver.rs b/apps/heft-native/src/config/real_path_resolver.rs deleted file mode 100644 index aba09cb27c..0000000000 --- a/apps/heft-native/src/config/real_path_resolver.rs +++ /dev/null @@ -1,104 +0,0 @@ -use std::collections::HashMap; -use std::fs; -use std::io; - -use super::fallback::{fallback, ConfigResult}; - -const MAXIMUM_FOLLOWED_SYMBOLIC_LINKS: u32 = 40; - -#[derive(Clone)] -enum PathComponentKind { - NotASymbolicLink, - SymbolicLink(String), -} - -#[derive(Default)] -pub struct RealPathResolver { - component_kinds: HashMap>, -} - -fn is_missing_entry_error(error: &io::Error) -> bool { - matches!( - error.kind(), - io::ErrorKind::NotFound | io::ErrorKind::NotADirectory - ) || matches!(error.raw_os_error(), Some(2) | Some(20)) -} - -fn classify_path_component(path: &str) -> ConfigResult> { - match fs::symlink_metadata(path) { - Ok(metadata) if metadata.file_type().is_symlink() => match fs::read_link(path) { - Ok(target) => match target.into_os_string().into_string() { - Ok(target) => Ok(Some(PathComponentKind::SymbolicLink(target))), - Err(_) => fallback("a symbolic link target is not valid UTF-8"), - }, - Err(error) if is_missing_entry_error(&error) => Ok(None), - Err(_) => fallback("readlink failed with an unexpected error"), - }, - Ok(_) => Ok(Some(PathComponentKind::NotASymbolicLink)), - Err(error) if is_missing_entry_error(&error) => Ok(None), - Err(_) => fallback("lstat failed with an unexpected error"), - } -} - -fn push_components_in_reverse(pending: &mut Vec, path: &str) { - pending.extend( - path.split('/') - .filter(|component| !component.is_empty()) - .rev() - .map(str::to_string), - ); -} - -impl RealPathResolver { - fn component_kind(&mut self, path: &str) -> ConfigResult> { - if let Some(kind) = self.component_kinds.get(path) { - return Ok(kind.clone()); - } - let kind: Option = classify_path_component(path)?; - self.component_kinds.insert(path.to_string(), kind.clone()); - Ok(kind) - } - - pub fn resolve_real_path(&mut self, path: &str) -> ConfigResult> { - if !path.starts_with('/') { - return fallback("realpath of a relative path"); - } - let mut pending: Vec = Vec::new(); - push_components_in_reverse(&mut pending, path); - let mut resolved: String = String::with_capacity(path.len()); - let mut followed_symbolic_links: u32 = 0; - while let Some(component) = pending.pop() { - if component == "." { - continue; - } - if component == ".." { - let parent_length: usize = resolved.rfind('/').unwrap_or(0); - resolved.truncate(parent_length); - continue; - } - let parent_length: usize = resolved.len(); - resolved.push('/'); - resolved.push_str(&component); - match self.component_kind(&resolved)? { - None => return Ok(None), - Some(PathComponentKind::NotASymbolicLink) => {} - Some(PathComponentKind::SymbolicLink(target)) => { - followed_symbolic_links += 1; - if followed_symbolic_links > MAXIMUM_FOLLOWED_SYMBOLIC_LINKS { - return fallback("too many levels of symbolic links"); - } - resolved.truncate(if target.starts_with('/') { - 0 - } else { - parent_length - }); - push_components_in_reverse(&mut pending, &target); - } - } - } - if resolved.is_empty() { - resolved.push('/'); - } - Ok(Some(resolved)) - } -} diff --git a/apps/heft-native/src/config/rig.rs b/apps/heft-native/src/config/rig.rs index 4ca009f00d..18e947169c 100644 --- a/apps/heft-native/src/config/rig.rs +++ b/apps/heft-native/src/config/rig.rs @@ -1,5 +1,5 @@ use super::fallback::{fallback, ConfigResult}; -use super::fs_probe::{exists_like_exists_sync, read_text_or_missing, FileSystemProbeCache}; +use super::fs_probe::FileSystemProbeCache; use super::node_path::{dirname, join, resolve_absolute}; use super::node_resolve::resolve_node_modules_file; use crate::json::{parse_json_with_comments_exactly_like_jju, JsonObject, JsonValue}; @@ -40,7 +40,10 @@ fn is_rig_profile_name(profile: &str) -> bool { .all(|word| !word.is_empty() && word.chars().all(is_word_char)) } -pub fn load_rig_config_data(project_folder_path: &str) -> ConfigResult { +pub fn load_rig_config_data( + file_system: &mut FileSystemProbeCache, + project_folder_path: &str, +) -> ConfigResult { let rig_config_file_path: String = join(project_folder_path, "config/rig.json"); let not_found: RigConfigData = RigConfigData { project_folder_original_path: project_folder_path.to_string(), @@ -51,7 +54,7 @@ pub fn load_rig_config_data(project_folder_path: &str) -> ConfigResult text, None => return Ok(not_found), }; @@ -107,7 +110,7 @@ pub fn resolve_rig_profile_folder( dirname(&rig_package_json_path), &rig.relative_profile_folder_path, ); - if !exists_like_exists_sync(&profile_folder) { + if !file_system.exists_like_exists_sync(&profile_folder)? { return fallback("the rig profile folder does not exist"); } Ok(profile_folder) diff --git a/apps/heft-native/src/config/tests_embedded_schemas.rs b/apps/heft-native/src/config/tests_embedded_schemas.rs new file mode 100644 index 0000000000..2103dba08f --- /dev/null +++ b/apps/heft-native/src/config/tests_embedded_schemas.rs @@ -0,0 +1,82 @@ +use super::embedded_schemas::{HEFT_JSON_SCHEMA_TEXT, HEFT_PLUGIN_JSON_SCHEMA_TEXT}; +use crate::json::{parse_json_with_comments_exactly_like_jju, JsonObject, JsonValue}; +use crate::schema::compile_json_schema_for_fast_validation; + +const ORIGINAL_HEFT_JSON_SCHEMA_TEXT: &str = + include_str!("../../../heft/src/schemas/heft.schema.json"); +const ORIGINAL_HEFT_PLUGIN_JSON_SCHEMA_TEXT: &str = + include_str!("../../../heft/src/schemas/heft-plugin.schema.json"); + +fn without_annotations(value: JsonValue<'_>) -> JsonValue<'_> { + match value { + JsonValue::Array(items) => { + JsonValue::Array(items.into_iter().map(without_annotations).collect()) + } + JsonValue::Object(object) => { + let mut stripped: JsonObject = JsonObject::with_capacity(object.len()); + for (key, member) in object.into_entries() { + let is_annotation: bool = matches!(key.as_ref(), "title" | "description") + && matches!(member, JsonValue::String(_)); + if is_annotation { + continue; + } + let member: JsonValue = match key.as_ref() { + "enum" | "const" | "default" | "examples" => member, + _ => without_annotations(member), + }; + stripped.set_keeping_first_position(key, member); + } + JsonValue::Object(stripped) + } + other => other, + } +} + +fn parse(text: &str) -> JsonValue<'_> { + parse_json_with_comments_exactly_like_jju(text).expect("the schema text parses") +} + +#[test] +fn embedded_schemas_are_the_heft_schemas_without_annotations() { + for (embedded, original) in [ + (HEFT_JSON_SCHEMA_TEXT, ORIGINAL_HEFT_JSON_SCHEMA_TEXT), + (HEFT_PLUGIN_JSON_SCHEMA_TEXT, ORIGINAL_HEFT_PLUGIN_JSON_SCHEMA_TEXT), + ] { + assert_eq!( + parse(embedded), + without_annotations(parse(original)), + "regenerate src/config/*_schema_without_annotations.json from apps/heft/src/schemas" + ); + assert!(!embedded.contains('\n') && embedded.len() * 2 < original.len()); + } +} + +#[test] +fn embedded_and_original_schemas_accept_the_same_documents() { + let documents: [&str; 8] = [ + r#"{}"#, + r#"{"phasesByName":{"build":{"tasksByName":{"t":{"taskPlugin":{"pluginPackage":"p"}}}}}}"#, + r#"{"phasesByName":{"Build":{}}}"#, + r#"{"heftPlugins":[{"pluginPackage":"a\\b"}]}"#, + r#"{"aliasesByName":{"a":{"actionName":"build","defaultParameters":["--x"]}}}"#, + r#"{"taskPlugins":[{"pluginName":"a","entryPoint":"./a"}]}"#, + r#"{"lifecyclePlugins":[{"pluginName":"a","entryPoint":"./a","parameterScope":"s"}]}"#, + r#"{"extends":5}"#, + ]; + for (embedded, original) in [ + (HEFT_JSON_SCHEMA_TEXT, ORIGINAL_HEFT_JSON_SCHEMA_TEXT), + (HEFT_PLUGIN_JSON_SCHEMA_TEXT, ORIGINAL_HEFT_PLUGIN_JSON_SCHEMA_TEXT), + ] { + let (embedded_document, original_document) = (parse(embedded), parse(original)); + let embedded_schema = compile_json_schema_for_fast_validation(&embedded_document).unwrap(); + let original_schema = compile_json_schema_for_fast_validation(&original_document).unwrap(); + for document in documents { + let value: JsonValue = parse(document); + assert_eq!( + embedded_schema.is_definitely_valid(&value), + original_schema.is_definitely_valid(&value), + "{document}" + ); + } + } +} diff --git a/apps/heft-native/src/config/tests_node_path_scan.rs b/apps/heft-native/src/config/tests_node_path_scan.rs new file mode 100644 index 0000000000..64500e7476 --- /dev/null +++ b/apps/heft-native/src/config/tests_node_path_scan.rs @@ -0,0 +1,40 @@ +use super::node_path::{join, normalize, resolve, resolve_absolute}; + +fn reference_resolve_absolute(path: &str) -> String { + let mut segments: Vec<&str> = Vec::new(); + for segment in path.split('/') { + match segment { + "" | "." => {} + ".." => { + segments.pop(); + } + _ => segments.push(segment), + } + } + format!("/{}", segments.join("/")) +} + +#[test] +fn random_absolute_paths_normalize_like_the_reference() { + let alphabet: [char; 3] = ['/', '.', 'a']; + let mut state: u64 = 0x2545_f491_4f6c_dd1d; + for _ in 0..20_000 { + state = state.wrapping_mul(6_364_136_223_846_793_005).wrapping_add(1); + let length: usize = (state >> 58) as usize % 41; + let mut path: String = String::from("/"); + for _ in 0..length { + state = state.wrapping_mul(6_364_136_223_846_793_005).wrapping_add(1); + path.push(alphabet[(state >> 33) as usize % 3]); + } + let expected: String = reference_resolve_absolute(&path); + assert_eq!(resolve_absolute(&path), expected, "{path}"); + let trailing: &str = if path.ends_with('/') && expected != "/" { "/" } else { "" }; + assert_eq!(normalize(&path), format!("{expected}{trailing}"), "{path}"); + assert_eq!(join("/", &path[1..]), format!("{expected}{trailing}"), "{path}"); + let dotted: String = format!("./{}", &path[1..]); + assert_eq!(resolve("/b", &dotted), reference_resolve_absolute(&format!("/b/{dotted}")), "{path}"); + let joined: String = format!("/b/{dotted}"); + let joined_trailing: &str = if joined.ends_with('/') && reference_resolve_absolute(&joined) != "/" { "/" } else { "" }; + assert_eq!(join("/b", &dotted), format!("{}{joined_trailing}", reference_resolve_absolute(&joined)), "{path}"); + } +} diff --git a/apps/heft-native/src/config/tests_node_paths.rs b/apps/heft-native/src/config/tests_node_paths.rs index 5ccf710233..a621a18fa3 100644 --- a/apps/heft-native/src/config/tests_node_paths.rs +++ b/apps/heft-native/src/config/tests_node_paths.rs @@ -1,4 +1,4 @@ -use super::node_path::{dirname, is_absolute, join, normalize, resolve}; +use super::node_path::{dirname, is_absolute, join, normalize, resolve, resolve_absolute}; use super::node_resolve::{is_definitely_valid_package_name, node_modules_folders}; const RESOLVE_CASES: [(&str, &str, &str); 12] = [ @@ -80,7 +80,7 @@ fn posix_paths_match_node_path_posix() { #[test] fn node_modules_folders_match_resolve_1_22() { assert_eq!( - node_modules_folders("/a/b/c"), + node_modules_folders("/a/b/c").collect::>(), [ "/a/b/c/node_modules", "/a/b/node_modules", @@ -89,7 +89,7 @@ fn node_modules_folders_match_resolve_1_22() { ] ); assert_eq!( - node_modules_folders("/a/node_modules/b"), + node_modules_folders("/a/node_modules/b").collect::>(), [ "/a/node_modules/b/node_modules", "/a/node_modules/node_modules", @@ -97,9 +97,9 @@ fn node_modules_folders_match_resolve_1_22() { "/node_modules" ] ); - assert_eq!(node_modules_folders("/"), ["/node_modules"]); + assert_eq!(node_modules_folders("/").collect::>(), ["/node_modules"]); assert_eq!( - node_modules_folders("/a/b/node_modules/@s/p/node_modules/q"), + node_modules_folders("/a/b/node_modules/@s/p/node_modules/q").collect::>(), [ "/a/b/node_modules/@s/p/node_modules/q/node_modules", "/a/b/node_modules/@s/p/node_modules/node_modules", @@ -145,3 +145,45 @@ fn package_names_that_resolve_accepts_without_doubt() { assert!(!is_definitely_valid_package_name("a b"), "a b"); assert!(!is_definitely_valid_package_name("x:y"), "x:y"); } + +fn reference_segments(path: &str) -> Vec<&str> { + let mut segments: Vec<&str> = Vec::new(); + for segment in path.split('/') { + match segment { + "" | "." => {} + ".." => { + segments.pop(); + } + _ => segments.push(segment), + } + } + segments +} + +#[test] +fn absolute_path_shortcuts_match_full_normalization() { + let names: [&str; 12] = ["", ".", "..", "...", "a", "bc", ".d", "e.", "..f", "abcdef", "abcdefg", "abcdef."]; + let mut paths: Vec = vec!["/".to_string(), "//".to_string()]; + for first in names { + for second in names { + for third in names { + paths.push(format!("/{first}/{second}/{third}")); + paths.push(format!("/{first}/{second}/{third}/")); + paths.push(format!("/{first}/{second}")); + paths.push(format!("/{first}")); + } + } + } + for path in &paths { + let joined: String = reference_segments(path).join("/"); + let expected: String = format!("/{joined}"); + assert_eq!(resolve_absolute(path), expected, "{path}"); + assert_eq!(resolve("/", path), expected, "{path}"); + let trailing: &str = if path.ends_with('/') && !joined.is_empty() { "/" } else { "" }; + assert_eq!(normalize(path), format!("{expected}{trailing}"), "{path}"); + assert_eq!(join("/", path), format!("{expected}{trailing}"), "{path}"); + let relative: &str = path.trim_start_matches('/'); + let combined: String = format!("/x/{relative}"); + assert_eq!(resolve("/x", relative), resolve_absolute(&combined), "{path}"); + } +} diff --git a/apps/heft-native/src/config/tests_real_path_resolver.rs b/apps/heft-native/src/config/tests_real_path_resolver.rs index 9f079484f6..46be02687d 100644 --- a/apps/heft-native/src/config/tests_real_path_resolver.rs +++ b/apps/heft-native/src/config/tests_real_path_resolver.rs @@ -2,7 +2,8 @@ use std::fs; use std::os::unix::fs::symlink; use std::path::PathBuf; -use super::real_path_resolver::RealPathResolver; +use super::path_component_cache::PathComponentCache; +use super::path_probes::EntryKind; fn canonicalize_or_missing(path: &str) -> Option { fs::canonicalize(path) @@ -36,7 +37,7 @@ fn real_paths_match_the_operating_system_realpath() { .to_str() .unwrap() .to_string(); - let mut resolver: RealPathResolver = RealPathResolver::default(); + let mut resolver: PathComponentCache = PathComponentCache::default(); for relative_path in [ "project/node_modules/@scope/pkg/package.json", "project/node_modules/@scope/pkg", @@ -49,24 +50,26 @@ fn real_paths_match_the_operating_system_realpath() { "", ] { let path: String = format!("{root}/{relative_path}"); - assert_eq!( - resolver.resolve_real_path(&path).unwrap(), - canonicalize_or_missing(&path), - "{path}" - ); - assert_eq!( - resolver.resolve_real_path(&path).unwrap(), - canonicalize_or_missing(&path), - "{path}" - ); + for _ in 0..2 { + let resolved = resolver.resolve(&path).unwrap(); + let real_path = resolved.as_ref().map(|entry| entry.real_path.clone()); + assert_eq!(real_path, canonicalize_or_missing(&path), "{path}"); + if let (Some(entry), Ok(metadata)) = (&resolved, fs::metadata(&path)) { + let expected_kind = if metadata.is_dir() { + EntryKind::Directory + } else { + EntryKind::File + }; + assert_eq!(entry.kind, expected_kind, "{path}"); + if metadata.is_file() { + assert_eq!(entry.size, metadata.len(), "{path}"); + } + } + } } - assert!(resolver - .resolve_real_path(&format!("{root}/loop-a/x")) - .is_err()); - assert!(resolver.resolve_real_path("relative/path").is_err()); - assert_eq!( - resolver.resolve_real_path("/").unwrap().as_deref(), - Some("/") - ); + assert!(resolver.resolve(&format!("{root}/loop-a/x")).is_err()); + assert!(resolver.resolve("relative/path").is_err()); + let root_entry = resolver.resolve("/").unwrap().map(|entry| entry.real_path); + assert_eq!(root_entry.as_deref(), Some("/")); let _ = fs::remove_dir_all(&root); } diff --git a/apps/heft-native/src/host_link/plan_file.rs b/apps/heft-native/src/host_link/plan_file.rs index 15cd9290c2..da24fed02b 100644 --- a/apps/heft-native/src/host_link/plan_file.rs +++ b/apps/heft-native/src/host_link/plan_file.rs @@ -34,11 +34,16 @@ fn write_plan_to_anonymous_file_in_folder( ) -> Option { use std::io::{Seek, SeekFrom, Write}; use std::os::unix::fs::OpenOptionsExt; - const OPEN_UNNAMED_TEMPORARY_FILE: i32 = 0o20200000; + #[cfg(any(target_arch = "x86_64", target_arch = "x86"))] + const OPEN_UNNAMED_TEMPORARY_FILE: Option = Some(0o20200000); + #[cfg(target_arch = "aarch64")] + const OPEN_UNNAMED_TEMPORARY_FILE: Option = Some(0o20040000); + #[cfg(not(any(target_arch = "x86_64", target_arch = "x86", target_arch = "aarch64")))] + const OPEN_UNNAMED_TEMPORARY_FILE: Option = None; let mut file = std::fs::OpenOptions::new() .read(true) .write(true) - .custom_flags(OPEN_UNNAMED_TEMPORARY_FILE) + .custom_flags(OPEN_UNNAMED_TEMPORARY_FILE?) .mode(0o600) .open(folder) .ok()?; diff --git a/apps/heft-native/src/host_link/warm_client.rs b/apps/heft-native/src/host_link/warm_client.rs index 8e584e88e8..4235357d32 100644 --- a/apps/heft-native/src/host_link/warm_client.rs +++ b/apps/heft-native/src/host_link/warm_client.rs @@ -2,7 +2,7 @@ use std::ffi::OsString; use std::io::{IsTerminal, Write}; use std::os::unix::net::UnixStream; use std::os::unix::process::CommandExt; -use std::path::PathBuf; +use std::path::{Path, PathBuf}; use std::process::{Command, Stdio}; use std::time::Duration; @@ -65,8 +65,10 @@ fn locate_warm_host_target(native_heft_context: &NativeHeftContext) -> Option Option Option Option { + use std::os::unix::fs::MetadataExt; + let file_metadata = std::fs::metadata(path).ok()?; + Some(format!("{}:{}", file_metadata.dev(), file_metadata.ino())) +} + fn connect_and_wait_for_acceptance( warm_host_target: &WarmHostTarget, node_host_plan: &NodeHostPlan, diff --git a/apps/heft-native/src/json/lexer_string.rs b/apps/heft-native/src/json/lexer_string.rs index e35ff1e474..f61e6cc2e7 100644 --- a/apps/heft-native/src/json/lexer_string.rs +++ b/apps/heft-native/src/json/lexer_string.rs @@ -1,6 +1,7 @@ use std::borrow::Cow; use super::cursor::{JsonCursor, JsonScanResult, JsonTextNeedsJavaScriptParser}; +use crate::simd::position_of_json_string_special_byte; const FIRST_HIGH_SURROGATE: u32 = 0xd800; const FIRST_LOW_SURROGATE: u32 = 0xdc00; @@ -11,6 +12,7 @@ impl<'text> JsonCursor<'text> { self.position += 1; let content_start = self.position; loop { + self.position = position_of_json_string_special_byte(self.bytes, self.position); match self.next_byte_or_refuse()? { b'"' => { let content = &self.text[content_start..self.position]; @@ -55,13 +57,7 @@ impl<'text> JsonCursor<'text> { fn copy_run_of_ordinary_string_bytes_into(&mut self, decoded: &mut String) { let run_start = self.position; - self.position += 1; - while let Some(byte) = self.peek_byte() { - if byte == b'"' || byte == b'\\' || byte < 0x20 || byte == 0xe2 { - break; - } - self.position += 1; - } + self.position = position_of_json_string_special_byte(self.bytes, self.position + 1); decoded.push_str(&self.text[run_start..self.position]); } diff --git a/apps/heft-native/src/json/lexer_trivia.rs b/apps/heft-native/src/json/lexer_trivia.rs index 81622f9b1b..9b7bce5d88 100644 --- a/apps/heft-native/src/json/lexer_trivia.rs +++ b/apps/heft-native/src/json/lexer_trivia.rs @@ -1,10 +1,20 @@ use super::cursor::{JsonCursor, JsonScanResult, JsonTextNeedsJavaScriptParser}; +use crate::simd::{ + position_after_json_whitespace, position_of_block_comment_star, + position_of_line_comment_end_or_separator_lead_byte, +}; impl JsonCursor<'_> { pub(super) fn skip_whitespace_and_comments(&mut self) -> JsonScanResult<()> { loop { match self.peek_byte() { - Some(b' ' | b'\t' | b'\n' | b'\r') => self.position += 1, + Some(b' ' | b'\t' | b'\n' | b'\r') => { + self.position += 1; + if let Some(b' ' | b'\t' | b'\n' | b'\r') = self.peek_byte() { + self.position = + position_after_json_whitespace(self.bytes, self.position + 1); + } + } Some(b'/') => { self.refuse_unless_comments_and_trailing_commas_are_allowed()?; match self.peek_byte_after(1) { @@ -20,24 +30,26 @@ impl JsonCursor<'_> { fn skip_line_comment(&mut self) -> JsonScanResult<()> { self.position += 2; - while let Some(byte) = self.peek_byte() { - if byte == b'\n' || byte == b'\r' { - return Ok(()); - } - if self.is_at_line_or_paragraph_separator() { - return Err(JsonTextNeedsJavaScriptParser); + loop { + self.position = + position_of_line_comment_end_or_separator_lead_byte(self.bytes, self.position); + match self.peek_byte() { + None | Some(b'\n' | b'\r') => return Ok(()), + Some(_) if self.is_at_line_or_paragraph_separator() => { + return Err(JsonTextNeedsJavaScriptParser) + } + Some(_) => self.position += 1, } - self.position += 1; } - Ok(()) } fn skip_block_comment(&mut self) -> JsonScanResult<()> { self.position += 2; loop { + self.position = position_of_block_comment_star(self.bytes, self.position); match self.peek_byte() { None => return Err(JsonTextNeedsJavaScriptParser), - Some(b'*') if self.peek_byte_after(1) == Some(b'/') => { + Some(_) if self.peek_byte_after(1) == Some(b'/') => { self.position += 2; return Ok(()); } diff --git a/apps/heft-native/src/json/writer.rs b/apps/heft-native/src/json/writer.rs index ab998e94c6..d4f4839a22 100644 --- a/apps/heft-native/src/json/writer.rs +++ b/apps/heft-native/src/json/writer.rs @@ -1,6 +1,7 @@ use std::fmt::{self, Write}; use super::value::JsonValue; +use crate::simd::position_of_json_string_special_byte; pub fn write_json_for_javascript( value: &JsonValue<'_>, @@ -42,32 +43,56 @@ pub fn write_json_for_javascript( const HEXADECIMAL_DIGITS: &[u8; 16] = b"0123456789abcdef"; +const SEPARATOR_LEAD_BYTE: u8 = 0xe2; +const LINE_SEPARATOR_LAST_BYTE: u8 = 0xa8; +const PARAGRAPH_SEPARATOR_LAST_BYTE: u8 = 0xa9; + +fn separator_last_byte_at(bytes: &[u8], position: usize) -> Option { + match bytes.get(position..position + 3) { + Some( + &[SEPARATOR_LEAD_BYTE, 0x80, last @ (LINE_SEPARATOR_LAST_BYTE | PARAGRAPH_SEPARATOR_LAST_BYTE)], + ) => Some(last), + _ => None, + } +} + +fn write_escape_sequence( + byte: u8, + separator_last_byte: Option, + output: &mut Output, +) -> fmt::Result { + match (byte, separator_last_byte) { + (_, Some(LINE_SEPARATOR_LAST_BYTE)) => output.write_str("\\u2028"), + (_, Some(_)) => output.write_str("\\u2029"), + (b'"', None) => output.write_str("\\\""), + (b'\\', None) => output.write_str("\\\\"), + (control, None) => { + output.write_str("\\u00")?; + output.write_char(HEXADECIMAL_DIGITS[(control >> 4) as usize] as char)?; + output.write_char(HEXADECIMAL_DIGITS[(control & 0xf) as usize] as char) + } + } +} + pub fn write_json_string_for_javascript( text: &str, output: &mut Output, ) -> fmt::Result { output.write_char('"')?; + let bytes = text.as_bytes(); let mut unescaped_run_start = 0; - for (index, character) in text.char_indices() { - let escape: Option<&str> = match character { - '"' => Some("\\\""), - '\\' => Some("\\\\"), - '\u{2028}' => Some("\\u2028"), - '\u{2029}' => Some("\\u2029"), - control if (control as u32) < 0x20 => None, - _ => continue, - }; - output.write_str(&text[unescaped_run_start..index])?; - match escape { - Some(sequence) => output.write_str(sequence)?, - None => { - let byte = character as u8; - output.write_str("\\u00")?; - output.write_char(HEXADECIMAL_DIGITS[(byte >> 4) as usize] as char)?; - output.write_char(HEXADECIMAL_DIGITS[(byte & 0xf) as usize] as char)?; - } + let mut position = position_of_json_string_special_byte(bytes, 0); + while let Some(&byte) = bytes.get(position) { + let separator_last_byte = separator_last_byte_at(bytes, position); + if byte == SEPARATOR_LEAD_BYTE && separator_last_byte.is_none() { + position = position_of_json_string_special_byte(bytes, position + 1); + continue; } - unescaped_run_start = index + character.len_utf8(); + output.write_str(&text[unescaped_run_start..position])?; + write_escape_sequence(byte, separator_last_byte, output)?; + position += if separator_last_byte.is_some() { 3 } else { 1 }; + unescaped_run_start = position; + position = position_of_json_string_special_byte(bytes, position); } output.write_str(&text[unescaped_run_start..])?; output.write_char('"') diff --git a/apps/heft-native/src/main.rs b/apps/heft-native/src/main.rs index 2a762f7608..64557cd6e7 100644 --- a/apps/heft-native/src/main.rs +++ b/apps/heft-native/src/main.rs @@ -1,4 +1,5 @@ #![deny(unsafe_code)] +#![cfg_attr(not(test), no_main)] mod builtin; mod cli; @@ -10,6 +11,7 @@ mod process; mod regex; mod run; mod schema; +mod simd; mod sys; mod terminal; mod version; @@ -19,7 +21,8 @@ use std::ffi::OsString; use run::HeftRunDecision; use version::HeftImplementationSelection; -fn main() { +#[cfg_attr(test, allow(dead_code))] +pub fn run_heft_command_line() -> ! { #[cfg(unix)] sys::reset_inherited_ignored_signals_like_node(); let command_line_arguments: Vec = std::env::args_os().skip(1).collect(); @@ -34,8 +37,7 @@ fn main() { process::exec_javascript_heft(&command_line_arguments) } HeftRunDecision::RunNatively(native_heft_run) => { - version::write_version_selector_banner(native_heft_context.version_selector_banner); - std::process::exit(run::run_heft_natively(native_heft_run)) + std::process::exit(run::run_heft_natively(native_heft_run, native_heft_context.version_selector_banner)) } HeftRunDecision::RunInNodeHost(node_host_plan) => host_link::run_in_node_host( &native_heft_context, diff --git a/apps/heft-native/src/run/mod.rs b/apps/heft-native/src/run/mod.rs index 0e76c1f66f..a60f3074ee 100644 --- a/apps/heft-native/src/run/mod.rs +++ b/apps/heft-native/src/run/mod.rs @@ -11,7 +11,7 @@ mod tests_tier_zero; use std::ffi::OsString; -use crate::builtin::{selections_are_deletable_without_permission_errors, AbsoluteFileSelection}; +use crate::builtin::{preflight_deletions, DeletionPlan}; use crate::cli::outcome::{CliOutcome, ParsedCommand, PrintedOutput}; use crate::config::cli_model_builder::build_cli_model; use crate::config::loader::{load_heft_configuration_and_then, HeftConfigurationRequest, LoadedHeftConfiguration}; @@ -19,7 +19,7 @@ use crate::config::model::{build_heft_configuration_model, HeftConfigurationMode use crate::config::package_json::PackageJsonLookup; use crate::host_link::NodeHostPlan; use crate::terminal::{console_supports_color_for_this_process, HeftConsole}; -use crate::version::NativeHeftContext; +use crate::version::{write_version_selector_banner, NativeHeftContext, VersionSelectorBanner}; use tier_zero_execution::TierZeroExit; use tier_zero_plan::TierZeroPlan; @@ -34,7 +34,7 @@ pub enum NativeHeftRun { PrintCliOutput(PrintedOutput), BuildWithBuiltinTasks { plan: Box, console: NativeConsoleSettings }, CleanProject { - selections: Vec, + deletion_plan: DeletionPlan, alias_expansion_message: Option, console: NativeConsoleSettings, }, @@ -64,11 +64,11 @@ fn decide_with_project(command_line_arguments: &[OsString], native_heft_context: return None; } let arguments: Vec<&str> = crate::cli::entry::command_line_strings(command_line_arguments)?; - let heft_package_folder = native_heft_context.companion_heft_package_folder.as_ref()?; - let heft_package_folder = std::fs::canonicalize(heft_package_folder).ok()?.to_str()?.to_owned(); - let heft_module_folder = format!("{heft_package_folder}/lib-commonjs/utilities"); let current_folder = std::env::current_dir().ok()?.to_str()?.to_owned(); - let mut lookup = PackageJsonLookup::default(); + let mut lookup = PackageJsonLookup::for_physical_current_folder(¤t_folder); + let heft_package_folder = native_heft_context.companion_heft_package_folder.as_ref()?.to_str()?; + let heft_package_folder = lookup.file_system.real_path(heft_package_folder).ok()?; + let heft_module_folder = format!("{heft_package_folder}/lib-commonjs/utilities"); let build_folder_path = lookup.try_get_package_folder_for(¤t_folder).ok()??; lookup.load_identity_for_folder(&build_folder_path).ok()?; let request = HeftConfigurationRequest { @@ -111,10 +111,12 @@ fn native_build_run( ) -> Option { if let Some(clean_options) = command_options::tier_zero_clean_options(command) { let selections = configured_project::native_clean_selections(model, &clean_options.selected_phase_indices)?; - let passes_preflight = process_environment::standard_input_is_the_null_device() - && selections_are_deletable_without_permission_errors(&selections); - return passes_preflight.then(|| NativeHeftRun::CleanProject { - selections, + if !process_environment::standard_input_is_the_null_device() { + return None; + } + let preflight_entries = Some(preflight_deletions(&selections)?); + return Some(NativeHeftRun::CleanProject { + deletion_plan: DeletionPlan { selections, preflight_entries }, alias_expansion_message: clean_options.alias_expansion_message, console: console_settings_for(invocation), }); @@ -143,21 +145,22 @@ fn console_settings_for(invocation: &HeftInvocation<'_>) -> NativeConsoleSetting } } -pub fn run_heft_natively(native_heft_run: NativeHeftRun) -> i32 { +pub fn run_heft_natively(native_heft_run: NativeHeftRun, version_selector_banner: VersionSelectorBanner) -> i32 { + let banner_text: &str = version_selector_banner.text_printed_by_javascript_version_selector(); match native_heft_run { - NativeHeftRun::PrintCliOutput(printed_output) => crate::cli::entry::write_printed_output(&printed_output), - NativeHeftRun::BuildWithBuiltinTasks { plan, console } => exit_code_of( - tier_zero_execution::execute_tier_zero_plan(*plan, &HeftConsole::new(console.supports_color)), - &console, - ), - NativeHeftRun::CleanProject { selections, alias_expansion_message, console } => exit_code_of( - tier_zero_execution::execute_tier_zero_clean( - &selections, - alias_expansion_message.as_deref(), - &HeftConsole::new(console.supports_color), - ), - &console, - ), + NativeHeftRun::PrintCliOutput(printed_output) => crate::cli::entry::write_printed_output_after(banner_text, &printed_output), + NativeHeftRun::BuildWithBuiltinTasks { plan, console } => { + write_version_selector_banner(version_selector_banner); + exit_code_of(tier_zero_execution::execute_tier_zero_plan(*plan, &HeftConsole::new(console.supports_color)), &console) + } + NativeHeftRun::CleanProject { deletion_plan, alias_expansion_message, console } => { + write_version_selector_banner(version_selector_banner); + let console_output: HeftConsole = HeftConsole::new(console.supports_color); + exit_code_of( + tier_zero_execution::execute_tier_zero_clean(deletion_plan, alias_expansion_message.as_deref(), &console_output), + &console, + ) + } } } diff --git a/apps/heft-native/src/run/tier_zero_execution.rs b/apps/heft-native/src/run/tier_zero_execution.rs index ad311f8534..b9602281b5 100644 --- a/apps/heft-native/src/run/tier_zero_execution.rs +++ b/apps/heft-native/src/run/tier_zero_execution.rs @@ -1,7 +1,7 @@ use std::time::Instant; use super::tier_zero_plan::{TierZeroPlan, TierZeroStep}; -use crate::builtin::{run_delete_operations, run_planned_builtin_task, AbsoluteFileSelection}; +use crate::builtin::{builtin_task_does_file_system_work, run_deletion_plan, run_planned_builtin_task, DeletionPlan}; use crate::terminal::{ bold, format_rounded_milliseconds_as_seconds, format_seconds_with_three_fraction_digits, green, red, ClosedOutput, HeftConsole, @@ -23,18 +23,19 @@ fn stop_if_output_closed(console: &HeftConsole) -> Result<(), TierZeroExit> { } pub fn execute_tier_zero_clean( - selections: &[AbsoluteFileSelection], + deletion_plan: DeletionPlan, alias_expansion_message: Option<&str>, console: &HeftConsole, ) -> TierZeroExit { if let Some(alias_expansion_message) = alias_expansion_message { console.write_line(alias_expansion_message); - if let Err(closed) = stop_if_output_closed(console) { - return closed; - } + } + console.flush(); + if let Err(closed) = stop_if_output_closed(console) { + return closed; } let run_started_at = Instant::now(); - let result = run_delete_operations(selections, &console.unprefixed_output()); + let result = run_deletion_plan(deletion_plan, &console.unprefixed_output()); if let Err(closed) = stop_if_output_closed(console) { return closed; } @@ -46,6 +47,7 @@ pub fn execute_tier_zero_clean( 1 } }; + console.flush(); stop_if_output_closed(console).map_or_else(|closed| closed, |()| TierZeroExit::Code(exit_code)) } @@ -69,18 +71,24 @@ fn execute_steps(plan: TierZeroPlan, console: &HeftConsole) -> Result { + TierZeroStep::StartPhase { phase_name, clean } => { phase_started_at = Instant::now(); console.write_line(&format!(" ---- {phase_name} started ---- ")); stop_if_output_closed(console)?; - match clean_selections { - Some(selections) => { - run_delete_operations(&selections, &console.scoped_logger_output(&format!("{phase_name}:clean"))) + match clean { + Some(deletion_plan) => { + if deletion_plan.does_file_system_work() { + flush_before_file_system_work(console)?; + } + run_deletion_plan(deletion_plan, &console.scoped_logger_output(&format!("{phase_name}:clean"))) } None => Ok(()), } } TierZeroStep::RunTask { logger_name, planned_task } => { + if builtin_task_does_file_system_work(&planned_task) { + flush_before_file_system_work(console)?; + } run_planned_builtin_task(planned_task, &console.scoped_logger_output(&logger_name)) } }; @@ -106,10 +114,16 @@ fn execute_steps(plan: TierZeroPlan, console: &HeftConsole) -> Result Result<(), TierZeroExit> { + console.flush(); + stop_if_output_closed(console) +} + fn write_summary(console: &HeftConsole, run_started_at: Instant, encountered_error: bool) { let run_duration_in_milliseconds = run_started_at.elapsed().as_secs_f64() * 1000.0; let finished_logging_word = if encountered_error { "Failed" } else { "Finished" }; diff --git a/apps/heft-native/src/run/tier_zero_plan.rs b/apps/heft-native/src/run/tier_zero_plan.rs index 1394ddaf43..e7fe9dc70a 100644 --- a/apps/heft-native/src/run/tier_zero_plan.rs +++ b/apps/heft-native/src/run/tier_zero_plan.rs @@ -1,7 +1,6 @@ use crate::builtin::{ - builtin_task_passes_preflight, builtin_task_touches_files, plan_builtin_task, plan_phase_clean, - selections_are_deletable_without_permission_errors, AbsoluteFileSelection, BuiltinTaskOptions, FileSelectionSpecifier, - PlannedBuiltinTask, + builtin_task_passes_preflight, plan_builtin_task, plan_phase_clean, plan_phase_clean_deletion, BuiltinTaskOptions, + DeletionPlan, FileSelectionSpecifier, ModifiedPaths, PlannedBuiltinTask, }; use crate::graph::{plan_sequential_operations, HeftOperation, PhaseShape}; @@ -27,7 +26,7 @@ pub struct NativeBuildRequest { } pub enum TierZeroStep { - StartPhase { phase_name: String, clean_selections: Option> }, + StartPhase { phase_name: String, clean: Option }, RunTask { logger_name: String, planned_task: PlannedBuiltinTask }, } @@ -72,33 +71,30 @@ pub fn plan_tier_zero_build(request: &NativeBuildRequest) -> Option = Vec::new(); - let mut files_may_have_changed = false; + let mut modified_paths = ModifiedPaths::default(); for operation in &sequential_plan.operations_in_execution_order { match *operation { HeftOperation::Phase { phase_index } => { let phase = &request.phases[phase_index]; - let clean_selections = if request.clean { + let clean = if request.clean { let selections = plan_phase_clean( &phase.clean_files, &request.build_folder_path, &temp_folder_path, &phase.phase_name, )?; - selections_are_deletable_without_permission_errors(&selections).then_some(())?; - files_may_have_changed = true; - Some(selections) + Some(plan_phase_clean_deletion(selections, &mut modified_paths)?) } else { None }; - steps.push((phase_index, TierZeroStep::StartPhase { phase_name: phase.phase_name.clone(), clean_selections })); + steps.push((phase_index, TierZeroStep::StartPhase { phase_name: phase.phase_name.clone(), clean })); } HeftOperation::Task { phase_index, task_index } => { let phase = &request.phases[phase_index]; let task = &phase.tasks[task_index]; let task_temp_folder_path = format!("{temp_folder_path}/{}/{}", phase.phase_name, task.task_name); let mut planned_task = plan_builtin_task(&task.options, &request.build_folder_path, &task_temp_folder_path)?; - builtin_task_passes_preflight(&mut planned_task, &temp_folder_path, !files_may_have_changed).then_some(())?; - files_may_have_changed |= builtin_task_touches_files(&planned_task); + builtin_task_passes_preflight(&mut planned_task, &temp_folder_path, &mut modified_paths).then_some(())?; let logger_name = format!("{}:{}", phase.phase_name, task.task_name); steps.push((phase_index, TierZeroStep::RunTask { logger_name, planned_task })); } diff --git a/apps/heft-native/src/simd/mod.rs b/apps/heft-native/src/simd/mod.rs new file mode 100644 index 0000000000..20c32256bf --- /dev/null +++ b/apps/heft-native/src/simd/mod.rs @@ -0,0 +1,68 @@ +use std::sync::OnceLock; + +mod scalar_scans; +#[cfg(all(test, target_arch = "x86_64"))] +mod tests_microbench; +#[cfg(all(test, target_arch = "x86_64"))] +mod tests_scan_equivalence; +#[cfg(all(test, target_arch = "x86_64"))] +mod tests_scan_fuzz; +#[cfg(target_arch = "x86_64")] +mod x86_64_avx2_scans; +#[cfg(target_arch = "x86_64")] +mod x86_64_dispatch; +#[cfg(target_arch = "x86_64")] +mod x86_64_sha256; +#[cfg(target_arch = "x86_64")] +mod x86_64_sse2_scans; +#[cfg(all(test, target_arch = "x86_64"))] +mod tests_sha256_equivalence; + +#[cfg_attr(not(target_arch = "x86_64"), allow(dead_code))] +const SIMD_KILL_SWITCH_VARIABLE: &str = "HEFT_NATIVE_NO_SIMD"; +#[cfg_attr(not(target_arch = "x86_64"), allow(dead_code))] +const SIMD_ENABLE_VARIABLE: &str = "HEFT_NATIVE_SIMD"; + +#[cfg_attr(not(target_arch = "x86_64"), allow(dead_code))] +pub fn simd_is_disabled_by_environment() -> bool { + static SIMD_IS_DISABLED: OnceLock = OnceLock::new(); + *SIMD_IS_DISABLED.get_or_init(|| { + std::env::var_os(SIMD_KILL_SWITCH_VARIABLE) + .is_some_and(|value| !value.is_empty() && value != "0") + || std::env::var_os(SIMD_ENABLE_VARIABLE).is_some_and(|value| value == "0") + }) +} + +#[cfg(target_arch = "x86_64")] +use x86_64_dispatch as selected_scans; + +#[cfg(not(target_arch = "x86_64"))] +use scalar_scans as selected_scans; + +#[inline] +pub fn position_after_json_whitespace(bytes: &[u8], from: usize) -> usize { + selected_scans::position_after_json_whitespace(bytes, from) +} + +#[inline] +pub fn position_of_json_string_special_byte(bytes: &[u8], from: usize) -> usize { + selected_scans::position_of_json_string_special_byte(bytes, from) +} + +#[inline] +pub fn position_of_line_comment_end_or_separator_lead_byte(bytes: &[u8], from: usize) -> usize { + selected_scans::position_of_line_comment_end_or_separator_lead_byte(bytes, from) +} + +#[inline] +pub fn position_of_block_comment_star(bytes: &[u8], from: usize) -> usize { + selected_scans::position_of_block_comment_star(bytes, from) +} + +#[cfg(target_arch = "x86_64")] +pub use x86_64_sha256::compress_sha256_blocks_if_the_cpu_can; + +#[cfg(not(target_arch = "x86_64"))] +pub fn compress_sha256_blocks_if_the_cpu_can(_state: &mut [u32; 8], _blocks: &[u8]) -> bool { + false +} diff --git a/apps/heft-native/src/simd/scalar_scans.rs b/apps/heft-native/src/simd/scalar_scans.rs new file mode 100644 index 0000000000..e66fdf025e --- /dev/null +++ b/apps/heft-native/src/simd/scalar_scans.rs @@ -0,0 +1,43 @@ +pub(super) const SEPARATOR_LEAD_BYTE: u8 = 0xe2; + +pub(super) fn is_json_whitespace(byte: u8) -> bool { + matches!(byte, b' ' | b'\t' | b'\n' | b'\r') +} + +pub(super) fn is_json_string_special_byte(byte: u8) -> bool { + byte == b'"' || byte == b'\\' || byte < 0x20 || byte == SEPARATOR_LEAD_BYTE +} + +pub(super) fn is_line_comment_end_or_separator_lead_byte(byte: u8) -> bool { + byte == b'\n' || byte == b'\r' || byte == SEPARATOR_LEAD_BYTE +} + +fn position_of_first_byte_matching(bytes: &[u8], from: usize, matches: fn(u8) -> bool) -> usize { + let mut position = from; + while let Some(&byte) = bytes.get(position) { + if matches(byte) { + break; + } + position += 1; + } + position +} + +pub(super) fn position_after_json_whitespace(bytes: &[u8], from: usize) -> usize { + position_of_first_byte_matching(bytes, from, |byte| !is_json_whitespace(byte)) +} + +pub(super) fn position_of_json_string_special_byte(bytes: &[u8], from: usize) -> usize { + position_of_first_byte_matching(bytes, from, is_json_string_special_byte) +} + +pub(super) fn position_of_line_comment_end_or_separator_lead_byte( + bytes: &[u8], + from: usize, +) -> usize { + position_of_first_byte_matching(bytes, from, is_line_comment_end_or_separator_lead_byte) +} + +pub(super) fn position_of_block_comment_star(bytes: &[u8], from: usize) -> usize { + position_of_first_byte_matching(bytes, from, |byte| byte == b'*') +} diff --git a/apps/heft-native/src/simd/tests_microbench.rs b/apps/heft-native/src/simd/tests_microbench.rs new file mode 100644 index 0000000000..f6a8d99d80 --- /dev/null +++ b/apps/heft-native/src/simd/tests_microbench.rs @@ -0,0 +1,80 @@ +#![allow(unsafe_code)] + +use std::time::Instant; + +use super::{scalar_scans, x86_64_avx2_scans, x86_64_sse2_scans}; + +type Scan = fn(&[u8], usize) -> usize; + +const BUFFER_LENGTH: usize = 1 << 20; +const MATCH_DISTANCES: [usize; 9] = [1, 2, 4, 8, 16, 32, 64, 256, 4096]; + +fn buffer_with_star_every(distance: usize) -> Vec { + (0..BUFFER_LENGTH) + .map(|index| { + if (index + 1) % distance == 0 { + b'*' + } else { + b'a' + } + }) + .collect() +} + +fn avx2_block_comment_star_scan(bytes: &[u8], from: usize) -> usize { + x86_64_avx2_scans::position_of_block_comment_star(bytes, from).unwrap_or(usize::MAX) +} + +fn timestamp_counter() -> u64 { + unsafe { std::arch::x86_64::_rdtsc() } +} + +fn nanoseconds_and_ticks_per_hop(scan: Scan, bytes: &[u8]) -> (f64, f64, usize) { + let mut best = (f64::MAX, f64::MAX); + let mut hops = 0; + for _ in 0..7 { + hops = 0; + let started = Instant::now(); + let started_ticks = timestamp_counter(); + let mut position = 0; + while position < bytes.len() { + position = std::hint::black_box(scan(std::hint::black_box(bytes), position)) + 1; + hops += 1; + } + let ticks = (timestamp_counter() - started_ticks) as f64; + let nanoseconds = started.elapsed().as_nanos() as f64; + if nanoseconds < best.0 { + best = (nanoseconds, ticks); + } + } + (best.0 / hops as f64, best.1 / hops as f64, hops) +} + +#[test] +#[ignore] +fn microbenchmark_scans_by_match_distance() { + let dispatched: Scan = super::position_of_block_comment_star; + let implementations: [(&str, Scan); 4] = [ + ("scalar", scalar_scans::position_of_block_comment_star), + ("sse2", x86_64_sse2_scans::position_of_block_comment_star), + ("avx2", avx2_block_comment_star_scan), + ("dispatched", dispatched), + ]; + for distance in MATCH_DISTANCES { + let bytes = buffer_with_star_every(distance); + let mut line = format!("distance {distance:5}:"); + let mut scalar_nanoseconds = 0.0; + for (name, scan) in implementations { + let (nanoseconds, ticks, _) = nanoseconds_and_ticks_per_hop(scan, &bytes); + if name == "scalar" { + scalar_nanoseconds = nanoseconds; + } + line.push_str(&format!( + " {name} {nanoseconds:8.2} ns/hop {:6.2} B/tick x{:5.2}", + distance as f64 / ticks, + scalar_nanoseconds / nanoseconds + )); + } + println!("{line}"); + } +} diff --git a/apps/heft-native/src/simd/tests_scan_equivalence.rs b/apps/heft-native/src/simd/tests_scan_equivalence.rs new file mode 100644 index 0000000000..3e9cb3be1b --- /dev/null +++ b/apps/heft-native/src/simd/tests_scan_equivalence.rs @@ -0,0 +1,113 @@ +use super::{scalar_scans, x86_64_avx2_scans, x86_64_sse2_scans}; + +type Scan = fn(&[u8], usize) -> usize; +type DetectedScan = fn(&[u8], usize) -> Option; + +const SCAN_IMPLEMENTATIONS: [(&str, Scan, Scan, DetectedScan, Scan); 4] = [ + ( + "whitespace", + scalar_scans::position_after_json_whitespace, + x86_64_sse2_scans::position_after_json_whitespace, + x86_64_avx2_scans::position_after_json_whitespace, + super::position_after_json_whitespace, + ), + ( + "string special", + scalar_scans::position_of_json_string_special_byte, + x86_64_sse2_scans::position_of_json_string_special_byte, + x86_64_avx2_scans::position_of_json_string_special_byte, + super::position_of_json_string_special_byte, + ), + ( + "line comment end", + scalar_scans::position_of_line_comment_end_or_separator_lead_byte, + x86_64_sse2_scans::position_of_line_comment_end_or_separator_lead_byte, + x86_64_avx2_scans::position_of_line_comment_end_or_separator_lead_byte, + super::position_of_line_comment_end_or_separator_lead_byte, + ), + ( + "block comment star", + scalar_scans::position_of_block_comment_star, + x86_64_sse2_scans::position_of_block_comment_star, + x86_64_avx2_scans::position_of_block_comment_star, + super::position_of_block_comment_star, + ), +]; + +const INTERESTING_BYTES: [u8; 24] = [ + b' ', b'\t', b'\n', b'\r', b'"', b'\\', b'*', b'/', 0x00, 0x08, 0x1f, 0x20, 0x21, b'a', 0x7f, + 0x80, 0xa8, 0xa9, 0xe1, 0xe2, 0xe3, 0xff, 0x0b, 0x0c, +]; + +struct PseudoRandomNumbers { + state: u64, +} + +impl PseudoRandomNumbers { + fn next_below(&mut self, bound: usize) -> usize { + self.state = self + .state + .wrapping_mul(6_364_136_223_846_793_005) + .wrapping_add(1_442_695_040_888_963_407); + ((self.state >> 33) as usize) % bound + } +} + +fn buffer_with_runs_of_interesting_bytes( + numbers: &mut PseudoRandomNumbers, + length: usize, +) -> Vec { + let mut buffer = Vec::with_capacity(length); + while buffer.len() < length { + let byte = INTERESTING_BYTES[numbers.next_below(INTERESTING_BYTES.len())]; + let run_length = (1 + numbers.next_below(70)).min(length - buffer.len()); + buffer.extend(std::iter::repeat_n(byte, run_length)); + } + buffer +} + +#[test] +fn simd_scans_find_the_same_position_as_the_scalar_scans_from_every_offset() { + let mut numbers = PseudoRandomNumbers { state: 0x5eed }; + for length in (0..=140).chain([255, 256, 257, 1023, 4099]) { + for _ in 0..8 { + let buffer = buffer_with_runs_of_interesting_bytes(&mut numbers, length); + let exact_size_buffer: Box<[u8]> = buffer.into_boxed_slice(); + for from in (0..=(length + 2).min(160)).chain([length.saturating_sub(33), length + 7]) { + for (name, scalar, sse2, avx2, dispatched) in SCAN_IMPLEMENTATIONS { + let expected = scalar(&exact_size_buffer, from); + assert_eq!( + sse2(&exact_size_buffer, from), + expected, + "{name} sse2 {length} {from}" + ); + if let Some(position) = avx2(&exact_size_buffer, from) { + assert_eq!(position, expected, "{name} avx2 {length} {from}"); + } + assert_eq!( + dispatched(&exact_size_buffer, from), + expected, + "{name} {length} {from}" + ); + } + } + } + } +} + +#[test] +fn the_selected_level_follows_the_kill_switch_and_cpu_detection() { + use super::x86_64_dispatch::{selected_level, LEVEL_AVX2, LEVEL_SCALAR, LEVEL_SSE2}; + let expected = if super::simd_is_disabled_by_environment() { + LEVEL_SCALAR + } else if std::arch::is_x86_feature_detected!("avx2") { + LEVEL_AVX2 + } else { + LEVEL_SSE2 + }; + println!( + "selected simd level {} (1 scalar, 2 sse2, 3 avx2)", + selected_level() + ); + assert_eq!(selected_level(), expected); +} diff --git a/apps/heft-native/src/simd/tests_scan_fuzz.rs b/apps/heft-native/src/simd/tests_scan_fuzz.rs new file mode 100644 index 0000000000..6906fad1db --- /dev/null +++ b/apps/heft-native/src/simd/tests_scan_fuzz.rs @@ -0,0 +1,87 @@ +use super::{scalar_scans, x86_64_avx2_scans, x86_64_sse2_scans}; + +type Scan = fn(&[u8], usize) -> usize; +type DetectedScan = fn(&[u8], usize) -> Option; + +const SCANS: [(Scan, Scan, DetectedScan, Scan); 4] = [ + ( + scalar_scans::position_after_json_whitespace, + x86_64_sse2_scans::position_after_json_whitespace, + x86_64_avx2_scans::position_after_json_whitespace, + super::position_after_json_whitespace, + ), + ( + scalar_scans::position_of_json_string_special_byte, + x86_64_sse2_scans::position_of_json_string_special_byte, + x86_64_avx2_scans::position_of_json_string_special_byte, + super::position_of_json_string_special_byte, + ), + ( + scalar_scans::position_of_line_comment_end_or_separator_lead_byte, + x86_64_sse2_scans::position_of_line_comment_end_or_separator_lead_byte, + x86_64_avx2_scans::position_of_line_comment_end_or_separator_lead_byte, + super::position_of_line_comment_end_or_separator_lead_byte, + ), + ( + scalar_scans::position_of_block_comment_star, + x86_64_sse2_scans::position_of_block_comment_star, + x86_64_avx2_scans::position_of_block_comment_star, + super::position_of_block_comment_star, + ), +]; + +const BYTE_DISTRIBUTIONS: [&[u8]; 5] = [ + b" \t\n\r", + b" \t\n\r\"\\*/a", + b"aaaaaaaaaaaaaaaaaaaaaaaaaaaaaa\"", + &[ + 0x00, 0x1f, 0x20, 0x7f, 0x80, 0xa8, 0xa9, 0xe1, 0xe2, 0xe3, 0xff, b'"', b'\\', b'*', b'\n', + ], + &[ + b'a', b'a', b'a', b'a', b'a', b'a', b'a', 0xe2, 0x80, 0xa8, b' ', b' ', b' ', b'\r', + ], +]; + +struct PseudoRandomNumbers { + state: u64, +} + +impl PseudoRandomNumbers { + fn next_below(&mut self, bound: usize) -> usize { + self.state ^= self.state << 13; + self.state ^= self.state >> 7; + self.state ^= self.state << 17; + (self.state % bound as u64) as usize + } +} + +#[test] +#[ignore] +fn ten_million_random_scans_agree_with_the_scalar_twins() { + let case_count: usize = std::env::var("A02_SCAN_FUZZ_CASES") + .ok() + .and_then(|value| value.parse().ok()) + .unwrap_or(10_000_000); + let mut numbers = PseudoRandomNumbers { + state: 0x2545_f491_4f6c_dd1d, + }; + let mut mismatches = 0; + for case in 0..case_count { + let distribution = BYTE_DISTRIBUTIONS[numbers.next_below(BYTE_DISTRIBUTIONS.len())]; + let length = numbers.next_below(300); + let buffer: Box<[u8]> = (0..length) + .map(|_| distribution[numbers.next_below(distribution.len())]) + .collect(); + let from = numbers.next_below(length + 3); + let (scalar, sse2, avx2, dispatched) = SCANS[case % SCANS.len()]; + let expected = scalar(&buffer, from); + let agrees = sse2(&buffer, from) == expected + && avx2(&buffer, from).is_none_or(|position| position == expected) + && dispatched(&buffer, from) == expected; + if !agrees { + mismatches += 1; + } + } + println!("scan fuzz: {case_count} cases, {mismatches} mismatches"); + assert_eq!(mismatches, 0); +} diff --git a/apps/heft-native/src/simd/tests_sha256_equivalence.rs b/apps/heft-native/src/simd/tests_sha256_equivalence.rs new file mode 100644 index 0000000000..0aee447591 --- /dev/null +++ b/apps/heft-native/src/simd/tests_sha256_equivalence.rs @@ -0,0 +1,61 @@ +use super::compress_sha256_blocks_if_the_cpu_can; +use crate::builtin::compress_sha256_blocks_without_simd; + +const INITIAL_STATE: [u32; 8] = [0x6a09e667, 0xbb67ae85, 0x3c6ef372, 0xa54ff53a, 0x510e527f, 0x9b05688c, 0x1f83d9ab, 0x5be0cd19]; + +fn pseudo_random_bytes(seed: u64, length: usize) -> Vec { + let mut generator_state = seed.wrapping_mul(0x9e37_79b9_7f4a_7c15) | 1; + (0..length) + .map(|_| { + generator_state ^= generator_state << 13; + generator_state ^= generator_state >> 7; + generator_state ^= generator_state << 17; + (generator_state >> 24) as u8 + }) + .collect() +} + +#[test] +fn sha_extensions_compress_every_block_count_exactly_like_the_scalar_twin() { + if !std::arch::is_x86_feature_detected!("sha") || super::simd_is_disabled_by_environment() { + return; + } + for block_count in 0..=40 { + for seed in 0..8u64 { + let whole_blocks = pseudo_random_bytes(seed * 1000 + block_count as u64, block_count * 64); + for trailing_byte_count in [0, 1, 63] { + let mut input = whole_blocks.clone(); + input.extend(std::iter::repeat_n(0xA5, trailing_byte_count)); + let mut simd_state = INITIAL_STATE; + let mut scalar_state = INITIAL_STATE; + assert!(compress_sha256_blocks_if_the_cpu_can(&mut simd_state, &input)); + compress_sha256_blocks_without_simd(&mut scalar_state, &input); + assert_eq!(simd_state, scalar_state, "{block_count} blocks + {trailing_byte_count} bytes, seed {seed}"); + } + } + } +} + +#[test] +fn sha_extensions_accept_every_input_offset_and_adversarial_words() { + if !std::arch::is_x86_feature_detected!("sha") || super::simd_is_disabled_by_environment() { + return; + } + let buffer = pseudo_random_bytes(7, 64 * 9 + 64); + for offset in 0..64 { + let blocks = &buffer[offset..offset + 64 * 9]; + let mut simd_state = INITIAL_STATE; + let mut scalar_state = INITIAL_STATE; + assert!(compress_sha256_blocks_if_the_cpu_can(&mut simd_state, blocks)); + compress_sha256_blocks_without_simd(&mut scalar_state, blocks); + assert_eq!(simd_state, scalar_state, "offset {offset}"); + } + for fill_byte in [0x00, 0xFF, 0x80, 0x7F] { + let blocks = vec![fill_byte; 64 * 3]; + let mut simd_state = [u32::MAX; 8]; + let mut scalar_state = [u32::MAX; 8]; + assert!(compress_sha256_blocks_if_the_cpu_can(&mut simd_state, &blocks)); + compress_sha256_blocks_without_simd(&mut scalar_state, &blocks); + assert_eq!(simd_state, scalar_state, "fill {fill_byte:#x}"); + } +} diff --git a/apps/heft-native/src/simd/x86_64_avx2_scans.rs b/apps/heft-native/src/simd/x86_64_avx2_scans.rs new file mode 100644 index 0000000000..5ff1964b2d --- /dev/null +++ b/apps/heft-native/src/simd/x86_64_avx2_scans.rs @@ -0,0 +1,108 @@ +#![allow(unsafe_code)] + +use std::arch::x86_64::{ + __m256i, _mm256_cmpeq_epi8, _mm256_loadu_si256, _mm256_max_epu8, _mm256_movemask_epi8, + _mm256_or_si256, _mm256_set1_epi8, +}; + +use super::scalar_scans::SEPARATOR_LEAD_BYTE; +use super::x86_64_sse2_scans; + +const LANE_COUNT: usize = 32; + +#[target_feature(enable = "avx2")] +fn load_thirty_two_bytes(chunk: &[u8; LANE_COUNT]) -> __m256i { + unsafe { _mm256_loadu_si256(chunk.as_ptr().cast()) } +} + +#[target_feature(enable = "avx2")] +fn lanes_equal_to(chunk: __m256i, byte: u8) -> __m256i { + _mm256_cmpeq_epi8(chunk, _mm256_set1_epi8(byte as i8)) +} + +#[target_feature(enable = "avx2")] +fn lane_bits(lanes: __m256i) -> u32 { + _mm256_movemask_epi8(lanes) as u32 +} + +#[target_feature(enable = "avx2")] +fn non_json_whitespace_lane_bits(chunk: __m256i) -> u32 { + let space_or_tab = _mm256_or_si256(lanes_equal_to(chunk, b' '), lanes_equal_to(chunk, b'\t')); + let line_breaks = _mm256_or_si256(lanes_equal_to(chunk, b'\n'), lanes_equal_to(chunk, b'\r')); + !lane_bits(_mm256_or_si256(space_or_tab, line_breaks)) +} + +#[target_feature(enable = "avx2")] +fn json_string_special_lane_bits(chunk: __m256i) -> u32 { + let highest_control = _mm256_set1_epi8(0x1f); + let control = _mm256_cmpeq_epi8(_mm256_max_epu8(chunk, highest_control), highest_control); + let quote_or_backslash = + _mm256_or_si256(lanes_equal_to(chunk, b'"'), lanes_equal_to(chunk, b'\\')); + let separator_lead = lanes_equal_to(chunk, SEPARATOR_LEAD_BYTE); + lane_bits(_mm256_or_si256( + _mm256_or_si256(control, quote_or_backslash), + separator_lead, + )) +} + +#[target_feature(enable = "avx2")] +fn line_comment_end_or_separator_lead_lane_bits(chunk: __m256i) -> u32 { + let line_breaks = _mm256_or_si256(lanes_equal_to(chunk, b'\n'), lanes_equal_to(chunk, b'\r')); + lane_bits(_mm256_or_si256( + line_breaks, + lanes_equal_to(chunk, SEPARATOR_LEAD_BYTE), + )) +} + +#[target_feature(enable = "avx2")] +fn star_lane_bits(chunk: __m256i) -> u32 { + lane_bits(lanes_equal_to(chunk, b'*')) +} + +macro_rules! scan_with_avx2_then_sse2 { + ($scan:ident, $scan_with_avx2:ident, $matching_lane_bits:ident) => { + #[target_feature(enable = "avx2")] + pub(super) fn $scan_with_avx2(bytes: &[u8], from: usize) -> usize { + let Some(rest) = bytes.get(from..) else { + return from; + }; + let (whole_chunks, _) = rest.as_chunks::(); + for (chunk_index, chunk) in whole_chunks.iter().enumerate() { + let matching = $matching_lane_bits(load_thirty_two_bytes(chunk)); + if matching != 0 { + return from + chunk_index * LANE_COUNT + matching.trailing_zeros() as usize; + } + } + x86_64_sse2_scans::$scan(bytes, from + whole_chunks.len() * LANE_COUNT) + } + + #[cfg(test)] + pub(super) fn $scan(bytes: &[u8], from: usize) -> Option { + if !std::arch::is_x86_feature_detected!("avx2") { + return None; + } + Some(unsafe { $scan_with_avx2(bytes, from) }) + } + }; +} + +scan_with_avx2_then_sse2!( + position_after_json_whitespace, + position_after_json_whitespace_with_avx2, + non_json_whitespace_lane_bits +); +scan_with_avx2_then_sse2!( + position_of_json_string_special_byte, + position_of_json_string_special_byte_with_avx2, + json_string_special_lane_bits +); +scan_with_avx2_then_sse2!( + position_of_line_comment_end_or_separator_lead_byte, + position_of_line_comment_end_or_separator_lead_byte_with_avx2, + line_comment_end_or_separator_lead_lane_bits +); +scan_with_avx2_then_sse2!( + position_of_block_comment_star, + position_of_block_comment_star_with_avx2, + star_lane_bits +); diff --git a/apps/heft-native/src/simd/x86_64_dispatch.rs b/apps/heft-native/src/simd/x86_64_dispatch.rs new file mode 100644 index 0000000000..4a852eadb7 --- /dev/null +++ b/apps/heft-native/src/simd/x86_64_dispatch.rs @@ -0,0 +1,82 @@ +#![allow(unsafe_code)] + +use std::sync::atomic::{AtomicU8, Ordering}; + +use super::{scalar_scans, x86_64_avx2_scans, x86_64_sse2_scans}; + +const LEVEL_NOT_SELECTED_YET: u8 = 0; +pub(super) const LEVEL_SCALAR: u8 = 1; +pub(super) const LEVEL_SSE2: u8 = 2; +pub(super) const LEVEL_AVX2: u8 = 3; + +static SELECTED_LEVEL: AtomicU8 = AtomicU8::new(LEVEL_NOT_SELECTED_YET); + +#[cold] +#[inline(never)] +fn select_level_once() -> u8 { + let level = if super::simd_is_disabled_by_environment() { + LEVEL_SCALAR + } else if std::arch::is_x86_feature_detected!("avx2") { + LEVEL_AVX2 + } else { + LEVEL_SSE2 + }; + SELECTED_LEVEL.store(level, Ordering::Relaxed); + level +} + +#[inline(always)] +pub(super) fn selected_level() -> u8 { + match SELECTED_LEVEL.load(Ordering::Relaxed) { + LEVEL_NOT_SELECTED_YET => select_level_once(), + level => level, + } +} + +macro_rules! dispatch_scan { + ($scan:ident, $stops_at:expr, $within_leading_chunks:ident, $scan_with_avx2:ident) => { + #[inline(always)] + pub(super) fn $scan(bytes: &[u8], from: usize) -> usize { + let level = selected_level(); + if level == LEVEL_SCALAR { + return scalar_scans::$scan(bytes, from); + } + match bytes.get(from) { + Some(&byte) if !$stops_at(byte) => {} + _ => return from, + } + match x86_64_sse2_scans::$within_leading_chunks(bytes, from + 1) { + Ok(position) => position, + Err(long_run_from) if level == LEVEL_AVX2 => unsafe { + x86_64_avx2_scans::$scan_with_avx2(bytes, long_run_from) + }, + Err(long_run_from) => x86_64_sse2_scans::$scan(bytes, long_run_from), + } + } + }; +} + +dispatch_scan!( + position_after_json_whitespace, + |byte| !scalar_scans::is_json_whitespace(byte), + position_after_json_whitespace_within_leading_chunks, + position_after_json_whitespace_with_avx2 +); +dispatch_scan!( + position_of_json_string_special_byte, + scalar_scans::is_json_string_special_byte, + position_of_json_string_special_byte_within_leading_chunks, + position_of_json_string_special_byte_with_avx2 +); +dispatch_scan!( + position_of_line_comment_end_or_separator_lead_byte, + scalar_scans::is_line_comment_end_or_separator_lead_byte, + position_of_line_comment_end_or_separator_lead_byte_within_leading_chunks, + position_of_line_comment_end_or_separator_lead_byte_with_avx2 +); +dispatch_scan!( + position_of_block_comment_star, + |byte| byte == b'*', + position_of_block_comment_star_within_leading_chunks, + position_of_block_comment_star_with_avx2 +); diff --git a/apps/heft-native/src/simd/x86_64_sha256.rs b/apps/heft-native/src/simd/x86_64_sha256.rs new file mode 100644 index 0000000000..70c656204c --- /dev/null +++ b/apps/heft-native/src/simd/x86_64_sha256.rs @@ -0,0 +1,124 @@ +#![allow(unsafe_code)] + +use std::arch::x86_64::{ + __m128i, _mm_add_epi32, _mm_alignr_epi8, _mm_blend_epi16, _mm_loadu_si128, _mm_set_epi32, _mm_set_epi64x, + _mm_sha256msg1_epu32, _mm_sha256msg2_epu32, _mm_sha256rnds2_epu32, _mm_shuffle_epi32, _mm_shuffle_epi8, + _mm_storeu_si128, +}; +use std::sync::atomic::{AtomicU8, Ordering}; + +const SHA_EXTENSIONS_NOT_DETECTED_YET: u8 = 0; +const SHA_EXTENSIONS_UNAVAILABLE: u8 = 1; +const SHA_EXTENSIONS_AVAILABLE: u8 = 2; + +static SHA_EXTENSIONS: AtomicU8 = AtomicU8::new(SHA_EXTENSIONS_NOT_DETECTED_YET); + +const ROUND_CONSTANTS: [u32; 64] = [ + 0x428a2f98, 0x71374491, 0xb5c0fbcf, 0xe9b5dba5, 0x3956c25b, 0x59f111f1, 0x923f82a4, 0xab1c5ed5, + 0xd807aa98, 0x12835b01, 0x243185be, 0x550c7dc3, 0x72be5d74, 0x80deb1fe, 0x9bdc06a7, 0xc19bf174, + 0xe49b69c1, 0xefbe4786, 0x0fc19dc6, 0x240ca1cc, 0x2de92c6f, 0x4a7484aa, 0x5cb0a9dc, 0x76f988da, + 0x983e5152, 0xa831c66d, 0xb00327c8, 0xbf597fc7, 0xc6e00bf3, 0xd5a79147, 0x06ca6351, 0x14292967, + 0x27b70a85, 0x2e1b2138, 0x4d2c6dfc, 0x53380d13, 0x650a7354, 0x766a0abb, 0x81c2c92e, 0x92722c85, + 0xa2bfe8a1, 0xa81a664b, 0xc24b8b70, 0xc76c51a3, 0xd192e819, 0xd6990624, 0xf40e3585, 0x106aa070, + 0x19a4c116, 0x1e376c08, 0x2748774c, 0x34b0bcb5, 0x391c0cb3, 0x4ed8aa4a, 0x5b9cca4f, 0x682e6ff3, + 0x748f82ee, 0x78a5636f, 0x84c87814, 0x8cc70208, 0x90befffa, 0xa4506ceb, 0xbef9a3f7, 0xc67178f2, +]; + +#[cold] +#[inline(never)] +fn detect_sha_extensions_once() -> u8 { + let sha_extensions = if !super::simd_is_disabled_by_environment() + && std::arch::is_x86_feature_detected!("sha") + && std::arch::is_x86_feature_detected!("sse2") + && std::arch::is_x86_feature_detected!("ssse3") + && std::arch::is_x86_feature_detected!("sse4.1") + { + SHA_EXTENSIONS_AVAILABLE + } else { + SHA_EXTENSIONS_UNAVAILABLE + }; + SHA_EXTENSIONS.store(sha_extensions, Ordering::Relaxed); + sha_extensions +} + +pub fn compress_sha256_blocks_if_the_cpu_can(state: &mut [u32; 8], blocks: &[u8]) -> bool { + let sha_extensions = match SHA_EXTENSIONS.load(Ordering::Relaxed) { + SHA_EXTENSIONS_NOT_DETECTED_YET => detect_sha_extensions_once(), + sha_extensions => sha_extensions, + }; + if sha_extensions != SHA_EXTENSIONS_AVAILABLE { + return false; + } + unsafe { compress_sha256_blocks_with_sha_extensions(state, blocks) }; + true +} + +#[target_feature(enable = "sha,sse2,ssse3,sse4.1")] +fn load_big_endian_words(group_bytes: &[u8; 16], byte_swap_mask: __m128i) -> __m128i { + let little_endian_words = unsafe { _mm_loadu_si128(group_bytes.as_ptr().cast()) }; + _mm_shuffle_epi8(little_endian_words, byte_swap_mask) +} + +#[target_feature(enable = "sha,sse2,ssse3,sse4.1")] +fn schedule_next_words(words_0: __m128i, words_1: __m128i, words_2: __m128i, words_3: __m128i) -> __m128i { + let partially_scheduled_words = _mm_sha256msg1_epu32(words_0, words_1); + let words_shifted_by_one = _mm_alignr_epi8(words_3, words_2, 4); + _mm_sha256msg2_epu32(_mm_add_epi32(partially_scheduled_words, words_shifted_by_one), words_3) +} + +#[target_feature(enable = "sha,sse2,ssse3,sse4.1")] +fn run_four_rounds(abef_state: &mut __m128i, cdgh_state: &mut __m128i, words: __m128i, word_group_index: usize) { + let constants = &ROUND_CONSTANTS[4 * word_group_index..4 * word_group_index + 4]; + let round_constants = + _mm_set_epi32(constants[3] as i32, constants[2] as i32, constants[1] as i32, constants[0] as i32); + let words_plus_constants = _mm_add_epi32(words, round_constants); + *cdgh_state = _mm_sha256rnds2_epu32(*cdgh_state, *abef_state, words_plus_constants); + let upper_words_plus_constants = _mm_shuffle_epi32(words_plus_constants, 0x0E); + *abef_state = _mm_sha256rnds2_epu32(*abef_state, *cdgh_state, upper_words_plus_constants); +} + +#[target_feature(enable = "sha,sse2,ssse3,sse4.1")] +fn compress_sha256_blocks_with_sha_extensions(state: &mut [u32; 8], blocks: &[u8]) { + let byte_swap_mask = _mm_set_epi64x(0x0c0d_0e0f_0809_0a0b, 0x0405_0607_0001_0203); + let (dcba_state, hgfe_state) = unsafe { + (_mm_loadu_si128(state[0..4].as_ptr().cast()), _mm_loadu_si128(state[4..8].as_ptr().cast())) + }; + let cdab_state = _mm_shuffle_epi32(dcba_state, 0xB1); + let efgh_state = _mm_shuffle_epi32(hgfe_state, 0x1B); + let mut abef_state = _mm_alignr_epi8(cdab_state, efgh_state, 8); + let mut cdgh_state = _mm_blend_epi16(efgh_state, cdab_state, 0xF0); + let (whole_blocks, _) = blocks.as_chunks::<64>(); + for block in whole_blocks { + let (abef_before_block, cdgh_before_block) = (abef_state, cdgh_state); + let ([first_group, second_group, third_group, fourth_group], _) = block.as_chunks::<16>() else { + break; + }; + let mut words = [ + load_big_endian_words(first_group, byte_swap_mask), + load_big_endian_words(second_group, byte_swap_mask), + load_big_endian_words(third_group, byte_swap_mask), + load_big_endian_words(fourth_group, byte_swap_mask), + ]; + for word_group_index in 0..16 { + if word_group_index >= 4 { + words[word_group_index % 4] = schedule_next_words( + words[word_group_index % 4], + words[(word_group_index + 1) % 4], + words[(word_group_index + 2) % 4], + words[(word_group_index + 3) % 4], + ); + } + run_four_rounds(&mut abef_state, &mut cdgh_state, words[word_group_index % 4], word_group_index); + } + abef_state = _mm_add_epi32(abef_state, abef_before_block); + cdgh_state = _mm_add_epi32(cdgh_state, cdgh_before_block); + } + let feba_state = _mm_shuffle_epi32(abef_state, 0x1B); + let dchg_state = _mm_shuffle_epi32(cdgh_state, 0xB1); + let dcba_state = _mm_blend_epi16(feba_state, dchg_state, 0xF0); + let hgfe_state = _mm_alignr_epi8(dchg_state, feba_state, 8); + unsafe { + _mm_storeu_si128(state[0..4].as_mut_ptr().cast(), dcba_state); + _mm_storeu_si128(state[4..8].as_mut_ptr().cast(), hgfe_state); + } +} diff --git a/apps/heft-native/src/simd/x86_64_sse2_scans.rs b/apps/heft-native/src/simd/x86_64_sse2_scans.rs new file mode 100644 index 0000000000..a441fe5630 --- /dev/null +++ b/apps/heft-native/src/simd/x86_64_sse2_scans.rs @@ -0,0 +1,198 @@ +#![allow(unsafe_code)] + +use std::arch::x86_64::{ + __m128i, _mm_cmpeq_epi8, _mm_loadu_si128, _mm_max_epu8, _mm_movemask_epi8, _mm_or_si128, + _mm_set1_epi8, +}; + +use super::scalar_scans; + +const LANE_COUNT: usize = 16; +const LEADING_CHUNK_COUNT: usize = 4; +const ALL_LANES: u32 = 0xffff; +const _: () = assert!(cfg!(target_feature = "sse2")); + +fn load_sixteen_bytes(chunk: &[u8; LANE_COUNT]) -> __m128i { + unsafe { _mm_loadu_si128(chunk.as_ptr().cast()) } +} + +fn lanes_equal_to(chunk: __m128i, byte: u8) -> __m128i { + unsafe { _mm_cmpeq_epi8(chunk, _mm_set1_epi8(byte as i8)) } +} + +fn lane_bits(lanes: __m128i) -> u32 { + unsafe { _mm_movemask_epi8(lanes) as u32 } +} + +fn either_lane(first: __m128i, second: __m128i) -> __m128i { + unsafe { _mm_or_si128(first, second) } +} + +fn lanes_at_most(chunk: __m128i, byte: u8) -> __m128i { + unsafe { + _mm_cmpeq_epi8( + _mm_max_epu8(chunk, _mm_set1_epi8(byte as i8)), + _mm_set1_epi8(byte as i8), + ) + } +} + +fn json_whitespace_lane_bits(chunk: __m128i) -> u32 { + let space_or_tab = either_lane(lanes_equal_to(chunk, b' '), lanes_equal_to(chunk, b'\t')); + let line_breaks = either_lane(lanes_equal_to(chunk, b'\n'), lanes_equal_to(chunk, b'\r')); + lane_bits(either_lane(space_or_tab, line_breaks)) +} + +fn json_string_special_lane_bits(chunk: __m128i) -> u32 { + let control = lanes_at_most(chunk, 0x1f); + let quote_or_backslash = either_lane(lanes_equal_to(chunk, b'"'), lanes_equal_to(chunk, b'\\')); + let separator_lead = lanes_equal_to(chunk, scalar_scans::SEPARATOR_LEAD_BYTE); + lane_bits(either_lane( + either_lane(control, quote_or_backslash), + separator_lead, + )) +} + +fn line_comment_end_or_separator_lead_lane_bits(chunk: __m128i) -> u32 { + let line_breaks = either_lane(lanes_equal_to(chunk, b'\n'), lanes_equal_to(chunk, b'\r')); + let separator_lead = lanes_equal_to(chunk, scalar_scans::SEPARATOR_LEAD_BYTE); + lane_bits(either_lane(line_breaks, separator_lead)) +} + +fn star_lane_bits(chunk: __m128i) -> u32 { + lane_bits(lanes_equal_to(chunk, b'*')) +} + +fn position_of_first_matching_lane( + bytes: &[u8], + from: usize, + matching_lane_bits: impl Fn(__m128i) -> u32, + scalar_scan: fn(&[u8], usize) -> usize, +) -> usize { + let Some(rest) = bytes.get(from..) else { + return from; + }; + let (whole_chunks, _) = rest.as_chunks::(); + for (chunk_index, chunk) in whole_chunks.iter().enumerate() { + let matching = matching_lane_bits(load_sixteen_bytes(chunk)); + if matching != 0 { + return from + chunk_index * LANE_COUNT + matching.trailing_zeros() as usize; + } + } + scalar_scan(bytes, from + whole_chunks.len() * LANE_COUNT) +} + +#[inline(always)] +fn position_within_leading_chunks( + bytes: &[u8], + from: usize, + matching_lane_bits: impl Fn(__m128i) -> u32, + scalar_scan: fn(&[u8], usize) -> usize, +) -> Result { + let mut position = from; + for _ in 0..LEADING_CHUNK_COUNT { + let Some(chunk) = bytes + .get(position..) + .and_then(<[u8]>::first_chunk::) + else { + return Ok(scalar_scan(bytes, position)); + }; + let matching = matching_lane_bits(load_sixteen_bytes(chunk)); + if matching != 0 { + return Ok(position + matching.trailing_zeros() as usize); + } + position += LANE_COUNT; + } + Err(position) +} + +#[inline(always)] +pub(super) fn position_after_json_whitespace_within_leading_chunks( + bytes: &[u8], + from: usize, +) -> Result { + position_within_leading_chunks( + bytes, + from, + |chunk| !json_whitespace_lane_bits(chunk) & ALL_LANES, + scalar_scans::position_after_json_whitespace, + ) +} + +#[inline(always)] +pub(super) fn position_of_json_string_special_byte_within_leading_chunks( + bytes: &[u8], + from: usize, +) -> Result { + position_within_leading_chunks( + bytes, + from, + json_string_special_lane_bits, + scalar_scans::position_of_json_string_special_byte, + ) +} + +#[inline(always)] +pub(super) fn position_of_line_comment_end_or_separator_lead_byte_within_leading_chunks( + bytes: &[u8], + from: usize, +) -> Result { + position_within_leading_chunks( + bytes, + from, + line_comment_end_or_separator_lead_lane_bits, + scalar_scans::position_of_line_comment_end_or_separator_lead_byte, + ) +} + +#[inline(always)] +pub(super) fn position_of_block_comment_star_within_leading_chunks( + bytes: &[u8], + from: usize, +) -> Result { + position_within_leading_chunks( + bytes, + from, + star_lane_bits, + scalar_scans::position_of_block_comment_star, + ) +} + +pub(super) fn position_after_json_whitespace(bytes: &[u8], from: usize) -> usize { + position_of_first_matching_lane( + bytes, + from, + |chunk| !json_whitespace_lane_bits(chunk) & ALL_LANES, + scalar_scans::position_after_json_whitespace, + ) +} + +pub(super) fn position_of_json_string_special_byte(bytes: &[u8], from: usize) -> usize { + position_of_first_matching_lane( + bytes, + from, + json_string_special_lane_bits, + scalar_scans::position_of_json_string_special_byte, + ) +} + +pub(super) fn position_of_line_comment_end_or_separator_lead_byte( + bytes: &[u8], + from: usize, +) -> usize { + position_of_first_matching_lane( + bytes, + from, + line_comment_end_or_separator_lead_lane_bits, + scalar_scans::position_of_line_comment_end_or_separator_lead_byte, + ) +} + +pub(super) fn position_of_block_comment_star(bytes: &[u8], from: usize) -> usize { + position_of_first_matching_lane( + bytes, + from, + star_lane_bits, + scalar_scans::position_of_block_comment_star, + ) +} diff --git a/apps/heft-native/src/sys/mod.rs b/apps/heft-native/src/sys/mod.rs index d372bb908a..50c5372b71 100644 --- a/apps/heft-native/src/sys/mod.rs +++ b/apps/heft-native/src/sys/mod.rs @@ -6,12 +6,23 @@ mod tests_allocation_regressions; #[cfg(target_os = "linux")] #[allow(unsafe_code)] mod file_descriptor_flags; +#[cfg(not(test))] +#[allow(unsafe_code)] +mod process_entry; #[cfg(unix)] #[allow(unsafe_code)] mod signal_dispositions; #[cfg(unix)] #[allow(unsafe_code)] mod signals_and_identity; +#[cfg(all(unix, not(test)))] +#[allow(unsafe_code)] +mod standard_streams; +#[cfg(not(target_os = "linux"))] +mod no_worker_processes; +#[cfg(target_os = "linux")] +#[allow(unsafe_code)] +mod worker_processes; #[cfg(target_os = "linux")] pub use file_descriptor_flags::let_executed_program_inherit_file; @@ -22,3 +33,15 @@ pub use signals_and_identity::{ effective_user_id, forward_interrupt_and_termination_signals_to_warm_host, signal_forwarded_to_warm_host, terminate_by_signal, }; +#[cfg(all(unix, not(test)))] +pub use standard_streams::reopen_closed_standard_streams_on_the_null_device; +#[cfg(not(target_os = "linux"))] +pub use no_worker_processes::{ + exit_worker_process_immediately, fork_worker_process_of_this_single_threaded_process, wait_for_worker_process, + ForkedWorker, +}; +#[cfg(target_os = "linux")] +pub use worker_processes::{ + exit_worker_process_immediately, fork_worker_process_of_this_single_threaded_process, wait_for_worker_process, + ForkedWorker, +}; diff --git a/apps/heft-native/src/sys/no_worker_processes.rs b/apps/heft-native/src/sys/no_worker_processes.rs new file mode 100644 index 0000000000..70f626f894 --- /dev/null +++ b/apps/heft-native/src/sys/no_worker_processes.rs @@ -0,0 +1,17 @@ +use std::io; + +#[allow(dead_code)] +pub enum ForkedWorker { + InParent { worker_process_id: i32 }, + InWorker, +} + +pub fn fork_worker_process_of_this_single_threaded_process() -> io::Result { + Err(io::Error::from(io::ErrorKind::Unsupported)) +} + +pub fn exit_worker_process_immediately(status: i32) -> ! { + std::process::exit(status) +} + +pub fn wait_for_worker_process(_worker_process_id: i32) {} diff --git a/apps/heft-native/src/sys/process_entry.rs b/apps/heft-native/src/sys/process_entry.rs new file mode 100644 index 0000000000..6006f2f6a5 --- /dev/null +++ b/apps/heft-native/src/sys/process_entry.rs @@ -0,0 +1,8 @@ +use std::ffi::{c_char, c_int}; + +#[no_mangle] +pub extern "C" fn main(_argument_count: c_int, _arguments: *const *const c_char) -> c_int { + #[cfg(unix)] + super::reopen_closed_standard_streams_on_the_null_device(); + crate::run_heft_command_line() +} diff --git a/apps/heft-native/src/sys/signal_dispositions.rs b/apps/heft-native/src/sys/signal_dispositions.rs index a6c2c128ff..11dfb73f3c 100644 --- a/apps/heft-native/src/sys/signal_dispositions.rs +++ b/apps/heft-native/src/sys/signal_dispositions.rs @@ -13,10 +13,7 @@ extern "C" { pub fn reset_inherited_ignored_signals_like_node() { for signal_number in 1..=HIGHEST_STANDARD_SIGNAL { - if signal_number == SIGNAL_BROKEN_PIPE { - continue; - } - if signal_number == SIGNAL_FILE_SIZE_LIMIT_EXCEEDED { + if signal_number == SIGNAL_BROKEN_PIPE || signal_number == SIGNAL_FILE_SIZE_LIMIT_EXCEEDED { unsafe { signal(signal_number, IGNORED_SIGNAL_DISPOSITION) }; continue; } diff --git a/apps/heft-native/src/sys/standard_streams.rs b/apps/heft-native/src/sys/standard_streams.rs new file mode 100644 index 0000000000..aff252fe10 --- /dev/null +++ b/apps/heft-native/src/sys/standard_streams.rs @@ -0,0 +1,76 @@ +use std::ffi::{c_char, c_int}; + +const OPEN_FOR_READING_AND_WRITING: c_int = 2; +const NULL_DEVICE_PATH: &[u8] = b"/dev/null\0"; +const STANDARD_STREAM_FILE_DESCRIPTORS: [c_int; 3] = [0, 1, 2]; + +extern "C" { + fn open(path: *const c_char, flags: c_int, ...) -> c_int; +} + +pub fn reopen_closed_standard_streams_on_the_null_device() { + for file_descriptor in closed_standard_streams().into_iter().flatten() { + let reopened_file_descriptor = unsafe { + open( + NULL_DEVICE_PATH.as_ptr().cast(), + OPEN_FOR_READING_AND_WRITING, + ) + }; + if reopened_file_descriptor != file_descriptor { + return; + } + } +} + +#[cfg(target_os = "linux")] +fn closed_standard_streams() -> [Option; 3] { + use std::ffi::{c_short, c_ulong}; + const POLL_INVALID_FILE_DESCRIPTOR: c_short = 0x20; + #[repr(C)] + struct PollFileDescriptor { + file_descriptor: c_int, + requested_events: c_short, + returned_events: c_short, + } + extern "C" { + fn poll( + descriptors: *mut PollFileDescriptor, + descriptor_count: c_ulong, + timeout: c_int, + ) -> c_int; + } + let mut standard_streams = + STANDARD_STREAM_FILE_DESCRIPTORS.map(|file_descriptor| PollFileDescriptor { + file_descriptor, + requested_events: 0, + returned_events: 0, + }); + while unsafe { poll(standard_streams.as_mut_ptr(), 3, 0) } < 0 { + if std::io::Error::last_os_error().kind() != std::io::ErrorKind::Interrupted { + return closed_standard_streams_by_descriptor_flags(); + } + } + standard_streams.map(|standard_stream| { + (standard_stream.returned_events & POLL_INVALID_FILE_DESCRIPTOR != 0) + .then_some(standard_stream.file_descriptor) + }) +} + +#[cfg(not(target_os = "linux"))] +fn closed_standard_streams() -> [Option; 3] { + closed_standard_streams_by_descriptor_flags() +} + +fn closed_standard_streams_by_descriptor_flags() -> [Option; 3] { + const GET_FILE_DESCRIPTOR_FLAGS: c_int = 1; + const BAD_FILE_DESCRIPTOR: i32 = 9; + extern "C" { + fn fcntl(file_descriptor: c_int, command: c_int, ...) -> c_int; + } + STANDARD_STREAM_FILE_DESCRIPTORS.map(|file_descriptor| { + let descriptor_is_closed = unsafe { fcntl(file_descriptor, GET_FILE_DESCRIPTOR_FLAGS) } + == -1 + && std::io::Error::last_os_error().raw_os_error() == Some(BAD_FILE_DESCRIPTOR); + descriptor_is_closed.then_some(file_descriptor) + }) +} diff --git a/apps/heft-native/src/sys/worker_processes.rs b/apps/heft-native/src/sys/worker_processes.rs new file mode 100644 index 0000000000..a9afe40c08 --- /dev/null +++ b/apps/heft-native/src/sys/worker_processes.rs @@ -0,0 +1,45 @@ +use std::ffi::{c_int, c_ulong}; +use std::io; + +const SET_PARENT_DEATH_SIGNAL: c_int = 1; +const SIGNAL_KILL: c_ulong = 9; +const EXIT_STATUS_AFTER_ORPHANING: c_int = 1; + +extern "C" { + fn fork() -> c_int; + fn _exit(status: c_int) -> !; + fn waitpid(process_id: c_int, status: *mut c_int, options: c_int) -> c_int; + fn prctl(option: c_int, ...) -> c_int; + fn getppid() -> c_int; +} + +pub enum ForkedWorker { + InParent { worker_process_id: i32 }, + InWorker, +} + +pub fn fork_worker_process_of_this_single_threaded_process() -> io::Result { + let parent_process_id = std::process::id() as c_int; + match unsafe { fork() } { + -1 => Err(io::Error::last_os_error()), + 0 => { + unsafe { prctl(SET_PARENT_DEATH_SIGNAL, SIGNAL_KILL) }; + if unsafe { getppid() } != parent_process_id { + exit_worker_process_immediately(EXIT_STATUS_AFTER_ORPHANING); + } + Ok(ForkedWorker::InWorker) + } + worker_process_id => Ok(ForkedWorker::InParent { worker_process_id }), + } +} + +pub fn exit_worker_process_immediately(status: i32) -> ! { + unsafe { _exit(status) } +} + +pub fn wait_for_worker_process(worker_process_id: i32) { + let mut status: c_int = 0; + while unsafe { waitpid(worker_process_id, &mut status, 0) } == -1 + && io::Error::last_os_error().kind() == io::ErrorKind::Interrupted + {} +} diff --git a/apps/heft-native/src/terminal/heft_console.rs b/apps/heft-native/src/terminal/heft_console.rs index ae7d337a07..9bf28d3771 100644 --- a/apps/heft-native/src/terminal/heft_console.rs +++ b/apps/heft-native/src/terminal/heft_console.rs @@ -19,16 +19,47 @@ pub struct HeftConsole { supports_color: bool, captured_output: Option>>, closed_output: Cell>, + pending_standard_output: RefCell, + first_pending_line_is_prefixed: Cell, } impl HeftConsole { pub fn new(supports_color: bool) -> HeftConsole { - HeftConsole { supports_color, captured_output: None, closed_output: Cell::new(None) } + HeftConsole::with_capture(supports_color, None) } #[cfg(test)] pub fn capturing(supports_color: bool) -> HeftConsole { - HeftConsole { supports_color, captured_output: Some(RefCell::new(Vec::new())), closed_output: Cell::new(None) } + HeftConsole::with_capture(supports_color, Some(RefCell::new(Vec::new()))) + } + + fn with_capture(supports_color: bool, captured_output: Option>>) -> HeftConsole { + HeftConsole { + supports_color, + captured_output, + closed_output: Cell::new(None), + pending_standard_output: RefCell::new(String::new()), + first_pending_line_is_prefixed: Cell::new(false), + } + } + + pub fn flush(&self) { + let mut pending_standard_output = self.pending_standard_output.borrow_mut(); + if pending_standard_output.is_empty() { + return; + } + if self.closed_output.get().is_none() + && write_to_stream(OutputSeverity::Log, &pending_standard_output).is_err_and(|error| error.kind() == ErrorKind::BrokenPipe) + { + let prefixed = self.first_pending_line_is_prefixed.get(); + self.closed_output.set(Some(ClosedOutput { severity: OutputSeverity::Log, prefixed })); + } + pending_standard_output.clear(); + } + + #[cfg(test)] + pub fn pending_standard_output_for_tests(&self) -> String { + self.pending_standard_output.borrow().clone() } pub fn closed_output(&self) -> Option { @@ -52,13 +83,21 @@ impl HeftConsole { if self.closed_output.get().is_some() { return; } - match &self.captured_output { - Some(captured) => captured.borrow_mut().push((severity, data.to_owned())), - None => { - if write_to_stream(severity, data).is_err_and(|error| error.kind() == ErrorKind::BrokenPipe) { - self.closed_output.set(Some(ClosedOutput { severity, prefixed })); - } + if let Some(captured) = &self.captured_output { + captured.borrow_mut().push((severity, data.to_owned())); + return; + } + if severity == OutputSeverity::Log { + let mut pending_standard_output = self.pending_standard_output.borrow_mut(); + if pending_standard_output.is_empty() { + self.first_pending_line_is_prefixed.set(prefixed); } + pending_standard_output.push_str(data); + return; + } + self.flush(); + if self.closed_output.get().is_none() && write_to_stream(severity, data).is_err_and(|error| error.kind() == ErrorKind::BrokenPipe) { + self.closed_output.set(Some(ClosedOutput { severity, prefixed })); } } @@ -93,6 +132,12 @@ impl HeftConsole { } } +impl Drop for HeftConsole { + fn drop(&mut self) { + self.flush(); + } +} + pub struct ScopedLoggerOutput<'console> { console: &'console HeftConsole, prefix: String, @@ -106,6 +151,7 @@ impl ScopedLoggerOutput<'_> { } pub fn output_is_closed(&self) -> bool { + self.console.flush(); self.console.closed_output().is_some() } @@ -141,28 +187,3 @@ fn write_to_stream(severity: OutputSeverity, data: &str) -> std::io::Result<()> std::io::stderr().lock().write_all(data.as_bytes()) } } - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn severity_colors_match_the_terminal_package() { - let colored = HeftConsole::new(true); - let plain = HeftConsole::new(false); - assert_eq!(colored.format_line("a\x1b[1mb", OutputSeverity::Error), "\x1b[31mab\x1b[39m\n"); - assert_eq!(plain.format_line("a\x1b[1mb", OutputSeverity::Error), "ab\n"); - assert_eq!(colored.format_line("\x1b[1mx\x1b[22m", OutputSeverity::Log), "\x1b[1mx\x1b[22m\n"); - assert_eq!(plain.format_line("\x1b[1mx\x1b[22m", OutputSeverity::Log), "x\n"); - } - - #[test] - fn scoped_output_prefixes_every_line() { - let console = HeftConsole::new(false); - let output = console.scoped_logger_output("build:set-env"); - assert_eq!(output.prefix_lines("a\nb\n"), "[build:set-env] a\n[build:set-env] b\n"); - assert_eq!(output.prefix_lines("partial"), "[build:set-env] partial"); - assert_eq!(output.prefix_lines(" rest\n"), " rest\n"); - assert_eq!(output.prefix_lines("\n"), "[build:set-env] \n"); - } -} diff --git a/apps/heft-native/src/terminal/heft_console_tests.rs b/apps/heft-native/src/terminal/heft_console_tests.rs new file mode 100644 index 0000000000..957abe3196 --- /dev/null +++ b/apps/heft-native/src/terminal/heft_console_tests.rs @@ -0,0 +1,33 @@ +use super::heft_console::{HeftConsole, OutputSeverity}; + +#[test] +fn severity_colors_match_the_terminal_package() { + let colored = HeftConsole::new(true); + let plain = HeftConsole::new(false); + assert_eq!(colored.format_line("a\x1b[1mb", OutputSeverity::Error), "\x1b[31mab\x1b[39m\n"); + assert_eq!(plain.format_line("a\x1b[1mb", OutputSeverity::Error), "ab\n"); + assert_eq!(colored.format_line("\x1b[1mx\x1b[22m", OutputSeverity::Log), "\x1b[1mx\x1b[22m\n"); + assert_eq!(plain.format_line("\x1b[1mx\x1b[22m", OutputSeverity::Log), "x\n"); +} + +#[test] +fn scoped_output_prefixes_every_line() { + let console = HeftConsole::new(false); + let output = console.scoped_logger_output("build:set-env"); + assert_eq!(output.prefix_lines("a\nb\n"), "[build:set-env] a\n[build:set-env] b\n"); + assert_eq!(output.prefix_lines("partial"), "[build:set-env] partial"); + assert_eq!(output.prefix_lines(" rest\n"), " rest\n"); + assert_eq!(output.prefix_lines("\n"), "[build:set-env] \n"); +} + +#[test] +fn log_lines_wait_for_a_flush_and_errors_flush_them_first() { + let console = HeftConsole::new(false); + console.write_line("pending"); + assert_eq!(console.pending_standard_output_for_tests(), "pending\n"); + console.scoped_logger_output("build:x").write_line("next"); + assert_eq!(console.pending_standard_output_for_tests(), "pending\n[build:x] next\n"); + console.flush(); + assert_eq!(console.pending_standard_output_for_tests(), ""); + assert!(console.closed_output().is_none()); +} diff --git a/apps/heft-native/src/terminal/mod.rs b/apps/heft-native/src/terminal/mod.rs index 1e8e0c2e56..ad51b33574 100644 --- a/apps/heft-native/src/terminal/mod.rs +++ b/apps/heft-native/src/terminal/mod.rs @@ -5,6 +5,8 @@ mod color_support_tests; mod force_color; mod has_flag; mod heft_console; +#[cfg(test)] +mod heft_console_tests; mod javascript_number_format; mod parse_int; mod term_patterns;