diff --git a/.github/actions/add-swap/action.yaml b/.github/actions/add-swap/action.yaml new file mode 100644 index 00000000..c467762b --- /dev/null +++ b/.github/actions/add-swap/action.yaml @@ -0,0 +1,32 @@ +name: Add swap +description: Add a swapfile on Linux, on top of whatever swap the runner image already has. + +# Adds to the image's swap rather than replacing it, and fails the job on any error. +# Carrying on without the swap is the failure to avoid: a build that needed it and ran with none would die the way +# it did before, with nothing in the log saying the swap was never there. +inputs: + size-gb: + description: Size of the swapfile to add, in GiB. + required: false + default: "8" + path: + description: Where to create the swapfile. Distinct from the image's own, which stays in use. + required: false + default: /et-swapfile + +runs: + using: composite + steps: + - name: Add swap on Linux + if: runner.os == 'Linux' + shell: bash --noprofile --norc -euo pipefail {0} + env: + SWAP_PATH: ${{ inputs.path }} + SWAP_SIZE_GB: ${{ inputs.size-gb }} + run: | + sudo fallocate -l "${SWAP_SIZE_GB}G" "$SWAP_PATH" + sudo chmod 600 "$SWAP_PATH" + sudo mkswap "$SWAP_PATH" + sudo swapon "$SWAP_PATH" + swapon --show + free -h diff --git a/.github/actions/free-disk-space/action.yaml b/.github/actions/free-disk-space/action.yaml index e88390e1..30b285dc 100644 --- a/.github/actions/free-disk-space/action.yaml +++ b/.github/actions/free-disk-space/action.yaml @@ -4,9 +4,21 @@ description: Reclaim runner disk before the long build and test steps runs: using: composite steps: + # Swap is kept rather than reclaimed. + # The action's default switches it off and deletes the swapfile, which on a 16 GiB runner leaves the link-heavy + # test build nothing to spill into, and the space that would free is not needed. - name: Free disk space on Linux if: runner.os == 'Linux' uses: jlumbroso/free-disk-space@v1.3.1 + with: + swap-storage: false + + # Swap on top of whatever the image ships, which on ubuntu-24.04-arm is a 3 GiB /swapfile. + # A test build that outgrows memory then slows down instead of taking the runner with it -- the hosted runner + # is killed outright on exhaustion, reporting a shutdown signal rather than any build error. Root has the room: + # it shows 119G free on the arm runner once the step above has run. + - name: Add swap + uses: ./.github/actions/add-swap - name: Free disk space on Windows if: runner.os == 'Windows' diff --git a/.github/workflows/k3s.yaml b/.github/workflows/k3s.yaml index bd3cd585..99279530 100644 --- a/.github/workflows/k3s.yaml +++ b/.github/workflows/k3s.yaml @@ -13,6 +13,9 @@ name: k3s "on": pull_request: paths: + # install-mise-tools and the two actions it runs on Linux before anything else. + - .github/actions/add-swap/** + - .github/actions/free-disk-space/** - .github/actions/install-mise-tools/** - .github/workflows/k3s.yaml # The `k3s-verify` task itself lives here. diff --git a/.github/workflows/test.yaml b/.github/workflows/test.yaml index edec0c4f..ccee98b6 100644 --- a/.github/workflows/test.yaml +++ b/.github/workflows/test.yaml @@ -46,6 +46,17 @@ jobs: packages: read runs-on: ${{ matrix.os }} timeout-minutes: ${{ matrix.timeout }} + # The test build's peak memory is what killed the ubuntu-24.04-arm lane, so this job trims it four ways. + # Line tables keep a panicking test's backtrace pointing at file and line while dropping the rest of the debug + # info the linker has to read and write; incremental caches are never reused by a one-shot build; the hosted + # Ubuntu runners, whose GCC is new enough, link through mold; and the arm lane alone runs three codegen and link + # jobs instead of four. That last one costs wall-clock, so it stays on the lane that needed it -- every other lane + # keeps cargo's own default, which is the one value besides a number that the variable accepts. + env: + CARGO_BUILD_JOBS: ${{ matrix.os == 'ubuntu-24.04-arm' && '3' || 'default' }} + CARGO_INCREMENTAL: "0" + CARGO_PROFILE_DEV_DEBUG: line-tables-only + ET_LINK_WITH_MOLD: ${{ startsWith(matrix.os, 'ubuntu') }} # Override the workflow-level `shell: bash` default for Windows runs. # This uses the busybox-w32 ash that mise installs in `_setup_all` (the same shell `MISE_BASH_PATH` already # points at for `shell = "bash"` tasks on the Windows Docker images). Avoids Git Bash's MSYS fork-emulation @@ -118,9 +129,18 @@ jobs: # macos-26-intel raises its own ceiling through the matrix, because it is the only lane that source-builds # vector and rustfs -- upstream ships no macos/x64 prebuilt for either. Those two overran the flat 60 even # while still failing early; completing them needs more. Every other lane falls back to 60 via the `||`. + # + # Linking through mold is off for this one step, because mold is among the tools it installs. The `cargo:` tools + # with no prebuilt for a lane are source-built here, before mold is on PATH, and asking GCC for it then fails each + # of those links with + # collect2: fatal error: cannot find 'ld' + # on commit https://github.com/edge-toolkit/core/commit/5d7dfaa10c0d10e980aa7a4a785476be059699fb at + # https://github.com/edge-toolkit/core/actions/runs/36812561314/job/110210608554. - name: Install mise + tools uses: ./.github/actions/install-mise-tools timeout-minutes: ${{ matrix.install_timeout || 60 }} + env: + ET_LINK_WITH_MOLD: "false" # GITHUB_TOKEN raises the rate-limit ceiling on mise's fetch of the rp-v rustpython_wasm tarball. # (The fetch is [tools."http:rp-wasm"], via mise's http backend against github.com release assets.) diff --git a/.mise/config.linux.toml b/.mise/config.linux.toml index c1deecd8..92c67048 100644 --- a/.mise/config.linux.toml +++ b/.mise/config.linux.toml @@ -9,6 +9,8 @@ "conda:clangxx" = "latest" # gpg for mise's gpg_verify signature checks; _setup_all installs it up front. "conda:gnupg" = "latest" +# mold, the linker a Linux lane that sets ET_LINK_WITH_MOLD links Rust with. +"aqua:rui314/mold" = "latest" # System packages `mise bootstrap packages status` verifies and `mise bootstrap packages apply` installs. # The same prerequisite set the Dockerfile's COMMON_PACKAGES + APT_PACKAGES / DNF_PACKAGES ARGs install; the @@ -56,6 +58,19 @@ a_conda_arch = "{% if arch() == 'arm64' %}aarch64{% else %}x86_64{% endif %}" a_conda_clangxx = "{{ env.HOME }}/.local/share/mise/installs/conda-clangxx/latest" b_clang_resource = "{{ vars.a_conda_clangxx }}/lib/clang/22" b_clang_sysroot = "{{ vars.a_conda_clangxx }}/{{ vars.a_conda_arch }}-conda-linux-gnu/sysroot" +# Link through mold rather than the system linker, for a lane that sets ET_LINK_WITH_MOLD=true. +# On aarch64 the system linker is GNU ld -- rustc only defaults to its bundled lld on x86_64 -- and linking the +# deno/V8 test binaries with it while the rest of the workspace still compiles exhausts the runner's memory. The +# arm lane of test.yaml died that way at the same compile step on 7 of 40 runs, the hosted runner killed rather +# than any build error: +# ##[error]The runner has received a shutdown signal. This can happen when the runner service is stopped, or a +# manually started runner is canceled. +# [cargo-test] ERROR sh exited with non-zero status: killed by SIGTERM +# on commit https://github.com/edge-toolkit/core/commit/674ca2bbc30ecb34db9cbf749e5ae3e3d609a01b at +# https://github.com/edge-toolkit/core/actions/runs/36792810690/job/110199819803. +# Opt-in rather than the default because `-fuse-ld=mold` needs GCC 12.1, and the opensuse, amazonlinux and +# ubuntu:22.04 image bases ship an older one. +mold_flag = '{% if env?.ET_LINK_WITH_MOLD == "true" %}-Clink-arg=-fuse-ld=mold{% endif %}' # bindgen loads libclang directly rather than driving the clang binary. # So it can't auto-find its resource dir (builtins like stddef.h) or conda's bundled sysroot. Pass `-resource-dir` # + `--sysroot` so header resolution is identical on every machine (a bare `-isystem` leaves builtin type @@ -75,8 +90,8 @@ BINDGEN_EXTRA_CLANG_ARGS = "{{ vars.c_bindgen_args }}" # OPENSSL_DIR points openssl-sys at conda's OpenSSL, so anything linking it records conda's libssl soname; # without an rpath the loader only finds it when conda's soname matches the system one (it broke when conda # bumped to libssl.so.4). pylib_flag does the same for the CPython lib dir. -CARGO_TARGET_AARCH64_UNKNOWN_LINUX_GNU_RUSTFLAGS = "{{ vars.rpath_flag }} {{ vars.pylib_flag }}" -CARGO_TARGET_X86_64_UNKNOWN_LINUX_GNU_RUSTFLAGS = "{{ vars.rpath_flag }} {{ vars.pylib_flag }}" +CARGO_TARGET_AARCH64_UNKNOWN_LINUX_GNU_RUSTFLAGS = "{{ vars.rpath_flag }} {{ vars.pylib_flag }} {{ vars.mold_flag }}" +CARGO_TARGET_X86_64_UNKNOWN_LINUX_GNU_RUSTFLAGS = "{{ vars.rpath_flag }} {{ vars.pylib_flag }} {{ vars.mold_flag }}" LIBCLANG_PATH = "{{ vars.a_conda_clangxx }}/lib" [tasks.preinstall] diff --git a/.mise/config.toml b/.mise/config.toml index 223509c7..eaa84522 100644 --- a/.mise/config.toml +++ b/.mise/config.toml @@ -1468,11 +1468,12 @@ if [ -z "${GITHUB_TOKEN:-}" ]; then exit 1 fi -# Storage is redirected so the run is self-contained. -# Unset, the hub resolves its default against the repository root it finds by walking up, so two runs and any -# stale bucket from ordinary development share one directory and the poll below can read a model this run never -# produced. -storage=$(coreutils mktemp -d) +# Storage is where the deployment itself keeps it, beside its `mise.toml`, and is emptied first. +# The deployment's own task env names that directory, which outranks anything exported into this process, so it is +# read from there rather than redirected. Emptying it keeps a stale model from an earlier run from satisfying the +# poll below; the directory is gitignored beside the deployment. +storage="$out/storage" +coreutils rm -rf "$storage" scenario_pid="" cleanup() { status=$? @@ -1504,7 +1505,7 @@ coreutils rm -f "$out/mise.lock" # fetched https://registry.npmjs.org/@edge-toolkit%2Fet-ws-math1 and installed nothing. npmrc="$PWD/$out/npmrc" (cd "$out" && NPM_CONFIG_USERCONFIG="$npmrc" mise install) -(cd "$out" && STORAGE_URL="file://$storage" mise run generated-scenario) & +(cd "$out" && mise run generated-scenario) & scenario_pid=$! # The model is what says the exchange ran, rather than the processes being up. diff --git a/.mise/mise.linux.lock b/.mise/mise.linux.lock index a416cc3f..027efb35 100644 --- a/.mise/mise.linux.lock +++ b/.mise/mise.linux.lock @@ -844,6 +844,20 @@ checksum = "sha256:731e043390c9457299484d39e427221fc868a9249540a498a5a4f6456c774 url = "https://conda.anaconda.org/conda-forge/win-64/zstd-1.5.7-h534d264_7.conda" checksum = "sha256:ca7daae4f218a11fab82cc2857f0ea518ec3f46acec60490485347a4c22c6b3e" +[[tools."aqua:rui314/mold"]] +version = "2.42.1" +backend = "aqua:rui314/mold" + +[tools."aqua:rui314/mold"."platforms.linux-arm64"] +checksum = "sha256:16b025652d3d7456689e6025a77e1903bb2a15e7630877c26cc133f5df95b9c6" +url = "https://github.com/rui314/mold/releases/download/v2.42.1/mold-2.42.1-aarch64-linux.tar.gz" +url_api = "https://api.github.com/repos/rui314/mold/releases/assets/557035958" + +[tools."aqua:rui314/mold"."platforms.linux-x64"] +checksum = "sha256:6ff270c9bf07d2bec5c98aa324eb7c4daf6a1a4d815c05ff1708049616047855" +url = "https://github.com/rui314/mold/releases/download/v2.42.1/mold-2.42.1-x86_64-linux.tar.gz" +url_api = "https://api.github.com/repos/rui314/mold/releases/assets/557036001" + [[tools."conda:clangxx"]] version = "22.1.6" backend = "conda:clangxx" diff --git a/CLAUDE.md b/CLAUDE.md index 2972ac38..56ee6ddf 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -840,6 +840,10 @@ only correlate: every sighting was a long job. So the remedy is `gh run rerun --job ` once the parent run completes, and there is nothing to fix. Do not spend a diagnosis on it; check for the empty log first, and if the siblings passed, re-run and move on. +The `ubuntu-24.04-arm` lane of test.yaml is the exception, and it disproves the rotation argument for that lane: +its sightings kept their logs, and all of them stop at the same compile step. That is memory running out, and +the lane is now built to stay under it. A shutdown signal there again is a regression to diagnose, not a rerun. + ### Fixed: vector_otlp_relay store-and-forward timing Root-caused and fixed on 2026-09-11 by capping the retry interval; kept here because the panic line is what a diff --git a/Cargo.lock b/Cargo.lock index 45d0d540..53fabbcc 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -4164,7 +4164,7 @@ dependencies = [ [[package]] name = "edge-toolkit" -version = "0.3.0" +version = "0.3.1" dependencies = [ "asyncapi-rust", "base64 0.23.1", @@ -4308,7 +4308,7 @@ checksum = "31ae425815400e5ed474178a7a22e275a9687086a12ca63ec793ff292d8fdae8" [[package]] name = "et-cli" -version = "0.1.0" +version = "0.1.1" dependencies = [ "clap", "clap-markdown", @@ -4878,7 +4878,7 @@ dependencies = [ [[package]] name = "et-ws-server" -version = "0.2.0" +version = "0.2.1" dependencies = [ "actix-rt", "actix-web", @@ -4917,7 +4917,7 @@ dependencies = [ [[package]] name = "et-ws-service" -version = "0.1.0" +version = "0.1.1" dependencies = [ "actix-web", "actix-ws", diff --git a/Cargo.toml b/Cargo.toml index 6a2e6935..fa074fbc 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -87,7 +87,7 @@ deno_resolver = "0.85" # instead of panicking when no SnapshotOptions struct is in OpState -- without # baking our own startup snapshot we'd otherwise hit the panic on first run. deno_runtime = { version = "0.262", features = ["transpile", "hmr"] } -edge-toolkit = { path = "libs/edge-toolkit", version = "0.3.0" } +edge-toolkit = { path = "libs/edge-toolkit", version = "0.3.1" } et-modules-service = { path = "services/modules", version = "0.2.0" } et-org = { path = "libs/org", version = "0.1.0" } et-otlp = { path = "libs/et-otlp", version = "0.2.0" } @@ -100,8 +100,8 @@ et-wasi-guest = { path = "libs/wasi-guest", version = "0.1.0" } et-web = { path = "libs/web", version = "0.2.0" } et-websockify-service = { path = "services/websockify", version = "0.1.0" } et-ws-runner-common = { path = "libs/ws-runner-common", version = "0.2.0" } -et-ws-server = { path = "services/ws-server", version = "0.2.0" } -et-ws-service = { path = "services/ws", version = "0.1.0" } +et-ws-server = { path = "services/ws-server", version = "0.2.1" } +et-ws-service = { path = "services/ws", version = "0.1.1" } et-ws-test-server = { path = "services/ws-test-server", version = "0.1.0" } et-ws-wasm-agent = { path = "services/ws-wasm-agent", version = "0.2.0" } fake = "5" diff --git a/config/jscpd-baseline.json b/config/jscpd-baseline.json index a7afb077..e5d8260b 100644 --- a/config/jscpd-baseline.json +++ b/config/jscpd-baseline.json @@ -119,7 +119,6 @@ "608f556637a1f4db": 1, "61012257ce19e116": 1, "61b28209f41edecf": 1, - "6214037b4acff9b5": 1, "62257586054b0689": 1, "635fb32c899e4df6": 1, "644491f2c34618c6": 1, @@ -172,6 +171,7 @@ "8beb95ee363cd6e3": 1, "8c912ef640675113": 1, "8cfbef0da485b326": 1, + "8dd8aa0822526cc2": 1, "8fa398e0937479af": 1, "928c39b341599636": 1, "92ed119fbc8ed830": 1, diff --git a/libs/edge-toolkit/Cargo.toml b/libs/edge-toolkit/Cargo.toml index 565393ba..131ae247 100644 --- a/libs/edge-toolkit/Cargo.toml +++ b/libs/edge-toolkit/Cargo.toml @@ -2,7 +2,7 @@ name = "edge-toolkit" publish = true description = "A collection of utilities and common code for Edge Toolkit services" -version = "0.3.0" +version = "0.3.1" edition.workspace = true license.workspace = true repository.workspace = true diff --git a/libs/edge-toolkit/src/ws_server.rs b/libs/edge-toolkit/src/ws_server.rs index 0789760e..395d2298 100644 --- a/libs/edge-toolkit/src/ws_server.rs +++ b/libs/edge-toolkit/src/ws_server.rs @@ -27,9 +27,8 @@ impl From> for RegistryError { /// Why an `acknowledge_message` call rejected the ack. /// -/// The variant itself describes *what* went wrong; the optional payload is -/// the recipient/sender id the caller can quote back in a wire-level status -/// message. +/// The variant itself describes *what* went wrong; the optional payload is the recipient/sender id the caller can quote +/// back in a wire-level status message. #[derive(Debug, Error)] #[non_exhaustive] pub enum AcknowledgeError { @@ -51,15 +50,32 @@ impl From> for AcknowledgeError { /// Take the lock, recovering from poison by returning the inner guard. /// -/// We never observe poisoned state in the wild -- every panic-prone path -/// holds the lock briefly around infallible map ops. Recovering keeps the -/// registry usable if a future change introduces a panic under the lock. +/// We never observe poisoned state in the wild -- every panic-prone path holds the lock briefly around infallible map +/// ops. Recovering keeps the registry usable if a future change introduces a panic under the lock. fn lock_agents( agents: &Mutex>>, ) -> MutexGuard<'_, BTreeMap>> { agents.lock().unwrap_or_else(PoisonError::into_inner) } +/// Longest agent id an agent may choose for itself; a generated UUID is 36 characters. +pub const MAX_AGENT_ID_LEN: usize = 64; + +/// Whether `id` is acceptable as an agent id the agent chose itself rather than one the hub generated. +/// +/// The id becomes the first segment of the agent's storage paths (`/`), so it is held to characters +/// that cannot address anything outside that bucket: ASCII letters, digits, `.`, `_` and `-`, with no `/`, and not a +/// dot-only segment such as `..`. Every id the hub generates passes, so a reconnecting agent is never affected. +#[must_use] +pub fn is_valid_agent_id(id: &str) -> bool { + !id.is_empty() + && id.len() <= MAX_AGENT_ID_LEN + && id.chars().any(|character| character != '.') + && id + .chars() + .all(|character| character.is_ascii_alphanumeric() || matches!(character, '.' | '_' | '-')) +} + #[derive(Debug, Clone, Serialize, Deserialize)] #[non_exhaustive] pub struct PendingDirectMessage { @@ -161,17 +177,21 @@ impl AgentRegistry { ) -> (String, ConnectStatus) { let mut agents = lock_agents(&self.agents); - if let Some(requested_id) = requested_id - && let Some(record) = agents.get_mut(&requested_id) + if let Some(requested) = requested_id.as_deref() + && let Some(record) = agents.get_mut(requested) { record.state = AgentConnectionState::Connected; record.last_known_ip = Some(client_ip.to_string()); record.session = Some(session); - return (requested_id, ConnectStatus::Reconnected); + return (requested.to_string(), ConnectStatus::Reconnected); } + // An id the hub has never seen is taken as the agent's own name for itself, so a headless agent can keep one + // identity -- and one storage bucket -- across hub restarts without the hub persisting its registry. One that + // is not safe to use as a storage path segment gets a generated id instead, as before. + let assigned_id = requested_id.filter(|id| is_valid_agent_id(id)).unwrap_or(new_id); let _previous: Option> = agents.insert( - new_id.clone(), + assigned_id.clone(), AgentRecord { state: AgentConnectionState::Connected, last_known_ip: Some(client_ip.to_string()), @@ -180,12 +200,31 @@ impl AgentRegistry { }, ); drop(agents); - (new_id, ConnectStatus::Assigned) + (assigned_id, ConnectStatus::Assigned) } pub fn mark_disconnected(&self, agent_id: &str) { + self.mark_disconnected_if(agent_id, &|_session| true); + } + + /// Mark `agent_id` disconnected only if `is_current` accepts the session the registry holds for it. + /// + /// A connection that takes over an id another connection still holds -- the newest claim wins, so a reconnect whose + /// old socket has not closed yet keeps its identity -- replaces that session. When the displaced connection later + /// closes, it must not disconnect its replacement, so a closing connection passes a test for its own session. A + /// record whose session is already gone is disconnected whatever the test says, since nothing live holds it. + /// + /// The test is a trait object rather than a generic so there is one copy of this per session type. Every closure is + /// its own type, and a generic would give each caller its own copy with only that caller's branches taken -- which + /// the branch-coverage gate scores copy by copy, so no single copy would ever read as fully covered: + /// `libs/edge-toolkit/src/ws_server.rs 19/20 branches` on commit + /// at + /// . + pub fn mark_disconnected_if(&self, agent_id: &str, is_current: &dyn Fn(&S) -> bool) { let mut agents = lock_agents(&self.agents); - if let Some(record) = agents.get_mut(agent_id) { + if let Some(record) = agents.get_mut(agent_id) + && record.session.as_ref().is_none_or(is_current) + { record.state = AgentConnectionState::Disconnected; record.session = None; } @@ -209,8 +248,8 @@ impl AgentRegistry { /// Queue a direct message for `to_agent_id`, returning the stored message and the recipient's session. /// - /// Returns `None` when `to_agent_id` is not in the registry. The inner `Option` is the recipient's live - /// session -- `Some` when connected, `None` when the message was queued for a disconnected agent. + /// Returns `None` when `to_agent_id` is not in the registry. The inner `Option` is the recipient's live session + /// -- `Some` when connected, `None` when the message was queued for a disconnected agent. #[must_use] pub fn queue_direct( &self, @@ -267,10 +306,9 @@ impl AgentRegistry { .iter() .find_map(|(id, pending)| (pending.message_id == message_id).then(|| id.clone())) .ok_or(AcknowledgeError::NoPendingMessage)?; - // The `find_map` above drops its iterator before we re-borrow - // `pending_direct_messages` mutably for the removal. The double - // lookup costs O(log n) but lets us share the `NoPendingMessage` - // error with the find-side case instead of asserting an invariant. + // The `find_map` above drops its iterator before we re-borrow `pending_direct_messages` mutably for the + // removal. The double lookup costs O(log n) but lets us share the `NoPendingMessage` error with the find-side + // case instead of asserting an invariant. let pending = recipient .pending_direct_messages .remove(&sender_agent_id) diff --git a/libs/edge-toolkit/tests/registry.rs b/libs/edge-toolkit/tests/registry.rs index b606a570..699fdc3e 100644 --- a/libs/edge-toolkit/tests/registry.rs +++ b/libs/edge-toolkit/tests/registry.rs @@ -1,13 +1,13 @@ //! Covers the `AgentRegistry` persistence + session-lookup paths the ws-server integration tests skip: -//! save/load round-trip (including the missing-file and populated branches), reconnect, `agent_session`, -//! and the `with_pending_direct_messages` builder. `S = String` stands in for the runtime session handle -//! (which is `#[serde(skip)]`, so it is never persisted). +//! save/load round-trip (including the missing-file and populated branches), reconnect, `agent_session`, and +//! the `with_pending_direct_messages` builder. `S = String` stands in for the runtime session handle (which is +//! `#[serde(skip)]`, so it is never persisted). #![cfg(test)] use std::collections::BTreeMap; use edge_toolkit::ws::{AgentConnectionState, ConnectStatus}; -use edge_toolkit::ws_server::{AgentRecord, AgentRegistry}; +use edge_toolkit::ws_server::{AgentRecord, AgentRegistry, MAX_AGENT_ID_LEN, is_valid_agent_id}; use tempfile::tempdir; #[test] @@ -43,14 +43,13 @@ fn save_load_roundtrip_and_session_lookup() { } #[test] -fn unknown_ids_fall_through_to_the_assign_and_no_op_paths() { +fn an_unsafe_id_is_replaced_and_an_unknown_disconnect_is_a_no_op() { let registry = AgentRegistry::::default(); - // A requested id the registry has never seen is not a reconnect: the lookup misses, so the caller's - // `new_id` is inserted fresh and the status comes back Assigned. Getting this wrong would hand a - // reconnecting client someone else's identity, or resurrect an id the server has forgotten. + // A requested id the registry has never seen, and that is not safe as a storage path segment, is not a reconnect: + // the lookup misses, so the caller's `new_id` is inserted fresh and the status comes back Assigned. let (agent_id, status) = registry.connect_agent( - Some("never-registered".to_string()), + Some("peer/escape".to_string()), "agent-fresh".to_string(), "127.0.0.1", "sess-a".to_string(), @@ -58,8 +57,8 @@ fn unknown_ids_fall_through_to_the_assign_and_no_op_paths() { assert_eq!(agent_id, "agent-fresh"); assert_eq!(status, ConnectStatus::Assigned); - // Disconnecting an id that was never registered is a no-op rather than an insert: the registry still - // holds exactly the one agent above, still Connected. + // Disconnecting an id that was never registered is a no-op rather than an insert: the registry still holds exactly + // the one agent above, still Connected. registry.mark_disconnected("never-registered"); let summaries = registry.list_agents(); assert_eq!( @@ -70,14 +69,14 @@ fn unknown_ids_fall_through_to_the_assign_and_no_op_paths() { assert_eq!(summaries[0].agent_id, "agent-fresh"); assert_eq!(summaries[0].state, AgentConnectionState::Connected); - // The known-id arm, asserted in this same `S = String` instantiation rather than left to the server tests - // that hit it for real. The branch gate scores a generic function by the instantiation that covers the most - // of it (llvm-cov merges an instantiation group's branch counts by taking the maximum, not the union), so - // the no-op arm above covered here and the update arm covered only under the server's session type read as - // one arm each, never both: `libs/edge-toolkit/src/ws_server.rs 9/10 branches` on commit + // The known-id arm, asserted in this same `S = String` instantiation rather than left to the server tests that hit + // it for real. The branch gate scores a generic function by the instantiation that covers the most of it (llvm-cov + // merges an instantiation group's branch counts by taking the maximum, not the union), so the no-op arm above + // covered here and the update arm covered only under the server's session type read as one arm each, never both: + // `libs/edge-toolkit/src/ws_server.rs 9/10 branches` on commit // https://github.com/edge-toolkit/core/commit/c6c4fce73dd25aa58754963867ccf9523caae1bb at - // https://github.com/edge-toolkit/core/actions/runs/35045764477/job/104635181514, while the lcov export, - // which sums instantiations, showed every branch taken. + // https://github.com/edge-toolkit/core/actions/runs/35045764477/job/104635181514, while the lcov export, which sums + // instantiations, showed every branch taken. registry.mark_disconnected("agent-fresh"); let summaries = registry.list_agents(); assert_eq!(summaries[0].state, AgentConnectionState::Disconnected); @@ -88,6 +87,82 @@ fn unknown_ids_fall_through_to_the_assign_and_no_op_paths() { ); } +#[test] +fn an_unknown_valid_id_is_adopted_as_the_agents_own() { + let registry = AgentRegistry::::default(); + + // An id the hub has never seen but that is safe as a storage path segment is kept, so a headless agent keeps one + // identity across hub restarts. It is still a first connection, so the status is Assigned. + let (agent_id, status) = registry.connect_agent( + Some("sensor-7".to_string()), + "agent-generated".to_string(), + "127.0.0.1", + "sess-a".to_string(), + ); + assert_eq!(agent_id, "sensor-7"); + assert_eq!(status, ConnectStatus::Assigned); + assert_eq!(registry.agent_session("sensor-7").as_deref(), Some("sess-a")); + assert_eq!(registry.agent_session("agent-generated"), None); +} + +#[test] +fn a_displaced_session_cannot_disconnect_the_one_that_replaced_it() { + let registry = AgentRegistry::::default(); + let (agent_id, _assigned) = registry.connect_agent( + Some("shared".to_string()), + "unused".to_string(), + "127.0.0.1", + "sess-a".to_string(), + ); + // A second connection claims the same id while the first still holds it, and takes it over. + let (_same_id, status) = registry.connect_agent( + Some(agent_id.clone()), + "unused".to_string(), + "127.0.0.2", + "sess-b".to_string(), + ); + assert_eq!(status, ConnectStatus::Reconnected); + + // The displaced connection closing leaves the id with its replacement. + registry.mark_disconnected_if(&agent_id, &|session| session == "sess-a"); + assert_eq!(registry.agent_session(&agent_id).as_deref(), Some("sess-b")); + assert_eq!(registry.list_agents()[0].state, AgentConnectionState::Connected); + + // The replacement closing does take it offline. + registry.mark_disconnected_if(&agent_id, &|session| session == "sess-b"); + assert_eq!(registry.agent_session(&agent_id), None); + assert_eq!(registry.list_agents()[0].state, AgentConnectionState::Disconnected); +} + +#[test] +fn a_record_with_no_session_is_disconnected_whatever_the_test_says() { + // Loaded from disk, a record carries no session, so no closing connection can prove it owns one. + let dir = tempdir().unwrap(); + let path = dir.path().join("registry.yaml"); + let registry = AgentRegistry::::default(); + let (agent_id, _assigned) = registry.connect_agent(None, "agent-1".to_string(), "127.0.0.1", "sess-a".to_string()); + registry.save(&path).unwrap(); + let reloaded = AgentRegistry::::load(&path).unwrap(); + + reloaded.mark_disconnected_if(&agent_id, &|_session| false); + assert_eq!(reloaded.list_agents()[0].state, AgentConnectionState::Disconnected); +} + +#[test] +fn agent_ids_are_held_to_a_single_safe_path_segment() { + assert!(is_valid_agent_id("sensor-7")); + assert!(is_valid_agent_id("a.b_c-d")); + assert!(is_valid_agent_id(&"x".repeat(MAX_AGENT_ID_LEN))); + assert!(!is_valid_agent_id("")); + // Built from the char so the dot-only segments are not read as the relative path literals they would spell. + let dot = '.'.to_string(); + assert!(!is_valid_agent_id(&dot)); + assert!(!is_valid_agent_id(&dot.repeat(2))); + assert!(!is_valid_agent_id("a/b")); + assert!(!is_valid_agent_id("a b")); + assert!(!is_valid_agent_id(&"x".repeat(MAX_AGENT_ID_LEN + 1))); +} + #[test] fn agent_record_with_pending_builder_replaces_the_map() { let record = AgentRecord::::new(AgentConnectionState::Disconnected, None, None) diff --git a/services/ws-modules/dart-comm1/pkg/package.json b/services/ws-modules/dart-comm1/pkg/package.json index ef4ffef5..ffd10f78 100644 --- a/services/ws-modules/dart-comm1/pkg/package.json +++ b/services/ws-modules/dart-comm1/pkg/package.json @@ -4,5 +4,8 @@ "description": "dart comm1", "version": "0.1.0", "license": "Apache-2.0 OR MIT", - "main": "et_ws_dart_comm1.js" + "main": "et_ws_dart_comm1.js", + "dependencies": { + "@edge-toolkit/et-ws-server-static": "*" + } } diff --git a/services/ws-modules/dart-data1/pkg/package.json b/services/ws-modules/dart-data1/pkg/package.json index f3526a77..0ddf2afa 100644 --- a/services/ws-modules/dart-data1/pkg/package.json +++ b/services/ws-modules/dart-data1/pkg/package.json @@ -4,5 +4,8 @@ "description": "dart data1", "version": "0.1.0", "license": "Apache-2.0 OR MIT", - "main": "et_ws_dart_data1.js" + "main": "et_ws_dart_data1.js", + "dependencies": { + "@edge-toolkit/et-ws-server-static": "*" + } } diff --git a/services/ws-modules/dart-math1/pkg/package.json b/services/ws-modules/dart-math1/pkg/package.json index 40af63d9..b87f1ae3 100644 --- a/services/ws-modules/dart-math1/pkg/package.json +++ b/services/ws-modules/dart-math1/pkg/package.json @@ -4,5 +4,8 @@ "description": "dart math1", "version": "0.1.0", "license": "Apache-2.0 OR MIT", - "main": "et_ws_dart_math1.js" + "main": "et_ws_dart_math1.js", + "dependencies": { + "@edge-toolkit/et-ws-server-static": "*" + } } diff --git a/services/ws-modules/js-math1/pkg/package.json b/services/ws-modules/js-math1/pkg/package.json index 062be6ca..d9b684db 100644 --- a/services/ws-modules/js-math1/pkg/package.json +++ b/services/ws-modules/js-math1/pkg/package.json @@ -4,5 +4,8 @@ "description": "js math1", "version": "0.1.0", "license": "Apache-2.0 OR MIT", - "main": "et_ws_js_math1.js" + "main": "et_ws_js_math1.js", + "dependencies": { + "@edge-toolkit/et-ws-server-static": "*" + } } diff --git a/services/ws-modules/pydata1/pkg/package.json b/services/ws-modules/pydata1/pkg/package.json index c19f10f8..b75169c7 100644 --- a/services/ws-modules/pydata1/pkg/package.json +++ b/services/ws-modules/pydata1/pkg/package.json @@ -1,6 +1,7 @@ { "dependencies": { "@edge-toolkit/et-rest-client": "*", + "@edge-toolkit/et-ws-server-static": "*", "pyodide": "*" }, "description": "Python data 1", diff --git a/services/ws-modules/pydata1/pyproject.toml b/services/ws-modules/pydata1/pyproject.toml index 2bad0a68..10ed8347 100644 --- a/services/ws-modules/pydata1/pyproject.toml +++ b/services/ws-modules/pydata1/pyproject.toml @@ -26,4 +26,5 @@ et-rest-client = { workspace = true } # at generated/python-rest/. [tool.ws-module.dependencies] et-rest-client = "*" +et-ws-server-static = "*" pyodide = "*" diff --git a/services/ws-modules/pydemo1/pkg/package.json b/services/ws-modules/pydemo1/pkg/package.json index 50123215..152074d4 100644 --- a/services/ws-modules/pydemo1/pkg/package.json +++ b/services/ws-modules/pydemo1/pkg/package.json @@ -5,6 +5,7 @@ "@edge-toolkit/et-ws": "*", "@edge-toolkit/et-ws-pyeye1": "*", "@edge-toolkit/et-ws-pyspeech1": "*", + "@edge-toolkit/et-ws-server-static": "*", "@mediapipe/tasks-vision": "*", "onnxruntime-web": "*", "pyodide": "*" diff --git a/services/ws-modules/pydemo1/pyproject.toml b/services/ws-modules/pydemo1/pyproject.toml index 88a45cbd..c22bf859 100644 --- a/services/ws-modules/pydemo1/pyproject.toml +++ b/services/ws-modules/pydemo1/pyproject.toml @@ -29,6 +29,7 @@ et-model-speech1 = "*" et-ws = "*" et-ws-pyeye1 = "*" et-ws-pyspeech1 = "*" +et-ws-server-static = "*" onnxruntime-web = "*" pyodide = "*" diff --git a/services/ws-modules/pyeye1/pyproject.toml b/services/ws-modules/pyeye1/pyproject.toml index 729ee3c1..0a63e976 100644 --- a/services/ws-modules/pyeye1/pyproject.toml +++ b/services/ws-modules/pyeye1/pyproject.toml @@ -27,6 +27,7 @@ et-ws = { workspace = true } "@mediapipe/tasks-vision" = "*" et-model-eye1 = "*" et-ws = "*" +et-ws-server-static = "*" pyodide = "*" [tool.pytest.ini_options] diff --git a/services/ws-modules/pyface1/pyproject.toml b/services/ws-modules/pyface1/pyproject.toml index c159711a..12a5fca3 100644 --- a/services/ws-modules/pyface1/pyproject.toml +++ b/services/ws-modules/pyface1/pyproject.toml @@ -27,6 +27,7 @@ et-ws = { workspace = true } [tool.ws-module.dependencies] et-model-face1 = "*" et-ws = "*" +et-ws-server-static = "*" onnxruntime-web = "*" pyodide = "*" diff --git a/services/ws-modules/pymath1/pyproject.toml b/services/ws-modules/pymath1/pyproject.toml index 708609ab..df637262 100644 --- a/services/ws-modules/pymath1/pyproject.toml +++ b/services/ws-modules/pymath1/pyproject.toml @@ -16,4 +16,5 @@ module-root = "" # Browser-side module dependency: the Pyodide runtime served at /modules/pyodide/. [tool.ws-module.dependencies] +et-ws-server-static = "*" pyodide = "*" diff --git a/services/ws-modules/pyspeech1/pkg/package.json b/services/ws-modules/pyspeech1/pkg/package.json index 02d53c17..37b3e292 100644 --- a/services/ws-modules/pyspeech1/pkg/package.json +++ b/services/ws-modules/pyspeech1/pkg/package.json @@ -2,6 +2,7 @@ "dependencies": { "@edge-toolkit/et-model-speech1": "*", "@edge-toolkit/et-ws": "*", + "@edge-toolkit/et-ws-server-static": "*", "onnxruntime-web": "*", "pyodide": "*" }, diff --git a/services/ws-modules/pyspeech1/pyproject.toml b/services/ws-modules/pyspeech1/pyproject.toml index 23f78ef0..482009f5 100644 --- a/services/ws-modules/pyspeech1/pyproject.toml +++ b/services/ws-modules/pyspeech1/pyproject.toml @@ -23,6 +23,7 @@ et-ws = { workspace = true } [tool.ws-module.dependencies] et-model-speech1 = "*" et-ws = "*" +et-ws-server-static = "*" onnxruntime-web = "*" pyodide = "*" diff --git a/services/ws-modules/rcomm1/pkg/package.json b/services/ws-modules/rcomm1/pkg/package.json index db2c698d..c5d10b0e 100644 --- a/services/ws-modules/rcomm1/pkg/package.json +++ b/services/ws-modules/rcomm1/pkg/package.json @@ -1,4 +1,7 @@ { + "dependencies": { + "@edge-toolkit/et-ws-server-static": "*" + }, "description": "R comm 1", "license": "Apache-2.0 OR MIT", "main": "et_ws_rcomm1.js", diff --git a/services/ws-modules/rdata1/pkg/package.json b/services/ws-modules/rdata1/pkg/package.json index df313db3..2ad28d53 100644 --- a/services/ws-modules/rdata1/pkg/package.json +++ b/services/ws-modules/rdata1/pkg/package.json @@ -1,4 +1,7 @@ { + "dependencies": { + "@edge-toolkit/et-ws-server-static": "*" + }, "description": "R data 1", "license": "Apache-2.0 OR MIT", "main": "et_ws_rdata1.js", diff --git a/services/ws-modules/rmath1/pkg/package.json b/services/ws-modules/rmath1/pkg/package.json index b6b9dd6f..51eb2081 100644 --- a/services/ws-modules/rmath1/pkg/package.json +++ b/services/ws-modules/rmath1/pkg/package.json @@ -1,4 +1,7 @@ { + "dependencies": { + "@edge-toolkit/et-ws-server-static": "*" + }, "description": "R math 1", "license": "Apache-2.0 OR MIT", "main": "et_ws_rmath1.js", diff --git a/services/ws-server/Cargo.toml b/services/ws-server/Cargo.toml index 41f4918b..1d5d8c09 100644 --- a/services/ws-server/Cargo.toml +++ b/services/ws-server/Cargo.toml @@ -2,7 +2,7 @@ name = "et-ws-server" publish = true description = "Actix-web entry point wiring the ws hub, storage and modules services together" -version = "0.2.0" +version = "0.2.1" edition.workspace = true license.workspace = true repository.workspace = true diff --git a/services/ws-test-server/src/lib.rs b/services/ws-test-server/src/lib.rs index e828efd6..e7ebe5a8 100644 --- a/services/ws-test-server/src/lib.rs +++ b/services/ws-test-server/src/lib.rs @@ -40,15 +40,15 @@ pub fn start() -> TestServer { /// Start an in-process ws-server bound to a specific `port` with a temporary storage directory. /// -/// Like [`start`], but for callers that must know the port ahead of time (e.g. a fixed-port launcher a separate -/// process connects to). Panics if the port is already in use. +/// Like [`start`], but for callers that must know the port ahead of time (e.g. a fixed-port launcher a separate process +/// connects to). Panics if the port is already in use. #[must_use] pub fn start_on(port: u16) -> TestServer { let storage_dir = TempDir::new().unwrap(); let storage_config = StorageConfig::local(storage_dir.path()); - // The default search paths, with the page module named outright. The server has no default for which - // module it serves at `/` -- that is the deployment's business, and this stand-in is a deployment. + // The default search paths, with the page module named outright. The server has no default for which module it + // serves at `/` -- that is the deployment's business, and this stand-in is a deployment. let mut modules_config = ModulesConfig::default(); modules_config.root = "@edge-toolkit/et-ws-server-static".to_string(); let modules_config = modules_config; @@ -61,9 +61,8 @@ pub fn start_on(port: u16) -> TestServer { let modules = modules_config; let ws_config = WsConfig::default(); HttpServer::new(move || { - // `TracingLogger` mirrors the real ws-server's pipeline: - // extracts `traceparent` from incoming requests so server - // spans are children of the caller's trace. + // `TracingLogger` mirrors the real ws-server's pipeline: extracts `traceparent` from incoming requests + // so server spans are children of the caller's trace. App::new() .wrap(TracingLogger::default()) .app_data(registry.clone()) @@ -98,14 +97,14 @@ const ROSTER_POLL_INTERVAL: Duration = Duration::from_millis(250); /// Block until `count` agents other than the waiter itself are connected to the hub at `ws_url`. /// -/// Returns the connected peer ids from the last roster the hub sent: at least `count` of them once the wait -/// succeeds, and whoever was present when the budget ran out otherwise, so a caller that gives up can report who -/// did come up rather than only that someone did not. +/// Returns the connected peer ids from the last roster the hub sent: at least `count` of them once the wait succeeds, +/// and whoever was present when the budget ran out otherwise, so a caller that gives up can report who did come up +/// rather than only that someone did not. /// /// Reading the roster means being an agent -- `et-list-agents` is a websocket request, not an HTTP route -- so the -/// waiter registers one of its own and filters itself out of every reply. Entries whose state is `Disconnected` -/// are filtered out too: the registry keeps listing an agent after its socket drops, so a runner that registered -/// and then exited would otherwise still read as up. +/// waiter registers one of its own and filters itself out of every reply. Entries whose state is `Disconnected` are +/// filtered out too: the registry keeps listing an agent after its socket drops, so a runner that registered and then +/// exited would otherwise still read as up. /// /// Synchronous, and owns the runtime it needs, so a plain `#[test]` can gate on hub state without taking a tokio /// dependency of its own. @@ -120,9 +119,9 @@ pub fn wait_for_connected_agents(ws_url: &str, count: usize, budget: Duration) - /// The async body of [`wait_for_connected_agents`], split out so the public helper can stay synchronous. /// -/// One socket serves the whole wait: re-asking on a fresh connection each round would leave a trail of -/// disconnected waiter entries in the registry, and a stale one still marked connected would be counted as a peer -/// by the very filter that exists to exclude it. +/// One socket serves the whole wait: re-asking on a fresh connection each round would leave a trail of disconnected +/// waiter entries in the registry, and a stale one still marked connected would be counted as a peer by the very filter +/// that exists to exclude it. #[expect( clippy::single_call_fn, reason = "async body of wait_for_connected_agents; separate so the public helper stays sync" @@ -178,20 +177,35 @@ fn connected_peers(agents: Vec, self_id: &str) -> Vec { /// Open a ws connection to `ws_url` and drive `et-connect` through its ack. /// -/// Returns `(stream, agent_id)` once the `et-connect-ack` has been observed. Lets a test drive the -/// hub as a websocket client. +/// Returns `(stream, agent_id)` once the `et-connect-ack` has been observed. Lets a test drive the hub as a websocket +/// client. pub async fn connect_agent( ws_url: &str, ) -> ( tokio_tungstenite::WebSocketStream>, String, +) { + connect_agent_as(ws_url, None).await +} + +/// Open a ws connection to `ws_url` and drive `et-connect` through its ack, asking for `requested_id`. +/// +/// The same as [`connect_agent`] for an agent that names itself, which is how a test puts two connections on one id. +/// `None` asks the hub to generate one, exactly as [`connect_agent`] does. +pub async fn connect_agent_as( + ws_url: &str, + requested_id: Option<&str>, +) -> ( + tokio_tungstenite::WebSocketStream>, + String, ) { let (mut stream, _) = connect_async(ws_url).await.unwrap(); - let connect_msg = serde_json::to_string(&ClientMessage::Connect { agent_id: None }).unwrap(); + let agent_id = requested_id.map(str::to_string); + let connect_msg = serde_json::to_string(&ClientMessage::Connect { agent_id }).unwrap(); stream.send(Message::text(connect_msg)).await.unwrap(); - // Bound the ack wait: a server that accepts the socket but never sends `et-connect-ack` (and never closes) - // must fail the test fast rather than hang. Non-ack frames simply fall through and the loop reads the next. + // Bound the ack wait: a server that accepts the socket but never sends `et-connect-ack` (and never closes) must + // fail the test fast rather than hang. Non-ack frames simply fall through and the loop reads the next. let deadline = tokio::time::Instant::now() + Duration::from_secs(5); while let Ok(Some(Ok(msg))) = tokio::time::timeout( deadline.saturating_duration_since(tokio::time::Instant::now()), @@ -210,8 +224,8 @@ pub async fn connect_agent( /// Pull the next non-ack frame from `stream`. /// -/// Skips known protocol acks (`et-connect-ack`, `et-message-status`, `et-response`) so callers see -/// the next "real" payload. +/// Skips known protocol acks (`et-connect-ack`, `et-message-status`, `et-response`) so callers see the next "real" +/// payload. pub async fn next_payload( stream: &mut tokio_tungstenite::WebSocketStream>, ) -> Message { @@ -235,8 +249,8 @@ pub async fn next_payload( return msg; } Message::Binary(_) => return msg, - // Ping/pong and any other control frame: skip until a real payload arrives (or the deadline - // elapses / the stream closes, which then surfaces through the `.unwrap()` above). + // Ping/pong and any other control frame: skip until a real payload arrives (or the deadline elapses / the + // stream closes, which then surfaces through the `.unwrap()` above). _ => continue, } } diff --git a/services/ws-test-server/tests/shared_agent_id.rs b/services/ws-test-server/tests/shared_agent_id.rs new file mode 100644 index 00000000..8a2508d5 --- /dev/null +++ b/services/ws-test-server/tests/shared_agent_id.rs @@ -0,0 +1,60 @@ +//! Two connections claiming one agent id: the newest claim wins, and the connection it displaced cannot take the id +//! offline when it later closes. +#![cfg(test)] + +use std::time::Duration; + +use edge_toolkit::ws::ServerMessage; +use et_ws_test_server::{connect_agent, connect_agent_as, next_payload}; +use futures_util::{SinkExt as _, StreamExt as _}; +use tokio_tungstenite::tungstenite::Message; + +/// The id both connections below claim, as an agent naming itself would. +const SHARED_ID: &str = "shared-agent"; + +type Stream = tokio_tungstenite::WebSocketStream>; + +/// Send `payload` from `sender` to the shared id, and return the payload `recipient` receives. +async fn relay_to_shared(sender: &mut Stream, recipient: &mut Stream, payload: u32) -> serde_json::Value { + let send = serde_json::json!({ + "type": "et-send-agent-message", + "to_agent_id": SHARED_ID, + "message": {"n": payload}, + }); + sender.send(Message::text(send.to_string())).await.unwrap(); + let Message::Text(text) = next_payload(recipient).await else { + panic!("expected an et-agent-message text frame"); + }; + let ServerMessage::AgentMessage { message, .. } = serde_json::from_str::(&text).unwrap() else { + panic!("expected ServerMessage::AgentMessage, got {text}"); + }; + message +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn the_newest_claim_on_an_id_survives_the_displaced_connection_closing() { + let server = et_ws_test_server::start(); + let (mut displaced, first_id) = connect_agent_as(&server.ws_url, Some(SHARED_ID)).await; + let (mut newest, second_id) = connect_agent_as(&server.ws_url, Some(SHARED_ID)).await; + let (mut sender, _sender_id) = connect_agent(&server.ws_url).await; + assert_eq!(first_id, SHARED_ID); + assert_eq!(second_id, SHARED_ID); + + // Direct messages to the id reach the connection that claimed it last. + assert_eq!( + relay_to_shared(&mut sender, &mut newest, 1).await, + serde_json::json!({"n": 1_u32}) + ); + + // The displaced connection closes. Reading it to the end waits for the hub to answer the close, after which its + // handler marks the id disconnected; the pause covers the few instructions between the two. + displaced.send(Message::Close(None)).await.unwrap(); + while let Ok(Some(Ok(_frame))) = tokio::time::timeout(Duration::from_secs(5), displaced.next()).await {} + tokio::time::sleep(Duration::from_millis(250)).await; + + // Taking the id offline here would drop the newest connection's session, and this message would sit queued. + assert_eq!( + relay_to_shared(&mut sender, &mut newest, 2).await, + serde_json::json!({"n": 2_u32}) + ); +} diff --git a/services/ws/Cargo.toml b/services/ws/Cargo.toml index f34d93bf..ef61fcfb 100644 --- a/services/ws/Cargo.toml +++ b/services/ws/Cargo.toml @@ -2,7 +2,7 @@ name = "et-ws-service" publish = true description = "Agent registry and WebSocket hub that routes messages between connected agents" -version = "0.1.0" +version = "0.1.1" edition.workspace = true license.workspace = true repository.workspace = true diff --git a/services/ws/src/lib.rs b/services/ws/src/lib.rs index 98178474..4f4a1f51 100644 --- a/services/ws/src/lib.rs +++ b/services/ws/src/lib.rs @@ -678,7 +678,9 @@ impl Connection { ACTIVE_CONNECTIONS.add(-1, &[]); if let Some(agent_id) = self.agent_id.as_deref() { - self.registry.mark_disconnected(agent_id); + // Only this connection's own session: another may have taken the id over since, and is still live. + self.registry + .mark_disconnected_if(agent_id, &|session| session.same_channel(&self.outbox)); info!("Agent {} disconnected; last known IP {}", agent_id, self.client_ip); } else { info!( diff --git a/utilities/cli/Cargo.toml b/utilities/cli/Cargo.toml index 7779967a..b47d15bc 100644 --- a/utilities/cli/Cargo.toml +++ b/utilities/cli/Cargo.toml @@ -2,7 +2,7 @@ name = "et-cli" publish = true description = "Edge Toolkit management CLI" -version = "0.1.0" +version = "0.1.1" edition.workspace = true license.workspace = true repository.workspace = true diff --git a/utilities/cli/src/deployment_types/docker_compose.rs b/utilities/cli/src/deployment_types/docker_compose.rs index aa04cce2..692a6e0b 100644 --- a/utilities/cli/src/deployment_types/docker_compose.rs +++ b/utilities/cli/src/deployment_types/docker_compose.rs @@ -1,4 +1,4 @@ -use std::path::Path; +use std::path::{Path, PathBuf}; use et_path::{absolute_from, relative_path_from}; use fs_err as fs; @@ -7,7 +7,8 @@ use crate::error::CliError; use crate::input::{ArtifactSource, ClusterInput}; use crate::{ COLLECTOR_SETTINGS, COLLECTOR_USERNAME, IMAGE_REGISTRY, OutputType, RunnerInstance, SECRETS_ENV_FILE, - cluster_module_names, hub_http_base, hub_ws_url, module_registry, resolve_cluster_runners, resolve_module_paths, + cluster_module_names, hub_http_base, hub_ws_url, module_path_contexts, module_registry, resolve_cluster_runners, + resolve_module_paths, }; pub fn generate_docker_compose_deployment(cluster: &ClusterInput, output_dir: &Path) -> Result<(), CliError> { @@ -20,7 +21,7 @@ pub fn generate_docker_compose_deployment(cluster: &ClusterInput, output_dir: &P let scenario_dockerfile_rel = format!("{}/Dockerfile", relative_path_from(&workspace_root, &output_abs)); let module_names = cluster_module_names(cluster); let serves_a_page = super::serves_a_page(cluster); - let module_paths = docker_image_module_paths(&module_names, serves_a_page)?; + let module_paths = docker_image_module_paths(&module_names, serves_a_page, &cluster.module_paths)?; let artifacts = cluster.artifact_source; let published = matches!(artifacts, ArtifactSource::Published); let mut services = vec![("openobserve".to_string(), openobserve_service())]; @@ -51,7 +52,10 @@ pub fn generate_docker_compose_deployment(cluster: &ClusterInput, output_dir: &P build: Some(ComposeBuild { context: workspace_rel, dockerfile: scenario_dockerfile_rel, - additional_contexts: vec![("hub".to_string(), hub_context)], + // Relative to this file, which is what compose resolves a context path against. + additional_contexts: std::iter::once(("hub".to_string(), hub_context)) + .chain(module_path_contexts(&cluster.module_paths, &output_abs)) + .collect(), }), network_mode: Some("host".to_string()), // Carries `OTLP_AUTH_PASSWORD`: the server authenticates its OTLP exports against the same root credential @@ -86,7 +90,11 @@ pub fn generate_docker_compose_deployment(cluster: &ClusterInput, output_dir: &P )); services.extend(runner_services( &resolve_cluster_runners( - &module_registry(&workspace_root, &workspace_root.join("services/ws-server")), + &module_registry( + &workspace_root, + &workspace_root.join("services/ws-server"), + &cluster.module_paths, + ), cluster, )?, &relative_path_from(&output_abs, &workspace_root), @@ -208,9 +216,9 @@ fn openobserve_service() -> ComposeService { /// The hub service's environment, in the alphabetical order the rest of the file is written in. /// /// `MODULES_ROOT` is stated even when it is empty, which is what a headless cluster needs rather than the entry -/// being left out. The hub image names the page it bundles, and a container inherits an image's environment for -/// every variable the compose file does not set -- so omitting it would serve a page this cluster was not given the -/// modules for, which the hub rejects at startup. Empty is read as unset. +/// being left out. The hub image names the page it bundles, and a container inherits an image's environment for every +/// variable the compose file does not set -- so omitting it would serve a page this cluster was not given the modules +/// for, which the hub rejects at startup. Empty is read as unset. #[expect( clippy::single_call_fn, reason = "distinct step of the hub service; separate so the root's two spellings are not inlined mid-literal" @@ -249,19 +257,21 @@ fn hub_environment( environment } -pub fn docker_image_module_paths(module_names: &[String], serves_a_page: bool) -> Result, CliError> { +pub fn docker_image_module_paths( + module_names: &[String], + serves_a_page: bool, + module_paths: &[PathBuf], +) -> Result, CliError> { let project_root = edge_toolkit::config::get_project_root(); let ws_server_dir = project_root.join("services/ws-server"); - let mut paths = Vec::with_capacity(module_names.len().saturating_add(2)); + // The page is resolved by name like any other module, so the agent and runtimes it declares come with it. + let mut names = Vec::with_capacity(module_names.len().saturating_add(1)); if serves_a_page { - paths.push("/app/services/ws-server/static".to_string()); + names.push("static".to_string()); } - paths.push("/app/services/ws-wasm-agent".to_string()); - let registry = module_registry(&project_root, &ws_server_dir); - paths.extend(resolve_module_paths(®istry, module_names, |entry| { - entry.docker_path.clone() - })?); - Ok(paths) + names.extend(module_names.iter().cloned()); + let registry = module_registry(&project_root, &ws_server_dir, module_paths); + resolve_module_paths(®istry, &names, |entry| entry.docker_path.clone()) } #[derive(Debug, Default)] @@ -424,10 +434,10 @@ impl ComposeRenderer { // its owner scope, so it begins with `@` -- which unquoted is a scanner error rather than a string, and // produced a compose file nothing could parse. // - // An empty value is quoted for a different reason: written bare it is YAML's null, and compose reads a - // null entry as "take this from the host environment" rather than as a value. The variable would then - // be absent from the container and the image's own default would stand -- the opposite of what stating - // it empty is for. + // An empty value is quoted for a different reason: written bare it is YAML's null, and compose reads a null + // entry as "take this from the host environment" rather than as a value. The variable would then be absent + // from the container and the image's own default would stand -- the opposite of what stating it empty is + // for. ComposeValue::Plain(value) => { let rendered = if value.is_empty() || value.starts_with(YAML_INDICATORS) { format!("\"{value}\"") @@ -442,7 +452,9 @@ impl ComposeRenderer { // per-segment trim expects. The backslash form is banned repo-wide, and this was the one generator still // emitting it. ComposeValue::WrappedDoubleQuoted(parts) => { - if let Some((first, rest)) = parts.split_first() { + if let [only] = parts.as_slice() { + self.push_line(3, &format!("{key}: \"{only}\"")); + } else if let Some((first, rest)) = parts.split_first() { self.push_line(3, &format!("{key}: \"{first},")); let last_index = rest.len().saturating_sub(1); for (index, part) in rest.iter().enumerate() { diff --git a/utilities/cli/src/deployment_types/k3s.rs b/utilities/cli/src/deployment_types/k3s.rs index 18f26d1a..11a1e24e 100644 --- a/utilities/cli/src/deployment_types/k3s.rs +++ b/utilities/cli/src/deployment_types/k3s.rs @@ -104,11 +104,15 @@ pub fn generate_k3s_deployment(cluster: &ClusterInput, output_dir: &Path) -> Res let namespace = namespace_name(&cluster.cluster_name); let workspace_root = edge_toolkit::config::get_project_root(); let runners = resolve_cluster_runners( - &module_registry(&workspace_root, &workspace_root.join("services/ws-server")), + &module_registry( + &workspace_root, + &workspace_root.join("services/ws-server"), + &cluster.module_paths, + ), cluster, )?; let serves_a_page = super::serves_a_page(cluster); - let module_paths = docker_image_module_paths(&cluster_module_names(cluster), serves_a_page)?; + let module_paths = docker_image_module_paths(&cluster_module_names(cluster), serves_a_page, &cluster.module_paths)?; // Empty for a headless cluster, which is not given the page: the hub rejects a root it cannot find among the // modules it was given, and serves nothing at `/` when named none. let root_module = if serves_a_page { @@ -476,10 +480,10 @@ fn collector_service(namespace: &str) -> Service { /// The paths under the runtime directory are the writable ones: a read-only root filesystem cannot take the self-signed /// certificate the hub generates on first start, so both halves are redirected to the scratch mount. fn hub_env(module_paths: &[String], root_module: &str) -> Vec { - // `MODULES_ROOT` is stated even when it is empty, which is what a headless cluster needs rather than the - // entry being left out. The hub image names the page it bundles, and a container inherits an image's - // environment for every variable the manifest does not set -- so omitting it would serve a page this - // cluster was not given the modules for, which the hub rejects at startup. Empty is read as unset. + // `MODULES_ROOT` is stated even when it is empty, which is what a headless cluster needs rather than the entry + // being left out. The hub image names the page it bundles, and a container inherits an image's environment for + // every variable the manifest does not set -- so omitting it would serve a page this cluster was not given the + // modules for, which the hub rejects at startup. Empty is read as unset. let mut vars = vec![ env("MODULES_PATHS", module_paths.join(",")), env("MODULES_ROOT", root_module.to_string()), diff --git a/utilities/cli/src/deployment_types/mise.rs b/utilities/cli/src/deployment_types/mise.rs index afc01e33..01bb9bb5 100644 --- a/utilities/cli/src/deployment_types/mise.rs +++ b/utilities/cli/src/deployment_types/mise.rs @@ -1,6 +1,6 @@ use std::collections::BTreeMap; use std::fmt::Write as _; -use std::path::Path; +use std::path::{Path, PathBuf}; use et_path::{absolute_from, relative_path_from}; use fs_err as fs; @@ -10,8 +10,8 @@ use crate::error::CliError; use crate::input::{ArtifactSource, ClusterInput}; use crate::{ COLLECTOR_SETTINGS, COLLECTOR_USERNAME, MODULE_SCOPE, ModuleRegistryEntry, ModuleSource, RunnerInstance, - cluster_module_names, hub_ws_url, module_registry, resolve_cluster_modules, resolve_cluster_runners, - resolve_module_paths, runner_crate, + cluster_module_names, escape_for_double_quotes, hub_ws_url, module_registry, resolve_cluster_modules, + resolve_cluster_runners, resolve_module_paths, runner_crate, }; /// Crate whose binary serves the hub, which a published deployment installs in place of building it. @@ -33,6 +33,9 @@ const STAGED_PREFIX: &str = "npm:"; /// reference that npm expands when it reads the file, so the deployment carries no secret. const NPMRC_FILE: &str = "npmrc"; +/// Directory, beside `mise.toml`, a published deployment's hub keeps agent storage in. +pub(crate) const STORAGE_DIR: &str = "storage"; + /// Version every `cargo:` tool the generated deployment declares is requested at. const LATEST: &str = "latest"; @@ -45,35 +48,19 @@ pub fn generate_mise_deployment(cluster: &ClusterInput, output_dir: &Path) -> Re let openobserve_run = openobserve_run_body(); let module_names = cluster_module_names(cluster); let artifacts = cluster.artifact_source; - let scenario = ScenarioModules::new(&ws_server_dir, &module_names, super::serves_a_page(cluster)); + let scenario = ScenarioModules::new(&ws_server_dir, &module_names, super::serves_a_page(cluster)) + .with_module_paths(&cluster.module_paths); let staged = if matches!(artifacts, ArtifactSource::Published) { staged_modules(&scenario)? } else { Vec::new() }; - let hub_command = if matches!(artifacts, ArtifactSource::Published) { - HUB_CRATE - } else { - "cargo run" - }; - // A published deployment names its modules once, as the `[tools]` that stage them, and says nothing about where - // they land: the hub asks mise, which is on `PATH` because the deployment is a mise config. Listing them again as - // paths would be the same set written twice, in a form the generator cannot produce anyway. A local deployment has - // no such list to lean on -- its modules are directories in a checkout -- so it still spells them out. - let ws_server_run = if matches!(artifacts, ArtifactSource::Published) { - format!("{hub_command}\n") - } else { - let module_paths = scenario_module_paths(&scenario)?; - let module_paths_lines = wrap_module_paths(&module_paths); - format!("{module_paths_lines}export MODULES_PATHS\n{hub_command}\n") - }; + let ws_server_run = ws_server_run_body(&scenario, artifacts, &output_abs)?; let ws_server_rel = relative_path_from(&output_abs, &ws_server_dir); // A published hub is handed absolute paths that `mise where` resolves, so it needs no directory of its own; a local // one is given paths relative to the server's source dir and has to start there. let ws_server_dir_entry = workspace_dir(&ws_server_rel, artifacts); - // Which module is the page served at `/`. The hub has no default for it -- that would be one project's module name - // carried by every other -- so the deployment that knows the answer states it. - let ws_server_env = Some(hub_root_env(&ws_server_dir, scenario.serves_a_page)); + let ws_server_env = Some(ws_server_env(&scenario, artifacts)); let mut root = Table::new(); let mut tasks = Table::new(); @@ -99,7 +86,10 @@ pub fn generate_mise_deployment(cluster: &ClusterInput, output_dir: &Path) -> Re // concurrently, which is what a cluster wants -- the hub and every runner are long-running peers, not a pipeline -- // but it also means a runner starts before the hub is listening. Each runner body therefore waits for the hub's own // health endpoint first; see `runner_run_body`. - let runners = resolve_cluster_runners(&module_registry(&workspace_root, &ws_server_dir), cluster)?; + let runners = resolve_cluster_runners( + &module_registry(&workspace_root, &ws_server_dir, &cluster.module_paths), + cluster, + )?; for runner in &runners { let _previous: Option = tasks.insert( runner.name.clone(), @@ -153,6 +143,59 @@ pub fn generate_mise_deployment(cluster: &ClusterInput, output_dir: &Path) -> Re Ok(()) } +/// The body of the task that starts the hub. +/// +/// A published deployment names its modules once, as the `[tools]` that stage them, and says nothing about where they +/// land: the hub asks mise, which is on `PATH` because the deployment is a mise config. Listing them again as paths +/// would be the same set written twice, in a form the generator cannot produce anyway. A local deployment has no such +/// list to lean on -- its modules are directories in a checkout -- so it still spells them out. The exception is a +/// scenario with modules of its own, which [`published_module_paths`] covers. +#[expect( + clippy::single_call_fn, + reason = "distinct step of generate_mise_deployment, split out to keep it readable" +)] +fn ws_server_run_body( + scenario: &ScenarioModules<'_>, + artifacts: ArtifactSource, + output_abs: &Path, +) -> Result { + let (hub_command, module_paths) = if matches!(artifacts, ArtifactSource::Published) { + (HUB_CRATE, published_module_paths(scenario, output_abs)?) + } else { + ("cargo run", scenario_module_paths(scenario)?) + }; + if module_paths.is_empty() { + return Ok(format!("{hub_command}\n")); + } + let module_paths_lines = wrap_module_paths(&module_paths); + Ok(format!("{module_paths_lines}export MODULES_PATHS\n{hub_command}\n")) +} + +/// The hub task's environment. +/// +/// It names the page served at `/`: the hub has no default for it -- that would be one project's module name carried by +/// every other -- so the deployment that knows the answer states it. +/// +/// A published deployment also names the agent store. The hub's own default is a directory of this repository's +/// layout (`services/ws-server/storage`) under whatever it takes for the project root -- outside this repository, +/// the directory it was started in. A published hub starts in the deployment directory, so the store is named there +/// directly, as the compose file does with its `/app/storage` volume. A local hub keeps the default, which is where +/// this repository's tests read agent output from. +#[expect( + clippy::single_call_fn, + reason = "distinct step of generate_mise_deployment, split out to keep it readable" +)] +fn ws_server_env(scenario: &ScenarioModules<'_>, artifacts: ArtifactSource) -> Table { + let mut env = hub_root_env(scenario.ws_server_dir, scenario.serves_a_page); + if matches!(artifacts, ArtifactSource::Published) { + let _previous: Option = env.insert( + "STORAGE_URL".to_string(), + Value::String(format!("file://{{{{ config_root }}}}/{STORAGE_DIR}")), + ); + } + env +} + /// The body of the task that starts the collector container. /// /// The image and the settings are lifted into shell variables rather than folded with continuations: inlining them @@ -177,26 +220,21 @@ fn openobserve_run_body() -> String { ) } -/// The agent every module talks to the hub through, served whatever the scenario declares. -/// -/// Named rather than prepended as a path, so it is resolved through the registry like any other module and the -/// dependencies it declares come with it. It is not conditional the way the page is: a module run by a headless runner -/// loads it too, and the modules that need it do not all declare it. -const BASE_MODULES: [&str; 1] = ["ws-wasm-agent"]; - /// The page served at `/`, which only a cluster a browser opens has any use for. /// /// `static` names the runtimes its page imports at boot, so a deployment that serves the page without it serves one /// whose first import 404s -- and one that serves neither is a headless cluster that was never going to load either. const PAGE_MODULE: &str = "static"; -/// The scenario's modules, the base ones first, each named as the hub serves it. +/// The scenario's modules, the page first when there is one, each named as the hub serves it. +/// +/// Nothing else is added: whatever a module needs at run time -- the wasm agent included, which it reaches through the +/// page it declares -- comes in as one of its declared dependencies. fn scenario_modules(module_names: &[String], serves_a_page: bool) -> Vec { let mut names: Vec = Vec::new(); if serves_a_page { names.push(PAGE_MODULE.to_string()); } - names.extend(BASE_MODULES.iter().map(|name| (*name).to_string())); names.extend(module_names.iter().cloned()); names } @@ -219,26 +257,68 @@ fn hub_root_env(ws_server_dir: &Path, serves_a_page: bool) -> Table { /// This is the whole of what a published deployment says about its modules: the tools install them, and the hub serves /// what the config staged. Nothing here records where a module lands, because that is decided per backend and platform /// when the tool installs and so cannot be written down as the deployment is generated. +/// +/// A module of this repository is staged as the package its `package.json` declares, which already carries the owner +/// scope the registry requires. Anything staged by a mise tool keeps the tool it was registered with, which is somebody +/// else's and must not be rewritten. A module from the scenario's own `module_paths:` is not staged at all: it has no +/// release, and is served from its directory instead. pub(crate) fn staged_modules(scenario: &ScenarioModules<'_>) -> Result, CliError> { let (registry, modules) = registry_and_modules(scenario); Ok(resolve_cluster_modules(®istry, &modules)? - .iter() - .map(staged_module) + .into_iter() + .filter_map(|entry| match entry.source { + ModuleSource::Repo(_) => { + let declared = entry.package_name.unwrap_or_default(); + Some(format!("{STAGED_PREFIX}{declared}")) + } + ModuleSource::MiseTool { tool, .. } => Some(tool), + ModuleSource::Scenario { .. } => None, + }) .collect()) } +/// The `MODULES_PATHS` a published hub is started with, in resolution order; empty when it needs none. +/// +/// Needed only once the scenario has a module from its own `module_paths:`, which has no release to stage and so is +/// handed over as a directory, relative to the deployment because a published hub runs from there. Setting the variable +/// replaces the hub's default paths, and of what those defaulted to, discovery restores only the `npm:` tools -- so +/// every other staged tool, `http:pyodide` among them, is named alongside the directories by the same `mise where` a +/// local deployment uses. +fn published_module_paths(scenario: &ScenarioModules<'_>, output_abs: &Path) -> Result, CliError> { + let (registry, modules) = registry_and_modules(scenario); + let resolved = resolve_cluster_modules(®istry, &modules)?; + if !resolved + .iter() + .any(|entry| matches!(entry.source, ModuleSource::Scenario { .. })) + { + return Ok(Vec::new()); + } + let mut paths: Vec = resolved + .into_iter() + .filter_map(|entry| match entry.source { + ModuleSource::Scenario { dir, .. } => Some(escape_for_double_quotes(&relative_path_from(output_abs, &dir))), + ModuleSource::MiseTool { tool, .. } if !tool.starts_with(STAGED_PREFIX) => Some(entry.mise_path), + ModuleSource::Repo(_) | ModuleSource::MiseTool { .. } => None, + }) + .collect(); + paths.sort(); + Ok(paths) +} + /// What a scenario's module resolution starts from, which is the same three answers either way it resolves. /// -/// One struct rather than three parameters because they are never apart: a module list means nothing without -/// the tree it was read from, and whether the page is among them is a property of the same scenario. +/// One struct rather than three parameters because they are never apart: a module list means nothing without the tree +/// it was read from, and whether the page is among them is a property of the same scenario. #[non_exhaustive] pub struct ScenarioModules<'scenario> { /// The hub's directory, which is where the page module and the relative paths are resolved from. pub ws_server_dir: &'scenario Path, - /// The modules the scenario itself declares, before the base ones are added. + /// The modules the scenario itself declares, before their dependencies are resolved. pub module_names: &'scenario [String], /// Whether a browser opens this cluster, and so whether the page module is one of them. pub serves_a_page: bool, + /// The scenario's own `module_paths:`, registered on top of this repository's modules. + pub module_paths: &'scenario [PathBuf], } impl<'scenario> ScenarioModules<'scenario> { @@ -249,34 +329,26 @@ impl<'scenario> ScenarioModules<'scenario> { ws_server_dir, module_names, serves_a_page, + module_paths: &[], } } + + /// The same scenario, also resolving against the module directories it names itself. + #[must_use] + pub const fn with_module_paths(self, module_paths: &'scenario [PathBuf]) -> Self { + Self { module_paths, ..self } + } } /// The registry to resolve against, and the module list to resolve -- what both resolutions here start from. fn registry_and_modules(scenario: &ScenarioModules<'_>) -> (BTreeMap, Vec) { let project_root = edge_toolkit::config::get_project_root(); ( - module_registry(&project_root, scenario.ws_server_dir), + module_registry(&project_root, scenario.ws_server_dir, scenario.module_paths), scenario_modules(scenario.module_names, scenario.serves_a_page), ) } -/// The tool that stages one resolved module, which differs by where the module comes from. -/// -/// A module of this repository is staged as the package its `package.json` declares, which already carries the owner -/// scope the registry requires. Anything staged by a mise tool keeps the tool it was registered with, which is somebody -/// else's and must not be rewritten. -fn staged_module(entry: &ModuleRegistryEntry) -> String { - match &entry.source { - ModuleSource::Repo(_) => { - let declared = entry.package_name.clone().unwrap_or_default(); - format!("{STAGED_PREFIX}{declared}") - } - ModuleSource::MiseTool { tool, .. } => tool.clone(), - } -} - pub fn scenario_module_paths(scenario: &ScenarioModules<'_>) -> Result, CliError> { let (registry, modules) = registry_and_modules(scenario); resolve_module_paths(®istry, &modules, |entry| entry.mise_path.clone()) @@ -472,10 +544,10 @@ fn module_tool_value(tool: &str, latest: &Value) -> Value { /// The same waiver, for a tool named outright rather than resolved from a scenario's module list. /// -/// The hub and the runners are this project's own binaries, published by the release that produced the -/// deployment, so the reasoning above applies to them unchanged: a deployment generated beside a fresh publish -/// would otherwise install yesterday's binary for a day and give no sign of it, since resolving `latest` to an -/// older release is what mise does rather than an error. `cargo:open` is somebody else's and keeps the delay. +/// The hub and the runners are this project's own binaries, published by the release that produced the deployment, so +/// the reasoning above applies to them unchanged: a deployment generated beside a fresh publish would otherwise install +/// yesterday's binary for a day and give no sign of it, since resolving `latest` to an older release is what mise does +/// rather than an error. `cargo:open` is somebody else's and keeps the delay. fn waived(latest: &Value) -> Value { let mut options = Table::new(); let _previous: Option = options.insert("version".to_string(), latest.clone()); diff --git a/utilities/cli/src/deployment_types/mod.rs b/utilities/cli/src/deployment_types/mod.rs index 0c08e722..e991bf11 100644 --- a/utilities/cli/src/deployment_types/mod.rs +++ b/utilities/cli/src/deployment_types/mod.rs @@ -24,9 +24,9 @@ pub(crate) fn serves_a_page(cluster: &ClusterInput) -> bool { /// The module the hub serves at `/`, named as its own `package.json` declares it. /// /// Read from that manifest rather than written out, so a generated deployment carries whatever the page module is -/// actually called -- including the owner scope publishing puts on it -- without this crate holding a second copy -/// of the name to drift from it. Every deployment format that serves the page needs it: the hub has no default for -/// which module is a deployment's front page, so the deployment that knows names it. +/// actually called -- including the owner scope publishing puts on it -- without this crate holding a second copy of +/// the name to drift from it. Every deployment format that serves the page needs it: the hub has no default for which +/// module is a deployment's front page, so the deployment that knows names it. pub(crate) fn hub_root_module(ws_server_dir: &Path) -> String { crate::module_package_json(&ws_server_dir.join("static")) .and_then(|package| package.name) @@ -35,5 +35,6 @@ pub(crate) fn hub_root_module(ws_server_dir: &Path) -> String { pub use self::docker_compose::{docker_image_module_paths, generate_docker_compose_deployment}; pub use self::k3s::generate_k3s_deployment; +pub(crate) use self::mise::STORAGE_DIR; pub use self::mise::{ScenarioModules, generate_mise_deployment, scenario_module_paths}; pub use self::scenario_image::generate_scenario_image; diff --git a/utilities/cli/src/deployment_types/scenario_image.rs b/utilities/cli/src/deployment_types/scenario_image.rs index 8ea24789..df312299 100644 --- a/utilities/cli/src/deployment_types/scenario_image.rs +++ b/utilities/cli/src/deployment_types/scenario_image.rs @@ -11,33 +11,49 @@ use crate::{ModuleSource, cluster_module_names, module_registry, resolve_module_ /// Build context path of the hub image's Dockerfile, which the scenario image layers on top of. const HUB_DOCKERFILE: &str = "services/ws-server/Dockerfile"; +/// Modules the hub image already carries at these paths, which a scenario image layered on it does not copy again. +/// +/// A module that depends on one of them still resolves it, so what it declares in turn is staged as usual; only the +/// module's own files are skipped. +const HUB_IMAGE_MODULES: [&str; 2] = ["/app/services/ws-server/static", "/app/services/ws-wasm-agent"]; + /// Guest languages the dependency stage loads so every `[tools]` pin is in scope. /// -/// A mise tool is only installable when the config declaring it is loaded, and the packages modules depend on -/// are spread across several guest configs (`npm:onnxruntime-web` in js, `http:pyodide` in python). Declaring -/// the full set costs nothing -- only the tools named on the `mise install` line are actually fetched. +/// A mise tool is only installable when the config declaring it is loaded, and the packages modules depend on are +/// spread across several guest configs (`npm:onnxruntime-web` in js, `http:pyodide` in python). Declaring the full set +/// costs nothing -- only the tools named on the `mise install` line are actually fetched. const DEPS_MISE_ENV: &str = "dart,dotnet,java,js,kotlin,python,r,rust,zig"; pub fn generate_scenario_image(cluster: &ClusterInput, output_dir: &Path) -> Result<(), CliError> { let project_root = edge_toolkit::config::get_project_root(); let ws_server_dir = project_root.join("services/ws-server"); - let registry = module_registry(&project_root, &ws_server_dir); + let registry = module_registry(&project_root, &ws_server_dir, &cluster.module_paths); let module_names = cluster_module_names(cluster); let sources = resolve_module_sources(®istry, &module_names)?; let mut repo_paths = Vec::new(); let mut mise_tools = Vec::new(); + let mut scenario_dirs = Vec::new(); for (docker_path, source) in sources { match source { + ModuleSource::Repo(_) if HUB_IMAGE_MODULES.contains(&docker_path.as_str()) => {} ModuleSource::Repo(repo_path) => repo_paths.push((repo_path, docker_path)), ModuleSource::MiseTool { tool, package } => mise_tools.push((tool, package, docker_path)), + ModuleSource::Scenario { + context, context_path, .. + } => scenario_dirs.push((context, context_path, docker_path)), } } + // Sorted out of resolution order, which moves a module whenever an unrelated one gains a dependency, so each list + // this file is rendered from stays put in a committed image: `SCENARIO_TOOLS`, the staging steps, and the COPYs. + repo_paths.sort(); + mise_tools.sort(); + scenario_dirs.sort(); - // Trimmed to exactly one trailing newline, because each section appends its own separating blank line - // and whichever section ends up last therefore leaves one behind. A scenario with no modules at all has - // no `COPY` sections, so the separator after the label is the end of the file. - let dockerfile = render_dockerfile(&cluster.cluster_name, &repo_paths, &mise_tools); + // Trimmed to exactly one trailing newline, because each section appends its own separating blank line and whichever + // section ends up last therefore leaves one behind. A scenario with no modules at all has no `COPY` sections, so + // the separator after the label is the end of the file. + let dockerfile = render_dockerfile(&cluster.cluster_name, &repo_paths, &mise_tools, &scenario_dirs); fs::write(output_dir.join("Dockerfile"), format!("{}\n", dockerfile.trim_end()))?; fs::write( output_dir.join("Dockerfile.dockerignore"), @@ -47,11 +63,21 @@ pub fn generate_scenario_image(cluster: &ClusterInput, output_dir: &Path) -> Res Ok(()) } +/// A path as one operand of a JSON-array `COPY`, which it has to be once the path is not this repository's own. +/// +/// A `module_paths:` directory is named by whoever laid out that tree, so it can hold a space -- which the plain form +/// splits into two operands -- or a quote or backslash, which JSON escapes. A `$` is escaped for the Dockerfile itself, +/// since variable substitution still runs on each operand of the JSON form. +fn dockerfile_word(path: &str) -> String { + serde_json::Value::String(path.replace('$', "\\$")).to_string() +} + /// Render the scenario image, which layers this cluster's modules onto the hub image. fn render_dockerfile( cluster_name: &str, repo_paths: &[(String, String)], mise_tools: &[(String, String, String)], + scenario_dirs: &[(String, String, String)], ) -> String { let mut out = String::default(); out.push_str("# syntax=docker/dockerfile:1\n\n"); @@ -79,8 +105,8 @@ fn render_dockerfile( out.push_str(&render_deps_stage(mise_tools)); } - // Each section above ends with its own blank line, so none of them start with one. - // Doing it the other way round doubles the separator wherever two sections meet. + // Each section above ends with its own blank line, so none of them start with one. Doing it the other way round + // doubles the separator wherever two sections meet. out.push_str(concat!( "# `hub` is a named build context, not a stage defined here.\n", "# The generated compose file wires it to the hub service with `service:`, so compose builds the hub\n", @@ -96,13 +122,12 @@ fn render_dockerfile( "\n", )); - // `pkg/` unconditionally, never probed for on disk. - // Every module serves out of `pkg/`, whether that directory is committed (the JS shims) or produced by a - // build (wasm-pack output, wheels). Probing for it makes this file depend on what happens to be built in the - // working tree: CI checks out unbuilt Rust modules, found no `services/ws-modules/har1/pkg`, and emitted the - // module root instead -- so the committed output and the CI regeneration disagreed, and the drift check - // failed. A module that genuinely has no `pkg/` now fails loudly at image build rather than silently - // generating a different Dockerfile per machine. + // `pkg/` unconditionally, never probed for on disk. Every module serves out of `pkg/`, whether that directory is + // committed (the JS shims) or produced by a build (wasm-pack output, wheels). Probing for it makes this file depend + // on what happens to be built in the working tree: CI checks out unbuilt Rust modules, found no `services/ws- + // modules/har1/pkg`, and emitted the module root instead -- so the committed output and the CI regeneration + // disagreed, and the drift check failed. A module that genuinely has no `pkg/` now fails loudly at image build + // rather than silently generating a different Dockerfile per machine. for (repo_path, docker_path) in repo_paths { let _write_result = writeln!(out, "COPY --chown=10001:10001 {repo_path}/pkg {docker_path}/pkg"); } @@ -112,14 +137,24 @@ fn render_dockerfile( "COPY --from=deps --chown=10001:10001 /staged{docker_path} {docker_path}" ); } + // Each from the named build context its `module_paths:` entry is handed to the build as, and whole rather than + // its `pkg/`: whether the module serves from `pkg/` or its root is the scenario author's layout, and the hub finds + // either once the directory is there. hadolint reads a context name as an undefined stage, as with `hub` above. + for (context, context_path, docker_path) in scenario_dirs { + let operands = format!("[{}, {}]", dockerfile_word(context_path), dockerfile_word(docker_path)); + let _write_result = writeln!( + out, + "# hadolint ignore=DL3022\nCOPY --from={context} --chown=10001:10001 {operands}" + ); + } out } /// The fixed head of the deps stage: the base image, the packages it installs, and a verified mise. /// -/// Held as a const rather than inlined so the rendering function below stays a short composer of named parts. -/// Nothing in this block varies with the cluster, so there is nothing here to parameterise. +/// Held as a const rather than inlined so the rendering function below stays a short composer of named parts. Nothing +/// in this block varies with the cluster, so there is nothing here to parameterise. const DEPS_BASE_STAGE: &str = concat!( "# Stage the packages that live outside the repository.\n", "# These are published packages the modules load at runtime, provisioned by mise rather than built\n", @@ -182,14 +217,13 @@ const DEPS_BASE_STAGE: &str = concat!( /// The per-tool copy step, driven entirely by the `STAGE_*` variables the loop emits ahead of it. /// -/// Locates the package rather than assuming where the backend put it. The npm backend nests it under one of -/// several layouts (`lib/node_modules/`, `node_modules/`, or an aube virtual store below a content-hashed -/// directory) that vary by platform, so this searches for the named directory instead. The named directory is -/// tried FIRST and the install root only as a fallback: the npm backend drops its own wrapper manifest -/// (`"name": "mise-npm-install"`) at that root, so a root-first probe stages the wrapper and the module is then -/// served under the wrapper's name. The fallback is what covers an archive-backed `http:` tool, which extracts -/// flat so the root really is the package. `-L` and `cp -L` resolve the symlinks the aube store is built from, -/// so real files land in the image. +/// Locates the package rather than assuming where the backend put it. The npm backend nests it under one of several +/// layouts (`lib/node_modules/`, `node_modules/`, or an aube virtual store below a content-hashed directory) that vary +/// by platform, so this searches for the named directory instead. The named directory is tried FIRST and the install +/// root only as a fallback: the npm backend drops its own wrapper manifest (`"name": "mise-npm-install"`) at that root, +/// so a root-first probe stages the wrapper and the module is then served under the wrapper's name. The fallback is +/// what covers an archive-backed `http:` tool, which extracts flat so the root really is the package. `-L` and `cp -L` +/// resolve the symlinks the aube store is built from, so real files land in the image. const DEPS_STAGE_TOOL_COPY: &str = concat!( "RUN bash <<'EOF'\n", "set -euo pipefail\n", @@ -211,8 +245,8 @@ const DEPS_STAGE_TOOL_COPY: &str = concat!( /// Render the stage that installs the mise-staged packages the cluster's modules depend on. /// -/// The repo's own `.mise/` configs and lockfiles are the version source, so nothing here pins a version that -/// could drift from what a workstation resolves; only the tool ids come from the module registry. +/// The repo's own `.mise/` configs and lockfiles are the version source, so nothing here pins a version that could +/// drift from what a workstation resolves; only the tool ids come from the module registry. fn render_deps_stage(mise_tools: &[(String, String, String)]) -> String { let tools = mise_tools .iter() @@ -223,12 +257,12 @@ fn render_deps_stage(mise_tools: &[(String, String, String)]) -> String { stage.push_str(DEPS_BASE_STAGE); let _write_result = writeln!(stage, "ENV MISE_ENV={DEPS_MISE_ENV}"); if mise_tools.iter().any(|(tool, _, _)| tool.starts_with("npm:")) { - // mise's npm backend shells out to npm, and refuses an `npm:` tool whose configured `node` dependency - // is not installed yet ("requires configured install dependency 'node@22', but its selected version is - // not installed"). Its version comes from the repo's own [tools] pin, like every other tool staged here. - // The blank line after `EOF` is load-bearing, not formatting. - // hadolint's parser reads the next instruction as a continuation of the heredoc without it and fails - // with `unexpected 'E' expecting a new line followed by the next instruction`. + // mise's npm backend shells out to npm, and refuses an `npm:` tool whose configured `node` dependency is + // not installed yet ("requires configured install dependency 'node@22', but its selected version is not + // installed"). Its version comes from the repo's own [tools] pin, like every other tool staged here. The + // blank line after `EOF` is load-bearing, not formatting. hadolint's parser reads the next instruction as a + // continuation of the heredoc without it and fails with `unexpected 'E' expecting a new line followed by the + // next instruction`. stage.push_str(concat!( "RUN bash <<'EOF'\n", "set -euo pipefail\n", @@ -259,9 +293,9 @@ fn render_deps_stage(mise_tools: &[(String, String, String)]) -> String { /// Render the ignore file `BuildKit` applies to this Dockerfile in place of the repository root one. /// /// It is the root file verbatim plus a re-include block, rather than a hand-picked subset, so a rule added to -/// `.gitignore` reaches this build too. Docker's matcher is last-match-wins and, unlike git's, still descends -/// into an excluded directory to honour a later negation, so the trailing block is enough to bring back the -/// built artifacts while every other exclusion keeps applying. +/// `.gitignore` reaches this build too. Docker's matcher is last-match-wins and, unlike git's, still descends into an +/// excluded directory to honour a later negation, so the trailing block is enough to bring back the built artifacts +/// while every other exclusion keeps applying. fn render_dockerignore( project_root: &Path, cluster_name: &str, diff --git a/utilities/cli/src/error.rs b/utilities/cli/src/error.rs index a252c8ab..bc0c5659 100644 --- a/utilities/cli/src/error.rs +++ b/utilities/cli/src/error.rs @@ -13,9 +13,9 @@ pub enum CliError { #[error(transparent)] Io(#[from] std::io::Error), - // The underlying message carries the whole of what is actionable -- which field, which line, and for a - // rejected enum the values that would have been accepted -- and a `source` nobody prints is a message - // that stops at "it did not parse". + // The underlying message carries the whole of what is actionable -- which field, which line, and for a rejected + // enum the values that would have been accepted -- and a `source` nobody prints is a message that stops at "it did + // not parse". #[error("Failed to parse cluster input YAML: {0}")] ParseClusterYaml(#[from] serde_yaml::Error), diff --git a/utilities/cli/src/input.rs b/utilities/cli/src/input.rs index dd72de02..22ac35da 100644 --- a/utilities/cli/src/input.rs +++ b/utilities/cli/src/input.rs @@ -1,4 +1,5 @@ use std::collections::BTreeMap; +use std::path::PathBuf; use clap::ValueEnum; use edge_toolkit::config::get_project_root; @@ -7,16 +8,16 @@ use serde::{Deserialize, Serialize}; /// This repository, as its own workspace manifest states it. /// -/// Compared against rather than merely looked for, because the question it answers is identity: whether the -/// tree this process resolved is the one that builds these artifacts, not whether some Rust project is nearby. +/// Compared against rather than merely looked for, because the question it answers is identity: whether the tree this +/// process resolved is the one that builds these artifacts, not whether some Rust project is nearby. const REPOSITORY_URL: &str = et_org::REPOSITORY_URL; /// The deployment format a scenario is rendered into. /// -/// An enum rather than the string it is written as, because the set is closed and every consumer switches on -/// it: as a string, a misspelling reaches the generator, and the three writers of these files each have to -/// re-decide what an unrecognised one means. Deserialization rejects anything outside the set instead, naming -/// the alternatives, at the point the input is read. +/// An enum rather than the string it is written as, because the set is closed and every consumer switches on it: as +/// a string, a misspelling reaches the generator, and the three writers of these files each have to re-decide what an +/// unrecognised one means. Deserialization rejects anything outside the set instead, naming the alternatives, at the +/// point the input is read. #[expect( clippy::exhaustive_enums, reason = "OutputType enumerates the supported deployment formats; downstream code matches exhaustively" @@ -53,18 +54,29 @@ pub struct ClusterInput { pub deployment_type: OutputType, /// Where the artifacts a generated deployment consumes come from, from `artifact_source:`. /// - /// Stated, it is honoured as written, in both directions. Left out, it is settled while the input is read - /// -- see [`infer_artifact_source`] -- so that by the time anything holds a `ClusterInput` the question is - /// answered and there is no second, later notion of "unset" for a consumer to re-resolve. + /// Stated, it is honoured as written, in both directions. Left out, it is settled while the input is read -- see + /// [`infer_artifact_source`] -- so that by the time anything holds a `ClusterInput` the question is answered and + /// there is no second, later notion of "unset" for a consumer to re-resolve. #[serde(default = "infer_artifact_source")] pub artifact_source: ArtifactSource, /// The agents this cluster runs, from `agents:`. /// - /// Defaults to none, which is a cluster of the hub and its collector and nothing else. That is a real - /// deployment rather than a degenerate one -- it serves the UI and accepts agents that connect to it -- - /// and it is what a scenario stating nothing but its name describes. + /// Defaults to none, which is a cluster of the hub and its collector and nothing else. That is a real deployment + /// rather than a degenerate one -- it serves the UI and accepts agents that connect to it -- and it is what a + /// scenario stating nothing but its name describes. #[serde(default)] pub agents: Vec, + /// Module directories the scenario names itself, from `module_paths:`, alongside the ones the registry knows. + /// + /// Each entry is either a module directory itself or a parent of several, the same two shapes the hub's own + /// `MODULES_PATHS` accepts. Written relative to the input file, so a scenario can name the modules it keeps beside + /// it wherever the CLI is run from; reading the input resolves them to absolute paths, so nothing downstream has to + /// know where the input was. + /// + /// These modules come from their directories whatever the artifact source: they belong to whoever wrote the + /// scenario, so there is no release of them for a published deployment to stage instead. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub module_paths: Vec, } /// The name a cluster takes when its input does not choose one. @@ -73,33 +85,32 @@ pub const DEFAULT_CLUSTER_NAME: &str = "default"; impl Default for ClusterInput { /// The smallest cluster that is still valid: named, no agents, rendered the way an unstated input is. /// - /// Not derived, because a derived `cluster_name` is the empty string and an empty name is one of the - /// things scenario validation exists to reject -- a `Default` that cannot be generated from would be a - /// trap rather than a starting point. `artifact_source` is the enum's own default rather than the - /// inferred one, because this is the value with nothing to infer from; reading an input is where a tree - /// gets a say. + /// Not derived, because a derived `cluster_name` is the empty string and an empty name is one of the things + /// scenario validation exists to reject -- a `Default` that cannot be generated from would be a trap rather than + /// a starting point. `artifact_source` is the enum's own default rather than the inferred one, because this is the + /// value with nothing to infer from; reading an input is where a tree gets a say. fn default() -> Self { Self { cluster_name: DEFAULT_CLUSTER_NAME.to_string(), deployment_type: OutputType::default(), artifact_source: ArtifactSource::default(), agents: Vec::new(), + module_paths: Vec::new(), } } } /// Which copy of this project's own artifacts a generated deployment addresses. /// -/// `Local` builds everything from the working tree: a Kubernetes manifest names image tags the deploying -/// machine has to produce and import into the node, and a `mise` deployment runs its binaries out of the cargo -/// workspace. `Published` addresses what has been released instead -- container images from the project's -/// registry, binaries from the crates.io releases -- so a deployment consumes the project rather than -/// rebuilding it. What a scenario cannot get either way is its own module set, which is built from the tree -/// because it is particular to that deployment. +/// `Local` builds everything from the working tree: a Kubernetes manifest names image tags the deploying machine +/// has to produce and import into the node, and a `mise` deployment runs its binaries out of the cargo workspace. +/// `Published` addresses what has been released instead -- container images from the project's registry, binaries from +/// the crates.io releases -- so a deployment consumes the project rather than rebuilding it. What a scenario cannot get +/// either way is its own module set, which is built from the tree because it is particular to that deployment. /// -/// `Published` is the default because it is the only one of the two that always works. Building from the -/// working tree needs a working tree, so `Local` is an answer available to almost nobody: everyone generating -/// a deployment for their own cluster has the releases and not this repository. +/// `Published` is the default because it is the only one of the two that always works. Building from the working tree +/// needs a working tree, so `Local` is an answer available to almost nobody: everyone generating a deployment for their +/// own cluster has the releases and not this repository. #[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)] #[non_exhaustive] #[serde(rename_all = "lowercase")] @@ -111,9 +122,9 @@ pub enum ArtifactSource { /// Settle where an input that did not say gets its artifacts from. /// -/// `Published`, except where there is proof it need not be: inside this repository every artifact can be -/// built from the tree, and building what you are working on is the whole point of generating a deployment -/// there. Anywhere else the releases are the only thing that exists. +/// `Published`, except where there is proof it need not be: inside this repository every artifact can be built from the +/// tree, and building what you are working on is the whole point of generating a deployment there. Anywhere else the +/// releases are the only thing that exists. #[must_use] pub fn infer_artifact_source() -> ArtifactSource { if running_in_this_repository() { @@ -125,10 +136,10 @@ pub fn infer_artifact_source() -> ArtifactSource { /// Whether this process is running inside the repository that builds these artifacts. /// -/// Two decisions turn on it, and they are the same question asked twice: whether a deployment can be built -/// from the tree, and whether a generated credential belongs to a committed fixture or to someone's real -/// deployment. A root with no readable manifest is not this repository, which is the same answer as a root -/// whose manifest belongs to something else, so the read failing needs no handling of its own. +/// Two decisions turn on it, and they are the same question asked twice: whether a deployment can be built from the +/// tree, and whether a generated credential belongs to a committed fixture or to someone's real deployment. A root with +/// no readable manifest is not this repository, which is the same answer as a root whose manifest belongs to something +/// else, so the read failing needs no handling of its own. #[must_use] pub fn running_in_this_repository() -> bool { let manifest = get_project_root().join("Cargo.toml"); @@ -137,10 +148,10 @@ pub fn running_in_this_repository() -> bool { /// Whether a workspace manifest is this repository's. /// -/// Split from the file read so the decision is a pure function of the text and can be exercised over every -/// shape a manifest takes -- another project's, one with no `[workspace.package]`, one that is not TOML at -/// all -- none of which is reachable through the caller, since the caller can only ever see whichever -/// manifest the running process happens to sit under. Public for that reason and no other. +/// Split from the file read so the decision is a pure function of the text and can be exercised over every shape a +/// manifest takes -- another project's, one with no `[workspace.package]`, one that is not TOML at all -- none of which +/// is reachable through the caller, since the caller can only ever see whichever manifest the running process happens +/// to sit under. Public for that reason and no other. #[must_use] pub fn manifest_declares_this_repository(manifest: &str) -> bool { let Ok(parsed) = manifest.parse::() else { @@ -160,19 +171,19 @@ pub struct Agent { pub name: String, /// Runner that executes this agent's modules, from `runner:`. /// - /// Unset means the modules are only served, for a browser to load and run itself -- which is what every - /// deployment did before this field existed. Set, the generated deployment also starts a runner process per - /// resource, so the cluster runs headless. + /// Unset means the modules are only served, for a browser to load and run itself -- which is what every deployment + /// did before this field existed. Set, the generated deployment also starts a runner process per resource, so the + /// cluster runs headless. #[serde(default, skip_serializing_if = "Option::is_none")] pub runner: Option, /// Extra environment for this agent's runner processes, from `env:`. /// /// Every deployment format has somewhere to put these -- a `mise` task's `[env]`, a compose service's - /// `environment:`, a Kubernetes container's `env:` -- so a scenario that needs a runner configured says - /// so once, here, rather than the operator editing three generated files that a regeneration overwrites. + /// `environment:`, a Kubernetes container's `env:` -- so a scenario that needs a runner configured says so once, + /// here, rather than the operator editing three generated files that a regeneration overwrites. /// - /// A `BTreeMap` so the rendering is ordered: the outputs are committed and diffed, and a hash map would - /// reorder them between runs for no reason anyone could act on. + /// A `BTreeMap` so the rendering is ordered: the outputs are committed and diffed, and a hash map would reorder + /// them between runs for no reason anyone could act on. #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] pub env: BTreeMap, pub resources: Vec, diff --git a/utilities/cli/src/lib.rs b/utilities/cli/src/lib.rs index 9f408139..5e8c9a83 100644 --- a/utilities/cli/src/lib.rs +++ b/utilities/cli/src/lib.rs @@ -8,7 +8,7 @@ use std::fmt::Write as _; use std::path::{Path, PathBuf}; use edge_toolkit::ports::Services; -use et_path::relative_path_from; +use et_path::{absolute_from, relative_path_from}; use fs_err as fs; use serde::Deserialize; @@ -19,11 +19,11 @@ mod input; mod module_package_json; mod scenario_password; -// `pub` here means "reachable from the binary or from `tests/`", and nothing else. -// This crate is a command line tool that happens to be split into a lib target so integration tests can drive -// it; no consumer outside this directory builds on it, and nothing exported is a promise. Everything the -// generators share among themselves is `pub(crate)`, so what remains below is the whole of the surface anyone -// could depend on -- short enough to read, which is what makes an accidental addition to it visible. +// `pub` here means "reachable from the binary or from `tests/`", and nothing else. This crate is a command line tool +// that happens to be split into a lib target so integration tests can drive it; no consumer outside this directory +// builds on it, and nothing exported is a promise. Everything the generators share among themselves is `pub(crate)`, so +// what remains below is the whole of the surface anyone could depend on -- short enough to read, which is what makes an +// accidental addition to it visible. pub use self::deployment_types::{ScenarioModules, docker_image_module_paths, scenario_module_paths}; pub(crate) use self::deployment_types::{ generate_docker_compose_deployment, generate_k3s_deployment, generate_mise_deployment, generate_scenario_image, @@ -119,9 +119,9 @@ struct CargoWsModule { /// Where a module's served files come from. /// -/// The deployment generators need this to tell apart the two provisioning routes: a repo directory can be -/// copied straight out of the Docker build context, whereas a mise-staged package exists only in the tool's -/// install dir and has to be installed before it can be staged. +/// The deployment generators need this to tell apart the two provisioning routes: a repo directory can be copied +/// straight out of the Docker build context, whereas a mise-staged package exists only in the tool's install dir and +/// has to be installed before it can be staged. #[derive(Debug, Clone, PartialEq, Eq)] #[non_exhaustive] pub(crate) enum ModuleSource { @@ -129,11 +129,40 @@ pub(crate) enum ModuleSource { Repo(String), /// A package staged by a mise tool. /// - /// Holds the backend-qualified tool id plus the published package name to locate beneath its install - /// directory. The name rather than a path because the npm backend has no single layout: a package lands - /// under `lib/node_modules/`, `node_modules/`, or an aube virtual store keyed by a content hash, - /// depending on backend and platform, so the directory has to be found rather than assumed. + /// Holds the backend-qualified tool id plus the published package name to locate beneath its install directory. The + /// name rather than a path because the npm backend has no single layout: a package lands under `lib/node_modules/`, + /// `node_modules/`, or an aube virtual store keyed by a content hash, depending on backend and platform, so the + /// directory has to be found rather than assumed. MiseTool { tool: String, package: String }, + /// A module directory the scenario names in its own `module_paths:`, wherever that directory is. + /// + /// It belongs to whoever wrote the scenario, so there is no release of it to stage. A deployment that runs where + /// the directory is reads it in place, from `dir`; an image copies it in from the named build context of the + /// `module_paths:` entry it came from, `context`, at `context_path` within it. + Scenario { + dir: PathBuf, + context: String, + context_path: String, + }, +} + +/// The named build context an image reads the `module_paths:` entry at `index` from. +/// +/// Indexed rather than named after the directory, since two entries can end in the same name and a context name is the +/// only handle a `COPY --from=` has. +pub(crate) fn module_path_context(index: usize) -> String { + format!("module-path-{index}") +} + +/// Every `module_paths:` entry as the named build context that supplies it, with its path relative to `base`. +/// +/// What a deployment building the scenario image hands to the build, alongside its main context. +pub(crate) fn module_path_contexts(module_paths: &[PathBuf], base: &Path) -> Vec<(String, String)> { + module_paths + .iter() + .enumerate() + .map(|(index, path)| (module_path_context(index), relative_path_from(base, path))) + .collect() } /// Where the hub serves pyodide from, whichever distribution was selected. @@ -141,43 +170,43 @@ const PYODIDE_DOCKER_PATH: &str = "/app/node_modules/pyodide"; /// Name of the generated env file that carries the scenario's derived credential. /// -/// The password is derived from the scenario input so a deployment is reproducible from it, which used to mean -/// writing the literal into `mise.toml` and `compose.yaml` -- both committed under `verification/`, where every -/// secret scanner duly found it. Collecting it into one file keeps the deployment reproducible while leaving -/// the rest of the generated output free of anything a scanner reads as a credential. +/// The password is derived from the scenario input so a deployment is reproducible from it, which used to mean writing +/// the literal into `mise.toml` and `compose.yaml` -- both committed under `verification/`, where every secret scanner +/// duly found it. Collecting it into one file keeps the deployment reproducible while leaving the rest of the generated +/// output free of anything a scanner reads as a credential. /// -/// Whether that one file is committed depends on where it was generated, and [`write_deployment_gitignore`] -/// decides: under `verification/` it is a fixture the drift check reads, and anywhere else it is a real -/// credential that gets an ignore file written beside it. +/// Whether that one file is committed depends on where it was generated, and [`write_deployment_gitignore`] decides: +/// under `verification/` it is a fixture the drift check reads, and anywhere else it is a real credential that the +/// ignore file written beside it lists. pub(crate) const SECRETS_ENV_FILE: &str = "secrets.env"; /// Account the collector is created with, and the one the hub authenticates its OTLP exports as. /// /// One constant because the two are the same account seen from either end: the collector is created with it as -/// `ZO_ROOT_USER_EMAIL` and the hub presents it as `OTLP_AUTH_USERNAME`, so a deployment where they disagree -/// comes up healthy and then rejects every export. Written in two files before this existed, with nothing -/// checking that they matched. +/// `ZO_ROOT_USER_EMAIL` and the hub presents it as `OTLP_AUTH_USERNAME`, so a deployment where they disagree comes up +/// healthy and then rejects every export. Written in two files before this existed, with nothing checking that they +/// matched. pub const COLLECTOR_USERNAME: &str = "root@example.com"; /// The collector's non-secret settings, which every generated deployment carries rather than sourcing. /// -/// Three formats render this -- a `ConfigMap`, a compose `environment:` block, a `docker run` flag list -- so -/// it is one list and a setting cannot reach some deployments and not others. Reading it from a file in this -/// repository instead is what made a generated deployment unable to leave the tree, for the sake of two values -/// neither secret nor scenario-specific. +/// Three formats render this -- a `ConfigMap`, a compose `environment:` block, a `docker run` flag list -- so it is +/// one list and a setting cannot reach some deployments and not others. Reading it from a file in this repository +/// instead is what made a generated deployment unable to leave the tree, for the sake of two values neither secret nor +/// scenario-specific. /// -/// Two things are deliberately absent. The root password is derived per scenario and reaches each format from -/// the generated env file. `ZO_DATA_DIR` is per format rather than shared, because it only means anything -/// alongside the storage that format declares -- a named volume, a claim, or nothing at all. +/// Two things are deliberately absent. The root password is derived per scenario and reaches each format from the +/// generated env file. `ZO_DATA_DIR` is per format rather than shared, because it only means anything alongside the +/// storage that format declares -- a named volume, a claim, or nothing at all. pub(crate) const COLLECTOR_SETTINGS: [(&str, &str); 2] = [("RUST_LOG", "warn"), ("ZO_ROOT_USER_EMAIL", COLLECTOR_USERNAME)]; /// Registry path the repository's own images are published under. /// -/// A runner image is the same for every deployment -- nothing in one varies by scenario -- so a scenario that -/// asks for published images names it here and the cluster pulls it, leaving the node with nothing to build or -/// import. The hub image is published alongside them but no manifest names it: it is the base a scenario image -/// is layered onto, so it reaches a deployment as a build context rather than as something a pod runs. +/// A runner image is the same for every deployment -- nothing in one varies by scenario -- so a scenario that asks for +/// published images names it here and the cluster pulls it, leaving the node with nothing to build or import. The hub +/// image is published alongside them but no manifest names it: it is the base a scenario image is layered onto, so it +/// reaches a deployment as a build context rather than as something a pod runs. pub(crate) const IMAGE_REGISTRY: &str = et_org::IMAGE_REGISTRY; /// Prefix that turns a bare image name into the one a scenario's `artifact_source` asks for. @@ -202,9 +231,9 @@ pub(crate) struct ModuleRegistryEntry { pub source: ModuleSource, /// The module's published package name, as `pkg/package.json` declares it. /// - /// This is what a runner has to be told: `RUNNER_MODULE` is resolved against the names the hub serves - /// modules under, which is the package name (`et-ws-math1`) and not the directory a scenario names it by - /// (`math1`). `None` for a mise-staged package, which is already keyed by its published name. + /// This is what a runner has to be told: `RUNNER_MODULE` is resolved against the names the hub serves modules + /// under, which is the package name (`et-ws-math1`) and not the directory a scenario names it by (`math1`). `None` + /// for a mise-staged package, which is already keyed by its published name. pub package_name: Option, } @@ -233,8 +262,19 @@ pub fn generate_deployment( /// formatting included, rather than a re-serialization of the parsed struct. pub(crate) fn load_cluster_input(input_file: &Path) -> Result<(ClusterInput, u64), CliError> { let content = fs::read(input_file)?; - let cluster: ClusterInput = serde_yaml::from_slice(&content)?; + let mut cluster: ClusterInput = serde_yaml::from_slice(&content)?; validate_cluster_name(&cluster.cluster_name)?; + // Relative to the input file, which is the one place a scenario can name its own modules from without knowing where + // the CLI is run. The input itself is resolved the way `fs::read` just resolved it, against the working directory, + // so the two cannot disagree when the CLI runs from a subdirectory of a repository. A path with no parent is a + // filesystem root, which is its own directory. + let input_abs = std::path::absolute(input_file)?; + let input_dir = input_abs.parent().unwrap_or(&input_abs); + cluster.module_paths = cluster + .module_paths + .iter() + .map(|path| absolute_from(input_dir, path)) + .collect(); Ok((cluster, scenario_seed(&content))) } @@ -246,15 +286,15 @@ const CLUSTER_NAME_MAX: usize = 60; /// Reject a `cluster_name` that is not an RFC 1123 label. /// -/// The name reaches three renderers that each read it as trusted text: it is interpolated into generated -/// comments, into shell commands in the generated README, and into Kubernetes object names. The comment case is -/// an instruction-injection vector on its own -- a name carrying a newline closes the scenario Dockerfile's -/// `# AUTO-GENERATED by ...` comment and everything after it is read by `BuildKit` as further instructions. -/// The README case is the same hazard against a shell, where `;` or a backtick would end the `scenario=` -/// assignment and start a command. Rather than escape per renderer, the name is held to the narrowest alphabet -/// any of them needs, which is the one Kubernetes already demands of a namespace: lowercase alphanumerics and -/// `-`, starting and ending alphanumeric. That leaves nothing to escape anywhere, and it turns a name the API -/// server would have rejected at `kubectl apply` into an error at generation time. +/// The name reaches three renderers that each read it as trusted text: it is interpolated into generated comments, +/// into shell commands in the generated README, and into Kubernetes object names. The comment case is an instruction- +/// injection vector on its own -- a name carrying a newline closes the scenario Dockerfile's `# AUTO-GENERATED by ...` +/// comment and everything after it is read by `BuildKit` as further instructions. The README case is the same hazard +/// against a shell, where `;` or a backtick would end the `scenario=` assignment and start a command. Rather +/// than escape per renderer, the name is held to the narrowest alphabet any of them needs, which is the one Kubernetes +/// already demands of a namespace: lowercase alphanumerics and `-`, starting and ending alphanumeric. That leaves +/// nothing to escape anywhere, and it turns a name the API server would have rejected at `kubectl apply` into an error +/// at generation time. fn validate_cluster_name(name: &str) -> Result<(), CliError> { let invalid = |reason: &str| { Err(CliError::InvalidClusterName { @@ -334,9 +374,8 @@ fn generate_deployment_outputs( fs::create_dir_all(output_dir)?; } - // One password per scenario, shared by both deployment formats. - // OpenObserve and the ws-server have to agree on it: the server authenticates its OTLP exports against the - // same root credentials the collector was started with. + // One password per scenario, shared by both deployment formats. OpenObserve and the ws-server have to agree on it: + // the server authenticates its OTLP exports against the same root credentials the collector was started with. let password = scenario_password(seed); fs::write(output_dir.join(SECRETS_ENV_FILE), secrets_env(&password))?; write_deployment_gitignore(output_dir)?; @@ -347,10 +386,10 @@ fn generate_deployment_outputs( generate_docker_compose_deployment(cluster, output_dir)?; generate_scenario_image(cluster, output_dir)?; } - // The scenario image is emitted here too, not just for compose. - // Both formats reference it: compose builds it as a service, and the manifests name it as the - // hub's image with the README giving the `docker build` for it. Generating k3s alone without it - // produced a README pointing at a Dockerfile that was never written. + // The scenario image is emitted here too, not just for compose. Both formats reference it: compose builds + // it as a service, and the manifests name it as the hub's image with the README giving the `docker build` + // for it. Generating k3s alone without it produced a README pointing at a Dockerfile that was never + // written. OutputType::K3s => { generate_k3s_deployment(cluster, output_dir)?; generate_scenario_image(cluster, output_dir)?; @@ -371,16 +410,15 @@ fn generate_deployment_outputs( /// The path the generated README tells a reader to build this scenario's image from. /// -/// Only the parent is rendered, so the command keeps naming the last segment through the `scenario` shell -/// variable it already sets rather than repeating the name. Joined from the path's own components rather than -/// displayed, because the result is committed: a `Display` of the same path writes `\` on Windows and `/` -/// everywhere else, which would make the file drift by platform and fail the check that holds it stable. -/// Regeneration passes a repository-relative path, which is what the command needs, since it runs from the -/// repository root. +/// Only the parent is rendered, so the command keeps naming the last segment through the `scenario` shell variable it +/// already sets rather than repeating the name. Joined from the path's own components rather than displayed, because +/// the result is committed: a `Display` of the same path writes `\` on Windows and `/` everywhere else, which would +/// make the file drift by platform and fail the check that holds it stable. Regeneration passes a repository-relative +/// path, which is what the command needs, since it runs from the repository root. /// -/// A single-component output directory has a parent, and it is the empty path rather than `None` -- so this -/// cannot lean on `unwrap_or` and has to test the rendered parent. Prefixing an empty one would produce -/// `/$scenario/Dockerfile`, an absolute path to a directory nobody has. +/// A single-component output directory has a parent, and it is the empty path rather than `None` -- so this cannot lean +/// on `unwrap_or` and has to test the rendered parent. Prefixing an empty one would produce `/$scenario/Dockerfile`, an +/// absolute path to a directory nobody has. #[must_use] pub fn scenario_dockerfile_path(output_dir: &Path) -> String { let parent = output_dir @@ -396,28 +434,44 @@ pub fn scenario_dockerfile_path(output_dir: &Path) -> String { format!("{parent}/$scenario/Dockerfile") } -/// Keep a generated deployment's credential out of whatever repository it was generated into. +/// Keep a generated deployment's credential and runtime state out of whatever repository it was generated into. /// -/// Written beside the file it covers rather than left to the operator, because the failure is silent and -/// permanent: a credential committed once stays in the history after it is deleted. A nested `.gitignore` is -/// what makes the deployment directory safe to drop anywhere, which is the point of generating it. +/// Written beside the files it covers rather than left to the operator, because the failure is silent and permanent: +/// a credential committed once stays in the history after it is deleted. A nested `.gitignore` is what makes the +/// deployment directory safe to drop anywhere, which is the point of generating it. /// -/// Not written inside this repository. Here the verification outputs are committed evidence -- their whole -/// purpose is to be diffed when a generator changes -- and the password is derived from an input this -/// repository also carries, so it is reproducible from what is already public rather than a secret the file -/// is keeping. An ignore file here would only hide them from the drift check that exists to read them. +/// Inside this repository the credential is left out. Here the verification outputs are committed evidence -- their +/// whole purpose is to be diffed when a generator changes -- and the password is derived from an input this repository +/// also carries, so it is reproducible from what is already public rather than a secret the file is keeping. Ignoring +/// it here would only hide it from the drift check that exists to read it. What the hub writes at runtime is ignored +/// everywhere, since it is never evidence of anything the generator did. fn write_deployment_gitignore(output_dir: &Path) -> Result<(), CliError> { - if running_in_this_repository() { - return Ok(()); - } + let credential = if running_in_this_repository() { + String::default() + } else { + format!( + concat!( + "# The password is derived from the scenario input, so regenerating this deployment rewrites it;\n", + "# committing it would publish the credential of every deployment generated from that input.\n", + "{file}\n", + "\n", + ), + file = SECRETS_ENV_FILE, + ) + }; let body = format!( concat!( - "# Written by `et-cli` beside the credential it covers.\n", - "# The password is derived from the scenario input, so regenerating this deployment rewrites it;\n", - "# committing it would publish the credential of every deployment generated from that input.\n", - "{file}\n", + "# Written by `et-cli`.\n", + "{credential}", + "# What the hub writes into the directory it runs from: its self-signed TLS pair (the key is a\n", + "# credential too), the agent registry it saves on shutdown, and agent storage.\n", + "cert.pem\n", + "key.pem\n", + "registry.yaml\n", + "{storage}/\n", ), - file = SECRETS_ENV_FILE + credential = credential, + storage = crate::deployment_types::STORAGE_DIR, ); fs::write(output_dir.join(".gitignore"), body)?; Ok(()) @@ -427,16 +481,16 @@ fn write_deployment_gitignore(output_dir: &Path) -> Result<(), CliError> { /// /// Two names for the one password because the services that share it read different variables: `OpenObserve` /// takes `ZO_ROOT_USER_PASSWORD` as its root credential, and the ws-server authenticates its OTLP exports with -/// `OTLP_AUTH_PASSWORD`. Nothing else belongs here. The account those two authenticate as is not a secret and -/// each format states it outright, so keeping it in this file would have meant an uncommitted file standing -/// between a reader and a value there was never any reason to withhold -- and, in the Kubernetes case, a -/// `Secret` holding something that is not one. +/// `OTLP_AUTH_PASSWORD`. Nothing else belongs here. The account those two authenticate as is not a secret and each +/// format states it outright, so keeping it in this file would have meant an uncommitted file standing between a reader +/// and a value there was never any reason to withhold -- and, in the Kubernetes case, a `Secret` holding something that +/// is not one. /// -/// Each value carries a `skipcq` pragma for `DeepSource SCT-A000`: the copies committed under -/// `verification/` are read as hardcoded credentials, and the path excludes that keep the rest of that tree -/// out of analysis do not reach a secrets scan. The pragma sits on the line above rather than at the end of -/// its own, because an env file has no inline comments: every consumer keeps what follows the `=` verbatim, -/// so a trailing marker would become part of the password. +/// Each value carries a `skipcq` pragma for `DeepSource SCT-A000`: the copies committed under `verification/` are +/// read as hardcoded credentials, and the path excludes that keep the rest of that tree out of analysis do not reach +/// a secrets scan. The pragma sits on the line above rather than at the end of its own, because an env file has no +/// inline comments: every consumer keeps what follows the `=` verbatim, so a trailing marker would become part of the +/// password. fn secrets_env(password: &str) -> String { format!( concat!( @@ -537,6 +591,15 @@ fn generated_readme( cluster.cluster_name, output_files ) }; + // Relative to the repository root, which is where the README's `docker build` runs from, and escaped because each + // lands inside a double-quoted shell argument. + let build_contexts = module_path_contexts(&cluster.module_paths, &edge_toolkit::config::get_project_root()) + .iter() + .fold(String::default(), |mut flags, (name, path)| { + let path = escape_for_double_quotes(path); + let _write_result = write!(flags, " --build-context \"{name}={path}\""); + flags + }); let run_instructions = output_types .iter() .map(|output_type| { @@ -546,6 +609,7 @@ fn generated_readme( dockerfile, &runner_kinds(cluster), cluster.artifact_source, + &build_contexts, ) }) .collect::>() @@ -571,11 +635,11 @@ fn generated_readme( /// State where this scenario's artifacts come from, for a scenario that does not build them. /// -/// Above the run sections rather than inside one, because it is true of every way this scenario starts: the -/// `mise` deployment runs released binaries, and the compose and Kubernetes ones run published images. Said -/// once per run mode it would be three copies of one fact, and said inside `mise` alone it would read as a -/// property of that mode. A local scenario says nothing here -- building what it runs is the unremarkable -/// case, and the run sections already show the builds. +/// Above the run sections rather than inside one, because it is true of every way this scenario starts: the `mise` +/// deployment runs released binaries, and the compose and Kubernetes ones run published images. Said once per run mode +/// it would be three copies of one fact, and said inside `mise` alone it would read as a property of that mode. A local +/// scenario says nothing here -- building what it runs is the unremarkable case, and the run sections already show +/// the builds. const fn artifact_source_note(artifacts: ArtifactSource) -> &'static str { if matches!(artifacts, ArtifactSource::Published) { return concat!( @@ -590,9 +654,9 @@ const fn artifact_source_note(artifacts: ArtifactSource) -> &'static str { /// Explain the credential file, since a deployment that reaches a new machine without it fails obscurely. /// -/// Worth saying out loud because its absence is silent: `mise` skips an `_.file` it cannot find without a -/// warning, and Docker Compose treats a missing `env_file` the same way, so a copy without it starts a -/// collector with no root password rather than failing. +/// Worth saying out loud because its absence is silent: `mise` skips an `_.file` it cannot find without a warning, +/// and Docker Compose treats a missing `env_file` the same way, so a copy without it starts a collector with no root +/// password rather than failing. fn secrets_note() -> String { format!( concat!( @@ -607,8 +671,8 @@ fn secrets_note() -> String { /// The distinct runner kinds a cluster names, in the order they first appear. /// -/// Deduplicated because a scenario with two agents on the same runner needs its image built once, and ordered -/// by first appearance rather than sorted so the generated instructions read in the order the input declares. +/// Deduplicated because a scenario with two agents on the same runner needs its image built once, and ordered by first +/// appearance rather than sorted so the generated instructions read in the order the input declares. fn runner_kinds(cluster: &ClusterInput) -> Vec { let mut kinds: Vec = Vec::new(); for agent in &cluster.agents { @@ -624,16 +688,16 @@ fn runner_kinds(cluster: &ClusterInput) -> Vec { /// Render the step that installs a published deployment's binaries, and nothing at all for a local one. /// -/// A local deployment compiles what it runs from the working tree, so `mise run` is the whole of it. A -/// published one declares its binaries as `cargo:` tools, and `task.run_auto_install` is off, so without this -/// step the first task dies on a command it cannot find rather than fetching it. Why the binaries are released -/// rather than built is said once, above the run sections, because it is true of every mode. +/// A local deployment compiles what it runs from the working tree, so `mise run` is the whole of it. A published one +/// declares its binaries as `cargo:` tools, and `task.run_auto_install` is off, so without this step the first task +/// dies on a command it cannot find rather than fetching it. Why the binaries are released rather than built is said +/// once, above the run sections, because it is true of every mode. /// -/// `NPM_CONFIG_USERCONFIG` is exported on the command line rather than left to the `[env]` beside it, which -/// carries the same value. mise computes that entry but does not apply it to its own tool resolution, so the -/// npm client it embeds reads no user config, falls back to registry.npmjs.org, and reports the scoped module -/// packages as `package not found` -- they exist only on GitHub Packages. Exported into the process, the same -/// file resolves against the right registry. The `[env]` entry stays because the tasks do get it. +/// `NPM_CONFIG_USERCONFIG` is exported on the command line rather than left to the `[env]` beside it, which carries the +/// same value. mise computes that entry but does not apply it to its own tool resolution, so the npm client it embeds +/// reads no user config, falls back to registry.npmjs.org, and reports the scoped module packages as `package not +/// found` -- they exist only on GitHub Packages. Exported into the process, the same file resolves against the right +/// registry. The `[env]` entry stays because the tasks do get it. const fn mise_install_note(artifacts: ArtifactSource) -> &'static str { if matches!(artifacts, ArtifactSource::Published) { return concat!( @@ -655,6 +719,7 @@ fn generated_run_instructions( dockerfile: &str, runners: &[String], artifacts: ArtifactSource, + build_contexts: &str, ) -> String { match output_type { OutputType::Mise => format!( @@ -693,17 +758,17 @@ fn generated_run_instructions( ), hub = compose_hub_note(artifacts) ), - OutputType::K3s => k3s_run_instructions(cluster_name, dockerfile, runners, artifacts), + OutputType::K3s => k3s_run_instructions(cluster_name, dockerfile, runners, artifacts, build_contexts), } } /// Explain where the module-less hub image the scenario layers onto comes from. /// -/// The two answers are structurally different rather than differently worded: a local scenario declares a -/// build-only service for the hub and takes that service as the named build context, so `docker compose up` -/// builds two images; a published one points the context straight at the released image and builds one. A -/// reader who does not know which shape they have is reading a `compose.yaml` with a service in it they -/// cannot account for, or missing one the other scenarios have. +/// The two answers are structurally different rather than differently worded: a local scenario declares a build-only +/// service for the hub and takes that service as the named build context, so `docker compose up` builds two images; +/// a published one points the context straight at the released image and builds one. A reader who does not know which +/// shape they have is reading a `compose.yaml` with a service in it they cannot account for, or missing one the other +/// scenarios have. const fn compose_hub_note(artifacts: ArtifactSource) -> &'static str { if matches!(artifacts, ArtifactSource::Published) { return concat!( @@ -721,13 +786,12 @@ const fn compose_hub_note(artifacts: ArtifactSource) -> &'static str { /// Render the section covering each runner image the scenario's manifests name. /// -/// Commands for a locally sourced scenario and a plain list for a published one, because that is the -/// difference the reader has to act on: in the first case a runner image the node does not have leaves its pod -/// in `ErrImagePull` and nothing in the deployment builds it, and in the second the cluster fetches it and -/// there is nothing to run at all. The published case still names the refs, since a reader who just built the -/// scenario image by hand will otherwise go looking for the step that produces these. A scenario whose agents -/// are all browser-side names no runner image, so the whole section including its heading sentence is omitted -/// rather than left as an empty code fence. +/// Commands for a locally sourced scenario and a plain list for a published one, because that is the difference the +/// reader has to act on: in the first case a runner image the node does not have leaves its pod in `ErrImagePull` and +/// nothing in the deployment builds it, and in the second the cluster fetches it and there is nothing to run at all. +/// The published case still names the refs, since a reader who just built the scenario image by hand will otherwise go +/// looking for the step that produces these. A scenario whose agents are all browser-side names no runner image, so the +/// whole section including its heading sentence is omitted rather than left as an empty code fence. fn runner_image_note(runners: &[String], images: ArtifactSource) -> String { if runners.is_empty() { return String::default(); @@ -768,9 +832,9 @@ fn runner_image_note(runners: &[String], images: ArtifactSource) -> String { /// Render the paragraph and the hub reference that differ between the two image sources. /// -/// Kept apart from the shell below rather than templating two whole sections, because everything else about -/// producing the scenario image is the same either way: only what supplies `FROM hub`, and whether that hub is -/// a build of its own, actually change. +/// Kept apart from the shell below rather than templating two whole sections, because everything else about producing +/// the scenario image is the same either way: only what supplies `FROM hub`, and whether that hub is a build of its +/// own, actually change. fn k3s_image_preamble(images: ArtifactSource) -> (&'static str, String, &'static str) { if matches!(images, ArtifactSource::Published) { return ( @@ -799,11 +863,17 @@ fn k3s_image_preamble(images: ArtifactSource) -> (&'static str, String, &'static /// Render the k3s half of the generated README. /// -/// Longer than the other two because a Kubernetes deployment needs two things done before `kubectl apply` -/// that neither `mise` nor compose does: the images have to exist on the node, since manifests reference -/// images rather than building them, and the credential has to be loaded as a `Secret`, since it is the one -/// generated file the repository does not carry. -fn k3s_run_instructions(cluster_name: &str, dockerfile: &str, runners: &[String], images: ArtifactSource) -> String { +/// Longer than the other two because a Kubernetes deployment needs two things done before `kubectl apply` that neither +/// `mise` nor compose does: the images have to exist on the node, since manifests reference images rather than building +/// them, and the credential has to be loaded as a `Secret`, since it is the one generated file the repository does +/// not carry. +fn k3s_run_instructions( + cluster_name: &str, + dockerfile: &str, + runners: &[String], + images: ArtifactSource, + build_contexts: &str, +) -> String { let (preamble, hub, hub_build) = k3s_image_preamble(images); format!( concat!( @@ -815,10 +885,34 @@ fn k3s_run_instructions(cluster_name: &str, dockerfile: &str, runners: &[String] "image=\"et-ws-server-$scenario:latest\"\n", "dockerfile=\"{dockerfile}\"\n", "{hub_build}", - "docker build --build-context \"hub=docker-image://$hub\" -t \"$image\" -f \"$dockerfile\" .\n", + "docker build --build-context \"hub=docker-image://$hub\"{build_contexts}", + " -t \"$image\" -f \"$dockerfile\" .\n", "docker save \"$image\" | sudo k3s ctr images import -\n", "```\n\n", "{runner_note}", + "{apply}", + ), + apply = k3s_apply_instructions(cluster_name), + build_contexts = build_contexts, + dockerfile = dockerfile, + hub = hub, + hub_build = hub_build, + name = cluster_name, + preamble = preamble, + runner_note = runner_image_note(runners, images) + ) +} + +/// The k3s steps after the images are on the node: load the credential, apply the manifests, and watch them settle. +/// +/// Apart from the image steps before them because nothing here depends on how the images were produced. +#[expect( + clippy::single_call_fn, + reason = "distinct half of k3s_run_instructions, split out to keep each readable" +)] +fn k3s_apply_instructions(cluster_name: &str) -> String { + format!( + concat!( "### Load The Credential\n\n", "The credential reaches the pods as a `Secret` created from `{file}`, rather than written into\n", "`k3s.yaml` where it would be committed alongside the manifests. From this directory:\n\n", @@ -837,66 +931,53 @@ fn k3s_run_instructions(cluster_name: &str, dockerfile: &str, runners: &[String] "kubectl get pods -n \"$ns\" --watch\n", "```\n" ), - dockerfile = dockerfile, file = SECRETS_ENV_FILE, - hub = hub, - hub_build = hub_build, name = cluster_name, - preamble = preamble, - runner_note = runner_image_note(runners, images) ) } +/// Directories of this repository whose every child holding a module is registered, relative to its root. +const REPO_MODULE_PARENTS: [&str; 2] = ["services/ws-modules", "data/model-modules"]; + +/// Single module directories of this repository, relative to its root, each registered as it is. +const REPO_MODULE_DIRS: [&str; 4] = [ + // Generated Python ws-modules: each generated/python-{ws,rest}/ holds its own pkg/package.json after `mise run + // build-et-{ws,rest-client}- wheel`. They're listed individually because the parent `generated/` also contains non- + // module artifacts (rust-rest, dart-ws, zig-rest, specs, docs). + "generated/python-ws", + "generated/python-rest", + // The two the hub serves whatever the scenario asks for: its own page, and the agent that page loads. Registered + // like any other module rather than prepended as bare paths by each generator, so the dependencies they declare are + // resolved too. `static` names the runtimes its page pulls at boot, and a deployment that omits them serves a page + // whose first import 404s. + "services/ws-server/static", + "services/ws-wasm-agent", +]; + #[must_use] -pub(crate) fn module_registry(project_root: &Path, ws_server_dir: &Path) -> BTreeMap { +pub(crate) fn module_registry( + project_root: &Path, + ws_server_dir: &Path, + module_paths: &[PathBuf], +) -> BTreeMap { let mut registry = BTreeMap::new(); - register_modules_under( - &mut registry, - &project_root.join("services/ws-modules"), - ws_server_dir, - "/app/services/ws-modules", - ); - register_modules_under( - &mut registry, - &project_root.join("data/model-modules"), - ws_server_dir, - "/app/data/model-modules", - ); - // Generated Python ws-modules: each generated/python-{ws,rest}/ holds - // its own pkg/package.json after `mise run build-et-{ws,rest-client}- - // wheel`. They're listed individually because the parent `generated/` - // also contains non-module artifacts (rust-rest, dart-ws, zig-rest, - // specs, docs). - register_module_at( - &mut registry, - &project_root.join("generated/python-ws"), - ws_server_dir, - "/app/generated/python-ws", - ); - register_module_at( - &mut registry, - &project_root.join("generated/python-rest"), - ws_server_dir, - "/app/generated/python-rest", - ); - - // The two the hub serves whatever the scenario asks for: its own page, and the agent that page loads. - // Registered like any other module rather than prepended as bare paths by each generator, so the - // dependencies they declare are resolved too. `static` names the runtimes its page pulls at boot, and a - // deployment that omits them serves a page whose first import 404s. - register_module_at( - &mut registry, - &project_root.join("services/ws-server/static"), - ws_server_dir, - "/app/services/ws-server/static", - ); - register_module_at( - &mut registry, - &project_root.join("services/ws-wasm-agent"), - ws_server_dir, - "/app/services/ws-wasm-agent", - ); + for parent in REPO_MODULE_PARENTS { + register_modules_under( + &mut registry, + &project_root.join(parent), + ws_server_dir, + &format!("/app/{parent}"), + ); + } + for module in REPO_MODULE_DIRS { + register_module_at( + &mut registry, + &project_root.join(module), + ws_server_dir, + &format!("/app/{module}"), + ); + } register_external_module( &mut registry, @@ -906,14 +987,91 @@ pub(crate) fn module_registry(project_root: &Path, ws_server_dir: &Path) -> BTre ); // The GPU utilisation overlay on the hub's page, declared by `static` alongside onnxruntime-web. register_external_module(&mut registry, "stats-gl", "npm:stats-gl", "/app/node_modules/stats-gl"); - // Registered as the full distribution, which `resolve_cluster_modules` narrows to the much smaller npm - // package for a cluster whose modules never call `micropip.install`. The full one comes from a GitHub - // release tarball that mise's http backend extracts flat, so its install dir is itself the module directory. + // Registered as the full distribution, which `resolve_cluster_modules` narrows to the much smaller npm package + // for a cluster whose modules never call `micropip.install`. The full one comes from a GitHub release tarball that + // mise's http backend extracts flat, so its install dir is itself the module directory. register_external_module(&mut registry, "pyodide", "http:pyodide", PYODIDE_DOCKER_PATH); + // Last, so a scenario's own module cannot be shadowed by one of this repository's of the same name -- the scenario + // named its directory deliberately, and silently serving something else would be the worse surprise. + for (index, path) in module_paths.iter().enumerate() { + register_scenario_paths(&mut registry, index, path, ws_server_dir); + } + registry } +/// Whether a path holds a module itself, rather than being a parent of module directories. +fn is_module_dir(path: &Path) -> bool { + path.join("pkg/package.json").is_file() + || path.join("package.json").is_file() + || path.join("Cargo.toml").is_file() + || path.join("pyproject.toml").is_file() +} + +/// A literal path spelled so a shell reads it back unchanged inside double quotes, as a generated command puts it. +/// +/// Applied to a `module_paths:` module's path only. Every other entry is this repository's own path or a command +/// substitution the generator writes on purpose, whereas a `module_paths:` directory is named by whoever laid out the +/// tree the scenario points at, and a `$(...)` in that name would otherwise run when the deployment starts. +pub(crate) fn escape_for_double_quotes(text: &str) -> String { + let mut escaped = String::with_capacity(text.len()); + for character in text.chars() { + if matches!(character, '\\' | '"' | '$' | '`') { + escaped.push('\\'); + } + escaped.push(character); + } + escaped +} + +/// Register the `module_paths:` entry at `index`, which is a module directory or a parent of several. +/// +/// Each module lands at `/app/module-paths/` in an image, below that at its own name when the entry is a parent. +/// The index keeps two entries ending in the same name apart, which matters beyond the image: the docker path is what +/// resolution deduplicates on. +fn register_scenario_paths( + registry: &mut BTreeMap, + index: usize, + path: &Path, + ws_server_dir: &Path, +) { + let context_root = format!("/app/module-paths/{index}"); + let module_dirs: Vec<(PathBuf, String, String)> = if is_module_dir(path) { + // The context is the module itself, so the whole of it is copied: its root, spelled as the one-character path. + vec![(path.to_path_buf(), '.'.to_string(), context_root)] + } else { + let Ok(entries) = fs::read_dir(path) else { + return; + }; + let mut dirs: Vec<(PathBuf, String, String)> = entries + .flatten() + .map(|entry| entry.path()) + .filter(|child| child.is_dir() && is_module_dir(child)) + .filter_map(|child| { + let name = child.file_name()?.to_str()?.to_string(); + let docker_path = format!("{context_root}/{name}"); + Some((child, name, docker_path)) + }) + .collect(); + dirs.sort(); + dirs + }; + for (module_path, context_path, docker_path) in module_dirs { + let Some(directory_name) = module_path.file_name().and_then(|name| name.to_str()) else { + continue; + }; + let mut entry = module_entry(&module_path, ws_server_dir, &docker_path); + entry.mise_path = escape_for_double_quotes(&entry.mise_path); + entry.source = ModuleSource::Scenario { + dir: module_path.clone(), + context: module_path_context(index), + context_path, + }; + insert_module(registry, directory_name, entry); + } +} + fn register_modules_under( registry: &mut BTreeMap, root: &Path, @@ -947,14 +1105,20 @@ fn register_modules_under( } /// Register a single module by its filesystem path (not a parent dir). -/// Used for modules that don't live under `services/ws-modules/` -- -/// currently the generated python clients under `generated/`. +/// +/// Used for modules that don't live under `services/ws-modules/` -- currently the generated python clients under +/// `generated/`. fn register_module_at( registry: &mut BTreeMap, module_path: &Path, ws_server_dir: &Path, docker_path: &str, ) { + // Absent outside this repository, where registering it anyway would record a module with no package name -- which a + // published deployment then stages as the bare tool `npm:`. + if !module_path.is_dir() { + return; + } let Some(directory_name) = module_path.file_name().and_then(|name| name.to_str()) else { return; }; @@ -971,27 +1135,48 @@ fn register_module( ws_server_dir: &Path, docker_path: &str, ) { + insert_module( + registry, + directory_name, + module_entry(module_path, ws_server_dir, docker_path), + ); +} + +/// Add a module under its directory name and, when it declares one, the package name it is served as. +fn insert_module( + registry: &mut BTreeMap, + directory_name: &str, + entry: ModuleRegistryEntry, +) { + if let Some(served_name) = entry.package_name.clone() { + let _previous: Option = registry.insert(served_name, entry.clone()); + } + let _previous: Option = registry.insert(directory_name.to_string(), entry); +} + +/// The registry entry for a module directory of this repository. +fn module_entry(module_path: &Path, ws_server_dir: &Path, docker_path: &str) -> ModuleRegistryEntry { let package = module_package_json(module_path); - // The docker path is always the repo-relative path under `/app`, which is where the hub image roots its - // module scan, so stripping that prefix recovers the path to copy out of the build context. + // The docker path is always the repo-relative path under `/app`, which is where the hub image roots its module + // scan, so stripping that prefix recovers the path to copy out of the build context. let repo_path = docker_path.strip_prefix("/app/").unwrap_or(docker_path).to_string(); - // The name a module is served, resolved and referred to by is the one its published `package.json` - // declares, scope and all -- and it has to be that name whether or not `pkg/` has been built, because - // the lanes that generate a deployment build no modules. A generated manifest already carries the scope; - // the source manifest standing in for it when `pkg/` is absent (`Cargo.toml`, `pyproject.toml`) names the - // crate unscoped. So the rule that scopes a dependency scopes the module's own name too, which leaves an - // already-scoped one untouched and keeps both sides of a dependency edge spelling the same key. + // The name a module is served, resolved and referred to by is the one its published `package.json` declares, scope + // and all -- and it has to be that name whether or not `pkg/` has been built, because the lanes that generate a + // deployment build no modules. A generated manifest already carries the scope; the source manifest standing in for + // it when `pkg/` is absent (`Cargo.toml`, `pyproject.toml`) names the crate unscoped. So the rule that scopes a + // dependency scopes the module's own name too, which leaves an already-scoped one untouched and keeps both sides of + // a dependency edge spelling the same key. let served_name = package .as_ref() .and_then(|package| package.name.clone()) .map(|name| module_package_json::scoped_dependency_name(&name)); - let entry = ModuleRegistryEntry { + ModuleRegistryEntry { mise_path: relative_path_from(ws_server_dir, module_path), docker_path: docker_path.to_string(), - // Through the same scoping the generator applies when it writes a `package.json`, because the two - // have to name the same module. A source manifest declares a dependency the way it declares its own - // crate -- unscoped -- and publishing scopes both; reading one side raw would leave a dependency - // naming something the registry has no key for. + // Through the same scoping the generator applies when it writes a `package.json`, because the two have to name + // the same module. A source manifest declares a dependency the way it declares its own crate -- unscoped -- and + // publishing scopes both; reading one side raw would leave a dependency naming something the registry has no + // key for. dependencies: package .as_ref() .map(|package| { @@ -1003,22 +1188,17 @@ fn register_module( }) .unwrap_or_default(), source: ModuleSource::Repo(repo_path), - package_name: served_name.clone(), - }; - - let _previous: Option = registry.insert(directory_name.to_string(), entry.clone()); - if let Some(served_name) = served_name { - let _previous: Option = registry.insert(served_name, entry); + package_name: served_name, } } /// Register a package that mise stages outside the repository, keyed by its published package name. /// -/// The mise path is a shell substitution rather than a literal, because where a tool's install dir keeps the -/// package is not knowable when the deployment is generated. An archive-backed `http:` tool extracts flat, so -/// `mise where` is already the answer; the npm backend spreads packages across several layouts that differ by -/// platform, so that case defers to `et-cli npm-module-path`, which resolves it through the same code the -/// ws-server uses to find these packages itself. +/// The mise path is a shell substitution rather than a literal, because where a tool's install dir keeps the package +/// is not knowable when the deployment is generated. An archive-backed `http:` tool extracts flat, so `mise where` +/// is already the answer; the npm backend spreads packages across several layouts that differ by platform, so that +/// case defers to `et-cli npm-module-path`, which resolves it through the same code the ws-server uses to find these +/// packages itself. fn register_external_module( registry: &mut BTreeMap, package_name: &str, @@ -1031,14 +1211,14 @@ fn register_external_module( /// Build the registry entry for a mise-staged package. /// -/// Separate from registration so the pyodide swap can rebuild an entry for a different tool without restating -/// how a mise path is spelled. +/// Separate from registration so the pyodide swap can rebuild an entry for a different tool without restating how a +/// mise path is spelled. fn external_module_entry(package_name: &str, tool: &str, docker_path: &str) -> ModuleRegistryEntry { - // Resolved at run time, because where mise's npm backend puts a package varies by backend and platform. - // A local deployment names the packages it wants rather than asking the hub to serve everything this - // config staged: the repository's own tools table mixes modules with development tooling, so serving all - // of it would serve things that are not modules at all. An archive-backed tool - // extracts flat, making its install directory the module directory, which `mise where` answers outright. + // Resolved at run time, because where mise's npm backend puts a package varies by backend and platform. A local + // deployment names the packages it wants rather than asking the hub to serve everything this config staged: the + // repository's own tools table mixes modules with development tooling, so serving all of it would serve things that + // are not modules at all. An archive-backed tool extracts flat, making its install directory the module directory, + // which `mise where` answers outright. let mise_path = if tool.starts_with("npm:") { format!("$(cargo run --quiet -p et-cli -- npm-module-path --package {package_name})") } else { @@ -1110,9 +1290,9 @@ fn read_cargo_package(path: &Path) -> Option { /// Walk `module_names` and everything they depend on, in breadth-first declaration order. /// -/// A module is registered under both its directory name and its `package.json` name, so the same entry is -/// reachable by two keys; de-duplicating on the docker path collapses those without disturbing the order the -/// generated files depend on. +/// A module is registered under both its directory name and its `package.json` name, so the same entry is reachable by +/// two keys; de-duplicating on the docker path collapses those without disturbing the order the generated files depend +/// on. fn resolve_module_entries<'registry>( registry: &'registry BTreeMap, module_names: &[String], @@ -1147,10 +1327,15 @@ pub(crate) fn resolve_module_paths( where F: Fn(&ModuleRegistryEntry) -> String, { - Ok(resolve_cluster_modules(registry, module_names)? + // Sorted rather than left in resolution order, which is breadth-first from whichever modules the scenario names, + // so a dependency lands wherever it was first reached and moves whenever an unrelated module gains one. The hub + // serves modules by name, so the order carries nothing, and sorted it stays put in a committed deployment. + let mut paths: Vec = resolve_cluster_modules(registry, module_names)? .iter() .map(path_for) - .collect()) + .collect(); + paths.sort(); + Ok(paths) } /// Resolve the cluster's modules to the docker path each is served from and how it is provisioned. @@ -1166,9 +1351,9 @@ pub(crate) fn resolve_module_sources( /// Resolve a cluster's modules, sized to what those modules actually need. /// -/// Everything is taken from the registry as-is except pyodide, whose distribution depends on the cluster: the -/// registry cannot decide that, because pyodide arrives as a dependency of whichever Python modules the cluster -/// happens to declare. +/// Everything is taken from the registry as-is except pyodide, whose distribution depends on the cluster: the registry +/// cannot decide that, because pyodide arrives as a dependency of whichever Python modules the cluster happens to +/// declare. pub(crate) fn resolve_cluster_modules( registry: &BTreeMap, module_names: &[String], @@ -1193,13 +1378,18 @@ pub(crate) fn resolve_cluster_modules( /// Whether a module pulls a non-stdlib wheel at runtime. /// -/// Decided from the module's served `pkg/`, which is the code the browser actually runs, rather than from its -/// Python sources: the `micropip.install` calls live in each Python module's JS loader shim. +/// Decided from the module's served `pkg/`, which is the code the browser actually runs, rather than from its Python +/// sources: the `micropip.install` calls live in each Python module's JS loader shim. fn module_installs_wheels(project_root: &Path, entry: &ModuleRegistryEntry) -> bool { - let ModuleSource::Repo(repo_path) = &entry.source else { - return false; + // A `module_paths:` module may be served from its own root rather than a `pkg/`, so its loader is looked for in + // whichever of the two the hub would serve. + let served_dir = match &entry.source { + ModuleSource::Repo(repo_path) => project_root.join(repo_path).join("pkg"), + ModuleSource::Scenario { dir, .. } if dir.join("pkg").is_dir() => dir.join("pkg"), + ModuleSource::Scenario { dir, .. } => dir.clone(), + ModuleSource::MiseTool { .. } => return false, }; - let Ok(files) = fs::read_dir(project_root.join(repo_path).join("pkg")) else { + let Ok(files) = fs::read_dir(served_dir) else { return false; }; @@ -1214,9 +1404,9 @@ fn module_installs_wheels(project_root: &Path, entry: &ModuleRegistryEntry) -> b /// Resolve the directory holding a mise-staged npm package. /// -/// Defers to the resolver the ws-server itself uses, which is the only place that knows the layouts mise's npm -/// backend produces. Generated deployments call back into this rather than embedding a path, because the layout -/// differs per platform and backend and so cannot be decided when the deployment is generated. +/// Defers to the resolver the ws-server itself uses, which is the only place that knows the layouts mise's npm backend +/// produces. Generated deployments call back into this rather than embedding a path, because the layout differs per +/// platform and backend and so cannot be decided when the deployment is generated. pub fn npm_module_path(package: &str) -> Result { edge_toolkit::config::mise_npm_package_path(package) .ok_or_else(|| CliError::UnresolvedNpmModule(package.to_string())) @@ -1243,9 +1433,8 @@ pub(crate) struct RunnerInstance { /// /// All three share one deployment shape, which is what lets one generator serve them: each takes the module's /// published name in `RUNNER_MODULE`, fetches it from the hub named by `WS_SERVER_URL`, and builds from -/// `services/ws--runner/Dockerfile`. Nothing else distinguishes a runner here, so a fourth is this line -/// plus an image. Rejecting a kind by name is what stops a scenario from asking for one and silently getting -/// nothing. +/// `services/ws--runner/Dockerfile`. Nothing else distinguishes a runner here, so a fourth is this line plus an +/// image. Rejecting a kind by name is what stops a scenario from asking for one and silently getting nothing. pub(crate) const SUPPORTED_RUNNERS: [(&str, &str); 3] = [ ("pyo3", "et-ws-pyo3-runner"), ("wasi", "et-ws-wasi-runner"), @@ -1254,10 +1443,10 @@ pub(crate) const SUPPORTED_RUNNERS: [(&str, &str); 3] = [ /// Names the generated deployment already uses for its own tasks, services and aliases. /// -/// A runner is named after the agent that declares it, and both generators key on that name: mise inserts each -/// task into a table and compose writes each service as a mapping key. Either way a collision replaces rather -/// than reports -- an agent called `ws-server` would quietly take the hub's place, and the deployment would come -/// up missing the thing it was meant to talk to. Rejecting the name is the only way that surfaces. +/// A runner is named after the agent that declares it, and both generators key on that name: mise inserts each task +/// into a table and compose writes each service as a mapping key. Either way a collision replaces rather than reports +/// -- an agent called `ws-server` would quietly take the hub's place, and the deployment would come up missing the +/// thing it was meant to talk to. Rejecting the name is the only way that surfaces. pub(crate) const RESERVED_RUNNER_NAMES: [&str; 6] = [ "generated-scenario", "o2", @@ -1269,9 +1458,9 @@ pub(crate) const RESERVED_RUNNER_NAMES: [&str; 6] = [ /// Resolve every agent that names a `runner:` into the processes the deployment has to start. /// -/// One process per resource rather than per agent, because a runner hosts exactly one module -- `RUNNER_MODULE` -/// is a single name. An agent with one resource (the usual shape) therefore keeps the agent's own name, and only -/// a multi-resource agent gets the resource suffixed, so the common case reads as the scenario wrote it. +/// One process per resource rather than per agent, because a runner hosts exactly one module -- `RUNNER_MODULE` is a +/// single name. An agent with one resource (the usual shape) therefore keeps the agent's own name, and only a multi- +/// resource agent gets the resource suffixed, so the common case reads as the scenario wrote it. pub(crate) fn resolve_cluster_runners( registry: &BTreeMap, cluster: &ClusterInput, @@ -1293,11 +1482,11 @@ pub(crate) fn resolve_cluster_runners( supported, }); } - // A scenario cannot set the two the deployment derives for it. - // `RUNNER_MODULE` and `WS_SERVER_URL` are what wire a runner to its module and its hub, and both are - // computed from the rest of the input. Letting `env:` win would mean a scenario whose generated files - // describe one deployment and whose runners join another; letting the derived value win would mean an - // `env:` entry that is silently ignored. Neither is worth allowing, so it is an error to write one. + // A scenario cannot set the two the deployment derives for it. `RUNNER_MODULE` and `WS_SERVER_URL` are what + // wire a runner to its module and its hub, and both are computed from the rest of the input. Letting `env:` win + // would mean a scenario whose generated files describe one deployment and whose runners join another; letting + // the derived value win would mean an `env:` entry that is silently ignored. Neither is worth allowing, so it + // is an error to write one. for variable in DERIVED_RUNNER_ENV { if agent.env.contains_key(variable) { return Err(CliError::ReservedRunnerEnv { @@ -1322,8 +1511,8 @@ pub(crate) fn resolve_cluster_runners( name, runner: runner.to_string(), module, - // Every resource of a multi-resource agent gets its own runner process, and the agent's - // environment describes the agent, so each of them carries it. + // Every resource of a multi-resource agent gets its own runner process, and the agent's environment + // describes the agent, so each of them carries it. env: agent.env.clone(), }); } @@ -1333,9 +1522,9 @@ pub(crate) fn resolve_cluster_runners( /// Base HTTP URL a generated deployment reaches the hub on. /// -/// Every generator addresses the hub by its standard insecure port, so spelling the URL out in each of them -/// meant writing the same format string more than once. One definition here serves the compose services, the -/// mise tasks and whatever is added next, and it is the only place that has to change if the port moves. +/// Every generator addresses the hub by its standard insecure port, so spelling the URL out in each of them meant +/// writing the same format string more than once. One definition here serves the compose services, the mise tasks and +/// whatever is added next, and it is the only place that has to change if the port moves. #[must_use] pub(crate) fn hub_http_base() -> String { format!("http://localhost:{}", Services::InsecureWebSocketServer.port()) @@ -1343,12 +1532,12 @@ pub(crate) fn hub_http_base() -> String { /// Derive one runner's deployment-unique name, rejecting the two ways it can collide. /// -/// An agent with a single resource keeps its own name, so the common case reads as the scenario wrote it; only a -/// multi-resource agent gets the resource suffixed, because each resource becomes its own process. +/// An agent with a single resource keeps its own name, so the common case reads as the scenario wrote it; only a multi- +/// resource agent gets the resource suffixed, because each resource becomes its own process. /// /// Both checks exist because a collision would otherwise be silent rather than wrong-looking: mise inserts each -/// task into a table and compose writes each service as a mapping key, so a repeated name replaces what was there. -/// A scenario could lose its hub and only find out when the runners had nothing to talk to. +/// task into a table and compose writes each service as a mapping key, so a repeated name replaces what was there. A +/// scenario could lose its hub and only find out when the runners had nothing to talk to. #[expect( clippy::single_call_fn, reason = "distinct step of resolve_cluster_runners; separate to keep that function within its complexity budget" diff --git a/utilities/cli/tests/scenario_generation.rs b/utilities/cli/tests/scenario_generation.rs index 9914246c..ead6dc10 100644 --- a/utilities/cli/tests/scenario_generation.rs +++ b/utilities/cli/tests/scenario_generation.rs @@ -121,9 +121,10 @@ agents: #[test] fn docker_image_module_paths_include_static_root_module() { - let paths = docker_image_module_paths(&["face-detection".to_string()], true).unwrap(); + let paths = docker_image_module_paths(&["face-detection".to_string()], true, &[]).unwrap(); - assert_eq!(paths[0], "/app/services/ws-server/static"); + assert!(paths.is_sorted(), "{paths:?}"); + assert!(paths.contains(&"/app/services/ws-server/static".to_string())); assert!(paths.contains(&"/app/services/ws-wasm-agent".to_string())); assert!(paths.contains(&"/app/data/model-modules/model-face1".to_string())); assert!(paths.contains(&"/app/node_modules/onnxruntime-web".to_string())); @@ -133,16 +134,17 @@ fn docker_image_module_paths_include_static_root_module() { #[test] fn a_headless_image_is_given_neither_the_page_nor_what_the_page_imports() { - // face-detection's own model still comes, because the module declares it. What goes is the page and the two - // packages only its `package.json` names -- a cluster nobody opens loads none of them. - let paths = docker_image_module_paths(&["face-detection".to_string()], false).unwrap(); + // face-detection's own model still comes, because the module declares it. What goes is the page and everything + // only its `package.json` names -- the agent included, which face-detection links in rather than loads -- since a + // cluster nobody opens loads none of them. + let paths = docker_image_module_paths(&["face-detection".to_string()], false, &[]).unwrap(); assert!( !paths.contains(&"/app/services/ws-server/static".to_string()), "{paths:?}" ); assert!(!paths.contains(&"/app/node_modules/stats-gl".to_string()), "{paths:?}"); - assert!(paths.contains(&"/app/services/ws-wasm-agent".to_string())); + assert!(!paths.contains(&"/app/services/ws-wasm-agent".to_string()), "{paths:?}"); assert!(paths.contains(&"/app/data/model-modules/model-face1".to_string())); assert!(paths.contains(&"/app/services/ws-modules/face-detection".to_string())); } @@ -154,20 +156,20 @@ fn scenario_module_paths_include_selected_modules_and_dependencies() { let modules = ["face-detection".to_string(), "har1".to_string()]; let paths = scenario_module_paths(&ScenarioModules::new(&ws_server_dir, &modules, true)).unwrap(); - // onnxruntime-web and stats-gl are here because the hub's own page declares them, not because either scenario - // module does -- they arrive in the first resolution wave, ahead of the model modules that face-detection and har1 - // pull in. A deployment that omitted them served a page whose first import 404d. + // The agent, onnxruntime-web and stats-gl are here because the hub's own page declares them, not because either + // scenario module does. A deployment that omitted them served a page whose first import 404d. The list is sorted, + // so where each was first reached in resolution does not show. assert_eq!( paths, vec![ - "static".to_string(), - "../ws-wasm-agent".to_string(), - "../ws-modules/face-detection".to_string(), - "../ws-modules/har1".to_string(), "$(cargo run --quiet -p et-cli -- npm-module-path --package onnxruntime-web)".to_string(), "$(cargo run --quiet -p et-cli -- npm-module-path --package stats-gl)".to_string(), "../../data/model-modules/model-face1".to_string(), "../../data/model-modules/model-har-motion1".to_string(), + "../ws-modules/face-detection".to_string(), + "../ws-modules/har1".to_string(), + "../ws-wasm-agent".to_string(), + "static".to_string(), ], ); assert!(!paths.contains(&"../ws-modules".to_string())); @@ -465,29 +467,59 @@ fn the_scenario_dockerfile_path_stays_relative_however_shallow_the_output_dir() ); } -#[test] -fn the_wrapped_module_list_folds_back_into_one_comma_separated_value() { - // The list is wrapped to stay inside the line limit, and it is wrapped by YAML folding rather than by a trailing - // `\`, which the repository bans. Folding is only correct if the breaks come back as separators the server accepts, - // so this parses the generated file rather than trusting the spelling. - let (_test_root, output_dir) = k3s_scenario_with(""); +/// The hub's `MODULES_PATHS` in the `compose.yaml` under `output_dir`, as YAML reads it back. +/// +/// Parsed rather than matched as text, because how the value is spelled -- folded, quoted, or on one line -- is what +/// these tests are checking, and only a parse says what it spells. A line continuation is refused on the way. +fn compose_modules_paths(output_dir: &std::path::Path) -> String { let text = fs::read_to_string(output_dir.join("compose.yaml")).unwrap(); - assert!(!text.contains('\\'), "no line continuations survive: {text}"); - let compose: serde_yaml::Value = serde_yaml::from_str(&text).unwrap(); - let paths = compose["services"]["ws-server"]["environment"]["MODULES_PATHS"] + compose["services"]["ws-server"]["environment"]["MODULES_PATHS"] .as_str() - .unwrap(); + .unwrap() + .to_string() +} + +#[test] +fn the_wrapped_module_list_folds_back_into_one_comma_separated_value() { + // The list is wrapped to stay inside the line limit, and it is wrapped by YAML folding rather than by a trailing + // `\`, which the repository bans. Folding is only correct if the breaks come back as separators the server accepts, + // so this parses the generated file rather than trusting the spelling. Two modules, so there is a break to fold. + let (_test_root, verification_root, output_dir) = scenario_tree( + r#"cluster_name: "folded" +agents: + - name: "math1-twin" + runner: "wasi" + resources: + - type: "wasi-math1" + - name: "math1-trigger" + runner: "wasi" + resources: + - type: "wasi-math1-sender" +"#, + ); + let _regenerated = regenerate_verification(&verification_root, None).unwrap(); + let paths = compose_modules_paths(&output_dir); assert!(!paths.contains('\n'), "folded to a single line: {paths}"); let segments: Vec<&str> = paths.split(',').map(str::trim).collect(); - // The agent leads, not the page: this fixture's one agent names a runner, so the cluster is headless and was never - // given a front page. - assert_eq!(segments.first().copied(), Some("/app/services/ws-wasm-agent")); - assert!( - segments.iter().all(|segment| segment.starts_with("/app/")), - "every segment is a path once trimmed: {segments:?}" + assert_eq!( + segments, + [ + "/app/services/ws-modules/wasi-math1", + "/app/services/ws-modules/wasi-math1-sender" + ] + ); +} + +#[test] +fn a_single_module_path_is_still_a_closed_quoted_value() { + // One path has no break to fold, and was once written as the opening line of a fold with nothing to close it. + let (_test_root, output_dir) = k3s_scenario_with(""); + assert_eq!( + compose_modules_paths(&output_dir), + "/app/services/ws-modules/wasi-math1" ); } @@ -547,10 +579,10 @@ fn the_artifact_source_decides_whether_the_mise_deployment_builds_what_it_runs() "a published deployment runs the released runner: {published}" ); assert!(!published.contains("cargo run"), "and builds nothing: {published}"); - // Each released binary is declared as a tool, which is what puts it on `PATH` for the task above, and - // each waives the release age. Without the waiver mise hides a release younger than a day and installs - // the one before it -- so a deployment generated beside the publish it was made for runs the previous - // binary, and says nothing: resolving `latest` to an older release is ordinary behaviour, not an error. + // Each released binary is declared as a tool, which is what puts it on `PATH` for the task above, and each waives + // the release age. Without the waiver mise hides a release younger than a day and installs the one before it -- so + // a deployment generated beside the publish it was made for runs the previous binary, and says nothing: resolving + // `latest` to an older release is ordinary behaviour, not an error. for crate_name in ["et-ws-server", "et-ws-wasi-runner"] { let declared = format!("[tools.\"cargo:{crate_name}\"]\nminimum_release_age = \"0\"\nversion = \"latest\""); assert!(published.contains(&declared), "expected {declared} in: {published}"); @@ -665,3 +697,134 @@ agents: [] assert!(ci_output_dir.join("mise.toml").exists()); assert!(ci_output_dir.join("compose.yaml").exists()); } + +/// Write a module directory at `dir` declaring `name`, depending on pyodide, with a loader whose source is `loader`. +fn scenario_module(dir: &std::path::Path, name: &str, loader: &str) { + fs::create_dir_all(dir.join("pkg")).unwrap(); + let package = format!(r#"{{"name": "{name}", "main": "loader.js", "dependencies": {{"pyodide": "*"}}}}"#); + fs::write(dir.join("pkg/package.json"), package).unwrap(); + fs::write(dir.join("pkg/loader.js"), loader).unwrap(); +} + +#[test] +fn module_paths_modules_keep_their_own_identity_and_cannot_inject_shell() { + let test_root = tempdir().unwrap(); + // Two directories of one name under different parents, which resolution used to fold into one module, and a + // directory whose name is shell syntax the generated task would otherwise run. + scenario_module(&test_root.path().join("a/module"), "@ext/a", ""); + scenario_module(&test_root.path().join("b/module"), "@ext/b", ""); + scenario_module(&test_root.path().join("c/mod$(id)"), "@ext/c", ""); + let module_paths = [ + test_root.path().join("a/module"), + test_root.path().join("b/module"), + test_root.path().join("c"), + ]; + let ws_server_dir = edge_toolkit::config::get_project_root().join("services/ws-server"); + let modules = ["@ext/a".to_string(), "@ext/b".to_string(), "@ext/c".to_string()]; + let scenario = ScenarioModules::new(&ws_server_dir, &modules, false).with_module_paths(&module_paths); + let paths = scenario_module_paths(&scenario).unwrap(); + + assert!(paths.iter().any(|path| path.ends_with("a/module")), "{paths:?}"); + assert!(paths.iter().any(|path| path.ends_with("b/module")), "{paths:?}"); + assert!(paths.iter().any(|path| path.ends_with(r"c/mod\$(id)")), "{paths:?}"); + assert!(!paths.iter().any(|path| path.ends_with("c/mod$(id)")), "{paths:?}"); +} + +#[test] +fn a_published_hub_serving_module_paths_modules_keeps_the_full_pyodide() { + let test_root = tempdir().unwrap(); + // The loader shim is what marks a module as needing wheels, which only the full distribution can install. + scenario_module( + &test_root.path().join("modules/wheels"), + "@ext/wheels", + "await pyodide.loadPackage('micropip'); await micropip.install('x');", + ); + let input_file = test_root.path().join("cluster.yaml"); + let input = concat!( + "cluster_name: \"external\"\n", + "artifact_source: \"published\"\n", + "module_paths:\n", + " - \"modules\"\n", + "agents:\n", + " - name: \"browser\"\n", + " resources:\n", + " - type: \"wheels\"\n", + ); + fs::write(&input_file, input).unwrap(); + let output_dir = test_root.path().join("deployment"); + let _summary = generate_deployment(&input_file, &output_dir, None).unwrap(); + let mise_toml = fs::read_to_string(output_dir.join("mise.toml")).unwrap(); + + // Setting `MODULES_PATHS` replaces the hub's defaults, which is where the full distribution came from. + assert!(mise_toml.contains("../modules/wheels"), "{mise_toml}"); + assert!(mise_toml.contains("$(mise where http:pyodide)"), "{mise_toml}"); +} + +/// Generate a `deployment_type` deployment of a scenario naming one module from its own `module_paths:`. +/// +/// The scenario's one entry is the directory `entry`, holding the module `@ext/mine` in its child `module`. Returns the +/// temp root with the output directory inside it, since dropping the root deletes the tree. +fn module_paths_deployment( + deployment_type: &str, + entry: &str, + module: &str, +) -> (tempfile::TempDir, std::path::PathBuf) { + let test_root = tempdir().unwrap(); + scenario_module(&test_root.path().join(entry).join(module), "@ext/mine", ""); + let input = format!( + concat!( + "cluster_name: \"own-modules\"\n", + "deployment_type: \"{deployment_type}\"\n", + "module_paths:\n", + " - \"{entry}\"\n", + "agents:\n", + " - name: \"browser\"\n", + " resources:\n", + " - type: \"@ext/mine\"\n", + ), + deployment_type = deployment_type, + entry = entry, + ); + let input_file = test_root.path().join("cluster.yaml"); + fs::write(&input_file, input).unwrap(); + let output_dir = test_root.path().join("deployment"); + let _summary = generate_deployment(&input_file, &output_dir, None).unwrap(); + (test_root, output_dir) +} + +#[test] +fn a_compose_deployment_builds_module_paths_modules_into_its_image() { + let (_test_root, output_dir) = module_paths_deployment("docker-compose", "modules", "mine"); + let compose = fs::read_to_string(output_dir.join("compose.yaml")).unwrap(); + let dockerfile = fs::read_to_string(output_dir.join("Dockerfile")).unwrap(); + + // The entry is handed to the build as its own context, relative to the compose file, and the image copies the + // module out of it to the path the hub is told to serve. + assert!(compose.contains("module-path-0: ../modules"), "{compose}"); + assert!(compose.contains("/app/module-paths/0/mine"), "{compose}"); + let copy = r#"COPY --from=module-path-0 --chown=10001:10001 ["mine", "/app/module-paths/0/mine"]"#; + assert!(dockerfile.contains(copy), "{dockerfile}"); +} + +#[test] +fn a_module_paths_directory_name_stays_one_operand_and_one_shell_word() { + // A space splits a plain-form `COPY` operand, and a `$` is substitution to both Docker and the shell: the entry + // carries the `$`, which the README's build context names, and the module under it the space. + let (_test_root, output_dir) = module_paths_deployment("k3s", "mods$HOME", "my mod"); + let dockerfile = fs::read_to_string(output_dir.join("Dockerfile")).unwrap(); + let readme = fs::read_to_string(output_dir.join("README.md")).unwrap(); + + let copy = r#"COPY --from=module-path-0 --chown=10001:10001 ["my mod", "/app/module-paths/0/my mod"]"#; + assert!(dockerfile.contains(copy), "{dockerfile}"); + assert!(readme.contains(r#"/mods\$HOME""#), "{readme}"); +} + +#[test] +fn a_k3s_deployment_names_the_module_paths_context_its_image_build_needs() { + let (_test_root, output_dir) = module_paths_deployment("k3s", "modules", "mine"); + let manifest = fs::read_to_string(output_dir.join("k3s.yaml")).unwrap(); + let readme = fs::read_to_string(output_dir.join("README.md")).unwrap(); + + assert!(manifest.contains("/app/module-paths/0/mine"), "{manifest}"); + assert!(readme.contains(" --build-context \"module-path-0="), "{readme}"); +} diff --git a/verification/local/output/default/.gitignore b/verification/local/output/default/.gitignore new file mode 100644 index 00000000..4b22f893 --- /dev/null +++ b/verification/local/output/default/.gitignore @@ -0,0 +1,7 @@ +# Written by `et-cli`. +# What the hub writes into the directory it runs from: its self-signed TLS pair (the key is a +# credential too), the agent registry it saves on shutdown, and agent storage. +cert.pem +key.pem +registry.yaml +storage/ diff --git a/verification/local/output/default/compose.yaml b/verification/local/output/default/compose.yaml index 914714b7..f79a7b81 100644 --- a/verification/local/output/default/compose.yaml +++ b/verification/local/output/default/compose.yaml @@ -46,7 +46,9 @@ services: hub: service:ws-server-hub network_mode: host environment: - MODULES_PATHS: "/app/services/ws-server/static, + MODULES_PATHS: "/app/node_modules/onnxruntime-web, + /app/node_modules/stats-gl, + /app/services/ws-server/static, /app/services/ws-wasm-agent" MODULES_ROOT: "@edge-toolkit/et-ws-server-static" OTLP_AUTH_USERNAME: root@example.com diff --git a/verification/local/output/default/k3s.yaml b/verification/local/output/default/k3s.yaml index 8e85b42c..8919af49 100644 --- a/verification/local/output/default/k3s.yaml +++ b/verification/local/output/default/k3s.yaml @@ -146,7 +146,7 @@ spec: - et-ws-server env: - name: MODULES_PATHS - value: /app/services/ws-server/static,/app/services/ws-wasm-agent + value: /app/node_modules/onnxruntime-web,/app/node_modules/stats-gl,/app/services/ws-server/static,/app/services/ws-wasm-agent - name: MODULES_ROOT value: "@edge-toolkit/et-ws-server-static" - name: OTLP_AUTH_USERNAME diff --git a/verification/local/output/default/mise.toml b/verification/local/output/default/mise.toml index 9fb88bec..95eb2fdd 100644 --- a/verification/local/output/default/mise.toml +++ b/verification/local/output/default/mise.toml @@ -25,8 +25,9 @@ docker run --rm --name openobserve -p 127.0.0.1:5080:5080 $settings -e ZO_ROOT_U description = "Run the WebSocket server" dir = "../../../../services/ws-server" run = """ -MODULES_PATHS="static, ../ws-wasm-agent, $(cargo run --quiet -p et-cli -- npm-module-path --package onnxruntime-web)" -MODULES_PATHS="$MODULES_PATHS, $(cargo run --quiet -p et-cli -- npm-module-path --package stats-gl)" +MODULES_PATHS="$(cargo run --quiet -p et-cli -- npm-module-path --package onnxruntime-web)" +MODULES_PATHS="$MODULES_PATHS, $(cargo run --quiet -p et-cli -- npm-module-path --package stats-gl), ../ws-wasm-agent" +MODULES_PATHS="$MODULES_PATHS, static" export MODULES_PATHS cargo run """ diff --git a/verification/local/output/facility-security-scenario/.gitignore b/verification/local/output/facility-security-scenario/.gitignore new file mode 100644 index 00000000..4b22f893 --- /dev/null +++ b/verification/local/output/facility-security-scenario/.gitignore @@ -0,0 +1,7 @@ +# Written by `et-cli`. +# What the hub writes into the directory it runs from: its self-signed TLS pair (the key is a +# credential too), the agent registry it saves on shutdown, and agent storage. +cert.pem +key.pem +registry.yaml +storage/ diff --git a/verification/local/output/facility-security-scenario/Dockerfile b/verification/local/output/facility-security-scenario/Dockerfile index a80d4661..07fa4533 100644 --- a/verification/local/output/facility-security-scenario/Dockerfile +++ b/verification/local/output/facility-security-scenario/Dockerfile @@ -76,7 +76,7 @@ set -euo pipefail mise install node EOF -ENV SCENARIO_TOOLS="npm:onnxruntime-web http:pyodide" +ENV SCENARIO_TOOLS="http:pyodide npm:onnxruntime-web npm:stats-gl" RUN bash <<'EOF' set -euo pipefail # Word-splitting SCENARIO_TOOLS is intentional; it is a space-separated tool list. @@ -84,6 +84,24 @@ set -euo pipefail mise install ${SCENARIO_TOOLS} EOF +ENV STAGE_TOOL="http:pyodide" +ENV STAGE_PACKAGE="pyodide" +ENV STAGE_DEST="/staged/app/node_modules/pyodide" +RUN bash <<'EOF' +set -euo pipefail +root="$(mise where "${STAGE_TOOL}")" +src="$(find -L "${root}" -type d -name "${STAGE_PACKAGE}" -exec test -f '{}/package.json' ';' -print -quit)" +if [ -z "${src}" ] && [ -f "${root}/package.json" ]; then + src="${root}" +fi +if [ -z "${src}" ]; then + echo "no ${STAGE_PACKAGE} package dir under the ${STAGE_TOOL} install at ${root}" >&2 + exit 1 +fi +mkdir -p "$(dirname "${STAGE_DEST}")" +cp -rL "${src}" "${STAGE_DEST}" +EOF + ENV STAGE_TOOL="npm:onnxruntime-web" ENV STAGE_PACKAGE="onnxruntime-web" ENV STAGE_DEST="/staged/app/node_modules/onnxruntime-web" @@ -102,9 +120,9 @@ mkdir -p "$(dirname "${STAGE_DEST}")" cp -rL "${src}" "${STAGE_DEST}" EOF -ENV STAGE_TOOL="http:pyodide" -ENV STAGE_PACKAGE="pyodide" -ENV STAGE_DEST="/staged/app/node_modules/pyodide" +ENV STAGE_TOOL="npm:stats-gl" +ENV STAGE_PACKAGE="stats-gl" +ENV STAGE_DEST="/staged/app/node_modules/stats-gl" RUN bash <<'EOF' set -euo pipefail root="$(mise where "${STAGE_TOOL}")" @@ -130,11 +148,12 @@ FROM hub LABEL org.opencontainers.image.source="https://github.com/edge-toolkit/core" -COPY --chown=10001:10001 services/ws-modules/face-detection/pkg /app/services/ws-modules/face-detection/pkg -COPY --chown=10001:10001 services/ws-modules/har1/pkg /app/services/ws-modules/har1/pkg -COPY --chown=10001:10001 services/ws-modules/pyface1/pkg /app/services/ws-modules/pyface1/pkg COPY --chown=10001:10001 data/model-modules/model-face1/pkg /app/data/model-modules/model-face1/pkg COPY --chown=10001:10001 data/model-modules/model-har-motion1/pkg /app/data/model-modules/model-har-motion1/pkg COPY --chown=10001:10001 generated/python-ws/pkg /app/generated/python-ws/pkg -COPY --from=deps --chown=10001:10001 /staged/app/node_modules/onnxruntime-web /app/node_modules/onnxruntime-web +COPY --chown=10001:10001 services/ws-modules/face-detection/pkg /app/services/ws-modules/face-detection/pkg +COPY --chown=10001:10001 services/ws-modules/har1/pkg /app/services/ws-modules/har1/pkg +COPY --chown=10001:10001 services/ws-modules/pyface1/pkg /app/services/ws-modules/pyface1/pkg COPY --from=deps --chown=10001:10001 /staged/app/node_modules/pyodide /app/node_modules/pyodide +COPY --from=deps --chown=10001:10001 /staged/app/node_modules/onnxruntime-web /app/node_modules/onnxruntime-web +COPY --from=deps --chown=10001:10001 /staged/app/node_modules/stats-gl /app/node_modules/stats-gl diff --git a/verification/local/output/facility-security-scenario/compose.yaml b/verification/local/output/facility-security-scenario/compose.yaml index 40a532fc..585645f8 100644 --- a/verification/local/output/facility-security-scenario/compose.yaml +++ b/verification/local/output/facility-security-scenario/compose.yaml @@ -46,16 +46,17 @@ services: hub: service:ws-server-hub network_mode: host environment: - MODULES_PATHS: "/app/services/ws-server/static, - /app/services/ws-wasm-agent, + MODULES_PATHS: "/app/data/model-modules/model-face1, + /app/data/model-modules/model-har-motion1, + /app/generated/python-ws, + /app/node_modules/onnxruntime-web, + /app/node_modules/pyodide, + /app/node_modules/stats-gl, /app/services/ws-modules/face-detection, /app/services/ws-modules/har1, /app/services/ws-modules/pyface1, - /app/data/model-modules/model-face1, - /app/node_modules/onnxruntime-web, - /app/data/model-modules/model-har-motion1, - /app/generated/python-ws, - /app/node_modules/pyodide" + /app/services/ws-server/static, + /app/services/ws-wasm-agent" MODULES_ROOT: "@edge-toolkit/et-ws-server-static" OTLP_AUTH_USERNAME: root@example.com OTLP_COLLECTOR_URL: http://127.0.0.1:5080/api/default/v1 diff --git a/verification/local/output/facility-security-scenario/k3s.yaml b/verification/local/output/facility-security-scenario/k3s.yaml index 56f67382..16777623 100644 --- a/verification/local/output/facility-security-scenario/k3s.yaml +++ b/verification/local/output/facility-security-scenario/k3s.yaml @@ -146,7 +146,7 @@ spec: - et-ws-server env: - name: MODULES_PATHS - value: /app/services/ws-server/static,/app/services/ws-wasm-agent,/app/services/ws-modules/face-detection,/app/services/ws-modules/har1,/app/services/ws-modules/pyface1,/app/data/model-modules/model-face1,/app/node_modules/onnxruntime-web,/app/data/model-modules/model-har-motion1,/app/generated/python-ws,/app/node_modules/pyodide + value: /app/data/model-modules/model-face1,/app/data/model-modules/model-har-motion1,/app/generated/python-ws,/app/node_modules/onnxruntime-web,/app/node_modules/pyodide,/app/node_modules/stats-gl,/app/services/ws-modules/face-detection,/app/services/ws-modules/har1,/app/services/ws-modules/pyface1,/app/services/ws-server/static,/app/services/ws-wasm-agent - name: MODULES_ROOT value: "@edge-toolkit/et-ws-server-static" - name: OTLP_AUTH_USERNAME diff --git a/verification/local/output/facility-security-scenario/mise.toml b/verification/local/output/facility-security-scenario/mise.toml index 46caed1d..4e7d2623 100644 --- a/verification/local/output/facility-security-scenario/mise.toml +++ b/verification/local/output/facility-security-scenario/mise.toml @@ -25,11 +25,12 @@ docker run --rm --name openobserve -p 127.0.0.1:5080:5080 $settings -e ZO_ROOT_U description = "Run the WebSocket server" dir = "../../../../services/ws-server" run = """ -MODULES_PATHS="static, ../ws-wasm-agent, ../ws-modules/face-detection, ../ws-modules/har1, ../ws-modules/pyface1" -MODULES_PATHS="$MODULES_PATHS, $(cargo run --quiet -p et-cli -- npm-module-path --package onnxruntime-web)" +MODULES_PATHS="$(cargo run --quiet -p et-cli -- npm-module-path --package onnxruntime-web)" MODULES_PATHS="$MODULES_PATHS, $(cargo run --quiet -p et-cli -- npm-module-path --package stats-gl)" -MODULES_PATHS="$MODULES_PATHS, ../../data/model-modules/model-face1, ../../data/model-modules/model-har-motion1" -MODULES_PATHS="$MODULES_PATHS, ../../generated/python-ws, $(mise where http:pyodide)" +MODULES_PATHS="$MODULES_PATHS, $(mise where http:pyodide), ../../data/model-modules/model-face1" +MODULES_PATHS="$MODULES_PATHS, ../../data/model-modules/model-har-motion1, ../../generated/python-ws" +MODULES_PATHS="$MODULES_PATHS, ../ws-modules/face-detection, ../ws-modules/har1, ../ws-modules/pyface1" +MODULES_PATHS="$MODULES_PATHS, ../ws-wasm-agent, static" export MODULES_PATHS cargo run """ diff --git a/verification/local/output/math1/.gitignore b/verification/local/output/math1/.gitignore new file mode 100644 index 00000000..4b22f893 --- /dev/null +++ b/verification/local/output/math1/.gitignore @@ -0,0 +1,7 @@ +# Written by `et-cli`. +# What the hub writes into the directory it runs from: its self-signed TLS pair (the key is a +# credential too), the agent registry it saves on shutdown, and agent storage. +cert.pem +key.pem +registry.yaml +storage/ diff --git a/verification/local/output/math1/compose.yaml b/verification/local/output/math1/compose.yaml index 28aa8d40..989ea37e 100644 --- a/verification/local/output/math1/compose.yaml +++ b/verification/local/output/math1/compose.yaml @@ -46,8 +46,7 @@ services: hub: service:ws-server-hub network_mode: host environment: - MODULES_PATHS: "/app/services/ws-wasm-agent, - /app/services/ws-modules/math1, + MODULES_PATHS: "/app/services/ws-modules/math1, /app/services/ws-modules/math1-sender" MODULES_ROOT: "" OTLP_AUTH_USERNAME: root@example.com diff --git a/verification/local/output/math1/k3s.yaml b/verification/local/output/math1/k3s.yaml index 93363fb3..1e1ea639 100644 --- a/verification/local/output/math1/k3s.yaml +++ b/verification/local/output/math1/k3s.yaml @@ -146,7 +146,7 @@ spec: - et-ws-server env: - name: MODULES_PATHS - value: /app/services/ws-wasm-agent,/app/services/ws-modules/math1,/app/services/ws-modules/math1-sender + value: /app/services/ws-modules/math1,/app/services/ws-modules/math1-sender - name: MODULES_ROOT value: "" - name: OTLP_AUTH_USERNAME diff --git a/verification/local/output/math1/mise.toml b/verification/local/output/math1/mise.toml index 9d20a766..f4c8c6fd 100644 --- a/verification/local/output/math1/mise.toml +++ b/verification/local/output/math1/mise.toml @@ -47,7 +47,7 @@ docker run --rm --name openobserve -p 127.0.0.1:5080:5080 $settings -e ZO_ROOT_U description = "Run the WebSocket server" dir = "../../../../services/ws-server" run = """ -MODULES_PATHS="../ws-wasm-agent, ../ws-modules/math1, ../ws-modules/math1-sender" +MODULES_PATHS="../ws-modules/math1, ../ws-modules/math1-sender" export MODULES_PATHS cargo run """ diff --git a/verification/local/output/pyo3-math1/.gitignore b/verification/local/output/pyo3-math1/.gitignore new file mode 100644 index 00000000..4b22f893 --- /dev/null +++ b/verification/local/output/pyo3-math1/.gitignore @@ -0,0 +1,7 @@ +# Written by `et-cli`. +# What the hub writes into the directory it runs from: its self-signed TLS pair (the key is a +# credential too), the agent registry it saves on shutdown, and agent storage. +cert.pem +key.pem +registry.yaml +storage/ diff --git a/verification/local/output/pyo3-math1/compose.yaml b/verification/local/output/pyo3-math1/compose.yaml index 1aa7701e..11509c7d 100644 --- a/verification/local/output/pyo3-math1/compose.yaml +++ b/verification/local/output/pyo3-math1/compose.yaml @@ -46,8 +46,7 @@ services: hub: service:ws-server-hub network_mode: host environment: - MODULES_PATHS: "/app/services/ws-wasm-agent, - /app/services/ws-modules/pyo3-math1, + MODULES_PATHS: "/app/services/ws-modules/pyo3-math1, /app/services/ws-modules/wasi-math1-sender" MODULES_ROOT: "" OTLP_AUTH_USERNAME: root@example.com diff --git a/verification/local/output/pyo3-math1/k3s.yaml b/verification/local/output/pyo3-math1/k3s.yaml index d63412a5..7b41d368 100644 --- a/verification/local/output/pyo3-math1/k3s.yaml +++ b/verification/local/output/pyo3-math1/k3s.yaml @@ -146,7 +146,7 @@ spec: - et-ws-server env: - name: MODULES_PATHS - value: /app/services/ws-wasm-agent,/app/services/ws-modules/pyo3-math1,/app/services/ws-modules/wasi-math1-sender + value: /app/services/ws-modules/pyo3-math1,/app/services/ws-modules/wasi-math1-sender - name: MODULES_ROOT value: "" - name: OTLP_AUTH_USERNAME diff --git a/verification/local/output/pyo3-math1/mise.toml b/verification/local/output/pyo3-math1/mise.toml index bfa8906d..f1a9619c 100644 --- a/verification/local/output/pyo3-math1/mise.toml +++ b/verification/local/output/pyo3-math1/mise.toml @@ -47,7 +47,7 @@ WS_SERVER_URL = "ws://localhost:8080/ws" description = "Run the WebSocket server" dir = "../../../../services/ws-server" run = """ -MODULES_PATHS="../ws-wasm-agent, ../ws-modules/pyo3-math1, ../ws-modules/wasi-math1-sender" +MODULES_PATHS="../ws-modules/pyo3-math1, ../ws-modules/wasi-math1-sender" export MODULES_PATHS cargo run """ diff --git a/verification/local/output/wasi-math1/.gitignore b/verification/local/output/wasi-math1/.gitignore new file mode 100644 index 00000000..4b22f893 --- /dev/null +++ b/verification/local/output/wasi-math1/.gitignore @@ -0,0 +1,7 @@ +# Written by `et-cli`. +# What the hub writes into the directory it runs from: its self-signed TLS pair (the key is a +# credential too), the agent registry it saves on shutdown, and agent storage. +cert.pem +key.pem +registry.yaml +storage/ diff --git a/verification/local/output/wasi-math1/compose.yaml b/verification/local/output/wasi-math1/compose.yaml index 1a2cfe7f..b06cecb9 100644 --- a/verification/local/output/wasi-math1/compose.yaml +++ b/verification/local/output/wasi-math1/compose.yaml @@ -46,8 +46,7 @@ services: hub: service:ws-server-hub network_mode: host environment: - MODULES_PATHS: "/app/services/ws-wasm-agent, - /app/services/ws-modules/wasi-math1, + MODULES_PATHS: "/app/services/ws-modules/wasi-math1, /app/services/ws-modules/wasi-math1-sender" MODULES_ROOT: "" OTLP_AUTH_USERNAME: root@example.com diff --git a/verification/local/output/wasi-math1/k3s.yaml b/verification/local/output/wasi-math1/k3s.yaml index 9f602c22..4c6d6331 100644 --- a/verification/local/output/wasi-math1/k3s.yaml +++ b/verification/local/output/wasi-math1/k3s.yaml @@ -146,7 +146,7 @@ spec: - et-ws-server env: - name: MODULES_PATHS - value: /app/services/ws-wasm-agent,/app/services/ws-modules/wasi-math1,/app/services/ws-modules/wasi-math1-sender + value: /app/services/ws-modules/wasi-math1,/app/services/ws-modules/wasi-math1-sender - name: MODULES_ROOT value: "" - name: OTLP_AUTH_USERNAME diff --git a/verification/local/output/wasi-math1/mise.toml b/verification/local/output/wasi-math1/mise.toml index 63a4a3e6..25e1186f 100644 --- a/verification/local/output/wasi-math1/mise.toml +++ b/verification/local/output/wasi-math1/mise.toml @@ -47,7 +47,7 @@ WS_SERVER_URL = "ws://localhost:8080/ws" description = "Run the WebSocket server" dir = "../../../../services/ws-server" run = """ -MODULES_PATHS="../ws-wasm-agent, ../ws-modules/wasi-math1, ../ws-modules/wasi-math1-sender" +MODULES_PATHS="../ws-modules/wasi-math1, ../ws-modules/wasi-math1-sender" export MODULES_PATHS cargo run """ diff --git a/verification/published/output/default/.gitignore b/verification/published/output/default/.gitignore new file mode 100644 index 00000000..4b22f893 --- /dev/null +++ b/verification/published/output/default/.gitignore @@ -0,0 +1,7 @@ +# Written by `et-cli`. +# What the hub writes into the directory it runs from: its self-signed TLS pair (the key is a +# credential too), the agent registry it saves on shutdown, and agent storage. +cert.pem +key.pem +registry.yaml +storage/ diff --git a/verification/published/output/default/compose.yaml b/verification/published/output/default/compose.yaml index 33411588..8b676e10 100644 --- a/verification/published/output/default/compose.yaml +++ b/verification/published/output/default/compose.yaml @@ -41,7 +41,9 @@ services: hub: docker-image://ghcr.io/edge-toolkit/core/et-ws-server:latest network_mode: host environment: - MODULES_PATHS: "/app/services/ws-server/static, + MODULES_PATHS: "/app/node_modules/onnxruntime-web, + /app/node_modules/stats-gl, + /app/services/ws-server/static, /app/services/ws-wasm-agent" MODULES_ROOT: "@edge-toolkit/et-ws-server-static" OTLP_AUTH_USERNAME: root@example.com diff --git a/verification/published/output/default/k3s.yaml b/verification/published/output/default/k3s.yaml index 8e85b42c..8919af49 100644 --- a/verification/published/output/default/k3s.yaml +++ b/verification/published/output/default/k3s.yaml @@ -146,7 +146,7 @@ spec: - et-ws-server env: - name: MODULES_PATHS - value: /app/services/ws-server/static,/app/services/ws-wasm-agent + value: /app/node_modules/onnxruntime-web,/app/node_modules/stats-gl,/app/services/ws-server/static,/app/services/ws-wasm-agent - name: MODULES_ROOT value: "@edge-toolkit/et-ws-server-static" - name: OTLP_AUTH_USERNAME diff --git a/verification/published/output/default/mise.toml b/verification/published/output/default/mise.toml index dc87a1fe..6555a867 100644 --- a/verification/published/output/default/mise.toml +++ b/verification/published/output/default/mise.toml @@ -30,6 +30,7 @@ et-ws-server [tasks.ws-server.env] MODULES_ROOT = "@edge-toolkit/et-ws-server-static" +STORAGE_URL = "file://{{ config_root }}/storage" [tools] "cargo:open" = "latest" diff --git a/verification/published/output/math1/.gitignore b/verification/published/output/math1/.gitignore new file mode 100644 index 00000000..4b22f893 --- /dev/null +++ b/verification/published/output/math1/.gitignore @@ -0,0 +1,7 @@ +# Written by `et-cli`. +# What the hub writes into the directory it runs from: its self-signed TLS pair (the key is a +# credential too), the agent registry it saves on shutdown, and agent storage. +cert.pem +key.pem +registry.yaml +storage/ diff --git a/verification/published/output/math1/compose.yaml b/verification/published/output/math1/compose.yaml index 2b8d2852..855ef5a4 100644 --- a/verification/published/output/math1/compose.yaml +++ b/verification/published/output/math1/compose.yaml @@ -41,8 +41,7 @@ services: hub: docker-image://ghcr.io/edge-toolkit/core/et-ws-server:latest network_mode: host environment: - MODULES_PATHS: "/app/services/ws-wasm-agent, - /app/services/ws-modules/math1, + MODULES_PATHS: "/app/services/ws-modules/math1, /app/services/ws-modules/math1-sender" MODULES_ROOT: "" OTLP_AUTH_USERNAME: root@example.com diff --git a/verification/published/output/math1/k3s.yaml b/verification/published/output/math1/k3s.yaml index dd7427cc..ec9bd54d 100644 --- a/verification/published/output/math1/k3s.yaml +++ b/verification/published/output/math1/k3s.yaml @@ -146,7 +146,7 @@ spec: - et-ws-server env: - name: MODULES_PATHS - value: /app/services/ws-wasm-agent,/app/services/ws-modules/math1,/app/services/ws-modules/math1-sender + value: /app/services/ws-modules/math1,/app/services/ws-modules/math1-sender - name: MODULES_ROOT value: "" - name: OTLP_AUTH_USERNAME diff --git a/verification/published/output/math1/mise.toml b/verification/published/output/math1/mise.toml index 936426e1..bc66f91a 100644 --- a/verification/published/output/math1/mise.toml +++ b/verification/published/output/math1/mise.toml @@ -49,6 +49,7 @@ et-ws-server """ [tasks.ws-server.env] +STORAGE_URL = "file://{{ config_root }}/storage" [tools] "cargo:open" = "latest" @@ -68,7 +69,3 @@ version = "latest" [tools."npm:@edge-toolkit/et-ws-math1-sender"] minimum_release_age = "0" version = "latest" - -[tools."npm:@edge-toolkit/et-ws-wasm-agent"] -minimum_release_age = "0" -version = "latest" diff --git a/verification/published/output/pyo3-math1/.gitignore b/verification/published/output/pyo3-math1/.gitignore new file mode 100644 index 00000000..4b22f893 --- /dev/null +++ b/verification/published/output/pyo3-math1/.gitignore @@ -0,0 +1,7 @@ +# Written by `et-cli`. +# What the hub writes into the directory it runs from: its self-signed TLS pair (the key is a +# credential too), the agent registry it saves on shutdown, and agent storage. +cert.pem +key.pem +registry.yaml +storage/ diff --git a/verification/published/output/pyo3-math1/compose.yaml b/verification/published/output/pyo3-math1/compose.yaml index 455e4cb0..52728dc0 100644 --- a/verification/published/output/pyo3-math1/compose.yaml +++ b/verification/published/output/pyo3-math1/compose.yaml @@ -41,8 +41,7 @@ services: hub: docker-image://ghcr.io/edge-toolkit/core/et-ws-server:latest network_mode: host environment: - MODULES_PATHS: "/app/services/ws-wasm-agent, - /app/services/ws-modules/pyo3-math1, + MODULES_PATHS: "/app/services/ws-modules/pyo3-math1, /app/services/ws-modules/wasi-math1-sender" MODULES_ROOT: "" OTLP_AUTH_USERNAME: root@example.com diff --git a/verification/published/output/pyo3-math1/k3s.yaml b/verification/published/output/pyo3-math1/k3s.yaml index 987af0ca..51622d6b 100644 --- a/verification/published/output/pyo3-math1/k3s.yaml +++ b/verification/published/output/pyo3-math1/k3s.yaml @@ -146,7 +146,7 @@ spec: - et-ws-server env: - name: MODULES_PATHS - value: /app/services/ws-wasm-agent,/app/services/ws-modules/pyo3-math1,/app/services/ws-modules/wasi-math1-sender + value: /app/services/ws-modules/pyo3-math1,/app/services/ws-modules/wasi-math1-sender - name: MODULES_ROOT value: "" - name: OTLP_AUTH_USERNAME diff --git a/verification/published/output/pyo3-math1/mise.toml b/verification/published/output/pyo3-math1/mise.toml index 277c0771..26ff9dc9 100644 --- a/verification/published/output/pyo3-math1/mise.toml +++ b/verification/published/output/pyo3-math1/mise.toml @@ -49,6 +49,7 @@ et-ws-server """ [tasks.ws-server.env] +STORAGE_URL = "file://{{ config_root }}/storage" [tools] "cargo:open" = "latest" @@ -72,7 +73,3 @@ version = "latest" [tools."npm:@edge-toolkit/et-ws-wasi-math1-sender"] minimum_release_age = "0" version = "latest" - -[tools."npm:@edge-toolkit/et-ws-wasm-agent"] -minimum_release_age = "0" -version = "latest" diff --git a/verification/published/output/wasi-math1/.gitignore b/verification/published/output/wasi-math1/.gitignore new file mode 100644 index 00000000..4b22f893 --- /dev/null +++ b/verification/published/output/wasi-math1/.gitignore @@ -0,0 +1,7 @@ +# Written by `et-cli`. +# What the hub writes into the directory it runs from: its self-signed TLS pair (the key is a +# credential too), the agent registry it saves on shutdown, and agent storage. +cert.pem +key.pem +registry.yaml +storage/ diff --git a/verification/published/output/wasi-math1/compose.yaml b/verification/published/output/wasi-math1/compose.yaml index 9d50aca4..433469d9 100644 --- a/verification/published/output/wasi-math1/compose.yaml +++ b/verification/published/output/wasi-math1/compose.yaml @@ -41,8 +41,7 @@ services: hub: docker-image://ghcr.io/edge-toolkit/core/et-ws-server:latest network_mode: host environment: - MODULES_PATHS: "/app/services/ws-wasm-agent, - /app/services/ws-modules/wasi-math1, + MODULES_PATHS: "/app/services/ws-modules/wasi-math1, /app/services/ws-modules/wasi-math1-sender" MODULES_ROOT: "" OTLP_AUTH_USERNAME: root@example.com diff --git a/verification/published/output/wasi-math1/k3s.yaml b/verification/published/output/wasi-math1/k3s.yaml index 0ff7f2a6..ddfd0bd1 100644 --- a/verification/published/output/wasi-math1/k3s.yaml +++ b/verification/published/output/wasi-math1/k3s.yaml @@ -146,7 +146,7 @@ spec: - et-ws-server env: - name: MODULES_PATHS - value: /app/services/ws-wasm-agent,/app/services/ws-modules/wasi-math1,/app/services/ws-modules/wasi-math1-sender + value: /app/services/ws-modules/wasi-math1,/app/services/ws-modules/wasi-math1-sender - name: MODULES_ROOT value: "" - name: OTLP_AUTH_USERNAME diff --git a/verification/published/output/wasi-math1/mise.toml b/verification/published/output/wasi-math1/mise.toml index dc4d4af8..28dd9f1d 100644 --- a/verification/published/output/wasi-math1/mise.toml +++ b/verification/published/output/wasi-math1/mise.toml @@ -49,6 +49,7 @@ et-ws-server """ [tasks.ws-server.env] +STORAGE_URL = "file://{{ config_root }}/storage" [tools] "cargo:open" = "latest" @@ -68,7 +69,3 @@ version = "latest" [tools."npm:@edge-toolkit/et-ws-wasi-math1-sender"] minimum_release_age = "0" version = "latest" - -[tools."npm:@edge-toolkit/et-ws-wasm-agent"] -minimum_release_age = "0" -version = "latest"