From 1747a8304b9aa21baede5a1d8ecfd2dfbfe96828 Mon Sep 17 00:00:00 2001 From: Vyncint Ng Date: Tue, 22 Sep 2026 13:11:15 +0700 Subject: [PATCH] test: cover the part that decides what the number is 94 tests for 7,571 lines, and the distribution was the finding: the best-tested crates were the ones that present a result or refuse a candidate, and the ones that produce it were thin. 123 now. The exit-code contract, which the README states and nothing checked. launchbound-cli had three tests, all about parse_budget, and no integration test at all. Exit 1 -- the fastest candidates were refused and the chosen one is slower -- is the product's argument in one integer, and the one a refactor can quietly turn into 0, since stdout is identical either way. The recorded Metal sweep has no refused candidate, so asserting against it as-is would have been a test that cannot fail; the test marks the two fastest refused -- two, because their 95% intervals overlap each other and rejected_faster needs the refused interval wholly below the chosen one's -- and requires exit 1. Replacing the branch with ExitCode::SUCCESS fails it and nothing else. Checkpoint and resume: run_plan needs a device, the two halves that decide whether resuming is safe do not. A checkpoint misread as "no checkpoint" silently re-measures candidates that cost GPU time, so each untrustworthy shape now fails with the path named, and write-then-rename leaves no .json.tmp. launchbound-metal had one test, about rewriting a constant in a string. Six now, and CI executes an actual dispatch on the macOS leg: one corpus kernel under a 60s budget, read back through `report`. It asserts a dispatch happened and produced a readable results.v1, not that the numbers mean anything -- timings from a shared runner are not measurements. The model's ranking is pinned against the real corpus kernel as a relation rather than as numbers: the same space ranks the same way twice, and sorting by cost is a total order with nothing infinite in it. Signed-off-by: Vyncint Ng --- .github/workflows/ci.yml | 35 +++ CHANGELOG.md | 56 ++++ crates/launchbound-bench/src/run.rs | 100 +++++- crates/launchbound-cli/tests/exit_codes.rs | 342 +++++++++++++++++++++ crates/launchbound-metal/src/lib.rs | 108 +++++++ crates/launchbound-model/src/lib.rs | 78 +++++ 6 files changed, 718 insertions(+), 1 deletion(-) create mode 100644 crates/launchbound-cli/tests/exit_codes.rs diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index c9fbbf0..f4e90de 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -36,6 +36,41 @@ jobs: - uses: taiki-e/install-action@94c31af3204a9f15ab40b35ad084410b905bbc73 # v2.87.17 with: tool: just,cargo-deny + # The Metal path, executed rather than merely compiled. + # + # `run_metal` is the function behind the README's "On Apple Silicon it + # measures on Metal", and until now nothing called it outside the CLI: + # its crate had one test, about rewriting a constant in a source + # string, and the macOS leg compiled the `cfg(target_os = "macos")` + # code and then ran that. The only evidence the path works was + # `runs/reduce-stable-metal`, one sweep on one machine, committed and + # never re-run. + # + # A smoke test, deliberately: it asserts that a dispatch happens and + # produces a readable `results.v1`, not that the numbers are anything. + # Timings from a shared runner are not measurements and this does not + # pretend otherwise -- `--budget 60s` bounds it, and `report` reading + # the run back is the assertion. + # + # `--cc` is required by the parser and ignored on this path, which has + # no gate to run at any capability. + - name: Measure one kernel on Metal + if: runner.os == 'macOS' + run: | + set -euo pipefail + launchbound() { cargo run -q -p launchbound-cli --bin launchbound -- "$@"; } + launchbound tune corpus/reduce-stable --cc 8.6 --backend metal \ + --out "${RUNNER_TEMP}/metal-smoke" --budget 60s + launchbound report "${RUNNER_TEMP}/metal-smoke" --json > "${RUNNER_TEMP}/metal.json" + python3 - "${RUNNER_TEMP}/metal.json" <<'PY' + import json, sys + report = json.load(open(sys.argv[1])) + chosen = report.get("chosen") + assert chosen, "the Metal sweep chose nothing" + assert chosen["summary"]["median_ms"] > 0, chosen + print("metal ok:", chosen["config"], chosen["summary"]["median_ms"], "ms") + PY + - run: just ci env: # Every screen a failing termlens wait embeds is also written here, diff --git a/CHANGELOG.md b/CHANGELOG.md index 61651fe..94aec89 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -29,6 +29,62 @@ change measured timings are marked `bench:`. is pointed at reconverge, which this project ships as a component and does not reimplement. +- **Tests for the part of this tool that decides what the number is.** 94 + tests for 7,571 lines, and the distribution was the finding: the + best-tested crates were the ones that *present* a result or *refuse* a + candidate, and the ones that produce it were the thin ones. 123 now. + + - **The exit-code contract**, which the README states and nothing checked. + `launchbound-cli` had three tests, all about `parse_budget`, and no + integration test at all. Exit 1 — "the fastest candidates were refused + and the chosen one is slower" — is the product's whole argument in one + integer, and it is the one a refactor can quietly turn into 0, because + the text on stdout is identical either way. + + The recorded Metal sweep has no refused candidate (`gate: "none"`), so + asserting against it as-is would have been exactly the kind of test that + cannot fail. The test marks the **two** fastest candidates refused — two, + because their 95% intervals overlap each other and `rejected_faster` + requires the refused interval to sit wholly below the chosen one's, which + is the same "indistinguishable, never ranked" rule the ranking uses — and + then requires exit 1. Replacing the branch with `ExitCode::SUCCESS` fails + it; nothing else in the suite notices. + + Also checked: `--allow-unsafe` without `--reason`, and with a blank one; + `--reason` without `--allow-unsafe`; that `tune` has no `--allow-unsafe` + at all; that a malformed `--cc` is refused before anything is spawned; + and that `--json` and the text report never disagree about the status. + + - **Checkpoint and resume.** `run_plan` needs a device, but the two halves + that decide whether a resume is safe do not. A checkpoint misread as "no + checkpoint" silently re-measures candidates that cost GPU time, so each + untrustworthy shape — not JSON, `results.v2`, no `schema` at all — is now + required to fail with the path named. And the write-then-rename leaves no + `.json.tmp` behind. + + - **The Metal path.** `launchbound-metal` had one test, about rewriting a + constant in a string. Six now: the no-gate notice is asserted as text + rather than trusted as a constant, the off-macOS stub must refuse rather + than return an empty sweep, a CUDA-only dimension with no MSL twin is + skipped rather than invented, and an unterminated constant is named + rather than silently truncating the source. + + - **The model's ranking**, pinned against the real corpus kernel rather + than a synthetic space, as a relation rather than as numbers: the same + space ranks the same way twice, and sorting by cost is a total order with + nothing infinite in it. Every other test there checked a piece of the + cost function; none checked that the pieces compose into the same order. + Plus: an unknown capability must list the ones that exist. + +- **CI executes a Metal dispatch.** The macOS leg compiled the + `cfg(target_os = "macos")` code and then ran a string-rewriting test; + `run_metal` — the function behind "On Apple Silicon it measures on Metal" — + had no caller outside the CLI, and the only evidence the path worked was + one committed sweep on one machine. A smoke step now tunes one corpus + kernel under a 60s budget and reads the run back. It asserts that a + dispatch happened and produced a readable `results.v1`, not that the + numbers mean anything: timings from a shared runner are not measurements. + ### Fixed - **The GitHub Release is created by the release workflow, and the semver diff --git a/crates/launchbound-bench/src/run.rs b/crates/launchbound-bench/src/run.rs index cbf8c5d..42fb519 100644 --- a/crates/launchbound-bench/src/run.rs +++ b/crates/launchbound-bench/src/run.rs @@ -596,7 +596,105 @@ pub fn parse_budget(flag: &str, text: &str) -> Result { #[cfg(test)] mod budget_tests { - use super::parse_budget; + use super::{Results, parse_budget}; + + // ---- checkpoint and resume ------------------------------------------- + // + // The checkpoint is the only record that a candidate was measured, and + // every one of them cost GPU time. `run_plan` cannot be exercised without + // a device, but the two halves that decide whether a resume is safe -- + // reading the file and writing it atomically -- can be, and they are + // where a wrong answer is expensive: a checkpoint misread as "start over" + // silently re-measures, and one misread as valid resumes into nonsense. + + fn scratch(name: &str) -> std::path::PathBuf { + let dir = std::path::Path::new(env!("CARGO_MANIFEST_DIR")) + .join("../../target/tmp") + .join(name); + let _ = std::fs::remove_dir_all(&dir); + std::fs::create_dir_all(&dir).expect("a scratch directory"); + dir + } + + fn sample_results() -> Results { + Results { + schema: "results.v1".into(), + kernel: "k".into(), + entry: "k".into(), + plan_cc: "8.6".into(), + device_name: "probe".into(), + device_cc: "8.6".into(), + driver_version: "0".into(), + candidates: Vec::new(), + total_gpu_seconds: 1.5, + strategy: Some("exhaustive".into()), + budget_exhausted: false, + } + } + + #[test] + fn no_checkpoint_is_a_fresh_start_not_an_error() { + let dir = scratch("results-absent"); + assert!( + Results::load(&dir.join("results.json")) + .expect("a missing file is not a failure") + .is_none() + ); + } + + #[test] + fn a_checkpoint_round_trips_and_leaves_no_temporary_behind() { + let dir = scratch("results-roundtrip"); + let path = dir.join("results.json"); + let written = sample_results(); + written.checkpoint(&path).expect("checkpointing"); + + let read = Results::load(&path).expect("readable").expect("present"); + assert_eq!(read.kernel, written.kernel); + assert_eq!(read.total_gpu_seconds, written.total_gpu_seconds); + assert_eq!(read.strategy, written.strategy); + + // Write-then-rename: the `.json.tmp` must not survive, or the next + // reader of the directory finds two documents and no rule for which. + let leftovers: Vec<_> = std::fs::read_dir(&dir) + .expect("readable directory") + .filter_map(Result::ok) + .map(|e| e.file_name().to_string_lossy().into_owned()) + .filter(|n| n.ends_with(".tmp")) + .collect(); + assert!(leftovers.is_empty(), "left behind: {leftovers:?}"); + } + + #[test] + fn a_checkpoint_that_cannot_be_trusted_stops_the_run() { + // Each of these used to be indistinguishable from "no checkpoint", + // and resuming from that discards measurements that cost GPU time. + // The error must name the path, because the operator is looking at a + // run directory, not at a stack trace. + let dir = scratch("results-untrustworthy"); + for (name, body, expected) in [ + ("not-json.json", "{ this is not json", "not JSON"), + ( + "wrong-schema.json", + r#"{"schema":"results.v2","kernel":"k"}"#, + "unsupported results schema", + ), + ( + "no-schema.json", + r#"{"kernel":"k","candidates":[]}"#, + "no `schema` field", + ), + ] { + let path = dir.join(name); + std::fs::write(&path, body).expect("writing the probe"); + let err = Results::load(&path).expect_err("must not be read as a fresh start"); + assert!(err.contains(expected), "{name}: {err}"); + assert!( + err.contains(&path.display().to_string()), + "{name}: the message must name the file: {err}" + ); + } + } #[test] fn accepted_forms_are_seconds() { diff --git a/crates/launchbound-cli/tests/exit_codes.rs b/crates/launchbound-cli/tests/exit_codes.rs new file mode 100644 index 0000000..081b2b7 --- /dev/null +++ b/crates/launchbound-cli/tests/exit_codes.rs @@ -0,0 +1,342 @@ +//! The exit-code contract, which the README states and nothing checked. +//! +//! > Exit codes: `0` a safe configuration was found; `1` the fastest +//! > candidates were refused and the chosen one is slower than a rejected +//! > candidate — notable, not an error; `2` tool error. +//! +//! That `1` is the product's whole argument in one integer: a caller can +//! tell "tuned, and the honest answer cost you something" from "tuned, and +//! nothing was refused" without parsing a report. It is also the one a +//! refactor can quietly turn into `0`, because every output-only test +//! passes either way — the text is identical, only the status differs. +//! +//! None of this needs a GPU. `report` reads a recorded run, and the usage +//! errors are decided before any device is opened. + +use std::path::{Path, PathBuf}; +use std::process::{Command, Output}; + +fn repo_root() -> PathBuf { + // CARGO_MANIFEST_DIR is crates/launchbound-cli. + Path::new(env!("CARGO_MANIFEST_DIR")) + .ancestors() + .nth(2) + .expect("the repository root is two levels above this crate") + .to_path_buf() +} + +fn launchbound(args: &[&str]) -> Output { + Command::new(env!("CARGO_BIN_EXE_launchbound")) + .args(args) + .current_dir(repo_root()) + .output() + .expect("the binary under test runs") +} + +fn code(out: &Output) -> i32 { + out.status + .code() + .expect("exited rather than being signalled") +} + +fn stderr(out: &Output) -> String { + String::from_utf8_lossy(&out.stderr).into_owned() +} + +/// The recorded Metal sweep committed under `runs/`, which is the only run in +/// the repository and the reason these tests need no hardware. +const RECORDED_RUN: &str = "runs/reduce-stable-metal"; + +#[test] +fn reporting_a_recorded_run_agrees_with_its_own_json() { + // The status is not asserted against a constant: whether this run has a + // refused-but-faster candidate is a property of the recording, and + // pinning the integer would make the test a copy of the fixture rather + // than a check of the rule. So the rule is checked directly -- the exit + // code must be 1 exactly when `rejected_faster` is non-empty. + let json = launchbound(&["report", RECORDED_RUN, "--json"]); + let report: serde_json::Value = + serde_json::from_slice(&json.stdout).expect("--json prints a JSON document"); + let refused_faster = report + .get("rejected_faster") + .and_then(|v| v.as_array()) + .expect("the report names the refused-but-faster set") + .len(); + + let text = launchbound(&["report", RECORDED_RUN]); + let expected = i32::from(refused_faster > 0); + assert_eq!( + code(&text), + expected, + "{refused_faster} refused-but-faster candidate(s) must mean exit {expected}" + ); + // The format a caller asked for must not change the verdict it is told. + assert_eq!( + code(&json), + code(&text), + "--json and the text report disagreed about the exit code" + ); +} + +/// Build a run directory from the recorded one with `disqualified` marked +/// refused, and return its path. Everything else is copied verbatim, so the +/// timings are real measurements from the Apple M4 Pro sweep. +fn run_with_a_refusal(name: &str, disqualified: &[String]) -> PathBuf { + let root = repo_root(); + let dst = root.join("target/tmp").join(name); + std::fs::create_dir_all(&dst).expect("a scratch run directory"); + std::fs::copy( + root.join(RECORDED_RUN).join("results.json"), + dst.join("results.json"), + ) + .expect("the recorded results"); + + let text = std::fs::read_to_string(root.join(RECORDED_RUN).join("verdicts.json")) + .expect("the recorded verdicts"); + let mut verdicts: serde_json::Value = serde_json::from_str(&text).expect("verdicts.v1 JSON"); + let mut marked = 0; + for c in verdicts["candidates"] + .as_array_mut() + .expect("verdicts name their candidates") + { + let config = c["config"].as_str().unwrap_or_default().to_string(); + if disqualified.contains(&config) { + c["verdict"] = serde_json::json!("disqualified"); + c["rules"] = serde_json::json!(["RC001"]); + marked += 1; + } + } + assert_eq!( + marked, + disqualified.len(), + "every named config must match exactly one candidate" + ); + std::fs::write( + dst.join("verdicts.json"), + serde_json::to_string_pretty(&verdicts).expect("serializable"), + ) + .expect("writing the doctored verdicts"); + dst +} + +/// The `n` fastest configurations in the recorded sweep, by median, read from +/// it rather than hard-coded so this cannot drift if the recording is remade. +fn fastest_configs(n: usize) -> Vec { + let text = std::fs::read_to_string(repo_root().join(RECORDED_RUN).join("results.json")) + .expect("the recorded results"); + let results: serde_json::Value = serde_json::from_str(&text).expect("results.v1 JSON"); + let mut measured: Vec<(f64, String)> = results["candidates"] + .as_array() + .expect("candidates") + .iter() + .filter_map(|c| { + Some(( + c["summary"]["median_ms"].as_f64()?, + c["config"].as_str()?.to_string(), + )) + }) + .collect(); + measured.sort_by(|a, b| a.0.total_cmp(&b.0)); + assert!( + measured.len() > n, + "the recording must have more than {n} measured candidates" + ); + measured.into_iter().take(n).map(|(_, c)| c).collect() +} + +/// Exit 1 is the contract that a refactor can quietly turn into 0: the text +/// on stdout is identical either way, so every output-only test passes. +/// +/// The recorded sweep has no refused candidate at all -- `gate: "none"`, the +/// Metal path has no convergence gate -- so asserting against it as-is would +/// have been exactly that kind of test. This marks the *fastest* candidate +/// refused, which is the shape the exit code exists for: the honest answer +/// is slower than something the tool would not hand you. +#[test] +fn a_refused_candidate_that_measured_faster_is_exit_one() { + // The two fastest, not one. Their 95% intervals overlap each other + // ([0.09679, 0.09688] both), and `rejected_faster` requires the refused + // candidate's whole interval to sit *below* the chosen one's -- the same + // "overlapping intervals are indistinguishable, never ranked" rule the + // ranking uses. Refusing only the first would leave the second as the + // chosen one and produce no gap, which is correct behaviour and would + // have made this test pass for the wrong reason. + let dir = run_with_a_refusal("report-refusal", &fastest_configs(2)); + let dir = dir.to_string_lossy().into_owned(); + + let json = launchbound(&["report", &dir, "--json"]); + let report: serde_json::Value = + serde_json::from_slice(&json.stdout).expect("--json prints a JSON document"); + assert!( + !report["rejected_faster"] + .as_array() + .expect("the refused-but-faster set") + .is_empty(), + "marking the fastest candidate refused must produce one" + ); + + let text = launchbound(&["report", &dir]); + assert_eq!( + code(&text), + 1, + "a refused candidate that measured faster is exit 1, not 0:\n{}", + String::from_utf8_lossy(&text.stdout) + ); + assert!( + String::from_utf8_lossy(&text.stdout).contains("REFUSED BUT FASTER"), + "and the reader is told which one" + ); + + // `--rejected` prints the same section, and `--json` the same document; + // both keep the status, because the format is not the verdict. + let only = launchbound(&["report", &dir, "--rejected"]); + assert_eq!(code(&only), 1); + assert_eq!(code(&json), 1, "--json must carry the same exit code"); +} + +#[test] +fn json_and_text_report_the_same_run() { + // A caller may read either; they must not disagree about the verdict. + let json = launchbound(&["report", RECORDED_RUN, "--json"]); + let text = launchbound(&["report", RECORDED_RUN]); + assert_eq!(code(&json), code(&text), "the two formats must agree"); + assert!( + !text.stdout.is_empty(), + "the text report is what a human reads; it must not be empty" + ); +} + +#[test] +fn a_missing_run_directory_is_a_tool_error_not_a_verdict() { + // Exit 2 rather than 1: nothing was measured, so there is no verdict to + // report, and a caller that treats 1 as "notable" must not see one here. + let out = launchbound(&["report", "runs/there-is-no-such-run"]); + assert_eq!(code(&out), 2, "stderr was:\n{}", stderr(&out)); + assert!( + stderr(&out).starts_with("error:"), + "the diagnosis must be the first thing on stderr:\n{}", + stderr(&out) + ); +} + +#[test] +fn allow_unsafe_without_a_reason_is_a_usage_error() { + // From the README: "it requires `--reason` with a non-empty string, + // recorded verbatim in the report. A missing reason is a usage error, + // not a warning." Measuring a configuration the gate refused is a + // deliberate act, and the deliberation is the string. + // `--out` is supplied because clap requires it, and a clap error about a + // missing `--out` would let this test pass without ever reaching the + // check it is about. Nothing is written there: the reason is validated + // first, which is the other half of the contract. + let out = launchbound(&[ + "stage", + "reduce-flip", + "--cc", + "8.6", + "--out", + "target/tmp/stage-probe", + "--allow-unsafe", + ]); + assert_eq!(code(&out), 2, "stderr was:\n{}", stderr(&out)); + assert!( + !Path::new("target/tmp/stage-probe").exists(), + "a refused usage must not have created the output directory" + ); + assert!( + stderr(&out).contains("--reason"), + "the message must name the flag that is missing:\n{}", + stderr(&out) + ); +} + +#[test] +fn a_blank_reason_is_the_same_as_no_reason() { + // Whitespace is not an explanation, and a report carrying `reason: " "` + // is worse than one that was never produced: it looks deliberated. + for blank in ["", " "] { + let out = launchbound(&[ + "stage", + "reduce-flip", + "--cc", + "8.6", + "--out", + "target/tmp/stage-probe", + "--allow-unsafe", + "--reason", + blank, + ]); + assert_eq!(code(&out), 2, "blank reason {blank:?} was accepted"); + assert!(stderr(&out).contains("--reason"), "{}", stderr(&out)); + } +} + +#[test] +fn a_reason_without_allow_unsafe_is_refused_too() { + // The other direction. A `--reason` alone means the caller believes they + // asked for something they did not ask for. + let out = launchbound(&[ + "stage", + "reduce-flip", + "--cc", + "8.6", + "--out", + "target/tmp/stage-probe", + "--reason", + "measuring the refusal", + ]); + assert_eq!(code(&out), 2, "stderr was:\n{}", stderr(&out)); +} + +#[test] +fn tune_has_no_allow_unsafe_at_all() { + // "`--allow-unsafe` is on `stage`, not on `tune`" -- the flag a user + // reaches for when a refusal is inconvenient must not exist on the + // command whose answer they act on. + let out = launchbound(&["tune", "reduce-flip", "--cc", "8.6", "--allow-unsafe"]); + assert_eq!(code(&out), 2, "stderr was:\n{}", stderr(&out)); + let err = stderr(&out); + assert!( + err.contains("unexpected argument") || err.contains("--allow-unsafe"), + "the refusal should be about the flag itself:\n{err}" + ); +} + +#[test] +fn a_malformed_cc_is_a_tool_error_before_anything_is_spawned() { + // `prune` used to hand whatever was typed straight to `cargo reconverge`, + // once per candidate: a mistyped `--cc 80` spawned eleven subprocesses + // and printed ninety lines in which the actual problem appeared nowhere. + let out = launchbound(&["prune", "corpus/reduce-flip", "--cc", "8.x"]); + assert_eq!(code(&out), 2, "stderr was:\n{}", stderr(&out)); + let err = stderr(&out); + assert!( + err.contains("compute capability"), + "the message must say what a --cc is:\n{err}" + ); + assert!( + err.contains("sm_86"), + "and name the spellings that would have worked:\n{err}" + ); +} + +#[test] +fn help_and_version_succeed() { + for args in [vec!["--help"], vec!["--version"], vec!["report", "--help"]] { + let out = launchbound(&args); + assert_eq!(code(&out), 0, "{args:?} failed:\n{}", stderr(&out)); + assert!(!out.stdout.is_empty(), "{args:?} printed nothing"); + } +} + +#[test] +fn enumerating_a_corpus_kernel_needs_no_gpu_and_no_analyzer() { + // `space` is the one command with no gate and no device behind it, so it + // is the cheapest end-to-end check that the corpus still parses. + let out = launchbound(&["space", "corpus/reduce-flip"]); + assert_eq!(code(&out), 0, "stderr was:\n{}", stderr(&out)); + assert!( + !out.stdout.is_empty(), + "the space is the output; it must not be empty" + ); +} diff --git a/crates/launchbound-metal/src/lib.rs b/crates/launchbound-metal/src/lib.rs index a1ab857..49740da 100644 --- a/crates/launchbound-metal/src/lib.rs +++ b/crates/launchbound-metal/src/lib.rs @@ -330,6 +330,114 @@ mod tests { use super::*; use launchbound_space::{KernelSpec, enumerate}; + /// A spec with one block dimension and one spec dimension. + fn spec(extra: &str) -> KernelSpec { + KernelSpec::from_toml_str( + "t", + &format!( + r#" + [kernel] + name = "t" + entry = "t" + domain = 1 + [dims.block_x] + values = [64] + [dims.tile] + role = "spec" + values = [512] + {extra} + "# + ), + ) + .unwrap() + } + + /// The whole Metal path is one function, `run_metal`, and until this + /// module grew nothing called it outside the CLI. It cannot be *run* + /// here — it needs a device, and these tests are the part that does not + /// — but the decisions it makes before touching one can be. + #[test] + fn the_notice_is_in_the_text_and_not_only_in_a_comment() { + // `docs/LIMITATIONS.md` promises this appears on every Metal + // surface. A constant nothing asserts is a promise nothing keeps. + assert!(METAL_NO_GATE_NOTICE.contains("NO convergence gate")); + assert!( + METAL_NO_GATE_NOTICE.contains("NOT checked"), + "the notice must say the bug class is not checked, not merely that a gate is absent" + ); + } + + #[test] + fn off_macos_the_backend_refuses_rather_than_returning_nothing() { + // The shape that matters: a Linux caller asking for Metal gets an + // error naming the platform, not an empty `Results` that reads like + // a sweep in which nothing happened to be measured. + #[cfg(not(target_os = "macos"))] + { + let spec = spec(""); + let err = run_metal(&spec, &[], None, &mut |_| {}) + .expect_err("Metal off macOS must fail loudly"); + assert!( + matches!(err, MetalError::Backend(ref m) if m.contains("macOS")), + "the message must name the platform: {err}" + ); + } + // On macOS the same call reaches the runtime, where "no device" is + // the honest failure; either way it is an error and never `Ok`. + #[cfg(target_os = "macos")] + { + let spec = spec(""); + assert!( + run_metal(&spec, &[], None, &mut |_| {}).is_err(), + "a kernel with no kernel.metal cannot be measured" + ); + } + } + + #[test] + fn a_dimension_with_no_msl_twin_is_skipped_rather_than_failing() { + // `lb_max` is a CUDA launch-bound and has no Metal meaning, so the + // MSL source will not declare it. Skipping is deliberate; erroring + // would make every kernel with a CUDA-only dimension untunable here. + let spec = spec("[dims.lb_max]\nrole = \"spec\"\nvalues = [256]"); + let config = enumerate(&spec).unwrap().remove(0); + let src = "constant constexpr uint TILE = 128;\nkernel void t() {}\n"; + let out = specialize_msl(src, &spec, &config).unwrap(); + assert!(out.contains("constant constexpr uint TILE = 512;")); + assert!( + !out.contains("LB_MAX"), + "a dimension the MSL does not declare must not be invented" + ); + } + + #[test] + fn an_unterminated_constant_is_named_rather_than_silently_truncating() { + // Without the `;` there is no end to replace up to. Rewriting to the + // end of the file would produce MSL that compiles into something + // else; the error names the constant instead. + let spec = spec(""); + let config = enumerate(&spec).unwrap().remove(0); + let src = "constant constexpr uint TILE = 128"; + let err = specialize_msl(src, &spec, &config).expect_err("must not truncate"); + assert!( + matches!(err, MetalError::Source(ref m) if m.contains("TILE")), + "the message must name the constant: {err}" + ); + } + + #[test] + fn a_block_dimension_is_not_rewritten_into_the_source() { + // Block size is a dispatch argument on Metal, not a compile-time + // constant, and the existing test below says "spec dims only" — + // this is the other half of that sentence, stated where it can fail. + let spec = spec(""); + let config = enumerate(&spec).unwrap().remove(0); + let src = "constant constexpr uint BLOCK_X = 1;\nconstant constexpr uint TILE = 128;\n"; + let out = specialize_msl(src, &spec, &config).unwrap(); + assert!(out.contains("constant constexpr uint BLOCK_X = 1;")); + assert!(out.contains("constant constexpr uint TILE = 512;")); + } + #[test] fn msl_specialization_rewrites_spec_dims_only() { let spec = KernelSpec::from_toml_str( diff --git a/crates/launchbound-model/src/lib.rs b/crates/launchbound-model/src/lib.rs index 7d36bbd..befec69 100644 --- a/crates/launchbound-model/src/lib.rs +++ b/crates/launchbound-model/src/lib.rs @@ -376,6 +376,84 @@ fn ranks(values: &[f64]) -> Vec { mod tests { use super::*; + /// The ranking the model produces for a real corpus kernel, pinned. + /// + /// `--backend model` is what runs when there is no GPU, and its output is + /// an *ordering* a user acts on. Every other test here checks a piece of + /// the cost function; none checked that the pieces compose into the same + /// order twice. A change to `cost` that looks like a refinement and + /// silently reverses two candidates would pass all of them. + /// + /// Pinned as a relation, not as numbers: the cost scale is the model's + /// own business and may be rescaled, but "more occupancy at the same + /// wave count ranks better" is the claim. + #[test] + fn the_ranking_is_stable_and_ordered_by_cost() { + use launchbound_space::enumerate; + + // The real corpus kernel, loaded from disk, rather than a spec + // written here: `smem_bytes` resolves `[model]` against the kernel's + // own directory, and a pin against a synthetic space would not + // notice the corpus changing under it. + let dir = std::path::Path::new(env!("CARGO_MANIFEST_DIR")).join("../../corpus/reduce-flip"); + let spec = launchbound_space::KernelSpec::load(&dir).expect("the corpus kernel"); + let dev = device("8.6").expect("the A10G row"); + + let rank = |()| -> Vec { + let mut est: Vec<_> = enumerate(&spec) + .expect("a space") + .iter() + .map(|c| estimate(&spec, c, &dev).expect("an estimate")) + .collect(); + est.sort_by(|a, b| { + a.cost + .total_cmp(&b.cost) + .then_with(|| a.config.cmp(&b.config)) + }); + est.into_iter().map(|e| e.config).collect() + }; + + let first = rank(()); + assert!( + first.len() >= 8, + "the corpus kernel has a space to rank: {first:?}" + ); + assert_eq!( + first, + rank(()), + "the same space must rank the same way twice" + ); + + // The ordering claim, checked against the costs it came from rather + // than against a copied list: sorted by cost means non-decreasing. + let mut costs: Vec = enumerate(&spec) + .expect("a space") + .iter() + .map(|c| estimate(&spec, c, &dev).expect("an estimate").cost) + .collect(); + costs.sort_by(f64::total_cmp); + assert!( + costs.windows(2).all(|w| w[0] <= w[1]), + "costs must be totally ordered and comparable" + ); + assert!( + costs.iter().all(|c| c.is_finite()), + "no candidate in this space is unlaunchable at cc 8.6, so none may cost infinity" + ); + } + + /// An unknown capability is an error that lists the known ones, never a + /// guess. A fabricated capacity still produces an occupancy number, and + /// an occupancy number is the sort of thing a reader believes. + #[test] + fn an_unknown_capability_names_the_ones_that_exist() { + let err = device("6.1").expect_err("Pascal is not in the table"); + let text = err.to_string(); + for known in ["7.5", "8.6", "9.0"] { + assert!(text.contains(known), "the error must list {known}: {text}"); + } + } + #[test] fn spearman_perfect_and_inverse_and_ties() { assert_eq!(