From 5666172159d45cb869f5ce16622ef1a8df06f965 Mon Sep 17 00:00:00 2001 From: utarafdar Date: Tue, 14 Jul 2026 23:27:57 +0530 Subject: [PATCH] Add --count N to radius test: repeat checks and report aggregate flakiness stats --- README.md | 49 ++++ cmd/authhound-probe/main.go | 71 ++++- cmd/authhound-probe/main_test.go | 20 ++ docs/json-schema.md | 43 ++- internal/check/common.go | 14 + internal/check/eaptls.go | 7 +- internal/check/eaptls_auth.go | 7 +- internal/check/pap.go | 2 +- internal/check/peap.go | 7 +- internal/check/reachability.go | 1 + internal/check/repeat.go | 256 ++++++++++++++++++ internal/check/repeat_test.go | 177 ++++++++++++ internal/check/secret.go | 4 +- internal/check/ttls.go | 7 +- internal/report/json.go | 3 + internal/report/repeat.go | 108 ++++++++ internal/report/repeat_test.go | 104 +++++++ .../testdata/schema-v1-repeat.golden.json | 76 ++++++ test/README.md | 10 + test/freeradius-smoke.sh | 64 ++++- test/lab/config/clients-flaky | 7 + test/lab/docker-compose.yml | 36 +++ 22 files changed, 1062 insertions(+), 11 deletions(-) create mode 100644 internal/check/repeat.go create mode 100644 internal/check/repeat_test.go create mode 100644 internal/report/repeat.go create mode 100644 internal/report/repeat_test.go create mode 100644 internal/report/testdata/schema-v1-repeat.golden.json create mode 100644 test/lab/config/clients-flaky diff --git a/README.md b/README.md index 78d561b..6c582c2 100644 --- a/README.md +++ b/README.md @@ -189,6 +189,8 @@ $ authhound-probe radsec test --server radius.corp.com \ | `--password-file FILE` | Password for a `user`-only `--pap/--peap/--ttls`, from a file (non-interactive). | | `--client-cert FILE` `--client-key FILE` | Run an EAP-TLS test with this client certificate + key (PEM). | | `--mtu` | Run the path-MTU / fragmentation probe (sends a few padded packets). | +| `--count N` | Run the checks `N` times (2–50) and report aggregate statistics — see [Chasing intermittent failures](#chasing-intermittent-failures). | +| `--interval DURATION` | Pause between `--count` iterations (default `2s`; a hard-coded safety floor applies). | | `--nas-port-type wireless\|ethernet\|virtual` | How the probe presents itself, so server policies match (default `wireless`). | | `--server-name NAME` | Expected server-certificate name (TLS SNI). | | `--nas-id NAME` | NAS-Identifier to send (default `authhound-probe`). | @@ -197,6 +199,53 @@ $ authhound-probe radsec test --server radius.corp.com \ | `--strict` | Exit non-zero on **warnings** too (e.g. a soon-to-expire cert), for scheduled monitoring. | | `--no-color` | Force plain output. Colour is auto-detected otherwise — see [Colour](#colour-and-windows-terminals). | +### Chasing intermittent failures + +A single-shot PASS proves nothing about the failure that hits one user in ten. +When the complaint is "Wi-Fi drops people randomly" or "auth works, except when +it doesn't", run the same checks repeatedly and look at the distribution: + +```console +$ export AUTHHOUND_SECRET='shared-secret' +$ authhound-probe radius test --server radius.corp.com --peap alice --count 10 + +Running 10 iterations, 2s apart + +run 1/10 reachability PASS 3ms · shared-secret PASS · peap-mschapv2 PASS +run 2/10 reachability PASS 41ms · shared-secret PASS · peap-mschapv2 LOST (no reply) +run 3/10 reachability PASS 2ms · shared-secret PASS · peap-mschapv2 PASS +... + +Aggregate over 10 runs: + +PASS reachability: 10/10 succeeded — stable + Latency over 10 answered runs: min 2ms, median 3ms, p95 41ms, max 41ms. +PASS shared-secret: 10/10 succeeded — stable +FAIL peap-mschapv2: 8/10 succeeded, 2/10 requests lost — consistent with an + unstable path or an overloaded/failing server, not a config error + Latency over 8 answered runs: min 9ms, median 12ms, p95 96ms, max 96ms. +``` + +How to read it: + +- **Requests lost** (timeouts) with the rest succeeding → the configuration is + fine; suspect the network path or an overloaded/failing server. A p95 far + above the median is the same story told by latency. +- **Failed every run** → not flaky at all; it's a configuration problem + (secret, credentials, policy) that a single run would also have caught. +- Exit code stays `1` if *any* iteration failed, so a flaky server fails a + scripted run loudly instead of depending on which iteration you got. + +Iterations run sequentially, `--interval` apart (default `2s`). The probe's +hard-coded rate ceiling still bounds everything: intervals below the safety +floor are stretched (and the stretch announced), and `--count` is capped at 50. +This is a diagnosis loop you babysit, not monitoring — it never schedules, +repeats forever, or stores anything between runs. + +With `--json`, each iteration's results plus per-check aggregate statistics +(success counts, timeouts, latency min/median/p95/max) appear in an additive +`repeat` block — see the [schema](docs/json-schema.md). + ### Where credentials come from This tool is meant to run on shared jump boxes, so it never *requires* a secret or diff --git a/cmd/authhound-probe/main.go b/cmd/authhound-probe/main.go index 68799f0..c990468 100644 --- a/cmd/authhound-probe/main.go +++ b/cmd/authhound-probe/main.go @@ -16,6 +16,7 @@ import ( "flag" "fmt" "os" + "os/signal" "runtime/debug" "strings" "time" @@ -186,6 +187,8 @@ func cmdRadiusTest(args []string) int { nasPortType := fs.String("nas-port-type", "wireless", "NAS-Port-Type: wireless, ethernet, or virtual") serverName := fs.String("server-name", "", "expected server certificate name (TLS SNI); optional") mtu := fs.Bool("mtu", false, "run the path-MTU / fragmentation probe (sends a few padded packets)") + count := fs.Int("count", 1, "run the checks N times (2..50) and report aggregate statistics — for chasing intermittent failures") + interval := fs.Duration("interval", 2*time.Second, "pause between --count iterations (a hard-coded safety floor applies)") timeout := fs.Duration("timeout", 5*time.Second, "per-request timeout") jsonOut := fs.Bool("json", false, "emit results as JSON instead of text") noColor := fs.Bool("no-color", false, "disable ANSI colour") @@ -216,6 +219,18 @@ func cmdRadiusTest(args []string) int { return 2 } + // --count 1 (the default) is exactly the classic single run; repeat mode is + // hard-capped at RepeatCountMax — this is a diagnosis loop a human watches, + // not a monitor (that's `connect`). + if *count != 1 && (*count < check.RepeatCountMin || *count > check.RepeatCountMax) { + fmt.Fprintf(os.Stderr, "error: --count must be between %d and %d\n", check.RepeatCountMin, check.RepeatCountMax) + return 2 + } + if provided["interval"] && *count == 1 { + fmt.Fprintln(os.Stderr, "error: --interval only makes sense with --count") + return 2 + } + prompter := credential.Default() secretValue, err := prompter.Resolve(credential.Spec{ Name: "shared secret", @@ -269,7 +284,6 @@ func cmdRadiusTest(args []string) int { sink = report.NewTextSink(os.Stdout, report.UseColor(os.Stdout, *noColor)) } - runner := check.Runner{Sink: sink} plan := check.Plan{ Target: target, Checks: []check.Check{ @@ -283,12 +297,67 @@ func cmdRadiusTest(args []string) int { check.MTUProbe{Enabled: *mtu}, }, } + + if *count != 1 { + return runRepeat(plan, sink, *count, *interval, *strict, *jsonOut) + } + + runner := check.Runner{Sink: sink} runner.Run(context.Background(), plan) _ = sink.Close() return exitCode(sink, *strict) } +// runRepeat drives --count: N sequential iterations, then the aggregate +// verdicts through the normal sink, so text/JSON rendering and the exit-code +// contract are identical to a single run. Ctrl-C mid-loop aggregates the +// iterations that completed instead of throwing them away. +func runRepeat(plan check.Plan, sink resultSink, count int, interval time.Duration, strict, jsonOut bool) int { + ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt) + defer stop() + + effective, stretched := check.EffectiveInterval(interval) + if stretched { + // Stderr in both modes: with --json, stdout stays pure JSON (the + // document carries interval_stretched for scripts). + fmt.Fprintf(os.Stderr, "note: --interval %s is below the probe's hard-coded safety floor; running %s apart instead\n", interval, effective) + } + opts := check.RepeatOptions{Count: count, Interval: interval} + if !jsonOut { + fmt.Printf("Running %d iterations, %s apart\n\n", count, effective) + opts.OnIteration = func(i int, results []check.Result) { + fmt.Println(report.IterationLine(i, count, results)) + } + } + + runner := check.Runner{} + run, err := check.RunRepeated(ctx, &runner, plan, opts) + if err != nil { + fmt.Fprintln(os.Stderr, "error:", err) + return 2 + } + if completed := len(run.Iterations); completed < count { + fmt.Fprintf(os.Stderr, "interrupted — aggregating the %d completed iteration(s)\n", completed) + if completed == 0 { + return 1 + } + } + + stats := check.AggregateRepeat(run) + if !jsonOut { + fmt.Printf("\nAggregate over %d runs:\n\n", len(run.Iterations)) + } + for _, s := range stats { + sink.Emit(s.Verdict()) + } + if js, ok := sink.(*report.JSONSink); ok { + js.SetRepeat(count, run, stats) + } + _ = sink.Close() + return exitCode(sink, strict) +} + // resultSink is the report-sink surface both subcommands use: the check.ResultSink // contract plus the tallies that drive the process exit code. type resultSink interface { diff --git a/cmd/authhound-probe/main_test.go b/cmd/authhound-probe/main_test.go index be9df34..b18c82f 100644 --- a/cmd/authhound-probe/main_test.go +++ b/cmd/authhound-probe/main_test.go @@ -38,6 +38,26 @@ func TestExitCode(t *testing.T) { } } +// TestCountFlagValidation pins the --count contract: out-of-range counts and a +// stray --interval are usage errors (exit 2) caught before any credential +// prompting or network I/O. +func TestCountFlagValidation(t *testing.T) { + cases := []struct { + name string + args []string + }{ + {"count too high", []string{"--server", "192.0.2.1", "--count", "51"}}, + {"count zero", []string{"--server", "192.0.2.1", "--count", "0"}}, + {"count negative", []string{"--server", "192.0.2.1", "--count", "-3"}}, + {"interval without count", []string{"--server", "192.0.2.1", "--interval", "5s"}}, + } + for _, c := range cases { + if got := cmdRadiusTest(c.args); got != 2 { + t.Errorf("%s: exit = %d, want 2", c.name, got) + } + } +} + func TestResolveVersion(t *testing.T) { orig := version t.Cleanup(func() { version = orig }) diff --git a/docs/json-schema.md b/docs/json-schema.md index b01b71b..8c394d8 100644 --- a/docs/json-schema.md +++ b/docs/json-schema.md @@ -42,8 +42,9 @@ is a deliberate, reviewed act rather than an accident. | Field | Type | Presence | Notes | |---|---|---|---| | `schema_version` | string | always | Major version of this document shape. Currently `"1"`. | -| `results` | array of result objects | always | One entry per check, in the order they ran. | +| `results` | array of result objects | always | One entry per check, in the order they ran. With `--count`, one **aggregate verdict** per check (see the `repeat` block below). | | `summary` | object | always | Per-status tally. **All five keys are always present**, including zeros — address `.summary.warn` without checking it exists first. | +| `repeat` | object | only with `--count` | Additive: per-iteration results and aggregate statistics for a repeated run. Absent on single runs, whose documents are unchanged. | ### `summary` object @@ -66,9 +67,44 @@ Each entry in `results`: | `summary` | string | always | One plain-English line describing the outcome. | | `detail` | string | when present | Extra context. Omitted when empty. | | `hint` | string | when present | Multi-line, paste-ready remediation. Newline formatting is significant. Omitted when empty. Never contains secrets. | -| `fields` | object (string→string) | when present | Structured extras such as `rtt_ms`, `tls_version`, `not_after`, `subject`, `san`, `chain_len`, `source_ip`. Keys vary by check; values are always strings. Omitted when there are none. | +| `fields` | object (string→string) | when present | Structured extras such as `rtt_ms`, `tls_version`, `not_after`, `subject`, `san`, `chain_len`, `source_ip`. `timeout: "true"` marks a request that got no reply at all (a *lost* request, as opposed to a processed rejection). Aggregate verdicts under `--count` add `success_rate`, `attempts`, `successes`, `timeouts`, and `latency_{min,median,p95,max}_ms`. Keys vary by check; values are always strings. Omitted when there are none. | | `duration_ns` | integer | when present | How long the check took, in nanoseconds. Omitted when zero. | +## `repeat` block (`--count`) + +`radius test --count N` runs the checks N times; the document then carries one +**aggregate verdict per check** in `results` (so `summary` and the exit code +reflect the whole run — any failed iteration fails the run), and this block with +the raw material: + +```json +"repeat": { + "count": 10, + "completed": 10, + "interval_ms": 2000, + "requested_interval_ms": 2000, + "interval_stretched": false, + "iterations": [ { "results": [ /* result objects, one per check */ ] } ], + "aggregate": [ + { + "check": "peap-mschapv2", + "attempts": 10, "successes": 8, "failures": 2, "timeouts": 2, "skipped": 0, + "latency_ms": { "min": 9, "median": 12, "p95": 96, "max": 96 } + } + ] +} +``` + +| Field | Type | Notes | +|---|---|---| +| `count` | integer | Requested iterations (2–50). | +| `completed` | integer | Iterations that actually finished (lower than `count` after Ctrl-C). | +| `interval_ms` | integer | Pause between iterations actually used. | +| `requested_interval_ms` | integer | The `--interval` that was asked for. | +| `interval_stretched` | boolean | `true` when the requested interval was below the hard-coded safety floor and got stretched. | +| `iterations` | array | One entry per completed iteration, each with its `results` (same result-object shape as the top level). | +| `aggregate` | array | Per-check tallies. `successes` = the server answered and processed the request (pass/warn/info); `timeouts` = the subset of `failures` where no reply arrived at all. `latency_ms` (nearest-rank percentiles over answered runs) is omitted when nothing was answered. | + ### `status` values | Value | Meaning | Effect on exit code | @@ -96,6 +132,9 @@ authhound-probe radius test --server r --json | jq '.summary.fail' # Pin to the schema you coded against: ... --json | jq -e '.schema_version=="1"' >/dev/null || echo "schema changed" + +# --count: which checks lost requests, from the aggregate block: +... --count 10 --json | jq -r '.repeat.aggregate[] | select(.timeouts > 0) | "\(.check): \(.timeouts) lost"' ``` ## Exit codes diff --git a/internal/check/common.go b/internal/check/common.go index 2838c8c..5c174dc 100644 --- a/internal/check/common.go +++ b/internal/check/common.go @@ -33,3 +33,17 @@ func addCommon(p *radius.Packet, t Target) { p.Add(a.Type, a.Value) } } + +// TimeoutField marks a Result whose underlying request got no reply, so +// aggregate reporting (--count) can count lost requests separately from +// processed rejections. Additive within schema major "1". +const TimeoutField = "timeout" + +// markTimeout tags r as a timeout result (see TimeoutField). +func markTimeout(r Result) Result { + if r.Fields == nil { + r.Fields = map[string]string{} + } + r.Fields[TimeoutField] = "true" + return r +} diff --git a/internal/check/eaptls.go b/internal/check/eaptls.go index d3b7ddc..7c37cba 100644 --- a/internal/check/eaptls.go +++ b/internal/check/eaptls.go @@ -2,6 +2,7 @@ package check import ( "context" + "errors" "github.com/authhound/probe/internal/radius" ) @@ -31,13 +32,17 @@ func (c ServerCert) Run(ctx context.Context, t Target) Result { captured, err := sess.InspectServerCert(ctx, c.ServerName) if err != nil { - return Result{ + r := Result{ Check: "server-cert", Status: StatusSkip, Summary: "Could not inspect the server certificate", Detail: "The PEAP/TLS handshake didn't get far enough to read the certificate: " + err.Error() + ". This is expected if the server doesn't offer PEAP or EAP-TLS, " + "or if reachability/secret checks above failed.", } + if errors.Is(err, radius.ErrTimeout) { + r = markTimeout(r) + } + return r } return analyzeCert("server-cert", captured.Chain, captured.TLSVersion) } diff --git a/internal/check/eaptls_auth.go b/internal/check/eaptls_auth.go index d1fed8b..517bbd2 100644 --- a/internal/check/eaptls_auth.go +++ b/internal/check/eaptls_auth.go @@ -4,6 +4,7 @@ import ( "context" "crypto/tls" "crypto/x509" + "errors" "fmt" "github.com/authhound/probe/internal/radius" @@ -64,7 +65,11 @@ func (c EAPTLS) Run(ctx context.Context, t Target) Result { res, err := sess.AuthEAPTLS(ctx, cert, c.ServerName) if err != nil { - return Result{Check: "eap-tls", Status: StatusFail, Summary: "EAP-TLS exchange failed: " + err.Error()} + r := Result{Check: "eap-tls", Status: StatusFail, Summary: "EAP-TLS exchange failed: " + err.Error()} + if errors.Is(err, radius.ErrTimeout) { + r = markTimeout(r) + } + return r } fields := map[string]string{"identity": identity} diff --git a/internal/check/pap.go b/internal/check/pap.go index a047324..b3e8d71 100644 --- a/internal/check/pap.go +++ b/internal/check/pap.go @@ -41,7 +41,7 @@ func (c PAP) Run(ctx context.Context, t Target) Result { reply, _, _, err := radius.Exchange(t.Address, t.Secret, p, t.Timeout) if err != nil { if errors.Is(err, radius.ErrTimeout) { - return Result{Check: "pap-auth", Status: StatusSkip, Summary: "No reply — resolve reachability first"} + return markTimeout(Result{Check: "pap-auth", Status: StatusSkip, Summary: "No reply — resolve reachability first"}) } return Result{Check: "pap-auth", Status: StatusFail, Summary: "PAP exchange failed: " + err.Error()} } diff --git a/internal/check/peap.go b/internal/check/peap.go index 95b9424..522421f 100644 --- a/internal/check/peap.go +++ b/internal/check/peap.go @@ -2,6 +2,7 @@ package check import ( "context" + "errors" "fmt" "strconv" @@ -41,13 +42,17 @@ func (c PEAPMSCHAPv2) Run(ctx context.Context, t Target) Result { res, err := sess.AuthPEAPMSCHAPv2(ctx, c.User, c.Pass, c.ServerName) if err != nil { - return Result{ + r := Result{ Check: "peap-mschapv2", Status: StatusFail, Summary: "PEAP-MSCHAPv2 exchange did not complete", Detail: "The tunnel or inner exchange broke before a verdict: " + err.Error() + ". If reachability/secret above failed, fix those first; otherwise the server " + "may not offer PEAP-MSCHAPv2.", } + if errors.Is(err, radius.ErrTimeout) { + r = markTimeout(r) + } + return r } fields := map[string]string{} diff --git a/internal/check/reachability.go b/internal/check/reachability.go index 30b00e1..0cb4763 100644 --- a/internal/check/reachability.go +++ b/internal/check/reachability.go @@ -37,6 +37,7 @@ func (Reachability) Run(ctx context.Context, t Target) Result { } } if errors.Is(err, radius.ErrTimeout) { + fields[TimeoutField] = "true" srcIP := "" var te *radius.TimeoutError if errors.As(err, &te) && te.LocalIP != "" { diff --git a/internal/check/repeat.go b/internal/check/repeat.go new file mode 100644 index 0000000..4904dbb --- /dev/null +++ b/internal/check/repeat.go @@ -0,0 +1,256 @@ +package check + +import ( + "context" + "fmt" + "math" + "sort" + "strconv" + "strings" + "time" +) + +// Repeat mode (`radius test --count N`) runs the same plan N times to expose +// intermittent failures a single-shot PASS can't see. It is a diagnosis loop a +// human babysits: strictly sequential, hard-capped at RepeatCountMax runs, no +// scheduling, no persistence — continuous monitoring stays out of this binary. + +const ( + RepeatCountMin = 2 + RepeatCountMax = 50 + + // repeatIntervalFloor is the hard minimum pause between iterations. It is + // derived from the runner's per-exchange ceiling (minInterval) so the two + // can never drift apart: the runner already enforces minInterval between + // exchanges inside an iteration, so any between-iteration pause >= minInterval + // keeps the sustained request rate under the ceiling; 4x adds margin. Like + // minInterval itself, this is deliberately not configurable by flag, env, + // or config. + repeatIntervalFloor = 4 * minInterval +) + +// RepeatOptions configures a repeated run. +type RepeatOptions struct { + Count int // iterations, RepeatCountMin..RepeatCountMax + Interval time.Duration // requested pause between iterations; floored to the safety minimum + // OnIteration, if set, is called after each completed iteration with its + // 1-based index and results — progress reporting for a human watching. + OnIteration func(iteration int, results []Result) +} + +// RepeatRun is the outcome of a repeated run. Iterations holds only fully +// completed iterations; a run cut short by ctx cancellation aggregates what +// finished. +type RepeatRun struct { + Iterations [][]Result + RequestedInterval time.Duration + Interval time.Duration // interval actually used (>= safety floor) + Stretched bool // requested interval was below the floor +} + +// EffectiveInterval returns the between-iteration interval a repeat run will +// actually use, and whether the requested one had to be stretched to satisfy +// the safety floor. +func EffectiveInterval(requested time.Duration) (effective time.Duration, stretched bool) { + if requested < repeatIntervalFloor { + return repeatIntervalFloor, true + } + return requested, false +} + +// RunRepeated executes plan Count times, pausing the effective interval between +// iterations. The Runner's own per-exchange ceiling still applies inside every +// iteration; the floor here only governs the gap between iterations. +func RunRepeated(ctx context.Context, r *Runner, plan Plan, opts RepeatOptions) (RepeatRun, error) { + if opts.Count < RepeatCountMin || opts.Count > RepeatCountMax { + return RepeatRun{}, fmt.Errorf("count must be between %d and %d", RepeatCountMin, RepeatCountMax) + } + interval, stretched := EffectiveInterval(opts.Interval) + run := RepeatRun{RequestedInterval: opts.Interval, Interval: interval, Stretched: stretched} + for i := 0; i < opts.Count; i++ { + if i > 0 { + select { + case <-time.After(interval): + case <-ctx.Done(): + return run, nil + } + } + results := r.Run(ctx, plan) + if ctx.Err() != nil && len(results) < len(plan.Checks) { + // Cancelled mid-iteration: a partial iteration would skew the + // per-check tallies, so drop it and report what completed. + return run, nil + } + run.Iterations = append(run.Iterations, results) + if opts.OnIteration != nil { + opts.OnIteration(i+1, results) + } + } + return run, nil +} + +// CheckStats aggregates one check's results across all iterations. +// +// A "success" is any iteration the server answered and processed (pass, warn, +// or info). A timeout is a lost request — no reply at all — counted separately +// from processed rejections because they point at different culprits: loss +// means path/overload, a consistent reject means configuration. +type CheckStats struct { + Check string + Attempts int // iterations where the check actually sent (not skipped) + Successes int // pass / warn / info + Failures int // fail or lost + Timeouts int // subset of Failures: no reply at all + Skipped int + Warns int + + // Latency over answered iterations (nearest-rank percentiles). Zero when + // nothing was answered. + LatMin, LatMedian, LatP95, LatMax time.Duration +} + +// Iterations returns how many iterations this check appeared in. +func (s CheckStats) Iterations() int { return s.Attempts + s.Skipped } + +// AggregateRepeat computes per-check statistics across the run's iterations, +// in the plan's check order. +func AggregateRepeat(run RepeatRun) []CheckStats { + if len(run.Iterations) == 0 { + return nil + } + byCheck := map[string]*CheckStats{} + var order []string + for _, r := range run.Iterations[0] { + byCheck[r.Check] = &CheckStats{Check: r.Check} + order = append(order, r.Check) + } + latencies := map[string][]time.Duration{} + for _, iter := range run.Iterations { + for _, r := range iter { + s, ok := byCheck[r.Check] + if !ok { // defensive: a check name not in iteration 1 + s = &CheckStats{Check: r.Check} + byCheck[r.Check] = s + order = append(order, r.Check) + } + switch { + case r.Fields[TimeoutField] == "true": + // A lost request is a failed attempt regardless of the status + // the check chose for single-run readability (pap reports + // timeout as SKIP to say "fix reachability first"). + s.Attempts++ + s.Failures++ + s.Timeouts++ + case r.Status == StatusSkip: + s.Skipped++ + case r.Status == StatusFail: + s.Attempts++ + s.Failures++ + default: // pass / warn / info: the server answered and processed it + s.Attempts++ + s.Successes++ + if r.Status == StatusWarn { + s.Warns++ + } + if r.Duration > 0 { + latencies[r.Check] = append(latencies[r.Check], r.Duration) + } + } + } + } + var out []CheckStats + for _, name := range order { + s := byCheck[name] + if lat := latencies[name]; len(lat) > 0 { + sort.Slice(lat, func(i, j int) bool { return lat[i] < lat[j] }) + s.LatMin = lat[0] + s.LatMax = lat[len(lat)-1] + s.LatMedian = percentile(lat, 0.5) + s.LatP95 = percentile(lat, 0.95) + } + out = append(out, *s) + } + return out +} + +// percentile is the nearest-rank percentile of an ascending-sorted slice. +func percentile(sorted []time.Duration, p float64) time.Duration { + if len(sorted) == 0 { + return 0 + } + rank := int(math.Ceil(p * float64(len(sorted)))) + if rank < 1 { + rank = 1 + } + if rank > len(sorted) { + rank = len(sorted) + } + return sorted[rank-1] +} + +// Verdict renders the aggregate as one Result per check, so repeat mode flows +// through the same sinks, tallies, and exit-code logic as a single run. The +// summary names flakiness explicitly: partial loss reads as a path/server +// problem, a 0% success rate as a configuration problem. +func (s CheckStats) Verdict() Result { + n := s.Iterations() + r := Result{Check: s.Check, Fields: map[string]string{ + "attempts": strconv.Itoa(s.Attempts), + "successes": strconv.Itoa(s.Successes), + "timeouts": strconv.Itoa(s.Timeouts), + }} + if s.Attempts > 0 { + r.Fields["success_rate"] = fmt.Sprintf("%d/%d", s.Successes, s.Attempts) + } + if s.Successes > 0 && s.LatMax > 0 { + r.Fields["latency_min_ms"] = ms(s.LatMin) + r.Fields["latency_median_ms"] = ms(s.LatMedian) + r.Fields["latency_p95_ms"] = ms(s.LatP95) + r.Fields["latency_max_ms"] = ms(s.LatMax) + r.Detail = fmt.Sprintf("Latency over %d answered runs: min %sms, median %sms, p95 %sms, max %sms.", + s.Successes, ms(s.LatMin), ms(s.LatMedian), ms(s.LatP95), ms(s.LatMax)) + } + + switch { + case s.Attempts == 0: + r.Status = StatusSkip + r.Summary = fmt.Sprintf("%s: skipped in all %d runs", s.Check, n) + case s.Failures == 0: + r.Status = StatusPass + if s.Warns > 0 { + r.Status = StatusWarn + } + r.Summary = fmt.Sprintf("%s: %d/%d succeeded — stable", s.Check, s.Successes, s.Attempts) + if s.Warns > 0 { + r.Summary += fmt.Sprintf(" (%d with warnings)", s.Warns) + } + case s.Successes == 0 && s.Timeouts == s.Attempts: + r.Status = StatusFail + r.Summary = fmt.Sprintf("%s: all %d requests lost — nothing answered; the server is down or unreachable, this probe isn't registered as a client, or the secret is wrong (silent drop) — not intermittent", + s.Check, s.Attempts) + case s.Successes == 0: + r.Status = StatusFail + r.Summary = fmt.Sprintf("%s: 0/%d succeeded — failed every run; a consistent failure points to configuration (secret, credentials, policy), not flakiness", + s.Check, s.Attempts) + default: + r.Status = StatusFail + r.Summary = fmt.Sprintf("%s: %d/%d succeeded, %s — consistent with an unstable path or an overloaded/failing server, not a config error", + s.Check, s.Successes, s.Attempts, lossPhrase(s)) + } + return r +} + +// lossPhrase describes the failed fraction of a flaky check: lost requests +// (timeouts) and processed-but-failed runs are named separately. +func lossPhrase(s CheckStats) string { + var parts []string + if s.Timeouts > 0 { + parts = append(parts, fmt.Sprintf("%d/%d requests lost", s.Timeouts, s.Attempts)) + } + if other := s.Failures - s.Timeouts; other > 0 { + parts = append(parts, fmt.Sprintf("%d/%d failed after an answer", other, s.Attempts)) + } + return strings.Join(parts, " and ") +} + +func ms(d time.Duration) string { return strconv.FormatInt(d.Milliseconds(), 10) } diff --git a/internal/check/repeat_test.go b/internal/check/repeat_test.go new file mode 100644 index 0000000..9d6197a --- /dev/null +++ b/internal/check/repeat_test.go @@ -0,0 +1,177 @@ +package check + +import ( + "context" + "strings" + "testing" + "time" +) + +// stubCheck returns a scripted sequence of results, one per iteration, so +// repeat behavior can be tested without a network. +type stubCheck struct { + name string + results []Result + calls int +} + +func (c *stubCheck) Name() string { return c.name } +func (c *stubCheck) Run(ctx context.Context, t Target) Result { + r := c.results[c.calls%len(c.results)] + c.calls++ + r.Check = c.name + return r +} + +func TestEffectiveIntervalStretchesToFloor(t *testing.T) { + // The floor is derived from the hard-coded rate ceiling and must win over + // any smaller request — this is the "cannot be bypassed" guarantee. + eff, stretched := EffectiveInterval(50 * time.Millisecond) + if !stretched || eff != repeatIntervalFloor { + t.Errorf("50ms: got (%s, %v), want (%s, true)", eff, stretched, repeatIntervalFloor) + } + eff, stretched = EffectiveInterval(0) + if !stretched || eff != repeatIntervalFloor { + t.Errorf("0: got (%s, %v), want (%s, true)", eff, stretched, repeatIntervalFloor) + } + eff, stretched = EffectiveInterval(2 * time.Second) + if stretched || eff != 2*time.Second { + t.Errorf("2s: got (%s, %v), want (2s, false)", eff, stretched) + } + if repeatIntervalFloor < minInterval { + t.Errorf("floor %s below the per-exchange ceiling %s", repeatIntervalFloor, minInterval) + } +} + +func TestRunRepeatedEnforcesFloor(t *testing.T) { + // Two iterations with a requested interval far below the floor must still + // be separated by at least the floor. + ok := Result{Status: StatusPass, Summary: "ok"} + plan := Plan{Checks: []Check{&stubCheck{name: "c", results: []Result{ok}}}} + start := time.Now() + run, err := RunRepeated(context.Background(), &Runner{}, plan, RepeatOptions{ + Count: 2, Interval: time.Millisecond, + }) + if err != nil { + t.Fatal(err) + } + if !run.Stretched || run.Interval != repeatIntervalFloor { + t.Errorf("got interval %s stretched=%v, want %s stretched=true", run.Interval, run.Stretched, repeatIntervalFloor) + } + if elapsed := time.Since(start); elapsed < repeatIntervalFloor { + t.Errorf("iterations only %s apart, want >= %s", elapsed, repeatIntervalFloor) + } + if len(run.Iterations) != 2 { + t.Errorf("got %d iterations, want 2", len(run.Iterations)) + } +} + +func TestRunRepeatedCountBounds(t *testing.T) { + plan := Plan{Checks: []Check{&stubCheck{name: "c", results: []Result{{Status: StatusPass}}}}} + for _, n := range []int{-1, 0, 1, RepeatCountMax + 1} { + if _, err := RunRepeated(context.Background(), &Runner{}, plan, RepeatOptions{Count: n, Interval: time.Second}); err == nil { + t.Errorf("count %d: expected an error", n) + } + } +} + +func TestRunRepeatedCancelKeepsCompletedIterations(t *testing.T) { + ok := Result{Status: StatusPass, Summary: "ok"} + plan := Plan{Checks: []Check{&stubCheck{name: "c", results: []Result{ok}}}} + ctx, cancel := context.WithCancel(context.Background()) + run, err := RunRepeated(ctx, &Runner{}, plan, RepeatOptions{ + Count: 10, Interval: time.Second, + OnIteration: func(i int, _ []Result) { + if i == 2 { + cancel() + } + }, + }) + if err != nil { + t.Fatal(err) + } + if len(run.Iterations) != 2 { + t.Errorf("got %d completed iterations after cancel, want 2", len(run.Iterations)) + } +} + +// mkIter builds one iteration's results for aggregate tests. +func mkIter(rs ...Result) []Result { return rs } + +func res(check string, st Status, d time.Duration, timeout bool) Result { + r := Result{Check: check, Status: st, Duration: d} + if timeout { + r = markTimeout(r) + } + return r +} + +func TestAggregateRepeatCountsAndLatency(t *testing.T) { + // 10 iterations of one check: 8 answered with known latencies, 2 lost. + // Latencies 10..80ms -> min 10, median (ceil(4)=4th) 40, p95 (ceil(7.6)=8th) 80, max 80. + var iters [][]Result + for i := 1; i <= 8; i++ { + iters = append(iters, mkIter(res("peap-mschapv2", StatusPass, time.Duration(i)*10*time.Millisecond, false))) + } + // One lost as FAIL, one lost as SKIP (pap reports timeouts as skip) — both + // must count as lost attempts, not skips. + iters = append(iters, + mkIter(res("peap-mschapv2", StatusFail, 5*time.Second, true)), + mkIter(res("peap-mschapv2", StatusSkip, 5*time.Second, true)), + ) + + stats := AggregateRepeat(RepeatRun{Iterations: iters}) + if len(stats) != 1 { + t.Fatalf("got %d stats, want 1", len(stats)) + } + s := stats[0] + if s.Attempts != 10 || s.Successes != 8 || s.Failures != 2 || s.Timeouts != 2 || s.Skipped != 0 { + t.Errorf("counts: attempts=%d successes=%d failures=%d timeouts=%d skipped=%d", + s.Attempts, s.Successes, s.Failures, s.Timeouts, s.Skipped) + } + if s.LatMin != 10*time.Millisecond || s.LatMedian != 40*time.Millisecond || + s.LatP95 != 80*time.Millisecond || s.LatMax != 80*time.Millisecond { + t.Errorf("latency: min=%s median=%s p95=%s max=%s", s.LatMin, s.LatMedian, s.LatP95, s.LatMax) + } + + v := s.Verdict() + if v.Status != StatusFail { + t.Errorf("flaky verdict status: got %s, want fail", v.Status) + } + for _, want := range []string{"8/10 succeeded", "2/10 requests lost", "unstable path", "not a config error"} { + if !strings.Contains(v.Summary, want) { + t.Errorf("flaky verdict missing %q: %s", want, v.Summary) + } + } + if v.Fields["success_rate"] != "8/10" || v.Fields["timeouts"] != "2" { + t.Errorf("verdict fields: %v", v.Fields) + } +} + +func TestAggregateVerdictWording(t *testing.T) { + pass := res("c", StatusPass, time.Millisecond, false) + fail := res("c", StatusFail, time.Millisecond, false) + skip := res("c", StatusSkip, 0, false) + warn := res("c", StatusWarn, time.Millisecond, false) + + cases := []struct { + name string + iters [][]Result + status Status + summary string + }{ + {"all pass", [][]Result{mkIter(pass), mkIter(pass)}, StatusPass, "2/2 succeeded — stable"}, + {"all fail", [][]Result{mkIter(fail), mkIter(fail)}, StatusFail, "points to configuration"}, + {"all skip", [][]Result{mkIter(skip), mkIter(skip)}, StatusSkip, "skipped in all 2 runs"}, + {"warns stay warn", [][]Result{mkIter(warn), mkIter(warn)}, StatusWarn, "stable"}, + {"reject flaky", [][]Result{mkIter(pass), mkIter(fail)}, StatusFail, "1/2 failed after an answer"}, + } + for _, tc := range cases { + stats := AggregateRepeat(RepeatRun{Iterations: tc.iters}) + v := stats[0].Verdict() + if v.Status != tc.status || !strings.Contains(v.Summary, tc.summary) { + t.Errorf("%s: got status=%s summary=%q, want status=%s containing %q", + tc.name, v.Status, v.Summary, tc.status, tc.summary) + } + } +} diff --git a/internal/check/secret.go b/internal/check/secret.go index 0ae984d..121a4d7 100644 --- a/internal/check/secret.go +++ b/internal/check/secret.go @@ -35,13 +35,13 @@ func (SharedSecret) Run(ctx context.Context, t Target) Result { _, raw, _, err := radius.Exchange(t.Address, t.Secret, p, t.Timeout) if err != nil { if errors.Is(err, radius.ErrTimeout) { - return Result{ + return markTimeout(Result{ Check: "shared-secret", Status: StatusSkip, Summary: "Could not verify the shared secret — no reply", Detail: "Most RADIUS servers silently drop requests when the secret is " + "wrong OR when the client isn't whitelisted, so a timeout alone can't " + "tell them apart. Fix reachability/whitelisting first, then re-run.", - } + }) } return Result{Check: "shared-secret", Status: StatusSkip, Summary: "Could not verify the shared secret: " + err.Error()} } diff --git a/internal/check/ttls.go b/internal/check/ttls.go index 64a5da5..08d3918 100644 --- a/internal/check/ttls.go +++ b/internal/check/ttls.go @@ -2,6 +2,7 @@ package check import ( "context" + "errors" "github.com/authhound/probe/internal/radius" ) @@ -37,11 +38,15 @@ func (c EAPTTLS) Run(ctx context.Context, t Target) Result { res, err := sess.AuthEAPTTLS(ctx, c.User, c.Pass, c.ServerName) if err != nil { - return Result{ + r := Result{ Check: "eap-ttls", Status: StatusFail, Summary: "EAP-TTLS exchange did not complete", Detail: "The tunnel or inner exchange broke before a verdict: " + err.Error() + ".", } + if errors.Is(err, radius.ErrTimeout) { + r = markTimeout(r) + } + return r } fields := map[string]string{} diff --git a/internal/report/json.go b/internal/report/json.go index c64db93..013d6bf 100644 --- a/internal/report/json.go +++ b/internal/report/json.go @@ -23,6 +23,7 @@ type JSONSink struct { w io.Writer results []check.Result counts map[check.Status]int + repeat *repeatDoc // non-nil only in repeat mode (--count); see SetRepeat } func NewJSONSink(w io.Writer) *JSONSink { @@ -50,9 +51,11 @@ func (s *JSONSink) Close() error { SchemaVersion string `json:"schema_version"` Results []check.Result `json:"results"` Summary summaryCounts `json:"summary"` + Repeat *repeatDoc `json:"repeat,omitempty"` // additive: only with --count }{ SchemaVersion: SchemaVersion, Results: s.results, + Repeat: s.repeat, Summary: summaryCounts{ Pass: s.counts[check.StatusPass], Fail: s.counts[check.StatusFail], diff --git a/internal/report/repeat.go b/internal/report/repeat.go new file mode 100644 index 0000000..234bac0 --- /dev/null +++ b/internal/report/repeat.go @@ -0,0 +1,108 @@ +package report + +import ( + "fmt" + "strings" + + "github.com/authhound/probe/internal/check" +) + +// This file renders repeat mode (`radius test --count N`): the per-iteration +// progress line for the text sink, and the additive `repeat` block for the +// JSON document. Everything here is additive within schema major "1" — the +// top-level results/summary keep their shape (one object per check; in repeat +// mode those are the aggregate verdicts). + +// repeatDoc is the JSON `repeat` block, present only when --count was used. +type repeatDoc struct { + Count int `json:"count"` + Completed int `json:"completed"` + IntervalMs int64 `json:"interval_ms"` + RequestedIntervalMs int64 `json:"requested_interval_ms"` + IntervalStretched bool `json:"interval_stretched"` + Iterations []iterationDoc `json:"iterations"` + Aggregate []aggregateDoc `json:"aggregate"` +} + +type iterationDoc struct { + Results []check.Result `json:"results"` +} + +type aggregateDoc struct { + Check string `json:"check"` + Attempts int `json:"attempts"` + Successes int `json:"successes"` + Failures int `json:"failures"` + Timeouts int `json:"timeouts"` + Skipped int `json:"skipped"` + LatencyMs *latencyDoc `json:"latency_ms,omitempty"` +} + +// latencyDoc is nearest-rank percentiles over answered runs, in milliseconds. +type latencyDoc struct { + Min int64 `json:"min"` + Median int64 `json:"median"` + P95 int64 `json:"p95"` + Max int64 `json:"max"` +} + +// SetRepeat attaches the repeat block to the JSON document. count is the +// requested iteration count (completed may be lower after an interrupt). +func (s *JSONSink) SetRepeat(count int, run check.RepeatRun, stats []check.CheckStats) { + doc := &repeatDoc{ + Count: count, + Completed: len(run.Iterations), + IntervalMs: run.Interval.Milliseconds(), + RequestedIntervalMs: run.RequestedInterval.Milliseconds(), + IntervalStretched: run.Stretched, + Iterations: []iterationDoc{}, + Aggregate: []aggregateDoc{}, + } + for _, iter := range run.Iterations { + doc.Iterations = append(doc.Iterations, iterationDoc{Results: iter}) + } + for _, st := range stats { + a := aggregateDoc{ + Check: st.Check, + Attempts: st.Attempts, + Successes: st.Successes, + Failures: st.Failures, + Timeouts: st.Timeouts, + Skipped: st.Skipped, + } + if st.Successes > 0 && st.LatMax > 0 { + a.LatencyMs = &latencyDoc{ + Min: st.LatMin.Milliseconds(), + Median: st.LatMedian.Milliseconds(), + P95: st.LatP95.Milliseconds(), + Max: st.LatMax.Milliseconds(), + } + } + doc.Aggregate = append(doc.Aggregate, a) + } + s.repeat = doc +} + +// IterationLine renders one repeat iteration as a single compact progress +// line. Skipped checks are omitted (they are identical every iteration and +// would drown the signal); lost requests are called out. +func IterationLine(iteration, total int, results []check.Result) string { + var parts []string + for _, r := range results { + if r.Status == check.StatusSkip && r.Fields[check.TimeoutField] != "true" { + continue + } + p := fmt.Sprintf("%s %s", r.Check, strings.ToUpper(string(r.Status))) + if r.Fields[check.TimeoutField] == "true" { + p = r.Check + " LOST (no reply)" + } else if rtt, ok := r.Fields["rtt_ms"]; ok { + p += " " + rtt + "ms" + } + parts = append(parts, p) + } + if parts == nil { + parts = []string{"all checks skipped"} + } + width := len(fmt.Sprint(total)) + return fmt.Sprintf("run %*d/%d %s", width, iteration, total, strings.Join(parts, " · ")) +} diff --git a/internal/report/repeat_test.go b/internal/report/repeat_test.go new file mode 100644 index 0000000..a2320ce --- /dev/null +++ b/internal/report/repeat_test.go @@ -0,0 +1,104 @@ +package report + +import ( + "bytes" + "os" + "path/filepath" + "strings" + "testing" + "time" + + "github.com/authhound/probe/internal/check" +) + +// repeatRun is a fixed two-iteration run (one flaky check) whose JSON render is +// pinned by testdata/schema-v1-repeat.golden.json — the shape guard for the +// additive `repeat` block. Regenerate with: +// `go test ./internal/report -run Golden -update`. +func repeatRun() (check.RepeatRun, []check.CheckStats) { + ok := check.Result{ + Check: "pap-auth", Status: check.StatusPass, + Summary: "PAP authentication accepted for alice", + Duration: 4_000_000, + } + lost := check.Result{ + Check: "pap-auth", Status: check.StatusFail, + Summary: "No reply from 127.0.0.1:1812 within 5s", + Fields: map[string]string{check.TimeoutField: "true"}, + Duration: 5_000_000_000, + } + run := check.RepeatRun{ + Iterations: [][]check.Result{{ok}, {lost}}, + RequestedInterval: 500 * time.Millisecond, + Interval: time.Second, + Stretched: true, + } + return run, check.AggregateRepeat(run) +} + +func TestJSONRepeatGolden(t *testing.T) { + run, stats := repeatRun() + var buf bytes.Buffer + sink := NewJSONSink(&buf) + for _, st := range stats { + sink.Emit(st.Verdict()) + } + sink.SetRepeat(2, run, stats) + if err := sink.Close(); err != nil { + t.Fatalf("Close: %v", err) + } + got := buf.Bytes() + + golden := filepath.Join("testdata", "schema-v1-repeat.golden.json") + if *update { + if err := os.WriteFile(golden, got, 0o644); err != nil { + t.Fatalf("update golden: %v", err) + } + t.Logf("updated %s", golden) + return + } + want, err := os.ReadFile(golden) + if err != nil { + t.Fatalf("read golden (run with -update to create it): %v", err) + } + if !bytes.Equal(got, want) { + t.Errorf("repeat --json shape changed vs %s.\n"+ + "If intentional and additive, regenerate with:\n"+ + " go test ./internal/report -run Golden -update\n"+ + "Otherwise bump report.SchemaVersion and document it in docs/json-schema.md."+ + "\n\n--- got ---\n%s\n--- want ---\n%s", golden, got, want) + } +} + +// TestJSONWithoutRepeatOmitsBlock pins that single runs are byte-identical to +// before this feature existed: no `repeat` key unless --count was used. +func TestJSONWithoutRepeatOmitsBlock(t *testing.T) { + var buf bytes.Buffer + sink := NewJSONSink(&buf) + sink.Emit(check.Result{Check: "reachability", Status: check.StatusPass, Summary: "ok"}) + if err := sink.Close(); err != nil { + t.Fatal(err) + } + if bytes.Contains(buf.Bytes(), []byte(`"repeat"`)) { + t.Errorf("single-run JSON must not contain a repeat block:\n%s", buf.String()) + } +} + +func TestIterationLine(t *testing.T) { + results := []check.Result{ + {Check: "reachability", Status: check.StatusPass, Fields: map[string]string{"rtt_ms": "2"}}, + {Check: "shared-secret", Status: check.StatusPass}, + {Check: "pap-auth", Status: check.StatusSkip, Fields: map[string]string{check.TimeoutField: "true"}}, + {Check: "eap-tls", Status: check.StatusSkip, Summary: "No client certificate given"}, + } + line := IterationLine(3, 10, results) + for _, want := range []string{"run 3/10", "reachability PASS 2ms", "shared-secret PASS", "pap-auth LOST (no reply)"} { + if !strings.Contains(line, want) { + t.Errorf("line missing %q: %s", want, line) + } + } + // Plain skips stay off the line; a timed-out check (whatever its status) is on it. + if strings.Contains(line, "eap-tls") { + t.Errorf("skipped check should not appear: %s", line) + } +} diff --git a/internal/report/testdata/schema-v1-repeat.golden.json b/internal/report/testdata/schema-v1-repeat.golden.json new file mode 100644 index 0000000..e18372b --- /dev/null +++ b/internal/report/testdata/schema-v1-repeat.golden.json @@ -0,0 +1,76 @@ +{ + "schema_version": "1", + "results": [ + { + "check": "pap-auth", + "status": "fail", + "summary": "pap-auth: 1/2 succeeded, 1/2 requests lost — consistent with an unstable path or an overloaded/failing server, not a config error", + "detail": "Latency over 1 answered runs: min 4ms, median 4ms, p95 4ms, max 4ms.", + "fields": { + "attempts": "2", + "latency_max_ms": "4", + "latency_median_ms": "4", + "latency_min_ms": "4", + "latency_p95_ms": "4", + "success_rate": "1/2", + "successes": "1", + "timeouts": "1" + } + } + ], + "summary": { + "pass": 0, + "fail": 1, + "warn": 0, + "info": 0, + "skip": 0 + }, + "repeat": { + "count": 2, + "completed": 2, + "interval_ms": 1000, + "requested_interval_ms": 500, + "interval_stretched": true, + "iterations": [ + { + "results": [ + { + "check": "pap-auth", + "status": "pass", + "summary": "PAP authentication accepted for alice", + "duration_ns": 4000000 + } + ] + }, + { + "results": [ + { + "check": "pap-auth", + "status": "fail", + "summary": "No reply from 127.0.0.1:1812 within 5s", + "fields": { + "timeout": "true" + }, + "duration_ns": 5000000000 + } + ] + } + ], + "aggregate": [ + { + "check": "pap-auth", + "attempts": 2, + "successes": 1, + "failures": 1, + "timeouts": 1, + "skipped": 0, + "latency_ms": { + "min": 4, + "median": 4, + "p95": 4, + "max": 4 + } + } + ] + } +} diff --git a/test/README.md b/test/README.md index ba38f5c..10f340a 100644 --- a/test/README.md +++ b/test/README.md @@ -45,6 +45,16 @@ $ docker compose -f test/lab/docker-compose.yml up # Ctrl-C to stop It serves classic RADIUS/UDP on `1812` and RadSec on `2083`. +With the `flaky` profile it also starts a second FreeRADIUS behind ~25% induced +packet loss and jitter (tc netem in a sidecar), published on `127.0.0.1:11812` — +the fixture for `radius test --count`: + +```console +$ docker compose -f test/lab/docker-compose.yml --profile flaky up +$ go run ./cmd/authhound-probe radius test \ + --server 127.0.0.1:11812 --secret testing123 --pap alice:pw --count 10 +``` + ### Run the UDP checks ```console diff --git a/test/freeradius-smoke.sh b/test/freeradius-smoke.sh index 167ffb8..4e0e30d 100755 --- a/test/freeradius-smoke.sh +++ b/test/freeradius-smoke.sh @@ -13,7 +13,7 @@ cd "$(dirname "$0")/.." SECRET="testing123" work="$(mktemp -d)" -trap 'rm -rf "$work"; docker rm -f ah-freeradius ah-freeradius-nc >/dev/null 2>&1 || true' EXIT +trap 'rm -rf "$work"; docker rm -f ah-freeradius ah-freeradius-nc ah-freeradius-flaky ah-netem >/dev/null 2>&1 || true' EXIT # One test user for the PAP check, plus a machine identity for the NPS-style # machine-auth check. This file replaces the default authorize file; a single @@ -209,5 +209,67 @@ echo "$json" | grep -q '"source_ip": "127.0.0.1"' || { echo "FAIL: --json missin if echo "$json" | grep -q "$SECRET"; then echo "FAIL: secret leaked into JSON output"; exit 1; fi echo "OK: registration hint present with detected IP; secret not leaked" +echo +echo "== flaky server: --count against induced packet loss (tc netem) ==" +# A second FreeRADIUS on a bridge network (so netem can shape just its traffic, +# published on 127.0.0.1:11812) with a sidecar dropping ~25% of its replies and +# adding 30ms±20ms delay. This is the fixture for `--count`: intermittent loss +# and jitter that a single-shot run can't see. +docker rm -f ah-freeradius-nc >/dev/null 2>&1 || true +cat > "$work/clients-flaky" <<'EOF' +client lab { + ipaddr = 0.0.0.0/0 + secret = testing123 +} +EOF +docker run -d --rm --name ah-freeradius-flaky -p 127.0.0.1:11812:1812/udp \ + -v "$work/authorize:/etc/raddb/mods-config/files/authorize:ro" \ + -v "$work/clients-flaky:/etc/raddb/clients.conf:ro" \ + freeradius/freeradius-server:latest -fxx -l stdout >/dev/null +docker run -d --rm --name ah-netem --network "container:ah-freeradius-flaky" \ + --cap-add NET_ADMIN alpine:3 sh -c \ + "apk add --no-cache iproute2 >/dev/null && tc qdisc replace dev eth0 root netem loss 25% delay 30ms 20ms && sleep infinity" >/dev/null +for i in $(seq 1 30); do + if docker logs ah-freeradius-flaky 2>&1 | grep -q "Ready to process requests"; then break; fi + sleep 0.5 +done +for i in $(seq 1 30); do + if docker exec ah-netem tc qdisc show dev eth0 2>/dev/null | grep -q netem; then break; fi + sleep 0.5 +done + +echo +echo "== --count 10 with 25% loss (expect lost requests, jitter, exit 1) ==" +set +e +count_out="$("$work/authhound-probe" radius test --server 127.0.0.1:11812 --secret "$SECRET" \ + --pap 'alice:pw' --count 10 --interval 1s --timeout 2s --no-color)"; count_rc=$? +set -e +echo "$count_out" +echo "$count_out" | grep -q "Aggregate over 10 runs" || { echo "FAIL: missing aggregate block"; exit 1; } +echo "$count_out" | grep -q "lost" || { echo "FAIL: expected lost requests under 25% packet loss"; exit 1; } +[ "$count_rc" -eq 1 ] || { echo "FAIL: lost requests should exit 1, got $count_rc"; exit 1; } +echo "OK: aggregate names the loss; exit 1" + +echo +echo "== --count --json exposes repeat block (iterations + aggregate) ==" +set +e +cjson="$("$work/authhound-probe" radius test --server 127.0.0.1:11812 --secret "$SECRET" \ + --pap 'alice:pw' --count 5 --interval 1s --timeout 2s --json)" +set -e +echo "$cjson" | grep -q '"repeat"' || { echo "FAIL: --json missing repeat block"; exit 1; } +echo "$cjson" | grep -q '"completed": 5' || { echo "FAIL: repeat.completed should be 5"; exit 1; } +echo "$cjson" | grep -q '"aggregate"' || { echo "FAIL: repeat.aggregate missing"; exit 1; } +echo "$cjson" | grep -q '"iterations"' || { echo "FAIL: repeat.iterations missing"; exit 1; } +echo "$cjson" | grep -q '"latency_ms"' || { echo "FAIL: aggregate latency stats missing"; exit 1; } +if echo "$cjson" | grep -q "$SECRET"; then echo "FAIL: secret leaked into --count JSON"; exit 1; fi +echo "OK: repeat block present with per-iteration results and aggregate stats" + +echo +echo "== --interval below the safety floor gets stretched (expect a note) ==" +stretch_note="$("$work/authhound-probe" radius test --server 127.0.0.1:11812 --secret "$SECRET" \ + --count 2 --interval 100ms --timeout 2s --no-color 2>&1 >/dev/null || true)" +echo "$stretch_note" | grep -q "safety floor" || { echo "FAIL: expected the interval-stretch note"; exit 1; } +echo "OK: sub-floor interval stretched and announced" + echo echo "== done; tearing down FreeRADIUS ==" diff --git a/test/lab/config/clients-flaky b/test/lab/config/clients-flaky new file mode 100644 index 0000000..b883965 --- /dev/null +++ b/test/lab/config/clients-flaky @@ -0,0 +1,7 @@ +# Client entry for the flaky-profile server. It runs on a bridge network, so +# the probe's requests arrive from the Docker gateway IP (not 127.0.0.1) — +# accept any source. Throwaway lab secret; never use this on a real server. +client lab { + ipaddr = 0.0.0.0/0 + secret = testing123 +} diff --git a/test/lab/docker-compose.yml b/test/lab/docker-compose.yml index 6aa7c8e..680ccd1 100644 --- a/test/lab/docker-compose.yml +++ b/test/lab/docker-compose.yml @@ -22,3 +22,39 @@ services: volumes: - ./config/authorize:/etc/raddb/mods-config/files/authorize:ro - ./config/radsec:/etc/raddb/sites-enabled/radsec:ro + + # Flaky profile: the same FreeRADIUS behind induced packet loss and jitter, + # for exercising `radius test --count` against genuinely intermittent + # failures. Runs on a bridge network (not host) so netem can shape just this + # server's traffic; published on 127.0.0.1:11812. + # + # docker compose -f test/lab/docker-compose.yml --profile flaky up + # authhound-probe radius test --server 127.0.0.1:11812 --secret testing123 \ + # --pap alice:pw --count 10 + # + # Requests reach it from the Docker bridge gateway, so clients-flaky accepts + # any source IP (throwaway lab secret, never production). + freeradius-flaky: + image: freeradius/freeradius-server:latest + profiles: [flaky] + command: ["-fxx", "-l", "stdout"] + ports: + - "127.0.0.1:11812:1812/udp" + volumes: + - ./config/authorize:/etc/raddb/mods-config/files/authorize:ro + - ./config/clients-flaky:/etc/raddb/clients.conf:ro + + # netem sidecar: joins the flaky server's network namespace and drops ~25% of + # its replies with 30ms±20ms delay — lost requests and latency jitter, the two + # things --count exists to expose. NET_ADMIN is scoped to the sidecar and that + # one namespace, not the host. + netem: + image: alpine:3 + profiles: [flaky] + network_mode: "service:freeradius-flaky" + cap_add: [NET_ADMIN] + command: > + sh -c "apk add --no-cache iproute2 >/dev/null + && tc qdisc replace dev eth0 root netem loss 25% delay 30ms 20ms + && echo 'netem active: 25% loss, 30ms±20ms delay' + && sleep infinity"