From 603d113cd23a60dbd0f31b95716a044c24ada8d0 Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Thu, 9 Jul 2026 16:29:43 -0700 Subject: [PATCH 01/32] feat(kvp): add diagnostics layer over KvpPoolStore --- libazureinit-kvp/Cargo.toml | 2 +- libazureinit-kvp/src/cli.rs | 231 +++++++++- libazureinit-kvp/src/diagnostics.rs | 631 ++++++++++++++++++++++++++ libazureinit-kvp/src/error.rs | 7 + libazureinit-kvp/src/lib.rs | 6 + libazureinit-kvp/tests/cli.rs | 198 ++++++++ libazureinit-kvp/tests/diagnostics.rs | 234 ++++++++++ 7 files changed, 1306 insertions(+), 3 deletions(-) create mode 100644 libazureinit-kvp/src/diagnostics.rs create mode 100644 libazureinit-kvp/tests/diagnostics.rs diff --git a/libazureinit-kvp/Cargo.toml b/libazureinit-kvp/Cargo.toml index d58f862d..cd0bb03c 100644 --- a/libazureinit-kvp/Cargo.toml +++ b/libazureinit-kvp/Cargo.toml @@ -15,7 +15,7 @@ csv = "1" libc = "0.2" serde_json = "1.0.96" tracing = "0.1.40" -uuid = "1.3" +uuid = { version = "1.3", features = ["v4"] } [dev-dependencies] rstest = { version = "0.26", default-features = false } diff --git a/libazureinit-kvp/src/cli.rs b/libazureinit-kvp/src/cli.rs index fa347cc7..08d3cc83 100644 --- a/libazureinit-kvp/src/cli.rs +++ b/libazureinit-kvp/src/cli.rs @@ -10,8 +10,8 @@ use clap::{Parser, Subcommand, ValueEnum}; use serde_json::json; use crate::{ - write_report, KvpError, KvpPool, KvpPoolStore, PoolMode, - ProvisioningReport, ReportPpsType, + write_report, DiagnosticEvent, DiagnosticRecord, DiagnosticsKvp, KvpError, + KvpPool, KvpPoolStore, PoolMode, ProvisioningReport, ReportPpsType, }; const EXIT_OK: u8 = 0; @@ -176,6 +176,50 @@ enum Command { #[arg(long, value_parser = parse_supporting_data)] supporting_data: Vec, }, + /// Inspect decoded diagnostic events, reassembling chunked records. + Diag { + #[command(subcommand)] + command: DiagCommand, + }, +} + +/// Diagnostics subcommands operating on the reassembled event view. +#[derive(Subcommand, Debug)] +enum DiagCommand { + /// Print every record, reassembling chunked events. Raw records + /// (e.g. PROVISIONING_REPORT) are hidden unless --include-raw. + Dump { + /// Also print raw (non-event) records. + #[arg(long)] + include_raw: bool, + }, + /// Print decoded events (oldest first), optionally filtered. + Events { + /// Only show events at this level (error, warn, info, debug, trace). + #[arg(long)] + level: Option, + /// Only show events whose name contains this substring. + #[arg(long)] + name: Option, + }, + /// Print the last COUNT events (default 20). + Tail { + /// Number of trailing events to print. + #[arg(short = 'n', long = "count", default_value_t = 20)] + count: usize, + }, + /// Remove this layer's events, leaving raw records intact. + Clear { + /// VM identifier whose events to remove. + #[arg(long)] + vm_id: String, + /// Event key prefix whose events to remove. + #[arg(long)] + event_prefix: String, + /// Confirm the removal (required). + #[arg(long)] + yes: bool, + }, } #[derive(ValueEnum, Clone, Copy, Debug, PartialEq, Eq)] @@ -267,6 +311,7 @@ fn dispatch(cli: Cli, stdout: &mut W) -> Result { documentation_url, supporting_data, ), + Command::Diag { command } => diag(&store, stdout, command, output), } } @@ -425,6 +470,188 @@ fn is_stale( Ok(if stale { EXIT_OK } else { EXIT_NOT_FOUND }) } +fn diag( + store: &KvpPoolStore, + stdout: &mut W, + command: DiagCommand, + output: OutputMode, +) -> Result { + match command { + DiagCommand::Dump { include_raw } => { + diag_dump(store, stdout, include_raw, output) + } + DiagCommand::Events { level, name } => { + diag_events(store, stdout, level, name, None, output) + } + DiagCommand::Tail { count } => { + diag_events(store, stdout, None, None, Some(count), output) + } + DiagCommand::Clear { + vm_id, + event_prefix, + yes, + } => diag_clear(store, stdout, vm_id, event_prefix, yes, output), + } +} + +fn diag_dump( + store: &KvpPoolStore, + stdout: &mut W, + include_raw: bool, + output: OutputMode, +) -> Result { + let diagnostics = DiagnosticsKvp::new(store.clone(), "", ""); + let records: Vec<_> = diagnostics + .records()? + .into_iter() + .filter(|record| { + include_raw || !matches!(record, DiagnosticRecord::Raw { .. }) + }) + .collect(); + + match output { + OutputMode::Text => { + for record in &records { + let line = match record { + DiagnosticRecord::Event { event, chunks } => format!( + "event level={} name={} event_id={} chunks={} \ + message={}", + event.level, + event.name, + event.event_id, + chunks, + event.message + ), + DiagnosticRecord::Raw { key, value } => { + format!("raw key={key} value={value}") + } + DiagnosticRecord::Malformed { key, value, reason } => { + format!( + "malformed key={key} reason={reason} \ + value={value}" + ) + } + }; + writeln!(stdout, "{line}")?; + } + } + OutputMode::Json => { + let array: Vec<_> = records.iter().map(diag_record_json).collect(); + writeln_json(stdout, &serde_json::Value::Array(array))?; + } + } + Ok(EXIT_OK) +} + +fn diag_events( + store: &KvpPoolStore, + stdout: &mut W, + level: Option, + name: Option, + tail: Option, + output: OutputMode, +) -> Result { + let level = level.as_deref().map(parse_level_filter).transpose()?; + + let diagnostics = DiagnosticsKvp::new(store.clone(), "", ""); + let mut events = diagnostics.events()?; + + if let Some(level) = level { + events.retain(|event| event.level == level); + } + if let Some(needle) = name.as_deref() { + events.retain(|event| event.name.contains(needle)); + } + if let Some(count) = tail { + let excess = events.len().saturating_sub(count); + events.drain(..excess); + } + + match output { + OutputMode::Text => { + for event in &events { + let line = format!( + "event level={} name={} event_id={} message={}", + event.level, event.name, event.event_id, event.message + ); + writeln!(stdout, "{line}")?; + } + } + OutputMode::Json => { + let array: Vec<_> = events.iter().map(diag_event_json).collect(); + writeln_json(stdout, &serde_json::Value::Array(array))?; + } + } + Ok(EXIT_OK) +} + +fn diag_clear( + store: &KvpPoolStore, + stdout: &mut W, + vm_id: String, + event_prefix: String, + yes: bool, + output: OutputMode, +) -> Result { + if !yes { + return Err(CliError::Usage( + "refusing to clear diagnostic events without --yes".to_string(), + )); + } + let diagnostics = DiagnosticsKvp::new(store.clone(), vm_id, event_prefix); + diagnostics.clear()?; + match output { + OutputMode::Text => writeln!(stdout, "cleared")?, + OutputMode::Json => writeln_json(stdout, &json!({ "cleared": true }))?, + } + Ok(EXIT_OK) +} + +/// Parse a `--level` filter argument into a [`tracing::Level`]. +fn parse_level_filter(level: &str) -> Result { + level.parse::().map_err(|_| { + CliError::Usage(format!( + "invalid level '{level}' (expected error, warn, info, debug, \ + or trace)" + )) + }) +} + +/// Render a [`DiagnosticRecord`] as a JSON object. +fn diag_record_json(record: &DiagnosticRecord) -> serde_json::Value { + match record { + DiagnosticRecord::Event { event, chunks } => { + let mut value = diag_event_json(event); + if let serde_json::Value::Object(map) = &mut value { + map.insert("chunks".to_string(), json!(chunks)); + } + value + } + DiagnosticRecord::Raw { key, value } => json!({ + "kind": "raw", + "key": key, + "value": value, + }), + DiagnosticRecord::Malformed { key, value, reason } => json!({ + "kind": "malformed", + "key": key, + "value": value, + "reason": reason, + }), + } +} + +/// Render a [`DiagnosticEvent`] as a JSON object (without chunk count). +fn diag_event_json(event: &DiagnosticEvent) -> serde_json::Value { + json!({ + "kind": "event", + "level": event.level.to_string(), + "name": event.name, + "event_id": event.event_id, + "message": event.message, + }) +} + fn report_success( store: &KvpPoolStore, vm_id: Option, diff --git a/libazureinit-kvp/src/diagnostics.rs b/libazureinit-kvp/src/diagnostics.rs new file mode 100644 index 00000000..e71f74bb --- /dev/null +++ b/libazureinit-kvp/src/diagnostics.rs @@ -0,0 +1,631 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +//! Typed diagnostics layer over the raw +//! [`KvpPoolStore`](crate::KvpPoolStore) key/value API. +//! +//! Where [`KvpPoolStore`](crate::KvpPoolStore) treats keys and values as +//! opaque bytes, [`DiagnosticsKvp`] understands the telemetry +//! conventions azure-init writes into the guest pool: +//! +//! - **Event keys** encode structured metadata as a five-segment, +//! pipe-delimited string +//! (`||||`). +//! - **Chunking**: a single KVP record caps the value at the store's +//! per-record limit (see [`MAX_CHUNK_BYTES`] for the safe-mode value). +//! Longer messages are split at UTF-8 codepoint boundaries into +//! multiple records that share one key, written atomically under a +//! single lock, and reassembled on read. +//! - **Classification**: [`records`](DiagnosticsKvp::records) sorts every +//! stored record into a [`DiagnosticRecord`] — a reassembled +//! [`DiagnosticEvent`], an unstructured [`Raw`](DiagnosticRecord::Raw) +//! record such as `PROVISIONING_REPORT`, or a +//! [`Malformed`](DiagnosticRecord::Malformed) event key. +//! +//! This module is policy only: all locking, size enforcement, and +//! on-disk encoding stay in [`KvpPoolStore`](crate::KvpPoolStore). +//! +//! # Example +//! +//! ``` +//! use libazureinit_kvp::{ +//! DiagnosticEvent, DiagnosticsKvp, KvpPool, KvpPoolStore, PoolMode, +//! MAX_CHUNK_BYTES, +//! }; +//! use tracing::Level; +//! +//! # fn main() -> Result<(), libazureinit_kvp::KvpError> { +//! let dir = std::env::temp_dir() +//! .join(format!("libazureinit-kvp-doc-{}", std::process::id())); +//! std::fs::create_dir_all(&dir)?; +//! let store = KvpPoolStore::new_in(KvpPool::Guest, &dir, PoolMode::Safe)?; +//! store.clear()?; +//! +//! let diagnostics = +//! DiagnosticsKvp::new(store, "vm-1234", "azure-init-doc"); +//! +//! // A short event lands in a single record. +//! diagnostics.emit(&DiagnosticEvent::new( +//! Level::INFO, +//! "user:create_user", +//! "Creating user azureuser", +//! ))?; +//! +//! // A long message is split across records that share one key and is +//! // reassembled transparently on read. +//! let long = "x".repeat(MAX_CHUNK_BYTES * 2 + 10); +//! diagnostics +//! .emit(&DiagnosticEvent::new(Level::DEBUG, "config:dump", &long))?; +//! +//! let events = diagnostics.events()?; +//! assert_eq!(events.len(), 2); +//! assert_eq!(events[1].message.len(), MAX_CHUNK_BYTES * 2 + 10); +//! +//! # std::fs::remove_dir_all(&dir).ok(); +//! # Ok(()) +//! # } +//! ``` + +use tracing::Level; +use uuid::Uuid; + +use crate::{KvpError, KvpPoolStore}; + +/// Maximum number of value bytes per KVP record under +/// [`PoolMode::Safe`](crate::PoolMode::Safe). +/// +/// This matches a safe-mode store's +/// [`KvpPoolStore::max_value_size`](crate::KvpPoolStore::max_value_size). +/// [`DiagnosticsKvp::emit`] splits messages longer than the store's +/// actual limit, so an [`Unsafe`](crate::PoolMode::Unsafe) store uses +/// its larger capacity; this constant is the conservative reference +/// value used throughout the diagnostics conventions. +pub const MAX_CHUNK_BYTES: usize = 1022; + +/// Delimiter separating the segments of a diagnostic event key. +const EVENT_KEY_DELIMITER: char = '|'; + +/// Format a diagnostic event key as its `|`-delimited on-disk string: +/// `||||`. +/// +/// [`classify_key`] is the inverse. For example: +/// +/// ```text +/// azure-init-0.1.0|3f2504e0-4f89-41d3-9a0c-0305e82c3301|INFO|user:create_user|8f3e9c4a-... +/// ``` +fn format_event_key( + prefix: &str, + vm_id: &str, + level: Level, + name: &str, + event_id: &str, +) -> String { + let d = EVENT_KEY_DELIMITER; + format!("{prefix}{d}{vm_id}{d}{level}{d}{name}{d}{event_id}") +} + +/// Outcome of inspecting a raw pool key. +enum KeyClass<'a> { + /// The key is a well-formed event key. + Event { + prefix: &'a str, + vm_id: &'a str, + level: Level, + name: &'a str, + event_id: &'a str, + }, + /// The key has five segments but is not a valid event. + Malformed { reason: String }, + /// The key is not an event key (e.g. `PROVISIONING_REPORT`). + Raw, +} + +/// Classify a raw pool key without allocating. +fn classify_key(key: &str) -> KeyClass<'_> { + let mut segments = key.split(EVENT_KEY_DELIMITER); + + // `str::split` always yields at least one element. + let prefix = segments.next().unwrap_or_default(); + let (Some(vm_id), Some(level), Some(name), Some(event_id)) = ( + segments.next(), + segments.next(), + segments.next(), + segments.next(), + ) else { + return KeyClass::Raw; + }; + if segments.next().is_some() { + // More than five segments: a `|` leaked into a field. + return KeyClass::Raw; + } + + match level.parse::() { + Ok(level) => KeyClass::Event { + prefix, + vm_id, + level, + name, + event_id, + }, + Err(_) => KeyClass::Malformed { + reason: format!("unrecognized level {level:?}"), + }, + } +} + +/// Split `value` into pieces of at most `max_bytes` bytes each, always +/// at UTF-8 codepoint boundaries. +/// +/// An empty input yields a single empty chunk so callers still write one +/// record. A codepoint wider than `max_bytes` (only possible for tiny +/// `max_bytes`, never for [`MAX_CHUNK_BYTES`]) is emitted whole so the +/// split always makes progress. +fn chunk_at_char_boundary(value: &str, max_bytes: usize) -> Vec<&str> { + debug_assert!(max_bytes > 0, "max_bytes must be positive"); + if value.is_empty() { + return vec![""]; + } + + let mut chunks = Vec::new(); + let mut start = 0; + while start < value.len() { + if value.len() - start <= max_bytes { + chunks.push(&value[start..]); + break; + } + + // Walk back from the byte limit to the nearest codepoint boundary. + let mut end = start + max_bytes; + while end > start && !value.is_char_boundary(end) { + end -= 1; + } + if end == start { + // One codepoint spans the whole window; take it whole. + end = start + max_bytes + 1; + while end < value.len() && !value.is_char_boundary(end) { + end += 1; + } + } + + chunks.push(&value[start..end]); + start = end; + } + chunks +} + +/// Reject the `|` key delimiter in an event field so the formatted key +/// round-trips through [`classify_key`]. +fn reject_delimiter(field: &'static str, value: &str) -> Result<(), KvpError> { + if value.contains(EVENT_KEY_DELIMITER) { + return Err(KvpError::EventFieldContainsDelimiter { field }); + } + Ok(()) +} + +/// A single diagnostic event. +/// +/// Construct one with [`new`](Self::new) (which generates a fresh +/// `event_id`) and hand it to [`DiagnosticsKvp::emit`]; events read back +/// via [`DiagnosticsKvp::records`] carry the same fields, decoded from +/// the pool. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct DiagnosticEvent { + /// Severity of the event. + pub level: Level, + /// Formatted event name, e.g. `user:create_user`. + pub name: String, + /// Per-emit identifier. [`new`](Self::new) generates a UUIDv4; + /// every chunk of one emitted event shares this value. + pub event_id: String, + /// Literal value bytes written to the pool. The diagnostics layer + /// imposes no format on this string. + pub message: String, +} + +impl DiagnosticEvent { + /// Create an event with a freshly generated `event_id` (UUIDv4). + pub fn new( + level: Level, + name: impl Into, + message: impl Into, + ) -> Self { + Self { + level, + name: name.into(), + event_id: Uuid::new_v4().to_string(), + message: message.into(), + } + } +} + +/// A single record read back from the pool and classified by +/// [`DiagnosticsKvp::records`]. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum DiagnosticRecord { + /// A reassembled diagnostic event. + Event { + /// The decoded event. + event: DiagnosticEvent, + /// Number of on-disk records the value spanned (1 when short). + chunks: usize, + }, + /// An unstructured record whose key is not an event key, such as + /// `PROVISIONING_REPORT`. + Raw { + /// The record key. + key: String, + /// The reassembled record value. + value: String, + }, + /// A record whose key has five segments but is not a valid event + /// (for example, an unrecognized level). + Malformed { + /// The record key. + key: String, + /// The reassembled record value. + value: String, + /// Why the key failed to parse as an event. + reason: String, + }, +} + +/// A typed diagnostics view over a [`KvpPoolStore`]. +/// +/// Owns the event-key `prefix` and `vm_id` used to format and +/// [`scope`](Self::clear) this layer's events. See the +/// [module documentation](self) for the on-disk format. +#[derive(Clone, Debug)] +pub struct DiagnosticsKvp { + store: KvpPoolStore, + vm_id: String, + event_prefix: String, +} + +impl DiagnosticsKvp { + /// Wrap `store` with the `vm_id` and `event_prefix` stamped into + /// this layer's event keys. + pub fn new( + store: KvpPoolStore, + vm_id: impl Into, + event_prefix: impl Into, + ) -> Self { + Self { + store, + vm_id: vm_id.into(), + event_prefix: event_prefix.into(), + } + } + + /// The underlying store. + pub fn store(&self) -> &KvpPoolStore { + &self.store + } + + /// The VM identifier stamped into event keys. + pub fn vm_id(&self) -> &str { + &self.vm_id + } + + /// The prefix stamped into event keys. + pub fn event_prefix(&self) -> &str { + &self.event_prefix + } + + /// Write `event`. + /// + /// Messages longer than the store's per-record value limit are split + /// at UTF-8 codepoint boundaries and written as multiple records that + /// share one key, atomically under a single lock via + /// [`KvpPoolStore::append_multiple`]. + /// + /// Returns [`KvpError::EventFieldContainsDelimiter`] if the + /// `event_prefix`, `vm_id`, `name`, or `event_id` contains the `|` + /// key delimiter, which would make the key ambiguous to + /// [`records`](Self::records). + pub fn emit(&self, event: &DiagnosticEvent) -> Result<(), KvpError> { + reject_delimiter("event_prefix", &self.event_prefix)?; + reject_delimiter("vm_id", &self.vm_id)?; + reject_delimiter("name", &event.name)?; + reject_delimiter("event_id", &event.event_id)?; + + let key = format_event_key( + &self.event_prefix, + &self.vm_id, + event.level, + &event.name, + &event.event_id, + ); + + self.write_chunked(&key, &event.message) + } + + /// Split `value` at the store's per-record limit and append every + /// chunk under the same `key` in one atomic batch. + fn write_chunked(&self, key: &str, value: &str) -> Result<(), KvpError> { + let chunks = chunk_at_char_boundary(value, self.store.max_value_size()); + self.store + .append_multiple(chunks.into_iter().map(|chunk| (key, chunk))) + } + + /// Read every record, reassembling chunked events and classifying + /// each into a [`DiagnosticRecord`]. + /// + /// Records are returned in on-disk order. Consecutive records that + /// share a key are one event; because [`emit`](Self::emit) writes an + /// event's chunks contiguously under a single lock, reassembly is + /// correct even under concurrent writers. + pub fn records(&self) -> Result, KvpError> { + Ok(reassemble(self.store.dump()?)) + } + + /// Read back only the records that decode as diagnostic events, in + /// on-disk order. + pub fn events(&self) -> Result, KvpError> { + Ok(self + .records()? + .into_iter() + .filter_map(|record| match record { + DiagnosticRecord::Event { event, .. } => Some(event), + DiagnosticRecord::Raw { .. } + | DiagnosticRecord::Malformed { .. } => None, + }) + .collect()) + } + + /// Remove this layer's events — keys that parse as an event key + /// whose `prefix` and `vm_id` match this instance — under a single + /// lock. Raw records such as `PROVISIONING_REPORT` are left intact. + pub fn clear(&self) -> Result<(), KvpError> { + let keys: Vec = self + .store + .dump()? + .into_iter() + .filter_map(|(key, _)| { + let is_mine = matches!( + classify_key(&key), + KeyClass::Event { prefix, vm_id, .. } + if prefix == self.event_prefix + && vm_id == self.vm_id + ); + is_mine.then_some(key) + }) + .collect(); + self.store.delete_multiple(keys)?; + Ok(()) + } +} + +/// Group consecutive same-key records from [`KvpPoolStore::dump`] and +/// classify each group into a [`DiagnosticRecord`]. +fn reassemble(dumped: Vec<(String, String)>) -> Vec { + let mut records = Vec::new(); + let mut dumped = dumped.into_iter().peekable(); + + while let Some((key, value)) = dumped.next() { + let mut message = value; + let mut chunks = 1; + while dumped.peek().is_some_and(|(next, _)| *next == key) { + let (_, next_value) = dumped.next().expect("peeked value exists"); + message.push_str(&next_value); + chunks += 1; + } + records.push(classify_record(key, message, chunks)); + } + + records +} + +/// Turn one reassembled key/value group into a [`DiagnosticRecord`]. +fn classify_record( + key: String, + value: String, + chunks: usize, +) -> DiagnosticRecord { + match classify_key(&key) { + KeyClass::Event { + level, + name, + event_id, + .. + } => DiagnosticRecord::Event { + event: DiagnosticEvent { + level, + name: name.to_string(), + event_id: event_id.to_string(), + message: value, + }, + chunks, + }, + KeyClass::Malformed { reason } => { + DiagnosticRecord::Malformed { key, value, reason } + } + KeyClass::Raw => DiagnosticRecord::Raw { key, value }, + } +} + +#[cfg(test)] +mod tests { + use super::*; + use rstest::rstest; + + const PREFIX: &str = "azure-init-0.1.0"; + const VM_ID: &str = "3f2504e0-4f89-41d3-9a0c-0305e82c3301"; + const EVENT_ID: &str = "8f3e9c4a-1b2c-4d5e-9f01-234567890abc"; + + #[test] + fn event_key_formats_and_classifies() { + let formatted = format_event_key( + PREFIX, + VM_ID, + Level::INFO, + "user:create_user", + EVENT_ID, + ); + assert_eq!( + formatted, + format!("{PREFIX}|{VM_ID}|INFO|user:create_user|{EVENT_ID}") + ); + assert!(matches!( + classify_key(&formatted), + KeyClass::Event { + prefix, + vm_id, + level, + name, + event_id, + } if prefix == PREFIX + && vm_id == VM_ID + && level == Level::INFO + && name == "user:create_user" + && event_id == EVENT_ID + )); + } + + #[test] + fn classify_round_trips_every_level() { + for expected in [ + Level::ERROR, + Level::WARN, + Level::INFO, + Level::DEBUG, + Level::TRACE, + ] { + let key = format_event_key( + PREFIX, + VM_ID, + expected, + "span:event", + EVENT_ID, + ); + assert!(matches!( + classify_key(&key), + KeyClass::Event { level, .. } if level == expected + )); + } + } + + /// Map a key to its [`KeyClass`] discriminant for table-driven tests. + fn class_of(key: &str) -> &'static str { + match classify_key(key) { + KeyClass::Event { .. } => "event", + KeyClass::Malformed { .. } => "malformed", + KeyClass::Raw => "raw", + } + } + + #[rstest] + #[case::event("p|vm|INFO|name|id", "event")] + #[case::raw_single_segment("PROVISIONING_REPORT", "raw")] + #[case::raw_too_few_segments("a|b|INFO|c", "raw")] + #[case::raw_too_many_segments("a|b|INFO|c|d|e", "raw")] + #[case::malformed_bad_level("p|vm|NOTALEVEL|name|id", "malformed")] + #[case::malformed_other_level("p|vm|NOPE|name|id", "malformed")] + fn classify_key_categorizes(#[case] key: &str, #[case] expected: &str) { + assert_eq!(class_of(key), expected); + } + + #[rstest] + #[case::empty("", 4, vec![""])] + #[case::shorter_than_max("abc", 8, vec!["abc"])] + #[case::exact_multiple("abcdef", 2, vec!["ab", "cd", "ef"])] + #[case::ascii_remainder("abcde", 2, vec!["ab", "cd", "e"])] + #[case::two_byte_boundary("aéb", 2, vec!["a", "é", "b"])] + #[case::oversized_three_byte("€", 1, vec!["€"])] + #[case::oversized_repeated("€€", 1, vec!["€", "€"])] + fn chunk_splits_at_utf8_boundaries( + #[case] input: &str, + #[case] max_bytes: usize, + #[case] expected: Vec<&str>, + ) { + assert_eq!(chunk_at_char_boundary(input, max_bytes), expected); + } + + #[test] + fn chunk_reassembles_multibyte_payload() { + let payload = "🚀".repeat(100); + let chunks = chunk_at_char_boundary(&payload, 7); + assert!(chunks.iter().all(|chunk| chunk.len() <= 7)); + assert_eq!(chunks.concat(), payload); + } + + #[test] + fn diagnostic_event_new_generates_uuid_event_id() { + let event = DiagnosticEvent::new(Level::INFO, "span:name", "message"); + assert!(Uuid::parse_str(&event.event_id).is_ok()); + } + + #[test] + fn reject_delimiter_flags_pipe() { + assert!(reject_delimiter("name", "no pipe here").is_ok()); + let err = reject_delimiter("name", "has|pipe").unwrap_err(); + assert!(matches!( + err, + KvpError::EventFieldContainsDelimiter { field: "name" } + )); + assert_eq!( + err.to_string(), + "event key field 'name' must not contain '|'" + ); + } + + #[test] + fn reassemble_groups_chunks_and_classifies() { + let key = format_event_key( + PREFIX, + VM_ID, + Level::INFO, + "config:dump", + EVENT_ID, + ); + + let dumped = vec![ + (key.clone(), "part-one/".to_string()), + (key.clone(), "part-two".to_string()), + ( + "PROVISIONING_REPORT".to_string(), + "result=success".to_string(), + ), + ("p|vm|NOPE|name|id".to_string(), "junk".to_string()), + ]; + + let records = reassemble(dumped); + assert_eq!(records.len(), 3); + + assert_eq!( + records[0], + DiagnosticRecord::Event { + event: DiagnosticEvent { + level: Level::INFO, + name: "config:dump".to_string(), + event_id: EVENT_ID.to_string(), + message: "part-one/part-two".to_string(), + }, + chunks: 2, + } + ); + assert!(matches!(&records[1], DiagnosticRecord::Raw { key, .. } + if key == "PROVISIONING_REPORT")); + assert!(matches!(&records[2], DiagnosticRecord::Malformed { .. })); + } + + #[test] + fn reassemble_keeps_distinct_adjacent_keys_separate() { + let make = |event_id: &str| { + format_event_key(PREFIX, VM_ID, Level::INFO, "span:name", event_id) + }; + let dumped = vec![ + (make("id-1"), "first".to_string()), + (make("id-2"), "second".to_string()), + ]; + let records = reassemble(dumped); + assert_eq!(records.len(), 2); + assert!(matches!( + &records[0], + DiagnosticRecord::Event { chunks: 1, .. } + )); + assert!(matches!( + &records[1], + DiagnosticRecord::Event { chunks: 1, .. } + )); + } +} diff --git a/libazureinit-kvp/src/error.rs b/libazureinit-kvp/src/error.rs index 02b9ec9c..71b75abb 100644 --- a/libazureinit-kvp/src/error.rs +++ b/libazureinit-kvp/src/error.rs @@ -11,6 +11,10 @@ pub enum KvpError { EmptyKey, /// An underlying I/O error. Io(io::Error), + /// An event key field (`event_prefix`, `vm_id`, `name`, or + /// `event_id`) contained the `|` delimiter, which would make the + /// formatted event key ambiguous to parse back. + EventFieldContainsDelimiter { field: &'static str }, /// The key contains a null byte, which is incompatible with the /// on-disk format (null-padded fixed-width fields). KeyContainsNull, @@ -29,6 +33,9 @@ impl fmt::Display for KvpError { fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { match self { Self::EmptyKey => write!(f, "KVP key must not be empty"), + Self::EventFieldContainsDelimiter { field } => { + write!(f, "event key field '{field}' must not contain '|'") + } Self::Io(e) => write!(f, "{e}"), Self::KeyContainsNull => { write!(f, "KVP key must not contain null bytes") diff --git a/libazureinit-kvp/src/lib.rs b/libazureinit-kvp/src/lib.rs index f17b66f7..2e94b14a 100644 --- a/libazureinit-kvp/src/lib.rs +++ b/libazureinit-kvp/src/lib.rs @@ -9,14 +9,20 @@ //! - [`ProvisioningReport`]: structured provisioning health report that //! is persisted as the single `PROVISIONING_REPORT` record with //! [`write_report`]. +//! - [`DiagnosticsKvp`]: typed view over [`KvpPoolStore`] that formats, +//! chunks, and reassembles azure-init diagnostic events. mod cli; +mod diagnostics; mod error; mod report; mod store; mod vm_id; pub use cli::run; +pub use diagnostics::{ + DiagnosticEvent, DiagnosticRecord, DiagnosticsKvp, MAX_CHUNK_BYTES, +}; pub use error::KvpError; pub use report::{ write_report, ProvisioningReport, ReportPpsType, PROVISIONING_REPORT_KEY, diff --git a/libazureinit-kvp/tests/cli.rs b/libazureinit-kvp/tests/cli.rs index ef485314..60416526 100644 --- a/libazureinit-kvp/tests/cli.rs +++ b/libazureinit-kvp/tests/cli.rs @@ -298,3 +298,201 @@ fn report_failure_rejects_invalid_supporting_data() { .unwrap() .contains("key=value")); } + +#[test] +fn diag_dump_json_reassembles_and_classifies() { + let dir = TempDir::new().unwrap(); + // A two-chunk event (same key repeated) plus a raw record. + assert_success(kvp(&with_dir( + &dir, + &["write", "--append", "azure-init-x|vm|INFO|a:b|id1", "one/"], + ))); + assert_success(kvp(&with_dir( + &dir, + &["write", "--append", "azure-init-x|vm|INFO|a:b|id1", "two"], + ))); + assert_success(kvp(&with_dir( + &dir, + &["write", "PROVISIONING_REPORT", "result=success"], + ))); + + let out = assert_success(kvp(&with_dir(&dir, &["--json", "diag", "dump"]))); + assert!(out.contains("\"kind\":\"event\"")); + assert!(out.contains("\"chunks\":2")); + assert!(out.contains("\"message\":\"one/two\"")); + // Raw records are hidden without --include-raw. + assert!(!out.contains("PROVISIONING_REPORT")); + + let out_raw = assert_success(kvp(&with_dir( + &dir, + &["--json", "diag", "dump", "--include-raw"], + ))); + assert!(out_raw.contains("\"kind\":\"raw\"")); + assert!(out_raw.contains("PROVISIONING_REPORT")); +} + +#[test] +fn diag_events_filters_by_level() { + let dir = TempDir::new().unwrap(); + assert_success(kvp(&with_dir( + &dir, + &["write", "--append", "p|vm|INFO|a:b|i1", "info-msg"], + ))); + assert_success(kvp(&with_dir( + &dir, + &["write", "--append", "p|vm|ERROR|c:d|i2", "err-msg"], + ))); + + let out = assert_success(kvp(&with_dir( + &dir, + &["diag", "events", "--level", "error"], + ))); + assert!(out.contains("err-msg")); + assert!(!out.contains("info-msg")); +} + +#[test] +fn diag_clear_requires_confirmation_then_scopes_removal() { + let dir = TempDir::new().unwrap(); + assert_success(kvp(&with_dir( + &dir, + &["write", "--append", "p|vm|INFO|a:b|i1", "msg"], + ))); + + // Without --yes the command refuses and exits with a usage error. + let refused = kvp(&with_dir( + &dir, + &["diag", "clear", "--vm-id", "vm", "--event-prefix", "p"], + )); + assert_eq!(refused.status.code(), Some(2)); + + // With --yes the matching event is removed. + assert_success(kvp(&with_dir( + &dir, + &[ + "diag", + "clear", + "--vm-id", + "vm", + "--event-prefix", + "p", + "--yes", + ], + ))); + let out = assert_success(kvp(&with_dir(&dir, &["diag", "dump"]))); + assert!(out.trim().is_empty()); +} + +#[test] +fn diag_dump_text_renders_all_record_kinds() { + let dir = TempDir::new().unwrap(); + assert_success(kvp(&with_dir( + &dir, + &["write", "--append", "p|vm|INFO|a:b|id1", "one/"], + ))); + assert_success(kvp(&with_dir( + &dir, + &["write", "--append", "p|vm|INFO|a:b|id1", "two"], + ))); + assert_success(kvp(&with_dir( + &dir, + &["write", "PROVISIONING_REPORT", "result=success"], + ))); + assert_success(kvp(&with_dir( + &dir, + &["write", "--append", "p|vm|NOPE|c:d|id2", "junk"], + ))); + + let out = assert_success(kvp(&with_dir( + &dir, + &["diag", "dump", "--include-raw"], + ))); + assert!(out.contains( + "event level=INFO name=a:b event_id=id1 chunks=2 message=one/two" + )); + assert!(out.contains("raw key=PROVISIONING_REPORT value=result=success")); + assert!(out.contains("malformed key=p|vm|NOPE|c:d|id2")); + assert!(out.contains("value=junk")); +} + +#[test] +fn diag_tail_limits_to_last_events() { + let dir = TempDir::new().unwrap(); + assert_success(kvp(&with_dir( + &dir, + &["write", "--append", "p|vm|INFO|a:b|i1", "first"], + ))); + assert_success(kvp(&with_dir( + &dir, + &["write", "--append", "p|vm|INFO|c:d|i2", "second"], + ))); + + let out = + assert_success(kvp(&with_dir(&dir, &["diag", "tail", "-n", "1"]))); + assert!(out.contains("second")); + assert!(!out.contains("first")); +} + +#[test] +fn diag_events_filters_by_name_substring() { + let dir = TempDir::new().unwrap(); + assert_success(kvp(&with_dir( + &dir, + &["write", "--append", "p|vm|INFO|user:add|i1", "u"], + ))); + assert_success(kvp(&with_dir( + &dir, + &["write", "--append", "p|vm|INFO|ssh:key|i2", "s"], + ))); + + let out = assert_success(kvp(&with_dir( + &dir, + &["diag", "events", "--name", "ssh"], + ))); + assert!(out.contains("ssh:key")); + assert!(!out.contains("user:add")); +} + +#[test] +fn diag_events_rejects_invalid_level() { + let dir = TempDir::new().unwrap(); + let output = kvp(&with_dir(&dir, &["diag", "events", "--level", "bogus"])); + assert_eq!(output.status.code(), Some(2)); +} + +#[test] +fn diag_json_covers_malformed_events_and_clear() { + let dir = TempDir::new().unwrap(); + assert_success(kvp(&with_dir( + &dir, + &["write", "--append", "p|vm|INFO|a:b|i1", "hello"], + ))); + assert_success(kvp(&with_dir( + &dir, + &["write", "--append", "p|vm|NOPE|c:d|i2", "junk"], + ))); + + let dump = + assert_success(kvp(&with_dir(&dir, &["--json", "diag", "dump"]))); + assert!(dump.contains("\"kind\":\"event\"")); + assert!(dump.contains("\"kind\":\"malformed\"")); + assert!(dump.contains("\"reason\":")); + + let events = + assert_success(kvp(&with_dir(&dir, &["--json", "diag", "events"]))); + assert!(events.contains("\"message\":\"hello\"")); + let cleared = assert_success(kvp(&with_dir( + &dir, + &[ + "--json", + "diag", + "clear", + "--vm-id", + "vm", + "--event-prefix", + "p", + "--yes", + ], + ))); + assert!(cleared.contains("\"cleared\":true")); +} diff --git a/libazureinit-kvp/tests/diagnostics.rs b/libazureinit-kvp/tests/diagnostics.rs new file mode 100644 index 00000000..d846aab0 --- /dev/null +++ b/libazureinit-kvp/tests/diagnostics.rs @@ -0,0 +1,234 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +//! Integration tests for the [`DiagnosticsKvp`] layer: emit/read +//! round-trips, chunk reassembly, classification, scoped clearing, and +//! the concurrent-write atomicity guarantee that keeps chunked events +//! from interleaving. + +use std::thread; + +use libazureinit_kvp::{ + DiagnosticEvent, DiagnosticRecord, DiagnosticsKvp, KvpPool, KvpPoolStore, + PoolMode, MAX_CHUNK_BYTES, +}; +use tempfile::TempDir; +use tracing::Level; + +const PREFIX: &str = "azure-init-test"; +const VM_ID: &str = "vm-abc"; + +fn diagnostics(dir: &TempDir) -> DiagnosticsKvp { + let store = + KvpPoolStore::new_in(KvpPool::Guest, dir.path(), PoolMode::Safe) + .unwrap(); + DiagnosticsKvp::new(store, VM_ID, PREFIX) +} + +#[test] +fn short_event_round_trips_as_single_record() { + let dir = TempDir::new().unwrap(); + let diag = diagnostics(&dir); + + assert_eq!(diag.vm_id(), VM_ID); + assert_eq!(diag.event_prefix(), PREFIX); + + let event = + DiagnosticEvent::new(Level::INFO, "user:create_user", "created"); + diag.emit(&event).unwrap(); + + assert_eq!(diag.store().dump().unwrap().len(), 1); + + let records = diag.records().unwrap(); + assert_eq!(records.len(), 1); + match &records[0] { + DiagnosticRecord::Event { + event: decoded, + chunks, + } => { + assert_eq!(*chunks, 1); + assert_eq!(decoded.level, Level::INFO); + assert_eq!(decoded.name, "user:create_user"); + assert_eq!(decoded.event_id, event.event_id); + assert_eq!(decoded.message, "created"); + } + other => panic!("expected event, got {other:?}"), + } +} + +#[test] +fn long_event_splits_across_records_and_reassembles() { + let dir = TempDir::new().unwrap(); + let diag = diagnostics(&dir); + + let message = "x".repeat(MAX_CHUNK_BYTES * 3 + 50); + let event = DiagnosticEvent::new(Level::DEBUG, "config:dump", &message); + diag.emit(&event).unwrap(); + + // Split across four records that all share one key (split keys). + let dumped = diag.store().dump().unwrap(); + assert_eq!(dumped.len(), 4); + let key = &dumped[0].0; + assert!(dumped.iter().all(|(k, _)| k == key)); + + let records = diag.records().unwrap(); + assert_eq!(records.len(), 1); + match &records[0] { + DiagnosticRecord::Event { + event: decoded, + chunks, + } => { + assert_eq!(*chunks, 4); + assert_eq!(decoded.message, message); + } + other => panic!("expected event, got {other:?}"), + } +} + +#[test] +fn injected_malformed_key_is_classified() { + let dir = TempDir::new().unwrap(); + let diag = diagnostics(&dir); + + // Five segments but an unrecognized level. + diag.store() + .append(&format!("{PREFIX}|{VM_ID}|NOPE|bad:level|id"), "junk") + .unwrap(); + + let records = diag.records().unwrap(); + assert_eq!(records.len(), 1); + assert!(matches!( + &records[0], + DiagnosticRecord::Malformed { reason, .. } if reason.contains("NOPE") + )); +} + +#[test] +fn mixed_records_round_trip_together() { + let dir = TempDir::new().unwrap(); + let diag = diagnostics(&dir); + + diag.emit(&DiagnosticEvent::new(Level::INFO, "a:b", "short")) + .unwrap(); + diag.emit(&DiagnosticEvent::new( + Level::WARN, + "c:d", + "y".repeat(MAX_CHUNK_BYTES + 5), + )) + .unwrap(); + diag.store() + .append("PROVISIONING_REPORT", "result=success") + .unwrap(); + diag.store() + .append(&format!("{PREFIX}|{VM_ID}|NOPE|e:f|id"), "junk") + .unwrap(); + + let records = diag.records().unwrap(); + // Two events + one raw + one malformed. + assert_eq!(records.len(), 4); + assert_eq!(diag.events().unwrap().len(), 2); +} + +#[test] +fn clear_removes_events_but_keeps_raw() { + let dir = TempDir::new().unwrap(); + let diag = diagnostics(&dir); + + diag.emit(&DiagnosticEvent::new(Level::INFO, "a:b", "e1")) + .unwrap(); + diag.emit(&DiagnosticEvent::new( + Level::DEBUG, + "c:d", + "z".repeat(MAX_CHUNK_BYTES * 2), + )) + .unwrap(); + diag.store() + .append("PROVISIONING_REPORT", "result=success") + .unwrap(); + + diag.clear().unwrap(); + + let records = diag.records().unwrap(); + assert_eq!(records.len(), 1); + assert!(matches!( + &records[0], + DiagnosticRecord::Raw { key, .. } if key == "PROVISIONING_REPORT" + )); + assert!(diag.events().unwrap().is_empty()); +} + +#[test] +fn clear_is_scoped_to_matching_prefix_and_vm_id() { + let dir = TempDir::new().unwrap(); + let diag = diagnostics(&dir); + + diag.emit(&DiagnosticEvent::new(Level::INFO, "a:b", "mine")) + .unwrap(); + // An event from a different agent/VM must survive clear(). + diag.store() + .append("other-agent|other-vm|INFO|x:y|id", "theirs") + .unwrap(); + + diag.clear().unwrap(); + + let events = diag.events().unwrap(); + assert_eq!(events.len(), 1); + assert_eq!(events[0].message, "theirs"); +} + +#[test] +fn emit_rejects_delimiter_in_event_fields() { + let dir = TempDir::new().unwrap(); + let diag = diagnostics(&dir); + + // A pipe in the name would produce an ambiguous six-segment key. + let event = DiagnosticEvent::new(Level::INFO, "a|b", "msg"); + assert!(diag.emit(&event).is_err()); + // Nothing was written. + assert!(diag.store().dump().unwrap().is_empty()); +} + +#[test] +fn concurrent_multichunk_emits_reassemble_without_interleaving() { + let dir = TempDir::new().unwrap(); + let diag = diagnostics(&dir); + + const THREADS: usize = 5; + const PER_THREAD: usize = 8; + // Force three chunks per event. + let len = MAX_CHUNK_BYTES * 2 + 7; + + let handles: Vec<_> = (0..THREADS) + .map(|t| { + let diag = diag.clone(); + let marker = (b'a' + t as u8) as char; + thread::spawn(move || { + for _ in 0..PER_THREAD { + let message = marker.to_string().repeat(len); + let event = DiagnosticEvent::new( + Level::INFO, + format!("thread:{marker}"), + message, + ); + diag.emit(&event).unwrap(); + } + }) + }) + .collect(); + for handle in handles { + handle.join().unwrap(); + } + + let events = diag.events().unwrap(); + // If any event's chunks had been split by an interleaving writer, the + // key would appear as multiple groups and the count would be wrong. + assert_eq!(events.len(), THREADS * PER_THREAD); + for event in &events { + // Each message is homogeneous and full length: chunks stayed + // contiguous on disk. + assert_eq!(event.message.len(), len); + let first = event.message.chars().next().unwrap(); + assert!(event.message.chars().all(|c| c == first)); + assert_eq!(event.name, format!("thread:{first}")); + } +} From a13158c802cb3c260690263217c67d19009186f8 Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Wed, 22 Jul 2026 16:52:25 -0700 Subject: [PATCH 02/32] fix(kvp): preserve chunked diagnostic events Append a chunk index to diagnostic event keys so the Hyper-V host retains all chunks. Reassembly and cleanup now operate on the base event key to correctly reconstruct and delete multi-record events. --- libazureinit-kvp/src/diagnostics.rs | 145 ++++++++++++++++++++++---- libazureinit-kvp/tests/diagnostics.rs | 41 +++++++- 2 files changed, 162 insertions(+), 24 deletions(-) diff --git a/libazureinit-kvp/src/diagnostics.rs b/libazureinit-kvp/src/diagnostics.rs index e71f74bb..e65f676e 100644 --- a/libazureinit-kvp/src/diagnostics.rs +++ b/libazureinit-kvp/src/diagnostics.rs @@ -14,8 +14,12 @@ //! - **Chunking**: a single KVP record caps the value at the store's //! per-record limit (see [`MAX_CHUNK_BYTES`] for the safe-mode value). //! Longer messages are split at UTF-8 codepoint boundaries into -//! multiple records that share one key, written atomically under a -//! single lock, and reassembled on read. +//! multiple records written atomically under a single lock. Each chunk +//! gets a unique key — the event key with a `|` suffix +//! (`|||||`), +//! matching cloud-init's naming — so the Hyper-V host, which keeps only +//! one record per key, retains every chunk. The chunks are regrouped +//! into one event on read. //! - **Classification**: [`records`](DiagnosticsKvp::records) sorts every //! stored record into a [`DiagnosticRecord`] — a reassembled //! [`DiagnosticEvent`], an unstructured [`Raw`](DiagnosticRecord::Raw) @@ -51,8 +55,8 @@ //! "Creating user azureuser", //! ))?; //! -//! // A long message is split across records that share one key and is -//! // reassembled transparently on read. +//! // A long message is split across records each with a unique +//! // `|`-suffixed key, and reassembled on read. //! let long = "x".repeat(MAX_CHUNK_BYTES * 2 + 10); //! diagnostics //! .emit(&DiagnosticEvent::new(Level::DEBUG, "config:dump", &long))?; @@ -314,9 +318,12 @@ impl DiagnosticsKvp { /// Write `event`. /// /// Messages longer than the store's per-record value limit are split - /// at UTF-8 codepoint boundaries and written as multiple records that - /// share one key, atomically under a single lock via - /// [`KvpPoolStore::append_multiple`]. + /// at UTF-8 codepoint boundaries and written as multiple records + /// atomically under a single lock via + /// [`KvpPoolStore::append_multiple`]. Each chunk is keyed by the event + /// key with a `|` suffix so every record is unique — + /// the Hyper-V host keeps only one record per key — and the chunks are + /// regrouped by [`records`](Self::records) on read. /// /// Returns [`KvpError::EventFieldContainsDelimiter`] if the /// `event_prefix`, `vm_id`, `name`, or `event_id` contains the `|` @@ -339,21 +346,41 @@ impl DiagnosticsKvp { self.write_chunked(&key, &event.message) } - /// Split `value` at the store's per-record limit and append every - /// chunk under the same `key` in one atomic batch. + /// Split `value` at the store's per-record limit and append the + /// chunks under `key` in one atomic batch. + /// + /// A single-record value keeps the bare event `key`. A value that + /// spans multiple records gets one record per chunk, each keyed + /// `|` (`0`, `1`, …) so no two records collide — + /// the Hyper-V host keeps only one record per key. [`reassemble`] + /// strips the subevent index to regroup the chunks on read. fn write_chunked(&self, key: &str, value: &str) -> Result<(), KvpError> { let chunks = chunk_at_char_boundary(value, self.store.max_value_size()); - self.store - .append_multiple(chunks.into_iter().map(|chunk| (key, chunk))) + if chunks.len() == 1 { + return self + .store + .append_multiple(chunks.into_iter().map(|chunk| (key, chunk))); + } + let records: Vec<(String, &str)> = chunks + .into_iter() + .enumerate() + .map(|(subevent_index, chunk)| { + let chunk_key = + format!("{key}{EVENT_KEY_DELIMITER}{subevent_index}"); + (chunk_key, chunk) + }) + .collect(); + self.store.append_multiple(records) } /// Read every record, reassembling chunked events and classifying /// each into a [`DiagnosticRecord`]. /// /// Records are returned in on-disk order. Consecutive records that - /// share a key are one event; because [`emit`](Self::emit) writes an - /// event's chunks contiguously under a single lock, reassembly is - /// correct even under concurrent writers. + /// share an event key — ignoring the `|` suffix — are + /// one event; because [`emit`](Self::emit) writes an event's chunks + /// contiguously under a single lock, reassembly is correct even under + /// concurrent writers. pub fn records(&self) -> Result, KvpError> { Ok(reassemble(self.store.dump()?)) } @@ -373,8 +400,10 @@ impl DiagnosticsKvp { } /// Remove this layer's events — keys that parse as an event key - /// whose `prefix` and `vm_id` match this instance — under a single - /// lock. Raw records such as `PROVISIONING_REPORT` are left intact. + /// (including every `|` chunk of a multi-record + /// event) whose `prefix` and `vm_id` match this instance — under a + /// single lock. Raw records such as `PROVISIONING_REPORT` are left + /// intact. pub fn clear(&self) -> Result<(), KvpError> { let keys: Vec = self .store @@ -382,7 +411,7 @@ impl DiagnosticsKvp { .into_iter() .filter_map(|(key, _)| { let is_mine = matches!( - classify_key(&key), + classify_key(base_event_key(&key)), KeyClass::Event { prefix, vm_id, .. } if prefix == self.event_prefix && vm_id == self.vm_id @@ -395,21 +424,47 @@ impl DiagnosticsKvp { } } -/// Group consecutive same-key records from [`KvpPoolStore::dump`] and -/// classify each group into a [`DiagnosticRecord`]. +/// The event key a chunk belongs to. +/// +/// [`DiagnosticsKvp::write_chunked`] gives each chunk of a multi-record +/// event a unique key by appending a `|` (cloud-init's +/// term) to the event key, so the Hyper-V host — which keeps only one +/// record per key — retains every chunk. This returns the shared event +/// key used to regroup them on read: for a chunk key +/// `|` it strips the trailing index; any other +/// key (a single-record event, `PROVISIONING_REPORT`, a malformed key, …) +/// is returned unchanged. +fn base_event_key(key: &str) -> &str { + if let Some((base, subevent_index)) = key.rsplit_once(EVENT_KEY_DELIMITER) { + if subevent_index.parse::().is_ok() + && matches!(classify_key(base), KeyClass::Event { .. }) + { + return base; + } + } + key +} + +/// Group consecutive records sharing an event key — chunk +/// `|` suffixes stripped — from [`KvpPoolStore::dump`] +/// and classify each group into a [`DiagnosticRecord`]. fn reassemble(dumped: Vec<(String, String)>) -> Vec { let mut records = Vec::new(); let mut dumped = dumped.into_iter().peekable(); while let Some((key, value)) = dumped.next() { + let base = base_event_key(&key).to_string(); let mut message = value; let mut chunks = 1; - while dumped.peek().is_some_and(|(next, _)| *next == key) { + while dumped + .peek() + .is_some_and(|(next, _)| base_event_key(next) == base) + { let (_, next_value) = dumped.next().expect("peeked value exists"); message.push_str(&next_value); chunks += 1; } - records.push(classify_record(key, message, chunks)); + records.push(classify_record(base, message, chunks)); } records @@ -628,4 +683,52 @@ mod tests { DiagnosticRecord::Event { chunks: 1, .. } )); } + + #[rstest] + #[case::indexed_chunk("p|vm|INFO|name|id|0", "p|vm|INFO|name|id")] + #[case::indexed_chunk_multi_digit( + "p|vm|INFO|name|id|12", + "p|vm|INFO|name|id" + )] + #[case::single_event_unchanged("p|vm|INFO|name|id", "p|vm|INFO|name|id")] + #[case::raw_unchanged("PROVISIONING_REPORT", "PROVISIONING_REPORT")] + #[case::non_event_numeric_tail_unchanged("foo|3", "foo|3")] + #[case::malformed_unchanged("p|vm|NOPE|name|id", "p|vm|NOPE|name|id")] + fn base_event_key_strips_event_subevent_index( + #[case] key: &str, + #[case] expected: &str, + ) { + assert_eq!(base_event_key(key), expected); + } + + #[test] + fn reassemble_groups_indexed_chunk_keys() { + let base = format_event_key( + PREFIX, + VM_ID, + Level::INFO, + "config:dump", + EVENT_ID, + ); + let dumped = vec![ + (format!("{base}|0"), "part-one/".to_string()), + (format!("{base}|1"), "part-two/".to_string()), + (format!("{base}|2"), "part-three".to_string()), + ]; + + let records = reassemble(dumped); + assert_eq!(records.len(), 1); + assert_eq!( + records[0], + DiagnosticRecord::Event { + event: DiagnosticEvent { + level: Level::INFO, + name: "config:dump".to_string(), + event_id: EVENT_ID.to_string(), + message: "part-one/part-two/part-three".to_string(), + }, + chunks: 3, + } + ); + } } diff --git a/libazureinit-kvp/tests/diagnostics.rs b/libazureinit-kvp/tests/diagnostics.rs index d846aab0..e1f5694a 100644 --- a/libazureinit-kvp/tests/diagnostics.rs +++ b/libazureinit-kvp/tests/diagnostics.rs @@ -65,11 +65,21 @@ fn long_event_splits_across_records_and_reassembles() { let event = DiagnosticEvent::new(Level::DEBUG, "config:dump", &message); diag.emit(&event).unwrap(); - // Split across four records that all share one key (split keys). + // Split across four records, each with a unique `|` + // key so the Hyper-V host (one record per key) keeps every chunk; + // they share one event-key base. let dumped = diag.store().dump().unwrap(); assert_eq!(dumped.len(), 4); - let key = &dumped[0].0; - assert!(dumped.iter().all(|(k, _)| k == key)); + let base_of = |k: &str| k.rsplit_once('|').unwrap().0.to_string(); + let base = base_of(&dumped[0].0); + assert!( + dumped.iter().all(|(k, _)| base_of(k) == base), + "all chunks share one event-key base" + ); + let mut keys: Vec = dumped.iter().map(|(k, _)| k.clone()).collect(); + keys.sort(); + keys.dedup(); + assert_eq!(keys.len(), 4, "each chunk must have a unique key"); let records = diag.records().unwrap(); assert_eq!(records.len(), 1); @@ -85,6 +95,31 @@ fn long_event_splits_across_records_and_reassembles() { } } +#[test] +fn multi_chunk_event_uses_unique_keys_so_host_keeps_all() { + let dir = TempDir::new().unwrap(); + let diag = diagnostics(&dir); + + let message = "z".repeat(MAX_CHUNK_BYTES * 2 + 1); + diag.emit(&DiagnosticEvent::new(Level::INFO, "big:event", &message)) + .unwrap(); + + // Three records, no two sharing a key: the Hyper-V host keeps only one + // record per key, so shared keys would silently drop chunks. + let dumped = diag.store().dump().unwrap(); + assert_eq!(dumped.len(), 3); + let total = dumped.len(); + let mut keys: Vec = dumped.into_iter().map(|(k, _)| k).collect(); + keys.sort(); + keys.dedup(); + assert_eq!(keys.len(), total, "chunk keys must be unique"); + + // The event still reassembles to the full message. + let events = diag.events().unwrap(); + assert_eq!(events.len(), 1); + assert_eq!(events[0].message, message); +} + #[test] fn injected_malformed_key_is_classified() { let dir = TempDir::new().unwrap(); From 7caa370204081daf416ae97e925675f2928f4047 Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Thu, 23 Jul 2026 11:01:17 -0700 Subject: [PATCH 03/32] refactor(kvp): fold diag subcommands into dump --parse-diagnostics and clear --diagnostics --- libazureinit-kvp/src/cli.rs | 235 ++++++++++++++------------ libazureinit-kvp/src/diagnostics.rs | 36 ++-- libazureinit-kvp/tests/cli.rs | 145 ++++++++++------ libazureinit-kvp/tests/diagnostics.rs | 20 ++- 4 files changed, 244 insertions(+), 192 deletions(-) diff --git a/libazureinit-kvp/src/cli.rs b/libazureinit-kvp/src/cli.rs index 08d3cc83..4a6b00b5 100644 --- a/libazureinit-kvp/src/cli.rs +++ b/libazureinit-kvp/src/cli.rs @@ -89,7 +89,41 @@ enum Command { /// Print store metadata. Info, /// Print every record in insertion order as KEY=VALUE lines. - Dump, + /// + /// With --parse-diagnostics, reassemble chunked diagnostic events and + /// decode each record instead of printing raw KEY=VALUE lines. + Dump { + /// Reassemble chunked diagnostic events and decode each record as + /// an event, raw, or malformed entry. + #[arg(long)] + parse_diagnostics: bool, + /// Also print raw (non-event) records such as PROVISIONING_REPORT. + /// Only applies to the unfiltered view; --level/--name/--tail + /// produce an events-only view where raw records never appear. + #[arg( + long, + requires = "parse_diagnostics", + conflicts_with_all = ["level", "name", "tail"] + )] + include_raw: bool, + /// Only show events at this level (error, warn, info, debug, + /// trace). + #[arg(long, requires = "parse_diagnostics")] + level: Option, + /// Only show events whose name contains this substring. + #[arg(long, requires = "parse_diagnostics")] + name: Option, + /// Print only the last COUNT events (default 20 when COUNT is + /// omitted). + #[arg( + short = 'n', + long = "tail", + num_args = 0..=1, + default_missing_value = "20", + requires = "parse_diagnostics" + )] + tail: Option, + }, /// Print key=last_value entries sorted by key. Entries, /// Print the last value for KEY (exit 1 if missing). @@ -121,11 +155,16 @@ enum Command { #[arg(required = true)] keys: Vec, }, - /// Clear the pool. Pass --if-stale to clear only when stale. + /// Clear the pool. Pass --if-stale to clear only when stale, or + /// --diagnostics to remove only diagnostic event keys. Clear { /// Only clear if the store is currently stale. - #[arg(long = "if-stale")] + #[arg(long = "if-stale", conflicts_with = "diagnostics")] if_stale: bool, + /// Remove every diagnostic event key (valid or malformed), + /// leaving raw records such as PROVISIONING_REPORT intact. + #[arg(long)] + diagnostics: bool, }, /// Print whether the pool is stale (exit 0 if stale, 1 otherwise). IsStale, @@ -176,50 +215,6 @@ enum Command { #[arg(long, value_parser = parse_supporting_data)] supporting_data: Vec, }, - /// Inspect decoded diagnostic events, reassembling chunked records. - Diag { - #[command(subcommand)] - command: DiagCommand, - }, -} - -/// Diagnostics subcommands operating on the reassembled event view. -#[derive(Subcommand, Debug)] -enum DiagCommand { - /// Print every record, reassembling chunked events. Raw records - /// (e.g. PROVISIONING_REPORT) are hidden unless --include-raw. - Dump { - /// Also print raw (non-event) records. - #[arg(long)] - include_raw: bool, - }, - /// Print decoded events (oldest first), optionally filtered. - Events { - /// Only show events at this level (error, warn, info, debug, trace). - #[arg(long)] - level: Option, - /// Only show events whose name contains this substring. - #[arg(long)] - name: Option, - }, - /// Print the last COUNT events (default 20). - Tail { - /// Number of trailing events to print. - #[arg(short = 'n', long = "count", default_value_t = 20)] - count: usize, - }, - /// Remove this layer's events, leaving raw records intact. - Clear { - /// VM identifier whose events to remove. - #[arg(long)] - vm_id: String, - /// Event key prefix whose events to remove. - #[arg(long)] - event_prefix: String, - /// Confirm the removal (required). - #[arg(long)] - yes: bool, - }, } #[derive(ValueEnum, Clone, Copy, Debug, PartialEq, Eq)] @@ -266,7 +261,21 @@ fn dispatch(cli: Cli, stdout: &mut W) -> Result { match cli.command { Command::Info => info(&store, stdout, output), - Command::Dump => dump(&store, stdout, output), + Command::Dump { + parse_diagnostics, + include_raw, + level, + name, + tail, + } => { + let parse = parse_diagnostics.then_some(ParseDiagnosticsArgs { + include_raw, + level, + name, + tail, + }); + dump(&store, stdout, parse, output) + } Command::Entries => entries(&store, stdout, output), Command::Read { key } => read(&store, stdout, &key, output), Command::Write { append, key, value } => { @@ -283,8 +292,13 @@ fn dispatch(cli: Cli, stdout: &mut W) -> Result { Command::DeleteMultiple { keys } => { delete_multiple(&store, stdout, keys, output) } - Command::Clear { if_stale } => { - if if_stale { + Command::Clear { + if_stale, + diagnostics, + } => { + if diagnostics { + DiagnosticsKvp::new(store.clone(), "", "").clear()?; + } else if if_stale { store.clear_if_stale()?; } else { store.clear()?; @@ -311,7 +325,6 @@ fn dispatch(cli: Cli, stdout: &mut W) -> Result { documentation_url, supporting_data, ), - Command::Diag { command } => diag(&store, stdout, command, output), } } @@ -357,11 +370,38 @@ fn info( Ok(EXIT_OK) } +/// Options for the `dump --parse-diagnostics` view; absent for a raw dump. +struct ParseDiagnosticsArgs { + include_raw: bool, + level: Option, + name: Option, + tail: Option, +} + fn dump( store: &KvpPoolStore, stdout: &mut W, + parse: Option, output: OutputMode, ) -> Result { + if let Some(parse) = parse { + // --level/--name/--tail select the decoded events-only view; + // otherwise every record is shown (raw hidden unless + // --include-raw). + if parse.level.is_some() || parse.name.is_some() || parse.tail.is_some() + { + return diagnostics_events( + store, + stdout, + parse.level, + parse.name, + parse.tail, + output, + ); + } + return diagnostics_records(store, stdout, parse.include_raw, output); + } + let records = store.dump()?; match output { OutputMode::Text => { @@ -470,31 +510,7 @@ fn is_stale( Ok(if stale { EXIT_OK } else { EXIT_NOT_FOUND }) } -fn diag( - store: &KvpPoolStore, - stdout: &mut W, - command: DiagCommand, - output: OutputMode, -) -> Result { - match command { - DiagCommand::Dump { include_raw } => { - diag_dump(store, stdout, include_raw, output) - } - DiagCommand::Events { level, name } => { - diag_events(store, stdout, level, name, None, output) - } - DiagCommand::Tail { count } => { - diag_events(store, stdout, None, None, Some(count), output) - } - DiagCommand::Clear { - vm_id, - event_prefix, - yes, - } => diag_clear(store, stdout, vm_id, event_prefix, yes, output), - } -} - -fn diag_dump( +fn diagnostics_records( store: &KvpPoolStore, stdout: &mut W, include_raw: bool, @@ -536,14 +552,15 @@ fn diag_dump( } } OutputMode::Json => { - let array: Vec<_> = records.iter().map(diag_record_json).collect(); + let array: Vec<_> = + records.iter().map(diagnostics_record_json).collect(); writeln_json(stdout, &serde_json::Value::Array(array))?; } } Ok(EXIT_OK) } -fn diag_events( +fn diagnostics_events( store: &KvpPoolStore, stdout: &mut W, level: Option, @@ -578,35 +595,14 @@ fn diag_events( } } OutputMode::Json => { - let array: Vec<_> = events.iter().map(diag_event_json).collect(); + let array: Vec<_> = + events.iter().map(diagnostics_event_json).collect(); writeln_json(stdout, &serde_json::Value::Array(array))?; } } Ok(EXIT_OK) } -fn diag_clear( - store: &KvpPoolStore, - stdout: &mut W, - vm_id: String, - event_prefix: String, - yes: bool, - output: OutputMode, -) -> Result { - if !yes { - return Err(CliError::Usage( - "refusing to clear diagnostic events without --yes".to_string(), - )); - } - let diagnostics = DiagnosticsKvp::new(store.clone(), vm_id, event_prefix); - diagnostics.clear()?; - match output { - OutputMode::Text => writeln!(stdout, "cleared")?, - OutputMode::Json => writeln_json(stdout, &json!({ "cleared": true }))?, - } - Ok(EXIT_OK) -} - /// Parse a `--level` filter argument into a [`tracing::Level`]. fn parse_level_filter(level: &str) -> Result { level.parse::().map_err(|_| { @@ -618,10 +614,10 @@ fn parse_level_filter(level: &str) -> Result { } /// Render a [`DiagnosticRecord`] as a JSON object. -fn diag_record_json(record: &DiagnosticRecord) -> serde_json::Value { +fn diagnostics_record_json(record: &DiagnosticRecord) -> serde_json::Value { match record { DiagnosticRecord::Event { event, chunks } => { - let mut value = diag_event_json(event); + let mut value = diagnostics_event_json(event); if let serde_json::Value::Object(map) = &mut value { map.insert("chunks".to_string(), json!(chunks)); } @@ -642,7 +638,7 @@ fn diag_record_json(record: &DiagnosticRecord) -> serde_json::Value { } /// Render a [`DiagnosticEvent`] as a JSON object (without chunk count). -fn diag_event_json(event: &DiagnosticEvent) -> serde_json::Value { +fn diagnostics_event_json(event: &DiagnosticEvent) -> serde_json::Value { json!({ "kind": "event", "level": event.level.to_string(), @@ -998,6 +994,17 @@ mod tests { (code, String::from_utf8(out).unwrap()) } + /// A plain `dump` command with no diagnostics parsing. + fn dump_cmd() -> Command { + Command::Dump { + parse_diagnostics: false, + include_raw: false, + level: None, + name: None, + tail: None, + } + } + fn set_mtime_to_epoch(path: &Path) { let c_path = CString::new(path.as_os_str().as_encoded_bytes()).unwrap(); let times = [libc::timeval { @@ -1037,7 +1044,7 @@ mod tests { assert_eq!(cli.dir, Some(PathBuf::from("/tmp/kvp"))); assert!(cli.unsafe_mode); assert!(cli.json); - assert!(matches!(cli.command, Command::Dump)); + assert!(matches!(cli.command, Command::Dump { .. })); } #[test] @@ -1533,7 +1540,7 @@ mod tests { }, )); - let (_, dumped) = run_dispatch(cli(&dir, Command::Dump)); + let (_, dumped) = run_dispatch(cli(&dir, dump_cmd())); assert_eq!(dumped, "k=1\nk=2\n"); } @@ -1557,7 +1564,7 @@ mod tests { store.insert("b", "two").unwrap(); store.insert("a", "one").unwrap(); - let (_, dumped) = run_dispatch(cli(&dir, Command::Dump)); + let (_, dumped) = run_dispatch(cli(&dir, dump_cmd())); assert!(dumped.contains("a=one")); assert!(dumped.contains("b=two")); @@ -1597,7 +1604,7 @@ mod tests { .unwrap(); assert_eq!(code, EXIT_OK); - let (_, dumped) = run_dispatch(cli(&dir, Command::Dump)); + let (_, dumped) = run_dispatch(cli(&dir, dump_cmd())); assert_eq!(dumped, "a=1\na=2\nb=3\n"); } @@ -1622,7 +1629,7 @@ mod tests { assert_eq!(code, EXIT_OK); assert_eq!(out, "2\n"); - let (_, dumped) = run_dispatch(cli(&dir, Command::Dump)); + let (_, dumped) = run_dispatch(cli(&dir, dump_cmd())); assert_eq!(dumped, "b=2\n"); } @@ -1636,7 +1643,13 @@ mod tests { let dir = TempDir::new().unwrap(); store_at(&dir).insert("k", "v").unwrap(); - let (code, _) = run_dispatch(cli(&dir, Command::Clear { if_stale })); + let (code, _) = run_dispatch(cli( + &dir, + Command::Clear { + if_stale, + diagnostics: false, + }, + )); assert_eq!(code, EXIT_OK); assert_eq!(store_at(&dir).is_empty().unwrap(), expect_empty_after); } @@ -1829,7 +1842,7 @@ mod tests { store.append("b", "two-prime").unwrap(); store.insert("a", "one").unwrap(); - let (_, out) = run_dispatch(cli_json(&dir, Command::Dump)); + let (_, out) = run_dispatch(cli_json(&dir, dump_cmd())); let json = parse_json(&out); let array = json.as_array().expect("dump --json returns array"); assert_eq!(array.len(), 3); diff --git a/libazureinit-kvp/src/diagnostics.rs b/libazureinit-kvp/src/diagnostics.rs index e65f676e..e9ee644e 100644 --- a/libazureinit-kvp/src/diagnostics.rs +++ b/libazureinit-kvp/src/diagnostics.rs @@ -112,8 +112,6 @@ fn format_event_key( enum KeyClass<'a> { /// The key is a well-formed event key. Event { - prefix: &'a str, - vm_id: &'a str, level: Level, name: &'a str, event_id: &'a str, @@ -129,8 +127,8 @@ fn classify_key(key: &str) -> KeyClass<'_> { let mut segments = key.split(EVENT_KEY_DELIMITER); // `str::split` always yields at least one element. - let prefix = segments.next().unwrap_or_default(); - let (Some(vm_id), Some(level), Some(name), Some(event_id)) = ( + let _prefix = segments.next(); + let (Some(_vm_id), Some(level), Some(name), Some(event_id)) = ( segments.next(), segments.next(), segments.next(), @@ -145,8 +143,6 @@ fn classify_key(key: &str) -> KeyClass<'_> { match level.parse::() { Ok(level) => KeyClass::Event { - prefix, - vm_id, level, name, event_id, @@ -275,8 +271,8 @@ pub enum DiagnosticRecord { /// A typed diagnostics view over a [`KvpPoolStore`]. /// -/// Owns the event-key `prefix` and `vm_id` used to format and -/// [`scope`](Self::clear) this layer's events. See the +/// Owns the event-key `prefix` and `vm_id` stamped into this layer's +/// events by [`emit`](Self::emit). See the /// [module documentation](self) for the on-disk format. #[derive(Clone, Debug)] pub struct DiagnosticsKvp { @@ -399,24 +395,21 @@ impl DiagnosticsKvp { .collect()) } - /// Remove this layer's events — keys that parse as an event key - /// (including every `|` chunk of a multi-record - /// event) whose `prefix` and `vm_id` match this instance — under a - /// single lock. Raw records such as `PROVISIONING_REPORT` are left - /// intact. + /// Remove every diagnostic key under a single lock: any key that + /// parses as an event key (including every `|` chunk + /// of a multi-record event) or a malformed event key. Raw records + /// such as `PROVISIONING_REPORT` are left intact. pub fn clear(&self) -> Result<(), KvpError> { let keys: Vec = self .store .dump()? .into_iter() .filter_map(|(key, _)| { - let is_mine = matches!( + let is_diagnostic = !matches!( classify_key(base_event_key(&key)), - KeyClass::Event { prefix, vm_id, .. } - if prefix == self.event_prefix - && vm_id == self.vm_id + KeyClass::Raw ); - is_mine.then_some(key) + is_diagnostic.then_some(key) }) .collect(); self.store.delete_multiple(keys)?; @@ -481,7 +474,6 @@ fn classify_record( level, name, event_id, - .. } => DiagnosticRecord::Event { event: DiagnosticEvent { level, @@ -523,14 +515,10 @@ mod tests { assert!(matches!( classify_key(&formatted), KeyClass::Event { - prefix, - vm_id, level, name, event_id, - } if prefix == PREFIX - && vm_id == VM_ID - && level == Level::INFO + } if level == Level::INFO && name == "user:create_user" && event_id == EVENT_ID )); diff --git a/libazureinit-kvp/tests/cli.rs b/libazureinit-kvp/tests/cli.rs index 60416526..e28ef570 100644 --- a/libazureinit-kvp/tests/cli.rs +++ b/libazureinit-kvp/tests/cli.rs @@ -300,7 +300,7 @@ fn report_failure_rejects_invalid_supporting_data() { } #[test] -fn diag_dump_json_reassembles_and_classifies() { +fn dump_parse_diagnostics_json_reassembles_and_classifies() { let dir = TempDir::new().unwrap(); // A two-chunk event (same key repeated) plus a raw record. assert_success(kvp(&with_dir( @@ -316,7 +316,10 @@ fn diag_dump_json_reassembles_and_classifies() { &["write", "PROVISIONING_REPORT", "result=success"], ))); - let out = assert_success(kvp(&with_dir(&dir, &["--json", "diag", "dump"]))); + let out = assert_success(kvp(&with_dir( + &dir, + &["--json", "dump", "--parse-diagnostics"], + ))); assert!(out.contains("\"kind\":\"event\"")); assert!(out.contains("\"chunks\":2")); assert!(out.contains("\"message\":\"one/two\"")); @@ -325,14 +328,14 @@ fn diag_dump_json_reassembles_and_classifies() { let out_raw = assert_success(kvp(&with_dir( &dir, - &["--json", "diag", "dump", "--include-raw"], + &["--json", "dump", "--parse-diagnostics", "--include-raw"], ))); assert!(out_raw.contains("\"kind\":\"raw\"")); assert!(out_raw.contains("PROVISIONING_REPORT")); } #[test] -fn diag_events_filters_by_level() { +fn dump_parse_diagnostics_filters_by_level() { let dir = TempDir::new().unwrap(); assert_success(kvp(&with_dir( &dir, @@ -345,46 +348,45 @@ fn diag_events_filters_by_level() { let out = assert_success(kvp(&with_dir( &dir, - &["diag", "events", "--level", "error"], + &["dump", "--parse-diagnostics", "--level", "error"], ))); assert!(out.contains("err-msg")); assert!(!out.contains("info-msg")); } #[test] -fn diag_clear_requires_confirmation_then_scopes_removal() { +fn clear_diagnostics_removes_events_and_malformed_keeps_raw() { let dir = TempDir::new().unwrap(); + // An event, a malformed event key, and a raw record. assert_success(kvp(&with_dir( &dir, &["write", "--append", "p|vm|INFO|a:b|i1", "msg"], ))); - - // Without --yes the command refuses and exits with a usage error. - let refused = kvp(&with_dir( + assert_success(kvp(&with_dir( &dir, - &["diag", "clear", "--vm-id", "vm", "--event-prefix", "p"], - )); - assert_eq!(refused.status.code(), Some(2)); - - // With --yes the matching event is removed. + &["write", "--append", "p|vm|NOPE|c:d|i2", "junk"], + ))); assert_success(kvp(&with_dir( &dir, - &[ - "diag", - "clear", - "--vm-id", - "vm", - "--event-prefix", - "p", - "--yes", - ], + &["write", "PROVISIONING_REPORT", "result=success"], ))); - let out = assert_success(kvp(&with_dir(&dir, &["diag", "dump"]))); - assert!(out.trim().is_empty()); + + assert_success(kvp(&with_dir(&dir, &["clear", "--diagnostics"]))); + + let out = assert_success(kvp(&with_dir(&dir, &["dump"]))); + assert_eq!(out, "PROVISIONING_REPORT=result=success\n"); +} + +#[test] +fn clear_diagnostics_conflicts_with_if_stale() { + let dir = TempDir::new().unwrap(); + let output = + kvp(&with_dir(&dir, &["clear", "--diagnostics", "--if-stale"])); + assert_eq!(output.status.code(), Some(2)); } #[test] -fn diag_dump_text_renders_all_record_kinds() { +fn dump_parse_diagnostics_text_renders_all_record_kinds() { let dir = TempDir::new().unwrap(); assert_success(kvp(&with_dir( &dir, @@ -405,7 +407,7 @@ fn diag_dump_text_renders_all_record_kinds() { let out = assert_success(kvp(&with_dir( &dir, - &["diag", "dump", "--include-raw"], + &["dump", "--parse-diagnostics", "--include-raw"], ))); assert!(out.contains( "event level=INFO name=a:b event_id=id1 chunks=2 message=one/two" @@ -416,7 +418,7 @@ fn diag_dump_text_renders_all_record_kinds() { } #[test] -fn diag_tail_limits_to_last_events() { +fn dump_parse_diagnostics_tail_limits_to_last_events() { let dir = TempDir::new().unwrap(); assert_success(kvp(&with_dir( &dir, @@ -427,14 +429,42 @@ fn diag_tail_limits_to_last_events() { &["write", "--append", "p|vm|INFO|c:d|i2", "second"], ))); - let out = - assert_success(kvp(&with_dir(&dir, &["diag", "tail", "-n", "1"]))); + let out = assert_success(kvp(&with_dir( + &dir, + &["dump", "--parse-diagnostics", "-n", "1"], + ))); assert!(out.contains("second")); assert!(!out.contains("first")); } #[test] -fn diag_events_filters_by_name_substring() { +fn dump_parse_diagnostics_tail_defaults_to_20_when_count_omitted() { + let dir = TempDir::new().unwrap(); + for i in 1..=25 { + assert_success(kvp(&with_dir( + &dir, + &[ + "write", + "--append", + &format!("p|vm|INFO|n:{i}|id{i}"), + &format!("msg{i}"), + ], + ))); + } + + // Bare --tail keeps the last 20 events (msg6..msg25). + let out = assert_success(kvp(&with_dir( + &dir, + &["dump", "--parse-diagnostics", "--tail"], + ))); + assert_eq!(out.lines().count(), 20); + assert!(out.contains("msg25")); + assert!(out.contains("msg6")); + assert!(!out.contains("msg5")); +} + +#[test] +fn dump_parse_diagnostics_filters_by_name_substring() { let dir = TempDir::new().unwrap(); assert_success(kvp(&with_dir( &dir, @@ -447,21 +477,40 @@ fn diag_events_filters_by_name_substring() { let out = assert_success(kvp(&with_dir( &dir, - &["diag", "events", "--name", "ssh"], + &["dump", "--parse-diagnostics", "--name", "ssh"], ))); assert!(out.contains("ssh:key")); assert!(!out.contains("user:add")); } #[test] -fn diag_events_rejects_invalid_level() { +fn dump_parse_diagnostics_rejects_invalid_level() { let dir = TempDir::new().unwrap(); - let output = kvp(&with_dir(&dir, &["diag", "events", "--level", "bogus"])); + let output = kvp(&with_dir( + &dir, + &["dump", "--parse-diagnostics", "--level", "bogus"], + )); assert_eq!(output.status.code(), Some(2)); } #[test] -fn diag_json_covers_malformed_events_and_clear() { +fn dump_parse_diagnostics_include_raw_conflicts_with_filters() { + let dir = TempDir::new().unwrap(); + let output = kvp(&with_dir( + &dir, + &[ + "dump", + "--parse-diagnostics", + "--include-raw", + "--level", + "info", + ], + )); + assert_eq!(output.status.code(), Some(2)); +} + +#[test] +fn dump_parse_diagnostics_json_covers_events_and_malformed() { let dir = TempDir::new().unwrap(); assert_success(kvp(&with_dir( &dir, @@ -472,27 +521,19 @@ fn diag_json_covers_malformed_events_and_clear() { &["write", "--append", "p|vm|NOPE|c:d|i2", "junk"], ))); - let dump = - assert_success(kvp(&with_dir(&dir, &["--json", "diag", "dump"]))); + let dump = assert_success(kvp(&with_dir( + &dir, + &["--json", "dump", "--parse-diagnostics"], + ))); assert!(dump.contains("\"kind\":\"event\"")); assert!(dump.contains("\"kind\":\"malformed\"")); assert!(dump.contains("\"reason\":")); - let events = - assert_success(kvp(&with_dir(&dir, &["--json", "diag", "events"]))); - assert!(events.contains("\"message\":\"hello\"")); - let cleared = assert_success(kvp(&with_dir( + // Filtering by name yields the events-only view. + let events = assert_success(kvp(&with_dir( &dir, - &[ - "--json", - "diag", - "clear", - "--vm-id", - "vm", - "--event-prefix", - "p", - "--yes", - ], + &["--json", "dump", "--parse-diagnostics", "--name", "a:b"], ))); - assert!(cleared.contains("\"cleared\":true")); + assert!(events.contains("\"message\":\"hello\"")); + assert!(!events.contains("\"kind\":\"malformed\"")); } diff --git a/libazureinit-kvp/tests/diagnostics.rs b/libazureinit-kvp/tests/diagnostics.rs index e1f5694a..052e924b 100644 --- a/libazureinit-kvp/tests/diagnostics.rs +++ b/libazureinit-kvp/tests/diagnostics.rs @@ -193,22 +193,32 @@ fn clear_removes_events_but_keeps_raw() { } #[test] -fn clear_is_scoped_to_matching_prefix_and_vm_id() { +fn clear_removes_all_diagnostics_regardless_of_scope() { let dir = TempDir::new().unwrap(); let diag = diagnostics(&dir); diag.emit(&DiagnosticEvent::new(Level::INFO, "a:b", "mine")) .unwrap(); - // An event from a different agent/VM must survive clear(). + // Events from a different agent/VM and a malformed event key are also + // diagnostic keys, so clear() removes them too. diag.store() .append("other-agent|other-vm|INFO|x:y|id", "theirs") .unwrap(); + diag.store().append("p|vm|NOPE|c:d|id", "junk").unwrap(); + // A raw record survives. + diag.store() + .append("PROVISIONING_REPORT", "result=success") + .unwrap(); diag.clear().unwrap(); - let events = diag.events().unwrap(); - assert_eq!(events.len(), 1); - assert_eq!(events[0].message, "theirs"); + let records = diag.records().unwrap(); + assert_eq!(records.len(), 1); + assert!(matches!( + &records[0], + DiagnosticRecord::Raw { key, .. } if key == "PROVISIONING_REPORT" + )); + assert!(diag.events().unwrap().is_empty()); } #[test] From 98f4490ed58ce75e8d5a43ad65729b0e31364fd0 Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Mon, 27 Jul 2026 15:55:50 -0700 Subject: [PATCH 04/32] feat(kvp): read and parse cloud-init KVP in diagnostics. --- libazureinit-kvp/Cargo.toml | 1 + libazureinit-kvp/src/cli.rs | 50 ++- libazureinit-kvp/src/diagnostics.rs | 431 +++++++++++++++++++++++--- libazureinit-kvp/src/lib.rs | 3 +- libazureinit-kvp/tests/diagnostics.rs | 67 ++++ 5 files changed, 513 insertions(+), 39 deletions(-) diff --git a/libazureinit-kvp/Cargo.toml b/libazureinit-kvp/Cargo.toml index cd0bb03c..7f81e102 100644 --- a/libazureinit-kvp/Cargo.toml +++ b/libazureinit-kvp/Cargo.toml @@ -13,6 +13,7 @@ chrono = { version = "0.4", default-features = false, features = ["clock", "std" clap = { version = "4.5.21", features = ["derive"] } csv = "1" libc = "0.2" +serde = { version = "1.0", features = ["derive"] } serde_json = "1.0.96" tracing = "0.1.40" uuid = { version = "1.3", features = ["v4"] } diff --git a/libazureinit-kvp/src/cli.rs b/libazureinit-kvp/src/cli.rs index 4a6b00b5..58ef644b 100644 --- a/libazureinit-kvp/src/cli.rs +++ b/libazureinit-kvp/src/cli.rs @@ -1,6 +1,7 @@ // Copyright (c) Microsoft Corporation. // Licensed under the MIT License. +use std::fmt::Write as _; use std::fs; use std::io::{self, Read, Write}; use std::path::PathBuf; @@ -94,7 +95,7 @@ enum Command { /// decode each record instead of printing raw KEY=VALUE lines. Dump { /// Reassemble chunked diagnostic events and decode each record as - /// an event, raw, or malformed entry. + /// an azure-init event, cloud-init event, raw, or malformed entry. #[arg(long)] parse_diagnostics: bool, /// Also print raw (non-event) records such as PROVISIONING_REPORT. @@ -538,6 +539,30 @@ fn diagnostics_records( chunks, event.message ), + DiagnosticRecord::CloudInit { event, chunks } => { + let mut line = format!( + "cloud-init-event type={} name={} uuid={}", + event.event_type, event.name, event.uuid + ); + if let Some(vm_id) = &event.vm_id { + let _ = write!(line, " vm_id={vm_id}"); + } + if let Some(result) = &event.result { + let _ = write!(line, " result={result}"); + } + if let Some(ts) = &event.timestamp { + let _ = write!(line, " ts={ts}"); + } + if let Some(duration) = event.duration { + let _ = write!(line, " duration={duration}"); + } + let _ = write!( + line, + " incarnation={} chunks={} message={}", + event.incarnation, chunks, event.message + ); + line + } DiagnosticRecord::Raw { key, value } => { format!("raw key={key} value={value}") } @@ -623,6 +648,29 @@ fn diagnostics_record_json(record: &DiagnosticRecord) -> serde_json::Value { } value } + DiagnosticRecord::CloudInit { event, chunks } => { + let mut map = serde_json::Map::new(); + map.insert("kind".to_string(), json!("cloud-init-event")); + map.insert("incarnation".to_string(), json!(event.incarnation)); + map.insert("type".to_string(), json!(event.event_type)); + map.insert("name".to_string(), json!(event.name)); + if let Some(vm_id) = &event.vm_id { + map.insert("vm_id".to_string(), json!(vm_id)); + } + map.insert("uuid".to_string(), json!(event.uuid)); + if let Some(ts) = &event.timestamp { + map.insert("ts".to_string(), json!(ts)); + } + if let Some(result) = &event.result { + map.insert("result".to_string(), json!(result)); + } + if let Some(duration) = event.duration { + map.insert("duration".to_string(), json!(duration)); + } + map.insert("chunks".to_string(), json!(chunks)); + map.insert("message".to_string(), json!(event.message)); + serde_json::Value::Object(map) + } DiagnosticRecord::Raw { key, value } => json!({ "kind": "raw", "key": key, diff --git a/libazureinit-kvp/src/diagnostics.rs b/libazureinit-kvp/src/diagnostics.rs index e9ee644e..fc29d64a 100644 --- a/libazureinit-kvp/src/diagnostics.rs +++ b/libazureinit-kvp/src/diagnostics.rs @@ -25,6 +25,11 @@ //! [`DiagnosticEvent`], an unstructured [`Raw`](DiagnosticRecord::Raw) //! record such as `PROVISIONING_REPORT`, or a //! [`Malformed`](DiagnosticRecord::Malformed) event key. +//! - **cloud-init**: the reader also decodes cloud-init reporting entries +//! (keys prefixed `CLOUD_INIT` with a JSON value) into +//! [`CloudInitEvent`]s so the diagnostics CLI can display telemetry +//! from either provisioning agent. This crate only *reads* that +//! format; it does not write it. //! //! This module is policy only: all locking, size enforcement, and //! on-disk encoding stay in [`KvpPoolStore`](crate::KvpPoolStore). @@ -70,11 +75,15 @@ //! # } //! ``` +use serde::Deserialize; use tracing::Level; use uuid::Uuid; use crate::{KvpError, KvpPoolStore}; +/// Literal prefix identifying a cloud-init reporting KVP key. +const CLOUD_INIT_PREFIX: &str = "CLOUD_INIT"; + /// Maximum number of value bytes per KVP record under /// [`PoolMode::Safe`](crate::PoolMode::Safe). /// @@ -110,20 +119,33 @@ fn format_event_key( /// Outcome of inspecting a raw pool key. enum KeyClass<'a> { - /// The key is a well-formed event key. + /// The key is a well-formed azure-init event key. Event { level: Level, name: &'a str, event_id: &'a str, }, + /// The key is a well-formed cloud-init reporting event key + /// (`CLOUD_INIT||||[|]`). + CloudInit { + incarnation: &'a str, + event_type: &'a str, + name: &'a str, + vm_id: Option<&'a str>, + uuid: &'a str, + }, /// The key has five segments but is not a valid event. Malformed { reason: String }, /// The key is not an event key (e.g. `PROVISIONING_REPORT`). Raw, } -/// Classify a raw pool key without allocating. +/// Classify a raw pool key. fn classify_key(key: &str) -> KeyClass<'_> { + if key.split(EVENT_KEY_DELIMITER).next() == Some(CLOUD_INIT_PREFIX) { + return classify_cloud_init_key(key); + } + let mut segments = key.split(EVENT_KEY_DELIMITER); // `str::split` always yields at least one element. @@ -153,6 +175,40 @@ fn classify_key(key: &str) -> KeyClass<'_> { } } +/// Classify a `CLOUD_INIT`-prefixed key into a [`KeyClass::CloudInit`]. +/// +/// Handles both the current layout +/// (`CLOUD_INIT|||||`) and the +/// older one that predates the `vm_id` segment +/// (`CLOUD_INIT||||`). Any other segment +/// count is treated as [`KeyClass::Raw`]. Parses by pulling segments +/// from the iterator so no intermediate collection is allocated. +fn classify_cloud_init_key(key: &str) -> KeyClass<'_> { + let mut segments = key.split(EVENT_KEY_DELIMITER); + // The caller matched the `CLOUD_INIT` prefix; skip it. + let _prefix = segments.next(); + let (Some(incarnation), Some(event_type), Some(name), Some(fourth)) = ( + segments.next(), + segments.next(), + segments.next(), + segments.next(), + ) else { + return KeyClass::Raw; + }; + let (vm_id, uuid) = match (segments.next(), segments.next()) { + (None, None) => (None, fourth), + (Some(uuid), None) => (Some(fourth), uuid), + _ => return KeyClass::Raw, + }; + KeyClass::CloudInit { + incarnation, + event_type, + name, + vm_id, + uuid, + } +} + /// Split `value` into pieces of at most `max_bytes` bytes each, always /// at UTF-8 codepoint boundaries. /// @@ -238,17 +294,72 @@ impl DiagnosticEvent { } } +/// The JSON payload cloud-init stores as a reporting event's KVP value. +/// +/// Only the fields the diagnostics reader surfaces are deserialized; +/// cloud-init also duplicates `name`/`type` here, but those are read +/// from the key. Unknown fields are ignored. +#[derive(Debug, Deserialize)] +struct CloudInitValue { + /// Human-readable message; defaults to empty when absent. + #[serde(default)] + msg: String, + /// ISO-8601 timestamp. + ts: Option, + /// Result string (e.g. `SUCCESS`), present on `finish` events. + result: Option, + /// Duration in seconds, present on `finish` events. + duration: Option, +} + +/// A single cloud-init reporting event decoded from a KVP entry. +/// +/// cloud-init encodes routing metadata in the key +/// (`CLOUD_INIT||||[|]`) and the +/// event details as a JSON value; this type is the decoded union of the +/// two. +#[derive(Clone, Debug, PartialEq)] +pub struct CloudInitEvent { + /// Boot-time incarnation stamp from the key. cloud-init uses it to + /// distinguish this boot's records from a previous boot's. + pub incarnation: String, + /// Provisioning phase from the key, e.g. `start` or `finish`. + pub event_type: String, + /// Event name from the key, e.g. `modules-final/config-scripts_user`. + pub name: String, + /// VM identifier from the key. Absent in cloud-init builds that + /// predate the `vm_id` key segment. + pub vm_id: Option, + /// Per-event UUID from the key. + pub uuid: String, + /// ISO-8601 timestamp from the JSON value, if present. + pub timestamp: Option, + /// Result from the JSON value (e.g. `SUCCESS`), if present. + pub result: Option, + /// Duration in seconds from the JSON value, if present. + pub duration: Option, + /// Human-readable message from the JSON value. + pub message: String, +} + /// A single record read back from the pool and classified by /// [`DiagnosticsKvp::records`]. -#[derive(Clone, Debug, PartialEq, Eq)] +#[derive(Clone, Debug, PartialEq)] pub enum DiagnosticRecord { - /// A reassembled diagnostic event. + /// A reassembled azure-init diagnostic event. Event { /// The decoded event. event: DiagnosticEvent, /// Number of on-disk records the value spanned (1 when short). chunks: usize, }, + /// A decoded cloud-init reporting event. + CloudInit { + /// The decoded cloud-init event. + event: CloudInitEvent, + /// Number of on-disk records the value spanned (1 when short). + chunks: usize, + }, /// An unstructured record whose key is not an event key, such as /// `PROVISIONING_REPORT`. Raw { @@ -381,15 +492,21 @@ impl DiagnosticsKvp { Ok(reassemble(self.store.dump()?)) } - /// Read back only the records that decode as diagnostic events, in - /// on-disk order. + /// Read back only the azure-init [`DiagnosticEvent`]s, in on-disk + /// order. + /// + /// This deliberately excludes cloud-init events — which decode to the + /// separate [`CloudInitEvent`] type — as well as raw and malformed + /// records. Use [`records`](Self::records) for the full cross-agent + /// view that includes cloud-init telemetry. pub fn events(&self) -> Result, KvpError> { Ok(self .records()? .into_iter() .filter_map(|record| match record { DiagnosticRecord::Event { event, .. } => Some(event), - DiagnosticRecord::Raw { .. } + DiagnosticRecord::CloudInit { .. } + | DiagnosticRecord::Raw { .. } | DiagnosticRecord::Malformed { .. } => None, }) .collect()) @@ -428,47 +545,72 @@ impl DiagnosticsKvp { /// key (a single-record event, `PROVISIONING_REPORT`, a malformed key, …) /// is returned unchanged. fn base_event_key(key: &str) -> &str { - if let Some((base, subevent_index)) = key.rsplit_once(EVENT_KEY_DELIMITER) { - if subevent_index.parse::().is_ok() - && matches!(classify_key(base), KeyClass::Event { .. }) - { - return base; + split_subevent_index(key).0 +} + +/// Split a key into its base event key and optional trailing subevent +/// index. A chunk key `|` returns `(, Some(i))`; +/// any other key (a single-record event, `PROVISIONING_REPORT`, a +/// malformed key, …) returns `(key, None)`. +/// +/// The subevent index is the same trailing `|` cloud-init and +/// azure-init append to give each chunk a unique key; [`reassemble`] uses +/// it both to regroup an event's chunks and to restore their write order. +fn split_subevent_index(key: &str) -> (&str, Option) { + if let Some((base, index)) = key.rsplit_once(EVENT_KEY_DELIMITER) { + if let Ok(index) = index.parse::() { + if matches!( + classify_key(base), + KeyClass::Event { .. } | KeyClass::CloudInit { .. } + ) { + return (base, Some(index)); + } } } - key + (key, None) } /// Group consecutive records sharing an event key — chunk /// `|` suffixes stripped — from [`KvpPoolStore::dump`] /// and classify each group into a [`DiagnosticRecord`]. fn reassemble(dumped: Vec<(String, String)>) -> Vec { + let mut parsed = dumped + .into_iter() + .map(|(key, value)| { + let (base, index) = split_subevent_index(&key); + (base.to_string(), index, value) + }) + .peekable(); + let mut records = Vec::new(); - let mut dumped = dumped.into_iter().peekable(); - - while let Some((key, value)) = dumped.next() { - let base = base_event_key(&key).to_string(); - let mut message = value; - let mut chunks = 1; - while dumped - .peek() - .is_some_and(|(next, _)| base_event_key(next) == base) - { - let (_, next_value) = dumped.next().expect("peeked value exists"); - message.push_str(&next_value); - chunks += 1; + while let Some((base, index, value)) = parsed.next() { + let mut indexed = vec![(index, value)]; + while parsed.peek().is_some_and(|(next, _, _)| *next == base) { + let (_, next_index, next_value) = + parsed.next().expect("peeked value exists"); + indexed.push((next_index, next_value)); } - records.push(classify_record(base, message, chunks)); + // Restore write order by subevent index. Stable, so a single + // record (index `None`) or any equal indices keep on-disk order. + indexed.sort_by_key(|(index, _)| *index); + let chunk_values = + indexed.into_iter().map(|(_, value)| value).collect(); + records.push(classify_record(base, chunk_values)); } records } -/// Turn one reassembled key/value group into a [`DiagnosticRecord`]. -fn classify_record( - key: String, - value: String, - chunks: usize, -) -> DiagnosticRecord { +/// Turn one reassembled group of chunk values into a +/// [`DiagnosticRecord`]. +/// +/// `chunk_values` holds every record that shared the base event key, +/// already ordered by subevent index by [`reassemble`], and is never +/// empty. azure-init events and raw records concatenate their values +/// directly. cloud-init writes a full JSON object per chunk, so those are +/// decoded and their `msg` slices joined into the full message. +fn classify_record(key: String, chunk_values: Vec) -> DiagnosticRecord { + let chunks = chunk_values.len(); match classify_key(&key) { KeyClass::Event { level, @@ -479,14 +621,68 @@ fn classify_record( level, name: name.to_string(), event_id: event_id.to_string(), - message: value, + message: chunk_values.concat(), }, chunks, }, - KeyClass::Malformed { reason } => { - DiagnosticRecord::Malformed { key, value, reason } + KeyClass::CloudInit { + incarnation, + event_type, + name, + vm_id, + uuid, + } => { + // Own the key-derived fields up front so `key` and + // `chunk_values` are free to move into a `Malformed` record + // when a chunk's JSON fails to parse. + let incarnation = incarnation.to_string(); + let event_type = event_type.to_string(); + let name = name.to_string(); + let vm_id = vm_id.map(str::to_string); + let uuid = uuid.to_string(); + match chunk_values + .iter() + .map(|value| serde_json::from_str::(value)) + .collect::, _>>() + { + Ok(parts) => { + // Chunks arrive already ordered by subevent index, so + // join the message slices directly. Every chunk repeats + // the same metadata, so read it from the first chunk. + let message: String = + parts.iter().map(|part| part.msg.as_str()).collect(); + let first = &parts[0]; + DiagnosticRecord::CloudInit { + event: CloudInitEvent { + incarnation, + event_type, + name, + vm_id, + uuid, + timestamp: first.ts.clone(), + result: first.result.clone(), + duration: first.duration, + message, + }, + chunks, + } + } + Err(err) => DiagnosticRecord::Malformed { + key, + value: chunk_values.concat(), + reason: format!("invalid cloud-init JSON value: {err}"), + }, + } } - KeyClass::Raw => DiagnosticRecord::Raw { key, value }, + KeyClass::Malformed { reason } => DiagnosticRecord::Malformed { + key, + value: chunk_values.concat(), + reason, + }, + KeyClass::Raw => DiagnosticRecord::Raw { + key, + value: chunk_values.concat(), + }, } } @@ -551,6 +747,7 @@ mod tests { fn class_of(key: &str) -> &'static str { match classify_key(key) { KeyClass::Event { .. } => "event", + KeyClass::CloudInit { .. } => "cloud-init", KeyClass::Malformed { .. } => "malformed", KeyClass::Raw => "raw", } @@ -679,6 +876,10 @@ mod tests { "p|vm|INFO|name|id" )] #[case::single_event_unchanged("p|vm|INFO|name|id", "p|vm|INFO|name|id")] + #[case::cloud_init_indexed_chunk( + "CLOUD_INIT|1785187982|finish|mod|vmid|uuid|0", + "CLOUD_INIT|1785187982|finish|mod|vmid|uuid" + )] #[case::raw_unchanged("PROVISIONING_REPORT", "PROVISIONING_REPORT")] #[case::non_event_numeric_tail_unchanged("foo|3", "foo|3")] #[case::malformed_unchanged("p|vm|NOPE|name|id", "p|vm|NOPE|name|id")] @@ -719,4 +920,160 @@ mod tests { } ); } + + // ---- cloud-init read support ---- + + const CLOUD_INIT_VM_ID: &str = "0e5e179d-5341-478b-8456-fbb90621bdf8"; + const CLOUD_INIT_KEY_FINISH: &str = "CLOUD_INIT|1785187982|finish|modules-final/config-scripts_user|0e5e179d-5341-478b-8456-fbb90621bdf8|e5f01809-a7a3-4279-aa64-1f18e21eda6e"; + const CLOUD_INIT_VALUE_FINISH: &str = r#"{"name":"modules-final/config-scripts_user","type":"finish","ts":"2026-07-27T21:33:24.339006+00:00","result":"SUCCESS","duration":0.0006448590000012189,"msg":"config-scripts_user ran successfully and took 0.001 seconds"}"#; + + #[rstest] + #[case::with_vm_id( + CLOUD_INIT_KEY_FINISH, + "1785187982", + "finish", + "modules-final/config-scripts_user", + Some(CLOUD_INIT_VM_ID), + "e5f01809-a7a3-4279-aa64-1f18e21eda6e" + )] + // Older cloud-init builds omit the vm_id key segment. + #[case::without_vm_id( + "CLOUD_INIT|1785187982|start|modules-config/foo|c4d4a08d-fe93-4c7a-9be6-9a38c212e212", + "1785187982", + "start", + "modules-config/foo", + None, + "c4d4a08d-fe93-4c7a-9be6-9a38c212e212" + )] + fn cloud_init_key_classifies( + #[case] key: &str, + #[case] incarnation: &str, + #[case] event_type: &str, + #[case] name: &str, + #[case] vm_id: Option<&str>, + #[case] uuid: &str, + ) { + assert!(matches!( + classify_key(key), + KeyClass::CloudInit { + incarnation: i, + event_type: t, + name: n, + vm_id: v, + uuid: u, + } if i == incarnation + && t == event_type + && n == name + && v == vm_id + && u == uuid + )); + } + + #[rstest] + #[case::too_many("CLOUD_INIT|a|b|c|d|e|f", "raw")] + #[case::too_few("CLOUD_INIT|a|b|c", "raw")] + #[case::prefix_only("CLOUD_INIT", "raw")] + fn cloud_init_bad_shapes_are_raw( + #[case] key: &str, + #[case] expected: &str, + ) { + assert_eq!(class_of(key), expected); + } + + #[test] + fn cloud_init_finish_record_decodes_all_fields() { + let record = classify_record( + CLOUD_INIT_KEY_FINISH.to_string(), + vec![CLOUD_INIT_VALUE_FINISH.to_string()], + ); + match record { + DiagnosticRecord::CloudInit { event, chunks } => { + assert_eq!(chunks, 1); + assert_eq!(event.incarnation, "1785187982"); + assert_eq!(event.event_type, "finish"); + assert_eq!(event.name, "modules-final/config-scripts_user"); + assert_eq!(event.vm_id.as_deref(), Some(CLOUD_INIT_VM_ID)); + assert_eq!(event.uuid, "e5f01809-a7a3-4279-aa64-1f18e21eda6e"); + assert_eq!( + event.timestamp.as_deref(), + Some("2026-07-27T21:33:24.339006+00:00") + ); + assert_eq!(event.result.as_deref(), Some("SUCCESS")); + assert!(event + .duration + .is_some_and( + |d| (d - 0.000_644_859_000_001_218_9).abs() < 1e-12 + )); + assert_eq!( + event.message, + "config-scripts_user ran successfully and took 0.001 \ + seconds" + ); + } + other => panic!("expected cloud-init event, got {other:?}"), + } + } + + #[test] + fn cloud_init_start_record_has_no_result_or_duration() { + let value = r#"{"name":"modules-final/config-keys_to_console","type":"start","ts":"2026-07-27T21:33:24.344349+00:00","msg":"running config-keys_to_console with frequency once-per-instance"}"#; + let key = "CLOUD_INIT|1785187982|start|modules-final/config-keys_to_console|0e5e179d-5341-478b-8456-fbb90621bdf8|7792621b-b339-4274-8b71-2a3dcbd2db4e"; + match classify_record(key.to_string(), vec![value.to_string()]) { + DiagnosticRecord::CloudInit { event, .. } => { + assert_eq!(event.event_type, "start"); + assert!(event.result.is_none()); + assert!(event.duration.is_none()); + assert_eq!( + event.message, + "running config-keys_to_console with frequency \ + once-per-instance" + ); + } + other => panic!("expected cloud-init event, got {other:?}"), + } + } + + #[test] + fn cloud_init_key_with_invalid_json_is_malformed() { + match classify_record( + CLOUD_INIT_KEY_FINISH.to_string(), + vec!["not json".to_string()], + ) { + DiagnosticRecord::Malformed { reason, .. } => { + assert!(reason.contains("cloud-init")); + } + other => panic!("expected malformed, got {other:?}"), + } + } + + #[test] + fn cloud_init_chunks_reassemble_by_subevent_index() { + let base = "CLOUD_INIT|1785187982|finish|modules-final/long|0e5e179d-5341-478b-8456-fbb90621bdf8|abc12345-1111-2222-3333-444455556666"; + // Each chunk carries its own `|` key suffix and a + // message slice. They are laid out on disk out of order to prove + // reassembly restores order by that index, not disk position. + let chunk = |i: u32, msg: &str| { + format!( + r#"{{"name":"modules-final/long","type":"finish","ts":"2026-07-27T21:33:24.339006+00:00","result":"SUCCESS","duration":0.5,"msg_i":{i},"msg":"{msg}"}}"# + ) + }; + let dumped = vec![ + (format!("{base}|1"), chunk(1, "two ")), + (format!("{base}|0"), chunk(0, "one ")), + (format!("{base}|2"), chunk(2, "three")), + ]; + + let records = reassemble(dumped); + assert_eq!(records.len(), 1); + match &records[0] { + DiagnosticRecord::CloudInit { event, chunks } => { + assert_eq!(*chunks, 3); + assert_eq!(event.message, "one two three"); + assert_eq!(event.event_type, "finish"); + assert_eq!(event.name, "modules-final/long"); + assert_eq!(event.result.as_deref(), Some("SUCCESS")); + } + other => panic!("expected cloud-init event, got {other:?}"), + } + } } diff --git a/libazureinit-kvp/src/lib.rs b/libazureinit-kvp/src/lib.rs index 2e94b14a..54dc57ca 100644 --- a/libazureinit-kvp/src/lib.rs +++ b/libazureinit-kvp/src/lib.rs @@ -21,7 +21,8 @@ mod vm_id; pub use cli::run; pub use diagnostics::{ - DiagnosticEvent, DiagnosticRecord, DiagnosticsKvp, MAX_CHUNK_BYTES, + CloudInitEvent, DiagnosticEvent, DiagnosticRecord, DiagnosticsKvp, + MAX_CHUNK_BYTES, }; pub use error::KvpError; pub use report::{ diff --git a/libazureinit-kvp/tests/diagnostics.rs b/libazureinit-kvp/tests/diagnostics.rs index 052e924b..033309a9 100644 --- a/libazureinit-kvp/tests/diagnostics.rs +++ b/libazureinit-kvp/tests/diagnostics.rs @@ -25,6 +25,73 @@ fn diagnostics(dir: &TempDir) -> DiagnosticsKvp { DiagnosticsKvp::new(store, VM_ID, PREFIX) } +/// Real cloud-init reporting entries captured from a guest pool 1 file. +/// Each tuple is one record's `(key, JSON value)`. +const CLOUD_INIT_RECORDS: &[(&str, &str)] = &[ + ( + "CLOUD_INIT|1785187982|finish|modules-final/config-scripts_user|0e5e179d-5341-478b-8456-fbb90621bdf8|e5f01809-a7a3-4279-aa64-1f18e21eda6e", + r#"{"name":"modules-final/config-scripts_user","type":"finish","ts":"2026-07-27T21:33:24.339006+00:00","result":"SUCCESS","duration":0.0006448590000012189,"msg":"config-scripts_user ran successfully and took 0.001 seconds"}"#, + ), + ( + "CLOUD_INIT|1785187982|start|modules-final/config-ssh_authkey_fingerprints|0e5e179d-5341-478b-8456-fbb90621bdf8|c4d4a08d-fe93-4c7a-9be6-9a38c212e212", + r#"{"name":"modules-final/config-ssh_authkey_fingerprints","type":"start","ts":"2026-07-27T21:33:24.339170+00:00","msg":"running config-ssh_authkey_fingerprints with frequency once-per-instance"}"#, + ), + ( + "CLOUD_INIT|1785187982|finish|modules-final|0e5e179d-5341-478b-8456-fbb90621bdf8|126f969f-13fd-4b4b-a136-b7114518491f", + r#"{"name":"modules-final","type":"finish","ts":"2026-07-27T21:33:24.431885+00:00","result":"SUCCESS","duration":0.340712044,"msg":"running modules for final"}"#, + ), +]; + +#[test] +fn reads_and_parses_real_cloud_init_pool() { + let dir = TempDir::new().unwrap(); + let store = + KvpPoolStore::new_in(KvpPool::Guest, dir.path(), PoolMode::Safe) + .unwrap(); + for &(key, value) in CLOUD_INIT_RECORDS { + store.append(key, value).unwrap(); + } + + // A DiagnosticsKvp with no azure-init identity still reads cloud-init + // entries written by another agent. + let diagnostics = DiagnosticsKvp::new(store, "", ""); + let records = diagnostics.records().unwrap(); + assert_eq!(records.len(), CLOUD_INIT_RECORDS.len()); + + // Every record decodes as a cloud-init event (none fall back to raw). + for record in &records { + assert!(matches!(record, DiagnosticRecord::CloudInit { .. })); + } + + match &records[0] { + DiagnosticRecord::CloudInit { event, chunks } => { + assert_eq!(*chunks, 1); + assert_eq!(event.event_type, "finish"); + assert_eq!(event.name, "modules-final/config-scripts_user"); + assert_eq!( + event.vm_id.as_deref(), + Some("0e5e179d-5341-478b-8456-fbb90621bdf8") + ); + assert_eq!(event.result.as_deref(), Some("SUCCESS")); + assert_eq!( + event.message, + "config-scripts_user ran successfully and took 0.001 seconds" + ); + } + other => panic!("expected cloud-init event, got {other:?}"), + } + + // A `start` event carries neither result nor duration. + match &records[1] { + DiagnosticRecord::CloudInit { event, .. } => { + assert_eq!(event.event_type, "start"); + assert!(event.result.is_none()); + assert!(event.duration.is_none()); + } + other => panic!("expected cloud-init event, got {other:?}"), + } +} + #[test] fn short_event_round_trips_as_single_record() { let dir = TempDir::new().unwrap(); From a0b4a2522eea2b469e319671894d4020a27709bf Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Mon, 27 Jul 2026 16:20:26 -0700 Subject: [PATCH 05/32] Updating tests to meet codecov standards --- libazureinit-kvp/src/diagnostics.rs | 136 +++++++++++++++------------- libazureinit-kvp/tests/cli.rs | 70 ++++++++++++++ 2 files changed, 144 insertions(+), 62 deletions(-) diff --git a/libazureinit-kvp/src/diagnostics.rs b/libazureinit-kvp/src/diagnostics.rs index fc29d64a..ee9a4d81 100644 --- a/libazureinit-kvp/src/diagnostics.rs +++ b/libazureinit-kvp/src/diagnostics.rs @@ -755,6 +755,10 @@ mod tests { #[rstest] #[case::event("p|vm|INFO|name|id", "event")] + #[case::cloud_init( + "CLOUD_INIT|1785187982|finish|name|vmid|uuid", + "cloud-init" + )] #[case::raw_single_segment("PROVISIONING_REPORT", "raw")] #[case::raw_too_few_segments("a|b|INFO|c", "raw")] #[case::raw_too_many_segments("a|b|INFO|c|d|e", "raw")] @@ -982,68 +986,68 @@ mod tests { #[test] fn cloud_init_finish_record_decodes_all_fields() { - let record = classify_record( - CLOUD_INIT_KEY_FINISH.to_string(), - vec![CLOUD_INIT_VALUE_FINISH.to_string()], - ); - match record { - DiagnosticRecord::CloudInit { event, chunks } => { - assert_eq!(chunks, 1); - assert_eq!(event.incarnation, "1785187982"); - assert_eq!(event.event_type, "finish"); - assert_eq!(event.name, "modules-final/config-scripts_user"); - assert_eq!(event.vm_id.as_deref(), Some(CLOUD_INIT_VM_ID)); - assert_eq!(event.uuid, "e5f01809-a7a3-4279-aa64-1f18e21eda6e"); - assert_eq!( - event.timestamp.as_deref(), - Some("2026-07-27T21:33:24.339006+00:00") - ); - assert_eq!(event.result.as_deref(), Some("SUCCESS")); - assert!(event - .duration - .is_some_and( - |d| (d - 0.000_644_859_000_001_218_9).abs() < 1e-12 - )); - assert_eq!( - event.message, - "config-scripts_user ran successfully and took 0.001 \ - seconds" - ); - } - other => panic!("expected cloud-init event, got {other:?}"), - } + // A single `matches!` covers every field (with a tolerance on the + // float duration) and leaves no unreachable arm to cover. + assert!(matches!( + classify_record( + CLOUD_INIT_KEY_FINISH.to_string(), + vec![CLOUD_INIT_VALUE_FINISH.to_string()], + ), + DiagnosticRecord::CloudInit { event, chunks: 1 } + if event.incarnation == "1785187982" + && event.event_type == "finish" + && event.name == "modules-final/config-scripts_user" + && event.vm_id.as_deref() == Some(CLOUD_INIT_VM_ID) + && event.uuid == "e5f01809-a7a3-4279-aa64-1f18e21eda6e" + && event.timestamp.as_deref() + == Some("2026-07-27T21:33:24.339006+00:00") + && event.result.as_deref() == Some("SUCCESS") + && event.duration.is_some_and(|d| { + (d - 0.000_644_859_000_001_218_9).abs() < 1e-12 + }) + && event.message + == "config-scripts_user ran successfully and took \ + 0.001 seconds" + )); } #[test] fn cloud_init_start_record_has_no_result_or_duration() { let value = r#"{"name":"modules-final/config-keys_to_console","type":"start","ts":"2026-07-27T21:33:24.344349+00:00","msg":"running config-keys_to_console with frequency once-per-instance"}"#; let key = "CLOUD_INIT|1785187982|start|modules-final/config-keys_to_console|0e5e179d-5341-478b-8456-fbb90621bdf8|7792621b-b339-4274-8b71-2a3dcbd2db4e"; - match classify_record(key.to_string(), vec![value.to_string()]) { - DiagnosticRecord::CloudInit { event, .. } => { - assert_eq!(event.event_type, "start"); - assert!(event.result.is_none()); - assert!(event.duration.is_none()); - assert_eq!( - event.message, - "running config-keys_to_console with frequency \ - once-per-instance" - ); + assert_eq!( + classify_record(key.to_string(), vec![value.to_string()]), + DiagnosticRecord::CloudInit { + event: CloudInitEvent { + incarnation: "1785187982".to_string(), + event_type: "start".to_string(), + name: "modules-final/config-keys_to_console".to_string(), + vm_id: Some(CLOUD_INIT_VM_ID.to_string()), + uuid: "7792621b-b339-4274-8b71-2a3dcbd2db4e".to_string(), + timestamp: Some( + "2026-07-27T21:33:24.344349+00:00".to_string() + ), + result: None, + duration: None, + message: "running config-keys_to_console with frequency \ + once-per-instance" + .to_string(), + }, + chunks: 1, } - other => panic!("expected cloud-init event, got {other:?}"), - } + ); } #[test] fn cloud_init_key_with_invalid_json_is_malformed() { - match classify_record( - CLOUD_INIT_KEY_FINISH.to_string(), - vec!["not json".to_string()], - ) { - DiagnosticRecord::Malformed { reason, .. } => { - assert!(reason.contains("cloud-init")); - } - other => panic!("expected malformed, got {other:?}"), - } + assert!(matches!( + classify_record( + CLOUD_INIT_KEY_FINISH.to_string(), + vec!["not json".to_string()], + ), + DiagnosticRecord::Malformed { reason, .. } + if reason.contains("cloud-init") + )); } #[test] @@ -1064,16 +1068,24 @@ mod tests { ]; let records = reassemble(dumped); - assert_eq!(records.len(), 1); - match &records[0] { - DiagnosticRecord::CloudInit { event, chunks } => { - assert_eq!(*chunks, 3); - assert_eq!(event.message, "one two three"); - assert_eq!(event.event_type, "finish"); - assert_eq!(event.name, "modules-final/long"); - assert_eq!(event.result.as_deref(), Some("SUCCESS")); - } - other => panic!("expected cloud-init event, got {other:?}"), - } + assert_eq!( + records, + vec![DiagnosticRecord::CloudInit { + event: CloudInitEvent { + incarnation: "1785187982".to_string(), + event_type: "finish".to_string(), + name: "modules-final/long".to_string(), + vm_id: Some(CLOUD_INIT_VM_ID.to_string()), + uuid: "abc12345-1111-2222-3333-444455556666".to_string(), + timestamp: Some( + "2026-07-27T21:33:24.339006+00:00".to_string() + ), + result: Some("SUCCESS".to_string()), + duration: Some(0.5), + message: "one two three".to_string(), + }, + chunks: 3, + }] + ); } } diff --git a/libazureinit-kvp/tests/cli.rs b/libazureinit-kvp/tests/cli.rs index e28ef570..03f56fd4 100644 --- a/libazureinit-kvp/tests/cli.rs +++ b/libazureinit-kvp/tests/cli.rs @@ -334,6 +334,76 @@ fn dump_parse_diagnostics_json_reassembles_and_classifies() { assert!(out_raw.contains("PROVISIONING_REPORT")); } +#[test] +fn dump_parse_diagnostics_text_renders_cloud_init_event() { + let dir = TempDir::new().unwrap(); + // A finish event carries every optional field (vm_id, result, ts, + // duration); a start event omits result and duration. + assert_success(kvp(&with_dir( + &dir, + &[ + "write", + "--append", + "CLOUD_INIT|1785187982|finish|modules-final/config-scripts_user|0e5e179d-5341-478b-8456-fbb90621bdf8|e5f01809-a7a3-4279-aa64-1f18e21eda6e", + r#"{"name":"modules-final/config-scripts_user","type":"finish","ts":"2026-07-27T21:33:24.339006+00:00","result":"SUCCESS","duration":0.5,"msg":"scripts ran"}"#, + ], + ))); + assert_success(kvp(&with_dir( + &dir, + &[ + "write", + "--append", + "CLOUD_INIT|1785187982|start|modules-final/config-keys_to_console|0e5e179d-5341-478b-8456-fbb90621bdf8|7792621b-b339-4274-8b71-2a3dcbd2db4e", + r#"{"name":"modules-final/config-keys_to_console","type":"start","ts":"2026-07-27T21:33:24.344349+00:00","msg":"running keys_to_console"}"#, + ], + ))); + + let out = + assert_success(kvp(&with_dir(&dir, &["dump", "--parse-diagnostics"]))); + // The finish event renders every optional field. + assert!(out.contains("cloud-init-event type=finish")); + assert!(out.contains("name=modules-final/config-scripts_user")); + assert!(out.contains("vm_id=0e5e179d-5341-478b-8456-fbb90621bdf8")); + assert!(out.contains("result=SUCCESS")); + assert!(out.contains("ts=2026-07-27T21:33:24.339006+00:00")); + assert!(out.contains("duration=0.5")); + assert!(out.contains("incarnation=1785187982")); + assert!(out.contains("chunks=1")); + assert!(out.contains("message=scripts ran")); + // The start event omits result and duration. + assert!(out.contains("cloud-init-event type=start")); +} + +#[test] +fn dump_parse_diagnostics_json_renders_cloud_init_event() { + let dir = TempDir::new().unwrap(); + assert_success(kvp(&with_dir( + &dir, + &[ + "write", + "--append", + "CLOUD_INIT|1785187982|finish|modules-final/config-scripts_user|0e5e179d-5341-478b-8456-fbb90621bdf8|e5f01809-a7a3-4279-aa64-1f18e21eda6e", + r#"{"name":"modules-final/config-scripts_user","type":"finish","ts":"2026-07-27T21:33:24.339006+00:00","result":"SUCCESS","duration":0.5,"msg":"scripts ran"}"#, + ], + ))); + + let out = assert_success(kvp(&with_dir( + &dir, + &["--json", "dump", "--parse-diagnostics"], + ))); + assert!(out.contains("\"kind\":\"cloud-init-event\"")); + assert!(out.contains("\"incarnation\":\"1785187982\"")); + assert!(out.contains("\"type\":\"finish\"")); + assert!(out.contains("\"name\":\"modules-final/config-scripts_user\"")); + assert!(out.contains("\"vm_id\":\"0e5e179d-5341-478b-8456-fbb90621bdf8\"")); + assert!(out.contains("\"uuid\":\"e5f01809-a7a3-4279-aa64-1f18e21eda6e\"")); + assert!(out.contains("\"ts\":\"2026-07-27T21:33:24.339006+00:00\"")); + assert!(out.contains("\"result\":\"SUCCESS\"")); + assert!(out.contains("\"duration\":0.5")); + assert!(out.contains("\"chunks\":1")); + assert!(out.contains("\"message\":\"scripts ran\"")); +} + #[test] fn dump_parse_diagnostics_filters_by_level() { let dir = TempDir::new().unwrap(); From 0d011f22bbe8513061984383ac506ae35b3a2742 Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Mon, 27 Jul 2026 16:32:57 -0700 Subject: [PATCH 06/32] Addressing CoPilot review --- libazureinit-kvp/src/cli.rs | 14 ++++++++------ libazureinit-kvp/src/diagnostics.rs | 23 +++++++++++++++-------- libazureinit-kvp/tests/diagnostics.rs | 4 ++++ 3 files changed, 27 insertions(+), 14 deletions(-) diff --git a/libazureinit-kvp/src/cli.rs b/libazureinit-kvp/src/cli.rs index 58ef644b..525f0361 100644 --- a/libazureinit-kvp/src/cli.rs +++ b/libazureinit-kvp/src/cli.rs @@ -100,22 +100,24 @@ enum Command { parse_diagnostics: bool, /// Also print raw (non-event) records such as PROVISIONING_REPORT. /// Only applies to the unfiltered view; --level/--name/--tail - /// produce an events-only view where raw records never appear. + /// produce an azure-init events-only view where raw records + /// never appear. #[arg( long, requires = "parse_diagnostics", conflicts_with_all = ["level", "name", "tail"] )] include_raw: bool, - /// Only show events at this level (error, warn, info, debug, - /// trace). + /// Only show azure-init events at this level (error, warn, info, + /// debug, trace). #[arg(long, requires = "parse_diagnostics")] level: Option, - /// Only show events whose name contains this substring. + /// Only show azure-init events whose name contains this + /// substring. #[arg(long, requires = "parse_diagnostics")] name: Option, - /// Print only the last COUNT events (default 20 when COUNT is - /// omitted). + /// Print only the last COUNT azure-init events (default 20 when + /// COUNT is omitted). #[arg( short = 'n', long = "tail", diff --git a/libazureinit-kvp/src/diagnostics.rs b/libazureinit-kvp/src/diagnostics.rs index ee9a4d81..3e1a128b 100644 --- a/libazureinit-kvp/src/diagnostics.rs +++ b/libazureinit-kvp/src/diagnostics.rs @@ -512,10 +512,10 @@ impl DiagnosticsKvp { .collect()) } - /// Remove every diagnostic key under a single lock: any key that - /// parses as an event key (including every `|` chunk - /// of a multi-record event) or a malformed event key. Raw records - /// such as `PROVISIONING_REPORT` are left intact. + /// Remove every diagnostic key: any key that parses as an event key + /// (including every `|` chunk of a multi-record event) + /// or a malformed event key. Raw records such as `PROVISIONING_REPORT` + /// are left intact. pub fn clear(&self) -> Result<(), KvpError> { let keys: Vec = self .store @@ -549,19 +549,25 @@ fn base_event_key(key: &str) -> &str { } /// Split a key into its base event key and optional trailing subevent -/// index. A chunk key `|` returns `(, Some(i))`; -/// any other key (a single-record event, `PROVISIONING_REPORT`, a -/// malformed key, …) returns `(key, None)`. +/// index. When the trailing segment is numeric and the base parses as an +/// event key — a valid azure-init or cloud-init event, or a malformed one +/// (event-shaped but with an unrecognized level) — returns +/// `(base, Some(index))`; any other key (a single-record event, +/// `PROVISIONING_REPORT`, …) returns `(key, None)`. /// /// The subevent index is the same trailing `|` cloud-init and /// azure-init append to give each chunk a unique key; [`reassemble`] uses /// it both to regroup an event's chunks and to restore their write order. +/// Malformed keys are included so a chunked malformed event still +/// regroups and is cleared consistently with a single-record one. fn split_subevent_index(key: &str) -> (&str, Option) { if let Some((base, index)) = key.rsplit_once(EVENT_KEY_DELIMITER) { if let Ok(index) = index.parse::() { if matches!( classify_key(base), - KeyClass::Event { .. } | KeyClass::CloudInit { .. } + KeyClass::Event { .. } + | KeyClass::CloudInit { .. } + | KeyClass::Malformed { .. } ) { return (base, Some(index)); } @@ -887,6 +893,7 @@ mod tests { #[case::raw_unchanged("PROVISIONING_REPORT", "PROVISIONING_REPORT")] #[case::non_event_numeric_tail_unchanged("foo|3", "foo|3")] #[case::malformed_unchanged("p|vm|NOPE|name|id", "p|vm|NOPE|name|id")] + #[case::malformed_indexed_chunk("p|vm|NOPE|name|id|0", "p|vm|NOPE|name|id")] fn base_event_key_strips_event_subevent_index( #[case] key: &str, #[case] expected: &str, diff --git a/libazureinit-kvp/tests/diagnostics.rs b/libazureinit-kvp/tests/diagnostics.rs index 033309a9..38b0a689 100644 --- a/libazureinit-kvp/tests/diagnostics.rs +++ b/libazureinit-kvp/tests/diagnostics.rs @@ -272,6 +272,10 @@ fn clear_removes_all_diagnostics_regardless_of_scope() { .append("other-agent|other-vm|INFO|x:y|id", "theirs") .unwrap(); diag.store().append("p|vm|NOPE|c:d|id", "junk").unwrap(); + // A chunked malformed event key (its base classifies as malformed) is + // also a diagnostic key, so clear() removes every chunk. + diag.store().append("p|vm|NOPE|c:d|id|0", "junk-0").unwrap(); + diag.store().append("p|vm|NOPE|c:d|id|1", "junk-1").unwrap(); // A raw record survives. diag.store() .append("PROVISIONING_REPORT", "result=success") From 732468382f8ce1d19230fd7d25258dcda1186c70 Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Thu, 13 Aug 2026 15:31:28 -0600 Subject: [PATCH 07/32] feat(kvp): align diagnostics schema with cloud-init and add emit CLI --- libazureinit-kvp/Cargo.toml | 1 + libazureinit-kvp/src/cli.rs | 68 ++++- libazureinit-kvp/src/diagnostics.rs | 417 ++++++++++++++++++-------- libazureinit-kvp/src/store.rs | 37 ++- libazureinit-kvp/tests/cli.rs | 85 ++++-- libazureinit-kvp/tests/diagnostics.rs | 82 +++-- 6 files changed, 497 insertions(+), 193 deletions(-) diff --git a/libazureinit-kvp/Cargo.toml b/libazureinit-kvp/Cargo.toml index 7f81e102..3c0381cd 100644 --- a/libazureinit-kvp/Cargo.toml +++ b/libazureinit-kvp/Cargo.toml @@ -21,6 +21,7 @@ uuid = { version = "1.3", features = ["v4"] } [dev-dependencies] rstest = { version = "0.26", default-features = false } tempfile = "3" +uuid = "1.3" [lib] name = "libazureinit_kvp" diff --git a/libazureinit-kvp/src/cli.rs b/libazureinit-kvp/src/cli.rs index 525f0361..97451195 100644 --- a/libazureinit-kvp/src/cli.rs +++ b/libazureinit-kvp/src/cli.rs @@ -139,6 +139,26 @@ enum Command { key: String, value: String, }, + /// Emit an azure-init diagnostic event: a structured KVP entry keyed + /// `|||||`. + /// Distinct from the raw `write` command. + Emit { + /// Event severity: error, warn, info, debug, or trace. + #[arg(long)] + level: String, + /// Event name, e.g. user:create_user. + #[arg(long)] + name: String, + /// Event message (stored as the record value). + #[arg(long)] + message: String, + /// VM identifier (defaults to the current VM's ID). + #[arg(long)] + vm_id: Option, + /// Event-key prefix (defaults to the reporting agent identifier). + #[arg(long)] + prefix: Option, + }, /// Replace the pool from KEY=VALUE lines read from --file or stdin. Load { /// Read records from PATH instead of stdin. @@ -289,6 +309,13 @@ fn dispatch(cli: Cli, stdout: &mut W) -> Result { } Ok(EXIT_OK) } + Command::Emit { + level, + name, + message, + vm_id, + prefix, + } => emit(&store, level, name, message, vm_id, prefix), Command::Load { file } => load(&store, file), Command::AppendMultiple { file } => append_multiple(&store, file), Command::Delete { key } => delete(&store, stdout, &key, output), @@ -533,10 +560,12 @@ fn diagnostics_records( for record in &records { let line = match record { DiagnosticRecord::Event { event, chunks } => format!( - "event level={} name={} event_id={} chunks={} \ - message={}", - event.level, + "event boot_epoch_time={} event_level={} name={} \ + vm_id={} event_id={} chunks={} message={}", + event.boot_epoch_time, + event.event_level, event.name, + event.vm_id, event.event_id, chunks, event.message @@ -601,7 +630,7 @@ fn diagnostics_events( let mut events = diagnostics.events()?; if let Some(level) = level { - events.retain(|event| event.level == level); + events.retain(|event| event.event_level == level); } if let Some(needle) = name.as_deref() { events.retain(|event| event.name.contains(needle)); @@ -615,8 +644,14 @@ fn diagnostics_events( OutputMode::Text => { for event in &events { let line = format!( - "event level={} name={} event_id={} message={}", - event.level, event.name, event.event_id, event.message + "event boot_epoch_time={} event_level={} name={} \ + vm_id={} event_id={} message={}", + event.boot_epoch_time, + event.event_level, + event.name, + event.vm_id, + event.event_id, + event.message ); writeln!(stdout, "{line}")?; } @@ -691,13 +726,32 @@ fn diagnostics_record_json(record: &DiagnosticRecord) -> serde_json::Value { fn diagnostics_event_json(event: &DiagnosticEvent) -> serde_json::Value { json!({ "kind": "event", - "level": event.level.to_string(), + "boot_epoch_time": event.boot_epoch_time, + "event_level": event.event_level.to_string(), "name": event.name, + "vm_id": event.vm_id, "event_id": event.event_id, "message": event.message, }) } +/// Emit an azure-init diagnostic event with the given fields. +fn emit( + store: &KvpPoolStore, + level: String, + name: String, + message: String, + vm_id: Option, + prefix: Option, +) -> Result { + let level = parse_level_filter(&level)?; + let vm_id = resolve_vm_id(vm_id)?; + let prefix = prefix.unwrap_or_else(|| DEFAULT_AGENT.to_string()); + let diagnostics = DiagnosticsKvp::new(store.clone(), vm_id, prefix); + diagnostics.emit(level, name, message)?; + Ok(EXIT_OK) +} + fn report_success( store: &KvpPoolStore, vm_id: Option, diff --git a/libazureinit-kvp/src/diagnostics.rs b/libazureinit-kvp/src/diagnostics.rs index 3e1a128b..72c54ffb 100644 --- a/libazureinit-kvp/src/diagnostics.rs +++ b/libazureinit-kvp/src/diagnostics.rs @@ -8,15 +8,15 @@ //! opaque bytes, [`DiagnosticsKvp`] understands the telemetry //! conventions azure-init writes into the guest pool: //! -//! - **Event keys** encode structured metadata as a five-segment, +//! - **Event keys** encode structured metadata as a six-segment, //! pipe-delimited string -//! (`||||`). +//! (`|||||`). //! - **Chunking**: a single KVP record caps the value at the store's //! per-record limit (see [`MAX_CHUNK_BYTES`] for the safe-mode value). //! Longer messages are split at UTF-8 codepoint boundaries into //! multiple records written atomically under a single lock. Each chunk //! gets a unique key — the event key with a `|` suffix -//! (`|||||`), +//! (`||||||`), //! matching cloud-init's naming — so the Hyper-V host, which keeps only //! one record per key, retains every chunk. The chunks are regrouped //! into one event on read. @@ -38,8 +38,7 @@ //! //! ``` //! use libazureinit_kvp::{ -//! DiagnosticEvent, DiagnosticsKvp, KvpPool, KvpPoolStore, PoolMode, -//! MAX_CHUNK_BYTES, +//! DiagnosticsKvp, KvpPool, KvpPoolStore, PoolMode, MAX_CHUNK_BYTES, //! }; //! use tracing::Level; //! @@ -54,17 +53,16 @@ //! DiagnosticsKvp::new(store, "vm-1234", "azure-init-doc"); //! //! // A short event lands in a single record. -//! diagnostics.emit(&DiagnosticEvent::new( +//! diagnostics.emit( //! Level::INFO, //! "user:create_user", //! "Creating user azureuser", -//! ))?; +//! )?; //! //! // A long message is split across records each with a unique //! // `|`-suffixed key, and reassembled on read. //! let long = "x".repeat(MAX_CHUNK_BYTES * 2 + 10); -//! diagnostics -//! .emit(&DiagnosticEvent::new(Level::DEBUG, "config:dump", &long))?; +//! diagnostics.emit(Level::DEBUG, "config:dump", &long)?; //! //! let events = diagnostics.events()?; //! assert_eq!(events.len(), 2); @@ -99,30 +97,41 @@ pub const MAX_CHUNK_BYTES: usize = 1022; const EVENT_KEY_DELIMITER: char = '|'; /// Format a diagnostic event key as its `|`-delimited on-disk string: -/// `||||`. +/// `|||||`. /// -/// [`classify_key`] is the inverse. For example: +/// `boot_epoch_time` is the Unix epoch second the system booted (see +/// [`KvpPoolStore::boot_epoch`](crate::KvpPoolStore::boot_epoch)); it sits +/// in the same slot as cloud-init's incarnation so the two agents' keys +/// line up. The `event_level`/`name`/`vm_id` order likewise mirrors +/// cloud-init's `type`/`name`/`vm_id` layout. [`classify_key`] is the +/// inverse. For example: /// /// ```text -/// azure-init-0.1.0|3f2504e0-4f89-41d3-9a0c-0305e82c3301|INFO|user:create_user|8f3e9c4a-... +/// azure-init-0.1.0|1785187982|INFO|user:create_user|3f2504e0-...|8f3e9c4a-... /// ``` fn format_event_key( prefix: &str, - vm_id: &str, - level: Level, + boot_epoch_time: i64, + event_level: Level, name: &str, + vm_id: &str, event_id: &str, ) -> String { let d = EVENT_KEY_DELIMITER; - format!("{prefix}{d}{vm_id}{d}{level}{d}{name}{d}{event_id}") + format!( + "{prefix}{d}{boot_epoch_time}{d}{event_level}{d}{name}{d}{vm_id}\ + {d}{event_id}" + ) } /// Outcome of inspecting a raw pool key. enum KeyClass<'a> { /// The key is a well-formed azure-init event key. Event { - level: Level, + boot_epoch_time: i64, + event_level: Level, name: &'a str, + vm_id: &'a str, event_id: &'a str, }, /// The key is a well-formed cloud-init reporting event key @@ -134,7 +143,8 @@ enum KeyClass<'a> { vm_id: Option<&'a str>, uuid: &'a str, }, - /// The key has five segments but is not a valid event. + /// The key has the azure-init six-segment shape but is not a valid + /// event (for example, an unrecognized level). Malformed { reason: String }, /// The key is not an event key (e.g. `PROVISIONING_REPORT`). Raw, @@ -146,31 +156,49 @@ fn classify_key(key: &str) -> KeyClass<'_> { return classify_cloud_init_key(key); } + // azure-init: + // `|||||` let mut segments = key.split(EVENT_KEY_DELIMITER); // `str::split` always yields at least one element. let _prefix = segments.next(); - let (Some(_vm_id), Some(level), Some(name), Some(event_id)) = ( + let ( + Some(boot_epoch), + Some(event_level), + Some(name), + Some(vm_id), + Some(event_id), + ) = ( segments.next(), segments.next(), segments.next(), segments.next(), - ) else { + segments.next(), + ) + else { return KeyClass::Raw; }; if segments.next().is_some() { - // More than five segments: a `|` leaked into a field. + // More than six segments: a `|` leaked into a field. return KeyClass::Raw; } - match level.parse::() { - Ok(level) => KeyClass::Event { - level, + // A non-numeric boot epoch means this is not an azure-init event key; + // keep it as an opaque record rather than a malformed event. + let Ok(boot_epoch_time) = boot_epoch.parse::() else { + return KeyClass::Raw; + }; + + match event_level.parse::() { + Ok(event_level) => KeyClass::Event { + boot_epoch_time, + event_level, name, + vm_id, event_id, }, Err(_) => KeyClass::Malformed { - reason: format!("unrecognized level {level:?}"), + reason: format!("unrecognized level {event_level:?}"), }, } } @@ -258,42 +286,36 @@ fn reject_delimiter(field: &'static str, value: &str) -> Result<(), KvpError> { Ok(()) } -/// A single diagnostic event. +/// A single azure-init diagnostic event — the decoded form of one +/// azure-init KVP entry. /// -/// Construct one with [`new`](Self::new) (which generates a fresh -/// `event_id`) and hand it to [`DiagnosticsKvp::emit`]; events read back -/// via [`DiagnosticsKvp::records`] carry the same fields, decoded from -/// the pool. +/// This is the crate's definition of what an azure-init KVP event looks +/// like: the boot-scoped `boot_epoch_time`/`vm_id`, the event-scoped +/// `event_level`/`name`/`event_id`, and the free-form `message`. Write one +/// with [`DiagnosticsKvp::emit`] (which stamps the boot-scoped fields from +/// the session and generates a fresh `event_id`); read events back, fully +/// populated, via [`records`](DiagnosticsKvp::records) / +/// [`events`](DiagnosticsKvp::events). #[derive(Clone, Debug, PartialEq, Eq)] +#[non_exhaustive] pub struct DiagnosticEvent { + /// Unix epoch second the system booted, shared by every event of one + /// boot. Distinguishes this boot's telemetry from a previous boot's. + pub boot_epoch_time: i64, /// Severity of the event. - pub level: Level, + pub event_level: Level, /// Formatted event name, e.g. `user:create_user`. pub name: String, - /// Per-emit identifier. [`new`](Self::new) generates a UUIDv4; - /// every chunk of one emitted event shares this value. + /// VM identifier the event was emitted under. + pub vm_id: String, + /// Per-emit identifier (UUIDv4); every chunk of one emitted event + /// shares this value. pub event_id: String, /// Literal value bytes written to the pool. The diagnostics layer /// imposes no format on this string. pub message: String, } -impl DiagnosticEvent { - /// Create an event with a freshly generated `event_id` (UUIDv4). - pub fn new( - level: Level, - name: impl Into, - message: impl Into, - ) -> Self { - Self { - level, - name: name.into(), - event_id: Uuid::new_v4().to_string(), - message: message.into(), - } - } -} - /// The JSON payload cloud-init stores as a reporting event's KVP value. /// /// Only the fields the diagnostics reader surfaces are deserialized; @@ -319,6 +341,7 @@ struct CloudInitValue { /// event details as a JSON value; this type is the decoded union of the /// two. #[derive(Clone, Debug, PartialEq)] +#[non_exhaustive] pub struct CloudInitEvent { /// Boot-time incarnation stamp from the key. cloud-init uses it to /// distinguish this boot's records from a previous boot's. @@ -345,6 +368,7 @@ pub struct CloudInitEvent { /// A single record read back from the pool and classified by /// [`DiagnosticsKvp::records`]. #[derive(Clone, Debug, PartialEq)] +#[non_exhaustive] pub enum DiagnosticRecord { /// A reassembled azure-init diagnostic event. Event { @@ -368,8 +392,8 @@ pub enum DiagnosticRecord { /// The reassembled record value. value: String, }, - /// A record whose key has five segments but is not a valid event - /// (for example, an unrecognized level). + /// A record whose key is event-shaped but is not a valid event (for + /// example, an unrecognized level or invalid cloud-init JSON). Malformed { /// The record key. key: String, @@ -422,7 +446,12 @@ impl DiagnosticsKvp { &self.event_prefix } - /// Write `event`. + /// Emit an azure-init diagnostic event: format the key + /// `|||||` + /// and write `message` as its value. `boot_epoch_time` (from + /// [`KvpPoolStore::boot_epoch`](crate::KvpPoolStore::boot_epoch)), + /// `vm_id`, and `event_prefix` come from this layer; the `event_id` is + /// a fresh UUIDv4. /// /// Messages longer than the store's per-record value limit are split /// at UTF-8 codepoint boundaries and written as multiple records @@ -433,24 +462,31 @@ impl DiagnosticsKvp { /// regrouped by [`records`](Self::records) on read. /// /// Returns [`KvpError::EventFieldContainsDelimiter`] if the - /// `event_prefix`, `vm_id`, `name`, or `event_id` contains the `|` - /// key delimiter, which would make the key ambiguous to - /// [`records`](Self::records). - pub fn emit(&self, event: &DiagnosticEvent) -> Result<(), KvpError> { + /// `event_prefix`, `vm_id`, or `name` contains the `|` key delimiter, + /// which would make the key ambiguous to [`records`](Self::records). + pub fn emit( + &self, + event_level: Level, + name: impl Into, + message: impl Into, + ) -> Result<(), KvpError> { + let name = name.into(); reject_delimiter("event_prefix", &self.event_prefix)?; reject_delimiter("vm_id", &self.vm_id)?; - reject_delimiter("name", &event.name)?; - reject_delimiter("event_id", &event.event_id)?; + reject_delimiter("name", &name)?; + let boot_epoch_time = self.store.boot_epoch()?; + let event_id = Uuid::new_v4().to_string(); let key = format_event_key( &self.event_prefix, + boot_epoch_time, + event_level, + &name, &self.vm_id, - event.level, - &event.name, - &event.event_id, + &event_id, ); - self.write_chunked(&key, &event.message) + self.write_chunked(&key, &message.into()) } /// Split `value` at the store's per-record limit and append the @@ -613,19 +649,24 @@ fn reassemble(dumped: Vec<(String, String)>) -> Vec { /// `chunk_values` holds every record that shared the base event key, /// already ordered by subevent index by [`reassemble`], and is never /// empty. azure-init events and raw records concatenate their values -/// directly. cloud-init writes a full JSON object per chunk, so those are -/// decoded and their `msg` slices joined into the full message. +/// directly. cloud-init writes each chunk as a metadata object carrying a +/// slice of the escaped message, so those are stitched back together and +/// decoded (see [`decode_cloud_init_value`]). fn classify_record(key: String, chunk_values: Vec) -> DiagnosticRecord { let chunks = chunk_values.len(); match classify_key(&key) { KeyClass::Event { - level, + boot_epoch_time, + event_level, name, + vm_id, event_id, } => DiagnosticRecord::Event { event: DiagnosticEvent { - level, + boot_epoch_time, + event_level, name: name.to_string(), + vm_id: vm_id.to_string(), event_id: event_id.to_string(), message: chunk_values.concat(), }, @@ -640,39 +681,27 @@ fn classify_record(key: String, chunk_values: Vec) -> DiagnosticRecord { } => { // Own the key-derived fields up front so `key` and // `chunk_values` are free to move into a `Malformed` record - // when a chunk's JSON fails to parse. + // when a chunk's value fails to decode. let incarnation = incarnation.to_string(); let event_type = event_type.to_string(); let name = name.to_string(); let vm_id = vm_id.map(str::to_string); let uuid = uuid.to_string(); - match chunk_values - .iter() - .map(|value| serde_json::from_str::(value)) - .collect::, _>>() - { - Ok(parts) => { - // Chunks arrive already ordered by subevent index, so - // join the message slices directly. Every chunk repeats - // the same metadata, so read it from the first chunk. - let message: String = - parts.iter().map(|part| part.msg.as_str()).collect(); - let first = &parts[0]; - DiagnosticRecord::CloudInit { - event: CloudInitEvent { - incarnation, - event_type, - name, - vm_id, - uuid, - timestamp: first.ts.clone(), - result: first.result.clone(), - duration: first.duration, - message, - }, - chunks, - } - } + match decode_cloud_init_value(&chunk_values) { + Ok(decoded) => DiagnosticRecord::CloudInit { + event: CloudInitEvent { + incarnation, + event_type, + name, + vm_id, + uuid, + timestamp: decoded.timestamp, + result: decoded.result, + duration: decoded.duration, + message: decoded.message, + }, + chunks, + }, Err(err) => DiagnosticRecord::Malformed { key, value: chunk_values.concat(), @@ -692,6 +721,93 @@ fn classify_record(key: String, chunk_values: Vec) -> DiagnosticRecord { } } +/// The fields [`classify_record`] pulls from a cloud-init event's +/// value(s): the reassembled `message` plus the metadata a `finish` event +/// carries. +struct DecodedCloudInit { + timestamp: Option, + result: Option, + duration: Option, + message: String, +} + +/// Marker preceding a cloud-init value's message field: `"msg":"`. +const CLOUD_INIT_MSG_MARKER: &str = "\"msg\":\""; + +/// Decode a cloud-init event's chunk value(s) into its metadata and full +/// message. +/// +/// A single-record event is a complete JSON object, parsed directly. A +/// multi-record event was split by cloud-init's `_break_down`, which +/// slices the JSON-*escaped* message at character boundaries — so an +/// individual chunk can end mid-escape (e.g. a `\n` split into `\` and +/// `n`) and is not valid JSON on its own. We therefore recover the raw +/// (still-escaped) `msg` slice from each chunk, concatenate the slices in +/// order, and unescape the whole once; the metadata is read from the +/// first chunk's non-`msg` fields. +fn decode_cloud_init_value( + chunks: &[String], +) -> Result { + if let [only] = chunks { + let value: CloudInitValue = + serde_json::from_str(only).map_err(|e| e.to_string())?; + return Ok(DecodedCloudInit { + timestamp: value.ts, + result: value.result, + duration: value.duration, + message: value.msg, + }); + } + + // Concatenate each chunk's raw escaped `msg` slice, then unescape the + // reassembled string once so escapes split across chunks are rejoined + // first. + let mut escaped = String::new(); + for chunk in chunks { + escaped.push_str(escaped_msg_slice(chunk)?); + } + let message: String = serde_json::from_str(&format!("\"{escaped}\"")) + .map_err(|e| e.to_string())?; + + // Metadata is identical across chunks; take it from the first, whose + // non-`msg` prefix is always valid JSON. + let meta = chunk_metadata(&chunks[0])?; + Ok(DecodedCloudInit { + timestamp: meta.ts, + result: meta.result, + duration: meta.duration, + message, + }) +} + +/// Recover a chunk's raw (still-escaped) `msg` slice — the bytes between +/// the `"msg":"` marker and the closing `"}` — without unescaping. +fn escaped_msg_slice(chunk: &str) -> Result<&str, String> { + let start = chunk + .find(CLOUD_INIT_MSG_MARKER) + .ok_or("chunk is missing a \"msg\" field")? + + CLOUD_INIT_MSG_MARKER.len(); + let end = chunk + .strip_suffix("\"}") + .map(str::len) + .ok_or("chunk does not end with '\"}'")?; + chunk + .get(start..end) + .ok_or_else(|| "chunk \"msg\" field is malformed".to_string()) +} + +/// Parse a chunk's non-`msg` metadata (`ts`/`result`/`duration`) from the +/// portion before its `,"msg":"` field. That prefix is always valid JSON +/// even when the trailing `msg` slice is not. +fn chunk_metadata(chunk: &str) -> Result { + let marker = format!(",{CLOUD_INIT_MSG_MARKER}"); + let end = chunk + .find(&marker) + .ok_or("chunk is missing a \"msg\" field")?; + serde_json::from_str(&format!("{}}}", &chunk[..end])) + .map_err(|e| e.to_string()) +} + #[cfg(test)] mod tests { use super::*; @@ -700,28 +816,37 @@ mod tests { const PREFIX: &str = "azure-init-0.1.0"; const VM_ID: &str = "3f2504e0-4f89-41d3-9a0c-0305e82c3301"; const EVENT_ID: &str = "8f3e9c4a-1b2c-4d5e-9f01-234567890abc"; + const BOOT_EPOCH: i64 = 1_700_000_000; #[test] fn event_key_formats_and_classifies() { let formatted = format_event_key( PREFIX, - VM_ID, + BOOT_EPOCH, Level::INFO, "user:create_user", + VM_ID, EVENT_ID, ); assert_eq!( formatted, - format!("{PREFIX}|{VM_ID}|INFO|user:create_user|{EVENT_ID}") + format!( + "{PREFIX}|{BOOT_EPOCH}|INFO|user:create_user|{VM_ID}|\ + {EVENT_ID}" + ) ); assert!(matches!( classify_key(&formatted), KeyClass::Event { - level, + boot_epoch_time, + event_level, name, + vm_id, event_id, - } if level == Level::INFO + } if boot_epoch_time == BOOT_EPOCH + && event_level == Level::INFO && name == "user:create_user" + && vm_id == VM_ID && event_id == EVENT_ID )); } @@ -737,14 +862,15 @@ mod tests { ] { let key = format_event_key( PREFIX, - VM_ID, + BOOT_EPOCH, expected, "span:event", + VM_ID, EVENT_ID, ); assert!(matches!( classify_key(&key), - KeyClass::Event { level, .. } if level == expected + KeyClass::Event { event_level, .. } if event_level == expected )); } } @@ -760,16 +886,17 @@ mod tests { } #[rstest] - #[case::event("p|vm|INFO|name|id", "event")] + #[case::event("p|100|INFO|name|vm|id", "event")] #[case::cloud_init( "CLOUD_INIT|1785187982|finish|name|vmid|uuid", "cloud-init" )] #[case::raw_single_segment("PROVISIONING_REPORT", "raw")] #[case::raw_too_few_segments("a|b|INFO|c", "raw")] - #[case::raw_too_many_segments("a|b|INFO|c|d|e", "raw")] - #[case::malformed_bad_level("p|vm|NOTALEVEL|name|id", "malformed")] - #[case::malformed_other_level("p|vm|NOPE|name|id", "malformed")] + #[case::raw_too_many_segments("a|100|b|INFO|c|d|e", "raw")] + #[case::raw_non_numeric_boot_epoch("p|notnum|INFO|name|vm|id", "raw")] + #[case::malformed_bad_level("p|100|NOTALEVEL|name|vm|id", "malformed")] + #[case::malformed_other_level("p|100|NOPE|name|vm|id", "malformed")] fn classify_key_categorizes(#[case] key: &str, #[case] expected: &str) { assert_eq!(class_of(key), expected); } @@ -798,12 +925,6 @@ mod tests { assert_eq!(chunks.concat(), payload); } - #[test] - fn diagnostic_event_new_generates_uuid_event_id() { - let event = DiagnosticEvent::new(Level::INFO, "span:name", "message"); - assert!(Uuid::parse_str(&event.event_id).is_ok()); - } - #[test] fn reject_delimiter_flags_pipe() { assert!(reject_delimiter("name", "no pipe here").is_ok()); @@ -822,9 +943,10 @@ mod tests { fn reassemble_groups_chunks_and_classifies() { let key = format_event_key( PREFIX, - VM_ID, + BOOT_EPOCH, Level::INFO, "config:dump", + VM_ID, EVENT_ID, ); @@ -835,7 +957,7 @@ mod tests { "PROVISIONING_REPORT".to_string(), "result=success".to_string(), ), - ("p|vm|NOPE|name|id".to_string(), "junk".to_string()), + ("p|100|NOPE|name|vm|id".to_string(), "junk".to_string()), ]; let records = reassemble(dumped); @@ -845,8 +967,10 @@ mod tests { records[0], DiagnosticRecord::Event { event: DiagnosticEvent { - level: Level::INFO, + boot_epoch_time: BOOT_EPOCH, + event_level: Level::INFO, name: "config:dump".to_string(), + vm_id: VM_ID.to_string(), event_id: EVENT_ID.to_string(), message: "part-one/part-two".to_string(), }, @@ -861,7 +985,14 @@ mod tests { #[test] fn reassemble_keeps_distinct_adjacent_keys_separate() { let make = |event_id: &str| { - format_event_key(PREFIX, VM_ID, Level::INFO, "span:name", event_id) + format_event_key( + PREFIX, + BOOT_EPOCH, + Level::INFO, + "span:name", + VM_ID, + event_id, + ) }; let dumped = vec![ (make("id-1"), "first".to_string()), @@ -880,20 +1011,29 @@ mod tests { } #[rstest] - #[case::indexed_chunk("p|vm|INFO|name|id|0", "p|vm|INFO|name|id")] + #[case::indexed_chunk("p|100|INFO|name|vm|id|0", "p|100|INFO|name|vm|id")] #[case::indexed_chunk_multi_digit( - "p|vm|INFO|name|id|12", - "p|vm|INFO|name|id" + "p|100|INFO|name|vm|id|12", + "p|100|INFO|name|vm|id" + )] + #[case::single_event_unchanged( + "p|100|INFO|name|vm|id", + "p|100|INFO|name|vm|id" )] - #[case::single_event_unchanged("p|vm|INFO|name|id", "p|vm|INFO|name|id")] #[case::cloud_init_indexed_chunk( "CLOUD_INIT|1785187982|finish|mod|vmid|uuid|0", "CLOUD_INIT|1785187982|finish|mod|vmid|uuid" )] #[case::raw_unchanged("PROVISIONING_REPORT", "PROVISIONING_REPORT")] #[case::non_event_numeric_tail_unchanged("foo|3", "foo|3")] - #[case::malformed_unchanged("p|vm|NOPE|name|id", "p|vm|NOPE|name|id")] - #[case::malformed_indexed_chunk("p|vm|NOPE|name|id|0", "p|vm|NOPE|name|id")] + #[case::malformed_unchanged( + "p|100|NOPE|name|vm|id", + "p|100|NOPE|name|vm|id" + )] + #[case::malformed_indexed_chunk( + "p|100|NOPE|name|vm|id|0", + "p|100|NOPE|name|vm|id" + )] fn base_event_key_strips_event_subevent_index( #[case] key: &str, #[case] expected: &str, @@ -905,9 +1045,10 @@ mod tests { fn reassemble_groups_indexed_chunk_keys() { let base = format_event_key( PREFIX, - VM_ID, + BOOT_EPOCH, Level::INFO, "config:dump", + VM_ID, EVENT_ID, ); let dumped = vec![ @@ -922,8 +1063,10 @@ mod tests { records[0], DiagnosticRecord::Event { event: DiagnosticEvent { - level: Level::INFO, + boot_epoch_time: BOOT_EPOCH, + event_level: Level::INFO, name: "config:dump".to_string(), + vm_id: VM_ID.to_string(), event_id: EVENT_ID.to_string(), message: "part-one/part-two/part-three".to_string(), }, @@ -1095,4 +1238,36 @@ mod tests { }] ); } + + #[test] + fn cloud_init_chunks_reassemble_split_json_escape() { + // Regression: cloud-init's `_break_down` re-emits the metadata + // (plus a `msg_i` chunk index) on every chunk and slices the + // JSON-escaped message at character boundaries, so a `\n` escape + // can straddle two chunks — the first ends in a lone backslash and + // is not valid JSON on its own. The reader must rejoin the raw + // slices before unescaping. + let base = "CLOUD_INIT|1785187982|finish|modules-final/x|0e5e179d-5341-478b-8456-fbb90621bdf8|abc12345-1111-2222-3333-444455556666"; + let dumped = vec![ + ( + format!("{base}|0"), + r#"{"name":"modules-final/x","type":"finish","ts":"2026-07-27T21:33:24.339006+00:00","result":"SUCCESS","duration":0.5,"msg_i":0,"msg":"line1\"}"# + .to_string(), + ), + ( + format!("{base}|1"), + r#"{"name":"modules-final/x","type":"finish","ts":"2026-07-27T21:33:24.339006+00:00","result":"SUCCESS","duration":0.5,"msg_i":1,"msg":"nline2"}"# + .to_string(), + ), + ]; + + let records = reassemble(dumped); + assert!(matches!( + &records[..], + [DiagnosticRecord::CloudInit { event, chunks: 2 }] + if event.message == "line1\nline2" + && event.result.as_deref() == Some("SUCCESS") + && event.duration == Some(0.5) + )); + } } diff --git a/libazureinit-kvp/src/store.rs b/libazureinit-kvp/src/store.rs index a067c980..1f722bab 100644 --- a/libazureinit-kvp/src/store.rs +++ b/libazureinit-kvp/src/store.rs @@ -211,7 +211,9 @@ impl KvpPoolStore { let boot_time = boot_time(&*self.ops)?; lock_for_writing(&mut *handle)?; - if handle.metadata()?.mtime <= boot_time { + // Strict `<`: a file whose mtime equals boot time was written + // this boot (during the boot second), so it is not stale. + if handle.metadata()?.mtime < boot_time { handle.set_len(0)?; } } @@ -399,7 +401,20 @@ impl KvpPoolStore { Err(ref e) if e.kind() == ErrorKind::NotFound => return Ok(false), Err(e) => return Err(e.into()), }; - Ok(metadata.mtime <= boot_time(&*self.ops)?) + // Strict `<`: mtime == boot_time means the file was written this + // boot (during the boot second), so it is not stale. + Ok(metadata.mtime < boot_time(&*self.ops)?) + } + + /// The system boot time as a Unix epoch timestamp in seconds, read + /// from `/proc/stat` `btime`. + /// + /// Diagnostic event keys stamp this value so records can be + /// attributed to a specific boot (mirroring cloud-init's + /// incarnation), letting readers tell this boot's telemetry from a + /// previous boot's. + pub fn boot_epoch(&self) -> Result { + boot_time(&*self.ops) } /// Variant of [`is_stale`](Self::is_stale) that takes an explicit @@ -412,7 +427,7 @@ impl KvpPoolStore { Err(ref e) if e.kind() == ErrorKind::NotFound => return Ok(false), Err(e) => return Err(e.into()), }; - Ok(metadata.mtime <= boot_time) + Ok(metadata.mtime < boot_time) } fn iter(&self) -> Result { @@ -3440,7 +3455,7 @@ mod tests { #[test] fn test_clear_if_stale_truncates_when_stale() { - // mtime (0) <= boot_time (10) → triggers set_len branch. + // mtime (0) < boot_time (10) → triggers set_len branch. let (store, ops, p) = mock_store(PoolMode::Safe); preload(&ops, &p, &[("a", "1")]); ops.set_boot_time(10); @@ -3468,6 +3483,20 @@ mod tests { assert_eq!(ops.lock().files.get(&p).unwrap().len(), RECORD_SIZE); } + #[test] + fn test_clear_if_stale_keeps_file_written_in_boot_second() { + // Boundary regression: mtime (10) == boot_time (10) means the + // file was written this boot (during the boot second), so it is + // NOT stale and must survive clear_if_stale (strict `<`). + let (store, ops, p) = mock_store(PoolMode::Safe); + ops.put_file(&p, vec![0u8; RECORD_SIZE], 10); + ops.set_boot_time(10); + + assert!(!store.is_stale().unwrap()); + store.clear_if_stale().unwrap(); + assert_eq!(ops.lock().files.get(&p).unwrap().len(), RECORD_SIZE); + } + #[test] fn test_delete_fails_when_iter_read_fails() { let (store, ops, p) = mock_store(PoolMode::Safe); diff --git a/libazureinit-kvp/tests/cli.rs b/libazureinit-kvp/tests/cli.rs index 03f56fd4..e5c20417 100644 --- a/libazureinit-kvp/tests/cli.rs +++ b/libazureinit-kvp/tests/cli.rs @@ -305,11 +305,21 @@ fn dump_parse_diagnostics_json_reassembles_and_classifies() { // A two-chunk event (same key repeated) plus a raw record. assert_success(kvp(&with_dir( &dir, - &["write", "--append", "azure-init-x|vm|INFO|a:b|id1", "one/"], + &[ + "write", + "--append", + "azure-init-x|100|INFO|a:b|vm|id1", + "one/", + ], ))); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "azure-init-x|vm|INFO|a:b|id1", "two"], + &[ + "write", + "--append", + "azure-init-x|100|INFO|a:b|vm|id1", + "two", + ], ))); assert_success(kvp(&with_dir( &dir, @@ -409,11 +419,11 @@ fn dump_parse_diagnostics_filters_by_level() { let dir = TempDir::new().unwrap(); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "p|vm|INFO|a:b|i1", "info-msg"], + &["write", "--append", "p|100|INFO|a:b|vm|i1", "info-msg"], ))); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "p|vm|ERROR|c:d|i2", "err-msg"], + &["write", "--append", "p|100|ERROR|c:d|vm|i2", "err-msg"], ))); let out = assert_success(kvp(&with_dir( @@ -430,11 +440,11 @@ fn clear_diagnostics_removes_events_and_malformed_keeps_raw() { // An event, a malformed event key, and a raw record. assert_success(kvp(&with_dir( &dir, - &["write", "--append", "p|vm|INFO|a:b|i1", "msg"], + &["write", "--append", "p|100|INFO|a:b|vm|i1", "msg"], ))); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "p|vm|NOPE|c:d|i2", "junk"], + &["write", "--append", "p|100|NOPE|c:d|vm|i2", "junk"], ))); assert_success(kvp(&with_dir( &dir, @@ -460,11 +470,11 @@ fn dump_parse_diagnostics_text_renders_all_record_kinds() { let dir = TempDir::new().unwrap(); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "p|vm|INFO|a:b|id1", "one/"], + &["write", "--append", "p|100|INFO|a:b|vm|id1", "one/"], ))); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "p|vm|INFO|a:b|id1", "two"], + &["write", "--append", "p|100|INFO|a:b|vm|id1", "two"], ))); assert_success(kvp(&with_dir( &dir, @@ -472,7 +482,7 @@ fn dump_parse_diagnostics_text_renders_all_record_kinds() { ))); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "p|vm|NOPE|c:d|id2", "junk"], + &["write", "--append", "p|100|NOPE|c:d|vm|id2", "junk"], ))); let out = assert_success(kvp(&with_dir( @@ -480,10 +490,11 @@ fn dump_parse_diagnostics_text_renders_all_record_kinds() { &["dump", "--parse-diagnostics", "--include-raw"], ))); assert!(out.contains( - "event level=INFO name=a:b event_id=id1 chunks=2 message=one/two" + "event boot_epoch_time=100 event_level=INFO name=a:b vm_id=vm \ + event_id=id1 chunks=2 message=one/two" )); assert!(out.contains("raw key=PROVISIONING_REPORT value=result=success")); - assert!(out.contains("malformed key=p|vm|NOPE|c:d|id2")); + assert!(out.contains("malformed key=p|100|NOPE|c:d|vm|id2")); assert!(out.contains("value=junk")); } @@ -492,11 +503,11 @@ fn dump_parse_diagnostics_tail_limits_to_last_events() { let dir = TempDir::new().unwrap(); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "p|vm|INFO|a:b|i1", "first"], + &["write", "--append", "p|100|INFO|a:b|vm|i1", "first"], ))); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "p|vm|INFO|c:d|i2", "second"], + &["write", "--append", "p|100|INFO|c:d|vm|i2", "second"], ))); let out = assert_success(kvp(&with_dir( @@ -516,7 +527,7 @@ fn dump_parse_diagnostics_tail_defaults_to_20_when_count_omitted() { &[ "write", "--append", - &format!("p|vm|INFO|n:{i}|id{i}"), + &format!("p|100|INFO|n:{i}|vm|id{i}"), &format!("msg{i}"), ], ))); @@ -538,11 +549,11 @@ fn dump_parse_diagnostics_filters_by_name_substring() { let dir = TempDir::new().unwrap(); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "p|vm|INFO|user:add|i1", "u"], + &["write", "--append", "p|100|INFO|user:add|vm|i1", "u"], ))); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "p|vm|INFO|ssh:key|i2", "s"], + &["write", "--append", "p|100|INFO|ssh:key|vm|i2", "s"], ))); let out = assert_success(kvp(&with_dir( @@ -584,11 +595,11 @@ fn dump_parse_diagnostics_json_covers_events_and_malformed() { let dir = TempDir::new().unwrap(); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "p|vm|INFO|a:b|i1", "hello"], + &["write", "--append", "p|100|INFO|a:b|vm|i1", "hello"], ))); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "p|vm|NOPE|c:d|i2", "junk"], + &["write", "--append", "p|100|NOPE|c:d|vm|i2", "junk"], ))); let dump = assert_success(kvp(&with_dir( @@ -607,3 +618,41 @@ fn dump_parse_diagnostics_json_covers_events_and_malformed() { assert!(events.contains("\"message\":\"hello\"")); assert!(!events.contains("\"kind\":\"malformed\"")); } + +#[test] +fn emit_writes_event_readable_by_dump() { + let dir = TempDir::new().unwrap(); + assert_success(kvp(&with_dir( + &dir, + &[ + "emit", + "--level", + "info", + "--name", + "user:create_user", + "--message", + "created azureuser", + "--vm-id", + "vm-emit", + "--prefix", + "azure-init-test", + ], + ))); + + let out = + assert_success(kvp(&with_dir(&dir, &["dump", "--parse-diagnostics"]))); + assert!(out.contains("event_level=INFO")); + assert!(out.contains("vm_id=vm-emit")); + assert!(out.contains("name=user:create_user")); + assert!(out.contains("message=created azureuser")); + + // The JSON view exposes the same structured fields. + let json = assert_success(kvp(&with_dir( + &dir, + &["--json", "dump", "--parse-diagnostics"], + ))); + assert!(json.contains("\"kind\":\"event\"")); + assert!(json.contains("\"event_level\":\"INFO\"")); + assert!(json.contains("\"vm_id\":\"vm-emit\"")); + assert!(json.contains("\"name\":\"user:create_user\"")); +} diff --git a/libazureinit-kvp/tests/diagnostics.rs b/libazureinit-kvp/tests/diagnostics.rs index 38b0a689..efe4dd47 100644 --- a/libazureinit-kvp/tests/diagnostics.rs +++ b/libazureinit-kvp/tests/diagnostics.rs @@ -9,8 +9,8 @@ use std::thread; use libazureinit_kvp::{ - DiagnosticEvent, DiagnosticRecord, DiagnosticsKvp, KvpPool, KvpPoolStore, - PoolMode, MAX_CHUNK_BYTES, + DiagnosticRecord, DiagnosticsKvp, KvpPool, KvpPoolStore, PoolMode, + MAX_CHUNK_BYTES, }; use tempfile::TempDir; use tracing::Level; @@ -100,9 +100,8 @@ fn short_event_round_trips_as_single_record() { assert_eq!(diag.vm_id(), VM_ID); assert_eq!(diag.event_prefix(), PREFIX); - let event = - DiagnosticEvent::new(Level::INFO, "user:create_user", "created"); - diag.emit(&event).unwrap(); + diag.emit(Level::INFO, "user:create_user", "created") + .unwrap(); assert_eq!(diag.store().dump().unwrap().len(), 1); @@ -114,9 +113,20 @@ fn short_event_round_trips_as_single_record() { chunks, } => { assert_eq!(*chunks, 1); - assert_eq!(decoded.level, Level::INFO); + assert_eq!(decoded.event_level, Level::INFO); + assert_eq!(decoded.vm_id, VM_ID); + assert_eq!( + decoded.boot_epoch_time, + diag.store().boot_epoch().unwrap() + ); assert_eq!(decoded.name, "user:create_user"); - assert_eq!(decoded.event_id, event.event_id); + let event_id = uuid::Uuid::parse_str(&decoded.event_id) + .expect("event_id should be a valid UUID"); + assert_eq!( + event_id.get_version_num(), + 4, + "event_id should be a UUIDv4" + ); assert_eq!(decoded.message, "created"); } other => panic!("expected event, got {other:?}"), @@ -129,8 +139,7 @@ fn long_event_splits_across_records_and_reassembles() { let diag = diagnostics(&dir); let message = "x".repeat(MAX_CHUNK_BYTES * 3 + 50); - let event = DiagnosticEvent::new(Level::DEBUG, "config:dump", &message); - diag.emit(&event).unwrap(); + diag.emit(Level::DEBUG, "config:dump", &message).unwrap(); // Split across four records, each with a unique `|` // key so the Hyper-V host (one record per key) keeps every chunk; @@ -168,8 +177,7 @@ fn multi_chunk_event_uses_unique_keys_so_host_keeps_all() { let diag = diagnostics(&dir); let message = "z".repeat(MAX_CHUNK_BYTES * 2 + 1); - diag.emit(&DiagnosticEvent::new(Level::INFO, "big:event", &message)) - .unwrap(); + diag.emit(Level::INFO, "big:event", &message).unwrap(); // Three records, no two sharing a key: the Hyper-V host keeps only one // record per key, so shared keys would silently drop chunks. @@ -192,9 +200,9 @@ fn injected_malformed_key_is_classified() { let dir = TempDir::new().unwrap(); let diag = diagnostics(&dir); - // Five segments but an unrecognized level. + // Six segments but an unrecognized level. diag.store() - .append(&format!("{PREFIX}|{VM_ID}|NOPE|bad:level|id"), "junk") + .append(&format!("{PREFIX}|100|NOPE|bad:level|{VM_ID}|id"), "junk") .unwrap(); let records = diag.records().unwrap(); @@ -210,19 +218,14 @@ fn mixed_records_round_trip_together() { let dir = TempDir::new().unwrap(); let diag = diagnostics(&dir); - diag.emit(&DiagnosticEvent::new(Level::INFO, "a:b", "short")) + diag.emit(Level::INFO, "a:b", "short").unwrap(); + diag.emit(Level::WARN, "c:d", "y".repeat(MAX_CHUNK_BYTES + 5)) .unwrap(); - diag.emit(&DiagnosticEvent::new( - Level::WARN, - "c:d", - "y".repeat(MAX_CHUNK_BYTES + 5), - )) - .unwrap(); diag.store() .append("PROVISIONING_REPORT", "result=success") .unwrap(); diag.store() - .append(&format!("{PREFIX}|{VM_ID}|NOPE|e:f|id"), "junk") + .append(&format!("{PREFIX}|100|NOPE|e:f|{VM_ID}|id"), "junk") .unwrap(); let records = diag.records().unwrap(); @@ -236,14 +239,9 @@ fn clear_removes_events_but_keeps_raw() { let dir = TempDir::new().unwrap(); let diag = diagnostics(&dir); - diag.emit(&DiagnosticEvent::new(Level::INFO, "a:b", "e1")) + diag.emit(Level::INFO, "a:b", "e1").unwrap(); + diag.emit(Level::DEBUG, "c:d", "z".repeat(MAX_CHUNK_BYTES * 2)) .unwrap(); - diag.emit(&DiagnosticEvent::new( - Level::DEBUG, - "c:d", - "z".repeat(MAX_CHUNK_BYTES * 2), - )) - .unwrap(); diag.store() .append("PROVISIONING_REPORT", "result=success") .unwrap(); @@ -264,18 +262,21 @@ fn clear_removes_all_diagnostics_regardless_of_scope() { let dir = TempDir::new().unwrap(); let diag = diagnostics(&dir); - diag.emit(&DiagnosticEvent::new(Level::INFO, "a:b", "mine")) - .unwrap(); + diag.emit(Level::INFO, "a:b", "mine").unwrap(); // Events from a different agent/VM and a malformed event key are also // diagnostic keys, so clear() removes them too. diag.store() - .append("other-agent|other-vm|INFO|x:y|id", "theirs") + .append("other-agent|100|INFO|x:y|other-vm|id", "theirs") .unwrap(); - diag.store().append("p|vm|NOPE|c:d|id", "junk").unwrap(); + diag.store().append("p|100|NOPE|c:d|vm|id", "junk").unwrap(); // A chunked malformed event key (its base classifies as malformed) is // also a diagnostic key, so clear() removes every chunk. - diag.store().append("p|vm|NOPE|c:d|id|0", "junk-0").unwrap(); - diag.store().append("p|vm|NOPE|c:d|id|1", "junk-1").unwrap(); + diag.store() + .append("p|100|NOPE|c:d|vm|id|0", "junk-0") + .unwrap(); + diag.store() + .append("p|100|NOPE|c:d|vm|id|1", "junk-1") + .unwrap(); // A raw record survives. diag.store() .append("PROVISIONING_REPORT", "result=success") @@ -297,9 +298,8 @@ fn emit_rejects_delimiter_in_event_fields() { let dir = TempDir::new().unwrap(); let diag = diagnostics(&dir); - // A pipe in the name would produce an ambiguous six-segment key. - let event = DiagnosticEvent::new(Level::INFO, "a|b", "msg"); - assert!(diag.emit(&event).is_err()); + // A pipe in the name would produce an ambiguous seven-segment key. + assert!(diag.emit(Level::INFO, "a|b", "msg").is_err()); // Nothing was written. assert!(diag.store().dump().unwrap().is_empty()); } @@ -321,12 +321,8 @@ fn concurrent_multichunk_emits_reassemble_without_interleaving() { thread::spawn(move || { for _ in 0..PER_THREAD { let message = marker.to_string().repeat(len); - let event = DiagnosticEvent::new( - Level::INFO, - format!("thread:{marker}"), - message, - ); - diag.emit(&event).unwrap(); + diag.emit(Level::INFO, format!("thread:{marker}"), message) + .unwrap(); } }) }) From 85bd9ee74447d650ad2b3536a83dcf873766500c Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Wed, 19 Aug 2026 14:05:52 -0600 Subject: [PATCH 08/32] refactor(kvp): unify azure-init and cloud-init diagnostics into one DiagnosticEvent Collapse the parallel azure-init/cloud-init types into a single, source- agnostic DiagnosticEvent decoded via a two-phase parse (key -> metadata, value -> payload). Removes CloudInitEvent/CloudInitValue and the CloudInit record variant. --- libazureinit-kvp/src/cli.rs | 168 ++--- libazureinit-kvp/src/diagnostics.rs | 929 ++++++++++++++------------ libazureinit-kvp/src/error.rs | 6 +- libazureinit-kvp/src/lib.rs | 2 +- libazureinit-kvp/src/store.rs | 44 +- libazureinit-kvp/src/vm_id.rs | 7 +- libazureinit-kvp/tests/cli.rs | 113 +--- libazureinit-kvp/tests/diagnostics.rs | 92 +-- 8 files changed, 628 insertions(+), 733 deletions(-) diff --git a/libazureinit-kvp/src/cli.rs b/libazureinit-kvp/src/cli.rs index 97451195..42a45fae 100644 --- a/libazureinit-kvp/src/cli.rs +++ b/libazureinit-kvp/src/cli.rs @@ -99,19 +99,14 @@ enum Command { #[arg(long)] parse_diagnostics: bool, /// Also print raw (non-event) records such as PROVISIONING_REPORT. - /// Only applies to the unfiltered view; --level/--name/--tail - /// produce an azure-init events-only view where raw records - /// never appear. + /// Only applies to the unfiltered view; --name/--tail produce an + /// azure-init events-only view where raw records never appear. #[arg( long, requires = "parse_diagnostics", - conflicts_with_all = ["level", "name", "tail"] + conflicts_with_all = ["name", "tail"] )] include_raw: bool, - /// Only show azure-init events at this level (error, warn, info, - /// debug, trace). - #[arg(long, requires = "parse_diagnostics")] - level: Option, /// Only show azure-init events whose name contains this /// substring. #[arg(long, requires = "parse_diagnostics")] @@ -140,12 +135,9 @@ enum Command { value: String, }, /// Emit an azure-init diagnostic event: a structured KVP entry keyed - /// `|||||`. + /// `||||||`. /// Distinct from the raw `write` command. Emit { - /// Event severity: error, warn, info, debug, or trace. - #[arg(long)] - level: String, /// Event name, e.g. user:create_user. #[arg(long)] name: String, @@ -287,13 +279,11 @@ fn dispatch(cli: Cli, stdout: &mut W) -> Result { Command::Dump { parse_diagnostics, include_raw, - level, name, tail, } => { let parse = parse_diagnostics.then_some(ParseDiagnosticsArgs { include_raw, - level, name, tail, }); @@ -310,12 +300,11 @@ fn dispatch(cli: Cli, stdout: &mut W) -> Result { Ok(EXIT_OK) } Command::Emit { - level, name, message, vm_id, prefix, - } => emit(&store, level, name, message, vm_id, prefix), + } => emit(&store, name, message, vm_id, prefix), Command::Load { file } => load(&store, file), Command::AppendMultiple { file } => append_multiple(&store, file), Command::Delete { key } => delete(&store, stdout, &key, output), @@ -399,11 +388,8 @@ fn info( } Ok(EXIT_OK) } - -/// Options for the `dump --parse-diagnostics` view; absent for a raw dump. struct ParseDiagnosticsArgs { include_raw: bool, - level: Option, name: Option, tail: Option, } @@ -415,18 +401,9 @@ fn dump( output: OutputMode, ) -> Result { if let Some(parse) = parse { - // --level/--name/--tail select the decoded events-only view; - // otherwise every record is shown (raw hidden unless - // --include-raw). - if parse.level.is_some() || parse.name.is_some() || parse.tail.is_some() - { + if parse.name.is_some() || parse.tail.is_some() { return diagnostics_events( - store, - stdout, - parse.level, - parse.name, - parse.tail, - output, + store, stdout, parse.name, parse.tail, output, ); } return diagnostics_records(store, stdout, parse.include_raw, output); @@ -559,38 +536,12 @@ fn diagnostics_records( OutputMode::Text => { for record in &records { let line = match record { - DiagnosticRecord::Event { event, chunks } => format!( - "event boot_epoch_time={} event_level={} name={} \ - vm_id={} event_id={} chunks={} message={}", - event.boot_epoch_time, - event.event_level, - event.name, - event.vm_id, - event.event_id, - chunks, - event.message - ), - DiagnosticRecord::CloudInit { event, chunks } => { - let mut line = format!( - "cloud-init-event type={} name={} uuid={}", - event.event_type, event.name, event.uuid - ); - if let Some(vm_id) = &event.vm_id { - let _ = write!(line, " vm_id={vm_id}"); - } - if let Some(result) = &event.result { - let _ = write!(line, " result={result}"); - } - if let Some(ts) = &event.timestamp { - let _ = write!(line, " ts={ts}"); - } - if let Some(duration) = event.duration { - let _ = write!(line, " duration={duration}"); - } + DiagnosticRecord::Decoded { event, chunks } => { + let mut line = diagnostics_event_text(event); let _ = write!( line, - " incarnation={} chunks={} message={}", - event.incarnation, chunks, event.message + " chunks={chunks} message={}", + event.message ); line } @@ -619,19 +570,13 @@ fn diagnostics_records( fn diagnostics_events( store: &KvpPoolStore, stdout: &mut W, - level: Option, name: Option, tail: Option, output: OutputMode, ) -> Result { - let level = level.as_deref().map(parse_level_filter).transpose()?; - let diagnostics = DiagnosticsKvp::new(store.clone(), "", ""); let mut events = diagnostics.events()?; - if let Some(level) = level { - events.retain(|event| event.event_level == level); - } if let Some(needle) = name.as_deref() { events.retain(|event| event.name.contains(needle)); } @@ -643,16 +588,8 @@ fn diagnostics_events( match output { OutputMode::Text => { for event in &events { - let line = format!( - "event boot_epoch_time={} event_level={} name={} \ - vm_id={} event_id={} message={}", - event.boot_epoch_time, - event.event_level, - event.name, - event.vm_id, - event.event_id, - event.message - ); + let mut line = diagnostics_event_text(event); + let _ = write!(line, " message={}", event.message); writeln!(stdout, "{line}")?; } } @@ -665,56 +602,47 @@ fn diagnostics_events( Ok(EXIT_OK) } -/// Parse a `--level` filter argument into a [`tracing::Level`]. -fn parse_level_filter(level: &str) -> Result { - level.parse::().map_err(|_| { - CliError::Usage(format!( - "invalid level '{level}' (expected error, warn, info, debug, \ - or trace)" - )) - }) +/// Render a [`DiagnosticEvent`] as a single text line of `key=value` +/// fields, omitting optional fields the source did not provide. +fn diagnostics_event_text(event: &DiagnosticEvent) -> String { + let mut line = format!( + "event kind={} agent={} boot_epoch={}", + event.kind, event.agent, event.boot_epoch + ); + if let Some(vm_id) = &event.vm_id { + let _ = write!(line, " vm_id={vm_id}"); + } + let _ = write!(line, " name={} event_id={}", event.name, event.event_id); + if let Some(ts) = &event.timestamp { + let _ = write!(line, " timestamp={ts}"); + } + if let Some(result) = &event.result { + let _ = write!(line, " result={result}"); + } + if let Some(duration) = event.duration { + let _ = write!(line, " duration={duration}"); + } + line } /// Render a [`DiagnosticRecord`] as a JSON object. fn diagnostics_record_json(record: &DiagnosticRecord) -> serde_json::Value { match record { - DiagnosticRecord::Event { event, chunks } => { + DiagnosticRecord::Decoded { event, chunks } => { let mut value = diagnostics_event_json(event); if let serde_json::Value::Object(map) = &mut value { + map.insert("record".to_string(), json!("event")); map.insert("chunks".to_string(), json!(chunks)); } value } - DiagnosticRecord::CloudInit { event, chunks } => { - let mut map = serde_json::Map::new(); - map.insert("kind".to_string(), json!("cloud-init-event")); - map.insert("incarnation".to_string(), json!(event.incarnation)); - map.insert("type".to_string(), json!(event.event_type)); - map.insert("name".to_string(), json!(event.name)); - if let Some(vm_id) = &event.vm_id { - map.insert("vm_id".to_string(), json!(vm_id)); - } - map.insert("uuid".to_string(), json!(event.uuid)); - if let Some(ts) = &event.timestamp { - map.insert("ts".to_string(), json!(ts)); - } - if let Some(result) = &event.result { - map.insert("result".to_string(), json!(result)); - } - if let Some(duration) = event.duration { - map.insert("duration".to_string(), json!(duration)); - } - map.insert("chunks".to_string(), json!(chunks)); - map.insert("message".to_string(), json!(event.message)); - serde_json::Value::Object(map) - } DiagnosticRecord::Raw { key, value } => json!({ - "kind": "raw", + "record": "raw", "key": key, "value": value, }), DiagnosticRecord::Malformed { key, value, reason } => json!({ - "kind": "malformed", + "record": "malformed", "key": key, "value": value, "reason": reason, @@ -722,33 +650,25 @@ fn diagnostics_record_json(record: &DiagnosticRecord) -> serde_json::Value { } } -/// Render a [`DiagnosticEvent`] as a JSON object (without chunk count). +/// Render a [`DiagnosticEvent`] as a JSON object (without chunk count), +/// omitting optional fields the source did not provide. fn diagnostics_event_json(event: &DiagnosticEvent) -> serde_json::Value { - json!({ - "kind": "event", - "boot_epoch_time": event.boot_epoch_time, - "event_level": event.event_level.to_string(), - "name": event.name, - "vm_id": event.vm_id, - "event_id": event.event_id, - "message": event.message, - }) + serde_json::to_value(event) + .expect("DiagnosticEvent always serializes to a JSON object") } /// Emit an azure-init diagnostic event with the given fields. fn emit( store: &KvpPoolStore, - level: String, name: String, message: String, vm_id: Option, prefix: Option, ) -> Result { - let level = parse_level_filter(&level)?; let vm_id = resolve_vm_id(vm_id)?; let prefix = prefix.unwrap_or_else(|| DEFAULT_AGENT.to_string()); let diagnostics = DiagnosticsKvp::new(store.clone(), vm_id, prefix); - diagnostics.emit(level, name, message)?; + diagnostics.emit_event(name, message)?; Ok(EXIT_OK) } @@ -1103,7 +1023,6 @@ mod tests { Command::Dump { parse_diagnostics: false, include_raw: false, - level: None, name: None, tail: None, } @@ -1673,7 +1592,6 @@ mod tests { assert!(dumped.contains("b=two")); let (_, entries) = run_dispatch(cli(&dir, Command::Entries)); - // entries are sorted by key assert_eq!(entries, "a=one\nb=two\n"); } diff --git a/libazureinit-kvp/src/diagnostics.rs b/libazureinit-kvp/src/diagnostics.rs index 72c54ffb..eb5e0c99 100644 --- a/libazureinit-kvp/src/diagnostics.rs +++ b/libazureinit-kvp/src/diagnostics.rs @@ -5,31 +5,31 @@ //! [`KvpPoolStore`](crate::KvpPoolStore) key/value API. //! //! Where [`KvpPoolStore`](crate::KvpPoolStore) treats keys and values as -//! opaque bytes, [`DiagnosticsKvp`] understands the telemetry -//! conventions azure-init writes into the guest pool: +//! opaque bytes, [`DiagnosticsKvp`] understands the telemetry conventions +//! azure-init writes into the guest pool and decodes cloud-init's +//! reporting entries into the same [`DiagnosticEvent`] shape. //! -//! - **Event keys** encode structured metadata as a six-segment, +//! - **azure-init keys** encode metadata as a seven-segment, //! pipe-delimited string -//! (`|||||`). -//! - **Chunking**: a single KVP record caps the value at the store's -//! per-record limit (see [`MAX_CHUNK_BYTES`] for the safe-mode value). -//! Longer messages are split at UTF-8 codepoint boundaries into -//! multiple records written atomically under a single lock. Each chunk -//! gets a unique key — the event key with a `|` suffix -//! (`||||||`), -//! matching cloud-init's naming — so the Hyper-V host, which keeps only -//! one record per key, retains every chunk. The chunks are regrouped -//! into one event on read. +//! (`||||||`). +//! The value is the record's message string, stored verbatim for +//! every kind. +//! - **cloud-init keys** +//! (`CLOUD_INIT||||[|]`) store a +//! JSON value; the reader pulls `ts`/`result`/`duration`/`msg` from it +//! and takes everything else from the key, decoding into the same +//! [`DiagnosticEvent`]. This crate only *reads* cloud-init. +//! - **Chunking**: values longer than [`MAX_CHUNK_BYTES`] are split at +//! UTF-8 codepoint boundaries into multiple records written atomically +//! under a single lock. Each chunk gets a unique key — the event key +//! with a `|` suffix — so the Hyper-V host, which keeps +//! only one record per key, retains every chunk. The chunks are +//! regrouped into one event on read. //! - **Classification**: [`records`](DiagnosticsKvp::records) sorts every //! stored record into a [`DiagnosticRecord`] — a reassembled -//! [`DiagnosticEvent`], an unstructured [`Raw`](DiagnosticRecord::Raw) -//! record such as `PROVISIONING_REPORT`, or a -//! [`Malformed`](DiagnosticRecord::Malformed) event key. -//! - **cloud-init**: the reader also decodes cloud-init reporting entries -//! (keys prefixed `CLOUD_INIT` with a JSON value) into -//! [`CloudInitEvent`]s so the diagnostics CLI can display telemetry -//! from either provisioning agent. This crate only *reads* that -//! format; it does not write it. +//! [`DiagnosticEvent`] (from either agent), an unstructured +//! [`Raw`](DiagnosticRecord::Raw) record such as `PROVISIONING_REPORT`, +//! or a [`Malformed`](DiagnosticRecord::Malformed) event key. //! //! This module is policy only: all locking, size enforcement, and //! on-disk encoding stay in [`KvpPoolStore`](crate::KvpPoolStore). @@ -40,7 +40,6 @@ //! use libazureinit_kvp::{ //! DiagnosticsKvp, KvpPool, KvpPoolStore, PoolMode, MAX_CHUNK_BYTES, //! }; -//! use tracing::Level; //! //! # fn main() -> Result<(), libazureinit_kvp::KvpError> { //! let dir = std::env::temp_dir() @@ -53,16 +52,11 @@ //! DiagnosticsKvp::new(store, "vm-1234", "azure-init-doc"); //! //! // A short event lands in a single record. -//! diagnostics.emit( -//! Level::INFO, -//! "user:create_user", -//! "Creating user azureuser", -//! )?; +//! diagnostics.emit_event("user:create_user", "Creating user azureuser")?; //! -//! // A long message is split across records each with a unique -//! // `|`-suffixed key, and reassembled on read. +//! // A long message is split across records and reassembled on read. //! let long = "x".repeat(MAX_CHUNK_BYTES * 2 + 10); -//! diagnostics.emit(Level::DEBUG, "config:dump", &long)?; +//! diagnostics.emit_event("config:dump", &long)?; //! //! let events = diagnostics.events()?; //! assert_eq!(events.len(), 2); @@ -73,8 +67,7 @@ //! # } //! ``` -use serde::Deserialize; -use tracing::Level; +use chrono::Utc; use uuid::Uuid; use crate::{KvpError, KvpPoolStore}; @@ -82,71 +75,116 @@ use crate::{KvpError, KvpPoolStore}; /// Literal prefix identifying a cloud-init reporting KVP key. const CLOUD_INIT_PREFIX: &str = "CLOUD_INIT"; -/// Maximum number of value bytes per KVP record under -/// [`PoolMode::Safe`](crate::PoolMode::Safe). +/// Maximum number of value bytes per diagnostic KVP record. /// -/// This matches a safe-mode store's -/// [`KvpPoolStore::max_value_size`](crate::KvpPoolStore::max_value_size). -/// [`DiagnosticsKvp::emit`] splits messages longer than the store's -/// actual limit, so an [`Unsafe`](crate::PoolMode::Unsafe) store uses -/// its larger capacity; this constant is the conservative reference -/// value used throughout the diagnostics conventions. +/// [`DiagnosticsKvp::emit_event`] splits messages longer than this into +/// multiple records, regardless of the store's +/// [`PoolMode`](crate::PoolMode). It is the conservative +/// [`Safe`](crate::PoolMode::Safe) limit (2 bytes under the Linux kernel +/// `HV_KVP_EXCHANGE_MAX_VALUE` maximum), so diagnostic records stay +/// readable by the Hyper-V host even on an +/// [`Unsafe`](crate::PoolMode::Unsafe) store — its larger capacity is +/// deliberately not used for diagnostics. pub const MAX_CHUNK_BYTES: usize = 1022; /// Delimiter separating the segments of a diagnostic event key. const EVENT_KEY_DELIMITER: char = '|'; -/// Format a diagnostic event key as its `|`-delimited on-disk string: -/// `|||||`. +/// The kind of a diagnostic record: a span boundary (`start`/`finish`) +/// or a point `event`. The `start`/`finish` tokens match cloud-init's, +/// so a span's boundaries read the same whichever agent emitted them. +#[derive(Clone, Copy, Debug, PartialEq, Eq, serde::Serialize)] +#[serde(rename_all = "lowercase")] +pub enum RecordKind { + /// The opening of a span (a function or stage begins), written + /// `start`. + Start, + /// The closing of a span (a function or stage ends), written + /// `finish`. + Finish, + /// A point-in-time event, written `event`. + Event, +} + +impl std::fmt::Display for RecordKind { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str(match self { + Self::Start => "start", + Self::Finish => "finish", + Self::Event => "event", + }) + } +} + +/// Parses the on-disk `kind`/`type` token; cloud-init emits only +/// `start`/`finish`. +impl std::str::FromStr for RecordKind { + type Err = (); + + fn from_str(token: &str) -> Result { + match token { + "start" => Ok(Self::Start), + "finish" => Ok(Self::Finish), + "event" => Ok(Self::Event), + _ => Err(()), + } + } +} + +/// The current time as an ISO-8601 UTC timestamp (millisecond precision). +fn now_timestamp() -> String { + Utc::now().format("%Y-%m-%dT%H:%M:%S%.3fZ").to_string() +} + +/// Format an azure-init diagnostic event key as its `|`-delimited on-disk +/// string: +/// `||||||`. /// -/// `boot_epoch_time` is the Unix epoch second the system booted (see +/// `boot_epoch` is the Unix epoch second the system booted (see /// [`KvpPoolStore::boot_epoch`](crate::KvpPoolStore::boot_epoch)); it sits -/// in the same slot as cloud-init's incarnation so the two agents' keys -/// line up. The `event_level`/`name`/`vm_id` order likewise mirrors -/// cloud-init's `type`/`name`/`vm_id` layout. [`classify_key`] is the -/// inverse. For example: +/// in the same slot as cloud-init's incarnation. `kind` records the +/// span/event shape. [`classify_key`] is the inverse. For example: /// /// ```text -/// azure-init-0.1.0|1785187982|INFO|user:create_user|3f2504e0-...|8f3e9c4a-... +/// azure-init-0.1.0|1785187982|3f2504e0-...|event|user:create_user|8f3e9c4a-...|2026-07-27T21:33:24.300Z /// ``` fn format_event_key( - prefix: &str, - boot_epoch_time: i64, - event_level: Level, - name: &str, + agent: &str, + boot_epoch: i64, vm_id: &str, + kind: RecordKind, + name: &str, event_id: &str, + timestamp: &str, ) -> String { let d = EVENT_KEY_DELIMITER; format!( - "{prefix}{d}{boot_epoch_time}{d}{event_level}{d}{name}{d}{vm_id}\ - {d}{event_id}" + "{agent}{d}{boot_epoch}{d}{vm_id}{d}{kind}{d}{name}{d}{event_id}\ + {d}{timestamp}" ) } - -/// Outcome of inspecting a raw pool key. enum KeyClass<'a> { - /// The key is a well-formed azure-init event key. Event { - boot_epoch_time: i64, - event_level: Level, - name: &'a str, + agent: &'a str, + boot_epoch: i64, vm_id: &'a str, + kind: RecordKind, + name: &'a str, event_id: &'a str, + timestamp: &'a str, }, /// The key is a well-formed cloud-init reporting event key /// (`CLOUD_INIT||||[|]`). CloudInit { - incarnation: &'a str, - event_type: &'a str, + boot_epoch: i64, + kind: RecordKind, name: &'a str, vm_id: Option<&'a str>, uuid: &'a str, }, - /// The key has the azure-init six-segment shape but is not a valid - /// event (for example, an unrecognized level). - Malformed { reason: String }, - /// The key is not an event key (e.g. `PROVISIONING_REPORT`). + Malformed { + reason: String, + }, Raw, } @@ -156,49 +194,47 @@ fn classify_key(key: &str) -> KeyClass<'_> { return classify_cloud_init_key(key); } - // azure-init: - // `|||||` let mut segments = key.split(EVENT_KEY_DELIMITER); - - // `str::split` always yields at least one element. - let _prefix = segments.next(); let ( + Some(agent), Some(boot_epoch), - Some(event_level), - Some(name), Some(vm_id), + Some(kind), + Some(name), Some(event_id), + Some(timestamp), ) = ( segments.next(), segments.next(), segments.next(), segments.next(), segments.next(), + segments.next(), + segments.next(), ) else { return KeyClass::Raw; }; if segments.next().is_some() { - // More than six segments: a `|` leaked into a field. return KeyClass::Raw; } - // A non-numeric boot epoch means this is not an azure-init event key; - // keep it as an opaque record rather than a malformed event. - let Ok(boot_epoch_time) = boot_epoch.parse::() else { + let Ok(boot_epoch) = boot_epoch.parse::() else { return KeyClass::Raw; }; - match event_level.parse::() { - Ok(event_level) => KeyClass::Event { - boot_epoch_time, - event_level, - name, + match kind.parse::() { + Ok(kind) => KeyClass::Event { + agent, + boot_epoch, vm_id, + kind, + name, event_id, + timestamp, }, - Err(_) => KeyClass::Malformed { - reason: format!("unrecognized level {event_level:?}"), + Err(()) => KeyClass::Malformed { + reason: format!("unrecognized kind {kind:?}"), }, } } @@ -209,11 +245,10 @@ fn classify_key(key: &str) -> KeyClass<'_> { /// (`CLOUD_INIT|||||`) and the /// older one that predates the `vm_id` segment /// (`CLOUD_INIT||||`). Any other segment -/// count is treated as [`KeyClass::Raw`]. Parses by pulling segments -/// from the iterator so no intermediate collection is allocated. +/// count is [`KeyClass::Raw`]; a right-shaped key with a non-numeric +/// incarnation or unrecognized type is [`KeyClass::Malformed`]. fn classify_cloud_init_key(key: &str) -> KeyClass<'_> { let mut segments = key.split(EVENT_KEY_DELIMITER); - // The caller matched the `CLOUD_INIT` prefix; skip it. let _prefix = segments.next(); let (Some(incarnation), Some(event_type), Some(name), Some(fourth)) = ( segments.next(), @@ -228,9 +263,21 @@ fn classify_cloud_init_key(key: &str) -> KeyClass<'_> { (Some(uuid), None) => (Some(fourth), uuid), _ => return KeyClass::Raw, }; + let Ok(boot_epoch) = incarnation.parse::() else { + return KeyClass::Malformed { + reason: format!( + "non-numeric cloud-init incarnation {incarnation:?}" + ), + }; + }; + let Ok(kind) = event_type.parse::() else { + return KeyClass::Malformed { + reason: format!("unrecognized cloud-init type {event_type:?}"), + }; + }; KeyClass::CloudInit { - incarnation, - event_type, + boot_epoch, + kind, name, vm_id, uuid, @@ -258,13 +305,11 @@ fn chunk_at_char_boundary(value: &str, max_bytes: usize) -> Vec<&str> { break; } - // Walk back from the byte limit to the nearest codepoint boundary. let mut end = start + max_bytes; while end > start && !value.is_char_boundary(end) { end -= 1; } if end == start { - // One codepoint spans the whole window; take it whole. end = start + max_bytes + 1; while end < value.len() && !value.is_char_boundary(end) { end += 1; @@ -286,82 +331,52 @@ fn reject_delimiter(field: &'static str, value: &str) -> Result<(), KvpError> { Ok(()) } -/// A single azure-init diagnostic event — the decoded form of one -/// azure-init KVP entry. +/// A single diagnostic event — the decoded, source-agnostic form of one +/// azure-init or cloud-init KVP entry. /// -/// This is the crate's definition of what an azure-init KVP event looks -/// like: the boot-scoped `boot_epoch_time`/`vm_id`, the event-scoped -/// `event_level`/`name`/`event_id`, and the free-form `message`. Write one -/// with [`DiagnosticsKvp::emit`] (which stamps the boot-scoped fields from -/// the session and generates a fresh `event_id`); read events back, fully -/// populated, via [`records`](DiagnosticsKvp::records) / +/// Metadata (`agent`, `boot_epoch`, `vm_id`, `kind`, `name`, `event_id`) +/// comes from the record key; the payload (`timestamp`, `result`, +/// `duration`, `message`) comes from the value. Optional fields are +/// populated only when the source provides them: azure-init events carry +/// a `timestamp`; cloud-init `finish` records carry a `result` and +/// `duration`. Write azure-init events with +/// [`DiagnosticsKvp::emit_event`]; read events back via +/// [`records`](DiagnosticsKvp::records) / /// [`events`](DiagnosticsKvp::events). -#[derive(Clone, Debug, PartialEq, Eq)] +#[derive(Clone, Debug, PartialEq, serde::Serialize)] #[non_exhaustive] pub struct DiagnosticEvent { - /// Unix epoch second the system booted, shared by every event of one - /// boot. Distinguishes this boot's telemetry from a previous boot's. - pub boot_epoch_time: i64, - /// Severity of the event. - pub event_level: Level, - /// Formatted event name, e.g. `user:create_user`. - pub name: String, - /// VM identifier the event was emitted under. - pub vm_id: String, - /// Per-emit identifier (UUIDv4); every chunk of one emitted event - /// shares this value. - pub event_id: String, - /// Literal value bytes written to the pool. The diagnostics layer - /// imposes no format on this string. - pub message: String, -} - -/// The JSON payload cloud-init stores as a reporting event's KVP value. -/// -/// Only the fields the diagnostics reader surfaces are deserialized; -/// cloud-init also duplicates `name`/`type` here, but those are read -/// from the key. Unknown fields are ignored. -#[derive(Debug, Deserialize)] -struct CloudInitValue { - /// Human-readable message; defaults to empty when absent. - #[serde(default)] - msg: String, - /// ISO-8601 timestamp. - ts: Option, - /// Result string (e.g. `SUCCESS`), present on `finish` events. - result: Option, - /// Duration in seconds, present on `finish` events. - duration: Option, -} - -/// A single cloud-init reporting event decoded from a KVP entry. -/// -/// cloud-init encodes routing metadata in the key -/// (`CLOUD_INIT||||[|]`) and the -/// event details as a JSON value; this type is the decoded union of the -/// two. -#[derive(Clone, Debug, PartialEq)] -#[non_exhaustive] -pub struct CloudInitEvent { - /// Boot-time incarnation stamp from the key. cloud-init uses it to - /// distinguish this boot's records from a previous boot's. - pub incarnation: String, - /// Provisioning phase from the key, e.g. `start` or `finish`. - pub event_type: String, - /// Event name from the key, e.g. `modules-final/config-scripts_user`. - pub name: String, + /// Reporting agent identifier from the key, e.g. `azure-init-0.1.0` + /// or `CLOUD_INIT`; also distinguishes the record's source. + pub agent: String, + /// Unix epoch second the system booted (cloud-init's incarnation), + /// shared by every record of one boot. + pub boot_epoch: i64, /// VM identifier from the key. Absent in cloud-init builds that /// predate the `vm_id` key segment. + #[serde(skip_serializing_if = "Option::is_none")] pub vm_id: Option, - /// Per-event UUID from the key. - pub uuid: String, - /// ISO-8601 timestamp from the JSON value, if present. + /// Whether this record opens a span, closes a span, or is a point + /// event. + pub kind: RecordKind, + /// Formatted event or span name, e.g. `user:create_user`. + pub name: String, + /// Per-record identifier (azure-init's UUIDv4 / cloud-init's uuid); + /// every chunk of one record shares it, as do a span's start and + /// finish. + pub event_id: String, + /// ISO-8601 timestamp, if the source provides one. + #[serde(skip_serializing_if = "Option::is_none")] pub timestamp: Option, - /// Result from the JSON value (e.g. `SUCCESS`), if present. + /// Result string (e.g. `SUCCESS`), present on cloud-init `finish` + /// records. + #[serde(skip_serializing_if = "Option::is_none")] pub result: Option, - /// Duration in seconds from the JSON value, if present. + /// Duration in seconds, present on cloud-init `finish` records. + #[serde(skip_serializing_if = "Option::is_none")] pub duration: Option, - /// Human-readable message from the JSON value. + /// Human-readable message. The diagnostics layer imposes no format + /// on this string. pub message: String, } @@ -370,20 +385,13 @@ pub struct CloudInitEvent { #[derive(Clone, Debug, PartialEq)] #[non_exhaustive] pub enum DiagnosticRecord { - /// A reassembled azure-init diagnostic event. - Event { + /// A reassembled diagnostic event, from either agent. + Decoded { /// The decoded event. event: DiagnosticEvent, /// Number of on-disk records the value spanned (1 when short). chunks: usize, }, - /// A decoded cloud-init reporting event. - CloudInit { - /// The decoded cloud-init event. - event: CloudInitEvent, - /// Number of on-disk records the value spanned (1 when short). - chunks: usize, - }, /// An unstructured record whose key is not an event key, such as /// `PROVISIONING_REPORT`. Raw { @@ -393,7 +401,7 @@ pub enum DiagnosticRecord { value: String, }, /// A record whose key is event-shaped but is not a valid event (for - /// example, an unrecognized level or invalid cloud-init JSON). + /// example, an unrecognized kind or invalid cloud-init JSON). Malformed { /// The record key. key: String, @@ -406,90 +414,98 @@ pub enum DiagnosticRecord { /// A typed diagnostics view over a [`KvpPoolStore`]. /// -/// Owns the event-key `prefix` and `vm_id` stamped into this layer's -/// events by [`emit`](Self::emit). See the -/// [module documentation](self) for the on-disk format. +/// Owns the `agent` and `vm_id` stamped into this layer's azure-init +/// event keys. See the [module documentation](self) for the on-disk +/// format. #[derive(Clone, Debug)] pub struct DiagnosticsKvp { store: KvpPoolStore, vm_id: String, - event_prefix: String, + agent: String, } impl DiagnosticsKvp { - /// Wrap `store` with the `vm_id` and `event_prefix` stamped into - /// this layer's event keys. pub fn new( store: KvpPoolStore, vm_id: impl Into, - event_prefix: impl Into, + agent: impl Into, ) -> Self { Self { store, vm_id: vm_id.into(), - event_prefix: event_prefix.into(), + agent: agent.into(), } } - - /// The underlying store. pub fn store(&self) -> &KvpPoolStore { &self.store } - - /// The VM identifier stamped into event keys. pub fn vm_id(&self) -> &str { &self.vm_id } + pub fn agent(&self) -> &str { + &self.agent + } - /// The prefix stamped into event keys. - pub fn event_prefix(&self) -> &str { - &self.event_prefix + /// Emit an azure-init point event with `name` and `message`. + /// + /// The `message` is stored verbatim as the record value. A fresh + /// `event_id` (UUIDv4) is generated. + pub fn emit_event( + &self, + name: impl Into, + message: impl AsRef, + ) -> Result<(), KvpError> { + let event_id = Uuid::new_v4().to_string(); + self.write_event( + RecordKind::Event, + &event_id, + &name.into(), + message.as_ref(), + ) } - /// Emit an azure-init diagnostic event: format the key - /// `|||||` - /// and write `message` as its value. `boot_epoch_time` (from - /// [`KvpPoolStore::boot_epoch`](crate::KvpPoolStore::boot_epoch)), - /// `vm_id`, and `event_prefix` come from this layer; the `event_id` is - /// a fresh UUIDv4. + /// Format the key + /// `||||||` + /// (stamping `boot_epoch`, `vm_id`, `agent`, and the current + /// `timestamp`) and write `value` as its message. /// - /// Messages longer than the store's per-record value limit are split + /// Messages longer than [`MAX_CHUNK_BYTES`] are split /// at UTF-8 codepoint boundaries and written as multiple records /// atomically under a single lock via - /// [`KvpPoolStore::append_multiple`]. Each chunk is keyed by the event - /// key with a `|` suffix so every record is unique — - /// the Hyper-V host keeps only one record per key — and the chunks are - /// regrouped by [`records`](Self::records) on read. + /// [`KvpPoolStore::append_multiple`]. Each chunk is keyed with a + /// `|` suffix so every record is unique, and the + /// chunks are regrouped by [`records`](Self::records) on read. /// - /// Returns [`KvpError::EventFieldContainsDelimiter`] if the - /// `event_prefix`, `vm_id`, or `name` contains the `|` key delimiter, - /// which would make the key ambiguous to [`records`](Self::records). - pub fn emit( + /// Returns [`KvpError::EventFieldContainsDelimiter`] if `agent`, + /// `vm_id`, `name`, or `event_id` contains the `|` key delimiter. + fn write_event( &self, - event_level: Level, - name: impl Into, - message: impl Into, + kind: RecordKind, + event_id: &str, + name: &str, + value: &str, ) -> Result<(), KvpError> { - let name = name.into(); - reject_delimiter("event_prefix", &self.event_prefix)?; + reject_delimiter("agent", &self.agent)?; reject_delimiter("vm_id", &self.vm_id)?; - reject_delimiter("name", &name)?; + reject_delimiter("name", name)?; + reject_delimiter("event_id", event_id)?; - let boot_epoch_time = self.store.boot_epoch()?; - let event_id = Uuid::new_v4().to_string(); + let boot_epoch = self.store.boot_epoch()?; + let timestamp = now_timestamp(); let key = format_event_key( - &self.event_prefix, - boot_epoch_time, - event_level, - &name, + &self.agent, + boot_epoch, &self.vm_id, - &event_id, + kind, + name, + event_id, + ×tamp, ); - self.write_chunked(&key, &message.into()) + self.write_chunked(&key, value) } - /// Split `value` at the store's per-record limit and append the + /// Split `value` at [`MAX_CHUNK_BYTES`] and append the /// chunks under `key` in one atomic batch. /// /// A single-record value keeps the bare event `key`. A value that @@ -498,7 +514,7 @@ impl DiagnosticsKvp { /// the Hyper-V host keeps only one record per key. [`reassemble`] /// strips the subevent index to regroup the chunks on read. fn write_chunked(&self, key: &str, value: &str) -> Result<(), KvpError> { - let chunks = chunk_at_char_boundary(value, self.store.max_value_size()); + let chunks = chunk_at_char_boundary(value, MAX_CHUNK_BYTES); if chunks.len() == 1 { return self .store @@ -521,28 +537,25 @@ impl DiagnosticsKvp { /// /// Records are returned in on-disk order. Consecutive records that /// share an event key — ignoring the `|` suffix — are - /// one event; because [`emit`](Self::emit) writes an event's chunks - /// contiguously under a single lock, reassembly is correct even under - /// concurrent writers. + /// one event; because [`emit_event`](Self::emit_event) writes an + /// event's chunks contiguously under a single lock, reassembly is + /// correct even under concurrent writers. pub fn records(&self) -> Result, KvpError> { Ok(reassemble(self.store.dump()?)) } - /// Read back only the azure-init [`DiagnosticEvent`]s, in on-disk - /// order. + /// Read back every decoded [`DiagnosticEvent`], from either agent, in + /// on-disk order. /// - /// This deliberately excludes cloud-init events — which decode to the - /// separate [`CloudInitEvent`] type — as well as raw and malformed - /// records. Use [`records`](Self::records) for the full cross-agent - /// view that includes cloud-init telemetry. + /// Raw and malformed records are excluded. Use + /// [`records`](Self::records) for the full view that includes them. pub fn events(&self) -> Result, KvpError> { Ok(self .records()? .into_iter() .filter_map(|record| match record { - DiagnosticRecord::Event { event, .. } => Some(event), - DiagnosticRecord::CloudInit { .. } - | DiagnosticRecord::Raw { .. } + DiagnosticRecord::Decoded { event, .. } => Some(event), + DiagnosticRecord::Raw { .. } | DiagnosticRecord::Malformed { .. } => None, }) .collect()) @@ -587,7 +600,7 @@ fn base_event_key(key: &str) -> &str { /// Split a key into its base event key and optional trailing subevent /// index. When the trailing segment is numeric and the base parses as an /// event key — a valid azure-init or cloud-init event, or a malformed one -/// (event-shaped but with an unrecognized level) — returns +/// (event-shaped but invalid) — returns /// `(base, Some(index))`; any other key (a single-record event, /// `PROVISIONING_REPORT`, …) returns `(key, None)`. /// @@ -656,49 +669,63 @@ fn classify_record(key: String, chunk_values: Vec) -> DiagnosticRecord { let chunks = chunk_values.len(); match classify_key(&key) { KeyClass::Event { - boot_epoch_time, - event_level, - name, + agent, + boot_epoch, vm_id, + kind, + name, event_id, - } => DiagnosticRecord::Event { - event: DiagnosticEvent { - boot_epoch_time, - event_level, - name: name.to_string(), - vm_id: vm_id.to_string(), - event_id: event_id.to_string(), - message: chunk_values.concat(), - }, - chunks, - }, + timestamp, + } => { + let message = chunk_values.concat(); + DiagnosticRecord::Decoded { + event: DiagnosticEvent { + agent: agent.to_string(), + boot_epoch, + vm_id: Some(vm_id.to_string()), + kind, + name: name.to_string(), + event_id: event_id.to_string(), + timestamp: Some(timestamp.to_string()), + result: None, + duration: None, + message, + }, + chunks, + } + } KeyClass::CloudInit { - incarnation, - event_type, + boot_epoch, + kind, name, vm_id, uuid, } => { // Own the key-derived fields up front so `key` and - // `chunk_values` are free to move into a `Malformed` record - // when a chunk's value fails to decode. - let incarnation = incarnation.to_string(); - let event_type = event_type.to_string(); + // `chunk_values` can move into a `Malformed` record when a + // chunk's value fails to decode. let name = name.to_string(); let vm_id = vm_id.map(str::to_string); let uuid = uuid.to_string(); match decode_cloud_init_value(&chunk_values) { - Ok(decoded) => DiagnosticRecord::CloudInit { - event: CloudInitEvent { - incarnation, - event_type, - name, + Ok((meta, message)) => DiagnosticRecord::Decoded { + event: DiagnosticEvent { + agent: CLOUD_INIT_PREFIX.to_string(), + boot_epoch, vm_id, - uuid, - timestamp: decoded.timestamp, - result: decoded.result, - duration: decoded.duration, - message: decoded.message, + kind, + name, + event_id: uuid, + timestamp: meta + .get("ts") + .and_then(|t| t.as_str()) + .map(str::to_string), + result: meta + .get("result") + .and_then(|r| r.as_str()) + .map(str::to_string), + duration: meta.get("duration").and_then(|d| d.as_f64()), + message, }, chunks, }, @@ -721,21 +748,14 @@ fn classify_record(key: String, chunk_values: Vec) -> DiagnosticRecord { } } -/// The fields [`classify_record`] pulls from a cloud-init event's -/// value(s): the reassembled `message` plus the metadata a `finish` event -/// carries. -struct DecodedCloudInit { - timestamp: Option, - result: Option, - duration: Option, - message: String, -} - /// Marker preceding a cloud-init value's message field: `"msg":"`. const CLOUD_INIT_MSG_MARKER: &str = "\"msg\":\""; -/// Decode a cloud-init event's chunk value(s) into its metadata and full -/// message. +/// Decode a cloud-init event's chunk value(s) into its reassembled +/// `message` and the JSON object carrying its `ts`/`result`/`duration` +/// metadata, read from an untyped [`serde_json::Value`] so no serde +/// deserialization target is needed. [`classify_record`] maps those +/// fields onto the shared [`DiagnosticEvent`]. /// /// A single-record event is a complete JSON object, parsed directly. A /// multi-record event was split by cloud-init's `_break_down`, which @@ -747,42 +767,31 @@ const CLOUD_INIT_MSG_MARKER: &str = "\"msg\":\""; /// first chunk's non-`msg` fields. fn decode_cloud_init_value( chunks: &[String], -) -> Result { +) -> Result<(serde_json::Value, String), String> { if let [only] = chunks { - let value: CloudInitValue = + let value: serde_json::Value = serde_json::from_str(only).map_err(|e| e.to_string())?; - return Ok(DecodedCloudInit { - timestamp: value.ts, - result: value.result, - duration: value.duration, - message: value.msg, - }); + let message = value + .get("msg") + .and_then(|m| m.as_str()) + .unwrap_or_default() + .to_string(); + return Ok((value, message)); } - // Concatenate each chunk's raw escaped `msg` slice, then unescape the - // reassembled string once so escapes split across chunks are rejoined - // first. let mut escaped = String::new(); for chunk in chunks { - escaped.push_str(escaped_msg_slice(chunk)?); + escaped.push_str(cloud_init_escaped_msg_slice(chunk)?); } let message: String = serde_json::from_str(&format!("\"{escaped}\"")) .map_err(|e| e.to_string())?; - // Metadata is identical across chunks; take it from the first, whose - // non-`msg` prefix is always valid JSON. - let meta = chunk_metadata(&chunks[0])?; - Ok(DecodedCloudInit { - timestamp: meta.ts, - result: meta.result, - duration: meta.duration, - message, - }) + Ok((cloud_init_chunk_metadata(&chunks[0])?, message)) } /// Recover a chunk's raw (still-escaped) `msg` slice — the bytes between /// the `"msg":"` marker and the closing `"}` — without unescaping. -fn escaped_msg_slice(chunk: &str) -> Result<&str, String> { +fn cloud_init_escaped_msg_slice(chunk: &str) -> Result<&str, String> { let start = chunk .find(CLOUD_INIT_MSG_MARKER) .ok_or("chunk is missing a \"msg\" field")? @@ -796,10 +805,9 @@ fn escaped_msg_slice(chunk: &str) -> Result<&str, String> { .ok_or_else(|| "chunk \"msg\" field is malformed".to_string()) } -/// Parse a chunk's non-`msg` metadata (`ts`/`result`/`duration`) from the -/// portion before its `,"msg":"` field. That prefix is always valid JSON -/// even when the trailing `msg` slice is not. -fn chunk_metadata(chunk: &str) -> Result { +/// Parse a chunk's non-`msg` prefix (the portion before its `,"msg":"` +/// field, which is always valid JSON) into a [`serde_json::Value`]. +fn cloud_init_chunk_metadata(chunk: &str) -> Result { let marker = format!(",{CLOUD_INIT_MSG_MARKER}"); let end = chunk .find(&marker) @@ -811,71 +819,74 @@ fn chunk_metadata(chunk: &str) -> Result { #[cfg(test)] mod tests { use super::*; + use crate::{KvpPool, PoolMode}; use rstest::rstest; - const PREFIX: &str = "azure-init-0.1.0"; + const AGENT: &str = "azure-init-0.1.0"; const VM_ID: &str = "3f2504e0-4f89-41d3-9a0c-0305e82c3301"; const EVENT_ID: &str = "8f3e9c4a-1b2c-4d5e-9f01-234567890abc"; const BOOT_EPOCH: i64 = 1_700_000_000; + const TIMESTAMP: &str = "2026-07-27T21:33:24.300Z"; #[test] fn event_key_formats_and_classifies() { let formatted = format_event_key( - PREFIX, + AGENT, BOOT_EPOCH, - Level::INFO, - "user:create_user", VM_ID, + RecordKind::Event, + "user:create_user", EVENT_ID, + TIMESTAMP, ); assert_eq!( formatted, format!( - "{PREFIX}|{BOOT_EPOCH}|INFO|user:create_user|{VM_ID}|\ - {EVENT_ID}" + "{AGENT}|{BOOT_EPOCH}|{VM_ID}|event|user:create_user|\ + {EVENT_ID}|{TIMESTAMP}" ) ); assert!(matches!( classify_key(&formatted), KeyClass::Event { - boot_epoch_time, - event_level, - name, + agent, + boot_epoch, vm_id, + kind, + name, event_id, - } if boot_epoch_time == BOOT_EPOCH - && event_level == Level::INFO - && name == "user:create_user" + timestamp, + } if agent == AGENT + && boot_epoch == BOOT_EPOCH && vm_id == VM_ID + && kind == RecordKind::Event + && name == "user:create_user" && event_id == EVENT_ID + && timestamp == TIMESTAMP )); } #[test] - fn classify_round_trips_every_level() { - for expected in [ - Level::ERROR, - Level::WARN, - Level::INFO, - Level::DEBUG, - Level::TRACE, - ] { + fn classify_round_trips_every_kind() { + for expected in + [RecordKind::Start, RecordKind::Finish, RecordKind::Event] + { let key = format_event_key( - PREFIX, + AGENT, BOOT_EPOCH, + VM_ID, expected, "span:event", - VM_ID, EVENT_ID, + TIMESTAMP, ); assert!(matches!( classify_key(&key), - KeyClass::Event { event_level, .. } if event_level == expected + KeyClass::Event { kind, .. } if kind == expected )); } } - /// Map a key to its [`KeyClass`] discriminant for table-driven tests. fn class_of(key: &str) -> &'static str { match classify_key(key) { KeyClass::Event { .. } => "event", @@ -886,17 +897,25 @@ mod tests { } #[rstest] - #[case::event("p|100|INFO|name|vm|id", "event")] + #[case::event("a|100|vm|event|name|id|ts", "event")] #[case::cloud_init( "CLOUD_INIT|1785187982|finish|name|vmid|uuid", "cloud-init" )] #[case::raw_single_segment("PROVISIONING_REPORT", "raw")] - #[case::raw_too_few_segments("a|b|INFO|c", "raw")] - #[case::raw_too_many_segments("a|100|b|INFO|c|d|e", "raw")] - #[case::raw_non_numeric_boot_epoch("p|notnum|INFO|name|vm|id", "raw")] - #[case::malformed_bad_level("p|100|NOTALEVEL|name|vm|id", "malformed")] - #[case::malformed_other_level("p|100|NOPE|name|vm|id", "malformed")] + #[case::raw_too_few_segments("a|100|vm|event|name|id", "raw")] + #[case::raw_too_many_segments("a|100|vm|event|name|id|ts|extra", "raw")] + #[case::raw_non_numeric_boot_epoch("a|notnum|vm|event|name|id|ts", "raw")] + #[case::malformed_bad_kind("a|100|vm|NOTAKIND|name|id|ts", "malformed")] + #[case::malformed_other_kind("a|100|vm|nope|name|id|ts", "malformed")] + #[case::cloud_init_bad_type( + "CLOUD_INIT|100|weird|name|vmid|uuid", + "malformed" + )] + #[case::cloud_init_non_numeric_incarnation( + "CLOUD_INIT|notnum|finish|name|vmid|uuid", + "malformed" + )] fn classify_key_categorizes(#[case] key: &str, #[case] expected: &str) { assert_eq!(class_of(key), expected); } @@ -942,12 +961,13 @@ mod tests { #[test] fn reassemble_groups_chunks_and_classifies() { let key = format_event_key( - PREFIX, + AGENT, BOOT_EPOCH, - Level::INFO, - "config:dump", VM_ID, + RecordKind::Start, + "config:dump", EVENT_ID, + TIMESTAMP, ); let dumped = vec![ @@ -957,7 +977,7 @@ mod tests { "PROVISIONING_REPORT".to_string(), "result=success".to_string(), ), - ("p|100|NOPE|name|vm|id".to_string(), "junk".to_string()), + ("a|100|vm|NOPE|name|id|ts".to_string(), "junk".to_string()), ]; let records = reassemble(dumped); @@ -965,13 +985,17 @@ mod tests { assert_eq!( records[0], - DiagnosticRecord::Event { + DiagnosticRecord::Decoded { event: DiagnosticEvent { - boot_epoch_time: BOOT_EPOCH, - event_level: Level::INFO, + agent: AGENT.to_string(), + boot_epoch: BOOT_EPOCH, + vm_id: Some(VM_ID.to_string()), + kind: RecordKind::Start, name: "config:dump".to_string(), - vm_id: VM_ID.to_string(), event_id: EVENT_ID.to_string(), + timestamp: Some(TIMESTAMP.to_string()), + result: None, + duration: None, message: "part-one/part-two".to_string(), }, chunks: 2, @@ -986,12 +1010,13 @@ mod tests { fn reassemble_keeps_distinct_adjacent_keys_separate() { let make = |event_id: &str| { format_event_key( - PREFIX, + AGENT, BOOT_EPOCH, - Level::INFO, - "span:name", VM_ID, + RecordKind::Start, + "span:name", event_id, + TIMESTAMP, ) }; let dumped = vec![ @@ -1002,23 +1027,26 @@ mod tests { assert_eq!(records.len(), 2); assert!(matches!( &records[0], - DiagnosticRecord::Event { chunks: 1, .. } + DiagnosticRecord::Decoded { chunks: 1, .. } )); assert!(matches!( &records[1], - DiagnosticRecord::Event { chunks: 1, .. } + DiagnosticRecord::Decoded { chunks: 1, .. } )); } #[rstest] - #[case::indexed_chunk("p|100|INFO|name|vm|id|0", "p|100|INFO|name|vm|id")] + #[case::indexed_chunk( + "a|100|vm|event|name|id|ts|0", + "a|100|vm|event|name|id|ts" + )] #[case::indexed_chunk_multi_digit( - "p|100|INFO|name|vm|id|12", - "p|100|INFO|name|vm|id" + "a|100|vm|event|name|id|ts|12", + "a|100|vm|event|name|id|ts" )] #[case::single_event_unchanged( - "p|100|INFO|name|vm|id", - "p|100|INFO|name|vm|id" + "a|100|vm|event|name|id|ts", + "a|100|vm|event|name|id|ts" )] #[case::cloud_init_indexed_chunk( "CLOUD_INIT|1785187982|finish|mod|vmid|uuid|0", @@ -1027,12 +1055,12 @@ mod tests { #[case::raw_unchanged("PROVISIONING_REPORT", "PROVISIONING_REPORT")] #[case::non_event_numeric_tail_unchanged("foo|3", "foo|3")] #[case::malformed_unchanged( - "p|100|NOPE|name|vm|id", - "p|100|NOPE|name|vm|id" + "a|100|vm|NOPE|name|id|ts", + "a|100|vm|NOPE|name|id|ts" )] #[case::malformed_indexed_chunk( - "p|100|NOPE|name|vm|id|0", - "p|100|NOPE|name|vm|id" + "a|100|vm|NOPE|name|id|ts|0", + "a|100|vm|NOPE|name|id|ts" )] fn base_event_key_strips_event_subevent_index( #[case] key: &str, @@ -1044,12 +1072,13 @@ mod tests { #[test] fn reassemble_groups_indexed_chunk_keys() { let base = format_event_key( - PREFIX, + AGENT, BOOT_EPOCH, - Level::INFO, - "config:dump", VM_ID, + RecordKind::Finish, + "config:dump", EVENT_ID, + TIMESTAMP, ); let dumped = vec![ (format!("{base}|0"), "part-one/".to_string()), @@ -1061,13 +1090,17 @@ mod tests { assert_eq!(records.len(), 1); assert_eq!( records[0], - DiagnosticRecord::Event { + DiagnosticRecord::Decoded { event: DiagnosticEvent { - boot_epoch_time: BOOT_EPOCH, - event_level: Level::INFO, + agent: AGENT.to_string(), + boot_epoch: BOOT_EPOCH, + vm_id: Some(VM_ID.to_string()), + kind: RecordKind::Finish, name: "config:dump".to_string(), - vm_id: VM_ID.to_string(), event_id: EVENT_ID.to_string(), + timestamp: Some(TIMESTAMP.to_string()), + result: None, + duration: None, message: "part-one/part-two/part-three".to_string(), }, chunks: 3, @@ -1075,51 +1108,113 @@ mod tests { ); } - // ---- cloud-init read support ---- + #[test] + fn azure_event_value_is_full_message() { + let key = format_event_key( + AGENT, + BOOT_EPOCH, + VM_ID, + RecordKind::Event, + "user:create_user", + EVENT_ID, + TIMESTAMP, + ); + assert!(matches!( + classify_record(key, vec!["boom".to_string()]), + DiagnosticRecord::Decoded { event, chunks: 1 } + if event.kind == RecordKind::Event + && event.message == "boom" + && event.agent == AGENT + && event.vm_id.as_deref() == Some(VM_ID) + && event.timestamp.as_deref() == Some(TIMESTAMP) + )); + } + + #[test] + fn azure_span_value_is_full_message() { + let key = format_event_key( + AGENT, + BOOT_EPOCH, + VM_ID, + RecordKind::Finish, + "config:write", + EVENT_ID, + TIMESTAMP, + ); + assert!(matches!( + classify_record(key, vec!["write_config completed".to_string()]), + DiagnosticRecord::Decoded { event, chunks: 1 } + if event.kind == RecordKind::Finish + && event.message == "write_config completed" + )); + } + + #[test] + fn azure_span_start_finish_pair_round_trips_through_writer() { + let dir = tempfile::TempDir::new().unwrap(); + let store = + KvpPoolStore::new_in(KvpPool::Guest, dir.path(), PoolMode::Safe) + .unwrap(); + let diag = DiagnosticsKvp::new(store, VM_ID, AGENT); + + // A span emits a start and a finish sharing one event_id. + diag.write_event( + RecordKind::Start, + EVENT_ID, + "provision:run", + "starting provision", + ) + .unwrap(); + diag.write_event( + RecordKind::Finish, + EVENT_ID, + "provision:run", + "provision completed", + ) + .unwrap(); + + let events = diag.events().unwrap(); + assert_eq!(events.len(), 2); + + assert_eq!(events[0].kind, RecordKind::Start); + assert_eq!(events[0].message, "starting provision"); + assert_eq!(events[1].kind, RecordKind::Finish); + assert_eq!(events[1].message, "provision completed"); + + for event in &events { + assert_eq!(event.event_id, EVENT_ID); + assert_eq!(event.name, "provision:run"); + assert_eq!(event.agent, AGENT); + assert_eq!(event.vm_id.as_deref(), Some(VM_ID)); + } + } const CLOUD_INIT_VM_ID: &str = "0e5e179d-5341-478b-8456-fbb90621bdf8"; const CLOUD_INIT_KEY_FINISH: &str = "CLOUD_INIT|1785187982|finish|modules-final/config-scripts_user|0e5e179d-5341-478b-8456-fbb90621bdf8|e5f01809-a7a3-4279-aa64-1f18e21eda6e"; const CLOUD_INIT_VALUE_FINISH: &str = r#"{"name":"modules-final/config-scripts_user","type":"finish","ts":"2026-07-27T21:33:24.339006+00:00","result":"SUCCESS","duration":0.0006448590000012189,"msg":"config-scripts_user ran successfully and took 0.001 seconds"}"#; - #[rstest] - #[case::with_vm_id( - CLOUD_INIT_KEY_FINISH, - "1785187982", - "finish", - "modules-final/config-scripts_user", - Some(CLOUD_INIT_VM_ID), - "e5f01809-a7a3-4279-aa64-1f18e21eda6e" - )] - // Older cloud-init builds omit the vm_id key segment. - #[case::without_vm_id( - "CLOUD_INIT|1785187982|start|modules-config/foo|c4d4a08d-fe93-4c7a-9be6-9a38c212e212", - "1785187982", - "start", - "modules-config/foo", - None, - "c4d4a08d-fe93-4c7a-9be6-9a38c212e212" - )] - fn cloud_init_key_classifies( - #[case] key: &str, - #[case] incarnation: &str, - #[case] event_type: &str, - #[case] name: &str, - #[case] vm_id: Option<&str>, - #[case] uuid: &str, - ) { + #[test] + fn cloud_init_key_classifies() { + assert!(matches!( + classify_key(CLOUD_INIT_KEY_FINISH), + KeyClass::CloudInit { boot_epoch, kind, name, vm_id, uuid } + if boot_epoch == 1785187982 + && kind == RecordKind::Finish + && name == "modules-final/config-scripts_user" + && vm_id == Some(CLOUD_INIT_VM_ID) + && uuid == "e5f01809-a7a3-4279-aa64-1f18e21eda6e" + )); assert!(matches!( - classify_key(key), - KeyClass::CloudInit { - incarnation: i, - event_type: t, - name: n, - vm_id: v, - uuid: u, - } if i == incarnation - && t == event_type - && n == name - && v == vm_id - && u == uuid + classify_key( + "CLOUD_INIT|1785187982|start|modules-config/foo|\ + c4d4a08d-fe93-4c7a-9be6-9a38c212e212" + ), + KeyClass::CloudInit { boot_epoch, kind, name, vm_id, uuid } + if boot_epoch == 1785187982 + && kind == RecordKind::Start + && name == "modules-config/foo" + && vm_id.is_none() + && uuid == "c4d4a08d-fe93-4c7a-9be6-9a38c212e212" )); } @@ -1136,19 +1231,18 @@ mod tests { #[test] fn cloud_init_finish_record_decodes_all_fields() { - // A single `matches!` covers every field (with a tolerance on the - // float duration) and leaves no unreachable arm to cover. assert!(matches!( classify_record( CLOUD_INIT_KEY_FINISH.to_string(), vec![CLOUD_INIT_VALUE_FINISH.to_string()], ), - DiagnosticRecord::CloudInit { event, chunks: 1 } - if event.incarnation == "1785187982" - && event.event_type == "finish" + DiagnosticRecord::Decoded { event, chunks: 1 } + if event.agent == "CLOUD_INIT" + && event.boot_epoch == 1785187982 + && event.kind == RecordKind::Finish && event.name == "modules-final/config-scripts_user" && event.vm_id.as_deref() == Some(CLOUD_INIT_VM_ID) - && event.uuid == "e5f01809-a7a3-4279-aa64-1f18e21eda6e" + && event.event_id == "e5f01809-a7a3-4279-aa64-1f18e21eda6e" && event.timestamp.as_deref() == Some("2026-07-27T21:33:24.339006+00:00") && event.result.as_deref() == Some("SUCCESS") @@ -1167,13 +1261,15 @@ mod tests { let key = "CLOUD_INIT|1785187982|start|modules-final/config-keys_to_console|0e5e179d-5341-478b-8456-fbb90621bdf8|7792621b-b339-4274-8b71-2a3dcbd2db4e"; assert_eq!( classify_record(key.to_string(), vec![value.to_string()]), - DiagnosticRecord::CloudInit { - event: CloudInitEvent { - incarnation: "1785187982".to_string(), - event_type: "start".to_string(), - name: "modules-final/config-keys_to_console".to_string(), + DiagnosticRecord::Decoded { + event: DiagnosticEvent { + agent: "CLOUD_INIT".to_string(), + boot_epoch: 1785187982, vm_id: Some(CLOUD_INIT_VM_ID.to_string()), - uuid: "7792621b-b339-4274-8b71-2a3dcbd2db4e".to_string(), + kind: RecordKind::Start, + name: "modules-final/config-keys_to_console".to_string(), + event_id: "7792621b-b339-4274-8b71-2a3dcbd2db4e" + .to_string(), timestamp: Some( "2026-07-27T21:33:24.344349+00:00".to_string() ), @@ -1203,9 +1299,6 @@ mod tests { #[test] fn cloud_init_chunks_reassemble_by_subevent_index() { let base = "CLOUD_INIT|1785187982|finish|modules-final/long|0e5e179d-5341-478b-8456-fbb90621bdf8|abc12345-1111-2222-3333-444455556666"; - // Each chunk carries its own `|` key suffix and a - // message slice. They are laid out on disk out of order to prove - // reassembly restores order by that index, not disk position. let chunk = |i: u32, msg: &str| { format!( r#"{{"name":"modules-final/long","type":"finish","ts":"2026-07-27T21:33:24.339006+00:00","result":"SUCCESS","duration":0.5,"msg_i":{i},"msg":"{msg}"}}"# @@ -1220,13 +1313,15 @@ mod tests { let records = reassemble(dumped); assert_eq!( records, - vec![DiagnosticRecord::CloudInit { - event: CloudInitEvent { - incarnation: "1785187982".to_string(), - event_type: "finish".to_string(), - name: "modules-final/long".to_string(), + vec![DiagnosticRecord::Decoded { + event: DiagnosticEvent { + agent: "CLOUD_INIT".to_string(), + boot_epoch: 1785187982, vm_id: Some(CLOUD_INIT_VM_ID.to_string()), - uuid: "abc12345-1111-2222-3333-444455556666".to_string(), + kind: RecordKind::Finish, + name: "modules-final/long".to_string(), + event_id: "abc12345-1111-2222-3333-444455556666" + .to_string(), timestamp: Some( "2026-07-27T21:33:24.339006+00:00".to_string() ), @@ -1241,12 +1336,6 @@ mod tests { #[test] fn cloud_init_chunks_reassemble_split_json_escape() { - // Regression: cloud-init's `_break_down` re-emits the metadata - // (plus a `msg_i` chunk index) on every chunk and slices the - // JSON-escaped message at character boundaries, so a `\n` escape - // can straddle two chunks — the first ends in a lone backslash and - // is not valid JSON on its own. The reader must rejoin the raw - // slices before unescaping. let base = "CLOUD_INIT|1785187982|finish|modules-final/x|0e5e179d-5341-478b-8456-fbb90621bdf8|abc12345-1111-2222-3333-444455556666"; let dumped = vec![ ( @@ -1264,7 +1353,7 @@ mod tests { let records = reassemble(dumped); assert!(matches!( &records[..], - [DiagnosticRecord::CloudInit { event, chunks: 2 }] + [DiagnosticRecord::Decoded { event, chunks: 2 }] if event.message == "line1\nline2" && event.result.as_deref() == Some("SUCCESS") && event.duration == Some(0.5) diff --git a/libazureinit-kvp/src/error.rs b/libazureinit-kvp/src/error.rs index 71b75abb..0fb171d0 100644 --- a/libazureinit-kvp/src/error.rs +++ b/libazureinit-kvp/src/error.rs @@ -11,9 +11,9 @@ pub enum KvpError { EmptyKey, /// An underlying I/O error. Io(io::Error), - /// An event key field (`event_prefix`, `vm_id`, `name`, or - /// `event_id`) contained the `|` delimiter, which would make the - /// formatted event key ambiguous to parse back. + /// An event key field (`agent`, `vm_id`, `name`, or `event_id`) + /// contained the `|` delimiter, which would make the formatted event + /// key ambiguous to parse back. EventFieldContainsDelimiter { field: &'static str }, /// The key contains a null byte, which is incompatible with the /// on-disk format (null-padded fixed-width fields). diff --git a/libazureinit-kvp/src/lib.rs b/libazureinit-kvp/src/lib.rs index 54dc57ca..197093f0 100644 --- a/libazureinit-kvp/src/lib.rs +++ b/libazureinit-kvp/src/lib.rs @@ -21,7 +21,7 @@ mod vm_id; pub use cli::run; pub use diagnostics::{ - CloudInitEvent, DiagnosticEvent, DiagnosticRecord, DiagnosticsKvp, + DiagnosticEvent, DiagnosticRecord, DiagnosticsKvp, RecordKind, MAX_CHUNK_BYTES, }; pub use error::KvpError; diff --git a/libazureinit-kvp/src/store.rs b/libazureinit-kvp/src/store.rs index 1f722bab..9c66569c 100644 --- a/libazureinit-kvp/src/store.rs +++ b/libazureinit-kvp/src/store.rs @@ -211,8 +211,6 @@ impl KvpPoolStore { let boot_time = boot_time(&*self.ops)?; lock_for_writing(&mut *handle)?; - // Strict `<`: a file whose mtime equals boot time was written - // this boot (during the boot second), so it is not stale. if handle.metadata()?.mtime < boot_time { handle.set_len(0)?; } @@ -401,8 +399,6 @@ impl KvpPoolStore { Err(ref e) if e.kind() == ErrorKind::NotFound => return Ok(false), Err(e) => return Err(e.into()), }; - // Strict `<`: mtime == boot_time means the file was written this - // boot (during the boot second), so it is not stale. Ok(metadata.mtime < boot_time(&*self.ops)?) } @@ -537,8 +533,6 @@ impl KvpPoolStore { iter.flush()?; Ok(()) } - - /// Return a reference to the pool file path. pub fn path(&self) -> &Path { &self.path } @@ -1374,7 +1368,6 @@ mod tests { assert_eq!(store.read(&long_key).unwrap(), Some("val".to_string())); - // Read is not size-capped: an oversized key simply misses. let too_long = "k".repeat(513); assert_eq!(store.read(&too_long).unwrap(), None); } @@ -1675,7 +1668,6 @@ mod tests { assert!(store.delete("k4").unwrap()); - // k9 takes k4's slot; the rest stays put. assert_eq!( store.dump().unwrap(), pairs([ @@ -1769,7 +1761,6 @@ mod tests { let store = safe_store(dir.path()); store.load(pairs([("keep", "me")])).unwrap(); - // Bad record mid-batch: file must be untouched on rejection. let bad_value = "v".repeat(1023); let err = store .load(vec![ @@ -1794,7 +1785,6 @@ mod tests { let err = store.load(too_many).unwrap_err(); assert!(is_max_keys(&err), "got {err:?}"); - // Cap is checked pre-lock; the file is never opened. assert!(!store.path().exists()); } @@ -1803,7 +1793,6 @@ mod tests { let dir = TempDir::new().unwrap(); let store = safe_store(dir.path()); - // 2 * MAX_UNIQUE_KEYS records, MAX_UNIQUE_KEYS unique keys. let mut records: Vec<(String, String)> = (0..MAX_UNIQUE_KEYS) .map(|i| (format!("k{i}"), "a".to_string())) .collect(); @@ -1828,7 +1817,6 @@ mod tests { let err = store.insert("overflow", "v").unwrap_err(); assert!(is_max_keys(&err), "got {err:?}"); - // Overwriting an existing key at the cap still works. store.insert("k0", "updated").unwrap(); assert_eq!(store.read("k0").unwrap(), Some("updated".to_string())); } @@ -1864,14 +1852,11 @@ mod tests { let dir = TempDir::new().unwrap(); let store = safe_store(dir.path()); - // Mirrors the chunked-event use case: many records sharing a - // single key, written atomically. let records = pairs([("chunk", "part1"), ("chunk", "part2"), ("chunk", "part3")]); store.append_multiple(records.clone()).unwrap(); assert_eq!(store.dump().unwrap(), records); - // `read` returns last-write-wins; entries() collapses to 1. assert_eq!(store.entries().unwrap().len(), 1); } @@ -1884,7 +1869,6 @@ mod tests { .append_multiple(Vec::<(String, String)>::new()) .unwrap(); - // No file created when the input is empty. assert!(!store.path().exists()); } @@ -1904,8 +1888,6 @@ mod tests { .unwrap_err(); assert!(is_value_too_large(&err), "got {err:?}"); - // Rejection is all-or-nothing: previously-written records - // are untouched, none of the new batch lands. assert_eq!(store.dump().unwrap(), pairs([("keep", "me")])); } @@ -1934,16 +1916,11 @@ mod tests { #[test] fn test_append_multiple_does_not_enforce_unique_key_cap() { - // Matches `append`'s contract: the bulk variant deliberately - // skips the unique-key cap so chunked writes (many records - // sharing one key) cannot accidentally trip it. let dir = TempDir::new().unwrap(); let store = safe_store(dir.path()); seed_unique_keys(&store, MAX_UNIQUE_KEYS); - // Adding a new unique key via append_multiple is allowed even - // when the pool is already at the cap. store.append_multiple(pairs([("extra", "v")])).unwrap(); assert_eq!(store.len().unwrap(), MAX_UNIQUE_KEYS + 1); } @@ -1973,7 +1950,6 @@ mod tests { let dir = TempDir::new().unwrap(); let store = safe_store(dir.path()); - // Two unique keys, three matching records. store .load(pairs([ ("k", "v1"), @@ -2022,7 +1998,6 @@ mod tests { let removed = store.delete_multiple(Vec::::new()).unwrap(); assert_eq!(removed, 0); - // Empty input never opens or creates the file. assert!(!store.path().exists()); } @@ -2043,8 +2018,6 @@ mod tests { store.load(pairs([("a", "1"), ("b", "2")])).unwrap(); - // Listing the same key twice still removes the (one) record - // exactly once. let removed = store.delete_multiple(vec!["a", "a", "a"]).unwrap(); assert_eq!(removed, 1); assert_eq!(store.dump().unwrap(), pairs([("b", "2")])); @@ -2061,7 +2034,6 @@ mod tests { .unwrap_err(); assert!(matches!(err, KvpError::EmptyKey), "got {err:?}"); - // Validation runs before any record is removed. assert_eq!(store.dump().unwrap(), pairs([("a", "1")])); } @@ -2078,8 +2050,6 @@ mod tests { #[test] fn test_delete_multiple_size_independent_of_mode() { - // Mirrors `delete`'s contract: keys longer than the safe-mode - // cap can be removed from a safe-mode store. let dir = TempDir::new().unwrap(); let store_unsafe = unsafe_store(dir.path()); let long_key = "k".repeat(SAFE_MAX_KEY_BYTES + 1); @@ -3078,21 +3048,14 @@ mod tests { let store = safe_store(dir.path()); store.load(pairs([("a", "1"), ("b", "2")])).unwrap(); - // Open a mutable iterator (exclusive lock) so we can manipulate the file. let mut iter = store.iter_mut().unwrap(); assert_eq!(iter.record_count(), 2); - // Read the first record successfully. let (k, _) = iter.next().unwrap().unwrap(); assert_eq!(k, "a"); - // Truncate via the iterator's own handle to remove the second record. - // The iterator still thinks record_count == 2, so the next - // read_exact will hit an unexpected EOF. iter.handle.set_len(0).unwrap(); - // The iterator's cached record_count (2) > current_index (1), - // so it attempts read_exact, which fails. let err = iter.next().unwrap().unwrap_err(); assert_eq!(err.kind(), ErrorKind::UnexpectedEof); } @@ -3360,12 +3323,15 @@ mod tests { fn is_io(e: &KvpError) -> bool { matches!(e, KvpError::Io(_)) } + fn is_max_keys(e: &KvpError) -> bool { matches!(e, KvpError::MaxUniqueKeysExceeded { .. }) } + fn is_value_too_large(e: &KvpError) -> bool { matches!(e, KvpError::ValueTooLarge { .. }) } + fn is_key_too_large(e: &KvpError) -> bool { matches!(e, KvpError::KeyTooLarge { .. }) } @@ -3455,7 +3421,6 @@ mod tests { #[test] fn test_clear_if_stale_truncates_when_stale() { - // mtime (0) < boot_time (10) → triggers set_len branch. let (store, ops, p) = mock_store(PoolMode::Safe); preload(&ops, &p, &[("a", "1")]); ops.set_boot_time(10); @@ -3485,9 +3450,6 @@ mod tests { #[test] fn test_clear_if_stale_keeps_file_written_in_boot_second() { - // Boundary regression: mtime (10) == boot_time (10) means the - // file was written this boot (during the boot second), so it is - // NOT stale and must survive clear_if_stale (strict `<`). let (store, ops, p) = mock_store(PoolMode::Safe); ops.put_file(&p, vec![0u8; RECORD_SIZE], 10); ops.set_boot_time(10); diff --git a/libazureinit-kvp/src/vm_id.rs b/libazureinit-kvp/src/vm_id.rs index a8c1edbd..7b1e5a85 100644 --- a/libazureinit-kvp/src/vm_id.rs +++ b/libazureinit-kvp/src/vm_id.rs @@ -67,7 +67,7 @@ fn is_vm_gen1( let sysfs_efi = sysfs_efi_path.unwrap_or("/sys/firmware/efi"); let dev_efi = dev_efi_path.unwrap_or("/dev/efi"); - // If *either* efi path exists, this is Gen2; if *neither* exist, Gen1. + // If either efi path exists, this is Gen2; if neither exist, Gen1. !Path::new(sysfs_efi).exists() && !Path::new(dev_efi).exists() } @@ -176,8 +176,6 @@ mod tests { let path = dir.path().join("product_uuid"); fs::write(&path, "not-a-uuid").unwrap(); - // Gen1 (no EFI paths) but the content cannot be parsed as a UUID, - // so the raw lowercased value is returned unchanged. let actual = private_get_vm_id( Some(path.to_str().unwrap()), Some("/nonexistent_sysfs_efi"), @@ -190,9 +188,6 @@ mod tests { #[test] fn get_vm_id_public_wrapper_is_callable() { - // Exercises the public entry point. It reads the host's - // product_uuid if present, so the result is environment dependent; - // we only assert that invoking it does not panic. let _ = get_vm_id(); } diff --git a/libazureinit-kvp/tests/cli.rs b/libazureinit-kvp/tests/cli.rs index e5c20417..6c6afd03 100644 --- a/libazureinit-kvp/tests/cli.rs +++ b/libazureinit-kvp/tests/cli.rs @@ -236,9 +236,6 @@ fn validation_errors_exit_two() { #[test] fn json_read_round_trips_value_with_equals_and_newline() { let dir = TempDir::new().unwrap(); - // A value containing both '=' and an embedded newline would be - // ambiguous in the default key=value text output but must survive - // round-tripping through JSON unchanged. let raw_value = "https://example.test/q=1\nline2"; let status = std::process::Command::new(env!("CARGO_BIN_EXE_libazureinit-kvp")) @@ -302,13 +299,12 @@ fn report_failure_rejects_invalid_supporting_data() { #[test] fn dump_parse_diagnostics_json_reassembles_and_classifies() { let dir = TempDir::new().unwrap(); - // A two-chunk event (same key repeated) plus a raw record. assert_success(kvp(&with_dir( &dir, &[ "write", "--append", - "azure-init-x|100|INFO|a:b|vm|id1", + "azure-init-x|100|vm|event|a:b|id1|ts", "one/", ], ))); @@ -317,7 +313,7 @@ fn dump_parse_diagnostics_json_reassembles_and_classifies() { &[ "write", "--append", - "azure-init-x|100|INFO|a:b|vm|id1", + "azure-init-x|100|vm|event|a:b|id1|ts", "two", ], ))); @@ -333,22 +329,19 @@ fn dump_parse_diagnostics_json_reassembles_and_classifies() { assert!(out.contains("\"kind\":\"event\"")); assert!(out.contains("\"chunks\":2")); assert!(out.contains("\"message\":\"one/two\"")); - // Raw records are hidden without --include-raw. assert!(!out.contains("PROVISIONING_REPORT")); let out_raw = assert_success(kvp(&with_dir( &dir, &["--json", "dump", "--parse-diagnostics", "--include-raw"], ))); - assert!(out_raw.contains("\"kind\":\"raw\"")); + assert!(out_raw.contains("\"record\":\"raw\"")); assert!(out_raw.contains("PROVISIONING_REPORT")); } #[test] fn dump_parse_diagnostics_text_renders_cloud_init_event() { let dir = TempDir::new().unwrap(); - // A finish event carries every optional field (vm_id, result, ts, - // duration); a start event omits result and duration. assert_success(kvp(&with_dir( &dir, &[ @@ -370,18 +363,17 @@ fn dump_parse_diagnostics_text_renders_cloud_init_event() { let out = assert_success(kvp(&with_dir(&dir, &["dump", "--parse-diagnostics"]))); - // The finish event renders every optional field. - assert!(out.contains("cloud-init-event type=finish")); + assert!(out.contains("event kind=finish")); + assert!(out.contains("agent=CLOUD_INIT")); + assert!(out.contains("boot_epoch=1785187982")); assert!(out.contains("name=modules-final/config-scripts_user")); assert!(out.contains("vm_id=0e5e179d-5341-478b-8456-fbb90621bdf8")); assert!(out.contains("result=SUCCESS")); - assert!(out.contains("ts=2026-07-27T21:33:24.339006+00:00")); + assert!(out.contains("timestamp=2026-07-27T21:33:24.339006+00:00")); assert!(out.contains("duration=0.5")); - assert!(out.contains("incarnation=1785187982")); assert!(out.contains("chunks=1")); assert!(out.contains("message=scripts ran")); - // The start event omits result and duration. - assert!(out.contains("cloud-init-event type=start")); + assert!(out.contains("event kind=start")); } #[test] @@ -401,50 +393,32 @@ fn dump_parse_diagnostics_json_renders_cloud_init_event() { &dir, &["--json", "dump", "--parse-diagnostics"], ))); - assert!(out.contains("\"kind\":\"cloud-init-event\"")); - assert!(out.contains("\"incarnation\":\"1785187982\"")); - assert!(out.contains("\"type\":\"finish\"")); + assert!(out.contains("\"record\":\"event\"")); + assert!(out.contains("\"kind\":\"finish\"")); + assert!(out.contains("\"agent\":\"CLOUD_INIT\"")); + assert!(out.contains("\"boot_epoch\":1785187982")); assert!(out.contains("\"name\":\"modules-final/config-scripts_user\"")); assert!(out.contains("\"vm_id\":\"0e5e179d-5341-478b-8456-fbb90621bdf8\"")); - assert!(out.contains("\"uuid\":\"e5f01809-a7a3-4279-aa64-1f18e21eda6e\"")); - assert!(out.contains("\"ts\":\"2026-07-27T21:33:24.339006+00:00\"")); + assert!( + out.contains("\"event_id\":\"e5f01809-a7a3-4279-aa64-1f18e21eda6e\"") + ); + assert!(out.contains("\"timestamp\":\"2026-07-27T21:33:24.339006+00:00\"")); assert!(out.contains("\"result\":\"SUCCESS\"")); assert!(out.contains("\"duration\":0.5")); assert!(out.contains("\"chunks\":1")); assert!(out.contains("\"message\":\"scripts ran\"")); } -#[test] -fn dump_parse_diagnostics_filters_by_level() { - let dir = TempDir::new().unwrap(); - assert_success(kvp(&with_dir( - &dir, - &["write", "--append", "p|100|INFO|a:b|vm|i1", "info-msg"], - ))); - assert_success(kvp(&with_dir( - &dir, - &["write", "--append", "p|100|ERROR|c:d|vm|i2", "err-msg"], - ))); - - let out = assert_success(kvp(&with_dir( - &dir, - &["dump", "--parse-diagnostics", "--level", "error"], - ))); - assert!(out.contains("err-msg")); - assert!(!out.contains("info-msg")); -} - #[test] fn clear_diagnostics_removes_events_and_malformed_keeps_raw() { let dir = TempDir::new().unwrap(); - // An event, a malformed event key, and a raw record. assert_success(kvp(&with_dir( &dir, - &["write", "--append", "p|100|INFO|a:b|vm|i1", "msg"], + &["write", "--append", "a|100|vm|event|a:b|i1|ts", "msg"], ))); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "p|100|NOPE|c:d|vm|i2", "junk"], + &["write", "--append", "a|100|vm|NOPE|c:d|i2|ts", "junk"], ))); assert_success(kvp(&with_dir( &dir, @@ -470,11 +444,11 @@ fn dump_parse_diagnostics_text_renders_all_record_kinds() { let dir = TempDir::new().unwrap(); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "p|100|INFO|a:b|vm|id1", "one/"], + &["write", "--append", "a|100|vm|event|a:b|id1|ts", "one/"], ))); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "p|100|INFO|a:b|vm|id1", "two"], + &["write", "--append", "a|100|vm|event|a:b|id1|ts", "two"], ))); assert_success(kvp(&with_dir( &dir, @@ -482,7 +456,7 @@ fn dump_parse_diagnostics_text_renders_all_record_kinds() { ))); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "p|100|NOPE|c:d|vm|id2", "junk"], + &["write", "--append", "a|100|vm|NOPE|c:d|id2|ts", "junk"], ))); let out = assert_success(kvp(&with_dir( @@ -490,11 +464,11 @@ fn dump_parse_diagnostics_text_renders_all_record_kinds() { &["dump", "--parse-diagnostics", "--include-raw"], ))); assert!(out.contains( - "event boot_epoch_time=100 event_level=INFO name=a:b vm_id=vm \ - event_id=id1 chunks=2 message=one/two" + "event kind=event agent=a boot_epoch=100 vm_id=vm \ + name=a:b event_id=id1 timestamp=ts chunks=2 message=one/two" )); assert!(out.contains("raw key=PROVISIONING_REPORT value=result=success")); - assert!(out.contains("malformed key=p|100|NOPE|c:d|vm|id2")); + assert!(out.contains("malformed key=a|100|vm|NOPE|c:d|id2|ts")); assert!(out.contains("value=junk")); } @@ -503,11 +477,11 @@ fn dump_parse_diagnostics_tail_limits_to_last_events() { let dir = TempDir::new().unwrap(); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "p|100|INFO|a:b|vm|i1", "first"], + &["write", "--append", "a|100|vm|event|a:b|i1|ts", "first"], ))); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "p|100|INFO|c:d|vm|i2", "second"], + &["write", "--append", "a|100|vm|event|c:d|i2|ts", "second"], ))); let out = assert_success(kvp(&with_dir( @@ -527,13 +501,12 @@ fn dump_parse_diagnostics_tail_defaults_to_20_when_count_omitted() { &[ "write", "--append", - &format!("p|100|INFO|n:{i}|vm|id{i}"), + &format!("a|100|vm|event|n:{i}|id{i}|ts"), &format!("msg{i}"), ], ))); } - // Bare --tail keeps the last 20 events (msg6..msg25). let out = assert_success(kvp(&with_dir( &dir, &["dump", "--parse-diagnostics", "--tail"], @@ -549,11 +522,11 @@ fn dump_parse_diagnostics_filters_by_name_substring() { let dir = TempDir::new().unwrap(); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "p|100|INFO|user:add|vm|i1", "u"], + &["write", "--append", "a|100|vm|event|user:add|i1|ts", "u"], ))); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "p|100|INFO|ssh:key|vm|i2", "s"], + &["write", "--append", "a|100|vm|event|ssh:key|i2|ts", "s"], ))); let out = assert_success(kvp(&with_dir( @@ -564,16 +537,6 @@ fn dump_parse_diagnostics_filters_by_name_substring() { assert!(!out.contains("user:add")); } -#[test] -fn dump_parse_diagnostics_rejects_invalid_level() { - let dir = TempDir::new().unwrap(); - let output = kvp(&with_dir( - &dir, - &["dump", "--parse-diagnostics", "--level", "bogus"], - )); - assert_eq!(output.status.code(), Some(2)); -} - #[test] fn dump_parse_diagnostics_include_raw_conflicts_with_filters() { let dir = TempDir::new().unwrap(); @@ -583,8 +546,8 @@ fn dump_parse_diagnostics_include_raw_conflicts_with_filters() { "dump", "--parse-diagnostics", "--include-raw", - "--level", - "info", + "--name", + "a:b", ], )); assert_eq!(output.status.code(), Some(2)); @@ -595,11 +558,11 @@ fn dump_parse_diagnostics_json_covers_events_and_malformed() { let dir = TempDir::new().unwrap(); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "p|100|INFO|a:b|vm|i1", "hello"], + &["write", "--append", "a|100|vm|event|a:b|i1|ts", "hello"], ))); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "p|100|NOPE|c:d|vm|i2", "junk"], + &["write", "--append", "a|100|vm|NOPE|c:d|i2|ts", "junk"], ))); let dump = assert_success(kvp(&with_dir( @@ -607,16 +570,15 @@ fn dump_parse_diagnostics_json_covers_events_and_malformed() { &["--json", "dump", "--parse-diagnostics"], ))); assert!(dump.contains("\"kind\":\"event\"")); - assert!(dump.contains("\"kind\":\"malformed\"")); + assert!(dump.contains("\"record\":\"malformed\"")); assert!(dump.contains("\"reason\":")); - // Filtering by name yields the events-only view. let events = assert_success(kvp(&with_dir( &dir, &["--json", "dump", "--parse-diagnostics", "--name", "a:b"], ))); assert!(events.contains("\"message\":\"hello\"")); - assert!(!events.contains("\"kind\":\"malformed\"")); + assert!(!events.contains("malformed")); } #[test] @@ -626,8 +588,6 @@ fn emit_writes_event_readable_by_dump() { &dir, &[ "emit", - "--level", - "info", "--name", "user:create_user", "--message", @@ -641,18 +601,15 @@ fn emit_writes_event_readable_by_dump() { let out = assert_success(kvp(&with_dir(&dir, &["dump", "--parse-diagnostics"]))); - assert!(out.contains("event_level=INFO")); assert!(out.contains("vm_id=vm-emit")); assert!(out.contains("name=user:create_user")); assert!(out.contains("message=created azureuser")); - // The JSON view exposes the same structured fields. let json = assert_success(kvp(&with_dir( &dir, &["--json", "dump", "--parse-diagnostics"], ))); assert!(json.contains("\"kind\":\"event\"")); - assert!(json.contains("\"event_level\":\"INFO\"")); assert!(json.contains("\"vm_id\":\"vm-emit\"")); assert!(json.contains("\"name\":\"user:create_user\"")); } diff --git a/libazureinit-kvp/tests/diagnostics.rs b/libazureinit-kvp/tests/diagnostics.rs index efe4dd47..2e40bad9 100644 --- a/libazureinit-kvp/tests/diagnostics.rs +++ b/libazureinit-kvp/tests/diagnostics.rs @@ -10,10 +10,9 @@ use std::thread; use libazureinit_kvp::{ DiagnosticRecord, DiagnosticsKvp, KvpPool, KvpPoolStore, PoolMode, - MAX_CHUNK_BYTES, + RecordKind, MAX_CHUNK_BYTES, }; use tempfile::TempDir; -use tracing::Level; const PREFIX: &str = "azure-init-test"; const VM_ID: &str = "vm-abc"; @@ -52,21 +51,19 @@ fn reads_and_parses_real_cloud_init_pool() { store.append(key, value).unwrap(); } - // A DiagnosticsKvp with no azure-init identity still reads cloud-init - // entries written by another agent. let diagnostics = DiagnosticsKvp::new(store, "", ""); let records = diagnostics.records().unwrap(); assert_eq!(records.len(), CLOUD_INIT_RECORDS.len()); - // Every record decodes as a cloud-init event (none fall back to raw). for record in &records { - assert!(matches!(record, DiagnosticRecord::CloudInit { .. })); + assert!(matches!(record, DiagnosticRecord::Decoded { .. })); } match &records[0] { - DiagnosticRecord::CloudInit { event, chunks } => { + DiagnosticRecord::Decoded { event, chunks } => { assert_eq!(*chunks, 1); - assert_eq!(event.event_type, "finish"); + assert_eq!(event.agent, "CLOUD_INIT"); + assert_eq!(event.kind, RecordKind::Finish); assert_eq!(event.name, "modules-final/config-scripts_user"); assert_eq!( event.vm_id.as_deref(), @@ -78,17 +75,16 @@ fn reads_and_parses_real_cloud_init_pool() { "config-scripts_user ran successfully and took 0.001 seconds" ); } - other => panic!("expected cloud-init event, got {other:?}"), + other => panic!("expected event, got {other:?}"), } - // A `start` event carries neither result nor duration. match &records[1] { - DiagnosticRecord::CloudInit { event, .. } => { - assert_eq!(event.event_type, "start"); + DiagnosticRecord::Decoded { event, .. } => { + assert_eq!(event.kind, RecordKind::Start); assert!(event.result.is_none()); assert!(event.duration.is_none()); } - other => panic!("expected cloud-init event, got {other:?}"), + other => panic!("expected event, got {other:?}"), } } @@ -98,27 +94,23 @@ fn short_event_round_trips_as_single_record() { let diag = diagnostics(&dir); assert_eq!(diag.vm_id(), VM_ID); - assert_eq!(diag.event_prefix(), PREFIX); + assert_eq!(diag.agent(), PREFIX); - diag.emit(Level::INFO, "user:create_user", "created") - .unwrap(); + diag.emit_event("user:create_user", "created").unwrap(); assert_eq!(diag.store().dump().unwrap().len(), 1); let records = diag.records().unwrap(); assert_eq!(records.len(), 1); match &records[0] { - DiagnosticRecord::Event { + DiagnosticRecord::Decoded { event: decoded, chunks, } => { assert_eq!(*chunks, 1); - assert_eq!(decoded.event_level, Level::INFO); - assert_eq!(decoded.vm_id, VM_ID); - assert_eq!( - decoded.boot_epoch_time, - diag.store().boot_epoch().unwrap() - ); + assert_eq!(decoded.kind, RecordKind::Event); + assert_eq!(decoded.vm_id.as_deref(), Some(VM_ID)); + assert_eq!(decoded.boot_epoch, diag.store().boot_epoch().unwrap()); assert_eq!(decoded.name, "user:create_user"); let event_id = uuid::Uuid::parse_str(&decoded.event_id) .expect("event_id should be a valid UUID"); @@ -139,11 +131,8 @@ fn long_event_splits_across_records_and_reassembles() { let diag = diagnostics(&dir); let message = "x".repeat(MAX_CHUNK_BYTES * 3 + 50); - diag.emit(Level::DEBUG, "config:dump", &message).unwrap(); + diag.emit_event("config:dump", &message).unwrap(); - // Split across four records, each with a unique `|` - // key so the Hyper-V host (one record per key) keeps every chunk; - // they share one event-key base. let dumped = diag.store().dump().unwrap(); assert_eq!(dumped.len(), 4); let base_of = |k: &str| k.rsplit_once('|').unwrap().0.to_string(); @@ -160,7 +149,7 @@ fn long_event_splits_across_records_and_reassembles() { let records = diag.records().unwrap(); assert_eq!(records.len(), 1); match &records[0] { - DiagnosticRecord::Event { + DiagnosticRecord::Decoded { event: decoded, chunks, } => { @@ -177,10 +166,8 @@ fn multi_chunk_event_uses_unique_keys_so_host_keeps_all() { let diag = diagnostics(&dir); let message = "z".repeat(MAX_CHUNK_BYTES * 2 + 1); - diag.emit(Level::INFO, "big:event", &message).unwrap(); + diag.emit_event("big:event", &message).unwrap(); - // Three records, no two sharing a key: the Hyper-V host keeps only one - // record per key, so shared keys would silently drop chunks. let dumped = diag.store().dump().unwrap(); assert_eq!(dumped.len(), 3); let total = dumped.len(); @@ -189,7 +176,6 @@ fn multi_chunk_event_uses_unique_keys_so_host_keeps_all() { keys.dedup(); assert_eq!(keys.len(), total, "chunk keys must be unique"); - // The event still reassembles to the full message. let events = diag.events().unwrap(); assert_eq!(events.len(), 1); assert_eq!(events[0].message, message); @@ -200,9 +186,8 @@ fn injected_malformed_key_is_classified() { let dir = TempDir::new().unwrap(); let diag = diagnostics(&dir); - // Six segments but an unrecognized level. diag.store() - .append(&format!("{PREFIX}|100|NOPE|bad:level|{VM_ID}|id"), "junk") + .append(&format!("{PREFIX}|100|{VM_ID}|NOPE|bad:kind|id|ts"), "junk") .unwrap(); let records = diag.records().unwrap(); @@ -218,18 +203,17 @@ fn mixed_records_round_trip_together() { let dir = TempDir::new().unwrap(); let diag = diagnostics(&dir); - diag.emit(Level::INFO, "a:b", "short").unwrap(); - diag.emit(Level::WARN, "c:d", "y".repeat(MAX_CHUNK_BYTES + 5)) + diag.emit_event("a:b", "short").unwrap(); + diag.emit_event("c:d", "y".repeat(MAX_CHUNK_BYTES + 5)) .unwrap(); diag.store() .append("PROVISIONING_REPORT", "result=success") .unwrap(); diag.store() - .append(&format!("{PREFIX}|100|NOPE|e:f|{VM_ID}|id"), "junk") + .append(&format!("{PREFIX}|100|{VM_ID}|NOPE|e:f|id|ts"), "junk") .unwrap(); let records = diag.records().unwrap(); - // Two events + one raw + one malformed. assert_eq!(records.len(), 4); assert_eq!(diag.events().unwrap().len(), 2); } @@ -239,8 +223,8 @@ fn clear_removes_events_but_keeps_raw() { let dir = TempDir::new().unwrap(); let diag = diagnostics(&dir); - diag.emit(Level::INFO, "a:b", "e1").unwrap(); - diag.emit(Level::DEBUG, "c:d", "z".repeat(MAX_CHUNK_BYTES * 2)) + diag.emit_event("a:b", "e1").unwrap(); + diag.emit_event("c:d", "z".repeat(MAX_CHUNK_BYTES * 2)) .unwrap(); diag.store() .append("PROVISIONING_REPORT", "result=success") @@ -262,22 +246,19 @@ fn clear_removes_all_diagnostics_regardless_of_scope() { let dir = TempDir::new().unwrap(); let diag = diagnostics(&dir); - diag.emit(Level::INFO, "a:b", "mine").unwrap(); - // Events from a different agent/VM and a malformed event key are also - // diagnostic keys, so clear() removes them too. + diag.emit_event("a:b", "mine").unwrap(); + diag.store() + .append("other-agent|100|other-vm|event|x:y|id|ts", "theirs") + .unwrap(); diag.store() - .append("other-agent|100|INFO|x:y|other-vm|id", "theirs") + .append("p|100|vm|NOPE|c:d|id|ts", "junk") .unwrap(); - diag.store().append("p|100|NOPE|c:d|vm|id", "junk").unwrap(); - // A chunked malformed event key (its base classifies as malformed) is - // also a diagnostic key, so clear() removes every chunk. diag.store() - .append("p|100|NOPE|c:d|vm|id|0", "junk-0") + .append("p|100|vm|NOPE|c:d|id|ts|0", "junk-0") .unwrap(); diag.store() - .append("p|100|NOPE|c:d|vm|id|1", "junk-1") + .append("p|100|vm|NOPE|c:d|id|ts|1", "junk-1") .unwrap(); - // A raw record survives. diag.store() .append("PROVISIONING_REPORT", "result=success") .unwrap(); @@ -298,9 +279,7 @@ fn emit_rejects_delimiter_in_event_fields() { let dir = TempDir::new().unwrap(); let diag = diagnostics(&dir); - // A pipe in the name would produce an ambiguous seven-segment key. - assert!(diag.emit(Level::INFO, "a|b", "msg").is_err()); - // Nothing was written. + assert!(diag.emit_event("a|b", "msg").is_err()); assert!(diag.store().dump().unwrap().is_empty()); } @@ -311,7 +290,6 @@ fn concurrent_multichunk_emits_reassemble_without_interleaving() { const THREADS: usize = 5; const PER_THREAD: usize = 8; - // Force three chunks per event. let len = MAX_CHUNK_BYTES * 2 + 7; let handles: Vec<_> = (0..THREADS) @@ -321,7 +299,7 @@ fn concurrent_multichunk_emits_reassemble_without_interleaving() { thread::spawn(move || { for _ in 0..PER_THREAD { let message = marker.to_string().repeat(len); - diag.emit(Level::INFO, format!("thread:{marker}"), message) + diag.emit_event(format!("thread:{marker}"), message) .unwrap(); } }) @@ -332,12 +310,8 @@ fn concurrent_multichunk_emits_reassemble_without_interleaving() { } let events = diag.events().unwrap(); - // If any event's chunks had been split by an interleaving writer, the - // key would appear as multiple groups and the count would be wrong. assert_eq!(events.len(), THREADS * PER_THREAD); for event in &events { - // Each message is homogeneous and full length: chunks stayed - // contiguous on disk. assert_eq!(event.message.len(), len); let first = event.message.chars().next().unwrap(); assert!(event.message.chars().all(|c| c == first)); From 785a83c814d940c6ce1650245cb4c0122065bb95 Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Thu, 20 Aug 2026 15:08:38 -0600 Subject: [PATCH 09/32] refactor(kvp): give every diagnostic record a subevent index and tighten rustdoc --- libazureinit-kvp/src/cli.rs | 27 ++---- libazureinit-kvp/src/diagnostics.rs | 133 +++++++++----------------- libazureinit-kvp/src/report.rs | 4 +- libazureinit-kvp/src/store.rs | 8 +- libazureinit-kvp/tests/diagnostics.rs | 4 +- 5 files changed, 62 insertions(+), 114 deletions(-) diff --git a/libazureinit-kvp/src/cli.rs b/libazureinit-kvp/src/cli.rs index 42a45fae..50558293 100644 --- a/libazureinit-kvp/src/cli.rs +++ b/libazureinit-kvp/src/cli.rs @@ -739,26 +739,15 @@ fn resolve_vm_id_with( #[derive(Clone, Debug, PartialEq, Eq)] struct SupportingData(Vec<(String, String)>); -/// Parse a `--supporting-data` argument into its `key=value` pairs. +/// Parse a `--supporting-data` argument into its comma-separated +/// `key=value` pairs. A value wrapped in matching single/double quotes may +/// contain literal commas (the quotes must wrap the whole value and are +/// stripped); empty fields are ignored. /// -/// Fields are comma-separated. A value may be wrapped in matching single or -/// double quotes so it can contain literal commas; the quotes are honored -/// only when they wrap the *entire* value (the opening quote immediately -/// follows `=` and the matching quote ends the field) and are stripped from -/// the stored value. Empty fields (such as a trailing comma) are ignored. -/// -/// Supported (input -> parsed pairs): -/// - `k=v` -> `k`=`v` -/// - `k1=v1,k2=v2` -> `k1`=`v1`, `k2`=`v2` -/// - `k='a,b'` or `k="a,b"` -> `k`=`a,b` (quotes protect the comma) -/// - `k=a'b` -> `k`=`a'b` (a quote not at the value start is literal) -/// - `k=v,` -> `k`=`v` (trailing/empty field ignored) -/// -/// Rejected: -/// - `novalue` -> missing `=` -/// - `=v` -> empty key -/// - `k='a,b` -> unterminated quote -/// - `k='a,b'x` -> characters after a quoted value +/// Supported: `k=v`; `k1=v1,k2=v2`; `k='a,b'` or `k="a,b"` -> `k`=`a,b`; +/// `k=a'b` -> literal quote; `k=v,` -> trailing field ignored. +/// Rejected: `novalue` (no `=`), `=v` (empty key), `k='a,b` +/// (unterminated quote), `k='a,b'x` (chars after a quoted value). fn parse_supporting_data(raw: &str) -> Result { let mut pairs = Vec::new(); for field in split_supporting_data_fields(raw)? { diff --git a/libazureinit-kvp/src/diagnostics.rs b/libazureinit-kvp/src/diagnostics.rs index eb5e0c99..bafe87dd 100644 --- a/libazureinit-kvp/src/diagnostics.rs +++ b/libazureinit-kvp/src/diagnostics.rs @@ -20,11 +20,10 @@ //! and takes everything else from the key, decoding into the same //! [`DiagnosticEvent`]. This crate only *reads* cloud-init. //! - **Chunking**: values longer than [`MAX_CHUNK_BYTES`] are split at -//! UTF-8 codepoint boundaries into multiple records written atomically -//! under a single lock. Each chunk gets a unique key — the event key -//! with a `|` suffix — so the Hyper-V host, which keeps -//! only one record per key, retains every chunk. The chunks are -//! regrouped into one event on read. +//! UTF-8 codepoint boundaries into multiple records under one lock, +//! each keyed with a unique `|` suffix (`0`, `1`, …) +//! since the Hyper-V host keeps only one record per key. Chunks are +//! regrouped on read. //! - **Classification**: [`records`](DiagnosticsKvp::records) sorts every //! stored record into a [`DiagnosticRecord`] — a reassembled //! [`DiagnosticEvent`] (from either agent), an unstructured @@ -136,8 +135,8 @@ fn now_timestamp() -> String { Utc::now().format("%Y-%m-%dT%H:%M:%S%.3fZ").to_string() } -/// Format an azure-init diagnostic event key as its `|`-delimited on-disk -/// string: +/// Format an azure-init diagnostic event's *shared* key as its +/// `|`-delimited on-disk string: /// `||||||`. /// /// `boot_epoch` is the Unix epoch second the system booted (see @@ -336,13 +335,9 @@ fn reject_delimiter(field: &'static str, value: &str) -> Result<(), KvpError> { /// /// Metadata (`agent`, `boot_epoch`, `vm_id`, `kind`, `name`, `event_id`) /// comes from the record key; the payload (`timestamp`, `result`, -/// `duration`, `message`) comes from the value. Optional fields are -/// populated only when the source provides them: azure-init events carry -/// a `timestamp`; cloud-init `finish` records carry a `result` and -/// `duration`. Write azure-init events with -/// [`DiagnosticsKvp::emit_event`]; read events back via -/// [`records`](DiagnosticsKvp::records) / -/// [`events`](DiagnosticsKvp::events). +/// `duration`, `message`) from the value. Optional fields are populated +/// only when the source provides them (e.g. cloud-init `finish` records +/// carry `result` and `duration`). #[derive(Clone, Debug, PartialEq, serde::Serialize)] #[non_exhaustive] pub struct DiagnosticEvent { @@ -415,7 +410,7 @@ pub enum DiagnosticRecord { /// A typed diagnostics view over a [`KvpPoolStore`]. /// /// Owns the `agent` and `vm_id` stamped into this layer's azure-init -/// event keys. See the [module documentation](self) for the on-disk +/// event keys. See the module-level documentation for the on-disk /// format. #[derive(Clone, Debug)] pub struct DiagnosticsKvp { @@ -469,12 +464,12 @@ impl DiagnosticsKvp { /// (stamping `boot_epoch`, `vm_id`, `agent`, and the current /// `timestamp`) and write `value` as its message. /// - /// Messages longer than [`MAX_CHUNK_BYTES`] are split - /// at UTF-8 codepoint boundaries and written as multiple records - /// atomically under a single lock via - /// [`KvpPoolStore::append_multiple`]. Each chunk is keyed with a - /// `|` suffix so every record is unique, and the - /// chunks are regrouped by [`records`](Self::records) on read. + /// The message is written under a single lock via + /// [`KvpPoolStore::append_multiple`]; values longer than + /// [`MAX_CHUNK_BYTES`] are split at UTF-8 codepoint boundaries into + /// multiple records. Every record is keyed with a `|` + /// suffix (`0`, `1`, …) so it is unique, and the chunks are regrouped + /// by [`records`](Self::records) on read. /// /// Returns [`KvpError::EventFieldContainsDelimiter`] if `agent`, /// `vm_id`, `name`, or `event_id` contains the `|` key delimiter. @@ -505,30 +500,20 @@ impl DiagnosticsKvp { self.write_chunked(&key, value) } - /// Split `value` at [`MAX_CHUNK_BYTES`] and append the - /// chunks under `key` in one atomic batch. - /// - /// A single-record value keeps the bare event `key`. A value that - /// spans multiple records gets one record per chunk, each keyed - /// `|` (`0`, `1`, …) so no two records collide — - /// the Hyper-V host keeps only one record per key. [`reassemble`] - /// strips the subevent index to regroup the chunks on read. + /// Split `value` at [`MAX_CHUNK_BYTES`] and append each chunk in one + /// atomic batch, keyed `|` (`0`, `1`, …) so every + /// record is unique. [`reassemble`] strips the index on read. fn write_chunked(&self, key: &str, value: &str) -> Result<(), KvpError> { - let chunks = chunk_at_char_boundary(value, MAX_CHUNK_BYTES); - if chunks.len() == 1 { - return self - .store - .append_multiple(chunks.into_iter().map(|chunk| (key, chunk))); - } - let records: Vec<(String, &str)> = chunks - .into_iter() - .enumerate() - .map(|(subevent_index, chunk)| { - let chunk_key = - format!("{key}{EVENT_KEY_DELIMITER}{subevent_index}"); - (chunk_key, chunk) - }) - .collect(); + let records: Vec<(String, &str)> = + chunk_at_char_boundary(value, MAX_CHUNK_BYTES) + .into_iter() + .enumerate() + .map(|(subevent_index, chunk)| { + let chunk_key = + format!("{key}{EVENT_KEY_DELIMITER}{subevent_index}"); + (chunk_key, chunk) + }) + .collect(); self.store.append_multiple(records) } @@ -583,32 +568,17 @@ impl DiagnosticsKvp { } } -/// The event key a chunk belongs to. -/// -/// [`DiagnosticsKvp::write_chunked`] gives each chunk of a multi-record -/// event a unique key by appending a `|` (cloud-init's -/// term) to the event key, so the Hyper-V host — which keeps only one -/// record per key — retains every chunk. This returns the shared event -/// key used to regroup them on read: for a chunk key -/// `|` it strips the trailing index; any other -/// key (a single-record event, `PROVISIONING_REPORT`, a malformed key, …) -/// is returned unchanged. +/// The shared event key a chunk belongs to: strips a trailing +/// `|`, or returns the key unchanged if it has none. fn base_event_key(key: &str) -> &str { split_subevent_index(key).0 } /// Split a key into its base event key and optional trailing subevent -/// index. When the trailing segment is numeric and the base parses as an -/// event key — a valid azure-init or cloud-init event, or a malformed one -/// (event-shaped but invalid) — returns -/// `(base, Some(index))`; any other key (a single-record event, -/// `PROVISIONING_REPORT`, …) returns `(key, None)`. -/// -/// The subevent index is the same trailing `|` cloud-init and -/// azure-init append to give each chunk a unique key; [`reassemble`] uses -/// it both to regroup an event's chunks and to restore their write order. -/// Malformed keys are included so a chunked malformed event still -/// regroups and is cleared consistently with a single-record one. +/// index: `(base, Some(index))` when a trailing numeric segment follows an +/// event-shaped base (valid or malformed), else `(key, None)`. +/// [`reassemble`] uses the index to regroup an event's chunks and restore +/// their write order. fn split_subevent_index(key: &str) -> (&str, Option) { if let Some((base, index)) = key.rsplit_once(EVENT_KEY_DELIMITER) { if let Ok(index) = index.parse::() { @@ -656,15 +626,10 @@ fn reassemble(dumped: Vec<(String, String)>) -> Vec { records } -/// Turn one reassembled group of chunk values into a -/// [`DiagnosticRecord`]. -/// -/// `chunk_values` holds every record that shared the base event key, -/// already ordered by subevent index by [`reassemble`], and is never -/// empty. azure-init events and raw records concatenate their values -/// directly. cloud-init writes each chunk as a metadata object carrying a -/// slice of the escaped message, so those are stitched back together and -/// decoded (see [`decode_cloud_init_value`]). +/// Classify one reassembled group of chunk values (ordered by subevent +/// index, never empty) into a [`DiagnosticRecord`]. azure-init and raw +/// records concatenate their values; cloud-init chunks are stitched and +/// decoded via [`decode_cloud_init_value`]. fn classify_record(key: String, chunk_values: Vec) -> DiagnosticRecord { let chunks = chunk_values.len(); match classify_key(&key) { @@ -751,20 +716,14 @@ fn classify_record(key: String, chunk_values: Vec) -> DiagnosticRecord { /// Marker preceding a cloud-init value's message field: `"msg":"`. const CLOUD_INIT_MSG_MARKER: &str = "\"msg\":\""; -/// Decode a cloud-init event's chunk value(s) into its reassembled -/// `message` and the JSON object carrying its `ts`/`result`/`duration` -/// metadata, read from an untyped [`serde_json::Value`] so no serde -/// deserialization target is needed. [`classify_record`] maps those -/// fields onto the shared [`DiagnosticEvent`]. +/// Decode a cloud-init event's chunk value(s) into `(metadata, message)`, +/// reading `ts`/`result`/`duration` from an untyped [`serde_json::Value`]. /// -/// A single-record event is a complete JSON object, parsed directly. A -/// multi-record event was split by cloud-init's `_break_down`, which -/// slices the JSON-*escaped* message at character boundaries — so an -/// individual chunk can end mid-escape (e.g. a `\n` split into `\` and -/// `n`) and is not valid JSON on its own. We therefore recover the raw -/// (still-escaped) `msg` slice from each chunk, concatenate the slices in -/// order, and unescape the whole once; the metadata is read from the -/// first chunk's non-`msg` fields. +/// A single record is complete JSON, parsed directly. A multi-record event +/// was split mid-escape by cloud-init's `_break_down` (e.g. a `\n` cut into +/// `\` and `n`), so no chunk is valid JSON alone: recover each chunk's raw +/// escaped `msg` slice, concatenate, and unescape once; metadata comes from +/// the first chunk. fn decode_cloud_init_value( chunks: &[String], ) -> Result<(serde_json::Value, String), String> { diff --git a/libazureinit-kvp/src/report.rs b/libazureinit-kvp/src/report.rs index 98274588..43b3a829 100644 --- a/libazureinit-kvp/src/report.rs +++ b/libazureinit-kvp/src/report.rs @@ -55,9 +55,7 @@ impl std::fmt::Display for ReportResult { /// Pre-provisioning (PPS) type reported in the `pps_type` field. /// /// Mirrors the values cloud-init reports for the platform's -/// `PreprovisionedVMType` / IMDS `ppsType`. Marked `#[non_exhaustive]` -/// so new platform PPS types can be added without breaking downstream -/// `match` statements. +/// `PreprovisionedVMType` / IMDS `ppsType`. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub enum ReportPpsType { /// Not pre-provisioned (`None`). diff --git a/libazureinit-kvp/src/store.rs b/libazureinit-kvp/src/store.rs index 9c66569c..64152a28 100644 --- a/libazureinit-kvp/src/store.rs +++ b/libazureinit-kvp/src/store.rs @@ -133,7 +133,7 @@ impl KvpPoolStore { /// checking for an existing key. /// /// This preserves any existing records, including duplicate keys, - /// and does not enforce [`MAX_UNIQUE_KEYS`]. Use + /// and does not enforce the `MAX_UNIQUE_KEYS` cap. Use /// [`insert`](Self::insert) when callers need upsert semantics. pub fn append(&self, key: &str, value: &str) -> Result<(), KvpError> { validate_key(key, self.mode.max_key_size())?; @@ -153,8 +153,8 @@ impl KvpPoolStore { /// interleaving with the batch. /// /// Existing records are kept and duplicate keys are preserved. - /// Like [`append`](Self::append), this does not enforce - /// [`MAX_UNIQUE_KEYS`]; use [`load`](Self::load) when the + /// Like [`append`](Self::append), this does not enforce the + /// `MAX_UNIQUE_KEYS` cap; use [`load`](Self::load) when the /// caller is replacing the entire pool and wants the unique-key cap /// enforced. Validation happens before the file is opened; empty /// input is a no-op and does not create the pool file. If an I/O @@ -547,7 +547,7 @@ impl KvpPoolStore { /// /// This is the inverse of [`dump`](Self::dump): duplicate keys are /// preserved exactly as provided, but the number of unique keys is - /// capped at [`MAX_UNIQUE_KEYS`]. Existing records are discarded; + /// capped at `MAX_UNIQUE_KEYS`. Existing records are discarded; /// empty input clears the pool. Use /// [`append_multiple`](Self::append_multiple) when callers need to /// extend the pool instead. diff --git a/libazureinit-kvp/tests/diagnostics.rs b/libazureinit-kvp/tests/diagnostics.rs index 2e40bad9..01cd1a5a 100644 --- a/libazureinit-kvp/tests/diagnostics.rs +++ b/libazureinit-kvp/tests/diagnostics.rs @@ -98,7 +98,9 @@ fn short_event_round_trips_as_single_record() { diag.emit_event("user:create_user", "created").unwrap(); - assert_eq!(diag.store().dump().unwrap().len(), 1); + let dumped = diag.store().dump().unwrap(); + assert_eq!(dumped.len(), 1); + assert!(dumped[0].0.ends_with("|0"), "key: {}", dumped[0].0); let records = diag.records().unwrap(); assert_eq!(records.len(), 1); From 720df647a9a3e329ed016110920fa29201d65f80 Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Mon, 24 Aug 2026 14:37:51 -0600 Subject: [PATCH 10/32] feat(kvp): decode cloud-init non-span reporting types (diagnostic, compressed, boot-telemetry, system-info) as RecordKind::Other instead of rejecting them as malformed --- libazureinit-kvp/src/diagnostics.rs | 94 +++++++++---- libazureinit-kvp/tests/diagnostics.rs | 186 +++++++++++++++++++++++++- 2 files changed, 253 insertions(+), 27 deletions(-) diff --git a/libazureinit-kvp/src/diagnostics.rs b/libazureinit-kvp/src/diagnostics.rs index bafe87dd..2f458639 100644 --- a/libazureinit-kvp/src/diagnostics.rs +++ b/libazureinit-kvp/src/diagnostics.rs @@ -89,20 +89,21 @@ pub const MAX_CHUNK_BYTES: usize = 1022; /// Delimiter separating the segments of a diagnostic event key. const EVENT_KEY_DELIMITER: char = '|'; -/// The kind of a diagnostic record: a span boundary (`start`/`finish`) -/// or a point `event`. The `start`/`finish` tokens match cloud-init's, -/// so a span's boundaries read the same whichever agent emitted them. -#[derive(Clone, Copy, Debug, PartialEq, Eq, serde::Serialize)] -#[serde(rename_all = "lowercase")] +/// The kind of a diagnostic record. `start`/`finish`/`event` are shared +/// with azure-init; cloud-init's other reporting types (`diagnostic`, +/// `compressed`, `boot-telemetry`, …) are kept verbatim as +/// [`Other`](RecordKind::Other). +#[derive(Clone, Debug, PartialEq, Eq)] pub enum RecordKind { - /// The opening of a span (a function or stage begins), written - /// `start`. + /// A span opening, written `start`. Start, - /// The closing of a span (a function or stage ends), written - /// `finish`. + /// A span closing, written `finish`. Finish, /// A point-in-time event, written `event`. Event, + /// Any other reporting type, kept verbatim (cloud-init only; + /// azure-init never writes it). + Other(String), } impl std::fmt::Display for RecordKind { @@ -111,12 +112,24 @@ impl std::fmt::Display for RecordKind { Self::Start => "start", Self::Finish => "finish", Self::Event => "event", + Self::Other(token) => token.as_str(), }) } } -/// Parses the on-disk `kind`/`type` token; cloud-init emits only -/// `start`/`finish`. +/// Serializes as the on-disk token (the `Display` form). +impl serde::Serialize for RecordKind { + fn serialize(&self, serializer: S) -> Result + where + S: serde::Serializer, + { + serializer.serialize_str(&self.to_string()) + } +} + +/// Parses an azure-init `kind` token — strictly `start`/`finish`/`event` +/// (a corrupt azure-init kind is rejected). cloud-init's wider `type` +/// space is mapped separately and falls back to [`RecordKind::Other`]. impl std::str::FromStr for RecordKind { type Err = (); @@ -240,12 +253,12 @@ fn classify_key(key: &str) -> KeyClass<'_> { /// Classify a `CLOUD_INIT`-prefixed key into a [`KeyClass::CloudInit`]. /// -/// Handles both the current layout +/// Handles the current layout /// (`CLOUD_INIT|||||`) and the -/// older one that predates the `vm_id` segment -/// (`CLOUD_INIT||||`). Any other segment -/// count is [`KeyClass::Raw`]; a right-shaped key with a non-numeric -/// incarnation or unrecognized type is [`KeyClass::Malformed`]. +/// older one without the `vm_id` segment. A wrong segment count is +/// [`KeyClass::Raw`]; a non-numeric incarnation is +/// [`KeyClass::Malformed`]. Any `type` other than `start`/`finish`/`event` +/// is preserved as [`RecordKind::Other`], not rejected. fn classify_cloud_init_key(key: &str) -> KeyClass<'_> { let mut segments = key.split(EVENT_KEY_DELIMITER); let _prefix = segments.next(); @@ -269,11 +282,10 @@ fn classify_cloud_init_key(key: &str) -> KeyClass<'_> { ), }; }; - let Ok(kind) = event_type.parse::() else { - return KeyClass::Malformed { - reason: format!("unrecognized cloud-init type {event_type:?}"), - }; - }; + // Non-span cloud-init types are kept verbatim, not rejected. + let kind = event_type + .parse::() + .unwrap_or_else(|()| RecordKind::Other(event_type.to_string())); KeyClass::CloudInit { boot_epoch, kind, @@ -834,7 +846,7 @@ mod tests { AGENT, BOOT_EPOCH, VM_ID, - expected, + expected.clone(), "span:event", EVENT_ID, TIMESTAMP, @@ -846,6 +858,22 @@ mod tests { } } + #[rstest] + #[case::start(RecordKind::Start, "start")] + #[case::finish(RecordKind::Finish, "finish")] + #[case::event(RecordKind::Event, "event")] + #[case::other(RecordKind::Other("compressed".to_string()), "compressed")] + fn record_kind_renders_as_its_token( + #[case] kind: RecordKind, + #[case] token: &str, + ) { + assert_eq!(kind.to_string(), token); + assert_eq!( + serde_json::to_value(&kind).unwrap(), + serde_json::json!(token) + ); + } + fn class_of(key: &str) -> &'static str { match classify_key(key) { KeyClass::Event { .. } => "event", @@ -867,9 +895,9 @@ mod tests { #[case::raw_non_numeric_boot_epoch("a|notnum|vm|event|name|id|ts", "raw")] #[case::malformed_bad_kind("a|100|vm|NOTAKIND|name|id|ts", "malformed")] #[case::malformed_other_kind("a|100|vm|nope|name|id|ts", "malformed")] - #[case::cloud_init_bad_type( - "CLOUD_INIT|100|weird|name|vmid|uuid", - "malformed" + #[case::cloud_init_custom_type( + "CLOUD_INIT|100|compressed|name|vmid|uuid", + "cloud-init" )] #[case::cloud_init_non_numeric_incarnation( "CLOUD_INIT|notnum|finish|name|vmid|uuid", @@ -1243,6 +1271,22 @@ mod tests { ); } + #[test] + fn cloud_init_non_span_type_decodes_to_other_kind() { + let key = format!( + "CLOUD_INIT|1785187982|compressed|cloud-init.log|\ + {CLOUD_INIT_VM_ID}|abc12345-1111-2222-3333-444455556666" + ); + let value = r#"{"name":"cloud-init.log","type":"compressed","ts":"2026-07-27T21:33:24.339006+00:00","msg":"payload"}"#; + assert!(matches!( + classify_record(key, vec![value.to_string()]), + DiagnosticRecord::Decoded { event, chunks: 1 } + if event.kind == RecordKind::Other("compressed".to_string()) + && event.agent == "CLOUD_INIT" + && event.message == "payload" + )); + } + #[test] fn cloud_init_key_with_invalid_json_is_malformed() { assert!(matches!( diff --git a/libazureinit-kvp/tests/diagnostics.rs b/libazureinit-kvp/tests/diagnostics.rs index 01cd1a5a..d763e404 100644 --- a/libazureinit-kvp/tests/diagnostics.rs +++ b/libazureinit-kvp/tests/diagnostics.rs @@ -9,9 +9,10 @@ use std::thread; use libazureinit_kvp::{ - DiagnosticRecord, DiagnosticsKvp, KvpPool, KvpPoolStore, PoolMode, - RecordKind, MAX_CHUNK_BYTES, + DiagnosticEvent, DiagnosticRecord, DiagnosticsKvp, KvpPool, KvpPoolStore, + PoolMode, RecordKind, MAX_CHUNK_BYTES, }; +use rstest::rstest; use tempfile::TempDir; const PREFIX: &str = "azure-init-test"; @@ -320,3 +321,184 @@ fn concurrent_multichunk_emits_reassemble_without_interleaving() { assert_eq!(event.name, format!("thread:{first}")); } } + +const COMPRESSED_LOG_CHUNKS: &[(&str, &str)] = &[ + ( + "CLOUD_INIT|1786047606|compressed|cloud-init.log|b7a822ba-4eea-46c0-b559-e84396101132|0", + "{\"name\":\"cloud-init.log\",\"type\":\"compressed\",\"ts\":\"2026-08-06T20:20:13.479078Z\",\"msg_i\":0,\"msg\":\"{\\\"encoding\\\": \\\"gz+b64\\\", \\\"data\\\": \\\"eJxtV0uObDUMnbOKt4GHYjuJbYYIxAoYIQZx7EhIDBjAgN1ziumz1K2uqtY9lWuf3+XB++uwr2N/\\\\n4fEDfsb48vXLP3//8ef3f/37208///jrL7//8OVq2K09s4zekFEn71Lyp2fiBZsU1wxZ5zv+BpAa\\\\nwIzY128A1jhGRMikPJSsPJym6x5ih681gNwA6jzvnvK4U+6dhTdqu/Z4OsaY0y9xutDOkeIZDaw0\\\\nsKTytj1N9x3zVFjMoKnFtGaZNDCzganSNHnj8Hg4Am1eK4fcY2Nn7djuLuvJ5Iq5zVP2PPdWA78a\\\\n+CcyRXBfig14WKnskaNU6UWe1cDsBiZ0LamKNLIJIPyJVDl3nrHyzdh7lvt42F4DqQ3kfPXY2e9h\\\\nMdUyfhqYQDCJvsJoa4VY3KG1j5F3E7UGeC0GE9Xm5p0X5EzeRxYdnG+BCrV0F7Ht7qTeANrZ59Sk\\\\new4YjeU6mIjxSQNAnUZoGU1i2eDc3Ya71KQpo7m8UwTRfc/3oUFKk4c9cPbMoho6nhn+iQ8r1iQn\\\\nrdfAdrrAeeja0gxPp9w3d0FwV+LiVy4YVwMUH9hTp17qVOGQpwHZDs60VwTfqINRsY17Q2l9pmcR\\\\nj+4+qztrp5LYy/KCfmfbXbX4rlBfonb1NEZArRaW1+XDdtRTL0Nm+9LQLaI15t2Yx/tY18WJa3oD\\\\n22mDRtLaV1QGtHAfdvCSyavRKHVKyLev7u34WrOSISCLQQRnPgJ0sBskTw7Ts+hutlOBqN85Rplz\\\\nXuWwj6iyaAhjjIL7P5qqfM9rhEWdDl48ObTXmF44zUznfd+kby/nTgVF70HnA0sQ0h1SL/zZilP1\\\\nsAm4C6UW7\"}", + ), + ( + "CLOUD_INIT|1786047606|compressed|cloud-init.log|b7a822ba-4eea-46c0-b559-e84396101132|1", + "{\"name\":\"cloud-init.log\",\"type\":\"compressed\",\"ts\":\"2026-08-06T20:20:13.479078Z\",\"msg_i\":1,\"msg\":\"ASEz7iBEzbQnUL8DD83hUtijQtvWjYOAogZfCGPdZMbtXKni0Ovzj2g/Ra4G3zZDVx5\\\\niKDbQHQqsHvXgQWB4ONOz21IHWKu3QB0fL8Yx8sNLlgG3Wv78IvN1xdpo0Xu+L6PrwejqVFX59R1\\\\nPgu86zGvSyDdXgQhra3I8Fmy2btddpwvQRbV2dc0RMaathBMrJCmyHIoynHqkQzv9VUb6x7Yazbw\\\\nnSIO3HFlbnWOElSMkncWwntin5KO1C5enuCOn6PdVDtN6COky/gkbjgh6su2R/qD/zFfHU0ccqcE\\\\nZKqmvPWK4Re+IovPxHi9tic+hU0DMO6zSfSpH9zcurRZYR+/NOS9ZwVxIRWLHzoFLB/rM4w469Nn\\\\nZoxN6ZMzpzdZIp1SRsbbG3Vj7YPocHpzudkUxxjxNQZCbIMxLKfAC1AuF0ljZdJpR8ZF3RvTEh0r\\\\nWN8q0Tu6y9tOlSOU5crBpfYxKA6y4ubyTjV8PsG639xy4Smw0oKp66B4oUdjolrFJ48uBFGjsVLp\\\\ndHQwaJ+JsISJokGIvIc6YBAkyLMPPIsnVo6CBKYf7QxVOhU5PDqPwe8hbFsuA2VSRx4nytmAtFqB\\\\n3SEDJbFIWM1aMI6gj82CNzJmV2ilU8dZajljvmFgSJrivtEeAaAo4rhlR/k5DLo1gK1KtvO8+TJj\\\\nfnxHH1wRwvj28tkp4RmthXGj7zAhqdCpBc8Ad+OMoIgQyTiINgGRdeNkzT5np4GHkNCZaFDCmDOC\\\\ntpYzTKFggG6JlhHPPiUykVQNaMf86yAGcgjJUQPlZI3/4xGPGPdsjsazZ6eAWHiUmugR8JI9HKx6\\\\n6NkPhfPDPopEZ1eFMTZbnZ0m8IQDlaPl/S8IG4GF8MJqcOD/AFeindw=\\\"}", + ), + ( + "CLOUD_INIT|1786047606|compressed|cloud-init.log|b7a822ba-4eea-46c0-b559-e84396101132|2", + "{\"name\":\"cloud-init.log\",\"type\":\"compressed\",\"ts\":\"2026-08-06T20:20:13.479078Z\",\"msg_i\":2,\"msg\":\"\\n\\\"}\"}", + ), +]; + +/// The reassembled `msg` across all three chunks. +const EXPECTED_COMPRESSED_MSG: &str = "{\"encoding\": \"gz+b64\", \"data\": \"eJxtV0uObDUMnbOKt4GHYjuJbYYIxAoYIQZx7EhIDBjAgN1ziumz1K2uqtY9lWuf3+XB++uwr2N/\\n4fEDfsb48vXLP3//8ef3f/37208///jrL7//8OVq2K09s4zekFEn71Lyp2fiBZsU1wxZ5zv+BpAa\\nwIzY128A1jhGRMikPJSsPJym6x5ih681gNwA6jzvnvK4U+6dhTdqu/Z4OsaY0y9xutDOkeIZDaw0\\nsKTytj1N9x3zVFjMoKnFtGaZNDCzganSNHnj8Hg4Am1eK4fcY2Nn7djuLuvJ5Iq5zVP2PPdWA78a\\n+CcyRXBfig14WKnskaNU6UWe1cDsBiZ0LamKNLIJIPyJVDl3nrHyzdh7lvt42F4DqQ3kfPXY2e9h\\nMdUyfhqYQDCJvsJoa4VY3KG1j5F3E7UGeC0GE9Xm5p0X5EzeRxYdnG+BCrV0F7Ht7qTeANrZ59Sk\\new4YjeU6mIjxSQNAnUZoGU1i2eDc3Ya71KQpo7m8UwTRfc/3oUFKk4c9cPbMoho6nhn+iQ8r1iQn\\nrdfAdrrAeeja0gxPp9w3d0FwV+LiVy4YVwMUH9hTp17qVOGQpwHZDs60VwTfqINRsY17Q2l9pmcR\\nj+4+qztrp5LYy/KCfmfbXbX4rlBfonb1NEZArRaW1+XDdtRTL0Nm+9LQLaI15t2Yx/tY18WJa3oD\\n22mDRtLaV1QGtHAfdvCSyavRKHVKyLev7u34WrOSISCLQQRnPgJ0sBskTw7Ts+hutlOBqN85Rplz\\nXuWwj6iyaAhjjIL7P5qqfM9rhEWdDl48ObTXmF44zUznfd+kby/nTgVF70HnA0sQ0h1SL/zZilP1\\nsAm4C6UW7ASEz7iBEzbQnUL8DD83hUtijQtvWjYOAogZfCGPdZMbtXKni0Ovzj2g/Ra4G3zZDVx5\\niKDbQHQqsHvXgQWB4ONOz21IHWKu3QB0fL8Yx8sNLlgG3Wv78IvN1xdpo0Xu+L6PrwejqVFX59R1\\nPgu86zGvSyDdXgQhra3I8Fmy2btddpwvQRbV2dc0RMaathBMrJCmyHIoynHqkQzv9VUb6x7Yazbw\\nnSIO3HFlbnWOElSMkncWwntin5KO1C5enuCOn6PdVDtN6COky/gkbjgh6su2R/qD/zFfHU0ccqcE\\nZKqmvPWK4Re+IovPxHi9tic+hU0DMO6zSfSpH9zcurRZYR+/NOS9ZwVxIRWLHzoFLB/rM4w469Nn\\nZoxN6ZMzpzdZIp1SRsbbG3Vj7YPocHpzudkUxxjxNQZCbIMxLKfAC1AuF0ljZdJpR8ZF3RvTEh0r\\nWN8q0Tu6y9tOlSOU5crBpfYxKA6y4ubyTjV8PsG639xy4Smw0oKp66B4oUdjolrFJ48uBFGjsVLp\\ndHQwaJ+JsISJokGIvIc6YBAkyLMPPIsnVo6CBKYf7QxVOhU5PDqPwe8hbFsuA2VSRx4nytmAtFqB\\n3SEDJbFIWM1aMI6gj82CNzJmV2ilU8dZajljvmFgSJrivtEeAaAo4rhlR/k5DLo1gK1KtvO8+TJj\\nfnxHH1wRwvj28tkp4RmthXGj7zAhqdCpBc8Ad+OMoIgQyTiINgGRdeNkzT5np4GHkNCZaFDCmDOC\\ntpYzTKFggG6JlhHPPiUykVQNaMf86yAGcgjJUQPlZI3/4xGPGPdsjsazZ6eAWHiUmugR8JI9HKx6\\n6NkPhfPDPopEZ1eFMTZbnZ0m8IQDlaPl/S8IG4GF8MJqcOD/AFeindw=\\n\"}"; + +const CLOUD_INIT_VM_ID: &str = "0e5e179d-5341-478b-8456-fbb90621bdf8"; + +/// Insert a `vm_id` segment after `name`, turning an old-format key into +/// the current layout (any trailing chunk index is preserved): +/// `CLOUD_INIT|inc|type|name|uuid[|i]` +/// -> `CLOUD_INIT|inc|type|name|vm_id|uuid[|i]`. +fn with_vm_id(old_key: &str, vm_id: &str) -> String { + let mut segments: Vec<&str> = old_key.split('|').collect(); + segments.insert(4, vm_id); // 0:CLOUD_INIT 1:inc 2:type 3:name | vm_id + segments.join("|") +} + +fn without_vm_id(current_key: &str) -> String { + let mut segments: Vec<&str> = current_key.split('|').collect(); + segments.remove(4); + segments.join("|") +} + +/// Append the given records to a fresh guest pool and classify them. +fn records_of, V: AsRef>( + pairs: &[(K, V)], +) -> Vec { + let dir = TempDir::new().unwrap(); + let store = + KvpPoolStore::new_in(KvpPool::Guest, dir.path(), PoolMode::Safe) + .unwrap(); + for (key, value) in pairs { + store.append(key.as_ref(), value.as_ref()).unwrap(); + } + DiagnosticsKvp::new(store, "", "").records().unwrap() +} + +/// Expect exactly one decoded event, returning it with its chunk count. +fn decode_single(records: Vec) -> (DiagnosticEvent, usize) { + assert_eq!(records.len(), 1, "expected one record, got: {records:?}"); + match records.into_iter().next().unwrap() { + DiagnosticRecord::Decoded { event, chunks } => (event, chunks), + other => panic!("expected a decoded event, got: {other:?}"), + } +} + +#[rstest] +#[case::start("start", "azure-ds", RecordKind::Start)] +#[case::finish("finish", "azure-ds/get-metadata", RecordKind::Finish)] +#[case::event("event", "user:create_user", RecordKind::Event)] +#[case::diagnostic( + "diagnostic", + "diagnostic message", + RecordKind::Other("diagnostic".to_string()) +)] +#[case::compressed( + "compressed", + "cloud-init.log", + RecordKind::Other("compressed".to_string()) +)] +#[case::boot_telemetry( + "boot-telemetry", + "boot-telemetry", + RecordKind::Other("boot-telemetry".to_string()) +)] +#[case::system_info( + "system-info", + "system information", + RecordKind::Other("system-info".to_string()) +)] +fn cloud_init_type_decodes_in_both_layouts( + #[case] event_type: &str, + #[case] name: &str, + #[case] expected: RecordKind, +) { + const TS: &str = "2026-08-06T20:20:13.479078Z"; + const UUID: &str = "b7a822ba-4eea-46c0-b559-e84396101132"; + let msg = format!("payload for {event_type}"); + let value = format!( + "{{\"name\":\"{name}\",\"type\":\"{event_type}\",\ + \"ts\":\"{TS}\",\"msg\":\"{msg}\"}}" + ); + let old_key = format!("CLOUD_INIT|1786047606|{event_type}|{name}|{UUID}"); + let current_key = with_vm_id(&old_key, CLOUD_INIT_VM_ID); + + let (event, chunks) = decode_single(records_of(&[(&old_key, &value)])); + assert_eq!(chunks, 1); + assert_eq!(event.agent, "CLOUD_INIT"); + assert_eq!(event.kind, expected); + assert_eq!(event.vm_id, None); + assert_eq!(event.name, name); + assert_eq!(event.message, msg); + + let (event, _) = decode_single(records_of(&[(¤t_key, &value)])); + assert_eq!(event.kind, expected); + assert_eq!(event.vm_id.as_deref(), Some(CLOUD_INIT_VM_ID)); + assert_eq!(event.message, msg); +} + +#[test] +fn cloud_init_finish_reports_result_and_duration_in_both_layouts() { + let value = "{\"name\":\"azure-ds/get-metadata\",\"type\":\"finish\",\ + \"ts\":\"2026-08-06T20:20:13.400000Z\",\"result\":\"SUCCESS\",\ + \"duration\":0.1234,\"msg\":\"finished\"}"; + let old_key = "CLOUD_INIT|1786047606|finish|azure-ds/get-metadata|\ + b7a822ba-4eea-46c0-b559-e84396101132"; + + for key in [old_key.to_string(), with_vm_id(old_key, CLOUD_INIT_VM_ID)] { + let (event, _) = decode_single(records_of(&[(key.as_str(), value)])); + assert_eq!(event.kind, RecordKind::Finish); + assert_eq!(event.result.as_deref(), Some("SUCCESS")); + assert_eq!(event.duration, Some(0.1234)); + } +} + +#[test] +fn real_cloud_init_samples_decode_without_vm_id_too() { + for &(key, value) in CLOUD_INIT_RECORDS { + let (event, _) = + decode_single(records_of(&[(without_vm_id(key), value)])); + assert!(event.vm_id.is_none(), "stripped sample kept a vm_id: {key}"); + } + let (event, _) = decode_single(records_of(&[CLOUD_INIT_RECORDS[0]])); + assert_eq!(event.vm_id.as_deref(), Some(CLOUD_INIT_VM_ID)); +} + +#[test] +fn old_compressed_log_reassembles_across_chunks() { + let (event, chunks) = decode_single(records_of(COMPRESSED_LOG_CHUNKS)); + assert_eq!(chunks, 3, "the three chunks must regroup into one event"); + assert_eq!(event.kind, RecordKind::Other("compressed".to_string())); + assert_eq!(event.vm_id, None); + assert_eq!(event.name, "cloud-init.log"); + assert_eq!(event.message, EXPECTED_COMPRESSED_MSG); +} + +#[test] +fn current_compressed_log_reassembles_across_chunks() { + let current: Vec<(String, &str)> = COMPRESSED_LOG_CHUNKS + .iter() + .map(|&(key, value)| (with_vm_id(key, CLOUD_INIT_VM_ID), value)) + .collect(); + let (event, chunks) = decode_single(records_of(¤t)); + assert_eq!(chunks, 3); + assert_eq!(event.kind, RecordKind::Other("compressed".to_string())); + assert_eq!(event.vm_id.as_deref(), Some(CLOUD_INIT_VM_ID)); + assert_eq!(event.message, EXPECTED_COMPRESSED_MSG); +} + +#[test] +fn cloud_init_event_with_invalid_json_is_still_flagged() { + let key = format!( + "CLOUD_INIT|1786047606|compressed|cloud-init.log|{CLOUD_INIT_VM_ID}|\ + b7a822ba-4eea-46c0-b559-e84396101132" + ); + let records = records_of(&[(key.as_str(), "not-json")]); + assert_eq!(records.len(), 1); + assert!( + matches!( + &records[0], + DiagnosticRecord::Malformed { reason, .. } + if reason.contains("invalid cloud-init JSON") + ), + "expected a Malformed record, got: {:?}", + records[0] + ); +} From 0fd7e0c4c94766a3931d9aaab0a268f3ddc7c4d7 Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Mon, 31 Aug 2026 17:31:47 -0700 Subject: [PATCH 11/32] Refactor KVP diagnostics into a typed, normalized azure-init/cloud-init interface. Simplify parsed output, add tracing-ready emission, and harden indexed chunk reassembly and validation. --- libazureinit-kvp/Cargo.toml | 2 +- libazureinit-kvp/src/cli.rs | 159 +-- libazureinit-kvp/src/diagnostics.rs | 1623 +++++++------------------ libazureinit-kvp/src/error.rs | 2 +- libazureinit-kvp/src/lib.rs | 7 +- libazureinit-kvp/tests/cli.rs | 151 ++- libazureinit-kvp/tests/diagnostics.rs | 359 +++--- 7 files changed, 746 insertions(+), 1557 deletions(-) diff --git a/libazureinit-kvp/Cargo.toml b/libazureinit-kvp/Cargo.toml index 3c0381cd..1e994906 100644 --- a/libazureinit-kvp/Cargo.toml +++ b/libazureinit-kvp/Cargo.toml @@ -9,7 +9,7 @@ license = "MIT" description = "Hyper-V KVP (Key-Value Pair) storage library for azure-init." [dependencies] -chrono = { version = "0.4", default-features = false, features = ["clock", "std"] } +chrono = { version = "0.4", default-features = false, features = ["clock", "serde", "std"] } clap = { version = "4.5.21", features = ["derive"] } csv = "1" libc = "0.2" diff --git a/libazureinit-kvp/src/cli.rs b/libazureinit-kvp/src/cli.rs index 50558293..4941f4ad 100644 --- a/libazureinit-kvp/src/cli.rs +++ b/libazureinit-kvp/src/cli.rs @@ -11,8 +11,8 @@ use clap::{Parser, Subcommand, ValueEnum}; use serde_json::json; use crate::{ - write_report, DiagnosticEvent, DiagnosticRecord, DiagnosticsKvp, KvpError, - KvpPool, KvpPoolStore, PoolMode, ProvisioningReport, ReportPpsType, + write_report, DiagnosticEvent, DiagnosticsKvp, KvpError, KvpPool, + KvpPoolStore, PoolMode, ProvisioningReport, ReportPpsType, }; const EXIT_OK: u8 = 0; @@ -92,26 +92,15 @@ enum Command { /// Print every record in insertion order as KEY=VALUE lines. /// /// With --parse-diagnostics, reassemble chunked diagnostic events and - /// decode each record instead of printing raw KEY=VALUE lines. + /// print only decodable diagnostics instead of raw KEY=VALUE lines. Dump { - /// Reassemble chunked diagnostic events and decode each record as - /// an azure-init event, cloud-init event, raw, or malformed entry. + /// Normalize decodable azure-init and cloud-init diagnostics. #[arg(long)] parse_diagnostics: bool, - /// Also print raw (non-event) records such as PROVISIONING_REPORT. - /// Only applies to the unfiltered view; --name/--tail produce an - /// azure-init events-only view where raw records never appear. - #[arg( - long, - requires = "parse_diagnostics", - conflicts_with_all = ["name", "tail"] - )] - include_raw: bool, - /// Only show azure-init events whose name contains this - /// substring. + /// Only show diagnostics whose name contains this substring. #[arg(long, requires = "parse_diagnostics")] name: Option, - /// Print only the last COUNT azure-init events (default 20 when + /// Print only the last COUNT diagnostics (default 20 when /// COUNT is omitted). #[arg( short = 'n', @@ -170,16 +159,11 @@ enum Command { #[arg(required = true)] keys: Vec, }, - /// Clear the pool. Pass --if-stale to clear only when stale, or - /// --diagnostics to remove only diagnostic event keys. + /// Clear the pool. Pass --if-stale to clear only when stale. Clear { /// Only clear if the store is currently stale. - #[arg(long = "if-stale", conflicts_with = "diagnostics")] + #[arg(long = "if-stale")] if_stale: bool, - /// Remove every diagnostic event key (valid or malformed), - /// leaving raw records such as PROVISIONING_REPORT intact. - #[arg(long)] - diagnostics: bool, }, /// Print whether the pool is stale (exit 0 if stale, 1 otherwise). IsStale, @@ -278,15 +262,11 @@ fn dispatch(cli: Cli, stdout: &mut W) -> Result { Command::Info => info(&store, stdout, output), Command::Dump { parse_diagnostics, - include_raw, name, tail, } => { - let parse = parse_diagnostics.then_some(ParseDiagnosticsArgs { - include_raw, - name, - tail, - }); + let parse = parse_diagnostics + .then_some(ParseDiagnosticsArgs { name, tail }); dump(&store, stdout, parse, output) } Command::Entries => entries(&store, stdout, output), @@ -311,13 +291,8 @@ fn dispatch(cli: Cli, stdout: &mut W) -> Result { Command::DeleteMultiple { keys } => { delete_multiple(&store, stdout, keys, output) } - Command::Clear { - if_stale, - diagnostics, - } => { - if diagnostics { - DiagnosticsKvp::new(store.clone(), "", "").clear()?; - } else if if_stale { + Command::Clear { if_stale } => { + if if_stale { store.clear_if_stale()?; } else { store.clear()?; @@ -389,7 +364,6 @@ fn info( Ok(EXIT_OK) } struct ParseDiagnosticsArgs { - include_raw: bool, name: Option, tail: Option, } @@ -401,12 +375,9 @@ fn dump( output: OutputMode, ) -> Result { if let Some(parse) = parse { - if parse.name.is_some() || parse.tail.is_some() { - return diagnostics_events( - store, stdout, parse.name, parse.tail, output, - ); - } - return diagnostics_records(store, stdout, parse.include_raw, output); + return diagnostics_entries( + store, stdout, parse.name, parse.tail, output, + ); } let records = store.dump()?; @@ -517,65 +488,15 @@ fn is_stale( Ok(if stale { EXIT_OK } else { EXIT_NOT_FOUND }) } -fn diagnostics_records( - store: &KvpPoolStore, - stdout: &mut W, - include_raw: bool, - output: OutputMode, -) -> Result { - let diagnostics = DiagnosticsKvp::new(store.clone(), "", ""); - let records: Vec<_> = diagnostics - .records()? - .into_iter() - .filter(|record| { - include_raw || !matches!(record, DiagnosticRecord::Raw { .. }) - }) - .collect(); - - match output { - OutputMode::Text => { - for record in &records { - let line = match record { - DiagnosticRecord::Decoded { event, chunks } => { - let mut line = diagnostics_event_text(event); - let _ = write!( - line, - " chunks={chunks} message={}", - event.message - ); - line - } - DiagnosticRecord::Raw { key, value } => { - format!("raw key={key} value={value}") - } - DiagnosticRecord::Malformed { key, value, reason } => { - format!( - "malformed key={key} reason={reason} \ - value={value}" - ) - } - }; - writeln!(stdout, "{line}")?; - } - } - OutputMode::Json => { - let array: Vec<_> = - records.iter().map(diagnostics_record_json).collect(); - writeln_json(stdout, &serde_json::Value::Array(array))?; - } - } - Ok(EXIT_OK) -} - -fn diagnostics_events( +fn diagnostics_entries( store: &KvpPoolStore, stdout: &mut W, name: Option, tail: Option, output: OutputMode, ) -> Result { - let diagnostics = DiagnosticsKvp::new(store.clone(), "", ""); - let mut events = diagnostics.events()?; + let diagnostics = DiagnosticsKvp::new(store.clone(), "", "")?; + let mut events = diagnostics.entries()?; if let Some(needle) = name.as_deref() { events.retain(|event| event.name.contains(needle)); @@ -613,9 +534,7 @@ fn diagnostics_event_text(event: &DiagnosticEvent) -> String { let _ = write!(line, " vm_id={vm_id}"); } let _ = write!(line, " name={} event_id={}", event.name, event.event_id); - if let Some(ts) = &event.timestamp { - let _ = write!(line, " timestamp={ts}"); - } + let _ = write!(line, " timestamp={}", event.timestamp); if let Some(result) = &event.result { let _ = write!(line, " result={result}"); } @@ -625,33 +544,8 @@ fn diagnostics_event_text(event: &DiagnosticEvent) -> String { line } -/// Render a [`DiagnosticRecord`] as a JSON object. -fn diagnostics_record_json(record: &DiagnosticRecord) -> serde_json::Value { - match record { - DiagnosticRecord::Decoded { event, chunks } => { - let mut value = diagnostics_event_json(event); - if let serde_json::Value::Object(map) = &mut value { - map.insert("record".to_string(), json!("event")); - map.insert("chunks".to_string(), json!(chunks)); - } - value - } - DiagnosticRecord::Raw { key, value } => json!({ - "record": "raw", - "key": key, - "value": value, - }), - DiagnosticRecord::Malformed { key, value, reason } => json!({ - "record": "malformed", - "key": key, - "value": value, - "reason": reason, - }), - } -} - -/// Render a [`DiagnosticEvent`] as a JSON object (without chunk count), -/// omitting optional fields the source did not provide. +/// Render a [`DiagnosticEvent`] as JSON, omitting optional fields the source +/// did not provide. fn diagnostics_event_json(event: &DiagnosticEvent) -> serde_json::Value { serde_json::to_value(event) .expect("DiagnosticEvent always serializes to a JSON object") @@ -667,7 +561,7 @@ fn emit( ) -> Result { let vm_id = resolve_vm_id(vm_id)?; let prefix = prefix.unwrap_or_else(|| DEFAULT_AGENT.to_string()); - let diagnostics = DiagnosticsKvp::new(store.clone(), vm_id, prefix); + let diagnostics = DiagnosticsKvp::new(store.clone(), vm_id, prefix)?; diagnostics.emit_event(name, message)?; Ok(EXIT_OK) } @@ -1011,7 +905,6 @@ mod tests { fn dump_cmd() -> Command { Command::Dump { parse_diagnostics: false, - include_raw: false, name: None, tail: None, } @@ -1654,13 +1547,7 @@ mod tests { let dir = TempDir::new().unwrap(); store_at(&dir).insert("k", "v").unwrap(); - let (code, _) = run_dispatch(cli( - &dir, - Command::Clear { - if_stale, - diagnostics: false, - }, - )); + let (code, _) = run_dispatch(cli(&dir, Command::Clear { if_stale })); assert_eq!(code, EXIT_OK); assert_eq!(store_at(&dir).is_empty().unwrap(), expect_empty_after); } diff --git a/libazureinit-kvp/src/diagnostics.rs b/libazureinit-kvp/src/diagnostics.rs index 2f458639..5e8ef3e8 100644 --- a/libazureinit-kvp/src/diagnostics.rs +++ b/libazureinit-kvp/src/diagnostics.rs @@ -1,124 +1,77 @@ // Copyright (c) Microsoft Corporation. // Licensed under the MIT License. -//! Typed diagnostics layer over the raw -//! [`KvpPoolStore`](crate::KvpPoolStore) key/value API. +//! Typed diagnostics over the raw [`KvpPoolStore`](crate::KvpPoolStore). //! -//! Where [`KvpPoolStore`](crate::KvpPoolStore) treats keys and values as -//! opaque bytes, [`DiagnosticsKvp`] understands the telemetry conventions -//! azure-init writes into the guest pool and decodes cloud-init's -//! reporting entries into the same [`DiagnosticEvent`] shape. +//! [`DiagnosticsKvp`] writes azure-init diagnostics and reads diagnostics from +//! both azure-init and cloud-init into one [`DiagnosticEvent`] schema. Pool +//! records that are unrelated to diagnostics or cannot be decoded are skipped; +//! callers that need a lossless view can use [`KvpPoolStore::dump`]. //! -//! - **azure-init keys** encode metadata as a seven-segment, -//! pipe-delimited string -//! (`||||||`). -//! The value is the record's message string, stored verbatim for -//! every kind. -//! - **cloud-init keys** -//! (`CLOUD_INIT||||[|]`) store a -//! JSON value; the reader pulls `ts`/`result`/`duration`/`msg` from it -//! and takes everything else from the key, decoding into the same -//! [`DiagnosticEvent`]. This crate only *reads* cloud-init. -//! - **Chunking**: values longer than [`MAX_CHUNK_BYTES`] are split at -//! UTF-8 codepoint boundaries into multiple records under one lock, -//! each keyed with a unique `|` suffix (`0`, `1`, …) -//! since the Hyper-V host keeps only one record per key. Chunks are -//! regrouped on read. -//! - **Classification**: [`records`](DiagnosticsKvp::records) sorts every -//! stored record into a [`DiagnosticRecord`] — a reassembled -//! [`DiagnosticEvent`] (from either agent), an unstructured -//! [`Raw`](DiagnosticRecord::Raw) record such as `PROVISIONING_REPORT`, -//! or a [`Malformed`](DiagnosticRecord::Malformed) event key. +//! Azure-init keys use this format: +//! `||||||`. +//! Every value chunk has a zero-based `|` suffix. //! -//! This module is policy only: all locking, size enforcement, and -//! on-disk encoding stay in [`KvpPoolStore`](crate::KvpPoolStore). -//! -//! # Example -//! -//! ``` -//! use libazureinit_kvp::{ -//! DiagnosticsKvp, KvpPool, KvpPoolStore, PoolMode, MAX_CHUNK_BYTES, -//! }; -//! -//! # fn main() -> Result<(), libazureinit_kvp::KvpError> { -//! let dir = std::env::temp_dir() -//! .join(format!("libazureinit-kvp-doc-{}", std::process::id())); -//! std::fs::create_dir_all(&dir)?; -//! let store = KvpPoolStore::new_in(KvpPool::Guest, &dir, PoolMode::Safe)?; -//! store.clear()?; -//! -//! let diagnostics = -//! DiagnosticsKvp::new(store, "vm-1234", "azure-init-doc"); -//! -//! // A short event lands in a single record. -//! diagnostics.emit_event("user:create_user", "Creating user azureuser")?; -//! -//! // A long message is split across records and reassembled on read. -//! let long = "x".repeat(MAX_CHUNK_BYTES * 2 + 10); -//! diagnostics.emit_event("config:dump", &long)?; -//! -//! let events = diagnostics.events()?; -//! assert_eq!(events.len(), 2); -//! assert_eq!(events[1].message.len(), MAX_CHUNK_BYTES * 2 + 10); -//! -//! # std::fs::remove_dir_all(&dir).ok(); -//! # Ok(()) -//! # } -//! ``` +//! Cloud-init keys use +//! `CLOUD_INIT||||[|]`, with the same +//! numeric suffix when chunked. A cloud-init chunk also stores its index in +//! the JSON `msg_i` field. The reader validates both indices before combining +//! the escaped `msg` fragments. + +use std::collections::HashMap; -use chrono::Utc; +use chrono::{DateTime, SecondsFormat, Utc}; use uuid::Uuid; use crate::{KvpError, KvpPoolStore}; -/// Literal prefix identifying a cloud-init reporting KVP key. const CLOUD_INIT_PREFIX: &str = "CLOUD_INIT"; +const EVENT_KEY_DELIMITER: char = '|'; +const CLOUD_INIT_MSG_MARKER: &str = "\"msg\":\""; -/// Maximum number of value bytes per diagnostic KVP record. +/// Maximum number of UTF-8 value bytes stored in one diagnostic record. /// -/// [`DiagnosticsKvp::emit_event`] splits messages longer than this into -/// multiple records, regardless of the store's -/// [`PoolMode`](crate::PoolMode). It is the conservative -/// [`Safe`](crate::PoolMode::Safe) limit (2 bytes under the Linux kernel -/// `HV_KVP_EXCHANGE_MAX_VALUE` maximum), so diagnostic records stay -/// readable by the Hyper-V host even on an -/// [`Unsafe`](crate::PoolMode::Unsafe) store — its larger capacity is -/// deliberately not used for diagnostics. +/// This conservative limit keeps records readable through the Hyper-V host +/// path. Longer messages are split at UTF-8 character boundaries. pub const MAX_CHUNK_BYTES: usize = 1022; -/// Delimiter separating the segments of a diagnostic event key. -const EVENT_KEY_DELIMITER: char = '|'; - -/// The kind of a diagnostic record. `start`/`finish`/`event` are shared -/// with azure-init; cloud-init's other reporting types (`diagnostic`, -/// `compressed`, `boot-telemetry`, …) are kept verbatim as -/// [`Other`](RecordKind::Other). +/// The lifecycle or reporting kind of a diagnostic entry. #[derive(Clone, Debug, PartialEq, Eq)] -pub enum RecordKind { - /// A span opening, written `start`. +pub enum DiagnosticKind { + /// A span opening. Start, - /// A span closing, written `finish`. + /// A span closing. Finish, - /// A point-in-time event, written `event`. + /// A point-in-time event. Event, - /// Any other reporting type, kept verbatim (cloud-init only; - /// azure-init never writes it). + /// Another source-specific reporting type, such as cloud-init's + /// `compressed` or `system-info`. Other(String), } -impl std::fmt::Display for RecordKind { +impl DiagnosticKind { + fn from_token(token: &str) -> Self { + match token { + "start" => Self::Start, + "finish" => Self::Finish, + "event" => Self::Event, + other => Self::Other(other.to_string()), + } + } +} + +impl std::fmt::Display for DiagnosticKind { fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { f.write_str(match self { Self::Start => "start", Self::Finish => "finish", Self::Event => "event", - Self::Other(token) => token.as_str(), + Self::Other(token) => token, }) } } -/// Serializes as the on-disk token (the `Display` form). -impl serde::Serialize for RecordKind { +impl serde::Serialize for DiagnosticKind { fn serialize(&self, serializer: S) -> Result where S: serde::Serializer, @@ -127,1239 +80,587 @@ impl serde::Serialize for RecordKind { } } -/// Parses an azure-init `kind` token — strictly `start`/`finish`/`event` -/// (a corrupt azure-init kind is rejected). cloud-init's wider `type` -/// space is mapped separately and falls back to [`RecordKind::Other`]. -impl std::str::FromStr for RecordKind { - type Err = (); - - fn from_str(token: &str) -> Result { - match token { - "start" => Ok(Self::Start), - "finish" => Ok(Self::Finish), - "event" => Ok(Self::Event), - _ => Err(()), - } - } -} - -/// The current time as an ISO-8601 UTC timestamp (millisecond precision). -fn now_timestamp() -> String { - Utc::now().format("%Y-%m-%dT%H:%M:%S%.3fZ").to_string() -} - -/// Format an azure-init diagnostic event's *shared* key as its -/// `|`-delimited on-disk string: -/// `||||||`. -/// -/// `boot_epoch` is the Unix epoch second the system booted (see -/// [`KvpPoolStore::boot_epoch`](crate::KvpPoolStore::boot_epoch)); it sits -/// in the same slot as cloud-init's incarnation. `kind` records the -/// span/event shape. [`classify_key`] is the inverse. For example: -/// -/// ```text -/// azure-init-0.1.0|1785187982|3f2504e0-...|event|user:create_user|8f3e9c4a-...|2026-07-27T21:33:24.300Z -/// ``` -fn format_event_key( - agent: &str, - boot_epoch: i64, - vm_id: &str, - kind: RecordKind, - name: &str, - event_id: &str, - timestamp: &str, -) -> String { - let d = EVENT_KEY_DELIMITER; - format!( - "{agent}{d}{boot_epoch}{d}{vm_id}{d}{kind}{d}{name}{d}{event_id}\ - {d}{timestamp}" - ) -} -enum KeyClass<'a> { - Event { - agent: &'a str, - boot_epoch: i64, - vm_id: &'a str, - kind: RecordKind, - name: &'a str, - event_id: &'a str, - timestamp: &'a str, - }, - /// The key is a well-formed cloud-init reporting event key - /// (`CLOUD_INIT||||[|]`). - CloudInit { - boot_epoch: i64, - kind: RecordKind, - name: &'a str, - vm_id: Option<&'a str>, - uuid: &'a str, - }, - Malformed { - reason: String, - }, - Raw, -} - -/// Classify a raw pool key. -fn classify_key(key: &str) -> KeyClass<'_> { - if key.split(EVENT_KEY_DELIMITER).next() == Some(CLOUD_INIT_PREFIX) { - return classify_cloud_init_key(key); - } - - let mut segments = key.split(EVENT_KEY_DELIMITER); - let ( - Some(agent), - Some(boot_epoch), - Some(vm_id), - Some(kind), - Some(name), - Some(event_id), - Some(timestamp), - ) = ( - segments.next(), - segments.next(), - segments.next(), - segments.next(), - segments.next(), - segments.next(), - segments.next(), - ) - else { - return KeyClass::Raw; - }; - if segments.next().is_some() { - return KeyClass::Raw; - } - - let Ok(boot_epoch) = boot_epoch.parse::() else { - return KeyClass::Raw; - }; - - match kind.parse::() { - Ok(kind) => KeyClass::Event { - agent, - boot_epoch, - vm_id, - kind, - name, - event_id, - timestamp, - }, - Err(()) => KeyClass::Malformed { - reason: format!("unrecognized kind {kind:?}"), - }, - } -} - -/// Classify a `CLOUD_INIT`-prefixed key into a [`KeyClass::CloudInit`]. -/// -/// Handles the current layout -/// (`CLOUD_INIT|||||`) and the -/// older one without the `vm_id` segment. A wrong segment count is -/// [`KeyClass::Raw`]; a non-numeric incarnation is -/// [`KeyClass::Malformed`]. Any `type` other than `start`/`finish`/`event` -/// is preserved as [`RecordKind::Other`], not rejected. -fn classify_cloud_init_key(key: &str) -> KeyClass<'_> { - let mut segments = key.split(EVENT_KEY_DELIMITER); - let _prefix = segments.next(); - let (Some(incarnation), Some(event_type), Some(name), Some(fourth)) = ( - segments.next(), - segments.next(), - segments.next(), - segments.next(), - ) else { - return KeyClass::Raw; - }; - let (vm_id, uuid) = match (segments.next(), segments.next()) { - (None, None) => (None, fourth), - (Some(uuid), None) => (Some(fourth), uuid), - _ => return KeyClass::Raw, - }; - let Ok(boot_epoch) = incarnation.parse::() else { - return KeyClass::Malformed { - reason: format!( - "non-numeric cloud-init incarnation {incarnation:?}" - ), - }; - }; - // Non-span cloud-init types are kept verbatim, not rejected. - let kind = event_type - .parse::() - .unwrap_or_else(|()| RecordKind::Other(event_type.to_string())); - KeyClass::CloudInit { - boot_epoch, - kind, - name, - vm_id, - uuid, - } -} - -/// Split `value` into pieces of at most `max_bytes` bytes each, always -/// at UTF-8 codepoint boundaries. -/// -/// An empty input yields a single empty chunk so callers still write one -/// record. A codepoint wider than `max_bytes` (only possible for tiny -/// `max_bytes`, never for [`MAX_CHUNK_BYTES`]) is emitted whole so the -/// split always makes progress. -fn chunk_at_char_boundary(value: &str, max_bytes: usize) -> Vec<&str> { - debug_assert!(max_bytes > 0, "max_bytes must be positive"); - if value.is_empty() { - return vec![""]; - } - - let mut chunks = Vec::new(); - let mut start = 0; - while start < value.len() { - if value.len() - start <= max_bytes { - chunks.push(&value[start..]); - break; - } - - let mut end = start + max_bytes; - while end > start && !value.is_char_boundary(end) { - end -= 1; - } - if end == start { - end = start + max_bytes + 1; - while end < value.len() && !value.is_char_boundary(end) { - end += 1; - } - } - - chunks.push(&value[start..end]); - start = end; - } - chunks -} - -/// Reject the `|` key delimiter in an event field so the formatted key -/// round-trips through [`classify_key`]. -fn reject_delimiter(field: &'static str, value: &str) -> Result<(), KvpError> { - if value.contains(EVENT_KEY_DELIMITER) { - return Err(KvpError::EventFieldContainsDelimiter { field }); - } - Ok(()) -} - -/// A single diagnostic event — the decoded, source-agnostic form of one -/// azure-init or cloud-init KVP entry. -/// -/// Metadata (`agent`, `boot_epoch`, `vm_id`, `kind`, `name`, `event_id`) -/// comes from the record key; the payload (`timestamp`, `result`, -/// `duration`, `message`) from the value. Optional fields are populated -/// only when the source provides them (e.g. cloud-init `finish` records -/// carry `result` and `duration`). +/// One normalized azure-init or cloud-init diagnostic entry. #[derive(Clone, Debug, PartialEq, serde::Serialize)] #[non_exhaustive] pub struct DiagnosticEvent { - /// Reporting agent identifier from the key, e.g. `azure-init-0.1.0` - /// or `CLOUD_INIT`; also distinguishes the record's source. + /// Reporting source, such as `azure-init-0.1.1` or `CLOUD_INIT`. pub agent: String, - /// Unix epoch second the system booted (cloud-init's incarnation), - /// shared by every record of one boot. + /// Unix epoch second at which this boot began. pub boot_epoch: i64, - /// VM identifier from the key. Absent in cloud-init builds that - /// predate the `vm_id` key segment. + /// VM identifier. Older cloud-init keys do not contain one. #[serde(skip_serializing_if = "Option::is_none")] pub vm_id: Option, - /// Whether this record opens a span, closes a span, or is a point - /// event. - pub kind: RecordKind, - /// Formatted event or span name, e.g. `user:create_user`. + /// Entry lifecycle or source-specific reporting kind. + pub kind: DiagnosticKind, + /// Logical event or span name. pub name: String, - /// Per-record identifier (azure-init's UUIDv4 / cloud-init's uuid); - /// every chunk of one record shares it, as do a span's start and - /// finish. + /// Identifier shared by all chunks and, for spans, related lifecycle + /// entries. pub event_id: String, - /// ISO-8601 timestamp, if the source provides one. - #[serde(skip_serializing_if = "Option::is_none")] - pub timestamp: Option, - /// Result string (e.g. `SUCCESS`), present on cloud-init `finish` - /// records. + /// Time at which the entry occurred. + pub timestamp: DateTime, + /// Source result, when supplied by cloud-init. #[serde(skip_serializing_if = "Option::is_none")] pub result: Option, - /// Duration in seconds, present on cloud-init `finish` records. + /// Source duration in seconds, when supplied by cloud-init. #[serde(skip_serializing_if = "Option::is_none")] pub duration: Option, - /// Human-readable message. The diagnostics layer imposes no format - /// on this string. + /// Reassembled human-readable payload. pub message: String, } -/// A single record read back from the pool and classified by -/// [`DiagnosticsKvp::records`]. -#[derive(Clone, Debug, PartialEq)] -#[non_exhaustive] -pub enum DiagnosticRecord { - /// A reassembled diagnostic event, from either agent. - Decoded { - /// The decoded event. - event: DiagnosticEvent, - /// Number of on-disk records the value spanned (1 when short). - chunks: usize, - }, - /// An unstructured record whose key is not an event key, such as - /// `PROVISIONING_REPORT`. - Raw { - /// The record key. - key: String, - /// The reassembled record value. - value: String, - }, - /// A record whose key is event-shaped but is not a valid event (for - /// example, an unrecognized kind or invalid cloud-init JSON). - Malformed { - /// The record key. - key: String, - /// The reassembled record value. - value: String, - /// Why the key failed to parse as an event. - reason: String, - }, -} - -/// A typed diagnostics view over a [`KvpPoolStore`]. +/// Typed diagnostic access over a KVP pool. /// -/// Owns the `agent` and `vm_id` stamped into this layer's azure-init -/// event keys. See the module-level documentation for the on-disk -/// format. +/// The boot epoch is resolved once at construction so emitting tracing events +/// does not read `/proc/stat` for every entry. #[derive(Clone, Debug)] pub struct DiagnosticsKvp { store: KvpPoolStore, vm_id: String, agent: String, + boot_epoch: i64, } impl DiagnosticsKvp { + /// Create an accessor for `store`, stamping new entries with `vm_id` and + /// `agent`. pub fn new( store: KvpPoolStore, vm_id: impl Into, agent: impl Into, - ) -> Self { - Self { + ) -> Result { + let boot_epoch = store.boot_epoch()?; + Ok(Self { store, vm_id: vm_id.into(), agent: agent.into(), - } + boot_epoch, + }) } + pub fn store(&self) -> &KvpPoolStore { &self.store } + pub fn vm_id(&self) -> &str { &self.vm_id } + pub fn agent(&self) -> &str { &self.agent } - /// Emit an azure-init point event with `name` and `message`. - /// - /// The `message` is stored verbatim as the record value. A fresh - /// `event_id` (UUIDv4) is generated. + pub fn boot_epoch(&self) -> i64 { + self.boot_epoch + } + + /// Emit a point event using a fresh UUID and the current timestamp. pub fn emit_event( &self, - name: impl Into, + name: impl AsRef, message: impl AsRef, ) -> Result<(), KvpError> { - let event_id = Uuid::new_v4().to_string(); - self.write_event( - RecordKind::Event, - &event_id, - &name.into(), - message.as_ref(), + self.emit( + DiagnosticKind::Event, + name, + Uuid::new_v4().to_string(), + Utc::now(), + message, ) } - /// Format the key - /// `||||||` - /// (stamping `boot_epoch`, `vm_id`, `agent`, and the current - /// `timestamp`) and write `value` as its message. + /// Emit an azure-init diagnostic with caller-supplied lifecycle metadata. /// - /// The message is written under a single lock via - /// [`KvpPoolStore::append_multiple`]; values longer than - /// [`MAX_CHUNK_BYTES`] are split at UTF-8 codepoint boundaries into - /// multiple records. Every record is keyed with a `|` - /// suffix (`0`, `1`, …) so it is unique, and the chunks are regrouped - /// by [`records`](Self::records) on read. - /// - /// Returns [`KvpError::EventFieldContainsDelimiter`] if `agent`, - /// `vm_id`, `name`, or `event_id` contains the `|` key delimiter. - fn write_event( + /// A tracing adapter can infer `kind` from its callback, retain one + /// `event_id` for a span, and pass the timestamp captured when the callback + /// occurred. Key formatting, chunking, and storage remain encapsulated + /// here. + pub fn emit( &self, - kind: RecordKind, - event_id: &str, - name: &str, - value: &str, + kind: DiagnosticKind, + name: impl AsRef, + event_id: impl AsRef, + timestamp: DateTime, + message: impl AsRef, ) -> Result<(), KvpError> { + let name = name.as_ref(); + let event_id = event_id.as_ref(); + let kind_token = kind.to_string(); + reject_delimiter("agent", &self.agent)?; reject_delimiter("vm_id", &self.vm_id)?; + reject_delimiter("kind", &kind_token)?; reject_delimiter("name", name)?; reject_delimiter("event_id", event_id)?; - let boot_epoch = self.store.boot_epoch()?; - let timestamp = now_timestamp(); + let timestamp = timestamp.to_rfc3339_opts(SecondsFormat::Millis, true); let key = format_event_key( &self.agent, - boot_epoch, + self.boot_epoch, &self.vm_id, - kind, + &kind_token, name, event_id, ×tamp, ); + self.write_chunked(&key, message.as_ref()) + } - self.write_chunked(&key, value) + /// Read all decodable diagnostics in first-seen pool order. + /// + /// Raw records, malformed entries, and incomplete chunk groups are omitted. + pub fn entries(&self) -> Result, KvpError> { + Ok(decode_entries(self.store.dump()?)) } - /// Split `value` at [`MAX_CHUNK_BYTES`] and append each chunk in one - /// atomic batch, keyed `|` (`0`, `1`, …) so every - /// record is unique. [`reassemble`] strips the index on read. fn write_chunked(&self, key: &str, value: &str) -> Result<(), KvpError> { - let records: Vec<(String, &str)> = - chunk_at_char_boundary(value, MAX_CHUNK_BYTES) - .into_iter() - .enumerate() - .map(|(subevent_index, chunk)| { - let chunk_key = - format!("{key}{EVENT_KEY_DELIMITER}{subevent_index}"); - (chunk_key, chunk) - }) - .collect(); + let records = chunk_at_char_boundary(value, MAX_CHUNK_BYTES) + .into_iter() + .enumerate() + .map(|(index, chunk)| { + (format!("{key}{EVENT_KEY_DELIMITER}{index}"), chunk) + }); self.store.append_multiple(records) } +} - /// Read every record, reassembling chunked events and classifying - /// each into a [`DiagnosticRecord`]. - /// - /// Records are returned in on-disk order. Consecutive records that - /// share an event key — ignoring the `|` suffix — are - /// one event; because [`emit_event`](Self::emit_event) writes an - /// event's chunks contiguously under a single lock, reassembly is - /// correct even under concurrent writers. - pub fn records(&self) -> Result, KvpError> { - Ok(reassemble(self.store.dump()?)) +fn reject_delimiter(field: &'static str, value: &str) -> Result<(), KvpError> { + if value.contains(EVENT_KEY_DELIMITER) { + return Err(KvpError::EventFieldContainsDelimiter { field }); } + Ok(()) +} - /// Read back every decoded [`DiagnosticEvent`], from either agent, in - /// on-disk order. - /// - /// Raw and malformed records are excluded. Use - /// [`records`](Self::records) for the full view that includes them. - pub fn events(&self) -> Result, KvpError> { - Ok(self - .records()? - .into_iter() - .filter_map(|record| match record { - DiagnosticRecord::Decoded { event, .. } => Some(event), - DiagnosticRecord::Raw { .. } - | DiagnosticRecord::Malformed { .. } => None, - }) - .collect()) +fn format_event_key( + agent: &str, + boot_epoch: i64, + vm_id: &str, + kind: &str, + name: &str, + event_id: &str, + timestamp: &str, +) -> String { + let d = EVENT_KEY_DELIMITER; + format!( + "{agent}{d}{boot_epoch}{d}{vm_id}{d}{kind}{d}{name}{d}{event_id}{d}{timestamp}" + ) +} + +fn chunk_at_char_boundary(value: &str, max_bytes: usize) -> Vec<&str> { + debug_assert!(max_bytes > 0, "max_bytes must be positive"); + if value.is_empty() { + return vec![""]; } - /// Remove every diagnostic key: any key that parses as an event key - /// (including every `|` chunk of a multi-record event) - /// or a malformed event key. Raw records such as `PROVISIONING_REPORT` - /// are left intact. - pub fn clear(&self) -> Result<(), KvpError> { - let keys: Vec = self - .store - .dump()? - .into_iter() - .filter_map(|(key, _)| { - let is_diagnostic = !matches!( - classify_key(base_event_key(&key)), - KeyClass::Raw - ); - is_diagnostic.then_some(key) - }) - .collect(); - self.store.delete_multiple(keys)?; - Ok(()) + let mut chunks = Vec::new(); + let mut start = 0; + while start < value.len() { + if value.len() - start <= max_bytes { + chunks.push(&value[start..]); + break; + } + + let mut end = start + max_bytes; + while end > start && !value.is_char_boundary(end) { + end -= 1; + } + if end == start { + end = start + max_bytes + 1; + while end < value.len() && !value.is_char_boundary(end) { + end += 1; + } + } + + chunks.push(&value[start..end]); + start = end; } + chunks } -/// The shared event key a chunk belongs to: strips a trailing -/// `|`, or returns the key unchanged if it has none. -fn base_event_key(key: &str) -> &str { - split_subevent_index(key).0 +struct AzureKey<'a> { + agent: &'a str, + boot_epoch: i64, + vm_id: &'a str, + kind: DiagnosticKind, + name: &'a str, + event_id: &'a str, + timestamp: DateTime, } -/// Split a key into its base event key and optional trailing subevent -/// index: `(base, Some(index))` when a trailing numeric segment follows an -/// event-shaped base (valid or malformed), else `(key, None)`. -/// [`reassemble`] uses the index to regroup an event's chunks and restore -/// their write order. -fn split_subevent_index(key: &str) -> (&str, Option) { - if let Some((base, index)) = key.rsplit_once(EVENT_KEY_DELIMITER) { - if let Ok(index) = index.parse::() { - if matches!( - classify_key(base), - KeyClass::Event { .. } - | KeyClass::CloudInit { .. } - | KeyClass::Malformed { .. } - ) { - return (base, Some(index)); - } - } +struct CloudInitKey<'a> { + boot_epoch: i64, + kind: DiagnosticKind, + name: &'a str, + vm_id: Option<&'a str>, + event_id: &'a str, +} + +enum ParsedKey<'a> { + Azure(AzureKey<'a>), + CloudInit(CloudInitKey<'a>), +} + +fn parse_diagnostic_key(key: &str) -> Option> { + if key.split(EVENT_KEY_DELIMITER).next()? == CLOUD_INIT_PREFIX { + parse_cloud_init_key(key).map(ParsedKey::CloudInit) + } else { + parse_azure_key(key).map(ParsedKey::Azure) } - (key, None) } -/// Group consecutive records sharing an event key — chunk -/// `|` suffixes stripped — from [`KvpPoolStore::dump`] -/// and classify each group into a [`DiagnosticRecord`]. -fn reassemble(dumped: Vec<(String, String)>) -> Vec { - let mut parsed = dumped - .into_iter() - .map(|(key, value)| { - let (base, index) = split_subevent_index(&key); - (base.to_string(), index, value) - }) - .peekable(); - - let mut records = Vec::new(); - while let Some((base, index, value)) = parsed.next() { - let mut indexed = vec![(index, value)]; - while parsed.peek().is_some_and(|(next, _, _)| *next == base) { - let (_, next_index, next_value) = - parsed.next().expect("peeked value exists"); - indexed.push((next_index, next_value)); - } - // Restore write order by subevent index. Stable, so a single - // record (index `None`) or any equal indices keep on-disk order. - indexed.sort_by_key(|(index, _)| *index); - let chunk_values = - indexed.into_iter().map(|(_, value)| value).collect(); - records.push(classify_record(base, chunk_values)); +fn parse_azure_key(key: &str) -> Option> { + let mut segments = key.split(EVENT_KEY_DELIMITER); + let agent = segments.next()?; + let boot_epoch = segments.next()?.parse().ok()?; + let vm_id = segments.next()?; + let kind = DiagnosticKind::from_token(segments.next()?); + let name = segments.next()?; + let event_id = segments.next()?; + let timestamp = parse_timestamp(segments.next()?)?; + if segments.next().is_some() { + return None; } - records + Some(AzureKey { + agent, + boot_epoch, + vm_id, + kind, + name, + event_id, + timestamp, + }) } -/// Classify one reassembled group of chunk values (ordered by subevent -/// index, never empty) into a [`DiagnosticRecord`]. azure-init and raw -/// records concatenate their values; cloud-init chunks are stitched and -/// decoded via [`decode_cloud_init_value`]. -fn classify_record(key: String, chunk_values: Vec) -> DiagnosticRecord { - let chunks = chunk_values.len(); - match classify_key(&key) { - KeyClass::Event { - agent, - boot_epoch, - vm_id, - kind, - name, - event_id, - timestamp, - } => { - let message = chunk_values.concat(); - DiagnosticRecord::Decoded { - event: DiagnosticEvent { - agent: agent.to_string(), - boot_epoch, - vm_id: Some(vm_id.to_string()), - kind, - name: name.to_string(), - event_id: event_id.to_string(), - timestamp: Some(timestamp.to_string()), - result: None, - duration: None, - message, - }, - chunks, +fn parse_cloud_init_key(key: &str) -> Option> { + let segments: Vec<_> = key.split(EVENT_KEY_DELIMITER).collect(); + let (vm_id, event_id) = match segments.as_slice() { + [CLOUD_INIT_PREFIX, _, _, _, event_id] => (None, *event_id), + [CLOUD_INIT_PREFIX, _, _, _, vm_id, event_id] => { + (Some(*vm_id), *event_id) + } + _ => return None, + }; + + Some(CloudInitKey { + boot_epoch: segments[1].parse().ok()?, + kind: DiagnosticKind::from_token(segments[2]), + name: segments[3], + vm_id, + event_id, + }) +} + +fn parse_timestamp(value: &str) -> Option> { + DateTime::parse_from_rfc3339(value) + .ok() + .map(|timestamp| timestamp.with_timezone(&Utc)) +} + +/// An indexed group is kept at the position where its first chunk appeared. +enum PendingEntry { + Standalone { + key: String, + value: String, + }, + Indexed { + key: String, + chunks: Vec<(u32, String)>, + }, +} + +fn decode_entries(dumped: Vec<(String, String)>) -> Vec { + let mut pending = Vec::::new(); + let mut indexed_groups = HashMap::::new(); + + for (key, value) in dumped { + if let Some((base, index)) = split_chunk_index(&key) { + let base = base.to_string(); + if let Some(position) = indexed_groups.get(&base).copied() { + if let PendingEntry::Indexed { chunks, .. } = + &mut pending[position] + { + chunks.push((index, value)); + } + } else { + indexed_groups.insert(base.clone(), pending.len()); + pending.push(PendingEntry::Indexed { + key: base, + chunks: vec![(index, value)], + }); } + } else if parse_diagnostic_key(&key).is_some() { + pending.push(PendingEntry::Standalone { key, value }); } - KeyClass::CloudInit { - boot_epoch, - kind, - name, - vm_id, - uuid, - } => { - // Own the key-derived fields up front so `key` and - // `chunk_values` can move into a `Malformed` record when a - // chunk's value fails to decode. - let name = name.to_string(); - let vm_id = vm_id.map(str::to_string); - let uuid = uuid.to_string(); - match decode_cloud_init_value(&chunk_values) { - Ok((meta, message)) => DiagnosticRecord::Decoded { - event: DiagnosticEvent { - agent: CLOUD_INIT_PREFIX.to_string(), - boot_epoch, - vm_id, - kind, - name, - event_id: uuid, - timestamp: meta - .get("ts") - .and_then(|t| t.as_str()) - .map(str::to_string), - result: meta - .get("result") - .and_then(|r| r.as_str()) - .map(str::to_string), - duration: meta.get("duration").and_then(|d| d.as_f64()), - message, - }, - chunks, - }, - Err(err) => DiagnosticRecord::Malformed { - key, - value: chunk_values.concat(), - reason: format!("invalid cloud-init JSON value: {err}"), - }, + } + + pending + .into_iter() + .filter_map(|entry| match entry { + PendingEntry::Standalone { key, value } => { + decode_standalone(&key, &value) } + PendingEntry::Indexed { key, chunks } => { + decode_indexed(&key, chunks) + } + }) + .collect() +} + +fn split_chunk_index(key: &str) -> Option<(&str, u32)> { + let (base, index) = key.rsplit_once(EVENT_KEY_DELIMITER)?; + let index = index.parse().ok()?; + parse_diagnostic_key(base)?; + Some((base, index)) +} + +fn order_chunks(mut chunks: Vec<(u32, String)>) -> Option> { + chunks.sort_by_key(|(index, _)| *index); + for (expected, (actual, _)) in chunks.iter().enumerate() { + if *actual != u32::try_from(expected).ok()? { + return None; } - KeyClass::Malformed { reason } => DiagnosticRecord::Malformed { - key, - value: chunk_values.concat(), - reason, - }, - KeyClass::Raw => DiagnosticRecord::Raw { + } + Some(chunks) +} + +fn decode_standalone(key: &str, value: &str) -> Option { + match parse_diagnostic_key(key)? { + ParsedKey::Azure(key) => Some(azure_event(key, value.to_string())), + ParsedKey::CloudInit(key) => decode_cloud_init_single(key, value), + } +} + +fn decode_indexed( + key: &str, + chunks: Vec<(u32, String)>, +) -> Option { + let chunks = order_chunks(chunks)?; + match parse_diagnostic_key(key)? { + ParsedKey::Azure(key) => Some(azure_event( key, - value: chunk_values.concat(), - }, + chunks.into_iter().map(|(_, value)| value).collect(), + )), + ParsedKey::CloudInit(key) => decode_cloud_init_chunks(key, &chunks), } } -/// Marker preceding a cloud-init value's message field: `"msg":"`. -const CLOUD_INIT_MSG_MARKER: &str = "\"msg\":\""; +fn azure_event(key: AzureKey<'_>, message: String) -> DiagnosticEvent { + DiagnosticEvent { + agent: key.agent.to_string(), + boot_epoch: key.boot_epoch, + vm_id: Some(key.vm_id.to_string()), + kind: key.kind, + name: key.name.to_string(), + event_id: key.event_id.to_string(), + timestamp: key.timestamp, + result: None, + duration: None, + message, + } +} -/// Decode a cloud-init event's chunk value(s) into `(metadata, message)`, -/// reading `ts`/`result`/`duration` from an untyped [`serde_json::Value`]. -/// -/// A single record is complete JSON, parsed directly. A multi-record event -/// was split mid-escape by cloud-init's `_break_down` (e.g. a `\n` cut into -/// `\` and `n`), so no chunk is valid JSON alone: recover each chunk's raw -/// escaped `msg` slice, concatenate, and unescape once; metadata comes from -/// the first chunk. -fn decode_cloud_init_value( - chunks: &[String], -) -> Result<(serde_json::Value, String), String> { - if let [only] = chunks { - let value: serde_json::Value = - serde_json::from_str(only).map_err(|e| e.to_string())?; - let message = value - .get("msg") - .and_then(|m| m.as_str()) - .unwrap_or_default() - .to_string(); - return Ok((value, message)); +fn decode_cloud_init_single( + key: CloudInitKey<'_>, + value: &str, +) -> Option { + let metadata: serde_json::Value = serde_json::from_str(value).ok()?; + if metadata.get("msg_i").is_some() { + return None; } + let message = metadata.get("msg")?.as_str()?.to_string(); + cloud_init_event(key, &metadata, message) +} - let mut escaped = String::new(); - for chunk in chunks { - escaped.push_str(cloud_init_escaped_msg_slice(chunk)?); +fn decode_cloud_init_chunks( + key: CloudInitKey<'_>, + chunks: &[(u32, String)], +) -> Option { + let mut metadata = None; + let mut escaped_message = String::new(); + + for (key_index, value) in chunks { + let chunk_metadata = cloud_init_chunk_metadata(value)?; + let value_index = chunk_metadata.get("msg_i")?.as_u64()?; + if value_index != u64::from(*key_index) { + return None; + } + if metadata.is_none() { + metadata = Some(chunk_metadata); + } + escaped_message.push_str(cloud_init_escaped_msg_slice(value)?); } - let message: String = serde_json::from_str(&format!("\"{escaped}\"")) - .map_err(|e| e.to_string())?; - Ok((cloud_init_chunk_metadata(&chunks[0])?, message)) + let message = + serde_json::from_str(&format!("\"{escaped_message}\"")).ok()?; + cloud_init_event(key, &metadata?, message) +} + +fn cloud_init_event( + key: CloudInitKey<'_>, + metadata: &serde_json::Value, + message: String, +) -> Option { + Some(DiagnosticEvent { + agent: CLOUD_INIT_PREFIX.to_string(), + boot_epoch: key.boot_epoch, + vm_id: key.vm_id.map(str::to_string), + kind: key.kind, + name: key.name.to_string(), + event_id: key.event_id.to_string(), + timestamp: parse_timestamp(metadata.get("ts")?.as_str()?)?, + result: metadata + .get("result") + .and_then(|result| result.as_str()) + .map(str::to_string), + duration: metadata.get("duration").and_then(|value| value.as_f64()), + message, + }) } -/// Recover a chunk's raw (still-escaped) `msg` slice — the bytes between -/// the `"msg":"` marker and the closing `"}` — without unescaping. -fn cloud_init_escaped_msg_slice(chunk: &str) -> Result<&str, String> { - let start = chunk - .find(CLOUD_INIT_MSG_MARKER) - .ok_or("chunk is missing a \"msg\" field")? - + CLOUD_INIT_MSG_MARKER.len(); - let end = chunk - .strip_suffix("\"}") - .map(str::len) - .ok_or("chunk does not end with '\"}'")?; - chunk - .get(start..end) - .ok_or_else(|| "chunk \"msg\" field is malformed".to_string()) +/// Recover a cloud-init chunk's raw, still-escaped `msg` fragment. +fn cloud_init_escaped_msg_slice(chunk: &str) -> Option<&str> { + let start = + chunk.find(CLOUD_INIT_MSG_MARKER)? + CLOUD_INIT_MSG_MARKER.len(); + let end = chunk.strip_suffix("\"}")?.len(); + chunk.get(start..end) } -/// Parse a chunk's non-`msg` prefix (the portion before its `,"msg":"` -/// field, which is always valid JSON) into a [`serde_json::Value`]. -fn cloud_init_chunk_metadata(chunk: &str) -> Result { +/// Parse the valid metadata prefix before cloud-init's final `msg` field. +fn cloud_init_chunk_metadata(chunk: &str) -> Option { let marker = format!(",{CLOUD_INIT_MSG_MARKER}"); - let end = chunk - .find(&marker) - .ok_or("chunk is missing a \"msg\" field")?; - serde_json::from_str(&format!("{}}}", &chunk[..end])) - .map_err(|e| e.to_string()) + let end = chunk.find(&marker)?; + serde_json::from_str(&format!("{}}}", &chunk[..end])).ok() } #[cfg(test)] mod tests { use super::*; - use crate::{KvpPool, PoolMode}; use rstest::rstest; - const AGENT: &str = "azure-init-0.1.0"; + const AGENT: &str = "azure-init-0.1.1"; const VM_ID: &str = "3f2504e0-4f89-41d3-9a0c-0305e82c3301"; const EVENT_ID: &str = "8f3e9c4a-1b2c-4d5e-9f01-234567890abc"; - const BOOT_EPOCH: i64 = 1_700_000_000; const TIMESTAMP: &str = "2026-07-27T21:33:24.300Z"; - #[test] - fn event_key_formats_and_classifies() { - let formatted = format_event_key( - AGENT, - BOOT_EPOCH, - VM_ID, - RecordKind::Event, - "user:create_user", - EVENT_ID, - TIMESTAMP, - ); - assert_eq!( - formatted, - format!( - "{AGENT}|{BOOT_EPOCH}|{VM_ID}|event|user:create_user|\ - {EVENT_ID}|{TIMESTAMP}" - ) - ); - assert!(matches!( - classify_key(&formatted), - KeyClass::Event { - agent, - boot_epoch, - vm_id, - kind, - name, - event_id, - timestamp, - } if agent == AGENT - && boot_epoch == BOOT_EPOCH - && vm_id == VM_ID - && kind == RecordKind::Event - && name == "user:create_user" - && event_id == EVENT_ID - && timestamp == TIMESTAMP - )); - } - - #[test] - fn classify_round_trips_every_kind() { - for expected in - [RecordKind::Start, RecordKind::Finish, RecordKind::Event] - { - let key = format_event_key( - AGENT, - BOOT_EPOCH, - VM_ID, - expected.clone(), - "span:event", - EVENT_ID, - TIMESTAMP, - ); - assert!(matches!( - classify_key(&key), - KeyClass::Event { kind, .. } if kind == expected - )); - } - } - #[rstest] - #[case::start(RecordKind::Start, "start")] - #[case::finish(RecordKind::Finish, "finish")] - #[case::event(RecordKind::Event, "event")] - #[case::other(RecordKind::Other("compressed".to_string()), "compressed")] - fn record_kind_renders_as_its_token( - #[case] kind: RecordKind, - #[case] token: &str, - ) { + #[case(DiagnosticKind::Start, "start")] + #[case(DiagnosticKind::Finish, "finish")] + #[case(DiagnosticKind::Event, "event")] + #[case(DiagnosticKind::Other("compressed".into()), "compressed")] + fn kind_uses_wire_token(#[case] kind: DiagnosticKind, #[case] token: &str) { assert_eq!(kind.to_string(), token); - assert_eq!( - serde_json::to_value(&kind).unwrap(), - serde_json::json!(token) - ); - } - - fn class_of(key: &str) -> &'static str { - match classify_key(key) { - KeyClass::Event { .. } => "event", - KeyClass::CloudInit { .. } => "cloud-init", - KeyClass::Malformed { .. } => "malformed", - KeyClass::Raw => "raw", - } - } - - #[rstest] - #[case::event("a|100|vm|event|name|id|ts", "event")] - #[case::cloud_init( - "CLOUD_INIT|1785187982|finish|name|vmid|uuid", - "cloud-init" - )] - #[case::raw_single_segment("PROVISIONING_REPORT", "raw")] - #[case::raw_too_few_segments("a|100|vm|event|name|id", "raw")] - #[case::raw_too_many_segments("a|100|vm|event|name|id|ts|extra", "raw")] - #[case::raw_non_numeric_boot_epoch("a|notnum|vm|event|name|id|ts", "raw")] - #[case::malformed_bad_kind("a|100|vm|NOTAKIND|name|id|ts", "malformed")] - #[case::malformed_other_kind("a|100|vm|nope|name|id|ts", "malformed")] - #[case::cloud_init_custom_type( - "CLOUD_INIT|100|compressed|name|vmid|uuid", - "cloud-init" - )] - #[case::cloud_init_non_numeric_incarnation( - "CLOUD_INIT|notnum|finish|name|vmid|uuid", - "malformed" - )] - fn classify_key_categorizes(#[case] key: &str, #[case] expected: &str) { - assert_eq!(class_of(key), expected); + assert_eq!(serde_json::to_value(kind).unwrap(), token); } #[rstest] - #[case::empty("", 4, vec![""])] - #[case::shorter_than_max("abc", 8, vec!["abc"])] - #[case::exact_multiple("abcdef", 2, vec!["ab", "cd", "ef"])] - #[case::ascii_remainder("abcde", 2, vec!["ab", "cd", "e"])] - #[case::two_byte_boundary("aéb", 2, vec!["a", "é", "b"])] - #[case::oversized_three_byte("€", 1, vec!["€"])] - #[case::oversized_repeated("€€", 1, vec!["€", "€"])] - fn chunk_splits_at_utf8_boundaries( + #[case("", 4, vec![""])] + #[case("abcdef", 2, vec!["ab", "cd", "ef"])] + #[case("aéb", 2, vec!["a", "é", "b"])] + #[case("€€", 1, vec!["€", "€"])] + fn chunks_on_utf8_boundaries( #[case] input: &str, - #[case] max_bytes: usize, + #[case] max: usize, #[case] expected: Vec<&str>, ) { - assert_eq!(chunk_at_char_boundary(input, max_bytes), expected); + assert_eq!(chunk_at_char_boundary(input, max), expected); } #[test] - fn chunk_reassembles_multibyte_payload() { - let payload = "🚀".repeat(100); - let chunks = chunk_at_char_boundary(&payload, 7); - assert!(chunks.iter().all(|chunk| chunk.len() <= 7)); - assert_eq!(chunks.concat(), payload); - } - - #[test] - fn reject_delimiter_flags_pipe() { - assert!(reject_delimiter("name", "no pipe here").is_ok()); - let err = reject_delimiter("name", "has|pipe").unwrap_err(); - assert!(matches!( - err, - KvpError::EventFieldContainsDelimiter { field: "name" } - )); - assert_eq!( - err.to_string(), - "event key field 'name' must not contain '|'" - ); - } - - #[test] - fn reassemble_groups_chunks_and_classifies() { + fn azure_key_round_trips() { let key = format_event_key( AGENT, - BOOT_EPOCH, + 1_700_000_000, VM_ID, - RecordKind::Start, - "config:dump", + "event", + "user:create_user", EVENT_ID, TIMESTAMP, ); - - let dumped = vec![ - (key.clone(), "part-one/".to_string()), - (key.clone(), "part-two".to_string()), - ( - "PROVISIONING_REPORT".to_string(), - "result=success".to_string(), - ), - ("a|100|vm|NOPE|name|id|ts".to_string(), "junk".to_string()), - ]; - - let records = reassemble(dumped); - assert_eq!(records.len(), 3); - - assert_eq!( - records[0], - DiagnosticRecord::Decoded { - event: DiagnosticEvent { - agent: AGENT.to_string(), - boot_epoch: BOOT_EPOCH, - vm_id: Some(VM_ID.to_string()), - kind: RecordKind::Start, - name: "config:dump".to_string(), - event_id: EVENT_ID.to_string(), - timestamp: Some(TIMESTAMP.to_string()), - result: None, - duration: None, - message: "part-one/part-two".to_string(), - }, - chunks: 2, - } - ); - assert!(matches!(&records[1], DiagnosticRecord::Raw { key, .. } - if key == "PROVISIONING_REPORT")); - assert!(matches!(&records[2], DiagnosticRecord::Malformed { .. })); - } - - #[test] - fn reassemble_keeps_distinct_adjacent_keys_separate() { - let make = |event_id: &str| { - format_event_key( - AGENT, - BOOT_EPOCH, - VM_ID, - RecordKind::Start, - "span:name", - event_id, - TIMESTAMP, - ) + let ParsedKey::Azure(parsed) = parse_diagnostic_key(&key).unwrap() + else { + panic!("expected azure-init key"); }; - let dumped = vec![ - (make("id-1"), "first".to_string()), - (make("id-2"), "second".to_string()), - ]; - let records = reassemble(dumped); - assert_eq!(records.len(), 2); - assert!(matches!( - &records[0], - DiagnosticRecord::Decoded { chunks: 1, .. } - )); - assert!(matches!( - &records[1], - DiagnosticRecord::Decoded { chunks: 1, .. } - )); - } - - #[rstest] - #[case::indexed_chunk( - "a|100|vm|event|name|id|ts|0", - "a|100|vm|event|name|id|ts" - )] - #[case::indexed_chunk_multi_digit( - "a|100|vm|event|name|id|ts|12", - "a|100|vm|event|name|id|ts" - )] - #[case::single_event_unchanged( - "a|100|vm|event|name|id|ts", - "a|100|vm|event|name|id|ts" - )] - #[case::cloud_init_indexed_chunk( - "CLOUD_INIT|1785187982|finish|mod|vmid|uuid|0", - "CLOUD_INIT|1785187982|finish|mod|vmid|uuid" - )] - #[case::raw_unchanged("PROVISIONING_REPORT", "PROVISIONING_REPORT")] - #[case::non_event_numeric_tail_unchanged("foo|3", "foo|3")] - #[case::malformed_unchanged( - "a|100|vm|NOPE|name|id|ts", - "a|100|vm|NOPE|name|id|ts" - )] - #[case::malformed_indexed_chunk( - "a|100|vm|NOPE|name|id|ts|0", - "a|100|vm|NOPE|name|id|ts" - )] - fn base_event_key_strips_event_subevent_index( - #[case] key: &str, - #[case] expected: &str, - ) { - assert_eq!(base_event_key(key), expected); + assert_eq!(parsed.agent, AGENT); + assert_eq!(parsed.kind, DiagnosticKind::Event); + assert_eq!(parsed.timestamp, parse_timestamp(TIMESTAMP).unwrap()); } #[test] - fn reassemble_groups_indexed_chunk_keys() { - let base = format_event_key( - AGENT, - BOOT_EPOCH, - VM_ID, - RecordKind::Finish, - "config:dump", - EVENT_ID, - TIMESTAMP, - ); - let dumped = vec![ - (format!("{base}|0"), "part-one/".to_string()), - (format!("{base}|1"), "part-two/".to_string()), - (format!("{base}|2"), "part-three".to_string()), - ]; - - let records = reassemble(dumped); - assert_eq!(records.len(), 1); - assert_eq!( - records[0], - DiagnosticRecord::Decoded { - event: DiagnosticEvent { - agent: AGENT.to_string(), - boot_epoch: BOOT_EPOCH, - vm_id: Some(VM_ID.to_string()), - kind: RecordKind::Finish, - name: "config:dump".to_string(), - event_id: EVENT_ID.to_string(), - timestamp: Some(TIMESTAMP.to_string()), - result: None, - duration: None, - message: "part-one/part-two/part-three".to_string(), - }, - chunks: 3, - } + fn invalid_and_raw_records_are_skipped() { + let valid = format_event_key( + AGENT, 100, VM_ID, "event", "valid", EVENT_ID, TIMESTAMP, ); + let events = decode_entries(vec![ + ("PROVISIONING_REPORT".into(), "result=success".into()), + ("a|not-a-boot|vm|event|name|id|timestamp".into(), "x".into()), + (valid, "message".into()), + ]); + assert_eq!(events.len(), 1); + assert_eq!(events[0].message, "message"); } #[test] - fn azure_event_value_is_full_message() { - let key = format_event_key( - AGENT, - BOOT_EPOCH, - VM_ID, - RecordKind::Event, - "user:create_user", - EVENT_ID, - TIMESTAMP, + fn indexed_chunks_group_globally_and_preserve_first_seen_order() { + let first = format_event_key( + AGENT, 100, VM_ID, "event", "first", "id-1", TIMESTAMP, ); - assert!(matches!( - classify_record(key, vec!["boom".to_string()]), - DiagnosticRecord::Decoded { event, chunks: 1 } - if event.kind == RecordKind::Event - && event.message == "boom" - && event.agent == AGENT - && event.vm_id.as_deref() == Some(VM_ID) - && event.timestamp.as_deref() == Some(TIMESTAMP) - )); - } - - #[test] - fn azure_span_value_is_full_message() { - let key = format_event_key( - AGENT, - BOOT_EPOCH, - VM_ID, - RecordKind::Finish, - "config:write", - EVENT_ID, - TIMESTAMP, + let second = format_event_key( + AGENT, 100, VM_ID, "event", "second", "id-2", TIMESTAMP, ); - assert!(matches!( - classify_record(key, vec!["write_config completed".to_string()]), - DiagnosticRecord::Decoded { event, chunks: 1 } - if event.kind == RecordKind::Finish - && event.message == "write_config completed" - )); - } - - #[test] - fn azure_span_start_finish_pair_round_trips_through_writer() { - let dir = tempfile::TempDir::new().unwrap(); - let store = - KvpPoolStore::new_in(KvpPool::Guest, dir.path(), PoolMode::Safe) - .unwrap(); - let diag = DiagnosticsKvp::new(store, VM_ID, AGENT); - - // A span emits a start and a finish sharing one event_id. - diag.write_event( - RecordKind::Start, - EVENT_ID, - "provision:run", - "starting provision", - ) - .unwrap(); - diag.write_event( - RecordKind::Finish, - EVENT_ID, - "provision:run", - "provision completed", - ) - .unwrap(); - - let events = diag.events().unwrap(); + let events = decode_entries(vec![ + (format!("{first}|1"), "b".into()), + (format!("{second}|0"), "second".into()), + (format!("{first}|0"), "a".into()), + ]); assert_eq!(events.len(), 2); - - assert_eq!(events[0].kind, RecordKind::Start); - assert_eq!(events[0].message, "starting provision"); - assert_eq!(events[1].kind, RecordKind::Finish); - assert_eq!(events[1].message, "provision completed"); - - for event in &events { - assert_eq!(event.event_id, EVENT_ID); - assert_eq!(event.name, "provision:run"); - assert_eq!(event.agent, AGENT); - assert_eq!(event.vm_id.as_deref(), Some(VM_ID)); - } - } - - const CLOUD_INIT_VM_ID: &str = "0e5e179d-5341-478b-8456-fbb90621bdf8"; - const CLOUD_INIT_KEY_FINISH: &str = "CLOUD_INIT|1785187982|finish|modules-final/config-scripts_user|0e5e179d-5341-478b-8456-fbb90621bdf8|e5f01809-a7a3-4279-aa64-1f18e21eda6e"; - const CLOUD_INIT_VALUE_FINISH: &str = r#"{"name":"modules-final/config-scripts_user","type":"finish","ts":"2026-07-27T21:33:24.339006+00:00","result":"SUCCESS","duration":0.0006448590000012189,"msg":"config-scripts_user ran successfully and took 0.001 seconds"}"#; - - #[test] - fn cloud_init_key_classifies() { - assert!(matches!( - classify_key(CLOUD_INIT_KEY_FINISH), - KeyClass::CloudInit { boot_epoch, kind, name, vm_id, uuid } - if boot_epoch == 1785187982 - && kind == RecordKind::Finish - && name == "modules-final/config-scripts_user" - && vm_id == Some(CLOUD_INIT_VM_ID) - && uuid == "e5f01809-a7a3-4279-aa64-1f18e21eda6e" - )); - assert!(matches!( - classify_key( - "CLOUD_INIT|1785187982|start|modules-config/foo|\ - c4d4a08d-fe93-4c7a-9be6-9a38c212e212" - ), - KeyClass::CloudInit { boot_epoch, kind, name, vm_id, uuid } - if boot_epoch == 1785187982 - && kind == RecordKind::Start - && name == "modules-config/foo" - && vm_id.is_none() - && uuid == "c4d4a08d-fe93-4c7a-9be6-9a38c212e212" - )); + assert_eq!(events[0].name, "first"); + assert_eq!(events[0].message, "ab"); + assert_eq!(events[1].name, "second"); } #[rstest] - #[case::too_many("CLOUD_INIT|a|b|c|d|e|f", "raw")] - #[case::too_few("CLOUD_INIT|a|b|c", "raw")] - #[case::prefix_only("CLOUD_INIT", "raw")] - fn cloud_init_bad_shapes_are_raw( - #[case] key: &str, - #[case] expected: &str, + #[case(vec![(0, "a"), (2, "c")])] + #[case(vec![(0, "a"), (0, "duplicate")])] + #[case(vec![(1, "b")])] + fn incomplete_or_duplicate_indices_are_skipped( + #[case] chunks: Vec<(u32, &str)>, ) { - assert_eq!(class_of(key), expected); - } - - #[test] - fn cloud_init_finish_record_decodes_all_fields() { - assert!(matches!( - classify_record( - CLOUD_INIT_KEY_FINISH.to_string(), - vec![CLOUD_INIT_VALUE_FINISH.to_string()], - ), - DiagnosticRecord::Decoded { event, chunks: 1 } - if event.agent == "CLOUD_INIT" - && event.boot_epoch == 1785187982 - && event.kind == RecordKind::Finish - && event.name == "modules-final/config-scripts_user" - && event.vm_id.as_deref() == Some(CLOUD_INIT_VM_ID) - && event.event_id == "e5f01809-a7a3-4279-aa64-1f18e21eda6e" - && event.timestamp.as_deref() - == Some("2026-07-27T21:33:24.339006+00:00") - && event.result.as_deref() == Some("SUCCESS") - && event.duration.is_some_and(|d| { - (d - 0.000_644_859_000_001_218_9).abs() < 1e-12 - }) - && event.message - == "config-scripts_user ran successfully and took \ - 0.001 seconds" - )); - } - - #[test] - fn cloud_init_start_record_has_no_result_or_duration() { - let value = r#"{"name":"modules-final/config-keys_to_console","type":"start","ts":"2026-07-27T21:33:24.344349+00:00","msg":"running config-keys_to_console with frequency once-per-instance"}"#; - let key = "CLOUD_INIT|1785187982|start|modules-final/config-keys_to_console|0e5e179d-5341-478b-8456-fbb90621bdf8|7792621b-b339-4274-8b71-2a3dcbd2db4e"; - assert_eq!( - classify_record(key.to_string(), vec![value.to_string()]), - DiagnosticRecord::Decoded { - event: DiagnosticEvent { - agent: "CLOUD_INIT".to_string(), - boot_epoch: 1785187982, - vm_id: Some(CLOUD_INIT_VM_ID.to_string()), - kind: RecordKind::Start, - name: "modules-final/config-keys_to_console".to_string(), - event_id: "7792621b-b339-4274-8b71-2a3dcbd2db4e" - .to_string(), - timestamp: Some( - "2026-07-27T21:33:24.344349+00:00".to_string() - ), - result: None, - duration: None, - message: "running config-keys_to_console with frequency \ - once-per-instance" - .to_string(), - }, - chunks: 1, - } - ); - } - - #[test] - fn cloud_init_non_span_type_decodes_to_other_kind() { - let key = format!( - "CLOUD_INIT|1785187982|compressed|cloud-init.log|\ - {CLOUD_INIT_VM_ID}|abc12345-1111-2222-3333-444455556666" + let key = format_event_key( + AGENT, 100, VM_ID, "event", "name", EVENT_ID, TIMESTAMP, ); - let value = r#"{"name":"cloud-init.log","type":"compressed","ts":"2026-07-27T21:33:24.339006+00:00","msg":"payload"}"#; - assert!(matches!( - classify_record(key, vec![value.to_string()]), - DiagnosticRecord::Decoded { event, chunks: 1 } - if event.kind == RecordKind::Other("compressed".to_string()) - && event.agent == "CLOUD_INIT" - && event.message == "payload" - )); - } - - #[test] - fn cloud_init_key_with_invalid_json_is_malformed() { - assert!(matches!( - classify_record( - CLOUD_INIT_KEY_FINISH.to_string(), - vec!["not json".to_string()], - ), - DiagnosticRecord::Malformed { reason, .. } - if reason.contains("cloud-init") - )); + let dumped = chunks + .into_iter() + .map(|(index, value)| (format!("{key}|{index}"), value.into())) + .collect(); + assert!(decode_entries(dumped).is_empty()); } #[test] - fn cloud_init_chunks_reassemble_by_subevent_index() { - let base = "CLOUD_INIT|1785187982|finish|modules-final/long|0e5e179d-5341-478b-8456-fbb90621bdf8|abc12345-1111-2222-3333-444455556666"; - let chunk = |i: u32, msg: &str| { - format!( - r#"{{"name":"modules-final/long","type":"finish","ts":"2026-07-27T21:33:24.339006+00:00","result":"SUCCESS","duration":0.5,"msg_i":{i},"msg":"{msg}"}}"# - ) - }; - let dumped = vec![ - (format!("{base}|1"), chunk(1, "two ")), - (format!("{base}|0"), chunk(0, "one ")), - (format!("{base}|2"), chunk(2, "three")), - ]; - - let records = reassemble(dumped); - assert_eq!( - records, - vec![DiagnosticRecord::Decoded { - event: DiagnosticEvent { - agent: "CLOUD_INIT".to_string(), - boot_epoch: 1785187982, - vm_id: Some(CLOUD_INIT_VM_ID.to_string()), - kind: RecordKind::Finish, - name: "modules-final/long".to_string(), - event_id: "abc12345-1111-2222-3333-444455556666" - .to_string(), - timestamp: Some( - "2026-07-27T21:33:24.339006+00:00".to_string() - ), - result: Some("SUCCESS".to_string()), - duration: Some(0.5), - message: "one two three".to_string(), - }, - chunks: 3, - }] - ); + fn cloud_init_msg_i_must_match_key_index() { + let base = "CLOUD_INIT|100|event|name|vm-id|event-id"; + let value = r#"{"name":"name","type":"event","ts":"2026-07-27T21:33:24Z","msg_i":1,"msg":"value"}"#; + assert!(decode_entries(vec![(format!("{base}|0"), value.into())]) + .is_empty()); } #[test] - fn cloud_init_chunks_reassemble_split_json_escape() { - let base = "CLOUD_INIT|1785187982|finish|modules-final/x|0e5e179d-5341-478b-8456-fbb90621bdf8|abc12345-1111-2222-3333-444455556666"; - let dumped = vec![ + fn cloud_init_split_escape_is_unescaped_after_reassembly() { + let base = "CLOUD_INIT|100|finish|name|vm-id|event-id"; + let events = decode_entries(vec![ ( format!("{base}|0"), - r#"{"name":"modules-final/x","type":"finish","ts":"2026-07-27T21:33:24.339006+00:00","result":"SUCCESS","duration":0.5,"msg_i":0,"msg":"line1\"}"# - .to_string(), + r#"{"name":"name","type":"finish","ts":"2026-07-27T21:33:24Z","msg_i":0,"msg":"line1\"}"# + .into(), ), ( format!("{base}|1"), - r#"{"name":"modules-final/x","type":"finish","ts":"2026-07-27T21:33:24.339006+00:00","result":"SUCCESS","duration":0.5,"msg_i":1,"msg":"nline2"}"# - .to_string(), + r#"{"name":"name","type":"finish","ts":"2026-07-27T21:33:24Z","msg_i":1,"msg":"nline2"}"# + .into(), ), - ]; - - let records = reassemble(dumped); - assert!(matches!( - &records[..], - [DiagnosticRecord::Decoded { event, chunks: 2 }] - if event.message == "line1\nline2" - && event.result.as_deref() == Some("SUCCESS") - && event.duration == Some(0.5) - )); + ]); + assert_eq!(events.len(), 1); + assert_eq!(events[0].message, "line1\nline2"); } } diff --git a/libazureinit-kvp/src/error.rs b/libazureinit-kvp/src/error.rs index 0fb171d0..405913e4 100644 --- a/libazureinit-kvp/src/error.rs +++ b/libazureinit-kvp/src/error.rs @@ -11,7 +11,7 @@ pub enum KvpError { EmptyKey, /// An underlying I/O error. Io(io::Error), - /// An event key field (`agent`, `vm_id`, `name`, or `event_id`) + /// An event key field (`agent`, `vm_id`, `kind`, `name`, or `event_id`) /// contained the `|` delimiter, which would make the formatted event /// key ambiguous to parse back. EventFieldContainsDelimiter { field: &'static str }, diff --git a/libazureinit-kvp/src/lib.rs b/libazureinit-kvp/src/lib.rs index 197093f0..204d68ed 100644 --- a/libazureinit-kvp/src/lib.rs +++ b/libazureinit-kvp/src/lib.rs @@ -9,8 +9,8 @@ //! - [`ProvisioningReport`]: structured provisioning health report that //! is persisted as the single `PROVISIONING_REPORT` record with //! [`write_report`]. -//! - [`DiagnosticsKvp`]: typed view over [`KvpPoolStore`] that formats, -//! chunks, and reassembles azure-init diagnostic events. +//! - [`DiagnosticsKvp`]: typed writer for azure-init diagnostics and normalized +//! reader for azure-init and cloud-init entries. mod cli; mod diagnostics; @@ -21,8 +21,7 @@ mod vm_id; pub use cli::run; pub use diagnostics::{ - DiagnosticEvent, DiagnosticRecord, DiagnosticsKvp, RecordKind, - MAX_CHUNK_BYTES, + DiagnosticEvent, DiagnosticKind, DiagnosticsKvp, MAX_CHUNK_BYTES, }; pub use error::KvpError; pub use report::{ diff --git a/libazureinit-kvp/tests/cli.rs b/libazureinit-kvp/tests/cli.rs index 6c6afd03..2f4b90c5 100644 --- a/libazureinit-kvp/tests/cli.rs +++ b/libazureinit-kvp/tests/cli.rs @@ -297,14 +297,14 @@ fn report_failure_rejects_invalid_supporting_data() { } #[test] -fn dump_parse_diagnostics_json_reassembles_and_classifies() { +fn dump_parse_diagnostics_json_reassembles_and_skips_raw() { let dir = TempDir::new().unwrap(); assert_success(kvp(&with_dir( &dir, &[ "write", "--append", - "azure-init-x|100|vm|event|a:b|id1|ts", + "azure-init-x|100|vm|event|a:b|id1|2026-08-31T00:00:00Z|0", "one/", ], ))); @@ -313,7 +313,7 @@ fn dump_parse_diagnostics_json_reassembles_and_classifies() { &[ "write", "--append", - "azure-init-x|100|vm|event|a:b|id1|ts", + "azure-init-x|100|vm|event|a:b|id1|2026-08-31T00:00:00Z|1", "two", ], ))); @@ -327,16 +327,8 @@ fn dump_parse_diagnostics_json_reassembles_and_classifies() { &["--json", "dump", "--parse-diagnostics"], ))); assert!(out.contains("\"kind\":\"event\"")); - assert!(out.contains("\"chunks\":2")); assert!(out.contains("\"message\":\"one/two\"")); assert!(!out.contains("PROVISIONING_REPORT")); - - let out_raw = assert_success(kvp(&with_dir( - &dir, - &["--json", "dump", "--parse-diagnostics", "--include-raw"], - ))); - assert!(out_raw.contains("\"record\":\"raw\"")); - assert!(out_raw.contains("PROVISIONING_REPORT")); } #[test] @@ -369,9 +361,8 @@ fn dump_parse_diagnostics_text_renders_cloud_init_event() { assert!(out.contains("name=modules-final/config-scripts_user")); assert!(out.contains("vm_id=0e5e179d-5341-478b-8456-fbb90621bdf8")); assert!(out.contains("result=SUCCESS")); - assert!(out.contains("timestamp=2026-07-27T21:33:24.339006+00:00")); + assert!(out.contains("timestamp=2026-07-27 21:33:24.339006 UTC")); assert!(out.contains("duration=0.5")); - assert!(out.contains("chunks=1")); assert!(out.contains("message=scripts ran")); assert!(out.contains("event kind=start")); } @@ -393,7 +384,6 @@ fn dump_parse_diagnostics_json_renders_cloud_init_event() { &dir, &["--json", "dump", "--parse-diagnostics"], ))); - assert!(out.contains("\"record\":\"event\"")); assert!(out.contains("\"kind\":\"finish\"")); assert!(out.contains("\"agent\":\"CLOUD_INIT\"")); assert!(out.contains("\"boot_epoch\":1785187982")); @@ -402,53 +392,38 @@ fn dump_parse_diagnostics_json_renders_cloud_init_event() { assert!( out.contains("\"event_id\":\"e5f01809-a7a3-4279-aa64-1f18e21eda6e\"") ); - assert!(out.contains("\"timestamp\":\"2026-07-27T21:33:24.339006+00:00\"")); + assert!(out.contains("\"timestamp\":\"2026-07-27T21:33:24.339006Z\"")); assert!(out.contains("\"result\":\"SUCCESS\"")); assert!(out.contains("\"duration\":0.5")); - assert!(out.contains("\"chunks\":1")); assert!(out.contains("\"message\":\"scripts ran\"")); } #[test] -fn clear_diagnostics_removes_events_and_malformed_keeps_raw() { - let dir = TempDir::new().unwrap(); - assert_success(kvp(&with_dir( - &dir, - &["write", "--append", "a|100|vm|event|a:b|i1|ts", "msg"], - ))); - assert_success(kvp(&with_dir( - &dir, - &["write", "--append", "a|100|vm|NOPE|c:d|i2|ts", "junk"], - ))); - assert_success(kvp(&with_dir( - &dir, - &["write", "PROVISIONING_REPORT", "result=success"], - ))); - - assert_success(kvp(&with_dir(&dir, &["clear", "--diagnostics"]))); - - let out = assert_success(kvp(&with_dir(&dir, &["dump"]))); - assert_eq!(out, "PROVISIONING_REPORT=result=success\n"); -} - -#[test] -fn clear_diagnostics_conflicts_with_if_stale() { - let dir = TempDir::new().unwrap(); - let output = - kvp(&with_dir(&dir, &["clear", "--diagnostics", "--if-stale"])); +fn clear_diagnostics_option_is_not_exposed() { + let output = kvp(&["clear", "--diagnostics"]); assert_eq!(output.status.code(), Some(2)); } #[test] -fn dump_parse_diagnostics_text_renders_all_record_kinds() { +fn dump_parse_diagnostics_text_renders_only_normalized_entries() { let dir = TempDir::new().unwrap(); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "a|100|vm|event|a:b|id1|ts", "one/"], + &[ + "write", + "--append", + "a|100|vm|event|a:b|id1|2026-08-31T00:00:00Z|0", + "one/", + ], ))); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "a|100|vm|event|a:b|id1|ts", "two"], + &[ + "write", + "--append", + "a|100|vm|event|a:b|id1|2026-08-31T00:00:00Z|1", + "two", + ], ))); assert_success(kvp(&with_dir( &dir, @@ -456,20 +431,23 @@ fn dump_parse_diagnostics_text_renders_all_record_kinds() { ))); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "a|100|vm|NOPE|c:d|id2|ts", "junk"], + &[ + "write", + "--append", + "a|not-a-boot|vm|event|c:d|id2|2026-08-31T00:00:00Z", + "junk", + ], ))); - let out = assert_success(kvp(&with_dir( - &dir, - &["dump", "--parse-diagnostics", "--include-raw"], - ))); + let out = + assert_success(kvp(&with_dir(&dir, &["dump", "--parse-diagnostics"]))); assert!(out.contains( "event kind=event agent=a boot_epoch=100 vm_id=vm \ - name=a:b event_id=id1 timestamp=ts chunks=2 message=one/two" + name=a:b event_id=id1 timestamp=2026-08-31 00:00:00 UTC \ + message=one/two" )); - assert!(out.contains("raw key=PROVISIONING_REPORT value=result=success")); - assert!(out.contains("malformed key=a|100|vm|NOPE|c:d|id2|ts")); - assert!(out.contains("value=junk")); + assert!(!out.contains("PROVISIONING_REPORT")); + assert!(!out.contains("junk")); } #[test] @@ -477,11 +455,21 @@ fn dump_parse_diagnostics_tail_limits_to_last_events() { let dir = TempDir::new().unwrap(); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "a|100|vm|event|a:b|i1|ts", "first"], + &[ + "write", + "--append", + "a|100|vm|event|a:b|i1|2026-08-31T00:00:00Z", + "first", + ], ))); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "a|100|vm|event|c:d|i2|ts", "second"], + &[ + "write", + "--append", + "a|100|vm|event|c:d|i2|2026-08-31T00:00:01Z", + "second", + ], ))); let out = assert_success(kvp(&with_dir( @@ -501,7 +489,7 @@ fn dump_parse_diagnostics_tail_defaults_to_20_when_count_omitted() { &[ "write", "--append", - &format!("a|100|vm|event|n:{i}|id{i}|ts"), + &format!("a|100|vm|event|n:{i}|id{i}|2026-08-31T00:00:00Z"), &format!("msg{i}"), ], ))); @@ -522,11 +510,21 @@ fn dump_parse_diagnostics_filters_by_name_substring() { let dir = TempDir::new().unwrap(); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "a|100|vm|event|user:add|i1|ts", "u"], + &[ + "write", + "--append", + "a|100|vm|event|user:add|i1|2026-08-31T00:00:00Z", + "u", + ], ))); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "a|100|vm|event|ssh:key|i2|ts", "s"], + &[ + "write", + "--append", + "a|100|vm|event|ssh:key|i2|2026-08-31T00:00:01Z", + "s", + ], ))); let out = assert_success(kvp(&with_dir( @@ -538,31 +536,31 @@ fn dump_parse_diagnostics_filters_by_name_substring() { } #[test] -fn dump_parse_diagnostics_include_raw_conflicts_with_filters() { - let dir = TempDir::new().unwrap(); - let output = kvp(&with_dir( - &dir, - &[ - "dump", - "--parse-diagnostics", - "--include-raw", - "--name", - "a:b", - ], - )); +fn dump_parse_diagnostics_include_raw_option_is_not_exposed() { + let output = kvp(&["dump", "--parse-diagnostics", "--include-raw"]); assert_eq!(output.status.code(), Some(2)); } #[test] -fn dump_parse_diagnostics_json_covers_events_and_malformed() { +fn dump_parse_diagnostics_json_skips_malformed_entries() { let dir = TempDir::new().unwrap(); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "a|100|vm|event|a:b|i1|ts", "hello"], + &[ + "write", + "--append", + "a|100|vm|event|a:b|i1|2026-08-31T00:00:00Z", + "hello", + ], ))); assert_success(kvp(&with_dir( &dir, - &["write", "--append", "a|100|vm|NOPE|c:d|i2|ts", "junk"], + &[ + "write", + "--append", + "a|not-a-boot|vm|event|c:d|i2|2026-08-31T00:00:00Z", + "junk", + ], ))); let dump = assert_success(kvp(&with_dir( @@ -570,15 +568,14 @@ fn dump_parse_diagnostics_json_covers_events_and_malformed() { &["--json", "dump", "--parse-diagnostics"], ))); assert!(dump.contains("\"kind\":\"event\"")); - assert!(dump.contains("\"record\":\"malformed\"")); - assert!(dump.contains("\"reason\":")); + assert!(!dump.contains("junk")); let events = assert_success(kvp(&with_dir( &dir, &["--json", "dump", "--parse-diagnostics", "--name", "a:b"], ))); assert!(events.contains("\"message\":\"hello\"")); - assert!(!events.contains("malformed")); + assert!(!events.contains("junk")); } #[test] diff --git a/libazureinit-kvp/tests/diagnostics.rs b/libazureinit-kvp/tests/diagnostics.rs index d763e404..de53c2ed 100644 --- a/libazureinit-kvp/tests/diagnostics.rs +++ b/libazureinit-kvp/tests/diagnostics.rs @@ -2,15 +2,15 @@ // Licensed under the MIT License. //! Integration tests for the [`DiagnosticsKvp`] layer: emit/read -//! round-trips, chunk reassembly, classification, scoped clearing, and -//! the concurrent-write atomicity guarantee that keeps chunked events -//! from interleaving. +//! round-trips, cloud-init normalization, chunk reassembly, and concurrent +//! writes. use std::thread; +use chrono::{DateTime, Utc}; use libazureinit_kvp::{ - DiagnosticEvent, DiagnosticRecord, DiagnosticsKvp, KvpPool, KvpPoolStore, - PoolMode, RecordKind, MAX_CHUNK_BYTES, + DiagnosticEvent, DiagnosticKind, DiagnosticsKvp, KvpPool, KvpPoolStore, + PoolMode, MAX_CHUNK_BYTES, }; use rstest::rstest; use tempfile::TempDir; @@ -22,7 +22,7 @@ fn diagnostics(dir: &TempDir) -> DiagnosticsKvp { let store = KvpPoolStore::new_in(KvpPool::Guest, dir.path(), PoolMode::Safe) .unwrap(); - DiagnosticsKvp::new(store, VM_ID, PREFIX) + DiagnosticsKvp::new(store, VM_ID, PREFIX).unwrap() } /// Real cloud-init reporting entries captured from a guest pool 1 file. @@ -52,41 +52,27 @@ fn reads_and_parses_real_cloud_init_pool() { store.append(key, value).unwrap(); } - let diagnostics = DiagnosticsKvp::new(store, "", ""); - let records = diagnostics.records().unwrap(); - assert_eq!(records.len(), CLOUD_INIT_RECORDS.len()); + let diagnostics = DiagnosticsKvp::new(store, "", "").unwrap(); + let entries = diagnostics.entries().unwrap(); + assert_eq!(entries.len(), CLOUD_INIT_RECORDS.len()); - for record in &records { - assert!(matches!(record, DiagnosticRecord::Decoded { .. })); - } - - match &records[0] { - DiagnosticRecord::Decoded { event, chunks } => { - assert_eq!(*chunks, 1); - assert_eq!(event.agent, "CLOUD_INIT"); - assert_eq!(event.kind, RecordKind::Finish); - assert_eq!(event.name, "modules-final/config-scripts_user"); - assert_eq!( - event.vm_id.as_deref(), - Some("0e5e179d-5341-478b-8456-fbb90621bdf8") - ); - assert_eq!(event.result.as_deref(), Some("SUCCESS")); - assert_eq!( - event.message, - "config-scripts_user ran successfully and took 0.001 seconds" - ); - } - other => panic!("expected event, got {other:?}"), - } + let event = &entries[0]; + assert_eq!(event.agent, "CLOUD_INIT"); + assert_eq!(event.kind, DiagnosticKind::Finish); + assert_eq!(event.name, "modules-final/config-scripts_user"); + assert_eq!( + event.vm_id.as_deref(), + Some("0e5e179d-5341-478b-8456-fbb90621bdf8") + ); + assert_eq!(event.result.as_deref(), Some("SUCCESS")); + assert_eq!( + event.message, + "config-scripts_user ran successfully and took 0.001 seconds" + ); - match &records[1] { - DiagnosticRecord::Decoded { event, .. } => { - assert_eq!(event.kind, RecordKind::Start); - assert!(event.result.is_none()); - assert!(event.duration.is_none()); - } - other => panic!("expected event, got {other:?}"), - } + assert_eq!(entries[1].kind, DiagnosticKind::Start); + assert!(entries[1].result.is_none()); + assert!(entries[1].duration.is_none()); } #[test] @@ -103,29 +89,17 @@ fn short_event_round_trips_as_single_record() { assert_eq!(dumped.len(), 1); assert!(dumped[0].0.ends_with("|0"), "key: {}", dumped[0].0); - let records = diag.records().unwrap(); - assert_eq!(records.len(), 1); - match &records[0] { - DiagnosticRecord::Decoded { - event: decoded, - chunks, - } => { - assert_eq!(*chunks, 1); - assert_eq!(decoded.kind, RecordKind::Event); - assert_eq!(decoded.vm_id.as_deref(), Some(VM_ID)); - assert_eq!(decoded.boot_epoch, diag.store().boot_epoch().unwrap()); - assert_eq!(decoded.name, "user:create_user"); - let event_id = uuid::Uuid::parse_str(&decoded.event_id) - .expect("event_id should be a valid UUID"); - assert_eq!( - event_id.get_version_num(), - 4, - "event_id should be a UUIDv4" - ); - assert_eq!(decoded.message, "created"); - } - other => panic!("expected event, got {other:?}"), - } + let entries = diag.entries().unwrap(); + assert_eq!(entries.len(), 1); + let decoded = &entries[0]; + assert_eq!(decoded.kind, DiagnosticKind::Event); + assert_eq!(decoded.vm_id.as_deref(), Some(VM_ID)); + assert_eq!(decoded.boot_epoch, diag.boot_epoch()); + assert_eq!(decoded.name, "user:create_user"); + let event_id = uuid::Uuid::parse_str(&decoded.event_id) + .expect("event_id should be a valid UUID"); + assert_eq!(event_id.get_version_num(), 4, "event_id should be a UUIDv4"); + assert_eq!(decoded.message, "created"); } #[test] @@ -149,18 +123,9 @@ fn long_event_splits_across_records_and_reassembles() { keys.dedup(); assert_eq!(keys.len(), 4, "each chunk must have a unique key"); - let records = diag.records().unwrap(); - assert_eq!(records.len(), 1); - match &records[0] { - DiagnosticRecord::Decoded { - event: decoded, - chunks, - } => { - assert_eq!(*chunks, 4); - assert_eq!(decoded.message, message); - } - other => panic!("expected event, got {other:?}"), - } + let entries = diag.entries().unwrap(); + assert_eq!(entries.len(), 1); + assert_eq!(entries[0].message, message); } #[test] @@ -179,26 +144,24 @@ fn multi_chunk_event_uses_unique_keys_so_host_keeps_all() { keys.dedup(); assert_eq!(keys.len(), total, "chunk keys must be unique"); - let events = diag.events().unwrap(); + let events = diag.entries().unwrap(); assert_eq!(events.len(), 1); assert_eq!(events[0].message, message); } #[test] -fn injected_malformed_key_is_classified() { +fn raw_and_malformed_records_are_skipped() { let dir = TempDir::new().unwrap(); let diag = diagnostics(&dir); diag.store() .append(&format!("{PREFIX}|100|{VM_ID}|NOPE|bad:kind|id|ts"), "junk") .unwrap(); + diag.store() + .append("PROVISIONING_REPORT", "result=success") + .unwrap(); - let records = diag.records().unwrap(); - assert_eq!(records.len(), 1); - assert!(matches!( - &records[0], - DiagnosticRecord::Malformed { reason, .. } if reason.contains("NOPE") - )); + assert!(diag.entries().unwrap().is_empty()); } #[test] @@ -216,65 +179,87 @@ fn mixed_records_round_trip_together() { .append(&format!("{PREFIX}|100|{VM_ID}|NOPE|e:f|id|ts"), "junk") .unwrap(); - let records = diag.records().unwrap(); - assert_eq!(records.len(), 4); - assert_eq!(diag.events().unwrap().len(), 2); + let entries = diag.entries().unwrap(); + assert_eq!(entries.len(), 2); } #[test] -fn clear_removes_events_but_keeps_raw() { +fn explicit_diagnostic_kinds_are_tracing_ready() { let dir = TempDir::new().unwrap(); let diag = diagnostics(&dir); - - diag.emit_event("a:b", "e1").unwrap(); - diag.emit_event("c:d", "z".repeat(MAX_CHUNK_BYTES * 2)) - .unwrap(); - diag.store() - .append("PROVISIONING_REPORT", "result=success") - .unwrap(); - - diag.clear().unwrap(); - - let records = diag.records().unwrap(); - assert_eq!(records.len(), 1); - assert!(matches!( - &records[0], - DiagnosticRecord::Raw { key, .. } if key == "PROVISIONING_REPORT" - )); - assert!(diag.events().unwrap().is_empty()); + let timestamp = DateTime::parse_from_rfc3339("2026-08-31T12:34:56.789Z") + .unwrap() + .with_timezone(&Utc); + + diag.emit( + DiagnosticKind::Start, + "provision:run", + "shared-span-id", + timestamp, + "starting", + ) + .unwrap(); + diag.emit( + DiagnosticKind::Event, + "provision:run", + "shared-span-id", + timestamp, + "progress", + ) + .unwrap(); + diag.emit( + DiagnosticKind::Finish, + "provision:run", + "shared-span-id", + timestamp, + "finished", + ) + .unwrap(); + diag.emit( + DiagnosticKind::Other("diagnostic".to_string()), + "support:bundle", + "diagnostic-id", + timestamp, + "collected", + ) + .unwrap(); + + let entries = diag.entries().unwrap(); + assert_eq!(entries.len(), 4); + assert_eq!(entries[0].kind, DiagnosticKind::Start); + assert_eq!(entries[1].kind, DiagnosticKind::Event); + assert_eq!(entries[2].kind, DiagnosticKind::Finish); + assert_eq!( + entries[3].kind, + DiagnosticKind::Other("diagnostic".to_string()) + ); + assert!(entries[..3] + .iter() + .all(|entry| entry.event_id == "shared-span-id")); + assert!(entries.iter().all(|entry| entry.timestamp == timestamp)); } #[test] -fn clear_removes_all_diagnostics_regardless_of_scope() { +fn chunked_entries_survive_store_swap_deletion() { let dir = TempDir::new().unwrap(); let diag = diagnostics(&dir); - - diag.emit_event("a:b", "mine").unwrap(); - diag.store() - .append("other-agent|100|other-vm|event|x:y|id|ts", "theirs") - .unwrap(); - diag.store() - .append("p|100|vm|NOPE|c:d|id|ts", "junk") - .unwrap(); - diag.store() - .append("p|100|vm|NOPE|c:d|id|ts|0", "junk-0") - .unwrap(); - diag.store() - .append("p|100|vm|NOPE|c:d|id|ts|1", "junk-1") - .unwrap(); - diag.store() - .append("PROVISIONING_REPORT", "result=success") - .unwrap(); - - diag.clear().unwrap(); - - let records = diag.records().unwrap(); - assert_eq!(records.len(), 1); - assert!(matches!( - &records[0], - DiagnosticRecord::Raw { key, .. } if key == "PROVISIONING_REPORT" - )); - assert!(diag.events().unwrap().is_empty()); + let first_message = "a".repeat(MAX_CHUNK_BYTES * 2 + 7); + let second_message = "b".repeat(MAX_CHUNK_BYTES * 2 + 7); + + diag.emit_event("first", &first_message).unwrap(); + diag.store().append("remove-me", "raw").unwrap(); + diag.emit_event("second", &second_message).unwrap(); + + // Deletion moves the final record into the removed slot, so the second + // event's chunks are no longer adjacent or in index order. + assert!(diag.store().delete("remove-me").unwrap()); + + let entries = diag.entries().unwrap(); + assert_eq!(entries.len(), 2); + assert_eq!(entries[0].name, "first"); + assert_eq!(entries[0].message, first_message); + assert_eq!(entries[1].name, "second"); + assert_eq!(entries[1].message, second_message); } #[test] @@ -312,7 +297,7 @@ fn concurrent_multichunk_emits_reassemble_without_interleaving() { handle.join().unwrap(); } - let events = diag.events().unwrap(); + let events = diag.entries().unwrap(); assert_eq!(events.len(), THREADS * PER_THREAD); for event in &events { assert_eq!(event.message.len(), len); @@ -358,10 +343,10 @@ fn without_vm_id(current_key: &str) -> String { segments.join("|") } -/// Append the given records to a fresh guest pool and classify them. -fn records_of, V: AsRef>( +/// Append the given records to a fresh guest pool and normalize them. +fn entries_of, V: AsRef>( pairs: &[(K, V)], -) -> Vec { +) -> Vec { let dir = TempDir::new().unwrap(); let store = KvpPoolStore::new_in(KvpPool::Guest, dir.path(), PoolMode::Safe) @@ -369,46 +354,45 @@ fn records_of, V: AsRef>( for (key, value) in pairs { store.append(key.as_ref(), value.as_ref()).unwrap(); } - DiagnosticsKvp::new(store, "", "").records().unwrap() + DiagnosticsKvp::new(store, "", "") + .unwrap() + .entries() + .unwrap() } -/// Expect exactly one decoded event, returning it with its chunk count. -fn decode_single(records: Vec) -> (DiagnosticEvent, usize) { - assert_eq!(records.len(), 1, "expected one record, got: {records:?}"); - match records.into_iter().next().unwrap() { - DiagnosticRecord::Decoded { event, chunks } => (event, chunks), - other => panic!("expected a decoded event, got: {other:?}"), - } +fn decode_single(entries: Vec) -> DiagnosticEvent { + assert_eq!(entries.len(), 1, "expected one entry, got: {entries:?}"); + entries.into_iter().next().unwrap() } #[rstest] -#[case::start("start", "azure-ds", RecordKind::Start)] -#[case::finish("finish", "azure-ds/get-metadata", RecordKind::Finish)] -#[case::event("event", "user:create_user", RecordKind::Event)] +#[case::start("start", "azure-ds", DiagnosticKind::Start)] +#[case::finish("finish", "azure-ds/get-metadata", DiagnosticKind::Finish)] +#[case::event("event", "user:create_user", DiagnosticKind::Event)] #[case::diagnostic( "diagnostic", "diagnostic message", - RecordKind::Other("diagnostic".to_string()) + DiagnosticKind::Other("diagnostic".to_string()) )] #[case::compressed( "compressed", "cloud-init.log", - RecordKind::Other("compressed".to_string()) + DiagnosticKind::Other("compressed".to_string()) )] #[case::boot_telemetry( "boot-telemetry", "boot-telemetry", - RecordKind::Other("boot-telemetry".to_string()) + DiagnosticKind::Other("boot-telemetry".to_string()) )] #[case::system_info( "system-info", "system information", - RecordKind::Other("system-info".to_string()) + DiagnosticKind::Other("system-info".to_string()) )] fn cloud_init_type_decodes_in_both_layouts( #[case] event_type: &str, #[case] name: &str, - #[case] expected: RecordKind, + #[case] expected: DiagnosticKind, ) { const TS: &str = "2026-08-06T20:20:13.479078Z"; const UUID: &str = "b7a822ba-4eea-46c0-b559-e84396101132"; @@ -420,15 +404,14 @@ fn cloud_init_type_decodes_in_both_layouts( let old_key = format!("CLOUD_INIT|1786047606|{event_type}|{name}|{UUID}"); let current_key = with_vm_id(&old_key, CLOUD_INIT_VM_ID); - let (event, chunks) = decode_single(records_of(&[(&old_key, &value)])); - assert_eq!(chunks, 1); + let event = decode_single(entries_of(&[(&old_key, &value)])); assert_eq!(event.agent, "CLOUD_INIT"); assert_eq!(event.kind, expected); assert_eq!(event.vm_id, None); assert_eq!(event.name, name); assert_eq!(event.message, msg); - let (event, _) = decode_single(records_of(&[(¤t_key, &value)])); + let event = decode_single(entries_of(&[(¤t_key, &value)])); assert_eq!(event.kind, expected); assert_eq!(event.vm_id.as_deref(), Some(CLOUD_INIT_VM_ID)); assert_eq!(event.message, msg); @@ -443,8 +426,8 @@ fn cloud_init_finish_reports_result_and_duration_in_both_layouts() { b7a822ba-4eea-46c0-b559-e84396101132"; for key in [old_key.to_string(), with_vm_id(old_key, CLOUD_INIT_VM_ID)] { - let (event, _) = decode_single(records_of(&[(key.as_str(), value)])); - assert_eq!(event.kind, RecordKind::Finish); + let event = decode_single(entries_of(&[(key.as_str(), value)])); + assert_eq!(event.kind, DiagnosticKind::Finish); assert_eq!(event.result.as_deref(), Some("SUCCESS")); assert_eq!(event.duration, Some(0.1234)); } @@ -453,19 +436,17 @@ fn cloud_init_finish_reports_result_and_duration_in_both_layouts() { #[test] fn real_cloud_init_samples_decode_without_vm_id_too() { for &(key, value) in CLOUD_INIT_RECORDS { - let (event, _) = - decode_single(records_of(&[(without_vm_id(key), value)])); + let event = decode_single(entries_of(&[(without_vm_id(key), value)])); assert!(event.vm_id.is_none(), "stripped sample kept a vm_id: {key}"); } - let (event, _) = decode_single(records_of(&[CLOUD_INIT_RECORDS[0]])); + let event = decode_single(entries_of(&[CLOUD_INIT_RECORDS[0]])); assert_eq!(event.vm_id.as_deref(), Some(CLOUD_INIT_VM_ID)); } #[test] fn old_compressed_log_reassembles_across_chunks() { - let (event, chunks) = decode_single(records_of(COMPRESSED_LOG_CHUNKS)); - assert_eq!(chunks, 3, "the three chunks must regroup into one event"); - assert_eq!(event.kind, RecordKind::Other("compressed".to_string())); + let event = decode_single(entries_of(COMPRESSED_LOG_CHUNKS)); + assert_eq!(event.kind, DiagnosticKind::Other("compressed".to_string())); assert_eq!(event.vm_id, None); assert_eq!(event.name, "cloud-init.log"); assert_eq!(event.message, EXPECTED_COMPRESSED_MSG); @@ -477,28 +458,52 @@ fn current_compressed_log_reassembles_across_chunks() { .iter() .map(|&(key, value)| (with_vm_id(key, CLOUD_INIT_VM_ID), value)) .collect(); - let (event, chunks) = decode_single(records_of(¤t)); - assert_eq!(chunks, 3); - assert_eq!(event.kind, RecordKind::Other("compressed".to_string())); + let event = decode_single(entries_of(¤t)); + assert_eq!(event.kind, DiagnosticKind::Other("compressed".to_string())); assert_eq!(event.vm_id.as_deref(), Some(CLOUD_INIT_VM_ID)); assert_eq!(event.message, EXPECTED_COMPRESSED_MSG); } #[test] -fn cloud_init_event_with_invalid_json_is_still_flagged() { +fn cloud_init_event_with_invalid_json_is_skipped() { let key = format!( "CLOUD_INIT|1786047606|compressed|cloud-init.log|{CLOUD_INIT_VM_ID}|\ b7a822ba-4eea-46c0-b559-e84396101132" ); - let records = records_of(&[(key.as_str(), "not-json")]); - assert_eq!(records.len(), 1); - assert!( - matches!( - &records[0], - DiagnosticRecord::Malformed { reason, .. } - if reason.contains("invalid cloud-init JSON") - ), - "expected a Malformed record, got: {:?}", - records[0] - ); + assert!(entries_of(&[(key.as_str(), "not-json")]).is_empty()); +} + +#[test] +fn cloud_init_chunk_index_mismatch_is_skipped() { + let base = "CLOUD_INIT|1786047606|event|test|\ + b7a822ba-4eea-46c0-b559-e84396101132"; + let value = r#"{"name":"test","type":"event","ts":"2026-08-06T20:20:13Z","msg_i":1,"msg":"payload"}"#; + + assert!(entries_of(&[(format!("{base}|0"), value)]).is_empty()); +} + +#[test] +fn cloud_init_chunk_without_key_index_is_skipped() { + let key = "CLOUD_INIT|1786047606|event|test|\ + b7a822ba-4eea-46c0-b559-e84396101132"; + let value = r#"{"name":"test","type":"event","ts":"2026-08-06T20:20:13Z","msg_i":0,"msg":"partial"}"#; + + assert!(entries_of(&[(key, value)]).is_empty()); +} + +#[test] +fn incomplete_cloud_init_chunk_group_is_skipped() { + let base = "CLOUD_INIT|1786047606|event|test|\ + b7a822ba-4eea-46c0-b559-e84396101132"; + let chunk = |index: u32, message: &str| { + format!( + r#"{{"name":"test","type":"event","ts":"2026-08-06T20:20:13Z","msg_i":{index},"msg":"{message}"}}"# + ) + }; + let records = vec![ + (format!("{base}|0"), chunk(0, "first")), + (format!("{base}|2"), chunk(2, "third")), + ]; + + assert!(entries_of(&records).is_empty()); } From a199587c45acf0d88da7d49348e82922c71d146d Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Tue, 1 Sep 2026 09:52:18 -0700 Subject: [PATCH 12/32] fix(kvp): decode fixed-width fields as C strings` --- libazureinit-kvp/src/store.rs | 33 +++++++++++++++++++++++++++++---- 1 file changed, 29 insertions(+), 4 deletions(-) diff --git a/libazureinit-kvp/src/store.rs b/libazureinit-kvp/src/store.rs index 64152a28..156228fc 100644 --- a/libazureinit-kvp/src/store.rs +++ b/libazureinit-kvp/src/store.rs @@ -662,13 +662,19 @@ fn decode_record(data: &[u8]) -> io::Result<(String, String)> { let (key_bytes, value_bytes) = data.split_at(WIRE_MAX_KEY_BYTES); - let key = std::str::from_utf8(key_bytes) + let key_end = key_bytes + .iter() + .position(|byte| *byte == 0) + .unwrap_or(key_bytes.len()); + let key = std::str::from_utf8(&key_bytes[..key_end]) .map_err(|e| io::Error::new(ErrorKind::InvalidData, e))? - .trim_end_matches('\0') .to_string(); - let value = std::str::from_utf8(value_bytes) + let value_end = value_bytes + .iter() + .position(|byte| *byte == 0) + .unwrap_or(value_bytes.len()); + let value = std::str::from_utf8(&value_bytes[..value_end]) .map_err(|e| io::Error::new(ErrorKind::InvalidData, e))? - .trim_end_matches('\0') .to_string(); Ok((key, value)) @@ -2145,6 +2151,25 @@ mod tests { assert_eq!(v, "val"); } + #[rstest] + #[case::key(Field::Key)] + #[case::value(Field::Value)] + fn test_decode_ignores_bytes_after_null_terminator(#[case] field: Field) { + let mut record = encode_record("key", "value"); + let tail = match field { + Field::Key => &mut record[4..WIRE_MAX_KEY_BYTES], + Field::Value => { + &mut record[WIRE_MAX_KEY_BYTES + 6 + ..WIRE_MAX_KEY_BYTES + WIRE_MAX_VALUE_BYTES] + } + }; + tail[..4].copy_from_slice(&[b'x', b'y', 0xFF, 0xFE]); + + let (key, value) = decode_record(&record).unwrap(); + assert_eq!(key, "key"); + assert_eq!(value, "value"); + } + /// Malformed buffers fed to `decode_record` produce the expected /// I/O error kind. #[rstest] From f74c78e1c3c9d75f7f49448d9ef92f239ec08b5a Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Wed, 2 Sep 2026 15:39:43 -0700 Subject: [PATCH 13/32] Add spec proposal for design feedback ahead of final implementation --- libazureinit-kvp/diagnostics-proposal.md | 271 +++++++++++++++++++++++ 1 file changed, 271 insertions(+) create mode 100644 libazureinit-kvp/diagnostics-proposal.md diff --git a/libazureinit-kvp/diagnostics-proposal.md b/libazureinit-kvp/diagnostics-proposal.md new file mode 100644 index 00000000..8b936481 --- /dev/null +++ b/libazureinit-kvp/diagnostics-proposal.md @@ -0,0 +1,271 @@ +# KVP Diagnostics Proposal + +## Goal + +Hyper-V KVP stores physical key/value records that a guest exposes to its host. Azure-init and cloud-init use those records for diagnostics. + +The proposal separates generic indexed values from diagnostic meaning, keeps each physical record available by default, and interprets payloads only when explicitly requested. + +## Decisions requested + +Please provide feedback on three decisions: + +1. **Layer boundary:** Keep KVP storage and diagnostics as the public layers. Keep indexed-value framing as a separate internal component until another use case needs it directly. Is that the intended separation? +2. **Type and name:** The proposal gives `type` first-class key-level meaning: it describes what is reported about the subject identified by `name`, while a common `kind` is derived from it. Should `type` have this independent meaning, or should the common diagnostic view expose only `kind` and `name` and leave the producer's exact classification in the raw key or value? +3. **Encoded content:** When content decoding is explicitly requested, choose one approach: + - **A — key-gated:** inspect an encoding envelope only when `type=compressed` indicates that encoded content is expected. + - **B — payload-directed:** inspect every requested payload for a validated `{encoding,data}` envelope. + +Under both approaches, the envelope determines the codec and ordinary reads do not decode content. If A is selected, should Azure-init also write `type=compressed`, and is the exact `type` sufficient or is a derived payload hint also useful? + +## Composition on top of KVP + +| Structure | Contains | Responsibility | +|---|---|---| +| Physical KVP record | Key and value | Preserve exactly what was written. | +| Indexed value | Base key plus physical parts numbered `0`, `1`, … | Split, identify, group, and order parts without interpreting their contents. | +| Diagnostic entry | Raw key, parsed metadata, and raw value | Explain the diagnostic meaning of one physical record. | +| Decoded diagnostic | Source entries plus decoded message or content | Optional source-specific combination and payload decoding. | + +```text +Diagnostics +├── uses the existing KVP record layer +├── uses indexed-value framing for |index +└── optionally interprets grouped diagnostic values +``` + +Indexed framing understands only the numeric suffix. It does not understand `start`, `finish`, JSON, compression, or diagnostic names. Diagnostics defines the base key and knows how values from that source can be combined. + +Use “indexed” or “chunked” for a value spread across KVP records. This is unrelated to an operation span represented by `start` and `finish`. + +## Key layouts + +- Azure-init: `||||||[|]` +- Current cloud-init: `CLOUD_INIT|||||[|]` +- Older cloud-init: `CLOUD_INIT||||[|]` + +The source prefix selects the key layout; `type` does not. A numeric suffix is an index only when the remaining base key is a valid diagnostic key. + +## Why each field exists + +| Field | Why it is meaningful | Example | +|---|---|---| +| `agent` | Identifies the writer and its key namespace. | `CLOUD_INIT`, `azure-init-0.1.1` | +| `boot_epoch` | Separates records produced by different boots. | `1788371515` | +| `vm_id` | Identifies the VM when records are collected outside the guest. | `e73baebd-...` | +| `type` | Preserves the producer's exact classification. It can describe lifecycle, data category, or representation. | `start`, `system-info`, `compressed` | +| `kind` | Gives consumers one common lifecycle view derived from `type`. It is not another encoded key field. | `Start`, `Finish`, `Diagnostic` | +| `name` | Identifies the operation or data the record concerns. | `provision:run`, `dmesg` | +| `event_id` | Identifies one emission and ties its physical parts together. Azure-init may reuse it to correlate a start and finish. | `operation-42` | +| `timestamp` | Identifies when the occurrence happened. Azure-init stores it in the key; cloud-init stores it in the value. | `2026-09-02T17:52:00Z` | +| `index` | Orders physical parts of one value. | `0`, `1`, `2` | +| `value` | Carries the producer-owned message or structured payload. | `starting provisioning` | + +## What `type` means compared with `name` + +`name` answers **“what is this record about?”** `type` adds context about **what is being reported about that subject**. + +| `type` | `name` | Meaning | Derived `kind` | +|---|---|---|---| +| `start` | `provision:run` | The `provision:run` operation began. | `Start` | +| `finish` | `provision:run` | The same operation completed. | `Finish` | +| `diagnostic` | `user:create_user` | Point-in-time data about creating a user. | `Diagnostic` | +| `system-info` | `system information` | Point-in-time system information. | `Diagnostic` | +| `compressed` | `dmesg` | Diagnostic data named `dmesg` is represented as encoded content. | `Diagnostic` | + +`type` changes the meaning of the **record**, but it does not rename, parse, or otherwise change `name`. The same name can have different lifecycle types, and the same type can be used with many names. + +The exact `type` is retained so `compressed`, `system-info`, and future producer-defined values are not lost. A normalized `kind` is derived as follows: + +- `start` → `Start` +- `finish` → `Finish` +- every other type → `Diagnostic` + +## Expected cloud-init telemetry + +The diagnostic view recognizes both current and older cloud-init key layouts and does not restrict `type` to a fixed allowlist. These are the known telemetry types supported explicitly: + +The VM and event IDs below are shortened for readability; cloud-init normally writes UUIDs. + +| Cloud-init `type` | Purpose | Expected value fields | Derived `kind` | +|---|---|---|---| +| `start` | An operation began. | `name`, `type`, `ts`, `msg` | `Start` | +| `finish` | An operation completed. | `name`, `type`, `ts`, `result`, `duration`, `msg` | `Finish` | +| `event` | Existing point-event format. | `name`, `type`, `ts`, `msg` | `Diagnostic` | +| `diagnostic` | Point-in-time diagnostic message. | `name`, `type`, `ts`, `msg` | `Diagnostic` | +| `system-info` | OS, kernel, distribution, and cloud-init information. | `name`, `type`, `ts`, `msg` | `Diagnostic` | +| `boot-telemetry` | Kernel, userspace, and cloud-init boot timing. | `name`, `type`, `ts`, `msg` | `Diagnostic` | +| `compressed` | Encoded diagnostic content such as `dmesg` or `cloud-init.log`. | `name`, `type`, `ts`, optional `msg_i`, `msg` | `Diagnostic` | +| Any other value | Future producer-defined telemetry. | Producer-defined | `Diagnostic` | + +The value-field list documents known cloud-init output; ordinary reads retain the complete value even when a field is missing or malformed. + +### Operation start + +```text +CLOUD_INIT|1788371515|start|modules-final/config-install_hotplug|vm-123|event-1 +``` + +```json +{"name":"modules-final/config-install_hotplug","type":"start","ts":"2026-09-02T17:52:24.256376Z","msg":"running config-install_hotplug"} +``` + +This means the named operation began. `type=start` supplies the lifecycle meaning; `name` identifies the operation. + +### Operation finish + +```text +CLOUD_INIT|1788371515|finish|modules-final/config-install_hotplug|vm-123|event-2 +``` + +```json +{"name":"modules-final/config-install_hotplug","type":"finish","ts":"2026-09-02T17:52:24.257560Z","result":"SUCCESS","duration":0.0012,"msg":"config-install_hotplug ran successfully"} +``` + +This means the same named operation completed. Cloud-init may assign start and finish different event IDs; matching names describe the same operation but do not by themselves prove correlation. + +### Diagnostic message + +```text +CLOUD_INIT|1788371515|diagnostic|diagnostic message|vm-123|event-3 +``` + +```json +{"name":"diagnostic message","type":"diagnostic","ts":"2026-09-02T17:52:25Z","msg":"Ephemeral resource disk exists."} +``` + +This is point-in-time data. Repeated diagnostic messages remain separate emissions because each has its own event ID. + +### System information + +```text +CLOUD_INIT|1788371515|system-info|system information|vm-123|event-4 +``` + +```json +{"name":"system information","type":"system-info","ts":"2026-09-02T17:52:26Z","msg":"cloudinit_version=26.1, kernel_version=6.8.0-azure, distro_name=ubuntu"} +``` + +The exact `system-info` type is retained while its common kind is `Diagnostic`. + +### Boot telemetry + +```text +CLOUD_INIT|1788371515|boot-telemetry|boot-telemetry|vm-123|event-5 +``` + +```json +{"name":"boot-telemetry","type":"boot-telemetry","ts":"2026-09-02T17:52:27Z","msg":"kernel_start=... user_start=... cloudinit_activation=..."} +``` + +This carries point-in-time boot timing data and also derives `kind=Diagnostic`. + +### Existing `event` compatibility + +```text +CLOUD_INIT|1788371515|event|user:create_user|vm-123|event-6 +``` + +```json +{"name":"user:create_user","type":"event","ts":"2026-09-02T17:52:28Z","msg":"Created user azureuser"} +``` + +`event` remains supported and maps to `Diagnostic`; it is not rejected when `diagnostic` is also supported. + +### Compressed, indexed content + +```text +CLOUD_INIT|1788371515|compressed|dmesg|vm-123|event-7|0 +CLOUD_INIT|1788371515|compressed|dmesg|vm-123|event-7|1 +``` + +```json +{"name":"dmesg","type":"compressed","ts":"2026-09-02T17:52:29Z","msg_i":0,"msg":"{\"encoding\":\"gz+b64\",\"data\":\"first-fragment"} +{"name":"dmesg","type":"compressed","ts":"2026-09-02T17:52:29Z","msg_i":1,"msg":"second-fragment\"}"} +``` + +Ordinary reads return both physical records. A real split may occur inside an escape and make one value invalid JSON by itself. Optional decoding validates the key and value indices, joins the escaped `msg` fragments, and then interprets the completed envelope according to Decision 3. + +### Older layout and unknown types + +Older cloud-init records omit `vm_id`: + +```text +CLOUD_INIT|1788371515|diagnostic|diagnostic message|event-8 +``` + +The older layout is accepted for every telemetry type listed above. + +Unknown types remain visible rather than being rejected: + +```text +CLOUD_INIT|1788371515|future-type|future diagnostic|vm-123|event-9 +``` + +The first record has no VM ID. The second retains `type=future-type` and derives `kind=Diagnostic`. + +## Clear use cases + +### 1. One operation starts and finishes + +```text +azure-init|1788371515|vm-123|start|provision:run|operation-42|2026-09-02T17:52:00Z|0 +value: starting provisioning + +azure-init|1788371515|vm-123|finish|provision:run|operation-42|2026-09-02T17:52:24Z|0 +value: provisioning completed +``` + +- `type` changes from `start` to `finish` because two different lifecycle occurrences are being reported. +- `name` stays `provision:run` because both records concern the same operation. +- `event_id` correlates the occurrences. +- `timestamp` says when each occurred. +- `index=0` says each value fits in one physical record. +- The values remain simple messages because identity and lifecycle metadata are already in the keys. + +### 2. The same type describes different diagnostics + +```text +azure-init|1788371515|vm-123|diagnostic|user:create_user|event-1|2026-09-02T17:52:25Z|0 +value: created user azureuser + +azure-init|1788371515|vm-123|diagnostic|network:configure|event-2|2026-09-02T17:52:26Z|0 +value: configured eth0 +``` + +Both records are point-in-time diagnostics, but their names identify different subjects. This demonstrates that `type` does not determine `name`. + +### 3. One value spans physical records + +```text +azure-init|1788371515|vm-123|diagnostic|config:dump|event-4|2026-09-02T17:52:28Z|0 +value: first fragment + +azure-init|1788371515|vm-123|diagnostic|config:dump|event-4|2026-09-02T17:52:28Z|1 +value: second fragment +``` + +The base key is identical and only the index changes. Indexed framing can group and order the records without knowing they are diagnostics. The diagnostic layer knows that Azure-init values can be concatenated. Ordinary reads still return both physical records independently. + +## Read and decode behavior + +Ordinary diagnostic reads: + +- Return one entry for each recognized physical diagnostic record, in pool order. +- Preserve the complete raw key and value. +- Parse available key metadata. +- Best-effort read cloud-init's timestamp without replacing or decoding its message. +- Retain duplicate, gapped, and incomplete indexed records. +- Omit unrelated records and records whose keys are malformed. The raw KVP view remains available when every pool record is needed. + +Optional decoding is a separate request: + +- Group only records with the same complete base key. +- Require unique, contiguous indices beginning at zero. +- Validate cloud-init's value index against the key index. +- Combine values according to their source: concatenate Azure-init fragments, or extract and combine cloud-init `msg` fragments. +- Return an explicit error while preserving the source records when combination or decoding fails. + +The format has no total-part count, so a missing final fragment cannot always be detected. A physical cloud-init part may also be invalid JSON because a split can occur inside an escaped `msg`; that does not prevent the physical record from being returned. + +Existing Azure-init `type=event` records remain readable and derive `kind=Diagnostic`; new point diagnostics use `type=diagnostic`. Unknown types are retained exactly and also derive `kind=Diagnostic`. \ No newline at end of file From bda2c7d739a93b6b927ff964582fcf5c9b522a15 Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Wed, 2 Sep 2026 16:20:44 -0700 Subject: [PATCH 14/32] docs(kvp): clarify diagnostics architecture proposal --- libazureinit-kvp/diagnostics-proposal.md | 327 +++++++++-------------- 1 file changed, 126 insertions(+), 201 deletions(-) diff --git a/libazureinit-kvp/diagnostics-proposal.md b/libazureinit-kvp/diagnostics-proposal.md index 8b936481..64d338a0 100644 --- a/libazureinit-kvp/diagnostics-proposal.md +++ b/libazureinit-kvp/diagnostics-proposal.md @@ -1,271 +1,196 @@ # KVP Diagnostics Proposal -## Goal +## What changes -Hyper-V KVP stores physical key/value records that a guest exposes to its host. Azure-init and cloud-init use those records for diagnostics. +Hyper-V KVP stores physical key/value records. It does not define chunking, diagnostics, or payload formats. This proposal does not change KVP. -The proposal separates generic indexed values from diagnostic meaning, keeps each physical record available by default, and interprets payloads only when explicitly requested. +Today, the diagnostic view reconstructs complete events and may omit incomplete groups. The proposal changes the default diagnostic view to return every recognized physical record with parsed metadata and its original value. Reconstruction and content decoding become optional. -## Decisions requested +The reader supports existing Azure-init and cloud-init records. The writer produces Azure-init records only. -Please provide feedback on three decisions: +## Components and composition -1. **Layer boundary:** Keep KVP storage and diagnostics as the public layers. Keep indexed-value framing as a separate internal component until another use case needs it directly. Is that the intended separation? -2. **Type and name:** The proposal gives `type` first-class key-level meaning: it describes what is reported about the subject identified by `name`, while a common `kind` is derived from it. Should `type` have this independent meaning, or should the common diagnostic view expose only `kind` and `name` and leave the producer's exact classification in the raw key or value? -3. **Encoded content:** When content decoding is explicitly requested, choose one approach: - - **A — key-gated:** inspect an encoding envelope only when `type=compressed` indicates that encoded content is expected. - - **B — payload-directed:** inspect every requested payload for a validated `{encoding,data}` envelope. - -Under both approaches, the envelope determines the codec and ordinary reads do not decode content. If A is selected, should Azure-init also write `type=compressed`, and is the exact `type` sufficient or is a derived payload hint also useful? - -## Composition on top of KVP - -| Structure | Contains | Responsibility | +| Component | Composed from | Responsibility | |---|---|---| -| Physical KVP record | Key and value | Preserve exactly what was written. | -| Indexed value | Base key plus physical parts numbered `0`, `1`, … | Split, identify, group, and order parts without interpreting their contents. | -| Diagnostic entry | Raw key, parsed metadata, and raw value | Explain the diagnostic meaning of one physical record. | -| Decoded diagnostic | Source entries plus decoded message or content | Optional source-specific combination and payload decoding. | - -```text -Diagnostics -├── uses the existing KVP record layer -├── uses indexed-value framing for |index -└── optionally interprets grouped diagnostic values -``` - -Indexed framing understands only the numeric suffix. It does not understand `start`, `finish`, JSON, compression, or diagnostic names. Diagnostics defines the base key and knows how values from that source can be combined. - -Use “indexed” or “chunked” for a value spread across KVP records. This is unrelated to an operation span represented by `start` and `finish`. - -## Key layouts - -- Azure-init: `||||||[|]` -- Current cloud-init: `CLOUD_INIT|||||[|]` -- Older cloud-init: `CLOUD_INIT||||[|]` - -The source prefix selects the key layout; `type` does not. A numeric suffix is an index only when the remaining base key is a valid diagnostic key. - -## Why each field exists - -| Field | Why it is meaningful | Example | -|---|---|---| -| `agent` | Identifies the writer and its key namespace. | `CLOUD_INIT`, `azure-init-0.1.1` | -| `boot_epoch` | Separates records produced by different boots. | `1788371515` | -| `vm_id` | Identifies the VM when records are collected outside the guest. | `e73baebd-...` | -| `type` | Preserves the producer's exact classification. It can describe lifecycle, data category, or representation. | `start`, `system-info`, `compressed` | -| `kind` | Gives consumers one common lifecycle view derived from `type`. It is not another encoded key field. | `Start`, `Finish`, `Diagnostic` | -| `name` | Identifies the operation or data the record concerns. | `provision:run`, `dmesg` | -| `event_id` | Identifies one emission and ties its physical parts together. Azure-init may reuse it to correlate a start and finish. | `operation-42` | -| `timestamp` | Identifies when the occurrence happened. Azure-init stores it in the key; cloud-init stores it in the value. | `2026-09-02T17:52:00Z` | -| `index` | Orders physical parts of one value. | `0`, `1`, `2` | -| `value` | Carries the producer-owned message or structured payload. | `starting provisioning` | - -## What `type` means compared with `name` +| `KvpPoolStore` | One Hyper-V KVP pool | Read and write physical key/value records without diagnostic meaning | +| Indexed framing | A base key and optional numeric `|index` suffixes | Split values for writing; group and order records only when reconstruction is requested | +| `DiagnosticsPool` | One `KvpPoolStore` plus Azure-init writer identity | Read both source formats, write Azure-init diagnostics, and expose diagnostic entries | +| `DiagnosticKey` | Metadata parsed from one Azure-init or cloud-init key | Provide one source-independent key model | +| `DiagnosticEntry` | Raw key, `DiagnosticKey`, and unchanged value | Represent one physical diagnostic record | +| Message and Content views | One or more `DiagnosticEntry` values | Optionally reconstruct a message and decode its declared content encoding | -`name` answers **“what is this record about?”** `type` adds context about **what is being reported about that subject**. +`DiagnosticsPool` is a diagnostic view over an existing `KvpPoolStore`; it does not replace the store or create another persisted format. It recognizes the Azure-init and cloud-init key layouts and converts either one into `DiagnosticKey`. -| `type` | `name` | Meaning | Derived `kind` | -|---|---|---|---| -| `start` | `provision:run` | The `provision:run` operation began. | `Start` | -| `finish` | `provision:run` | The same operation completed. | `Finish` | -| `diagnostic` | `user:create_user` | Point-in-time data about creating a user. | `Diagnostic` | -| `system-info` | `system information` | Point-in-time system information. | `Diagnostic` | -| `compressed` | `dmesg` | Diagnostic data named `dmesg` is represented as encoded content. | `Diagnostic` | +Indexed framing is an internal rule, not diagnostic metadata. It understands only a base key, an index, and value fragments. It does not understand `kind`, `name`, JSON, or compression. -`type` changes the meaning of the **record**, but it does not rename, parse, or otherwise change `name`. The same name can have different lifecycle types, and the same type can be used with many names. - -The exact `type` is retained so `compressed`, `system-info`, and future producer-defined values are not lost. A normalized `kind` is derived as follows: - -- `start` → `Start` -- `finish` → `Finish` -- every other type → `Diagnostic` - -## Expected cloud-init telemetry +```text +Read: +KvpPoolStore + -> physical records + -> DiagnosticsPool parses each source key + -> DiagnosticEntry values [default] + -> indexed framing groups and orders entries [Message requested] + -> DiagnosticsPool reconstructs the source value + -> encoding envelope is decoded [Content requested] -The diagnostic view recognizes both current and older cloud-init key layouts and does not restrict `type` to a fixed allowlist. These are the known telemetry types supported explicitly: +Write: +caller + -> DiagnosticsPool creates an Azure-init base key + -> indexed framing splits the value and adds indexes + -> KvpPoolStore writes the physical records +``` -The VM and event IDs below are shortened for readability; cloud-init normally writes UUIDs. +The default `DiagnosticsPool` read returns one `DiagnosticEntry` for each recognized physical record; it does not group records. Its point-diagnostic write generates the event ID and timestamp. A lifecycle-aware write accepts `Start`, `Finish`, or `Diagnostic` and the caller's event metadata. -| Cloud-init `type` | Purpose | Expected value fields | Derived `kind` | -|---|---|---|---| -| `start` | An operation began. | `name`, `type`, `ts`, `msg` | `Start` | -| `finish` | An operation completed. | `name`, `type`, `ts`, `result`, `duration`, `msg` | `Finish` | -| `event` | Existing point-event format. | `name`, `type`, `ts`, `msg` | `Diagnostic` | -| `diagnostic` | Point-in-time diagnostic message. | `name`, `type`, `ts`, `msg` | `Diagnostic` | -| `system-info` | OS, kernel, distribution, and cloud-init information. | `name`, `type`, `ts`, `msg` | `Diagnostic` | -| `boot-telemetry` | Kernel, userspace, and cloud-init boot timing. | `name`, `type`, `ts`, `msg` | `Diagnostic` | -| `compressed` | Encoded diagnostic content such as `dmesg` or `cloud-init.log`. | `name`, `type`, `ts`, optional `msg_i`, `msg` | `Diagnostic` | -| Any other value | Future producer-defined telemetry. | Producer-defined | `Diagnostic` | +Raw KVP commands use `KvpPoolStore` directly. Parsed diagnostic commands use `DiagnosticsPool`. Message reconstruction and content decoding are requested views, not additional storage layers or stored data structures. -The value-field list documents known cloud-init output; ordinary reads retain the complete value even when a field is missing or malformed. +## One normalized `DiagnosticKey` -### Operation start +New Azure-init records use: ```text -CLOUD_INIT|1788371515|start|modules-final/config-install_hotplug|vm-123|event-1 -``` - -```json -{"name":"modules-final/config-install_hotplug","type":"start","ts":"2026-09-02T17:52:24.256376Z","msg":"running config-install_hotplug"} +||||||[|] ``` -This means the named operation began. `type=start` supplies the lifecycle meaning; `name` identifies the operation. - -### Operation finish +Existing cloud-init records use: ```text -CLOUD_INIT|1788371515|finish|modules-final/config-install_hotplug|vm-123|event-2 -``` - -```json -{"name":"modules-final/config-install_hotplug","type":"finish","ts":"2026-09-02T17:52:24.257560Z","result":"SUCCESS","duration":0.0012,"msg":"config-install_hotplug ran successfully"} +CLOUD_INIT|||||[|] +CLOUD_INIT||||[|] ``` -This means the same named operation completed. Cloud-init may assign start and finish different event IDs; matching names describe the same operation but do not by themselves prove correlation. - -### Diagnostic message +Both formats produce the same key model: ```text -CLOUD_INIT|1788371515|diagnostic|diagnostic message|vm-123|event-3 -``` +DiagnosticKey { + agent, boot_epoch, vm_id?, kind, name, + event_id, timestamp?, chunk_index? +} -```json -{"name":"diagnostic message","type":"diagnostic","ts":"2026-09-02T17:52:25Z","msg":"Ephemeral resource disk exists."} +DiagnosticEntry { raw_key, key: DiagnosticKey, value } ``` -This is point-in-time data. Repeated diagnostic messages remain separate emissions because each has its own event ID. - -### System information - -```text -CLOUD_INIT|1788371515|system-info|system information|vm-123|event-4 -``` - -```json -{"name":"system information","type":"system-info","ts":"2026-09-02T17:52:26Z","msg":"cloudinit_version=26.1, kernel_version=6.8.0-azure, distro_name=ubuntu"} -``` +| `DiagnosticKey` field | Azure-init source | Cloud-init source | +|---|---|---| +| `agent` | First key field | `CLOUD_INIT` | +| `boot_epoch` | Key | Key | +| `vm_id` | Key | Key, or absent in the older layout | +| `kind` | Derived from key `kind` | Derived from key `type` | +| `name` | Key | Key | +| `event_id` | Key | Key | +| `timestamp` | Key | Best-effort `ts` from the value | +| `chunk_index` | Trailing numeric suffix | Trailing numeric suffix | + +The value is preserved unchanged. In the default view, cloud-init value parsing is used only to obtain its timestamp. Message and Content views interpret more of the value only when requested. + +| Field | Meaning and use | Example | +|---|---|---| +| `raw_key` | Preserve the exact source key for compatibility and inspection | Original `CLOUD_INIT|...` key | +| `agent` | Identify the producer and its key namespace | `CLOUD_INIT`, `azure-init` | +| `boot_epoch` | Identify when the VM boot began so records from different boots can be separated | `1788371515` | +| `vm_id` | Identify the VM when records are exported or aggregated; absent from older cloud-init keys | `vm-123` | +| `kind` | State the record's lifecycle role | `Start`, `Finish`, `Diagnostic` | +| `name` | Identify the producer-defined operation or subject for filtering | `provision:run`, `dmesg` | +| `event_id` | Identify one logical emission and tie all of its physical parts together | `event-1` | +| `timestamp` | State when the occurrence was reported; optional when unavailable | `2026-09-02T17:52:25Z` | +| `chunk_index` | Order the physical parts of one value | `0`, `1`, `2` | +| `value` | Preserve the producer-owned message or structured payload | `Retrieved 1 key from IMDS` | -The exact `system-info` type is retained while its common kind is `Diagnostic`. +An `event_id` always groups parts of one emission. It correlates separate `Start` and `Finish` records only when the producer explicitly guarantees that convention. -### Boot telemetry +## `kind` compared with `name` -```text -CLOUD_INIT|1788371515|boot-telemetry|boot-telemetry|vm-123|event-5 -``` +`kind` answers “what role does this record have?” `name` answers “what is this record about?” -```json -{"name":"boot-telemetry","type":"boot-telemetry","ts":"2026-09-02T17:52:27Z","msg":"kernel_start=... user_start=... cloudinit_activation=..."} -``` +| `kind` | `name` | Meaning | +|---|---|---| +| `Start` | `provision:run` | The named operation began | +| `Diagnostic` | `provision:run` | A point-in-time observation about the operation | +| `Finish` | `provision:run` | The named operation finished | +| `Diagnostic` | `dmesg` | Point-in-time diagnostic data named `dmesg` | -This carries point-in-time boot timing data and also derives `kind=Diagnostic`. +`kind` does not rename `name` and does not select a value parser. Results, durations, and other details remain in the producer-owned value. -### Existing `event` compatibility +Azure-init calls its classification `kind`; cloud-init calls it `type`. Each known token means: -```text -CLOUD_INIT|1788371515|event|user:create_user|vm-123|event-6 -``` +| Source token | Exact source meaning | Common `kind` | +|---|---|---| +| `start` | The operation identified by `name` began | `Start` | +| `finish` | The operation identified by `name` ended | `Finish` | +| `event` | Legacy point-in-time event | `Diagnostic` | +| `diagnostic` | Point-in-time diagnostic observation | `Diagnostic` | +| `system-info` | System, OS, or agent information | `Diagnostic` | +| `boot-telemetry` | Boot timing or boot-related information | `Diagnostic` | +| `compressed` | Legacy cloud-init label for an encoded diagnostic payload | `Diagnostic` | +| Any other value | Producer-defined or unknown classification | `Diagnostic` | -```json -{"name":"user:create_user","type":"event","ts":"2026-09-02T17:52:28Z","msg":"Created user azureuser"} -``` +The token changes the record's lifecycle meaning only for `start` and `finish`. It never changes how `name` is interpreted. The original source token remains available in `raw_key`; it does not select a value parser in the common model. In particular, compression is a payload encoding, not a lifecycle kind. -`event` remains supported and maps to `Diagnostic`; it is not rejected when `diagnostic` is also supported. +## Proposed read behavior -### Compressed, indexed content +| Command | Result | +|---|---| +| `read ` | The raw value for that exact KVP key | +| `dump` | Every physical key/value record in pool order | +| `dump --parse-diagnostics` | One parsed record per recognized physical record, including its chunk index and unchanged value | +| `dump --parse-diagnostics --view message` | Reconstruct the source message when its indexed records are available | +| `dump --parse-diagnostics --view content` | Decode a validated encoding envelope such as `{"encoding":"gz+b64","data":"..."}` | -```text -CLOUD_INIT|1788371515|compressed|dmesg|vm-123|event-7|0 -CLOUD_INIT|1788371515|compressed|dmesg|vm-123|event-7|1 -``` +Raw parsed diagnostics are the default. `--name` filters by diagnostic name, and `--tail` counts physical records in this view. JSON output keeps the original value as a string. -```json -{"name":"dmesg","type":"compressed","ts":"2026-09-02T17:52:29Z","msg_i":0,"msg":"{\"encoding\":\"gz+b64\",\"data\":\"first-fragment"} -{"name":"dmesg","type":"compressed","ts":"2026-09-02T17:52:29Z","msg_i":1,"msg":"second-fragment\"}"} -``` +If Message or Content processing fails, that request falls back to the available Raw records rather than hiding them. -Ordinary reads return both physical records. A real split may occur inside an escape and make one value invalid JSON by itself. Optional decoding validates the key and value indices, joins the escaped `msg` fragments, and then interprets the completed envelope according to Decision 3. +The existing suffix format has no total-part count. Gaps and duplicate indexes can be detected, but a missing final part cannot always be detected; reconstruction of existing records is therefore best effort. -### Older layout and unknown types +## End-to-end example -Older cloud-init records omit `vm_id`: +Write one Azure-init point diagnostic: ```text -CLOUD_INIT|1788371515|diagnostic|diagnostic message|event-8 +libazureinit-kvp emit --prefix azure-init --vm-id vm-123 --name imds --message "Retrieved 1 key from IMDS" ``` -The older layout is accepted for every telemetry type listed above. - -Unknown types remain visible rather than being rejected: +The command uses the supplied agent and VM ID, generates the boot epoch, event ID, and timestamp, then writes: ```text -CLOUD_INIT|1788371515|future-type|future diagnostic|vm-123|event-9 +azure-init|1788371515|vm-123|diagnostic|imds|event-1|2026-09-02T17:52:25Z|0 +value: Retrieved 1 key from IMDS ``` -The first record has no VM ID. The second retains `type=future-type` and derives `kind=Diagnostic`. - -## Clear use cases - -### 1. One operation starts and finishes +An exact-key read returns: ```text -azure-init|1788371515|vm-123|start|provision:run|operation-42|2026-09-02T17:52:00Z|0 -value: starting provisioning - -azure-init|1788371515|vm-123|finish|provision:run|operation-42|2026-09-02T17:52:24Z|0 -value: provisioning completed +Retrieved 1 key from IMDS ``` -- `type` changes from `start` to `finish` because two different lifecycle occurrences are being reported. -- `name` stays `provision:run` because both records concern the same operation. -- `event_id` correlates the occurrences. -- `timestamp` says when each occurred. -- `index=0` says each value fits in one physical record. -- The values remain simple messages because identity and lifecycle metadata are already in the keys. - -### 2. The same type describes different diagnostics +A parsed diagnostic read returns: ```text -azure-init|1788371515|vm-123|diagnostic|user:create_user|event-1|2026-09-02T17:52:25Z|0 -value: created user azureuser - -azure-init|1788371515|vm-123|diagnostic|network:configure|event-2|2026-09-02T17:52:26Z|0 -value: configured eth0 +kind=Diagnostic agent=azure-init boot_epoch=1788371515 vm_id=vm-123 name=imds event_id=event-1 timestamp=2026-09-02T17:52:25Z chunk_index=0 value="Retrieved 1 key from IMDS" ``` -Both records are point-in-time diagnostics, but their names identify different subjects. This demonstrates that `type` does not determine `name`. - -### 3. One value spans physical records +A long value may produce two physical records: ```text -azure-init|1788371515|vm-123|diagnostic|config:dump|event-4|2026-09-02T17:52:28Z|0 -value: first fragment - -azure-init|1788371515|vm-123|diagnostic|config:dump|event-4|2026-09-02T17:52:28Z|1 -value: second fragment +|0 = "Retrieved " +|1 = "1 key from IMDS" ``` -The base key is identical and only the index changes. Indexed framing can group and order the records without knowing they are diagnostics. The diagnostic layer knows that Azure-init values can be concatenated. Ordinary reads still return both physical records independently. +The default diagnostic view returns both records. Message view returns one value: -## Read and decode behavior - -Ordinary diagnostic reads: +```text +Retrieved 1 key from IMDS +``` -- Return one entry for each recognized physical diagnostic record, in pool order. -- Preserve the complete raw key and value. -- Parse available key metadata. -- Best-effort read cloud-init's timestamp without replacing or decoding its message. -- Retain duplicate, gapped, and incomplete indexed records. -- Omit unrelated records and records whose keys are malformed. The raw KVP view remains available when every pool record is needed. +If part 0 is missing, part 1 remains visible in the default view and no reconstructed message is claimed. -Optional decoding is a separate request: +Existing cloud-init records follow the same read contract. Their key metadata is normalized, while their original key and JSON value remain unchanged. -- Group only records with the same complete base key. -- Require unique, contiguous indices beginning at zero. -- Validate cloud-init's value index against the key index. -- Combine values according to their source: concatenate Azure-init fragments, or extract and combine cloud-init `msg` fragments. -- Return an explicit error while preserving the source records when combination or decoding fails. +## Feedback requested -The format has no total-part count, so a missing final fragment cannot always be detected. A physical cloud-init part may also be invalid JSON because a split can occur inside an escaped `msg`; that does not prevent the physical record from being returned. +Please confirm these four decisions: -Existing Azure-init `type=event` records remain readable and derive `kind=Diagnostic`; new point diagnostics use `type=diagnostic`. Unknown types are retained exactly and also derive `kind=Diagnostic`. \ No newline at end of file +1. Keep physical KVP, indexed framing, and diagnostics as separate concepts, while leaving indexed framing internal for now. +2. Use only `Start`, `Finish`, and `Diagnostic` in the common model. Preserve cloud-init's exact `type` in the raw key rather than assigning it cross-source semantics. +3. Make one physical diagnostic record the default result and request higher-level processing with `--view message` or `--view content`. +4. Treat encoding as a validated value envelope rather than `type=compressed`; if a higher-level view fails, preserve the available physical records. From c3a1961e7c0304f141330cc14870aaeaf71963d3 Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Wed, 2 Sep 2026 17:06:26 -0700 Subject: [PATCH 15/32] docs(kvp): restructure diagnostics proposal around layers and decoding --- libazureinit-kvp/diagnostics-proposal.md | 322 +++++++++++++++-------- 1 file changed, 208 insertions(+), 114 deletions(-) diff --git a/libazureinit-kvp/diagnostics-proposal.md b/libazureinit-kvp/diagnostics-proposal.md index 64d338a0..9653206f 100644 --- a/libazureinit-kvp/diagnostics-proposal.md +++ b/libazureinit-kvp/diagnostics-proposal.md @@ -1,148 +1,209 @@ # KVP Diagnostics Proposal -## What changes +## Purpose -Hyper-V KVP stores physical key/value records. It does not define chunking, diagnostics, or payload formats. This proposal does not change KVP. +Hyper-V KVP is a guest-to-host exchange of physical key/value records. Each record has a size limit. KVP preserves records but does not define diagnostics, splitting, reconstruction, or payload formats. -Today, the diagnostic view reconstructs complete events and may omit incomplete groups. The proposal changes the default diagnostic view to return every recognized physical record with parsed metadata and its original value. Reconstruction and content decoding become optional. +Azure-init and cloud-init are guest provisioning agents that write diagnostic records to KVP. This proposal defines one way to read both formats and one Azure-init format for new writes. -The reader supports existing Azure-init and cloud-init records. The writer produces Azure-init records only. +This proposal leaves KVP unchanged and changes the diagnostic view: -## Components and composition +- The default result is one entry per recognized physical diagnostic record, with its raw key and value unchanged. +- Splitting and grouping records is separate from diagnostic meaning. +- Message reconstruction and content decoding happen only when requested. -| Component | Composed from | Responsibility | -|---|---|---| -| `KvpPoolStore` | One Hyper-V KVP pool | Read and write physical key/value records without diagnostic meaning | -| Indexed framing | A base key and optional numeric `|index` suffixes | Split values for writing; group and order records only when reconstruction is requested | -| `DiagnosticsPool` | One `KvpPoolStore` plus Azure-init writer identity | Read both source formats, write Azure-init diagnostics, and expose diagnostic entries | -| `DiagnosticKey` | Metadata parsed from one Azure-init or cloud-init key | Provide one source-independent key model | -| `DiagnosticEntry` | Raw key, `DiagnosticKey`, and unchanged value | Represent one physical diagnostic record | -| Message and Content views | One or more `DiagnosticEntry` values | Optionally reconstruct a message and decode its declared content encoding | +## Architecture + +### Existing foundation: `KvpPoolStore` + +`KvpPoolStore` reads and writes one Hyper-V KVP pool. It returns physical key/value records exactly as stored and has no diagnostic meaning. + +### Layer 1: indexed values + +A value that exceeds one KVP record is split across records. Each part uses the same base key plus a numeric suffix: + +```text +|0 = first value fragment +|1 = second value fragment +``` + +Indexed framing splits values for writes and identifies, groups, and orders parts for reconstruction. It understands only a caller-defined base key, index, and value fragments. It does not understand diagnostic fields, JSON, or compression. + +Indexed framing is decoupled from diagnostics in the code, but it is not a separate public API. Callers use raw KVP or diagnostics; nothing else needs to span values today. + +### Layer 2: diagnostics + +Diagnostics defines the base-key formats and how ordered source values become a logical message. It uses three components: -`DiagnosticsPool` is a diagnostic view over an existing `KvpPoolStore`; it does not replace the store or create another persisted format. It recognizes the Azure-init and cloud-init key layouts and converts either one into `DiagnosticKey`. +#### `DiagnosticKey`: normalized metadata -Indexed framing is an internal rule, not diagnostic metadata. It understands only a base key, an index, and value fragments. It does not understand `kind`, `name`, JSON, or compression. +`DiagnosticKey` contains diagnostic metadata parsed from one Azure-init or cloud-init key: + +```text +DiagnosticKey { + agent, boot_epoch, vm_id?, kind, name, + event_id, timestamp?, chunk_index? +} +``` + +Its purpose is to give both source formats one common key model. + +#### `DiagnosticEntry`: one physical diagnostic record + +`DiagnosticEntry` is the default unit returned by a diagnostic read: + +```text +DiagnosticEntry { + raw_key, + key: DiagnosticKey, + value +} +``` + +The raw key and value remain unchanged. `DiagnosticKey` provides the normalized metadata. A value split into four KVP records produces four `DiagnosticEntry` values. + +#### `DiagnosticsPool`: diagnostic access + +`DiagnosticsPool` is the diagnostic interface composed over one `KvpPoolStore` and the indexed framing rules. + +On read, it returns one `DiagnosticEntry` per recognized physical record, in pool order. A record is recognized when its key matches the Azure-init or cloud-init layout; other keys, such as `PROVISIONING_REPORT`, are skipped in the diagnostic view but remain visible through `dump`. Optional fields (`vm_id`, `timestamp`, `chunk_index`) are omitted from output when absent rather than shown as empty. On write, it creates Azure-init keys, uses indexed framing when a value must be split, and writes the resulting records through `KvpPoolStore`. ```text Read: -KvpPoolStore - -> physical records - -> DiagnosticsPool parses each source key - -> DiagnosticEntry values [default] - -> indexed framing groups and orders entries [Message requested] - -> DiagnosticsPool reconstructs the source value - -> encoding envelope is decoded [Content requested] +KvpPoolStore -> physical records -> DiagnosticsPool -> DiagnosticEntry values Write: -caller - -> DiagnosticsPool creates an Azure-init base key - -> indexed framing splits the value and adds indexes - -> KvpPoolStore writes the physical records +caller -> DiagnosticsPool -> indexed framing -> KvpPoolStore ``` -The default `DiagnosticsPool` read returns one `DiagnosticEntry` for each recognized physical record; it does not group records. Its point-diagnostic write generates the event ID and timestamp. A lifecycle-aware write accepts `Start`, `Finish`, or `Diagnostic` and the caller's event metadata. +## Reading a value back: Raw, Message, Content + +A diagnostic value can be wrapped up to three times before it lands in KVP: + +1. Split — a value larger than one record is spread across several physical records (indexed values). +2. Wrapped — the producer stores the message inside its own structure, such as cloud-init's JSON `{"...","msg":"..."}`. +3. Encoded — the message itself may be compressed and base64-encoded, such as a captured `dmesg` or `cloud-init.log`. -Raw KVP commands use `KvpPoolStore` directly. Parsed diagnostic commands use `DiagnosticsPool`. Message reconstruction and content decoding are requested views, not additional storage layers or stored data structures. +Reading a value back means choosing how far to unwrap it. The three views are read-time transformations, not stored data: + +| View | Unwraps | Result | +|---|---|---| +| Raw | Nothing | Each physical record exactly as stored | +| Message | Split and producer wrapper | The producer's logical message | +| Content | Split, wrapper, and encoding | The decoded, decompressed payload | -## One normalized `DiagnosticKey` +### Example: a compressed `dmesg` -New Azure-init records use: +Stored as two physical records. Each is a cloud-init JSON wrapper carrying one fragment of the message: ```text -||||||[|] +Raw: + CLOUD_INIT|...|compressed|dmesg|...|0 = {"...","msg_i":0,"msg":"{\"encoding\":\"gz+b64\",\"da"} + CLOUD_INIT|...|compressed|dmesg|...|1 = {"...","msg_i":1,"msg":"ta\":\"H4sIA...\"}"} ``` -Existing cloud-init records use: +Reassemble the fragments into the producer's message. For a compressed event that message is an encoding envelope, not yet readable text: ```text -CLOUD_INIT|||||[|] -CLOUD_INIT||||[|] +Message: + {"encoding":"gz+b64","data":"H4sIA..."} ``` -Both formats produce the same key model: +Decode and decompress the envelope to get the actual artifact: ```text -DiagnosticKey { - agent, boot_epoch, vm_id?, kind, name, - event_id, timestamp?, chunk_index? -} +Content: + [ 0.000000] Linux version 6.8.0-azure ... +``` + +### First-class encoding mechanism -DiagnosticEntry { raw_key, key: DiagnosticKey, value } +Undoing the split and the producer wrapper is mechanical. Decoding is not: it needs a defined envelope, a codec, and a decompression step. Today no supported operation does this, so a compressed diagnostic can only be read back as an opaque `{encoding,data}` blob. The proposal is to make decoding a first-class, explicitly requested step — the Content view — driven by a validated `{encoding,data}` envelope in the value rather than guessed from the key. + +For any value that is not encoded there is nothing to decode, so Message and Content are identical: + +```text +Raw: |0 = "Retrieved ", |1 = "1 key from IMDS" +Message: "Retrieved 1 key from IMDS" +Content: "Retrieved 1 key from IMDS" (no envelope, nothing to decode) ``` -| `DiagnosticKey` field | Azure-init source | Cloud-init source | -|---|---|---| -| `agent` | First key field | `CLOUD_INIT` | -| `boot_epoch` | Key | Key | -| `vm_id` | Key | Key, or absent in the older layout | -| `kind` | Derived from key `kind` | Derived from key `type` | -| `name` | Key | Key | -| `event_id` | Key | Key | -| `timestamp` | Key | Best-effort `ts` from the value | -| `chunk_index` | Trailing numeric suffix | Trailing numeric suffix | - -The value is preserved unchanged. In the default view, cloud-init value parsing is used only to obtain its timestamp. Message and Content views interpret more of the value only when requested. - -| Field | Meaning and use | Example | -|---|---|---| -| `raw_key` | Preserve the exact source key for compatibility and inspection | Original `CLOUD_INIT|...` key | -| `agent` | Identify the producer and its key namespace | `CLOUD_INIT`, `azure-init` | -| `boot_epoch` | Identify when the VM boot began so records from different boots can be separated | `1788371515` | -| `vm_id` | Identify the VM when records are exported or aggregated; absent from older cloud-init keys | `vm-123` | -| `kind` | State the record's lifecycle role | `Start`, `Finish`, `Diagnostic` | -| `name` | Identify the producer-defined operation or subject for filtering | `provision:run`, `dmesg` | -| `event_id` | Identify one logical emission and tie all of its physical parts together | `event-1` | -| `timestamp` | State when the occurrence was reported; optional when unavailable | `2026-09-02T17:52:25Z` | -| `chunk_index` | Order the physical parts of one value | `0`, `1`, `2` | -| `value` | Preserve the producer-owned message or structured payload | `Retrieved 1 key from IMDS` | +Raw returns one result per physical record. Message and Content return one result per event, that is, per set of records sharing a base key. Both require a contiguous run of parts beginning at `chunk_index` 0; if the first or a middle part is missing, no message is produced and the available parts remain as Raw entries. The suffix format has no total-part count, so a missing final part cannot always be detected; reconstruction of existing records is therefore best effort. -An `event_id` always groups parts of one emission. It correlates separate `Start` and `Finish` records only when the producer explicitly guarantees that convention. +## Normalizing Azure-init and cloud-init -## `kind` compared with `name` +Azure-init keys: -`kind` answers “what role does this record have?” `name` answers “what is this record about?” +```text +||||||[|] +``` -| `kind` | `name` | Meaning | -|---|---|---| -| `Start` | `provision:run` | The named operation began | -| `Diagnostic` | `provision:run` | A point-in-time observation about the operation | -| `Finish` | `provision:run` | The named operation finished | -| `Diagnostic` | `dmesg` | Point-in-time diagnostic data named `dmesg` | +Current cloud-init keys: -`kind` does not rename `name` and does not select a value parser. Results, durations, and other details remain in the producer-owned value. +```text +CLOUD_INIT|||||[|] +``` -Azure-init calls its classification `kind`; cloud-init calls it `type`. Each known token means: +Older cloud-init keys omit `vm_id`: -| Source token | Exact source meaning | Common `kind` | -|---|---|---| -| `start` | The operation identified by `name` began | `Start` | -| `finish` | The operation identified by `name` ended | `Finish` | -| `event` | Legacy point-in-time event | `Diagnostic` | -| `diagnostic` | Point-in-time diagnostic observation | `Diagnostic` | -| `system-info` | System, OS, or agent information | `Diagnostic` | -| `boot-telemetry` | Boot timing or boot-related information | `Diagnostic` | -| `compressed` | Legacy cloud-init label for an encoded diagnostic payload | `Diagnostic` | -| Any other value | Producer-defined or unknown classification | `Diagnostic` | +```text +CLOUD_INIT||||[|] +``` -The token changes the record's lifecycle meaning only for `start` and `finish`. It never changes how `name` is interpreted. The original source token remains available in `raw_key`; it does not select a value parser in the common model. In particular, compression is a payload encoding, not a lifecycle kind. +Both formats map to `DiagnosticKey`: -## Proposed read behavior +| Field | Meaning and use | Azure-init source | Cloud-init source | +|---|---|---|---| +| `agent` | Producer and key namespace | First key field | `CLOUD_INIT` | +| `boot_epoch` | Separates records from different VM boots | Key | Key | +| `vm_id` | Identifies the VM when records are exported or aggregated | Key | Key or absent | +| `kind` | Lifecycle role of the record | Key `kind` | Derived from key `type` | +| `name` | Producer-defined operation or subject used for filtering | Key | Key | +| `event_id` | Correlates records chosen by the producer; all parts of one value share it | Key | Key | +| `timestamp` | Time the occurrence was reported | Key (always present) | Best-effort `ts` from value; omitted if missing or unparsable | +| `chunk_index` | Orders physical parts of one value | Numeric suffix (always present) | Numeric suffix, present only when split | -| Command | Result | +`DiagnosticEntry.raw_key` preserves source details that are not in the common model. `DiagnosticEntry.value` preserves the producer-owned message or structured payload. + +## What `kind`, cloud-init `type`, and `name` mean + +`kind` answers “what role does this record have?” `name` answers “what is this record about?” + +| Common `kind` | Meaning | |---|---| -| `read ` | The raw value for that exact KVP key | -| `dump` | Every physical key/value record in pool order | -| `dump --parse-diagnostics` | One parsed record per recognized physical record, including its chunk index and unchanged value | -| `dump --parse-diagnostics --view message` | Reconstruct the source message when its indexed records are available | -| `dump --parse-diagnostics --view content` | Decode a validated encoding envelope such as `{"encoding":"gz+b64","data":"..."}` | +| `Start` | The operation identified by `name` began | +| `Finish` | The operation identified by `name` ended | +| `Diagnostic` | A point-in-time observation about `name` | -Raw parsed diagnostics are the default. `--name` filters by diagnostic name, and `--tail` counts physical records in this view. JSON output keeps the original value as a string. +The same name can appear with different lifecycle roles: -If Message or Content processing fails, that request falls back to the available Raw records rather than hiding them. +| `kind` | `name` | What the record answers | +|---|---|---| +| `Start` | `provision:run` | When did provisioning begin? | +| `Diagnostic` | `provision:run` | What was observed during provisioning? | +| `Finish` | `provision:run` | When did provisioning end? | +| `Diagnostic` | `dmesg` | What `dmesg` data was reported? | + +The source prefix selects the Azure-init or cloud-init key layout. Common `kind` does not select a key layout, rename or parse `name`, or define the value format. Results, durations, messages, and other producer-defined details remain in the value. + +Azure-init writes `start`, `finish`, and `diagnostic` directly and maps them to `Start`, `Finish`, and `Diagnostic`. + +Cloud-init calls its broader source classification `type`. Some values describe lifecycle, some describe a data category, and `compressed` describes representation. They normalize as follows: + +| Cloud-init `type` | Source meaning | Common `kind` | +|---|---|---| +| `start` | Named operation began | `Start` | +| `finish` | Named operation ended | `Finish` | +| `event` | Legacy point event | `Diagnostic` | +| `diagnostic` | Diagnostic point event | `Diagnostic` | +| `system-info` | System, OS, or agent information | `Diagnostic` | +| `boot-telemetry` | Boot timing or boot information | `Diagnostic` | +| `compressed` | Legacy label for encoded diagnostic data | `Diagnostic` | +| Any other value | Producer-defined or unknown classification | `Diagnostic` | -The existing suffix format has no total-part count. Gaps and duplicate indexes can be detected, but a missing final part cannot always be detected; reconstruction of existing records is therefore best effort. +Only `start` and `finish` have common lifecycle meaning. Every other source type is a point-in-time `Diagnostic`. The original cloud-init `type` remains in `raw_key`, but it does not become another common field under this proposal. -## End-to-end example +## Read and write example Write one Azure-init point diagnostic: @@ -150,47 +211,80 @@ Write one Azure-init point diagnostic: libazureinit-kvp emit --prefix azure-init --vm-id vm-123 --name imds --message "Retrieved 1 key from IMDS" ``` -The command uses the supplied agent and VM ID, generates the boot epoch, event ID, and timestamp, then writes: +The command generates the boot epoch, event ID, and timestamp and writes: ```text azure-init|1788371515|vm-123|diagnostic|imds|event-1|2026-09-02T17:52:25Z|0 value: Retrieved 1 key from IMDS ``` -An exact-key read returns: +An exact KVP read returns only the raw value: ```text Retrieved 1 key from IMDS ``` -A parsed diagnostic read returns: +The default diagnostic view returns one entry: ```text -kind=Diagnostic agent=azure-init boot_epoch=1788371515 vm_id=vm-123 name=imds event_id=event-1 timestamp=2026-09-02T17:52:25Z chunk_index=0 value="Retrieved 1 key from IMDS" +raw_key="azure-init|1788371515|vm-123|diagnostic|imds|event-1|2026-09-02T17:52:25Z|0" +kind=Diagnostic agent=azure-init boot_epoch=1788371515 vm_id=vm-123 name=imds event_id=event-1 timestamp=2026-09-02T17:52:25Z chunk_index=0 +value="Retrieved 1 key from IMDS" ``` -A long value may produce two physical records: +A longer value may produce two entries: ```text |0 = "Retrieved " |1 = "1 key from IMDS" ``` -The default diagnostic view returns both records. Message view returns one value: +Raw returns both entries. Message returns `Retrieved 1 key from IMDS`. If part 0 is missing, part 1 remains visible and no reconstructed message is claimed. -```text -Retrieved 1 key from IMDS -``` +## Proposed commands -If part 0 is missing, part 1 remains visible in the default view and no reconstructed message is claimed. +| Command | Result | +|---|---| +| `read ` | Raw value for one exact KVP key | +| `dump` | Every physical KVP record in pool order | +| `dump --parse-diagnostics` | Raw diagnostic view: one entry per recognized physical record | +| `dump --parse-diagnostics --view message` | Reconstructed Message view | +| `dump --parse-diagnostics --view content` | Decoded Content view | -Existing cloud-init records follow the same read contract. Their key metadata is normalized, while their original key and JSON value remain unchanged. +`--name` filters by diagnostic name and `--tail` counts physical records. Both select records before any `--view` transformation, so Message and Content are built from the selected records. JSON output keeps the original value as a string. ## Feedback requested -Please confirm these four decisions: +Please confirm these four design choices. + +### 1. Common `kind` versus source `type` + +Proposed: `DiagnosticKey` exposes only `Start`, `Finish`, or `Diagnostic`. The exact cloud-init `type` remains available in `raw_key` but does not control `name`, key parsing, or value parsing. + +Effect: A consumer that wants to distinguish `system-info`, `boot-telemetry`, or another cloud-init classification must inspect the source key. + +Alternative: add `source_type` to `DiagnosticKey`. Is that distinction important enough to expose directly, or is the normalized lifecycle view sufficient? + +### 2. Default read unit + +Proposed: `dump --parse-diagnostics` returns one `DiagnosticEntry` for every recognized physical record. Message and Content are explicit views. + +Effect: partial payloads remain inspectable. Callers that want one reconstructed event must request it. Filters and tail counts apply to physical records in the default view. + +Alternative: continue returning reconstructed events by default, which is simpler for complete data but can hide incomplete records. Which result should be the primary contract? + +### 3. Transformation failure + +Proposed: if Message reconstruction or Content decoding fails, return the available Raw entries. + +Effect: no physical record disappears because a higher-level view could not be produced, and callers can still inspect the source data. + +Alternative: return an error for the requested view. Should failure fall back to Raw entries or fail the request? + +### 4. Encoding as a first-class step + +Proposed: decoding is an explicit operation (the Content view), directed by a validated `{encoding,data}` envelope in the reconstructed value rather than guessed from the key. + +Effect: encoding remains independent of lifecycle `kind`. Existing cloud-init `type=compressed` remains readable, but new Azure-init data does not need a `compressed` kind merely to carry encoded content. -1. Keep physical KVP, indexed framing, and diagnostics as separate concepts, while leaving indexed framing internal for now. -2. Use only `Start`, `Finish`, and `Diagnostic` in the common model. Preserve cloud-init's exact `type` in the raw key rather than assigning it cross-source semantics. -3. Make one physical diagnostic record the default result and request higher-level processing with `--view message` or `--view content`. -4. Treat encoding as a validated value envelope rather than `type=compressed`; if a higher-level view fails, preserve the available physical records. +Alternative: inspect the envelope only when a source token such as `type=compressed` indicates encoded content. That would require exposing or deriving that hint and deciding whether Azure-init also writes `compressed`. Should encoding be payload-directed or source-token-gated? A related question is whether to keep Raw, Message, and Content as three views or collapse Message and Content into a single Decoded view, since they differ only for encoded payloads. From 3586085e77365ada669b5cae17dd0071c4172e16 Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Wed, 2 Sep 2026 17:38:38 -0700 Subject: [PATCH 16/32] docs(kvp): define physical diagnostics and decoding proposal Clarify the KVP and diagnostics boundaries, normalized public model, and physical-record read, write, and CLI contracts. --- libazureinit-kvp/diagnostics-proposal.md | 375 ++++++++++++----------- 1 file changed, 198 insertions(+), 177 deletions(-) diff --git a/libazureinit-kvp/diagnostics-proposal.md b/libazureinit-kvp/diagnostics-proposal.md index 9653206f..845be2ff 100644 --- a/libazureinit-kvp/diagnostics-proposal.md +++ b/libazureinit-kvp/diagnostics-proposal.md @@ -1,290 +1,311 @@ # KVP Diagnostics Proposal -## Purpose +## Summary -Hyper-V KVP is a guest-to-host exchange of physical key/value records. Each record has a size limit. KVP preserves records but does not define diagnostics, splitting, reconstruction, or payload formats. +The base proposal is a typed view of physical KVP records. -Azure-init and cloud-init are guest provisioning agents that write diagnostic records to KVP. This proposal defines one way to read both formats and one Azure-init format for new writes. - -This proposal leaves KVP unchanged and changes the diagnostic view: - -- The default result is one entry per recognized physical diagnostic record, with its raw key and value unchanged. -- Splitting and grouping records is separate from diagnostic meaning. -- Message reconstruction and content decoding happen only when requested. +Every recognized Azure-init or cloud-init record produces one `DiagnosticEntry`. Its key is normalized into `DiagnosticKey`, and its physical value is returned unchanged. Reading diagnostics does not automatically group chunks, extract messages, or decompress content. ## Architecture -### Existing foundation: `KvpPoolStore` +`KvpPoolStore` is the existing foundation. Above it, the proposal separates the two concerns that are currently combined: -`KvpPoolStore` reads and writes one Hyper-V KVP pool. It returns physical key/value records exactly as stored and has no diagnostic meaning. +1. Spanned values handle `base-key|chunk-index`, splitting, grouping, and ordering. They do not know what the key or value means. +2. Diagnostics defines the base key, normalizes its fields, and knows how source values would be combined when payload decoding is requested. -### Layer 1: indexed values +These are separate responsibilities, but spanned values are not a new public API. The public API remains `KvpPoolStore` for raw KVP and `DiagnosticsPool` for diagnostics. `DiagnosticKey` and `DiagnosticEntry` are the public data model returned by the diagnostics layer. -A value that exceeds one KVP record is split across records. Each part uses the same base key plus a numeric suffix: +### API layers -```text -|0 = first value fragment -|1 = second value fragment -``` +| Component | Input and output | Responsibility | Why it exists | +|---|---|---|---| +| `KvpPoolStore` | Reads and writes physical key/value records | Preserve the KVP pool exactly, in pool order | Existing general-purpose KVP access with no diagnostic meaning | +| `DiagnosticsPool` | Uses one `KvpPoolStore`; returns `DiagnosticEntry` values and writes Azure-init diagnostics | Recognize diagnostic keys, normalize their metadata, and preserve each physical value | Give callers one diagnostic view across Azure-init and cloud-init | -Indexed framing splits values for writes and identifies, groups, and orders parts for reconstruction. It understands only a caller-defined base key, index, and value fragments. It does not understand diagnostic fields, JSON, or compression. +`KvpPoolStore` owns storage; `DiagnosticsPool` owns diagnostic meaning and delegates all physical reads and writes to the store. Raw KVP callers use the store directly. Diagnostic callers use the pool layered over it. -Indexed framing is decoupled from diagnostics in the code, but it is not a separate public API. Callers use raw KVP or diagnostics; nothing else needs to span values today. +### Internal operations -### Layer 2: diagnostics +| Operation | Used when | Responsibility | Not responsible for | +|---|---|---|---| +| Key normalization | Every diagnostic read | Convert either source key into `DiagnosticKey` and best-effort obtain cloud-init's timestamp | Grouping chunks or interpreting payload content | +| Chunk framing | Writing a long value | Split the value and add numeric key suffixes | Diagnostic meaning | +| Chunk grouping | Payload decoding is explicitly requested | Group, validate, and order entries from one event | The default physical-entry read | +| Payload transformation | Message or Content is explicitly requested | Extract a logical message and optionally decode compressed content | Key parsing or timestamp extraction | + +Chunk framing and grouping remain internal. They are separate from diagnostic meaning, but there is no separate public indexed-value API in this proposal. -Diagnostics defines the base-key formats and how ordered source values become a logical message. It uses three components: +`DiagnosticKey` and `DiagnosticEntry` are data models within `DiagnosticsPool`, not additional architecture layers. -#### `DiagnosticKey`: normalized metadata +### How the pieces interact -`DiagnosticKey` contains diagnostic metadata parsed from one Azure-init or cloud-init key: +Default read: ```text -DiagnosticKey { - agent, boot_epoch, vm_id?, kind, name, - event_id, timestamp?, chunk_index? -} +KvpPoolStore.dump() + -> one physical key/value record + -> DiagnosticsPool normalizes its key + -> one DiagnosticEntry with the unchanged physical value ``` -Its purpose is to give both source formats one common key model. - -#### `DiagnosticEntry`: one physical diagnostic record +This repeats independently for every record. No grouping occurs. -`DiagnosticEntry` is the default unit returned by a diagnostic read: +Write: ```text -DiagnosticEntry { - raw_key, - key: DiagnosticKey, - value -} +caller + -> DiagnosticsPool creates an Azure-init key and value + -> internal chunk framing splits the value when necessary + -> KvpPoolStore appends the physical record or records ``` -The raw key and value remain unchanged. `DiagnosticKey` provides the normalized metadata. A value split into four KVP records produces four `DiagnosticEntry` values. +Optional payload decoding: -#### `DiagnosticsPool`: diagnostic access +```text +DiagnosticEntry values + -> internal chunk grouping + -> source-specific message reconstruction + -> optional decoding and decompression +``` -`DiagnosticsPool` is the diagnostic interface composed over one `KvpPoolStore` and the indexed framing rules. +This optional path transforms payloads only. It does not change the default entries or the metadata already extracted from their keys. -On read, it returns one `DiagnosticEntry` per recognized physical record, in pool order. A record is recognized when its key matches the Azure-init or cloud-init layout; other keys, such as `PROVISIONING_REPORT`, are skipped in the diagnostic view but remain visible through `dump`. Optional fields (`vm_id`, `timestamp`, `chunk_index`) are omitted from output when absent rather than shown as empty. On write, it creates Azure-init keys, uses indexed framing when a value must be split, and writes the resulting records through `KvpPoolStore`. +## Base public model ```text -Read: -KvpPoolStore -> physical records -> DiagnosticsPool -> DiagnosticEntry values +DiagnosticKind = Start | Finish | Diagnostic -Write: -caller -> DiagnosticsPool -> indexed framing -> KvpPoolStore -``` - -## Reading a value back: Raw, Message, Content +DiagnosticKey { + agent, + boot_epoch, + vm_id?, + kind, + name, + event_id, + timestamp?, + chunk_index? +} -A diagnostic value can be wrapped up to three times before it lands in KVP: +DiagnosticEntry { + key: DiagnosticKey, + value +} +``` -1. Split — a value larger than one record is spread across several physical records (indexed values). -2. Wrapped — the producer stores the message inside its own structure, such as cloud-init's JSON `{"...","msg":"..."}`. -3. Encoded — the message itself may be compressed and base64-encoded, such as a captured `dmesg` or `cloud-init.log`. +`DiagnosticKey` describes the record. `DiagnosticEntry` combines that description with the unchanged physical value. -Reading a value back means choosing how far to unwrap it. The three views are read-time transformations, not stored data: +| Field | Required | Source | Meaning and concrete use | +|---|---|---|---| +| `agent` | Yes | Key | Producer of the record; distinguishes `CLOUD_INIT` from Azure-init | +| `boot_epoch` | Yes | Key | Boot that produced the record; separates current-boot from previous-boot data | +| `vm_id` | No | Key | VM identity; supports host-side collection while remaining compatible with older cloud-init keys that omit it | +| `kind` | Yes | Key `type` token | Common lifecycle meaning; tells a reader whether `provision:run` began, ended, or emitted a point observation | +| `name` | Yes | Key | Actual producer-defined subject; lets a caller filter for `dmesg` or `provision:run` without parsing the value | +| `event_id` | Yes | Key | Identity shared by physical chunks from one emission; associates those chunks without inspecting their values | +| `timestamp` | No | Azure-init key or cloud-init `ts` | Time the occurrence was reported; remains absent when cloud-init metadata is unreadable | +| `chunk_index` | No | Numeric key suffix | Position of this physical fragment; makes chunks `16` through `19` useful even when earlier chunks are gone | +| `value` | Yes | Physical KVP value | Original fragment exactly as stored; preserves incomplete or invalid JSON for inspection | -| View | Unwraps | Result | -|---|---|---| -| Raw | Nothing | Each physical record exactly as stored | -| Message | Split and producer wrapper | The producer's logical message | -| Content | Split, wrapper, and encoding | The decoded, decompressed payload | +The optional fields are optional in the common model because the supported source formats do not always provide them. Their absence never causes an otherwise valid physical entry to be discarded. -### Example: a compressed `dmesg` +## Key normalization -Stored as two physical records. Each is a cloud-init JSON wrapper carrying one fragment of the message: +Azure-init keys: ```text -Raw: - CLOUD_INIT|...|compressed|dmesg|...|0 = {"...","msg_i":0,"msg":"{\"encoding\":\"gz+b64\",\"da"} - CLOUD_INIT|...|compressed|dmesg|...|1 = {"...","msg_i":1,"msg":"ta\":\"H4sIA...\"}"} +||||||[|] ``` -Reassemble the fragments into the producer's message. For a compressed event that message is an encoding envelope, not yet readable text: +Current cloud-init keys: ```text -Message: - {"encoding":"gz+b64","data":"H4sIA..."} +CLOUD_INIT|||||[|] ``` -Decode and decompress the envelope to get the actual artifact: +Older cloud-init keys omit `vm_id`: ```text -Content: - [ 0.000000] Linux version 6.8.0-azure ... +CLOUD_INIT||||[|] ``` -### First-class encoding mechanism +All three layouts map into the same `DiagnosticKey`. -Undoing the split and the producer wrapper is mechanical. Decoding is not: it needs a defined envelope, a codec, and a decompression step. Today no supported operation does this, so a compressed diagnostic can only be read back as an opaque `{encoding,data}` blob. The proposal is to make decoding a first-class, explicitly requested step — the Content view — driven by a validated `{encoding,data}` envelope in the value rather than guessed from the key. +Azure-init provides its timestamp in the key. Cloud-init provides `ts` in the value, so `DiagnosticsPool` reads only enough metadata to find it. If `ts` is missing or unreadable, `timestamp` is absent and the entry is still returned. -For any value that is not encoded there is nothing to decode, so Message and Content are identical: +## What `type` and `name` mean -```text -Raw: |0 = "Retrieved ", |1 = "1 key from IMDS" -Message: "Retrieved 1 key from IMDS" -Content: "Retrieved 1 key from IMDS" (no envelope, nothing to decode) -``` +The source key contains a `type` token and a `name`. -Raw returns one result per physical record. Message and Content return one result per event, that is, per set of records sharing a base key. Both require a contiguous run of parts beginning at `chunk_index` 0; if the first or a middle part is missing, no message is produced and the available parts remain as Raw entries. The suffix format has no total-part count, so a missing final part cannot always be detected; reconstruction of existing records is therefore best effort. +- `name` answers: what is this record about? +- `type` answers: what did the source report about that name? -## Normalizing Azure-init and cloud-init +The base proposal retains only the common meaning needed by readers: `kind`. It does not retain the exact source type as another public field. -Azure-init keys: +Under this proposal, `type` has a narrow key-level role: -```text -||||||[|] -``` +- It does not select the Azure-init or cloud-init key layout; the key shape and source prefix do that. +- It does not define, parse, or validate the client-owned value during the default read. +- It derives `kind`, so lifecycle can be understood without parsing the value. -Current cloud-init keys: + +| Source `type` | Common `kind` | Meaning | +|---|---|---| +| `start` | `Start` | The operation identified by `name` began | +| `finish` | `Finish` | The operation identified by `name` ended | +| Any other type, including `compressed` | `Diagnostic` | A point-in-time observation about `name` | + +`type` does not rename or parse `name`. For example: ```text -CLOUD_INIT|||||[|] +type=start name=provision:run -> provisioning began +type=finish name=provision:run -> provisioning ended +type=system-info name=system information -> system information was reported +type=compressed name=dmesg -> dmesg data was reported ``` -Older cloud-init keys omit `vm_id`: +The common results are: ```text -CLOUD_INIT||||[|] +Start name=provision:run +Finish name=provision:run +Diagnostic name=system information +Diagnostic name=dmesg ``` -Both formats map to `DiagnosticKey`: +Existing `event`, `diagnostic`, `system-info`, `boot-telemetry`, and unknown types all become `Diagnostic`. Their actual `name` remains unchanged. Existing Azure-init `event` records remain readable; new Azure-init point records use `diagnostic`. -| Field | Meaning and use | Azure-init source | Cloud-init source | -|---|---|---|---| -| `agent` | Producer and key namespace | First key field | `CLOUD_INIT` | -| `boot_epoch` | Separates records from different VM boots | Key | Key | -| `vm_id` | Identifies the VM when records are exported or aggregated | Key | Key or absent | -| `kind` | Lifecycle role of the record | Key `kind` | Derived from key `type` | -| `name` | Producer-defined operation or subject used for filtering | Key | Key | -| `event_id` | Correlates records chosen by the producer; all parts of one value share it | Key | Key | -| `timestamp` | Time the occurrence was reported | Key (always present) | Best-effort `ts` from value; omitted if missing or unparsable | -| `chunk_index` | Orders physical parts of one value | Numeric suffix (always present) | Numeric suffix, present only when split | +## Default read contract -`DiagnosticEntry.raw_key` preserves source details that are not in the common model. `DiagnosticEntry.value` preserves the producer-owned message or structured payload. +`DiagnosticsPool::entries()` returns one `DiagnosticEntry` for each recognized physical record, in pool order. -## What `kind`, cloud-init `type`, and `name` mean +It does: -`kind` answers “what role does this record have?” `name` answers “what is this record about?” +- Parse key metadata into `DiagnosticKey`. +- Best-effort extract cloud-init's `ts`. +- Preserve the physical value unchanged. +- Return duplicate, partial, and out-of-sequence chunks independently. -| Common `kind` | Meaning | -|---|---| -| `Start` | The operation identified by `name` began | -| `Finish` | The operation identified by `name` ended | -| `Diagnostic` | A point-in-time observation about `name` | +It does not: -The same name can appear with different lifecycle roles: +- Group or order chunks. +- Require a complete chunk sequence. +- Validate cloud-init `msg_i`. +- Extract or unescape `msg`. +- Extract `result` or `duration`. +- Decode or decompress content. -| `kind` | `name` | What the record answers | -|---|---|---| -| `Start` | `provision:run` | When did provisioning begin? | -| `Diagnostic` | `provision:run` | What was observed during provisioning? | -| `Finish` | `provision:run` | When did provisioning end? | -| `Diagnostic` | `dmesg` | What `dmesg` data was reported? | +Example: if only `dmesg` chunks 16 through 19 remain, the read returns four entries: -The source prefix selects the Azure-init or cloud-init key layout. Common `kind` does not select a key layout, rename or parse `name`, or define the value format. Results, durations, messages, and other producer-defined details remain in the value. +```text +Diagnostic name=dmesg chunk_index=16 value= +Diagnostic name=dmesg chunk_index=17 value= +Diagnostic name=dmesg chunk_index=18 value= +Diagnostic name=dmesg chunk_index=19 value= +``` -Azure-init writes `start`, `finish`, and `diagnostic` directly and maps them to `Start`, `Finish`, and `Diagnostic`. +Missing chunks 0 through 15 do not hide the records that are present. A cloud-init chunk also remains visible when a split inside `msg` makes that individual value invalid JSON. -Cloud-init calls its broader source classification `type`. Some values describe lifecycle, some describe a data category, and `compressed` describes representation. They normalize as follows: +## Write contract -| Cloud-init `type` | Source meaning | Common `kind` | -|---|---|---| -| `start` | Named operation began | `Start` | -| `finish` | Named operation ended | `Finish` | -| `event` | Legacy point event | `Diagnostic` | -| `diagnostic` | Diagnostic point event | `Diagnostic` | -| `system-info` | System, OS, or agent information | `Diagnostic` | -| `boot-telemetry` | Boot timing or boot information | `Diagnostic` | -| `compressed` | Legacy label for encoded diagnostic data | `Diagnostic` | -| Any other value | Producer-defined or unknown classification | `Diagnostic` | +`DiagnosticsPool` writes Azure-init records only. -Only `start` and `finish` have common lifecycle meaning. Every other source type is a point-in-time `Diagnostic`. The original cloud-init `type` remains in `raw_key`, but it does not become another common field under this proposal. +- A point diagnostic uses the `diagnostic` token and generates its event ID and timestamp. +- Explicit writes may use `Start`, `Finish`, or `Diagnostic` with caller-supplied event metadata. +- Values larger than one record are split at safe text boundaries and written with numeric chunk suffixes. +- All chunks from one write are appended together. -## Read and write example - -Write one Azure-init point diagnostic: +Example: ```text -libazureinit-kvp emit --prefix azure-init --vm-id vm-123 --name imds --message "Retrieved 1 key from IMDS" -``` +emit diagnostic: name=imds, value="Retrieved 1 key from IMDS" -The command generates the boot epoch, event ID, and timestamp and writes: - -```text azure-init|1788371515|vm-123|diagnostic|imds|event-1|2026-09-02T17:52:25Z|0 -value: Retrieved 1 key from IMDS += Retrieved 1 key from IMDS ``` -An exact KVP read returns only the raw value: +A long value produces multiple physical records. The default read returns each one as a separate `DiagnosticEntry`. -```text -Retrieved 1 key from IMDS -``` +## CLI contract -The default diagnostic view returns one entry: +`dump --parse-diagnostics` means classify physical diagnostic records. -```text -raw_key="azure-init|1788371515|vm-123|diagnostic|imds|event-1|2026-09-02T17:52:25Z|0" -kind=Diagnostic agent=azure-init boot_epoch=1788371515 vm_id=vm-123 name=imds event_id=event-1 timestamp=2026-09-02T17:52:25Z chunk_index=0 -value="Retrieved 1 key from IMDS" -``` +Each output item contains the fields from `DiagnosticKey` plus the unchanged physical value. -A longer value may produce two entries: +- `--name` filters by the actual `name`. +- `--tail` counts physical records. +- JSON output represents the KVP value as a string, even when that string contains JSON. +- Decoded `message`, `result`, and `duration` are not part of this default output. -```text -|0 = "Retrieved " -|1 = "1 key from IMDS" -``` +## Separate proposal: first-class payload decoding -Raw returns both entries. Message returns `Retrieved 1 key from IMDS`. If part 0 is missing, part 1 remains visible and no reconstructed message is claimed. +The base design above always returns physical entries. Some callers may also need one logical message or the decompressed artifact contained by that message. That transformation should be explicit and must not change key parsing or timestamp extraction. -## Proposed commands +There are three possible levels: -| Command | Result | +| Level | Result | |---|---| -| `read ` | Raw value for one exact KVP key | -| `dump` | Every physical KVP record in pool order | -| `dump --parse-diagnostics` | Raw diagnostic view: one entry per recognized physical record | -| `dump --parse-diagnostics --view message` | Reconstructed Message view | -| `dump --parse-diagnostics --view content` | Decoded Content view | +| Raw | Every physical `DiagnosticEntry`, unchanged | +| Message | Group a complete chunk sequence and reconstruct the logical message | +| Content | Perform Message processing, then decode compressed content | -`--name` filters by diagnostic name and `--tail` counts physical records. Both select records before any `--view` transformation, so Message and Content are built from the selected records. JSON output keeps the original value as a string. +For Azure-init, Message concatenates value chunks. For cloud-init, Message joins the `msg` fragments and removes the outer telemetry JSON. -## Feedback requested +For an ordinary diagnostic: + +```text +Raw: {"name":"imds","msg":"Retrieved 1 key from IMDS",...} +Message: Retrieved 1 key from IMDS +Content: Retrieved 1 key from IMDS +``` -Please confirm these four design choices. +For a complete compressed `dmesg` event: -### 1. Common `kind` versus source `type` +```text +Raw: separate physical cloud-init chunks +Message: {"encoding":"gz+b64","data":"..."} +Content: decompressed dmesg bytes or text +``` -Proposed: `DiagnosticKey` exposes only `Start`, `Finish`, or `Diagnostic`. The exact cloud-init `type` remains available in `raw_key` but does not control `name`, key parsing, or value parsing. +The base `DiagnosticKey` does not identify encoded content. If Content decoding is approved, one of these signals must also be selected: -Effect: A consumer that wants to distinguish `system-info`, `boot-telemetry`, or another cloud-init classification must inspect the source key. +| Option | Design | Effect | +|---|---|---| +| Key-derived flag | Add `compressed` to `DiagnosticKey`, derived from `type=compressed` | Identifies partial chunks before reconstruction, but gives one source token special public meaning | +| Complete source type | Add `source_type` to `DiagnosticKey` | Preserves `compressed` and every other producer classification, but expands the common model | +| Reconstructed envelope | Keep the base model unchanged and inspect the completed `{encoding,data}` message | Keeps encoding in the payload, but cannot identify incomplete chunks as encoded | -Alternative: add `source_type` to `DiagnosticKey`. Is that distinction important enough to expose directly, or is the normalized lifecycle view sufficient? +In all three options, a valid envelope determines the codec. The signal only determines whether Content decoding should be attempted. Content then base64-decodes the data and applies the declared decompression. -### 2. Default read unit +If a requested transformation cannot be completed, the proposed behavior is to return the original Raw entries rather than discard them. -Proposed: `dump --parse-diagnostics` returns one `DiagnosticEntry` for every recognized physical record. Message and Content are explicit views. +## Feedback requested -Effect: partial payloads remain inspectable. Callers that want one reconstructed event must request it. Filters and tail counts apply to physical records in the default view. +Sign-off is needed on both the base physical-entry design and the optional payload-decoding mechanism. -Alternative: continue returning reconstructed events by default, which is simpler for complete data but can hide incomplete records. Which result should be the primary contract? +“First-class” means callers select one named payload level. They do not manually combine separate grouping, message-extraction, base64, and decompression flags. The proposed public shape is: + +```text +Default API read: entries() -> Raw DiagnosticEntry records +Optional API read: payload view = Message | Content + +CLI: + dump --parse-diagnostics -> Raw + dump --parse-diagnostics --view message -> Message + dump --parse-diagnostics --view content -> Content +``` -### 3. Transformation failure +Message and Content automatically perform their required earlier stages. This public control is part of the proposal and also requires approval. -Proposed: if Message reconstruction or Content decoding fails, return the available Raw entries. +1. Base design: Does one physical `DiagnosticEntry` per record, with the model and normalization above, match the intended diagnostics contract? Does keeping spanned-value handling separate but internal, with diagnostics layered on top of it, match the intended architecture? -Effect: no physical record disappears because a higher-level view could not be produced, and callers can still inspect the source data. +2. Purpose of `type`: Should `type` remain a key field whose base public effect is deriving `kind`, while the value schema remains client-owned? Should lifecycle continue to use `type=start` and `type=finish` in the key, or should Start and Finish use a different key structure? -Alternative: return an error for the requested view. Should failure fall back to Raw entries or fail the request? +3. Decoding levels: Should the first-class mechanism expose Raw, Message, and Content, or only Raw and fully decoded Content? -### 4. Encoding as a first-class step +4. Public control: Should `entries()` remain the unchanged Raw read while one separate payload-view operation and matching `--view` option select Message or Content? If only Raw and Content are needed, should the option be named `--decode` instead of `--view content`? -Proposed: decoding is an explicit operation (the Content view), directed by a validated `{encoding,data}` envelope in the reconstructed value rather than guessed from the key. +5. Decoded content type: Should Content return bytes, or require UTF-8 text? -Effect: encoding remains independent of lifecycle `kind`. Existing cloud-init `type=compressed` remains readable, but new Azure-init data does not need a `compressed` kind merely to carry encoded content. +6. Failure behavior: When chunks are incomplete or decoding fails, should the operation return the original Raw entries or an error? -Alternative: inspect the envelope only when a source token such as `type=compressed` indicates encoded content. That would require exposing or deriving that hint and deciding whether Azure-init also writes `compressed`. Should encoding be payload-directed or source-token-gated? A related question is whether to keep Raw, Message, and Content as three views or collapse Message and Content into a single Decoded view, since they differ only for encoded payloads. +7. Encoded-content signal: If Content decoding is approved, should it use a key-derived `compressed` flag, retain the complete `source_type`, or inspect only the reconstructed `{encoding,data}` envelope? Is this proposal limited to decoding existing encoded content, or should it also define how Azure-init compresses and writes new encoded content? \ No newline at end of file From afca2d3ab4d1a7912a18918dc6e774e84c315ed2 Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Fri, 4 Sep 2026 09:31:02 -0700 Subject: [PATCH 17/32] docs(kvp): add versioned diagnostic reader and writer design --- libazureinit-kvp/diagnostics-proposal.md | 553 ++++++++++++++--------- 1 file changed, 351 insertions(+), 202 deletions(-) diff --git a/libazureinit-kvp/diagnostics-proposal.md b/libazureinit-kvp/diagnostics-proposal.md index 845be2ff..d154c81e 100644 --- a/libazureinit-kvp/diagnostics-proposal.md +++ b/libazureinit-kvp/diagnostics-proposal.md @@ -1,311 +1,460 @@ -# KVP Diagnostics Proposal +# KVP Diagnostics Specification -## Summary +## Background -The base proposal is a typed view of physical KVP records. +Diagnostics are the records a provisioning client emits to explain what a boot did: spans that mark when an operation such as `provision:run` starts and finishes, and point observations that capture a single event or artifact, such as an IMDS result or a `dmesg` snapshot. They let an operator or the host reconstruct a boot and triage a failure after the fact, including when the guest is no longer reachable. -Every recognized Azure-init or cloud-init record produces one `DiagnosticEntry`. Its key is normalized into `DiagnosticKey`, and its physical value is returned unchanged. Reading diagnostics does not automatically group chunks, extract messages, or decompress content. +A Hyper-V guest carries these records to the host through a KVP pool: a flat namespace of key=value records that the host copies out of the running guest. We already have `KvpPoolStore`, which gives raw, in-order access to that pool but no notion of diagnostics. -## Architecture +The pool constrains the design in four ways: -`KvpPoolStore` is the existing foundation. Above it, the proposal separates the two concerns that are currently combined: +- A record is a flat key=value pair. The key is the only structured metadata; the value is opaque bytes to the pool. +- Keys and values are size-capped. The safe key limit is 254 bytes and the safe value limit is 1022 bytes, so a message larger than the value limit cannot fit in one record and must span several records that share a key and differ only by a trailing index. +- The host may copy or truncate the pool at any moment. A reader can find a group whose members are missing, duplicated, or out of order. +- There is no native grouping or typing. A diagnostic event exists only as a convention encoded in the key and the value. -1. Spanned values handle `base-key|chunk-index`, splitting, grouping, and ordering. They do not know what the key or value means. -2. Diagnostics defines the base key, normalizes its fields, and knows how source values would be combined when payload decoding is requested. +This spec defines a next-generation diagnostics format for a provisioning client (azure-init), the writer that produces it, and the reader that interprets the pool. cloud-init also writes diagnostics into the same pool in its own format; reading those is a separate read-only concern (see Compatibility). -These are separate responsibilities, but spanned values are not a new public API. The public API remains `KvpPoolStore` for raw KVP and `DiagnosticsPool` for diagnostics. `DiagnosticKey` and `DiagnosticEntry` are the public data model returned by the diagnostics layer. +### Records today -### API layers +Both producers already write diagnostics into the pool, in two different shapes. -| Component | Input and output | Responsibility | Why it exists | -|---|---|---|---| -| `KvpPoolStore` | Reads and writes physical key/value records | Preserve the KVP pool exactly, in pool order | Existing general-purpose KVP access with no diagnostic meaning | -| `DiagnosticsPool` | Uses one `KvpPoolStore`; returns `DiagnosticEntry` values and writes Azure-init diagnostics | Recognize diagnostic keys, normalize their metadata, and preserve each physical value | Give callers one diagnostic view across Azure-init and cloud-init | +azure-init uses a pipe-delimited key with a plain-text value and no encoding field. The `type` token is `start`, `finish`, or `event`; a value over the size limit is split with a trailing chunk index: -`KvpPoolStore` owns storage; `DiagnosticsPool` owns diagnostic meaning and delegates all physical reads and writes to the store. Raw KVP callers use the store directly. Diagnostic callers use the pool layered over it. - -### Internal operations +```text +||||||| -| Operation | Used when | Responsibility | Not responsible for | -|---|---|---|---| -| Key normalization | Every diagnostic read | Convert either source key into `DiagnosticKey` and best-effort obtain cloud-init's timestamp | Grouping chunks or interpreting payload content | -| Chunk framing | Writing a long value | Split the value and add numeric key suffixes | Diagnostic meaning | -| Chunk grouping | Payload decoding is explicitly requested | Group, validate, and order entries from one event | The default physical-entry read | -| Payload transformation | Message or Content is explicitly requested | Extract a logical message and optionally decode compressed content | Key parsing or timestamp extraction | +# point event, one record +azure-init-0.1.1|1700000000|vm-abc|event|imds|8f3e9c4a-1b2c-4d5e-9f01-234567890abc|2026-07-27T21:33:24.300Z|0 + value: Retrieved 1 key from IMDS -Chunk framing and grouping remain internal. They are separate from diagnostic meaning, but there is no separate public indexed-value API in this proposal. +# span start +azure-init-0.1.1|1700000000|vm-abc|start|provision:run|9c1d2e3f-4a5b-6c7d-8e9f-0a1b2c3d4e5f|2026-08-31T12:34:56.789Z|0 + value: starting -`DiagnosticKey` and `DiagnosticEntry` are data models within `DiagnosticsPool`, not additional architecture layers. +# span finish, same event_id as its start +azure-init-0.1.1|1700000000|vm-abc|finish|provision:run|9c1d2e3f-4a5b-6c7d-8e9f-0a1b2c3d4e5f|2026-08-31T12:34:57.101Z|0 + value: provisioning succeeded -### How the pieces interact +# long value split across records, one event_id, indices 0..N +azure-init-0.1.1|1700000000|vm-abc|event|config:dump|1a2b3c4d-5e6f-7a8b-9c0d-1e2f3a4b5c6d||0 value: +azure-init-0.1.1|1700000000|vm-abc|event|config:dump|1a2b3c4d-5e6f-7a8b-9c0d-1e2f3a4b5c6d||1 value: +``` -Default read: +cloud-init puts the type and name before the identifiers; a current key carries a `vm_id` that an older key omits. The value is JSON with `ts` and `msg`, `result` and `duration` on a finish, `msg_i` per chunk on a split value, and an `{encoding, data}` envelope inside `msg` when compressed: ```text -KvpPoolStore.dump() - -> one physical key/value record - -> DiagnosticsPool normalizes its key - -> one DiagnosticEntry with the unchanged physical value +current CLOUD_INIT|||||[|] +older CLOUD_INIT||||[|] + +# span finish with result and duration (current key) +CLOUD_INIT|1785187982|finish|modules-final/config-scripts_user|0e5e179d-5341-478b-8456-fbb90621bdf8|e5f01809-a7a3-4279-aa64-1f18e21eda6e + value: {"name":"modules-final/config-scripts_user","type":"finish","ts":"2026-07-27T21:33:24.339006+00:00","result":"SUCCESS","duration":0.00064,"msg":"config-scripts_user ran successfully and took 0.001 seconds"} + msg -> "config-scripts_user ran successfully and took 0.001 seconds" + +# span start, no result or duration +CLOUD_INIT|1785187982|start|modules-final/config-ssh_authkey_fingerprints|0e5e179d-5341-478b-8456-fbb90621bdf8|c4d4a08d-fe93-4c7a-9be6-9a38c212e212 + value: {"name":"modules-final/config-ssh_authkey_fingerprints","type":"start","ts":"2026-07-27T21:33:24.339170+00:00","msg":"running config-ssh_authkey_fingerprints with frequency once-per-instance"} + msg -> "running config-ssh_authkey_fingerprints with frequency once-per-instance" + +# older key without vm_id (five base fields) +CLOUD_INIT|1785187982|finish|modules-final|126f969f-13fd-4b4b-a136-b7114518491f + value: {"name":"modules-final","type":"finish","ts":"2026-07-27T21:33:24.431885+00:00","result":"SUCCESS","duration":0.340712044,"msg":"running modules for final"} + msg -> "running modules for final" + +# point event +CLOUD_INIT|1785187982|event|network-config|0e5e179d-5341-478b-8456-fbb90621bdf8|a1b2c3d4-e5f6-7a8b-9c0d-1e2f3a4b5c6d + value: {"name":"network-config","type":"event","ts":"2026-07-27T21:33:20.100000+00:00","msg":"applied fallback network configuration"} + msg -> "applied fallback network configuration" + +# system-info (not a timeline position; the subject is in name) +CLOUD_INIT|1785187982|system-info|system information|0e5e179d-5341-478b-8456-fbb90621bdf8|b2c3d4e5-f6a7-8b9c-0d1e-2f3a4b5c6d7e + value: {"name":"system information","type":"system-info","ts":"2026-07-27T21:33:19.500000+00:00","msg":"cloud-init running on Ubuntu"} + msg -> "cloud-init running on Ubuntu" + +# split value: each chunk carries msg_i, and a JSON \n escape is split across the boundary +CLOUD_INIT|1785187982|finish|modules-final|0e5e179d-5341-478b-8456-fbb90621bdf8|c3d4e5f6-a7b8-9c0d-1e2f-3a4b5c6d7e8f|0 + value: {"name":"modules-final","type":"finish","ts":"2026-07-27T21:33:24.43Z","msg_i":0,"msg":"line1\"} +CLOUD_INIT|1785187982|finish|modules-final|0e5e179d-5341-478b-8456-fbb90621bdf8|c3d4e5f6-a7b8-9c0d-1e2f-3a4b5c6d7e8f|1 + value: {"name":"modules-final","type":"finish","ts":"2026-07-27T21:33:24.43Z","msg_i":1,"msg":"nline2"} + msg -> "line1\nline2" (reassembled from the two chunks) + +# compressed artifact: type=compressed, msg holds an {encoding, data} envelope; a large one splits like the value above +CLOUD_INIT|1785187982|compressed|dmesg|0e5e179d-5341-478b-8456-fbb90621bdf8|d4e5f6a7-b8c9-0d1e-2f3a-4b5c6d7e8f90|0 + value: {"name":"dmesg","type":"compressed","ts":"2026-07-27T21:33:25.00Z","msg_i":0,"msg":"{\"encoding\":\"gz+b64\",\"data\":\"H4sIAAAA...\"}"} + msg -> {"encoding":"gz+b64","data":"H4sIAAAA..."} ``` -This repeats independently for every record. No grouping occurs. +The two disagree on field order, on where the timestamp and encoding live, and on whether the value is text or JSON. -Write: +## Proposed design -```text -caller - -> DiagnosticsPool creates an Azure-init key and value - -> internal chunk framing splits the value when necessary - -> KvpPoolStore appends the physical record or records -``` +This proposal defines one versioned record format for azure-init and separate reader and writer interfaces over `KvpPoolStore`. The reader interprets the pool; the writer produces only the azure-init format. Untyped access remains on `KvpPoolStore` itself. + +### Diagnostics format -Optional payload decoding: +The proposed key is the current azure-init key with a few changes, all keeping metadata in the key. It is pipe-delimited with `|` reserved, begins with a diagnostic schema-version identifier, and always ends with the chunk index, so a single-record value still ends in `|0`: ```text -DiagnosticEntry values - -> internal chunk grouping - -> source-specific message reconstruction - -> optional decoding and decompression +AZURE_INIT_V1||||||||||| ``` -This optional path transforms payloads only. It does not change the default entries or the metadata already extracted from their keys. +- `AZURE_INIT_V1` is the `diagnostic_version_id`: `AZURE_INIT` identifies the diagnostics family and `V1` identifies version 1 of its wire schema. That schema covers the field layout, required and optional fields, token meanings, units, encoding rules, and chunk framing. It is one self-identifying token rather than a bare number because the pool also contains unrelated keys: a reader can recognize an unsupported future `AZURE_INIT_V*` record without mistaking an arbitrary numeric key for a diagnostic. It is separate from `agent`, which remains the producer identifier and whose version identifies the producer binary. A schema change requires a new `diagnostic_version_id`; an agent release by itself does not. +- `type` becomes `kind`, narrowed to the three timeline positions (`start`/`finish`/`event`); any other category is carried by `name`, `encoding`, or `result`, not a token. +- `encoding` names the payload encoding in the key so a large artifact can be compressed. +- `result` and `duration` are the finish's verdict fields, carried in the key: `result` is a `success`/`fail` token and `duration` is the elapsed milliseconds. Both are required on a finish, optional on an event, and empty on a start. Everything else, including the plain-text value, is unchanged. -## Base public model +| Field | Meaning | +|---|---| +| diagnostic_version_id | `AZURE_INIT_V1`; identifies the wire schema and selects its parser before any later field is interpreted | +| agent | Producer identifier, such as `azure-init-0.1.1` | +| boot_epoch | Unix seconds of the boot that produced the record | +| vm_id | VM identity | +| kind | `start` or `finish` for a span, `event` for a point observation | +| name | Subject, such as `provision:run` or `dmesg` | +| event_id | Shared by a span's start and finish, and by every chunk of one value | +| timestamp | RFC 3339 (ISO 8601), UTC with a `Z` suffix, millisecond precision, e.g. `2026-08-31T12:34:56.789Z` | +| encoding | How the value is encoded: `none`, `b64`, or `gz+b64` | +| result | `success` or `fail` on a finish, optionally on an event; empty otherwise | +| duration | Elapsed milliseconds on a finish, optionally on a timed event; empty otherwise | +| chunk_index | Chunk position, from 0 | + +#### Key size + +The whole key is one string, and the host silently truncates a guest-written key past 254 UTF-8 bytes; safe-mode `KvpPoolStore` rejects it first. The fixed fields spend most of that budget: `vm_id` and `event_id` at 36 bytes each plus the 24-byte `timestamp` are already 96 bytes, and `AZURE_INIT_V1` adds another 13 before the enums, numbers, and delimiters. A representative finish key runs about 190 bytes, leaving roughly 64 for the two free-form fields. + +Only `agent` and `name` are free-form; every other field is bounded by its format or its enum. The writer caps the two so the whole key cannot exceed 254 bytes: + +| Field | Cap | Bounded by | +|---|---|---| +| diagnostic_version_id | 13 B | fixed `AZURE_INIT_V1` token | +| agent | 32 B | free-form producer id | +| name | 48 B | free-form subject | +| vm_id, event_id | 36 B each | GUID / UUID | +| timestamp | 24 B | fixed format | +| boot_epoch, duration | 10 B each | digits | +| result, encoding, kind | ≤ 7 B each | enum token | +| chunk_index | 4 B | at most 1023 records | -```text -DiagnosticKind = Start | Finish | Diagnostic - -DiagnosticKey { - agent, - boot_epoch, - vm_id?, - kind, - name, - event_id, - timestamp?, - chunk_index? -} +With those caps the worst-case key is 243 bytes, leaving 11 bytes inside the limit. cloud-init reads are never capped; the bridge takes names as they are. -DiagnosticEntry { - key: DiagnosticKey, - value -} -``` +### Kinds -`DiagnosticKey` describes the record. `DiagnosticEntry` combines that description with the unchanged physical value. +`kind` marks where a record sits in an operation's timeline, and only that: `start` opens a timed operation, `finish` closes one, and `event` is a one-off that opens and closes nothing. Those three cover every position. A would-be fourth kind is really something the key already records: how the bytes are packed (`encoding`), what the record is about (`name`), or how it turned out (`result`). No kind carries a structured payload; the value is always a plain message or an artifact read per `encoding`. -| Field | Required | Source | Meaning and concrete use | -|---|---|---|---| -| `agent` | Yes | Key | Producer of the record; distinguishes `CLOUD_INIT` from Azure-init | -| `boot_epoch` | Yes | Key | Boot that produced the record; separates current-boot from previous-boot data | -| `vm_id` | No | Key | VM identity; supports host-side collection while remaining compatible with older cloud-init keys that omit it | -| `kind` | Yes | Key `type` token | Common lifecycle meaning; tells a reader whether `provision:run` began, ended, or emitted a point observation | -| `name` | Yes | Key | Actual producer-defined subject; lets a caller filter for `dmesg` or `provision:run` without parsing the value | -| `event_id` | Yes | Key | Identity shared by physical chunks from one emission; associates those chunks without inspecting their values | -| `timestamp` | No | Azure-init key or cloud-init `ts` | Time the occurrence was reported; remains absent when cloud-init metadata is unreadable | -| `chunk_index` | No | Numeric key suffix | Position of this physical fragment; makes chunks `16` through `19` useful even when earlier chunks are gone | -| `value` | Yes | Physical KVP value | Original fragment exactly as stored; preserves incomplete or invalid JSON for inspection | +For a concrete case, cloud-init's `type` field carries `compressed` and `system-info` tokens beside the same three. Neither is a timeline position: `compressed` is how the bytes are packed, an `encoding` here, and `system-info` is a subject, a `name`. How cloud-init's tokens map onto these three is in Compatibility. -The optional fields are optional in the common model because the supported source formats do not always provide them. Their absence never causes an otherwise valid physical entry to be discarded. +#### start -## Key normalization +A `start` opens a span, an operation that takes measurable time such as `provision:run`. Its `timestamp` marks when the operation began, and its value is a short opening message. -Azure-init keys: +It shares one `event_id` with its `finish`; that shared id is what ties the pair, so a `start` whose `finish` never arrives stands out as an operation that began but never ended, exactly the signal an operator wants after a hang or crash. ```text -||||||[|] +AZURE_INIT_V1|azure-init-0.1.1|1700000000|vm-abc|start|provision:run|9c1d2e3f-...|2026-08-31T12:34:56.789Z|none|||0 value: starting ``` -Current cloud-init keys: +#### finish + +A `finish` closes the span it shares an `event_id` with. Its `timestamp` is later than the start's, and it reports the operation's outcome directly in the key: `result` is `success` or `fail`, and `duration` is the elapsed milliseconds. The writer holds the start instant, so it stamps `duration` at emit time rather than making a reader pair the two records to recover it, and a truncated pool that kept only the finish still carries both the verdict and the elapsed time. The value stays a human message such as `provisioning succeeded` or `provisioning failed: `. cloud-init carries the same fields in its value JSON, which the bridge maps across (see Compatibility). ```text -CLOUD_INIT|||||[|] +AZURE_INIT_V1|azure-init-0.1.1|1700000000|vm-abc|finish|provision:run|9c1d2e3f-...|2026-08-31T12:34:57.101Z|none|success|312|0 value: provisioning succeeded ``` -Older cloud-init keys omit `vm_id`: +#### event + +An `event` is a point observation with no open or close: a single fact captured at one moment, such as an IMDS result or a `dmesg` snapshot. It has its own `event_id`, with no start or finish to pair. Most diagnostics are events. + +Its value is the observed payload, read per `encoding`: a short text as `none`, or a large artifact the caller marks for compression as `gz+b64`, split across chunks that share the `event_id` and differ only by `chunk_index`. An event may also set `result` (`success`/`fail`) when the observation is itself a pass or failure, and `duration` when it measured how long something took with no start and finish to bracket it; it leaves either empty when it does not apply. ```text -CLOUD_INIT||||[|] +# plain text, one record +AZURE_INIT_V1|azure-init-0.1.1|1700000000|vm-abc|event|imds|8f3e...|2026-07-27T21:33:24.300Z|none|||0 value: Retrieved 1 key from IMDS + +# an event that is itself a failure sets result +AZURE_INIT_V1|azure-init-0.1.1|1700000000|vm-abc|event|imds|7b2c...|2026-07-27T21:33:24.400Z|none|fail||0 value: IMDS unreachable + +# a self-contained timing sets duration but no result +AZURE_INIT_V1|azure-init-0.1.1|1700000000|vm-abc|event|imds:probe|5d6e...|2026-07-27T21:33:24.500Z|none||52|0 value: probed IMDS in 52ms + +# compressed artifact, split across records, indices 0..N +AZURE_INIT_V1|azure-init-0.1.1|1700000000|vm-abc|event|dmesg|9a1b...||gz+b64|||0 value: +AZURE_INIT_V1|azure-init-0.1.1|1700000000|vm-abc|event|dmesg|9a1b...||gz+b64|||1 value: ``` -All three layouts map into the same `DiagnosticKey`. +### Encodings -Azure-init provides its timestamp in the key. Cloud-init provides `ts` in the value, so `DiagnosticsPool` reads only enough metadata to find it. If `ts` is missing or unreadable, `timestamp` is absent and the entry is still returned. +`encoding` names how the value bytes are formed, chosen by the caller when it emits a diagnostic rather than guessed from size, so the choice is deterministic. It lives in the key so a reader knows it without parsing the value, and the value stays a single opaque payload rather than a wrapper. base64 is required for any binary, because the store holds values as UTF-8 and trims trailing nulls, so raw bytes would not survive a round trip. Because the encoding is a key token, a new one can be added later without a format change; cloud-init instead declares its encoding inside the value (see Compatibility). -## What `type` and `name` mean +Whatever the encoding, a value over the 1022-byte limit is split across chunks that share the value's `event_id` and `kind` and are ordered by `chunk_index`; decoding joins them in order before anything else. -The source key contains a `type` token and a `name`. +#### none -- `name` answers: what is this record about? -- `type` answers: what did the source report about that name? +Plain UTF-8 text, the default and the common case. The value is the message as stored, so decoding is a no-op; a split value is just the text slices joined in index order. Span messages and short observations use `none`. -The base proposal retains only the common meaning needed by readers: `kind`. It does not retain the exact source type as another public field. +#### b64 -Under this proposal, `type` has a narrow key-level role: +Standard base64 of the raw payload bytes, for binary that will not shrink usefully. Decoding joins the chunks and base64-decodes them back to the original bytes. The key stays text and the binary survives the UTF-8 store. -- It does not select the Azure-init or cloud-init key layout; the key shape and source prefix do that. -- It does not define, parse, or validate the client-owned value during the default read. -- It derives `kind`, so lifecycle can be understood without parsing the value. +#### gz+b64 +base64 of gzip of the raw payload bytes, for a large compressible artifact such as `dmesg`. Writing gzips the payload, base64-encodes the result, then chunks it; decoding joins the chunks, base64-decodes, and gunzips. It is one gzip stream spread across the chunks, so any missing chunk makes the whole artifact `Undecodable` (see Reads and writes), the cost it trades for far fewer records. -| Source `type` | Common `kind` | Meaning | -|---|---|---| -| `start` | `Start` | The operation identified by `name` began | -| `finish` | `Finish` | The operation identified by `name` ended | -| Any other type, including `compressed` | `Diagnostic` | A point-in-time observation about `name` | +### Reads and writes -`type` does not rename or parse `name`. For example: +A pool holds many diagnostics. Each is a start, finish, or event, and each is stored as one or more chunks (`chunk_index` 0 to N): ```text -type=start name=provision:run -> provisioning began -type=finish name=provision:run -> provisioning ended -type=system-info name=system information -> system information was reported -type=compressed name=dmesg -> dmesg data was reported +Diagnostics pool +├─ start provision:run +│ └─ chunk 0 +├─ event imds +│ └─ chunk 0 +├─ event dmesg (gz+b64, 3 chunks) +│ ├─ chunk 0 +│ ├─ chunk 1 +│ └─ chunk 2 +└─ finish provision:run + └─ chunk 0 ``` -The common results are: +`DiagnosticReader::entries()` reads the pool once and interprets every record, returning one `Entry` each: it combines and decodes a diagnostic into a `Diagnostic`, parses the `PROVISIONING_REPORT` into a `ProvisioningReport`, and leaves anything else as `Raw` (its key and value). A recognized record that will not parse — a broken diagnostic group, an unsupported azure-init diagnostics version, or a malformed report — also falls back to `Raw`, carrying the `DecodeError`, so nothing is dropped. A caller that wants the untouched records reads `KvpPoolStore` directly. + +Writing is the inverse: `DiagnosticWriter` stamps `AZURE_INIT_V1`, frames a diagnostic into records, and appends them to the `KvpPoolStore`. + +Reading the pool like a log is the `Diagnostic` entries in timestamp order. The host may reorder the pool, so a reader sorts on the timestamp each carries, and each becomes one line: + +```text +2026-08-31T12:34:56.789Z start provision:run +2026-08-31T12:34:57.020Z event imds ok +2026-08-31T12:34:57.101Z finish provision:run success 312ms +``` + +Splitting the type by kind is what keeps each line honest: a `finish` always carries its `result` and `duration`, a `start` carries neither, and an `event` carries either when the caller measured it, so the renderer never second-guesses an optional field. An operation that began but never finished is a `Start` with no matching `Finish`, still printed in place, which is the signal an operator wants after a crash: ```text -Start name=provision:run -Finish name=provision:run -Diagnostic name=system information -Diagnostic name=dmesg +2026-08-31T12:35:10.000Z start provision:run +(no finish) ``` -Existing `event`, `diagnostic`, `system-info`, `boot-telemetry`, and unknown types all become `Diagnostic`. Their actual `name` remains unchanged. Existing Azure-init `event` records remain readable; new Azure-init point records use `diagnostic`. +A finished operation already carries its `duration` on its finish line, so there is nothing to roll up. Pairing a start to its finish is a plain group-by on `event_id` when a reader wants both ends at once, and a start with no matching finish is an operation that began but never ended. -## Default read contract +Write: -`DiagnosticsPool::entries()` returns one `DiagnosticEntry` for each recognized physical record, in pool order. +```mermaid +flowchart TD + emit["DiagnosticWriter::emit_*
kind, name, payload"] --> encsel{"encoding
(caller's choice)"} + encsel -->|"gz+b64"| gz["gzip then base64"] + encsel -->|"b64"| b64["base64"] + encsel -->|"none"| plain["text as-is"] + gz --> frame["frame on UTF-8 boundaries
1022-byte value cap"] + b64 --> frame + plain --> frame + frame --> keys["stamp AZURE_INIT_V1
and format chunk keys"] + keys --> append["append all chunks under one store lock"] + append --> store[("KvpPoolStore
flat key=value pool")] +``` -It does: +Read (`DiagnosticReader::entries()`): + +```mermaid +flowchart TD + store[("KvpPoolStore
flat key=value pool")] --> reader["DiagnosticReader::entries()"] + reader --> cls{"first key field"} + cls -->|"AZURE_INIT_V1"| dec["group chunks,
decode by encoding"] + cls -->|"unsupported AZURE_INIT_V*"| rawver["Raw {key, value,
UnsupportedVersion}"] + cls -->|"CLOUD_INIT"| bridge["cloud-init bridge"] + bridge --> dec + cls -->|"PROVISIONING_REPORT"| rep["parse report"] + cls -->|"neither"| raw["Raw {key, value}"] + dec -->|"ok"| diag["Entry::Diagnostic"] + dec -->|"incomplete / duplicate / undecodable"| rawerr["Raw {key, value, error}"] + rep -->|"ok"| repe["Entry::Report"] + rep -->|"malformed"| rawerr +``` -- Parse key metadata into `DiagnosticKey`. -- Best-effort extract cloud-init's `ts`. -- Preserve the physical value unchanged. -- Return duplicate, partial, and out-of-sequence chunks independently. +The `diagnostic_version_id` is part of every azure-init group key, so chunks from different schemas can never combine. The rest of the group key includes `kind` because a span's start and finish share an `event_id`. There is no decode-time size limit either: the producer is trusted and the pool bounds the input, so an artifact either fits when it is written or is never written. -It does not: +## Crate design -- Group or order chunks. -- Require a complete chunk sequence. -- Validate cloud-init `msg_i`. -- Extract or unescape `msg`. -- Extract `result` or `duration`. -- Decode or decompress content. +The crate exposes two interfaces over `KvpPoolStore`; neither holds files or locks, and both delegate all IO to the store. A provisioning client writes through `DiagnosticWriter`, while a diagnostic consumer reads through `DiagnosticReader`; a caller that wants untyped records uses the store directly. The writer produces azure-init records only. The reader understands supported azure-init versions and reads cloud-init through the bridge. The format and behavior are in Proposed design; the types, API, and CLI are here. -Example: if only `dmesg` chunks 16 through 19 remain, the read returns four entries: +Their initialization is deliberately asymmetric. `DiagnosticReader` needs only a store because every source, identity, boot, and format decision comes from the records it reads; constructing one performs no IO. `DiagnosticWriter` needs the store plus the local `agent` and `vm_id`, validates that stable producer identity once, and obtains the boot epoch once for every record it will emit. It always writes the crate's current `AZURE_INIT_V1` format; callers cannot select a version or ask it to write cloud-init records. A process that needs both interfaces constructs them from clones of the same `KvpPoolStore`. -```text -Diagnostic name=dmesg chunk_index=16 value= -Diagnostic name=dmesg chunk_index=17 value= -Diagnostic name=dmesg chunk_index=18 value= -Diagnostic name=dmesg chunk_index=19 value= -``` +```rust +const AZURE_INIT_DIAGNOSTIC_VERSION_ID: &str = "AZURE_INIT_V1"; -Missing chunks 0 through 15 do not hide the records that are present. A cloud-init chunk also remains visible when a split inside `msg` makes that individual value invalid JSON. +enum Kind { Start, Finish, Event } -## Write contract +/// Plain text is `None`; a value names an encoding only when it has one. +/// `Other` keeps an unknown token so it decodes to `Undecodable`, never a panic. +enum Encoding { B64, GzB64, Other(String) } -`DiagnosticsPool` writes Azure-init records only. +enum Outcome { Success, Failure } -- A point diagnostic uses the `diagnostic` token and generates its event ID and timestamp. -- Explicit writes may use `Start`, `Finish`, or `Diagnostic` with caller-supplied event metadata. -- Values larger than one record are split at safe text boundaries and written with numeric chunk suffixes. -- All chunks from one write are appended together. +/// Why a recognized record could not be parsed, carried by the `Raw` it falls back to. +/// Implements `Error`, serialized as a snake_case reason. +enum DecodeError { + /// The key identifies the azure-init diagnostics family, but not a version this reader supports. + UnsupportedVersion, + /// Chunks are missing: not a contiguous run from 0. + IncompleteGroup, + /// A `chunk_index` appears more than once. + DuplicateChunk, + /// Unknown encoding, bad base64, or truncated gzip. + Undecodable, + /// A recognized value did not parse, such as a malformed `PROVISIONING_REPORT`. + Malformed, +} -Example: +/// The identity the three kinds share. Not the `diagnostic_version_id`, `kind`, `result`, `duration`, or `chunk_index`. +/// The reader consumes the schema ID while selecting a parser, then every supported source maps here. +struct DiagnosticKey { + agent: String, + boot_epoch: i64, + /// Older cloud-init keys omit it. + vm_id: Option, + name: String, + /// One per span (start and finish share it) or standalone event. + event_id: String, + /// RFC 3339, UTC, millisecond precision. + timestamp: DateTime, + encoding: Option, +} -```text -emit diagnostic: name=imds, value="Retrieved 1 key from IMDS" +/// Opens a span. +struct DiagnosticStart { key: DiagnosticKey, payload: Vec } +/// Closes a span; carries its verdict and elapsed milliseconds. +struct DiagnosticFinish { key: DiagnosticKey, payload: Vec, result: Outcome, duration_ms: u64 } +/// A point observation; may carry a verdict or a self-contained timing. +struct DiagnosticEvent { key: DiagnosticKey, payload: Vec, result: Option, duration_ms: Option } + +/// One decoded emission, typed by kind. +enum Diagnostic { + Start(DiagnosticStart), + Finish(DiagnosticFinish), + Event(DiagnosticEvent), +} -azure-init|1788371515|vm-123|diagnostic|imds|event-1|2026-09-02T17:52:25Z|0 -= Retrieved 1 key from IMDS -``` +/// A record left as key and value: unknown to the parser, or a recognized one +/// that failed to decode, in which case `error` says why. +struct RawKeyValue { + key: String, + value: String, + error: Option, +} -A long value produces multiple physical records. The default read returns each one as a separate `DiagnosticEntry`. +/// One interpreted item from `entries()`. A new known type is a new variant; +/// everything else stays `Raw`, so a reader never drops a record. +enum Entry { + Diagnostic(Diagnostic), + Report(ProvisioningReport), + Raw(RawKeyValue), +} -## CLI contract +/// Interprets a pool without any local producer identity. +struct DiagnosticReader { + store: KvpPoolStore, +} -`dump --parse-diagnostics` means classify physical diagnostic records. +/// Produces the current azure-init diagnostics format. +struct DiagnosticWriter { + store: KvpPoolStore, + agent: String, + vm_id: String, + boot_epoch: i64, +} -Each output item contains the fields from `DiagnosticKey` plus the unchanged physical value. +impl DiagnosticReader { + /// Constructing a reader performs no IO; the pool is read by `entries()`. + pub fn new(store: KvpPoolStore) -> Self; -- `--name` filters by the actual `name`. -- `--tail` counts physical records. -- JSON output represents the KVP value as a string, even when that string contains JSON. -- Decoded `message`, `result`, and `duration` are not part of this default output. + /// One `Entry` per item: each diagnostic is combined and decoded, the `PROVISIONING_REPORT` + /// is parsed into a `ProvisioningReport`, and everything else is `Raw`. A recognized record + /// that will not parse is `Raw` with its `DecodeError`, so nothing is dropped. + pub fn entries(&self) -> Result, KvpError>; +} -## Separate proposal: first-class payload decoding +impl DiagnosticWriter { + /// Fix the local producer identity and boot epoch used by every emitted record. + /// The writer always emits `AZURE_INIT_V1`. + pub fn new(store: KvpPoolStore, agent: impl Into, vm_id: impl Into) -> Result; -The base design above always returns physical entries. Some callers may also need one logical message or the decompressed artifact contained by that message. That transformation should be explicit and must not change key parsing or timestamp extraction. + /// Open a span. `event_id` links this start to the finish that closes it. + pub fn emit_start(&self, event_id: &str, name: &str, payload: &[u8], encoding: Option) -> Result<(), KvpError>; -There are three possible levels: + /// Close the span opened under `event_id`, recording its `result` and elapsed `duration_ms`. + pub fn emit_finish(&self, event_id: &str, name: &str, payload: &[u8], encoding: Option, result: Outcome, duration_ms: u64) -> Result<(), KvpError>; -| Level | Result | -|---|---| -| Raw | Every physical `DiagnosticEntry`, unchanged | -| Message | Group a complete chunk sequence and reconstruct the logical message | -| Content | Perform Message processing, then decode compressed content | + /// Record a standalone point observation; the writer assigns its `event_id`. + /// `result` and `duration_ms` are set only when measured. + pub fn emit_event(&self, name: &str, payload: &[u8], encoding: Option, result: Option, duration_ms: Option) -> Result<(), KvpError>; +} +``` -For Azure-init, Message concatenates value chunks. For cloud-init, Message joins the `msg` fragments and removes the outer telemetry JSON. +### CLI -For an ordinary diagnostic: +`dump` prints every record in pool order as raw `KEY=VALUE`. `--parse` is an add-on: it interprets each record, decoding diagnostics and parsing the `PROVISIONING_REPORT` into a `ProvisioningReport`, and leaves anything it does not recognize as its raw key and value. Nothing is dropped. ```text -Raw: {"name":"imds","msg":"Retrieved 1 key from IMDS",...} -Message: Retrieved 1 key from IMDS -Content: Retrieved 1 key from IMDS +dump -> every record, raw KEY=VALUE +dump --parse -> known records parsed (Diagnostic, ProvisioningReport); the rest stay Raw ``` -For a complete compressed `dmesg` event: +- `--parse` calls `DiagnosticReader::entries()`, one `Entry` per item. A recognized record that will not parse — a broken diagnostic, an unsupported azure-init diagnostics version, or a malformed report — stays `Raw` with its `DecodeError`, so its key, value, and the reason are all shown. +- `--name` filters the parsed diagnostics by name; other entries are unaffected. + +Examples: ```text -Raw: separate physical cloud-init chunks -Message: {"encoding":"gz+b64","data":"..."} -Content: decompressed dmesg bytes or text +# dump: every record raw, including the report and a dmesg chunk whose group was truncated +$ dump +AZURE_INIT_V1|azure-init-0.1.1|1700000000|vm-abc|finish|provision:run|9c1d...|2026-08-31T12:34:57.101Z|none|success|312|0=provisioning succeeded +AZURE_INIT_V1|azure-init-0.1.1|1700000000|vm-abc|event|dmesg|d4e5...|2026-07-27T21:33:25.00Z|gz+b64|||17= +PROVISIONING_REPORT=result=success|agent=azure-init-0.1.1|pps_type=None|vm_id=vm-abc|timestamp=2026-08-31T12:34:57.500Z + +# --parse: the diagnostic decodes, the report parses, the unassemblable dmesg chunk stays raw with its reason +$ dump --parse +{"type":"diagnostic","kind":"finish","name":"provision:run","event_id":"9c1d...","timestamp":"2026-08-31T12:34:57.101Z","result":"success","duration":312,"payload":"provisioning succeeded"} +{"type":"raw","key":"AZURE_INIT_V1|azure-init-0.1.1|1700000000|vm-abc|event|dmesg|d4e5...|gz+b64|||17","value":"","error":"incomplete_group"} +{"type":"PROVISIONING_REPORT","result":"success","agent":"azure-init-0.1.1","vm_id":"vm-abc","timestamp":"2026-08-31T12:34:57.500Z","pps_type":"None"} ``` -The base `DiagnosticKey` does not identify encoded content. If Content decoding is approved, one of these signals must also be selected: +## Adoption -| Option | Design | Effect | -|---|---|---| -| Key-derived flag | Add `compressed` to `DiagnosticKey`, derived from `type=compressed` | Identifies partial chunks before reconstruction, but gives one source token special public meaning | -| Complete source type | Add `source_type` to `DiagnosticKey` | Preserves `compressed` and every other producer classification, but expands the common model | -| Reconstructed envelope | Keep the base model unchanged and inspect the completed `{encoding,data}` message | Keeps encoding in the payload, but cannot identify incomplete chunks as encoded | - -In all three options, a valid envelope determines the codec. The signal only determines whether Content decoding should be attempted. Content then base64-decodes the data and applies the declared decompression. +azure-init adopts `AZURE_INIT_V1` by cutting over to it when it switches to the kvp crate; this is the first supported azure-init diagnostics format. The unversioned azure-init shape under Records today is pre-adoption rather than a compatibility contract: `DiagnosticReader` leaves one of those records as `Raw` instead of guessing which schema it follows. There are no prior records to migrate. -If a requested transformation cannot be completed, the proposed behavior is to return the original Raw entries rather than discard them. +Updating cloud-init to emit this format is a non-goal for now and may be revisited; until then cloud-init is read-only through the compatibility bridge. -## Feedback requested +## Compatibility -Sign-off is needed on both the base physical-entry design and the optional payload-decoding mechanism. +cloud-init writes diagnostics into the same pool in its own format. `DiagnosticReader` dispatches its records to a read-only bridge that maps them onto the same model; `DiagnosticWriter` never writes cloud-init's format, and the `AZURE_INIT_V1` parser never parses a cloud-init value. -“First-class” means callers select one named payload level. They do not manually combine separate grouping, message-extraction, base64, and decompression flags. The proposed public shape is: +cloud-init keys put the type and name before the identifiers, and current keys include a `vm_id` that older keys omit: ```text -Default API read: entries() -> Raw DiagnosticEntry records -Optional API read: payload view = Message | Content - -CLI: - dump --parse-diagnostics -> Raw - dump --parse-diagnostics --view message -> Message - dump --parse-diagnostics --view content -> Content +current CLOUD_INIT|||||[|] +older CLOUD_INIT||||[|] ``` -Message and Content automatically perform their required earlier stages. This public control is part of the proposal and also requires approval. +The value is JSON telemetry carrying `name`, `type`, `ts`, and `msg`, with `result` and `duration` on a span finish, a per-chunk `msg_i` on each chunk of a split value, and, for a compressed artifact, a `{encoding, data}` envelope inside `msg`. -1. Base design: Does one physical `DiagnosticEntry` per record, with the model and normalization above, match the intended diagnostics contract? Does keeping spanned-value handling separate but internal, with diagnostics layered on top of it, match the intended architecture? +The bridge maps fields onto the model: -2. Purpose of `type`: Should `type` remain a key field whose base public effect is deriving `kind`, while the value schema remains client-owned? Should lifecycle continue to use `type=start` and `type=finish` in the key, or should Start and Finish use a different key structure? - -3. Decoding levels: Should the first-class mechanism expose Raw, Message, and Content, or only Raw and fully decoded Content? - -4. Public control: Should `entries()` remain the unchanged Raw read while one separate payload-view operation and matching `--view` option select Message or Content? If only Raw and Content are needed, should the option be named `--decode` instead of `--view content`? - -5. Decoded content type: Should Content return bytes, or require UTF-8 text? - -6. Failure behavior: When chunks are incomplete or decoding fails, should the operation return the original Raw entries or an error? - -7. Encoded-content signal: If Content decoding is approved, should it use a key-derived `compressed` flag, retain the complete `source_type`, or inspect only the reconstructed `{encoding,data}` envelope? Is this proposal limited to decoding existing encoded content, or should it also define how Azure-init compresses and writes new encoded content? \ No newline at end of file +| Model field | cloud-init source | +|---|---| +| agent | the literal `CLOUD_INIT` | +| boot_epoch | `incarnation` | +| kind | `type` (`start`, `finish`, else `event`) | +| name | `name` | +| vm_id | present only on current keys | +| event_id | trailing key identifier | +| timestamp | value `ts`, read by the bridge when it maps the record | +| result | value field on a finish, mapped to the model's `result` | +| duration | value field on a finish (seconds; the bridge converts to milliseconds) | +| encoding | the value `{encoding, data}` envelope, not the key | + +cloud-init has no `AZURE_INIT_V1` field and the bridge does not invent one. Its `CLOUD_INIT` prefix selects the bridge, which performs its source-specific parsing before constructing the same version-independent `DiagnosticKey` as the azure-init parser. + +Reassembly is cloud-init specific: chunks are JSON objects, so the bridge validates each `msg_i` against the chunk index, concatenates the still-escaped `msg` slices, and unescapes the joined string once. If the result is an `{encoding, data}` envelope, the bridge decodes it with the same encodings as the core; otherwise the message is the payload. cloud-init declares its encoding in the value, which is why it is read there and not from the key. \ No newline at end of file From 6ec08be72327dff27be93898f25db4a2345021e0 Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Fri, 4 Sep 2026 09:38:52 -0700 Subject: [PATCH 18/32] docs(kvp): improve error handling paths in diagrams --- libazureinit-kvp/diagnostics-proposal.md | 52 ++++++++++++++++++------ 1 file changed, 40 insertions(+), 12 deletions(-) diff --git a/libazureinit-kvp/diagnostics-proposal.md b/libazureinit-kvp/diagnostics-proposal.md index d154c81e..7d3d93d4 100644 --- a/libazureinit-kvp/diagnostics-proposal.md +++ b/libazureinit-kvp/diagnostics-proposal.md @@ -199,7 +199,7 @@ Standard base64 of the raw payload bytes, for binary that will not shrink useful #### gz+b64 -base64 of gzip of the raw payload bytes, for a large compressible artifact such as `dmesg`. Writing gzips the payload, base64-encodes the result, then chunks it; decoding joins the chunks, base64-decodes, and gunzips. It is one gzip stream spread across the chunks, so any missing chunk makes the whole artifact `Undecodable` (see Reads and writes), the cost it trades for far fewer records. +base64 of gzip of the raw payload bytes, for a large compressible artifact such as `dmesg`. Writing gzips the payload, base64-encodes the result, then chunks it; decoding joins the chunks, base64-decodes, and gunzips. It is one gzip stream spread across the chunks, so a missing chunk makes the whole artifact unavailable: a visible index gap is `IncompleteGroup`, while a contiguous truncation is detected during decoding as `Undecodable`. That is the cost it trades for far fewer records. ### Reads and writes @@ -244,7 +244,13 @@ Write: ```mermaid flowchart TD - emit["DiagnosticWriter::emit_*
kind, name, payload"] --> encsel{"encoding
(caller's choice)"} + client["Provisioning client"] --> new["DiagnosticWriter::new
store, agent, vm_id"] + new --> init{"identity valid and
boot epoch readable?"} + init -->|"no"| initerr["Err(KvpError)
writer not constructed"] + init -->|"yes"| emit["DiagnosticWriter::emit_*
kind, name, payload"] + emit --> valid{"fields, kind invariants,
and encoding valid?"} + valid -->|"no"| inputerr["Err(KvpError)
nothing written"] + valid -->|"yes"| encsel{"encoding
(caller's choice)"} encsel -->|"gz+b64"| gz["gzip then base64"] encsel -->|"b64"| b64["base64"] encsel -->|"none"| plain["text as-is"] @@ -252,33 +258,55 @@ flowchart TD b64 --> frame plain --> frame frame --> keys["stamp AZURE_INIT_V1
and format chunk keys"] - keys --> append["append all chunks under one store lock"] - append --> store[("KvpPoolStore
flat key=value pool")] + keys --> limits{"key and chunk-count
limits satisfied?"} + limits -->|"no"| inputerr + limits -->|"yes"| append["KvpPoolStore::append_multiple
all chunks under one lock"] + append -->|"ok"| store[("KvpPoolStore
flat key=value pool")] + store --> ok["Ok(())"] + append -->|"lock / write / flush error"| writeerr["Err(KvpError)
batch may be partial"] ``` Read (`DiagnosticReader::entries()`): ```mermaid flowchart TD - store[("KvpPoolStore
flat key=value pool")] --> reader["DiagnosticReader::entries()"] - reader --> cls{"first key field"} - cls -->|"AZURE_INIT_V1"| dec["group chunks,
decode by encoding"] - cls -->|"unsupported AZURE_INIT_V*"| rawver["Raw {key, value,
UnsupportedVersion}"] + client["Diagnostic consumer / CLI"] --> new["DiagnosticReader::new(store)
no IO, cannot fail"] + new --> entries["DiagnosticReader::entries()"] + entries --> dump["KvpPoolStore::dump()"] + dump -->|"lock / read error"| readerr["Err(KvpError)
no entries returned"] + dump -->|"ok"| cls{"first key field"} + cls -->|"AZURE_INIT_V1"| dec["source parser
parse, group, decode"] + cls -->|"unsupported AZURE_INIT_V*"| rawver["Entry::Raw
UnsupportedVersion"] cls -->|"CLOUD_INIT"| bridge["cloud-init bridge"] bridge --> dec cls -->|"PROVISIONING_REPORT"| rep["parse report"] - cls -->|"neither"| raw["Raw {key, value}"] + cls -->|"neither"| raw["Entry::Raw
error: None"] dec -->|"ok"| diag["Entry::Diagnostic"] - dec -->|"incomplete / duplicate / undecodable"| rawerr["Raw {key, value, error}"] + dec -->|"bad key or source value"| malformed["Entry::Raw
Malformed"] + dec -->|"missing chunk"| incomplete["Entry::Raw
IncompleteGroup"] + dec -->|"duplicate index"| duplicate["Entry::Raw
DuplicateChunk"] + dec -->|"unknown encoding / bad data"| undecodable["Entry::Raw
Undecodable"] rep -->|"ok"| repe["Entry::Report"] - rep -->|"malformed"| rawerr + rep -->|"malformed"| malformed + rawver --> out["Ok(Vec<Entry>)"] + raw --> out + diag --> out + malformed --> out + incomplete --> out + duplicate --> out + undecodable --> out + repe --> out ``` +`KvpError` and `DecodeError` mark different boundaries. A `KvpError` means the requested construction, read, or write could not complete and is returned by the method. A `DecodeError` means the pool read succeeded but stored data could not be interpreted; `entries()` still succeeds and preserves that data as `Entry::Raw` with the reason. + +The writer prepares and validates the complete batch before calling the store, so an identity, field, encoding, or size error writes nothing. Once `append_multiple` begins, a lock, write, or flush error returns `KvpError` but may leave part of the batch in the pool. A reader reports a visible index gap as `IncompleteGroup` and invalid encoded content as `Undecodable`. A contiguous prefix of a `none` payload has neither condition and cannot be distinguished from a complete value because this format carries no total chunk count. + The `diagnostic_version_id` is part of every azure-init group key, so chunks from different schemas can never combine. The rest of the group key includes `kind` because a span's start and finish share an `event_id`. There is no decode-time size limit either: the producer is trusted and the pool bounds the input, so an artifact either fits when it is written or is never written. ## Crate design -The crate exposes two interfaces over `KvpPoolStore`; neither holds files or locks, and both delegate all IO to the store. A provisioning client writes through `DiagnosticWriter`, while a diagnostic consumer reads through `DiagnosticReader`; a caller that wants untyped records uses the store directly. The writer produces azure-init records only. The reader understands supported azure-init versions and reads cloud-init through the bridge. The format and behavior are in Proposed design; the types, API, and CLI are here. +The crate exposes two interfaces over `KvpPoolStore`; neither holds files or locks, and both delegate all IO to the store. `DiagnosticWriter` is the provisioning clients' producer interface: clients provide diagnostic meaning and payload, but do not construct keys, frame chunks, or append diagnostic records directly. A diagnostic consumer reads through `DiagnosticReader`; a caller that wants untyped records uses the store directly. The writer produces azure-init records only. The reader understands supported azure-init versions and reads cloud-init through the bridge. The format and behavior are in Proposed design; the types, API, and CLI are here. Their initialization is deliberately asymmetric. `DiagnosticReader` needs only a store because every source, identity, boot, and format decision comes from the records it reads; constructing one performs no IO. `DiagnosticWriter` needs the store plus the local `agent` and `vm_id`, validates that stable producer identity once, and obtains the boot epoch once for every record it will emit. It always writes the crate's current `AZURE_INIT_V1` format; callers cannot select a version or ask it to write cloud-init records. A process that needs both interfaces constructs them from clones of the same `KvpPoolStore`. From cd4dc89a01df5ddbcca8c780db4eb97be70351a1 Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Fri, 4 Sep 2026 10:14:54 -0700 Subject: [PATCH 19/32] docs(kvp): finalize diagnostics schema and CLI contract --- libazureinit-kvp/diagnostics-proposal.md | 147 +++++++++++++---------- 1 file changed, 86 insertions(+), 61 deletions(-) diff --git a/libazureinit-kvp/diagnostics-proposal.md b/libazureinit-kvp/diagnostics-proposal.md index 7d3d93d4..a846dd59 100644 --- a/libazureinit-kvp/diagnostics-proposal.md +++ b/libazureinit-kvp/diagnostics-proposal.md @@ -89,24 +89,24 @@ The two disagree on field order, on where the timestamp and encoding live, and o ## Proposed design -This proposal defines one versioned record format for azure-init and separate reader and writer interfaces over `KvpPoolStore`. The reader interprets the pool; the writer produces only the azure-init format. Untyped access remains on `KvpPoolStore` itself. +This proposal defines one versioned diagnostic record format, emitted here by azure-init, and separate reader and writer interfaces over `KvpPoolStore`. The reader interprets the pool; the writer produces only this format. Untyped access remains on `KvpPoolStore` itself. ### Diagnostics format The proposed key is the current azure-init key with a few changes, all keeping metadata in the key. It is pipe-delimited with `|` reserved, begins with a diagnostic schema-version identifier, and always ends with the chunk index, so a single-record value still ends in `|0`: ```text -AZURE_INIT_V1||||||||||| +DIAG_V1||||||||||| ``` -- `AZURE_INIT_V1` is the `diagnostic_version_id`: `AZURE_INIT` identifies the diagnostics family and `V1` identifies version 1 of its wire schema. That schema covers the field layout, required and optional fields, token meanings, units, encoding rules, and chunk framing. It is one self-identifying token rather than a bare number because the pool also contains unrelated keys: a reader can recognize an unsupported future `AZURE_INIT_V*` record without mistaking an arbitrary numeric key for a diagnostic. It is separate from `agent`, which remains the producer identifier and whose version identifies the producer binary. A schema change requires a new `diagnostic_version_id`; an agent release by itself does not. +- `DIAG_V1` is the `diagnostic_version_id`: `DIAG` identifies the general diagnostics family and `V1` identifies version 1 of its wire schema. It does not identify the producer; `agent` continues to do that. The schema covers the field layout, required and optional fields, token meanings, units, encoding rules, and chunk framing. It is one self-identifying token rather than a bare number because the pool also contains unrelated keys: a reader can recognize an unsupported future `DIAG_V*` record without mistaking an arbitrary numeric key for a diagnostic. A schema change requires a new `diagnostic_version_id`; an agent release by itself does not. - `type` becomes `kind`, narrowed to the three timeline positions (`start`/`finish`/`event`); any other category is carried by `name`, `encoding`, or `result`, not a token. - `encoding` names the payload encoding in the key so a large artifact can be compressed. - `result` and `duration` are the finish's verdict fields, carried in the key: `result` is a `success`/`fail` token and `duration` is the elapsed milliseconds. Both are required on a finish, optional on an event, and empty on a start. Everything else, including the plain-text value, is unchanged. | Field | Meaning | |---|---| -| diagnostic_version_id | `AZURE_INIT_V1`; identifies the wire schema and selects its parser before any later field is interpreted | +| diagnostic_version_id | `DIAG_V1`; identifies the wire schema and selects its parser before any later field is interpreted | | agent | Producer identifier, such as `azure-init-0.1.1` | | boot_epoch | Unix seconds of the boot that produced the record | | vm_id | VM identity | @@ -114,20 +114,20 @@ AZURE_INIT_V1|||||||| | name | Subject, such as `provision:run` or `dmesg` | | event_id | Shared by a span's start and finish, and by every chunk of one value | | timestamp | RFC 3339 (ISO 8601), UTC with a `Z` suffix, millisecond precision, e.g. `2026-08-31T12:34:56.789Z` | -| encoding | How the value is encoded: `none`, `b64`, or `gz+b64` | +| encoding | How the value is encoded: `none` or `gz+b64` | | result | `success` or `fail` on a finish, optionally on an event; empty otherwise | | duration | Elapsed milliseconds on a finish, optionally on a timed event; empty otherwise | | chunk_index | Chunk position, from 0 | #### Key size -The whole key is one string, and the host silently truncates a guest-written key past 254 UTF-8 bytes; safe-mode `KvpPoolStore` rejects it first. The fixed fields spend most of that budget: `vm_id` and `event_id` at 36 bytes each plus the 24-byte `timestamp` are already 96 bytes, and `AZURE_INIT_V1` adds another 13 before the enums, numbers, and delimiters. A representative finish key runs about 190 bytes, leaving roughly 64 for the two free-form fields. +The whole key is one string, and the host silently truncates a guest-written key past 254 UTF-8 bytes; safe-mode `KvpPoolStore` rejects it first. The fixed fields spend most of that budget: `vm_id` and `event_id` at 36 bytes each plus the 24-byte `timestamp` are already 96 bytes, and `DIAG_V1` adds another 7 before the enums, numbers, and delimiters. A representative finish key runs about 184 bytes, leaving roughly 70 for the two free-form fields. Only `agent` and `name` are free-form; every other field is bounded by its format or its enum. The writer caps the two so the whole key cannot exceed 254 bytes: | Field | Cap | Bounded by | |---|---|---| -| diagnostic_version_id | 13 B | fixed `AZURE_INIT_V1` token | +| diagnostic_version_id | 7 B | fixed `DIAG_V1` token | | agent | 32 B | free-form producer id | | name | 48 B | free-form subject | | vm_id, event_id | 36 B each | GUID / UUID | @@ -136,7 +136,7 @@ Only `agent` and `name` are free-form; every other field is bounded by its forma | result, encoding, kind | ≤ 7 B each | enum token | | chunk_index | 4 B | at most 1023 records | -With those caps the worst-case key is 243 bytes, leaving 11 bytes inside the limit. cloud-init reads are never capped; the bridge takes names as they are. +With those caps the worst-case key is 237 bytes, leaving 17 bytes inside the limit. cloud-init reads are never capped; the bridge takes names as they are. ### Kinds @@ -151,7 +151,7 @@ A `start` opens a span, an operation that takes measurable time such as `provisi It shares one `event_id` with its `finish`; that shared id is what ties the pair, so a `start` whose `finish` never arrives stands out as an operation that began but never ended, exactly the signal an operator wants after a hang or crash. ```text -AZURE_INIT_V1|azure-init-0.1.1|1700000000|vm-abc|start|provision:run|9c1d2e3f-...|2026-08-31T12:34:56.789Z|none|||0 value: starting +DIAG_V1|azure-init-0.1.1|1700000000|vm-abc|start|provision:run|9c1d2e3f-...|2026-08-31T12:34:56.789Z|none|||0 value: starting ``` #### finish @@ -159,7 +159,7 @@ AZURE_INIT_V1|azure-init-0.1.1|1700000000|vm-abc|start|provision:run|9c1d2e3f-.. A `finish` closes the span it shares an `event_id` with. Its `timestamp` is later than the start's, and it reports the operation's outcome directly in the key: `result` is `success` or `fail`, and `duration` is the elapsed milliseconds. The writer holds the start instant, so it stamps `duration` at emit time rather than making a reader pair the two records to recover it, and a truncated pool that kept only the finish still carries both the verdict and the elapsed time. The value stays a human message such as `provisioning succeeded` or `provisioning failed: `. cloud-init carries the same fields in its value JSON, which the bridge maps across (see Compatibility). ```text -AZURE_INIT_V1|azure-init-0.1.1|1700000000|vm-abc|finish|provision:run|9c1d2e3f-...|2026-08-31T12:34:57.101Z|none|success|312|0 value: provisioning succeeded +DIAG_V1|azure-init-0.1.1|1700000000|vm-abc|finish|provision:run|9c1d2e3f-...|2026-08-31T12:34:57.101Z|none|success|312|0 value: provisioning succeeded ``` #### event @@ -170,22 +170,37 @@ Its value is the observed payload, read per `encoding`: a short text as `none`, ```text # plain text, one record -AZURE_INIT_V1|azure-init-0.1.1|1700000000|vm-abc|event|imds|8f3e...|2026-07-27T21:33:24.300Z|none|||0 value: Retrieved 1 key from IMDS +DIAG_V1|azure-init-0.1.1|1700000000|vm-abc|event|imds|8f3e...|2026-07-27T21:33:24.300Z|none|||0 value: Retrieved 1 key from IMDS # an event that is itself a failure sets result -AZURE_INIT_V1|azure-init-0.1.1|1700000000|vm-abc|event|imds|7b2c...|2026-07-27T21:33:24.400Z|none|fail||0 value: IMDS unreachable +DIAG_V1|azure-init-0.1.1|1700000000|vm-abc|event|imds|7b2c...|2026-07-27T21:33:24.400Z|none|fail||0 value: IMDS unreachable # a self-contained timing sets duration but no result -AZURE_INIT_V1|azure-init-0.1.1|1700000000|vm-abc|event|imds:probe|5d6e...|2026-07-27T21:33:24.500Z|none||52|0 value: probed IMDS in 52ms +DIAG_V1|azure-init-0.1.1|1700000000|vm-abc|event|imds:probe|5d6e...|2026-07-27T21:33:24.500Z|none||52|0 value: probed IMDS in 52ms # compressed artifact, split across records, indices 0..N -AZURE_INIT_V1|azure-init-0.1.1|1700000000|vm-abc|event|dmesg|9a1b...||gz+b64|||0 value: -AZURE_INIT_V1|azure-init-0.1.1|1700000000|vm-abc|event|dmesg|9a1b...||gz+b64|||1 value: +DIAG_V1|azure-init-0.1.1|1700000000|vm-abc|event|dmesg|9a1b...||gz+b64|||0 value: +DIAG_V1|azure-init-0.1.1|1700000000|vm-abc|event|dmesg|9a1b...||gz+b64|||1 value: ``` +### Payloads + +The public API distinguishes text from arbitrary bytes instead of representing every decoded payload as `Vec`. `DiagnosticPayload::Text` carries a Rust `String`, which is already valid UTF-8; `DiagnosticPayload::Bytes` carries arbitrary bytes. Writer methods accept `impl Into`, with conversions from `&str`, `String`, `&[u8]`, and `Vec`, so callers can pass either form without separate method names. + +The caller still chooses the wire `encoding`; the input type does not guess whether compression is useful. The writer handles each combination as follows: + +| Input | `encoding=None` | `gz+b64` | +|---|---|---| +| Text | Store its UTF-8 bytes directly | Encode its UTF-8 bytes | +| Bytes | Validate UTF-8, then store directly; reject invalid UTF-8 | Encode the arbitrary bytes | + +On read, `none` must decode as UTF-8 and produces `DiagnosticPayload::Text`; invalid UTF-8 is `Undecodable`. `gz+b64` produces `DiagnosticPayload::Bytes` after decoding, even when those bytes happen to be valid UTF-8, because the wire schema does not declare their content type. + +In JSON output, `Text` is a JSON string. `Bytes` is `{ "type": "bytes", "encoding": "base64", "data": "..." }`, because JSON cannot carry arbitrary bytes; this presentation encoding does not change the diagnostic's wire `encoding`. + ### Encodings -`encoding` names how the value bytes are formed, chosen by the caller when it emits a diagnostic rather than guessed from size, so the choice is deterministic. It lives in the key so a reader knows it without parsing the value, and the value stays a single opaque payload rather than a wrapper. base64 is required for any binary, because the store holds values as UTF-8 and trims trailing nulls, so raw bytes would not survive a round trip. Because the encoding is a key token, a new one can be added later without a format change; cloud-init instead declares its encoding inside the value (see Compatibility). +`encoding` names how the payload is represented in the KVP value, chosen by the caller when it emits a diagnostic rather than guessed from size, so the choice is deterministic. It lives in the key so a reader knows it without parsing the value, and the value stays a single opaque payload rather than a wrapper. Arbitrary binary uses `gz+b64` because the store holds values as UTF-8 and trims trailing nulls, so raw bytes would not survive a round trip. `DIAG_V1` does not define standalone base64; a later schema can add another encoding without changing the key's overall shape. cloud-init instead declares its encoding inside the value (see Compatibility). Whatever the encoding, a value over the 1022-byte limit is split across chunks that share the value's `event_id` and `kind` and are ordered by `chunk_index`; decoding joins them in order before anything else. @@ -193,10 +208,6 @@ Whatever the encoding, a value over the 1022-byte limit is split across chunks t Plain UTF-8 text, the default and the common case. The value is the message as stored, so decoding is a no-op; a split value is just the text slices joined in index order. Span messages and short observations use `none`. -#### b64 - -Standard base64 of the raw payload bytes, for binary that will not shrink usefully. Decoding joins the chunks and base64-decodes them back to the original bytes. The key stays text and the binary survives the UTF-8 store. - #### gz+b64 base64 of gzip of the raw payload bytes, for a large compressible artifact such as `dmesg`. Writing gzips the payload, base64-encodes the result, then chunks it; decoding joins the chunks, base64-decodes, and gunzips. It is one gzip stream spread across the chunks, so a missing chunk makes the whole artifact unavailable: a visible index gap is `IncompleteGroup`, while a contiguous truncation is detected during decoding as `Undecodable`. That is the cost it trades for far fewer records. @@ -219,11 +230,11 @@ Diagnostics pool └─ chunk 0 ``` -`DiagnosticReader::entries()` reads the pool once and interprets every record, returning one `Entry` each: it combines and decodes a diagnostic into a `Diagnostic`, parses the `PROVISIONING_REPORT` into a `ProvisioningReport`, and leaves anything else as `Raw` (its key and value). A recognized record that will not parse — a broken diagnostic group, an unsupported azure-init diagnostics version, or a malformed report — also falls back to `Raw`, carrying the `DecodeError`, so nothing is dropped. A caller that wants the untouched records reads `KvpPoolStore` directly. +`DiagnosticReader::entries()` reads the pool once and interprets every record, returning one `Entry` each: it combines and decodes a diagnostic into a `Diagnostic`, parses the `PROVISIONING_REPORT` into a `ProvisioningReport`, and leaves anything else as `Raw` (its key and value). A recognized record that will not parse — a broken diagnostic group, an unsupported diagnostics version, or a malformed report — also falls back to `Raw`, carrying the `DecodeError`, so nothing is dropped. A caller that wants the untouched records reads `KvpPoolStore` directly. -Writing is the inverse: `DiagnosticWriter` stamps `AZURE_INIT_V1`, frames a diagnostic into records, and appends them to the `KvpPoolStore`. +Writing is the inverse: `DiagnosticWriter` stamps `DIAG_V1`, converts the typed payload according to `encoding`, frames it into records, and appends them to the `KvpPoolStore`. -Reading the pool like a log is the `Diagnostic` entries in timestamp order. The host may reorder the pool, so a reader sorts on the timestamp each carries, and each becomes one line: +Reading the pool like a log is the `Diagnostic` entries in timestamp order. The host may reorder the pool, so a reader sorts on the timestamp each carries. Text output renders each as one line: ```text 2026-08-31T12:34:56.789Z start provision:run @@ -247,17 +258,15 @@ flowchart TD client["Provisioning client"] --> new["DiagnosticWriter::new
store, agent, vm_id"] new --> init{"identity valid and
boot epoch readable?"} init -->|"no"| initerr["Err(KvpError)
writer not constructed"] - init -->|"yes"| emit["DiagnosticWriter::emit_*
kind, name, payload"] - emit --> valid{"fields, kind invariants,
and encoding valid?"} + init -->|"yes"| emit["DiagnosticWriter::emit_*
text or byte payload"] + emit --> valid{"fields, kind invariants,
payload, and encoding valid?"} valid -->|"no"| inputerr["Err(KvpError)
nothing written"] valid -->|"yes"| encsel{"encoding
(caller's choice)"} encsel -->|"gz+b64"| gz["gzip then base64"] - encsel -->|"b64"| b64["base64"] encsel -->|"none"| plain["text as-is"] gz --> frame["frame on UTF-8 boundaries
1022-byte value cap"] - b64 --> frame plain --> frame - frame --> keys["stamp AZURE_INIT_V1
and format chunk keys"] + frame --> keys["stamp DIAG_V1
and format chunk keys"] keys --> limits{"key and chunk-count
limits satisfied?"} limits -->|"no"| inputerr limits -->|"yes"| append["KvpPoolStore::append_multiple
all chunks under one lock"] @@ -275,8 +284,8 @@ flowchart TD entries --> dump["KvpPoolStore::dump()"] dump -->|"lock / read error"| readerr["Err(KvpError)
no entries returned"] dump -->|"ok"| cls{"first key field"} - cls -->|"AZURE_INIT_V1"| dec["source parser
parse, group, decode"] - cls -->|"unsupported AZURE_INIT_V*"| rawver["Entry::Raw
UnsupportedVersion"] + cls -->|"DIAG_V1"| dec["source parser
parse, group, decode"] + cls -->|"unsupported DIAG_V*"| rawver["Entry::Raw
UnsupportedVersion"] cls -->|"CLOUD_INIT"| bridge["cloud-init bridge"] bridge --> dec cls -->|"PROVISIONING_REPORT"| rep["parse report"] @@ -300,31 +309,39 @@ flowchart TD `KvpError` and `DecodeError` mark different boundaries. A `KvpError` means the requested construction, read, or write could not complete and is returned by the method. A `DecodeError` means the pool read succeeded but stored data could not be interpreted; `entries()` still succeeds and preserves that data as `Entry::Raw` with the reason. -The writer prepares and validates the complete batch before calling the store, so an identity, field, encoding, or size error writes nothing. Once `append_multiple` begins, a lock, write, or flush error returns `KvpError` but may leave part of the batch in the pool. A reader reports a visible index gap as `IncompleteGroup` and invalid encoded content as `Undecodable`. A contiguous prefix of a `none` payload has neither condition and cannot be distinguished from a complete value because this format carries no total chunk count. +The writer prepares and validates the complete batch before calling the store, so an identity, field, payload, encoding, or size error writes nothing. This includes byte input that is not valid UTF-8 when `encoding=None`. Once `append_multiple` begins, a lock, write, or flush error returns `KvpError` but may leave part of the batch in the pool. A reader reports a visible index gap as `IncompleteGroup` and invalid encoded content as `Undecodable`. A contiguous prefix of a `none` payload has neither condition and cannot be distinguished from a complete value because this format carries no total chunk count. -The `diagnostic_version_id` is part of every azure-init group key, so chunks from different schemas can never combine. The rest of the group key includes `kind` because a span's start and finish share an `event_id`. There is no decode-time size limit either: the producer is trusted and the pool bounds the input, so an artifact either fits when it is written or is never written. +The `diagnostic_version_id` is part of every diagnostic group key, so chunks from different schemas can never combine. The rest of the group key includes `kind` because a span's start and finish share an `event_id`. There is no decode-time size limit either: the producer is trusted and the pool bounds the input, so an artifact either fits when it is written or is never written. ## Crate design -The crate exposes two interfaces over `KvpPoolStore`; neither holds files or locks, and both delegate all IO to the store. `DiagnosticWriter` is the provisioning clients' producer interface: clients provide diagnostic meaning and payload, but do not construct keys, frame chunks, or append diagnostic records directly. A diagnostic consumer reads through `DiagnosticReader`; a caller that wants untyped records uses the store directly. The writer produces azure-init records only. The reader understands supported azure-init versions and reads cloud-init through the bridge. The format and behavior are in Proposed design; the types, API, and CLI are here. +The crate exposes two interfaces over `KvpPoolStore`; neither holds files or locks, and both delegate all IO to the store. `DiagnosticWriter` is the provisioning clients' producer interface: clients provide diagnostic meaning and payload, but do not construct keys, frame chunks, or append diagnostic records directly. A diagnostic consumer reads through `DiagnosticReader`; a caller that wants untyped records uses the store directly. The writer produces azure-init records only. The reader understands supported diagnostics versions and reads cloud-init through the bridge. The format and behavior are in Proposed design; the types, API, and CLI are here. -Their initialization is deliberately asymmetric. `DiagnosticReader` needs only a store because every source, identity, boot, and format decision comes from the records it reads; constructing one performs no IO. `DiagnosticWriter` needs the store plus the local `agent` and `vm_id`, validates that stable producer identity once, and obtains the boot epoch once for every record it will emit. It always writes the crate's current `AZURE_INIT_V1` format; callers cannot select a version or ask it to write cloud-init records. A process that needs both interfaces constructs them from clones of the same `KvpPoolStore`. +Their initialization is deliberately asymmetric. `DiagnosticReader` needs only a store because every source, identity, boot, and format decision comes from the records it reads; constructing one performs no IO. `DiagnosticWriter` needs the store plus the local `agent` and `vm_id`, validates that stable producer identity once, and obtains the boot epoch once for every record it will emit. It always writes the crate's current `DIAG_V1` format; callers cannot select a version or ask it to write cloud-init records. A process that needs both interfaces constructs them from clones of the same `KvpPoolStore`. ```rust -const AZURE_INIT_DIAGNOSTIC_VERSION_ID: &str = "AZURE_INIT_V1"; +const DIAGNOSTIC_VERSION_ID: &str = "DIAG_V1"; enum Kind { Start, Finish, Event } -/// Plain text is `None`; a value names an encoding only when it has one. +/// Plain text is `None`; a compressed value is `GzB64`. /// `Other` keeps an unknown token so it decodes to `Undecodable`, never a panic. -enum Encoding { B64, GzB64, Other(String) } +enum Encoding { GzB64, Other(String) } enum Outcome { Success, Failure } +/// The decoded payload. Rust strings guarantee UTF-8; bytes make no text claim. +/// `From` implementations map `&str` and `String` to `Text`, and `&[u8]` +/// and `Vec` to `Bytes`. +enum DiagnosticPayload { + Text(String), + Bytes(Vec), +} + /// Why a recognized record could not be parsed, carried by the `Raw` it falls back to. /// Implements `Error`, serialized as a snake_case reason. enum DecodeError { - /// The key identifies the azure-init diagnostics family, but not a version this reader supports. + /// The key identifies the diagnostics family, but not a version this reader supports. UnsupportedVersion, /// Chunks are missing: not a contiguous run from 0. IncompleteGroup, @@ -352,11 +369,11 @@ struct DiagnosticKey { } /// Opens a span. -struct DiagnosticStart { key: DiagnosticKey, payload: Vec } +struct DiagnosticStart { key: DiagnosticKey, payload: DiagnosticPayload } /// Closes a span; carries its verdict and elapsed milliseconds. -struct DiagnosticFinish { key: DiagnosticKey, payload: Vec, result: Outcome, duration_ms: u64 } +struct DiagnosticFinish { key: DiagnosticKey, payload: DiagnosticPayload, result: Outcome, duration_ms: u64 } /// A point observation; may carry a verdict or a self-contained timing. -struct DiagnosticEvent { key: DiagnosticKey, payload: Vec, result: Option, duration_ms: Option } +struct DiagnosticEvent { key: DiagnosticKey, payload: DiagnosticPayload, result: Option, duration_ms: Option } /// One decoded emission, typed by kind. enum Diagnostic { @@ -406,58 +423,66 @@ impl DiagnosticReader { impl DiagnosticWriter { /// Fix the local producer identity and boot epoch used by every emitted record. - /// The writer always emits `AZURE_INIT_V1`. + /// The writer always emits `DIAG_V1`. pub fn new(store: KvpPoolStore, agent: impl Into, vm_id: impl Into) -> Result; /// Open a span. `event_id` links this start to the finish that closes it. - pub fn emit_start(&self, event_id: &str, name: &str, payload: &[u8], encoding: Option) -> Result<(), KvpError>; + pub fn emit_start(&self, event_id: &str, name: &str, payload: impl Into, encoding: Option) -> Result<(), KvpError>; /// Close the span opened under `event_id`, recording its `result` and elapsed `duration_ms`. - pub fn emit_finish(&self, event_id: &str, name: &str, payload: &[u8], encoding: Option, result: Outcome, duration_ms: u64) -> Result<(), KvpError>; + pub fn emit_finish(&self, event_id: &str, name: &str, payload: impl Into, encoding: Option, result: Outcome, duration_ms: u64) -> Result<(), KvpError>; /// Record a standalone point observation; the writer assigns its `event_id`. /// `result` and `duration_ms` are set only when measured. - pub fn emit_event(&self, name: &str, payload: &[u8], encoding: Option, result: Option, duration_ms: Option) -> Result<(), KvpError>; + pub fn emit_event(&self, name: &str, payload: impl Into, encoding: Option, result: Option, duration_ms: Option) -> Result<(), KvpError>; } ``` ### CLI -`dump` prints every record in pool order as raw `KEY=VALUE`. `--parse` is an add-on: it interprets each record, decoding diagnostics and parsing the `PROVISIONING_REPORT` into a `ProvisioningReport`, and leaves anything it does not recognize as its raw key and value. Nothing is dropped. +JSON is the default output mode for `dump`; `--json` may state it explicitly, and the mutually exclusive `--text` selects human-readable output. `dump` returns every physical record in pool order as a JSON array of `{ "key", "value" }` objects. `--parse` changes what is represented, not the output mode: it decodes diagnostics, parses the `PROVISIONING_REPORT`, and leaves anything it does not recognize as `Raw`. Nothing is dropped. ```text -dump -> every record, raw KEY=VALUE -dump --parse -> known records parsed (Diagnostic, ProvisioningReport); the rest stay Raw +dump -> JSON array of every physical {key, value} record +dump --parse -> JSON array of Diagnostic, ProvisioningReport, and Raw entries +dump --text -> every physical record as raw KEY=VALUE +dump --parse --text -> one human-readable line per interpreted entry ``` -- `--parse` calls `DiagnosticReader::entries()`, one `Entry` per item. A recognized record that will not parse — a broken diagnostic, an unsupported azure-init diagnostics version, or a malformed report — stays `Raw` with its `DecodeError`, so its key, value, and the reason are all shown. +- `--parse` calls `DiagnosticReader::entries()`, one `Entry` per item. A recognized record that will not parse — a broken diagnostic, an unsupported diagnostics version, or a malformed report — stays `Raw` with its `DecodeError`, so its key, value, and the reason are all shown. - `--name` filters the parsed diagnostics by name; other entries are unaffected. +- `--json` and `--text` are mutually exclusive; omitting both is equivalent to `--json`. +- In parsed text output, `Text` payloads print directly and `Bytes` payloads print as standard base64 under `payload_b64`. Examples: ```text -# dump: every record raw, including the report and a dmesg chunk whose group was truncated +# default dump: every physical record as JSON, including the report and a truncated dmesg group $ dump -AZURE_INIT_V1|azure-init-0.1.1|1700000000|vm-abc|finish|provision:run|9c1d...|2026-08-31T12:34:57.101Z|none|success|312|0=provisioning succeeded -AZURE_INIT_V1|azure-init-0.1.1|1700000000|vm-abc|event|dmesg|d4e5...|2026-07-27T21:33:25.00Z|gz+b64|||17= -PROVISIONING_REPORT=result=success|agent=azure-init-0.1.1|pps_type=None|vm_id=vm-abc|timestamp=2026-08-31T12:34:57.500Z +[ + {"key":"DIAG_V1|azure-init-0.1.1|1700000000|vm-abc|finish|provision:run|9c1d...|2026-08-31T12:34:57.101Z|none|success|312|0","value":"provisioning succeeded"}, + {"key":"DIAG_V1|azure-init-0.1.1|1700000000|vm-abc|event|dmesg|d4e5...|2026-07-27T21:33:25.00Z|gz+b64|||17","value":""}, + {"key":"PROVISIONING_REPORT","value":"result=success|agent=azure-init-0.1.1|pps_type=None|vm_id=vm-abc|timestamp=2026-08-31T12:34:57.500Z"} +] -# --parse: the diagnostic decodes, the report parses, the unassemblable dmesg chunk stays raw with its reason +# --parse remains JSON: the diagnostic decodes, the report parses, and the dmesg chunk stays Raw $ dump --parse -{"type":"diagnostic","kind":"finish","name":"provision:run","event_id":"9c1d...","timestamp":"2026-08-31T12:34:57.101Z","result":"success","duration":312,"payload":"provisioning succeeded"} -{"type":"raw","key":"AZURE_INIT_V1|azure-init-0.1.1|1700000000|vm-abc|event|dmesg|d4e5...|gz+b64|||17","value":"","error":"incomplete_group"} -{"type":"PROVISIONING_REPORT","result":"success","agent":"azure-init-0.1.1","vm_id":"vm-abc","timestamp":"2026-08-31T12:34:57.500Z","pps_type":"None"} +[ + {"type":"diagnostic","kind":"finish","name":"provision:run","event_id":"9c1d...","timestamp":"2026-08-31T12:34:57.101Z","result":"success","duration":312,"payload":"provisioning succeeded"}, + {"type":"raw","key":"DIAG_V1|azure-init-0.1.1|1700000000|vm-abc|event|dmesg|d4e5...|gz+b64|||17","value":"","error":"incomplete_group"}, + {"type":"PROVISIONING_REPORT","result":"success","agent":"azure-init-0.1.1","vm_id":"vm-abc","timestamp":"2026-08-31T12:34:57.500Z","pps_type":"None"} +] ``` ## Adoption -azure-init adopts `AZURE_INIT_V1` by cutting over to it when it switches to the kvp crate; this is the first supported azure-init diagnostics format. The unversioned azure-init shape under Records today is pre-adoption rather than a compatibility contract: `DiagnosticReader` leaves one of those records as `Raw` instead of guessing which schema it follows. There are no prior records to migrate. +azure-init adopts `DIAG_V1` by cutting over to it when it switches to the kvp crate; this is the first supported diagnostics schema. The unversioned azure-init shape under Records today is pre-adoption rather than a compatibility contract: `DiagnosticReader` leaves one of those records as `Raw` instead of guessing which schema it follows. There are no prior records to migrate. Updating cloud-init to emit this format is a non-goal for now and may be revisited; until then cloud-init is read-only through the compatibility bridge. ## Compatibility -cloud-init writes diagnostics into the same pool in its own format. `DiagnosticReader` dispatches its records to a read-only bridge that maps them onto the same model; `DiagnosticWriter` never writes cloud-init's format, and the `AZURE_INIT_V1` parser never parses a cloud-init value. +cloud-init writes diagnostics into the same pool in its own format. `DiagnosticReader` dispatches its records to a read-only bridge that maps them onto the same model; `DiagnosticWriter` never writes cloud-init's format, and the `DIAG_V1` parser never parses a cloud-init value. cloud-init keys put the type and name before the identifiers, and current keys include a `vm_id` that older keys omit: @@ -483,6 +508,6 @@ The bridge maps fields onto the model: | duration | value field on a finish (seconds; the bridge converts to milliseconds) | | encoding | the value `{encoding, data}` envelope, not the key | -cloud-init has no `AZURE_INIT_V1` field and the bridge does not invent one. Its `CLOUD_INIT` prefix selects the bridge, which performs its source-specific parsing before constructing the same version-independent `DiagnosticKey` as the azure-init parser. +cloud-init has no `DIAG_V1` field and the bridge does not invent one. Its `CLOUD_INIT` prefix selects the bridge, which performs its source-specific parsing before constructing the same version-independent `DiagnosticKey` as the `DIAG_V1` parser. -Reassembly is cloud-init specific: chunks are JSON objects, so the bridge validates each `msg_i` against the chunk index, concatenates the still-escaped `msg` slices, and unescapes the joined string once. If the result is an `{encoding, data}` envelope, the bridge decodes it with the same encodings as the core; otherwise the message is the payload. cloud-init declares its encoding in the value, which is why it is read there and not from the key. \ No newline at end of file +Reassembly is cloud-init specific: chunks are JSON objects, so the bridge validates each `msg_i` against the chunk index, concatenates the still-escaped `msg` slices, and unescapes the joined string once. If the result is an `{encoding, data}` envelope, the bridge decodes it with the same encodings as the core into `DiagnosticPayload::Bytes`; otherwise the message becomes `DiagnosticPayload::Text`. cloud-init declares its encoding in the value, which is why it is read there and not from the key. \ No newline at end of file From a2537c9a1f43292635d529f5a96ae06da9779a56 Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Fri, 4 Sep 2026 11:10:14 -0700 Subject: [PATCH 20/32] docs(kvp): remove boot epoch from diagnostics schema --- libazureinit-kvp/diagnostics-proposal.md | 41 ++++++++++++------------ 1 file changed, 20 insertions(+), 21 deletions(-) diff --git a/libazureinit-kvp/diagnostics-proposal.md b/libazureinit-kvp/diagnostics-proposal.md index a846dd59..56fa989c 100644 --- a/libazureinit-kvp/diagnostics-proposal.md +++ b/libazureinit-kvp/diagnostics-proposal.md @@ -96,10 +96,11 @@ This proposal defines one versioned diagnostic record format, emitted here by az The proposed key is the current azure-init key with a few changes, all keeping metadata in the key. It is pipe-delimited with `|` reserved, begins with a diagnostic schema-version identifier, and always ends with the chunk index, so a single-record value still ends in `|0`: ```text -DIAG_V1||||||||||| +DIAG_V1|||||||||| ``` - `DIAG_V1` is the `diagnostic_version_id`: `DIAG` identifies the general diagnostics family and `V1` identifies version 1 of its wire schema. It does not identify the producer; `agent` continues to do that. The schema covers the field layout, required and optional fields, token meanings, units, encoding rules, and chunk framing. It is one self-identifying token rather than a bare number because the pool also contains unrelated keys: a reader can recognize an unsupported future `DIAG_V*` record without mistaking an arbitrary numeric key for a diagnostic. A schema change requires a new `diagnostic_version_id`; an agent release by itself does not. +- The previous `boot_epoch` field is removed. Each diagnostic already carries an absolute timestamp, while stale-pool cleanup owns removal of records from prior boots; the schema does not duplicate that boot identity. - `type` becomes `kind`, narrowed to the three timeline positions (`start`/`finish`/`event`); any other category is carried by `name`, `encoding`, or `result`, not a token. - `encoding` names the payload encoding in the key so a large artifact can be compressed. - `result` and `duration` are the finish's verdict fields, carried in the key: `result` is a `success`/`fail` token and `duration` is the elapsed milliseconds. Both are required on a finish, optional on an event, and empty on a start. Everything else, including the plain-text value, is unchanged. @@ -108,7 +109,6 @@ DIAG_V1|||||||||||||||`. cloud-init carries the same fields in its value JSON, which the bridge maps across (see Compatibility). ```text -DIAG_V1|azure-init-0.1.1|1700000000|vm-abc|finish|provision:run|9c1d2e3f-...|2026-08-31T12:34:57.101Z|none|success|312|0 value: provisioning succeeded +DIAG_V1|azure-init-0.1.1|vm-abc|finish|provision:run|9c1d2e3f-...|2026-08-31T12:34:57.101Z|none|success|312|0 value: provisioning succeeded ``` #### event @@ -170,17 +170,17 @@ Its value is the observed payload, read per `encoding`: a short text as `none`, ```text # plain text, one record -DIAG_V1|azure-init-0.1.1|1700000000|vm-abc|event|imds|8f3e...|2026-07-27T21:33:24.300Z|none|||0 value: Retrieved 1 key from IMDS +DIAG_V1|azure-init-0.1.1|vm-abc|event|imds|8f3e...|2026-07-27T21:33:24.300Z|none|||0 value: Retrieved 1 key from IMDS # an event that is itself a failure sets result -DIAG_V1|azure-init-0.1.1|1700000000|vm-abc|event|imds|7b2c...|2026-07-27T21:33:24.400Z|none|fail||0 value: IMDS unreachable +DIAG_V1|azure-init-0.1.1|vm-abc|event|imds|7b2c...|2026-07-27T21:33:24.400Z|none|fail||0 value: IMDS unreachable # a self-contained timing sets duration but no result -DIAG_V1|azure-init-0.1.1|1700000000|vm-abc|event|imds:probe|5d6e...|2026-07-27T21:33:24.500Z|none||52|0 value: probed IMDS in 52ms +DIAG_V1|azure-init-0.1.1|vm-abc|event|imds:probe|5d6e...|2026-07-27T21:33:24.500Z|none||52|0 value: probed IMDS in 52ms # compressed artifact, split across records, indices 0..N -DIAG_V1|azure-init-0.1.1|1700000000|vm-abc|event|dmesg|9a1b...||gz+b64|||0 value: -DIAG_V1|azure-init-0.1.1|1700000000|vm-abc|event|dmesg|9a1b...||gz+b64|||1 value: +DIAG_V1|azure-init-0.1.1|vm-abc|event|dmesg|9a1b...||gz+b64|||0 value: +DIAG_V1|azure-init-0.1.1|vm-abc|event|dmesg|9a1b...||gz+b64|||1 value: ``` ### Payloads @@ -256,7 +256,7 @@ Write: ```mermaid flowchart TD client["Provisioning client"] --> new["DiagnosticWriter::new
store, agent, vm_id"] - new --> init{"identity valid and
boot epoch readable?"} + new --> init{"producer identity valid?"} init -->|"no"| initerr["Err(KvpError)
writer not constructed"] init -->|"yes"| emit["DiagnosticWriter::emit_*
text or byte payload"] emit --> valid{"fields, kind invariants,
payload, and encoding valid?"} @@ -317,7 +317,7 @@ The `diagnostic_version_id` is part of every diagnostic group key, so chunks fro The crate exposes two interfaces over `KvpPoolStore`; neither holds files or locks, and both delegate all IO to the store. `DiagnosticWriter` is the provisioning clients' producer interface: clients provide diagnostic meaning and payload, but do not construct keys, frame chunks, or append diagnostic records directly. A diagnostic consumer reads through `DiagnosticReader`; a caller that wants untyped records uses the store directly. The writer produces azure-init records only. The reader understands supported diagnostics versions and reads cloud-init through the bridge. The format and behavior are in Proposed design; the types, API, and CLI are here. -Their initialization is deliberately asymmetric. `DiagnosticReader` needs only a store because every source, identity, boot, and format decision comes from the records it reads; constructing one performs no IO. `DiagnosticWriter` needs the store plus the local `agent` and `vm_id`, validates that stable producer identity once, and obtains the boot epoch once for every record it will emit. It always writes the crate's current `DIAG_V1` format; callers cannot select a version or ask it to write cloud-init records. A process that needs both interfaces constructs them from clones of the same `KvpPoolStore`. +Their initialization is deliberately asymmetric. `DiagnosticReader` needs only a store because every source, identity, and format decision comes from the records it reads. `DiagnosticWriter` also needs the local `agent` and `vm_id` and validates that stable producer identity once. Neither constructor reads the pool or resolves boot state. The writer always writes the crate's current `DIAG_V1` format; callers cannot select a version or ask it to write cloud-init records. A process that needs both interfaces constructs them from clones of the same `KvpPoolStore`. ```rust const DIAGNOSTIC_VERSION_ID: &str = "DIAG_V1"; @@ -357,7 +357,6 @@ enum DecodeError { /// The reader consumes the schema ID while selecting a parser, then every supported source maps here. struct DiagnosticKey { agent: String, - boot_epoch: i64, /// Older cloud-init keys omit it. vm_id: Option, name: String, @@ -408,7 +407,6 @@ struct DiagnosticWriter { store: KvpPoolStore, agent: String, vm_id: String, - boot_epoch: i64, } impl DiagnosticReader { @@ -422,7 +420,7 @@ impl DiagnosticReader { } impl DiagnosticWriter { - /// Fix the local producer identity and boot epoch used by every emitted record. + /// Fix the local producer identity used by every emitted record. /// The writer always emits `DIAG_V1`. pub fn new(store: KvpPoolStore, agent: impl Into, vm_id: impl Into) -> Result; @@ -460,8 +458,8 @@ Examples: # default dump: every physical record as JSON, including the report and a truncated dmesg group $ dump [ - {"key":"DIAG_V1|azure-init-0.1.1|1700000000|vm-abc|finish|provision:run|9c1d...|2026-08-31T12:34:57.101Z|none|success|312|0","value":"provisioning succeeded"}, - {"key":"DIAG_V1|azure-init-0.1.1|1700000000|vm-abc|event|dmesg|d4e5...|2026-07-27T21:33:25.00Z|gz+b64|||17","value":""}, + {"key":"DIAG_V1|azure-init-0.1.1|vm-abc|finish|provision:run|9c1d...|2026-08-31T12:34:57.101Z|none|success|312|0","value":"provisioning succeeded"}, + {"key":"DIAG_V1|azure-init-0.1.1|vm-abc|event|dmesg|d4e5...|2026-07-27T21:33:25.00Z|gz+b64|||17","value":""}, {"key":"PROVISIONING_REPORT","value":"result=success|agent=azure-init-0.1.1|pps_type=None|vm_id=vm-abc|timestamp=2026-08-31T12:34:57.500Z"} ] @@ -469,7 +467,7 @@ $ dump $ dump --parse [ {"type":"diagnostic","kind":"finish","name":"provision:run","event_id":"9c1d...","timestamp":"2026-08-31T12:34:57.101Z","result":"success","duration":312,"payload":"provisioning succeeded"}, - {"type":"raw","key":"DIAG_V1|azure-init-0.1.1|1700000000|vm-abc|event|dmesg|d4e5...|gz+b64|||17","value":"","error":"incomplete_group"}, + {"type":"raw","key":"DIAG_V1|azure-init-0.1.1|vm-abc|event|dmesg|d4e5...|gz+b64|||17","value":"","error":"incomplete_group"}, {"type":"PROVISIONING_REPORT","result":"success","agent":"azure-init-0.1.1","vm_id":"vm-abc","timestamp":"2026-08-31T12:34:57.500Z","pps_type":"None"} ] ``` @@ -498,7 +496,6 @@ The bridge maps fields onto the model: | Model field | cloud-init source | |---|---| | agent | the literal `CLOUD_INIT` | -| boot_epoch | `incarnation` | | kind | `type` (`start`, `finish`, else `event`) | | name | `name` | | vm_id | present only on current keys | @@ -508,6 +505,8 @@ The bridge maps fields onto the model: | duration | value field on a finish (seconds; the bridge converts to milliseconds) | | encoding | the value `{encoding, data}` envelope, not the key | +`incarnation` remains part of cloud-init's source key because that is the format cloud-init writes. The bridge includes it while grouping cloud-init chunks so records from different incarnations cannot combine, then discards it; it is not part of `DiagnosticKey`. + cloud-init has no `DIAG_V1` field and the bridge does not invent one. Its `CLOUD_INIT` prefix selects the bridge, which performs its source-specific parsing before constructing the same version-independent `DiagnosticKey` as the `DIAG_V1` parser. Reassembly is cloud-init specific: chunks are JSON objects, so the bridge validates each `msg_i` against the chunk index, concatenates the still-escaped `msg` slices, and unescapes the joined string once. If the result is an `{encoding, data}` envelope, the bridge decodes it with the same encodings as the core into `DiagnosticPayload::Bytes`; otherwise the message becomes `DiagnosticPayload::Text`. cloud-init declares its encoding in the value, which is why it is read there and not from the key. \ No newline at end of file From db40dab77963b8f68791fa50beccbfa853785d6f Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Tue, 8 Sep 2026 08:58:33 -0700 Subject: [PATCH 21/32] docs(kvp): trim and cut redundancy --- libazureinit-kvp/diagnostics-proposal.md | 107 ++++++++++------------- 1 file changed, 48 insertions(+), 59 deletions(-) diff --git a/libazureinit-kvp/diagnostics-proposal.md b/libazureinit-kvp/diagnostics-proposal.md index 56fa989c..b7d61a27 100644 --- a/libazureinit-kvp/diagnostics-proposal.md +++ b/libazureinit-kvp/diagnostics-proposal.md @@ -2,24 +2,22 @@ ## Background -Diagnostics are the records a provisioning client emits to explain what a boot did: spans that mark when an operation such as `provision:run` starts and finishes, and point observations that capture a single event or artifact, such as an IMDS result or a `dmesg` snapshot. They let an operator or the host reconstruct a boot and triage a failure after the fact, including when the guest is no longer reachable. +Diagnostics explain what a provisioning client did during boot: spans mark the start and finish of operations such as `provision:run`, while point observations capture events or artifacts such as an IMDS result or `dmesg` snapshot. They let operators reconstruct a boot and triage failures even when the guest is unreachable. -A Hyper-V guest carries these records to the host through a KVP pool: a flat namespace of key=value records that the host copies out of the running guest. We already have `KvpPoolStore`, which gives raw, in-order access to that pool but no notion of diagnostics. +A Hyper-V guest exposes these records to the host through a KVP pool, a flat key=value namespace. `KvpPoolStore` provides raw, in-order access but no diagnostic semantics. The pool constrains the design in four ways: -- A record is a flat key=value pair. The key is the only structured metadata; the value is opaque bytes to the pool. -- Keys and values are size-capped. The safe key limit is 254 bytes and the safe value limit is 1022 bytes, so a message larger than the value limit cannot fit in one record and must span several records that share a key and differ only by a trailing index. -- The host may copy or truncate the pool at any moment. A reader can find a group whose members are missing, duplicated, or out of order. -- There is no native grouping or typing. A diagnostic event exists only as a convention encoded in the key and the value. +- Records are flat key=value pairs; keys hold structured metadata and values are opaque bytes. +- Safe limits are 254-byte keys and 1022-byte values. Larger values span records whose keys differ by a trailing index. +- The host may copy or truncate the pool at any moment, leaving group members missing, duplicated, or out of order. +- Grouping and typing exist only by convention in the key and value. -This spec defines a next-generation diagnostics format for a provisioning client (azure-init), the writer that produces it, and the reader that interprets the pool. cloud-init also writes diagnostics into the same pool in its own format; reading those is a separate read-only concern (see Compatibility). +This spec defines the versioned format emitted by azure-init, its writer, and a reader for the pool. cloud-init uses a separate format supported read-only through the compatibility bridge. ### Records today -Both producers already write diagnostics into the pool, in two different shapes. - -azure-init uses a pipe-delimited key with a plain-text value and no encoding field. The `type` token is `start`, `finish`, or `event`; a value over the size limit is split with a trailing chunk index: +azure-init uses a pipe-delimited key with a plain-text value and no encoding field. Its `type` is `start`, `finish`, or `event`; oversized values use a trailing chunk index: ```text ||||||| @@ -41,7 +39,7 @@ azure-init-0.1.1|1700000000|vm-abc|event|config:dump|1a2b3c4d-5e6f-7a8b-9c0d-1e2 azure-init-0.1.1|1700000000|vm-abc|event|config:dump|1a2b3c4d-5e6f-7a8b-9c0d-1e2f3a4b5c6d||1 value: ``` -cloud-init puts the type and name before the identifiers; a current key carries a `vm_id` that an older key omits. The value is JSON with `ts` and `msg`, `result` and `duration` on a finish, `msg_i` per chunk on a split value, and an `{encoding, data}` envelope inside `msg` when compressed: +cloud-init puts type and name before the identifiers; current keys carry a `vm_id` that older keys omit. Values are JSON with `ts` and `msg`; finishes add `result` and `duration`, splits add `msg_i`, and compressed artifacts embed `{encoding, data}` in `msg`: ```text current CLOUD_INIT|||||[|] @@ -85,25 +83,25 @@ CLOUD_INIT|1785187982|compressed|dmesg|0e5e179d-5341-478b-8456-fbb90621bdf8|d4e5 msg -> {"encoding":"gz+b64","data":"H4sIAAAA..."} ``` -The two disagree on field order, on where the timestamp and encoding live, and on whether the value is text or JSON. +The formats differ in field order, metadata placement, and value representation. ## Proposed design -This proposal defines one versioned diagnostic record format, emitted here by azure-init, and separate reader and writer interfaces over `KvpPoolStore`. The reader interprets the pool; the writer produces only this format. Untyped access remains on `KvpPoolStore` itself. +The design adds separate interfaces over `KvpPoolStore`: a reader interprets the pool, and a writer emits the versioned format. Untyped callers use `KvpPoolStore` directly. ### Diagnostics format -The proposed key is the current azure-init key with a few changes, all keeping metadata in the key. It is pipe-delimited with `|` reserved, begins with a diagnostic schema-version identifier, and always ends with the chunk index, so a single-record value still ends in `|0`: +The proposed key keeps metadata in a pipe-delimited key with `|` reserved. It begins with a schema ID and ends with the chunk index, including `|0` for a single-record value: ```text DIAG_V1|||||||||| ``` -- `DIAG_V1` is the `diagnostic_version_id`: `DIAG` identifies the general diagnostics family and `V1` identifies version 1 of its wire schema. It does not identify the producer; `agent` continues to do that. The schema covers the field layout, required and optional fields, token meanings, units, encoding rules, and chunk framing. It is one self-identifying token rather than a bare number because the pool also contains unrelated keys: a reader can recognize an unsupported future `DIAG_V*` record without mistaking an arbitrary numeric key for a diagnostic. A schema change requires a new `diagnostic_version_id`; an agent release by itself does not. -- The previous `boot_epoch` field is removed. Each diagnostic already carries an absolute timestamp, while stale-pool cleanup owns removal of records from prior boots; the schema does not duplicate that boot identity. -- `type` becomes `kind`, narrowed to the three timeline positions (`start`/`finish`/`event`); any other category is carried by `name`, `encoding`, or `result`, not a token. -- `encoding` names the payload encoding in the key so a large artifact can be compressed. -- `result` and `duration` are the finish's verdict fields, carried in the key: `result` is a `success`/`fail` token and `duration` is the elapsed milliseconds. Both are required on a finish, optional on an event, and empty on a start. Everything else, including the plain-text value, is unchanged. +- `DIAG_V1` is the `diagnostic_version_id`: `DIAG` identifies the format family and `V1` its schema. The schema defines layout, field requirements, tokens, units, encoding, and chunking; `agent` separately identifies the producer. The combined token distinguishes unsupported `DIAG_V*` records from unrelated keys. Schema changes require a new ID; agent releases do not. +- `boot_epoch` is removed because timestamps identify occurrences and stale-pool cleanup removes prior-boot records. +- `type` becomes `kind`, limited to the timeline positions `start`, `finish`, and `event`; `name`, `encoding`, and `result` carry other classifications. +- `encoding` in the key supports compressed payloads. +- `result` (`success` or `fail`) and `duration` (milliseconds) are required on finishes, optional on events, and empty on starts. Values otherwise remain plain text or encoded artifacts. | Field | Meaning | |---|---| @@ -121,9 +119,7 @@ DIAG_V1|||||||| #### Key size -The whole key is one string, and the host silently truncates a guest-written key past 254 UTF-8 bytes; safe-mode `KvpPoolStore` rejects it first. The fixed fields spend most of that budget: `vm_id` and `event_id` at 36 bytes each plus the 24-byte `timestamp` are already 96 bytes, and `DIAG_V1` adds another 7 before the enums, numbers, and delimiters. A representative finish key runs about 173 bytes, leaving roughly 81 before the limit. - -Only `agent` and `name` are free-form; every other field is bounded by its format or its enum. The writer caps the two so the whole key cannot exceed 254 bytes: +The host silently truncates keys past 254 UTF-8 bytes; safe-mode `KvpPoolStore` rejects them first. Only `agent` and `name` are free-form, so the writer caps them to bound the full key: | Field | Cap | Bounded by | |---|---|---| @@ -140,15 +136,11 @@ With those caps the worst-case key is 226 bytes, leaving 28 bytes inside the lim ### Kinds -`kind` marks where a record sits in an operation's timeline, and only that: `start` opens a timed operation, `finish` closes one, and `event` is a one-off that opens and closes nothing. Those three cover every position. A would-be fourth kind is really something the key already records: how the bytes are packed (`encoding`), what the record is about (`name`), or how it turned out (`result`). No kind carries a structured payload; the value is always a plain message or an artifact read per `encoding`. - -For a concrete case, cloud-init's `type` field carries `compressed` and `system-info` tokens beside the same three. Neither is a timeline position: `compressed` is how the bytes are packed, an `encoding` here, and `system-info` is a subject, a `name`. How cloud-init's tokens map onto these three is in Compatibility. +`kind` marks timeline position only: `start` opens an operation, `finish` closes one, and `event` is a one-off. These cover every timeline position. Other categories belong in `encoding`, `name`, or `result`; values are messages or artifacts, not kind-specific structures. For cloud-init, `compressed` maps to encoding and `system-info` to name (see Compatibility). #### start -A `start` opens a span, an operation that takes measurable time such as `provision:run`. Its `timestamp` marks when the operation began, and its value is a short opening message. - -It shares one `event_id` with its `finish`; that shared id is what ties the pair, so a `start` whose `finish` never arrives stands out as an operation that began but never ended, exactly the signal an operator wants after a hang or crash. +A `start` opens a measurable operation such as `provision:run`. Its timestamp marks when it began, and its value is a short message. It shares an `event_id` with its finish; an unmatched start signals an incomplete operation after a hang or crash. ```text DIAG_V1|azure-init-0.1.1|vm-abc|start|provision:run|9c1d2e3f-...|2026-08-31T12:34:56.789Z|none|||0 value: starting @@ -156,7 +148,7 @@ DIAG_V1|azure-init-0.1.1|vm-abc|start|provision:run|9c1d2e3f-...|2026-08-31T12:3 #### finish -A `finish` closes the span it shares an `event_id` with. Its `timestamp` is later than the start's, and it reports the operation's outcome directly in the key: `result` is `success` or `fail`, and `duration` is the elapsed milliseconds. The writer holds the start instant, so it stamps `duration` at emit time rather than making a reader pair the two records to recover it, and a truncated pool that kept only the finish still carries both the verdict and the elapsed time. The value stays a human message such as `provisioning succeeded` or `provisioning failed: `. cloud-init carries the same fields in its value JSON, which the bridge maps across (see Compatibility). +A `finish` closes the span sharing its `event_id`. It records the later timestamp, `result`, and elapsed `duration` at emit time, so it remains self-contained if the start is lost. Its value is a message such as `provisioning succeeded` or `provisioning failed: `. The cloud-init bridge maps equivalent value fields (see Compatibility). ```text DIAG_V1|azure-init-0.1.1|vm-abc|finish|provision:run|9c1d2e3f-...|2026-08-31T12:34:57.101Z|none|success|312|0 value: provisioning succeeded @@ -164,9 +156,7 @@ DIAG_V1|azure-init-0.1.1|vm-abc|finish|provision:run|9c1d2e3f-...|2026-08-31T12: #### event -An `event` is a point observation with no open or close: a single fact captured at one moment, such as an IMDS result or a `dmesg` snapshot. It has its own `event_id`, with no start or finish to pair. Most diagnostics are events. - -Its value is the observed payload, read per `encoding`: a short text as `none`, or a large artifact the caller marks for compression as `gz+b64`, split across chunks that share the `event_id` and differ only by `chunk_index`. An event may also set `result` (`success`/`fail`) when the observation is itself a pass or failure, and `duration` when it measured how long something took with no start and finish to bracket it; it leaves either empty when it does not apply. +An `event` is a point observation such as an IMDS result or `dmesg` snapshot. It has its own `event_id` and no span pair. Its payload is text (`none`) or a compressed artifact (`gz+b64`), split by `chunk_index` when needed. It may set `result` or `duration` when applicable. ```text # plain text, one record @@ -185,32 +175,32 @@ DIAG_V1|azure-init-0.1.1|vm-abc|event|dmesg|9a1b...||gz+b64|||1 value: `. `DiagnosticPayload::Text` carries a Rust `String`, which is already valid UTF-8; `DiagnosticPayload::Bytes` carries arbitrary bytes. Writer methods accept `impl Into`, with conversions from `&str`, `String`, `&[u8]`, and `Vec`, so callers can pass either form without separate method names. +`DiagnosticPayload` distinguishes valid UTF-8 `Text(String)` from arbitrary `Bytes(Vec)`. Writer methods accept `impl Into` with conversions from `&str`, `String`, `&[u8]`, and `Vec`, avoiding separate method names. -The caller still chooses the wire `encoding`; the input type does not guess whether compression is useful. The writer handles each combination as follows: +The caller chooses the wire `encoding`; the input type does not infer whether compression is useful: | Input | `encoding=None` | `gz+b64` | |---|---|---| | Text | Store its UTF-8 bytes directly | Encode its UTF-8 bytes | | Bytes | Validate UTF-8, then store directly; reject invalid UTF-8 | Encode the arbitrary bytes | -On read, `none` must decode as UTF-8 and produces `DiagnosticPayload::Text`; invalid UTF-8 is `Undecodable`. `gz+b64` produces `DiagnosticPayload::Bytes` after decoding, even when those bytes happen to be valid UTF-8, because the wire schema does not declare their content type. +On read, `none` produces `Text` after UTF-8 validation; invalid UTF-8 is `Undecodable`. `gz+b64` produces `Bytes` even when the decoded bytes are valid UTF-8, because the wire does not declare their content type. -In JSON output, `Text` is a JSON string. `Bytes` is `{ "type": "bytes", "encoding": "base64", "data": "..." }`, because JSON cannot carry arbitrary bytes; this presentation encoding does not change the diagnostic's wire `encoding`. +JSON renders `Text` as a string and `Bytes` as `{ "type": "bytes", "encoding": "base64", "data": "..." }`; this presentation does not change the wire `encoding`. ### Encodings -`encoding` names how the payload is represented in the KVP value, chosen by the caller when it emits a diagnostic rather than guessed from size, so the choice is deterministic. It lives in the key so a reader knows it without parsing the value, and the value stays a single opaque payload rather than a wrapper. Arbitrary binary uses `gz+b64` because the store holds values as UTF-8 and trims trailing nulls, so raw bytes would not survive a round trip. `DIAG_V1` does not define standalone base64; a later schema can add another encoding without changing the key's overall shape. cloud-init instead declares its encoding inside the value (see Compatibility). +`encoding` names the KVP value representation and is chosen by the caller, not inferred from size. Keeping it in the key leaves the value opaque. Arbitrary binary uses `gz+b64` because the UTF-8 store trims trailing nulls. `DIAG_V1` omits standalone base64; a later schema can add encodings without changing the key shape. cloud-init declares encoding inside its value (see Compatibility). -Whatever the encoding, a value over the 1022-byte limit is split across chunks that share the value's `event_id` and `kind` and are ordered by `chunk_index`; decoding joins them in order before anything else. +Values over 1022 bytes are split across chunks sharing `event_id` and `kind`, ordered by `chunk_index`, and joined before decoding. #### none -Plain UTF-8 text, the default and the common case. The value is the message as stored, so decoding is a no-op; a split value is just the text slices joined in index order. Span messages and short observations use `none`. +Plain UTF-8 text and the default, used for span messages and short observations. Decoding joins split text in index order. #### gz+b64 -base64 of gzip of the raw payload bytes, for a large compressible artifact such as `dmesg`. Writing gzips the payload, base64-encodes the result, then chunks it; decoding joins the chunks, base64-decodes, and gunzips. It is one gzip stream spread across the chunks, so a missing chunk makes the whole artifact unavailable: a visible index gap is `IncompleteGroup`, while a contiguous truncation is detected during decoding as `Undecodable`. That is the cost it trades for far fewer records. +Base64 of one gzip stream, used for compressible artifacts such as `dmesg`. Writing compresses, encodes, then chunks; decoding reverses the process. A visible index gap is `IncompleteGroup`; contiguous truncation is `Undecodable`. This sacrifices partial readability for fewer records. ### Reads and writes @@ -230,11 +220,11 @@ Diagnostics pool └─ chunk 0 ``` -`DiagnosticReader::entries()` reads the pool once and interprets every record, returning one `Entry` each: it combines and decodes a diagnostic into a `Diagnostic`, parses the `PROVISIONING_REPORT` into a `ProvisioningReport`, and leaves anything else as `Raw` (its key and value). A recognized record that will not parse — a broken diagnostic group, an unsupported diagnostics version, or a malformed report — also falls back to `Raw`, carrying the `DecodeError`, so nothing is dropped. A caller that wants the untouched records reads `KvpPoolStore` directly. +`DiagnosticReader::entries()` reads the pool once and returns each logical item as a decoded `Diagnostic`, parsed `ProvisioningReport`, or `Raw` key and value. Unrecognized records are `Raw`; recognized but invalid records are `Raw` with a `DecodeError`, so nothing is dropped. `KvpPoolStore` provides the untouched physical view. Writing is the inverse: `DiagnosticWriter` stamps `DIAG_V1`, converts the typed payload according to `encoding`, frames it into records, and appends them to the `KvpPoolStore`. -Reading the pool like a log is the `Diagnostic` entries in timestamp order. The host may reorder the pool, so a reader sorts on the timestamp each carries. Text output renders each as one line: +The reader sorts diagnostics by timestamp because the host may reorder the pool. Text output renders one line per diagnostic: ```text 2026-08-31T12:34:56.789Z start provision:run @@ -242,14 +232,14 @@ Reading the pool like a log is the `Diagnostic` entries in timestamp order. The 2026-08-31T12:34:57.101Z finish provision:run success 312ms ``` -Splitting the type by kind is what keeps each line honest: a `finish` always carries its `result` and `duration`, a `start` carries neither, and an `event` carries either when the caller measured it, so the renderer never second-guesses an optional field. An operation that began but never finished is a `Start` with no matching `Finish`, still printed in place, which is the signal an operator wants after a crash: +Fields follow `kind`: a finish always carries `result` and `duration`, a start carries neither, and an event carries either when measured. An unmatched start remains visible as an incomplete operation: ```text 2026-08-31T12:35:10.000Z start provision:run (no finish) ``` -A finished operation already carries its `duration` on its finish line, so there is nothing to roll up. Pairing a start to its finish is a plain group-by on `event_id` when a reader wants both ends at once, and a start with no matching finish is an operation that began but never ended. +No duration rollup is needed; pairing starts and finishes is a group-by on `event_id`. Write: @@ -307,17 +297,17 @@ flowchart TD repe --> out ``` -`KvpError` and `DecodeError` mark different boundaries. A `KvpError` means the requested construction, read, or write could not complete and is returned by the method. A `DecodeError` means the pool read succeeded but stored data could not be interpreted; `entries()` still succeeds and preserves that data as `Entry::Raw` with the reason. +`KvpError` means construction, reading, or writing failed and is returned by the method. `DecodeError` describes uninterpretable stored data; `entries()` still succeeds and preserves it as `Entry::Raw`. -The writer prepares and validates the complete batch before calling the store, so an identity, field, payload, encoding, or size error writes nothing. This includes byte input that is not valid UTF-8 when `encoding=None`. Once `append_multiple` begins, a lock, write, or flush error returns `KvpError` but may leave part of the batch in the pool. A reader reports a visible index gap as `IncompleteGroup` and invalid encoded content as `Undecodable`. A contiguous prefix of a `none` payload has neither condition and cannot be distinguished from a complete value because this format carries no total chunk count. +The writer validates the full batch before storage, so identity, field, payload, encoding, and size errors write nothing; this includes non-UTF-8 bytes with `encoding=None`. Once `append_multiple` starts, an error may leave a partial batch. Readers report visible index gaps as `IncompleteGroup` and invalid encoded content as `Undecodable`. Without a total chunk count, a contiguous prefix of a `none` payload is indistinguishable from a complete value. -The `diagnostic_version_id` is part of every diagnostic group key, so chunks from different schemas can never combine. The rest of the group key includes `kind` because a span's start and finish share an `event_id`. There is no decode-time size limit either: the producer is trusted and the pool bounds the input, so an artifact either fits when it is written or is never written. +Every group key includes `diagnostic_version_id` and `kind`, preventing chunks from different schemas or span ends from combining. There is no decode-time size limit: the pool already bounds producer input. ## Crate design -The crate exposes two interfaces over `KvpPoolStore`; neither holds files or locks, and both delegate all IO to the store. `DiagnosticWriter` is the provisioning clients' producer interface: clients provide diagnostic meaning and payload, but do not construct keys, frame chunks, or append diagnostic records directly. A diagnostic consumer reads through `DiagnosticReader`; a caller that wants untyped records uses the store directly. The writer produces azure-init records only. The reader understands supported diagnostics versions and reads cloud-init through the bridge. The format and behavior are in Proposed design; the types, API, and CLI are here. +Both interfaces hold no files or locks and delegate IO to `KvpPoolStore`. Provisioning clients use `DiagnosticWriter`, supplying diagnostic meaning and payload rather than constructing keys or chunks. Consumers use `DiagnosticReader`; untyped callers use the store directly. The writer emits azure-init records, while the reader supports known schemas and cloud-init through a bridge. -Their initialization is deliberately asymmetric. `DiagnosticReader` needs only a store because every source, identity, and format decision comes from the records it reads. `DiagnosticWriter` also needs the local `agent` and `vm_id` and validates that stable producer identity once. Neither constructor reads the pool or resolves boot state. The writer always writes the crate's current `DIAG_V1` format; callers cannot select a version or ask it to write cloud-init records. A process that needs both interfaces constructs them from clones of the same `KvpPoolStore`. +Initialization is asymmetric: reader identity comes from stored records, so `DiagnosticReader` needs only a store; `DiagnosticWriter` also needs and validates the local `agent` and `vm_id`. Neither constructor reads the pool or boot state. The writer always emits `DIAG_V1`; callers cannot select a version or cloud-init format. Both may share clones of one `KvpPoolStore`. ```rust const DIAGNOSTIC_VERSION_ID: &str = "DIAG_V1"; @@ -438,7 +428,7 @@ impl DiagnosticWriter { ### CLI -JSON is the default output mode for `dump`; `--json` may state it explicitly, and the mutually exclusive `--text` selects human-readable output. `dump` returns every physical record in pool order as a JSON array of `{ "key", "value" }` objects. `--parse` changes what is represented, not the output mode: it decodes diagnostics, parses the `PROVISIONING_REPORT`, and leaves anything it does not recognize as `Raw`. Nothing is dropped. +`dump` defaults to JSON; `--json` makes that explicit and `--text` selects human-readable output. Without `--parse`, it returns every physical record in pool order. `--parse` calls `DiagnosticReader::entries()`, returning typed diagnostics and reports while preserving other or invalid records as `Raw`. ```text dump -> JSON array of every physical {key, value} record @@ -447,7 +437,6 @@ dump --text -> every physical record as raw KEY=VALUE dump --parse --text -> one human-readable line per interpreted entry ``` -- `--parse` calls `DiagnosticReader::entries()`, one `Entry` per item. A recognized record that will not parse — a broken diagnostic, an unsupported diagnostics version, or a malformed report — stays `Raw` with its `DecodeError`, so its key, value, and the reason are all shown. - `--name` filters the parsed diagnostics by name; other entries are unaffected. - `--json` and `--text` are mutually exclusive; omitting both is equivalent to `--json`. - In parsed text output, `Text` payloads print directly and `Bytes` payloads print as standard base64 under `payload_b64`. @@ -474,22 +463,22 @@ $ dump --parse ## Adoption -azure-init adopts `DIAG_V1` by cutting over to it when it switches to the kvp crate; this is the first supported diagnostics schema. The unversioned azure-init shape under Records today is pre-adoption rather than a compatibility contract: `DiagnosticReader` leaves one of those records as `Raw` instead of guessing which schema it follows. There are no prior records to migrate. +azure-init adopts `DIAG_V1` as its first supported schema when it switches to this crate. Earlier unversioned records are pre-adoption, not a compatibility contract; `DiagnosticReader` treats them as `Raw`. No migration is required. -Updating cloud-init to emit this format is a non-goal for now and may be revisited; until then cloud-init is read-only through the compatibility bridge. +Updating cloud-init to emit this format is a non-goal; it remains read-only through the compatibility bridge. ## Compatibility -cloud-init writes diagnostics into the same pool in its own format. `DiagnosticReader` dispatches its records to a read-only bridge that maps them onto the same model; `DiagnosticWriter` never writes cloud-init's format, and the `DIAG_V1` parser never parses a cloud-init value. +cloud-init uses its own format in the same pool. `DiagnosticReader` maps it through a read-only bridge; `DiagnosticWriter` never emits it, and the `DIAG_V1` parser never parses it. -cloud-init keys put the type and name before the identifiers, and current keys include a `vm_id` that older keys omit: +cloud-init keys put type and name before the identifiers; current keys include a `vm_id` that older keys omit: ```text current CLOUD_INIT|||||[|] older CLOUD_INIT||||[|] ``` -The value is JSON telemetry carrying `name`, `type`, `ts`, and `msg`, with `result` and `duration` on a span finish, a per-chunk `msg_i` on each chunk of a split value, and, for a compressed artifact, a `{encoding, data}` envelope inside `msg`. +Values are JSON with `name`, `type`, `ts`, and `msg`; finishes add `result` and `duration`, splits add `msg_i`, and compressed artifacts embed `{encoding, data}` in `msg`. The bridge maps fields onto the model: @@ -505,8 +494,8 @@ The bridge maps fields onto the model: | duration | value field on a finish (seconds; the bridge converts to milliseconds) | | encoding | the value `{encoding, data}` envelope, not the key | -`incarnation` remains part of cloud-init's source key because that is the format cloud-init writes. The bridge includes it while grouping cloud-init chunks so records from different incarnations cannot combine, then discards it; it is not part of `DiagnosticKey`. +The bridge uses cloud-init's `incarnation` to keep chunk groups separate, then discards it; `DiagnosticKey` does not expose it. -cloud-init has no `DIAG_V1` field and the bridge does not invent one. Its `CLOUD_INIT` prefix selects the bridge, which performs its source-specific parsing before constructing the same version-independent `DiagnosticKey` as the `DIAG_V1` parser. +The `CLOUD_INIT` prefix selects the bridge; no `DIAG_V1` field is invented. After source-specific parsing, the bridge constructs the same version-independent `DiagnosticKey` as the `DIAG_V1` parser. -Reassembly is cloud-init specific: chunks are JSON objects, so the bridge validates each `msg_i` against the chunk index, concatenates the still-escaped `msg` slices, and unescapes the joined string once. If the result is an `{encoding, data}` envelope, the bridge decodes it with the same encodings as the core into `DiagnosticPayload::Bytes`; otherwise the message becomes `DiagnosticPayload::Text`. cloud-init declares its encoding in the value, which is why it is read there and not from the key. \ No newline at end of file +For cloud-init chunks, the bridge validates each `msg_i`, joins the still-escaped `msg` slices, then unescapes once. An `{encoding, data}` envelope decodes to `DiagnosticPayload::Bytes`; otherwise `msg` becomes `DiagnosticPayload::Text`. Its encoding comes from the value, not the key. \ No newline at end of file From d785313dc5819717e20b596dbdebeb4eee79b65e Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Fri, 11 Sep 2026 12:17:16 -0700 Subject: [PATCH 22/32] feat(kvp): rewrite diagnostics per DIAG_V1 design proposal Reimplement the libazureinit-kvp diagnostics layer to match diagnostics-proposal.md: a versioned wire format with separate reader and writer over the raw KvpPoolStore. - Split src/diagnostics.rs into diagnostics/{diagnostic,encoding,writer, reader,cloud_init,mod} - DIAG_V1 key format (drops boot_epoch; adds encoding/result/duration) - DiagnosticWriter: typed emit_start/emit_finish/emit_event, validation before I/O, UTF-8 chunk framing at safe wire limits - DiagnosticReader: single-snapshot, lossless entries() -> Vec (Diagnostic | ProvisioningReport | Raw), first-seen order - DiagnosticPayload Text/Bytes with none and gz+b64 encodings - Read-only cloud-init compatibility bridge (both key layouts) - ProvisioningReport FromStr parsing surfaced as Entry::Report; wire format unchanged - CLI cutover: dump defaults to JSON with --parse/--text, emit --agent; remove --parse-diagnostics/--prefix/--tail - Add diagnostic KvpError variants; update docs (doc/kvp.md, proposal) Tracing integration is deferred to a follow-up PR. --- doc/kvp.md | 47 + libazureinit-kvp/Cargo.toml | 2 + libazureinit-kvp/diagnostics-proposal.md | 55 +- libazureinit-kvp/src/cli.rs | 350 ++++-- libazureinit-kvp/src/diagnostics.rs | 666 ----------- .../src/diagnostics/cloud_init.rs | 733 ++++++++++++ .../src/diagnostics/diagnostic.rs | 587 +++++++++ libazureinit-kvp/src/diagnostics/encoding.rs | 354 ++++++ libazureinit-kvp/src/diagnostics/mod.rs | 36 + libazureinit-kvp/src/diagnostics/reader.rs | 1047 +++++++++++++++++ libazureinit-kvp/src/diagnostics/writer.rs | 948 +++++++++++++++ libazureinit-kvp/src/error.rs | 107 +- libazureinit-kvp/src/lib.rs | 41 +- libazureinit-kvp/src/report.rs | 405 ++++++- libazureinit-kvp/src/store.rs | 107 +- libazureinit-kvp/tests/cli.rs | 551 +++++---- libazureinit-kvp/tests/diagnostics.rs | 672 +++++------ libazureinit-kvp/tests/fixtures/cloud_init.rs | 17 + 18 files changed, 5249 insertions(+), 1476 deletions(-) delete mode 100644 libazureinit-kvp/src/diagnostics.rs create mode 100644 libazureinit-kvp/src/diagnostics/cloud_init.rs create mode 100644 libazureinit-kvp/src/diagnostics/diagnostic.rs create mode 100644 libazureinit-kvp/src/diagnostics/encoding.rs create mode 100644 libazureinit-kvp/src/diagnostics/mod.rs create mode 100644 libazureinit-kvp/src/diagnostics/reader.rs create mode 100644 libazureinit-kvp/src/diagnostics/writer.rs create mode 100644 libazureinit-kvp/tests/fixtures/cloud_init.rs diff --git a/doc/kvp.md b/doc/kvp.md index 26ef7f9b..f2c89b04 100644 --- a/doc/kvp.md +++ b/doc/kvp.md @@ -239,3 +239,50 @@ no truncation. | `HV_KVP_EXCHANGE_MAX_RECORDS` | 1,024 | Max records per pool file | | `HV_KVP_SAFE_MAX_UTF8_KEY_SIZE` | 255 | 254 UTF-8 bytes + null-terminator; no kernel truncation on write path | | `HV_KVP_SAFE_MAX_UTF8_VALUE_SIZE` | 1,023 | 1,022 UTF-8 bytes + null-terminator; no kernel truncation on write path | + +--- + +## Diagnostics API and CLI + +`libazureinit-kvp` exports `DiagnosticWriter` for `DIAG_V1` records and +`DiagnosticReader` for diagnostics, provisioning reports, and raw records. +The writer validates the agent identifier and VM UUID at construction. +`emit_start` and `emit_finish` share a caller-supplied event UUID; +`emit_event` generates its own. Durations are integer milliseconds. +Payloads are plain UTF-8 text or gzip plus base64 (`Encoding::GzB64`). + +Each reader call takes one fresh snapshot and returns `Entry::Diagnostic`, +`Entry::Report`, or `Entry::Raw` in first-seen pool order. Unknown records +remain raw; recognized but invalid records also carry a `DecodeError`. +Cloud-init records are supported through a read-only compatibility bridge. +Invalid physical UTF-8 fails the entire snapshot without partial output. +Neither reading nor emitting diagnostics clears stale pool data implicitly. + +| Command | Output | +|---------|--------| +| `libazureinit-kvp dump` | JSON array of physical key/value records in pool order, including duplicates | +| `libazureinit-kvp dump --parse` | JSON array of diagnostics and reports in timestamp order, then raw entries | +| `libazureinit-kvp dump --text` | Physical records as `KEY=VALUE` lines | +| `libazureinit-kvp dump --parse --text` | The same timestamp ordering, with binary payloads under `payload_b64` | +| `libazureinit-kvp dump --parse --name ssh` | Filter diagnostic names by substring; retain reports and raw entries | + +Parsed CLI output is oldest-first. Equal timestamps keep first-seen order; +raw entries retain their relative pool order at the end. This presentation +does not change `DiagnosticReader::entries()` or the pool file. + +`--json` and `--text` are mutually exclusive. Only `dump` defaults to JSON; +other commands retain their text defaults. Global `--dir` and `--pool` +options select the store. + +```sh +libazureinit-kvp emit --agent azure-init \ + --vm-id 3f2504e0-4f89-41d3-9a0c-0305e82c3301 \ + --name user:create_user --message "created azureuser" +``` + +The separate reader and writer replace `DiagnosticsKvp`. `--parse` replaces +`--parse-diagnostics`, and `emit --agent` replaces `--prefix`. The `--tail` +and `-n` options are removed. + +See the [diagnostics specification](../libazureinit-kvp/diagnostics-proposal.md) +for the wire format, validation limits, and cloud-init compatibility rules. diff --git a/libazureinit-kvp/Cargo.toml b/libazureinit-kvp/Cargo.toml index 1e994906..3f1c05fd 100644 --- a/libazureinit-kvp/Cargo.toml +++ b/libazureinit-kvp/Cargo.toml @@ -9,9 +9,11 @@ license = "MIT" description = "Hyper-V KVP (Key-Value Pair) storage library for azure-init." [dependencies] +base64 = "0.22" chrono = { version = "0.4", default-features = false, features = ["clock", "serde", "std"] } clap = { version = "4.5.21", features = ["derive"] } csv = "1" +flate2 = "1.0" libc = "0.2" serde = { version = "1.0", features = ["derive"] } serde_json = "1.0.96" diff --git a/libazureinit-kvp/diagnostics-proposal.md b/libazureinit-kvp/diagnostics-proposal.md index b7d61a27..0b5e91cd 100644 --- a/libazureinit-kvp/diagnostics-proposal.md +++ b/libazureinit-kvp/diagnostics-proposal.md @@ -134,6 +134,8 @@ The host silently truncates keys past 254 UTF-8 bytes; safe-mode `KvpPoolStore` With those caps the worst-case key is 226 bytes, leaving 28 bytes inside the limit. cloud-init reads are never capped; the bridge takes names as they are. +Wire examples with shortened UUIDs or `` are schematic. Stored `DIAG_V1` records require valid UUIDs and the exact timestamp format above. + ### Kinds `kind` marks timeline position only: `start` opens an operation, `finish` closes one, and `event` is a one-off. These cover every timeline position. Other categories belong in `encoding`, `name`, or `result`; values are messages or artifacts, not kind-specific structures. For cloud-init, `compressed` maps to encoding and `system-info` to name (see Compatibility). @@ -184,7 +186,9 @@ The caller chooses the wire `encoding`; the input type does not infer whether co | Text | Store its UTF-8 bytes directly | Encode its UTF-8 bytes | | Bytes | Validate UTF-8, then store directly; reject invalid UTF-8 | Encode the arbitrary bytes | -On read, `none` produces `Text` after UTF-8 validation; invalid UTF-8 is `Undecodable`. `gz+b64` produces `Bytes` even when the decoded bytes are valid UTF-8, because the wire does not declare their content type. +On read, `KvpPoolStore::dump()` validates the physical keys and values as UTF-8 before diagnostic parsing. Invalid UTF-8 in any field before its terminating NUL fails the entire snapshot with `KvpError::Io` carrying `InvalidData`; no entries are returned. Record preservation applies only to a successful string-based snapshot, not to arbitrary physical bytes. + +Within a successful snapshot, `none` produces `Text`. `gz+b64` produces `Bytes` even when the decoded bytes are valid UTF-8, because the wire does not declare their content type. Its stored base64 is UTF-8 text, so arbitrary decoded bytes do not violate the store contract. Invalid base64 or gzip remains `Undecodable` and falls back to `Raw`. JSON renders `Text` as a string and `Bytes` as `{ "type": "bytes", "encoding": "base64", "data": "..." }`; this presentation does not change the wire `encoding`. @@ -220,11 +224,15 @@ Diagnostics pool └─ chunk 0 ``` -`DiagnosticReader::entries()` reads the pool once and returns each logical item as a decoded `Diagnostic`, parsed `ProvisioningReport`, or `Raw` key and value. Unrecognized records are `Raw`; recognized but invalid records are `Raw` with a `DecodeError`, so nothing is dropped. `KvpPoolStore` provides the untouched physical view. +`DiagnosticReader::entries()` reads the pool once through `KvpPoolStore::dump()`. If that snapshot succeeds, each logical item becomes a decoded `Diagnostic`, parsed `ProvisioningReport`, or `Raw` key and value. Unrecognized records are `Raw`; recognized but invalid records are `Raw` with a `DecodeError`, so no record from the successful snapshot is dropped. `KvpPoolStore::dump()` exposes the physical key/value records as UTF-8 strings without diagnostic interpretation. + +Any snapshot failure, including invalid physical UTF-8 or malformed physical record framing, returns `KvpError` with no entries, even when other records are valid. Reads do not modify the pool; they neither skip unreadable records nor replace invalid bytes with lossy text. The existing string-based store and `RawKeyValue` interfaces remain unchanged. + +Only the exact `PROVISIONING_REPORT` key selects report parsing. Its value is one pipe-delimited CSV record of `key=value` fields, with double-quote escaping for embedded pipes, quotes, or newlines. `result` (`success` or `error`), `agent`, `vm_id`, `pps_type`, and an RFC 3339 `timestamp` are required; error reports also require `reason` and may include `documentation_url`. Field order is not significant. Empty text values remain supported, and identities and timestamps are preserved rather than normalized. Other fields remain ordered supporting data, including duplicate supporting-data keys. Duplicate standard fields, unknown enum tokens, missing required fields, or malformed CSV are `Raw` with `Malformed`, not silently repaired or interpreted with first/last-write-wins semantics. Writing is the inverse: `DiagnosticWriter` stamps `DIAG_V1`, converts the typed payload according to `encoding`, frames it into records, and appends them to the `KvpPoolStore`. -The reader sorts diagnostics by timestamp because the host may reorder the pool. Text output renders one line per diagnostic: +The reader preserves first-seen pool order: a complete chunk group occupies its first physical position, and failed groups remain raw records at their original positions. Only the CLI's `dump --parse` sorts diagnostics and reports by timestamp, oldest first, with stable ties and raw entries last. A span's timeline can be summarized as: ```text 2026-08-31T12:34:56.789Z start provision:run @@ -272,7 +280,7 @@ flowchart TD client["Diagnostic consumer / CLI"] --> new["DiagnosticReader::new(store)
no IO, cannot fail"] new --> entries["DiagnosticReader::entries()"] entries --> dump["KvpPoolStore::dump()"] - dump -->|"lock / read error"| readerr["Err(KvpError)
no entries returned"] + dump -->|"lock / read / framing / UTF-8 error"| readerr["Err(KvpError)
no entries returned"] dump -->|"ok"| cls{"first key field"} cls -->|"DIAG_V1"| dec["source parser
parse, group, decode"] cls -->|"unsupported DIAG_V*"| rawver["Entry::Raw
UnsupportedVersion"] @@ -297,11 +305,11 @@ flowchart TD repe --> out ``` -`KvpError` means construction, reading, or writing failed and is returned by the method. `DecodeError` describes uninterpretable stored data; `entries()` still succeeds and preserves it as `Entry::Raw`. +`KvpError` means construction, reading, or writing failed and is returned by the method; snapshot failures include invalid physical UTF-8. `DecodeError` describes uninterpretable records within a successful string-based snapshot; `entries()` still succeeds and preserves those records as `Entry::Raw`. The writer validates the full batch before storage, so identity, field, payload, encoding, and size errors write nothing; this includes non-UTF-8 bytes with `encoding=None`. Once `append_multiple` starts, an error may leave a partial batch. Readers report visible index gaps as `IncompleteGroup` and invalid encoded content as `Undecodable`. Without a total chunk count, a contiguous prefix of a `none` payload is indistinguishable from a complete value. -Every group key includes `diagnostic_version_id` and `kind`, preventing chunks from different schemas or span ends from combining. There is no decode-time size limit: the pool already bounds producer input. +Every group key includes `diagnostic_version_id` and `kind`, preventing chunks from different schemas or span ends from combining. There is no decode-time size limit. Writer-side field and chunk caps do not bound reader input: raw pool appends have no record-count cap, and compressed payloads may expand substantially. Reading a large pool or highly compressed payload can therefore require substantial memory. ## Crate design @@ -371,16 +379,14 @@ enum Diagnostic { Event(DiagnosticEvent), } -/// A record left as key and value: unknown to the parser, or a recognized one -/// that failed to decode, in which case `error` says why. +/// An uninterpreted record from a successful UTF-8 snapshot. struct RawKeyValue { key: String, value: String, error: Option, } -/// One interpreted item from `entries()`. A new known type is a new variant; -/// everything else stays `Raw`, so a reader never drops a record. +/// Every item from a successful snapshot is interpreted or preserved as `Raw`. enum Entry { Diagnostic(Diagnostic), Report(ProvisioningReport), @@ -403,9 +409,7 @@ impl DiagnosticReader { /// Constructing a reader performs no IO; the pool is read by `entries()`. pub fn new(store: KvpPoolStore) -> Self; - /// One `Entry` per item: each diagnostic is combined and decoded, the `PROVISIONING_REPORT` - /// is parsed into a `ProvisioningReport`, and everything else is `Raw`. A recognized record - /// that will not parse is `Raw` with its `DecodeError`, so nothing is dropped. + /// Interprets one UTF-8 snapshot; a store failure returns no entries. pub fn entries(&self) -> Result, KvpError>; } @@ -428,7 +432,9 @@ impl DiagnosticWriter { ### CLI -`dump` defaults to JSON; `--json` makes that explicit and `--text` selects human-readable output. Without `--parse`, it returns every physical record in pool order. `--parse` calls `DiagnosticReader::entries()`, returning typed diagnostics and reports while preserving other or invalid records as `Raw`. +`dump` defaults to JSON; `--json` makes that explicit and `--text` selects human-readable output. Without `--parse`, it returns every physical record in pool order. `--parse` calls `DiagnosticReader::entries()`, returning typed diagnostics and reports while preserving other or invalid records as `Raw`. Both modes require a successful string-based snapshot; invalid physical UTF-8 fails the command without returning records. + +Parsed JSON and text output sort diagnostics and reports by their timestamps as instants, oldest first, including timezone offsets and available fractional precision. Equal timestamps retain first-seen pool order. Raw entries follow the timestamped entries in their original relative pool order; the CLI does not infer timestamps from malformed or unrecognized records. This presentation does not change the reader API's ordering or write to the pool. ```text dump -> JSON array of every physical {key, value} record @@ -438,8 +444,9 @@ dump --parse --text -> one human-readable line per interpreted entry ``` - `--name` filters the parsed diagnostics by name; other entries are unaffected. -- `--json` and `--text` are mutually exclusive; omitting both is equivalent to `--json`. +- `--json` and `--text` are mutually exclusive; for `dump`, omitting both is equivalent to `--json`. Other commands retain their text defaults. - In parsed text output, `Text` payloads print directly and `Bytes` payloads print as standard base64 under `payload_b64`. +- `--parse` replaces `--parse-diagnostics`; `emit --agent` replaces `--prefix`. `--tail` and `-n` are removed. Examples: @@ -447,17 +454,17 @@ Examples: # default dump: every physical record as JSON, including the report and a truncated dmesg group $ dump [ - {"key":"DIAG_V1|azure-init-0.1.1|vm-abc|finish|provision:run|9c1d...|2026-08-31T12:34:57.101Z|none|success|312|0","value":"provisioning succeeded"}, - {"key":"DIAG_V1|azure-init-0.1.1|vm-abc|event|dmesg|d4e5...|2026-07-27T21:33:25.00Z|gz+b64|||17","value":""}, - {"key":"PROVISIONING_REPORT","value":"result=success|agent=azure-init-0.1.1|pps_type=None|vm_id=vm-abc|timestamp=2026-08-31T12:34:57.500Z"} + {"key":"DIAG_V1|azure-init-0.1.1|3f2504e0-4f89-41d3-9a0c-0305e82c3301|finish|provision:run|9c1d2e3f-4a5b-6c7d-8e9f-0a1b2c3d4e5f|2026-08-31T12:34:57.101Z|none|success|312|0","value":"provisioning succeeded"}, + {"key":"DIAG_V1|azure-init-0.1.1|3f2504e0-4f89-41d3-9a0c-0305e82c3301|event|dmesg|d4e5f6a7-b8c9-0d1e-2f3a-4b5c6d7e8f90|2026-07-27T21:33:25.000Z|gz+b64|||17","value":""}, + {"key":"PROVISIONING_REPORT","value":"result=success|agent=azure-init-0.1.1|pps_type=None|vm_id=3f2504e0-4f89-41d3-9a0c-0305e82c3301|timestamp=2026-08-31T12:34:57.500Z"} ] # --parse remains JSON: the diagnostic decodes, the report parses, and the dmesg chunk stays Raw $ dump --parse [ - {"type":"diagnostic","kind":"finish","name":"provision:run","event_id":"9c1d...","timestamp":"2026-08-31T12:34:57.101Z","result":"success","duration":312,"payload":"provisioning succeeded"}, - {"type":"raw","key":"DIAG_V1|azure-init-0.1.1|vm-abc|event|dmesg|d4e5...|gz+b64|||17","value":"","error":"incomplete_group"}, - {"type":"PROVISIONING_REPORT","result":"success","agent":"azure-init-0.1.1","vm_id":"vm-abc","timestamp":"2026-08-31T12:34:57.500Z","pps_type":"None"} + {"type":"diagnostic","kind":"finish","agent":"azure-init-0.1.1","vm_id":"3f2504e0-4f89-41d3-9a0c-0305e82c3301","name":"provision:run","event_id":"9c1d2e3f-4a5b-6c7d-8e9f-0a1b2c3d4e5f","timestamp":"2026-08-31T12:34:57.101Z","encoding":"none","result":"success","duration":312,"payload":"provisioning succeeded"}, + {"type":"PROVISIONING_REPORT","result":"success","agent":"azure-init-0.1.1","vm_id":"3f2504e0-4f89-41d3-9a0c-0305e82c3301","timestamp":"2026-08-31T12:34:57.500Z","pps_type":"None"}, + {"type":"raw","key":"DIAG_V1|azure-init-0.1.1|3f2504e0-4f89-41d3-9a0c-0305e82c3301|event|dmesg|d4e5f6a7-b8c9-0d1e-2f3a-4b5c6d7e8f90|2026-07-27T21:33:25.000Z|gz+b64|||17","value":"","error":"incomplete_group"} ] ``` @@ -494,8 +501,12 @@ The bridge maps fields onto the model: | duration | value field on a finish (seconds; the bridge converts to milliseconds) | | encoding | the value `{encoding, data}` envelope, not the key | +Finish results `SUCCESS` and `FAIL` map to `success` and `fail`. An unmappable result such as `WARN` is preserved as `Raw` with `Malformed`; it is not coerced to a verdict or reclassified as an event. Source types other than `start` and `finish`, including standalone warnings, map to `event` without a span outcome. Duration conversion truncates fractional milliseconds and rejects negative or overflowing values. + The bridge uses cloud-init's `incarnation` to keep chunk groups separate, then discards it; `DiagnosticKey` does not expose it. The `CLOUD_INIT` prefix selects the bridge; no `DIAG_V1` field is invented. After source-specific parsing, the bridge constructs the same version-independent `DiagnosticKey` as the `DIAG_V1` parser. -For cloud-init chunks, the bridge validates each `msg_i`, joins the still-escaped `msg` slices, then unescapes once. An `{encoding, data}` envelope decodes to `DiagnosticPayload::Bytes`; otherwise `msg` becomes `DiagnosticPayload::Text`. Its encoding comes from the value, not the key. \ No newline at end of file +For cloud-init chunks, the bridge validates each `msg_i` and the consistency of source metadata, joins the still-escaped `msg` slices, then unescapes once. An `{encoding, data}` envelope decodes to `DiagnosticPayload::Bytes`; otherwise `msg` becomes `DiagnosticPayload::Text`. Its encoding comes from the value, not the key. + +cloud-init artifacts labeled `gz+b64` may contain zlib-wrapped data and line-wrapped base64. The bridge accepts zlib or gzip under that label and removes base64 whitespace before decoding. This read-only compatibility rule does not change `DIAG_V1`, which remains gzip-only. Other envelope encodings, including standalone `b64`, are `Undecodable`. \ No newline at end of file diff --git a/libazureinit-kvp/src/cli.rs b/libazureinit-kvp/src/cli.rs index 4941f4ad..53d9b389 100644 --- a/libazureinit-kvp/src/cli.rs +++ b/libazureinit-kvp/src/cli.rs @@ -7,12 +7,15 @@ use std::io::{self, Read, Write}; use std::path::PathBuf; use std::process::ExitCode; +use base64::{engine::general_purpose::STANDARD, Engine as _}; +use chrono::SecondsFormat; use clap::{Parser, Subcommand, ValueEnum}; use serde_json::json; use crate::{ - write_report, DiagnosticEvent, DiagnosticsKvp, KvpError, KvpPool, - KvpPoolStore, PoolMode, ProvisioningReport, ReportPpsType, + write_report, Diagnostic, DiagnosticPayload, DiagnosticReader, + DiagnosticWriter, Entry, KvpError, KvpPool, KvpPoolStore, PoolMode, + ProvisioningReport, ReportPpsType, PROVISIONING_REPORT_KEY, }; const EXIT_OK: u8 = 0; @@ -61,9 +64,13 @@ struct Cli { unsafe_mode: bool, /// Emit machine-readable JSON for commands that produce output. - #[arg(long, global = true)] + #[arg(long, global = true, conflicts_with = "text")] json: bool, + /// Emit human-readable text for commands that produce output. + #[arg(long, global = true, conflicts_with = "json")] + text: bool, + #[command(subcommand)] command: Command, } @@ -89,27 +96,17 @@ impl OutputMode { enum Command { /// Print store metadata. Info, - /// Print every record in insertion order as KEY=VALUE lines. + /// Print every record in pool order (JSON by default; --text for KEY=VALUE). /// - /// With --parse-diagnostics, reassemble chunked diagnostic events and - /// print only decodable diagnostics instead of raw KEY=VALUE lines. + /// With --parse, sort diagnostics and reports by timestamp, oldest first. + /// Equal timestamps keep pool order; raw entries follow in pool order. Dump { - /// Normalize decodable azure-init and cloud-init diagnostics. + /// Decode diagnostics and reports in oldest-first timestamp order. #[arg(long)] - parse_diagnostics: bool, - /// Only show diagnostics whose name contains this substring. - #[arg(long, requires = "parse_diagnostics")] + parse: bool, + /// Filter diagnostic names by substring; retain reports and raw entries. + #[arg(long, requires = "parse")] name: Option, - /// Print only the last COUNT diagnostics (default 20 when - /// COUNT is omitted). - #[arg( - short = 'n', - long = "tail", - num_args = 0..=1, - default_missing_value = "20", - requires = "parse_diagnostics" - )] - tail: Option, }, /// Print key=last_value entries sorted by key. Entries, @@ -123,9 +120,7 @@ enum Command { key: String, value: String, }, - /// Emit an azure-init diagnostic event: a structured KVP entry keyed - /// `||||||`. - /// Distinct from the raw `write` command. + /// Emit a DIAG_V1 point event with a fresh UUID and current timestamp. Emit { /// Event name, e.g. user:create_user. #[arg(long)] @@ -133,12 +128,12 @@ enum Command { /// Event message (stored as the record value). #[arg(long)] message: String, - /// VM identifier (defaults to the current VM's ID). + /// VM UUID (defaults to the current VM's ID). #[arg(long)] vm_id: Option, - /// Event-key prefix (defaults to the reporting agent identifier). - #[arg(long)] - prefix: Option, + /// Reporting agent identifier (at most 32 UTF-8 bytes). + #[arg(long, default_value = DEFAULT_AGENT)] + agent: String, }, /// Replace the pool from KEY=VALUE lines read from --file or stdin. Load { @@ -244,6 +239,11 @@ impl From for KvpPool { } fn dispatch(cli: Cli, stdout: &mut W) -> Result { + if cli.json && cli.text { + return Err(CliError::Usage( + "--json and --text cannot be used together".to_owned(), + )); + } let pool: KvpPool = cli.pool.into(); let mode = if cli.unsafe_mode { PoolMode::Unsafe @@ -260,14 +260,13 @@ fn dispatch(cli: Cli, stdout: &mut W) -> Result { match cli.command { Command::Info => info(&store, stdout, output), - Command::Dump { - parse_diagnostics, - name, - tail, - } => { - let parse = parse_diagnostics - .then_some(ParseDiagnosticsArgs { name, tail }); - dump(&store, stdout, parse, output) + Command::Dump { parse, name } => { + let output = if cli.text { + OutputMode::Text + } else { + OutputMode::Json + }; + dump(&store, stdout, parse, name.as_deref(), output) } Command::Entries => entries(&store, stdout, output), Command::Read { key } => read(&store, stdout, &key, output), @@ -283,8 +282,8 @@ fn dispatch(cli: Cli, stdout: &mut W) -> Result { name, message, vm_id, - prefix, - } => emit(&store, name, message, vm_id, prefix), + agent, + } => emit(&store, name, message, vm_id, agent), Command::Load { file } => load(&store, file), Command::AppendMultiple { file } => append_multiple(&store, file), Command::Delete { key } => delete(&store, stdout, &key, output), @@ -363,21 +362,16 @@ fn info( } Ok(EXIT_OK) } -struct ParseDiagnosticsArgs { - name: Option, - tail: Option, -} fn dump( store: &KvpPoolStore, stdout: &mut W, - parse: Option, + parse: bool, + name: Option<&str>, output: OutputMode, ) -> Result { - if let Some(parse) = parse { - return diagnostics_entries( - store, stdout, parse.name, parse.tail, output, - ); + if parse { + return diagnostics_entries(store, stdout, name, output); } let records = store.dump()?; @@ -491,78 +485,109 @@ fn is_stale( fn diagnostics_entries( store: &KvpPoolStore, stdout: &mut W, - name: Option, - tail: Option, + name: Option<&str>, output: OutputMode, ) -> Result { - let diagnostics = DiagnosticsKvp::new(store.clone(), "", "")?; - let mut events = diagnostics.entries()?; + let mut entries = DiagnosticReader::new(store.clone()).entries()?; - if let Some(needle) = name.as_deref() { - events.retain(|event| event.name.contains(needle)); - } - if let Some(count) = tail { - let excess = events.len().saturating_sub(count); - events.drain(..excess); + if let Some(needle) = name { + entries.retain(|entry| match entry { + Entry::Diagnostic(diagnostic) => { + diagnostic.key().name.contains(needle) + } + Entry::Report(_) | Entry::Raw(_) => true, + }); } + entries.sort_by_cached_key(|entry| { + let timestamp = match entry { + Entry::Diagnostic(diagnostic) => Some(diagnostic.key().timestamp), + Entry::Report(report) => Some(report.timestamp()), + Entry::Raw(_) => None, + }; + (timestamp.is_none(), timestamp) + }); + match output { OutputMode::Text => { - for event in &events { - let mut line = diagnostics_event_text(event); - let _ = write!(line, " message={}", event.message); - writeln!(stdout, "{line}")?; + for entry in &entries { + writeln!(stdout, "{}", diagnostic_entry_text(entry))?; } } OutputMode::Json => { - let array: Vec<_> = - events.iter().map(diagnostics_event_json).collect(); - writeln_json(stdout, &serde_json::Value::Array(array))?; + let value = serde_json::to_value(&entries) + .expect("diagnostic entries always serialize to JSON"); + writeln_json(stdout, &value)?; } } Ok(EXIT_OK) } -/// Render a [`DiagnosticEvent`] as a single text line of `key=value` -/// fields, omitting optional fields the source did not provide. -fn diagnostics_event_text(event: &DiagnosticEvent) -> String { - let mut line = format!( - "event kind={} agent={} boot_epoch={}", - event.kind, event.agent, event.boot_epoch - ); - if let Some(vm_id) = &event.vm_id { +fn diagnostic_entry_text(entry: &Entry) -> String { + match entry { + Entry::Diagnostic(diagnostic) => diagnostic_text(diagnostic), + Entry::Report(report) => { + format!("{PROVISIONING_REPORT_KEY}={}", report.encode()) + } + Entry::Raw(raw) => { + let mut line = format!("raw key={} value={}", raw.key, raw.value); + if let Some(error) = raw.error { + let _ = write!(line, " error={error}"); + } + line + } + } +} + +fn diagnostic_text(diagnostic: &Diagnostic) -> String { + let key = diagnostic.key(); + let mut line = + format!("diagnostic kind={} agent={}", diagnostic.kind(), key.agent); + if let Some(vm_id) = &key.vm_id { let _ = write!(line, " vm_id={vm_id}"); } - let _ = write!(line, " name={} event_id={}", event.name, event.event_id); - let _ = write!(line, " timestamp={}", event.timestamp); - if let Some(result) = &event.result { + let _ = write!(line, " name={} event_id={}", key.name, key.event_id); + let timestamp = key.timestamp.to_rfc3339_opts(SecondsFormat::Millis, true); + let encoding = key + .encoding + .as_ref() + .map(ToString::to_string) + .unwrap_or_else(|| "none".to_owned()); + let _ = write!(line, " timestamp={timestamp} encoding={encoding}"); + let (result, duration_ms) = match diagnostic { + Diagnostic::Start(_) => (None, None), + Diagnostic::Finish(finish) => { + (Some(finish.result), Some(finish.duration_ms)) + } + Diagnostic::Event(event) => (event.result, event.duration_ms), + }; + if let Some(result) = result { let _ = write!(line, " result={result}"); } - if let Some(duration) = event.duration { - let _ = write!(line, " duration={duration}"); + if let Some(duration_ms) = duration_ms { + let _ = write!(line, " duration={duration_ms}ms"); + } + match diagnostic.payload() { + DiagnosticPayload::Text(text) => { + let _ = write!(line, " payload={text}"); + } + DiagnosticPayload::Bytes(bytes) => { + let _ = write!(line, " payload_b64={}", STANDARD.encode(bytes)); + } } line } -/// Render a [`DiagnosticEvent`] as JSON, omitting optional fields the source -/// did not provide. -fn diagnostics_event_json(event: &DiagnosticEvent) -> serde_json::Value { - serde_json::to_value(event) - .expect("DiagnosticEvent always serializes to a JSON object") -} - -/// Emit an azure-init diagnostic event with the given fields. fn emit( store: &KvpPoolStore, name: String, message: String, vm_id: Option, - prefix: Option, + agent: String, ) -> Result { let vm_id = resolve_vm_id(vm_id)?; - let prefix = prefix.unwrap_or_else(|| DEFAULT_AGENT.to_string()); - let diagnostics = DiagnosticsKvp::new(store.clone(), vm_id, prefix)?; - diagnostics.emit_event(name, message)?; + let writer = DiagnosticWriter::new(store.clone(), agent, vm_id)?; + writer.emit_event(&name, message, None, None, None)?; Ok(EXIT_OK) } @@ -876,6 +901,7 @@ mod tests { dir: Some(dir.path().to_path_buf()), unsafe_mode: false, json: false, + text: false, command, } } @@ -886,6 +912,7 @@ mod tests { dir: Some(dir.path().to_path_buf()), unsafe_mode: false, json: true, + text: false, command, } } @@ -904,9 +931,8 @@ mod tests { /// A plain `dump` command with no diagnostics parsing. fn dump_cmd() -> Command { Command::Dump { - parse_diagnostics: false, + parse: false, name: None, - tail: None, } } @@ -952,6 +978,56 @@ mod tests { assert!(matches!(cli.command, Command::Dump { .. })); } + #[rstest] + #[case::default(None, false)] + #[case::json(Some("--json"), false)] + #[case::text(Some("--text"), true)] + fn dump_output_mode(#[case] flag: Option<&str>, #[case] text: bool) { + let dir = TempDir::new().unwrap(); + store_at(&dir).append("key", "value").unwrap(); + let mut args = vec![ + "libazureinit-kvp", + "--dir", + dir.path().to_str().unwrap(), + "dump", + ]; + args.extend(flag); + let (code, output) = run_dispatch(Cli::parse_from(args)); + assert_eq!(code, EXIT_OK); + if text { + assert_eq!(output, "key=value\n"); + } else { + assert_eq!( + parse_json(&output), + json!([{"key":"key","value":"value"}]) + ); + } + } + + #[rstest] + #[case::same_scope(vec!["dump", "--json", "--text"])] + #[case::global_json(vec!["--json", "dump", "--text"])] + #[case::global_text(vec!["--text", "dump", "--json"])] + fn json_and_text_are_mutually_exclusive(#[case] args: Vec<&str>) { + match Cli::try_parse_from( + std::iter::once("libazureinit-kvp").chain(args), + ) { + Err(error) => { + assert_eq!( + error.kind(), + clap::error::ErrorKind::ArgumentConflict + ); + } + Ok(cli) => { + let mut output = Vec::new(); + let error = dispatch(cli, &mut output).unwrap_err(); + assert!(matches!(error, CliError::Usage(_))); + assert_eq!(error.exit_code(), EXIT_USAGE_OR_VALIDATION); + assert!(output.is_empty()); + } + } + } + #[test] fn parse_pool_numeric_aliases() { let cases = [ @@ -1398,6 +1474,7 @@ mod tests { dir: Some(dir.path().to_path_buf()), unsafe_mode: true, json: false, + text: false, command: Command::Info, }; let (code, out) = run_dispatch(invocation); @@ -1446,7 +1523,13 @@ mod tests { )); let (_, dumped) = run_dispatch(cli(&dir, dump_cmd())); - assert_eq!(dumped, "k=1\nk=2\n"); + assert_eq!( + parse_json(&dumped), + json!([ + {"key": "k", "value": "1"}, + {"key": "k", "value": "2"}, + ]) + ); } #[test] @@ -1470,8 +1553,13 @@ mod tests { store.insert("a", "one").unwrap(); let (_, dumped) = run_dispatch(cli(&dir, dump_cmd())); - assert!(dumped.contains("a=one")); - assert!(dumped.contains("b=two")); + assert_eq!( + parse_json(&dumped), + json!([ + {"key": "b", "value": "two"}, + {"key": "a", "value": "one"}, + ]) + ); let (_, entries) = run_dispatch(cli(&dir, Command::Entries)); assert_eq!(entries, "a=one\nb=two\n"); @@ -1509,7 +1597,14 @@ mod tests { assert_eq!(code, EXIT_OK); let (_, dumped) = run_dispatch(cli(&dir, dump_cmd())); - assert_eq!(dumped, "a=1\na=2\nb=3\n"); + assert_eq!( + parse_json(&dumped), + json!([ + {"key": "a", "value": "1"}, + {"key": "a", "value": "2"}, + {"key": "b", "value": "3"}, + ]) + ); } #[test] @@ -1534,7 +1629,7 @@ mod tests { assert_eq!(out, "2\n"); let (_, dumped) = run_dispatch(cli(&dir, dump_cmd())); - assert_eq!(dumped, "b=2\n"); + assert_eq!(parse_json(&dumped), json!([{"key": "b", "value": "2"}])); } #[rstest] @@ -1616,6 +1711,7 @@ mod tests { dir: Some(blocker), unsafe_mode: false, json: false, + text: false, command: Command::Info, }; let mut out = Vec::new(); @@ -1701,6 +1797,18 @@ mod tests { assert_eq!(io_err.exit_code(), EXIT_IO); } + #[rstest] + #[case(KvpError::EmptyEventField { field: "name" })] + #[case(KvpError::EventFieldTooLong { field: "agent", max: 32, actual: 33 })] + #[case(KvpError::InvalidUuid { field: "event_id" })] + #[case(KvpError::DurationTooLarge { max_ms: 9_999_999_999, actual_ms: u64::MAX })] + #[case(KvpError::TooManyChunks { max: 1023 })] + #[case(KvpError::PayloadNotUtf8)] + #[case(KvpError::UnsupportedEncoding { token: "zstd+b64".into() })] + fn diagnostic_errors_use_validation_exit_code(#[case] error: KvpError) { + assert_eq!(CliError::from(error).exit_code(), EXIT_USAGE_OR_VALIDATION); + } + #[test] fn cli_error_from_conversions() { let from_kvp: CliError = KvpError::EmptyKey.into(); @@ -1752,6 +1860,58 @@ mod tests { assert_eq!(array[2]["value"], "one"); } + #[test] + fn dispatch_parsed_dump_filters_only_diagnostics() { + let dir = TempDir::new().unwrap(); + let store = store_at(&dir); + store.append("note", "raw value").unwrap(); + let writer = DiagnosticWriter::new( + store.clone(), + "agent", + "00000000-0000-0000-0000-000000000abc", + ) + .unwrap(); + writer + .emit_event("skip", "hidden", None, None, None) + .unwrap(); + writer + .emit_event("keep", "visible", None, None, None) + .unwrap(); + let report = + ProvisioningReport::success("agent", "vm-id", ReportPpsType::None); + write_report(&store, &report).unwrap(); + store.append("DIAG_V2|future", "preserved").unwrap(); + + let (_, output) = run_dispatch(cli( + &dir, + Command::Dump { + parse: true, + name: Some("keep".into()), + }, + )); + let entries = parse_json(&output); + let entries = entries.as_array().unwrap(); + assert_eq!(entries.len(), 4); + assert_eq!(entries[0]["type"], "diagnostic"); + assert_eq!(entries[0]["name"], "keep"); + assert_eq!(entries[0]["payload"], "visible"); + assert_eq!( + entries[1], + serde_json::to_value(Entry::Report(report)).unwrap() + ); + assert_eq!( + entries[2], + json!({"type": "raw", "key": "note", "value": "raw value"}) + ); + assert_eq!( + entries[3], + json!({ + "type": "raw", "key": "DIAG_V2|future", "value": "preserved", + "error": "unsupported_version", + }) + ); + } + #[test] fn dispatch_entries_json_emits_sorted_object() { let dir = TempDir::new().unwrap(); diff --git a/libazureinit-kvp/src/diagnostics.rs b/libazureinit-kvp/src/diagnostics.rs deleted file mode 100644 index 5e8ef3e8..00000000 --- a/libazureinit-kvp/src/diagnostics.rs +++ /dev/null @@ -1,666 +0,0 @@ -// Copyright (c) Microsoft Corporation. -// Licensed under the MIT License. - -//! Typed diagnostics over the raw [`KvpPoolStore`](crate::KvpPoolStore). -//! -//! [`DiagnosticsKvp`] writes azure-init diagnostics and reads diagnostics from -//! both azure-init and cloud-init into one [`DiagnosticEvent`] schema. Pool -//! records that are unrelated to diagnostics or cannot be decoded are skipped; -//! callers that need a lossless view can use [`KvpPoolStore::dump`]. -//! -//! Azure-init keys use this format: -//! `||||||`. -//! Every value chunk has a zero-based `|` suffix. -//! -//! Cloud-init keys use -//! `CLOUD_INIT||||[|]`, with the same -//! numeric suffix when chunked. A cloud-init chunk also stores its index in -//! the JSON `msg_i` field. The reader validates both indices before combining -//! the escaped `msg` fragments. - -use std::collections::HashMap; - -use chrono::{DateTime, SecondsFormat, Utc}; -use uuid::Uuid; - -use crate::{KvpError, KvpPoolStore}; - -const CLOUD_INIT_PREFIX: &str = "CLOUD_INIT"; -const EVENT_KEY_DELIMITER: char = '|'; -const CLOUD_INIT_MSG_MARKER: &str = "\"msg\":\""; - -/// Maximum number of UTF-8 value bytes stored in one diagnostic record. -/// -/// This conservative limit keeps records readable through the Hyper-V host -/// path. Longer messages are split at UTF-8 character boundaries. -pub const MAX_CHUNK_BYTES: usize = 1022; - -/// The lifecycle or reporting kind of a diagnostic entry. -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum DiagnosticKind { - /// A span opening. - Start, - /// A span closing. - Finish, - /// A point-in-time event. - Event, - /// Another source-specific reporting type, such as cloud-init's - /// `compressed` or `system-info`. - Other(String), -} - -impl DiagnosticKind { - fn from_token(token: &str) -> Self { - match token { - "start" => Self::Start, - "finish" => Self::Finish, - "event" => Self::Event, - other => Self::Other(other.to_string()), - } - } -} - -impl std::fmt::Display for DiagnosticKind { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - f.write_str(match self { - Self::Start => "start", - Self::Finish => "finish", - Self::Event => "event", - Self::Other(token) => token, - }) - } -} - -impl serde::Serialize for DiagnosticKind { - fn serialize(&self, serializer: S) -> Result - where - S: serde::Serializer, - { - serializer.serialize_str(&self.to_string()) - } -} - -/// One normalized azure-init or cloud-init diagnostic entry. -#[derive(Clone, Debug, PartialEq, serde::Serialize)] -#[non_exhaustive] -pub struct DiagnosticEvent { - /// Reporting source, such as `azure-init-0.1.1` or `CLOUD_INIT`. - pub agent: String, - /// Unix epoch second at which this boot began. - pub boot_epoch: i64, - /// VM identifier. Older cloud-init keys do not contain one. - #[serde(skip_serializing_if = "Option::is_none")] - pub vm_id: Option, - /// Entry lifecycle or source-specific reporting kind. - pub kind: DiagnosticKind, - /// Logical event or span name. - pub name: String, - /// Identifier shared by all chunks and, for spans, related lifecycle - /// entries. - pub event_id: String, - /// Time at which the entry occurred. - pub timestamp: DateTime, - /// Source result, when supplied by cloud-init. - #[serde(skip_serializing_if = "Option::is_none")] - pub result: Option, - /// Source duration in seconds, when supplied by cloud-init. - #[serde(skip_serializing_if = "Option::is_none")] - pub duration: Option, - /// Reassembled human-readable payload. - pub message: String, -} - -/// Typed diagnostic access over a KVP pool. -/// -/// The boot epoch is resolved once at construction so emitting tracing events -/// does not read `/proc/stat` for every entry. -#[derive(Clone, Debug)] -pub struct DiagnosticsKvp { - store: KvpPoolStore, - vm_id: String, - agent: String, - boot_epoch: i64, -} - -impl DiagnosticsKvp { - /// Create an accessor for `store`, stamping new entries with `vm_id` and - /// `agent`. - pub fn new( - store: KvpPoolStore, - vm_id: impl Into, - agent: impl Into, - ) -> Result { - let boot_epoch = store.boot_epoch()?; - Ok(Self { - store, - vm_id: vm_id.into(), - agent: agent.into(), - boot_epoch, - }) - } - - pub fn store(&self) -> &KvpPoolStore { - &self.store - } - - pub fn vm_id(&self) -> &str { - &self.vm_id - } - - pub fn agent(&self) -> &str { - &self.agent - } - - pub fn boot_epoch(&self) -> i64 { - self.boot_epoch - } - - /// Emit a point event using a fresh UUID and the current timestamp. - pub fn emit_event( - &self, - name: impl AsRef, - message: impl AsRef, - ) -> Result<(), KvpError> { - self.emit( - DiagnosticKind::Event, - name, - Uuid::new_v4().to_string(), - Utc::now(), - message, - ) - } - - /// Emit an azure-init diagnostic with caller-supplied lifecycle metadata. - /// - /// A tracing adapter can infer `kind` from its callback, retain one - /// `event_id` for a span, and pass the timestamp captured when the callback - /// occurred. Key formatting, chunking, and storage remain encapsulated - /// here. - pub fn emit( - &self, - kind: DiagnosticKind, - name: impl AsRef, - event_id: impl AsRef, - timestamp: DateTime, - message: impl AsRef, - ) -> Result<(), KvpError> { - let name = name.as_ref(); - let event_id = event_id.as_ref(); - let kind_token = kind.to_string(); - - reject_delimiter("agent", &self.agent)?; - reject_delimiter("vm_id", &self.vm_id)?; - reject_delimiter("kind", &kind_token)?; - reject_delimiter("name", name)?; - reject_delimiter("event_id", event_id)?; - - let timestamp = timestamp.to_rfc3339_opts(SecondsFormat::Millis, true); - let key = format_event_key( - &self.agent, - self.boot_epoch, - &self.vm_id, - &kind_token, - name, - event_id, - ×tamp, - ); - self.write_chunked(&key, message.as_ref()) - } - - /// Read all decodable diagnostics in first-seen pool order. - /// - /// Raw records, malformed entries, and incomplete chunk groups are omitted. - pub fn entries(&self) -> Result, KvpError> { - Ok(decode_entries(self.store.dump()?)) - } - - fn write_chunked(&self, key: &str, value: &str) -> Result<(), KvpError> { - let records = chunk_at_char_boundary(value, MAX_CHUNK_BYTES) - .into_iter() - .enumerate() - .map(|(index, chunk)| { - (format!("{key}{EVENT_KEY_DELIMITER}{index}"), chunk) - }); - self.store.append_multiple(records) - } -} - -fn reject_delimiter(field: &'static str, value: &str) -> Result<(), KvpError> { - if value.contains(EVENT_KEY_DELIMITER) { - return Err(KvpError::EventFieldContainsDelimiter { field }); - } - Ok(()) -} - -fn format_event_key( - agent: &str, - boot_epoch: i64, - vm_id: &str, - kind: &str, - name: &str, - event_id: &str, - timestamp: &str, -) -> String { - let d = EVENT_KEY_DELIMITER; - format!( - "{agent}{d}{boot_epoch}{d}{vm_id}{d}{kind}{d}{name}{d}{event_id}{d}{timestamp}" - ) -} - -fn chunk_at_char_boundary(value: &str, max_bytes: usize) -> Vec<&str> { - debug_assert!(max_bytes > 0, "max_bytes must be positive"); - if value.is_empty() { - return vec![""]; - } - - let mut chunks = Vec::new(); - let mut start = 0; - while start < value.len() { - if value.len() - start <= max_bytes { - chunks.push(&value[start..]); - break; - } - - let mut end = start + max_bytes; - while end > start && !value.is_char_boundary(end) { - end -= 1; - } - if end == start { - end = start + max_bytes + 1; - while end < value.len() && !value.is_char_boundary(end) { - end += 1; - } - } - - chunks.push(&value[start..end]); - start = end; - } - chunks -} - -struct AzureKey<'a> { - agent: &'a str, - boot_epoch: i64, - vm_id: &'a str, - kind: DiagnosticKind, - name: &'a str, - event_id: &'a str, - timestamp: DateTime, -} - -struct CloudInitKey<'a> { - boot_epoch: i64, - kind: DiagnosticKind, - name: &'a str, - vm_id: Option<&'a str>, - event_id: &'a str, -} - -enum ParsedKey<'a> { - Azure(AzureKey<'a>), - CloudInit(CloudInitKey<'a>), -} - -fn parse_diagnostic_key(key: &str) -> Option> { - if key.split(EVENT_KEY_DELIMITER).next()? == CLOUD_INIT_PREFIX { - parse_cloud_init_key(key).map(ParsedKey::CloudInit) - } else { - parse_azure_key(key).map(ParsedKey::Azure) - } -} - -fn parse_azure_key(key: &str) -> Option> { - let mut segments = key.split(EVENT_KEY_DELIMITER); - let agent = segments.next()?; - let boot_epoch = segments.next()?.parse().ok()?; - let vm_id = segments.next()?; - let kind = DiagnosticKind::from_token(segments.next()?); - let name = segments.next()?; - let event_id = segments.next()?; - let timestamp = parse_timestamp(segments.next()?)?; - if segments.next().is_some() { - return None; - } - - Some(AzureKey { - agent, - boot_epoch, - vm_id, - kind, - name, - event_id, - timestamp, - }) -} - -fn parse_cloud_init_key(key: &str) -> Option> { - let segments: Vec<_> = key.split(EVENT_KEY_DELIMITER).collect(); - let (vm_id, event_id) = match segments.as_slice() { - [CLOUD_INIT_PREFIX, _, _, _, event_id] => (None, *event_id), - [CLOUD_INIT_PREFIX, _, _, _, vm_id, event_id] => { - (Some(*vm_id), *event_id) - } - _ => return None, - }; - - Some(CloudInitKey { - boot_epoch: segments[1].parse().ok()?, - kind: DiagnosticKind::from_token(segments[2]), - name: segments[3], - vm_id, - event_id, - }) -} - -fn parse_timestamp(value: &str) -> Option> { - DateTime::parse_from_rfc3339(value) - .ok() - .map(|timestamp| timestamp.with_timezone(&Utc)) -} - -/// An indexed group is kept at the position where its first chunk appeared. -enum PendingEntry { - Standalone { - key: String, - value: String, - }, - Indexed { - key: String, - chunks: Vec<(u32, String)>, - }, -} - -fn decode_entries(dumped: Vec<(String, String)>) -> Vec { - let mut pending = Vec::::new(); - let mut indexed_groups = HashMap::::new(); - - for (key, value) in dumped { - if let Some((base, index)) = split_chunk_index(&key) { - let base = base.to_string(); - if let Some(position) = indexed_groups.get(&base).copied() { - if let PendingEntry::Indexed { chunks, .. } = - &mut pending[position] - { - chunks.push((index, value)); - } - } else { - indexed_groups.insert(base.clone(), pending.len()); - pending.push(PendingEntry::Indexed { - key: base, - chunks: vec![(index, value)], - }); - } - } else if parse_diagnostic_key(&key).is_some() { - pending.push(PendingEntry::Standalone { key, value }); - } - } - - pending - .into_iter() - .filter_map(|entry| match entry { - PendingEntry::Standalone { key, value } => { - decode_standalone(&key, &value) - } - PendingEntry::Indexed { key, chunks } => { - decode_indexed(&key, chunks) - } - }) - .collect() -} - -fn split_chunk_index(key: &str) -> Option<(&str, u32)> { - let (base, index) = key.rsplit_once(EVENT_KEY_DELIMITER)?; - let index = index.parse().ok()?; - parse_diagnostic_key(base)?; - Some((base, index)) -} - -fn order_chunks(mut chunks: Vec<(u32, String)>) -> Option> { - chunks.sort_by_key(|(index, _)| *index); - for (expected, (actual, _)) in chunks.iter().enumerate() { - if *actual != u32::try_from(expected).ok()? { - return None; - } - } - Some(chunks) -} - -fn decode_standalone(key: &str, value: &str) -> Option { - match parse_diagnostic_key(key)? { - ParsedKey::Azure(key) => Some(azure_event(key, value.to_string())), - ParsedKey::CloudInit(key) => decode_cloud_init_single(key, value), - } -} - -fn decode_indexed( - key: &str, - chunks: Vec<(u32, String)>, -) -> Option { - let chunks = order_chunks(chunks)?; - match parse_diagnostic_key(key)? { - ParsedKey::Azure(key) => Some(azure_event( - key, - chunks.into_iter().map(|(_, value)| value).collect(), - )), - ParsedKey::CloudInit(key) => decode_cloud_init_chunks(key, &chunks), - } -} - -fn azure_event(key: AzureKey<'_>, message: String) -> DiagnosticEvent { - DiagnosticEvent { - agent: key.agent.to_string(), - boot_epoch: key.boot_epoch, - vm_id: Some(key.vm_id.to_string()), - kind: key.kind, - name: key.name.to_string(), - event_id: key.event_id.to_string(), - timestamp: key.timestamp, - result: None, - duration: None, - message, - } -} - -fn decode_cloud_init_single( - key: CloudInitKey<'_>, - value: &str, -) -> Option { - let metadata: serde_json::Value = serde_json::from_str(value).ok()?; - if metadata.get("msg_i").is_some() { - return None; - } - let message = metadata.get("msg")?.as_str()?.to_string(); - cloud_init_event(key, &metadata, message) -} - -fn decode_cloud_init_chunks( - key: CloudInitKey<'_>, - chunks: &[(u32, String)], -) -> Option { - let mut metadata = None; - let mut escaped_message = String::new(); - - for (key_index, value) in chunks { - let chunk_metadata = cloud_init_chunk_metadata(value)?; - let value_index = chunk_metadata.get("msg_i")?.as_u64()?; - if value_index != u64::from(*key_index) { - return None; - } - if metadata.is_none() { - metadata = Some(chunk_metadata); - } - escaped_message.push_str(cloud_init_escaped_msg_slice(value)?); - } - - let message = - serde_json::from_str(&format!("\"{escaped_message}\"")).ok()?; - cloud_init_event(key, &metadata?, message) -} - -fn cloud_init_event( - key: CloudInitKey<'_>, - metadata: &serde_json::Value, - message: String, -) -> Option { - Some(DiagnosticEvent { - agent: CLOUD_INIT_PREFIX.to_string(), - boot_epoch: key.boot_epoch, - vm_id: key.vm_id.map(str::to_string), - kind: key.kind, - name: key.name.to_string(), - event_id: key.event_id.to_string(), - timestamp: parse_timestamp(metadata.get("ts")?.as_str()?)?, - result: metadata - .get("result") - .and_then(|result| result.as_str()) - .map(str::to_string), - duration: metadata.get("duration").and_then(|value| value.as_f64()), - message, - }) -} - -/// Recover a cloud-init chunk's raw, still-escaped `msg` fragment. -fn cloud_init_escaped_msg_slice(chunk: &str) -> Option<&str> { - let start = - chunk.find(CLOUD_INIT_MSG_MARKER)? + CLOUD_INIT_MSG_MARKER.len(); - let end = chunk.strip_suffix("\"}")?.len(); - chunk.get(start..end) -} - -/// Parse the valid metadata prefix before cloud-init's final `msg` field. -fn cloud_init_chunk_metadata(chunk: &str) -> Option { - let marker = format!(",{CLOUD_INIT_MSG_MARKER}"); - let end = chunk.find(&marker)?; - serde_json::from_str(&format!("{}}}", &chunk[..end])).ok() -} - -#[cfg(test)] -mod tests { - use super::*; - use rstest::rstest; - - const AGENT: &str = "azure-init-0.1.1"; - const VM_ID: &str = "3f2504e0-4f89-41d3-9a0c-0305e82c3301"; - const EVENT_ID: &str = "8f3e9c4a-1b2c-4d5e-9f01-234567890abc"; - const TIMESTAMP: &str = "2026-07-27T21:33:24.300Z"; - - #[rstest] - #[case(DiagnosticKind::Start, "start")] - #[case(DiagnosticKind::Finish, "finish")] - #[case(DiagnosticKind::Event, "event")] - #[case(DiagnosticKind::Other("compressed".into()), "compressed")] - fn kind_uses_wire_token(#[case] kind: DiagnosticKind, #[case] token: &str) { - assert_eq!(kind.to_string(), token); - assert_eq!(serde_json::to_value(kind).unwrap(), token); - } - - #[rstest] - #[case("", 4, vec![""])] - #[case("abcdef", 2, vec!["ab", "cd", "ef"])] - #[case("aéb", 2, vec!["a", "é", "b"])] - #[case("€€", 1, vec!["€", "€"])] - fn chunks_on_utf8_boundaries( - #[case] input: &str, - #[case] max: usize, - #[case] expected: Vec<&str>, - ) { - assert_eq!(chunk_at_char_boundary(input, max), expected); - } - - #[test] - fn azure_key_round_trips() { - let key = format_event_key( - AGENT, - 1_700_000_000, - VM_ID, - "event", - "user:create_user", - EVENT_ID, - TIMESTAMP, - ); - let ParsedKey::Azure(parsed) = parse_diagnostic_key(&key).unwrap() - else { - panic!("expected azure-init key"); - }; - assert_eq!(parsed.agent, AGENT); - assert_eq!(parsed.kind, DiagnosticKind::Event); - assert_eq!(parsed.timestamp, parse_timestamp(TIMESTAMP).unwrap()); - } - - #[test] - fn invalid_and_raw_records_are_skipped() { - let valid = format_event_key( - AGENT, 100, VM_ID, "event", "valid", EVENT_ID, TIMESTAMP, - ); - let events = decode_entries(vec![ - ("PROVISIONING_REPORT".into(), "result=success".into()), - ("a|not-a-boot|vm|event|name|id|timestamp".into(), "x".into()), - (valid, "message".into()), - ]); - assert_eq!(events.len(), 1); - assert_eq!(events[0].message, "message"); - } - - #[test] - fn indexed_chunks_group_globally_and_preserve_first_seen_order() { - let first = format_event_key( - AGENT, 100, VM_ID, "event", "first", "id-1", TIMESTAMP, - ); - let second = format_event_key( - AGENT, 100, VM_ID, "event", "second", "id-2", TIMESTAMP, - ); - let events = decode_entries(vec![ - (format!("{first}|1"), "b".into()), - (format!("{second}|0"), "second".into()), - (format!("{first}|0"), "a".into()), - ]); - assert_eq!(events.len(), 2); - assert_eq!(events[0].name, "first"); - assert_eq!(events[0].message, "ab"); - assert_eq!(events[1].name, "second"); - } - - #[rstest] - #[case(vec![(0, "a"), (2, "c")])] - #[case(vec![(0, "a"), (0, "duplicate")])] - #[case(vec![(1, "b")])] - fn incomplete_or_duplicate_indices_are_skipped( - #[case] chunks: Vec<(u32, &str)>, - ) { - let key = format_event_key( - AGENT, 100, VM_ID, "event", "name", EVENT_ID, TIMESTAMP, - ); - let dumped = chunks - .into_iter() - .map(|(index, value)| (format!("{key}|{index}"), value.into())) - .collect(); - assert!(decode_entries(dumped).is_empty()); - } - - #[test] - fn cloud_init_msg_i_must_match_key_index() { - let base = "CLOUD_INIT|100|event|name|vm-id|event-id"; - let value = r#"{"name":"name","type":"event","ts":"2026-07-27T21:33:24Z","msg_i":1,"msg":"value"}"#; - assert!(decode_entries(vec![(format!("{base}|0"), value.into())]) - .is_empty()); - } - - #[test] - fn cloud_init_split_escape_is_unescaped_after_reassembly() { - let base = "CLOUD_INIT|100|finish|name|vm-id|event-id"; - let events = decode_entries(vec![ - ( - format!("{base}|0"), - r#"{"name":"name","type":"finish","ts":"2026-07-27T21:33:24Z","msg_i":0,"msg":"line1\"}"# - .into(), - ), - ( - format!("{base}|1"), - r#"{"name":"name","type":"finish","ts":"2026-07-27T21:33:24Z","msg_i":1,"msg":"nline2"}"# - .into(), - ), - ]); - assert_eq!(events.len(), 1); - assert_eq!(events[0].message, "line1\nline2"); - } -} diff --git a/libazureinit-kvp/src/diagnostics/cloud_init.rs b/libazureinit-kvp/src/diagnostics/cloud_init.rs new file mode 100644 index 00000000..0c51a976 --- /dev/null +++ b/libazureinit-kvp/src/diagnostics/cloud_init.rs @@ -0,0 +1,733 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +use std::io::Read; + +use base64::{engine::general_purpose::STANDARD, Engine as _}; +use chrono::{DateTime, Utc}; +use flate2::bufread::{GzDecoder, ZlibDecoder}; +use serde_json::{Number, Value}; +use uuid::Uuid; + +use super::diagnostic::{ + DecodeError, Diagnostic, DiagnosticEvent, DiagnosticFinish, DiagnosticKey, + DiagnosticPayload, DiagnosticStart, Encoding, Outcome, +}; + +pub(super) const PREFIX: &str = "CLOUD_INIT"; + +struct CloudInitKey<'a> { + kind: &'a str, + name: &'a str, + vm_id: Option<&'a str>, + event_id: &'a str, +} + +pub(super) fn split_key(key: &str) -> Result<(&str, Option), DecodeError> { + let fields: Vec<_> = key.split('|').collect(); + let indexed = match fields.len() { + 5 => false, + 6 => Uuid::parse_str(fields[5]).is_err(), + 7 => true, + _ => return Err(DecodeError::Malformed), + }; + let (base, index) = if indexed { + let (base, index) = + key.rsplit_once('|').ok_or(DecodeError::Malformed)?; + (base, Some(super::parse_unsigned(index)?)) + } else { + (key, None) + }; + parse_key(base)?; + Ok((base, index)) +} + +fn parse_key(base: &str) -> Result, DecodeError> { + let fields: Vec<_> = base.split('|').collect(); + let (prefix, incarnation, kind, name, vm_id, event_id) = + match fields.as_slice() { + [prefix, incarnation, kind, name, event_id] => { + (*prefix, *incarnation, *kind, *name, None, *event_id) + } + [prefix, incarnation, kind, name, vm_id, event_id] => { + (*prefix, *incarnation, *kind, *name, Some(*vm_id), *event_id) + } + _ => return Err(DecodeError::Malformed), + }; + if prefix != PREFIX + || base.contains('\0') + || fields.iter().any(|field| field.is_empty()) + || !incarnation.bytes().all(|byte| byte.is_ascii_digit()) + { + return Err(DecodeError::Malformed); + } + if let Some(vm_id) = vm_id { + Uuid::parse_str(vm_id).map_err(|_| DecodeError::Malformed)?; + } + Uuid::parse_str(event_id).map_err(|_| DecodeError::Malformed)?; + Ok(CloudInitKey { + kind, + name, + vm_id, + event_id, + }) +} + +pub(super) fn decode_single( + base: &str, + value: &str, +) -> Result { + let key = parse_key(base)?; + let metadata: Value = + serde_json::from_str(value).map_err(|_| DecodeError::Malformed)?; + if metadata.get("msg_i").is_some() { + return Err(DecodeError::Malformed); + } + let message = metadata + .get("msg") + .and_then(Value::as_str) + .ok_or(DecodeError::Malformed)?; + diagnostic(key, &metadata, message.to_owned()) +} + +/// Chunk indices are already ordered and checked by the reader. +pub(super) fn decode_chunks<'a>( + base: &str, + chunks: impl Iterator, +) -> Result { + let key = parse_key(base)?; + let mut metadata = None; + let mut escaped_message = String::new(); + for (index, value) in chunks { + let (mut current, fragment) = chunk_parts(value)?; + if current.get("msg_i").and_then(Value::as_u64) != Some(index) { + return Err(DecodeError::Malformed); + } + current + .as_object_mut() + .ok_or(DecodeError::Malformed)? + .remove("msg_i"); + if metadata.as_ref().is_some_and(|first| first != ¤t) { + return Err(DecodeError::Malformed); + } + metadata.get_or_insert(current); + escaped_message.push_str(fragment); + } + let message = serde_json::from_str(&format!("\"{escaped_message}\"")) + .map_err(|_| DecodeError::Malformed)?; + diagnostic(key, &metadata.ok_or(DecodeError::Malformed)?, message) +} + +fn chunk_parts(value: &str) -> Result<(Value, &str), DecodeError> { + // cloud-init puts `msg` last and slices its escaped JSON string verbatim. + let (prefix, message) = value + .rsplit_once(",\"msg\":") + .ok_or(DecodeError::Malformed)?; + let fragment = message + .strip_prefix('"') + .and_then(|message| message.strip_suffix("\"}")) + .ok_or(DecodeError::Malformed)?; + let metadata = serde_json::from_str(&format!("{prefix}}}")) + .map_err(|_| DecodeError::Malformed)?; + Ok((metadata, fragment)) +} + +fn diagnostic( + source: CloudInitKey<'_>, + metadata: &Value, + message: String, +) -> Result { + if metadata.get("name").and_then(Value::as_str) != Some(source.name) + || metadata.get("type").and_then(Value::as_str) != Some(source.kind) + { + return Err(DecodeError::Malformed); + } + let timestamp = metadata + .get("ts") + .and_then(Value::as_str) + .ok_or(DecodeError::Malformed)?; + let timestamp = DateTime::parse_from_rfc3339(timestamp) + .map_err(|_| DecodeError::Malformed)? + .with_timezone(&Utc); + let (payload, encoding) = + decode_message(message, source.kind == "compressed")?; + let key = DiagnosticKey { + agent: PREFIX.to_owned(), + vm_id: source.vm_id.map(str::to_owned), + name: source.name.to_owned(), + event_id: source.event_id.to_owned(), + timestamp, + encoding, + }; + match source.kind { + "start" => { + if metadata.get("result").is_some() + || metadata.get("duration").is_some() + { + return Err(DecodeError::Malformed); + } + Ok(Diagnostic::Start(DiagnosticStart { key, payload })) + } + "finish" => { + let result = match metadata.get("result").and_then(Value::as_str) { + Some("SUCCESS") => Outcome::Success, + Some("FAIL") => Outcome::Failure, + _ => return Err(DecodeError::Malformed), + }; + let duration = metadata + .get("duration") + .and_then(Value::as_number) + .ok_or(DecodeError::Malformed)?; + Ok(Diagnostic::Finish(DiagnosticFinish { + key, + payload, + result, + duration_ms: duration_ms(duration)?, + })) + } + _ => Ok(Diagnostic::Event(DiagnosticEvent { + key, + payload, + result: None, + duration_ms: None, + })), + } +} + +fn duration_ms(seconds: &Number) -> Result { + if let Some(seconds) = seconds.as_u64() { + return seconds.checked_mul(1000).ok_or(DecodeError::Malformed); + } + let millis = seconds.as_f64().ok_or(DecodeError::Malformed)? * 1000.0; + // The exclusive upper bound avoids a saturating float-to-integer cast. + if !(0.0..u64::MAX as f64).contains(&millis) { + return Err(DecodeError::Malformed); + } + Ok(millis as u64) +} + +fn decode_message( + message: String, + compressed: bool, +) -> Result<(DiagnosticPayload, Option), DecodeError> { + let envelope = serde_json::from_str::(&message).ok(); + let is_envelope = envelope.as_ref().is_some_and(|value| { + value.get("encoding").is_some() && value.get("data").is_some() + }); + if !compressed && !is_envelope { + return Ok((DiagnosticPayload::Text(message), None)); + } + let envelope = envelope.ok_or(DecodeError::Malformed)?; + let encoding = envelope + .get("encoding") + .and_then(Value::as_str) + .ok_or(DecodeError::Malformed)?; + if encoding != "gz+b64" { + return Err(DecodeError::Undecodable); + } + let data = envelope + .get("data") + .and_then(Value::as_str) + .ok_or(DecodeError::Malformed)?; + Ok((decode_compressed(data)?, Some(Encoding::GzB64))) +} + +fn decode_compressed(data: &str) -> Result { + let compact: Vec<_> = data + .bytes() + .filter(|byte| !byte.is_ascii_whitespace()) + .collect(); + let compressed = STANDARD + .decode(compact) + .map_err(|_| DecodeError::Undecodable)?; + let mut bytes = Vec::new(); + // cloud-init labels zlib streams and line-wrapped base64 as `gz+b64`. + let remaining = if compressed.starts_with(&[0x1f, 0x8b]) { + let mut decoder = GzDecoder::new(compressed.as_slice()); + decoder + .read_to_end(&mut bytes) + .map_err(|_| DecodeError::Undecodable)?; + decoder.into_inner() + } else { + let mut decoder = ZlibDecoder::new(compressed.as_slice()); + decoder + .read_to_end(&mut bytes) + .map_err(|_| DecodeError::Undecodable)?; + decoder.into_inner() + }; + if !remaining.is_empty() { + return Err(DecodeError::Undecodable); + } + Ok(DiagnosticPayload::Bytes(bytes)) +} + +#[cfg(test)] +mod tests { + use super::*; + use std::fs; + + use rstest::rstest; + use serde_json::json; + use tempfile::TempDir; + + use super::super::diagnostic::{Entry, Kind, RawKeyValue}; + use super::super::encoding::decode_payload; + use super::super::reader::DiagnosticReader; + use crate::{KvpPool, KvpPoolStore, PoolMode}; + + const VM_ID: &str = "0e5e179d-5341-478b-8456-fbb90621bdf8"; + const EVENT_ID: &str = "b7a822ba-4eea-46c0-b559-e84396101132"; + const TIMESTAMP: &str = "2026-07-27T23:33:24.339006+02:00"; + const ZLIB_DATA: &str = "eJxLzskvTdHNzMss4WL4DwAi1AUC"; + const GZIP_DATA: &str = "H4sIAAAAAAAC/0vOyS9N0c3MyyzhYvgPACZ1n10NAAAA"; + + fn key(kind: &str, current: bool, index: Option) -> String { + let identity = if current { + format!("{VM_ID}|{EVENT_ID}") + } else { + EVENT_ID.to_owned() + }; + let base = format!("CLOUD_INIT|100|{kind}|test|{identity}"); + match index { + Some(index) => format!("{base}|{index}"), + None => base, + } + } + + fn value(kind: &str, message: &str) -> Value { + json!({"name": "test", "type": kind, "ts": TIMESTAMP, "msg": message}) + } + + fn chunk(kind: &str, index: u64, fragment: &str) -> String { + format!( + r#"{{"name":"test","type":"{kind}","ts":"{TIMESTAMP}","msg_i":{index},"msg":"{fragment}"}}"# + ) + } + + fn entries(records: &[(String, String)]) -> Vec { + let dir = TempDir::new().unwrap(); + let store = + KvpPoolStore::new_in(KvpPool::Guest, dir.path(), PoolMode::Safe) + .unwrap(); + store.append_multiple(records.iter().cloned()).unwrap(); + let before = fs::read(store.path()).unwrap(); + let entries = DiagnosticReader::new(store.clone()).entries().unwrap(); + assert_eq!(fs::read(store.path()).unwrap(), before); + entries + } + + fn only_diagnostic(entries: Vec) -> Diagnostic { + assert_eq!(entries.len(), 1); + match entries.into_iter().next().unwrap() { + Entry::Diagnostic(diagnostic) => diagnostic, + other => panic!("expected a diagnostic, got {other:?}"), + } + } + + fn assert_raw(records: &[(String, String)], error: DecodeError) { + let expected: Vec<_> = records + .iter() + .map(|(key, value)| { + Entry::Raw(RawKeyValue { + key: key.clone(), + value: value.clone(), + error: Some(error), + }) + }) + .collect(); + assert_eq!(entries(records), expected); + } + + #[rstest] + #[case::old_single(false, None)] + #[case::current_single(true, None)] + #[case::old_chunk(false, Some(0))] + #[case::current_chunk(true, Some(17))] + fn key_layouts_identify_chunk_suffixes( + #[case] current: bool, + #[case] index: Option, + ) { + let key = key("event", current, index); + let (base, parsed_index) = split_key(&key).unwrap(); + let parsed = parse_key(base).unwrap(); + assert_eq!(parsed_index, index); + assert_eq!(parsed.vm_id, current.then_some(VM_ID)); + assert_eq!(parsed.event_id, EVENT_ID); + } + + #[rstest] + #[case::layout("CLOUD_INIT|100|event".into())] + #[case::vm(key("event", true, None).replace(VM_ID, "invalid"))] + #[case::incarnation(key("event", false, None).replace("|100|", "|bad|"))] + #[case::index(format!("{}|-1", key("event", true, None)))] + fn malformed_cloud_keys_remain_raw(#[case] key: String) { + assert_raw( + &[(key, value("event", "message").to_string())], + DecodeError::Malformed, + ); + } + + #[rstest] + #[case::older(false)] + #[case::current(true)] + fn source_identity_and_timestamp_are_mapped(#[case] current: bool) { + let diagnostic = only_diagnostic(entries(&[( + key("event", current, None), + value("event", "message").to_string(), + )])); + assert_eq!(diagnostic.key().agent, "CLOUD_INIT"); + assert_eq!(diagnostic.key().vm_id.as_deref(), current.then_some(VM_ID)); + assert_eq!(diagnostic.key().name, "test"); + assert_eq!(diagnostic.key().event_id, EVENT_ID); + let rendered = serde_json::to_value(diagnostic).unwrap(); + assert_eq!(rendered["timestamp"], "2026-07-27T21:33:24.339Z"); + assert!(rendered.get("boot_epoch").is_none()); + assert!(rendered.get("diagnostic_version_id").is_none()); + } + + #[rstest] + #[case::start("start", Kind::Start)] + #[case::event("event", Kind::Event)] + #[case::system_info("system-info", Kind::Event)] + #[case::warning("warning", Kind::Event)] + fn source_type_determines_timeline_kind( + #[case] source: &str, + #[case] kind: Kind, + ) { + let diagnostic = only_diagnostic(entries(&[( + key(source, true, None), + value(source, "observed").to_string(), + )])); + assert_eq!(diagnostic.kind(), kind); + assert_eq!( + diagnostic.payload(), + &DiagnosticPayload::Text("observed".into()) + ); + } + + #[test] + fn finish_closes_its_span_with_mapped_outcome_and_milliseconds() { + let mut finish = value("finish", "finished with failure"); + finish["result"] = json!("FAIL"); + finish["duration"] = json!(0.1234); + let entries = entries(&[ + ( + key("start", true, None), + value("start", "starting").to_string(), + ), + (key("finish", true, None), finish.to_string()), + ]); + let [Entry::Diagnostic(Diagnostic::Start(start)), Entry::Diagnostic(Diagnostic::Finish(finish))] = + entries.as_slice() + else { + panic!("expected a start and finish"); + }; + assert_eq!(start.key.event_id, finish.key.event_id); + assert_eq!(finish.result, Outcome::Failure); + assert_eq!(finish.duration_ms, 123); + } + + #[test] + fn captured_successful_finish_decodes() { + let records = [( + "CLOUD_INIT|1785187982|finish|modules-final/config-scripts_user|0e5e179d-5341-478b-8456-fbb90621bdf8|e5f01809-a7a3-4279-aa64-1f18e21eda6e".into(), + r#"{"name":"modules-final/config-scripts_user","type":"finish","ts":"2026-07-27T21:33:24.339006+00:00","result":"SUCCESS","duration":0.0006448590000012189,"msg":"config-scripts_user ran successfully and took 0.001 seconds"}"#.into(), + )]; + let Diagnostic::Finish(finish) = only_diagnostic(entries(&records)) + else { + panic!("expected a finish"); + }; + assert_eq!(finish.result, Outcome::Success); + assert_eq!(finish.duration_ms, 0); + assert_eq!(finish.key.name, "modules-final/config-scripts_user"); + } + + #[test] + fn warn_finish_is_preserved_without_reclassifying_it_as_an_event() { + let mut warning = value("finish", "completed with warnings"); + warning["result"] = json!("WARN"); + warning["duration"] = json!(1); + let raw = RawKeyValue { + key: key("finish", true, None), + value: warning.to_string(), + error: Some(DecodeError::Malformed), + }; + let entries = entries(&[ + ( + key("start", true, None), + value("start", "starting").to_string(), + ), + (raw.key.clone(), raw.value.clone()), + ]); + assert!(matches!( + &entries[0], + Entry::Diagnostic(Diagnostic::Start(_)) + )); + assert_eq!(entries[1], Entry::Raw(raw)); + } + + #[rstest] + #[case::result("result")] + #[case::duration("duration")] + fn finish_requires_result_and_duration(#[case] missing: &str) { + let mut metadata = value("finish", "finished"); + metadata["result"] = json!("SUCCESS"); + metadata["duration"] = json!(1); + metadata.as_object_mut().unwrap().remove(missing); + assert_raw( + &[(key("finish", true, None), metadata.to_string())], + DecodeError::Malformed, + ); + } + + #[rstest] + #[case::invalid_json("not json".into())] + #[case::timestamp(value("event", "message").to_string().replace(TIMESTAMP, "bad"))] + #[case::name(value("event", "message").to_string().replace("\"test\"", "\"other\""))] + #[case::message(json!({"name":"test", "type":"event", "ts":TIMESTAMP, "msg":7}).to_string())] + fn malformed_source_values_remain_raw(#[case] value: String) { + assert_raw( + &[(key("event", true, None), value)], + DecodeError::Malformed, + ); + } + + #[test] + fn chunk_without_a_key_index_is_malformed() { + assert_raw( + &[(key("event", false, None), chunk("event", 0, "partial"))], + DecodeError::Malformed, + ); + } + + #[rstest] + #[case::newline(r#"line1\"#, "nline2", "line1\nline2")] + #[case::unicode(r"\ud83", r"d\ude00", "😀")] + fn escaped_fragments_are_unescaped_only_after_reassembly( + #[case] first: &str, + #[case] second: &str, + #[case] expected: &str, + ) { + let first = chunk("event", 0, first); + assert!(serde_json::from_str::(&first).is_err()); + let diagnostic = only_diagnostic(entries(&[ + (key("event", true, Some(1)), chunk("event", 1, second)), + (key("event", true, Some(0)), first), + ])); + assert_eq!( + diagnostic.payload(), + &DiagnosticPayload::Text(expected.into()) + ); + } + + #[rstest] + #[case::index(chunk("event", 0, "second"))] + #[case::metadata(chunk("event", 1, "second").replace(TIMESTAMP, "2026-07-27T21:34:00Z"))] + #[case::escape(chunk("event", 1, r"\x"))] + fn invalid_chunk_values_preserve_the_entire_group(#[case] second: String) { + assert_raw( + &[ + (key("event", true, Some(1)), second), + (key("event", true, Some(0)), chunk("event", 0, "first")), + ], + DecodeError::Malformed, + ); + } + + #[rstest] + #[case::gap(2, DecodeError::IncompleteGroup)] + #[case::duplicate(0, DecodeError::DuplicateChunk)] + fn cloud_groups_use_shared_index_validation( + #[case] second: u64, + #[case] error: DecodeError, + ) { + assert_raw( + &[ + (key("event", false, Some(0)), chunk("event", 0, "first")), + ( + key("event", false, Some(second)), + chunk("event", second, "second"), + ), + ], + error, + ); + } + + #[test] + fn incarnation_keeps_otherwise_identical_groups_separate() { + let records = [ + (key("event", false, Some(0)), chunk("event", 0, "a")), + ( + key("event", false, Some(0)).replace("|100|", "|101|"), + chunk("event", 0, "x"), + ), + (key("event", false, Some(1)), chunk("event", 1, "b")), + ( + key("event", false, Some(1)).replace("|100|", "|101|"), + chunk("event", 1, "y"), + ), + ]; + let entries = entries(&records); + let [Entry::Diagnostic(first), Entry::Diagnostic(second)] = + entries.as_slice() + else { + panic!("expected two separate groups"); + }; + assert_eq!(first.key(), second.key()); + assert_eq!(first.payload(), &DiagnosticPayload::Text("ab".into())); + assert_eq!(second.payload(), &DiagnosticPayload::Text("xy".into())); + } + + #[test] + fn mixed_sources_keep_first_seen_order_and_unrelated_raw_records() { + let v1 = format!("DIAG_V1|azure-init|{VM_ID}|event|test|{EVENT_ID}|2026-07-27T21:33:00.000Z|none|||0"); + let entries = entries(&[ + (key("event", true, Some(1)), chunk("event", 1, "b")), + ("unrelated".into(), "value".into()), + (v1, "v1 message".into()), + (key("event", true, Some(0)), chunk("event", 0, "a")), + ]); + assert_eq!(entries.len(), 3); + assert!( + matches!(&entries[0], Entry::Diagnostic(d) if d.key().agent == PREFIX) + ); + assert!( + matches!(&entries[1], Entry::Raw(raw) if raw.key == "unrelated" && raw.error.is_none()) + ); + assert!( + matches!(&entries[2], Entry::Diagnostic(d) if d.key().agent == "azure-init") + ); + } + + #[test] + fn cloud_names_are_not_capped_by_writer_budgets() { + let name = "long-subject".repeat(8); + let mut metadata = value("system-info", "message"); + metadata["name"] = json!(name); + let diagnostic = only_diagnostic(entries(&[( + key("system-info", true, None) + .replace("|test|", &format!("|{name}|")), + metadata.to_string(), + )])); + assert_eq!(diagnostic.key().name, name); + } + + #[rstest] + #[case::whole_seconds(json!(2), 2000)] + #[case::zero(json!(0), 0)] + #[case::fraction(json!(0.1234), 123)] + #[case::sub_millisecond(json!(0.00064), 0)] + fn duration_conversion_truncates_to_milliseconds( + #[case] seconds: Value, + #[case] expected: u64, + ) { + assert_eq!( + duration_ms(seconds.as_number().unwrap()).unwrap(), + expected + ); + } + + #[rstest] + #[case::negative(json!(-1))] + #[case::integer_overflow(json!(u64::MAX))] + #[case::float_overflow(json!(1e30))] + fn duration_conversion_rejects_invalid_ranges(#[case] seconds: Value) { + assert_eq!( + duration_ms(seconds.as_number().unwrap()), + Err(DecodeError::Malformed) + ); + } + + #[rstest] + #[case::zlib(ZLIB_DATA)] + #[case::gzip(GZIP_DATA)] + fn compressed_envelopes_decode_arbitrary_bytes(#[case] data: &str) { + let message = json!({"encoding": "gz+b64", "data": data}).to_string(); + let diagnostic = only_diagnostic(entries(&[( + key("compressed", true, None), + value("compressed", &message).to_string(), + )])); + assert_eq!(diagnostic.kind(), Kind::Event); + assert_eq!(diagnostic.key().encoding, Some(Encoding::GzB64)); + assert_eq!( + diagnostic.payload(), + &DiagnosticPayload::Bytes(b"cloud-init\n\x00\xff".to_vec()) + ); + } + + #[test] + fn all_truncated_zlib_prefixes_are_undecodable() { + let compressed = STANDARD.decode(ZLIB_DATA).unwrap(); + for end in 0..compressed.len() { + assert_eq!( + decode_compressed(&STANDARD.encode(&compressed[..end])), + Err(DecodeError::Undecodable), + "prefix {end}" + ); + } + } + + #[test] + fn zlib_checksum_and_trailing_data_are_validated() { + let compressed = STANDARD.decode(ZLIB_DATA).unwrap(); + let mut corrupt = compressed.clone(); + *corrupt.last_mut().unwrap() ^= 1; + let mut trailing = compressed; + trailing.push(0); + for bytes in [corrupt, trailing] { + assert_eq!( + decode_compressed(&STANDARD.encode(bytes)), + Err(DecodeError::Undecodable) + ); + } + } + + #[test] + fn compatibility_does_not_relax_v1_gzip_validation() { + assert_eq!( + decode_payload(ZLIB_DATA.as_bytes(), Some(&Encoding::GzB64)), + Err(DecodeError::Undecodable) + ); + } + + #[rstest] + #[case::standalone_base64("b64")] + #[case::unknown("zstd")] + fn other_encoding_labels_are_not_supported(#[case] encoding: &str) { + let message = + json!({"encoding": encoding, "data": GZIP_DATA}).to_string(); + assert_raw( + &[( + key("compressed", true, None), + value("compressed", &message).to_string(), + )], + DecodeError::Undecodable, + ); + } + + #[test] + fn compressed_envelope_requires_string_data() { + let message = json!({"encoding": "gz+b64", "data": 7}).to_string(); + assert_raw( + &[( + key("compressed", true, None), + value("compressed", &message).to_string(), + )], + DecodeError::Malformed, + ); + } + + #[test] + fn ordinary_json_messages_are_not_assumed_to_be_encoded() { + let message = r#"{"encoding":"utf-8","description":"ordinary JSON"}"#; + let diagnostic = only_diagnostic(entries(&[( + key("event", false, None), + value("event", message).to_string(), + )])); + assert_eq!( + diagnostic.payload(), + &DiagnosticPayload::Text(message.into()) + ); + assert_eq!(diagnostic.key().encoding, None); + } +} diff --git a/libazureinit-kvp/src/diagnostics/diagnostic.rs b/libazureinit-kvp/src/diagnostics/diagnostic.rs new file mode 100644 index 00000000..9a3f6219 --- /dev/null +++ b/libazureinit-kvp/src/diagnostics/diagnostic.rs @@ -0,0 +1,587 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +use std::fmt; + +use base64::{engine::general_purpose::STANDARD, Engine as _}; +use chrono::{DateTime, SecondsFormat, Utc}; +use serde::ser::SerializeStruct; +use serde::{Serialize, Serializer}; + +use crate::ProvisioningReport; + +pub const DIAGNOSTIC_VERSION_ID: &str = "DIAG_V1"; + +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize)] +#[serde(rename_all = "lowercase")] +pub enum Kind { + Start, + Finish, + Event, +} + +impl fmt::Display for Kind { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.write_str(match self { + Self::Start => "start", + Self::Finish => "finish", + Self::Event => "event", + }) + } +} + +/// Unencoded text uses `None` in `DiagnosticKey::encoding`. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum Encoding { + GzB64, + Other(String), +} + +impl fmt::Display for Encoding { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.write_str(match self { + Self::GzB64 => "gz+b64", + Self::Other(token) => token, + }) + } +} + +impl Serialize for Encoding { + fn serialize(&self, serializer: S) -> Result + where + S: Serializer, + { + serializer.collect_str(self) + } +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize)] +#[serde(rename_all = "lowercase")] +pub enum Outcome { + Success, + #[serde(rename = "fail")] + Failure, +} + +impl fmt::Display for Outcome { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.write_str(match self { + Self::Success => "success", + Self::Failure => "fail", + }) + } +} + +/// Decoded bytes retain their type even when they contain valid UTF-8. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum DiagnosticPayload { + Text(String), + Bytes(Vec), +} + +impl From<&str> for DiagnosticPayload { + fn from(value: &str) -> Self { + Self::Text(value.to_owned()) + } +} + +impl From for DiagnosticPayload { + fn from(value: String) -> Self { + Self::Text(value) + } +} + +impl From<&[u8]> for DiagnosticPayload { + fn from(value: &[u8]) -> Self { + Self::Bytes(value.to_vec()) + } +} + +impl From> for DiagnosticPayload { + fn from(value: Vec) -> Self { + Self::Bytes(value) + } +} + +impl Serialize for DiagnosticPayload { + fn serialize(&self, serializer: S) -> Result + where + S: Serializer, + { + match self { + Self::Text(text) => serializer.serialize_str(text), + Self::Bytes(bytes) => { + let mut payload = + serializer.serialize_struct("DiagnosticPayload", 3)?; + payload.serialize_field("type", "bytes")?; + payload.serialize_field("encoding", "base64")?; + payload.serialize_field("data", &STANDARD.encode(bytes))?; + payload.end() + } + } + } +} + +/// Describes uninterpretable stored data, not a failed I/O operation. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize)] +#[serde(rename_all = "snake_case")] +pub enum DecodeError { + UnsupportedVersion, + IncompleteGroup, + DuplicateChunk, + Undecodable, + Malformed, +} + +impl fmt::Display for DecodeError { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.write_str(match self { + Self::UnsupportedVersion => "unsupported diagnostic schema version", + Self::IncompleteGroup => { + "diagnostic chunks are not contiguous from index 0" + } + Self::DuplicateChunk => "duplicate diagnostic chunk index", + Self::Undecodable => "diagnostic payload could not be decoded", + Self::Malformed => "malformed diagnostic or provisioning report", + }) + } +} + +impl std::error::Error for DecodeError {} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize)] +pub struct DiagnosticKey { + pub agent: String, + /// Older cloud-init records do not include a VM identity. + pub vm_id: Option, + pub name: String, + /// Span endpoints share this ID; standalone events have their own. + pub event_id: String, + #[serde(serialize_with = "serialize_timestamp")] + pub timestamp: DateTime, + #[serde(serialize_with = "serialize_encoding")] + pub encoding: Option, +} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize)] +pub struct DiagnosticStart { + #[serde(flatten)] + pub key: DiagnosticKey, + pub payload: DiagnosticPayload, +} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize)] +pub struct DiagnosticFinish { + #[serde(flatten)] + pub key: DiagnosticKey, + pub payload: DiagnosticPayload, + pub result: Outcome, + #[serde(rename = "duration")] + pub duration_ms: u64, +} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize)] +pub struct DiagnosticEvent { + #[serde(flatten)] + pub key: DiagnosticKey, + pub payload: DiagnosticPayload, + #[serde(skip_serializing_if = "Option::is_none")] + pub result: Option, + #[serde(rename = "duration", skip_serializing_if = "Option::is_none")] + pub duration_ms: Option, +} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize)] +#[serde(tag = "kind", rename_all = "lowercase")] +pub enum Diagnostic { + Start(DiagnosticStart), + Finish(DiagnosticFinish), + Event(DiagnosticEvent), +} + +impl Diagnostic { + pub fn key(&self) -> &DiagnosticKey { + match self { + Self::Start(start) => &start.key, + Self::Finish(finish) => &finish.key, + Self::Event(event) => &event.key, + } + } + + pub fn kind(&self) -> Kind { + match self { + Self::Start(_) => Kind::Start, + Self::Finish(_) => Kind::Finish, + Self::Event(_) => Kind::Event, + } + } + + pub fn payload(&self) -> &DiagnosticPayload { + match self { + Self::Start(start) => &start.payload, + Self::Finish(finish) => &finish.payload, + Self::Event(event) => &event.payload, + } + } +} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize)] +pub struct RawKeyValue { + pub key: String, + pub value: String, + #[serde(skip_serializing_if = "Option::is_none")] + pub error: Option, +} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize)] +#[serde(tag = "type")] +pub enum Entry { + #[serde(rename = "diagnostic")] + Diagnostic(Diagnostic), + #[serde(rename = "PROVISIONING_REPORT")] + Report(ProvisioningReport), + #[serde(rename = "raw")] + Raw(RawKeyValue), +} + +fn serialize_timestamp( + timestamp: &DateTime, + serializer: S, +) -> Result +where + S: Serializer, +{ + serializer + .serialize_str(×tamp.to_rfc3339_opts(SecondsFormat::Millis, true)) +} + +fn serialize_encoding( + encoding: &Option, + serializer: S, +) -> Result +where + S: Serializer, +{ + match encoding { + Some(encoding) => encoding.serialize(serializer), + None => serializer.serialize_str("none"), + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::ReportPpsType; + use rstest::rstest; + use serde_json::{json, Value}; + + const AGENT: &str = "azure-init-0.1.1"; + const VM_ID: &str = "3f2504e0-4f89-41d3-9a0c-0305e82c3301"; + const EVENT_ID: &str = "8f3e9c4a-1b2c-4d5e-9f01-234567890abc"; + const TIMESTAMP: &str = "2026-08-31T12:34:56.789Z"; + + fn key() -> DiagnosticKey { + DiagnosticKey { + agent: AGENT.into(), + vm_id: Some(VM_ID.into()), + name: "provision:run".into(), + event_id: EVENT_ID.into(), + timestamp: DateTime::parse_from_rfc3339(TIMESTAMP) + .unwrap() + .with_timezone(&Utc), + encoding: None, + } + } + + fn expected_diagnostic(kind: &str, payload: Value) -> Value { + json!({ + "type": "diagnostic", + "kind": kind, + "agent": AGENT, + "vm_id": VM_ID, + "name": "provision:run", + "event_id": EVENT_ID, + "timestamp": TIMESTAMP, + "encoding": "none", + "payload": payload, + }) + } + + #[rstest] + #[case(Kind::Start, "start")] + #[case(Kind::Finish, "finish")] + #[case(Kind::Event, "event")] + fn kind_uses_wire_token(#[case] kind: Kind, #[case] token: &str) { + assert_eq!(kind.to_string(), token); + assert_eq!(serde_json::to_value(kind).unwrap(), token); + } + + #[rstest] + #[case(Encoding::GzB64, "gz+b64")] + #[case(Encoding::Other("zstd+b64".into()), "zstd+b64")] + #[case(Encoding::Other("".into()), "")] + fn encoding_preserves_token( + #[case] encoding: Encoding, + #[case] token: &str, + ) { + assert_eq!(encoding.to_string(), token); + assert_eq!(serde_json::to_value(encoding).unwrap(), token); + } + + #[rstest] + #[case(Outcome::Success, "success")] + #[case(Outcome::Failure, "fail")] + fn outcome_uses_wire_token(#[case] result: Outcome, #[case] token: &str) { + assert_eq!(result.to_string(), token); + assert_eq!(serde_json::to_value(result).unwrap(), token); + } + + #[rstest] + #[case(DecodeError::UnsupportedVersion, "unsupported_version")] + #[case(DecodeError::IncompleteGroup, "incomplete_group")] + #[case(DecodeError::DuplicateChunk, "duplicate_chunk")] + #[case(DecodeError::Undecodable, "undecodable")] + #[case(DecodeError::Malformed, "malformed")] + fn decode_error_has_serializable_reason( + #[case] reason: DecodeError, + #[case] token: &str, + ) { + assert_eq!(serde_json::to_value(reason).unwrap(), token); + let error: &dyn std::error::Error = &reason; + assert!(!error.to_string().is_empty()); + assert!(error.source().is_none()); + } + + #[test] + fn string_conversions_preserve_text() { + let text = "héllo\n\"world\"\u{0}"; + let expected = DiagnosticPayload::Text(text.to_owned()); + assert_eq!(DiagnosticPayload::from(text), expected); + assert_eq!(DiagnosticPayload::from(text.to_owned()), expected); + } + + #[rstest] + #[case(b"")] + #[case(b"hello")] + #[case(&[0, 255, 128, 0])] + fn byte_inputs_remain_bytes(#[case] bytes: &[u8]) { + let expected = DiagnosticPayload::Bytes(bytes.to_vec()); + assert_eq!(DiagnosticPayload::from(bytes), expected); + assert_eq!(DiagnosticPayload::from(bytes.to_vec()), expected); + } + + #[rstest] + #[case("")] + #[case("héllo\n\"world\"\u{0}")] + fn text_serializes_as_a_string(#[case] text: &str) { + let payload = DiagnosticPayload::Text(text.to_owned()); + assert_eq!(serde_json::to_value(payload).unwrap(), json!(text)); + } + + #[rstest] + #[case(b"", "")] + #[case(b"hello", "aGVsbG8=")] + #[case(&[0], "AA==")] + #[case(&[251, 255], "+/8=")] + fn bytes_serialize_as_standard_base64( + #[case] bytes: &[u8], + #[case] encoded: &str, + ) { + assert_eq!( + serde_json::to_value(DiagnosticPayload::Bytes(bytes.to_vec())) + .unwrap(), + json!({"type": "bytes", "encoding": "base64", "data": encoded}) + ); + } + + #[test] + fn start_serializes_without_result_or_duration() { + let entry = Entry::Diagnostic(Diagnostic::Start(DiagnosticStart { + key: key(), + payload: "starting".into(), + })); + assert_eq!( + serde_json::to_value(entry).unwrap(), + expected_diagnostic("start", json!("starting")) + ); + } + + #[rstest] + #[case(Outcome::Success, "success", 312)] + #[case(Outcome::Failure, "fail", 0)] + #[case(Outcome::Success, "success", u64::MAX)] + fn finish_serializes_result_and_milliseconds( + #[case] result: Outcome, + #[case] token: &str, + #[case] duration_ms: u64, + ) { + let entry = Entry::Diagnostic(Diagnostic::Finish(DiagnosticFinish { + key: key(), + payload: "finished".into(), + result, + duration_ms, + })); + let mut expected = expected_diagnostic("finish", json!("finished")); + expected["result"] = json!(token); + expected["duration"] = json!(duration_ms); + assert_eq!(serde_json::to_value(entry).unwrap(), expected); + } + + #[rstest] + #[case::neither(None, None)] + #[case::result_only(Some(Outcome::Success), None)] + #[case::zero_duration(None, Some(0))] + #[case::both(Some(Outcome::Failure), Some(52))] + fn event_serializes_only_measured_fields( + #[case] result: Option, + #[case] duration_ms: Option, + ) { + let entry = Entry::Diagnostic(Diagnostic::Event(DiagnosticEvent { + key: key(), + payload: "observed".into(), + result, + duration_ms, + })); + let mut expected = expected_diagnostic("event", json!("observed")); + if let Some(result) = result { + expected["result"] = json!(result.to_string()); + } + if let Some(duration_ms) = duration_ms { + expected["duration"] = json!(duration_ms); + } + assert_eq!(serde_json::to_value(entry).unwrap(), expected); + } + + #[test] + fn wire_encoding_is_distinct_from_payload_presentation() { + let entry = Entry::Diagnostic(Diagnostic::Event(DiagnosticEvent { + key: DiagnosticKey { + encoding: Some(Encoding::GzB64), + ..key() + }, + payload: b"hello".as_slice().into(), + result: None, + duration_ms: None, + })); + let mut expected = expected_diagnostic( + "event", + json!({ + "type": "bytes", + "encoding": "base64", + "data": "aGVsbG8=", + }), + ); + expected["encoding"] = json!("gz+b64"); + assert_eq!(serde_json::to_value(entry).unwrap(), expected); + } + + #[rstest] + #[case(None, "none")] + #[case(Some(Encoding::GzB64), "gz+b64")] + #[case(Some(Encoding::Other("zstd+b64".into())), "zstd+b64")] + fn key_always_serializes_encoding( + #[case] encoding: Option, + #[case] token: &str, + ) { + let key = DiagnosticKey { encoding, ..key() }; + assert_eq!(serde_json::to_value(key).unwrap()["encoding"], token); + } + + #[test] + fn missing_cloud_init_vm_id_serializes_as_null() { + let entry = Entry::Diagnostic(Diagnostic::Start(DiagnosticStart { + key: DiagnosticKey { + agent: "CLOUD_INIT".into(), + vm_id: None, + ..key() + }, + payload: "starting".into(), + })); + let mut expected = expected_diagnostic("start", json!("starting")); + expected["agent"] = json!("CLOUD_INIT"); + expected["vm_id"] = Value::Null; + assert_eq!(serde_json::to_value(entry).unwrap(), expected); + } + + #[rstest] + #[case("2026-08-31T12:34:56Z", "2026-08-31T12:34:56.000Z")] + #[case("2026-08-31T12:34:56.3Z", "2026-08-31T12:34:56.300Z")] + #[case("2026-08-31T12:34:56.789999Z", TIMESTAMP)] + #[case("2026-08-31T14:34:56.789+02:00", TIMESTAMP)] + fn timestamp_serializes_in_utc_milliseconds( + #[case] timestamp: &str, + #[case] expected: &str, + ) { + let key = DiagnosticKey { + timestamp: DateTime::parse_from_rfc3339(timestamp) + .unwrap() + .with_timezone(&Utc), + ..key() + }; + assert_eq!(serde_json::to_value(key).unwrap()["timestamp"], expected); + } + + #[rstest] + fn diagnostic_accessors_cover_all_kinds( + #[values(Kind::Start, Kind::Finish, Kind::Event)] kind: Kind, + ) { + let key = key(); + let payload = DiagnosticPayload::from("message"); + let diagnostic = match kind { + Kind::Start => Diagnostic::Start(DiagnosticStart { + key: key.clone(), + payload: payload.clone(), + }), + Kind::Finish => Diagnostic::Finish(DiagnosticFinish { + key: key.clone(), + payload: payload.clone(), + result: Outcome::Success, + duration_ms: 0, + }), + Kind::Event => Diagnostic::Event(DiagnosticEvent { + key: key.clone(), + payload: payload.clone(), + result: None, + duration_ms: None, + }), + }; + assert_eq!(diagnostic.key(), &key); + assert_eq!(diagnostic.kind(), kind); + assert_eq!(diagnostic.payload(), &payload); + } + + #[rstest] + #[case(None, None)] + #[case(Some(DecodeError::Malformed), Some("malformed"))] + fn raw_entry_preserves_key_value_and_optional_error( + #[case] error: Option, + #[case] token: Option<&str>, + ) { + let entry = Entry::Raw(RawKeyValue { + key: "original|key|0".into(), + value: "original \"value\"\nwith Unicode: é".into(), + error, + }); + let mut expected = json!({ + "type": "raw", + "key": "original|key|0", + "value": "original \"value\"\nwith Unicode: é", + }); + if let Some(token) = token { + expected["error"] = json!(token); + } + assert_eq!(serde_json::to_value(entry).unwrap(), expected); + } + + #[test] + fn report_entry_adds_type_without_nesting_report_fields() { + let report = + ProvisioningReport::success(AGENT, VM_ID, ReportPpsType::None); + let mut expected = serde_json::to_value(&report).unwrap(); + expected["type"] = json!("PROVISIONING_REPORT"); + assert_eq!( + serde_json::to_value(Entry::Report(report)).unwrap(), + expected + ); + } +} diff --git a/libazureinit-kvp/src/diagnostics/encoding.rs b/libazureinit-kvp/src/diagnostics/encoding.rs new file mode 100644 index 00000000..a0a6ec8d --- /dev/null +++ b/libazureinit-kvp/src/diagnostics/encoding.rs @@ -0,0 +1,354 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +use std::io::{Read, Write}; + +use base64::{engine::general_purpose::STANDARD, Engine as _}; +use flate2::{bufread::GzDecoder, write::GzEncoder, Compression}; + +use super::diagnostic::{DecodeError, DiagnosticPayload, Encoding}; +use crate::KvpError; + +/// Produces a complete wire value; the writer handles chunk framing. +pub(super) fn encode_payload( + payload: DiagnosticPayload, + encoding: Option<&Encoding>, +) -> Result { + match encoding { + None => { + let text = match payload { + DiagnosticPayload::Text(text) => text, + DiagnosticPayload::Bytes(bytes) => String::from_utf8(bytes) + .map_err(|_| KvpError::PayloadNotUtf8)?, + }; + if text.contains('\0') { + return Err(KvpError::ValueContainsNull); + } + Ok(text) + } + Some(Encoding::GzB64) => { + let bytes = match &payload { + DiagnosticPayload::Text(text) => text.as_bytes(), + DiagnosticPayload::Bytes(bytes) => bytes.as_slice(), + }; + let mut encoder = + GzEncoder::new(Vec::new(), Compression::default()); + encoder.write_all(bytes)?; + Ok(STANDARD.encode(encoder.finish()?)) + } + Some(Encoding::Other(token)) => Err(KvpError::UnsupportedEncoding { + token: token.clone(), + }), + } +} + +/// The reader must reassemble and validate chunk indices before decoding. +pub(super) fn decode_payload( + value: &[u8], + encoding: Option<&Encoding>, +) -> Result { + match encoding { + None => std::str::from_utf8(value) + .map(|text| DiagnosticPayload::Text(text.to_owned())) + .map_err(|_| DecodeError::Undecodable), + Some(Encoding::GzB64) => { + let compressed = STANDARD + .decode(value) + .map_err(|_| DecodeError::Undecodable)?; + let mut decoder = GzDecoder::new(compressed.as_slice()); + let mut bytes = Vec::new(); + decoder + .read_to_end(&mut bytes) + .map_err(|_| DecodeError::Undecodable)?; + // The buffered decoder leaves any data after the gzip member unread. + if !decoder.get_ref().is_empty() { + return Err(DecodeError::Undecodable); + } + Ok(DiagnosticPayload::Bytes(bytes)) + } + Some(Encoding::Other(_)) => Err(DecodeError::Undecodable), + } +} + +#[cfg(test)] +mod tests { + use super::*; + use rstest::rstest; + + // Python gzip fixtures keep decoding tests independent of our encoder. + const PYTHON_EMPTY: &str = "H4sIAAAAAAAC/wMAAAAAAAAAAAA="; + const PYTHON_HELLO: &str = "H4sIAAAAAAAC/8tIzcnJBwCGphA2BQAAAA=="; + const PYTHON_BINARY: &str = + "H4sIAAAAAAAC/0vOyS9N0c3MyyxRSMusKCktSuVi+A8AokCfWhUAAAA="; + const PYTHON_FILENAME: &str = + "H4sICAAAAAAC/2RtZXNnAEvOyS9N0c3MyyxRSMusKCktSuVi+A8AokCfWhUAAAA="; + + #[rstest] + #[case::empty_text("", false)] + #[case::empty_bytes("", true)] + #[case::unicode_text("héllo\n\"value\" | = 😀", false)] + #[case::unicode_bytes("héllo", true)] + #[case::not_inferred(PYTHON_HELLO, false)] + fn none_preserves_text(#[case] text: &str, #[case] as_bytes: bool) { + let payload = if as_bytes { + DiagnosticPayload::Bytes(text.as_bytes().to_vec()) + } else { + DiagnosticPayload::Text(text.to_owned()) + }; + let value = encode_payload(payload, None).unwrap(); + assert_eq!(value, text); + assert_eq!( + decode_payload(value.as_bytes(), None).unwrap(), + DiagnosticPayload::Text(text.to_owned()) + ); + } + + #[test] + fn none_keeps_large_payload_unencoded() { + let text = "é".repeat(4096); + let value = encode_payload(text.clone().into(), None).unwrap(); + assert_eq!(value, text); + assert_eq!( + decode_payload(value.as_bytes(), None).unwrap(), + DiagnosticPayload::Text(text) + ); + } + + #[rstest] + #[case(&[0xff])] + #[case(&[0xe2, 0x82])] + #[case(&[0xc0, 0xaf])] + #[case(&[0xed, 0xa0, 0x80])] + fn none_rejects_invalid_utf8(#[case] bytes: &[u8]) { + assert!(matches!( + encode_payload(bytes.into(), None), + Err(KvpError::PayloadNotUtf8) + )); + assert_eq!(decode_payload(bytes, None), Err(DecodeError::Undecodable)); + } + + #[rstest] + #[case::nul_only("\0", false)] + #[case::embedded_bytes("prefix\0suffix", true)] + #[case::trailing_text("trailing\0", false)] + fn none_rejects_nul_on_write(#[case] text: &str, #[case] as_bytes: bool) { + let payload = if as_bytes { + DiagnosticPayload::Bytes(text.as_bytes().to_vec()) + } else { + DiagnosticPayload::Text(text.to_owned()) + }; + assert!(matches!( + encode_payload(payload, None), + Err(KvpError::ValueContainsNull) + )); + } + + #[rstest] + #[case("")] + #[case("hello")] + #[case("héllo\n\0 | 😀")] + fn gz_b64_text_decodes_to_bytes(#[case] text: &str) { + let encoding = Some(&Encoding::GzB64); + let value = encode_payload(text.into(), encoding).unwrap(); + assert!(value.is_ascii()); + assert!(!value.contains('\0')); + assert_eq!( + decode_payload(value.as_bytes(), encoding).unwrap(), + DiagnosticPayload::Bytes(text.as_bytes().to_vec()) + ); + } + + #[rstest] + #[case(vec![])] + #[case(b"hello".to_vec())] + #[case(vec![0xff, 0x80, 0, 0])] + #[case((0u8..=255).collect())] + fn gz_b64_preserves_arbitrary_bytes(#[case] bytes: Vec) { + let encoding = Some(&Encoding::GzB64); + let value = encode_payload(bytes.clone().into(), encoding).unwrap(); + assert_eq!( + decode_payload(value.as_bytes(), encoding).unwrap(), + DiagnosticPayload::Bytes(bytes) + ); + } + + #[test] + fn gz_b64_writer_emits_gzip_header_crc_and_size() { + let value = + encode_payload("hello".into(), Some(&Encoding::GzB64)).unwrap(); + let gzip = STANDARD.decode(value).unwrap(); + assert!(gzip.starts_with(&[0x1f, 0x8b, 8])); + assert_eq!( + &gzip[gzip.len() - 8..], + &[0x86, 0xa6, 0x10, 0x36, 5, 0, 0, 0] + ); + } + + #[rstest] + #[case::empty(PYTHON_EMPTY, b"")] + #[case::text(PYTHON_HELLO, b"hello")] + #[case::binary(PYTHON_BINARY, b"cloud-init fixture\n\x00\xff")] + #[case::filename(PYTHON_FILENAME, b"cloud-init fixture\n\x00\xff")] + fn decodes_python_gzip_fixtures( + #[case] value: &str, + #[case] expected: &[u8], + ) { + assert_eq!( + decode_payload(value.as_bytes(), Some(&Encoding::GzB64)).unwrap(), + DiagnosticPayload::Bytes(expected.to_vec()) + ); + } + + #[rstest] + #[case("!")] + #[case("====")] + #[case("AA=A")] + #[case("é")] + fn gz_b64_rejects_invalid_base64(#[case] value: &str) { + assert_eq!( + decode_payload(value.as_bytes(), Some(&Encoding::GzB64)), + Err(DecodeError::Undecodable) + ); + } + + #[test] + fn gz_b64_requires_canonical_base64() { + for value in [ + format!("{PYTHON_HELLO}\n"), + PYTHON_HELLO.trim_end_matches('=').to_owned(), + PYTHON_HELLO.replace("AA==", "AB=="), + ] { + assert_eq!( + decode_payload(value.as_bytes(), Some(&Encoding::GzB64)), + Err(DecodeError::Undecodable), + "{value}" + ); + } + } + + #[rstest] + #[case(b"")] + #[case(b"not gzip")] + #[case(b"\x78\x9c\x03\x00\x00\x00\x00\x01")] + fn gz_b64_rejects_non_gzip_data(#[case] bytes: &[u8]) { + let value = STANDARD.encode(bytes); + assert_eq!( + decode_payload(value.as_bytes(), Some(&Encoding::GzB64)), + Err(DecodeError::Undecodable) + ); + } + + #[rstest] + #[case(PYTHON_EMPTY)] + #[case(PYTHON_HELLO)] + #[case(PYTHON_BINARY)] + #[case(PYTHON_FILENAME)] + fn every_truncated_gzip_prefix_is_undecodable(#[case] value: &str) { + let gzip = STANDARD.decode(value).unwrap(); + for end in 0..gzip.len() { + let truncated = STANDARD.encode(&gzip[..end]); + assert_eq!( + decode_payload(truncated.as_bytes(), Some(&Encoding::GzB64)), + Err(DecodeError::Undecodable), + "gzip truncated at byte {end}" + ); + } + } + + #[test] + fn every_truncated_base64_prefix_is_undecodable() { + for end in 0..PYTHON_HELLO.len() { + assert_eq!( + decode_payload( + &PYTHON_HELLO.as_bytes()[..end], + Some(&Encoding::GzB64) + ), + Err(DecodeError::Undecodable), + "base64 truncated at byte {end}" + ); + } + } + + #[rstest] + #[case::magic(0)] + #[case::compression_method(2)] + #[case::deflate_body(10)] + fn gz_b64_rejects_corrupt_gzip(#[case] offset: usize) { + let mut gzip = STANDARD.decode(PYTHON_HELLO).unwrap(); + gzip[offset] ^= 1; + let value = STANDARD.encode(gzip); + assert_eq!( + decode_payload(value.as_bytes(), Some(&Encoding::GzB64)), + Err(DecodeError::Undecodable) + ); + } + + #[rstest] + #[case::crc(8)] + #[case::size(4)] + fn gz_b64_validates_crc_and_size(#[case] trailer_offset: usize) { + let mut gzip = STANDARD.decode(PYTHON_HELLO).unwrap(); + let offset = gzip.len() - trailer_offset; + gzip[offset] ^= 1; + let value = STANDARD.encode(gzip); + assert_eq!( + decode_payload(value.as_bytes(), Some(&Encoding::GzB64)), + Err(DecodeError::Undecodable) + ); + } + + #[rstest] + #[case(b"\0")] + #[case(b"trailing data")] + fn gz_b64_rejects_trailing_data(#[case] suffix: &[u8]) { + let mut gzip = STANDARD.decode(PYTHON_HELLO).unwrap(); + gzip.extend_from_slice(suffix); + let value = STANDARD.encode(gzip); + assert_eq!( + decode_payload(value.as_bytes(), Some(&Encoding::GzB64)), + Err(DecodeError::Undecodable) + ); + } + + #[rstest] + #[case(PYTHON_EMPTY)] + #[case(PYTHON_HELLO)] + fn gz_b64_rejects_concatenated_members(#[case] second: &str) { + let mut gzip = STANDARD.decode(PYTHON_HELLO).unwrap(); + gzip.extend(STANDARD.decode(second).unwrap()); + let value = STANDARD.encode(gzip); + assert_eq!( + decode_payload(value.as_bytes(), Some(&Encoding::GzB64)), + Err(DecodeError::Undecodable) + ); + } + + #[test] + fn gz_b64_handles_large_expansion() { + let bytes = vec![0; 2 * 1024 * 1024]; + let encoding = Some(&Encoding::GzB64); + let value = encode_payload(bytes.clone().into(), encoding).unwrap(); + assert!(value.len() < bytes.len()); + assert_eq!( + decode_payload(value.as_bytes(), encoding).unwrap(), + DiagnosticPayload::Bytes(bytes) + ); + } + + #[rstest] + #[case::unknown("zstd+b64")] + #[case::plain_token("none")] + #[case::gzip_token("gz+b64")] + fn other_encoding_is_never_inferred(#[case] token: &str) { + let encoding = Encoding::Other(token.into()); + assert!(matches!( + encode_payload("payload".into(), Some(&encoding)), + Err(KvpError::UnsupportedEncoding { token: actual }) + if actual == token + )); + assert_eq!( + decode_payload(PYTHON_HELLO.as_bytes(), Some(&encoding)), + Err(DecodeError::Undecodable) + ); + } +} diff --git a/libazureinit-kvp/src/diagnostics/mod.rs b/libazureinit-kvp/src/diagnostics/mod.rs new file mode 100644 index 00000000..923b8b44 --- /dev/null +++ b/libazureinit-kvp/src/diagnostics/mod.rs @@ -0,0 +1,36 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +//! Typed diagnostics over the raw [`crate::KvpPoolStore`]. +//! +//! [`DiagnosticWriter`] emits versioned `DIAG_V1` records. [`DiagnosticReader`] +//! reads diagnostics, provisioning reports, and raw records from a successful +//! string snapshot, including cloud-init diagnostics through a read-only bridge. + +mod cloud_init; +mod diagnostic; +mod encoding; +mod reader; +mod writer; + +pub use diagnostic::{ + DecodeError, Diagnostic, DiagnosticEvent, DiagnosticFinish, DiagnosticKey, + DiagnosticPayload, DiagnosticStart, Encoding, Entry, Kind, Outcome, + RawKeyValue, DIAGNOSTIC_VERSION_ID, +}; +pub use reader::DiagnosticReader; +pub use writer::DiagnosticWriter; + +/// Maximum number of UTF-8 value bytes stored in one diagnostic record. +/// +/// This conservative limit keeps records readable through the Hyper-V host +/// path. Longer messages are split at UTF-8 character boundaries. +pub const MAX_CHUNK_BYTES: usize = 1022; + +/// Parse a non-empty run of ASCII digits (a chunk index or duration) as `u64`. +fn parse_unsigned(value: &str) -> Result { + if value.is_empty() || !value.bytes().all(|byte| byte.is_ascii_digit()) { + return Err(DecodeError::Malformed); + } + value.parse().map_err(|_| DecodeError::Malformed) +} diff --git a/libazureinit-kvp/src/diagnostics/reader.rs b/libazureinit-kvp/src/diagnostics/reader.rs new file mode 100644 index 00000000..8300246d --- /dev/null +++ b/libazureinit-kvp/src/diagnostics/reader.rs @@ -0,0 +1,1047 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +use std::collections::HashMap; + +use chrono::{DateTime, SecondsFormat, Utc}; +use uuid::Uuid; + +use super::cloud_init; +use super::diagnostic::{ + DecodeError, Diagnostic, DiagnosticEvent, DiagnosticFinish, DiagnosticKey, + DiagnosticPayload, DiagnosticStart, Encoding, Entry, Outcome, RawKeyValue, + DIAGNOSTIC_VERSION_ID, +}; +use super::encoding::decode_payload; +use super::parse_unsigned; +use crate::{ + KvpError, KvpPoolStore, ProvisioningReport, PROVISIONING_REPORT_KEY, +}; + +#[derive(Clone, Debug)] +pub struct DiagnosticReader { + store: KvpPoolStore, +} + +impl DiagnosticReader { + /// Construction performs no I/O and needs no local producer identity. + pub fn new(store: KvpPoolStore) -> Self { + Self { store } + } + + /// Invalid pool UTF-8 fails the snapshot; entries keep first-seen order. + pub fn entries(&self) -> Result, KvpError> { + Ok(decode_entries(self.store.dump()?)) + } +} + +struct Chunk { + position: usize, + index: u64, + raw: RawKeyValue, +} + +struct ChunkGroup { + first_position: usize, + chunks: Vec, +} + +fn decode_entries(records: Vec<(String, String)>) -> Vec { + let mut entries = Vec::new(); + let mut groups = HashMap::::new(); + + for (position, (key, value)) in records.into_iter().enumerate() { + let mut raw = RawKeyValue { + key, + value, + error: None, + }; + let prefix = raw.key.split('|').next().unwrap_or_default(); + let parsed = match prefix { + DIAGNOSTIC_VERSION_ID => split_chunk_key(&raw.key) + .map(|(base, index)| (base, Some(index))), + cloud_init::PREFIX => cloud_init::split_key(&raw.key), + PROVISIONING_REPORT_KEY if raw.key == PROVISIONING_REPORT_KEY => { + match raw.value.parse::() { + Ok(report) => { + entries.push((position, Entry::Report(report))) + } + Err(error) => { + raw.error = Some(error); + entries.push((position, Entry::Raw(raw))); + } + } + continue; + } + version if version.starts_with("DIAG_V") => { + Err(DecodeError::UnsupportedVersion) + } + _ => { + entries.push((position, Entry::Raw(raw))); + continue; + } + }; + match parsed { + Ok((base, Some(index))) => { + groups + .entry(base.to_owned()) + .or_insert_with(|| ChunkGroup { + first_position: position, + chunks: Vec::new(), + }) + .chunks + .push(Chunk { + position, + index, + raw, + }); + } + Ok((base, None)) => { + match cloud_init::decode_single(base, &raw.value) { + Ok(diagnostic) => { + entries.push((position, Entry::Diagnostic(diagnostic))) + } + Err(error) => { + raw.error = Some(error); + entries.push((position, Entry::Raw(raw))); + } + } + } + Err(error) => { + raw.error = Some(error); + entries.push((position, Entry::Raw(raw))); + } + } + } + + for (base, mut group) in groups { + let decoded = if base.split('|').next() == Some(cloud_init::PREFIX) { + order_chunks(&mut group.chunks).and_then(|()| { + cloud_init::decode_chunks( + &base, + group + .chunks + .iter() + .map(|chunk| (chunk.index, chunk.raw.value.as_str())), + ) + }) + } else { + decode_v1_group(&base, &mut group.chunks) + }; + match decoded { + Ok(diagnostic) => { + entries.push(( + group.first_position, + Entry::Diagnostic(diagnostic), + )); + } + Err(error) => { + for mut chunk in group.chunks { + chunk.raw.error = Some(error); + entries.push((chunk.position, Entry::Raw(chunk.raw))); + } + } + } + } + + // Failed groups retain every physical record at its original position. + entries.sort_by_key(|(position, _)| *position); + entries.into_iter().map(|(_, entry)| entry).collect() +} + +fn split_chunk_key(key: &str) -> Result<(&str, u64), DecodeError> { + let (base, index) = key.rsplit_once('|').ok_or(DecodeError::Malformed)?; + Ok((base, parse_unsigned(index)?)) +} + +fn decode_v1_group( + base: &str, + chunks: &mut [Chunk], +) -> Result { + let [version, agent, vm_id, kind, name, event_id, timestamp, encoding, result, duration]: [&str; 10] = + base.split('|') + .collect::>() + .try_into() + .map_err(|_| DecodeError::Malformed)?; + if version != DIAGNOSTIC_VERSION_ID + || base.contains('\0') + || [agent, vm_id, name, event_id, timestamp, encoding] + .iter() + .any(|field| field.is_empty()) + { + return Err(DecodeError::Malformed); + } + Uuid::parse_str(vm_id).map_err(|_| DecodeError::Malformed)?; + Uuid::parse_str(event_id).map_err(|_| DecodeError::Malformed)?; + let parsed_timestamp = DateTime::parse_from_rfc3339(timestamp) + .map_err(|_| DecodeError::Malformed)? + .with_timezone(&Utc); + if timestamp.len() != 24 + || parsed_timestamp.to_rfc3339_opts(SecondsFormat::Millis, true) + != timestamp + { + return Err(DecodeError::Malformed); + } + let result = match result { + "" => None, + "success" => Some(Outcome::Success), + "fail" => Some(Outcome::Failure), + _ => return Err(DecodeError::Malformed), + }; + let duration_ms = if duration.is_empty() { + None + } else { + Some(parse_unsigned(duration)?) + }; + let encoding = match encoding { + "none" => None, + "gz+b64" => Some(Encoding::GzB64), + other => Some(Encoding::Other(other.to_owned())), + }; + let key = DiagnosticKey { + agent: agent.to_owned(), + vm_id: Some(vm_id.to_owned()), + name: name.to_owned(), + event_id: event_id.to_owned(), + timestamp: parsed_timestamp, + encoding, + }; + + match (kind, result, duration_ms) { + ("start", None, None) => { + let payload = decode_chunks(chunks, key.encoding.as_ref())?; + Ok(Diagnostic::Start(DiagnosticStart { key, payload })) + } + ("finish", Some(result), Some(duration_ms)) => { + let payload = decode_chunks(chunks, key.encoding.as_ref())?; + Ok(Diagnostic::Finish(DiagnosticFinish { + key, + payload, + result, + duration_ms, + })) + } + ("event", result, duration_ms) => { + let payload = decode_chunks(chunks, key.encoding.as_ref())?; + Ok(Diagnostic::Event(DiagnosticEvent { + key, + payload, + result, + duration_ms, + })) + } + _ => Err(DecodeError::Malformed), + } +} + +fn decode_chunks( + chunks: &mut [Chunk], + encoding: Option<&Encoding>, +) -> Result { + order_chunks(chunks)?; + let value: String = chunks + .iter() + .map(|chunk| chunk.raw.value.as_str()) + .collect(); + decode_payload(value.as_bytes(), encoding) +} + +fn order_chunks(chunks: &mut [Chunk]) -> Result<(), DecodeError> { + chunks.sort_by_key(|chunk| chunk.index); + if chunks.windows(2).any(|pair| pair[0].index == pair[1].index) { + return Err(DecodeError::DuplicateChunk); + } + if chunks + .iter() + .enumerate() + .any(|(expected, chunk)| usize::try_from(chunk.index) != Ok(expected)) + { + return Err(DecodeError::IncompleteGroup); + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + use std::fs; + use std::io; + use std::path::Path; + use std::sync::atomic::{AtomicUsize, Ordering}; + use std::sync::Arc; + + use base64::{engine::general_purpose::STANDARD, Engine as _}; + use rstest::rstest; + use tempfile::TempDir; + + use super::super::diagnostic::Kind; + use crate::store::{Handle, OsSysOps, StatInfo, SysOps}; + use crate::{write_report, KvpPool, PoolMode, ReportPpsType}; + + const AGENT: &str = "azure-init-0.1.1"; + const VM_ID: &str = "3f2504e0-4f89-41d3-9a0c-0305e82c3301"; + const EVENT_ID: &str = "8f3e9c4a-1b2c-4d5e-9f01-234567890abc"; + const TIMESTAMP: &str = "2026-08-31T12:34:56.789Z"; + const GZIP_HELLO: &str = "H4sIAAAAAAAC/8tIzcnJBwCGphA2BQAAAA=="; + + fn key(index: u64) -> String { + format!( + "DIAG_V1|{AGENT}|{VM_ID}|event|test|{EVENT_ID}|{TIMESTAMP}|none|||{index}" + ) + } + + fn with_field(key: &str, index: usize, value: &str) -> String { + let mut fields: Vec<_> = key.split('|').collect(); + fields[index] = value; + fields.join("|") + } + + fn raw_entries( + records: &[(String, String)], + error: Option, + ) -> Vec { + records + .iter() + .map(|(key, value)| { + Entry::Raw(RawKeyValue { + key: key.clone(), + value: value.clone(), + error, + }) + }) + .collect() + } + + fn only_diagnostic(entries: Vec) -> Diagnostic { + assert_eq!(entries.len(), 1); + match entries.into_iter().next().unwrap() { + Entry::Diagnostic(diagnostic) => diagnostic, + other => panic!("expected a diagnostic, got {other:?}"), + } + } + + fn store(dir: &TempDir) -> KvpPoolStore { + KvpPoolStore::new_in(KvpPool::Guest, dir.path(), PoolMode::Safe) + .unwrap() + } + + #[derive(Debug, Default)] + struct ReadOnlyOps { + os: OsSysOps, + reads: AtomicUsize, + open_error: Option, + } + + impl SysOps for ReadOnlyOps { + fn open_read(&self, path: &Path) -> io::Result> { + self.reads.fetch_add(1, Ordering::SeqCst); + if let Some(error) = self.open_error { + return Err(error.into()); + } + self.os.open_read(path) + } + + fn open_read_write(&self, _: &Path) -> io::Result> { + panic!("reader must not write the pool") + } + + fn open_read_write_create( + &self, + _: &Path, + ) -> io::Result> { + panic!("reader must not create the pool") + } + + fn path_metadata(&self, _: &Path) -> io::Result { + panic!("reader must not inspect pool staleness") + } + + fn boot_time(&self) -> io::Result { + panic!("reader must not read boot state") + } + } + + fn observed_reader(dir: &TempDir) -> (DiagnosticReader, Arc) { + let ops = Arc::new(ReadOnlyOps::default()); + let observed = KvpPoolStore::with_ops( + KvpPool::Guest, + dir.path(), + PoolMode::Safe, + ops.clone(), + ) + .unwrap(); + (DiagnosticReader::new(observed), ops) + } + + #[test] + fn constructor_does_no_io() { + let dir = TempDir::new().unwrap(); + let (reader, ops) = observed_reader(&dir); + assert_eq!(ops.reads.load(Ordering::SeqCst), 0); + assert!(!reader.store.path().exists()); + } + + #[test] + fn entries_reads_one_fresh_snapshot_without_modifying_the_pool() { + let dir = TempDir::new().unwrap(); + let pool = store(&dir); + let (reader, ops) = observed_reader(&dir); + assert!(reader.entries().unwrap().is_empty()); + assert_eq!(ops.reads.load(Ordering::SeqCst), 1); + assert!(!pool.path().exists()); + + pool.append("unrelated", "unchanged").unwrap(); + let before = fs::read(pool.path()).unwrap(); + assert_eq!( + reader.entries().unwrap(), + raw_entries(&[("unrelated".into(), "unchanged".into())], None) + ); + assert_eq!(ops.reads.load(Ordering::SeqCst), 2); + assert_eq!(fs::read(pool.path()).unwrap(), before); + } + + #[test] + fn snapshot_open_errors_propagate() { + let dir = TempDir::new().unwrap(); + let ops = Arc::new(ReadOnlyOps { + open_error: Some(io::ErrorKind::PermissionDenied), + ..ReadOnlyOps::default() + }); + let pool = KvpPoolStore::with_ops( + KvpPool::Guest, + dir.path(), + PoolMode::Safe, + ops.clone(), + ) + .unwrap(); + let reader = DiagnosticReader::new(pool); + assert_eq!(ops.reads.load(Ordering::SeqCst), 0); + assert!(matches!( + reader.entries(), + Err(KvpError::Io(error)) + if error.kind() == io::ErrorKind::PermissionDenied + )); + assert_eq!(ops.reads.load(Ordering::SeqCst), 1); + } + + #[test] + fn truncated_physical_pool_is_a_snapshot_error() { + let dir = TempDir::new().unwrap(); + let pool = store(&dir); + fs::write(pool.path(), b"partial record").unwrap(); + let reader = DiagnosticReader::new(pool); + assert!(matches!( + reader.entries(), + Err(KvpError::Io(error)) if error.kind() == io::ErrorKind::Other + )); + } + + #[rstest] + #[case::key(0)] + #[case::value(512)] + fn non_utf8_record_fails_the_entire_snapshot(#[case] offset: usize) { + let dir = TempDir::new().unwrap(); + let pool = store(&dir); + pool.append(&with_field(&key(0), 4, "before"), "valid") + .unwrap(); + let record_start = fs::read(pool.path()).unwrap().len(); + pool.append(&key(0), "value").unwrap(); + pool.append(&with_field(&key(0), 4, "after"), "valid") + .unwrap(); + let reader = DiagnosticReader::new(pool.clone()); + assert_eq!(reader.entries().unwrap().len(), 3); + + let mut bytes = fs::read(pool.path()).unwrap(); + bytes[record_start + offset] = 0xff; + fs::write(pool.path(), &bytes).unwrap(); + assert!(matches!( + reader.entries(), + Err(KvpError::Io(error)) if error.kind() == io::ErrorKind::InvalidData + )); + assert_eq!(fs::read(pool.path()).unwrap(), bytes); + } + + #[rstest] + #[case("")] + #[case("unrelated")] + #[case("other|0")] + #[case("DIAG")] + #[case("DIAG_OTHER|0")] + #[case("diag_v1|0")] + #[case("prefixDIAG_V2|0")] + #[case("azure-init-0.1.1|1700000000|vm-abc|event|imds|id|2026-08-31T12:34:56.789Z|0")] + fn unrelated_and_pre_adoption_records_remain_raw(#[case] key: &str) { + let records = vec![ + (key.into(), "first\nvalue".into()), + (key.into(), "second \"value\"".into()), + ]; + assert_eq!( + decode_entries(records.clone()), + raw_entries(&records, None) + ); + } + + #[rstest] + #[case::missing_fields("result=success".into())] + #[case::conflicting_results(format!( + "result=success|agent={AGENT}|pps_type=None|vm_id={VM_ID}|timestamp={TIMESTAMP}|result=error" + ))] + fn malformed_report_is_preserved_without_losing_other_entries( + #[case] value: String, + ) { + let records = vec![(PROVISIONING_REPORT_KEY.into(), value)]; + let mut mixed = records.clone(); + mixed.push((key(0), "valid diagnostic".into())); + let entries = decode_entries(mixed); + assert_eq!(entries.len(), 2); + assert_eq!( + entries[0], + raw_entries(&records, Some(DecodeError::Malformed))[0] + ); + assert!(matches!(entries[1], Entry::Diagnostic(_))); + } + + #[test] + fn report_is_typed_in_first_seen_order() { + let dir = TempDir::new().unwrap(); + let pool = store(&dir); + let report = ProvisioningReport::failure( + AGENT, + VM_ID, + "failed | with details", + ReportPpsType::None, + ) + .with_extra("detail", "first") + .with_extra("detail", "second"); + pool.append(&key(1), "second chunk").unwrap(); + write_report(&pool, &report).unwrap(); + pool.append(&key(0), "first chunk").unwrap(); + let before = fs::read(pool.path()).unwrap(); + let entries = DiagnosticReader::new(pool.clone()).entries().unwrap(); + assert_eq!(entries.len(), 2); + assert!(matches!(entries[0], Entry::Diagnostic(_))); + assert_eq!(entries[1], Entry::Report(report)); + assert_eq!(fs::read(pool.path()).unwrap(), before); + } + + #[test] + fn repeated_report_records_are_not_deduplicated() { + let dir = TempDir::new().unwrap(); + let pool = store(&dir); + let report = + ProvisioningReport::success(AGENT, VM_ID, ReportPpsType::None); + write_report(&pool, &report).unwrap(); + let value = pool.read(PROVISIONING_REPORT_KEY).unwrap().unwrap(); + pool.append(PROVISIONING_REPORT_KEY, &value).unwrap(); + assert_eq!( + DiagnosticReader::new(pool).entries().unwrap(), + vec![Entry::Report(report.clone()), Entry::Report(report)] + ); + } + + #[test] + fn report_prefix_is_not_a_chunked_report_key() { + let records = vec![("PROVISIONING_REPORT|0".into(), "value".into())]; + assert_eq!( + decode_entries(records.clone()), + raw_entries(&records, None) + ); + } + + #[rstest] + #[case("DIAG_V0")] + #[case("DIAG_V2")] + #[case("DIAG_V999")] + #[case("DIAG_V1_extra")] + #[case("DIAG_V")] + fn unsupported_versions_do_not_interpret_later_fields( + #[case] version: &str, + ) { + let records = vec![ + (version.into(), "value".into()), + (format!("{version}|bad metadata|0"), "first".into()), + (format!("{version}|bad metadata|0"), "duplicate".into()), + (format!("{version}|bad metadata|2"), "gap".into()), + ]; + assert_eq!( + decode_entries(records.clone()), + raw_entries(&records, Some(DecodeError::UnsupportedVersion)) + ); + } + + #[test] + fn v1_event_decodes_without_local_identity() { + let payload = "héllo\n\"message\" | ="; + let diagnostic = + only_diagnostic(decode_entries(vec![(key(0), payload.into())])); + assert_eq!(diagnostic.kind(), Kind::Event); + assert_eq!(diagnostic.key().agent, AGENT); + assert_eq!(diagnostic.key().vm_id.as_deref(), Some(VM_ID)); + assert_eq!(diagnostic.key().name, "test"); + assert_eq!(diagnostic.key().event_id, EVENT_ID); + assert_eq!(diagnostic.key().encoding, None); + assert_eq!( + diagnostic + .key() + .timestamp + .to_rfc3339_opts(SecondsFormat::Millis, true), + TIMESTAMP + ); + assert_eq!( + diagnostic.payload(), + &DiagnosticPayload::Text(payload.into()) + ); + } + + #[rstest] + #[case::unmatched_start("start", "", "")] + #[case::orphan_finish("finish", "fail", "312")] + fn isolated_span_endpoint_decodes( + #[case] kind: &str, + #[case] result: &str, + #[case] duration: &str, + ) { + let key = with_field(&key(0), 3, kind); + let key = with_field(&key, 8, result); + let key = with_field(&key, 9, duration); + let diagnostic = + only_diagnostic(decode_entries(vec![(key, "message".into())])); + assert_eq!(diagnostic.kind().to_string(), kind); + } + + #[test] + fn span_endpoints_with_a_shared_id_decode_separately() { + let start = with_field(&key(0), 3, "start"); + let finish = with_field(&key(0), 3, "finish"); + let finish = with_field(&finish, 8, "fail"); + let finish = with_field(&finish, 9, "312"); + let entries = decode_entries(vec![ + (start, "starting".into()), + (finish, "failed".into()), + ]); + let [Entry::Diagnostic(Diagnostic::Start(start)), Entry::Diagnostic(Diagnostic::Finish(finish))] = + entries.as_slice() + else { + panic!("expected separate start and finish entries"); + }; + assert_eq!(start.key.event_id, EVENT_ID); + assert_eq!(finish.key.event_id, EVENT_ID); + assert_eq!(start.payload, DiagnosticPayload::Text("starting".into())); + assert_eq!(finish.payload, DiagnosticPayload::Text("failed".into())); + assert_eq!(finish.result, Outcome::Failure); + assert_eq!(finish.duration_ms, 312); + } + + #[rstest] + #[case::neither(None, None)] + #[case::result_only(Some(Outcome::Success), None)] + #[case::zero_duration(None, Some(0))] + #[case::both(Some(Outcome::Failure), Some(52))] + fn event_result_and_duration_are_independent( + #[case] result: Option, + #[case] duration: Option, + ) { + let result_token = result.map_or_else(String::new, |v| v.to_string()); + let duration_token = + duration.map_or_else(String::new, |v| v.to_string()); + let key = with_field(&key(0), 8, &result_token); + let key = with_field(&key, 9, &duration_token); + let diagnostic = + only_diagnostic(decode_entries(vec![(key, "value".into())])); + let Diagnostic::Event(event) = diagnostic else { + panic!("expected an event"); + }; + assert_eq!(event.result, result); + assert_eq!(event.duration_ms, duration); + } + + #[rstest] + #[case("start", "success", "")] + #[case("start", "", "0")] + #[case("start", "fail", "52")] + #[case("finish", "", "")] + #[case("finish", "success", "")] + #[case("finish", "", "0")] + #[case("finish", "error", "52")] + fn invalid_kind_fields_are_malformed( + #[case] kind: &str, + #[case] result: &str, + #[case] duration: &str, + ) { + let base = with_field(&key(0), 3, kind); + let base = with_field(&base, 8, result); + let base = with_field(&base, 9, duration); + let records = vec![(base, "value".into())]; + assert_eq!( + decode_entries(records.clone()), + raw_entries(&records, Some(DecodeError::Malformed)) + ); + } + + #[rstest] + #[case::empty_agent(1, "")] + #[case::invalid_vm_id(2, "vm-abc")] + #[case::invalid_kind(3, "compressed")] + #[case::empty_name(4, "")] + #[case::null_in_name(4, "bad\0name")] + #[case::invalid_event_id(5, "bad-uuid")] + #[case::empty_encoding(7, "")] + #[case::invalid_result(8, "SUCCESS")] + #[case::signed_duration(9, "+1")] + #[case::numeric_overflow(9, "18446744073709551616")] + #[case::missing_chunk_index(10, "")] + #[case::non_numeric_chunk_index(10, "x")] + fn malformed_v1_fields_preserve_the_original_record( + #[case] field: usize, + #[case] value: &str, + ) { + let records = vec![(with_field(&key(0), field, value), "value".into())]; + assert_eq!( + decode_entries(records.clone()), + raw_entries(&records, Some(DecodeError::Malformed)) + ); + } + + #[test] + fn v1_requires_the_exact_key_layout_and_an_index() { + let complete = key(0); + let records = vec![ + ("DIAG_V1".into(), "value".into()), + (complete.rsplit_once('|').unwrap().0.into(), "value".into()), + (format!("{complete}|1"), "value".into()), + ]; + assert_eq!( + decode_entries(records.clone()), + raw_entries(&records, Some(DecodeError::Malformed)) + ); + } + + #[rstest] + #[case("not a timestamp")] + #[case("2026-08-31T12:34:56Z")] + #[case("2026-08-31T12:34:56.78Z")] + #[case("2026-08-31T12:34:56.789000Z")] + #[case("2026-08-31T12:34:56.789+00:00")] + #[case("2026-08-31t12:34:56.789z")] + #[case("2026-13-31T12:34:56.789Z")] + fn v1_requires_utc_millisecond_timestamps(#[case] timestamp: &str) { + let records = vec![(with_field(&key(0), 6, timestamp), "value".into())]; + assert_eq!( + decode_entries(records.clone()), + raw_entries(&records, Some(DecodeError::Malformed)) + ); + } + + #[rstest] + #[case([1, 0, 2])] + #[case([2, 1, 0])] + fn chunk_permutations_decode_in_index_order(#[case] order: [u64; 3]) { + let records = order + .into_iter() + .map(|index| (key(index), ["a", "é", "c"][index as usize].into())) + .collect(); + let diagnostic = only_diagnostic(decode_entries(records)); + assert_eq!( + diagnostic.payload(), + &DiagnosticPayload::Text("aéc".into()) + ); + } + + #[test] + fn logical_groups_use_first_seen_order_not_timestamps() { + let later = with_field(&key(0), 6, "2026-08-31T12:35:00.000Z"); + let earlier = with_field(&key(0), 4, "earlier"); + let records = vec![ + (with_field(&later, 10, "1"), "b".into()), + ("raw".into(), "untouched".into()), + (earlier, "early".into()), + (later, "a".into()), + ]; + let entries = decode_entries(records); + assert_eq!(entries.len(), 3); + let Entry::Diagnostic(first) = &entries[0] else { + panic!("expected first-seen group"); + }; + assert_eq!(first.payload(), &DiagnosticPayload::Text("ab".into())); + assert!(matches!(&entries[1], Entry::Raw(raw) if raw.key == "raw")); + let Entry::Diagnostic(last) = &entries[2] else { + panic!("expected earlier-timestamp group"); + }; + assert!(first.key().timestamp > last.key().timestamp); + } + + #[rstest] + #[case(vec![1])] + #[case(vec![0, 2])] + #[case(vec![2, 0])] + #[case(vec![u64::MAX])] + #[case(vec![0, u64::MAX])] + fn missing_chunks_preserve_all_members(#[case] indices: Vec) { + let records: Vec<_> = indices + .into_iter() + .map(|index| (key(index), format!("chunk {index}"))) + .collect(); + assert_eq!( + decode_entries(records.clone()), + raw_entries(&records, Some(DecodeError::IncompleteGroup)) + ); + } + + #[rstest] + #[case(vec![0, 0])] + #[case(vec![1, 1])] + #[case(vec![0, 1, 1])] + #[case(vec![2, 0, 2])] + fn duplicates_take_precedence_over_gaps(#[case] indices: Vec) { + let records: Vec<_> = indices + .into_iter() + .enumerate() + .map(|(position, index)| { + (key(index), format!("original {position}")) + }) + .collect(); + assert_eq!( + decode_entries(records.clone()), + raw_entries(&records, Some(DecodeError::DuplicateChunk)) + ); + } + + #[test] + fn duplicate_index_spellings_preserve_the_exact_keys() { + let records = vec![ + (key(0), "first".into()), + (with_field(&key(0), 10, "00"), "second".into()), + ]; + assert_eq!( + decode_entries(records.clone()), + raw_entries(&records, Some(DecodeError::DuplicateChunk)) + ); + } + + #[test] + fn failed_groups_preserve_physical_positions_among_other_entries() { + let records = vec![ + (key(2), "first seen".into()), + ("unrelated".into(), "unchanged".into()), + (key(0), "zero".into()), + (with_field(&key(0), 4, "valid"), "good".into()), + (key(2), "duplicate".into()), + ]; + let entries = decode_entries(records.clone()); + assert_eq!(entries.len(), records.len()); + for position in [0, 2, 4] { + let expected = raw_entries( + &records[position..=position], + Some(DecodeError::DuplicateChunk), + ); + assert_eq!(entries[position], expected[0]); + } + assert_eq!(entries[1], raw_entries(&records[1..2], None)[0]); + assert!( + matches!(&entries[3], Entry::Diagnostic(d) if d.key().name == "valid") + ); + } + + #[rstest] + #[case(1, "other-agent")] + #[case(2, "00000000-0000-0000-0000-000000000000")] + #[case(3, "start")] + #[case(4, "other-name")] + #[case(5, "00000000-0000-0000-0000-000000000000")] + #[case(6, "2026-08-31T12:34:56.790Z")] + #[case(7, "gz+b64")] + #[case(8, "success")] + #[case(9, "52")] + fn grouping_does_not_combine_different_metadata( + #[case] field: usize, + #[case] value: &str, + ) { + let detached = (with_field(&key(1), field, value), "detached".into()); + let entries = decode_entries(vec![ + (key(0), "a".into()), + detached.clone(), + (key(1), "b".into()), + ]); + assert_eq!(entries.len(), 2); + assert!(matches!( + &entries[0], + Entry::Diagnostic(d) if d.payload() == &DiagnosticPayload::Text("ab".into()) + )); + assert_eq!( + entries[1], + raw_entries(&[detached], Some(DecodeError::IncompleteGroup))[0] + ); + } + + #[rstest] + #[case(2, VM_ID, "3F2504E0-4F89-41D3-9A0C-0305E82C3301")] + #[case(5, EVENT_ID, "8f3e9c4a1b2c4d5e9f01234567890abc")] + #[case(9, "52", "052")] + fn equivalent_metadata_spellings_remain_separate_groups( + #[case] field: usize, + #[case] first: &str, + #[case] second: &str, + ) { + let records = vec![ + (with_field(&key(0), field, first), "a".into()), + (with_field(&key(0), field, second), "b".into()), + (with_field(&key(1), field, first), "c".into()), + (with_field(&key(1), field, second), "d".into()), + ]; + let entries = decode_entries(records); + assert_eq!(entries.len(), 2); + let Entry::Diagnostic(first) = &entries[0] else { + panic!("expected first group"); + }; + let Entry::Diagnostic(second) = &entries[1] else { + panic!("expected second group"); + }; + assert_eq!(first.payload(), &DiagnosticPayload::Text("ac".into())); + assert_eq!(second.payload(), &DiagnosticPayload::Text("bd".into())); + } + + #[test] + fn versions_are_never_grouped_together() { + let future = (with_field(&key(1), 0, "DIAG_V2"), "future chunk".into()); + let entries = decode_entries(vec![ + (key(0), "a".into()), + future.clone(), + (key(1), "b".into()), + ]); + assert_eq!(entries.len(), 2); + assert!(matches!( + &entries[0], + Entry::Diagnostic(d) if d.payload() == &DiagnosticPayload::Text("ab".into()) + )); + assert_eq!( + entries[1], + raw_entries(&[future], Some(DecodeError::UnsupportedVersion))[0] + ); + } + + #[rstest] + #[case("base64")] + #[case("zstd+b64")] + #[case("GZ+B64")] + fn unknown_encodings_preserve_every_chunk(#[case] encoding: &str) { + let records = vec![ + (with_field(&key(1), 7, encoding), "b".into()), + (with_field(&key(0), 7, encoding), "a".into()), + ]; + assert_eq!( + decode_entries(records.clone()), + raw_entries(&records, Some(DecodeError::Undecodable)) + ); + } + + #[test] + fn compressed_payload_is_joined_before_decoding() { + let records = vec![ + (with_field(&key(1), 7, "gz+b64"), GZIP_HELLO[5..].into()), + (with_field(&key(0), 7, "gz+b64"), GZIP_HELLO[..5].into()), + ]; + let diagnostic = only_diagnostic(decode_entries(records)); + assert_eq!(diagnostic.key().encoding, Some(Encoding::GzB64)); + assert_eq!( + diagnostic.payload(), + &DiagnosticPayload::Bytes(b"hello".to_vec()) + ); + } + + #[test] + fn corrupt_gzip_preserves_every_original_chunk() { + let mut gzip = STANDARD.decode(GZIP_HELLO).unwrap(); + let checksum_offset = gzip.len() - 8; + gzip[checksum_offset] ^= 1; + let value = STANDARD.encode(gzip); + let records = vec![ + (with_field(&key(1), 7, "gz+b64"), value[5..].into()), + (with_field(&key(0), 7, "gz+b64"), value[..5].into()), + ]; + assert_eq!( + decode_entries(records.clone()), + raw_entries(&records, Some(DecodeError::Undecodable)) + ); + } + + #[test] + fn visible_gaps_are_reported_before_payload_errors() { + let records = + vec![(with_field(&key(1), 7, "gz+b64"), "not base64".into())]; + assert_eq!( + decode_entries(records.clone()), + raw_entries(&records, Some(DecodeError::IncompleteGroup)) + ); + } + + #[test] + fn empty_text_payload_decodes() { + let diagnostic = + only_diagnostic(decode_entries(vec![(key(0), String::new())])); + assert_eq!( + diagnostic.payload(), + &DiagnosticPayload::Text(String::new()) + ); + } + + #[rstest] + #[case::agent(1, "a".repeat(64))] + #[case::name(4, "n".repeat(160))] + fn reader_accepts_fields_above_writer_budgets( + #[case] field: usize, + #[case] value: String, + ) { + let dir = TempDir::new().unwrap(); + let pool = + KvpPoolStore::new_in(KvpPool::Guest, dir.path(), PoolMode::Unsafe) + .unwrap(); + pool.append(&with_field(&key(0), field, &value), "payload") + .unwrap(); + let diagnostic = only_diagnostic( + DiagnosticReader::new(store(&dir)).entries().unwrap(), + ); + let actual = match field { + 1 => &diagnostic.key().agent, + _ => &diagnostic.key().name, + }; + assert_eq!(actual, &value); + } + + #[test] + fn reader_accepts_full_u64_duration() { + let key = with_field(&key(0), 9, &u64::MAX.to_string()); + let diagnostic = + only_diagnostic(decode_entries(vec![(key, "payload".into())])); + let Diagnostic::Event(event) = diagnostic else { + panic!("expected an event"); + }; + assert_eq!(event.duration_ms, Some(u64::MAX)); + } + + #[test] + fn reader_accepts_values_above_safe_write_limit() { + let dir = TempDir::new().unwrap(); + let pool = + KvpPoolStore::new_in(KvpPool::Guest, dir.path(), PoolMode::Unsafe) + .unwrap(); + let value = "v".repeat(1500); + pool.append(&key(0), &value).unwrap(); + let diagnostic = only_diagnostic( + DiagnosticReader::new(store(&dir)).entries().unwrap(), + ); + assert_eq!(diagnostic.payload(), &DiagnosticPayload::Text(value)); + } + + #[test] + fn reader_does_not_limit_contiguous_group_size() { + let records = (0..1024) + .rev() + .map(|index| (key(index), "x".into())) + .collect(); + let diagnostic = only_diagnostic(decode_entries(records)); + assert_eq!( + diagnostic.payload(), + &DiagnosticPayload::Text("x".repeat(1024)) + ); + } +} diff --git a/libazureinit-kvp/src/diagnostics/writer.rs b/libazureinit-kvp/src/diagnostics/writer.rs new file mode 100644 index 00000000..4b488735 --- /dev/null +++ b/libazureinit-kvp/src/diagnostics/writer.rs @@ -0,0 +1,948 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +use chrono::{SecondsFormat, Utc}; +use uuid::Uuid; + +use super::diagnostic::{ + Diagnostic, DiagnosticEvent, DiagnosticFinish, DiagnosticKey, + DiagnosticPayload, DiagnosticStart, Encoding, Outcome, + DIAGNOSTIC_VERSION_ID, +}; +use super::encoding::encode_payload; +use super::MAX_CHUNK_BYTES; +use crate::{KvpError, KvpPoolStore}; + +const MAX_AGENT_BYTES: usize = 32; +const MAX_NAME_BYTES: usize = 48; +const MAX_UUID_BYTES: usize = 36; +const MAX_TIMESTAMP_BYTES: usize = 24; +const MAX_KEY_BYTES: usize = 254; +const MAX_DURATION_MS: u64 = 9_999_999_999; +const MAX_CHUNKS: usize = 1023; + +/// Validation precedes I/O; a storage failure can leave a partial batch. +#[derive(Clone, Debug)] +pub struct DiagnosticWriter { + store: KvpPoolStore, + agent: String, + vm_id: String, +} + +impl DiagnosticWriter { + /// Validates producer identity without reading the pool or boot state. + pub fn new( + store: KvpPoolStore, + agent: impl Into, + vm_id: impl Into, + ) -> Result { + let agent = agent.into(); + let vm_id = vm_id.into(); + validate_field("agent", &agent, MAX_AGENT_BYTES)?; + validate_uuid("vm_id", &vm_id)?; + Ok(Self { + store, + agent, + vm_id, + }) + } + + /// The caller retains `event_id` to correlate the corresponding finish. + pub fn emit_start( + &self, + event_id: &str, + name: &str, + payload: impl Into, + encoding: Option, + ) -> Result<(), KvpError> { + self.emit(Diagnostic::Start(DiagnosticStart { + key: self.key(event_id, name, encoding), + payload: payload.into(), + })) + } + + /// Duration is caller-measured milliseconds, not inferred from the pool. + pub fn emit_finish( + &self, + event_id: &str, + name: &str, + payload: impl Into, + encoding: Option, + result: Outcome, + duration_ms: u64, + ) -> Result<(), KvpError> { + self.emit(Diagnostic::Finish(DiagnosticFinish { + key: self.key(event_id, name, encoding), + payload: payload.into(), + result, + duration_ms, + })) + } + + /// Each standalone event receives a fresh UUID. + pub fn emit_event( + &self, + name: &str, + payload: impl Into, + encoding: Option, + result: Option, + duration_ms: Option, + ) -> Result<(), KvpError> { + self.emit(Diagnostic::Event(DiagnosticEvent { + key: self.key(&Uuid::new_v4().to_string(), name, encoding), + payload: payload.into(), + result, + duration_ms, + })) + } + + fn key( + &self, + event_id: &str, + name: &str, + encoding: Option, + ) -> DiagnosticKey { + DiagnosticKey { + agent: self.agent.clone(), + vm_id: Some(self.vm_id.clone()), + name: name.to_owned(), + event_id: event_id.to_owned(), + timestamp: Utc::now(), + encoding, + } + } + + fn emit(&self, diagnostic: Diagnostic) -> Result<(), KvpError> { + self.store.append_multiple(prepare_records(diagnostic)?) + } +} + +fn prepare_records( + diagnostic: Diagnostic, +) -> Result, KvpError> { + let kind = diagnostic.kind(); + let (key, payload, result, duration_ms) = match diagnostic { + Diagnostic::Start(start) => (start.key, start.payload, None, None), + Diagnostic::Finish(finish) => ( + finish.key, + finish.payload, + Some(finish.result), + Some(finish.duration_ms), + ), + Diagnostic::Event(event) => { + (event.key, event.payload, event.result, event.duration_ms) + } + }; + + validate_field("agent", &key.agent, MAX_AGENT_BYTES)?; + let vm_id = key + .vm_id + .as_deref() + .ok_or(KvpError::EmptyEventField { field: "vm_id" })?; + validate_uuid("vm_id", vm_id)?; + validate_field("name", &key.name, MAX_NAME_BYTES)?; + validate_uuid("event_id", &key.event_id)?; + + let duration = match duration_ms { + Some(actual) if actual > MAX_DURATION_MS => { + return Err(KvpError::DurationTooLarge { + max_ms: MAX_DURATION_MS, + actual_ms: actual, + }); + } + Some(duration) => duration.to_string(), + None => String::new(), + }; + let timestamp = key.timestamp.to_rfc3339_opts(SecondsFormat::Millis, true); + validate_field("timestamp", ×tamp, MAX_TIMESTAMP_BYTES)?; + + let value = encode_payload(payload, key.encoding.as_ref())?; + let encoding = key + .encoding + .as_ref() + .map_or_else(|| "none".to_owned(), ToString::to_string); + let result = result.map_or_else(String::new, |result| result.to_string()); + let base_key = format!( + "{DIAGNOSTIC_VERSION_ID}|{}|{vm_id}|{kind}|{}|{}|{timestamp}|{encoding}|{result}|{duration}", + key.agent, key.name, key.event_id, + ); + frame_records(&base_key, &value) +} + +fn validate_field( + field: &'static str, + value: &str, + max: usize, +) -> Result<(), KvpError> { + if value.is_empty() { + return Err(KvpError::EmptyEventField { field }); + } + if value.contains('|') { + return Err(KvpError::EventFieldContainsDelimiter { field }); + } + if value.contains('\0') { + return Err(KvpError::KeyContainsNull); + } + if value.len() > max { + return Err(KvpError::EventFieldTooLong { + field, + max, + actual: value.len(), + }); + } + Ok(()) +} + +fn validate_uuid(field: &'static str, value: &str) -> Result<(), KvpError> { + validate_field(field, value, MAX_UUID_BYTES)?; + Uuid::parse_str(value) + .map(|_| ()) + .map_err(|_| KvpError::InvalidUuid { field }) +} + +fn frame_records( + base_key: &str, + mut value: &str, +) -> Result, KvpError> { + let mut records = Vec::new(); + loop { + if records.len() == MAX_CHUNKS { + return Err(KvpError::TooManyChunks { max: MAX_CHUNKS }); + } + let key = format!("{base_key}|{}", records.len()); + if key.len() > MAX_KEY_BYTES { + return Err(KvpError::KeyTooLarge { + max: MAX_KEY_BYTES, + actual: key.len(), + }); + } + + let mut end = value.len().min(MAX_CHUNK_BYTES); + while !value.is_char_boundary(end) { + end -= 1; + } + let (chunk, remaining) = value.split_at(end); + records.push((key, chunk.to_owned())); + if remaining.is_empty() { + return Ok(records); + } + value = remaining; + } +} + +#[cfg(test)] +mod tests { + use super::*; + use std::fs; + use std::io; + use std::path::Path; + use std::sync::atomic::{AtomicUsize, Ordering}; + use std::sync::Arc; + + use chrono::DateTime; + use rstest::rstest; + use tempfile::TempDir; + + use super::super::diagnostic::Kind; + use super::super::encoding::decode_payload; + use crate::store::{Handle, OsSysOps, StatInfo, SysOps}; + use crate::{KvpPool, PoolMode}; + + const AGENT: &str = "azure-init-0.1.1"; + const VM_ID: &str = "3f2504e0-4f89-41d3-9a0c-0305e82c3301"; + const EVENT_ID: &str = "8f3e9c4a-1b2c-4d5e-9f01-234567890abc"; + const TIMESTAMP: &str = "2026-08-31T12:34:56.789Z"; + + fn store(dir: &TempDir, mode: PoolMode) -> KvpPoolStore { + KvpPoolStore::new_in(KvpPool::Guest, dir.path(), mode).unwrap() + } + + fn key() -> DiagnosticKey { + DiagnosticKey { + agent: AGENT.into(), + vm_id: Some(VM_ID.into()), + name: "provision:run".into(), + event_id: EVENT_ID.into(), + timestamp: DateTime::parse_from_rfc3339(TIMESTAMP) + .unwrap() + .with_timezone(&Utc), + encoding: None, + } + } + + fn event(key: DiagnosticKey, payload: DiagnosticPayload) -> Diagnostic { + Diagnostic::Event(DiagnosticEvent { + key, + payload, + result: None, + duration_ms: None, + }) + } + + #[derive(Debug, Default)] + struct WriteOnlyOps { + os: OsSysOps, + opens: AtomicUsize, + } + + impl SysOps for WriteOnlyOps { + fn open_read(&self, _: &Path) -> io::Result> { + panic!("writer must not read the pool") + } + + fn open_read_write(&self, _: &Path) -> io::Result> { + panic!("writer must only append") + } + + fn open_read_write_create( + &self, + path: &Path, + ) -> io::Result> { + self.opens.fetch_add(1, Ordering::SeqCst); + self.os.open_read_write_create(path) + } + + fn path_metadata(&self, _: &Path) -> io::Result { + panic!("writer must not inspect pool staleness") + } + + fn boot_time(&self) -> io::Result { + panic!("writer must not read boot state") + } + } + + fn observed_writer(dir: &TempDir) -> (DiagnosticWriter, Arc) { + let ops = Arc::new(WriteOnlyOps::default()); + let store = KvpPoolStore::with_ops( + KvpPool::Guest, + dir.path(), + PoolMode::Safe, + ops.clone(), + ) + .unwrap(); + (DiagnosticWriter::new(store, AGENT, VM_ID).unwrap(), ops) + } + + fn assert_rejected_without_writes( + operation: impl FnOnce(&DiagnosticWriter) -> Result<(), KvpError>, + ) -> KvpError { + let dir = TempDir::new().unwrap(); + let pool = store(&dir, PoolMode::Safe); + pool.append("existing", "value").unwrap(); + let before = fs::read(pool.path()).unwrap(); + let (writer, ops) = observed_writer(&dir); + let error = operation(&writer).unwrap_err(); + assert_eq!(ops.opens.load(Ordering::SeqCst), 0); + assert_eq!(fs::read(pool.path()).unwrap(), before); + error + } + + #[test] + fn constructor_does_no_io_and_emission_opens_once() { + let dir = TempDir::new().unwrap(); + let pool = store(&dir, PoolMode::Safe); + let (writer, ops) = observed_writer(&dir); + assert_eq!(ops.opens.load(Ordering::SeqCst), 0); + assert!(!pool.path().exists()); + + let payload = "x".repeat(MAX_CHUNK_BYTES * 3 + 1); + writer + .emit_event("test", payload, None, None, None) + .unwrap(); + assert_eq!(ops.opens.load(Ordering::SeqCst), 1); + assert_eq!(pool.dump().unwrap().len(), 4); + } + + #[test] + fn constructor_rejects_empty_agent_without_io() { + let dir = TempDir::new().unwrap(); + let (writer, ops) = observed_writer(&dir); + assert!(matches!( + DiagnosticWriter::new(writer.store, "", VM_ID), + Err(KvpError::EmptyEventField { field: "agent" }) + )); + assert_eq!(ops.opens.load(Ordering::SeqCst), 0); + } + + #[test] + fn constructor_rejects_invalid_vm_id_without_io() { + let dir = TempDir::new().unwrap(); + let (writer, ops) = observed_writer(&dir); + assert!(matches!( + DiagnosticWriter::new(writer.store, AGENT, "vm-abc"), + Err(KvpError::InvalidUuid { field: "vm_id" }) + )); + assert_eq!(ops.opens.load(Ordering::SeqCst), 0); + } + + #[rstest] + #[case("invalid")] + #[case("3f2504e0-4f89-41d3-9a0c-0305e82c330z")] + fn uuid_validation_rejects_invalid_syntax(#[case] value: &str) { + assert!(matches!( + validate_uuid("event_id", value), + Err(KvpError::InvalidUuid { field: "event_id" }) + )); + } + + #[test] + fn uuid_validation_preserves_field_errors() { + assert!(matches!( + validate_uuid("event_id", ""), + Err(KvpError::EmptyEventField { field: "event_id" }) + )); + assert!(matches!( + validate_uuid("event_id", "bad|id"), + Err(KvpError::EventFieldContainsDelimiter { field: "event_id" }) + )); + assert!(matches!( + validate_uuid("event_id", "bad\0id"), + Err(KvpError::KeyContainsNull) + )); + } + + #[test] + fn uuid_validation_rejects_oversized_representations() { + assert!(matches!( + validate_uuid("vm_id", &format!("{{{VM_ID}}}")), + Err(KvpError::EventFieldTooLong { + field: "vm_id", + max: MAX_UUID_BYTES, + actual: 38, + }) + )); + } + + #[rstest] + #[case(VM_ID)] + #[case("3F2504E0-4F89-41D3-9A0C-0305E82C3301")] + #[case("3f2504e04f8941d39a0c0305e82c3301")] + #[case("00000000-0000-0000-0000-000000000000")] + fn valid_uuid_spellings_are_preserved(#[case] id: &str) { + let dir = TempDir::new().unwrap(); + let pool = store(&dir, PoolMode::Safe); + let writer = DiagnosticWriter::new(pool.clone(), AGENT, id).unwrap(); + writer.emit_start(id, "test", "starting", None).unwrap(); + let records = pool.dump().unwrap(); + let fields: Vec<_> = records[0].0.split('|').collect(); + assert_eq!(fields[2], id); + assert_eq!(fields[5], id); + } + + #[rstest] + #[case("agent", "a".repeat(32), 32)] + #[case("agent", "é".repeat(16), 32)] + #[case("name", "n".repeat(48), 48)] + fn freeform_caps_count_bytes_without_truncation( + #[case] field: &'static str, + #[case] value: String, + #[case] max: usize, + #[values(PoolMode::Safe, PoolMode::Unsafe)] mode: PoolMode, + ) { + let dir = TempDir::new().unwrap(); + let pool = store(&dir, mode); + let (agent, name) = match field { + "agent" => (value.as_str(), "test"), + _ => (AGENT, value.as_str()), + }; + let writer = DiagnosticWriter::new(pool.clone(), agent, VM_ID).unwrap(); + writer.emit_event(name, "ok", None, None, None).unwrap(); + let records = pool.dump().unwrap(); + let fields: Vec<_> = records[0].0.split('|').collect(); + assert_eq!(fields[1], agent); + assert_eq!(fields[4], name); + + let before = fs::read(pool.path()).unwrap(); + let oversized = format!("{value}x"); + let error = match field { + "agent" => DiagnosticWriter::new(pool.clone(), oversized, VM_ID) + .unwrap_err(), + _ => writer + .emit_event(&oversized, "bad", None, None, None) + .unwrap_err(), + }; + assert!(matches!( + error, + KvpError::EventFieldTooLong { field: actual_field, max: cap, actual } + if actual_field == field && cap == max && actual == max + 1 + )); + assert_eq!(fs::read(pool.path()).unwrap(), before); + } + + #[rstest] + #[case("", "empty")] + #[case("bad|field", "delimiter")] + #[case("bad\0field", "null")] + fn field_validation_reports_the_reason( + #[case] value: &str, + #[case] reason: &str, + ) { + let error = validate_field("name", value, MAX_NAME_BYTES).unwrap_err(); + match reason { + "empty" => { + assert!(matches!( + error, + KvpError::EmptyEventField { field: "name" } + )) + } + "delimiter" => assert!(matches!( + error, + KvpError::EventFieldContainsDelimiter { field: "name" } + )), + _ => assert!(matches!(error, KvpError::KeyContainsNull)), + } + } + + #[test] + fn exact_start_key_matches_v1_layout() { + let records = prepare_records(Diagnostic::Start(DiagnosticStart { + key: key(), + payload: "starting".into(), + })) + .unwrap(); + assert_eq!( + records, + vec![( + format!( + "DIAG_V1|{AGENT}|{VM_ID}|start|provision:run|{EVENT_ID}|{TIMESTAMP}|none|||0" + ), + "starting".to_owned(), + )] + ); + } + + #[rstest] + #[case::success(Outcome::Success, "success", 312)] + #[case::failure(Outcome::Failure, "fail", 0)] + fn exact_finish_key_matches_v1_layout( + #[case] result: Outcome, + #[case] token: &str, + #[case] duration_ms: u64, + ) { + let records = prepare_records(Diagnostic::Finish(DiagnosticFinish { + key: key(), + payload: "finished".into(), + result, + duration_ms, + })) + .unwrap(); + assert_eq!( + records[0].0, + format!( + "DIAG_V1|{AGENT}|{VM_ID}|finish|provision:run|{EVENT_ID}|{TIMESTAMP}|none|{token}|{duration_ms}|0" + ) + ); + } + + #[rstest] + #[case::neither(None, None)] + #[case::result_only(Some(Outcome::Success), None)] + #[case::zero_duration(None, Some(0))] + #[case::both(Some(Outcome::Failure), Some(MAX_DURATION_MS))] + fn event_optional_fields_are_independent( + #[case] result: Option, + #[case] duration_ms: Option, + ) { + let dir = TempDir::new().unwrap(); + let pool = store(&dir, PoolMode::Safe); + let writer = DiagnosticWriter::new(pool.clone(), AGENT, VM_ID).unwrap(); + writer + .emit_event("test", "ok", None, result, duration_ms) + .unwrap(); + let records = pool.dump().unwrap(); + let fields: Vec<_> = records[0].0.split('|').collect(); + assert_eq!(fields.len(), 11); + assert_eq!(fields[3], "event"); + assert_eq!( + fields[8], + result.map_or_else(String::new, |v| v.to_string()) + ); + assert_eq!( + fields[9], + duration_ms.map_or_else(String::new, |v| v.to_string()) + ); + } + + #[test] + fn span_endpoints_reuse_the_supplied_event_id() { + let dir = TempDir::new().unwrap(); + let pool = store(&dir, PoolMode::Safe); + let writer = DiagnosticWriter::new(pool.clone(), AGENT, VM_ID).unwrap(); + writer + .emit_start(EVENT_ID, "test", "starting", None) + .unwrap(); + writer + .emit_finish( + EVENT_ID, + "test", + "finished", + None, + Outcome::Failure, + 17, + ) + .unwrap(); + let records = pool.dump().unwrap(); + assert_eq!(records.len(), 2); + for (record, kind) in records.iter().zip(["start", "finish"]) { + let fields: Vec<_> = record.0.split('|').collect(); + assert_eq!(fields[3], kind); + assert_eq!(fields[5], EVENT_ID); + } + } + + #[test] + fn events_receive_distinct_v4_ids() { + let dir = TempDir::new().unwrap(); + let pool = store(&dir, PoolMode::Safe); + let writer = DiagnosticWriter::new(pool.clone(), AGENT, VM_ID).unwrap(); + for _ in 0..2 { + writer + .emit_event("test", "message", None, None, None) + .unwrap(); + } + let ids: Vec<_> = pool + .dump() + .unwrap() + .iter() + .map(|(key, _)| { + Uuid::parse_str(key.split('|').nth(5).unwrap()).unwrap() + }) + .collect(); + assert_eq!(ids.len(), 2); + assert_ne!(ids[0], ids[1]); + for id in ids { + assert_eq!(id.get_version_num(), 4); + } + } + + #[test] + fn emission_uses_a_current_utc_millisecond_timestamp() { + let dir = TempDir::new().unwrap(); + let pool = store(&dir, PoolMode::Safe); + let writer = DiagnosticWriter::new(pool.clone(), AGENT, VM_ID).unwrap(); + let before = Utc::now().timestamp_millis(); + writer + .emit_event("test", "message", None, None, None) + .unwrap(); + let after = Utc::now().timestamp_millis(); + let records = pool.dump().unwrap(); + assert_eq!(records.len(), 1); + let timestamp = records[0].0.split('|').nth(6).unwrap(); + assert_eq!(timestamp.len(), 24); + assert!(timestamp.ends_with('Z')); + let parsed = DateTime::parse_from_rfc3339(timestamp).unwrap(); + assert!((before..=after).contains(&parsed.timestamp_millis())); + } + + #[rstest] + #[case(String::new(), vec![0])] + #[case("x".repeat(MAX_CHUNK_BYTES), vec![MAX_CHUNK_BYTES])] + #[case("x".repeat(MAX_CHUNK_BYTES + 1), vec![MAX_CHUNK_BYTES, 1])] + #[case(format!("{}é", "x".repeat(1021)), vec![1021, 2])] + #[case(format!("{}€", "x".repeat(1021)), vec![1021, 3])] + #[case(format!("{}😀", "x".repeat(1021)), vec![1021, 4])] + #[case("€".repeat(1000), vec![1020, 1020, 960])] + fn framing_preserves_utf8_and_empty_values( + #[case] value: String, + #[case] lengths: Vec, + ) { + let records = frame_records("base", &value).unwrap(); + assert_eq!( + records + .iter() + .map(|(_, value)| value.len()) + .collect::>(), + lengths + ); + assert_eq!( + records + .iter() + .map(|(_, value)| value.as_str()) + .collect::(), + value + ); + for (index, (key, value)) in records.iter().enumerate() { + assert_eq!(key, &format!("base|{index}")); + assert!(value.len() <= MAX_CHUNK_BYTES); + } + } + + #[rstest] + fn compression_precedes_safe_framing_in_both_modes( + #[values(PoolMode::Safe, PoolMode::Unsafe)] mode: PoolMode, + ) { + let dir = TempDir::new().unwrap(); + let pool = store(&dir, mode); + let writer = DiagnosticWriter::new(pool.clone(), AGENT, VM_ID).unwrap(); + let bytes: Vec<_> = (0..4096u32).flat_map(u32::to_le_bytes).collect(); + writer + .emit_event( + "provision:run", + bytes.clone(), + Some(Encoding::GzB64), + None, + None, + ) + .unwrap(); + let records = pool.dump().unwrap(); + assert!(records.len() > 1); + let base = records[0].0.rsplit_once('|').unwrap().0; + for (index, (key, value)) in records.iter().enumerate() { + assert_eq!(key, &format!("{base}|{index}")); + let fields: Vec<_> = key.split('|').collect(); + assert_eq!(fields[3], "event"); + assert_eq!(fields[7], "gz+b64"); + assert!(key.len() <= MAX_KEY_BYTES); + assert!(value.len() <= MAX_CHUNK_BYTES); + } + let value: String = records.iter().map(|(_, v)| v.as_str()).collect(); + assert_eq!( + decode_payload(value.as_bytes(), Some(&Encoding::GzB64)).unwrap(), + DiagnosticPayload::Bytes(bytes) + ); + } + + #[test] + fn maximum_chunk_count_is_accepted() { + let dir = TempDir::new().unwrap(); + let pool = store(&dir, PoolMode::Safe); + let writer = DiagnosticWriter::new(pool.clone(), AGENT, VM_ID).unwrap(); + let payload = "x".repeat(MAX_CHUNK_BYTES * MAX_CHUNKS); + writer + .emit_event("test", payload, None, None, None) + .unwrap(); + let records = pool.dump().unwrap(); + assert_eq!(records.len(), MAX_CHUNKS); + assert!(records.last().unwrap().0.ends_with("|1022")); + } + + #[rstest] + #[case("x".repeat(MAX_CHUNK_BYTES * MAX_CHUNKS + 1))] + #[case("€".repeat((MAX_CHUNK_BYTES / 3) * MAX_CHUNKS + 1))] + fn excess_chunks_do_not_open_or_modify_the_pool(#[case] payload: String) { + let dir = TempDir::new().unwrap(); + let pool = store(&dir, PoolMode::Safe); + pool.append("existing", "value").unwrap(); + let before = fs::read(pool.path()).unwrap(); + let (writer, ops) = observed_writer(&dir); + assert!(matches!( + writer.emit_event("test", payload, None, None, None), + Err(KvpError::TooManyChunks { max: MAX_CHUNKS }) + )); + assert_eq!(ops.opens.load(Ordering::SeqCst), 0); + assert_eq!(fs::read(pool.path()).unwrap(), before); + } + + #[test] + fn chunk_limit_applies_after_compression() { + let dir = TempDir::new().unwrap(); + let pool = store(&dir, PoolMode::Safe); + let writer = DiagnosticWriter::new(pool.clone(), AGENT, VM_ID).unwrap(); + let bytes = vec![0; MAX_CHUNK_BYTES * MAX_CHUNKS + 1]; + writer + .emit_event( + "test", + bytes.clone(), + Some(Encoding::GzB64), + None, + None, + ) + .unwrap(); + let records = pool.dump().unwrap(); + assert!(records.len() < MAX_CHUNKS); + let value: String = records.iter().map(|(_, v)| v.as_str()).collect(); + assert_eq!( + decode_payload(value.as_bytes(), Some(&Encoding::GzB64)).unwrap(), + DiagnosticPayload::Bytes(bytes) + ); + } + + #[test] + fn key_limit_includes_every_chunk_suffix() { + let base = "x".repeat(MAX_KEY_BYTES - 2); + let records = frame_records(&base, "").unwrap(); + assert_eq!(records[0].0.len(), MAX_KEY_BYTES); + let payload = "v".repeat(MAX_CHUNK_BYTES * 11); + assert!(matches!( + frame_records(&base, &payload), + Err(KvpError::KeyTooLarge { + max: MAX_KEY_BYTES, + actual: 255 + }) + )); + } + + #[test] + fn largest_permitted_fields_fit_with_four_digit_index() { + let diagnostic = Diagnostic::Finish(DiagnosticFinish { + key: DiagnosticKey { + agent: "a".repeat(MAX_AGENT_BYTES), + name: "n".repeat(MAX_NAME_BYTES), + encoding: Some(Encoding::GzB64), + ..key() + }, + payload: "test".into(), + result: Outcome::Success, + duration_ms: MAX_DURATION_MS, + }); + let records = prepare_records(diagnostic).unwrap(); + let base = records[0].0.rsplit_once('|').unwrap().0; + let longest = format!("{base}|1022"); + assert_eq!(longest.len(), 226); + assert!(longest.len() <= MAX_KEY_BYTES); + } + + #[test] + fn start_rejects_invalid_event_id_before_io() { + let error = assert_rejected_without_writes(|writer| { + writer.emit_start("invalid", "test", "message", None) + }); + assert!(matches!(error, KvpError::InvalidUuid { field: "event_id" })); + } + + #[test] + fn finish_rejects_invalid_name_before_io() { + let error = assert_rejected_without_writes(|writer| { + writer.emit_finish( + EVENT_ID, + "bad|name", + "message", + None, + Outcome::Failure, + 0, + ) + }); + assert!(matches!( + error, + KvpError::EventFieldContainsDelimiter { field: "name" } + )); + } + + #[test] + fn event_rejects_empty_name_before_io() { + let error = assert_rejected_without_writes(|writer| { + writer.emit_event("", "message", None, None, None) + }); + assert!(matches!(error, KvpError::EmptyEventField { field: "name" })); + } + + #[test] + fn start_rejects_invalid_utf8_before_io() { + let error = assert_rejected_without_writes(|writer| { + writer.emit_start(EVENT_ID, "test", vec![0xff], None) + }); + assert!(matches!(error, KvpError::PayloadNotUtf8)); + } + + #[test] + fn finish_rejects_late_nul_before_writing_any_chunks() { + let payload = format!("{}\0", "x".repeat(MAX_CHUNK_BYTES)); + let error = assert_rejected_without_writes(|writer| { + writer.emit_finish( + EVENT_ID, + "test", + payload, + None, + Outcome::Failure, + 0, + ) + }); + assert!(matches!(error, KvpError::ValueContainsNull)); + } + + #[test] + fn event_rejects_unsupported_encoding_before_io() { + let error = assert_rejected_without_writes(|writer| { + writer.emit_event( + "test", + "message", + Some(Encoding::Other("zstd".into())), + None, + None, + ) + }); + assert!(matches!( + error, + KvpError::UnsupportedEncoding { token } if token == "zstd" + )); + } + + #[rstest] + fn excessive_durations_are_rejected_before_io( + #[values(MAX_DURATION_MS + 1, u64::MAX)] duration_ms: u64, + #[values(Kind::Finish, Kind::Event)] kind: Kind, + ) { + let dir = TempDir::new().unwrap(); + let (writer, ops) = observed_writer(&dir); + let result = match kind { + Kind::Finish => writer.emit_finish( + EVENT_ID, + "test", + "bad", + None, + Outcome::Failure, + duration_ms, + ), + _ => { + writer.emit_event("test", "bad", None, None, Some(duration_ms)) + } + }; + assert!(matches!( + result, + Err(KvpError::DurationTooLarge { max_ms: MAX_DURATION_MS, actual_ms }) + if actual_ms == duration_ms + )); + assert_eq!(ops.opens.load(Ordering::SeqCst), 0); + } + + #[test] + fn writer_rejects_absent_vm_identity() { + let mut missing_vm = key(); + missing_vm.vm_id = None; + assert!(matches!( + prepare_records(event(missing_vm, "bad".into())), + Err(KvpError::EmptyEventField { field: "vm_id" }) + )); + } + + #[test] + fn writer_rejects_expanded_timestamp_years() { + let mut expanded_year = key(); + expanded_year.timestamp = + DateTime::from_timestamp(253_402_300_800, 0).unwrap(); + assert!(matches!( + prepare_records(event(expanded_year, "bad".into())), + Err(KvpError::EventFieldTooLong { + field: "timestamp", + max: 24, + .. + }) + )); + } + + #[test] + fn appends_preserve_existing_records_and_duplicates() { + let dir = TempDir::new().unwrap(); + let pool = store(&dir, PoolMode::Safe); + pool.append("existing", "first").unwrap(); + pool.append("existing", "second").unwrap(); + let writer = DiagnosticWriter::new(pool.clone(), AGENT, VM_ID).unwrap(); + writer.emit_event("test", "new", None, None, None).unwrap(); + let records = pool.dump().unwrap(); + assert_eq!(records.len(), 3); + assert_eq!(records[0], ("existing".into(), "first".into())); + assert_eq!(records[1], ("existing".into(), "second".into())); + } + + #[test] + fn constructor_defers_storage_errors_until_emission() { + let dir = TempDir::new().unwrap(); + let pool = store(&dir, PoolMode::Safe); + fs::create_dir(pool.path()).unwrap(); + let writer = DiagnosticWriter::new(pool, AGENT, VM_ID).unwrap(); + assert!(matches!( + writer.emit_event("test", "value", None, None, None), + Err(KvpError::Io(_)) + )); + } +} diff --git a/libazureinit-kvp/src/error.rs b/libazureinit-kvp/src/error.rs index 405913e4..db8fd060 100644 --- a/libazureinit-kvp/src/error.rs +++ b/libazureinit-kvp/src/error.rs @@ -4,26 +4,58 @@ use std::fmt; use std::io; -/// Errors returned by [`KvpPoolStore`](crate::KvpPoolStore) operations. +/// Errors returned by KVP storage and diagnostic writing. #[derive(Debug)] pub enum KvpError { /// The key was empty. EmptyKey, + EmptyEventField { + field: &'static str, + }, /// An underlying I/O error. Io(io::Error), /// An event key field (`agent`, `vm_id`, `kind`, `name`, or `event_id`) /// contained the `|` delimiter, which would make the formatted event /// key ambiguous to parse back. - EventFieldContainsDelimiter { field: &'static str }, + EventFieldContainsDelimiter { + field: &'static str, + }, + EventFieldTooLong { + field: &'static str, + max: usize, + actual: usize, + }, + InvalidUuid { + field: &'static str, + }, + DurationTooLarge { + max_ms: u64, + actual_ms: u64, + }, + TooManyChunks { + max: usize, + }, /// The key contains a null byte, which is incompatible with the /// on-disk format (null-padded fixed-width fields). KeyContainsNull, /// The key exceeds the store's maximum key size. - KeyTooLarge { max: usize, actual: usize }, + KeyTooLarge { + max: usize, + actual: usize, + }, /// The store already has the maximum allowed number of unique keys. - MaxUniqueKeysExceeded { max: usize }, + MaxUniqueKeysExceeded { + max: usize, + }, + PayloadNotUtf8, + UnsupportedEncoding { + token: String, + }, /// The value exceeds the store's maximum value size. - ValueTooLarge { max: usize, actual: usize }, + ValueTooLarge { + max: usize, + actual: usize, + }, /// The value contains a null byte, which is incompatible with the /// null-padded KVP wire format. ValueContainsNull, @@ -33,9 +65,24 @@ impl fmt::Display for KvpError { fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { match self { Self::EmptyKey => write!(f, "KVP key must not be empty"), + Self::EmptyEventField { field } => { + write!(f, "event key field '{field}' must not be empty") + } Self::EventFieldContainsDelimiter { field } => { write!(f, "event key field '{field}' must not contain '|'") } + Self::EventFieldTooLong { field, max, actual } => { + write!(f, "event key field '{field}' length ({actual}) exceeds maximum ({max})") + } + Self::InvalidUuid { field } => { + write!(f, "event key field '{field}' must be a UUID") + } + Self::DurationTooLarge { max_ms, actual_ms } => { + write!(f, "diagnostic duration ({actual_ms}ms) exceeds maximum ({max_ms}ms)") + } + Self::TooManyChunks { max } => { + write!(f, "diagnostic chunk count exceeds maximum ({max})") + } Self::Io(e) => write!(f, "{e}"), Self::KeyContainsNull => { write!(f, "KVP key must not contain null bytes") @@ -46,6 +93,12 @@ impl fmt::Display for KvpError { Self::MaxUniqueKeysExceeded { max } => { write!(f, "KVP unique key count exceeded maximum ({max})") } + Self::PayloadNotUtf8 => { + write!(f, "diagnostic payload must be valid UTF-8 for encoding 'none'") + } + Self::UnsupportedEncoding { token } => { + write!(f, "diagnostic encoding '{token}' is not supported") + } Self::ValueTooLarge { max, actual } => { write!(f, "KVP value length ({actual}) exceeds maximum ({max})") } @@ -70,3 +123,47 @@ impl From for KvpError { Self::Io(err) } } + +#[cfg(test)] +mod tests { + use super::*; + use rstest::rstest; + + #[rstest] + #[case( + KvpError::EmptyEventField { field: "name" }, + "event key field 'name' must not be empty" + )] + #[case( + KvpError::EventFieldTooLong { field: "name", max: 48, actual: 49 }, + "event key field 'name' length (49) exceeds maximum (48)" + )] + #[case( + KvpError::InvalidUuid { field: "vm_id" }, + "event key field 'vm_id' must be a UUID" + )] + #[case( + KvpError::DurationTooLarge { max_ms: 9_999_999_999, actual_ms: 10_000_000_000 }, + "diagnostic duration (10000000000ms) exceeds maximum (9999999999ms)" + )] + #[case( + KvpError::TooManyChunks { max: 1023 }, + "diagnostic chunk count exceeds maximum (1023)" + )] + #[case( + KvpError::PayloadNotUtf8, + "diagnostic payload must be valid UTF-8 for encoding 'none'" + )] + #[case( + KvpError::UnsupportedEncoding { token: "zstd+b64".into() }, + "diagnostic encoding 'zstd+b64' is not supported" + )] + fn diagnostic_validation_errors_explain_the_failure( + #[case] error: KvpError, + #[case] expected: &str, + ) { + assert_eq!(error.to_string(), expected); + let error: &dyn std::error::Error = &error; + assert!(error.source().is_none()); + } +} diff --git a/libazureinit-kvp/src/lib.rs b/libazureinit-kvp/src/lib.rs index 204d68ed..c50d0e08 100644 --- a/libazureinit-kvp/src/lib.rs +++ b/libazureinit-kvp/src/lib.rs @@ -9,8 +9,40 @@ //! - [`ProvisioningReport`]: structured provisioning health report that //! is persisted as the single `PROVISIONING_REPORT` record with //! [`write_report`]. -//! - [`DiagnosticsKvp`]: typed writer for azure-init diagnostics and normalized -//! reader for azure-init and cloud-init entries. +//! - [`DiagnosticWriter`]: typed writer for versioned diagnostics. +//! - [`DiagnosticReader`]: reader for diagnostics, provisioning reports, and +//! raw records, including a read-only cloud-init compatibility bridge. +//! +//! # Diagnostics +//! +//! The reader preserves first-seen pool order. Unknown or invalid records +//! remain [`Entry::Raw`] within a successful snapshot; a failed snapshot, +//! including invalid physical UTF-8, returns an error without entries. +//! The CLI's `dump --parse` sorts diagnostics and reports by timestamp, +//! oldest first, with stable ties and raw entries last. +//! +//! ```no_run +//! use libazureinit_kvp::{ +//! DiagnosticReader, DiagnosticWriter, KvpPool, KvpPoolStore, Outcome, +//! PoolMode, +//! }; +//! +//! # fn main() -> Result<(), Box> { +//! let store = KvpPoolStore::new(KvpPool::Guest, PoolMode::Safe)?; +//! let writer = DiagnosticWriter::new( +//! store.clone(), +//! "azure-init", +//! "3f2504e0-4f89-41d3-9a0c-0305e82c3301", +//! )?; +//! writer.emit_event( +//! "imds", "metadata retrieved", None, Some(Outcome::Success), None, +//! )?; +//! +//! let entries = DiagnosticReader::new(store).entries()?; +//! println!("{}", serde_json::to_string(&entries)?); +//! # Ok(()) +//! # } +//! ``` mod cli; mod diagnostics; @@ -21,7 +53,10 @@ mod vm_id; pub use cli::run; pub use diagnostics::{ - DiagnosticEvent, DiagnosticKind, DiagnosticsKvp, MAX_CHUNK_BYTES, + DecodeError, Diagnostic, DiagnosticEvent, DiagnosticFinish, DiagnosticKey, + DiagnosticPayload, DiagnosticReader, DiagnosticStart, DiagnosticWriter, + Encoding, Entry, Kind, Outcome, RawKeyValue, DIAGNOSTIC_VERSION_ID, + MAX_CHUNK_BYTES, }; pub use error::KvpError; pub use report::{ diff --git a/libazureinit-kvp/src/report.rs b/libazureinit-kvp/src/report.rs index 43b3a829..ae0b307c 100644 --- a/libazureinit-kvp/src/report.rs +++ b/libazureinit-kvp/src/report.rs @@ -2,7 +2,7 @@ // Licensed under the MIT License. //! Structured provisioning report abstraction layered over the raw -//! [`KvpPoolStore`](crate::KvpPoolStore) key/value API. +//! [`KvpPoolStore`] key/value API. //! //! [`ProvisioningReport`] is a strongly-typed representation of a //! provisioning health report. Instead of building ad-hoc key/value @@ -11,9 +11,11 @@ //! pipe-delimited `PROVISIONING_REPORT` KVP record that the Azure/Hyper-V //! host parses. -use chrono::Utc; +use std::str::FromStr; -use crate::{KvpError, KvpPoolStore}; +use chrono::{DateTime, Utc}; + +use crate::{DecodeError, KvpError, KvpPoolStore}; /// KVP key under which the encoded provisioning health report is stored. /// @@ -28,7 +30,8 @@ fn now_rfc3339() -> String { } /// Outcome of a provisioning attempt. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] +#[derive(Clone, Copy, Debug, PartialEq, Eq, serde::Serialize)] +#[serde(rename_all = "lowercase")] enum ReportResult { /// Provisioning completed successfully. Success, @@ -56,11 +59,12 @@ impl std::fmt::Display for ReportResult { /// /// Mirrors the values cloud-init reports for the platform's /// `PreprovisionedVMType` / IMDS `ppsType`. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] +#[derive(Clone, Copy, Debug, PartialEq, Eq, serde::Serialize)] pub enum ReportPpsType { /// Not pre-provisioned (`None`). None, /// Pre-provisioned OS disk (`PreprovisionedOSDisk`). + #[serde(rename = "PreprovisionedOSDisk")] OsDisk, /// Running pre-provisioning (`Running`). Running, @@ -71,6 +75,17 @@ pub enum ReportPpsType { } impl ReportPpsType { + fn from_wire(value: &str) -> Result { + match value { + "None" => Ok(Self::None), + "PreprovisionedOSDisk" => Ok(Self::OsDisk), + "Running" => Ok(Self::Running), + "Savable" => Ok(Self::Savable), + "Unknown" => Ok(Self::Unknown), + _ => Err(DecodeError::Malformed), + } + } + /// The wire string used in the `pps_type` KVP field. fn as_str(self) -> &'static str { match self { @@ -116,7 +131,7 @@ impl std::fmt::Display for ReportPpsType { /// # Ok(()) /// # } /// ``` -#[derive(Clone, Debug, PartialEq, Eq)] +#[derive(Clone, Debug, PartialEq, Eq, serde::Serialize)] pub struct ProvisioningReport { /// Provisioning outcome (`result` field). result: ReportResult, @@ -130,10 +145,13 @@ pub struct ProvisioningReport { /// Pre-provisioning type (`pps_type` field). pps_type: ReportPpsType, /// Failure reason (`reason` field). Present for error reports. + #[serde(skip_serializing_if = "Option::is_none")] reason: Option, /// Documentation URL (`documentation_url` field), if applicable. + #[serde(skip_serializing_if = "Option::is_none")] documentation_url: Option, /// Additional ordered key/value context (e.g. supporting data). + #[serde(skip_serializing_if = "Vec::is_empty")] extra: Vec<(String, String)>, } @@ -191,6 +209,135 @@ impl ProvisioningReport { self.extra.push((key.into(), value.into())); self } + + pub(crate) fn timestamp(&self) -> DateTime { + DateTime::parse_from_rfc3339(&self.timestamp) + .expect( + "report timestamps are constructed or validated as RFC 3339", + ) + .with_timezone(&Utc) + } +} + +impl FromStr for ProvisioningReport { + type Err = DecodeError; + + /// Parses one report without generating a timestamp or accessing storage. + fn from_str(value: &str) -> Result { + let value = value + .strip_suffix("\r\n") + .or_else(|| value.strip_suffix('\n')) + .unwrap_or(value); + validate_report_quoting(value)?; + let mut reader = csv::ReaderBuilder::new() + .delimiter(b'|') + .has_headers(false) + .from_reader(value.as_bytes()); + let record = reader + .records() + .next() + .ok_or(DecodeError::Malformed)? + .map_err(|_| DecodeError::Malformed)?; + let mut fields = record + .iter() + .map(|field| { + let (key, value) = + field.split_once('=').ok_or(DecodeError::Malformed)?; + if key.is_empty() { + return Err(DecodeError::Malformed); + } + Ok((key.to_owned(), value.to_owned())) + }) + .collect::, DecodeError>>()?; + + let result = match take_field(&mut fields, "result")? + .ok_or(DecodeError::Malformed)? + .as_str() + { + "success" => ReportResult::Success, + "error" => ReportResult::Error, + _ => return Err(DecodeError::Malformed), + }; + let agent = + take_field(&mut fields, "agent")?.ok_or(DecodeError::Malformed)?; + let vm_id = + take_field(&mut fields, "vm_id")?.ok_or(DecodeError::Malformed)?; + let timestamp = take_field(&mut fields, "timestamp")? + .ok_or(DecodeError::Malformed)?; + DateTime::parse_from_rfc3339(×tamp) + .map_err(|_| DecodeError::Malformed)?; + let pps_type = ReportPpsType::from_wire( + &take_field(&mut fields, "pps_type")? + .ok_or(DecodeError::Malformed)?, + )?; + let (reason, documentation_url) = match result { + ReportResult::Success => (None, None), + ReportResult::Error => ( + Some( + take_field(&mut fields, "reason")? + .ok_or(DecodeError::Malformed)?, + ), + take_field(&mut fields, "documentation_url")?, + ), + }; + Ok(Self { + result, + agent, + vm_id, + timestamp, + pps_type, + reason, + documentation_url, + extra: fields, + }) + } +} + +fn take_field( + fields: &mut Vec<(String, String)>, + name: &str, +) -> Result, DecodeError> { + let mut matches = fields + .iter() + .enumerate() + .filter(|(_, (key, _))| key == name); + let position = matches.next().map(|(index, _)| index); + if matches.next().is_some() { + return Err(DecodeError::Malformed); + } + Ok(position.map(|index| fields.remove(index).1)) +} + +fn validate_report_quoting(value: &str) -> Result<(), DecodeError> { + enum State { + Start, + Unquoted, + Quoted, + Closed, + } + use State::*; + + // The CSV reader tolerates broken quoting, but reports must be unambiguous. + let mut state = Start; + for byte in value.bytes() { + if byte == 0 { + return Err(DecodeError::Malformed); + } + state = match (state, byte) { + (Start | Closed, b'"') => Quoted, + (Quoted, b'"') => Closed, + (Quoted, _) => Quoted, + (Start | Unquoted | Closed, b'|') => Start, + (Closed, _) | (_, b'"' | b'\r' | b'\n') => { + return Err(DecodeError::Malformed); + } + _ => Unquoted, + }; + } + if matches!(state, Quoted) { + return Err(DecodeError::Malformed); + } + Ok(()) } impl ProvisioningReport { @@ -201,7 +348,7 @@ impl ProvisioningReport { /// - Failure: `result`, `reason`, `agent`, extras in insertion /// order, `pps_type`, `vm_id`, `timestamp`, then /// `documentation_url` (if any). - fn encode(&self) -> String { + pub(crate) fn encode(&self) -> String { let mut data = Vec::with_capacity(7 + self.extra.len()); data.push(format!("result={}", self.result)); @@ -284,6 +431,15 @@ mod tests { report } + fn success_wire() -> String { + with_ts(ProvisioningReport::success( + AGENT, + VM_ID, + ReportPpsType::None, + )) + .encode() + } + #[rstest] #[case::success( with_ts(ProvisioningReport::success(AGENT, VM_ID, ReportPpsType::None)), @@ -326,11 +482,12 @@ mod tests { )), "result=error|reason=boom|agent=Azure-Init/0.0.0|pps_type=None|vm_id=00000000-0000-0000-0000-000000000abc|timestamp=2026-06-17T00:00:00+00:00", )] - fn encode_emits_expected_pipe_string( + fn report_wire_format_round_trips( #[case] report: ProvisioningReport, #[case] expected: &str, ) { assert_eq!(report.encode(), expected); + assert_eq!(expected.parse::().unwrap(), report); } /// Pins each [`ReportPpsType`] variant to its exact wire string. @@ -340,11 +497,65 @@ mod tests { #[case(ReportPpsType::Running, "Running")] #[case(ReportPpsType::Savable, "Savable")] #[case(ReportPpsType::Unknown, "Unknown")] - fn pps_type_display_matches_wire_string( + fn pps_type_wire_tokens_match_the_model( #[case] pps_type: ReportPpsType, #[case] expected: &str, ) { assert_eq!(pps_type.to_string(), expected); + assert_eq!(serde_json::to_value(pps_type).unwrap(), expected); + assert_eq!(ReportPpsType::from_wire(expected).unwrap(), pps_type); + } + + #[test] + fn success_serializes_without_absent_fields() { + let report = with_ts(ProvisioningReport::success( + AGENT, + VM_ID, + ReportPpsType::None, + )); + assert_eq!( + serde_json::to_value(report).unwrap(), + serde_json::json!({ + "result": "success", + "agent": AGENT, + "vm_id": VM_ID, + "timestamp": TS, + "pps_type": "None", + }) + ); + } + + #[test] + fn failure_serialization_preserves_ordered_extras() { + let report = with_ts( + ProvisioningReport::failure( + AGENT, + VM_ID, + "boom", + ReportPpsType::OsDisk, + ) + .with_extra("detail", "first") + .with_extra("detail", "second") + .with_extra("result", "extra context") + .with_documentation_url("https://aka.ms/linuxprovisioningerror"), + ); + assert_eq!( + serde_json::to_value(report).unwrap(), + serde_json::json!({ + "result": "error", + "agent": AGENT, + "vm_id": VM_ID, + "timestamp": TS, + "pps_type": "PreprovisionedOSDisk", + "reason": "boom", + "documentation_url": "https://aka.ms/linuxprovisioningerror", + "extra": [ + ["detail", "first"], + ["detail", "second"], + ["result", "extra context"], + ], + }) + ); } /// The success layout lists the standard fields first, then any @@ -452,4 +663,180 @@ mod tests { let result = write_report(&store, &report); assert!(result.is_err()); } + + #[test] + fn parses_reordered_fields_without_normalizing_identity_or_timestamp() { + let timestamp = "2026-06-17T02:00:00.123456+02:00"; + let value = format!( + "timestamp={timestamp}|vm_id=vm-abc|pps_type=None|agent={AGENT}|result=success" + ); + let mut expected = + ProvisioningReport::success(AGENT, "vm-abc", ReportPpsType::None); + expected.timestamp = timestamp.into(); + assert_eq!(value.parse::().unwrap(), expected); + } + + #[test] + fn failure_round_trips_quoted_fields_and_documentation_url() { + let report = with_ts( + ProvisioningReport::failure( + AGENT, + VM_ID, + "failed | \"quoted\"\r\nnext line", + ReportPpsType::Running, + ) + .with_extra("context", "key=value|details\nmore") + .with_documentation_url("https://example.invalid/?key=a=b"), + ); + assert_eq!( + report.encode().parse::().unwrap(), + report + ); + } + + #[test] + fn supporting_data_preserves_order_duplicates_and_empty_values() { + let value = + format!("{}|detail=first|detail=second|empty=", success_wire()); + let report = value.parse::().unwrap(); + assert_eq!( + report.extra, + vec![ + ("detail".into(), "first".into()), + ("detail".into(), "second".into()), + ("empty".into(), String::new()), + ] + ); + } + + #[test] + fn success_keeps_failure_only_fields_as_supporting_data() { + let report = with_ts( + ProvisioningReport::success(AGENT, VM_ID, ReportPpsType::None) + .with_extra("reason", "additional context") + .with_extra("documentation_url", "https://example.invalid/"), + ); + assert_eq!( + report.encode().parse::().unwrap(), + report + ); + } + + #[rstest] + #[case::result("result")] + #[case::agent("agent")] + #[case::vm_id("vm_id")] + #[case::timestamp("timestamp")] + #[case::pps_type("pps_type")] + fn report_requires_each_standard_field(#[case] missing: &str) { + let value = success_wire() + .split('|') + .filter(|field| !field.starts_with(&format!("{missing}="))) + .collect::>() + .join("|"); + assert_eq!( + value.parse::(), + Err(DecodeError::Malformed) + ); + } + + #[test] + fn failure_requires_a_reason() { + let value = success_wire().replace("result=success", "result=error"); + assert_eq!( + value.parse::(), + Err(DecodeError::Malformed) + ); + } + + #[rstest] + #[case::success(with_ts(ProvisioningReport::success( + "", + "", + ReportPpsType::None + )))] + #[case::failure(with_ts(ProvisioningReport::failure( + AGENT, + VM_ID, + "", + ReportPpsType::None + )))] + fn empty_values_supported_by_the_writer_remain_readable( + #[case] report: ProvisioningReport, + ) { + assert_eq!( + report.encode().parse::().unwrap(), + report + ); + } + + #[rstest] + #[case::result("result=success", "result=fail")] + #[case::pps_type("pps_type=None", "pps_type=FutureType")] + #[case::timestamp(TS, "not-a-timestamp")] + fn invalid_standard_values_are_malformed( + #[case] from: &str, + #[case] to: &str, + ) { + let value = success_wire().replace(from, to); + assert_eq!( + value.parse::(), + Err(DecodeError::Malformed) + ); + } + + #[rstest] + #[case::unclosed_quote("\"extra=value")] + #[case::characters_after_quote("\"extra=value\"suffix")] + #[case::unquoted_quote("extra=va\"lue")] + #[case::missing_equals("extra")] + #[case::empty_key("=value")] + #[case::null("extra=va\0lue")] + fn malformed_supporting_fields_are_not_silently_repaired( + #[case] extra: &str, + ) { + let value = format!("{}|{extra}", success_wire()); + assert_eq!( + value.parse::(), + Err(DecodeError::Malformed) + ); + } + + #[test] + fn multiple_csv_records_are_rejected() { + let value = format!("{}\n{}", success_wire(), success_wire()); + assert_eq!( + value.parse::(), + Err(DecodeError::Malformed) + ); + } + + #[test] + fn optional_csv_quotes_and_record_terminators_are_accepted() { + let quoted = success_wire() + .split('|') + .map(|field| format!("\"{field}\"")) + .collect::>() + .join("|"); + let expected = success_wire().parse::().unwrap(); + for ending in ["", "\n", "\r\n"] { + assert_eq!( + format!("{quoted}{ending}") + .parse::() + .unwrap(), + expected + ); + } + } + + #[rstest] + #[case::conflicting_result("result=error")] + #[case::repeated_agent("agent=Azure-Init/0.0.0")] + fn duplicate_standard_fields_are_ambiguous(#[case] duplicate: &str) { + let value = format!("{}|{duplicate}", success_wire()); + assert_eq!( + value.parse::(), + Err(DecodeError::Malformed) + ); + } } diff --git a/libazureinit-kvp/src/store.rs b/libazureinit-kvp/src/store.rs index 156228fc..05404027 100644 --- a/libazureinit-kvp/src/store.rs +++ b/libazureinit-kvp/src/store.rs @@ -211,7 +211,7 @@ impl KvpPoolStore { let boot_time = boot_time(&*self.ops)?; lock_for_writing(&mut *handle)?; - if handle.metadata()?.mtime < boot_time { + if handle.metadata()?.mtime <= boot_time { handle.set_len(0)?; } } @@ -399,18 +399,7 @@ impl KvpPoolStore { Err(ref e) if e.kind() == ErrorKind::NotFound => return Ok(false), Err(e) => return Err(e.into()), }; - Ok(metadata.mtime < boot_time(&*self.ops)?) - } - - /// The system boot time as a Unix epoch timestamp in seconds, read - /// from `/proc/stat` `btime`. - /// - /// Diagnostic event keys stamp this value so records can be - /// attributed to a specific boot (mirroring cloud-init's - /// incarnation), letting readers tell this boot's telemetry from a - /// previous boot's. - pub fn boot_epoch(&self) -> Result { - boot_time(&*self.ops) + Ok(metadata.mtime <= boot_time(&*self.ops)?) } /// Variant of [`is_stale`](Self::is_stale) that takes an explicit @@ -423,7 +412,7 @@ impl KvpPoolStore { Err(ref e) if e.kind() == ErrorKind::NotFound => return Ok(false), Err(e) => return Err(e.into()), }; - Ok(metadata.mtime < boot_time) + Ok(metadata.mtime <= boot_time) } fn iter(&self) -> Result { @@ -1249,6 +1238,18 @@ mod tests { KvpErrKind::ValueContainsNull )] #[case::load_empty_key(WriteOp::Load, "", "v", KvpErrKind::EmptyKey)] + #[case::append_multiple_empty_key( + WriteOp::AppendMultiple, + "", + "v", + KvpErrKind::EmptyKey + )] + #[case::append_multiple_null_key( + WriteOp::AppendMultiple, + "bad\0key", + "v", + KvpErrKind::KeyContainsNull + )] fn test_write_rejects_invalid_input( #[case] op: WriteOp, #[case] key: &str, @@ -1261,9 +1262,13 @@ mod tests { WriteOp::Insert => store.insert(key, value), WriteOp::Append => store.append(key, value), WriteOp::Load => store.load(pairs([(key, value)])), + WriteOp::AppendMultiple => { + store.append_multiple(pairs([(key, value)])) + } } .unwrap_err(); assert!(expected.matches(&err), "got: {err:?}"); + assert!(!store.path().exists(), "no file on invalid input"); } #[derive(Clone, Copy)] @@ -1283,6 +1288,7 @@ mod tests { Insert, Append, Load, + AppendMultiple, } #[derive(Clone, Copy)] @@ -1599,6 +1605,16 @@ mod tests { "bad\0key", KvpErrKind::KeyContainsNull )] + #[case::delete_multiple_empty( + ReadOp::DeleteMultiple, + "", + KvpErrKind::EmptyKey + )] + #[case::delete_multiple_null( + ReadOp::DeleteMultiple, + "bad\0key", + KvpErrKind::KeyContainsNull + )] fn test_bad_key_is_rejected( #[case] op: ReadOp, #[case] bad_key: &str, @@ -1611,6 +1627,9 @@ mod tests { let err = match op { ReadOp::Read => store.read(bad_key).unwrap_err(), ReadOp::Delete => store.delete(bad_key).unwrap_err(), + ReadOp::DeleteMultiple => { + store.delete_multiple(vec![bad_key]).unwrap_err() + } }; assert!(expected.matches(&err), "got {err:?}"); assert_eq!(store.read("k1").unwrap(), Some("v1".to_string())); @@ -1621,6 +1640,7 @@ mod tests { enum ReadOp { Read, Delete, + DeleteMultiple, } #[test] @@ -1897,29 +1917,6 @@ mod tests { assert_eq!(store.dump().unwrap(), pairs([("keep", "me")])); } - #[test] - fn test_append_multiple_rejects_empty_key() { - let dir = TempDir::new().unwrap(); - let store = safe_store(dir.path()); - - let err = store - .append_multiple(pairs([("ok", "v"), ("", "bad")])) - .unwrap_err(); - assert!(matches!(err, KvpError::EmptyKey), "got {err:?}"); - assert!(!store.path().exists()); - } - - #[test] - fn test_append_multiple_rejects_null_in_key() { - let dir = TempDir::new().unwrap(); - let store = safe_store(dir.path()); - - let err = store - .append_multiple(pairs([("ok\0bad", "v")])) - .unwrap_err(); - assert!(matches!(err, KvpError::KeyContainsNull), "got {err:?}"); - } - #[test] fn test_append_multiple_does_not_enforce_unique_key_cap() { let dir = TempDir::new().unwrap(); @@ -2029,31 +2026,6 @@ mod tests { assert_eq!(store.dump().unwrap(), pairs([("b", "2")])); } - #[test] - fn test_delete_multiple_rejects_empty_key() { - let dir = TempDir::new().unwrap(); - let store = safe_store(dir.path()); - store.load(pairs([("a", "1")])).unwrap(); - - let err = store - .delete_multiple(vec!["a".to_string(), "".to_string()]) - .unwrap_err(); - assert!(matches!(err, KvpError::EmptyKey), "got {err:?}"); - - assert_eq!(store.dump().unwrap(), pairs([("a", "1")])); - } - - #[test] - fn test_delete_multiple_rejects_null_in_key() { - let dir = TempDir::new().unwrap(); - let store = safe_store(dir.path()); - store.load(pairs([("a", "1")])).unwrap(); - - let err = store.delete_multiple(vec!["bad\0key"]).unwrap_err(); - assert!(matches!(err, KvpError::KeyContainsNull), "got {err:?}"); - assert_eq!(store.dump().unwrap(), pairs([("a", "1")])); - } - #[test] fn test_delete_multiple_size_independent_of_mode() { let dir = TempDir::new().unwrap(); @@ -3473,17 +3445,6 @@ mod tests { assert_eq!(ops.lock().files.get(&p).unwrap().len(), RECORD_SIZE); } - #[test] - fn test_clear_if_stale_keeps_file_written_in_boot_second() { - let (store, ops, p) = mock_store(PoolMode::Safe); - ops.put_file(&p, vec![0u8; RECORD_SIZE], 10); - ops.set_boot_time(10); - - assert!(!store.is_stale().unwrap()); - store.clear_if_stale().unwrap(); - assert_eq!(ops.lock().files.get(&p).unwrap().len(), RECORD_SIZE); - } - #[test] fn test_delete_fails_when_iter_read_fails() { let (store, ops, p) = mock_store(PoolMode::Safe); diff --git a/libazureinit-kvp/tests/cli.rs b/libazureinit-kvp/tests/cli.rs index 2f4b90c5..fc91ef4a 100644 --- a/libazureinit-kvp/tests/cli.rs +++ b/libazureinit-kvp/tests/cli.rs @@ -5,8 +5,16 @@ use std::fs; use std::io::Write; use std::process::{Command, Output}; +use libazureinit_kvp::{ + DiagnosticWriter, Encoding, KvpPool, KvpPoolStore, Outcome, PoolMode, + PROVISIONING_REPORT_KEY, +}; +use rstest::rstest; +use serde_json::{json, Value}; use tempfile::TempDir; +const VM_ID: &str = "0e5e179d-5341-478b-8456-fbb90621bdf8"; + fn kvp(args: &[&str]) -> Output { Command::new(env!("CARGO_BIN_EXE_libazureinit-kvp")) .args(args) @@ -48,6 +56,14 @@ fn assert_success(output: Output) -> String { String::from_utf8(output.stdout).unwrap() } +fn assert_json(output: Output) -> Value { + serde_json::from_str(&assert_success(output)).unwrap() +} + +fn store_at(dir: &TempDir) -> KvpPoolStore { + KvpPoolStore::new_in(KvpPool::Guest, dir.path(), PoolMode::Safe).unwrap() +} + #[test] fn help_lists_commands() { let stdout = assert_success(kvp(&["--help"])); @@ -114,19 +130,19 @@ fn write_append_read_dump_entries_delete_and_clear() { assert_eq!(assert_success(kvp(&with_dir(&dir, &["read", "a"]))), "2\n"); assert_eq!( - assert_success(kvp(&with_dir(&dir, &["dump"]))), - "a=1\na=2\n" + assert_json(kvp(&with_dir(&dir, &["dump"]))), + json!([{"key": "a", "value": "1"}, {"key": "a", "value": "2"}]) ); assert_eq!(assert_success(kvp(&with_dir(&dir, &["entries"]))), "a=2\n"); assert_eq!( assert_success(kvp(&with_dir(&dir, &["delete", "a"]))), "true\n" ); - assert_eq!(assert_success(kvp(&with_dir(&dir, &["dump"]))), ""); + assert_eq!(assert_json(kvp(&with_dir(&dir, &["dump"]))), json!([])); assert_success(kvp(&with_dir(&dir, &["write", "b", "3"]))); assert_success(kvp(&with_dir(&dir, &["clear"]))); - assert_eq!(assert_success(kvp(&with_dir(&dir, &["dump"]))), ""); + assert_eq!(assert_json(kvp(&with_dir(&dir, &["dump"]))), json!([])); } #[test] @@ -140,7 +156,7 @@ fn load_replaces_pool_from_file() { &["load", "--file", input.to_str().unwrap()], ))); assert_eq!( - assert_success(kvp(&with_dir(&dir, &["dump"]))), + assert_success(kvp(&with_dir(&dir, &["dump", "--text"]))), "a=1\nb=2\n" ); } @@ -166,7 +182,7 @@ fn append_multiple_can_read_from_stdin() { )); assert_eq!( - assert_success(kvp(&with_dir(&dir, &["dump"]))), + assert_success(kvp(&with_dir(&dir, &["dump", "--text"]))), "x=1\nx=2\ny=3\n" ); } @@ -182,7 +198,7 @@ fn append_multiple_can_read_from_file() { &["append-multiple", "--file", input.to_str().unwrap()], ))); assert_eq!( - assert_success(kvp(&with_dir(&dir, &["dump"]))), + assert_success(kvp(&with_dir(&dir, &["dump", "--text"]))), "a=1\nb=2\n" ); } @@ -297,42 +313,144 @@ fn report_failure_rejects_invalid_supporting_data() { } #[test] -fn dump_parse_diagnostics_json_reassembles_and_skips_raw() { +fn parsed_dump_reassembles_and_filters_without_dropping_other_entries() { let dir = TempDir::new().unwrap(); + let base = format!( + "DIAG_V1|agent|{VM_ID}|event|a:b|e5f01809-a7a3-4279-aa64-1f18e21eda6e|2026-08-31T00:00:00.000Z|none||" + ); + for (key, value) in [ + (format!("{base}|1"), "two"), + ("note".into(), "raw value"), + (format!("{base}|0"), "one/"), + ("DIAG_V2|future".into(), "preserved"), + ] { + assert_success(kvp(&with_dir( + &dir, + &["write", "--append", &key, value], + ))); + } + assert_success(kvp(&with_dir(&dir, &["report-success", "--vm-id", VM_ID]))); assert_success(kvp(&with_dir( &dir, &[ - "write", - "--append", - "azure-init-x|100|vm|event|a:b|id1|2026-08-31T00:00:00Z|0", - "one/", - ], - ))); - assert_success(kvp(&with_dir( - &dir, - &[ - "write", - "--append", - "azure-init-x|100|vm|event|a:b|id1|2026-08-31T00:00:00Z|1", - "two", + "emit", + "--name", + "ssh:key", + "--message", + "added", + "--vm-id", + VM_ID, ], ))); - assert_success(kvp(&with_dir( - &dir, - &["write", "PROVISIONING_REPORT", "result=success"], - ))); + let path = store_at(&dir).path().to_path_buf(); + let before = fs::read(&path).unwrap(); + + let entries = assert_json(kvp(&with_dir(&dir, &["dump", "--parse"]))); + assert_eq!(entries.as_array().unwrap().len(), 5); + assert_eq!(entries[0]["type"], "diagnostic"); + assert_eq!(entries[0]["kind"], "event"); + assert_eq!(entries[0]["name"], "a:b"); + assert_eq!(entries[0]["payload"], "one/two"); + let timed_entries = &entries.as_array().unwrap()[1..3]; + assert!(timed_entries.iter().any(|entry| { + entry["type"] == "PROVISIONING_REPORT" && entry["result"] == "success" + })); + assert!(timed_entries.iter().any(|entry| entry["name"] == "ssh:key")); + assert_eq!( + entries[3], + json!({"type": "raw", "key": "note", "value": "raw value"}) + ); + assert_eq!( + entries[4], + json!({ + "type": "raw", "key": "DIAG_V2|future", "value": "preserved", + "error": "unsupported_version", + }) + ); - let out = assert_success(kvp(&with_dir( + let filtered = assert_json(kvp(&with_dir( &dir, - &["--json", "dump", "--parse-diagnostics"], + &["dump", "--parse", "--name", "ssh"], ))); - assert!(out.contains("\"kind\":\"event\"")); - assert!(out.contains("\"message\":\"one/two\"")); - assert!(!out.contains("PROVISIONING_REPORT")); + assert_eq!( + filtered, + Value::Array(entries.as_array().unwrap()[1..].to_vec()) + ); + assert_eq!(fs::read(path).unwrap(), before); +} + +#[test] +fn parsed_dump_sorts_timestamps_without_reordering_reader_or_raw_dump() { + let dir = TempDir::new().unwrap(); + let store = store_at(&dir); + let event_id = "e5f01809-a7a3-4279-aa64-1f18e21eda6e"; + let key = |name: &str, timestamp: &str| { + format!("DIAG_V1|agent|{VM_ID}|event|{name}|{event_id}|{timestamp}|none|||0") + }; + let records = vec![ + ("note".into(), "raw first"), + (key("latest", "2026-08-31T00:00:03.000Z"), "latest"), + ( + format!("CLOUD_INIT|100|event|cloud|{VM_ID}|{event_id}"), + r#"{"name":"cloud","type":"event","ts":"2026-08-31T00:00:02.000500Z","msg":"cloud"}"#, + ), + ( + PROVISIONING_REPORT_KEY.into(), + "result=success|agent=agent|vm_id=vm|pps_type=None|timestamp=2026-08-31T02:00:02+02:00", + ), + ("DIAG_V2|future".into(), "raw second"), + (key("tie-first", "2026-08-31T00:00:02.000Z"), "first tie"), + (key("earliest", "2026-08-31T00:00:01.000Z"), "earliest"), + (key("tie-second", "2026-08-31T00:00:02.000Z"), "second tie"), + ]; + store + .append_multiple(records.iter().map(|(key, value)| (key, *value))) + .unwrap(); + let before = fs::read(store.path()).unwrap(); + let reader_entries = serde_json::to_value( + libazureinit_kvp::DiagnosticReader::new(store.clone()) + .entries() + .unwrap(), + ) + .unwrap(); + let expected = Value::Array( + [6, 3, 5, 7, 2, 1, 0, 4] + .into_iter() + .map(|position| reader_entries[position].clone()) + .collect(), + ); + let parsed = assert_json(kvp(&with_dir(&dir, &["dump", "--parse"]))); + assert_eq!(parsed, expected); + assert_eq!(parsed[1]["timestamp"], "2026-08-31T02:00:02+02:00"); + + let text = + assert_success(kvp(&with_dir(&dir, &["dump", "--parse", "--text"]))); + let expected_labels = [ + "name=earliest ", + "PROVISIONING_REPORT=", + "name=tie-first ", + "name=tie-second ", + "name=cloud ", + "name=latest ", + "raw key=note ", + "raw key=DIAG_V2|future ", + ]; + assert_eq!(text.lines().count(), expected_labels.len()); + for (line, label) in text.lines().zip(expected_labels) { + assert!(line.contains(label), "expected {label:?} in {line:?}"); + } + + let physical = assert_json(kvp(&with_dir(&dir, &["dump"]))); + let expected_physical: Vec<_> = records + .iter() + .map(|(key, value)| json!({"key": key, "value": value})) + .collect(); + assert_eq!(physical, Value::Array(expected_physical)); + assert_eq!(fs::read(store.path()).unwrap(), before); } #[test] -fn dump_parse_diagnostics_text_renders_cloud_init_event() { +fn parsed_dump_normalizes_cloud_init_in_json_and_text() { let dir = TempDir::new().unwrap(); assert_success(kvp(&with_dir( &dir, @@ -353,260 +471,209 @@ fn dump_parse_diagnostics_text_renders_cloud_init_event() { ], ))); + let entries = assert_json(kvp(&with_dir(&dir, &["dump", "--parse"]))); + assert_eq!( + entries[0], + json!({ + "type": "diagnostic", "kind": "finish", "agent": "CLOUD_INIT", + "name": "modules-final/config-scripts_user", "vm_id": VM_ID, + "event_id": "e5f01809-a7a3-4279-aa64-1f18e21eda6e", + "timestamp": "2026-07-27T21:33:24.339Z", "encoding": "none", + "result": "success", "duration": 500, "payload": "scripts ran", + }) + ); + assert_eq!(entries[1]["kind"], "start"); + assert!(entries[1].get("result").is_none()); + assert!(entries[1].get("duration").is_none()); + let out = - assert_success(kvp(&with_dir(&dir, &["dump", "--parse-diagnostics"]))); - assert!(out.contains("event kind=finish")); + assert_success(kvp(&with_dir(&dir, &["dump", "--parse", "--text"]))); + assert!(out.contains("diagnostic kind=finish")); assert!(out.contains("agent=CLOUD_INIT")); - assert!(out.contains("boot_epoch=1785187982")); + assert!(!out.contains("boot_epoch")); assert!(out.contains("name=modules-final/config-scripts_user")); assert!(out.contains("vm_id=0e5e179d-5341-478b-8456-fbb90621bdf8")); - assert!(out.contains("result=SUCCESS")); - assert!(out.contains("timestamp=2026-07-27 21:33:24.339006 UTC")); - assert!(out.contains("duration=0.5")); - assert!(out.contains("message=scripts ran")); - assert!(out.contains("event kind=start")); + assert!(out.contains("result=success")); + assert!(out.contains("timestamp=2026-07-27T21:33:24.339Z")); + assert!(out.contains("duration=500ms")); + assert!(out.contains("payload=scripts ran")); + let start = out.lines().nth(1).unwrap(); + assert!(start.contains("diagnostic kind=start")); + assert!(!start.contains("result=")); + assert!(!start.contains("duration=")); } #[test] -fn dump_parse_diagnostics_json_renders_cloud_init_event() { +fn parsed_dump_renders_bytes_reports_and_raw_errors() { let dir = TempDir::new().unwrap(); + let store = store_at(&dir); + DiagnosticWriter::new(store.clone(), "agent", VM_ID) + .unwrap() + .emit_event( + "artifact", + vec![0, 255], + Some(Encoding::GzB64), + Some(Outcome::Failure), + Some(7), + ) + .unwrap(); + store.append("note", "raw value").unwrap(); + store.append("DIAG_V1|bad", "junk").unwrap(); assert_success(kvp(&with_dir( &dir, - &[ - "write", - "--append", - "CLOUD_INIT|1785187982|finish|modules-final/config-scripts_user|0e5e179d-5341-478b-8456-fbb90621bdf8|e5f01809-a7a3-4279-aa64-1f18e21eda6e", - r#"{"name":"modules-final/config-scripts_user","type":"finish","ts":"2026-07-27T21:33:24.339006+00:00","result":"SUCCESS","duration":0.5,"msg":"scripts ran"}"#, - ], + &["report-failure", "--vm-id", VM_ID, "--reason", "bad input"], ))); + let out = + assert_success(kvp(&with_dir(&dir, &["dump", "--parse", "--text"]))); + let lines: Vec<_> = out.lines().collect(); + assert_eq!(lines.len(), 4); + assert!(lines[0] + .contains("encoding=gz+b64 result=fail duration=7ms payload_b64=AP8=")); + assert_eq!( + lines[1], + format!( + "PROVISIONING_REPORT={}", + store.read(PROVISIONING_REPORT_KEY).unwrap().unwrap() + ) + ); + assert_eq!(lines[2], "raw key=note value=raw value"); + assert_eq!(lines[3], "raw key=DIAG_V1|bad value=junk error=malformed diagnostic or provisioning report"); - let out = assert_success(kvp(&with_dir( - &dir, - &["--json", "dump", "--parse-diagnostics"], - ))); - assert!(out.contains("\"kind\":\"finish\"")); - assert!(out.contains("\"agent\":\"CLOUD_INIT\"")); - assert!(out.contains("\"boot_epoch\":1785187982")); - assert!(out.contains("\"name\":\"modules-final/config-scripts_user\"")); - assert!(out.contains("\"vm_id\":\"0e5e179d-5341-478b-8456-fbb90621bdf8\"")); - assert!( - out.contains("\"event_id\":\"e5f01809-a7a3-4279-aa64-1f18e21eda6e\"") + let entries = + assert_json(kvp(&with_dir(&dir, &["dump", "--parse", "--json"]))); + assert_eq!( + entries[0]["payload"], + json!({"type": "bytes", "encoding": "base64", "data": "AP8="}) ); - assert!(out.contains("\"timestamp\":\"2026-07-27T21:33:24.339006Z\"")); - assert!(out.contains("\"result\":\"SUCCESS\"")); - assert!(out.contains("\"duration\":0.5")); - assert!(out.contains("\"message\":\"scripts ran\"")); + assert_eq!(entries[1]["reason"], "bad input"); + assert_eq!(entries[3]["error"], "malformed"); } #[test] -fn clear_diagnostics_option_is_not_exposed() { - let output = kvp(&["clear", "--diagnostics"]); +fn dump_name_requires_parse() { + let output = kvp(&["dump", "--name", "ssh"]); assert_eq!(output.status.code(), Some(2)); + assert!(String::from_utf8(output.stderr) + .unwrap() + .contains("--parse")); } #[test] -fn dump_parse_diagnostics_text_renders_only_normalized_entries() { - let dir = TempDir::new().unwrap(); - assert_success(kvp(&with_dir( - &dir, - &[ - "write", - "--append", - "a|100|vm|event|a:b|id1|2026-08-31T00:00:00Z|0", - "one/", - ], - ))); - assert_success(kvp(&with_dir( - &dir, - &[ - "write", - "--append", - "a|100|vm|event|a:b|id1|2026-08-31T00:00:00Z|1", - "two", - ], - ))); - assert_success(kvp(&with_dir( - &dir, - &["write", "PROVISIONING_REPORT", "result=success"], - ))); - assert_success(kvp(&with_dir( - &dir, - &[ - "write", - "--append", - "a|not-a-boot|vm|event|c:d|id2|2026-08-31T00:00:00Z", - "junk", - ], - ))); - - let out = - assert_success(kvp(&with_dir(&dir, &["dump", "--parse-diagnostics"]))); - assert!(out.contains( - "event kind=event agent=a boot_epoch=100 vm_id=vm \ - name=a:b event_id=id1 timestamp=2026-08-31 00:00:00 UTC \ - message=one/two" - )); - assert!(!out.contains("PROVISIONING_REPORT")); - assert!(!out.contains("junk")); -} - -#[test] -fn dump_parse_diagnostics_tail_limits_to_last_events() { +fn conflicting_output_flags_fail_before_pool_access() { let dir = TempDir::new().unwrap(); - assert_success(kvp(&with_dir( - &dir, - &[ - "write", - "--append", - "a|100|vm|event|a:b|i1|2026-08-31T00:00:00Z", - "first", - ], - ))); - assert_success(kvp(&with_dir( - &dir, - &[ - "write", - "--append", - "a|100|vm|event|c:d|i2|2026-08-31T00:00:01Z", - "second", - ], - ))); - - let out = assert_success(kvp(&with_dir( - &dir, - &["dump", "--parse-diagnostics", "-n", "1"], - ))); - assert!(out.contains("second")); - assert!(!out.contains("first")); + let output = kvp(&with_dir(&dir, &["--json", "dump", "--text"])); + assert_eq!(output.status.code(), Some(2)); + assert!(output.stdout.is_empty()); + assert!(String::from_utf8(output.stderr) + .unwrap() + .contains("--json and --text cannot be used together")); + assert!(!store_at(&dir).path().exists()); } #[test] -fn dump_parse_diagnostics_tail_defaults_to_20_when_count_omitted() { - let dir = TempDir::new().unwrap(); - for i in 1..=25 { - assert_success(kvp(&with_dir( - &dir, - &[ - "write", - "--append", - &format!("a|100|vm|event|n:{i}|id{i}|2026-08-31T00:00:00Z"), - &format!("msg{i}"), - ], - ))); +fn removed_diagnostic_options_are_rejected() { + let cases: &[(&[&str], &str)] = &[ + (&["dump", "--parse-diagnostics"], "--parse-diagnostics"), + (&["dump", "--parse", "--tail"], "--tail"), + (&["dump", "--parse", "-n", "1"], "-n"), + (&["emit", "--prefix", "agent"], "--prefix"), + ]; + for (args, flag) in cases { + let output = kvp(args); + assert_eq!(output.status.code(), Some(2)); + assert!(String::from_utf8(output.stderr) + .unwrap() + .contains(&format!("unexpected argument '{flag}'"))); } - - let out = assert_success(kvp(&with_dir( - &dir, - &["dump", "--parse-diagnostics", "--tail"], - ))); - assert_eq!(out.lines().count(), 20); - assert!(out.contains("msg25")); - assert!(out.contains("msg6")); - assert!(!out.contains("msg5")); } #[test] -fn dump_parse_diagnostics_filters_by_name_substring() { +fn dumps_fail_without_partial_output_for_invalid_physical_utf8() { let dir = TempDir::new().unwrap(); - assert_success(kvp(&with_dir( - &dir, - &[ - "write", - "--append", - "a|100|vm|event|user:add|i1|2026-08-31T00:00:00Z", - "u", - ], - ))); - assert_success(kvp(&with_dir( - &dir, - &[ - "write", - "--append", - "a|100|vm|event|ssh:key|i2|2026-08-31T00:00:01Z", - "s", - ], - ))); - - let out = assert_success(kvp(&with_dir( - &dir, - &["dump", "--parse-diagnostics", "--name", "ssh"], - ))); - assert!(out.contains("ssh:key")); - assert!(!out.contains("user:add")); -} - -#[test] -fn dump_parse_diagnostics_include_raw_option_is_not_exposed() { - let output = kvp(&["dump", "--parse-diagnostics", "--include-raw"]); - assert_eq!(output.status.code(), Some(2)); + let store = store_at(&dir); + store + .append_multiple([("good", "ok"), ("bad", "value")]) + .unwrap(); + let mut bytes = fs::read(store.path()).unwrap(); + let record_size = bytes.len() / 2; + bytes[record_size] = 0xff; + fs::write(store.path(), &bytes).unwrap(); + + for args in [ + &["dump"][..], + &["dump", "--parse"], + &["dump", "--parse", "--text"], + ] { + let output = kvp(&with_dir(&dir, args)); + assert_eq!(output.status.code(), Some(3)); + assert!(output.stdout.is_empty()); + assert!(!output.stderr.is_empty()); + } + assert_eq!(fs::read(store.path()).unwrap(), bytes); } -#[test] -fn dump_parse_diagnostics_json_skips_malformed_entries() { +#[rstest] +#[case::default_agent(None)] +#[case::custom_agent(Some("azure-init-test"))] +fn emit_writes_event_readable_by_dump(#[case] agent: Option<&str>) { let dir = TempDir::new().unwrap(); - assert_success(kvp(&with_dir( - &dir, - &[ - "write", - "--append", - "a|100|vm|event|a:b|i1|2026-08-31T00:00:00Z", - "hello", - ], - ))); - assert_success(kvp(&with_dir( - &dir, - &[ - "write", - "--append", - "a|not-a-boot|vm|event|c:d|i2|2026-08-31T00:00:00Z", - "junk", - ], - ))); - - let dump = assert_success(kvp(&with_dir( - &dir, - &["--json", "dump", "--parse-diagnostics"], - ))); - assert!(dump.contains("\"kind\":\"event\"")); - assert!(!dump.contains("junk")); + let mut args = vec![ + "emit", + "--name", + "user:create_user", + "--message", + "created azureuser", + "--vm-id", + VM_ID, + ]; + if let Some(agent) = agent { + args.extend(["--agent", agent]); + } + assert_success(kvp(&with_dir(&dir, &args))); + + let entries = assert_json(kvp(&with_dir(&dir, &["dump", "--parse"]))); + let expected_agent = agent + .unwrap_or(concat!("libazureinit-kvp/", env!("CARGO_PKG_VERSION"))); + assert_eq!(entries.as_array().unwrap().len(), 1); + assert_eq!(entries[0]["type"], "diagnostic"); + assert_eq!(entries[0]["kind"], "event"); + assert_eq!(entries[0]["agent"], expected_agent); + assert_eq!(entries[0]["vm_id"], VM_ID); + assert_eq!(entries[0]["name"], "user:create_user"); + assert_eq!(entries[0]["payload"], "created azureuser"); + assert_eq!(entries[0]["encoding"], "none"); + assert!(entries[0].get("result").is_none()); + assert!(entries[0].get("duration").is_none()); + let event_id = + uuid::Uuid::parse_str(entries[0]["event_id"].as_str().unwrap()) + .unwrap(); + assert_eq!(event_id.get_version_num(), 4); - let events = assert_success(kvp(&with_dir( - &dir, - &["--json", "dump", "--parse-diagnostics", "--name", "a:b"], - ))); - assert!(events.contains("\"message\":\"hello\"")); - assert!(!events.contains("junk")); + let raw = assert_json(kvp(&with_dir(&dir, &["dump"]))); + let key = raw[0]["key"].as_str().unwrap(); + assert!( + key.starts_with(&format!("DIAG_V1|{expected_agent}|{VM_ID}|event|")) + ); + assert!(key.ends_with("|none|||0")); } #[test] -fn emit_writes_event_readable_by_dump() { +fn emit_rejects_invalid_uuid_without_creating_pool() { let dir = TempDir::new().unwrap(); - assert_success(kvp(&with_dir( + let output = kvp(&with_dir( &dir, &[ "emit", "--name", - "user:create_user", + "event", "--message", - "created azureuser", + "test", "--vm-id", "vm-emit", - "--prefix", - "azure-init-test", ], - ))); - - let out = - assert_success(kvp(&with_dir(&dir, &["dump", "--parse-diagnostics"]))); - assert!(out.contains("vm_id=vm-emit")); - assert!(out.contains("name=user:create_user")); - assert!(out.contains("message=created azureuser")); - - let json = assert_success(kvp(&with_dir( - &dir, - &["--json", "dump", "--parse-diagnostics"], - ))); - assert!(json.contains("\"kind\":\"event\"")); - assert!(json.contains("\"vm_id\":\"vm-emit\"")); - assert!(json.contains("\"name\":\"user:create_user\"")); + )); + assert_eq!(output.status.code(), Some(2)); + assert!(output.stdout.is_empty()); + assert!(String::from_utf8(output.stderr).unwrap().contains("UUID")); + assert!(!store_at(&dir).path().exists()); } diff --git a/libazureinit-kvp/tests/diagnostics.rs b/libazureinit-kvp/tests/diagnostics.rs index de53c2ed..a9d25527 100644 --- a/libazureinit-kvp/tests/diagnostics.rs +++ b/libazureinit-kvp/tests/diagnostics.rs @@ -1,28 +1,40 @@ // Copyright (c) Microsoft Corporation. // Licensed under the MIT License. -//! Integration tests for the [`DiagnosticsKvp`] layer: emit/read -//! round-trips, cloud-init normalization, chunk reassembly, and concurrent -//! writes. +//! Public diagnostics API round trips, cloud-init compatibility, and store +//! behavior. +use std::io::Read; use std::thread; -use chrono::{DateTime, Utc}; +use base64::{engine::general_purpose::STANDARD, Engine as _}; +use flate2::read::ZlibDecoder; use libazureinit_kvp::{ - DiagnosticEvent, DiagnosticKind, DiagnosticsKvp, KvpPool, KvpPoolStore, - PoolMode, MAX_CHUNK_BYTES, + write_report, DecodeError, Diagnostic, DiagnosticPayload, DiagnosticReader, + DiagnosticWriter, Encoding, Entry, Kind, KvpError, KvpPool, KvpPoolStore, + Outcome, PoolMode, ProvisioningReport, RawKeyValue, ReportPpsType, + DIAGNOSTIC_VERSION_ID, MAX_CHUNK_BYTES, }; use rstest::rstest; use tempfile::TempDir; -const PREFIX: &str = "azure-init-test"; -const VM_ID: &str = "vm-abc"; +#[path = "fixtures/cloud_init.rs"] +mod cloud_init_fixtures; +use cloud_init_fixtures::COMPRESSED_LOG_CHUNKS; -fn diagnostics(dir: &TempDir) -> DiagnosticsKvp { - let store = - KvpPoolStore::new_in(KvpPool::Guest, dir.path(), PoolMode::Safe) - .unwrap(); - DiagnosticsKvp::new(store, VM_ID, PREFIX).unwrap() +const AGENT: &str = "azure-init-test"; +const VM_ID: &str = "3f2504e0-4f89-41d3-9a0c-0305e82c3301"; +const EVENT_ID: &str = "8f3e9c4a-1b2c-4d5e-9f01-234567890abc"; + +fn store_at(dir: &TempDir) -> KvpPoolStore { + KvpPoolStore::new_in(KvpPool::Guest, dir.path(), PoolMode::Safe).unwrap() +} + +fn diagnostic(entry: &Entry) -> &Diagnostic { + let Entry::Diagnostic(diagnostic) = entry else { + panic!("expected a diagnostic, got {entry:?}"); + }; + diagnostic } /// Real cloud-init reporting entries captured from a guest pool 1 file. @@ -42,252 +54,266 @@ const CLOUD_INIT_RECORDS: &[(&str, &str)] = &[ ), ]; -#[test] -fn reads_and_parses_real_cloud_init_pool() { +#[rstest] +#[case::legacy(false)] +#[case::current(true)] +fn reads_real_cloud_init_pool_in_both_layouts(#[case] include_vm_id: bool) { let dir = TempDir::new().unwrap(); - let store = - KvpPoolStore::new_in(KvpPool::Guest, dir.path(), PoolMode::Safe) - .unwrap(); - for &(key, value) in CLOUD_INIT_RECORDS { - store.append(key, value).unwrap(); - } + let store = store_at(&dir); + store + .append_multiple(CLOUD_INIT_RECORDS.iter().map(|&(key, value)| { + let key = if include_vm_id { + key.to_owned() + } else { + without_vm_id(key) + }; + (key, value) + })) + .unwrap(); - let diagnostics = DiagnosticsKvp::new(store, "", "").unwrap(); - let entries = diagnostics.entries().unwrap(); + let entries = DiagnosticReader::new(store).entries().unwrap(); assert_eq!(entries.len(), CLOUD_INIT_RECORDS.len()); - let event = &entries[0]; - assert_eq!(event.agent, "CLOUD_INIT"); - assert_eq!(event.kind, DiagnosticKind::Finish); - assert_eq!(event.name, "modules-final/config-scripts_user"); + let Diagnostic::Finish(finish) = diagnostic(&entries[0]) else { + panic!("expected a finish"); + }; + assert_eq!(finish.key.agent, "CLOUD_INIT"); + assert_eq!(finish.key.name, "modules-final/config-scripts_user"); assert_eq!( - event.vm_id.as_deref(), - Some("0e5e179d-5341-478b-8456-fbb90621bdf8") + finish.key.vm_id.as_deref(), + include_vm_id.then_some(CLOUD_INIT_VM_ID) ); - assert_eq!(event.result.as_deref(), Some("SUCCESS")); + assert_eq!(finish.result, Outcome::Success); + assert_eq!(finish.duration_ms, 0); assert_eq!( - event.message, - "config-scripts_user ran successfully and took 0.001 seconds" - ); - - assert_eq!(entries[1].kind, DiagnosticKind::Start); - assert!(entries[1].result.is_none()); - assert!(entries[1].duration.is_none()); -} - -#[test] -fn short_event_round_trips_as_single_record() { - let dir = TempDir::new().unwrap(); - let diag = diagnostics(&dir); - - assert_eq!(diag.vm_id(), VM_ID); - assert_eq!(diag.agent(), PREFIX); - - diag.emit_event("user:create_user", "created").unwrap(); - - let dumped = diag.store().dump().unwrap(); - assert_eq!(dumped.len(), 1); - assert!(dumped[0].0.ends_with("|0"), "key: {}", dumped[0].0); - - let entries = diag.entries().unwrap(); - assert_eq!(entries.len(), 1); - let decoded = &entries[0]; - assert_eq!(decoded.kind, DiagnosticKind::Event); - assert_eq!(decoded.vm_id.as_deref(), Some(VM_ID)); - assert_eq!(decoded.boot_epoch, diag.boot_epoch()); - assert_eq!(decoded.name, "user:create_user"); - let event_id = uuid::Uuid::parse_str(&decoded.event_id) - .expect("event_id should be a valid UUID"); - assert_eq!(event_id.get_version_num(), 4, "event_id should be a UUIDv4"); - assert_eq!(decoded.message, "created"); -} - -#[test] -fn long_event_splits_across_records_and_reassembles() { - let dir = TempDir::new().unwrap(); - let diag = diagnostics(&dir); - - let message = "x".repeat(MAX_CHUNK_BYTES * 3 + 50); - diag.emit_event("config:dump", &message).unwrap(); - - let dumped = diag.store().dump().unwrap(); - assert_eq!(dumped.len(), 4); - let base_of = |k: &str| k.rsplit_once('|').unwrap().0.to_string(); - let base = base_of(&dumped[0].0); - assert!( - dumped.iter().all(|(k, _)| base_of(k) == base), - "all chunks share one event-key base" + finish.payload, + DiagnosticPayload::from( + "config-scripts_user ran successfully and took 0.001 seconds" + ) ); - let mut keys: Vec = dumped.iter().map(|(k, _)| k.clone()).collect(); - keys.sort(); - keys.dedup(); - assert_eq!(keys.len(), 4, "each chunk must have a unique key"); - - let entries = diag.entries().unwrap(); - assert_eq!(entries.len(), 1); - assert_eq!(entries[0].message, message); -} - -#[test] -fn multi_chunk_event_uses_unique_keys_so_host_keeps_all() { - let dir = TempDir::new().unwrap(); - let diag = diagnostics(&dir); - - let message = "z".repeat(MAX_CHUNK_BYTES * 2 + 1); - diag.emit_event("big:event", &message).unwrap(); - - let dumped = diag.store().dump().unwrap(); - assert_eq!(dumped.len(), 3); - let total = dumped.len(); - let mut keys: Vec = dumped.into_iter().map(|(k, _)| k).collect(); - keys.sort(); - keys.dedup(); - assert_eq!(keys.len(), total, "chunk keys must be unique"); - - let events = diag.entries().unwrap(); - assert_eq!(events.len(), 1); - assert_eq!(events[0].message, message); + assert!(matches!(diagnostic(&entries[1]), Diagnostic::Start(_))); + assert!(matches!(diagnostic(&entries[2]), Diagnostic::Finish(finish) + if finish.result == Outcome::Success && finish.duration_ms == 340)); } #[test] -fn raw_and_malformed_records_are_skipped() { +fn span_and_point_events_round_trip_with_a_report() { let dir = TempDir::new().unwrap(); - let diag = diagnostics(&dir); - - diag.store() - .append(&format!("{PREFIX}|100|{VM_ID}|NOPE|bad:kind|id|ts"), "junk") + let store = store_at(&dir); + let writer = DiagnosticWriter::new(store.clone(), AGENT, VM_ID).unwrap(); + let reader = DiagnosticReader::new(store.clone()); + assert!(!store.path().exists()); + assert!(reader.entries().unwrap().is_empty()); + + writer + .emit_start(EVENT_ID, "provision:run", "starting", None) .unwrap(); - diag.store() - .append("PROVISIONING_REPORT", "result=success") + writer + .emit_event("imds", "ok", None, Some(Outcome::Success), Some(17)) .unwrap(); - - assert!(diag.entries().unwrap().is_empty()); + writer + .emit_finish( + EVENT_ID, + "provision:run", + "finished", + None, + Outcome::Success, + 120, + ) + .unwrap(); + let report = ProvisioningReport::success(AGENT, VM_ID, ReportPpsType::None) + .with_extra("build", "test-123"); + write_report(&store, &report).unwrap(); + + let entries = reader.entries().unwrap(); + let [Entry::Diagnostic(Diagnostic::Start(start)), Entry::Diagnostic(Diagnostic::Event(event)), Entry::Diagnostic(Diagnostic::Finish(finish)), Entry::Report(decoded_report)] = + entries.as_slice() + else { + panic!("unexpected entries: {entries:?}"); + }; + assert_eq!(start.key.event_id, EVENT_ID); + assert_eq!(finish.key.event_id, EVENT_ID); + assert_eq!(start.key.name, finish.key.name); + assert_eq!(start.key.agent, AGENT); + assert_eq!(start.key.vm_id.as_deref(), Some(VM_ID)); + assert_eq!(start.payload, DiagnosticPayload::from("starting")); + assert_eq!(event.key.name, "imds"); + assert_eq!(event.payload, DiagnosticPayload::from("ok")); + assert_eq!(event.result, Some(Outcome::Success)); + assert_eq!(event.duration_ms, Some(17)); + assert_eq!( + uuid::Uuid::parse_str(&event.key.event_id) + .unwrap() + .get_version_num(), + 4 + ); + assert_ne!(event.key.event_id, EVENT_ID); + assert_eq!(finish.payload, DiagnosticPayload::from("finished")); + assert_eq!(finish.result, Outcome::Success); + assert_eq!(finish.duration_ms, 120); + assert_eq!(decoded_report, &report); + + let dumped = store.dump().unwrap(); + for (key, _) in &dumped[..3] { + assert!(key + .starts_with(&format!("{DIAGNOSTIC_VERSION_ID}|{AGENT}|{VM_ID}|"))); + assert!(key.ends_with("|0")); + } } -#[test] -fn mixed_records_round_trip_together() { +#[rstest] +#[case::text(None)] +#[case::compressed(Some(Encoding::GzB64))] +fn long_payload_round_trips_through_host_visible_records( + #[case] encoding: Option, +) { let dir = TempDir::new().unwrap(); - let diag = diagnostics(&dir); - - diag.emit_event("a:b", "short").unwrap(); - diag.emit_event("c:d", "y".repeat(MAX_CHUNK_BYTES + 5)) - .unwrap(); - diag.store() - .append("PROVISIONING_REPORT", "result=success") - .unwrap(); - diag.store() - .append(&format!("{PREFIX}|100|{VM_ID}|NOPE|e:f|id|ts"), "junk") + let store = store_at(&dir); + let writer = DiagnosticWriter::new(store.clone(), AGENT, VM_ID).unwrap(); + let message: String = (0..MAX_CHUNK_BYTES) + .map(|index| format!("{index:04x}\u{20ac};")) + .collect(); + writer + .emit_event( + "config:dump", + message.as_str(), + encoding.clone(), + None, + None, + ) .unwrap(); - let entries = diag.entries().unwrap(); - assert_eq!(entries.len(), 2); + let dumped = store.dump().unwrap(); + assert!(dumped.len() > 1); + assert_eq!(store.entries().unwrap().len(), dumped.len()); + let base = dumped[0].0.rsplit_once('|').unwrap().0; + for (index, (key, value)) in dumped.iter().enumerate() { + assert_eq!(key, &format!("{base}|{index}")); + assert!(key.len() <= 254); + assert!(value.len() <= MAX_CHUNK_BYTES); + } + + let decoded = + decode_single(DiagnosticReader::new(store).entries().unwrap()); + assert_eq!(decoded.kind(), Kind::Event); + assert_eq!(decoded.key().encoding, encoding); + let expected = if encoding.is_some() { + DiagnosticPayload::Bytes(message.into_bytes()) + } else { + DiagnosticPayload::Text(message) + }; + assert_eq!(decoded.payload(), &expected); } #[test] -fn explicit_diagnostic_kinds_are_tracing_ready() { +fn raw_and_malformed_records_are_preserved_beside_diagnostics() { let dir = TempDir::new().unwrap(); - let diag = diagnostics(&dir); - let timestamp = DateTime::parse_from_rfc3339("2026-08-31T12:34:56.789Z") + let store = store_at(&dir); + let records = vec![ + (format!("{AGENT}|100|{VM_ID}|event|legacy|{EVENT_ID}|2026-08-31T00:00:00Z|0"), "legacy", None), + ("DIAG_V1|bad".into(), "junk", Some(DecodeError::Malformed)), + (format!("CLOUD_INIT|100|event|broken|{EVENT_ID}"), "not-json", Some(DecodeError::Malformed)), + ("PROVISIONING_REPORT".into(), "result=success", Some(DecodeError::Malformed)), + ("DIAG_V2|future".into(), "unknown", Some(DecodeError::UnsupportedVersion)), + ]; + store + .append_multiple(records.iter().map(|(key, value, _)| (key, *value))) + .unwrap(); + DiagnosticWriter::new(store.clone(), AGENT, VM_ID) .unwrap() - .with_timezone(&Utc); - - diag.emit( - DiagnosticKind::Start, - "provision:run", - "shared-span-id", - timestamp, - "starting", - ) - .unwrap(); - diag.emit( - DiagnosticKind::Event, - "provision:run", - "shared-span-id", - timestamp, - "progress", - ) - .unwrap(); - diag.emit( - DiagnosticKind::Finish, - "provision:run", - "shared-span-id", - timestamp, - "finished", - ) - .unwrap(); - diag.emit( - DiagnosticKind::Other("diagnostic".to_string()), - "support:bundle", - "diagnostic-id", - timestamp, - "collected", - ) - .unwrap(); - - let entries = diag.entries().unwrap(); - assert_eq!(entries.len(), 4); - assert_eq!(entries[0].kind, DiagnosticKind::Start); - assert_eq!(entries[1].kind, DiagnosticKind::Event); - assert_eq!(entries[2].kind, DiagnosticKind::Finish); + .emit_event("valid", "visible", None, None, None) + .unwrap(); + let before = std::fs::read(store.path()).unwrap(); + + let entries = DiagnosticReader::new(store.clone()).entries().unwrap(); + assert_eq!(entries.len(), records.len() + 1); + for (entry, (key, value, error)) in entries.iter().zip(&records) { + assert_eq!( + entry, + &Entry::Raw(RawKeyValue { + key: key.clone(), + value: (*value).to_owned(), + error: *error, + }) + ); + } assert_eq!( - entries[3].kind, - DiagnosticKind::Other("diagnostic".to_string()) + diagnostic(entries.last().unwrap()).payload(), + &DiagnosticPayload::from("visible") ); - assert!(entries[..3] - .iter() - .all(|entry| entry.event_id == "shared-span-id")); - assert!(entries.iter().all(|entry| entry.timestamp == timestamp)); + assert_eq!(std::fs::read(store.path()).unwrap(), before); } #[test] fn chunked_entries_survive_store_swap_deletion() { let dir = TempDir::new().unwrap(); - let diag = diagnostics(&dir); + let store = store_at(&dir); + let writer = DiagnosticWriter::new(store.clone(), AGENT, VM_ID).unwrap(); let first_message = "a".repeat(MAX_CHUNK_BYTES * 2 + 7); let second_message = "b".repeat(MAX_CHUNK_BYTES * 2 + 7); - diag.emit_event("first", &first_message).unwrap(); - diag.store().append("remove-me", "raw").unwrap(); - diag.emit_event("second", &second_message).unwrap(); + store.append("remove-me", "raw").unwrap(); + writer + .emit_event("first", first_message.as_str(), None, None, None) + .unwrap(); + writer + .emit_event("second", second_message.as_str(), None, None, None) + .unwrap(); - // Deletion moves the final record into the removed slot, so the second - // event's chunks are no longer adjacent or in index order. - assert!(diag.store().delete("remove-me").unwrap()); + assert!(store.delete("remove-me").unwrap()); - let entries = diag.entries().unwrap(); + let entries = DiagnosticReader::new(store).entries().unwrap(); assert_eq!(entries.len(), 2); - assert_eq!(entries[0].name, "first"); - assert_eq!(entries[0].message, first_message); - assert_eq!(entries[1].name, "second"); - assert_eq!(entries[1].message, second_message); + assert_eq!(diagnostic(&entries[0]).key().name, "second"); + assert_eq!( + diagnostic(&entries[0]).payload(), + &DiagnosticPayload::Text(second_message) + ); + assert_eq!(diagnostic(&entries[1]).key().name, "first"); + assert_eq!( + diagnostic(&entries[1]).payload(), + &DiagnosticPayload::Text(first_message) + ); } #[test] fn emit_rejects_delimiter_in_event_fields() { let dir = TempDir::new().unwrap(); - let diag = diagnostics(&dir); - - assert!(diag.emit_event("a|b", "msg").is_err()); - assert!(diag.store().dump().unwrap().is_empty()); + let store = store_at(&dir); + let writer = DiagnosticWriter::new(store.clone(), AGENT, VM_ID).unwrap(); + + assert!(matches!( + writer.emit_event("a|b", "msg", None, None, None), + Err(KvpError::EventFieldContainsDelimiter { field: "name" }) + )); + assert!(!store.path().exists()); } #[test] fn concurrent_multichunk_emits_reassemble_without_interleaving() { let dir = TempDir::new().unwrap(); - let diag = diagnostics(&dir); + let store = store_at(&dir); + let writer = DiagnosticWriter::new(store.clone(), AGENT, VM_ID).unwrap(); const THREADS: usize = 5; const PER_THREAD: usize = 8; let len = MAX_CHUNK_BYTES * 2 + 7; let handles: Vec<_> = (0..THREADS) - .map(|t| { - let diag = diag.clone(); - let marker = (b'a' + t as u8) as char; + .map(|thread_index| { + let writer = writer.clone(); + let marker = (b'a' + u8::try_from(thread_index).unwrap()) as char; thread::spawn(move || { for _ in 0..PER_THREAD { let message = marker.to_string().repeat(len); - diag.emit_event(format!("thread:{marker}"), message) + writer + .emit_event( + &format!("thread:{marker}"), + message, + None, + None, + None, + ) .unwrap(); } }) @@ -297,31 +323,29 @@ fn concurrent_multichunk_emits_reassemble_without_interleaving() { handle.join().unwrap(); } - let events = diag.entries().unwrap(); + let dumped = store.dump().unwrap(); + assert_eq!(dumped.len(), THREADS * PER_THREAD * 3); + for group in dumped.chunks_exact(3) { + let base = group[0].0.rsplit_once('|').unwrap().0; + for (index, (key, _)) in group.iter().enumerate() { + assert_eq!(key, &format!("{base}|{index}")); + } + } + + let events = DiagnosticReader::new(store).entries().unwrap(); assert_eq!(events.len(), THREADS * PER_THREAD); - for event in &events { - assert_eq!(event.message.len(), len); - let first = event.message.chars().next().unwrap(); - assert!(event.message.chars().all(|c| c == first)); - assert_eq!(event.name, format!("thread:{first}")); + for entry in &events { + let event = diagnostic(entry); + let DiagnosticPayload::Text(message) = event.payload() else { + panic!("expected text payload"); + }; + assert_eq!(message.len(), len); + let first = message.chars().next().unwrap(); + assert!(message.chars().all(|character| character == first)); + assert_eq!(event.key().name, format!("thread:{first}")); } } -const COMPRESSED_LOG_CHUNKS: &[(&str, &str)] = &[ - ( - "CLOUD_INIT|1786047606|compressed|cloud-init.log|b7a822ba-4eea-46c0-b559-e84396101132|0", - "{\"name\":\"cloud-init.log\",\"type\":\"compressed\",\"ts\":\"2026-08-06T20:20:13.479078Z\",\"msg_i\":0,\"msg\":\"{\\\"encoding\\\": \\\"gz+b64\\\", \\\"data\\\": \\\"eJxtV0uObDUMnbOKt4GHYjuJbYYIxAoYIQZx7EhIDBjAgN1ziumz1K2uqtY9lWuf3+XB++uwr2N/\\\\n4fEDfsb48vXLP3//8ef3f/37208///jrL7//8OVq2K09s4zekFEn71Lyp2fiBZsU1wxZ5zv+BpAa\\\\nwIzY128A1jhGRMikPJSsPJym6x5ih681gNwA6jzvnvK4U+6dhTdqu/Z4OsaY0y9xutDOkeIZDaw0\\\\nsKTytj1N9x3zVFjMoKnFtGaZNDCzganSNHnj8Hg4Am1eK4fcY2Nn7djuLuvJ5Iq5zVP2PPdWA78a\\\\n+CcyRXBfig14WKnskaNU6UWe1cDsBiZ0LamKNLIJIPyJVDl3nrHyzdh7lvt42F4DqQ3kfPXY2e9h\\\\nMdUyfhqYQDCJvsJoa4VY3KG1j5F3E7UGeC0GE9Xm5p0X5EzeRxYdnG+BCrV0F7Ht7qTeANrZ59Sk\\\\new4YjeU6mIjxSQNAnUZoGU1i2eDc3Ya71KQpo7m8UwTRfc/3oUFKk4c9cPbMoho6nhn+iQ8r1iQn\\\\nrdfAdrrAeeja0gxPp9w3d0FwV+LiVy4YVwMUH9hTp17qVOGQpwHZDs60VwTfqINRsY17Q2l9pmcR\\\\nj+4+qztrp5LYy/KCfmfbXbX4rlBfonb1NEZArRaW1+XDdtRTL0Nm+9LQLaI15t2Yx/tY18WJa3oD\\\\n22mDRtLaV1QGtHAfdvCSyavRKHVKyLev7u34WrOSISCLQQRnPgJ0sBskTw7Ts+hutlOBqN85Rplz\\\\nXuWwj6iyaAhjjIL7P5qqfM9rhEWdDl48ObTXmF44zUznfd+kby/nTgVF70HnA0sQ0h1SL/zZilP1\\\\nsAm4C6UW7\"}", - ), - ( - "CLOUD_INIT|1786047606|compressed|cloud-init.log|b7a822ba-4eea-46c0-b559-e84396101132|1", - "{\"name\":\"cloud-init.log\",\"type\":\"compressed\",\"ts\":\"2026-08-06T20:20:13.479078Z\",\"msg_i\":1,\"msg\":\"ASEz7iBEzbQnUL8DD83hUtijQtvWjYOAogZfCGPdZMbtXKni0Ovzj2g/Ra4G3zZDVx5\\\\niKDbQHQqsHvXgQWB4ONOz21IHWKu3QB0fL8Yx8sNLlgG3Wv78IvN1xdpo0Xu+L6PrwejqVFX59R1\\\\nPgu86zGvSyDdXgQhra3I8Fmy2btddpwvQRbV2dc0RMaathBMrJCmyHIoynHqkQzv9VUb6x7Yazbw\\\\nnSIO3HFlbnWOElSMkncWwntin5KO1C5enuCOn6PdVDtN6COky/gkbjgh6su2R/qD/zFfHU0ccqcE\\\\nZKqmvPWK4Re+IovPxHi9tic+hU0DMO6zSfSpH9zcurRZYR+/NOS9ZwVxIRWLHzoFLB/rM4w469Nn\\\\nZoxN6ZMzpzdZIp1SRsbbG3Vj7YPocHpzudkUxxjxNQZCbIMxLKfAC1AuF0ljZdJpR8ZF3RvTEh0r\\\\nWN8q0Tu6y9tOlSOU5crBpfYxKA6y4ubyTjV8PsG639xy4Smw0oKp66B4oUdjolrFJ48uBFGjsVLp\\\\ndHQwaJ+JsISJokGIvIc6YBAkyLMPPIsnVo6CBKYf7QxVOhU5PDqPwe8hbFsuA2VSRx4nytmAtFqB\\\\n3SEDJbFIWM1aMI6gj82CNzJmV2ilU8dZajljvmFgSJrivtEeAaAo4rhlR/k5DLo1gK1KtvO8+TJj\\\\nfnxHH1wRwvj28tkp4RmthXGj7zAhqdCpBc8Ad+OMoIgQyTiINgGRdeNkzT5np4GHkNCZaFDCmDOC\\\\ntpYzTKFggG6JlhHPPiUykVQNaMf86yAGcgjJUQPlZI3/4xGPGPdsjsazZ6eAWHiUmugR8JI9HKx6\\\\n6NkPhfPDPopEZ1eFMTZbnZ0m8IQDlaPl/S8IG4GF8MJqcOD/AFeindw=\\\"}", - ), - ( - "CLOUD_INIT|1786047606|compressed|cloud-init.log|b7a822ba-4eea-46c0-b559-e84396101132|2", - "{\"name\":\"cloud-init.log\",\"type\":\"compressed\",\"ts\":\"2026-08-06T20:20:13.479078Z\",\"msg_i\":2,\"msg\":\"\\n\\\"}\"}", - ), -]; - /// The reassembled `msg` across all three chunks. const EXPECTED_COMPRESSED_MSG: &str = "{\"encoding\": \"gz+b64\", \"data\": \"eJxtV0uObDUMnbOKt4GHYjuJbYYIxAoYIQZx7EhIDBjAgN1ziumz1K2uqtY9lWuf3+XB++uwr2N/\\n4fEDfsb48vXLP3//8ef3f/37208///jrL7//8OVq2K09s4zekFEn71Lyp2fiBZsU1wxZ5zv+BpAa\\nwIzY128A1jhGRMikPJSsPJym6x5ih681gNwA6jzvnvK4U+6dhTdqu/Z4OsaY0y9xutDOkeIZDaw0\\nsKTytj1N9x3zVFjMoKnFtGaZNDCzganSNHnj8Hg4Am1eK4fcY2Nn7djuLuvJ5Iq5zVP2PPdWA78a\\n+CcyRXBfig14WKnskaNU6UWe1cDsBiZ0LamKNLIJIPyJVDl3nrHyzdh7lvt42F4DqQ3kfPXY2e9h\\nMdUyfhqYQDCJvsJoa4VY3KG1j5F3E7UGeC0GE9Xm5p0X5EzeRxYdnG+BCrV0F7Ht7qTeANrZ59Sk\\new4YjeU6mIjxSQNAnUZoGU1i2eDc3Ya71KQpo7m8UwTRfc/3oUFKk4c9cPbMoho6nhn+iQ8r1iQn\\nrdfAdrrAeeja0gxPp9w3d0FwV+LiVy4YVwMUH9hTp17qVOGQpwHZDs60VwTfqINRsY17Q2l9pmcR\\nj+4+qztrp5LYy/KCfmfbXbX4rlBfonb1NEZArRaW1+XDdtRTL0Nm+9LQLaI15t2Yx/tY18WJa3oD\\n22mDRtLaV1QGtHAfdvCSyavRKHVKyLev7u34WrOSISCLQQRnPgJ0sBskTw7Ts+hutlOBqN85Rplz\\nXuWwj6iyaAhjjIL7P5qqfM9rhEWdDl48ObTXmF44zUznfd+kby/nTgVF70HnA0sQ0h1SL/zZilP1\\nsAm4C6UW7ASEz7iBEzbQnUL8DD83hUtijQtvWjYOAogZfCGPdZMbtXKni0Ovzj2g/Ra4G3zZDVx5\\niKDbQHQqsHvXgQWB4ONOz21IHWKu3QB0fL8Yx8sNLlgG3Wv78IvN1xdpo0Xu+L6PrwejqVFX59R1\\nPgu86zGvSyDdXgQhra3I8Fmy2btddpwvQRbV2dc0RMaathBMrJCmyHIoynHqkQzv9VUb6x7Yazbw\\nnSIO3HFlbnWOElSMkncWwntin5KO1C5enuCOn6PdVDtN6COky/gkbjgh6su2R/qD/zFfHU0ccqcE\\nZKqmvPWK4Re+IovPxHi9tic+hU0DMO6zSfSpH9zcurRZYR+/NOS9ZwVxIRWLHzoFLB/rM4w469Nn\\nZoxN6ZMzpzdZIp1SRsbbG3Vj7YPocHpzudkUxxjxNQZCbIMxLKfAC1AuF0ljZdJpR8ZF3RvTEh0r\\nWN8q0Tu6y9tOlSOU5crBpfYxKA6y4ubyTjV8PsG639xy4Smw0oKp66B4oUdjolrFJ48uBFGjsVLp\\ndHQwaJ+JsISJokGIvIc6YBAkyLMPPIsnVo6CBKYf7QxVOhU5PDqPwe8hbFsuA2VSRx4nytmAtFqB\\n3SEDJbFIWM1aMI6gj82CNzJmV2ilU8dZajljvmFgSJrivtEeAaAo4rhlR/k5DLo1gK1KtvO8+TJj\\nfnxHH1wRwvj28tkp4RmthXGj7zAhqdCpBc8Ad+OMoIgQyTiINgGRdeNkzT5np4GHkNCZaFDCmDOC\\ntpYzTKFggG6JlhHPPiUykVQNaMf86yAGcgjJUQPlZI3/4xGPGPdsjsazZ6eAWHiUmugR8JI9HKx6\\n6NkPhfPDPopEZ1eFMTZbnZ0m8IQDlaPl/S8IG4GF8MJqcOD/AFeindw=\\n\"}"; @@ -333,7 +357,7 @@ const CLOUD_INIT_VM_ID: &str = "0e5e179d-5341-478b-8456-fbb90621bdf8"; /// -> `CLOUD_INIT|inc|type|name|vm_id|uuid[|i]`. fn with_vm_id(old_key: &str, vm_id: &str) -> String { let mut segments: Vec<&str> = old_key.split('|').collect(); - segments.insert(4, vm_id); // 0:CLOUD_INIT 1:inc 2:type 3:name | vm_id + segments.insert(4, vm_id); segments.join("|") } @@ -344,155 +368,72 @@ fn without_vm_id(current_key: &str) -> String { } /// Append the given records to a fresh guest pool and normalize them. -fn entries_of, V: AsRef>( - pairs: &[(K, V)], -) -> Vec { +fn entries_of, V: AsRef>(pairs: &[(K, V)]) -> Vec { let dir = TempDir::new().unwrap(); - let store = - KvpPoolStore::new_in(KvpPool::Guest, dir.path(), PoolMode::Safe) - .unwrap(); - for (key, value) in pairs { - store.append(key.as_ref(), value.as_ref()).unwrap(); - } - DiagnosticsKvp::new(store, "", "") - .unwrap() - .entries() - .unwrap() + let store = store_at(&dir); + store + .append_multiple( + pairs + .iter() + .map(|(key, value)| (key.as_ref(), value.as_ref())), + ) + .unwrap(); + DiagnosticReader::new(store).entries().unwrap() } -fn decode_single(entries: Vec) -> DiagnosticEvent { +fn decode_single(entries: Vec) -> Diagnostic { assert_eq!(entries.len(), 1, "expected one entry, got: {entries:?}"); - entries.into_iter().next().unwrap() + match entries.into_iter().next().unwrap() { + Entry::Diagnostic(diagnostic) => diagnostic, + entry => panic!("expected a diagnostic, got {entry:?}"), + } } #[rstest] -#[case::start("start", "azure-ds", DiagnosticKind::Start)] -#[case::finish("finish", "azure-ds/get-metadata", DiagnosticKind::Finish)] -#[case::event("event", "user:create_user", DiagnosticKind::Event)] -#[case::diagnostic( - "diagnostic", - "diagnostic message", - DiagnosticKind::Other("diagnostic".to_string()) -)] -#[case::compressed( - "compressed", - "cloud-init.log", - DiagnosticKind::Other("compressed".to_string()) -)] -#[case::boot_telemetry( - "boot-telemetry", - "boot-telemetry", - DiagnosticKind::Other("boot-telemetry".to_string()) -)] -#[case::system_info( - "system-info", - "system information", - DiagnosticKind::Other("system-info".to_string()) -)] -fn cloud_init_type_decodes_in_both_layouts( - #[case] event_type: &str, - #[case] name: &str, - #[case] expected: DiagnosticKind, +#[case::legacy(false)] +#[case::current(true)] +fn captured_compressed_log_reassembles_and_decodes( + #[case] include_vm_id: bool, ) { - const TS: &str = "2026-08-06T20:20:13.479078Z"; - const UUID: &str = "b7a822ba-4eea-46c0-b559-e84396101132"; - let msg = format!("payload for {event_type}"); - let value = format!( - "{{\"name\":\"{name}\",\"type\":\"{event_type}\",\ - \"ts\":\"{TS}\",\"msg\":\"{msg}\"}}" - ); - let old_key = format!("CLOUD_INIT|1786047606|{event_type}|{name}|{UUID}"); - let current_key = with_vm_id(&old_key, CLOUD_INIT_VM_ID); - - let event = decode_single(entries_of(&[(&old_key, &value)])); - assert_eq!(event.agent, "CLOUD_INIT"); - assert_eq!(event.kind, expected); - assert_eq!(event.vm_id, None); - assert_eq!(event.name, name); - assert_eq!(event.message, msg); - - let event = decode_single(entries_of(&[(¤t_key, &value)])); - assert_eq!(event.kind, expected); - assert_eq!(event.vm_id.as_deref(), Some(CLOUD_INIT_VM_ID)); - assert_eq!(event.message, msg); -} - -#[test] -fn cloud_init_finish_reports_result_and_duration_in_both_layouts() { - let value = "{\"name\":\"azure-ds/get-metadata\",\"type\":\"finish\",\ - \"ts\":\"2026-08-06T20:20:13.400000Z\",\"result\":\"SUCCESS\",\ - \"duration\":0.1234,\"msg\":\"finished\"}"; - let old_key = "CLOUD_INIT|1786047606|finish|azure-ds/get-metadata|\ - b7a822ba-4eea-46c0-b559-e84396101132"; - - for key in [old_key.to_string(), with_vm_id(old_key, CLOUD_INIT_VM_ID)] { - let event = decode_single(entries_of(&[(key.as_str(), value)])); - assert_eq!(event.kind, DiagnosticKind::Finish); - assert_eq!(event.result.as_deref(), Some("SUCCESS")); - assert_eq!(event.duration, Some(0.1234)); - } -} - -#[test] -fn real_cloud_init_samples_decode_without_vm_id_too() { - for &(key, value) in CLOUD_INIT_RECORDS { - let event = decode_single(entries_of(&[(without_vm_id(key), value)])); - assert!(event.vm_id.is_none(), "stripped sample kept a vm_id: {key}"); - } - let event = decode_single(entries_of(&[CLOUD_INIT_RECORDS[0]])); - assert_eq!(event.vm_id.as_deref(), Some(CLOUD_INIT_VM_ID)); -} - -#[test] -fn old_compressed_log_reassembles_across_chunks() { - let event = decode_single(entries_of(COMPRESSED_LOG_CHUNKS)); - assert_eq!(event.kind, DiagnosticKind::Other("compressed".to_string())); - assert_eq!(event.vm_id, None); - assert_eq!(event.name, "cloud-init.log"); - assert_eq!(event.message, EXPECTED_COMPRESSED_MSG); -} - -#[test] -fn current_compressed_log_reassembles_across_chunks() { - let current: Vec<(String, &str)> = COMPRESSED_LOG_CHUNKS + let records: Vec<(String, &str)> = COMPRESSED_LOG_CHUNKS .iter() - .map(|&(key, value)| (with_vm_id(key, CLOUD_INIT_VM_ID), value)) + .rev() + .map(|&(key, value)| { + let key = if include_vm_id { + with_vm_id(key, CLOUD_INIT_VM_ID) + } else { + key.to_owned() + }; + (key, value) + }) .collect(); - let event = decode_single(entries_of(¤t)); - assert_eq!(event.kind, DiagnosticKind::Other("compressed".to_string())); - assert_eq!(event.vm_id.as_deref(), Some(CLOUD_INIT_VM_ID)); - assert_eq!(event.message, EXPECTED_COMPRESSED_MSG); -} - -#[test] -fn cloud_init_event_with_invalid_json_is_skipped() { - let key = format!( - "CLOUD_INIT|1786047606|compressed|cloud-init.log|{CLOUD_INIT_VM_ID}|\ - b7a822ba-4eea-46c0-b559-e84396101132" + let event = decode_single(entries_of(&records)); + assert_eq!(event.kind(), Kind::Event); + assert_eq!( + event.key().vm_id.as_deref(), + include_vm_id.then_some(CLOUD_INIT_VM_ID) ); - assert!(entries_of(&[(key.as_str(), "not-json")]).is_empty()); -} + assert_eq!(event.key().name, "cloud-init.log"); + assert_eq!(event.key().encoding, Some(Encoding::GzB64)); -#[test] -fn cloud_init_chunk_index_mismatch_is_skipped() { - let base = "CLOUD_INIT|1786047606|event|test|\ - b7a822ba-4eea-46c0-b559-e84396101132"; - let value = r#"{"name":"test","type":"event","ts":"2026-08-06T20:20:13Z","msg_i":1,"msg":"payload"}"#; - - assert!(entries_of(&[(format!("{base}|0"), value)]).is_empty()); -} - -#[test] -fn cloud_init_chunk_without_key_index_is_skipped() { - let key = "CLOUD_INIT|1786047606|event|test|\ - b7a822ba-4eea-46c0-b559-e84396101132"; - let value = r#"{"name":"test","type":"event","ts":"2026-08-06T20:20:13Z","msg_i":0,"msg":"partial"}"#; - - assert!(entries_of(&[(key, value)]).is_empty()); + let envelope: serde_json::Value = + serde_json::from_str(EXPECTED_COMPRESSED_MSG).unwrap(); + let data: String = envelope["data"] + .as_str() + .unwrap() + .split_ascii_whitespace() + .collect(); + let compressed = STANDARD.decode(data).unwrap(); + let mut expected = Vec::new(); + ZlibDecoder::new(compressed.as_slice()) + .read_to_end(&mut expected) + .unwrap(); + assert_eq!(expected.len(), 3554); + assert_eq!(event.payload(), &DiagnosticPayload::Bytes(expected)); } #[test] -fn incomplete_cloud_init_chunk_group_is_skipped() { +fn incomplete_cloud_init_group_preserves_each_physical_record() { let base = "CLOUD_INIT|1786047606|event|test|\ b7a822ba-4eea-46c0-b559-e84396101132"; let chunk = |index: u32, message: &str| { @@ -501,9 +442,18 @@ fn incomplete_cloud_init_chunk_group_is_skipped() { ) }; let records = vec![ - (format!("{base}|0"), chunk(0, "first")), (format!("{base}|2"), chunk(2, "third")), + ("note".to_owned(), "untouched".to_owned()), + (format!("{base}|0"), chunk(0, "first")), ]; - assert!(entries_of(&records).is_empty()); + let entries = entries_of(&records); + let expected: Vec<_> = records + .into_iter() + .map(|(key, value)| { + let error = (key != "note").then_some(DecodeError::IncompleteGroup); + Entry::Raw(RawKeyValue { key, value, error }) + }) + .collect(); + assert_eq!(entries, expected); } diff --git a/libazureinit-kvp/tests/fixtures/cloud_init.rs b/libazureinit-kvp/tests/fixtures/cloud_init.rs new file mode 100644 index 00000000..3eafd5e3 --- /dev/null +++ b/libazureinit-kvp/tests/fixtures/cloud_init.rs @@ -0,0 +1,17 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +pub const COMPRESSED_LOG_CHUNKS: &[(&str, &str)] = &[ + ( + "CLOUD_INIT|1786047606|compressed|cloud-init.log|b7a822ba-4eea-46c0-b559-e84396101132|0", + "{\"name\":\"cloud-init.log\",\"type\":\"compressed\",\"ts\":\"2026-08-06T20:20:13.479078Z\",\"msg_i\":0,\"msg\":\"{\\\"encoding\\\": \\\"gz+b64\\\", \\\"data\\\": \\\"eJxtV0uObDUMnbOKt4GHYjuJbYYIxAoYIQZx7EhIDBjAgN1ziumz1K2uqtY9lWuf3+XB++uwr2N/\\\\n4fEDfsb48vXLP3//8ef3f/37208///jrL7//8OVq2K09s4zekFEn71Lyp2fiBZsU1wxZ5zv+BpAa\\\\nwIzY128A1jhGRMikPJSsPJym6x5ih681gNwA6jzvnvK4U+6dhTdqu/Z4OsaY0y9xutDOkeIZDaw0\\\\nsKTytj1N9x3zVFjMoKnFtGaZNDCzganSNHnj8Hg4Am1eK4fcY2Nn7djuLuvJ5Iq5zVP2PPdWA78a\\\\n+CcyRXBfig14WKnskaNU6UWe1cDsBiZ0LamKNLIJIPyJVDl3nrHyzdh7lvt42F4DqQ3kfPXY2e9h\\\\nMdUyfhqYQDCJvsJoa4VY3KG1j5F3E7UGeC0GE9Xm5p0X5EzeRxYdnG+BCrV0F7Ht7qTeANrZ59Sk\\\\new4YjeU6mIjxSQNAnUZoGU1i2eDc3Ya71KQpo7m8UwTRfc/3oUFKk4c9cPbMoho6nhn+iQ8r1iQn\\\\nrdfAdrrAeeja0gxPp9w3d0FwV+LiVy4YVwMUH9hTp17qVOGQpwHZDs60VwTfqINRsY17Q2l9pmcR\\\\nj+4+qztrp5LYy/KCfmfbXbX4rlBfonb1NEZArRaW1+XDdtRTL0Nm+9LQLaI15t2Yx/tY18WJa3oD\\\\n22mDRtLaV1QGtHAfdvCSyavRKHVKyLev7u34WrOSISCLQQRnPgJ0sBskTw7Ts+hutlOBqN85Rplz\\\\nXuWwj6iyaAhjjIL7P5qqfM9rhEWdDl48ObTXmF44zUznfd+kby/nTgVF70HnA0sQ0h1SL/zZilP1\\\\nsAm4C6UW7\"}", + ), + ( + "CLOUD_INIT|1786047606|compressed|cloud-init.log|b7a822ba-4eea-46c0-b559-e84396101132|1", + "{\"name\":\"cloud-init.log\",\"type\":\"compressed\",\"ts\":\"2026-08-06T20:20:13.479078Z\",\"msg_i\":1,\"msg\":\"ASEz7iBEzbQnUL8DD83hUtijQtvWjYOAogZfCGPdZMbtXKni0Ovzj2g/Ra4G3zZDVx5\\\\niKDbQHQqsHvXgQWB4ONOz21IHWKu3QB0fL8Yx8sNLlgG3Wv78IvN1xdpo0Xu+L6PrwejqVFX59R1\\\\nPgu86zGvSyDdXgQhra3I8Fmy2btddpwvQRbV2dc0RMaathBMrJCmyHIoynHqkQzv9VUb6x7Yazbw\\\\nnSIO3HFlbnWOElSMkncWwntin5KO1C5enuCOn6PdVDtN6COky/gkbjgh6su2R/qD/zFfHU0ccqcE\\\\nZKqmvPWK4Re+IovPxHi9tic+hU0DMO6zSfSpH9zcurRZYR+/NOS9ZwVxIRWLHzoFLB/rM4w469Nn\\\\nZoxN6ZMzpzdZIp1SRsbbG3Vj7YPocHpzudkUxxjxNQZCbIMxLKfAC1AuF0ljZdJpR8ZF3RvTEh0r\\\\nWN8q0Tu6y9tOlSOU5crBpfYxKA6y4ubyTjV8PsG639xy4Smw0oKp66B4oUdjolrFJ48uBFGjsVLp\\\\ndHQwaJ+JsISJokGIvIc6YBAkyLMPPIsnVo6CBKYf7QxVOhU5PDqPwe8hbFsuA2VSRx4nytmAtFqB\\\\n3SEDJbFIWM1aMI6gj82CNzJmV2ilU8dZajljvmFgSJrivtEeAaAo4rhlR/k5DLo1gK1KtvO8+TJj\\\\nfnxHH1wRwvj28tkp4RmthXGj7zAhqdCpBc8Ad+OMoIgQyTiINgGRdeNkzT5np4GHkNCZaFDCmDOC\\\\ntpYzTKFggG6JlhHPPiUykVQNaMf86yAGcgjJUQPlZI3/4xGPGPdsjsazZ6eAWHiUmugR8JI9HKx6\\\\n6NkPhfPDPopEZ1eFMTZbnZ0m8IQDlaPl/S8IG4GF8MJqcOD/AFeindw=\\\"}", + ), + ( + "CLOUD_INIT|1786047606|compressed|cloud-init.log|b7a822ba-4eea-46c0-b559-e84396101132|2", + "{\"name\":\"cloud-init.log\",\"type\":\"compressed\",\"ts\":\"2026-08-06T20:20:13.479078Z\",\"msg_i\":2,\"msg\":\"\\n\\\"}\"}", + ), +]; From b2840afe3f8015dac32bbfdca204aaf3af2633c2 Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Mon, 14 Sep 2026 11:47:18 -0700 Subject: [PATCH 23/32] Improve test coverage to 100 percent per repo standards --- .../src/diagnostics/cloud_init.rs | 92 ++++++------ libazureinit-kvp/src/diagnostics/reader.rs | 132 ++++++++++-------- libazureinit-kvp/src/diagnostics/writer.rs | 56 +++++--- libazureinit-kvp/src/error.rs | 4 + 4 files changed, 168 insertions(+), 116 deletions(-) diff --git a/libazureinit-kvp/src/diagnostics/cloud_init.rs b/libazureinit-kvp/src/diagnostics/cloud_init.rs index 0c51a976..6fccbcdf 100644 --- a/libazureinit-kvp/src/diagnostics/cloud_init.rs +++ b/libazureinit-kvp/src/diagnostics/cloud_init.rs @@ -316,12 +316,13 @@ mod tests { entries } - fn only_diagnostic(entries: Vec) -> Diagnostic { + fn only_diagnostic(mut entries: Vec) -> Diagnostic { assert_eq!(entries.len(), 1); - match entries.into_iter().next().unwrap() { - Entry::Diagnostic(diagnostic) => diagnostic, - other => panic!("expected a diagnostic, got {other:?}"), + let mut diagnostic = None; + if let Some(Entry::Diagnostic(value)) = entries.pop() { + diagnostic = Some(value); } + diagnostic.expect("expected a diagnostic") } fn assert_raw(records: &[(String, String)], error: DecodeError) { @@ -355,6 +356,14 @@ mod tests { assert_eq!(parsed.event_id, EVENT_ID); } + #[test] + fn malformed_base_key_layout_is_rejected() { + assert!(matches!( + parse_key("CLOUD_INIT|100|event|test"), + Err(DecodeError::Malformed) + )); + } + #[rstest] #[case::layout("CLOUD_INIT|100|event".into())] #[case::vm(key("event", true, None).replace(VM_ID, "invalid"))] @@ -417,14 +426,13 @@ mod tests { ), (key("finish", true, None), finish.to_string()), ]); - let [Entry::Diagnostic(Diagnostic::Start(start)), Entry::Diagnostic(Diagnostic::Finish(finish))] = - entries.as_slice() - else { - panic!("expected a start and finish"); - }; - assert_eq!(start.key.event_id, finish.key.event_id); - assert_eq!(finish.result, Outcome::Failure); - assert_eq!(finish.duration_ms, 123); + assert!(matches!( + entries.as_slice(), + [Entry::Diagnostic(Diagnostic::Start(start)), Entry::Diagnostic(Diagnostic::Finish(finish))] + if start.key.event_id == finish.key.event_id + && finish.result == Outcome::Failure + && finish.duration_ms == 123 + )); } #[test] @@ -433,13 +441,11 @@ mod tests { "CLOUD_INIT|1785187982|finish|modules-final/config-scripts_user|0e5e179d-5341-478b-8456-fbb90621bdf8|e5f01809-a7a3-4279-aa64-1f18e21eda6e".into(), r#"{"name":"modules-final/config-scripts_user","type":"finish","ts":"2026-07-27T21:33:24.339006+00:00","result":"SUCCESS","duration":0.0006448590000012189,"msg":"config-scripts_user ran successfully and took 0.001 seconds"}"#.into(), )]; - let Diagnostic::Finish(finish) = only_diagnostic(entries(&records)) - else { - panic!("expected a finish"); - }; - assert_eq!(finish.result, Outcome::Success); - assert_eq!(finish.duration_ms, 0); - assert_eq!(finish.key.name, "modules-final/config-scripts_user"); + let diagnostic = only_diagnostic(entries(&records)); + assert!(matches!(&diagnostic, Diagnostic::Finish(finish) + if finish.result == Outcome::Success + && finish.duration_ms == 0 + && finish.key.name == "modules-final/config-scripts_user")); } #[test] @@ -481,15 +487,25 @@ mod tests { } #[rstest] - #[case::invalid_json("not json".into())] - #[case::timestamp(value("event", "message").to_string().replace(TIMESTAMP, "bad"))] - #[case::name(value("event", "message").to_string().replace("\"test\"", "\"other\""))] - #[case::message(json!({"name":"test", "type":"event", "ts":TIMESTAMP, "msg":7}).to_string())] - fn malformed_source_values_remain_raw(#[case] value: String) { - assert_raw( - &[(key("event", true, None), value)], - DecodeError::Malformed, - ); + #[case::invalid_json("event", "not json".into())] + #[case::timestamp("event", value("event", "message").to_string().replace(TIMESTAMP, "bad"))] + #[case::name("event", value("event", "message").to_string().replace("\"test\"", "\"other\""))] + #[case::message("event", json!({"name":"test", "type":"event", "ts":TIMESTAMP, "msg":7}).to_string())] + #[case::start_result("start", { + let mut metadata = value("start", "message"); + metadata["result"] = json!("SUCCESS"); + metadata.to_string() + })] + #[case::start_duration("start", { + let mut metadata = value("start", "message"); + metadata["duration"] = json!(1); + metadata.to_string() + })] + fn malformed_source_values_remain_raw( + #[case] kind: &str, + #[case] value: String, + ) { + assert_raw(&[(key(kind, true, None), value)], DecodeError::Malformed); } #[test] @@ -568,14 +584,13 @@ mod tests { ), ]; let entries = entries(&records); - let [Entry::Diagnostic(first), Entry::Diagnostic(second)] = - entries.as_slice() - else { - panic!("expected two separate groups"); - }; - assert_eq!(first.key(), second.key()); - assert_eq!(first.payload(), &DiagnosticPayload::Text("ab".into())); - assert_eq!(second.payload(), &DiagnosticPayload::Text("xy".into())); + assert!(matches!( + entries.as_slice(), + [Entry::Diagnostic(first), Entry::Diagnostic(second)] + if first.key() == second.key() + && first.payload() == &DiagnosticPayload::Text("ab".into()) + && second.payload() == &DiagnosticPayload::Text("xy".into()) + )); } #[test] @@ -660,9 +675,8 @@ mod tests { let compressed = STANDARD.decode(ZLIB_DATA).unwrap(); for end in 0..compressed.len() { assert_eq!( - decode_compressed(&STANDARD.encode(&compressed[..end])), - Err(DecodeError::Undecodable), - "prefix {end}" + (end, decode_compressed(&STANDARD.encode(&compressed[..end]))), + (end, Err(DecodeError::Undecodable)) ); } } diff --git a/libazureinit-kvp/src/diagnostics/reader.rs b/libazureinit-kvp/src/diagnostics/reader.rs index 8300246d..3d404d7a 100644 --- a/libazureinit-kvp/src/diagnostics/reader.rs +++ b/libazureinit-kvp/src/diagnostics/reader.rs @@ -312,12 +312,13 @@ mod tests { .collect() } - fn only_diagnostic(entries: Vec) -> Diagnostic { + fn only_diagnostic(mut entries: Vec) -> Diagnostic { assert_eq!(entries.len(), 1); - match entries.into_iter().next().unwrap() { - Entry::Diagnostic(diagnostic) => diagnostic, - other => panic!("expected a diagnostic, got {other:?}"), + let mut diagnostic = None; + if let Some(Entry::Diagnostic(value)) = entries.pop() { + diagnostic = Some(value); } + diagnostic.expect("expected a diagnostic") } fn store(dir: &TempDir) -> KvpPoolStore { @@ -326,15 +327,15 @@ mod tests { } #[derive(Debug, Default)] - struct ReadOnlyOps { + struct ReaderOps { os: OsSysOps, - reads: AtomicUsize, + calls: AtomicUsize, open_error: Option, } - impl SysOps for ReadOnlyOps { + impl SysOps for ReaderOps { fn open_read(&self, path: &Path) -> io::Result> { - self.reads.fetch_add(1, Ordering::SeqCst); + self.calls.fetch_add(1, Ordering::SeqCst); if let Some(error) = self.open_error { return Err(error.into()); } @@ -342,27 +343,31 @@ mod tests { } fn open_read_write(&self, _: &Path) -> io::Result> { - panic!("reader must not write the pool") + self.calls.fetch_add(1, Ordering::SeqCst); + Err(io::ErrorKind::Unsupported.into()) } fn open_read_write_create( &self, _: &Path, ) -> io::Result> { - panic!("reader must not create the pool") + self.calls.fetch_add(1, Ordering::SeqCst); + Err(io::ErrorKind::Unsupported.into()) } fn path_metadata(&self, _: &Path) -> io::Result { - panic!("reader must not inspect pool staleness") + self.calls.fetch_add(1, Ordering::SeqCst); + Err(io::ErrorKind::Unsupported.into()) } fn boot_time(&self) -> io::Result { - panic!("reader must not read boot state") + self.calls.fetch_add(1, Ordering::SeqCst); + Err(io::ErrorKind::Unsupported.into()) } } - fn observed_reader(dir: &TempDir) -> (DiagnosticReader, Arc) { - let ops = Arc::new(ReadOnlyOps::default()); + fn observed_reader(dir: &TempDir) -> (DiagnosticReader, Arc) { + let ops = Arc::new(ReaderOps::default()); let observed = KvpPoolStore::with_ops( KvpPool::Guest, dir.path(), @@ -373,11 +378,29 @@ mod tests { (DiagnosticReader::new(observed), ops) } + #[test] + fn reader_ops_rejects_non_read_operations() { + let dir = TempDir::new().unwrap(); + let pool = store(&dir); + let ops = ReaderOps::default(); + assert_eq!( + [ + ops.open_read_write(pool.path()).unwrap_err().kind(), + ops.open_read_write_create(pool.path()).unwrap_err().kind(), + ops.path_metadata(pool.path()).unwrap_err().kind(), + ops.boot_time().unwrap_err().kind(), + ], + [io::ErrorKind::Unsupported; 4] + ); + assert_eq!(ops.calls.load(Ordering::SeqCst), 4); + assert!(!pool.path().exists()); + } + #[test] fn constructor_does_no_io() { let dir = TempDir::new().unwrap(); let (reader, ops) = observed_reader(&dir); - assert_eq!(ops.reads.load(Ordering::SeqCst), 0); + assert_eq!(ops.calls.load(Ordering::SeqCst), 0); assert!(!reader.store.path().exists()); } @@ -386,8 +409,9 @@ mod tests { let dir = TempDir::new().unwrap(); let pool = store(&dir); let (reader, ops) = observed_reader(&dir); + assert_eq!(ops.calls.load(Ordering::SeqCst), 0); assert!(reader.entries().unwrap().is_empty()); - assert_eq!(ops.reads.load(Ordering::SeqCst), 1); + assert_eq!(ops.calls.load(Ordering::SeqCst), 1); assert!(!pool.path().exists()); pool.append("unrelated", "unchanged").unwrap(); @@ -396,16 +420,16 @@ mod tests { reader.entries().unwrap(), raw_entries(&[("unrelated".into(), "unchanged".into())], None) ); - assert_eq!(ops.reads.load(Ordering::SeqCst), 2); + assert_eq!(ops.calls.load(Ordering::SeqCst), 2); assert_eq!(fs::read(pool.path()).unwrap(), before); } #[test] fn snapshot_open_errors_propagate() { let dir = TempDir::new().unwrap(); - let ops = Arc::new(ReadOnlyOps { + let ops = Arc::new(ReaderOps { open_error: Some(io::ErrorKind::PermissionDenied), - ..ReadOnlyOps::default() + ..ReaderOps::default() }); let pool = KvpPoolStore::with_ops( KvpPool::Guest, @@ -415,13 +439,13 @@ mod tests { ) .unwrap(); let reader = DiagnosticReader::new(pool); - assert_eq!(ops.reads.load(Ordering::SeqCst), 0); + assert_eq!(ops.calls.load(Ordering::SeqCst), 0); assert!(matches!( reader.entries(), Err(KvpError::Io(error)) if error.kind() == io::ErrorKind::PermissionDenied )); - assert_eq!(ops.reads.load(Ordering::SeqCst), 1); + assert_eq!(ops.calls.load(Ordering::SeqCst), 1); } #[test] @@ -619,17 +643,16 @@ mod tests { (start, "starting".into()), (finish, "failed".into()), ]); - let [Entry::Diagnostic(Diagnostic::Start(start)), Entry::Diagnostic(Diagnostic::Finish(finish))] = - entries.as_slice() - else { - panic!("expected separate start and finish entries"); - }; - assert_eq!(start.key.event_id, EVENT_ID); - assert_eq!(finish.key.event_id, EVENT_ID); - assert_eq!(start.payload, DiagnosticPayload::Text("starting".into())); - assert_eq!(finish.payload, DiagnosticPayload::Text("failed".into())); - assert_eq!(finish.result, Outcome::Failure); - assert_eq!(finish.duration_ms, 312); + assert!(matches!( + entries.as_slice(), + [Entry::Diagnostic(Diagnostic::Start(start)), Entry::Diagnostic(Diagnostic::Finish(finish))] + if start.key.event_id == EVENT_ID + && finish.key.event_id == EVENT_ID + && start.payload == DiagnosticPayload::Text("starting".into()) + && finish.payload == DiagnosticPayload::Text("failed".into()) + && finish.result == Outcome::Failure + && finish.duration_ms == 312 + )); } #[rstest] @@ -648,11 +671,8 @@ mod tests { let key = with_field(&key, 9, &duration_token); let diagnostic = only_diagnostic(decode_entries(vec![(key, "value".into())])); - let Diagnostic::Event(event) = diagnostic else { - panic!("expected an event"); - }; - assert_eq!(event.result, result); - assert_eq!(event.duration_ms, duration); + assert!(matches!(&diagnostic, Diagnostic::Event(event) + if event.result == result && event.duration_ms == duration)); } #[rstest] @@ -758,16 +778,13 @@ mod tests { (later, "a".into()), ]; let entries = decode_entries(records); - assert_eq!(entries.len(), 3); - let Entry::Diagnostic(first) = &entries[0] else { - panic!("expected first-seen group"); - }; - assert_eq!(first.payload(), &DiagnosticPayload::Text("ab".into())); - assert!(matches!(&entries[1], Entry::Raw(raw) if raw.key == "raw")); - let Entry::Diagnostic(last) = &entries[2] else { - panic!("expected earlier-timestamp group"); - }; - assert!(first.key().timestamp > last.key().timestamp); + assert!(matches!( + entries.as_slice(), + [Entry::Diagnostic(first), Entry::Raw(raw), Entry::Diagnostic(last)] + if first.payload() == &DiagnosticPayload::Text("ab".into()) + && raw.key == "raw" + && first.key().timestamp > last.key().timestamp + )); } #[rstest] @@ -889,15 +906,12 @@ mod tests { (with_field(&key(1), field, second), "d".into()), ]; let entries = decode_entries(records); - assert_eq!(entries.len(), 2); - let Entry::Diagnostic(first) = &entries[0] else { - panic!("expected first group"); - }; - let Entry::Diagnostic(second) = &entries[1] else { - panic!("expected second group"); - }; - assert_eq!(first.payload(), &DiagnosticPayload::Text("ac".into())); - assert_eq!(second.payload(), &DiagnosticPayload::Text("bd".into())); + assert!(matches!( + entries.as_slice(), + [Entry::Diagnostic(first), Entry::Diagnostic(second)] + if first.payload() == &DiagnosticPayload::Text("ac".into()) + && second.payload() == &DiagnosticPayload::Text("bd".into()) + )); } #[test] @@ -1012,10 +1026,8 @@ mod tests { let key = with_field(&key(0), 9, &u64::MAX.to_string()); let diagnostic = only_diagnostic(decode_entries(vec![(key, "payload".into())])); - let Diagnostic::Event(event) = diagnostic else { - panic!("expected an event"); - }; - assert_eq!(event.duration_ms, Some(u64::MAX)); + assert!(matches!(&diagnostic, Diagnostic::Event(event) + if event.duration_ms == Some(u64::MAX))); } #[test] diff --git a/libazureinit-kvp/src/diagnostics/writer.rs b/libazureinit-kvp/src/diagnostics/writer.rs index 4b488735..0a2ef171 100644 --- a/libazureinit-kvp/src/diagnostics/writer.rs +++ b/libazureinit-kvp/src/diagnostics/writer.rs @@ -280,39 +280,43 @@ mod tests { } #[derive(Debug, Default)] - struct WriteOnlyOps { + struct WriterOps { os: OsSysOps, - opens: AtomicUsize, + calls: AtomicUsize, } - impl SysOps for WriteOnlyOps { + impl SysOps for WriterOps { fn open_read(&self, _: &Path) -> io::Result> { - panic!("writer must not read the pool") + self.calls.fetch_add(1, Ordering::SeqCst); + Err(io::ErrorKind::Unsupported.into()) } fn open_read_write(&self, _: &Path) -> io::Result> { - panic!("writer must only append") + self.calls.fetch_add(1, Ordering::SeqCst); + Err(io::ErrorKind::Unsupported.into()) } fn open_read_write_create( &self, path: &Path, ) -> io::Result> { - self.opens.fetch_add(1, Ordering::SeqCst); + self.calls.fetch_add(1, Ordering::SeqCst); self.os.open_read_write_create(path) } fn path_metadata(&self, _: &Path) -> io::Result { - panic!("writer must not inspect pool staleness") + self.calls.fetch_add(1, Ordering::SeqCst); + Err(io::ErrorKind::Unsupported.into()) } fn boot_time(&self) -> io::Result { - panic!("writer must not read boot state") + self.calls.fetch_add(1, Ordering::SeqCst); + Err(io::ErrorKind::Unsupported.into()) } } - fn observed_writer(dir: &TempDir) -> (DiagnosticWriter, Arc) { - let ops = Arc::new(WriteOnlyOps::default()); + fn observed_writer(dir: &TempDir) -> (DiagnosticWriter, Arc) { + let ops = Arc::new(WriterOps::default()); let store = KvpPoolStore::with_ops( KvpPool::Guest, dir.path(), @@ -323,6 +327,24 @@ mod tests { (DiagnosticWriter::new(store, AGENT, VM_ID).unwrap(), ops) } + #[test] + fn writer_ops_rejects_non_append_operations() { + let dir = TempDir::new().unwrap(); + let pool = store(&dir, PoolMode::Safe); + let ops = WriterOps::default(); + assert_eq!( + [ + ops.open_read(pool.path()).unwrap_err().kind(), + ops.open_read_write(pool.path()).unwrap_err().kind(), + ops.path_metadata(pool.path()).unwrap_err().kind(), + ops.boot_time().unwrap_err().kind(), + ], + [io::ErrorKind::Unsupported; 4] + ); + assert_eq!(ops.calls.load(Ordering::SeqCst), 4); + assert!(!pool.path().exists()); + } + fn assert_rejected_without_writes( operation: impl FnOnce(&DiagnosticWriter) -> Result<(), KvpError>, ) -> KvpError { @@ -332,7 +354,7 @@ mod tests { let before = fs::read(pool.path()).unwrap(); let (writer, ops) = observed_writer(&dir); let error = operation(&writer).unwrap_err(); - assert_eq!(ops.opens.load(Ordering::SeqCst), 0); + assert_eq!(ops.calls.load(Ordering::SeqCst), 0); assert_eq!(fs::read(pool.path()).unwrap(), before); error } @@ -342,14 +364,14 @@ mod tests { let dir = TempDir::new().unwrap(); let pool = store(&dir, PoolMode::Safe); let (writer, ops) = observed_writer(&dir); - assert_eq!(ops.opens.load(Ordering::SeqCst), 0); + assert_eq!(ops.calls.load(Ordering::SeqCst), 0); assert!(!pool.path().exists()); let payload = "x".repeat(MAX_CHUNK_BYTES * 3 + 1); writer .emit_event("test", payload, None, None, None) .unwrap(); - assert_eq!(ops.opens.load(Ordering::SeqCst), 1); + assert_eq!(ops.calls.load(Ordering::SeqCst), 1); assert_eq!(pool.dump().unwrap().len(), 4); } @@ -361,7 +383,7 @@ mod tests { DiagnosticWriter::new(writer.store, "", VM_ID), Err(KvpError::EmptyEventField { field: "agent" }) )); - assert_eq!(ops.opens.load(Ordering::SeqCst), 0); + assert_eq!(ops.calls.load(Ordering::SeqCst), 0); } #[test] @@ -372,7 +394,7 @@ mod tests { DiagnosticWriter::new(writer.store, AGENT, "vm-abc"), Err(KvpError::InvalidUuid { field: "vm_id" }) )); - assert_eq!(ops.opens.load(Ordering::SeqCst), 0); + assert_eq!(ops.calls.load(Ordering::SeqCst), 0); } #[rstest] @@ -729,7 +751,7 @@ mod tests { writer.emit_event("test", payload, None, None, None), Err(KvpError::TooManyChunks { max: MAX_CHUNKS }) )); - assert_eq!(ops.opens.load(Ordering::SeqCst), 0); + assert_eq!(ops.calls.load(Ordering::SeqCst), 0); assert_eq!(fs::read(pool.path()).unwrap(), before); } @@ -892,7 +914,7 @@ mod tests { Err(KvpError::DurationTooLarge { max_ms: MAX_DURATION_MS, actual_ms }) if actual_ms == duration_ms )); - assert_eq!(ops.opens.load(Ordering::SeqCst), 0); + assert_eq!(ops.calls.load(Ordering::SeqCst), 0); } #[test] diff --git a/libazureinit-kvp/src/error.rs b/libazureinit-kvp/src/error.rs index db8fd060..ea7ebaef 100644 --- a/libazureinit-kvp/src/error.rs +++ b/libazureinit-kvp/src/error.rs @@ -134,6 +134,10 @@ mod tests { KvpError::EmptyEventField { field: "name" }, "event key field 'name' must not be empty" )] + #[case( + KvpError::EventFieldContainsDelimiter { field: "name" }, + "event key field 'name' must not contain '|'" + )] #[case( KvpError::EventFieldTooLong { field: "name", max: 48, actual: 49 }, "event key field 'name' length (49) exceeds maximum (48)" From 6c3914d74d205a23e1410c3115ae71ae9ceb9346 Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Mon, 14 Sep 2026 15:54:10 -0700 Subject: [PATCH 24/32] =?UTF-8?q?fix(kvp):=20address=20review=20=E2=80=94?= =?UTF-8?q?=20restore=20pool=20order=20for=20dump=20--parse,=20add=20--kin?= =?UTF-8?q?d,=20drop=20dev=20uuid?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- doc/kvp.md | 11 +- libazureinit-kvp/Cargo.toml | 1 - libazureinit-kvp/diagnostics-proposal.md | 4 +- libazureinit-kvp/src/cli.rs | 105 ++++++++++++++---- libazureinit-kvp/src/lib.rs | 3 +- libazureinit-kvp/src/report.rs | 8 -- libazureinit-kvp/tests/cli.rs | 134 +++++++++++++++-------- 7 files changed, 178 insertions(+), 88 deletions(-) diff --git a/doc/kvp.md b/doc/kvp.md index f2c89b04..fba954bd 100644 --- a/doc/kvp.md +++ b/doc/kvp.md @@ -261,14 +261,15 @@ Neither reading nor emitting diagnostics clears stale pool data implicitly. | Command | Output | |---------|--------| | `libazureinit-kvp dump` | JSON array of physical key/value records in pool order, including duplicates | -| `libazureinit-kvp dump --parse` | JSON array of diagnostics and reports in timestamp order, then raw entries | +| `libazureinit-kvp dump --parse` | JSON array of decoded diagnostics, reports, and raw entries in pool order | | `libazureinit-kvp dump --text` | Physical records as `KEY=VALUE` lines | -| `libazureinit-kvp dump --parse --text` | The same timestamp ordering, with binary payloads under `payload_b64` | +| `libazureinit-kvp dump --parse --text` | One line per entry in pool order; binary payloads render as `payload_b64=` | | `libazureinit-kvp dump --parse --name ssh` | Filter diagnostic names by substring; retain reports and raw entries | +| `libazureinit-kvp dump --parse --kind finish` | Filter diagnostics by kind (`start`/`finish`/`event`); adding `--name` keeps only diagnostics that match both filters | -Parsed CLI output is oldest-first. Equal timestamps keep first-seen order; -raw entries retain their relative pool order at the end. This presentation -does not change `DiagnosticReader::entries()` or the pool file. +Parsed CLI output preserves the pool order returned by +`DiagnosticReader::entries()`: a complete chunk group appears at its first +record's position, and every other record stays where it sits in the pool. `--json` and `--text` are mutually exclusive. Only `dump` defaults to JSON; other commands retain their text defaults. Global `--dir` and `--pool` diff --git a/libazureinit-kvp/Cargo.toml b/libazureinit-kvp/Cargo.toml index 3f1c05fd..108a4ddf 100644 --- a/libazureinit-kvp/Cargo.toml +++ b/libazureinit-kvp/Cargo.toml @@ -23,7 +23,6 @@ uuid = { version = "1.3", features = ["v4"] } [dev-dependencies] rstest = { version = "0.26", default-features = false } tempfile = "3" -uuid = "1.3" [lib] name = "libazureinit_kvp" diff --git a/libazureinit-kvp/diagnostics-proposal.md b/libazureinit-kvp/diagnostics-proposal.md index 0b5e91cd..6209015b 100644 --- a/libazureinit-kvp/diagnostics-proposal.md +++ b/libazureinit-kvp/diagnostics-proposal.md @@ -232,7 +232,7 @@ Only the exact `PROVISIONING_REPORT` key selects report parsing. Its value is on Writing is the inverse: `DiagnosticWriter` stamps `DIAG_V1`, converts the typed payload according to `encoding`, frames it into records, and appends them to the `KvpPoolStore`. -The reader preserves first-seen pool order: a complete chunk group occupies its first physical position, and failed groups remain raw records at their original positions. Only the CLI's `dump --parse` sorts diagnostics and reports by timestamp, oldest first, with stable ties and raw entries last. A span's timeline can be summarized as: +The reader preserves first-seen pool order: a complete chunk group occupies its first physical position, and failed groups remain raw records at their original positions. The CLI's `dump --parse` renders entries in that same pool order. A span's timeline can be summarized as: ```text 2026-08-31T12:34:56.789Z start provision:run @@ -434,7 +434,7 @@ impl DiagnosticWriter { `dump` defaults to JSON; `--json` makes that explicit and `--text` selects human-readable output. Without `--parse`, it returns every physical record in pool order. `--parse` calls `DiagnosticReader::entries()`, returning typed diagnostics and reports while preserving other or invalid records as `Raw`. Both modes require a successful string-based snapshot; invalid physical UTF-8 fails the command without returning records. -Parsed JSON and text output sort diagnostics and reports by their timestamps as instants, oldest first, including timezone offsets and available fractional precision. Equal timestamps retain first-seen pool order. Raw entries follow the timestamped entries in their original relative pool order; the CLI does not infer timestamps from malformed or unrecognized records. This presentation does not change the reader API's ordering or write to the pool. +Parsed JSON and text output preserve the reader's first-seen pool order; the CLI does not reorder entries. In parsed text output, binary payloads render as `payload_b64=`. This presentation does not change the reader API's ordering or write to the pool. ```text dump -> JSON array of every physical {key, value} record diff --git a/libazureinit-kvp/src/cli.rs b/libazureinit-kvp/src/cli.rs index 53d9b389..af1c7861 100644 --- a/libazureinit-kvp/src/cli.rs +++ b/libazureinit-kvp/src/cli.rs @@ -14,7 +14,7 @@ use serde_json::json; use crate::{ write_report, Diagnostic, DiagnosticPayload, DiagnosticReader, - DiagnosticWriter, Entry, KvpError, KvpPool, KvpPoolStore, PoolMode, + DiagnosticWriter, Entry, Kind, KvpError, KvpPool, KvpPoolStore, PoolMode, ProvisioningReport, ReportPpsType, PROVISIONING_REPORT_KEY, }; @@ -98,15 +98,18 @@ enum Command { Info, /// Print every record in pool order (JSON by default; --text for KEY=VALUE). /// - /// With --parse, sort diagnostics and reports by timestamp, oldest first. - /// Equal timestamps keep pool order; raw entries follow in pool order. + /// With --parse, decode diagnostics and reports, preserving other or + /// invalid records as raw entries. Entries stay in pool order. Dump { - /// Decode diagnostics and reports in oldest-first timestamp order. + /// Decode diagnostics and provisioning reports (kept in pool order). #[arg(long)] parse: bool, /// Filter diagnostic names by substring; retain reports and raw entries. #[arg(long, requires = "parse")] name: Option, + /// Filter diagnostics by kind; retain reports and raw entries. + #[arg(long, value_enum, requires = "parse")] + kind: Option, }, /// Print key=last_value entries sorted by key. Entries, @@ -238,6 +241,24 @@ impl From for KvpPool { } } +/// Diagnostic kind accepted by `dump --parse --kind`. +#[derive(ValueEnum, Clone, Copy, Debug)] +enum KindArg { + Start, + Finish, + Event, +} + +impl From for Kind { + fn from(value: KindArg) -> Self { + match value { + KindArg::Start => Self::Start, + KindArg::Finish => Self::Finish, + KindArg::Event => Self::Event, + } + } +} + fn dispatch(cli: Cli, stdout: &mut W) -> Result { if cli.json && cli.text { return Err(CliError::Usage( @@ -260,13 +281,20 @@ fn dispatch(cli: Cli, stdout: &mut W) -> Result { match cli.command { Command::Info => info(&store, stdout, output), - Command::Dump { parse, name } => { + Command::Dump { parse, name, kind } => { let output = if cli.text { OutputMode::Text } else { OutputMode::Json }; - dump(&store, stdout, parse, name.as_deref(), output) + dump( + &store, + stdout, + parse, + name.as_deref(), + kind.map(Kind::from), + output, + ) } Command::Entries => entries(&store, stdout, output), Command::Read { key } => read(&store, stdout, &key, output), @@ -368,10 +396,11 @@ fn dump( stdout: &mut W, parse: bool, name: Option<&str>, + kind: Option, output: OutputMode, ) -> Result { if parse { - return diagnostics_entries(store, stdout, name, output); + return diagnostics_entries(store, stdout, name, kind, output); } let records = store.dump()?; @@ -486,28 +515,21 @@ fn diagnostics_entries( store: &KvpPoolStore, stdout: &mut W, name: Option<&str>, + kind: Option, output: OutputMode, ) -> Result { let mut entries = DiagnosticReader::new(store.clone()).entries()?; - if let Some(needle) = name { + if name.is_some() || kind.is_some() { entries.retain(|entry| match entry { Entry::Diagnostic(diagnostic) => { - diagnostic.key().name.contains(needle) + name.is_none_or(|needle| diagnostic.key().name.contains(needle)) + && kind.is_none_or(|wanted| diagnostic.kind() == wanted) } Entry::Report(_) | Entry::Raw(_) => true, }); } - entries.sort_by_cached_key(|entry| { - let timestamp = match entry { - Entry::Diagnostic(diagnostic) => Some(diagnostic.key().timestamp), - Entry::Report(report) => Some(report.timestamp()), - Entry::Raw(_) => None, - }; - (timestamp.is_none(), timestamp) - }); - match output { OutputMode::Text => { for entry in &entries { @@ -933,6 +955,7 @@ mod tests { Command::Dump { parse: false, name: None, + kind: None, } } @@ -1887,21 +1910,22 @@ mod tests { Command::Dump { parse: true, name: Some("keep".into()), + kind: None, }, )); let entries = parse_json(&output); let entries = entries.as_array().unwrap(); assert_eq!(entries.len(), 4); - assert_eq!(entries[0]["type"], "diagnostic"); - assert_eq!(entries[0]["name"], "keep"); - assert_eq!(entries[0]["payload"], "visible"); assert_eq!( - entries[1], - serde_json::to_value(Entry::Report(report)).unwrap() + entries[0], + json!({"type": "raw", "key": "note", "value": "raw value"}) ); + assert_eq!(entries[1]["type"], "diagnostic"); + assert_eq!(entries[1]["name"], "keep"); + assert_eq!(entries[1]["payload"], "visible"); assert_eq!( entries[2], - json!({"type": "raw", "key": "note", "value": "raw value"}) + serde_json::to_value(Entry::Report(report)).unwrap() ); assert_eq!( entries[3], @@ -1912,6 +1936,39 @@ mod tests { ); } + #[test] + fn dispatch_parsed_dump_filters_by_kind() { + let dir = TempDir::new().unwrap(); + let store = store_at(&dir); + let event_id = "8f3e9c4a-1b2c-4d5e-9f01-234567890abc"; + let ts = "2026-08-31T12:34:56.789Z"; + let vm = "00000000-0000-0000-0000-000000000abc"; + let diag = |kind: &str| { + format!("DIAG_V1|agent|{vm}|{kind}|span|{event_id}|{ts}|none|||0") + }; + store.append(&diag("start"), "starting").unwrap(); + store.append(&diag("event"), "obs").unwrap(); + store.append("note", "raw").unwrap(); + + let (_, output) = run_dispatch(cli( + &dir, + Command::Dump { + parse: true, + name: None, + kind: Some(KindArg::Start), + }, + )); + let entries = parse_json(&output); + let entries = entries.as_array().unwrap(); + assert_eq!(entries.len(), 2); + assert_eq!(entries[0]["kind"], "start"); + assert_eq!(entries[0]["name"], "span"); + assert_eq!( + entries[1], + json!({"type": "raw", "key": "note", "value": "raw"}) + ); + } + #[test] fn dispatch_entries_json_emits_sorted_object() { let dir = TempDir::new().unwrap(); diff --git a/libazureinit-kvp/src/lib.rs b/libazureinit-kvp/src/lib.rs index c50d0e08..812216d7 100644 --- a/libazureinit-kvp/src/lib.rs +++ b/libazureinit-kvp/src/lib.rs @@ -18,8 +18,7 @@ //! The reader preserves first-seen pool order. Unknown or invalid records //! remain [`Entry::Raw`] within a successful snapshot; a failed snapshot, //! including invalid physical UTF-8, returns an error without entries. -//! The CLI's `dump --parse` sorts diagnostics and reports by timestamp, -//! oldest first, with stable ties and raw entries last. +//! The CLI's `dump --parse` renders those entries in the same pool order. //! //! ```no_run //! use libazureinit_kvp::{ diff --git a/libazureinit-kvp/src/report.rs b/libazureinit-kvp/src/report.rs index ae0b307c..cb3f9a3a 100644 --- a/libazureinit-kvp/src/report.rs +++ b/libazureinit-kvp/src/report.rs @@ -209,14 +209,6 @@ impl ProvisioningReport { self.extra.push((key.into(), value.into())); self } - - pub(crate) fn timestamp(&self) -> DateTime { - DateTime::parse_from_rfc3339(&self.timestamp) - .expect( - "report timestamps are constructed or validated as RFC 3339", - ) - .with_timezone(&Utc) - } } impl FromStr for ProvisioningReport { diff --git a/libazureinit-kvp/tests/cli.rs b/libazureinit-kvp/tests/cli.rs index fc91ef4a..2d2e27d5 100644 --- a/libazureinit-kvp/tests/cli.rs +++ b/libazureinit-kvp/tests/cli.rs @@ -351,22 +351,21 @@ fn parsed_dump_reassembles_and_filters_without_dropping_other_entries() { assert_eq!(entries[0]["kind"], "event"); assert_eq!(entries[0]["name"], "a:b"); assert_eq!(entries[0]["payload"], "one/two"); - let timed_entries = &entries.as_array().unwrap()[1..3]; - assert!(timed_entries.iter().any(|entry| { - entry["type"] == "PROVISIONING_REPORT" && entry["result"] == "success" - })); - assert!(timed_entries.iter().any(|entry| entry["name"] == "ssh:key")); assert_eq!( - entries[3], + entries[1], json!({"type": "raw", "key": "note", "value": "raw value"}) ); assert_eq!( - entries[4], + entries[2], json!({ "type": "raw", "key": "DIAG_V2|future", "value": "preserved", "error": "unsupported_version", }) ); + assert_eq!(entries[3]["type"], "PROVISIONING_REPORT"); + assert_eq!(entries[3]["result"], "success"); + assert_eq!(entries[4]["type"], "diagnostic"); + assert_eq!(entries[4]["name"], "ssh:key"); let filtered = assert_json(kvp(&with_dir( &dir, @@ -380,66 +379,40 @@ fn parsed_dump_reassembles_and_filters_without_dropping_other_entries() { } #[test] -fn parsed_dump_sorts_timestamps_without_reordering_reader_or_raw_dump() { +fn parsed_dump_preserves_reader_pool_order() { let dir = TempDir::new().unwrap(); let store = store_at(&dir); let event_id = "e5f01809-a7a3-4279-aa64-1f18e21eda6e"; let key = |name: &str, timestamp: &str| { format!("DIAG_V1|agent|{VM_ID}|event|{name}|{event_id}|{timestamp}|none|||0") }; - let records = vec![ + // Timestamps are deliberately out of order to prove the CLI does not sort. + let records = [ ("note".into(), "raw first"), (key("latest", "2026-08-31T00:00:03.000Z"), "latest"), - ( - format!("CLOUD_INIT|100|event|cloud|{VM_ID}|{event_id}"), - r#"{"name":"cloud","type":"event","ts":"2026-08-31T00:00:02.000500Z","msg":"cloud"}"#, - ), ( PROVISIONING_REPORT_KEY.into(), - "result=success|agent=agent|vm_id=vm|pps_type=None|timestamp=2026-08-31T02:00:02+02:00", + "result=success|agent=agent|vm_id=vm|pps_type=None|timestamp=2026-08-31T00:00:02Z", ), ("DIAG_V2|future".into(), "raw second"), - (key("tie-first", "2026-08-31T00:00:02.000Z"), "first tie"), (key("earliest", "2026-08-31T00:00:01.000Z"), "earliest"), - (key("tie-second", "2026-08-31T00:00:02.000Z"), "second tie"), ]; store .append_multiple(records.iter().map(|(key, value)| (key, *value))) .unwrap(); let before = fs::read(store.path()).unwrap(); + + // Parsed output matches the reader entries verbatim (first-seen pool order). let reader_entries = serde_json::to_value( libazureinit_kvp::DiagnosticReader::new(store.clone()) .entries() .unwrap(), ) .unwrap(); - let expected = Value::Array( - [6, 3, 5, 7, 2, 1, 0, 4] - .into_iter() - .map(|position| reader_entries[position].clone()) - .collect(), - ); let parsed = assert_json(kvp(&with_dir(&dir, &["dump", "--parse"]))); - assert_eq!(parsed, expected); - assert_eq!(parsed[1]["timestamp"], "2026-08-31T02:00:02+02:00"); - - let text = - assert_success(kvp(&with_dir(&dir, &["dump", "--parse", "--text"]))); - let expected_labels = [ - "name=earliest ", - "PROVISIONING_REPORT=", - "name=tie-first ", - "name=tie-second ", - "name=cloud ", - "name=latest ", - "raw key=note ", - "raw key=DIAG_V2|future ", - ]; - assert_eq!(text.lines().count(), expected_labels.len()); - for (line, label) in text.lines().zip(expected_labels) { - assert!(line.contains(label), "expected {label:?} in {line:?}"); - } + assert_eq!(parsed, reader_entries); + // Physical dump keeps every record in file order. let physical = assert_json(kvp(&with_dir(&dir, &["dump"]))); let expected_physical: Vec<_> = records .iter() @@ -529,15 +502,15 @@ fn parsed_dump_renders_bytes_reports_and_raw_errors() { assert_eq!(lines.len(), 4); assert!(lines[0] .contains("encoding=gz+b64 result=fail duration=7ms payload_b64=AP8=")); + assert_eq!(lines[1], "raw key=note value=raw value"); + assert_eq!(lines[2], "raw key=DIAG_V1|bad value=junk error=malformed diagnostic or provisioning report"); assert_eq!( - lines[1], + lines[3], format!( "PROVISIONING_REPORT={}", store.read(PROVISIONING_REPORT_KEY).unwrap().unwrap() ) ); - assert_eq!(lines[2], "raw key=note value=raw value"); - assert_eq!(lines[3], "raw key=DIAG_V1|bad value=junk error=malformed diagnostic or provisioning report"); let entries = assert_json(kvp(&with_dir(&dir, &["dump", "--parse", "--json"]))); @@ -545,8 +518,8 @@ fn parsed_dump_renders_bytes_reports_and_raw_errors() { entries[0]["payload"], json!({"type": "bytes", "encoding": "base64", "data": "AP8="}) ); - assert_eq!(entries[1]["reason"], "bad input"); - assert_eq!(entries[3]["error"], "malformed"); + assert_eq!(entries[2]["error"], "malformed"); + assert_eq!(entries[3]["reason"], "bad input"); } #[test] @@ -558,6 +531,75 @@ fn dump_name_requires_parse() { .contains("--parse")); } +#[test] +fn parsed_dump_filters_by_kind() { + let dir = TempDir::new().unwrap(); + let event_id = "e5f01809-a7a3-4279-aa64-1f18e21eda6e"; + let ts = "2026-08-31T00:00:00.000Z"; + let diag = |kind: &str, name: &str, result: &str, duration: &str| { + format!( + "DIAG_V1|agent|{VM_ID}|{kind}|{name}|{event_id}|{ts}|none|{result}|{duration}|0" + ) + }; + for (key, value) in [ + (diag("start", "provision:run", "", ""), "starting"), + (diag("finish", "provision:run", "success", "312"), "done"), + (diag("event", "imds", "", ""), "ok"), + ("note".to_string(), "raw value"), + ] { + assert_success(kvp(&with_dir( + &dir, + &["write", "--append", &key, value], + ))); + } + assert_success(kvp(&with_dir(&dir, &["report-success", "--vm-id", VM_ID]))); + + // --kind keeps only diagnostics of that kind; reports and raw remain. + let finish = assert_json(kvp(&with_dir( + &dir, + &["dump", "--parse", "--kind", "finish"], + ))); + let finish = finish.as_array().unwrap(); + assert_eq!(finish.len(), 3); + assert_eq!(finish[0]["kind"], "finish"); + assert_eq!(finish[0]["name"], "provision:run"); + assert_eq!( + finish[1], + json!({"type": "raw", "key": "note", "value": "raw value"}) + ); + assert_eq!(finish[2]["type"], "PROVISIONING_REPORT"); + + // --name and --kind combine with AND semantics. + let combined = assert_json(kvp(&with_dir( + &dir, + &["dump", "--parse", "--kind", "start", "--name", "provision"], + ))); + let combined = combined.as_array().unwrap(); + assert_eq!(combined.len(), 3); + assert_eq!(combined[0]["kind"], "start"); + + // A kind/name pair matching no diagnostic keeps only reports and raw. + let empty = assert_json(kvp(&with_dir( + &dir, + &["dump", "--parse", "--kind", "start", "--name", "imds"], + ))); + let empty = empty.as_array().unwrap(); + assert_eq!(empty.len(), 2); + assert!(empty.iter().all(|entry| entry["type"] != "diagnostic")); +} + +#[test] +fn dump_kind_requires_parse_and_rejects_unknown_value() { + let requires_parse = kvp(&["dump", "--kind", "finish"]); + assert_eq!(requires_parse.status.code(), Some(2)); + assert!(String::from_utf8(requires_parse.stderr) + .unwrap() + .contains("--parse")); + + let unknown = kvp(&["dump", "--parse", "--kind", "bogus"]); + assert_eq!(unknown.status.code(), Some(2)); +} + #[test] fn conflicting_output_flags_fail_before_pool_access() { let dir = TempDir::new().unwrap(); From 401167de99de273b7ba9c4fc36e9bc1f1530bccc Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Tue, 15 Sep 2026 13:46:35 -0700 Subject: [PATCH 25/32] feat(kvp): record DIAG_V1 duration in microseconds and accept canonical timestamp precisions --- doc/kvp.md | 6 +- libazureinit-kvp/diagnostics-proposal.md | 32 +-- libazureinit-kvp/src/cli.rs | 14 +- .../src/diagnostics/cloud_init.rs | 40 ++-- .../src/diagnostics/diagnostic.rs | 72 +++++-- libazureinit-kvp/src/diagnostics/mod.rs | 2 +- libazureinit-kvp/src/diagnostics/reader.rs | 76 ++++--- libazureinit-kvp/src/diagnostics/writer.rs | 191 +++++++++++++----- libazureinit-kvp/src/error.rs | 12 +- libazureinit-kvp/src/lib.rs | 4 +- libazureinit-kvp/tests/cli.rs | 13 +- libazureinit-kvp/tests/diagnostics.rs | 20 +- 12 files changed, 324 insertions(+), 158 deletions(-) diff --git a/doc/kvp.md b/doc/kvp.md index fba954bd..d12ac266 100644 --- a/doc/kvp.md +++ b/doc/kvp.md @@ -248,7 +248,11 @@ no truncation. `DiagnosticReader` for diagnostics, provisioning reports, and raw records. The writer validates the agent identifier and VM UUID at construction. `emit_start` and `emit_finish` share a caller-supplied event UUID; -`emit_event` generates its own. Durations are integer milliseconds. +`emit_event` generates its own. Durations are `std::time::Duration` +values stored as integer microseconds. Timestamps are RFC 3339 UTC (`Z`) +at millisecond precision by default; `DiagnosticWriter::with_timestamp_precision` +selects second, microsecond, or nanosecond precision, and the reader +accepts any of those canonical precisions. Payloads are plain UTF-8 text or gzip plus base64 (`Encoding::GzB64`). Each reader call takes one fresh snapshot and returns `Entry::Diagnostic`, diff --git a/libazureinit-kvp/diagnostics-proposal.md b/libazureinit-kvp/diagnostics-proposal.md index 6209015b..ba3e5733 100644 --- a/libazureinit-kvp/diagnostics-proposal.md +++ b/libazureinit-kvp/diagnostics-proposal.md @@ -101,7 +101,7 @@ DIAG_V1|||||||| - `boot_epoch` is removed because timestamps identify occurrences and stale-pool cleanup removes prior-boot records. - `type` becomes `kind`, limited to the timeline positions `start`, `finish`, and `event`; `name`, `encoding`, and `result` carry other classifications. - `encoding` in the key supports compressed payloads. -- `result` (`success` or `fail`) and `duration` (milliseconds) are required on finishes, optional on events, and empty on starts. Values otherwise remain plain text or encoded artifacts. +- `result` (`success` or `fail`) and `duration` (microseconds) are required on finishes, optional on events, and empty on starts. Values otherwise remain plain text or encoded artifacts. | Field | Meaning | |---|---| @@ -111,10 +111,10 @@ DIAG_V1|||||||| | kind | `start` or `finish` for a span, `event` for a point observation | | name | Subject, such as `provision:run` or `dmesg` | | event_id | Shared by a span's start and finish, and by every chunk of one value | -| timestamp | RFC 3339 (ISO 8601), UTC with a `Z` suffix, millisecond precision, e.g. `2026-08-31T12:34:56.789Z` | +| timestamp | RFC 3339 (ISO 8601), UTC with a `Z` suffix, at second/millisecond/microsecond/nanosecond precision (azure-init emits millisecond by default), e.g. `2026-08-31T12:34:56.789Z` | | encoding | How the value is encoded: `none` or `gz+b64` | | result | `success` or `fail` on a finish, optionally on an event; empty otherwise | -| duration | Elapsed milliseconds on a finish, optionally on a timed event; empty otherwise | +| duration | Elapsed microseconds on a finish, optionally on a timed event; empty otherwise | | chunk_index | Chunk position, from 0 | #### Key size @@ -127,12 +127,12 @@ The host silently truncates keys past 254 UTF-8 bytes; safe-mode `KvpPoolStore` | agent | 32 B | free-form producer id | | name | 48 B | free-form subject | | vm_id, event_id | 36 B each | GUID / UUID | -| timestamp | 24 B | fixed format | -| duration | 10 B | digits | +| timestamp | 30 B | up to nanosecond precision | +| duration | 13 B | digits | | result, encoding, kind | ≤ 7 B each | enum token | | chunk_index | 4 B | at most 1023 records | -With those caps the worst-case key is 226 bytes, leaving 28 bytes inside the limit. cloud-init reads are never capped; the bridge takes names as they are. +With those caps the worst-case key is 235 bytes, leaving 19 bytes inside the limit. cloud-init reads are never capped; the bridge takes names as they are. Wire examples with shortened UUIDs or `` are schematic. Stored `DIAG_V1` records require valid UUIDs and the exact timestamp format above. @@ -360,17 +360,17 @@ struct DiagnosticKey { name: String, /// One per span (start and finish share it) or standalone event. event_id: String, - /// RFC 3339, UTC, millisecond precision. + /// RFC 3339, UTC; millisecond precision by default, configurable per writer. timestamp: DateTime, encoding: Option, } /// Opens a span. struct DiagnosticStart { key: DiagnosticKey, payload: DiagnosticPayload } -/// Closes a span; carries its verdict and elapsed milliseconds. -struct DiagnosticFinish { key: DiagnosticKey, payload: DiagnosticPayload, result: Outcome, duration_ms: u64 } +/// Closes a span; carries its verdict and elapsed duration (microseconds on the wire). +struct DiagnosticFinish { key: DiagnosticKey, payload: DiagnosticPayload, result: Outcome, duration: Duration } /// A point observation; may carry a verdict or a self-contained timing. -struct DiagnosticEvent { key: DiagnosticKey, payload: DiagnosticPayload, result: Option, duration_ms: Option } +struct DiagnosticEvent { key: DiagnosticKey, payload: DiagnosticPayload, result: Option, duration: Option } /// One decoded emission, typed by kind. enum Diagnostic { @@ -421,12 +421,12 @@ impl DiagnosticWriter { /// Open a span. `event_id` links this start to the finish that closes it. pub fn emit_start(&self, event_id: &str, name: &str, payload: impl Into, encoding: Option) -> Result<(), KvpError>; - /// Close the span opened under `event_id`, recording its `result` and elapsed `duration_ms`. - pub fn emit_finish(&self, event_id: &str, name: &str, payload: impl Into, encoding: Option, result: Outcome, duration_ms: u64) -> Result<(), KvpError>; + /// Close the span opened under `event_id`, recording its `result` and elapsed `duration`. + pub fn emit_finish(&self, event_id: &str, name: &str, payload: impl Into, encoding: Option, result: Outcome, duration: Duration) -> Result<(), KvpError>; /// Record a standalone point observation; the writer assigns its `event_id`. - /// `result` and `duration_ms` are set only when measured. - pub fn emit_event(&self, name: &str, payload: impl Into, encoding: Option, result: Option, duration_ms: Option) -> Result<(), KvpError>; + /// `result` and `duration` are set only when measured. + pub fn emit_event(&self, name: &str, payload: impl Into, encoding: Option, result: Option, duration: Option) -> Result<(), KvpError>; } ``` @@ -498,10 +498,10 @@ The bridge maps fields onto the model: | event_id | trailing key identifier | | timestamp | value `ts`, read by the bridge when it maps the record | | result | value field on a finish, mapped to the model's `result` | -| duration | value field on a finish (seconds; the bridge converts to milliseconds) | +| duration | value field on a finish (seconds; the bridge converts to microseconds) | | encoding | the value `{encoding, data}` envelope, not the key | -Finish results `SUCCESS` and `FAIL` map to `success` and `fail`. An unmappable result such as `WARN` is preserved as `Raw` with `Malformed`; it is not coerced to a verdict or reclassified as an event. Source types other than `start` and `finish`, including standalone warnings, map to `event` without a span outcome. Duration conversion truncates fractional milliseconds and rejects negative or overflowing values. +Finish results `SUCCESS` and `FAIL` map to `success` and `fail`. An unmappable result such as `WARN` is preserved as `Raw` with `Malformed`; it is not coerced to a verdict or reclassified as an event. Source types other than `start` and `finish`, including standalone warnings, map to `event` without a span outcome. Duration conversion rounds to whole microseconds and rejects negative or overflowing values. The bridge uses cloud-init's `incarnation` to keep chunk groups separate, then discards it; `DiagnosticKey` does not expose it. diff --git a/libazureinit-kvp/src/cli.rs b/libazureinit-kvp/src/cli.rs index af1c7861..a24ab141 100644 --- a/libazureinit-kvp/src/cli.rs +++ b/libazureinit-kvp/src/cli.rs @@ -569,25 +569,25 @@ fn diagnostic_text(diagnostic: &Diagnostic) -> String { let _ = write!(line, " vm_id={vm_id}"); } let _ = write!(line, " name={} event_id={}", key.name, key.event_id); - let timestamp = key.timestamp.to_rfc3339_opts(SecondsFormat::Millis, true); + let timestamp = key.timestamp.to_rfc3339_opts(SecondsFormat::AutoSi, true); let encoding = key .encoding .as_ref() .map(ToString::to_string) .unwrap_or_else(|| "none".to_owned()); let _ = write!(line, " timestamp={timestamp} encoding={encoding}"); - let (result, duration_ms) = match diagnostic { + let (result, duration) = match diagnostic { Diagnostic::Start(_) => (None, None), Diagnostic::Finish(finish) => { - (Some(finish.result), Some(finish.duration_ms)) + (Some(finish.result), Some(finish.duration)) } - Diagnostic::Event(event) => (event.result, event.duration_ms), + Diagnostic::Event(event) => (event.result, event.duration), }; if let Some(result) = result { let _ = write!(line, " result={result}"); } - if let Some(duration_ms) = duration_ms { - let _ = write!(line, " duration={duration_ms}ms"); + if let Some(duration) = duration { + let _ = write!(line, " duration={}us", duration.as_micros()); } match diagnostic.payload() { DiagnosticPayload::Text(text) => { @@ -1824,7 +1824,7 @@ mod tests { #[case(KvpError::EmptyEventField { field: "name" })] #[case(KvpError::EventFieldTooLong { field: "agent", max: 32, actual: 33 })] #[case(KvpError::InvalidUuid { field: "event_id" })] - #[case(KvpError::DurationTooLarge { max_ms: 9_999_999_999, actual_ms: u64::MAX })] + #[case(KvpError::DurationTooLarge { max_us: 9_999_999_999_999, actual_us: u64::MAX })] #[case(KvpError::TooManyChunks { max: 1023 })] #[case(KvpError::PayloadNotUtf8)] #[case(KvpError::UnsupportedEncoding { token: "zstd+b64".into() })] diff --git a/libazureinit-kvp/src/diagnostics/cloud_init.rs b/libazureinit-kvp/src/diagnostics/cloud_init.rs index 6fccbcdf..22806ff5 100644 --- a/libazureinit-kvp/src/diagnostics/cloud_init.rs +++ b/libazureinit-kvp/src/diagnostics/cloud_init.rs @@ -2,6 +2,7 @@ // Licensed under the MIT License. use std::io::Read; +use std::time::Duration; use base64::{engine::general_purpose::STANDARD, Engine as _}; use chrono::{DateTime, Utc}; @@ -182,28 +183,31 @@ fn diagnostic( key, payload, result, - duration_ms: duration_ms(duration)?, + duration: duration_us(duration)?, })) } _ => Ok(Diagnostic::Event(DiagnosticEvent { key, payload, result: None, - duration_ms: None, + duration: None, })), } } -fn duration_ms(seconds: &Number) -> Result { +fn duration_us(seconds: &Number) -> Result { if let Some(seconds) = seconds.as_u64() { - return seconds.checked_mul(1000).ok_or(DecodeError::Malformed); + let micros = seconds + .checked_mul(1_000_000) + .ok_or(DecodeError::Malformed)?; + return Ok(Duration::from_micros(micros)); } - let millis = seconds.as_f64().ok_or(DecodeError::Malformed)? * 1000.0; + let micros = seconds.as_f64().ok_or(DecodeError::Malformed)? * 1_000_000.0; // The exclusive upper bound avoids a saturating float-to-integer cast. - if !(0.0..u64::MAX as f64).contains(&millis) { + if !(0.0..u64::MAX as f64).contains(µs) { return Err(DecodeError::Malformed); } - Ok(millis as u64) + Ok(Duration::from_micros(micros.round() as u64)) } fn decode_message( @@ -389,7 +393,7 @@ mod tests { assert_eq!(diagnostic.key().name, "test"); assert_eq!(diagnostic.key().event_id, EVENT_ID); let rendered = serde_json::to_value(diagnostic).unwrap(); - assert_eq!(rendered["timestamp"], "2026-07-27T21:33:24.339Z"); + assert_eq!(rendered["timestamp"], "2026-07-27T21:33:24.339006Z"); assert!(rendered.get("boot_epoch").is_none()); assert!(rendered.get("diagnostic_version_id").is_none()); } @@ -415,7 +419,7 @@ mod tests { } #[test] - fn finish_closes_its_span_with_mapped_outcome_and_milliseconds() { + fn finish_closes_its_span_with_mapped_outcome_and_microseconds() { let mut finish = value("finish", "finished with failure"); finish["result"] = json!("FAIL"); finish["duration"] = json!(0.1234); @@ -431,7 +435,7 @@ mod tests { [Entry::Diagnostic(Diagnostic::Start(start)), Entry::Diagnostic(Diagnostic::Finish(finish))] if start.key.event_id == finish.key.event_id && finish.result == Outcome::Failure - && finish.duration_ms == 123 + && finish.duration == Duration::from_micros(123_400) )); } @@ -444,7 +448,7 @@ mod tests { let diagnostic = only_diagnostic(entries(&records)); assert!(matches!(&diagnostic, Diagnostic::Finish(finish) if finish.result == Outcome::Success - && finish.duration_ms == 0 + && finish.duration == Duration::from_micros(645) && finish.key.name == "modules-final/config-scripts_user")); } @@ -628,17 +632,17 @@ mod tests { } #[rstest] - #[case::whole_seconds(json!(2), 2000)] + #[case::whole_seconds(json!(2), 2_000_000)] #[case::zero(json!(0), 0)] - #[case::fraction(json!(0.1234), 123)] - #[case::sub_millisecond(json!(0.00064), 0)] - fn duration_conversion_truncates_to_milliseconds( + #[case::fraction(json!(0.1234), 123_400)] + #[case::sub_millisecond(json!(0.00064), 640)] + fn duration_conversion_rounds_to_microseconds( #[case] seconds: Value, #[case] expected: u64, ) { assert_eq!( - duration_ms(seconds.as_number().unwrap()).unwrap(), - expected + duration_us(seconds.as_number().unwrap()).unwrap(), + Duration::from_micros(expected) ); } @@ -648,7 +652,7 @@ mod tests { #[case::float_overflow(json!(1e30))] fn duration_conversion_rejects_invalid_ranges(#[case] seconds: Value) { assert_eq!( - duration_ms(seconds.as_number().unwrap()), + duration_us(seconds.as_number().unwrap()), Err(DecodeError::Malformed) ); } diff --git a/libazureinit-kvp/src/diagnostics/diagnostic.rs b/libazureinit-kvp/src/diagnostics/diagnostic.rs index 9a3f6219..df592b81 100644 --- a/libazureinit-kvp/src/diagnostics/diagnostic.rs +++ b/libazureinit-kvp/src/diagnostics/diagnostic.rs @@ -2,6 +2,7 @@ // Licensed under the MIT License. use std::fmt; +use std::time::Duration; use base64::{engine::general_purpose::STANDARD, Engine as _}; use chrono::{DateTime, SecondsFormat, Utc}; @@ -176,8 +177,8 @@ pub struct DiagnosticFinish { pub key: DiagnosticKey, pub payload: DiagnosticPayload, pub result: Outcome, - #[serde(rename = "duration")] - pub duration_ms: u64, + #[serde(rename = "duration", serialize_with = "serialize_duration_us")] + pub duration: Duration, } #[derive(Clone, Debug, PartialEq, Eq, Serialize)] @@ -187,8 +188,12 @@ pub struct DiagnosticEvent { pub payload: DiagnosticPayload, #[serde(skip_serializing_if = "Option::is_none")] pub result: Option, - #[serde(rename = "duration", skip_serializing_if = "Option::is_none")] - pub duration_ms: Option, + #[serde( + rename = "duration", + skip_serializing_if = "Option::is_none", + serialize_with = "serialize_opt_duration_us" + )] + pub duration: Option, } #[derive(Clone, Debug, PartialEq, Eq, Serialize)] @@ -252,7 +257,7 @@ where S: Serializer, { serializer - .serialize_str(×tamp.to_rfc3339_opts(SecondsFormat::Millis, true)) + .serialize_str(×tamp.to_rfc3339_opts(SecondsFormat::AutoSi, true)) } fn serialize_encoding( @@ -268,6 +273,34 @@ where } } +/// DIAG_V1 stores elapsed time as integer microseconds. +fn duration_micros(duration: &Duration) -> u64 { + u64::try_from(duration.as_micros()).unwrap_or(u64::MAX) +} + +fn serialize_duration_us( + duration: &Duration, + serializer: S, +) -> Result +where + S: Serializer, +{ + serializer.serialize_u64(duration_micros(duration)) +} + +fn serialize_opt_duration_us( + duration: &Option, + serializer: S, +) -> Result +where + S: Serializer, +{ + match duration { + Some(duration) => serializer.serialize_u64(duration_micros(duration)), + None => serializer.serialize_none(), + } +} + #[cfg(test)] mod tests { use super::*; @@ -410,20 +443,20 @@ mod tests { #[case(Outcome::Success, "success", 312)] #[case(Outcome::Failure, "fail", 0)] #[case(Outcome::Success, "success", u64::MAX)] - fn finish_serializes_result_and_milliseconds( + fn finish_serializes_result_and_microseconds( #[case] result: Outcome, #[case] token: &str, - #[case] duration_ms: u64, + #[case] duration_us: u64, ) { let entry = Entry::Diagnostic(Diagnostic::Finish(DiagnosticFinish { key: key(), payload: "finished".into(), result, - duration_ms, + duration: Duration::from_micros(duration_us), })); let mut expected = expected_diagnostic("finish", json!("finished")); expected["result"] = json!(token); - expected["duration"] = json!(duration_ms); + expected["duration"] = json!(duration_us); assert_eq!(serde_json::to_value(entry).unwrap(), expected); } @@ -434,20 +467,20 @@ mod tests { #[case::both(Some(Outcome::Failure), Some(52))] fn event_serializes_only_measured_fields( #[case] result: Option, - #[case] duration_ms: Option, + #[case] duration_us: Option, ) { let entry = Entry::Diagnostic(Diagnostic::Event(DiagnosticEvent { key: key(), payload: "observed".into(), result, - duration_ms, + duration: duration_us.map(Duration::from_micros), })); let mut expected = expected_diagnostic("event", json!("observed")); if let Some(result) = result { expected["result"] = json!(result.to_string()); } - if let Some(duration_ms) = duration_ms { - expected["duration"] = json!(duration_ms); + if let Some(duration_us) = duration_us { + expected["duration"] = json!(duration_us); } assert_eq!(serde_json::to_value(entry).unwrap(), expected); } @@ -461,7 +494,7 @@ mod tests { }, payload: b"hello".as_slice().into(), result: None, - duration_ms: None, + duration: None, })); let mut expected = expected_diagnostic( "event", @@ -504,11 +537,12 @@ mod tests { } #[rstest] - #[case("2026-08-31T12:34:56Z", "2026-08-31T12:34:56.000Z")] + #[case("2026-08-31T12:34:56Z", "2026-08-31T12:34:56Z")] #[case("2026-08-31T12:34:56.3Z", "2026-08-31T12:34:56.300Z")] - #[case("2026-08-31T12:34:56.789999Z", TIMESTAMP)] + #[case("2026-08-31T12:34:56.789999Z", "2026-08-31T12:34:56.789999Z")] + #[case("2026-08-31T12:34:56.789123456Z", "2026-08-31T12:34:56.789123456Z")] #[case("2026-08-31T14:34:56.789+02:00", TIMESTAMP)] - fn timestamp_serializes_in_utc_milliseconds( + fn timestamp_serializes_in_utc_without_padding( #[case] timestamp: &str, #[case] expected: &str, ) { @@ -536,13 +570,13 @@ mod tests { key: key.clone(), payload: payload.clone(), result: Outcome::Success, - duration_ms: 0, + duration: Duration::ZERO, }), Kind::Event => Diagnostic::Event(DiagnosticEvent { key: key.clone(), payload: payload.clone(), result: None, - duration_ms: None, + duration: None, }), }; assert_eq!(diagnostic.key(), &key); diff --git a/libazureinit-kvp/src/diagnostics/mod.rs b/libazureinit-kvp/src/diagnostics/mod.rs index 923b8b44..80336126 100644 --- a/libazureinit-kvp/src/diagnostics/mod.rs +++ b/libazureinit-kvp/src/diagnostics/mod.rs @@ -19,7 +19,7 @@ pub use diagnostic::{ RawKeyValue, DIAGNOSTIC_VERSION_ID, }; pub use reader::DiagnosticReader; -pub use writer::DiagnosticWriter; +pub use writer::{DiagnosticWriter, TimestampPrecision}; /// Maximum number of UTF-8 value bytes stored in one diagnostic record. /// diff --git a/libazureinit-kvp/src/diagnostics/reader.rs b/libazureinit-kvp/src/diagnostics/reader.rs index 3d404d7a..5293e4f6 100644 --- a/libazureinit-kvp/src/diagnostics/reader.rs +++ b/libazureinit-kvp/src/diagnostics/reader.rs @@ -2,6 +2,7 @@ // Licensed under the MIT License. use std::collections::HashMap; +use std::time::Duration; use chrono::{DateTime, SecondsFormat, Utc}; use uuid::Uuid; @@ -176,10 +177,18 @@ fn decode_v1_group( let parsed_timestamp = DateTime::parse_from_rfc3339(timestamp) .map_err(|_| DecodeError::Malformed)? .with_timezone(&Utc); - if timestamp.len() != 24 - || parsed_timestamp.to_rfc3339_opts(SecondsFormat::Millis, true) - != timestamp - { + // Accept only canonical UTC `Z` timestamps at second/ms/us/ns precision. + let canonical = [ + SecondsFormat::Secs, + SecondsFormat::Millis, + SecondsFormat::Micros, + SecondsFormat::Nanos, + ] + .into_iter() + .any(|precision| { + parsed_timestamp.to_rfc3339_opts(precision, true) == timestamp + }); + if !canonical { return Err(DecodeError::Malformed); } let result = match result { @@ -188,10 +197,10 @@ fn decode_v1_group( "fail" => Some(Outcome::Failure), _ => return Err(DecodeError::Malformed), }; - let duration_ms = if duration.is_empty() { + let duration = if duration.is_empty() { None } else { - Some(parse_unsigned(duration)?) + Some(Duration::from_micros(parse_unsigned(duration)?)) }; let encoding = match encoding { "none" => None, @@ -207,27 +216,27 @@ fn decode_v1_group( encoding, }; - match (kind, result, duration_ms) { + match (kind, result, duration) { ("start", None, None) => { let payload = decode_chunks(chunks, key.encoding.as_ref())?; Ok(Diagnostic::Start(DiagnosticStart { key, payload })) } - ("finish", Some(result), Some(duration_ms)) => { + ("finish", Some(result), Some(duration)) => { let payload = decode_chunks(chunks, key.encoding.as_ref())?; Ok(Diagnostic::Finish(DiagnosticFinish { key, payload, result, - duration_ms, + duration, })) } - ("event", result, duration_ms) => { + ("event", result, duration) => { let payload = decode_chunks(chunks, key.encoding.as_ref())?; Ok(Diagnostic::Event(DiagnosticEvent { key, payload, result, - duration_ms, + duration, })) } _ => Err(DecodeError::Malformed), @@ -651,7 +660,7 @@ mod tests { && start.payload == DiagnosticPayload::Text("starting".into()) && finish.payload == DiagnosticPayload::Text("failed".into()) && finish.result == Outcome::Failure - && finish.duration_ms == 312 + && finish.duration == Duration::from_micros(312) )); } @@ -662,17 +671,18 @@ mod tests { #[case::both(Some(Outcome::Failure), Some(52))] fn event_result_and_duration_are_independent( #[case] result: Option, - #[case] duration: Option, + #[case] duration_us: Option, ) { let result_token = result.map_or_else(String::new, |v| v.to_string()); let duration_token = - duration.map_or_else(String::new, |v| v.to_string()); + duration_us.map_or_else(String::new, |v| v.to_string()); let key = with_field(&key(0), 8, &result_token); let key = with_field(&key, 9, &duration_token); let diagnostic = only_diagnostic(decode_entries(vec![(key, "value".into())])); assert!(matches!(&diagnostic, Diagnostic::Event(event) - if event.result == result && event.duration_ms == duration)); + if event.result == result + && event.duration == duration_us.map(Duration::from_micros))); } #[rstest] @@ -737,14 +747,13 @@ mod tests { } #[rstest] - #[case("not a timestamp")] - #[case("2026-08-31T12:34:56Z")] - #[case("2026-08-31T12:34:56.78Z")] - #[case("2026-08-31T12:34:56.789000Z")] - #[case("2026-08-31T12:34:56.789+00:00")] - #[case("2026-08-31t12:34:56.789z")] - #[case("2026-13-31T12:34:56.789Z")] - fn v1_requires_utc_millisecond_timestamps(#[case] timestamp: &str) { + #[case::unparsable("not a timestamp")] + #[case::two_fraction_digits("2026-08-31T12:34:56.78Z")] + #[case::four_fraction_digits("2026-08-31T12:34:56.7890Z")] + #[case::numeric_offset("2026-08-31T12:34:56.789+00:00")] + #[case::lowercase("2026-08-31t12:34:56.789z")] + #[case::invalid_month("2026-13-31T12:34:56.789Z")] + fn v1_rejects_non_canonical_timestamps(#[case] timestamp: &str) { let records = vec![(with_field(&key(0), 6, timestamp), "value".into())]; assert_eq!( decode_entries(records.clone()), @@ -752,6 +761,25 @@ mod tests { ); } + #[rstest] + #[case::seconds("2026-08-31T12:34:56Z")] + #[case::milliseconds("2026-08-31T12:34:56.789Z")] + #[case::millisecond_whole("2026-08-31T12:34:56.000Z")] + #[case::microseconds("2026-08-31T12:34:56.789123Z")] + #[case::microsecond_trailing_zeros("2026-08-31T12:34:56.789000Z")] + #[case::nanoseconds("2026-08-31T12:34:56.789123456Z")] + fn v1_accepts_canonical_timestamp_precisions(#[case] timestamp: &str) { + let key = with_field(&key(0), 6, timestamp); + let diagnostic = + only_diagnostic(decode_entries(vec![(key, "value".into())])); + assert_eq!( + diagnostic.key().timestamp, + DateTime::parse_from_rfc3339(timestamp) + .unwrap() + .with_timezone(&Utc) + ); + } + #[rstest] #[case([1, 0, 2])] #[case([2, 1, 0])] @@ -1027,7 +1055,7 @@ mod tests { let diagnostic = only_diagnostic(decode_entries(vec![(key, "payload".into())])); assert!(matches!(&diagnostic, Diagnostic::Event(event) - if event.duration_ms == Some(u64::MAX))); + if event.duration == Some(Duration::from_micros(u64::MAX)))); } #[test] diff --git a/libazureinit-kvp/src/diagnostics/writer.rs b/libazureinit-kvp/src/diagnostics/writer.rs index 0a2ef171..a6dabd09 100644 --- a/libazureinit-kvp/src/diagnostics/writer.rs +++ b/libazureinit-kvp/src/diagnostics/writer.rs @@ -1,6 +1,8 @@ // Copyright (c) Microsoft Corporation. // Licensed under the MIT License. +use std::time::Duration; + use chrono::{SecondsFormat, Utc}; use uuid::Uuid; @@ -16,17 +18,43 @@ use crate::{KvpError, KvpPoolStore}; const MAX_AGENT_BYTES: usize = 32; const MAX_NAME_BYTES: usize = 48; const MAX_UUID_BYTES: usize = 36; -const MAX_TIMESTAMP_BYTES: usize = 24; +const MAX_TIMESTAMP_BYTES: usize = 30; const MAX_KEY_BYTES: usize = 254; -const MAX_DURATION_MS: u64 = 9_999_999_999; +const MAX_DURATION_US: u64 = 9_999_999_999_999; const MAX_CHUNKS: usize = 1023; +/// Fractional-second precision for emitted DIAG_V1 timestamps. +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] +pub enum TimestampPrecision { + /// Whole seconds, no fractional digits. + Seconds, + /// Millisecond precision (three fractional digits). The default. + #[default] + Millis, + /// Microsecond precision (six fractional digits). + Micros, + /// Nanosecond precision (nine fractional digits). + Nanos, +} + +impl TimestampPrecision { + fn seconds_format(self) -> SecondsFormat { + match self { + Self::Seconds => SecondsFormat::Secs, + Self::Millis => SecondsFormat::Millis, + Self::Micros => SecondsFormat::Micros, + Self::Nanos => SecondsFormat::Nanos, + } + } +} + /// Validation precedes I/O; a storage failure can leave a partial batch. #[derive(Clone, Debug)] pub struct DiagnosticWriter { store: KvpPoolStore, agent: String, vm_id: String, + timestamp_precision: TimestampPrecision, } impl DiagnosticWriter { @@ -44,9 +72,19 @@ impl DiagnosticWriter { store, agent, vm_id, + timestamp_precision: TimestampPrecision::default(), }) } + /// Overrides the default millisecond precision for emitted timestamps. + pub fn with_timestamp_precision( + mut self, + precision: TimestampPrecision, + ) -> Self { + self.timestamp_precision = precision; + self + } + /// The caller retains `event_id` to correlate the corresponding finish. pub fn emit_start( &self, @@ -61,7 +99,7 @@ impl DiagnosticWriter { })) } - /// Duration is caller-measured milliseconds, not inferred from the pool. + /// Duration is caller-measured, not inferred from the pool. pub fn emit_finish( &self, event_id: &str, @@ -69,13 +107,13 @@ impl DiagnosticWriter { payload: impl Into, encoding: Option, result: Outcome, - duration_ms: u64, + duration: Duration, ) -> Result<(), KvpError> { self.emit(Diagnostic::Finish(DiagnosticFinish { key: self.key(event_id, name, encoding), payload: payload.into(), result, - duration_ms, + duration, })) } @@ -86,13 +124,13 @@ impl DiagnosticWriter { payload: impl Into, encoding: Option, result: Option, - duration_ms: Option, + duration: Option, ) -> Result<(), KvpError> { self.emit(Diagnostic::Event(DiagnosticEvent { key: self.key(&Uuid::new_v4().to_string(), name, encoding), payload: payload.into(), result, - duration_ms, + duration, })) } @@ -113,24 +151,28 @@ impl DiagnosticWriter { } fn emit(&self, diagnostic: Diagnostic) -> Result<(), KvpError> { - self.store.append_multiple(prepare_records(diagnostic)?) + self.store.append_multiple(prepare_records( + diagnostic, + self.timestamp_precision, + )?) } } fn prepare_records( diagnostic: Diagnostic, + precision: TimestampPrecision, ) -> Result, KvpError> { let kind = diagnostic.kind(); - let (key, payload, result, duration_ms) = match diagnostic { + let (key, payload, result, duration) = match diagnostic { Diagnostic::Start(start) => (start.key, start.payload, None, None), Diagnostic::Finish(finish) => ( finish.key, finish.payload, Some(finish.result), - Some(finish.duration_ms), + Some(finish.duration), ), Diagnostic::Event(event) => { - (event.key, event.payload, event.result, event.duration_ms) + (event.key, event.payload, event.result, event.duration) } }; @@ -143,17 +185,23 @@ fn prepare_records( validate_field("name", &key.name, MAX_NAME_BYTES)?; validate_uuid("event_id", &key.event_id)?; - let duration = match duration_ms { - Some(actual) if actual > MAX_DURATION_MS => { - return Err(KvpError::DurationTooLarge { - max_ms: MAX_DURATION_MS, - actual_ms: actual, - }); + let duration = match duration { + Some(duration) => { + let micros = + u64::try_from(duration.as_micros()).unwrap_or(u64::MAX); + if micros > MAX_DURATION_US { + return Err(KvpError::DurationTooLarge { + max_us: MAX_DURATION_US, + actual_us: micros, + }); + } + micros.to_string() } - Some(duration) => duration.to_string(), None => String::new(), }; - let timestamp = key.timestamp.to_rfc3339_opts(SecondsFormat::Millis, true); + let timestamp = key + .timestamp + .to_rfc3339_opts(precision.seconds_format(), true); validate_field("timestamp", ×tamp, MAX_TIMESTAMP_BYTES)?; let value = encode_payload(payload, key.encoding.as_ref())?; @@ -275,7 +323,7 @@ mod tests { key, payload, result: None, - duration_ms: None, + duration: None, }) } @@ -517,10 +565,13 @@ mod tests { #[test] fn exact_start_key_matches_v1_layout() { - let records = prepare_records(Diagnostic::Start(DiagnosticStart { - key: key(), - payload: "starting".into(), - })) + let records = prepare_records( + Diagnostic::Start(DiagnosticStart { + key: key(), + payload: "starting".into(), + }), + TimestampPrecision::Millis, + ) .unwrap(); assert_eq!( records, @@ -539,19 +590,22 @@ mod tests { fn exact_finish_key_matches_v1_layout( #[case] result: Outcome, #[case] token: &str, - #[case] duration_ms: u64, + #[case] duration_us: u64, ) { - let records = prepare_records(Diagnostic::Finish(DiagnosticFinish { - key: key(), - payload: "finished".into(), - result, - duration_ms, - })) + let records = prepare_records( + Diagnostic::Finish(DiagnosticFinish { + key: key(), + payload: "finished".into(), + result, + duration: Duration::from_micros(duration_us), + }), + TimestampPrecision::Millis, + ) .unwrap(); assert_eq!( records[0].0, format!( - "DIAG_V1|{AGENT}|{VM_ID}|finish|provision:run|{EVENT_ID}|{TIMESTAMP}|none|{token}|{duration_ms}|0" + "DIAG_V1|{AGENT}|{VM_ID}|finish|provision:run|{EVENT_ID}|{TIMESTAMP}|none|{token}|{duration_us}|0" ) ); } @@ -560,16 +614,22 @@ mod tests { #[case::neither(None, None)] #[case::result_only(Some(Outcome::Success), None)] #[case::zero_duration(None, Some(0))] - #[case::both(Some(Outcome::Failure), Some(MAX_DURATION_MS))] + #[case::both(Some(Outcome::Failure), Some(MAX_DURATION_US))] fn event_optional_fields_are_independent( #[case] result: Option, - #[case] duration_ms: Option, + #[case] duration_us: Option, ) { let dir = TempDir::new().unwrap(); let pool = store(&dir, PoolMode::Safe); let writer = DiagnosticWriter::new(pool.clone(), AGENT, VM_ID).unwrap(); writer - .emit_event("test", "ok", None, result, duration_ms) + .emit_event( + "test", + "ok", + None, + result, + duration_us.map(Duration::from_micros), + ) .unwrap(); let records = pool.dump().unwrap(); let fields: Vec<_> = records[0].0.split('|').collect(); @@ -581,7 +641,7 @@ mod tests { ); assert_eq!( fields[9], - duration_ms.map_or_else(String::new, |v| v.to_string()) + duration_us.map_or_else(String::new, |v| v.to_string()) ); } @@ -600,7 +660,7 @@ mod tests { "finished", None, Outcome::Failure, - 17, + Duration::from_micros(17), ) .unwrap(); let records = pool.dump().unwrap(); @@ -656,6 +716,27 @@ mod tests { assert!((before..=after).contains(&parsed.timestamp_millis())); } + #[rstest] + #[case(TimestampPrecision::Seconds, 20)] + #[case(TimestampPrecision::Millis, 24)] + #[case(TimestampPrecision::Micros, 27)] + #[case(TimestampPrecision::Nanos, 30)] + fn timestamp_precision_controls_emitted_width( + #[case] precision: TimestampPrecision, + #[case] expected_len: usize, + ) { + let dir = TempDir::new().unwrap(); + let pool = store(&dir, PoolMode::Safe); + let writer = DiagnosticWriter::new(pool.clone(), AGENT, VM_ID) + .unwrap() + .with_timestamp_precision(precision); + writer.emit_event("test", "ok", None, None, None).unwrap(); + let records = pool.dump().unwrap(); + let timestamp = records[0].0.split('|').nth(6).unwrap(); + assert_eq!(timestamp.len(), expected_len); + assert!(timestamp.ends_with('Z')); + } + #[rstest] #[case(String::new(), vec![0])] #[case("x".repeat(MAX_CHUNK_BYTES), vec![MAX_CHUNK_BYTES])] @@ -805,12 +886,13 @@ mod tests { }, payload: "test".into(), result: Outcome::Success, - duration_ms: MAX_DURATION_MS, + duration: Duration::from_micros(MAX_DURATION_US), }); - let records = prepare_records(diagnostic).unwrap(); + let records = + prepare_records(diagnostic, TimestampPrecision::Millis).unwrap(); let base = records[0].0.rsplit_once('|').unwrap().0; let longest = format!("{base}|1022"); - assert_eq!(longest.len(), 226); + assert_eq!(longest.len(), 229); assert!(longest.len() <= MAX_KEY_BYTES); } @@ -831,7 +913,7 @@ mod tests { "message", None, Outcome::Failure, - 0, + Duration::ZERO, ) }); assert!(matches!( @@ -866,7 +948,7 @@ mod tests { payload, None, Outcome::Failure, - 0, + Duration::ZERO, ) }); assert!(matches!(error, KvpError::ValueContainsNull)); @@ -891,11 +973,12 @@ mod tests { #[rstest] fn excessive_durations_are_rejected_before_io( - #[values(MAX_DURATION_MS + 1, u64::MAX)] duration_ms: u64, + #[values(MAX_DURATION_US + 1, u64::MAX)] duration_us: u64, #[values(Kind::Finish, Kind::Event)] kind: Kind, ) { let dir = TempDir::new().unwrap(); let (writer, ops) = observed_writer(&dir); + let duration = Duration::from_micros(duration_us); let result = match kind { Kind::Finish => writer.emit_finish( EVENT_ID, @@ -903,16 +986,14 @@ mod tests { "bad", None, Outcome::Failure, - duration_ms, + duration, ), - _ => { - writer.emit_event("test", "bad", None, None, Some(duration_ms)) - } + _ => writer.emit_event("test", "bad", None, None, Some(duration)), }; assert!(matches!( result, - Err(KvpError::DurationTooLarge { max_ms: MAX_DURATION_MS, actual_ms }) - if actual_ms == duration_ms + Err(KvpError::DurationTooLarge { max_us: MAX_DURATION_US, actual_us }) + if actual_us == duration_us )); assert_eq!(ops.calls.load(Ordering::SeqCst), 0); } @@ -922,7 +1003,10 @@ mod tests { let mut missing_vm = key(); missing_vm.vm_id = None; assert!(matches!( - prepare_records(event(missing_vm, "bad".into())), + prepare_records( + event(missing_vm, "bad".into()), + TimestampPrecision::Millis + ), Err(KvpError::EmptyEventField { field: "vm_id" }) )); } @@ -933,10 +1017,13 @@ mod tests { expanded_year.timestamp = DateTime::from_timestamp(253_402_300_800, 0).unwrap(); assert!(matches!( - prepare_records(event(expanded_year, "bad".into())), + prepare_records( + event(expanded_year, "bad".into()), + TimestampPrecision::Nanos + ), Err(KvpError::EventFieldTooLong { field: "timestamp", - max: 24, + max: 30, .. }) )); diff --git a/libazureinit-kvp/src/error.rs b/libazureinit-kvp/src/error.rs index ea7ebaef..f0a39437 100644 --- a/libazureinit-kvp/src/error.rs +++ b/libazureinit-kvp/src/error.rs @@ -29,8 +29,8 @@ pub enum KvpError { field: &'static str, }, DurationTooLarge { - max_ms: u64, - actual_ms: u64, + max_us: u64, + actual_us: u64, }, TooManyChunks { max: usize, @@ -77,8 +77,8 @@ impl fmt::Display for KvpError { Self::InvalidUuid { field } => { write!(f, "event key field '{field}' must be a UUID") } - Self::DurationTooLarge { max_ms, actual_ms } => { - write!(f, "diagnostic duration ({actual_ms}ms) exceeds maximum ({max_ms}ms)") + Self::DurationTooLarge { max_us, actual_us } => { + write!(f, "diagnostic duration ({actual_us}us) exceeds maximum ({max_us}us)") } Self::TooManyChunks { max } => { write!(f, "diagnostic chunk count exceeds maximum ({max})") @@ -147,8 +147,8 @@ mod tests { "event key field 'vm_id' must be a UUID" )] #[case( - KvpError::DurationTooLarge { max_ms: 9_999_999_999, actual_ms: 10_000_000_000 }, - "diagnostic duration (10000000000ms) exceeds maximum (9999999999ms)" + KvpError::DurationTooLarge { max_us: 9_999_999_999_999, actual_us: 10_000_000_000_000 }, + "diagnostic duration (10000000000000us) exceeds maximum (9999999999999us)" )] #[case( KvpError::TooManyChunks { max: 1023 }, diff --git a/libazureinit-kvp/src/lib.rs b/libazureinit-kvp/src/lib.rs index 812216d7..f5028008 100644 --- a/libazureinit-kvp/src/lib.rs +++ b/libazureinit-kvp/src/lib.rs @@ -54,8 +54,8 @@ pub use cli::run; pub use diagnostics::{ DecodeError, Diagnostic, DiagnosticEvent, DiagnosticFinish, DiagnosticKey, DiagnosticPayload, DiagnosticReader, DiagnosticStart, DiagnosticWriter, - Encoding, Entry, Kind, Outcome, RawKeyValue, DIAGNOSTIC_VERSION_ID, - MAX_CHUNK_BYTES, + Encoding, Entry, Kind, Outcome, RawKeyValue, TimestampPrecision, + DIAGNOSTIC_VERSION_ID, MAX_CHUNK_BYTES, }; pub use error::KvpError; pub use report::{ diff --git a/libazureinit-kvp/tests/cli.rs b/libazureinit-kvp/tests/cli.rs index 2d2e27d5..24c67867 100644 --- a/libazureinit-kvp/tests/cli.rs +++ b/libazureinit-kvp/tests/cli.rs @@ -4,6 +4,7 @@ use std::fs; use std::io::Write; use std::process::{Command, Output}; +use std::time::Duration; use libazureinit_kvp::{ DiagnosticWriter, Encoding, KvpPool, KvpPoolStore, Outcome, PoolMode, @@ -451,8 +452,8 @@ fn parsed_dump_normalizes_cloud_init_in_json_and_text() { "type": "diagnostic", "kind": "finish", "agent": "CLOUD_INIT", "name": "modules-final/config-scripts_user", "vm_id": VM_ID, "event_id": "e5f01809-a7a3-4279-aa64-1f18e21eda6e", - "timestamp": "2026-07-27T21:33:24.339Z", "encoding": "none", - "result": "success", "duration": 500, "payload": "scripts ran", + "timestamp": "2026-07-27T21:33:24.339006Z", "encoding": "none", + "result": "success", "duration": 500000, "payload": "scripts ran", }) ); assert_eq!(entries[1]["kind"], "start"); @@ -467,8 +468,8 @@ fn parsed_dump_normalizes_cloud_init_in_json_and_text() { assert!(out.contains("name=modules-final/config-scripts_user")); assert!(out.contains("vm_id=0e5e179d-5341-478b-8456-fbb90621bdf8")); assert!(out.contains("result=success")); - assert!(out.contains("timestamp=2026-07-27T21:33:24.339Z")); - assert!(out.contains("duration=500ms")); + assert!(out.contains("timestamp=2026-07-27T21:33:24.339006Z")); + assert!(out.contains("duration=500000us")); assert!(out.contains("payload=scripts ran")); let start = out.lines().nth(1).unwrap(); assert!(start.contains("diagnostic kind=start")); @@ -487,7 +488,7 @@ fn parsed_dump_renders_bytes_reports_and_raw_errors() { vec![0, 255], Some(Encoding::GzB64), Some(Outcome::Failure), - Some(7), + Some(Duration::from_micros(7)), ) .unwrap(); store.append("note", "raw value").unwrap(); @@ -501,7 +502,7 @@ fn parsed_dump_renders_bytes_reports_and_raw_errors() { let lines: Vec<_> = out.lines().collect(); assert_eq!(lines.len(), 4); assert!(lines[0] - .contains("encoding=gz+b64 result=fail duration=7ms payload_b64=AP8=")); + .contains("encoding=gz+b64 result=fail duration=7us payload_b64=AP8=")); assert_eq!(lines[1], "raw key=note value=raw value"); assert_eq!(lines[2], "raw key=DIAG_V1|bad value=junk error=malformed diagnostic or provisioning report"); assert_eq!( diff --git a/libazureinit-kvp/tests/diagnostics.rs b/libazureinit-kvp/tests/diagnostics.rs index a9d25527..f1a70d92 100644 --- a/libazureinit-kvp/tests/diagnostics.rs +++ b/libazureinit-kvp/tests/diagnostics.rs @@ -6,6 +6,7 @@ use std::io::Read; use std::thread; +use std::time::Duration; use base64::{engine::general_purpose::STANDARD, Engine as _}; use flate2::read::ZlibDecoder; @@ -84,7 +85,7 @@ fn reads_real_cloud_init_pool_in_both_layouts(#[case] include_vm_id: bool) { include_vm_id.then_some(CLOUD_INIT_VM_ID) ); assert_eq!(finish.result, Outcome::Success); - assert_eq!(finish.duration_ms, 0); + assert_eq!(finish.duration, Duration::from_micros(645)); assert_eq!( finish.payload, DiagnosticPayload::from( @@ -93,7 +94,8 @@ fn reads_real_cloud_init_pool_in_both_layouts(#[case] include_vm_id: bool) { ); assert!(matches!(diagnostic(&entries[1]), Diagnostic::Start(_))); assert!(matches!(diagnostic(&entries[2]), Diagnostic::Finish(finish) - if finish.result == Outcome::Success && finish.duration_ms == 340)); + if finish.result == Outcome::Success + && finish.duration == Duration::from_micros(340712))); } #[test] @@ -109,7 +111,13 @@ fn span_and_point_events_round_trip_with_a_report() { .emit_start(EVENT_ID, "provision:run", "starting", None) .unwrap(); writer - .emit_event("imds", "ok", None, Some(Outcome::Success), Some(17)) + .emit_event( + "imds", + "ok", + None, + Some(Outcome::Success), + Some(Duration::from_micros(17)), + ) .unwrap(); writer .emit_finish( @@ -118,7 +126,7 @@ fn span_and_point_events_round_trip_with_a_report() { "finished", None, Outcome::Success, - 120, + Duration::from_micros(120), ) .unwrap(); let report = ProvisioningReport::success(AGENT, VM_ID, ReportPpsType::None) @@ -140,7 +148,7 @@ fn span_and_point_events_round_trip_with_a_report() { assert_eq!(event.key.name, "imds"); assert_eq!(event.payload, DiagnosticPayload::from("ok")); assert_eq!(event.result, Some(Outcome::Success)); - assert_eq!(event.duration_ms, Some(17)); + assert_eq!(event.duration, Some(Duration::from_micros(17))); assert_eq!( uuid::Uuid::parse_str(&event.key.event_id) .unwrap() @@ -150,7 +158,7 @@ fn span_and_point_events_round_trip_with_a_report() { assert_ne!(event.key.event_id, EVENT_ID); assert_eq!(finish.payload, DiagnosticPayload::from("finished")); assert_eq!(finish.result, Outcome::Success); - assert_eq!(finish.duration_ms, 120); + assert_eq!(finish.duration, Duration::from_micros(120)); assert_eq!(decoded_report, &report); let dumped = store.dump().unwrap(); From 9cb5a27cd3415a1ec6605c3519c2b25700eb189b Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Tue, 15 Sep 2026 14:30:29 -0700 Subject: [PATCH 26/32] Add coverage for missing line in cli.rs --- libazureinit-kvp/src/cli.rs | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) diff --git a/libazureinit-kvp/src/cli.rs b/libazureinit-kvp/src/cli.rs index a24ab141..a0387bad 100644 --- a/libazureinit-kvp/src/cli.rs +++ b/libazureinit-kvp/src/cli.rs @@ -1967,6 +1967,23 @@ mod tests { entries[1], json!({"type": "raw", "key": "note", "value": "raw"}) ); + + let (_, output) = run_dispatch(cli( + &dir, + Command::Dump { + parse: true, + name: None, + kind: Some(KindArg::Event), + }, + )); + let entries = parse_json(&output); + let entries = entries.as_array().unwrap(); + assert_eq!(entries.len(), 2); + assert_eq!(entries[0]["kind"], "event"); + assert_eq!( + entries[1], + json!({"type": "raw", "key": "note", "value": "raw"}) + ); } #[test] From e9c37bc19cc69aa07db137802e68287dea319277 Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Tue, 15 Sep 2026 14:42:29 -0700 Subject: [PATCH 27/32] Add coverage for missing line in diagnostics.rs --- libazureinit-kvp/src/diagnostics/diagnostic.rs | 5 +---- 1 file changed, 1 insertion(+), 4 deletions(-) diff --git a/libazureinit-kvp/src/diagnostics/diagnostic.rs b/libazureinit-kvp/src/diagnostics/diagnostic.rs index df592b81..20e5a5eb 100644 --- a/libazureinit-kvp/src/diagnostics/diagnostic.rs +++ b/libazureinit-kvp/src/diagnostics/diagnostic.rs @@ -295,10 +295,7 @@ fn serialize_opt_duration_us( where S: Serializer, { - match duration { - Some(duration) => serializer.serialize_u64(duration_micros(duration)), - None => serializer.serialize_none(), - } + duration.as_ref().map(duration_micros).serialize(serializer) } #[cfg(test)] From 6337db359d55d33791839e94e7c02c6b7d2cee4a Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Thu, 17 Sep 2026 13:07:51 -0700 Subject: [PATCH 28/32] Overhaul KVP Rustdoc and consolidate the layer contracts into root documentation. Adopt DIAG and name/VERSION identifiers, add configurable decimal-second durations and zlib+b64 encoding, and update tests while preserving gzip and cloud-init compatibility. --- README.md | 3 +- doc/diagnostics.md | 273 ++++++++++ doc/getting_started.md | 3 +- doc/kvp.md | 75 +-- libazureinit-kvp/diagnostics-proposal.md | 512 ------------------ libazureinit-kvp/src/cli.rs | 25 +- .../src/diagnostics/cloud_init.rs | 29 +- .../src/diagnostics/diagnostic.rs | 122 ++++- libazureinit-kvp/src/diagnostics/encoding.rs | 125 ++++- libazureinit-kvp/src/diagnostics/mod.rs | 13 +- libazureinit-kvp/src/diagnostics/reader.rs | 104 +++- libazureinit-kvp/src/diagnostics/writer.rs | 261 ++++++--- libazureinit-kvp/src/error.rs | 46 +- libazureinit-kvp/src/lib.rs | 66 ++- libazureinit-kvp/src/report.rs | 94 ++-- libazureinit-kvp/src/store.rs | 181 +++---- libazureinit-kvp/src/vm_id.rs | 15 +- libazureinit-kvp/tests/cli.rs | 36 +- libazureinit-kvp/tests/diagnostics.rs | 5 +- 19 files changed, 1010 insertions(+), 978 deletions(-) create mode 100644 doc/diagnostics.md delete mode 100644 libazureinit-kvp/diagnostics-proposal.md diff --git a/README.md b/README.md index d07ec4d4..f979f6a3 100644 --- a/README.md +++ b/README.md @@ -74,7 +74,8 @@ To run the program, you must enter the command `cargo run --bin ` a |----------|-------------| | [Getting Started Guide](doc/getting_started.md) | Step-by-step instructions for new users | | [Configuration Guide](doc/configuration.md) | Detailed configuration options and file structure | -| [Tracing System](doc/libazurekvp.md) | Understanding the telemetry and tracing capabilities | +| [KVP Data Exchange](doc/kvp.md) | Pool-file format and Hyper-V interfaces | +| [Diagnostics Contract](doc/diagnostics.md) | Telemetry fields, encodings and consumer behavior | | [End-to-End Testing](doc/e2e_testing.md) | How to perform comprehensive system testing | | [Library Documentation](libazureinit/README.md) | Documentation for the libazureinit library | diff --git a/doc/diagnostics.md b/doc/diagnostics.md new file mode 100644 index 00000000..138fb657 --- /dev/null +++ b/doc/diagnostics.md @@ -0,0 +1,273 @@ +# KVP Diagnostics Contract + +This contract defines provisioning telemetry for host-side consumers, including +EG and LPA: operations, observations and artifacts such as logs. The +[KVP contract](kvp.md) defines the underlying pool files and transport. + +## Record Format + +Native diagnostics are appended to guest pool 1. A diagnostic occupies one or +more records. Each key has exactly eleven pipe-delimited fields: + +```text +DIAG|||||||||| +``` + +`DIAG` selects the current format; `DIAG_V*` is reserved for other schemas. +Do not parse an unsupported schema using this layout. Agent versions identify +the producer, not the schema. + +| Field | Meaning | +|---|---| +| `DIAG` | Native diagnostic format identifier | +| `agent` | Reporting agent, conventionally `name/VERSION`, such as `azure-init/0.1.1` | +| `vm_id` | VM UUID | +| `kind` | `start`, `finish`, or `event` | +| `name` | Operation or observation, such as `provision:run`, `imds`, or `dmesg` | +| `event_id` | UUID shared by an operation's start and finish; a standalone event has its own UUID | +| `timestamp` | Emission time as RFC 3339 UTC with a `Z` suffix | +| `encoding` | Value representation: `none`, `zlib+b64`, or `gz+b64` | +| `result` | `success` or `fail`, when applicable | +| `duration` | Nonnegative elapsed seconds, when measured | +| `chunk_index` | Decimal chunk index starting at zero, including for a single record | + +All fields except `result` and `duration` are required and nonempty. Key fields +cannot contain `|` or NUL; there is no key-field escaping. Agent strings remain +opaque to readers, including unversioned names and older naming conventions. +UUID spellings are preserved rather than rewritten during reading. + +## Timing and Correlation + +| Kind | Result | Duration | Meaning | +|---|---|---|---| +| `start` | Empty | Empty | An operation began | +| `finish` | Required | Required | An operation ended with a reported outcome and elapsed time | +| `event` | Optional | Optional | A standalone observation; outcome and timing are independent | + +A start and finish share an event ID, agent, VM and operation name. A finish +carries its own elapsed duration; do not derive it by subtracting wall-clock +timestamps. Either endpoint remains valid if its counterpart is missing. + +### Timestamps and Durations + +Timestamps use RFC 3339 UTC `Z` form, with zero, three, six or nine fractional +digits. The default is milliseconds, for example `2026-08-31T12:34:56.789Z`. +Numeric offsets and other fractional widths are invalid in native records. + +Durations are decimal **seconds**: `0.312000` is 312 milliseconds. Emission +defaults to six fractional digits, independently of timestamp precision. +Producers can select zero, three, six or nine digits; lower digits are discarded. + +Accept decimal digits with an optional decimal point and one to nine fractional +digits. The whole-seconds component is at most `18446744073709551615`. Signs, +exponents, NaN, infinity and empty fractional parts are invalid. An empty field +means absent; `0` is a measured zero. + +### Examples + +An operation and a compressed observation, shown as stored key/value records: + +```json +[ + { + "key": "DIAG|azure-init/0.1.1|3f2504e0-4f89-41d3-9a0c-0305e82c3301|start|provision:run|8f3e9c4a-1b2c-4d5e-9f01-234567890abc|2026-08-31T12:34:56.789Z|none|||0", + "value": "starting" + }, + { + "key": "DIAG|azure-init/0.1.1|3f2504e0-4f89-41d3-9a0c-0305e82c3301|finish|provision:run|8f3e9c4a-1b2c-4d5e-9f01-234567890abc|2026-08-31T12:34:57.101Z|none|success|0.312000|0", + "value": "provisioning succeeded" + }, + { + "key": "DIAG|azure-init/0.1.1|3f2504e0-4f89-41d3-9a0c-0305e82c3301|event|example|9f3e9c4a-1b2c-4d5e-9f01-234567890abc|2026-08-31T12:34:57.102Z|zlib+b64|||0", + "value": "eJwLSS0uUSguKcrMS1cwNDIGACxqBQ4=" + } +] +``` + +## Reassembly and Decoding + +The producer encodes the whole payload before splitting it into records. +Group members need not be adjacent or ordered in the pool. + +1. Select the schema from the first key field and validate its metadata. +2. Group by the full key excluding only `chunk_index`. Event ID alone is not + sufficient: it would combine start and finish records. +3. Sort indices numerically and require a unique, contiguous sequence from zero. +4. Concatenate values in that order, then decode according to `encoding`. + +### Encodings + +| Token | Stored value | Decoded content | +|---|---|---| +| `none` | Plain UTF-8 without NUL | Text | +| `zlib+b64` | Standard padded base64 of one zlib stream (RFC 1950) | Bytes | +| `gz+b64` | Standard padded base64 of one gzip member (RFC 1952) | Bytes | + +For compressed encodings, decode base64 before decompressing. Native base64 has +no whitespace. Zlib uses DEFLATE with a 32 KiB window (`wbits=15`) and no preset +dictionary; gzip uses a basic header without optional fields. The tokens are +not aliases. Compression does not imply that the decoded content is text. + +### Kusto Consumption + +Use `zlib+b64` for new compressed telemetry targeting Kusto's +[zlib function](https://learn.microsoft.com/en-us/kusto/query/zlib-base64-decompress-function), +which requires window size 15. Pass the complete reassembled base64 value: + +```kusto +print message = zlib_decompress_from_base64_string("eJwLSS0uUSguKcrMS1cwNDIGACxqBQ4=") +``` + +The result is `Test string 123`. Actual gzip data requires the separate +[gzip function](https://learn.microsoft.com/en-us/kusto/query/gzip-base64-decompress), +which does not support optional gzip header fields. Both return strings, not +arbitrary binary artifacts. Cloud-init needs the source-specific extraction +described [below](#cloud-init-compatibility). + +## Limits and Invalid Records + +Native producer limits count UTF-8 bytes after encoding: + +| Item | Limit | +|---|---| +| Complete key, including delimiters and index | 254 bytes | +| Encoded value per chunk | 1,022 bytes | +| Chunks per payload | 1,023, indexed 0 through 1022 | +| Agent / name | 32 / 48 bytes | +| Each UUID | 36 bytes | + +These budgets do not impose equivalent read limits on other producers' records. +Encoded size also does not bound decompressed size. + +Invalid metadata, unsupported encodings, duplicate or missing indices, invalid +base64, corrupt or truncated compressed streams, and trailing bytes after a +completed stream prevent decoding. Preserve the affected records rather than +silently dropping or repairing them. Unrelated keys are not diagnostic errors. + +There is no total-chunk-count field. Missing trailing records in a plain-text +payload cannot be detected if the remaining indices are contiguous from zero. +Compressed streams additionally allow completion and checksum validation. + +## Provisioning Reports + +The exact key `PROVISIONING_REPORT` identifies a separate, single-record health +report. Its value is pipe-delimited CSV whose fields contain `key=value`. +Fields containing pipes, quotes or newlines use CSV double-quote escaping; +split each decoded field on its first `=`. Field order is not significant for +reading. + +Required fields are `result`, `agent`, `vm_id`, `pps_type` and an RFC 3339 +`timestamp`. Report results are `success` or `error`, not the diagnostic +`success`/`fail` tokens. Error reports also require `reason` and may include +`documentation_url`. Pre-provisioning types are `None`, `PreprovisionedOSDisk`, +`Running`, `Savable` and `Unknown`. + +Other fields are ordered supporting data, including duplicate supporting-data +keys. Required fields cannot be duplicated. On error reports, `reason` and +`documentation_url` cannot be duplicated either. Empty text values remain +supported. Report identities and timestamp spellings are preserved; native +diagnostic UUID and timestamp restrictions are not imposed on reports. + +Reports replace the prior provisioning result, are not chunked, and must fit +one value. + +## Cloud-init Compatibility + +Cloud-init uses a separate format in the same pool. Current keys include a VM +identity that older keys omit: + +```text +CLOUD_INIT|||||[|] +CLOUD_INIT||||[|] +``` + +The incarnation is numeric, and the VM and event identities are UUIDs. Values +contain `name`, `type`, `ts` and `msg`; finishes add `result` and `duration`, +and chunks add `msg_i`. The value's name and type must match the key. + +Apply the grouping and index checks above to the complete cloud-init base key, +including incarnation and source type. `msg_i` must match the key index, and +other metadata must agree. Join the still-escaped `msg` fragments before +unescaping once: cloud-init can split a JSON escape sequence across records, +so individual chunks need not be valid standalone JSON. + +| Normalized field | Cloud-init source | +|---|---| +| Agent | Literal `CLOUD_INIT` | +| Kind | Source `start` or `finish`; all other types become `event` | +| Name | Source name | +| VM identity | Key field when present; otherwise absent | +| Event identity | Key event UUID | +| Timestamp | `ts`, RFC 3339 with offsets allowed; convert to UTC | +| Outcome | Finish `SUCCESS` becomes `success`; `FAIL` becomes `fail` | +| Duration | Finish's numeric seconds | +| Encoding | Embedded `msg` envelope, when present | + +Do not turn a finish result such as `WARN` into a success or failure; retain it +as unparsed data. Other source types become events without a normalized outcome +or duration. Negative or overflowing finish durations are invalid. + +Compressed artifacts put a JSON object with `encoding` and `data` inside `msg`; +extract `data` after reassembly and unescaping. +Cloud-init's [Azure producer](https://github.com/canonical/cloud-init/blob/main/cloudinit/sources/helpers/azure.py) +uses zlib compression and line-wrapped base64 while labeling the envelope +`gz+b64`. Remove ASCII base64 whitespace and accept zlib or gzip under that +source label, using the actual stream format to select the decompressor. +This exception does not apply to native `gz+b64` records. Other envelope +encodings are unsupported; messages without an encoding envelope are text. + +## Implementation Notes + +The following describe `libazureinit-kvp`, not requirements for a consumer's API. + +### Reader Output + +Each decoded group appears at its first record's position. Failed groups retain +their physical records and positions. Reads do not modify the pool or sort by +timestamp. Decoding errors are attached to preserved records: + +| Condition | Error token | +|---|---| +| Unsupported `DIAG_V*` schema | `unsupported_version` | +| Missing index zero or an index gap | `incomplete_group` | +| Repeated index | `duplicate_chunk` | +| Unsupported encoding or invalid payload | `undecodable` | +| Invalid metadata or report | `malformed` | + +Unrelated keys have no decoding error. Invalid pool framing or UTF-8 fails the +whole read. Grouping uses memory proportional to input and decoded content; +there is no decoded-size cap. + +Parsed JSON represents decoded binary payloads as base64 objects. Those bytes +are already decompressed; do not send them to a decompressor again. Durations +serialize as numeric seconds, with possible floating-point precision loss at +large values. Stored durations use integer arithmetic to retain the selected +precision. Cloud-init durations are rounded to microseconds during conversion, +and the source encoding label is retained. + +### Writer Choices + +Validation precedes writing, but I/O failure may leave a partial batch. Pool +cleanup is explicit. The maximum native key is 251 bytes, within the 254-byte +budget even at nanosecond precision and the largest supported duration: + +| Key component | Maximum bytes | +|---|---:| +| `DIAG` | 4 | +| Agent | 32 | +| VM UUID | 36 | +| Kind | 6 | +| Name | 48 | +| Event UUID | 36 | +| Timestamp | 30 | +| Encoding | 8 | +| Result | 7 | +| Duration | 30 | +| Chunk index | 4 | +| Ten pipe delimiters | 10 | +| Total | 251 | + +Report writers emit success fields as `result`, `agent`, `pps_type`, `vm_id`, +`timestamp`, then extras. Failure order is `result`, `reason`, `agent`, extras, +`pps_type`, `vm_id`, `timestamp`, then the optional documentation URL. Consumers +must not depend on that order. \ No newline at end of file diff --git a/doc/getting_started.md b/doc/getting_started.md index 941888c1..0292e69f 100644 --- a/doc/getting_started.md +++ b/doc/getting_started.md @@ -134,5 +134,6 @@ Both containers will output all logs they have access to in order to better debu ## Next Steps - Review the [Configuration Guide](configuration.md) for detailed configuration options -- Understand the [Tracing System](libazurekvp.md) for monitoring and debugging +- Read the [KVP contract](kvp.md) for pool storage and Hyper-V transport behavior +- Read the [diagnostics contract](diagnostics.md) for telemetry formats and consumer behavior - Explore [End-to-End Testing](e2e_testing.md) for comprehensive testing diff --git a/doc/kvp.md b/doc/kvp.md index d12ac266..98ab1caa 100644 --- a/doc/kvp.md +++ b/doc/kvp.md @@ -130,7 +130,7 @@ full details. | hv_kvp_daemon | Upsert + full rewrite | `fcntl` | 512 B (field width) | 2,048 B (field width) | N/A | None | Not checked | Yes (shift + rewrite) | Yes (`kvp_update_mem_state`) | 0–3 | | cloud-init | Append-only | `flock()` | 512 B (field width) | 1,024 B (1,023 + null-terminator) | Truncates | Truncate if `mtime` < boot | Yes | No | No | Pool 1 only | | azure-init (current) | Append-only, batched | `flock()` (via `fs2`) | 512 B (field width) | 1,022 B/chunk | Splits across records | Truncate if `mtime` < boot (no lock) | Zero-padded (implicit) | No | No | Pool 1 only (hardcoded) | -| libazureinit-kvp (planned) | Upsert | `flock()` + `fcntl` | Error if > 254 B | Error if > 1,022 B | Error | Option to truncate if `mtime` < boot (with lock) | Explicit null-terminator | Planned | N/A (direct file I/O) | Any pool (configurable) | +| libazureinit-kvp | Append / upsert / replace | `flock()` + `fcntl` | Error if > 254 B | Error if > 1,022 B | Error | Option to truncate if `mtime` < boot (with lock) | Explicit null-terminator | Yes | N/A (direct file I/O) | Any pool (configurable) | #### flock vs fcntl @@ -242,52 +242,31 @@ no truncation. --- -## Diagnostics API and CLI - -`libazureinit-kvp` exports `DiagnosticWriter` for `DIAG_V1` records and -`DiagnosticReader` for diagnostics, provisioning reports, and raw records. -The writer validates the agent identifier and VM UUID at construction. -`emit_start` and `emit_finish` share a caller-supplied event UUID; -`emit_event` generates its own. Durations are `std::time::Duration` -values stored as integer microseconds. Timestamps are RFC 3339 UTC (`Z`) -at millisecond precision by default; `DiagnosticWriter::with_timestamp_precision` -selects second, microsecond, or nanosecond precision, and the reader -accepts any of those canonical precisions. -Payloads are plain UTF-8 text or gzip plus base64 (`Encoding::GzB64`). - -Each reader call takes one fresh snapshot and returns `Entry::Diagnostic`, -`Entry::Report`, or `Entry::Raw` in first-seen pool order. Unknown records -remain raw; recognized but invalid records also carry a `DecodeError`. -Cloud-init records are supported through a read-only compatibility bridge. -Invalid physical UTF-8 fails the entire snapshot without partial output. -Neither reading nor emitting diagnostics clears stale pool data implicitly. - -| Command | Output | -|---------|--------| -| `libazureinit-kvp dump` | JSON array of physical key/value records in pool order, including duplicates | -| `libazureinit-kvp dump --parse` | JSON array of decoded diagnostics, reports, and raw entries in pool order | -| `libazureinit-kvp dump --text` | Physical records as `KEY=VALUE` lines | -| `libazureinit-kvp dump --parse --text` | One line per entry in pool order; binary payloads render as `payload_b64=` | -| `libazureinit-kvp dump --parse --name ssh` | Filter diagnostic names by substring; retain reports and raw entries | -| `libazureinit-kvp dump --parse --kind finish` | Filter diagnostics by kind (`start`/`finish`/`event`); adding `--name` keeps only diagnostics that match both filters | - -Parsed CLI output preserves the pool order returned by -`DiagnosticReader::entries()`: a complete chunk group appears at its first -record's position, and every other record stays where it sits in the pool. - -`--json` and `--text` are mutually exclusive. Only `dump` defaults to JSON; -other commands retain their text defaults. Global `--dir` and `--pool` -options select the store. - -```sh -libazureinit-kvp emit --agent azure-init \ - --vm-id 3f2504e0-4f89-41d3-9a0c-0305e82c3301 \ - --name user:create_user --message "created azureuser" -``` +## Implementation Notes + +The `libazureinit-kvp` store acquires both `flock` and open-file-description +`fcntl` locks, in that order. Its safe policy enforces the conservative byte +budgets above. Its full-width policy allows the physical field sizes, including +fields without a terminator; the reader accepts a full-width UTF-8 field in +that case. Full-width writes are not guaranteed to survive host transport. + +Appending retains duplicates without a record-count cap. Inserting updates a +key and collapses its duplicates; inserting new keys and replacing the pool +enforce a limit of 1,024 distinct keys. That limit is a library policy, not an +extra field or universal constraint of the KVP format. Map-style reads use the +last stored value, while physical reads retain every record. + +Deletion may swap a record with the file's tail, changing order. Stale-data +cleanup is explicit and compares modification time with system boot time under +the write lock. Invalid framing or invalid UTF-8 content fails a read; padding +after a NUL is ignored rather than interpreted as text. + +The [diagnostics contract](diagnostics.md) defines telemetry carried in KVP records. Rust API usage is covered by the crate's generated documentation. -The separate reader and writer replace `DiagnosticsKvp`. `--parse` replaces -`--parse-diagnostics`, and `emit --agent` replaces `--prefix`. The `--tail` -and `-n` options are removed. +## References -See the [diagnostics specification](../libazureinit-kvp/diagnostics-proposal.md) -for the wire format, validation limits, and cloud-init compatibility rules. +- [Linux kernel UAPI](https://github.com/torvalds/linux/blob/master/include/uapi/linux/hyperv.h) +- [Linux KVP driver](https://github.com/torvalds/linux/blob/master/drivers/hv/hv_kvp.c) +- [Linux KVP daemon](https://github.com/torvalds/linux/blob/master/tools/hv/hv_kvp_daemon.c) +- [Cloud-init KVP reporting](https://github.com/canonical/cloud-init/blob/main/cloudinit/reporting/handlers.py) +- [Microsoft Data Exchange overview](https://learn.microsoft.com/en-us/windows-server/virtualization/hyper-v/integration-services-data-exchange) diff --git a/libazureinit-kvp/diagnostics-proposal.md b/libazureinit-kvp/diagnostics-proposal.md deleted file mode 100644 index ba3e5733..00000000 --- a/libazureinit-kvp/diagnostics-proposal.md +++ /dev/null @@ -1,512 +0,0 @@ -# KVP Diagnostics Specification - -## Background - -Diagnostics explain what a provisioning client did during boot: spans mark the start and finish of operations such as `provision:run`, while point observations capture events or artifacts such as an IMDS result or `dmesg` snapshot. They let operators reconstruct a boot and triage failures even when the guest is unreachable. - -A Hyper-V guest exposes these records to the host through a KVP pool, a flat key=value namespace. `KvpPoolStore` provides raw, in-order access but no diagnostic semantics. - -The pool constrains the design in four ways: - -- Records are flat key=value pairs; keys hold structured metadata and values are opaque bytes. -- Safe limits are 254-byte keys and 1022-byte values. Larger values span records whose keys differ by a trailing index. -- The host may copy or truncate the pool at any moment, leaving group members missing, duplicated, or out of order. -- Grouping and typing exist only by convention in the key and value. - -This spec defines the versioned format emitted by azure-init, its writer, and a reader for the pool. cloud-init uses a separate format supported read-only through the compatibility bridge. - -### Records today - -azure-init uses a pipe-delimited key with a plain-text value and no encoding field. Its `type` is `start`, `finish`, or `event`; oversized values use a trailing chunk index: - -```text -||||||| - -# point event, one record -azure-init-0.1.1|1700000000|vm-abc|event|imds|8f3e9c4a-1b2c-4d5e-9f01-234567890abc|2026-07-27T21:33:24.300Z|0 - value: Retrieved 1 key from IMDS - -# span start -azure-init-0.1.1|1700000000|vm-abc|start|provision:run|9c1d2e3f-4a5b-6c7d-8e9f-0a1b2c3d4e5f|2026-08-31T12:34:56.789Z|0 - value: starting - -# span finish, same event_id as its start -azure-init-0.1.1|1700000000|vm-abc|finish|provision:run|9c1d2e3f-4a5b-6c7d-8e9f-0a1b2c3d4e5f|2026-08-31T12:34:57.101Z|0 - value: provisioning succeeded - -# long value split across records, one event_id, indices 0..N -azure-init-0.1.1|1700000000|vm-abc|event|config:dump|1a2b3c4d-5e6f-7a8b-9c0d-1e2f3a4b5c6d||0 value: -azure-init-0.1.1|1700000000|vm-abc|event|config:dump|1a2b3c4d-5e6f-7a8b-9c0d-1e2f3a4b5c6d||1 value: -``` - -cloud-init puts type and name before the identifiers; current keys carry a `vm_id` that older keys omit. Values are JSON with `ts` and `msg`; finishes add `result` and `duration`, splits add `msg_i`, and compressed artifacts embed `{encoding, data}` in `msg`: - -```text -current CLOUD_INIT|||||[|] -older CLOUD_INIT||||[|] - -# span finish with result and duration (current key) -CLOUD_INIT|1785187982|finish|modules-final/config-scripts_user|0e5e179d-5341-478b-8456-fbb90621bdf8|e5f01809-a7a3-4279-aa64-1f18e21eda6e - value: {"name":"modules-final/config-scripts_user","type":"finish","ts":"2026-07-27T21:33:24.339006+00:00","result":"SUCCESS","duration":0.00064,"msg":"config-scripts_user ran successfully and took 0.001 seconds"} - msg -> "config-scripts_user ran successfully and took 0.001 seconds" - -# span start, no result or duration -CLOUD_INIT|1785187982|start|modules-final/config-ssh_authkey_fingerprints|0e5e179d-5341-478b-8456-fbb90621bdf8|c4d4a08d-fe93-4c7a-9be6-9a38c212e212 - value: {"name":"modules-final/config-ssh_authkey_fingerprints","type":"start","ts":"2026-07-27T21:33:24.339170+00:00","msg":"running config-ssh_authkey_fingerprints with frequency once-per-instance"} - msg -> "running config-ssh_authkey_fingerprints with frequency once-per-instance" - -# older key without vm_id (five base fields) -CLOUD_INIT|1785187982|finish|modules-final|126f969f-13fd-4b4b-a136-b7114518491f - value: {"name":"modules-final","type":"finish","ts":"2026-07-27T21:33:24.431885+00:00","result":"SUCCESS","duration":0.340712044,"msg":"running modules for final"} - msg -> "running modules for final" - -# point event -CLOUD_INIT|1785187982|event|network-config|0e5e179d-5341-478b-8456-fbb90621bdf8|a1b2c3d4-e5f6-7a8b-9c0d-1e2f3a4b5c6d - value: {"name":"network-config","type":"event","ts":"2026-07-27T21:33:20.100000+00:00","msg":"applied fallback network configuration"} - msg -> "applied fallback network configuration" - -# system-info (not a timeline position; the subject is in name) -CLOUD_INIT|1785187982|system-info|system information|0e5e179d-5341-478b-8456-fbb90621bdf8|b2c3d4e5-f6a7-8b9c-0d1e-2f3a4b5c6d7e - value: {"name":"system information","type":"system-info","ts":"2026-07-27T21:33:19.500000+00:00","msg":"cloud-init running on Ubuntu"} - msg -> "cloud-init running on Ubuntu" - -# split value: each chunk carries msg_i, and a JSON \n escape is split across the boundary -CLOUD_INIT|1785187982|finish|modules-final|0e5e179d-5341-478b-8456-fbb90621bdf8|c3d4e5f6-a7b8-9c0d-1e2f-3a4b5c6d7e8f|0 - value: {"name":"modules-final","type":"finish","ts":"2026-07-27T21:33:24.43Z","msg_i":0,"msg":"line1\"} -CLOUD_INIT|1785187982|finish|modules-final|0e5e179d-5341-478b-8456-fbb90621bdf8|c3d4e5f6-a7b8-9c0d-1e2f-3a4b5c6d7e8f|1 - value: {"name":"modules-final","type":"finish","ts":"2026-07-27T21:33:24.43Z","msg_i":1,"msg":"nline2"} - msg -> "line1\nline2" (reassembled from the two chunks) - -# compressed artifact: type=compressed, msg holds an {encoding, data} envelope; a large one splits like the value above -CLOUD_INIT|1785187982|compressed|dmesg|0e5e179d-5341-478b-8456-fbb90621bdf8|d4e5f6a7-b8c9-0d1e-2f3a-4b5c6d7e8f90|0 - value: {"name":"dmesg","type":"compressed","ts":"2026-07-27T21:33:25.00Z","msg_i":0,"msg":"{\"encoding\":\"gz+b64\",\"data\":\"H4sIAAAA...\"}"} - msg -> {"encoding":"gz+b64","data":"H4sIAAAA..."} -``` - -The formats differ in field order, metadata placement, and value representation. - -## Proposed design - -The design adds separate interfaces over `KvpPoolStore`: a reader interprets the pool, and a writer emits the versioned format. Untyped callers use `KvpPoolStore` directly. - -### Diagnostics format - -The proposed key keeps metadata in a pipe-delimited key with `|` reserved. It begins with a schema ID and ends with the chunk index, including `|0` for a single-record value: - -```text -DIAG_V1|||||||||| -``` - -- `DIAG_V1` is the `diagnostic_version_id`: `DIAG` identifies the format family and `V1` its schema. The schema defines layout, field requirements, tokens, units, encoding, and chunking; `agent` separately identifies the producer. The combined token distinguishes unsupported `DIAG_V*` records from unrelated keys. Schema changes require a new ID; agent releases do not. -- `boot_epoch` is removed because timestamps identify occurrences and stale-pool cleanup removes prior-boot records. -- `type` becomes `kind`, limited to the timeline positions `start`, `finish`, and `event`; `name`, `encoding`, and `result` carry other classifications. -- `encoding` in the key supports compressed payloads. -- `result` (`success` or `fail`) and `duration` (microseconds) are required on finishes, optional on events, and empty on starts. Values otherwise remain plain text or encoded artifacts. - -| Field | Meaning | -|---|---| -| diagnostic_version_id | `DIAG_V1`; identifies the wire schema and selects its parser before any later field is interpreted | -| agent | Producer identifier, such as `azure-init-0.1.1` | -| vm_id | VM identity | -| kind | `start` or `finish` for a span, `event` for a point observation | -| name | Subject, such as `provision:run` or `dmesg` | -| event_id | Shared by a span's start and finish, and by every chunk of one value | -| timestamp | RFC 3339 (ISO 8601), UTC with a `Z` suffix, at second/millisecond/microsecond/nanosecond precision (azure-init emits millisecond by default), e.g. `2026-08-31T12:34:56.789Z` | -| encoding | How the value is encoded: `none` or `gz+b64` | -| result | `success` or `fail` on a finish, optionally on an event; empty otherwise | -| duration | Elapsed microseconds on a finish, optionally on a timed event; empty otherwise | -| chunk_index | Chunk position, from 0 | - -#### Key size - -The host silently truncates keys past 254 UTF-8 bytes; safe-mode `KvpPoolStore` rejects them first. Only `agent` and `name` are free-form, so the writer caps them to bound the full key: - -| Field | Cap | Bounded by | -|---|---|---| -| diagnostic_version_id | 7 B | fixed `DIAG_V1` token | -| agent | 32 B | free-form producer id | -| name | 48 B | free-form subject | -| vm_id, event_id | 36 B each | GUID / UUID | -| timestamp | 30 B | up to nanosecond precision | -| duration | 13 B | digits | -| result, encoding, kind | ≤ 7 B each | enum token | -| chunk_index | 4 B | at most 1023 records | - -With those caps the worst-case key is 235 bytes, leaving 19 bytes inside the limit. cloud-init reads are never capped; the bridge takes names as they are. - -Wire examples with shortened UUIDs or `` are schematic. Stored `DIAG_V1` records require valid UUIDs and the exact timestamp format above. - -### Kinds - -`kind` marks timeline position only: `start` opens an operation, `finish` closes one, and `event` is a one-off. These cover every timeline position. Other categories belong in `encoding`, `name`, or `result`; values are messages or artifacts, not kind-specific structures. For cloud-init, `compressed` maps to encoding and `system-info` to name (see Compatibility). - -#### start - -A `start` opens a measurable operation such as `provision:run`. Its timestamp marks when it began, and its value is a short message. It shares an `event_id` with its finish; an unmatched start signals an incomplete operation after a hang or crash. - -```text -DIAG_V1|azure-init-0.1.1|vm-abc|start|provision:run|9c1d2e3f-...|2026-08-31T12:34:56.789Z|none|||0 value: starting -``` - -#### finish - -A `finish` closes the span sharing its `event_id`. It records the later timestamp, `result`, and elapsed `duration` at emit time, so it remains self-contained if the start is lost. Its value is a message such as `provisioning succeeded` or `provisioning failed: `. The cloud-init bridge maps equivalent value fields (see Compatibility). - -```text -DIAG_V1|azure-init-0.1.1|vm-abc|finish|provision:run|9c1d2e3f-...|2026-08-31T12:34:57.101Z|none|success|312|0 value: provisioning succeeded -``` - -#### event - -An `event` is a point observation such as an IMDS result or `dmesg` snapshot. It has its own `event_id` and no span pair. Its payload is text (`none`) or a compressed artifact (`gz+b64`), split by `chunk_index` when needed. It may set `result` or `duration` when applicable. - -```text -# plain text, one record -DIAG_V1|azure-init-0.1.1|vm-abc|event|imds|8f3e...|2026-07-27T21:33:24.300Z|none|||0 value: Retrieved 1 key from IMDS - -# an event that is itself a failure sets result -DIAG_V1|azure-init-0.1.1|vm-abc|event|imds|7b2c...|2026-07-27T21:33:24.400Z|none|fail||0 value: IMDS unreachable - -# a self-contained timing sets duration but no result -DIAG_V1|azure-init-0.1.1|vm-abc|event|imds:probe|5d6e...|2026-07-27T21:33:24.500Z|none||52|0 value: probed IMDS in 52ms - -# compressed artifact, split across records, indices 0..N -DIAG_V1|azure-init-0.1.1|vm-abc|event|dmesg|9a1b...||gz+b64|||0 value: -DIAG_V1|azure-init-0.1.1|vm-abc|event|dmesg|9a1b...||gz+b64|||1 value: -``` - -### Payloads - -`DiagnosticPayload` distinguishes valid UTF-8 `Text(String)` from arbitrary `Bytes(Vec)`. Writer methods accept `impl Into` with conversions from `&str`, `String`, `&[u8]`, and `Vec`, avoiding separate method names. - -The caller chooses the wire `encoding`; the input type does not infer whether compression is useful: - -| Input | `encoding=None` | `gz+b64` | -|---|---|---| -| Text | Store its UTF-8 bytes directly | Encode its UTF-8 bytes | -| Bytes | Validate UTF-8, then store directly; reject invalid UTF-8 | Encode the arbitrary bytes | - -On read, `KvpPoolStore::dump()` validates the physical keys and values as UTF-8 before diagnostic parsing. Invalid UTF-8 in any field before its terminating NUL fails the entire snapshot with `KvpError::Io` carrying `InvalidData`; no entries are returned. Record preservation applies only to a successful string-based snapshot, not to arbitrary physical bytes. - -Within a successful snapshot, `none` produces `Text`. `gz+b64` produces `Bytes` even when the decoded bytes are valid UTF-8, because the wire does not declare their content type. Its stored base64 is UTF-8 text, so arbitrary decoded bytes do not violate the store contract. Invalid base64 or gzip remains `Undecodable` and falls back to `Raw`. - -JSON renders `Text` as a string and `Bytes` as `{ "type": "bytes", "encoding": "base64", "data": "..." }`; this presentation does not change the wire `encoding`. - -### Encodings - -`encoding` names the KVP value representation and is chosen by the caller, not inferred from size. Keeping it in the key leaves the value opaque. Arbitrary binary uses `gz+b64` because the UTF-8 store trims trailing nulls. `DIAG_V1` omits standalone base64; a later schema can add encodings without changing the key shape. cloud-init declares encoding inside its value (see Compatibility). - -Values over 1022 bytes are split across chunks sharing `event_id` and `kind`, ordered by `chunk_index`, and joined before decoding. - -#### none - -Plain UTF-8 text and the default, used for span messages and short observations. Decoding joins split text in index order. - -#### gz+b64 - -Base64 of one gzip stream, used for compressible artifacts such as `dmesg`. Writing compresses, encodes, then chunks; decoding reverses the process. A visible index gap is `IncompleteGroup`; contiguous truncation is `Undecodable`. This sacrifices partial readability for fewer records. - -### Reads and writes - -A pool holds many diagnostics. Each is a start, finish, or event, and each is stored as one or more chunks (`chunk_index` 0 to N): - -```text -Diagnostics pool -├─ start provision:run -│ └─ chunk 0 -├─ event imds -│ └─ chunk 0 -├─ event dmesg (gz+b64, 3 chunks) -│ ├─ chunk 0 -│ ├─ chunk 1 -│ └─ chunk 2 -└─ finish provision:run - └─ chunk 0 -``` - -`DiagnosticReader::entries()` reads the pool once through `KvpPoolStore::dump()`. If that snapshot succeeds, each logical item becomes a decoded `Diagnostic`, parsed `ProvisioningReport`, or `Raw` key and value. Unrecognized records are `Raw`; recognized but invalid records are `Raw` with a `DecodeError`, so no record from the successful snapshot is dropped. `KvpPoolStore::dump()` exposes the physical key/value records as UTF-8 strings without diagnostic interpretation. - -Any snapshot failure, including invalid physical UTF-8 or malformed physical record framing, returns `KvpError` with no entries, even when other records are valid. Reads do not modify the pool; they neither skip unreadable records nor replace invalid bytes with lossy text. The existing string-based store and `RawKeyValue` interfaces remain unchanged. - -Only the exact `PROVISIONING_REPORT` key selects report parsing. Its value is one pipe-delimited CSV record of `key=value` fields, with double-quote escaping for embedded pipes, quotes, or newlines. `result` (`success` or `error`), `agent`, `vm_id`, `pps_type`, and an RFC 3339 `timestamp` are required; error reports also require `reason` and may include `documentation_url`. Field order is not significant. Empty text values remain supported, and identities and timestamps are preserved rather than normalized. Other fields remain ordered supporting data, including duplicate supporting-data keys. Duplicate standard fields, unknown enum tokens, missing required fields, or malformed CSV are `Raw` with `Malformed`, not silently repaired or interpreted with first/last-write-wins semantics. - -Writing is the inverse: `DiagnosticWriter` stamps `DIAG_V1`, converts the typed payload according to `encoding`, frames it into records, and appends them to the `KvpPoolStore`. - -The reader preserves first-seen pool order: a complete chunk group occupies its first physical position, and failed groups remain raw records at their original positions. The CLI's `dump --parse` renders entries in that same pool order. A span's timeline can be summarized as: - -```text -2026-08-31T12:34:56.789Z start provision:run -2026-08-31T12:34:57.020Z event imds ok -2026-08-31T12:34:57.101Z finish provision:run success 312ms -``` - -Fields follow `kind`: a finish always carries `result` and `duration`, a start carries neither, and an event carries either when measured. An unmatched start remains visible as an incomplete operation: - -```text -2026-08-31T12:35:10.000Z start provision:run -(no finish) -``` - -No duration rollup is needed; pairing starts and finishes is a group-by on `event_id`. - -Write: - -```mermaid -flowchart TD - client["Provisioning client"] --> new["DiagnosticWriter::new
store, agent, vm_id"] - new --> init{"producer identity valid?"} - init -->|"no"| initerr["Err(KvpError)
writer not constructed"] - init -->|"yes"| emit["DiagnosticWriter::emit_*
text or byte payload"] - emit --> valid{"fields, kind invariants,
payload, and encoding valid?"} - valid -->|"no"| inputerr["Err(KvpError)
nothing written"] - valid -->|"yes"| encsel{"encoding
(caller's choice)"} - encsel -->|"gz+b64"| gz["gzip then base64"] - encsel -->|"none"| plain["text as-is"] - gz --> frame["frame on UTF-8 boundaries
1022-byte value cap"] - plain --> frame - frame --> keys["stamp DIAG_V1
and format chunk keys"] - keys --> limits{"key and chunk-count
limits satisfied?"} - limits -->|"no"| inputerr - limits -->|"yes"| append["KvpPoolStore::append_multiple
all chunks under one lock"] - append -->|"ok"| store[("KvpPoolStore
flat key=value pool")] - store --> ok["Ok(())"] - append -->|"lock / write / flush error"| writeerr["Err(KvpError)
batch may be partial"] -``` - -Read (`DiagnosticReader::entries()`): - -```mermaid -flowchart TD - client["Diagnostic consumer / CLI"] --> new["DiagnosticReader::new(store)
no IO, cannot fail"] - new --> entries["DiagnosticReader::entries()"] - entries --> dump["KvpPoolStore::dump()"] - dump -->|"lock / read / framing / UTF-8 error"| readerr["Err(KvpError)
no entries returned"] - dump -->|"ok"| cls{"first key field"} - cls -->|"DIAG_V1"| dec["source parser
parse, group, decode"] - cls -->|"unsupported DIAG_V*"| rawver["Entry::Raw
UnsupportedVersion"] - cls -->|"CLOUD_INIT"| bridge["cloud-init bridge"] - bridge --> dec - cls -->|"PROVISIONING_REPORT"| rep["parse report"] - cls -->|"neither"| raw["Entry::Raw
error: None"] - dec -->|"ok"| diag["Entry::Diagnostic"] - dec -->|"bad key or source value"| malformed["Entry::Raw
Malformed"] - dec -->|"missing chunk"| incomplete["Entry::Raw
IncompleteGroup"] - dec -->|"duplicate index"| duplicate["Entry::Raw
DuplicateChunk"] - dec -->|"unknown encoding / bad data"| undecodable["Entry::Raw
Undecodable"] - rep -->|"ok"| repe["Entry::Report"] - rep -->|"malformed"| malformed - rawver --> out["Ok(Vec<Entry>)"] - raw --> out - diag --> out - malformed --> out - incomplete --> out - duplicate --> out - undecodable --> out - repe --> out -``` - -`KvpError` means construction, reading, or writing failed and is returned by the method; snapshot failures include invalid physical UTF-8. `DecodeError` describes uninterpretable records within a successful string-based snapshot; `entries()` still succeeds and preserves those records as `Entry::Raw`. - -The writer validates the full batch before storage, so identity, field, payload, encoding, and size errors write nothing; this includes non-UTF-8 bytes with `encoding=None`. Once `append_multiple` starts, an error may leave a partial batch. Readers report visible index gaps as `IncompleteGroup` and invalid encoded content as `Undecodable`. Without a total chunk count, a contiguous prefix of a `none` payload is indistinguishable from a complete value. - -Every group key includes `diagnostic_version_id` and `kind`, preventing chunks from different schemas or span ends from combining. There is no decode-time size limit. Writer-side field and chunk caps do not bound reader input: raw pool appends have no record-count cap, and compressed payloads may expand substantially. Reading a large pool or highly compressed payload can therefore require substantial memory. - -## Crate design - -Both interfaces hold no files or locks and delegate IO to `KvpPoolStore`. Provisioning clients use `DiagnosticWriter`, supplying diagnostic meaning and payload rather than constructing keys or chunks. Consumers use `DiagnosticReader`; untyped callers use the store directly. The writer emits azure-init records, while the reader supports known schemas and cloud-init through a bridge. - -Initialization is asymmetric: reader identity comes from stored records, so `DiagnosticReader` needs only a store; `DiagnosticWriter` also needs and validates the local `agent` and `vm_id`. Neither constructor reads the pool or boot state. The writer always emits `DIAG_V1`; callers cannot select a version or cloud-init format. Both may share clones of one `KvpPoolStore`. - -```rust -const DIAGNOSTIC_VERSION_ID: &str = "DIAG_V1"; - -enum Kind { Start, Finish, Event } - -/// Plain text is `None`; a compressed value is `GzB64`. -/// `Other` keeps an unknown token so it decodes to `Undecodable`, never a panic. -enum Encoding { GzB64, Other(String) } - -enum Outcome { Success, Failure } - -/// The decoded payload. Rust strings guarantee UTF-8; bytes make no text claim. -/// `From` implementations map `&str` and `String` to `Text`, and `&[u8]` -/// and `Vec` to `Bytes`. -enum DiagnosticPayload { - Text(String), - Bytes(Vec), -} - -/// Why a recognized record could not be parsed, carried by the `Raw` it falls back to. -/// Implements `Error`, serialized as a snake_case reason. -enum DecodeError { - /// The key identifies the diagnostics family, but not a version this reader supports. - UnsupportedVersion, - /// Chunks are missing: not a contiguous run from 0. - IncompleteGroup, - /// A `chunk_index` appears more than once. - DuplicateChunk, - /// Unknown encoding, bad base64, or truncated gzip. - Undecodable, - /// A recognized value did not parse, such as a malformed `PROVISIONING_REPORT`. - Malformed, -} - -/// The identity the three kinds share. Not the `diagnostic_version_id`, `kind`, `result`, `duration`, or `chunk_index`. -/// The reader consumes the schema ID while selecting a parser, then every supported source maps here. -struct DiagnosticKey { - agent: String, - /// Older cloud-init keys omit it. - vm_id: Option, - name: String, - /// One per span (start and finish share it) or standalone event. - event_id: String, - /// RFC 3339, UTC; millisecond precision by default, configurable per writer. - timestamp: DateTime, - encoding: Option, -} - -/// Opens a span. -struct DiagnosticStart { key: DiagnosticKey, payload: DiagnosticPayload } -/// Closes a span; carries its verdict and elapsed duration (microseconds on the wire). -struct DiagnosticFinish { key: DiagnosticKey, payload: DiagnosticPayload, result: Outcome, duration: Duration } -/// A point observation; may carry a verdict or a self-contained timing. -struct DiagnosticEvent { key: DiagnosticKey, payload: DiagnosticPayload, result: Option, duration: Option } - -/// One decoded emission, typed by kind. -enum Diagnostic { - Start(DiagnosticStart), - Finish(DiagnosticFinish), - Event(DiagnosticEvent), -} - -/// An uninterpreted record from a successful UTF-8 snapshot. -struct RawKeyValue { - key: String, - value: String, - error: Option, -} - -/// Every item from a successful snapshot is interpreted or preserved as `Raw`. -enum Entry { - Diagnostic(Diagnostic), - Report(ProvisioningReport), - Raw(RawKeyValue), -} - -/// Interprets a pool without any local producer identity. -struct DiagnosticReader { - store: KvpPoolStore, -} - -/// Produces the current azure-init diagnostics format. -struct DiagnosticWriter { - store: KvpPoolStore, - agent: String, - vm_id: String, -} - -impl DiagnosticReader { - /// Constructing a reader performs no IO; the pool is read by `entries()`. - pub fn new(store: KvpPoolStore) -> Self; - - /// Interprets one UTF-8 snapshot; a store failure returns no entries. - pub fn entries(&self) -> Result, KvpError>; -} - -impl DiagnosticWriter { - /// Fix the local producer identity used by every emitted record. - /// The writer always emits `DIAG_V1`. - pub fn new(store: KvpPoolStore, agent: impl Into, vm_id: impl Into) -> Result; - - /// Open a span. `event_id` links this start to the finish that closes it. - pub fn emit_start(&self, event_id: &str, name: &str, payload: impl Into, encoding: Option) -> Result<(), KvpError>; - - /// Close the span opened under `event_id`, recording its `result` and elapsed `duration`. - pub fn emit_finish(&self, event_id: &str, name: &str, payload: impl Into, encoding: Option, result: Outcome, duration: Duration) -> Result<(), KvpError>; - - /// Record a standalone point observation; the writer assigns its `event_id`. - /// `result` and `duration` are set only when measured. - pub fn emit_event(&self, name: &str, payload: impl Into, encoding: Option, result: Option, duration: Option) -> Result<(), KvpError>; -} -``` - -### CLI - -`dump` defaults to JSON; `--json` makes that explicit and `--text` selects human-readable output. Without `--parse`, it returns every physical record in pool order. `--parse` calls `DiagnosticReader::entries()`, returning typed diagnostics and reports while preserving other or invalid records as `Raw`. Both modes require a successful string-based snapshot; invalid physical UTF-8 fails the command without returning records. - -Parsed JSON and text output preserve the reader's first-seen pool order; the CLI does not reorder entries. In parsed text output, binary payloads render as `payload_b64=`. This presentation does not change the reader API's ordering or write to the pool. - -```text -dump -> JSON array of every physical {key, value} record -dump --parse -> JSON array of Diagnostic, ProvisioningReport, and Raw entries -dump --text -> every physical record as raw KEY=VALUE -dump --parse --text -> one human-readable line per interpreted entry -``` - -- `--name` filters the parsed diagnostics by name; other entries are unaffected. -- `--json` and `--text` are mutually exclusive; for `dump`, omitting both is equivalent to `--json`. Other commands retain their text defaults. -- In parsed text output, `Text` payloads print directly and `Bytes` payloads print as standard base64 under `payload_b64`. -- `--parse` replaces `--parse-diagnostics`; `emit --agent` replaces `--prefix`. `--tail` and `-n` are removed. - -Examples: - -```text -# default dump: every physical record as JSON, including the report and a truncated dmesg group -$ dump -[ - {"key":"DIAG_V1|azure-init-0.1.1|3f2504e0-4f89-41d3-9a0c-0305e82c3301|finish|provision:run|9c1d2e3f-4a5b-6c7d-8e9f-0a1b2c3d4e5f|2026-08-31T12:34:57.101Z|none|success|312|0","value":"provisioning succeeded"}, - {"key":"DIAG_V1|azure-init-0.1.1|3f2504e0-4f89-41d3-9a0c-0305e82c3301|event|dmesg|d4e5f6a7-b8c9-0d1e-2f3a-4b5c6d7e8f90|2026-07-27T21:33:25.000Z|gz+b64|||17","value":""}, - {"key":"PROVISIONING_REPORT","value":"result=success|agent=azure-init-0.1.1|pps_type=None|vm_id=3f2504e0-4f89-41d3-9a0c-0305e82c3301|timestamp=2026-08-31T12:34:57.500Z"} -] - -# --parse remains JSON: the diagnostic decodes, the report parses, and the dmesg chunk stays Raw -$ dump --parse -[ - {"type":"diagnostic","kind":"finish","agent":"azure-init-0.1.1","vm_id":"3f2504e0-4f89-41d3-9a0c-0305e82c3301","name":"provision:run","event_id":"9c1d2e3f-4a5b-6c7d-8e9f-0a1b2c3d4e5f","timestamp":"2026-08-31T12:34:57.101Z","encoding":"none","result":"success","duration":312,"payload":"provisioning succeeded"}, - {"type":"PROVISIONING_REPORT","result":"success","agent":"azure-init-0.1.1","vm_id":"3f2504e0-4f89-41d3-9a0c-0305e82c3301","timestamp":"2026-08-31T12:34:57.500Z","pps_type":"None"}, - {"type":"raw","key":"DIAG_V1|azure-init-0.1.1|3f2504e0-4f89-41d3-9a0c-0305e82c3301|event|dmesg|d4e5f6a7-b8c9-0d1e-2f3a-4b5c6d7e8f90|2026-07-27T21:33:25.000Z|gz+b64|||17","value":"","error":"incomplete_group"} -] -``` - -## Adoption - -azure-init adopts `DIAG_V1` as its first supported schema when it switches to this crate. Earlier unversioned records are pre-adoption, not a compatibility contract; `DiagnosticReader` treats them as `Raw`. No migration is required. - -Updating cloud-init to emit this format is a non-goal; it remains read-only through the compatibility bridge. - -## Compatibility - -cloud-init uses its own format in the same pool. `DiagnosticReader` maps it through a read-only bridge; `DiagnosticWriter` never emits it, and the `DIAG_V1` parser never parses it. - -cloud-init keys put type and name before the identifiers; current keys include a `vm_id` that older keys omit: - -```text -current CLOUD_INIT|||||[|] -older CLOUD_INIT||||[|] -``` - -Values are JSON with `name`, `type`, `ts`, and `msg`; finishes add `result` and `duration`, splits add `msg_i`, and compressed artifacts embed `{encoding, data}` in `msg`. - -The bridge maps fields onto the model: - -| Model field | cloud-init source | -|---|---| -| agent | the literal `CLOUD_INIT` | -| kind | `type` (`start`, `finish`, else `event`) | -| name | `name` | -| vm_id | present only on current keys | -| event_id | trailing key identifier | -| timestamp | value `ts`, read by the bridge when it maps the record | -| result | value field on a finish, mapped to the model's `result` | -| duration | value field on a finish (seconds; the bridge converts to microseconds) | -| encoding | the value `{encoding, data}` envelope, not the key | - -Finish results `SUCCESS` and `FAIL` map to `success` and `fail`. An unmappable result such as `WARN` is preserved as `Raw` with `Malformed`; it is not coerced to a verdict or reclassified as an event. Source types other than `start` and `finish`, including standalone warnings, map to `event` without a span outcome. Duration conversion rounds to whole microseconds and rejects negative or overflowing values. - -The bridge uses cloud-init's `incarnation` to keep chunk groups separate, then discards it; `DiagnosticKey` does not expose it. - -The `CLOUD_INIT` prefix selects the bridge; no `DIAG_V1` field is invented. After source-specific parsing, the bridge constructs the same version-independent `DiagnosticKey` as the `DIAG_V1` parser. - -For cloud-init chunks, the bridge validates each `msg_i` and the consistency of source metadata, joins the still-escaped `msg` slices, then unescapes once. An `{encoding, data}` envelope decodes to `DiagnosticPayload::Bytes`; otherwise `msg` becomes `DiagnosticPayload::Text`. Its encoding comes from the value, not the key. - -cloud-init artifacts labeled `gz+b64` may contain zlib-wrapped data and line-wrapped base64. The bridge accepts zlib or gzip under that label and removes base64 whitespace before decoding. This read-only compatibility rule does not change `DIAG_V1`, which remains gzip-only. Other envelope encodings, including standalone `b64`, are `Undecodable`. \ No newline at end of file diff --git a/libazureinit-kvp/src/cli.rs b/libazureinit-kvp/src/cli.rs index a0387bad..0fd2e50f 100644 --- a/libazureinit-kvp/src/cli.rs +++ b/libazureinit-kvp/src/cli.rs @@ -14,8 +14,9 @@ use serde_json::json; use crate::{ write_report, Diagnostic, DiagnosticPayload, DiagnosticReader, - DiagnosticWriter, Entry, Kind, KvpError, KvpPool, KvpPoolStore, PoolMode, - ProvisioningReport, ReportPpsType, PROVISIONING_REPORT_KEY, + DiagnosticWriter, DurationPrecision, Entry, Kind, KvpError, KvpPool, + KvpPoolStore, PoolMode, ProvisioningReport, ReportPpsType, + PROVISIONING_REPORT_KEY, }; const EXIT_OK: u8 = 0; @@ -23,11 +24,14 @@ const EXIT_NOT_FOUND: u8 = 1; const EXIT_USAGE_OR_VALIDATION: u8 = 2; const EXIT_IO: u8 = 3; -/// Default reporting agent identifier, derived from this crate's version +/// Default reporting agent identifier, derived from this crate's version. const DEFAULT_AGENT: &str = concat!("libazureinit-kvp/", env!("CARGO_PKG_VERSION")); -/// Entry point for the `libazureinit-kvp` binary. +/// Runs the command-line interface using process arguments and standard I/O. +/// +/// Returns the command's exit status. Library callers should use the store, +/// diagnostic or report APIs directly. pub fn run() -> ExitCode { let cli = Cli::parse(); let stdout = io::stdout(); @@ -59,7 +63,7 @@ struct Cli { #[arg(long, global = true)] dir: Option, - /// Use full wire-format key/value limits instead of the safe profile. + /// Use larger key/value limits that may be truncated in host transport. #[arg(long = "unsafe", global = true)] unsafe_mode: bool, @@ -123,12 +127,12 @@ enum Command { key: String, value: String, }, - /// Emit a DIAG_V1 point event with a fresh UUID and current timestamp. + /// Emit a standalone diagnostic with a fresh UUID and current timestamp. Emit { /// Event name, e.g. user:create_user. #[arg(long)] name: String, - /// Event message (stored as the record value). + /// Event payload. #[arg(long)] message: String, /// VM UUID (defaults to the current VM's ID). @@ -587,7 +591,9 @@ fn diagnostic_text(diagnostic: &Diagnostic) -> String { let _ = write!(line, " result={result}"); } if let Some(duration) = duration { - let _ = write!(line, " duration={}us", duration.as_micros()); + let seconds = DurationPrecision::Nanos.format(duration); + let seconds = seconds.trim_end_matches('0').trim_end_matches('.'); + let _ = write!(line, " duration={seconds}s"); } match diagnostic.payload() { DiagnosticPayload::Text(text) => { @@ -1824,7 +1830,6 @@ mod tests { #[case(KvpError::EmptyEventField { field: "name" })] #[case(KvpError::EventFieldTooLong { field: "agent", max: 32, actual: 33 })] #[case(KvpError::InvalidUuid { field: "event_id" })] - #[case(KvpError::DurationTooLarge { max_us: 9_999_999_999_999, actual_us: u64::MAX })] #[case(KvpError::TooManyChunks { max: 1023 })] #[case(KvpError::PayloadNotUtf8)] #[case(KvpError::UnsupportedEncoding { token: "zstd+b64".into() })] @@ -1944,7 +1949,7 @@ mod tests { let ts = "2026-08-31T12:34:56.789Z"; let vm = "00000000-0000-0000-0000-000000000abc"; let diag = |kind: &str| { - format!("DIAG_V1|agent|{vm}|{kind}|span|{event_id}|{ts}|none|||0") + format!("DIAG|agent|{vm}|{kind}|span|{event_id}|{ts}|none|||0") }; store.append(&diag("start"), "starting").unwrap(); store.append(&diag("event"), "obs").unwrap(); diff --git a/libazureinit-kvp/src/diagnostics/cloud_init.rs b/libazureinit-kvp/src/diagnostics/cloud_init.rs index 22806ff5..c65ccbd7 100644 --- a/libazureinit-kvp/src/diagnostics/cloud_init.rs +++ b/libazureinit-kvp/src/diagnostics/cloud_init.rs @@ -1,12 +1,10 @@ // Copyright (c) Microsoft Corporation. // Licensed under the MIT License. -use std::io::Read; use std::time::Duration; use base64::{engine::general_purpose::STANDARD, Engine as _}; use chrono::{DateTime, Utc}; -use flate2::bufread::{GzDecoder, ZlibDecoder}; use serde_json::{Number, Value}; use uuid::Uuid; @@ -14,6 +12,7 @@ use super::diagnostic::{ DecodeError, Diagnostic, DiagnosticEvent, DiagnosticFinish, DiagnosticKey, DiagnosticPayload, DiagnosticStart, Encoding, Outcome, }; +use super::encoding::decompress; pub(super) const PREFIX: &str = "CLOUD_INIT"; @@ -244,25 +243,13 @@ fn decode_compressed(data: &str) -> Result { let compressed = STANDARD .decode(compact) .map_err(|_| DecodeError::Undecodable)?; - let mut bytes = Vec::new(); - // cloud-init labels zlib streams and line-wrapped base64 as `gz+b64`. - let remaining = if compressed.starts_with(&[0x1f, 0x8b]) { - let mut decoder = GzDecoder::new(compressed.as_slice()); - decoder - .read_to_end(&mut bytes) - .map_err(|_| DecodeError::Undecodable)?; - decoder.into_inner() + // cloud-init's report_compressed_event uses zlib.compress under `gz+b64`. + let encoding = if compressed.starts_with(&[0x1f, 0x8b]) { + Encoding::GzB64 } else { - let mut decoder = ZlibDecoder::new(compressed.as_slice()); - decoder - .read_to_end(&mut bytes) - .map_err(|_| DecodeError::Undecodable)?; - decoder.into_inner() + Encoding::ZlibB64 }; - if !remaining.is_empty() { - return Err(DecodeError::Undecodable); - } - Ok(DiagnosticPayload::Bytes(bytes)) + decompress(&compressed, &encoding) } #[cfg(test)] @@ -599,7 +586,7 @@ mod tests { #[test] fn mixed_sources_keep_first_seen_order_and_unrelated_raw_records() { - let v1 = format!("DIAG_V1|azure-init|{VM_ID}|event|test|{EVENT_ID}|2026-07-27T21:33:00.000Z|none|||0"); + let v1 = format!("DIAG|azure-init|{VM_ID}|event|test|{EVENT_ID}|2026-07-27T21:33:00.000Z|none|||0"); let entries = entries(&[ (key("event", true, Some(1)), chunk("event", 1, "b")), ("unrelated".into(), "value".into()), @@ -701,7 +688,7 @@ mod tests { } #[test] - fn compatibility_does_not_relax_v1_gzip_validation() { + fn compatibility_does_not_relax_native_gzip_validation() { assert_eq!( decode_payload(ZLIB_DATA.as_bytes(), Some(&Encoding::GzB64)), Err(DecodeError::Undecodable) diff --git a/libazureinit-kvp/src/diagnostics/diagnostic.rs b/libazureinit-kvp/src/diagnostics/diagnostic.rs index 20e5a5eb..833c4ffd 100644 --- a/libazureinit-kvp/src/diagnostics/diagnostic.rs +++ b/libazureinit-kvp/src/diagnostics/diagnostic.rs @@ -11,13 +11,19 @@ use serde::{Serialize, Serializer}; use crate::ProvisioningReport; -pub const DIAGNOSTIC_VERSION_ID: &str = "DIAG_V1"; +/// Prefix used for diagnostics emitted by +/// [`DiagnosticWriter`](crate::DiagnosticWriter). +pub const DIAGNOSTIC_VERSION_ID: &str = "DIAG"; +/// Whether a diagnostic starts or finishes an operation, or records an event. #[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize)] #[serde(rename_all = "lowercase")] pub enum Kind { + /// An operation began. Start, + /// An operation ended. Finish, + /// A standalone observation. Event, } @@ -31,16 +37,25 @@ impl fmt::Display for Kind { } } -/// Unencoded text uses `None` in `DiagnosticKey::encoding`. +/// Encoding used to compress a diagnostic payload for storage. +/// +/// Pass `None` to [`DiagnosticWriter`](crate::DiagnosticWriter) for plain text. +/// Prefer [`ZlibB64`](Self::ZlibB64) for telemetry consumed by Kusto. #[derive(Clone, Debug, PartialEq, Eq)] pub enum Encoding { + /// Zlib followed by standard base64 (`zlib+b64`). + ZlibB64, + /// Gzip followed by standard base64 (`gz+b64`). The reader also accepts + /// cloud-init's zlib data under this label. GzB64, + /// An unsupported encoding token; writers reject it. Other(String), } impl fmt::Display for Encoding { fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { f.write_str(match self { + Self::ZlibB64 => "zlib+b64", Self::GzB64 => "gz+b64", Self::Other(token) => token, }) @@ -56,10 +71,13 @@ impl Serialize for Encoding { } } +/// Reported result of an operation or observation. #[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize)] #[serde(rename_all = "lowercase")] pub enum Outcome { + /// The operation succeeded. Success, + /// The operation failed, represented as `fail` in serialized output. #[serde(rename = "fail")] Failure, } @@ -73,10 +91,16 @@ impl fmt::Display for Outcome { } } -/// Decoded bytes retain their type even when they contain valid UTF-8. +/// The text or bytes carried by a diagnostic. +/// +/// Strings convert to [`Text`](Self::Text); byte slices and vectors convert to +/// [`Bytes`](Self::Bytes). Reading a compressed payload always returns bytes, +/// even when its contents are valid text. #[derive(Clone, Debug, PartialEq, Eq)] pub enum DiagnosticPayload { + /// A UTF-8 message. Text(String), + /// Binary content, serialized to JSON as a base64-encoded object. Bytes(Vec), } @@ -123,14 +147,23 @@ impl Serialize for DiagnosticPayload { } } -/// Describes uninterpretable stored data, not a failed I/O operation. +/// Why a recognized diagnostic or report could not be decoded. +/// +/// The reader preserves the original record and attaches this error to +/// [`RawKeyValue::error`]. Failures to read the pool instead return +/// [`KvpError`](crate::KvpError). #[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize)] #[serde(rename_all = "snake_case")] pub enum DecodeError { + /// The record belongs to an unsupported diagnostic schema. UnsupportedVersion, + /// Chunk indices are missing or do not start at zero. IncompleteGroup, + /// More than one record uses the same chunk index. DuplicateChunk, + /// The payload encoding is unsupported or its contents are invalid. Undecodable, + /// Diagnostic or report metadata does not match its format. Malformed, } @@ -150,61 +183,87 @@ impl fmt::Display for DecodeError { impl std::error::Error for DecodeError {} +/// Producer, identity and timestamp shared by all diagnostic kinds. #[derive(Clone, Debug, PartialEq, Eq, Serialize)] pub struct DiagnosticKey { + /// Reporting agent, conventionally `name/VERSION`. pub agent: String, - /// Older cloud-init records do not include a VM identity. + /// VM UUID; `None` for older cloud-init records. pub vm_id: Option, + /// Operation or observation name, such as `provision:run` or `dmesg`. pub name: String, - /// Span endpoints share this ID; standalone events have their own. + /// UUID shared by an operation's start and finish; unique for a standalone event. pub event_id: String, + /// When this diagnostic was emitted, in UTC. #[serde(serialize_with = "serialize_timestamp")] pub timestamp: DateTime, + /// Stored payload encoding; `None` means plain text. #[serde(serialize_with = "serialize_encoding")] pub encoding: Option, } +/// The beginning of an operation whose finish uses the same event ID. #[derive(Clone, Debug, PartialEq, Eq, Serialize)] pub struct DiagnosticStart { + /// Producer, identity and time of the start. #[serde(flatten)] pub key: DiagnosticKey, + /// Message or artifact associated with the start. pub payload: DiagnosticPayload, } +/// The end of an operation, including its outcome and elapsed time. #[derive(Clone, Debug, PartialEq, Eq, Serialize)] pub struct DiagnosticFinish { + /// Producer, identity and time of the finish. #[serde(flatten)] pub key: DiagnosticKey, + /// Message or artifact associated with the finish. pub payload: DiagnosticPayload, + /// Reported outcome of the operation. pub result: Outcome, - #[serde(rename = "duration", serialize_with = "serialize_duration_us")] + /// Elapsed time, serialized to JSON as a number of seconds. + #[serde( + rename = "duration", + serialize_with = "serialize_duration_seconds" + )] pub duration: Duration, } +/// A standalone observation, optionally with an outcome or elapsed time. #[derive(Clone, Debug, PartialEq, Eq, Serialize)] pub struct DiagnosticEvent { + /// Producer, identity and time of the observation. #[serde(flatten)] pub key: DiagnosticKey, + /// Message or artifact captured by the event. pub payload: DiagnosticPayload, + /// Reported outcome, when applicable. #[serde(skip_serializing_if = "Option::is_none")] pub result: Option, + /// Elapsed time if measured, serialized to JSON as a number of seconds. #[serde( rename = "duration", skip_serializing_if = "Option::is_none", - serialize_with = "serialize_opt_duration_us" + serialize_with = "serialize_opt_duration_seconds" )] pub duration: Option, } +/// A diagnostic record with its metadata and decoded payload. #[derive(Clone, Debug, PartialEq, Eq, Serialize)] #[serde(tag = "kind", rename_all = "lowercase")] pub enum Diagnostic { + /// An operation began. Start(DiagnosticStart), + /// An operation ended with a reported outcome and duration. Finish(DiagnosticFinish), + /// A standalone observation. Event(DiagnosticEvent), } impl Diagnostic { + /// Returns the metadata shared by all diagnostic kinds. pub fn key(&self) -> &DiagnosticKey { match self { Self::Start(start) => &start.key, @@ -213,6 +272,7 @@ impl Diagnostic { } } + /// Returns whether this is a start, finish or standalone event. pub fn kind(&self) -> Kind { match self { Self::Start(_) => Kind::Start, @@ -221,6 +281,7 @@ impl Diagnostic { } } + /// Returns the decoded payload; encoded data has already been decompressed. pub fn payload(&self) -> &DiagnosticPayload { match self { Self::Start(start) => &start.payload, @@ -230,21 +291,29 @@ impl Diagnostic { } } +/// A stored record that was not decoded as a diagnostic or report. #[derive(Clone, Debug, PartialEq, Eq, Serialize)] pub struct RawKeyValue { + /// Original key. pub key: String, + /// Original value, without payload decoding. pub value: String, + /// Why a recognized record could not be decoded; `None` for unrelated keys. #[serde(skip_serializing_if = "Option::is_none")] pub error: Option, } +/// One item returned by [`DiagnosticReader::entries`](crate::DiagnosticReader::entries). #[derive(Clone, Debug, PartialEq, Eq, Serialize)] #[serde(tag = "type")] pub enum Entry { + /// A decoded diagnostic. #[serde(rename = "diagnostic")] Diagnostic(Diagnostic), + /// A provisioning health report. #[serde(rename = "PROVISIONING_REPORT")] Report(ProvisioningReport), + /// An unrelated or invalid record, preserved without interpretation. #[serde(rename = "raw")] Raw(RawKeyValue), } @@ -273,29 +342,27 @@ where } } -/// DIAG_V1 stores elapsed time as integer microseconds. -fn duration_micros(duration: &Duration) -> u64 { - u64::try_from(duration.as_micros()).unwrap_or(u64::MAX) -} - -fn serialize_duration_us( +fn serialize_duration_seconds( duration: &Duration, serializer: S, ) -> Result where S: Serializer, { - serializer.serialize_u64(duration_micros(duration)) + serializer.serialize_f64(duration.as_secs_f64()) } -fn serialize_opt_duration_us( +fn serialize_opt_duration_seconds( duration: &Option, serializer: S, ) -> Result where S: Serializer, { - duration.as_ref().map(duration_micros).serialize(serializer) + duration + .as_ref() + .map(Duration::as_secs_f64) + .serialize(serializer) } #[cfg(test)] @@ -305,7 +372,7 @@ mod tests { use rstest::rstest; use serde_json::{json, Value}; - const AGENT: &str = "azure-init-0.1.1"; + const AGENT: &str = "azure-init/0.1.1"; const VM_ID: &str = "3f2504e0-4f89-41d3-9a0c-0305e82c3301"; const EVENT_ID: &str = "8f3e9c4a-1b2c-4d5e-9f01-234567890abc"; const TIMESTAMP: &str = "2026-08-31T12:34:56.789Z"; @@ -347,6 +414,7 @@ mod tests { } #[rstest] + #[case(Encoding::ZlibB64, "zlib+b64")] #[case(Encoding::GzB64, "gz+b64")] #[case(Encoding::Other("zstd+b64".into()), "zstd+b64")] #[case(Encoding::Other("".into()), "")] @@ -437,23 +505,24 @@ mod tests { } #[rstest] - #[case(Outcome::Success, "success", 312)] - #[case(Outcome::Failure, "fail", 0)] - #[case(Outcome::Success, "success", u64::MAX)] - fn finish_serializes_result_and_microseconds( + #[case(Outcome::Success, "success", Duration::from_micros(312), 0.000312)] + #[case(Outcome::Failure, "fail", Duration::ZERO, 0.0)] + #[case(Outcome::Success, "success", Duration::MAX, u64::MAX as f64)] + fn finish_serializes_result_and_seconds( #[case] result: Outcome, #[case] token: &str, - #[case] duration_us: u64, + #[case] duration: Duration, + #[case] seconds: f64, ) { let entry = Entry::Diagnostic(Diagnostic::Finish(DiagnosticFinish { key: key(), payload: "finished".into(), result, - duration: Duration::from_micros(duration_us), + duration, })); let mut expected = expected_diagnostic("finish", json!("finished")); expected["result"] = json!(token); - expected["duration"] = json!(duration_us); + expected["duration"] = json!(seconds); assert_eq!(serde_json::to_value(entry).unwrap(), expected); } @@ -477,7 +546,7 @@ mod tests { expected["result"] = json!(result.to_string()); } if let Some(duration_us) = duration_us { - expected["duration"] = json!(duration_us); + expected["duration"] = json!(duration_us as f64 / 1_000_000.0); } assert_eq!(serde_json::to_value(entry).unwrap(), expected); } @@ -507,6 +576,7 @@ mod tests { #[rstest] #[case(None, "none")] + #[case(Some(Encoding::ZlibB64), "zlib+b64")] #[case(Some(Encoding::GzB64), "gz+b64")] #[case(Some(Encoding::Other("zstd+b64".into())), "zstd+b64")] fn key_always_serializes_encoding( diff --git a/libazureinit-kvp/src/diagnostics/encoding.rs b/libazureinit-kvp/src/diagnostics/encoding.rs index a0a6ec8d..868cd4d2 100644 --- a/libazureinit-kvp/src/diagnostics/encoding.rs +++ b/libazureinit-kvp/src/diagnostics/encoding.rs @@ -4,7 +4,11 @@ use std::io::{Read, Write}; use base64::{engine::general_purpose::STANDARD, Engine as _}; -use flate2::{bufread::GzDecoder, write::GzEncoder, Compression}; +use flate2::{ + bufread::{GzDecoder, ZlibDecoder}, + write::{GzEncoder, ZlibEncoder}, + Compression, +}; use super::diagnostic::{DecodeError, DiagnosticPayload, Encoding}; use crate::KvpError; @@ -14,6 +18,10 @@ pub(super) fn encode_payload( payload: DiagnosticPayload, encoding: Option<&Encoding>, ) -> Result { + let bytes = match &payload { + DiagnosticPayload::Text(text) => text.as_bytes(), + DiagnosticPayload::Bytes(bytes) => bytes.as_slice(), + }; match encoding { None => { let text = match payload { @@ -27,15 +35,17 @@ pub(super) fn encode_payload( Ok(text) } Some(Encoding::GzB64) => { - let bytes = match &payload { - DiagnosticPayload::Text(text) => text.as_bytes(), - DiagnosticPayload::Bytes(bytes) => bytes.as_slice(), - }; let mut encoder = GzEncoder::new(Vec::new(), Compression::default()); encoder.write_all(bytes)?; Ok(STANDARD.encode(encoder.finish()?)) } + Some(Encoding::ZlibB64) => { + let mut encoder = + ZlibEncoder::new(Vec::new(), Compression::default()); + encoder.write_all(bytes)?; + Ok(STANDARD.encode(encoder.finish()?)) + } Some(Encoding::Other(token)) => Err(KvpError::UnsupportedEncoding { token: token.clone(), }), @@ -51,23 +61,42 @@ pub(super) fn decode_payload( None => std::str::from_utf8(value) .map(|text| DiagnosticPayload::Text(text.to_owned())) .map_err(|_| DecodeError::Undecodable), - Some(Encoding::GzB64) => { + Some(encoding @ (Encoding::GzB64 | Encoding::ZlibB64)) => { let compressed = STANDARD .decode(value) .map_err(|_| DecodeError::Undecodable)?; - let mut decoder = GzDecoder::new(compressed.as_slice()); - let mut bytes = Vec::new(); + decompress(&compressed, encoding) + } + Some(Encoding::Other(_)) => Err(DecodeError::Undecodable), + } +} + +pub(super) fn decompress( + compressed: &[u8], + encoding: &Encoding, +) -> Result { + let mut bytes = Vec::new(); + let remaining = match encoding { + Encoding::GzB64 => { + let mut decoder = GzDecoder::new(compressed); decoder .read_to_end(&mut bytes) .map_err(|_| DecodeError::Undecodable)?; - // The buffered decoder leaves any data after the gzip member unread. - if !decoder.get_ref().is_empty() { - return Err(DecodeError::Undecodable); - } - Ok(DiagnosticPayload::Bytes(bytes)) + decoder.into_inner() } - Some(Encoding::Other(_)) => Err(DecodeError::Undecodable), + Encoding::ZlibB64 => { + let mut decoder = ZlibDecoder::new(compressed); + decoder + .read_to_end(&mut bytes) + .map_err(|_| DecodeError::Undecodable)?; + decoder.into_inner() + } + Encoding::Other(_) => return Err(DecodeError::Undecodable), + }; + if !remaining.is_empty() { + return Err(DecodeError::Undecodable); } + Ok(DiagnosticPayload::Bytes(bytes)) } #[cfg(test)] @@ -82,6 +111,7 @@ mod tests { "H4sIAAAAAAAC/0vOyS9N0c3MyyxRSMusKCktSuVi+A8AokCfWhUAAAA="; const PYTHON_FILENAME: &str = "H4sICAAAAAAC/2RtZXNnAEvOyS9N0c3MyyxRSMusKCktSuVi+A8AokCfWhUAAAA="; + const KUSTO_ZLIB: &str = "eJwLSS0uUSguKcrMS1cwNDIGACxqBQ4="; #[rstest] #[case::empty_text("", false)] @@ -147,13 +177,15 @@ mod tests { #[case("")] #[case("hello")] #[case("héllo\n\0 | 😀")] - fn gz_b64_text_decodes_to_bytes(#[case] text: &str) { - let encoding = Some(&Encoding::GzB64); - let value = encode_payload(text.into(), encoding).unwrap(); + fn compressed_text_decodes_to_bytes( + #[case] text: &str, + #[values(Encoding::GzB64, Encoding::ZlibB64)] encoding: Encoding, + ) { + let value = encode_payload(text.into(), Some(&encoding)).unwrap(); assert!(value.is_ascii()); assert!(!value.contains('\0')); assert_eq!( - decode_payload(value.as_bytes(), encoding).unwrap(), + decode_payload(value.as_bytes(), Some(&encoding)).unwrap(), DiagnosticPayload::Bytes(text.as_bytes().to_vec()) ); } @@ -163,11 +195,14 @@ mod tests { #[case(b"hello".to_vec())] #[case(vec![0xff, 0x80, 0, 0])] #[case((0u8..=255).collect())] - fn gz_b64_preserves_arbitrary_bytes(#[case] bytes: Vec) { - let encoding = Some(&Encoding::GzB64); - let value = encode_payload(bytes.clone().into(), encoding).unwrap(); + fn compression_preserves_arbitrary_bytes( + #[case] bytes: Vec, + #[values(Encoding::GzB64, Encoding::ZlibB64)] encoding: Encoding, + ) { + let value = + encode_payload(bytes.clone().into(), Some(&encoding)).unwrap(); assert_eq!( - decode_payload(value.as_bytes(), encoding).unwrap(), + decode_payload(value.as_bytes(), Some(&encoding)).unwrap(), DiagnosticPayload::Bytes(bytes) ); } @@ -177,13 +212,51 @@ mod tests { let value = encode_payload("hello".into(), Some(&Encoding::GzB64)).unwrap(); let gzip = STANDARD.decode(value).unwrap(); - assert!(gzip.starts_with(&[0x1f, 0x8b, 8])); + assert!(gzip.starts_with(&[0x1f, 0x8b, 8, 0])); assert_eq!( &gzip[gzip.len() - 8..], &[0x86, 0xa6, 0x10, 0x36, 5, 0, 0, 0] ); } + #[test] + fn zlib_b64_matches_kusto_format() { + assert_eq!( + decode_payload(KUSTO_ZLIB.as_bytes(), Some(&Encoding::ZlibB64)) + .unwrap(), + DiagnosticPayload::Bytes(b"Test string 123".to_vec()), + ); + let value = + encode_payload("Test string 123".into(), Some(&Encoding::ZlibB64)) + .unwrap(); + let compressed = STANDARD.decode(value).unwrap(); + assert_eq!(compressed[0], 0x78); + assert_eq!(compressed[1] & 0x20, 0); + } + + #[test] + fn zlib_b64_requires_a_complete_valid_zlib_stream() { + let compressed = STANDARD.decode(KUSTO_ZLIB).unwrap(); + let mut corrupt = compressed.clone(); + *corrupt.last_mut().unwrap() ^= 1; + let mut trailing = compressed.clone(); + trailing.push(0); + for bytes in [ + STANDARD.decode(PYTHON_HELLO).unwrap(), + compressed[..compressed.len() - 1].to_vec(), + corrupt, + trailing, + ] { + assert_eq!( + decode_payload( + STANDARD.encode(bytes).as_bytes(), + Some(&Encoding::ZlibB64) + ), + Err(DecodeError::Undecodable), + ); + } + } + #[rstest] #[case::empty(PYTHON_EMPTY, b"")] #[case::text(PYTHON_HELLO, b"hello")] @@ -339,6 +412,7 @@ mod tests { #[case::unknown("zstd+b64")] #[case::plain_token("none")] #[case::gzip_token("gz+b64")] + #[case::zlib_token("zlib+b64")] fn other_encoding_is_never_inferred(#[case] token: &str) { let encoding = Encoding::Other(token.into()); assert!(matches!( @@ -350,5 +424,10 @@ mod tests { decode_payload(PYTHON_HELLO.as_bytes(), Some(&encoding)), Err(DecodeError::Undecodable) ); + let compressed = STANDARD.decode(PYTHON_HELLO).unwrap(); + assert_eq!( + decompress(&compressed, &encoding), + Err(DecodeError::Undecodable) + ); } } diff --git a/libazureinit-kvp/src/diagnostics/mod.rs b/libazureinit-kvp/src/diagnostics/mod.rs index 80336126..293441d7 100644 --- a/libazureinit-kvp/src/diagnostics/mod.rs +++ b/libazureinit-kvp/src/diagnostics/mod.rs @@ -1,11 +1,7 @@ // Copyright (c) Microsoft Corporation. // Licensed under the MIT License. -//! Typed diagnostics over the raw [`crate::KvpPoolStore`]. -//! -//! [`DiagnosticWriter`] emits versioned `DIAG_V1` records. [`DiagnosticReader`] -//! reads diagnostics, provisioning reports, and raw records from a successful -//! string snapshot, including cloud-init diagnostics through a read-only bridge. +//! Diagnostic types and KVP pool access. mod cloud_init; mod diagnostic; @@ -19,12 +15,11 @@ pub use diagnostic::{ RawKeyValue, DIAGNOSTIC_VERSION_ID, }; pub use reader::DiagnosticReader; -pub use writer::{DiagnosticWriter, TimestampPrecision}; +pub use writer::{DiagnosticWriter, DurationPrecision, TimestampPrecision}; -/// Maximum number of UTF-8 value bytes stored in one diagnostic record. +/// Maximum encoded payload bytes stored in one diagnostic record. /// -/// This conservative limit keeps records readable through the Hyper-V host -/// path. Longer messages are split at UTF-8 character boundaries. +/// [`DiagnosticWriter`] splits larger payloads automatically. pub const MAX_CHUNK_BYTES: usize = 1022; /// Parse a non-empty run of ASCII digits (a chunk index or duration) as `u64`. diff --git a/libazureinit-kvp/src/diagnostics/reader.rs b/libazureinit-kvp/src/diagnostics/reader.rs index 5293e4f6..3217f451 100644 --- a/libazureinit-kvp/src/diagnostics/reader.rs +++ b/libazureinit-kvp/src/diagnostics/reader.rs @@ -19,18 +19,31 @@ use crate::{ KvpError, KvpPoolStore, ProvisioningReport, PROVISIONING_REPORT_KEY, }; +/// Reads diagnostics and provisioning reports from a KVP pool. +/// +/// Native and cloud-init diagnostics are decoded. Other or invalid records are +/// returned as [`Entry::Raw`]. Agent and VM identities come from the records, +/// so no local identity is required. #[derive(Clone, Debug)] pub struct DiagnosticReader { store: KvpPoolStore, } impl DiagnosticReader { - /// Construction performs no I/O and needs no local producer identity. + /// Creates a reader for `store`; the pool is accessed by [`entries`](Self::entries). pub fn new(store: KvpPoolStore) -> Self { Self { store } } - /// Invalid pool UTF-8 fails the snapshot; entries keep first-seen order. + /// Reads the current pool contents without modifying them. + /// + /// Returns an empty list if the pool file does not exist. Entries retain + /// pool order, and unrelated or invalid records remain [`Entry::Raw`]. + /// + /// # Errors + /// Returns [`KvpError`] if the pool cannot be read or has invalid storage + /// layout or UTF-8; no entries are returned. Per-record decoding failures + /// are returned as [`Entry::Raw`] values with a [`DecodeError`]. pub fn entries(&self) -> Result, KvpError> { Ok(decode_entries(self.store.dump()?)) } @@ -127,7 +140,7 @@ fn decode_entries(records: Vec<(String, String)>) -> Vec { ) }) } else { - decode_v1_group(&base, &mut group.chunks) + decode_diag_group(&base, &mut group.chunks) }; match decoded { Ok(diagnostic) => { @@ -155,7 +168,7 @@ fn split_chunk_key(key: &str) -> Result<(&str, u64), DecodeError> { Ok((base, parse_unsigned(index)?)) } -fn decode_v1_group( +fn decode_diag_group( base: &str, chunks: &mut [Chunk], ) -> Result { @@ -200,10 +213,11 @@ fn decode_v1_group( let duration = if duration.is_empty() { None } else { - Some(Duration::from_micros(parse_unsigned(duration)?)) + Some(parse_duration(duration)?) }; let encoding = match encoding { "none" => None, + "zlib+b64" => Some(Encoding::ZlibB64), "gz+b64" => Some(Encoding::GzB64), other => Some(Encoding::Other(other.to_owned())), }; @@ -243,6 +257,31 @@ fn decode_v1_group( } } +fn parse_duration(value: &str) -> Result { + let (seconds, fraction) = value + .split_once('.') + .map_or((value, None), |(seconds, fraction)| { + (seconds, Some(fraction)) + }); + let seconds = parse_unsigned(seconds)?; + let nanos = match fraction { + None => 0, + Some(fraction) => { + if fraction.is_empty() + || fraction.len() > 9 + || !fraction.bytes().all(|byte| byte.is_ascii_digit()) + { + return Err(DecodeError::Malformed); + } + fraction + .parse::() + .map_err(|_| DecodeError::Malformed)? + * 10u32.pow(9 - fraction.len() as u32) + } + }; + Ok(Duration::new(seconds, nanos)) +} + fn decode_chunks( chunks: &mut [Chunk], encoding: Option<&Encoding>, @@ -287,7 +326,7 @@ mod tests { use crate::store::{Handle, OsSysOps, StatInfo, SysOps}; use crate::{write_report, KvpPool, PoolMode, ReportPpsType}; - const AGENT: &str = "azure-init-0.1.1"; + const AGENT: &str = "azure-init/0.1.1"; const VM_ID: &str = "3f2504e0-4f89-41d3-9a0c-0305e82c3301"; const EVENT_ID: &str = "8f3e9c4a-1b2c-4d5e-9f01-234567890abc"; const TIMESTAMP: &str = "2026-08-31T12:34:56.789Z"; @@ -295,7 +334,7 @@ mod tests { fn key(index: u64) -> String { format!( - "DIAG_V1|{AGENT}|{VM_ID}|event|test|{EVENT_ID}|{TIMESTAMP}|none|||{index}" + "DIAG|{AGENT}|{VM_ID}|event|test|{EVENT_ID}|{TIMESTAMP}|none|||{index}" ) } @@ -498,9 +537,8 @@ mod tests { #[case("")] #[case("unrelated")] #[case("other|0")] - #[case("DIAG")] #[case("DIAG_OTHER|0")] - #[case("diag_v1|0")] + #[case("diag|0")] #[case("prefixDIAG_V2|0")] #[case("azure-init-0.1.1|1700000000|vm-abc|event|imds|id|2026-08-31T12:34:56.789Z|0")] fn unrelated_and_pre_adoption_records_remain_raw(#[case] key: &str) { @@ -583,6 +621,7 @@ mod tests { #[rstest] #[case("DIAG_V0")] + #[case("DIAG_V1")] #[case("DIAG_V2")] #[case("DIAG_V999")] #[case("DIAG_V1_extra")] @@ -603,7 +642,7 @@ mod tests { } #[test] - fn v1_event_decodes_without_local_identity() { + fn native_event_decodes_without_local_identity() { let payload = "héllo\n\"message\" | ="; let diagnostic = only_diagnostic(decode_entries(vec![(key(0), payload.into())])); @@ -628,7 +667,7 @@ mod tests { #[rstest] #[case::unmatched_start("start", "", "")] - #[case::orphan_finish("finish", "fail", "312")] + #[case::orphan_finish("finish", "fail", "0.000312")] fn isolated_span_endpoint_decodes( #[case] kind: &str, #[case] result: &str, @@ -647,7 +686,7 @@ mod tests { let start = with_field(&key(0), 3, "start"); let finish = with_field(&key(0), 3, "finish"); let finish = with_field(&finish, 8, "fail"); - let finish = with_field(&finish, 9, "312"); + let finish = with_field(&finish, 9, "0.000312"); let entries = decode_entries(vec![ (start, "starting".into()), (finish, "failed".into()), @@ -671,18 +710,18 @@ mod tests { #[case::both(Some(Outcome::Failure), Some(52))] fn event_result_and_duration_are_independent( #[case] result: Option, - #[case] duration_us: Option, + #[case] duration_secs: Option, ) { let result_token = result.map_or_else(String::new, |v| v.to_string()); let duration_token = - duration_us.map_or_else(String::new, |v| v.to_string()); + duration_secs.map_or_else(String::new, |value| value.to_string()); let key = with_field(&key(0), 8, &result_token); let key = with_field(&key, 9, &duration_token); let diagnostic = only_diagnostic(decode_entries(vec![(key, "value".into())])); assert!(matches!(&diagnostic, Diagnostic::Event(event) if event.result == result - && event.duration == duration_us.map(Duration::from_micros))); + && event.duration == duration_secs.map(Duration::from_secs))); } #[rstest] @@ -718,10 +757,18 @@ mod tests { #[case::empty_encoding(7, "")] #[case::invalid_result(8, "SUCCESS")] #[case::signed_duration(9, "+1")] + #[case::negative_duration(9, "-0.1")] + #[case::non_finite_duration(9, "NaN")] + #[case::infinite_duration(9, "inf")] + #[case::exponent_duration(9, "1e3")] + #[case::missing_seconds(9, ".5")] + #[case::missing_fraction(9, "1.")] + #[case::signed_fraction(9, "1.+2")] + #[case::excess_precision(9, "1.1234567890")] #[case::numeric_overflow(9, "18446744073709551616")] #[case::missing_chunk_index(10, "")] #[case::non_numeric_chunk_index(10, "x")] - fn malformed_v1_fields_preserve_the_original_record( + fn malformed_native_fields_preserve_the_original_record( #[case] field: usize, #[case] value: &str, ) { @@ -733,10 +780,10 @@ mod tests { } #[test] - fn v1_requires_the_exact_key_layout_and_an_index() { + fn native_format_requires_the_exact_key_layout_and_an_index() { let complete = key(0); let records = vec![ - ("DIAG_V1".into(), "value".into()), + ("DIAG".into(), "value".into()), (complete.rsplit_once('|').unwrap().0.into(), "value".into()), (format!("{complete}|1"), "value".into()), ]; @@ -753,7 +800,7 @@ mod tests { #[case::numeric_offset("2026-08-31T12:34:56.789+00:00")] #[case::lowercase("2026-08-31t12:34:56.789z")] #[case::invalid_month("2026-13-31T12:34:56.789Z")] - fn v1_rejects_non_canonical_timestamps(#[case] timestamp: &str) { + fn native_format_rejects_non_canonical_timestamps(#[case] timestamp: &str) { let records = vec![(with_field(&key(0), 6, timestamp), "value".into())]; assert_eq!( decode_entries(records.clone()), @@ -768,7 +815,9 @@ mod tests { #[case::microseconds("2026-08-31T12:34:56.789123Z")] #[case::microsecond_trailing_zeros("2026-08-31T12:34:56.789000Z")] #[case::nanoseconds("2026-08-31T12:34:56.789123456Z")] - fn v1_accepts_canonical_timestamp_precisions(#[case] timestamp: &str) { + fn native_format_accepts_canonical_timestamp_precisions( + #[case] timestamp: &str, + ) { let key = with_field(&key(0), 6, timestamp); let diagnostic = only_diagnostic(decode_entries(vec![(key, "value".into())])); @@ -1049,13 +1098,20 @@ mod tests { assert_eq!(actual, &value); } - #[test] - fn reader_accepts_full_u64_duration() { - let key = with_field(&key(0), 9, &u64::MAX.to_string()); + #[rstest] + #[case("0", Duration::ZERO)] + #[case("1.5", Duration::from_millis(1500))] + #[case("1.000000001", Duration::new(1, 1))] + #[case("18446744073709551615.999999999", Duration::MAX)] + fn reader_accepts_decimal_seconds( + #[case] seconds: &str, + #[case] expected: Duration, + ) { + let key = with_field(&key(0), 9, seconds); let diagnostic = only_diagnostic(decode_entries(vec![(key, "payload".into())])); assert!(matches!(&diagnostic, Diagnostic::Event(event) - if event.duration == Some(Duration::from_micros(u64::MAX)))); + if event.duration == Some(expected))); } #[test] diff --git a/libazureinit-kvp/src/diagnostics/writer.rs b/libazureinit-kvp/src/diagnostics/writer.rs index a6dabd09..bc3984a4 100644 --- a/libazureinit-kvp/src/diagnostics/writer.rs +++ b/libazureinit-kvp/src/diagnostics/writer.rs @@ -20,10 +20,9 @@ const MAX_NAME_BYTES: usize = 48; const MAX_UUID_BYTES: usize = 36; const MAX_TIMESTAMP_BYTES: usize = 30; const MAX_KEY_BYTES: usize = 254; -const MAX_DURATION_US: u64 = 9_999_999_999_999; const MAX_CHUNKS: usize = 1023; -/// Fractional-second precision for emitted DIAG_V1 timestamps. +/// Fractional-second precision for emitted diagnostic timestamps. #[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] pub enum TimestampPrecision { /// Whole seconds, no fractional digits. @@ -48,17 +47,93 @@ impl TimestampPrecision { } } -/// Validation precedes I/O; a storage failure can leave a partial batch. +/// Fractional-second precision for emitted durations. +/// +/// Digits below the selected precision are discarded. +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] +pub enum DurationPrecision { + /// Whole seconds, no fractional digits. + Seconds, + /// Milliseconds (three fractional digits). + Millis, + /// Microseconds (six fractional digits). The default. + #[default] + Micros, + /// Nanoseconds (nine fractional digits). + Nanos, +} + +impl DurationPrecision { + pub(crate) fn format(self, duration: Duration) -> String { + let digits = match self { + Self::Seconds => return duration.as_secs().to_string(), + Self::Millis => 3, + Self::Micros => 6, + Self::Nanos => 9, + }; + let fraction = duration.subsec_nanos() / 10u32.pow(9 - digits); + let width = digits as usize; + format!("{}.{fraction:0width$}", duration.as_secs()) + } +} + +/// Emits diagnostics for one reporting agent and VM. +/// +/// Use [`emit_event`](Self::emit_event) for a standalone observation. To record +/// an operation, call [`emit_start`](Self::emit_start) and +/// [`emit_finish`](Self::emit_finish) with the same event UUID. +/// +/// Timestamps and event UUIDs for standalone observations are generated +/// automatically. The caller supplies operation IDs, outcomes and durations. +/// Pass `None` for plain text or an [`Encoding`] for compressed text or bytes. +/// +/// Emission appends to the pool and splits large payloads automatically; it +/// never clears existing records. Invalid input writes nothing, while an I/O +/// error may leave part of a payload in the pool. See the +/// [diagnostics contract] for format details and limits. +/// +/// [diagnostics contract]: https://github.com/Azure/azure-init/blob/main/doc/diagnostics.md +/// +/// # Example +/// ```no_run +/// use std::time::Instant; +/// use libazureinit_kvp::{ +/// DiagnosticWriter, KvpPool, KvpPoolStore, Outcome, PoolMode, +/// }; +/// use uuid::Uuid; +/// +/// # fn main() -> Result<(), Box> { +/// let store = KvpPoolStore::new(KvpPool::Guest, PoolMode::Safe)?; +/// let writer = DiagnosticWriter::new( +/// store, "azure-init/0.1.1", "3f2504e0-4f89-41d3-9a0c-0305e82c3301", +/// )?; +/// let event_id = Uuid::new_v4().to_string(); +/// writer.emit_start(&event_id, "read:os-release", "reading OS information", None)?; +/// let started = Instant::now(); +/// let contents = std::fs::read_to_string("/etc/os-release")?; +/// writer.emit_finish( +/// &event_id, "read:os-release", format!("read {} bytes", contents.len()), +/// None, Outcome::Success, started.elapsed(), +/// )?; +/// # Ok(()) +/// # } +/// ``` #[derive(Clone, Debug)] pub struct DiagnosticWriter { store: KvpPoolStore, agent: String, vm_id: String, timestamp_precision: TimestampPrecision, + duration_precision: DurationPrecision, } impl DiagnosticWriter { - /// Validates producer identity without reading the pool or boot state. + /// Creates a writer with a fixed reporting agent and VM UUID. + /// + /// Versioned agents conventionally use `name/VERSION`. Invalid agent or + /// VM identifiers return [`KvpError`]; construction does not access the pool. + /// Timestamp precision defaults to milliseconds and duration precision to + /// microseconds. pub fn new( store: KvpPoolStore, agent: impl Into, @@ -73,6 +148,7 @@ impl DiagnosticWriter { agent, vm_id, timestamp_precision: TimestampPrecision::default(), + duration_precision: DurationPrecision::default(), }) } @@ -85,7 +161,19 @@ impl DiagnosticWriter { self } - /// The caller retains `event_id` to correlate the corresponding finish. + /// Sets duration precision independently of timestamp precision. + pub fn with_duration_precision( + mut self, + precision: DurationPrecision, + ) -> Self { + self.duration_precision = precision; + self + } + + /// Records the beginning of an operation. + /// + /// Pass the same event UUID and name to + /// [`emit_finish`](Self::emit_finish). pub fn emit_start( &self, event_id: &str, @@ -99,7 +187,10 @@ impl DiagnosticWriter { })) } - /// Duration is caller-measured, not inferred from the pool. + /// Records an operation's outcome and caller-measured elapsed time. + /// + /// Use the event UUID and name passed to [`emit_start`](Self::emit_start). + /// The writer does not verify that a start exists or calculate the duration. pub fn emit_finish( &self, event_id: &str, @@ -117,7 +208,9 @@ impl DiagnosticWriter { })) } - /// Each standalone event receives a fresh UUID. + /// Records a standalone observation with a generated event UUID. + /// + /// `result` and `duration` may be supplied independently. pub fn emit_event( &self, name: &str, @@ -154,13 +247,15 @@ impl DiagnosticWriter { self.store.append_multiple(prepare_records( diagnostic, self.timestamp_precision, + self.duration_precision, )?) } } fn prepare_records( diagnostic: Diagnostic, - precision: TimestampPrecision, + timestamp_precision: TimestampPrecision, + duration_precision: DurationPrecision, ) -> Result, KvpError> { let kind = diagnostic.kind(); let (key, payload, result, duration) = match diagnostic { @@ -185,23 +280,12 @@ fn prepare_records( validate_field("name", &key.name, MAX_NAME_BYTES)?; validate_uuid("event_id", &key.event_id)?; - let duration = match duration { - Some(duration) => { - let micros = - u64::try_from(duration.as_micros()).unwrap_or(u64::MAX); - if micros > MAX_DURATION_US { - return Err(KvpError::DurationTooLarge { - max_us: MAX_DURATION_US, - actual_us: micros, - }); - } - micros.to_string() - } - None => String::new(), - }; + let duration = duration.map_or_else(String::new, |duration| { + duration_precision.format(duration) + }); let timestamp = key .timestamp - .to_rfc3339_opts(precision.seconds_format(), true); + .to_rfc3339_opts(timestamp_precision.seconds_format(), true); validate_field("timestamp", ×tamp, MAX_TIMESTAMP_BYTES)?; let value = encode_payload(payload, key.encoding.as_ref())?; @@ -291,12 +375,11 @@ mod tests { use rstest::rstest; use tempfile::TempDir; - use super::super::diagnostic::Kind; use super::super::encoding::decode_payload; use crate::store::{Handle, OsSysOps, StatInfo, SysOps}; - use crate::{KvpPool, PoolMode}; + use crate::{DiagnosticReader, Entry, KvpPool, PoolMode}; - const AGENT: &str = "azure-init-0.1.1"; + const AGENT: &str = "azure-init/0.1.1"; const VM_ID: &str = "3f2504e0-4f89-41d3-9a0c-0305e82c3301"; const EVENT_ID: &str = "8f3e9c4a-1b2c-4d5e-9f01-234567890abc"; const TIMESTAMP: &str = "2026-08-31T12:34:56.789Z"; @@ -564,20 +647,21 @@ mod tests { } #[test] - fn exact_start_key_matches_v1_layout() { + fn exact_start_key_matches_diag_layout() { let records = prepare_records( Diagnostic::Start(DiagnosticStart { key: key(), payload: "starting".into(), }), TimestampPrecision::Millis, + DurationPrecision::default(), ) .unwrap(); assert_eq!( records, vec![( format!( - "DIAG_V1|{AGENT}|{VM_ID}|start|provision:run|{EVENT_ID}|{TIMESTAMP}|none|||0" + "DIAG|{AGENT}|{VM_ID}|start|provision:run|{EVENT_ID}|{TIMESTAMP}|none|||0" ), "starting".to_owned(), )] @@ -585,12 +669,13 @@ mod tests { } #[rstest] - #[case::success(Outcome::Success, "success", 312)] - #[case::failure(Outcome::Failure, "fail", 0)] - fn exact_finish_key_matches_v1_layout( + #[case::success(Outcome::Success, "success", 312_000, "0.312000")] + #[case::failure(Outcome::Failure, "fail", 0, "0.000000")] + fn exact_finish_key_matches_diag_layout( #[case] result: Outcome, #[case] token: &str, #[case] duration_us: u64, + #[case] seconds: &str, ) { let records = prepare_records( Diagnostic::Finish(DiagnosticFinish { @@ -600,24 +685,26 @@ mod tests { duration: Duration::from_micros(duration_us), }), TimestampPrecision::Millis, + DurationPrecision::default(), ) .unwrap(); assert_eq!( records[0].0, format!( - "DIAG_V1|{AGENT}|{VM_ID}|finish|provision:run|{EVENT_ID}|{TIMESTAMP}|none|{token}|{duration_us}|0" + "DIAG|{AGENT}|{VM_ID}|finish|provision:run|{EVENT_ID}|{TIMESTAMP}|none|{token}|{seconds}|0" ) ); } #[rstest] - #[case::neither(None, None)] - #[case::result_only(Some(Outcome::Success), None)] - #[case::zero_duration(None, Some(0))] - #[case::both(Some(Outcome::Failure), Some(MAX_DURATION_US))] + #[case::neither(None, None, "")] + #[case::result_only(Some(Outcome::Success), None, "")] + #[case::zero_duration(None, Some(0), "0.000000")] + #[case::both(Some(Outcome::Failure), Some(52), "0.000052")] fn event_optional_fields_are_independent( #[case] result: Option, #[case] duration_us: Option, + #[case] seconds: &str, ) { let dir = TempDir::new().unwrap(); let pool = store(&dir, PoolMode::Safe); @@ -639,10 +726,7 @@ mod tests { fields[8], result.map_or_else(String::new, |v| v.to_string()) ); - assert_eq!( - fields[9], - duration_us.map_or_else(String::new, |v| v.to_string()) - ); + assert_eq!(fields[9], seconds); } #[test] @@ -737,6 +821,49 @@ mod tests { assert!(timestamp.ends_with('Z')); } + #[rstest] + #[case(DurationPrecision::Seconds, "1", Duration::from_secs(1))] + #[case(DurationPrecision::Millis, "1.123", Duration::from_millis(1123))] + #[case( + DurationPrecision::Micros, + "1.123456", + Duration::from_micros(1_123_456) + )] + #[case( + DurationPrecision::Nanos, + "1.123456789", + Duration::new(1, 123_456_789) + )] + fn duration_precision_controls_emitted_seconds( + #[case] precision: DurationPrecision, + #[case] seconds: &str, + #[case] expected: Duration, + ) { + let dir = TempDir::new().unwrap(); + let pool = store(&dir, PoolMode::Safe); + let writer = DiagnosticWriter::new(pool.clone(), AGENT, VM_ID) + .unwrap() + .with_duration_precision(precision); + writer + .emit_event( + "timed", + "ok", + None, + None, + Some(Duration::new(1, 123_456_789)), + ) + .unwrap(); + let records = pool.dump().unwrap(); + let fields: Vec<_> = records[0].0.split('|').collect(); + assert_eq!(fields[9], seconds); + assert_eq!(fields[6].len(), 24); + let entries = DiagnosticReader::new(pool).entries().unwrap(); + assert!( + matches!(&entries[0], Entry::Diagnostic(Diagnostic::Event(event)) + if event.duration == Some(expected)) + ); + } + #[rstest] #[case(String::new(), vec![0])] #[case("x".repeat(MAX_CHUNK_BYTES), vec![MAX_CHUNK_BYTES])] @@ -773,6 +900,7 @@ mod tests { #[rstest] fn compression_precedes_safe_framing_in_both_modes( #[values(PoolMode::Safe, PoolMode::Unsafe)] mode: PoolMode, + #[values(Encoding::GzB64, Encoding::ZlibB64)] encoding: Encoding, ) { let dir = TempDir::new().unwrap(); let pool = store(&dir, mode); @@ -782,7 +910,7 @@ mod tests { .emit_event( "provision:run", bytes.clone(), - Some(Encoding::GzB64), + Some(encoding.clone()), None, None, ) @@ -794,13 +922,13 @@ mod tests { assert_eq!(key, &format!("{base}|{index}")); let fields: Vec<_> = key.split('|').collect(); assert_eq!(fields[3], "event"); - assert_eq!(fields[7], "gz+b64"); + assert_eq!(fields[7], encoding.to_string()); assert!(key.len() <= MAX_KEY_BYTES); assert!(value.len() <= MAX_CHUNK_BYTES); } let value: String = records.iter().map(|(_, v)| v.as_str()).collect(); assert_eq!( - decode_payload(value.as_bytes(), Some(&Encoding::GzB64)).unwrap(), + decode_payload(value.as_bytes(), Some(&encoding)).unwrap(), DiagnosticPayload::Bytes(bytes) ); } @@ -881,18 +1009,22 @@ mod tests { key: DiagnosticKey { agent: "a".repeat(MAX_AGENT_BYTES), name: "n".repeat(MAX_NAME_BYTES), - encoding: Some(Encoding::GzB64), + encoding: Some(Encoding::ZlibB64), ..key() }, payload: "test".into(), result: Outcome::Success, - duration: Duration::from_micros(MAX_DURATION_US), + duration: Duration::MAX, }); - let records = - prepare_records(diagnostic, TimestampPrecision::Millis).unwrap(); + let records = prepare_records( + diagnostic, + TimestampPrecision::Nanos, + DurationPrecision::Nanos, + ) + .unwrap(); let base = records[0].0.rsplit_once('|').unwrap().0; let longest = format!("{base}|1022"); - assert_eq!(longest.len(), 229); + assert_eq!(longest.len(), 251); assert!(longest.len() <= MAX_KEY_BYTES); } @@ -971,33 +1103,6 @@ mod tests { )); } - #[rstest] - fn excessive_durations_are_rejected_before_io( - #[values(MAX_DURATION_US + 1, u64::MAX)] duration_us: u64, - #[values(Kind::Finish, Kind::Event)] kind: Kind, - ) { - let dir = TempDir::new().unwrap(); - let (writer, ops) = observed_writer(&dir); - let duration = Duration::from_micros(duration_us); - let result = match kind { - Kind::Finish => writer.emit_finish( - EVENT_ID, - "test", - "bad", - None, - Outcome::Failure, - duration, - ), - _ => writer.emit_event("test", "bad", None, None, Some(duration)), - }; - assert!(matches!( - result, - Err(KvpError::DurationTooLarge { max_us: MAX_DURATION_US, actual_us }) - if actual_us == duration_us - )); - assert_eq!(ops.calls.load(Ordering::SeqCst), 0); - } - #[test] fn writer_rejects_absent_vm_identity() { let mut missing_vm = key(); @@ -1005,7 +1110,8 @@ mod tests { assert!(matches!( prepare_records( event(missing_vm, "bad".into()), - TimestampPrecision::Millis + TimestampPrecision::Millis, + DurationPrecision::default(), ), Err(KvpError::EmptyEventField { field: "vm_id" }) )); @@ -1019,7 +1125,8 @@ mod tests { assert!(matches!( prepare_records( event(expanded_year, "bad".into()), - TimestampPrecision::Nanos + TimestampPrecision::Nanos, + DurationPrecision::default(), ), Err(KvpError::EventFieldTooLong { field: "timestamp", diff --git a/libazureinit-kvp/src/error.rs b/libazureinit-kvp/src/error.rs index f0a39437..e2ed734b 100644 --- a/libazureinit-kvp/src/error.rs +++ b/libazureinit-kvp/src/error.rs @@ -4,60 +4,71 @@ use std::fmt; use std::io; -/// Errors returned by KVP storage and diagnostic writing. +/// Errors returned by KVP storage and telemetry writers. #[derive(Debug)] pub enum KvpError { /// The key was empty. EmptyKey, + /// A required diagnostic field was empty. EmptyEventField { + /// Name of the rejected field. field: &'static str, }, - /// An underlying I/O error. + /// An I/O operation failed, or stored pool data was invalid. Io(io::Error), - /// An event key field (`agent`, `vm_id`, `kind`, `name`, or `event_id`) - /// contained the `|` delimiter, which would make the formatted event - /// key ambiguous to parse back. + /// A diagnostic field contained the reserved `|` delimiter. EventFieldContainsDelimiter { + /// Name of the rejected field. field: &'static str, }, + /// A diagnostic field exceeded its UTF-8 byte limit. EventFieldTooLong { + /// Name of the rejected field. field: &'static str, + /// Maximum allowed bytes. max: usize, + /// Supplied bytes. actual: usize, }, + /// A VM or event identifier was not a valid UUID. InvalidUuid { + /// Name of the rejected field. field: &'static str, }, - DurationTooLarge { - max_us: u64, - actual_us: u64, - }, + /// An encoded payload needed too many records. TooManyChunks { + /// Maximum allowed records for one payload. max: usize, }, - /// The key contains a null byte, which is incompatible with the - /// on-disk format (null-padded fixed-width fields). + /// A key or diagnostic key field contained a NUL byte. KeyContainsNull, /// The key exceeds the store's maximum key size. KeyTooLarge { + /// Maximum allowed UTF-8 bytes. max: usize, + /// Supplied UTF-8 bytes. actual: usize, }, - /// The store already has the maximum allowed number of unique keys. + /// An insert or replacement would exceed the unique-key limit. MaxUniqueKeysExceeded { + /// Maximum allowed distinct keys. max: usize, }, + /// A byte payload could not be written as unencoded UTF-8 text. PayloadNotUtf8, + /// The requested payload encoding is not supported. UnsupportedEncoding { + /// Requested encoding name. token: String, }, /// The value exceeds the store's maximum value size. ValueTooLarge { + /// Maximum allowed UTF-8 bytes. max: usize, + /// Supplied UTF-8 bytes. actual: usize, }, - /// The value contains a null byte, which is incompatible with the - /// null-padded KVP wire format. + /// A stored value contained a NUL byte. ValueContainsNull, } @@ -77,9 +88,6 @@ impl fmt::Display for KvpError { Self::InvalidUuid { field } => { write!(f, "event key field '{field}' must be a UUID") } - Self::DurationTooLarge { max_us, actual_us } => { - write!(f, "diagnostic duration ({actual_us}us) exceeds maximum ({max_us}us)") - } Self::TooManyChunks { max } => { write!(f, "diagnostic chunk count exceeds maximum ({max})") } @@ -146,10 +154,6 @@ mod tests { KvpError::InvalidUuid { field: "vm_id" }, "event key field 'vm_id' must be a UUID" )] - #[case( - KvpError::DurationTooLarge { max_us: 9_999_999_999_999, actual_us: 10_000_000_000_000 }, - "diagnostic duration (10000000000000us) exceeds maximum (9999999999999us)" - )] #[case( KvpError::TooManyChunks { max: 1023 }, "diagnostic chunk count exceeds maximum (1023)" diff --git a/libazureinit-kvp/src/lib.rs b/libazureinit-kvp/src/lib.rs index f5028008..32c040d5 100644 --- a/libazureinit-kvp/src/lib.rs +++ b/libazureinit-kvp/src/lib.rs @@ -1,47 +1,65 @@ // Copyright (c) Microsoft Corporation. // Licensed under the MIT License. -//! `libazureinit-kvp` provides a unified KVP pool file store for -//! Hyper-V/Azure guests. +//! Read and write Hyper-V KVP metadata on Linux guests. //! -//! - [`KvpPoolStore`]: KVP pool file store with -//! [`PoolMode`]-based policy. -//! - [`ProvisioningReport`]: structured provisioning health report that -//! is persisted as the single `PROVISIONING_REPORT` record with -//! [`write_report`]. -//! - [`DiagnosticWriter`]: typed writer for versioned diagnostics. -//! - [`DiagnosticReader`]: reader for diagnostics, provisioning reports, and -//! raw records, including a read-only cloud-init compatibility bridge. +//! KVP (Key-Value Pair) data exchange lets a VM share small text records with +//! its host without network connectivity. This library operates on the local +//! pool files, normally in `/var/lib/hyperv`. The Hyper-V daemon and kernel +//! handle transport; writing a file does not itself notify the host. //! -//! # Diagnostics +//! # Choose an API +//! - [`DiagnosticWriter`] emits operations and observations, handling +//! timestamps, compression and splitting long payloads into records. +//! - [`DiagnosticReader`] reads native and cloud-init diagnostics, provisioning +//! reports, and unrecognized records as [`Entry`] values. +//! - [`ProvisioningReport`] and [`write_report`] publish one replaceable +//! provisioning result, rather than a history of diagnostic events. +//! - [`KvpPoolStore`] reads and writes ordinary key/value records. Use +//! [`insert`](KvpPoolStore::insert) to update a key, or +//! [`append`](KvpPoolStore::append) to preserve duplicates. //! -//! The reader preserves first-seen pool order. Unknown or invalid records -//! remain [`Entry::Raw`] within a successful snapshot; a failed snapshot, -//! including invalid physical UTF-8, returns an error without entries. -//! The CLI's `dump --parse` renders those entries in the same pool order. +//! Use [`KvpPool::Guest`] for guest-produced telemetry and [`PoolMode::Safe`] +//! for host-readable writes. [`KvpPool::AutoExternal`] contains host-provided +//! data. The selected directory must exist and permit the requested file I/O; +//! [`KvpPoolStore::new_in`] selects an alternate directory. +//! +//! # Emit Telemetry //! //! ```no_run //! use libazureinit_kvp::{ -//! DiagnosticReader, DiagnosticWriter, KvpPool, KvpPoolStore, Outcome, -//! PoolMode, +//! DiagnosticWriter, Encoding, KvpPool, KvpPoolStore, Outcome, PoolMode, //! }; //! //! # fn main() -> Result<(), Box> { //! let store = KvpPoolStore::new(KvpPool::Guest, PoolMode::Safe)?; //! let writer = DiagnosticWriter::new( -//! store.clone(), -//! "azure-init", +//! store, +//! "azure-init/0.1.1", //! "3f2504e0-4f89-41d3-9a0c-0305e82c3301", //! )?; //! writer.emit_event( //! "imds", "metadata retrieved", None, Some(Outcome::Success), None, //! )?; -//! -//! let entries = DiagnosticReader::new(store).entries()?; -//! println!("{}", serde_json::to_string(&entries)?); +//! writer.emit_event( +//! "os:release", std::fs::read("/etc/os-release")?, +//! Some(Encoding::ZlibB64), None, None, +//! )?; //! # Ok(()) //! # } //! ``` +//! +//! See [`DiagnosticWriter`] for recording operations, [`DiagnosticReader`] for +//! consuming telemetry, and [`ProvisioningReport`] for reporting provisioning +//! success or failure. +//! +//! # Format References +//! The [KVP contract] describes pool files and Hyper-V interfaces. The +//! [diagnostics contract] describes record fields and encodings for consumers +//! that read telemetry independently of this library. +//! +//! [KVP contract]: https://github.com/Azure/azure-init/blob/main/doc/kvp.md +//! [diagnostics contract]: https://github.com/Azure/azure-init/blob/main/doc/diagnostics.md mod cli; mod diagnostics; @@ -54,8 +72,8 @@ pub use cli::run; pub use diagnostics::{ DecodeError, Diagnostic, DiagnosticEvent, DiagnosticFinish, DiagnosticKey, DiagnosticPayload, DiagnosticReader, DiagnosticStart, DiagnosticWriter, - Encoding, Entry, Kind, Outcome, RawKeyValue, TimestampPrecision, - DIAGNOSTIC_VERSION_ID, MAX_CHUNK_BYTES, + DurationPrecision, Encoding, Entry, Kind, Outcome, RawKeyValue, + TimestampPrecision, DIAGNOSTIC_VERSION_ID, MAX_CHUNK_BYTES, }; pub use error::KvpError; pub use report::{ diff --git a/libazureinit-kvp/src/report.rs b/libazureinit-kvp/src/report.rs index cb3f9a3a..3690c3b7 100644 --- a/libazureinit-kvp/src/report.rs +++ b/libazureinit-kvp/src/report.rs @@ -1,15 +1,7 @@ // Copyright (c) Microsoft Corporation. // Licensed under the MIT License. -//! Structured provisioning report abstraction layered over the raw -//! [`KvpPoolStore`] key/value API. -//! -//! [`ProvisioningReport`] is a strongly-typed representation of a -//! provisioning health report. Instead of building ad-hoc key/value -//! strings at the call site, callers construct a report and persist it -//! with [`write_report`], which serializes it into the single -//! pipe-delimited `PROVISIONING_REPORT` KVP record that the Azure/Hyper-V -//! host parses. +//! Provisioning report creation and storage. use std::str::FromStr; @@ -17,14 +9,9 @@ use chrono::{DateTime, Utc}; use crate::{DecodeError, KvpError, KvpPoolStore}; -/// KVP key under which the encoded provisioning health report is stored. -/// -/// The Azure/Hyper-V host parses this single key; its value is the -/// pipe-delimited `key=value|key=value|...` report produced by -/// [`write_report`]. +/// Key used by [`write_report`] to store the provisioning result. pub const PROVISIONING_REPORT_KEY: &str = "PROVISIONING_REPORT"; -/// The current time formatted as an RFC 3339 string. fn now_rfc3339() -> String { Utc::now().to_rfc3339() } @@ -55,22 +42,19 @@ impl std::fmt::Display for ReportResult { } } -/// Pre-provisioning (PPS) type reported in the `pps_type` field. -/// -/// Mirrors the values cloud-init reports for the platform's -/// `PreprovisionedVMType` / IMDS `ppsType`. +/// Pre-provisioning state included in a provisioning report. #[derive(Clone, Copy, Debug, PartialEq, Eq, serde::Serialize)] pub enum ReportPpsType { - /// Not pre-provisioned (`None`). + /// Not pre-provisioned. None, - /// Pre-provisioned OS disk (`PreprovisionedOSDisk`). + /// A pre-provisioned OS disk. #[serde(rename = "PreprovisionedOSDisk")] OsDisk, - /// Running pre-provisioning (`Running`). + /// Pre-provisioning is running. Running, - /// Savable pre-provisioning (`Savable`). + /// Pre-provisioning can be saved. Savable, - /// Unknown pre-provisioning type (`Unknown`). + /// The pre-provisioning state is unknown. Unknown, } @@ -104,11 +88,16 @@ impl std::fmt::Display for ReportPpsType { } } -/// A strongly-typed provisioning health report. +/// A provisioning result for host telemetry. +/// +/// [`success`](Self::success) and [`failure`](Self::failure) capture the current +/// time. Add optional context with the builder methods, then call +/// [`write_report`] to persist it. Existing stored reports can be parsed with +/// [`str::parse`]; parsing preserves their timestamps. +/// +/// See the [provisioning report contract] for the stored format. /// -/// Construct one with [`ProvisioningReport::success`] or -/// [`ProvisioningReport::failure`], optionally attach extra context with -/// the builder methods, then persist it with [`write_report`]. +/// [provisioning report contract]: https://github.com/Azure/azure-init/blob/main/doc/diagnostics.md#provisioning-reports /// /// # Example /// ```no_run @@ -133,30 +122,29 @@ impl std::fmt::Display for ReportPpsType { /// ``` #[derive(Clone, Debug, PartialEq, Eq, serde::Serialize)] pub struct ProvisioningReport { - /// Provisioning outcome (`result` field). + /// Provisioning outcome. result: ReportResult, - /// Reporting agent identifier (`agent` field). + /// Reporting agent identifier. agent: String, - /// Virtual machine identifier (`vm_id` field). + /// Virtual machine identifier. vm_id: String, - /// Report timestamp (`timestamp` field), set to the current time - /// (RFC 3339) when the report is constructed. + /// RFC 3339 timestamp captured when the report is constructed. timestamp: String, - /// Pre-provisioning type (`pps_type` field). + /// Pre-provisioning state. pps_type: ReportPpsType, - /// Failure reason (`reason` field). Present for error reports. + /// Failure reason, present only for error reports. #[serde(skip_serializing_if = "Option::is_none")] reason: Option, - /// Documentation URL (`documentation_url` field), if applicable. + /// Help URL, if applicable. #[serde(skip_serializing_if = "Option::is_none")] documentation_url: Option, - /// Additional ordered key/value context (e.g. supporting data). + /// Additional ordered key/value context. #[serde(skip_serializing_if = "Vec::is_empty")] extra: Vec<(String, String)>, } impl ProvisioningReport { - /// Create a successful provisioning report. + /// Creates a successful provisioning report. pub fn success( agent: impl Into, vm_id: impl Into, @@ -174,7 +162,7 @@ impl ProvisioningReport { } } - /// Create a failed provisioning report with a failure reason. + /// Creates a failed provisioning report with a reason. pub fn failure( agent: impl Into, vm_id: impl Into, @@ -193,14 +181,16 @@ impl ProvisioningReport { } } - /// Attach a documentation URL. + /// Sets a help URL, included in the stored value for failure reports only. pub fn with_documentation_url(mut self, url: impl Into) -> Self { self.documentation_url = Some(url.into()); self } - /// Append an additional key/value pair. Extras are emitted in the - /// order they were added. + /// Adds context to the report. + /// + /// Entries retain insertion order and duplicate keys. Do not use standard + /// report field names such as `result` or `timestamp`. pub fn with_extra( mut self, key: impl Into, @@ -214,7 +204,7 @@ impl ProvisioningReport { impl FromStr for ProvisioningReport { type Err = DecodeError; - /// Parses one report without generating a timestamp or accessing storage. + /// Parses a stored report without changing its timestamp. fn from_str(value: &str) -> Result { let value = value .strip_suffix("\r\n") @@ -333,13 +323,7 @@ fn validate_report_quoting(value: &str) -> Result<(), DecodeError> { } impl ProvisioningReport { - /// Encode the report as a single pipe-delimited `key=value` string. - /// - /// - Success: `result`, `agent`, `pps_type`, `vm_id`, `timestamp`, - /// then any extras in insertion order. - /// - Failure: `result`, `reason`, `agent`, extras in insertion - /// order, `pps_type`, `vm_id`, `timestamp`, then - /// `documentation_url` (if any). + /// Encodes the report for storage. pub(crate) fn encode(&self) -> String { let mut data = Vec::with_capacity(7 + self.extra.len()); @@ -388,12 +372,12 @@ impl ProvisioningReport { } } -/// Persist a report to the KVP store under [`PROVISIONING_REPORT_KEY`]. +/// Stores a provisioning result, replacing any existing report in the pool. +/// +/// A report must fit in one value under the store's configured size policy. /// -/// The report is encoded into a single pipe-delimited value and written -/// with [`KvpPoolStore::insert`] (upsert / last-write-wins), so it -/// overrides any existing `PROVISIONING_REPORT` record rather than -/// accumulating duplicates. +/// # Errors +/// Returns validation and I/O errors from the store. pub fn write_report( store: &KvpPoolStore, report: &ProvisioningReport, diff --git a/libazureinit-kvp/src/store.rs b/libazureinit-kvp/src/store.rs index 05404027..de06c40e 100644 --- a/libazureinit-kvp/src/store.rs +++ b/libazureinit-kvp/src/store.rs @@ -1,26 +1,7 @@ // Copyright (c) Microsoft Corporation. // Licensed under the MIT License. -//! Unified KVP pool file backend for Hyper-V and Azure guests. -//! -//! All sizes are UTF-8 byte counts — the on-disk pool stores keys and -//! values as zero-padded UTF-8. The Windows-side Hyper-V wire spec is -//! UTF-16 char-based (256 / 1024), but the Linux `hv_kvp_daemon` -//! exchanges UTF-8 bytes with the kernel. -//! -//! Fixed-width record format (matches Linux kernel -//! `HV_KVP_EXCHANGE_MAX_*`): -//! - key field: 512 bytes -//! - value field: 2048 bytes -//! - record size: 2560 bytes -//! -//! [`PoolMode`] selects which size limits are enforced on writes: -//! - [`Safe`](PoolMode::Safe): key <= 254 bytes, -//! value <= 1022 bytes (recommended for Linux kernel compatibility) -//! - [`Unsafe`](PoolMode::Unsafe): key <= 512 bytes, value <= 2048 bytes -//! -//! ## Reference -//! - [Hyper-V Data Exchange Service (KVP)](https://learn.microsoft.com/en-us/virtualization/hyper-v-on-windows/reference/integration-services#hyper-v-data-exchange-service-kvp) +//! KVP pool storage and file locking. use std::collections::{HashMap, HashSet}; use std::fs::{File, OpenOptions}; @@ -40,12 +21,15 @@ const WIRE_MAX_VALUE_BYTES: usize = 2048; const SAFE_MAX_KEY_BYTES: usize = 254; const SAFE_MAX_VALUE_BYTES: usize = 1022; -/// Maximum number of unique keys allowed in the pool. +/// Unique-key limit for insert and load operations. const MAX_UNIQUE_KEYS: usize = 1024; const RECORD_SIZE: usize = WIRE_MAX_KEY_BYTES + WIRE_MAX_VALUE_BYTES; -/// Hyper-V KVP pool indices. +/// Selects the Hyper-V pool file to access. +/// +/// Use [`Guest`](Self::Guest) for guest-produced diagnostics and reports. +/// Other pools are generally populated by the host or integration daemon. #[derive(Clone, Copy, Debug, PartialEq, Eq)] #[repr(u8)] pub enum KvpPool { @@ -53,11 +37,11 @@ pub enum KvpPool { External = 0, /// Guest-to-host data; cloud-init / azure-init write here (`.kvp_pool_1`). Guest = 1, - /// Guest intrinsics generated by the daemon (`.kvp_pool_2`). + /// Guest details generated by the daemon, not read from `.kvp_pool_2`. Auto = 2, /// Host-originated data describing the host (`.kvp_pool_3`). AutoExternal = 3, - /// Undocumented; no pool file exists (`.kvp_pool_4`). + /// Internal pool (`.kvp_pool_4`); not intended for application telemetry. AutoInternal = 4, } @@ -84,22 +68,17 @@ impl KvpPool { } } -/// Policy mode controlling key/value size limits for writes. +/// Size policy for keys and values written through [`KvpPoolStore`]. /// -/// All limits are **UTF-8 byte counts** measured against the on-disk -/// pool format. The Windows-side Hyper-V wire spec expresses limits -/// in UTF-16 code units (256 / 1024), which equal 512 / 2048 bytes -/// when stored as UTF-8 by `hv_kvp_daemon` on Linux. +/// Limits count UTF-8 bytes, not characters or UTF-16 code units. This policy +/// does not restrict reading records written with a different mode. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub enum PoolMode { - /// Conservative limits for Linux kernel compatibility: - /// key <= 254 UTF-8 bytes, value <= 1022 UTF-8 bytes — 2 bytes - /// under the kernel `HV_KVP_EXCHANGE_MAX_*` maximums to leave room - /// for a NUL terminator on either side of the boundary. + /// Host-compatible limits: keys up to 254 bytes and values up to 1,022 bytes. + /// Recommended for data exchanged with the host. Safe, - /// Full Hyper-V wire-format limits expressed in UTF-8 bytes: - /// key <= 512 bytes (== 256 UTF-16 code units), - /// value <= 2048 bytes (== 1024 UTF-16 code units). + /// Full pool-field widths: keys up to 512 bytes and values up to 2,048 bytes. + /// Such records may be truncated during transport to the host. Unsafe, } @@ -119,7 +98,23 @@ impl PoolMode { } } -/// Unified KVP pool file store. +/// Reads and writes UTF-8 key/value records in one Linux Hyper-V pool file. +/// +/// Constructors select a path and size policy without performing I/O. Each +/// operation opens and locks the file as needed. The parent directory must +/// exist before writing. +/// +/// [`insert`](Self::insert) updates a key, while [`append`](Self::append) +/// preserves duplicates. Use [`dump`](Self::dump) for every physical record +/// in pool order, or [`entries`](Self::entries) for the last value of each key. +/// [`load`](Self::load) replaces the entire pool. +/// +/// Records persist across boots until explicitly cleared. Writes reject empty +/// keys and NUL bytes. An I/O failure during a write may leave partial changes. +/// +/// See the [KVP contract] for record layout and transport details. +/// +/// [KVP contract]: https://github.com/Azure/azure-init/blob/main/doc/kvp.md #[derive(Clone, Debug)] pub struct KvpPoolStore { pool: KvpPool, @@ -129,12 +124,10 @@ pub struct KvpPoolStore { } impl KvpPoolStore { - /// Append one key-value pair to the tail of the pool without - /// checking for an existing key. + /// Appends a key/value pair, preserving existing records and duplicates. /// - /// This preserves any existing records, including duplicate keys, - /// and does not enforce the `MAX_UNIQUE_KEYS` cap. Use - /// [`insert`](Self::insert) when callers need upsert semantics. + /// Record count is not capped. Use [`insert`](Self::insert) to replace a + /// key's existing values. pub fn append(&self, key: &str, value: &str) -> Result<(), KvpError> { validate_key(key, self.mode.max_key_size())?; validate_value(value, self.mode.max_value_size())?; @@ -148,18 +141,12 @@ impl KvpPoolStore { Ok(()) } - /// Append records to the tail of the pool under one exclusive - /// lock, preserving order and preventing cooperating writers from - /// interleaving with the batch. + /// Appends records in iteration order under one exclusive lock. /// - /// Existing records are kept and duplicate keys are preserved. - /// Like [`append`](Self::append), this does not enforce the - /// `MAX_UNIQUE_KEYS` cap; use [`load`](Self::load) when the - /// caller is replacing the entire pool and wants the unique-key cap - /// enforced. Validation happens before the file is opened; empty - /// input is a no-op and does not create the pool file. If an I/O - /// error occurs after writing starts, the file may contain a - /// partial batch. + /// Existing records and duplicates are kept, and record count is not + /// capped. Empty input is a no-op. Inputs are validated before writing. + /// Cooperating writers cannot interleave with the batch; an I/O error may + /// leave only part of it written. pub fn append_multiple(&self, records: I) -> Result<(), KvpError> where I: IntoIterator, @@ -185,7 +172,7 @@ impl KvpPoolStore { self.write_records_locked(&records, false) } - /// Remove all entries from the store. + /// Removes all records. A missing pool file is left absent. pub fn clear(&self) -> Result<(), KvpError> { { let mut handle = match self.open_for_read_write() { @@ -200,7 +187,9 @@ impl KvpPoolStore { Ok(()) } - /// Clear the pool file if it contains data from a previous boot. + /// Clears the pool if its modification time is at or before system boot. + /// + /// The check and removal hold an exclusive lock. A missing file is a no-op. pub fn clear_if_stale(&self) -> Result<(), KvpError> { { let mut handle = match self.open_for_read_write() { @@ -218,12 +207,12 @@ impl KvpPoolStore { Ok(()) } - /// Remove all records matching the key. Returns `true` if at least - /// one record was present. + /// Removes all records matching `key`, returning whether any were found. /// /// Rejects empty keys and keys containing null bytes; key size is /// not capped, so a [`Safe`](PoolMode::Safe)-mode store can delete /// keys written in [`Unsafe`](PoolMode::Unsafe) mode. + /// Removal may change the order of remaining records. pub fn delete(&self, key: &str) -> Result { validate_key(key, usize::MAX)?; @@ -252,16 +241,13 @@ impl KvpPoolStore { Ok(found) } - /// Delete all records whose key appears in `keys`, under one - /// exclusive lock. + /// Removes all records whose key appears in `keys` under one exclusive lock. /// - /// Returns the number of records removed, counting duplicate - /// on-disk keys separately. Like [`delete`](Self::delete), key - /// size is not capped. Empty input is a no-op; a missing file - /// returns `Ok(0)`. + /// Returns the number of records removed, counting duplicates separately. + /// Like [`delete`](Self::delete), key size is not capped. Empty input is a + /// no-op; a missing file returns `Ok(0)`. /// - /// Removal swaps deleted records with the tail, so remaining - /// record order is not preserved. + /// Remaining records may be reordered. pub fn delete_multiple(&self, keys: I) -> Result where I: IntoIterator, @@ -303,8 +289,10 @@ impl KvpPoolStore { Ok(removed) } - /// Return all key-value records in on-disk order, including - /// duplicates from [`append`](Self::append) calls. + /// Returns every key/value record in pool order, including duplicates. + /// + /// A missing file returns an empty list. Invalid record layout or UTF-8, + /// or a file-access error, fails the call without returning partial results. pub fn dump(&self) -> Result, KvpError> { let records = { let mut iter = match self.iter() { @@ -324,7 +312,10 @@ impl KvpPoolStore { Ok(records) } - /// Return all key-value pairs (deduplicated, last-write-wins). + /// Returns one value per key, using the last record when keys repeat. + /// + /// A missing file returns an empty map. Use [`dump`](Self::dump) to retain + /// duplicate records and their order. pub fn entries(&self) -> Result, KvpError> { let map = { let mut iter = match self.iter() { @@ -345,7 +336,10 @@ impl KvpPoolStore { Ok(map) } - /// Insert a new key-value pair or update an existing key's value. + /// Inserts a key/value pair or replaces an existing key's value. + /// + /// Existing duplicates of the key are collapsed into one record. Adding + /// a new key fails if the pool already contains 1,024 distinct keys. pub fn insert(&self, key: &str, value: &str) -> Result<(), KvpError> { validate_key(key, self.mode.max_key_size())?; validate_value(value, self.mode.max_value_size())?; @@ -387,12 +381,13 @@ impl KvpPoolStore { Ok(()) } - /// Return whether the store is empty. + /// Returns whether the store is empty. pub fn is_empty(&self) -> Result { Ok(self.len()? == 0) } - /// Whether the store's data is stale (e.g. predates current boot). + /// Returns whether the pool's modification time is at or before system boot. + /// A missing file is not stale. pub fn is_stale(&self) -> Result { let metadata = match self.ops.path_metadata(&self.path) { Ok(m) => m, @@ -425,12 +420,7 @@ impl KvpPoolStore { KvpPoolIter::new(handle, false) } - /// Return the number of records in the store. - /// - /// This counts on-disk records, not unique keys. If - /// [`append`](Self::append) was used to write duplicate keys, this - /// may exceed the number of unique keys returned by - /// [`entries`](Self::entries). + /// Returns the number of physical records, including duplicate keys. pub fn len(&self) -> Result { match self.ops.path_metadata(&self.path) { Ok(m) => record_count_from_len(m.len as usize).map_err(Into::into), @@ -454,13 +444,15 @@ impl KvpPoolStore { self.mode } - /// Open using the default Hyper-V directory - /// ([`/var/lib/hyperv`](KvpPool::default_dir)). + /// Selects a pool in the default `/var/lib/hyperv` directory without I/O. pub fn new(pool: KvpPool, mode: PoolMode) -> Result { Self::new_in(pool, KvpPool::default_dir(), mode) } - /// Open using a custom directory (file name derived from `pool`). + /// Selects a pool in `dir` without opening or creating it. + /// + /// The file name is derived from [`KvpPool::file_name`]. The directory must + /// exist before performing a write that creates the file. pub fn new_in( pool: KvpPool, dir: impl AsRef, @@ -522,6 +514,7 @@ impl KvpPoolStore { iter.flush()?; Ok(()) } + /// Returns the configured pool-file path. pub fn path(&self) -> &Path { &self.path } @@ -531,23 +524,16 @@ impl KvpPoolStore { self.pool } - /// Replace the entire pool with the given pairs, in iteration - /// order, under one exclusive lock. + /// Replaces the entire pool in iteration order under one exclusive lock. /// - /// This is the inverse of [`dump`](Self::dump): duplicate keys are - /// preserved exactly as provided, but the number of unique keys is - /// capped at `MAX_UNIQUE_KEYS`. Existing records are discarded; - /// empty input clears the pool. Use - /// [`append_multiple`](Self::append_multiple) when callers need to - /// extend the pool instead. + /// Duplicates are preserved, with at most 1,024 distinct keys. Empty input + /// clears the pool. Use [`append_multiple`](Self::append_multiple) to keep + /// existing records instead. /// - /// Validation happens before locking, so a rejected batch never - /// blocks other writers. The file is truncated and rewritten - /// under one exclusive lock; an I/O error mid-write may leave - /// the file partially written. Every successful call updates - /// mtime (relevant to [`is_stale`](Self::is_stale)). Rejects - /// malformed pool files (size not a multiple of the record - /// size); call [`clear`](Self::clear) first to recover. + /// Invalid input leaves the pool unchanged; a storage error after truncation + /// may leave a partial replacement. A successful call updates modification + /// time. An existing file with invalid record framing must be explicitly + /// cleared with [`clear`](Self::clear) before it can be replaced. pub fn load(&self, records: I) -> Result<(), KvpError> where I: IntoIterator, @@ -576,10 +562,7 @@ impl KvpPoolStore { self.write_records_locked(&records, true) } - /// Read the value for a key. Returns `Ok(None)` when absent. - /// - /// If multiple records share the same key (e.g. via - /// [`append`](Self::append)), the last (most recent) match wins. + /// Returns the last value for `key`, or `None` when absent. /// /// Rejects empty keys and keys containing null bytes; key size is /// not capped, so a [`Safe`](PoolMode::Safe)-mode store can read diff --git a/libazureinit-kvp/src/vm_id.rs b/libazureinit-kvp/src/vm_id.rs index 7b1e5a85..88aafac1 100644 --- a/libazureinit-kvp/src/vm_id.rs +++ b/libazureinit-kvp/src/vm_id.rs @@ -1,22 +1,18 @@ // Copyright (c) Microsoft Corporation. // Licensed under the MIT License. -//! Self-contained VM ID lookup used to auto-populate provisioning reports. +//! Current VM ID lookup for command-line defaults. //! -//! The VM ID is read from `/sys/class/dmi/id/product_uuid` and, on Gen1 VMs, -//! the first three UUID fields are byte-swapped from big-endian to native endianness. +//! Reads `/sys/class/dmi/id/product_uuid` and adjusts Gen1 UUID byte order. use std::fs; use std::path::Path; use uuid::Uuid; -/// Retrieves the current VM ID by reading `/sys/class/dmi/id/product_uuid` -/// and byte-swapping the result if the VM is Gen1. +/// Returns the current VM ID, adjusting UUID byte order on Gen1 VMs. /// -/// # Returns -/// - `Some(String)` containing the VM ID if retrieval is successful. -/// - `None` if the file is missing, empty, or cannot be read. +/// Returns `None` if the DMI file is missing, unreadable or empty. pub fn get_vm_id() -> Option { private_get_vm_id(None, None, None) } @@ -58,8 +54,7 @@ fn private_get_vm_id( } } -/// Determines whether the VM is Gen1 (i.e. not UEFI/Gen2) based on EFI -/// detection. Returns `true` when neither EFI path exists. +/// Returns `true` when neither EFI path exists, identifying a Gen1 VM. fn is_vm_gen1( sysfs_efi_path: Option<&str>, dev_efi_path: Option<&str>, diff --git a/libazureinit-kvp/tests/cli.rs b/libazureinit-kvp/tests/cli.rs index 24c67867..a5642641 100644 --- a/libazureinit-kvp/tests/cli.rs +++ b/libazureinit-kvp/tests/cli.rs @@ -317,7 +317,7 @@ fn report_failure_rejects_invalid_supporting_data() { fn parsed_dump_reassembles_and_filters_without_dropping_other_entries() { let dir = TempDir::new().unwrap(); let base = format!( - "DIAG_V1|agent|{VM_ID}|event|a:b|e5f01809-a7a3-4279-aa64-1f18e21eda6e|2026-08-31T00:00:00.000Z|none||" + "DIAG|agent|{VM_ID}|event|a:b|e5f01809-a7a3-4279-aa64-1f18e21eda6e|2026-08-31T00:00:00.000Z|none||" ); for (key, value) in [ (format!("{base}|1"), "two"), @@ -385,7 +385,9 @@ fn parsed_dump_preserves_reader_pool_order() { let store = store_at(&dir); let event_id = "e5f01809-a7a3-4279-aa64-1f18e21eda6e"; let key = |name: &str, timestamp: &str| { - format!("DIAG_V1|agent|{VM_ID}|event|{name}|{event_id}|{timestamp}|none|||0") + format!( + "DIAG|agent|{VM_ID}|event|{name}|{event_id}|{timestamp}|none|||0" + ) }; // Timestamps are deliberately out of order to prove the CLI does not sort. let records = [ @@ -453,7 +455,7 @@ fn parsed_dump_normalizes_cloud_init_in_json_and_text() { "name": "modules-final/config-scripts_user", "vm_id": VM_ID, "event_id": "e5f01809-a7a3-4279-aa64-1f18e21eda6e", "timestamp": "2026-07-27T21:33:24.339006Z", "encoding": "none", - "result": "success", "duration": 500000, "payload": "scripts ran", + "result": "success", "duration": 0.5, "payload": "scripts ran", }) ); assert_eq!(entries[1]["kind"], "start"); @@ -469,7 +471,7 @@ fn parsed_dump_normalizes_cloud_init_in_json_and_text() { assert!(out.contains("vm_id=0e5e179d-5341-478b-8456-fbb90621bdf8")); assert!(out.contains("result=success")); assert!(out.contains("timestamp=2026-07-27T21:33:24.339006Z")); - assert!(out.contains("duration=500000us")); + assert!(out.contains("duration=0.5s")); assert!(out.contains("payload=scripts ran")); let start = out.lines().nth(1).unwrap(); assert!(start.contains("diagnostic kind=start")); @@ -477,8 +479,13 @@ fn parsed_dump_normalizes_cloud_init_in_json_and_text() { assert!(!start.contains("duration=")); } -#[test] -fn parsed_dump_renders_bytes_reports_and_raw_errors() { +#[rstest] +#[case(Encoding::GzB64, "gz+b64")] +#[case(Encoding::ZlibB64, "zlib+b64")] +fn parsed_dump_renders_bytes_reports_and_raw_errors( + #[case] encoding: Encoding, + #[case] token: &str, +) { let dir = TempDir::new().unwrap(); let store = store_at(&dir); DiagnosticWriter::new(store.clone(), "agent", VM_ID) @@ -486,13 +493,13 @@ fn parsed_dump_renders_bytes_reports_and_raw_errors() { .emit_event( "artifact", vec![0, 255], - Some(Encoding::GzB64), + Some(encoding), Some(Outcome::Failure), Some(Duration::from_micros(7)), ) .unwrap(); store.append("note", "raw value").unwrap(); - store.append("DIAG_V1|bad", "junk").unwrap(); + store.append("DIAG|bad", "junk").unwrap(); assert_success(kvp(&with_dir( &dir, &["report-failure", "--vm-id", VM_ID, "--reason", "bad input"], @@ -501,10 +508,11 @@ fn parsed_dump_renders_bytes_reports_and_raw_errors() { assert_success(kvp(&with_dir(&dir, &["dump", "--parse", "--text"]))); let lines: Vec<_> = out.lines().collect(); assert_eq!(lines.len(), 4); - assert!(lines[0] - .contains("encoding=gz+b64 result=fail duration=7us payload_b64=AP8=")); + assert!(lines[0].contains(&format!( + "encoding={token} result=fail duration=0.000007s payload_b64=AP8=" + ))); assert_eq!(lines[1], "raw key=note value=raw value"); - assert_eq!(lines[2], "raw key=DIAG_V1|bad value=junk error=malformed diagnostic or provisioning report"); + assert_eq!(lines[2], "raw key=DIAG|bad value=junk error=malformed diagnostic or provisioning report"); assert_eq!( lines[3], format!( @@ -539,7 +547,7 @@ fn parsed_dump_filters_by_kind() { let ts = "2026-08-31T00:00:00.000Z"; let diag = |kind: &str, name: &str, result: &str, duration: &str| { format!( - "DIAG_V1|agent|{VM_ID}|{kind}|{name}|{event_id}|{ts}|none|{result}|{duration}|0" + "DIAG|agent|{VM_ID}|{kind}|{name}|{event_id}|{ts}|none|{result}|{duration}|0" ) }; for (key, value) in [ @@ -694,9 +702,7 @@ fn emit_writes_event_readable_by_dump(#[case] agent: Option<&str>) { let raw = assert_json(kvp(&with_dir(&dir, &["dump"]))); let key = raw[0]["key"].as_str().unwrap(); - assert!( - key.starts_with(&format!("DIAG_V1|{expected_agent}|{VM_ID}|event|")) - ); + assert!(key.starts_with(&format!("DIAG|{expected_agent}|{VM_ID}|event|"))); assert!(key.ends_with("|none|||0")); } diff --git a/libazureinit-kvp/tests/diagnostics.rs b/libazureinit-kvp/tests/diagnostics.rs index f1a70d92..8adbfee6 100644 --- a/libazureinit-kvp/tests/diagnostics.rs +++ b/libazureinit-kvp/tests/diagnostics.rs @@ -171,7 +171,8 @@ fn span_and_point_events_round_trip_with_a_report() { #[rstest] #[case::text(None)] -#[case::compressed(Some(Encoding::GzB64))] +#[case::gzip(Some(Encoding::GzB64))] +#[case::zlib(Some(Encoding::ZlibB64))] fn long_payload_round_trips_through_host_visible_records( #[case] encoding: Option, ) { @@ -219,7 +220,7 @@ fn raw_and_malformed_records_are_preserved_beside_diagnostics() { let store = store_at(&dir); let records = vec![ (format!("{AGENT}|100|{VM_ID}|event|legacy|{EVENT_ID}|2026-08-31T00:00:00Z|0"), "legacy", None), - ("DIAG_V1|bad".into(), "junk", Some(DecodeError::Malformed)), + ("DIAG|bad".into(), "junk", Some(DecodeError::Malformed)), (format!("CLOUD_INIT|100|event|broken|{EVENT_ID}"), "not-json", Some(DecodeError::Malformed)), ("PROVISIONING_REPORT".into(), "result=success", Some(DecodeError::Malformed)), ("DIAG_V2|future".into(), "unknown", Some(DecodeError::UnsupportedVersion)), From 5ee6eebd222acbb6640f47307a0f1a14d6f5222b Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Thu, 17 Sep 2026 17:04:57 -0700 Subject: [PATCH 29/32] Clarify DIAG schema references, UTF-8 encoding, and unknown key/value documentation. --- doc/diagnostics.md | 7 +++---- libazureinit-kvp/src/diagnostics/diagnostic.rs | 5 +++-- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/doc/diagnostics.md b/doc/diagnostics.md index 138fb657..46901d6c 100644 --- a/doc/diagnostics.md +++ b/doc/diagnostics.md @@ -13,9 +13,8 @@ more records. Each key has exactly eleven pipe-delimited fields: DIAG|||||||||| ``` -`DIAG` selects the current format; `DIAG_V*` is reserved for other schemas. -Do not parse an unsupported schema using this layout. Agent versions identify -the producer, not the schema. +`DIAG` selects the current format. Do not parse an unsupported diagnostic schema +using this layout. Agent versions identify the producer, not the schema. | Field | Meaning | |---|---| @@ -228,7 +227,7 @@ timestamp. Decoding errors are attached to preserved records: | Condition | Error token | |---|---| -| Unsupported `DIAG_V*` schema | `unsupported_version` | +| Unsupported `DIAG` schema | `unsupported_version` | | Missing index zero or an index gap | `incomplete_group` | | Repeated index | `duplicate_chunk` | | Unsupported encoding or invalid payload | `undecodable` | diff --git a/libazureinit-kvp/src/diagnostics/diagnostic.rs b/libazureinit-kvp/src/diagnostics/diagnostic.rs index 833c4ffd..feebbbd9 100644 --- a/libazureinit-kvp/src/diagnostics/diagnostic.rs +++ b/libazureinit-kvp/src/diagnostics/diagnostic.rs @@ -197,7 +197,7 @@ pub struct DiagnosticKey { /// When this diagnostic was emitted, in UTC. #[serde(serialize_with = "serialize_timestamp")] pub timestamp: DateTime, - /// Stored payload encoding; `None` means plain text. + /// Stored payload encoding; `None` means UTF-8-encoded plain text. #[serde(serialize_with = "serialize_encoding")] pub encoding: Option, } @@ -298,7 +298,8 @@ pub struct RawKeyValue { pub key: String, /// Original value, without payload decoding. pub value: String, - /// Why a recognized record could not be decoded; `None` for unrelated keys. + /// Why a recognized record could not be decoded; `None` for unknown or + /// non-diagnostic key/value pairs. #[serde(skip_serializing_if = "Option::is_none")] pub error: Option, } From 7eb27f8788320e6ef36861e480d5941e89b306d3 Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Fri, 18 Sep 2026 10:02:28 -0700 Subject: [PATCH 30/32] Broaden diagnostic timestamp and duration parsing and clarify KVP contracts. --- doc/diagnostics.md | 195 ++++++++++----------- doc/kvp.md | 45 +++-- libazureinit-kvp/src/diagnostics/reader.rs | 85 +++++---- 3 files changed, 170 insertions(+), 155 deletions(-) diff --git a/doc/diagnostics.md b/doc/diagnostics.md index 46901d6c..59e4892d 100644 --- a/doc/diagnostics.md +++ b/doc/diagnostics.md @@ -16,24 +16,22 @@ DIAG||||||||| None, "success" => Some(Outcome::Success), @@ -258,6 +244,14 @@ fn decode_diag_group( } fn parse_duration(value: &str) -> Result { + parse_decimal_duration(value).or_else(|_| { + let seconds = + value.parse::().map_err(|_| DecodeError::Malformed)?; + Duration::try_from_secs_f64(seconds).map_err(|_| DecodeError::Malformed) + }) +} + +fn parse_decimal_duration(value: &str) -> Result { let (seconds, fraction) = value .split_once('.') .map_or((value, None), |(seconds, fraction)| { @@ -319,6 +313,7 @@ mod tests { use std::sync::Arc; use base64::{engine::general_purpose::STANDARD, Engine as _}; + use chrono::SecondsFormat; use rstest::rstest; use tempfile::TempDir; @@ -756,15 +751,12 @@ mod tests { #[case::invalid_event_id(5, "bad-uuid")] #[case::empty_encoding(7, "")] #[case::invalid_result(8, "SUCCESS")] - #[case::signed_duration(9, "+1")] #[case::negative_duration(9, "-0.1")] #[case::non_finite_duration(9, "NaN")] #[case::infinite_duration(9, "inf")] - #[case::exponent_duration(9, "1e3")] - #[case::missing_seconds(9, ".5")] - #[case::missing_fraction(9, "1.")] + #[case::duration_whitespace(9, " 1")] + #[case::float_overflow(9, "1e100")] #[case::signed_fraction(9, "1.+2")] - #[case::excess_precision(9, "1.1234567890")] #[case::numeric_overflow(9, "18446744073709551616")] #[case::missing_chunk_index(10, "")] #[case::non_numeric_chunk_index(10, "x")] @@ -795,12 +787,9 @@ mod tests { #[rstest] #[case::unparsable("not a timestamp")] - #[case::two_fraction_digits("2026-08-31T12:34:56.78Z")] - #[case::four_fraction_digits("2026-08-31T12:34:56.7890Z")] - #[case::numeric_offset("2026-08-31T12:34:56.789+00:00")] - #[case::lowercase("2026-08-31t12:34:56.789z")] + #[case::missing_offset("2026-08-31T12:34:56.789")] #[case::invalid_month("2026-13-31T12:34:56.789Z")] - fn native_format_rejects_non_canonical_timestamps(#[case] timestamp: &str) { + fn native_format_rejects_invalid_timestamps(#[case] timestamp: &str) { let records = vec![(with_field(&key(0), 6, timestamp), "value".into())]; assert_eq!( decode_entries(records.clone()), @@ -809,23 +798,33 @@ mod tests { } #[rstest] - #[case::seconds("2026-08-31T12:34:56Z")] - #[case::milliseconds("2026-08-31T12:34:56.789Z")] - #[case::millisecond_whole("2026-08-31T12:34:56.000Z")] - #[case::microseconds("2026-08-31T12:34:56.789123Z")] - #[case::microsecond_trailing_zeros("2026-08-31T12:34:56.789000Z")] - #[case::nanoseconds("2026-08-31T12:34:56.789123456Z")] - fn native_format_accepts_canonical_timestamp_precisions( + #[case::seconds("2026-08-31T12:34:56Z", 0)] + #[case::milliseconds("2026-08-31T12:34:56.789Z", 789_000_000)] + #[case::millisecond_whole("2026-08-31T12:34:56.000Z", 0)] + #[case::microseconds("2026-08-31T12:34:56.789123Z", 789_123_000)] + #[case::microsecond_trailing_zeros( + "2026-08-31T12:34:56.789000Z", + 789_000_000 + )] + #[case::nanoseconds("2026-08-31T12:34:56.789123456Z", 789_123_456)] + #[case::two_fraction_digits("2026-08-31T12:34:56.78Z", 780_000_000)] + #[case::four_fraction_digits("2026-08-31T12:34:56.7890Z", 789_000_000)] + #[case::numeric_offset("2026-08-31T14:34:56.789+02:00", 789_000_000)] + #[case::lowercase("2026-08-31t12:34:56.789z", 789_000_000)] + #[case::subnanoseconds("2026-08-31T12:34:56.7891234567Z", 789_123_456)] + fn native_format_accepts_rfc3339_timestamps( #[case] timestamp: &str, + #[case] expected_nanos: u32, ) { let key = with_field(&key(0), 6, timestamp); let diagnostic = only_diagnostic(decode_entries(vec![(key, "value".into())])); assert_eq!( - diagnostic.key().timestamp, - DateTime::parse_from_rfc3339(timestamp) - .unwrap() - .with_timezone(&Utc) + diagnostic + .key() + .timestamp + .to_rfc3339_opts(SecondsFormat::Nanos, true), + format!("2026-08-31T12:34:56.{expected_nanos:09}Z") ); } @@ -1100,10 +1099,22 @@ mod tests { #[rstest] #[case("0", Duration::ZERO)] + #[case("-0.0", Duration::ZERO)] + #[case("+1", Duration::from_secs(1))] + #[case("1.", Duration::from_secs(1))] + #[case(".5", Duration::from_millis(500))] #[case("1.5", Duration::from_millis(1500))] + #[case("0.312000", Duration::from_millis(312))] + #[case("3.12e-1", Duration::from_millis(312))] #[case("1.000000001", Duration::new(1, 1))] + #[case("0.9999999996", Duration::from_secs(1))] + #[case("1e-12", Duration::ZERO)] + #[case( + "9007199254740993.000000001", + Duration::new(9_007_199_254_740_993, 1) + )] #[case("18446744073709551615.999999999", Duration::MAX)] - fn reader_accepts_decimal_seconds( + fn reader_accepts_duration_seconds( #[case] seconds: &str, #[case] expected: Duration, ) { From b9140793e9e1968da1e11bcba64ecef143fff2c5 Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Fri, 18 Sep 2026 12:11:16 -0700 Subject: [PATCH 31/32] Support opaque event ID reads and configurable writer name limits --- doc/diagnostics.md | 42 ++++++----- .../src/diagnostics/diagnostic.rs | 3 +- libazureinit-kvp/src/diagnostics/reader.rs | 18 +++-- libazureinit-kvp/src/diagnostics/writer.rs | 75 ++++++++++++++++--- 4 files changed, 102 insertions(+), 36 deletions(-) diff --git a/doc/diagnostics.md b/doc/diagnostics.md index 59e4892d..f3637305 100644 --- a/doc/diagnostics.md +++ b/doc/diagnostics.md @@ -23,7 +23,7 @@ using this layout. Agent versions identify the producer, not the schema. | `vm_id` | VM identity | UUID | | `kind` | Operation endpoint or standalone observation | `start`, `finish`, or `event` | | `name` | Operation or observation, such as `provision:run` or `dmesg` | UTF-8 text | -| `event_id` | Shared by an operation's start and finish; unique to a standalone event | UUID | +| `event_id` | Shared by an operation's start and finish; unique to a standalone event | Opaque UTF-8 identifier | | `timestamp` | Emission time | RFC 3339 timestamp | | `encoding` | Stored value representation | `none`, `zlib+b64`, or `gz+b64` | | `result` | Reported outcome, when applicable | `success`, `fail`, or empty | @@ -33,6 +33,8 @@ using this layout. Agent versions identify the producer, not the schema. All fields except `result` and `duration` are required and nonempty. Key fields cannot contain `|` or NUL; there is no key-field escaping. +Preserve event IDs as text: `0000000001` and `1` are distinct identifiers. + ## Timing and Correlation | Kind | Result | Duration | Meaning | @@ -53,10 +55,8 @@ second, millisecond (default), microsecond or nanosecond precision. Durations are finite, nonnegative IEEE 754 double-precision seconds, in decimal or exponent notation. Empty means absent; zero is a measured duration. - -The writer emits fixed-point durations with 0, 3, 6 (default) or 9 fractional -digits, independently of timestamp precision, discarding finer digits. See -[Implementation Notes](#implementation-notes) for reader limits and rounding. +See [Implementation Notes](#implementation-notes) for reader limits and +rounding. ### Examples @@ -106,17 +106,17 @@ not aliases. Compression does not imply that the decoded content is text. ## Limits The complete key must fit in 254 UTF-8 bytes, including separators and the -chunk index. The table accounts for this writer's output. Example widths use -the finish record above, with default precision and a single chunk; they are -illustrative, not measured production averages. +chunk index. The table accounts for this writer's output with the default name +limit. Example widths use the finish record above, with default precision and +a single chunk; they are illustrative, not measured production averages. -| Field | Example bytes | Maximum emitted bytes | Basis | +| Field | Example bytes | Field bound (bytes) | Basis | |---|---:|---:|---| | `DIAG` | 4 | 4 | Fixed token | | `agent` | 16 | 32 | `azure-init/0.1.1`; producer text limit | | `vm_id` | 36 | 36 | Hyphenated UUID in the example; writer UUID limit | | `kind` | 6 | 6 | `start`/`event`: 5; `finish`: 6 | -| `name` | 13 | 48 | `provision:run`; producer text limit | +| `name` | 13 | 64 | `provision:run`; configurable producer limit, default 64 | | `event_id` | 36 | 36 | Hyphenated UUID in the example; writer UUID limit | | `timestamp` | 24 | 30 | UTC `Z` output: 20/24/27/30 for seconds/ms/us/ns | | `encoding` | 4 | 8 | `none`: 4; `gz+b64`: 6; `zlib+b64`: 8 | @@ -124,13 +124,14 @@ illustrative, not measured production averages. | `duration` | 8 | 30 | `0.312000`; up to 20 whole-second digits, a point and 9 fractional digits | | `chunk_index` | 1 | 4 | `0` through `1022` | | Ten pipe separators | 10 | 10 | One byte each | -| **Total** | **165** | **251** | | -| **Space remaining** | **89** | **3** | Within 254 bytes | +| **Total** | **165** | **267** | Sum of field widths | +| **Space remaining** | **89** | **13 over limit** | Against the 254-byte limit | The start and compressed-event examples use 149 and 147 key bytes respectively. -With both default precisions, the maximum emitted key is 242 bytes. Widths -count text bytes, not the in-memory size of a double. These are writer budgets, -not universal widths for every RFC 3339 timestamp or numeric spelling. +The field bounds total 258 bytes with both default precisions, so not every +combination fits; oversized keys are rejected before writing. Widths count text +bytes, not the in-memory size of a double. These are writer budgets, not +universal widths for every RFC 3339 timestamp or numeric spelling. Encoded values are limited to 1,022 bytes per chunk, with at most 1,023 chunks per payload. Producer budgets do not impose equivalent read limits on other @@ -256,10 +257,15 @@ Cloud-init durations round to microseconds; source encoding labels are preserved ### Writer Choices +The writer requires UUID event IDs. The name limit defaults to 64 UTF-8 bytes +and is configurable with `DiagnosticWriter::with_max_name_bytes`; oversized +names are rejected, not truncated. Durations use fixed-point seconds with +microsecond precision by default, independently of timestamp precision; +finer digits are discarded. + Validation precedes writing, but I/O failure may leave a partial batch. Pool -cleanup is explicit. Free-form agent and name limits leave room for the other -key fields; the full key length is checked for every chunk. Timestamp and -duration formatting choices keep emitted keys within the [budget](#limits). +cleanup is explicit. The complete key, including each chunk index, must fit +the [budget](#limits) regardless of the configured name limit. Report writers emit success fields as `result`, `agent`, `pps_type`, `vm_id`, `timestamp`, then extras. Failure order is `result`, `reason`, `agent`, extras, diff --git a/libazureinit-kvp/src/diagnostics/diagnostic.rs b/libazureinit-kvp/src/diagnostics/diagnostic.rs index feebbbd9..33b14f78 100644 --- a/libazureinit-kvp/src/diagnostics/diagnostic.rs +++ b/libazureinit-kvp/src/diagnostics/diagnostic.rs @@ -192,7 +192,8 @@ pub struct DiagnosticKey { pub vm_id: Option, /// Operation or observation name, such as `provision:run` or `dmesg`. pub name: String, - /// UUID shared by an operation's start and finish; unique for a standalone event. + /// Opaque identifier shared by an operation's start and finish; + /// unique for a standalone event. pub event_id: String, /// When this diagnostic was emitted, in UTC. #[serde(serialize_with = "serialize_timestamp")] diff --git a/libazureinit-kvp/src/diagnostics/reader.rs b/libazureinit-kvp/src/diagnostics/reader.rs index eb551d5e..c2cd7fd6 100644 --- a/libazureinit-kvp/src/diagnostics/reader.rs +++ b/libazureinit-kvp/src/diagnostics/reader.rs @@ -186,7 +186,6 @@ fn decode_diag_group( return Err(DecodeError::Malformed); } Uuid::parse_str(vm_id).map_err(|_| DecodeError::Malformed)?; - Uuid::parse_str(event_id).map_err(|_| DecodeError::Malformed)?; let parsed_timestamp = DateTime::parse_from_rfc3339(timestamp) .map_err(|_| DecodeError::Malformed)? .with_timezone(&Utc); @@ -636,16 +635,21 @@ mod tests { ); } - #[test] - fn native_event_decodes_without_local_identity() { + #[rstest] + #[case::uuid(EVENT_ID)] + #[case::ten_digits("0000000001")] + #[case::opaque("bad-uuid")] + fn native_event_decodes_without_local_identity(#[case] event_id: &str) { let payload = "héllo\n\"message\" | ="; - let diagnostic = - only_diagnostic(decode_entries(vec![(key(0), payload.into())])); + let diagnostic = only_diagnostic(decode_entries(vec![( + with_field(&key(0), 5, event_id), + payload.into(), + )])); assert_eq!(diagnostic.kind(), Kind::Event); assert_eq!(diagnostic.key().agent, AGENT); assert_eq!(diagnostic.key().vm_id.as_deref(), Some(VM_ID)); assert_eq!(diagnostic.key().name, "test"); - assert_eq!(diagnostic.key().event_id, EVENT_ID); + assert_eq!(diagnostic.key().event_id, event_id); assert_eq!(diagnostic.key().encoding, None); assert_eq!( diagnostic @@ -748,7 +752,7 @@ mod tests { #[case::invalid_kind(3, "compressed")] #[case::empty_name(4, "")] #[case::null_in_name(4, "bad\0name")] - #[case::invalid_event_id(5, "bad-uuid")] + #[case::empty_event_id(5, "")] #[case::empty_encoding(7, "")] #[case::invalid_result(8, "SUCCESS")] #[case::negative_duration(9, "-0.1")] diff --git a/libazureinit-kvp/src/diagnostics/writer.rs b/libazureinit-kvp/src/diagnostics/writer.rs index bc3984a4..8ac68b48 100644 --- a/libazureinit-kvp/src/diagnostics/writer.rs +++ b/libazureinit-kvp/src/diagnostics/writer.rs @@ -16,7 +16,7 @@ use super::MAX_CHUNK_BYTES; use crate::{KvpError, KvpPoolStore}; const MAX_AGENT_BYTES: usize = 32; -const MAX_NAME_BYTES: usize = 48; +const DEFAULT_MAX_NAME_BYTES: usize = 64; const MAX_UUID_BYTES: usize = 36; const MAX_TIMESTAMP_BYTES: usize = 30; const MAX_KEY_BYTES: usize = 254; @@ -123,6 +123,7 @@ pub struct DiagnosticWriter { store: KvpPoolStore, agent: String, vm_id: String, + max_name_bytes: usize, timestamp_precision: TimestampPrecision, duration_precision: DurationPrecision, } @@ -147,11 +148,21 @@ impl DiagnosticWriter { store, agent, vm_id, + max_name_bytes: DEFAULT_MAX_NAME_BYTES, timestamp_precision: TimestampPrecision::default(), duration_precision: DurationPrecision::default(), }) } + /// Sets the maximum name length in UTF-8 bytes (default: 64). + /// + /// Names over the limit are rejected, never truncated. The complete key + /// must still fit within 254 bytes, including its chunk index. + pub fn with_max_name_bytes(mut self, max: usize) -> Self { + self.max_name_bytes = max; + self + } + /// Overrides the default millisecond precision for emitted timestamps. pub fn with_timestamp_precision( mut self, @@ -248,6 +259,7 @@ impl DiagnosticWriter { diagnostic, self.timestamp_precision, self.duration_precision, + self.max_name_bytes, )?) } } @@ -256,6 +268,7 @@ fn prepare_records( diagnostic: Diagnostic, timestamp_precision: TimestampPrecision, duration_precision: DurationPrecision, + max_name_bytes: usize, ) -> Result, KvpError> { let kind = diagnostic.kind(); let (key, payload, result, duration) = match diagnostic { @@ -277,7 +290,7 @@ fn prepare_records( .as_deref() .ok_or(KvpError::EmptyEventField { field: "vm_id" })?; validate_uuid("vm_id", vm_id)?; - validate_field("name", &key.name, MAX_NAME_BYTES)?; + validate_field("name", &key.name, max_name_bytes)?; validate_uuid("event_id", &key.event_id)?; let duration = duration.map_or_else(String::new, |duration| { @@ -583,13 +596,17 @@ mod tests { } #[rstest] - #[case("agent", "a".repeat(32), 32)] - #[case("agent", "é".repeat(16), 32)] - #[case("name", "n".repeat(48), 48)] + #[case("agent", "a".repeat(32), 32, None)] + #[case("agent", "é".repeat(16), 32, None)] + #[case("name", "n".repeat(64), 64, None)] + #[case("name", "é".repeat(32), 64, None)] + #[case("name", "n".repeat(32), 32, Some(32))] + #[case("name", "n".repeat(96), 96, Some(96))] fn freeform_caps_count_bytes_without_truncation( #[case] field: &'static str, #[case] value: String, #[case] max: usize, + #[case] max_name_bytes: Option, #[values(PoolMode::Safe, PoolMode::Unsafe)] mode: PoolMode, ) { let dir = TempDir::new().unwrap(); @@ -599,6 +616,10 @@ mod tests { _ => (AGENT, value.as_str()), }; let writer = DiagnosticWriter::new(pool.clone(), agent, VM_ID).unwrap(); + let writer = match max_name_bytes { + Some(max) => writer.with_max_name_bytes(max), + None => writer, + }; writer.emit_event(name, "ok", None, None, None).unwrap(); let records = pool.dump().unwrap(); let fields: Vec<_> = records[0].0.split('|').collect(); @@ -630,7 +651,8 @@ mod tests { #[case] value: &str, #[case] reason: &str, ) { - let error = validate_field("name", value, MAX_NAME_BYTES).unwrap_err(); + let error = + validate_field("name", value, DEFAULT_MAX_NAME_BYTES).unwrap_err(); match reason { "empty" => { assert!(matches!( @@ -655,6 +677,7 @@ mod tests { }), TimestampPrecision::Millis, DurationPrecision::default(), + DEFAULT_MAX_NAME_BYTES, ) .unwrap(); assert_eq!( @@ -686,6 +709,7 @@ mod tests { }), TimestampPrecision::Millis, DurationPrecision::default(), + DEFAULT_MAX_NAME_BYTES, ) .unwrap(); assert_eq!( @@ -1004,30 +1028,59 @@ mod tests { } #[test] - fn largest_permitted_fields_fit_with_four_digit_index() { + fn default_name_limit_fits_with_four_digit_index() { let diagnostic = Diagnostic::Finish(DiagnosticFinish { key: DiagnosticKey { agent: "a".repeat(MAX_AGENT_BYTES), - name: "n".repeat(MAX_NAME_BYTES), + name: "n".repeat(DEFAULT_MAX_NAME_BYTES), encoding: Some(Encoding::ZlibB64), ..key() }, payload: "test".into(), result: Outcome::Success, - duration: Duration::MAX, + duration: Duration::from_millis(312), }); let records = prepare_records( diagnostic, TimestampPrecision::Nanos, DurationPrecision::Nanos, + DEFAULT_MAX_NAME_BYTES, ) .unwrap(); let base = records[0].0.rsplit_once('|').unwrap().0; let longest = format!("{base}|1022"); - assert_eq!(longest.len(), 251); + assert_eq!(longest.len(), 248); assert!(longest.len() <= MAX_KEY_BYTES); } + #[test] + fn name_allowance_does_not_override_full_key_limit() { + let error = assert_rejected_without_writes(|writer| { + DiagnosticWriter::new( + writer.store.clone(), + "a".repeat(MAX_AGENT_BYTES), + VM_ID, + )? + .with_timestamp_precision(TimestampPrecision::Nanos) + .with_duration_precision(DurationPrecision::Nanos) + .emit_finish( + EVENT_ID, + &"n".repeat(DEFAULT_MAX_NAME_BYTES), + "message", + Some(Encoding::ZlibB64), + Outcome::Success, + Duration::MAX, + ) + }); + assert!(matches!( + error, + KvpError::KeyTooLarge { + max: MAX_KEY_BYTES, + actual: 264 + } + )); + } + #[test] fn start_rejects_invalid_event_id_before_io() { let error = assert_rejected_without_writes(|writer| { @@ -1112,6 +1165,7 @@ mod tests { event(missing_vm, "bad".into()), TimestampPrecision::Millis, DurationPrecision::default(), + DEFAULT_MAX_NAME_BYTES, ), Err(KvpError::EmptyEventField { field: "vm_id" }) )); @@ -1127,6 +1181,7 @@ mod tests { event(expanded_year, "bad".into()), TimestampPrecision::Nanos, DurationPrecision::default(), + DEFAULT_MAX_NAME_BYTES, ), Err(KvpError::EventFieldTooLong { field: "timestamp", From 6c72046773d66b95e776c8d515fd4318cdf210ce Mon Sep 17 00:00:00 2001 From: peytonr18 Date: Fri, 18 Sep 2026 12:19:41 -0700 Subject: [PATCH 32/32] Clarify diagnostic event ID wording in spec and Rustdoc --- doc/diagnostics.md | 6 ++---- libazureinit-kvp/src/diagnostics/diagnostic.rs | 2 +- 2 files changed, 3 insertions(+), 5 deletions(-) diff --git a/doc/diagnostics.md b/doc/diagnostics.md index f3637305..02e8ee7f 100644 --- a/doc/diagnostics.md +++ b/doc/diagnostics.md @@ -23,7 +23,7 @@ using this layout. Agent versions identify the producer, not the schema. | `vm_id` | VM identity | UUID | | `kind` | Operation endpoint or standalone observation | `start`, `finish`, or `event` | | `name` | Operation or observation, such as `provision:run` or `dmesg` | UTF-8 text | -| `event_id` | Shared by an operation's start and finish; unique to a standalone event | Opaque UTF-8 identifier | +| `event_id` | Shared by an operation's start and finish; unique to a standalone event | UTF-8 identifier, typically UUID | | `timestamp` | Emission time | RFC 3339 timestamp | | `encoding` | Stored value representation | `none`, `zlib+b64`, or `gz+b64` | | `result` | Reported outcome, when applicable | `success`, `fail`, or empty | @@ -33,8 +33,6 @@ using this layout. Agent versions identify the producer, not the schema. All fields except `result` and `duration` are required and nonempty. Key fields cannot contain `|` or NUL; there is no key-field escaping. -Preserve event IDs as text: `0000000001` and `1` are distinct identifiers. - ## Timing and Correlation | Kind | Result | Duration | Meaning | @@ -110,7 +108,7 @@ chunk index. The table accounts for this writer's output with the default name limit. Example widths use the finish record above, with default precision and a single chunk; they are illustrative, not measured production averages. -| Field | Example bytes | Field bound (bytes) | Basis | +| Field | Example bytes | Maximum emitted bytes | Basis | |---|---:|---:|---| | `DIAG` | 4 | 4 | Fixed token | | `agent` | 16 | 32 | `azure-init/0.1.1`; producer text limit | diff --git a/libazureinit-kvp/src/diagnostics/diagnostic.rs b/libazureinit-kvp/src/diagnostics/diagnostic.rs index 33b14f78..c568f891 100644 --- a/libazureinit-kvp/src/diagnostics/diagnostic.rs +++ b/libazureinit-kvp/src/diagnostics/diagnostic.rs @@ -192,7 +192,7 @@ pub struct DiagnosticKey { pub vm_id: Option, /// Operation or observation name, such as `provision:run` or `dmesg`. pub name: String, - /// Opaque identifier shared by an operation's start and finish; + /// UTF-8 identifier, typically UUID, shared by an operation's start and finish; /// unique for a standalone event. pub event_id: String, /// When this diagnostic was emitted, in UTC.