From 317df8919f98030f30299b952a89874bc5fb79d2 Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 00:30:04 +0900 Subject: [PATCH 01/55] =?UTF-8?q?deps:=20bump=20epics-rs=200.20.4=20?= =?UTF-8?q?=E2=86=92=200.28.0?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit pvmonitor_handle now takes an on_conn callback; wire it to conn_info and the disconnect counters the way the CA monitor-end block does, so a reconnect's first sample re-arms first_after_connect and bypasses the drift filter. DBF_UCHAR lands in ScalarByte: CA promotes it to DBR_CHAR with the same 1-byte payload. --- Cargo.lock | 53 +++++++++++++++---- Cargo.toml | 2 +- crates/archiver-engine/src/channel_manager.rs | 50 ++++++++++++++++- tests/api_pva.rs | 4 +- 4 files changed, 95 insertions(+), 14 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 4146147..bac1cdc 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -843,14 +843,18 @@ dependencies = [ [[package]] name = "epics-base-rs" -version = "0.20.3" +version = "0.28.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e509a355ac8d5fa81450df13b08b9aca691dd12a67dbabce4e0b4e60c282bac7" +checksum = "ffc5fb8c868e1114e589680551fd5dcc48df339a51c042e92a5415285a83cb3e" dependencies = [ + "arc-swap", + "async-trait", "bytes", "chrono", "clap", + "epics-libcom-rs", "epics-macros-rs", + "epics-rtems-boot", "hostname", "if-addrs", "libc", @@ -865,9 +869,9 @@ dependencies = [ [[package]] name = "epics-ca-rs" -version = "0.20.3" +version = "0.28.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "511fefc270b9874dad78a8d57b7eb0b187b72c8e6797731cde9360b117dad048" +checksum = "07d093efafb055a711069a2e2985004f29a55198167892e52410a1d053eeea51" dependencies = [ "arc-swap", "async-trait", @@ -875,6 +879,7 @@ dependencies = [ "clap", "dashmap", "epics-base-rs", + "epics-rtems-boot", "futures-util", "if-addrs", "libc", @@ -884,13 +889,31 @@ dependencies = [ "thiserror 1.0.69", "tokio", "tracing", + "windows-sys 0.59.0", +] + +[[package]] +name = "epics-libcom-rs" +version = "0.28.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6a5794b250cda69b1fe724af5826d37831f8c9bf2a333359025fc9bae47d4b2c" +dependencies = [ + "chrono", + "epics-rtems-boot", + "hostname", + "if-addrs", + "libc", + "parking_lot", + "socket2 0.5.10", + "tokio", + "tracing", ] [[package]] name = "epics-macros-rs" -version = "0.20.3" +version = "0.28.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "afece7651f6cc6b6b52286bbf99774b968297891e532d03bf046d0d16dc0ffc3" +checksum = "d4574f2e2c8c5317e6f0983276feccae4cc5d45535620f9d9bb42987c663e229" dependencies = [ "proc-macro-crate", "proc-macro2", @@ -900,9 +923,9 @@ dependencies = [ [[package]] name = "epics-pva-rs" -version = "0.20.3" +version = "0.28.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a0e58ef561ffb4f66b739bc0ecb78e36b7c795ce014a4713c30a0bdfbc8099a4" +checksum = "67defd7fb25c002a023aa1750617c10ace4eba91400c82c61b7632fcf0525352" dependencies = [ "bytes", "chrono", @@ -939,15 +962,25 @@ dependencies = [ [[package]] name = "epics-rs" -version = "0.20.4" +version = "0.28.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "758bf8455add8561007f93470538a52683ef064d6d2fa39edb524c486ffd8546" +checksum = "17a8d56c5d94ee6c8cf7c7e206427244a485638809c870f8a5007dd8c896805c" dependencies = [ "epics-base-rs", "epics-ca-rs", "epics-pva-rs", ] +[[package]] +name = "epics-rtems-boot" +version = "0.28.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ac25b14a69d693209a239812b356096de7c8051ec87af3f98da29ac21fdd7ff8" +dependencies = [ + "cc", + "libc", +] + [[package]] name = "equivalent" version = "1.0.2" diff --git a/Cargo.toml b/Cargo.toml index 62068ab..083d4af 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -74,7 +74,7 @@ tokio-util = { version = "0.7", features = ["rt"] } rusqlite = { version = "0.34", features = ["bundled"] } # EPICS Channel Access -epics-rs = { version = "0.20.4", features = ["pva"] } +epics-rs = { version = "0.28.0", features = ["pva"] } # URL encoding urlencoding = "2" diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index d818529..a11aa93 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -8,6 +8,7 @@ use epics_rs::base::server::snapshot::DbrClass; use epics_rs::base::types::{DbFieldType, EpicsValue}; use epics_rs::ca::client::{CaChannel, CaClient, ConnectionEvent}; use epics_rs::pva::client_native::PvaClient; +use epics_rs::pva::client_native::ops_v2::MonitorConnEvent; use epics_rs::pva::proto::ByteOrder; use epics_rs::pva::pvdata::encode::{encode_pv_field, encode_type_desc}; use epics_rs::pva::pvdata::{PvField, ScalarType, ScalarValue, TypedScalarArray}; @@ -1492,7 +1493,46 @@ async fn monitor_loop_pva( drift_secs, ); }; - let handle = match pva_client.pvmonitor_handle(&pv_name, cb).await { + // Connection transitions come from the handle's + // `MonitorConnEvent` — the reactor's sanctioned signal, + // replacing the watchdog's event-gap approximation on this + // path. Mirrors the CA monitor-end block: resetting + // `last_event_time` re-arms `first_after_connect` so a + // reconnect backfill sample bypasses the drift filter + // instead of being dropped. The `was_connected` guard keeps + // the counter single-owner against the watchdog, which may + // have marked the PV disconnected already. + let conn_info_ev = conn_info.clone(); + let counters_ev = counters.clone(); + let pv_name_ev = pv_name.clone(); + let on_conn = move |ev: MonitorConnEvent| match ev { + MonitorConnEvent::Connected { .. } => { + let mut ci = conn_info_ev.lock().unwrap_or_else(|e| e.into_inner()); + if ci.connected_since.is_none() { + ci.connected_since = Some(SystemTime::now()); + } + ci.is_connected = true; + ci.state = PvConnectionState::Connected; + } + MonitorConnEvent::Disconnected | MonitorConnEvent::Finished => { + let was_connected = { + let mut ci = conn_info_ev.lock().unwrap_or_else(|e| e.into_inner()); + let prev = ci.is_connected; + ci.is_connected = false; + ci.last_event_time = None; + ci.state = PvConnectionState::Disconnected; + prev + }; + if was_connected { + counters_ev.disconnect_count.fetch_add(1, Ordering::Relaxed); + counters_ev + .last_disconnect_unix_secs + .store(unix_secs(SystemTime::now()), Ordering::Relaxed); + } + debug!(pv = pv_name_ev, "PVA monitor connection lost"); + } + }; + let handle = match pva_client.pvmonitor_handle(&pv_name, cb, on_conn).await { Ok(h) => h, Err(e) => { counters @@ -2228,6 +2268,7 @@ fn epics_value_to_field_string(val: &EpicsValue) -> String { // Transient NTEnum carrier: render the index, same as `Enum`. EpicsValue::EnumWithChoices { index, .. } => index.to_string(), EpicsValue::Char(v) => v.to_string(), + EpicsValue::UChar(v) => v.to_string(), EpicsValue::Long(v) => v.to_string(), EpicsValue::Int64(v) => v.to_string(), EpicsValue::UInt64(v) => v.to_string(), @@ -2244,6 +2285,7 @@ fn epics_value_to_field_string(val: &EpicsValue) -> String { EpicsValue::UShortArray(v) => format!("{v:?}"), EpicsValue::ULongArray(v) => format!("{v:?}"), EpicsValue::CharArray(v) => String::from_utf8_lossy(v).into_owned(), + EpicsValue::UCharArray(v) => String::from_utf8_lossy(v).into_owned(), EpicsValue::StringArray(v) => format!("{v:?}"), } } @@ -2833,7 +2875,9 @@ fn dbr_field_to_arch_type(field_type: DbFieldType) -> ArchDbType { DbFieldType::Short => ArchDbType::ScalarShort, DbFieldType::Float => ArchDbType::ScalarFloat, DbFieldType::Enum => ArchDbType::ScalarEnum, - DbFieldType::Char => ArchDbType::ScalarByte, + // DBF_UCHAR promotes to DBR_CHAR over CA (identical 1-byte + // payload), so both land in the ScalarByte slot. + DbFieldType::Char | DbFieldType::UChar => ArchDbType::ScalarByte, DbFieldType::Long => ArchDbType::ScalarInt, // PB PayloadType has no SCALAR_LONG (i64) — values outside i32 range // are truncated by the i32 cast in epics_value_to_archiver. The @@ -2858,6 +2902,7 @@ fn epics_value_to_archiver(val: &EpicsValue) -> ArchiverValue { // Transient NTEnum carrier: archive the index, same as `Enum`. EpicsValue::EnumWithChoices { index, .. } => ArchiverValue::ScalarEnum(*index as i32), EpicsValue::Char(v) => ArchiverValue::ScalarByte(vec![*v]), + EpicsValue::UChar(v) => ArchiverValue::ScalarByte(vec![*v]), EpicsValue::Long(v) => ArchiverValue::ScalarInt(*v), // i64/u64/u32 fold into the i32 ScalarInt slot (out-of-range values // truncate) — matches the `DbFieldType` registration mapping above. @@ -2888,6 +2933,7 @@ fn epics_value_to_archiver(val: &EpicsValue) -> ArchiverValue { ArchiverValue::VectorShort(v.iter().map(|x| *x as i32).collect()) } EpicsValue::CharArray(v) => ArchiverValue::VectorChar(v.clone()), + EpicsValue::UCharArray(v) => ArchiverValue::VectorChar(v.clone()), EpicsValue::StringArray(v) => { ArchiverValue::VectorString(v.iter().map(|s| s.to_string()).collect()) } diff --git a/tests/api_pva.rs b/tests/api_pva.rs index 6681968..effcdaf 100644 --- a/tests/api_pva.rs +++ b/tests/api_pva.rs @@ -111,7 +111,9 @@ async fn pva_rpc_get_data_end_to_end() { let (_resp_desc, resp_value) = client .pvrpc("archappl/getData", &req_desc, &req_value) .await - .expect("rpc call ok"); + .expect("rpc call ok") + .into_value() + .expect("getData reply carries an NTTable, never RpcReply::Empty"); // ── Inspect the NTTable: value.value column should contain 1.5, 2.5 ── let PvField::Structure(resp) = resp_value else { From 74ae5981e7b023f8f49d3bb9cc33b36af25c5d05 Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 00:30:04 +0900 Subject: [PATCH 02/55] engine(pva): re-subscribe when the pvmonitor_handle task ends A server-side SharedPV::close() ends the handle's reactor task with Finished (pvxs sends FINISH before DESTROY_CHANNEL), and fatal/remote errors end it behind a plain Disconnected; the fast path parked on cancel_token and never re-subscribed. SubscriptionHandle has no completion future, so a drop guard captured by on_conn signals the task end and the loop retries like the custom-request path. --- crates/archiver-engine/src/channel_manager.rs | 109 +++-- tests/save_path_live.rs | 386 ++++++++++++++++++ 2 files changed, 463 insertions(+), 32 deletions(-) create mode 100644 tests/save_path_live.rs diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index a11aa93..d8ef52c 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -715,9 +715,11 @@ impl ChannelManager { } /// PVA equivalent of `start_archiving_internal`. Spawns a single - /// callback-driven monitor task; pvAccess auto-reconnect inside - /// `pvmonitor_handle` removes the explicit reconnect loop the CA - /// path needs. PVA records use `channel: None` on the [`PvHandle`] + /// callback-driven monitor task. `pvmonitor_handle` reconnects on + /// its own only across circuit loss; a server-side stream end + /// (`Finished`, fatal/remote error) ends its reactor task, and + /// `monitor_loop_pva` re-subscribes from there. PVA records use + /// `channel: None` on the [`PvHandle`] /// and skip per-field "extras" subscriptions (`.HIHI`/`.LOLO`/…) /// since pvAccess wraps those in the NTScalar value structure /// rather than separate channels. @@ -1393,8 +1395,10 @@ async fn monitor_loop_pva( // Subscribe loop with retry. The custom-request path uses // `pvmonitor_with_request` (no SubscriptionHandle) inside a // tokio::select! so cancellation drops the future cleanly; the - // empty-request path keeps the original SubscriptionHandle form - // since that's the simpler path for the common case. + // empty-request path keeps the SubscriptionHandle form and + // watches its reactor task through `SubscriptionEnd`. Both paths + // follow one rule: a subscription that ends for any reason other + // than our own cancel waits `CA_RETRY_DELAY` and re-subscribes. loop { if let Some(ref req) = request_expr { // Custom-request path: 1-arg callback. `pvmonitor_with_request` @@ -1502,34 +1506,44 @@ async fn monitor_loop_pva( // instead of being dropped. The `was_connected` guard keeps // the counter single-owner against the watchdog, which may // have marked the PV disconnected already. + // One `Notify` per subscription generation, so a permit + // left by a generation that failed to subscribe cannot + // fire into the next one. + let ended = Arc::new(tokio::sync::Notify::new()); + let end_guard = SubscriptionEnd(ended.clone()); let conn_info_ev = conn_info.clone(); let counters_ev = counters.clone(); let pv_name_ev = pv_name.clone(); - let on_conn = move |ev: MonitorConnEvent| match ev { - MonitorConnEvent::Connected { .. } => { - let mut ci = conn_info_ev.lock().unwrap_or_else(|e| e.into_inner()); - if ci.connected_since.is_none() { - ci.connected_since = Some(SystemTime::now()); - } - ci.is_connected = true; - ci.state = PvConnectionState::Connected; - } - MonitorConnEvent::Disconnected | MonitorConnEvent::Finished => { - let was_connected = { + let on_conn = move |ev: MonitorConnEvent| { + // Captured so the guard lives in the reactor task and + // drops when the task ends. + let _ = &end_guard; + match ev { + MonitorConnEvent::Connected { .. } => { let mut ci = conn_info_ev.lock().unwrap_or_else(|e| e.into_inner()); - let prev = ci.is_connected; - ci.is_connected = false; - ci.last_event_time = None; - ci.state = PvConnectionState::Disconnected; - prev - }; - if was_connected { - counters_ev.disconnect_count.fetch_add(1, Ordering::Relaxed); - counters_ev - .last_disconnect_unix_secs - .store(unix_secs(SystemTime::now()), Ordering::Relaxed); + if ci.connected_since.is_none() { + ci.connected_since = Some(SystemTime::now()); + } + ci.is_connected = true; + ci.state = PvConnectionState::Connected; + } + MonitorConnEvent::Disconnected | MonitorConnEvent::Finished => { + let was_connected = { + let mut ci = conn_info_ev.lock().unwrap_or_else(|e| e.into_inner()); + let prev = ci.is_connected; + ci.is_connected = false; + ci.last_event_time = None; + ci.state = PvConnectionState::Disconnected; + prev + }; + if was_connected { + counters_ev.disconnect_count.fetch_add(1, Ordering::Relaxed); + counters_ev + .last_disconnect_unix_secs + .store(unix_secs(SystemTime::now()), Ordering::Relaxed); + } + debug!(pv = pv_name_ev, "PVA monitor connection lost"); } - debug!(pv = pv_name_ev, "PVA monitor connection lost"); } }; let handle = match pva_client.pvmonitor_handle(&pv_name, cb, on_conn).await { @@ -1546,14 +1560,45 @@ async fn monitor_loop_pva( } }; debug!(pv = pv_name, "PVA monitor active"); - cancel_token.cancelled().await; - debug!(pv = pv_name, "PVA monitor cancelled; dropping subscription"); - drop(handle); - return; + tokio::select! { + _ = cancel_token.cancelled() => { + debug!(pv = pv_name, "PVA monitor cancelled; dropping subscription"); + drop(handle); + return; + } + _ = ended.notified() => { + debug!(pv = pv_name, "PVA monitor subscription ended; resubscribing"); + drop(handle); + } + } + tokio::select! { + _ = cancel_token.cancelled() => return, + _ = tokio::time::sleep(CA_RETRY_DELAY) => continue, + } } } } +/// Fires when the reactor task behind a `pvmonitor_handle` +/// subscription stops running. `SubscriptionHandle` exposes no +/// completion future, but the task owns the `on_conn` closure, so +/// the closure's captures drop exactly when the task ends — on +/// `Finished` (server closed the stream), on a fatal/remote error +/// (surfaced only as `Disconnected`, indistinguishable from a +/// circuit loss the task retries itself), or on channel close. +/// `monitor_loop_pva` selects on this next to `cancel_token`, so +/// every task end re-subscribes through the same retry rule as the +/// custom-request path. +struct SubscriptionEnd(Arc); + +impl Drop for SubscriptionEnd { + fn drop(&mut self) { + // Stores a permit when nobody is waiting yet, so an end that + // races ahead of `notified()` is not lost. + self.0.notify_one(); + } +} + /// PVA Scan loop: periodic `pvget` instead of streaming monitor. /// Mirrors the CA scan_loop's per-tick connection check + sample emit. #[allow(clippy::too_many_arguments)] diff --git a/tests/save_path_live.rs b/tests/save_path_live.rs new file mode 100644 index 0000000..9c8f2aa --- /dev/null +++ b/tests/save_path_live.rs @@ -0,0 +1,386 @@ +//! Live save-path smoke against the epics-rs PVA stack: an in-process +//! `SharedPV` NTScalar served by an isolated PVA server → +//! `ChannelManager` PVA monitor → sharded write pool → PlainPB on disk +//! + on-disk SQLite registry → read-back through `query_data`. +//! +//! The archiver's own `PvaClient` is built from the environment, so the +//! isolated server is reached through `EPICS_PVA_NAME_SERVERS`. Each test +//! sets process env; run under nextest (one process per test). + +use std::sync::Arc; +use std::time::{Duration, SystemTime, UNIX_EPOCH}; + +use archiver_core::registry::{Protocol, PvRegistry, PvStatus, SampleMode}; +use archiver_core::retrieval::query::query_data; +use archiver_core::storage::partition::PartitionGranularity; +use archiver_core::storage::plainpb::PlainPbStoragePlugin; +use archiver_core::storage::traits::StoragePlugin; +use archiver_core::types::{ArchDbType, ArchiverValue}; +use archiver_engine::channel_manager::{ + ChannelManager, PvCountersSnapshot, ShardedWritePoolConfig, WriteLoopConfig, + run_sharded_write_pool, +}; +use epics_rs::pva::nt::NTScalar; +use epics_rs::pva::pvdata::{FieldDesc, PvField, ScalarType, ScalarValue}; +use epics_rs::pva::server_native::{PvaServer, SharedPV, SharedSource}; + +const PV: &str = "SMOKE:PVA:DBL"; + +fn set_scalar(field: &mut PvField, path: &[&str], val: ScalarValue) { + match path { + [] => *field = PvField::Scalar(val), + [head, rest @ ..] => { + let PvField::Structure(s) = field else { + panic!("{head}: parent is not a structure"); + }; + let child = s + .get_field_mut(head) + .unwrap_or_else(|| panic!("no field {head}")); + set_scalar(child, rest, val); + } + } +} + +fn nt_desc() -> FieldDesc { + NTScalar::new(ScalarType::Double).with_display().build() +} + +fn nt_double(value: f64, ts: SystemTime) -> PvField { + let mut v = NTScalar::new(ScalarType::Double).with_display().create(); + set_scalar(&mut v, &["value"], ScalarValue::Double(value)); + let d = ts.duration_since(UNIX_EPOCH).unwrap(); + set_scalar( + &mut v, + &["timeStamp", "secondsPastEpoch"], + ScalarValue::Long(d.as_secs() as i64), + ); + set_scalar( + &mut v, + &["timeStamp", "nanoseconds"], + ScalarValue::Int(d.subsec_nanos() as i32), + ); + v +} + +fn secs(ts: SystemTime) -> u64 { + ts.duration_since(UNIX_EPOCH).unwrap().as_secs() +} + +struct Live { + _dir: tempfile::TempDir, + storage: Arc, + registry: Arc, + mgr: Arc, + pv: SharedPV, + server: PvaServer, + shutdown_tx: tokio::sync::watch::Sender, + pool: tokio::task::JoinHandle<()>, +} + +impl Live { + async fn start(initial: PvField) -> Self { + // Quiet unless RUST_LOG is set (e.g. archiver_engine=debug). + let _ = tracing_subscriber::fmt() + .with_env_filter(tracing_subscriber::EnvFilter::from_default_env()) + .with_test_writer() + .try_init(); + let dir = tempfile::tempdir().unwrap(); + let storage: Arc = Arc::new(PlainPbStoragePlugin::new( + "sts", + dir.path().join("sts"), + PartitionGranularity::Year, + )); + let registry = Arc::new(PvRegistry::open(&dir.path().join("registry.db")).unwrap()); + + let pv = SharedPV::build_readonly(); + pv.open(nt_desc(), initial).unwrap(); + let source = SharedSource::new(); + source.add(PV, pv.clone()); + let server = PvaServer::isolated(Arc::new(source)).expect("isolated PVA server"); + + // SAFETY: nextest runs each test in its own process; nothing else + // reads these variables concurrently. + unsafe { + std::env::set_var("EPICS_PVA_NAME_SERVERS", server.tcp_addr().to_string()); + std::env::set_var("EPICS_PVA_AUTO_ADDR_LIST", "NO"); + std::env::set_var("EPICS_PVA_ADDR_LIST", ""); + } + + let (mgr, rx) = ChannelManager::new(storage.clone(), registry.clone(), None) + .await + .unwrap(); + let mgr = Arc::new(mgr); + let (shutdown_tx, shutdown_rx) = tokio::sync::watch::channel(false); + let pool = tokio::spawn(run_sharded_write_pool( + storage.clone(), + registry.clone(), + rx, + shutdown_rx, + ShardedWritePoolConfig { + shards: 2, + per_shard_buffer: 1024, + write_loop: WriteLoopConfig { + flush_period: Duration::from_millis(300), + ..Default::default() + }, + }, + )); + Self { + _dir: dir, + storage, + registry, + mgr, + pv, + server, + shutdown_tx, + pool, + } + } + + /// Doubles stored on disk for `PV` in `[start, end)`, in stream order. + async fn stored(&self, start: SystemTime, end: SystemTime) -> Vec<(u64, f64)> { + let mut stream = query_data(&*self.storage, PV, start, end, None) + .await + .expect("query_data"); + let mut out = Vec::new(); + while let Some(s) = stream.next_event().expect("next_event") { + if let ArchiverValue::ScalarDouble(d) = s.value { + out.push((secs(s.timestamp), d)); + } else { + panic!("unexpected stored value {:?}", s.value); + } + } + out + } + + async fn wait_stored( + &self, + start: SystemTime, + end: SystemTime, + want: &[f64], + budget: Duration, + ) -> Vec<(u64, f64)> { + let deadline = tokio::time::Instant::now() + budget; + loop { + let got = self.stored(start, end).await; + if want.iter().all(|w| got.iter().any(|(_, v)| v == w)) { + return got; + } + assert!( + tokio::time::Instant::now() < deadline, + "timed out waiting for {want:?} on disk; have {got:?}; counters {:?}", + self.counters() + ); + tokio::time::sleep(Duration::from_millis(200)).await; + } + } + + fn counters(&self) -> PvCountersSnapshot { + self.mgr + .all_pv_counters() + .into_iter() + .find(|(n, _)| n == PV) + .map(|(_, c)| c) + .expect("counters for PV") + } + + async fn wait_connected(&self, want: bool, budget: Duration) { + let deadline = tokio::time::Instant::now() + budget; + loop { + let ci = self.mgr.get_connection_info(PV).expect("conn info"); + if ci.is_connected == want { + return; + } + assert!( + tokio::time::Instant::now() < deadline, + "timed out waiting for is_connected == {want}; state {:?}", + ci.state + ); + tokio::time::sleep(Duration::from_millis(100)).await; + } + } + + async fn shutdown(self) { + self.mgr.stop_pv(PV).await.unwrap(); + self.shutdown_tx.send(true).unwrap(); + tokio::time::timeout(Duration::from_secs(10), self.pool) + .await + .expect("write pool exits on shutdown") + .unwrap(); + self.server.stop(); + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn pva_monitor_samples_land_in_plainpb_and_registry() { + let t0 = SystemTime::now(); + let live = Live::start(nt_double(0.0, t0)).await; + + live.mgr + .archive_pv(PV, &SampleMode::Monitor, Protocol::Pva) + .await + .expect("archive_pv over PVA"); + + let values = [1.0, 2.0, 3.0, 4.0, 5.0]; + let mut last_ts = t0; + for (i, v) in values.iter().enumerate() { + last_ts = t0 + Duration::from_secs(i as u64 + 1); + let delivered = live.pv.try_post(nt_double(*v, last_ts)); + assert!(delivered > 0 || i == 0, "post {v} reached no subscriber"); + tokio::time::sleep(Duration::from_millis(50)).await; + } + + let got = live + .wait_stored( + t0 - Duration::from_secs(60), + t0 + Duration::from_secs(60), + &values, + Duration::from_secs(20), + ) + .await; + // Stream order is timestamp order; the connect-time value 0.0 at t0 + // may or may not precede them depending on when the monitor attached. + let doubles: Vec = got.iter().map(|(_, v)| *v).filter(|v| *v > 0.0).collect(); + assert_eq!(doubles, values, "on-disk sample order/content"); + for (ts, v) in &got { + if *v > 0.0 { + assert_eq!(*ts, secs(t0) + *v as u64, "IOC timestamp preserved for {v}"); + } + } + + // Registry (on-disk SQLite) reflects the PV and its newest committed sample. + let deadline = tokio::time::Instant::now() + Duration::from_secs(10); + let rec = loop { + let rec = live.registry.get_pv(PV).unwrap().expect("registry row"); + if rec.last_timestamp.map(secs) == Some(secs(last_ts)) { + break rec; + } + assert!( + tokio::time::Instant::now() < deadline, + "registry last_timestamp never reached {}: {:?}", + secs(last_ts), + rec.last_timestamp.map(secs) + ); + tokio::time::sleep(Duration::from_millis(200)).await; + }; + assert_eq!(rec.status, PvStatus::Active); + assert_eq!(rec.protocol, Protocol::Pva); + assert_eq!(rec.dbr_type, ArchDbType::ScalarDouble); + + let c = live.counters(); + assert_eq!(c.transient_error_count, 0, "{c:?}"); + assert_eq!(c.type_change_drops, 0, "{c:?}"); + assert_eq!(c.timestamp_drops, 0, "{c:?}"); + assert_eq!(c.buffer_overflow_drops, 0, "{c:?}"); + assert_eq!(c.storage_append_timeouts, 0, "{c:?}"); + assert_eq!(c.disconnect_count, 0, "{c:?}"); + assert!(c.events_stored >= values.len() as u64, "{c:?}"); + + live.shutdown().await; +} + +/// Server-side close/reopen against an IOC whose clock runs two hours +/// behind. The monitor's `MonitorConnEvent::Finished` must flip the PV +/// to disconnected without waiting for the 60 s watchdog, the fast +/// path must re-subscribe on its own, and the first sample after the +/// reconnect — outside the 30-minute drift window like every sample +/// from this IOC — must be stored through the `first_after_connect` +/// bypass instead of dropped. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn pva_reconnect_first_sample_bypasses_drift_filter() { + let t0 = SystemTime::now(); + let ioc_t0 = t0 - Duration::from_secs(2 * 3600); + let live = Live::start(nt_double(0.0, ioc_t0)).await; + live.mgr + .archive_pv(PV, &SampleMode::Monitor, Protocol::Pva) + .await + .expect("archive_pv over PVA"); + + let from = t0 - Duration::from_secs(3 * 3600); + let to = t0 + Duration::from_secs(60); + live.wait_stored(from, to, &[0.0], Duration::from_secs(20)) + .await; + live.wait_connected(true, Duration::from_secs(5)).await; + + live.pv.close(); + live.wait_connected(false, Duration::from_secs(15)).await; + let c = live.counters(); + assert_eq!(c.disconnect_count, 1, "{c:?}"); + + let ioc_t1 = ioc_t0 + Duration::from_secs(10); + live.pv.open(nt_desc(), nt_double(100.0, ioc_t1)).unwrap(); + live.wait_connected(true, Duration::from_secs(30)).await; + + let got = live + .wait_stored(from, to, &[0.0, 100.0], Duration::from_secs(20)) + .await; + assert!( + got.iter().any(|(ts, v)| *v == 100.0 && *ts == secs(ioc_t1)), + "reconnect sample stored with its own IOC timestamp: {got:?}" + ); + let c = live.counters(); + assert_eq!(c.timestamp_drops, 0, "{c:?}"); + assert_eq!(c.disconnect_count, 1, "{c:?}"); + + live.shutdown().await; +} + +/// Library contract the fast path's re-subscribe relies on: a +/// server-side `SharedPV::close()` ends the `pvmonitor_handle` stream +/// with `MonitorConnEvent::Finished` (pvxs parity: `SharedPV::close()` +/// destroys its `MonitorControlOp`s, whose destructor sends FINISH +/// before DESTROY_CHANNEL). The handle does not re-subscribe on its +/// own after `open()`; `monitor_loop_pva` does. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn library_pvmonitor_handle_finishes_on_server_close() { + use epics_rs::pva::client_native::ops_v2::MonitorConnEvent; + use std::sync::Mutex; + + let t0 = SystemTime::now(); + let pv = SharedPV::build_readonly(); + pv.open(nt_desc(), nt_double(0.0, t0)).unwrap(); + let source = SharedSource::new(); + source.add(PV, pv.clone()); + let server = PvaServer::isolated(Arc::new(source)).expect("isolated PVA server"); + let client = server.client_config(); + + let events: Arc>> = Arc::new(Mutex::new(Vec::new())); + let ev = events.clone(); + let handle = client + .pvmonitor_handle(PV, |_desc, _field| {}, move |e| ev.lock().unwrap().push(e)) + .await + .expect("pvmonitor_handle"); + + let wait_for = |pred: fn(&[MonitorConnEvent]) -> bool, budget: Duration, what: &'static str| { + let events = events.clone(); + async move { + let deadline = tokio::time::Instant::now() + budget; + while !pred(&events.lock().unwrap()) { + assert!( + tokio::time::Instant::now() < deadline, + "timed out waiting for {what}; events {:?}", + events.lock().unwrap() + ); + tokio::time::sleep(Duration::from_millis(100)).await; + } + } + }; + wait_for( + |e| { + e.iter() + .any(|x| matches!(x, MonitorConnEvent::Connected { .. })) + }, + Duration::from_secs(10), + "Connected", + ) + .await; + pv.close(); + wait_for( + |e| e.iter().any(|x| matches!(x, MonitorConnEvent::Finished)), + Duration::from_secs(15), + "Finished after close", + ) + .await; + drop(handle); + server.stop(); +} From 27eecab7b96237308d61b668c9b13a580b8108ee Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 00:33:53 +0900 Subject: [PATCH 03/55] api(mgmt): sort get_pvs_by_storage_consumed with sort_by_key Rust 1.97 clippy flags the reversed sort_by closure as unnecessary_sort_by, which failed the workspace -D warnings gate. --- crates/archiver-api/src/handlers/mgmt/reports.rs | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/crates/archiver-api/src/handlers/mgmt/reports.rs b/crates/archiver-api/src/handlers/mgmt/reports.rs index 439daf3..5002562 100644 --- a/crates/archiver-api/src/handlers/mgmt/reports.rs +++ b/crates/archiver-api/src/handlers/mgmt/reports.rs @@ -174,7 +174,7 @@ pub async fn get_pvs_by_storage_consumed( let files: u64 = summaries.iter().filter_map(|s| s.pv_file_count).sum(); entries.push((pv, bytes, files)); } - entries.sort_by(|a, b| b.1.cmp(&a.1)); + entries.sort_by_key(|e| std::cmp::Reverse(e.1)); entries.truncate(limit); let json: Vec = entries From 91d39c3551887da632866cd40175af8bb31ddfec Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 01:01:31 +0900 Subject: [PATCH 04/55] engine: waive only the past side of the drift window for the first sample The first-after-connect waiver exists for reconnect backfill, but it also let a far-future stamp through (an IOC booted before NTP sync), and shard_handle_sample then rejected every corrected sample as out-of-order until restart. accept_ioc_timestamp is now the single gate: the 1991 floor and the future bound hold for every sample. --- crates/archiver-engine/src/channel_manager.rs | 106 +++++++++++++----- tests/save_path_live.rs | 35 ++++++ 2 files changed, 114 insertions(+), 27 deletions(-) diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index d8ef52c..ea4203c 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -34,15 +34,26 @@ const CA_RETRY_DELAY: Duration = Duration::from_secs(5); /// stale uninitialised IOC clock or a sentinel. const PAST_CUTOFF_UNIX_SECS: i64 = 662_688_000; // 1991-01-01 00:00:00 UTC -/// Filter a freshly-received sample timestamp against the wall clock and -/// the floor. Returns `Some(ts)` if accepted, `None` if it should be -/// dropped (caller bumps `timestamp_drops`). +/// The one timestamp gate for every ingest path that carries an IOC +/// timestamp. Returns `true` if the sample is accepted, `false` if the +/// caller must drop it (and bump `timestamp_drops`). /// -/// `drift_secs` is the configured `server_ioc_drift_secs` (Java parity -/// 6538631), so per-site tuning doesn't require recompiling. `now` is -/// passed in (rather than calling `SystemTime::now()` here) so the test -/// suite can pin time deterministically. -fn ioc_timestamp_in_window(ts: SystemTime, now: SystemTime, drift_secs: u64) -> bool { +/// Three bounds: the 1991 floor, and `now ± drift_secs` +/// (`server_ioc_drift_secs`, Java parity 6538631). `first_after_connect` +/// waives only the past side — a reconnect backfill legitimately +/// carries older stamps. The floor and the future bound hold for every +/// sample: the write pool's per-PV monotonic guard rejects anything +/// older than the last accepted stamp, so a single accepted future +/// stamp (an IOC booted before NTP sync) would reject every correctly +/// stamped sample that follows until restart. +/// +/// `now` is passed in so tests can pin time deterministically. +fn accept_ioc_timestamp( + ts: SystemTime, + now: SystemTime, + drift_secs: u64, + first_after_connect: bool, +) -> bool { let unix = ts .duration_since(SystemTime::UNIX_EPOCH) .map(|d| d.as_secs() as i64) @@ -50,13 +61,58 @@ fn ioc_timestamp_in_window(ts: SystemTime, now: SystemTime, drift_secs: u64) -> if unix < PAST_CUTOFF_UNIX_SECS { return false; } - // Within ±drift_secs of `now`? let now_unix = now .duration_since(SystemTime::UNIX_EPOCH) .map(|d| d.as_secs() as i64) .unwrap_or(0); - let delta = (unix - now_unix).unsigned_abs(); - delta <= drift_secs + let drift = i64::try_from(drift_secs).unwrap_or(i64::MAX); + if unix > now_unix.saturating_add(drift) { + return false; + } + first_after_connect || unix >= now_unix.saturating_sub(drift) +} + +#[cfg(test)] +mod ioc_timestamp_gate_tests { + use super::*; + + const DRIFT: i64 = 1800; + const NOW: i64 = 1_700_000_000; + + fn at(unix: i64) -> SystemTime { + SystemTime::UNIX_EPOCH + Duration::from_secs(unix as u64) + } + + fn accept(unix: i64, first_after_connect: bool) -> bool { + accept_ioc_timestamp(at(unix), at(NOW), DRIFT as u64, first_after_connect) + } + + #[test] + fn window_edges_inclusive_either_way() { + for first in [false, true] { + assert!(accept(NOW - DRIFT, first)); + assert!(accept(NOW, first)); + assert!(accept(NOW + DRIFT, first)); + } + } + + #[test] + fn past_beyond_window_only_first_after_connect() { + assert!(!accept(NOW - DRIFT - 1, false)); + assert!(accept(NOW - DRIFT - 1, true)); + } + + #[test] + fn future_beyond_window_rejected_even_first_after_connect() { + assert!(!accept(NOW + DRIFT + 1, false)); + assert!(!accept(NOW + DRIFT + 1, true)); + } + + #[test] + fn below_1991_floor_rejected_even_first_after_connect() { + assert!(!accept(PAST_CUTOFF_UNIX_SECS - 1, true)); + assert!(accept(PAST_CUTOFF_UNIX_SECS, true)); + } } /// Discrete connection state for `getPVDetails` (Java parity dea7acb). @@ -1308,7 +1364,7 @@ fn pva_handle_event( }; let ts = pv_field_extract_timestamp(field); - if !first_after_connect && !ioc_timestamp_in_window(ts, now, drift_secs) { + if !accept_ioc_timestamp(ts, now, drift_secs, first_after_connect) { counters.timestamp_drops.fetch_add(1, Ordering::Relaxed); debug!( pv = pv_name, @@ -2017,14 +2073,11 @@ async fn monitor_loop( let now = SystemTime::now(); // Java parity (11e554d0): use the IOC-reported // timestamp, not receive-time, so latency - // doesn't smear sample times. First sample - // after connect is accepted unconditionally - // — legitimate backfill on reconnect can - // include older timestamps. Subsequent - // samples whose IOC clock is more than - // SERVER_IOC_DRIFT_SECS away from wall clock, - // or earlier than the 1991 floor, are - // dropped + counted. + // doesn't smear sample times. The first + // sample after connect may be older than the + // drift window (reconnect backfill); the 1991 + // floor and the future bound apply to every + // sample — see `accept_ioc_timestamp`. let first_after_connect = { let mut ci = conn_info.lock().unwrap_or_else(|e| e.into_inner()); let first = ci.last_event_time.is_none(); @@ -2036,13 +2089,12 @@ async fn monitor_loop( ci.state = PvConnectionState::Connected; first }; - if !first_after_connect - && !ioc_timestamp_in_window( - snapshot.timestamp.into(), - now, - server_ioc_drift_secs, - ) - { + if !accept_ioc_timestamp( + snapshot.timestamp.into(), + now, + server_ioc_drift_secs, + first_after_connect, + ) { counters.timestamp_drops.fetch_add(1, Ordering::Relaxed); debug!( pv = pv_name, diff --git a/tests/save_path_live.rs b/tests/save_path_live.rs index 9c8f2aa..8e17deb 100644 --- a/tests/save_path_live.rs +++ b/tests/save_path_live.rs @@ -325,6 +325,41 @@ async fn pva_reconnect_first_sample_bypasses_drift_filter() { live.shutdown().await; } +/// An IOC that connects with a far-future clock (booted before NTP +/// sync) and is corrected afterwards. The first-after-connect waiver +/// covers only the past side of the drift window; a future stamp is +/// dropped even on the first sample, because the write pool's per-PV +/// monotonic guard would otherwise reject every later, correctly +/// stamped sample until restart. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn future_first_sample_must_not_block_corrected_clock() { + let t0 = SystemTime::now(); + let future = t0 + Duration::from_secs(2 * 3600); + let live = Live::start(nt_double(0.0, future)).await; + live.mgr + .archive_pv(PV, &SampleMode::Monitor, Protocol::Pva) + .await + .expect("archive_pv over PVA"); + live.wait_connected(true, Duration::from_secs(5)).await; + tokio::time::sleep(Duration::from_secs(1)).await; + + live.pv + .try_post(nt_double(1.0, t0 + Duration::from_secs(1))); + let from = t0 - Duration::from_secs(60); + let to = t0 + Duration::from_secs(3 * 3600); + let got = live + .wait_stored(from, to, &[1.0], Duration::from_secs(15)) + .await; + assert!( + !got.iter().any(|(_, v)| *v == 0.0), + "future-stamped sample must not be archived: {got:?}" + ); + let c = live.counters(); + assert_eq!(c.timestamp_drops, 1, "{c:?}"); + + live.shutdown().await; +} + /// Library contract the fast path's re-subscribe relies on: a /// server-side `SharedPV::close()` ends the `pvmonitor_handle` stream /// with `MonitorConnEvent::Finished` (pvxs parity: `SharedPV::close()` From 5a38cb079189619ebe7218b9efaaf36c4b4f6f44 Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 01:08:29 +0900 Subject: [PATCH 05/55] engine: count storage append errors in PvCounters::storage_write_errors A failed or panicked append_event_with_meta was only an error! line, so a PV losing every sample to ENOSPC or a codec mismatch looked healthy on getPVStatus and in Prometheus. Both terminal branches of shard_handle_sample now route through record_write_error; the API DTO and the engine JSON carry the field as storageWriteErrors. --- .../archiver-api/src/handlers/mgmt/engine.rs | 1 + .../src/services/fakes/archiver_control.rs | 1 + .../src/services/impls/archiver_control.rs | 1 + crates/archiver-api/src/services/traits.rs | 3 +++ crates/archiver-engine/src/channel_manager.rs | 22 +++++++++++++++++++ tests/write_loop_failure.rs | 7 ++++++ 6 files changed, 35 insertions(+) diff --git a/crates/archiver-api/src/handlers/mgmt/engine.rs b/crates/archiver-api/src/handlers/mgmt/engine.rs index 5f00d6d..767ecf6 100644 --- a/crates/archiver-api/src/handlers/mgmt/engine.rs +++ b/crates/archiver-api/src/handlers/mgmt/engine.rs @@ -314,6 +314,7 @@ pub async fn pv_status_action( "bufferOverflowDrops": c.buffer_overflow_drops, "timestampDrops": c.timestamp_drops, "typeChangeDrops": c.type_change_drops, + "storageWriteErrors": c.storage_write_errors, "shardClosedDrops": c.shard_closed_drops, "shutdownAbandonedDrops": c.shutdown_abandoned_drops, "disconnectCount": c.disconnect_count, diff --git a/crates/archiver-api/src/services/fakes/archiver_control.rs b/crates/archiver-api/src/services/fakes/archiver_control.rs index a1ce70a..37f0437 100644 --- a/crates/archiver-api/src/services/fakes/archiver_control.rs +++ b/crates/archiver-api/src/services/fakes/archiver_control.rs @@ -105,6 +105,7 @@ impl ArchiverQuery for FakeArchiverControl { latest_observed_dbr: None, metadata_fetch_failures: 0, storage_append_timeouts: 0, + storage_write_errors: 0, shard_closed_drops: 0, shutdown_abandoned_drops: 0, }, diff --git a/crates/archiver-api/src/services/impls/archiver_control.rs b/crates/archiver-api/src/services/impls/archiver_control.rs index 1348831..d909470 100644 --- a/crates/archiver-api/src/services/impls/archiver_control.rs +++ b/crates/archiver-api/src/services/impls/archiver_control.rs @@ -58,6 +58,7 @@ impl ArchiverQuery for ChannelArchiverControl { latest_observed_dbr: c.latest_observed_dbr, metadata_fetch_failures: c.metadata_fetch_failures, storage_append_timeouts: c.storage_append_timeouts, + storage_write_errors: c.storage_write_errors, shard_closed_drops: c.shard_closed_drops, shutdown_abandoned_drops: c.shutdown_abandoned_drops, }, diff --git a/crates/archiver-api/src/services/traits.rs b/crates/archiver-api/src/services/traits.rs index 374f758..1815175 100644 --- a/crates/archiver-api/src/services/traits.rs +++ b/crates/archiver-api/src/services/traits.rs @@ -117,6 +117,9 @@ pub struct PvCountersDto { /// so operators can tell a wedged storage tier apart from a /// slow writer. pub storage_append_timeouts: u64, + /// Storage `append_event_with_meta` calls that returned an error or + /// panicked — the sample is lost. + pub storage_write_errors: u64, /// Events dropped because the shard's channel was closed (worker /// death / respawn budget spent) — distinct from /// `buffer_overflow_drops` (channel full but the worker is alive). diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index ea4203c..582e6da 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -210,6 +210,12 @@ pub struct PvCounters { /// operator can tell "the storage tier is wedged" apart from /// "the writer can't keep up with the producer". pub storage_append_timeouts: AtomicU64, + /// Number of `storage.append_event_with_meta` calls that returned an + /// error or panicked — the sample is gone (disk full, EIO, codec + /// mismatch). Distinct from `storage_append_timeouts` (the write may + /// still land late) and `buffer_overflow_drops` (never reached + /// storage). Aggregated as `archiver_storage_write_errors_total`. + pub storage_write_errors: AtomicU64, /// Number of events dropped because the shard's channel was closed — /// the worker task died (loop panic) or its respawn budget was spent. /// Distinct from `buffer_overflow_drops` (channel full but the worker @@ -240,6 +246,7 @@ impl Default for PvCounters { latest_observed_dbr: AtomicI32::new(-1), metadata_fetch_failures: AtomicU64::new(0), storage_append_timeouts: AtomicU64::new(0), + storage_write_errors: AtomicU64::new(0), shard_closed_drops: AtomicU64::new(0), shutdown_abandoned_drops: AtomicU64::new(0), } @@ -264,6 +271,7 @@ pub struct PvCountersSnapshot { pub latest_observed_dbr: Option, pub metadata_fetch_failures: u64, pub storage_append_timeouts: u64, + pub storage_write_errors: u64, pub shard_closed_drops: u64, pub shutdown_abandoned_drops: u64, } @@ -292,6 +300,7 @@ impl From<&PvCounters> for PvCountersSnapshot { }, metadata_fetch_failures: c.metadata_fetch_failures.load(Ordering::Relaxed), storage_append_timeouts: c.storage_append_timeouts.load(Ordering::Relaxed), + storage_write_errors: c.storage_write_errors.load(Ordering::Relaxed), shard_closed_drops: c.shard_closed_drops.load(Ordering::Relaxed), shutdown_abandoned_drops: c.shutdown_abandoned_drops.load(Ordering::Relaxed), } @@ -4070,9 +4079,11 @@ async fn shard_handle_sample( // Report already went out from inside the closure. } Ok(Ok(Err(e))) => { + record_write_error(&counters_for_post); error!(shard = shard_idx, pv = pv_name_for_post, "Write error: {e}"); } Ok(Err(join_err)) => { + record_write_error(&counters_for_post); error!( shard = shard_idx, pv = pv_name_for_post, @@ -4099,6 +4110,17 @@ async fn shard_handle_sample( } } +/// Account a sample lost to a storage append error or panic. Single +/// owner of `storage_write_errors`: both terminal failure branches of +/// `shard_handle_sample` route through here, so the loss is visible +/// on the PV's counters and in Prometheus, not only in the log. +fn record_write_error(counters: &Option>) { + if let Some(c) = counters.as_ref() { + c.storage_write_errors.fetch_add(1, Ordering::Relaxed); + } + metrics::counter!("archiver_storage_write_errors_total").increment(1); +} + /// Drain the shard's remaining buffered samples on shutdown, /// applying the same bounded-timeout discipline as the hot path. async fn shard_drain_on_shutdown( diff --git a/tests/write_loop_failure.rs b/tests/write_loop_failure.rs index 8e52e0f..1b0807e 100644 --- a/tests/write_loop_failure.rs +++ b/tests/write_loop_failure.rs @@ -1034,6 +1034,13 @@ async fn shard_error_bumps_last_ts_for_ordering() { .unwrap(); tokio::time::sleep(Duration::from_millis(100)).await; + // The errored append is a lost sample; it must be visible on the + // PV's own counters, not only in the error log. + assert_eq!( + counters.storage_write_errors.load(Ordering::Relaxed), + 1, + "storage append error must increment storage_write_errors" + ); // The older sample must have been dropped at the shard's // ordering check, not reached storage. assert!( From a8823dfde014b83150db77e4ed310e30db86342b Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 01:11:32 +0900 Subject: [PATCH 06/55] engine: make pv_field_extract_timestamp overflow-safe A negative secondsPastEpoch was reinterpreted as a huge u64, and Duration::new / UNIX_EPOCH + d panic on overflow. The panic unwinds inside the epics-rs reactor task that owns the subscription, so one malformed timeStamp from one IOC silently ended the monitor. Treat any unrepresentable timeStamp as "no usable timeStamp" (fall back to now), which is the documented contract of this function. --- crates/archiver-engine/src/channel_manager.rs | 108 ++++++++++++++++-- 1 file changed, 99 insertions(+), 9 deletions(-) diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index 582e6da..8c35841 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -2959,19 +2959,31 @@ fn pv_field_extract_timestamp(field: &PvField) -> SystemTime { let Some(PvField::Structure(ts)) = s.get_field("timeStamp") else { return SystemTime::now(); }; + // Checked conversions throughout: a negative or absurd + // `secondsPastEpoch` (an IOC with an unset clock, a corrupt + // structure) is "not a usable timeStamp" per the contract above, + // not a reason to panic the reactor task — `Duration::new` and + // `UNIX_EPOCH + d` both panic on overflow, and a panic here kills + // the subscription for every PV sharing the reactor. let secs = match ts.get_field("secondsPastEpoch") { - Some(PvField::Scalar(ScalarValue::Long(v))) => *v as u64, - Some(PvField::Scalar(ScalarValue::ULong(v))) => *v, - Some(PvField::Scalar(ScalarValue::Int(v))) => *v as u64, - Some(PvField::Scalar(ScalarValue::UInt(v))) => *v as u64, - _ => return SystemTime::now(), + Some(PvField::Scalar(ScalarValue::Long(v))) => u64::try_from(*v).ok(), + Some(PvField::Scalar(ScalarValue::ULong(v))) => Some(*v), + Some(PvField::Scalar(ScalarValue::Int(v))) => u64::try_from(*v).ok(), + Some(PvField::Scalar(ScalarValue::UInt(v))) => Some(u64::from(*v)), + _ => None, }; let nanos = match ts.get_field("nanoseconds") { - Some(PvField::Scalar(ScalarValue::Int(v))) => *v as u32, - Some(PvField::Scalar(ScalarValue::UInt(v))) => *v, - _ => 0, + Some(PvField::Scalar(ScalarValue::Int(v))) => u32::try_from(*v).ok(), + Some(PvField::Scalar(ScalarValue::UInt(v))) => Some(*v), + _ => Some(0), + }; + let (Some(secs), Some(nanos)) = (secs, nanos) else { + return SystemTime::now(); }; - SystemTime::UNIX_EPOCH + Duration::new(secs, nanos) + Duration::from_secs(secs) + .checked_add(Duration::from_nanos(u64::from(nanos))) + .and_then(|d| SystemTime::UNIX_EPOCH.checked_add(d)) + .unwrap_or_else(SystemTime::now) } /// Convert epics-base-rs DbFieldType to archiver ArchDbType. @@ -4323,6 +4335,84 @@ mod pva_mapping_tests { PvField::Structure(table) } + /// NTScalar with an explicit `timeStamp` sub-structure; `nanos` + /// `None` omits the `nanoseconds` field entirely. + fn nt_with_timestamp(secs: ScalarValue, nanos: Option) -> PvField { + let mut ts = PvStructure::new("time_t"); + ts.fields + .push(("secondsPastEpoch".into(), PvField::Scalar(secs))); + if let Some(n) = nanos { + ts.fields.push(("nanoseconds".into(), PvField::Scalar(n))); + } + let mut nt = PvStructure::new("epics:nt/NTScalar:1.0"); + nt.fields + .push(("value".into(), PvField::Scalar(ScalarValue::Double(1.0)))); + nt.fields.push(("timeStamp".into(), PvField::Structure(ts))); + PvField::Structure(nt) + } + + fn assert_about_now(t: SystemTime, what: &str) { + let skew = t + .duration_since(SystemTime::now()) + .or_else(|e| Ok::<_, ()>(e.duration())) + .unwrap(); + assert!( + skew < Duration::from_secs(5), + "{what}: expected ~now, got {t:?}" + ); + } + + #[test] + fn timestamp_well_formed_is_exact() { + let t = pv_field_extract_timestamp(&nt_with_timestamp( + ScalarValue::Long(1_700_000_000), + Some(ScalarValue::Int(250_000_000)), + )); + assert_eq!( + t, + SystemTime::UNIX_EPOCH + Duration::new(1_700_000_000, 250_000_000) + ); + // Sub-second overflow carries into seconds rather than panicking. + let t = pv_field_extract_timestamp(&nt_with_timestamp( + ScalarValue::ULong(1_700_000_000), + Some(ScalarValue::UInt(1_500_000_000)), + )); + assert_eq!( + t, + SystemTime::UNIX_EPOCH + Duration::new(1_700_000_001, 500_000_000) + ); + } + + #[test] + fn timestamp_negative_seconds_falls_back_to_now() { + let t = pv_field_extract_timestamp(&nt_with_timestamp(ScalarValue::Long(-1), None)); + assert_about_now(t, "Long(-1)"); + let t = pv_field_extract_timestamp(&nt_with_timestamp(ScalarValue::Int(-7), None)); + assert_about_now(t, "Int(-7)"); + } + + #[test] + fn timestamp_negative_nanos_falls_back_to_now() { + let t = pv_field_extract_timestamp(&nt_with_timestamp( + ScalarValue::Long(1_700_000_000), + Some(ScalarValue::Int(-5)), + )); + assert_about_now(t, "nanoseconds Int(-5)"); + } + + #[test] + fn timestamp_overflowing_seconds_does_not_panic() { + // Would overflow `SystemTime` (i64 seconds on Linux) — the old + // `UNIX_EPOCH + Duration::new(..)` panicked in the reactor task. + let t = pv_field_extract_timestamp(&nt_with_timestamp(ScalarValue::ULong(u64::MAX), None)); + assert_about_now(t, "ULong(u64::MAX)"); + let t = pv_field_extract_timestamp(&nt_with_timestamp( + ScalarValue::Long(i64::MAX), + Some(ScalarValue::UInt(u32::MAX)), + )); + assert_about_now(t, "Long(i64::MAX)"); + } + #[test] fn nttable_classifies_as_v4_generic_bytes() { let pv = make_nttable(); From ffc777167df993a9cfe1a62cad4943fe105f8b5f Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 01:11:37 +0900 Subject: [PATCH 07/55] engine: log PVA samples dropped by pva_handle_event's try_send The Full case was counted but never logged and the Closed case was neither, so a PVA PV losing samples to a saturated write channel left no trace in the log to correlate with buffer_overflow_drops. --- crates/archiver-engine/src/channel_manager.rs | 20 +++++++++++++++---- 1 file changed, 16 insertions(+), 4 deletions(-) diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index 8c35841..528ddda 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -1401,10 +1401,22 @@ fn pva_handle_event( element_count: Some(elem_count), counters: Some(counters.clone()), }; - if let Err(tokio::sync::mpsc::error::TrySendError::Full(_)) = tx.try_send(pv_sample) { - counters - .buffer_overflow_drops - .fetch_add(1, Ordering::Relaxed); + // Non-blocking by design (this runs on the epics-rs reactor task), + // so a full channel drops the sample: count it AND say so. A + // closed channel means the write pool is gone (shutdown); the + // sample is dropped either way, and a silent drop of both cases + // was indistinguishable from a healthy PV in the logs. + match tx.try_send(pv_sample) { + Ok(()) => {} + Err(tokio::sync::mpsc::error::TrySendError::Full(_)) => { + counters + .buffer_overflow_drops + .fetch_add(1, Ordering::Relaxed); + debug!(pv = pv_name, "write channel full; PVA sample dropped"); + } + Err(tokio::sync::mpsc::error::TrySendError::Closed(_)) => { + debug!(pv = pv_name, "write channel closed; PVA sample dropped"); + } } } From 8b00d1669ec5d3b7fbb98490e2ee3cc1bc74e96c Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 01:14:42 +0900 Subject: [PATCH 08/55] storage: truncate in file_needs_header only on an undecodable header Any PbFileReader::open failure was treated as a corrupt header and the partition truncated, so a transient EMFILE/EACCES/EIO on the probe wiped every sample already in the file. Only a header that was read and failed to decode is corrupt; stat/open/read errors and a failed truncate or trim now fail the append instead. --- .../archiver-core/src/storage/plainpb/mod.rs | 131 +++++++++++++----- 1 file changed, 99 insertions(+), 32 deletions(-) diff --git a/crates/archiver-core/src/storage/plainpb/mod.rs b/crates/archiver-core/src/storage/plainpb/mod.rs index 9774875..017d1d7 100644 --- a/crates/archiver-core/src/storage/plainpb/mod.rs +++ b/crates/archiver-core/src/storage/plainpb/mod.rs @@ -1379,7 +1379,7 @@ impl PlainPbStoragePlugin { } if slot.writer.is_none() { - let needs_header = file_needs_header(path); + let needs_header = file_needs_header(path)?; // Whether the open below will CREATE the file (add a new // directory entry). Reliable under the per-PV slot lock: no // other writer targets this path. Gates the parent-dir @@ -1677,48 +1677,63 @@ fn is_too_many_open_files(e: &std::io::Error) -> bool { } /// Decide whether a file at `path` needs a fresh PayloadInfo header -/// written before sample data is appended. Returns `true` when: +/// written before sample data is appended. Returns `Ok(true)` when: /// 1. the file doesn't exist, /// 2. the file exists but is 0 bytes (Java parity 651c3a6b: a crash /// mid-create would otherwise leave a header-less file), OR -/// 3. the file exists with bytes but `PbFileReader::open` cannot -/// parse the header — in that case we **truncate** the file so -/// the caller's `create+append` opens cleanly. Without this -/// third branch, a partial-header crash makes the file forever -/// unreadable AND every subsequent append silently piles garbage -/// onto a corrupt prefix. -fn file_needs_header(path: &Path) -> bool { - if !path.exists() { - return true; - } - let size = std::fs::metadata(path).map(|m| m.len()).unwrap_or(0); +/// 3. the file is readable but its header line does not decode — in +/// that case we **truncate** the file so the caller's +/// `create+append` opens cleanly. Without this third branch, a +/// partial-header crash makes the file forever unreadable AND +/// every subsequent append silently piles garbage onto a corrupt +/// prefix. +/// +/// Only branch 3 — a header that was READ and failed to DECODE — may +/// truncate. An I/O error while stat-ing or reading the file (EACCES, +/// EIO, EMFILE, a hung NFS mount) is propagated instead: the file's +/// contents are unknown, and treating "cannot read" as "corrupt" wiped +/// a whole partition of good samples every time the fd table or the +/// mount hiccupped. The caller fails this one append; the next sample +/// re-probes. A failed truncate is likewise an error, not `true`: +/// returning `true` would append a second header mid-file. +fn file_needs_header(path: &Path) -> anyhow::Result { + let size = match std::fs::metadata(path) { + Ok(m) => m.len(), + Err(e) if e.kind() == std::io::ErrorKind::NotFound => return Ok(true), + Err(e) => return Err(anyhow::Error::from(e).context(format!("stat {path:?}"))), + }; if size == 0 { - return true; + return Ok(true); } - if PbFileReader::open(path).is_err() { - tracing::warn!( - ?path, - "PB file has unreadable header; truncating so a fresh \ - header gets written" - ); - if let Err(e) = std::fs::OpenOptions::new() - .write(true) - .truncate(true) - .open(path) - { - tracing::warn!(?path, "Failed to truncate corrupt PB file: {e}"); + match PbFileReader::open(path) { + Ok(_) => {} + Err(e) if e.downcast_ref::().is_some() => { + return Err(e.context(format!("cannot read PB header of {path:?}"))); + } + Err(e) => { + tracing::warn!( + ?path, + "PB file has undecodable header ({e}); truncating so a \ + fresh header gets written" + ); + std::fs::OpenOptions::new() + .write(true) + .truncate(true) + .open(path) + .map_err(|e| anyhow::anyhow!("failed to truncate corrupt PB file {path:?}: {e}"))?; + return Ok(true); } - return true; } // Header is valid; defend against a tail with a partial sample // (writer killed mid-flush). Truncating to the last NEWLINE // boundary loses at most one record but keeps the file readable // — without this, a reader hits the partial record and stops - // returning every sample after that point. - if let Err(e) = trim_to_last_newline(path) { - tracing::warn!(?path, "Failed to trim partial trailing record: {e}"); - } - false + // returning every sample after that point. A failed trim leaves + // the partial record in place, so the append must not proceed + // (it would fuse the new frame onto the partial one). + trim_to_last_newline(path) + .map_err(|e| anyhow::anyhow!("failed to trim partial trailing record in {path:?}: {e}"))?; + Ok(false) } /// Truncate `path` to end at its last NEWLINE byte (inclusive). Used @@ -2448,6 +2463,58 @@ fn total_pb_stats(root: &Path) -> (u64, u64) { mod tests { use super::*; + /// A header that cannot be READ is not a corrupt header: the stat + /// and the open/read error paths must propagate and leave the + /// file untouched, not truncate it. + #[cfg(unix)] + #[test] + fn file_needs_header_io_error_propagates_without_truncating() { + let dir = tempfile::tempdir().unwrap(); + + // stat error (ELOOP): a self-referential symlink. + let looped = dir.path().join("loop:2024.pb"); + std::os::unix::fs::symlink(&looped, &looped).unwrap(); + let err = file_needs_header(&looped).expect_err("ELOOP must propagate"); + assert!( + err.to_string().contains("stat"), + "unexpected error: {err:#}" + ); + assert!( + std::fs::symlink_metadata(&looped).is_ok(), + "symlink must survive an unreadable-header probe" + ); + + // read error (EISDIR): a directory in the partition's place. + // `File::open` succeeds on a directory; the header read fails. + let dirp = dir.path().join("dir:2024.pb"); + std::fs::create_dir(&dirp).unwrap(); + std::fs::write(dirp.join("marker"), b"x").unwrap(); + assert!( + std::fs::metadata(&dirp).unwrap().len() > 0, + "test premise: a non-empty directory reports a non-zero size" + ); + let err = file_needs_header(&dirp).expect_err("EISDIR must propagate"); + assert!( + err.to_string().contains("cannot read PB header"), + "unexpected error: {err:#}" + ); + assert!( + dirp.join("marker").exists(), + "directory must survive an unreadable-header probe" + ); + } + + /// The one case that MAY truncate: bytes were read and the header + /// does not decode. + #[test] + fn file_needs_header_truncates_only_undecodable_header() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("bad:2024.pb"); + std::fs::write(&path, b"\xff\xff\xffnot a payload info\n").unwrap(); + assert!(file_needs_header(&path).unwrap()); + assert_eq!(std::fs::metadata(&path).unwrap().len(), 0); + } + #[test] fn canonical_pv_key_matches_path_derived_name() { let plugin = From 60fc606c57faea36ace8f88372fbfdb2673e4d3b Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 01:15:43 +0900 Subject: [PATCH 09/55] storage: treat only a definite absence as a ghost file in write_cached Path::exists folds a stat error into false, so an EACCES/EIO/ENOTDIR blip on the cached writer's path discarded its dirty buffer as "file gone" although the inode was still there. try_exists keeps the buffer unless the path is positively absent. --- .../archiver-core/src/storage/plainpb/mod.rs | 7 +- tests/plainpb_compat.rs | 64 +++++++++++++++++++ 2 files changed, 70 insertions(+), 1 deletion(-) diff --git a/crates/archiver-core/src/storage/plainpb/mod.rs b/crates/archiver-core/src/storage/plainpb/mod.rs index 017d1d7..938c10e 100644 --- a/crates/archiver-core/src/storage/plainpb/mod.rs +++ b/crates/archiver-core/src/storage/plainpb/mod.rs @@ -1365,8 +1365,13 @@ impl PlainPbStoragePlugin { // Drop via `drop_writer_file_gone` so dirty bytes get a loss marker // (flushing to a deleted inode is meaningless; bytes are lost // regardless). + // + // Only a definite `Ok(false)` counts as gone. A stat ERROR + // (EACCES, EIO, ENOTDIR from a mount flap) says nothing about + // the inode, and `Path::exists` folds it into `false` — which + // discarded the dirty buffer of a file that was still there. if let Some(existing) = slot.writer.as_ref() - && !existing.path.exists() + && matches!(existing.path.try_exists(), Ok(false)) { tracing::warn!( pv, diff --git a/tests/plainpb_compat.rs b/tests/plainpb_compat.rs index 7c0e8c5..8760a83 100644 --- a/tests/plainpb_compat.rs +++ b/tests/plainpb_compat.rs @@ -4,7 +4,9 @@ use std::time::SystemTime; use archiver_core::storage::partition::PartitionGranularity; use archiver_core::storage::plainpb::codec; +use archiver_core::storage::plainpb::reader::PbFileReader; use archiver_core::storage::plainpb::{FdBudget, PlainPbStoragePlugin}; +use archiver_core::storage::traits::EventStream; use archiver_core::storage::traits::StoragePlugin; use archiver_core::types::{ArchDbType, ArchiverSample, ArchiverValue}; @@ -793,6 +795,68 @@ async fn ghost_file_path_records_loss() { ); } +/// A stat ERROR on the cached writer's path is not "file gone": the +/// dirty buffer must be kept and land once the path is stat-able +/// again. Provoked with ENOTDIR — the partition's parent directory is +/// swapped for a regular file while the writer's fd stays open. +#[cfg(unix)] +#[tokio::test] +async fn ghost_check_stat_error_keeps_dirty_writer() { + let dir = temp_dir(); + let plugin = + PlainPbStoragePlugin::new("test", dir.path().to_path_buf(), PartitionGranularity::Hour); + + let ts1: SystemTime = Utc.with_ymd_and_hms(2026, 9, 1, 12, 0, 0).unwrap().into(); + let ts2 = ts1 + std::time::Duration::from_secs(60); + let pv = "TEST:GhostStat"; + + let s1 = ArchiverSample::new(ts1, ArchiverValue::ScalarDouble(1.0)); + plugin + .append_event(pv, ArchDbType::ScalarDouble, &s1) + .await + .unwrap(); + + // Make stat(path) fail with ENOTDIR without touching the inode. + let path = plugin.file_path_for(pv, ts1); + let parent = path.parent().unwrap().to_path_buf(); + let aside = dir.path().join("aside"); + std::fs::rename(&parent, &aside).unwrap(); + std::fs::write(&parent, b"not a directory").unwrap(); + assert!( + path.try_exists().is_err(), + "test premise: stat must error, not report absent" + ); + + let s2 = ArchiverSample::new(ts2, ArchiverValue::ScalarDouble(2.0)); + plugin + .append_event(pv, ArchDbType::ScalarDouble, &s2) + .await + .unwrap(); + + // Restore the directory; the writer's fd still points at the file. + std::fs::remove_file(&parent).unwrap(); + std::fs::rename(&aside, &parent).unwrap(); + plugin.flush_writes().await.unwrap(); + + assert!( + plugin.take_loss_markers().is_empty(), + "a stat error must not be classified as loss" + ); + let mut rdr = PbFileReader::open(&path).unwrap(); + let mut vals = Vec::new(); + while let Some(s) = rdr.next_event().unwrap() { + vals.push(s.value); + } + assert_eq!( + vals, + vec![ + ArchiverValue::ScalarDouble(1.0), + ArchiverValue::ScalarDouble(2.0) + ], + "both samples must land in the file that never went away" + ); +} + /// `evict_writer_for_path`: when ETL removes a `.pb` file and /// then evicts its writer, dirty bytes must surface as loss. /// Mirrors the ghost-file path but with the writer eviction From c2fc72e900ebfeaf03e56eb2cfc84616313a4302 Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 01:17:21 +0900 Subject: [PATCH 10/55] storage: recreate a vanished partition directory in write_cached known_dirs is a positive-only cache, so once an operator removed a PV's directory every append under that prefix failed with NotFound until restart. The open is the one place that learns the truth: on NotFound it forgets the cached parent, recreates the chain and retries once. --- .../archiver-core/src/storage/plainpb/mod.rs | 30 ++++++++++++++ tests/plainpb_compat.rs | 39 +++++++++++++++++++ 2 files changed, 69 insertions(+) diff --git a/crates/archiver-core/src/storage/plainpb/mod.rs b/crates/archiver-core/src/storage/plainpb/mod.rs index 938c10e..d8dcddb 100644 --- a/crates/archiver-core/src/storage/plainpb/mod.rs +++ b/crates/archiver-core/src/storage/plainpb/mod.rs @@ -1209,6 +1209,17 @@ impl PlainPbStoragePlugin { } /// Ensure a parent directory exists, using a cached set to skip repeated syscalls. + /// Drop `path`'s parent from the `known_dirs` cache. The cache is + /// positive-only (a directory once created is assumed to persist); + /// this is its single invalidation point, used when an open reports + /// NotFound for a path whose parent the cache still claims. + fn forget_parent_dir(&self, path: &Path) { + if let Some(parent) = path.parent() { + let mut dirs = self.known_dirs.lock().unwrap_or_else(|e| e.into_inner()); + dirs.remove(parent); + } + } + fn ensure_parent_dir(&self, path: &Path) -> anyhow::Result<()> { if let Some(parent) = path.parent() { let needs_create = { @@ -1438,6 +1449,25 @@ impl PlainPbStoragePlugin { .append(true) .open(path)? } + Err(e) if e.kind() == std::io::ErrorKind::NotFound => { + // `create(true)` can only report NotFound when the + // parent directory is gone — removed after + // `known_dirs` recorded it (operator cleanup, a + // remount). The cache is positive-only, so without + // this every append under that prefix failed until + // restart. Forget the entry, recreate, retry once. + tracing::warn!( + ?path, + "partition directory vanished; recreating and \ + retrying open" + ); + self.forget_parent_dir(path); + self.ensure_parent_dir(path)?; + std::fs::OpenOptions::new() + .create(true) + .append(true) + .open(path)? + } Err(e) => return Err(e.into()), }; diff --git a/tests/plainpb_compat.rs b/tests/plainpb_compat.rs index 8760a83..d28fe61 100644 --- a/tests/plainpb_compat.rs +++ b/tests/plainpb_compat.rs @@ -795,6 +795,45 @@ async fn ghost_file_path_records_loss() { ); } +/// The partition directory is removed out from under a plugin that +/// already created it (operator cleanup). The next append must +/// recreate it and land, not fail on the stale `known_dirs` entry +/// until restart. +#[tokio::test] +async fn append_recreates_partition_dir_removed_after_creation() { + let dir = temp_dir(); + let plugin = + PlainPbStoragePlugin::new("test", dir.path().to_path_buf(), PartitionGranularity::Hour); + + let ts1: SystemTime = Utc.with_ymd_and_hms(2026, 9, 1, 12, 0, 0).unwrap().into(); + let ts2 = ts1 + std::time::Duration::from_secs(60); + let pv = "TEST:DirGone"; + + let s1 = ArchiverSample::new(ts1, ArchiverValue::ScalarDouble(1.0)); + plugin + .append_event(pv, ArchDbType::ScalarDouble, &s1) + .await + .unwrap(); + plugin.flush_writes().await.unwrap(); + + let path = plugin.file_path_for(pv, ts1); + std::fs::remove_dir_all(path.parent().unwrap()).unwrap(); + + let s2 = ArchiverSample::new(ts2, ArchiverValue::ScalarDouble(2.0)); + plugin + .append_event(pv, ArchDbType::ScalarDouble, &s2) + .await + .expect("append must recreate the vanished partition directory"); + plugin.flush_writes().await.unwrap(); + + let mut rdr = PbFileReader::open(&path).expect("partition recreated with a header"); + let mut vals = Vec::new(); + while let Some(s) = rdr.next_event().unwrap() { + vals.push(s.value); + } + assert_eq!(vals, vec![ArchiverValue::ScalarDouble(2.0)]); +} + /// A stat ERROR on the cached writer's path is not "file gone": the /// dirty buffer must be kept and land once the path is stat-able /// again. Provoked with ENOTDIR — the partition's parent directory is From bc2c5deed2e47814c4f8cb0f336411a40b60f7f3 Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 01:18:20 +0900 Subject: [PATCH 11/55] registry: upsert in register_pv_with_protocol instead of INSERT OR REPLACE REPLACE deletes the row and re-inserts only the named columns, so a repeated archivePV for an existing PV nulled last_timestamp, prec, egu, alias_for, archive_fields and policy_name. ON CONFLICT DO UPDATE touches only the columns this call owns. --- crates/archiver-core/src/registry.rs | 45 ++++++++++++++++++++++++++-- 1 file changed, 43 insertions(+), 2 deletions(-) diff --git a/crates/archiver-core/src/registry.rs b/crates/archiver-core/src/registry.rs index 2d2a1bb..575a627 100644 --- a/crates/archiver-core/src/registry.rs +++ b/crates/archiver-core/src/registry.rs @@ -350,10 +350,24 @@ impl PvRegistry { let now = Utc::now().to_rfc3339(); let (mode_str, period) = sample_mode.to_db(); + // UPSERT, not INSERT OR REPLACE: REPLACE deletes the row and + // re-inserts only the columns named here, which nulled + // last_timestamp, prec, egu, alias_for, archive_fields and + // policy_name every time an already-archived PV was + // re-registered (a repeated archivePV, a restart-time + // re-archive). Only the columns this call owns are updated. conn.execute( - "INSERT OR REPLACE INTO pv_info + "INSERT INTO pv_info (pv_name, dbr_type, sample_mode, sample_period, status, element_count, created_at, updated_at, protocol) - VALUES (?1, ?2, ?3, ?4, 'active', ?5, COALESCE((SELECT created_at FROM pv_info WHERE pv_name = ?1), ?6), ?6, ?7)", + VALUES (?1, ?2, ?3, ?4, 'active', ?5, ?6, ?6, ?7) + ON CONFLICT(pv_name) DO UPDATE SET + dbr_type = excluded.dbr_type, + sample_mode = excluded.sample_mode, + sample_period = excluded.sample_period, + status = 'active', + element_count = excluded.element_count, + updated_at = excluded.updated_at, + protocol = excluded.protocol", params![pv_name, dbr_type as i32, mode_str, period, element_count, now, protocol.as_str()], )?; Ok(()) @@ -950,6 +964,33 @@ fn is_duplicate_column_error(e: &rusqlite::Error) -> bool { mod tests { use super::*; + /// Re-registering an existing PV updates only the columns + /// register_pv owns; metadata and the committed last_timestamp + /// survive. + #[test] + fn reregister_preserves_columns_register_pv_does_not_own() { + let reg = PvRegistry::in_memory().unwrap(); + reg.register_pv("PV:Re", ArchDbType::ScalarDouble, &SampleMode::Monitor, 1) + .unwrap(); + reg.update_metadata("PV:Re", Some("3"), Some("mm")).unwrap(); + let ts = SystemTime::UNIX_EPOCH + std::time::Duration::from_secs(1_700_000_000); + reg.batch_update_timestamps(&[("PV:Re", ts)]).unwrap(); + reg.set_status("PV:Re", PvStatus::Paused).unwrap(); + let before = reg.get_pv("PV:Re").unwrap().unwrap(); + assert_eq!(before.last_timestamp, Some(ts), "test premise"); + + let scan = SampleMode::Scan { period_secs: 2.0 }; + reg.register_pv("PV:Re", ArchDbType::ScalarDouble, &scan, 1) + .unwrap(); + let after = reg.get_pv("PV:Re").unwrap().unwrap(); + assert_eq!(after.prec.as_deref(), Some("3")); + assert_eq!(after.egu.as_deref(), Some("mm")); + assert_eq!(after.last_timestamp, Some(ts)); + assert_eq!(after.created_at, before.created_at); + assert_eq!(after.sample_mode, scan); + assert_eq!(after.status, PvStatus::Active); + } + #[test] fn invalid_pv_names_rejected() { // path traversal From 6141e344f18730c7fa4f5fc378e0cd5069a27cee Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 01:18:20 +0900 Subject: [PATCH 12/55] etl: route move_file by the path-derived PV name, not the header pvname rename_pv moves partitions without rewriting their PayloadInfo, so a renamed PV's files still name the old PV inside. Routing the dest partition by that header migrated the renamed PV's data under the old name, where no reader looks. The path is the identity every other ETL step already uses. --- crates/archiver-core/src/etl/executor.rs | 57 ++++++++++++++++++++++-- 1 file changed, 53 insertions(+), 4 deletions(-) diff --git a/crates/archiver-core/src/etl/executor.rs b/crates/archiver-core/src/etl/executor.rs index 9ca4d58..11e336b 100644 --- a/crates/archiver-core/src/etl/executor.rs +++ b/crates/archiver-core/src/etl/executor.rs @@ -525,6 +525,17 @@ impl EtlExecutor { let source_name = self.source.name().to_string(); let dest_name = self.dest.name().to_string(); + // Route by the PATH-derived name, never by the header's + // `pvname`. The path is the identity every other step uses + // (grouping, the pause check, writer-slot lookups), and + // `rename_pv` moves partitions without rewriting headers, so + // after renamePV A→B a B partition still says A inside — + // routing by it migrated B's data under A, where no reader + // looks for it. + let pv_name = source.pv_name_from_path(&source_path).ok_or_else(|| { + anyhow::anyhow!("ETL: cannot derive PV name from partition path {source_path:?}") + })?; + let mut reader = match PbFileReader::open(&source_path) { Ok(r) => r, Err(e) @@ -574,7 +585,7 @@ impl EtlExecutor { return Ok(removed); }; - let d_path = dest.file_path_for(&desc.pv_name, first_sample.timestamp); + let d_path = dest.file_path_for(&pv_name, first_sample.timestamp); let ckpt = d_path.with_extension("pb.etl_ckpt"); let s_filename = source_path .file_name() @@ -851,11 +862,11 @@ impl EtlExecutor { let already_in_dest = |ts: SystemTime| dest_ts.as_ref().is_some_and(|set| set.contains(&ts)); if !already_in_dest(first_sample.timestamp) { - dest.append_event_with_meta(&desc.pv_name, dbr_type, &first_sample, &append_meta) + dest.append_event_with_meta(&pv_name, dbr_type, &first_sample, &append_meta) .await?; } while let Some(sample) = reader.next_event()? { - if dest.file_path_for(&desc.pv_name, sample.timestamp) != d_path { + if dest.file_path_for(&pv_name, sample.timestamp) != d_path { return Err(anyhow::anyhow!( "ETL: a sample timestamp in {source_path:?} maps to a different dest \ partition than {d_path:?}; refusing to split the copy" @@ -864,7 +875,7 @@ impl EtlExecutor { if already_in_dest(sample.timestamp) { continue; } - dest.append_event_with_meta(&desc.pv_name, dbr_type, &sample, &append_meta) + dest.append_event_with_meta(&pv_name, dbr_type, &sample, &append_meta) .await?; } // Durability BEFORE the commit. Flush all dest writers, then fsync @@ -1296,6 +1307,44 @@ mod tests { assert!(!ckpt.exists(), "checkpoint cleared on commit"); } + /// renamePV moves partitions without rewriting their headers, so a + /// renamed partition's header still names the OLD PV. The move must + /// route by the path (the new name), not the header. + #[tokio::test] + async fn move_file_routes_by_path_name_not_header_pvname() { + let tmp = tempfile::tempdir().unwrap(); + let src = plugin(tmp.path().join("sts"), "STS", PartitionGranularity::Hour); + let dest = plugin(tmp.path().join("mts"), "MTS", PartitionGranularity::Day); + let exec = EtlExecutor::new(src.clone(), dest.clone(), 3600, 0, 100); + + let samples: Vec<_> = (0..20) + .map(|i| sample_at(BASE_SECS + i, i as f64)) + .collect(); + let old_path = write_source_partition(&src, "Old:PV", &samples).await; + // What rename_pv does on disk: move the file, header untouched. + let new_path = src.file_path_for("New:PV", samples[0].timestamp); + std::fs::create_dir_all(new_path.parent().unwrap()).unwrap(); + std::fs::rename(&old_path, &new_path).unwrap(); + assert_eq!( + PbFileReader::open(&new_path).unwrap().description().pv_name, + "Old:PV", + "test premise: header still carries the old name" + ); + + assert!(exec.move_file(&new_path).await.unwrap()); + + let d_new = dest.file_path_for("New:PV", samples[0].timestamp); + let d_old = dest.file_path_for("Old:PV", samples[0].timestamp); + assert_eq!(count_samples(&d_new), 20, "data lands under the NEW name"); + assert!(!d_old.exists(), "nothing may land under the old name"); + assert_eq!( + PbFileReader::open(&d_new).unwrap().description().pv_name, + "New:PV", + "dest header is written with the routed name" + ); + assert!(!new_path.exists(), "source deleted after durable copy"); + } + #[tokio::test] async fn move_file_owner_retry_truncates_partial_not_duplicated() { // Simulates a crash after a prior attempt appended a PREFIX of S to From 1cd24680ade5ca4cc54697958fb2fed745f01e7f Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 01:19:47 +0900 Subject: [PATCH 13/55] storage: refuse appends whose type differs from the partition header Readers decode every frame of a partition with the header's type, so a sample of another type (PV retyped at the IOC while the archiver was down, or changeTypeForPV without a partition roll) became an undecodable frame that also hid every later sample in the file. partition_header now returns the declared type and write_cached rejects the append, which surfaces as a counted write error. --- .../archiver-core/src/storage/plainpb/mod.rs | 59 +++++++++++++------ tests/plainpb_compat.rs | 50 ++++++++++++++++ 2 files changed, 90 insertions(+), 19 deletions(-) diff --git a/crates/archiver-core/src/storage/plainpb/mod.rs b/crates/archiver-core/src/storage/plainpb/mod.rs index d8dcddb..1a41825 100644 --- a/crates/archiver-core/src/storage/plainpb/mod.rs +++ b/crates/archiver-core/src/storage/plainpb/mod.rs @@ -1395,7 +1395,25 @@ impl PlainPbStoragePlugin { } if slot.writer.is_none() { - let needs_header = file_needs_header(path)?; + // A partition's header fixes the type readers decode every + // frame with. Appending a sample of another type (the PV + // was retyped at the IOC, or changeTypeForPV ran without a + // partition roll) would write frames the reader cannot + // decode — and, worse, ones it stops at, hiding every + // later sample in the file. Refuse here, at the one place + // partitions are opened for append, so the file stays + // decodable and the loss is a counted write error. + let header_type = partition_header(path)?; + if let Some(existing) = header_type + && existing != dbr_type + { + return Err(anyhow::anyhow!( + "PB partition {path:?} holds {existing:?} samples; refusing to \ + append a {dbr_type:?} sample (type changed — the partition \ + keeps its header type until it rolls)" + )); + } + let needs_header = header_type.is_none(); // Whether the open below will CREATE the file (add a new // directory entry). Reliable under the per-PV slot lock: no // other writer targets this path. Gates the parent-dir @@ -1519,7 +1537,7 @@ impl PlainPbStoragePlugin { // the bytes as lost. (Principle: failure- // classified resource never goes through the // normal destructor path.) The created file - // exists on disk; `file_needs_header` on the + // exists on disk; `partition_header` on the // next attempt sees the unreadable header and // truncates. let (_file, _buffered) = bw.into_parts(); @@ -1563,7 +1581,7 @@ impl PlainPbStoragePlugin { // would compound the corruption (tail garbage, repeated // failures). Evict so the next call goes through the // create+append+header path which validates the file - // and reopens fresh. Tail-trim runs in `file_needs_header` + // and reopens fresh. Tail-trim runs in `partition_header` // on that next open and removes any partial record. // // Use `into_parts` to discard buffered bytes WITHOUT @@ -1711,8 +1729,11 @@ fn is_too_many_open_files(e: &std::io::Error) -> bool { matches!(e.raw_os_error(), Some(23) | Some(24)) } -/// Decide whether a file at `path` needs a fresh PayloadInfo header -/// written before sample data is appended. Returns `Ok(true)` when: +/// Inspect the partition at `path` before an append. Returns +/// `Ok(Some(type))` when a valid PayloadInfo header is present — +/// `type` is what the header declares, and every frame in the file +/// is decoded by readers as that type — or `Ok(None)` when a fresh +/// header must be written first, i.e. when: /// 1. the file doesn't exist, /// 2. the file exists but is 0 bytes (Java parity 651c3a6b: a crash /// mid-create would otherwise leave a header-less file), OR @@ -1730,18 +1751,18 @@ fn is_too_many_open_files(e: &std::io::Error) -> bool { /// a whole partition of good samples every time the fd table or the /// mount hiccupped. The caller fails this one append; the next sample /// re-probes. A failed truncate is likewise an error, not `true`: -/// returning `true` would append a second header mid-file. -fn file_needs_header(path: &Path) -> anyhow::Result { +/// returning `None` would append a second header mid-file. +fn partition_header(path: &Path) -> anyhow::Result> { let size = match std::fs::metadata(path) { Ok(m) => m.len(), - Err(e) if e.kind() == std::io::ErrorKind::NotFound => return Ok(true), + Err(e) if e.kind() == std::io::ErrorKind::NotFound => return Ok(None), Err(e) => return Err(anyhow::Error::from(e).context(format!("stat {path:?}"))), }; if size == 0 { - return Ok(true); + return Ok(None); } - match PbFileReader::open(path) { - Ok(_) => {} + let header_type = match PbFileReader::open(path) { + Ok(r) => r.description().db_type, Err(e) if e.downcast_ref::().is_some() => { return Err(e.context(format!("cannot read PB header of {path:?}"))); } @@ -1756,9 +1777,9 @@ fn file_needs_header(path: &Path) -> anyhow::Result { .truncate(true) .open(path) .map_err(|e| anyhow::anyhow!("failed to truncate corrupt PB file {path:?}: {e}"))?; - return Ok(true); + return Ok(None); } - } + }; // Header is valid; defend against a tail with a partial sample // (writer killed mid-flush). Truncating to the last NEWLINE // boundary loses at most one record but keeps the file readable @@ -1768,7 +1789,7 @@ fn file_needs_header(path: &Path) -> anyhow::Result { // (it would fuse the new frame onto the partial one). trim_to_last_newline(path) .map_err(|e| anyhow::anyhow!("failed to trim partial trailing record in {path:?}: {e}"))?; - Ok(false) + Ok(Some(header_type)) } /// Truncate `path` to end at its last NEWLINE byte (inclusive). Used @@ -2503,13 +2524,13 @@ mod tests { /// file untouched, not truncate it. #[cfg(unix)] #[test] - fn file_needs_header_io_error_propagates_without_truncating() { + fn partition_header_io_error_propagates_without_truncating() { let dir = tempfile::tempdir().unwrap(); // stat error (ELOOP): a self-referential symlink. let looped = dir.path().join("loop:2024.pb"); std::os::unix::fs::symlink(&looped, &looped).unwrap(); - let err = file_needs_header(&looped).expect_err("ELOOP must propagate"); + let err = partition_header(&looped).expect_err("ELOOP must propagate"); assert!( err.to_string().contains("stat"), "unexpected error: {err:#}" @@ -2528,7 +2549,7 @@ mod tests { std::fs::metadata(&dirp).unwrap().len() > 0, "test premise: a non-empty directory reports a non-zero size" ); - let err = file_needs_header(&dirp).expect_err("EISDIR must propagate"); + let err = partition_header(&dirp).expect_err("EISDIR must propagate"); assert!( err.to_string().contains("cannot read PB header"), "unexpected error: {err:#}" @@ -2542,11 +2563,11 @@ mod tests { /// The one case that MAY truncate: bytes were read and the header /// does not decode. #[test] - fn file_needs_header_truncates_only_undecodable_header() { + fn partition_header_truncates_only_undecodable_header() { let dir = tempfile::tempdir().unwrap(); let path = dir.path().join("bad:2024.pb"); std::fs::write(&path, b"\xff\xff\xffnot a payload info\n").unwrap(); - assert!(file_needs_header(&path).unwrap()); + assert!(partition_header(&path).unwrap().is_none()); assert_eq!(std::fs::metadata(&path).unwrap().len(), 0); } diff --git a/tests/plainpb_compat.rs b/tests/plainpb_compat.rs index d28fe61..8e75a49 100644 --- a/tests/plainpb_compat.rs +++ b/tests/plainpb_compat.rs @@ -795,6 +795,56 @@ async fn ghost_file_path_records_loss() { ); } +/// A partition's header fixes its type. After a restart (fresh plugin, +/// no cached writer) a sample of another type for the same partition +/// must be refused, not appended as a frame the reader decodes with +/// the header's type. +#[tokio::test] +async fn append_refuses_type_that_differs_from_partition_header() { + let dir = temp_dir(); + let ts1: SystemTime = Utc.with_ymd_and_hms(2026, 9, 1, 12, 0, 0).unwrap().into(); + let ts2 = ts1 + std::time::Duration::from_secs(60); + let pv = "TEST:Retyped"; + + let path = { + let plugin = + PlainPbStoragePlugin::new("test", dir.path().to_path_buf(), PartitionGranularity::Hour); + let s1 = ArchiverSample::new(ts1, ArchiverValue::ScalarDouble(1.0)); + plugin + .append_event(pv, ArchDbType::ScalarDouble, &s1) + .await + .unwrap(); + plugin.flush_writes().await.unwrap(); + plugin.file_path_for(pv, ts1) + }; + let len_before = std::fs::metadata(&path).unwrap().len(); + + // "Restart": a new plugin over the same root, PV now typed Int. + let plugin = + PlainPbStoragePlugin::new("test", dir.path().to_path_buf(), PartitionGranularity::Hour); + let s2 = ArchiverSample::new(ts2, ArchiverValue::ScalarInt(2)); + let err = plugin + .append_event(pv, ArchDbType::ScalarInt, &s2) + .await + .expect_err("a foreign-type append must be refused"); + assert!( + err.to_string().contains("holds ScalarDouble"), + "unexpected error: {err:#}" + ); + let _ = plugin.flush_writes().await; + assert_eq!( + std::fs::metadata(&path).unwrap().len(), + len_before, + "refused append must leave the partition untouched" + ); + let mut rdr = PbFileReader::open(&path).unwrap(); + let mut vals = Vec::new(); + while let Some(s) = rdr.next_event().unwrap() { + vals.push(s.value); + } + assert_eq!(vals, vec![ArchiverValue::ScalarDouble(1.0)]); +} + /// The partition directory is removed out from under a plugin that /// already created it (operator cleanup). The next append must /// recreate it and land, not fail on the stale `known_dirs` entry From 0c88c02e86c7d257db46d548da4a2eab36bed9c5 Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 01:20:08 +0900 Subject: [PATCH 14/55] registry: route update_last_timestamp through batch_update_timestamps The single-PV form ran an unguarded UPDATE, so it could move the last_timestamp watermark backwards while the batch form refused to. One owner of the commit rule now. --- crates/archiver-core/src/registry.rs | 34 +++++++++++++++++++++------- 1 file changed, 26 insertions(+), 8 deletions(-) diff --git a/crates/archiver-core/src/registry.rs b/crates/archiver-core/src/registry.rs index 575a627..9e132ae 100644 --- a/crates/archiver-core/src/registry.rs +++ b/crates/archiver-core/src/registry.rs @@ -390,14 +390,11 @@ impl PvRegistry { pv_name: &str, timestamp: SystemTime, ) -> anyhow::Result<()> { - let conn = self.lock_conn()?; - let dt = DateTime::::from(timestamp).to_rfc3339(); - let now = Utc::now().to_rfc3339(); - conn.execute( - "UPDATE pv_info SET last_timestamp = ?1, updated_at = ?2 WHERE pv_name = ?3", - params![dt, now, pv_name], - )?; - Ok(()) + // Single-PV form of `batch_update_timestamps`, which is the one + // owner of the last_timestamp commit rule (never regress). A + // second, unguarded UPDATE here would let any caller move the + // watermark backwards. + self.batch_update_timestamps(&[(pv_name, timestamp)]) } /// Remove a PV from the registry entirely. @@ -991,6 +988,27 @@ mod tests { assert_eq!(after.status, PvStatus::Active); } + #[test] + fn update_last_timestamp_never_regresses() { + let reg = PvRegistry::in_memory().unwrap(); + reg.register_pv( + "PV:Single", + ArchDbType::ScalarDouble, + &SampleMode::Monitor, + 1, + ) + .unwrap(); + let newer = SystemTime::UNIX_EPOCH + Duration::from_secs(1_700_000_100); + let older = SystemTime::UNIX_EPOCH + Duration::from_secs(1_700_000_000); + reg.update_last_timestamp("PV:Single", newer).unwrap(); + reg.update_last_timestamp("PV:Single", older).unwrap(); + assert_eq!( + reg.get_pv("PV:Single").unwrap().unwrap().last_timestamp, + Some(newer), + "single-PV update must obey the same monotonic rule as the batch" + ); + } + #[test] fn invalid_pv_names_rejected() { // path traversal From c667e44cde9cbab338c59a1ee8d967df78598d9d Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 01:21:30 +0900 Subject: [PATCH 15/55] storage: refuse rename_pv onto an existing destination partition std::fs::rename replaces an existing file, so a partition already present under the new name (stale earlier rename, manual copy, a registry/disk skew the API's registry-only guard cannot see) was destroyed. Every destination is checked before the first rename so a refusal leaves the source set whole. --- .../archiver-core/src/storage/plainpb/mod.rs | 18 ++++++- tests/plainpb_compat.rs | 50 +++++++++++++++++++ 2 files changed, 67 insertions(+), 1 deletion(-) diff --git a/crates/archiver-core/src/storage/plainpb/mod.rs b/crates/archiver-core/src/storage/plainpb/mod.rs index 1a41825..4ad045d 100644 --- a/crates/archiver-core/src/storage/plainpb/mod.rs +++ b/crates/archiver-core/src/storage/plainpb/mod.rs @@ -2417,7 +2417,13 @@ impl StoragePlugin for PlainPbStoragePlugin { } } - let mut moved = 0u64; + // Plan every move first and refuse if any destination already + // exists: std::fs::rename silently replaces an existing file, + // so a partition already under `to` (a stale earlier rename, a + // manual copy, a registry/disk skew the API's registry-only + // guard cannot see) would be destroyed. Checking up front + // keeps a refusal from leaving `from` half-renamed. + let mut planned: Vec<(&PathBuf, PathBuf)> = Vec::with_capacity(from_files.len()); for src in &from_files { let file_name = src .file_name() @@ -2432,6 +2438,16 @@ impl StoragePlugin for PlainPbStoragePlugin { })?; let new_name = format!("{to_leaf}:{suffix}"); let dst = to_dir.join(new_name); + if dst.try_exists()? { + anyhow::bail!( + "rename {from} -> {to}: destination partition {dst:?} already \ + exists; refusing to overwrite it" + ); + } + planned.push((src, dst)); + } + let mut moved = 0u64; + for (src, dst) in planned { std::fs::rename(src, &dst)?; moved += 1; } diff --git a/tests/plainpb_compat.rs b/tests/plainpb_compat.rs index 8e75a49..790aa58 100644 --- a/tests/plainpb_compat.rs +++ b/tests/plainpb_compat.rs @@ -795,6 +795,56 @@ async fn ghost_file_path_records_loss() { ); } +/// rename_pv must not clobber a partition that already exists under +/// the destination name; both PVs' data must survive the refusal. +#[tokio::test] +async fn rename_pv_refuses_to_overwrite_existing_destination_partition() { + let dir = temp_dir(); + let plugin = + PlainPbStoragePlugin::new("test", dir.path().to_path_buf(), PartitionGranularity::Hour); + let ts1: SystemTime = Utc.with_ymd_and_hms(2026, 9, 1, 12, 0, 0).unwrap().into(); + let (src, dst) = ("TEST:RenSrc", "TEST:RenDst"); + for (pv, v) in [(src, 1.0), (dst, 2.0)] { + let s = ArchiverSample::new(ts1, ArchiverValue::ScalarDouble(v)); + plugin + .append_event(pv, ArchDbType::ScalarDouble, &s) + .await + .unwrap(); + } + plugin.flush_writes().await.unwrap(); + + let err = plugin + .rename_pv(src, dst) + .await + .expect_err("rename onto an existing partition must be refused"); + assert!( + err.to_string().contains("already exists"), + "unexpected error: {err:#}" + ); + + for (pv, v) in [(src, 1.0), (dst, 2.0)] { + let mut rdr = PbFileReader::open(&plugin.file_path_for(pv, ts1)).unwrap(); + let mut vals = Vec::new(); + while let Some(s) = rdr.next_event().unwrap() { + vals.push(s.value); + } + assert_eq!( + vals, + vec![ArchiverValue::ScalarDouble(v)], + "{pv} must be intact" + ); + } + // The refused rename must not leave the source slot tombstoned. + let s = ArchiverSample::new( + ts1 + std::time::Duration::from_secs(1), + ArchiverValue::ScalarDouble(3.0), + ); + plugin + .append_event(src, ArchDbType::ScalarDouble, &s) + .await + .expect("source PV must stay writable after a refused rename"); +} + /// A partition's header fixes its type. After a restart (fresh plugin, /// no cached writer) a sample of another type for the same partition /// must be refused, not appended as a frame the reader decodes with From 0ec50d53fa4d3314c0928a7cfd76ad81e0a852ab Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 01:24:22 +0900 Subject: [PATCH 16/55] engine: run the flush owner's final flush after the shard drain flush_owner_loop started its shutdown grace on the shutdown signal while the shards were still draining, so every sample the drain appended after the ~200 ms grace stayed in BufWriters with its ts_update never committed. run_sharded_write_pool now fires shards_done once every shard has returned and the owner waits for it (bounded by drain_total_budget) before the final flush. main's supervisor budget is derived from the same two timeouts so it cannot abort that final flush. --- crates/archiver-engine/src/channel_manager.rs | 41 ++++++++++++--- src/main.rs | 9 +++- tests/write_loop_failure.rs | 50 +++++++++++++++++++ 3 files changed, 91 insertions(+), 9 deletions(-) diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index 528ddda..43a9a7e 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -3208,11 +3208,16 @@ pub async fn run_sharded_write_pool( // Spawn the global flush owner first so it's ready to flush // as soon as shards start appending. + // Fired once every shard worker has returned, so the flush + // owner's final flush + commit runs AFTER the drain, not + // concurrently with it. + let (shards_done_tx, shards_done_rx) = tokio::sync::oneshot::channel::<()>(); let flush_owner_handle = tokio::spawn(flush_owner_loop( storage.clone(), registry.clone(), pending.clone(), shutdown.clone(), + shards_done_rx, cfg.write_loop.clone(), )); @@ -3248,9 +3253,9 @@ pub async fn run_sharded_write_pool( .await; } - // Flush owner uses its own shutdown-watch + drain_total_budget - // grace timer, not an EOS signal — see flush_owner_loop. Wait - // for it to finish. + // Every shard has returned: nothing will append or post to + // `pending` again. Release the flush owner's final flush. + let _ = shards_done_tx.send(()); let _ = flush_owner_handle.await; } @@ -4208,6 +4213,7 @@ async fn flush_owner_loop( registry: Arc, pending: Arc, mut shutdown: tokio::sync::watch::Receiver, + shards_done: tokio::sync::oneshot::Receiver<()>, cfg: WriteLoopConfig, ) { let flush_period = cfg.flush_period; @@ -4262,7 +4268,16 @@ async fn flush_owner_loop( } } - // Phase 2: shutdown grace. Two windows in one budget: + // Phase 2: shutdown grace. Three windows in one budget: + // + // 0. Wait for the shard workers to finish draining. A shard + // keeps appending — and posting ts_updates into `pending` + // — for as long as its drain runs, so a final flush taken + // while the drain was still going left the tail of the + // drain sitting in BufWriters with its ts_updates never + // committed. `run_sharded_write_pool` fires `shards_done` + // once every shard has returned; a dropped sender (the + // pool unwound) counts as done. // // 1. A brief minimum sleep so any in-flight spawn_blocking // late-success closures can coalesce their reports into @@ -4275,13 +4290,23 @@ async fn flush_owner_loop( // short-circuit, and skip every entry in `pending` — // data on disk but no registry commit. // - // Both windows fit inside the operator-configured - // `drain_total_budget`. If a flush is genuinely wedged past - // the budget, we proceed to final flush; that call will see - // `in_flight` still set and short-circuit, but at least we + // All windows fit inside the operator-configured + // `drain_total_budget`. If a drain or flush is genuinely wedged + // past the budget, we proceed to final flush; that call will + // see `in_flight` still set and short-circuit, but at least we // didn't pin the process forever. Future samples on restart // will re-populate `pending` and the registry catches up. let phase2_deadline = std::time::Instant::now() + drain_total_budget; + if tokio::time::timeout(drain_total_budget, shards_done) + .await + .is_err() + { + warn!( + "Shutdown drain budget ({drain_total_budget:?}) exhausted before the \ + shard workers finished; final flush will not cover the rest of \ + the drain" + ); + } let min_grace = std::cmp::min(drain_total_budget, Duration::from_millis(200)); if !min_grace.is_zero() { tokio::time::sleep(min_grace).await; diff --git a/src/main.rs b/src/main.rs index 6a026b6..dd776e4 100644 --- a/src/main.rs +++ b/src/main.rs @@ -180,6 +180,13 @@ async fn main() -> anyhow::Result<()> { ..Default::default() }, }; + // The write pool's worst-case shutdown is the shard drain budget + // followed by the final flush + registry commit. The supervisor + // must outlast both: aborting the pool mid-final-flush leaves the + // drained tail on disk with its timestamps never committed. + let write_pool_shutdown_budget = pool_cfg.write_loop.drain_total_budget + + pool_cfg.write_loop.shutdown_flush_timeout + + Duration::from_secs(5); info!( shards = pool_cfg.shards, per_shard_buffer = pool_cfg.per_shard_buffer, @@ -415,7 +422,7 @@ async fn main() -> anyhow::Result<()> { server.await?; } - supervisor.shutdown(Duration::from_secs(30)).await; + supervisor.shutdown(write_pool_shutdown_budget).await; info!("Archiver stopped"); Ok(()) } diff --git a/tests/write_loop_failure.rs b/tests/write_loop_failure.rs index 1b0807e..b442332 100644 --- a/tests/write_loop_failure.rs +++ b/tests/write_loop_failure.rs @@ -929,6 +929,56 @@ async fn flush_timeout_does_not_consume_loss_markers_then_no_overcommit() { /// The final flush should wait for the in-flight flush to /// complete (within budget), then commit the new sample's /// timestamp to the registry. +/// Shutdown ordering: the final flush + commit must cover every +/// sample the shard drain appends, however long the drain takes +/// within `drain_total_budget`. Before, the owner's final flush ran +/// ~200 ms after the shutdown signal while the shard was still +/// draining, so the tail of the drain was appended but its +/// timestamps were never committed to the registry. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn shutdown_final_flush_runs_after_shard_drain() { + let storage = InjectingStorage::new(); + let registry = Arc::new(PvRegistry::in_memory().unwrap()); + let counters = Arc::new(PvCounters::default()); + let n = 30u64; + let pvs: Vec = (0..n).map(|i| format!("PV:Drain{i}")).collect(); + for pv in &pvs { + registry + .register_pv(pv, ArchDbType::ScalarDouble, &SampleMode::Monitor, 1) + .unwrap(); + // 30 × 30 ms ≈ 900 ms of sequential drain — far past the + // owner's 200 ms grace, well inside fast_cfg's 2 s budget. + storage.set_append_hang(pv, Duration::from_millis(30)); + } + let cfg = fast_cfg(); + let (tx, sd_tx, join) = spawn_loop(storage.clone(), registry.clone(), cfg); + + for (i, pv) in pvs.iter().enumerate() { + tx.send(pv_sample(pv, ts(i as u64), 1.0, &counters)) + .await + .unwrap(); + } + sd_tx.send(true).unwrap(); + join.await.unwrap(); + + let uncommitted: Vec<&String> = pvs + .iter() + .filter(|pv| { + registry + .get_pv(pv) + .unwrap() + .unwrap() + .last_timestamp + .is_none() + }) + .collect(); + assert!( + uncommitted.is_empty(), + "every drained sample must be committed by the final flush; \ + uncommitted: {uncommitted:?}" + ); +} + #[tokio::test(flavor = "multi_thread", worker_threads = 4)] async fn shutdown_final_flush_waits_for_in_flight_to_clear() { // Tighter timing for a fast test, but still within fast_cfg From b99f7ea4dc22d1bf7f34fb2958b0901bbfe34a50 Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 09:48:18 +0900 Subject: [PATCH 17/55] registry: reject '/' in is_valid_pv_name MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit pv_name_to_key encodes ':' as '/' and left a literal '/' untouched, so A/B and A:B mapped to one partition file under two writer slots, and pv_name_from_path could not invert the key — every path-derived lookup (ETL grouping, evict, truncate) missed the raw '/' slot. Rejecting '/' makes the encoding injective; canonical_pv_key, which papered over the lossy case for the ETL paused set, goes with it. --- crates/archiver-core/src/etl/executor.rs | 18 +++----- crates/archiver-core/src/registry.rs | 27 ++++++++---- .../archiver-core/src/storage/plainpb/mod.rs | 42 ++++++------------- 3 files changed, 36 insertions(+), 51 deletions(-) diff --git a/crates/archiver-core/src/etl/executor.rs b/crates/archiver-core/src/etl/executor.rs index 11e336b..7ba3bd1 100644 --- a/crates/archiver-core/src/etl/executor.rs +++ b/crates/archiver-core/src/etl/executor.rs @@ -188,21 +188,15 @@ impl EtlExecutor { } // Java parity (92db337): skip files whose owning PV is paused. - // `grouped`'s keys come from `pv_name_from_path`, whose on-disk - // encoding is lossy (`:` and `/` both collapse to `/`), so the - // registry name must be routed through the SAME canonicalization - // (`canonical_pv_key`) or a paused PV whose name contains `/` - // (e.g. `RING/DCCT` → grouped key `RING:DCCT`) would never match - // and its files would migrate despite the pause. Computed once - // per tick to keep the registry lookup off the per-file path. + // `grouped`'s keys come from `pv_name_from_path`, which inverts + // `pv_name_to_key` exactly (`is_valid_pv_name` rejects `/`, so + // the `:` → `/` encoding is injective), so the raw registry + // name is the matching key. Computed once per tick to keep the + // registry lookup off the per-file path. let paused: HashSet = match self.pv_registry.as_ref() { Some(reg) => reg .pvs_by_status(PvStatus::Paused) - .map(|recs| { - recs.into_iter() - .map(|r| crate::storage::plainpb::canonical_pv_key(&r.pv_name)) - .collect() - }) + .map(|recs| recs.into_iter().map(|r| r.pv_name).collect()) .unwrap_or_else(|e| { warn!("ETL: failed to read paused PVs from registry: {e}"); HashSet::new() diff --git a/crates/archiver-core/src/registry.rs b/crates/archiver-core/src/registry.rs index 9e132ae..9c06d38 100644 --- a/crates/archiver-core/src/registry.rs +++ b/crates/archiver-core/src/registry.rs @@ -109,22 +109,27 @@ pub fn is_valid_pv_name(name: &str) -> bool { if name.is_empty() || name.len() > 256 { return false; } + // `/` is the on-disk form of `:` (`pv_name_to_key` maps `:` → `/`), + // so a `/` in a PV name would alias another PV's partition files + // (`A/B` and `A:B` → one file, two writer slots) and could not be + // inverted by `pv_name_from_path`. Rejecting it keeps the key + // encoding injective, which every path-derived lookup (ETL, evict, + // truncate) relies on. + if name.contains('/') { + return false; + } // Leading separator → after pv_name_to_key the result begins with // `/`, and `Path::join(root, "/abs/path")` ignores `root` entirely // (Rust semantics) → escapes the storage root. Same for the `.` / // `-` cases. - if name.starts_with('.') - || name.starts_with('-') - || name.starts_with('/') - || name.starts_with(':') - { + if name.starts_with('.') || name.starts_with('-') || name.starts_with(':') { return false; } - for component in name.split([':', '/']) { + for component in name.split(':') { // `..` / `.` are obvious traversal. An empty segment (`A::B`, - // `A//B`, trailing `:`/`/`) maps to a `//` in the filesystem - // path which most OSes collapse but some path-relative tools - // re-split, so just reject. + // trailing `:`) maps to a `//` in the filesystem path which + // most OSes collapse but some path-relative tools re-split, so + // just reject. if component.is_empty() || component == ".." || component == "." { return false; } @@ -1026,6 +1031,10 @@ mod tests { assert!(!is_valid_pv_name("foo//bar")); assert!(!is_valid_pv_name("foo:")); assert!(!is_valid_pv_name("foo/")); + // `/` is the on-disk form of `:`; a name containing it would + // alias another PV's files. + assert!(!is_valid_pv_name("RING/DCCT")); + assert!(!is_valid_pv_name("A/B:C")); // shell metacharacters assert!(!is_valid_pv_name("foo;rm -rf /")); assert!(!is_valid_pv_name("foo|bar")); diff --git a/crates/archiver-core/src/storage/plainpb/mod.rs b/crates/archiver-core/src/storage/plainpb/mod.rs index 4ad045d..cb58ed8 100644 --- a/crates/archiver-core/src/storage/plainpb/mod.rs +++ b/crates/archiver-core/src/storage/plainpb/mod.rs @@ -1874,22 +1874,6 @@ pub(crate) fn pv_name_to_key(pv: &str) -> String { pv.replace(':', "/") } -/// Canonical PV key for comparing a registry PV name against the name -/// `pv_name_from_path` derives from a storage path. -/// -/// The on-disk key encoding is lossy: `pv_name_to_key` maps `:` → `/` -/// while leaving any existing `/` untouched, so both `A:B` and `A/B` -/// share the key `A/B`, and `pv_name_from_path` maps that back to `A:B`. -/// A raw registry name that contains `/` (a valid PV-name separator) is -/// therefore NOT equal to its path-derived form — matching them by raw -/// string silently fails (e.g. a paused `RING/DCCT` would not match the -/// grouped key `RING:DCCT`). Routing both sides through this function — -/// which reproduces `pv_name_from_path`'s final `/`→`:` step over the -/// on-disk key — makes them comparable. -pub(crate) fn canonical_pv_key(pv: &str) -> String { - pv_name_to_key(pv).replace('/', ":") -} - /// Read the last sample from a PB file by seeking near the end. /// Falls back to full sequential read for edge cases (e.g., very large single sample). fn read_last_sample_from_file(path: &Path) -> anyhow::Result> { @@ -2587,28 +2571,26 @@ mod tests { assert_eq!(std::fs::metadata(&path).unwrap().len(), 0); } + /// The on-disk key encoding must be injective and exactly + /// inverted by `pv_name_from_path`: ETL grouping, the paused-set + /// check and every path-derived slot lookup compare the raw + /// registry name against the path-derived one. `/` in a name is + /// rejected by `is_valid_pv_name` precisely because it would + /// collide with the `:` → `/` encoding. #[test] - fn canonical_pv_key_matches_path_derived_name() { + fn pv_key_round_trips_exactly_for_valid_names() { let plugin = PlainPbStoragePlugin::new("t", PathBuf::from("/root"), PartitionGranularity::Hour); - // For every PV name, the key that ETL's `paused` set must use - // (canonical_pv_key) must equal the key `grouped` uses - // (pv_name_from_path over the PV's storage path) — including - // valid names that contain '/', where the raw registry name - // differs from the path-derived form. - for pv in ["SIM:Sine", "RING/DCCT", "A/B:C", "PLAIN"] { + for pv in ["SIM:Sine", "RING:DCCT:AVG", "PLAIN", "IOC[A]:val"] { let key = pv_name_to_key(pv); let path = PathBuf::from(format!("/root/{key}:2024_01_01_00.pb")); - let from_path = plugin.pv_name_from_path(&path).unwrap(); assert_eq!( - canonical_pv_key(pv), - from_path, - "canonical_pv_key must equal pv_name_from_path for {pv:?}" + plugin.pv_name_from_path(&path).as_deref(), + Some(pv), + "pv_name_from_path must invert pv_name_to_key for {pv:?}" ); } - // The raw-name comparison ETL used before is exactly what breaks - // for a '/'-containing name. - assert_ne!("RING/DCCT", canonical_pv_key("RING/DCCT")); + assert!(!crate::registry::is_valid_pv_name("RING/DCCT")); } #[tokio::test] From 4aab7439a3218ffc44356e2d11c135ebe9968e42 Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 09:50:30 +0900 Subject: [PATCH 18/55] core: make decompose_timestamp fallible and partition arithmetic saturating MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit chrono's From unwraps once the year leaves ±262143, so a sample with an absurd timestamp panicked inside the shard's append, the ETL move or the getDataAtTime handler. utc_datetime_checked is the one non-panicking conversion; decompose_timestamp reports the range error and the partition helpers clamp/saturate so file_path_for and the range walks stay total. --- crates/archiver-api/src/pv_data.rs | 20 +-- crates/archiver-core/src/storage/partition.rs | 143 ++++++++++-------- .../archiver-core/src/storage/plainpb/mod.rs | 2 +- .../src/storage/plainpb/writer.rs | 4 +- crates/archiver-core/src/types.rs | 85 ++++++++++- tests/plainpb_compat.rs | 19 +++ 6 files changed, 196 insertions(+), 77 deletions(-) diff --git a/crates/archiver-api/src/pv_data.rs b/crates/archiver-api/src/pv_data.rs index ce51115..656a42c 100644 --- a/crates/archiver-api/src/pv_data.rs +++ b/crates/archiver-api/src/pv_data.rs @@ -244,9 +244,12 @@ async fn get_data_at_time( } }; - let entry = match pick { - Some(s) => { - let (year, secs, nanos) = s.decompose_timestamp(); + let entry = match pick.map(|s| (s.decompose_timestamp(), s)) { + Some((Err(e), _)) => { + tracing::warn!(pv, "getDataAtTime: sample timestamp unrepresentable: {e}"); + serde_json::Value::Null + } + Some((Ok((year, secs, nanos)), s)) => { let val = archiver_value_to_json_v4(&s.value); // Java parity (9b55268): include a "meta" object carrying the // PB field values (EGU, PREC, cnxlost markers, etc.) so @@ -257,12 +260,11 @@ async fn get_data_at_time( .map(|(k, v)| (k.clone(), serde_json::Value::String(v.clone()))) .collect(); serde_json::json!({ - "secs": SystemTime::from( - chrono::DateTime::::from(s.timestamp) - ) - .duration_since(SystemTime::UNIX_EPOCH) - .unwrap_or_default() - .as_secs(), + "secs": s + .timestamp + .duration_since(SystemTime::UNIX_EPOCH) + .unwrap_or_default() + .as_secs(), "nanos": nanos, "year": year, "secondsIntoYear": secs, diff --git a/crates/archiver-core/src/storage/partition.rs b/crates/archiver-core/src/storage/partition.rs index 7f9bf63..05a95a7 100644 --- a/crates/archiver-core/src/storage/partition.rs +++ b/crates/archiver-core/src/storage/partition.rs @@ -40,6 +40,19 @@ impl PartitionGranularity { } } +/// `SystemTime` → `DateTime` for partition arithmetic. Timestamps +/// beyond chrono's calendar clamp to `MIN_UTC` / `MAX_UTC` instead of +/// panicking in `From`; the storage write path rejects such samples +/// separately (`ArchiverSample::decompose_timestamp`), so clamping +/// here only keeps path derivation and range walks total. +fn utc_datetime_clamped(ts: SystemTime) -> chrono::DateTime { + crate::types::utc_datetime_checked(ts).unwrap_or(if ts < SystemTime::UNIX_EPOCH { + chrono::DateTime::::MIN_UTC + } else { + chrono::DateTime::::MAX_UTC + }) +} + /// Generate the partition name string for a given timestamp. /// Matches Java archiver's TimeUtils.getPartitionName(). /// @@ -50,7 +63,7 @@ impl PartitionGranularity { /// Hour → "2024_03_15_09" /// 5Min → "2024_03_15_09_30" pub fn partition_name(ts: SystemTime, granularity: PartitionGranularity) -> String { - let dt = chrono::DateTime::::from(ts); + let dt = utc_datetime_clamped(ts); let y = dt.year(); let m = dt.month(); let d = dt.day(); @@ -120,7 +133,7 @@ pub fn recent_partitions( /// Compute the start of the partition CONTAINING `ts` — the lower bound of /// that partition's half-open time range `[partition_start, next_partition_start)`. pub fn partition_start(ts: SystemTime, granularity: PartitionGranularity) -> SystemTime { - let dt = chrono::DateTime::::from(ts); + let dt = utc_datetime_clamped(ts); let start = match granularity { PartitionGranularity::Year => NaiveDate::from_ymd_opt(dt.year(), 1, 1) @@ -159,87 +172,74 @@ pub fn partition_start(ts: SystemTime, granularity: PartitionGranularity) -> Sys } /// Compute the start of the next partition after the one containing `ts`. +/// +/// Saturates at the end of chrono's calendar instead of panicking: +/// past `MAX_UTC` there is no next partition, and callers that walk +/// forward (`partitions_in_range`) stop when the value no longer +/// advances. pub fn next_partition_start(ts: SystemTime, granularity: PartitionGranularity) -> SystemTime { - let dt = chrono::DateTime::::from(ts); + let dt = utc_datetime_clamped(ts); + let midnight = |d: NaiveDate| d.and_hms_opt(0, 0, 0).map(|n| n.and_utc()); let next = match granularity { - PartitionGranularity::Year => NaiveDate::from_ymd_opt(dt.year() + 1, 1, 1) - .expect("Jan 1 is always valid") - .and_hms_opt(0, 0, 0) - .expect("midnight is always valid") - .and_utc(), + PartitionGranularity::Year => { + NaiveDate::from_ymd_opt(dt.year() + 1, 1, 1).and_then(midnight) + } PartitionGranularity::Month => { let (y, m) = if dt.month() == 12 { (dt.year() + 1, 1) } else { (dt.year(), dt.month() + 1) }; - NaiveDate::from_ymd_opt(y, m, 1) - .expect("1st of month is always valid") - .and_hms_opt(0, 0, 0) - .expect("midnight is always valid") - .and_utc() - } - PartitionGranularity::Day => (dt.date_naive() + Duration::days(1)) - .and_hms_opt(0, 0, 0) - .expect("midnight is always valid") - .and_utc(), - PartitionGranularity::Hour => { - let current_hour = dt - .date_naive() - .and_hms_opt(dt.hour(), 0, 0) - .expect("hour from valid DateTime") - .and_utc(); - current_hour + Duration::hours(1) + NaiveDate::from_ymd_opt(y, m, 1).and_then(midnight) } + PartitionGranularity::Day => dt + .date_naive() + .checked_add_signed(Duration::days(1)) + .and_then(midnight), + PartitionGranularity::Hour => dt + .date_naive() + .and_hms_opt(dt.hour(), 0, 0) + .map(|n| n.and_utc()) + .and_then(|h| h.checked_add_signed(Duration::hours(1))), PartitionGranularity::FiveMin | PartitionGranularity::FifteenMin | PartitionGranularity::ThirtyMin => { let approx_min = granularity.approx_minutes(); let start_min = (dt.minute() / approx_min) * approx_min; - let current_start = dt - .date_naive() + dt.date_naive() .and_hms_opt(dt.hour(), start_min, 0) - .expect("aligned minute from valid DateTime") - .and_utc(); - current_start + Duration::minutes(approx_min as i64) + .map(|n| n.and_utc()) + .and_then(|s| s.checked_add_signed(Duration::minutes(approx_min as i64))) } }; - next.into() + next.unwrap_or(chrono::DateTime::::MAX_UTC).into() } /// Compute the last moment of the previous partition before `ts`. +/// Saturates at `MIN_UTC` instead of panicking (see +/// `next_partition_start`). fn prev_partition_end(ts: SystemTime, granularity: PartitionGranularity) -> SystemTime { - let dt = chrono::DateTime::::from(ts); + let dt = utc_datetime_clamped(ts); + let last_second = |d: NaiveDate| d.and_hms_opt(23, 59, 59).map(|n| n.and_utc()); let prev_end = match granularity { - PartitionGranularity::Year => NaiveDate::from_ymd_opt(dt.year() - 1, 12, 31) - .expect("Dec 31 is always valid") - .and_hms_opt(23, 59, 59) - .expect("23:59:59 is always valid") - .and_utc(), - PartitionGranularity::Month => { - let first_of_month = NaiveDate::from_ymd_opt(dt.year(), dt.month(), 1) - .expect("1st of month is always valid"); - let prev = first_of_month - Duration::days(1); - prev.and_hms_opt(23, 59, 59) - .expect("23:59:59 is always valid") - .and_utc() - } - PartitionGranularity::Day => { - let prev = dt.date_naive() - Duration::days(1); - prev.and_hms_opt(23, 59, 59) - .expect("23:59:59 is always valid") - .and_utc() - } - PartitionGranularity::Hour => { - dt.date_naive() - .and_hms_opt(dt.hour(), 0, 0) - .expect("hour from valid DateTime") - .and_utc() - - Duration::seconds(1) + PartitionGranularity::Year => { + NaiveDate::from_ymd_opt(dt.year() - 1, 12, 31).and_then(last_second) } + PartitionGranularity::Month => NaiveDate::from_ymd_opt(dt.year(), dt.month(), 1) + .and_then(|d| d.checked_sub_signed(Duration::days(1))) + .and_then(last_second), + PartitionGranularity::Day => dt + .date_naive() + .checked_sub_signed(Duration::days(1)) + .and_then(last_second), + PartitionGranularity::Hour => dt + .date_naive() + .and_hms_opt(dt.hour(), 0, 0) + .map(|n| n.and_utc()) + .and_then(|h| h.checked_sub_signed(Duration::seconds(1))), PartitionGranularity::FiveMin | PartitionGranularity::FifteenMin | PartitionGranularity::ThirtyMin => { @@ -247,13 +247,12 @@ fn prev_partition_end(ts: SystemTime, granularity: PartitionGranularity) -> Syst let start_min = (dt.minute() / approx_min) * approx_min; dt.date_naive() .and_hms_opt(dt.hour(), start_min, 0) - .expect("aligned minute from valid DateTime") - .and_utc() - - Duration::seconds(1) + .map(|n| n.and_utc()) + .and_then(|s| s.checked_sub_signed(Duration::seconds(1))) } }; - prev_end.into() + prev_end.unwrap_or(chrono::DateTime::::MIN_UTC).into() } #[cfg(test)] @@ -261,6 +260,28 @@ mod tests { use super::*; use chrono::TimeZone; + /// Out-of-calendar timestamps must clamp/saturate, never panic: + /// `file_path_for` runs inside the shard's append and the ETL move. + #[test] + fn partition_arithmetic_saturates_out_of_range() { + let huge = SystemTime::UNIX_EPOCH + std::time::Duration::from_secs(1 << 62); + let max: SystemTime = chrono::DateTime::::MAX_UTC.into(); + let min: SystemTime = chrono::DateTime::::MIN_UTC.into(); + for g in [ + PartitionGranularity::Year, + PartitionGranularity::Month, + PartitionGranularity::Day, + PartitionGranularity::Hour, + PartitionGranularity::FiveMin, + ] { + assert_eq!(partition_name(huge, g), partition_name(max, g)); + assert!(next_partition_start(max, g) >= max, "{g:?}"); + assert!(prev_partition_end(min, g) <= min, "{g:?}"); + assert!(partition_start(huge, g) <= huge, "{g:?}"); + assert_eq!(partitions_in_range(max, max, g).len(), 1, "{g:?}"); + } + } + #[test] fn test_partition_name_year() { let ts: SystemTime = Utc.with_ymd_and_hms(2024, 6, 15, 10, 30, 0).unwrap().into(); diff --git a/crates/archiver-core/src/storage/plainpb/mod.rs b/crates/archiver-core/src/storage/plainpb/mod.rs index cb58ed8..53c4706 100644 --- a/crates/archiver-core/src/storage/plainpb/mod.rs +++ b/crates/archiver-core/src/storage/plainpb/mod.rs @@ -1512,7 +1512,7 @@ impl PlainPbStoragePlugin { let mut bw = BufWriter::with_capacity(64 * 1024, file); if needs_header { - let (year, _, _) = sample.decompose_timestamp(); + let (year, _, _) = sample.decompose_timestamp()?; let header = writer::build_payload_info( pv, dbr_type, diff --git a/crates/archiver-core/src/storage/plainpb/writer.rs b/crates/archiver-core/src/storage/plainpb/writer.rs index f2f8591..2ae6a31 100644 --- a/crates/archiver-core/src/storage/plainpb/writer.rs +++ b/crates/archiver-core/src/storage/plainpb/writer.rs @@ -52,7 +52,7 @@ impl<'a> PbFileWriter<'a> { .await?; if !file_exists { - let (year, _, _) = sample.decompose_timestamp(); + let (year, _, _) = sample.decompose_timestamp()?; let header = build_payload_info(pv, dbr_type, year, element_count, headers); let header_bytes = header.encode_to_vec(); let escaped = codec::escape(&header_bytes); @@ -115,7 +115,7 @@ fn field_values_to_pb(fvs: &[(String, String)]) -> Vec { /// Encode an ArchiverSample into protobuf bytes (before line-escaping). pub fn encode_sample(dbr_type: ArchDbType, sample: &ArchiverSample) -> anyhow::Result> { - let (_, secs, nanos) = sample.decompose_timestamp(); + let (_, secs, nanos) = sample.decompose_timestamp()?; let fvs = field_values_to_pb(&sample.field_values); let severity = if sample.severity != 0 { Some(sample.severity) diff --git a/crates/archiver-core/src/types.rs b/crates/archiver-core/src/types.rs index ab035c0..66e3640 100644 --- a/crates/archiver-core/src/types.rs +++ b/crates/archiver-core/src/types.rs @@ -145,18 +145,30 @@ impl ArchiverSample { } /// Decompose timestamp into (year, seconds_into_year, nanos). - pub fn decompose_timestamp(&self) -> (i32, u32, u32) { - let datetime = chrono::DateTime::::from(self.timestamp); + /// + /// Errors instead of panicking when the timestamp lies outside the + /// calendar range chrono can represent (|year| > 262143): the PB + /// frame and the partition name both need a civil year, and a + /// panic here unwinds whichever task carried the sample (a shard's + /// append, an ETL move, a retrieval handler). + pub fn decompose_timestamp(&self) -> anyhow::Result<(i32, u32, u32)> { + let datetime = utc_datetime_checked(self.timestamp).ok_or_else(|| { + anyhow::anyhow!( + "sample timestamp {:?} is outside the representable calendar range", + self.timestamp + ) + })?; let year = datetime.year(); + // Jan 1 00:00:00 of a year chrono already represents is valid. let year_start = NaiveDateTime::new( - chrono::NaiveDate::from_ymd_opt(year, 1, 1).expect("year from valid SystemTime"), + chrono::NaiveDate::from_ymd_opt(year, 1, 1).expect("Jan 1 of an in-range year"), chrono::NaiveTime::from_hms_opt(0, 0, 0).expect("midnight"), ) .and_utc(); let duration = datetime.signed_duration_since(year_start); let seconds_into_year = duration.num_seconds() as u32; let nanos = datetime.timestamp_subsec_nanos(); - (year, seconds_into_year, nanos) + Ok((year, seconds_into_year, nanos)) } /// Reconstruct a SystemTime from year + seconds_into_year + nanos. @@ -183,6 +195,30 @@ impl ArchiverSample { } } +/// Non-panicking `SystemTime` → `DateTime` conversion. `From` in +/// chrono unwraps and panics once the year leaves ±262143; every +/// timestamp that reaches storage or retrieval goes through here (or +/// the clamping variant in `partition`) instead. +pub fn utc_datetime_checked(ts: SystemTime) -> Option> { + let (secs, nanos) = match ts.duration_since(SystemTime::UNIX_EPOCH) { + Ok(d) => (i64::try_from(d.as_secs()).ok()?, d.subsec_nanos()), + Err(e) => { + // Before the epoch: borrow one second so nanos stay positive. + let d = e.duration(); + let s = i64::try_from(d.as_secs()).ok()?; + if d.subsec_nanos() == 0 { + (s.checked_neg()?, 0) + } else { + ( + s.checked_neg()?.checked_sub(1)?, + 1_000_000_000 - d.subsec_nanos(), + ) + } + } + }; + chrono::DateTime::::from_timestamp(secs, nanos) +} + /// Description of an event stream (used in reader). #[derive(Debug, Clone)] pub struct EventStreamDesc { @@ -253,3 +289,44 @@ pub fn finite_or_null(f: f64) -> serde_json::Value { serde_json::Value::Null } } + +#[cfg(test)] +mod tests { + use super::*; + use std::time::Duration; + + #[test] + fn decompose_timestamp_round_trips_through_epoch_parts() { + let ts = SystemTime::UNIX_EPOCH + Duration::new(1_700_000_000, 250); + let s = ArchiverSample::new(ts, ArchiverValue::ScalarDouble(0.0)); + let (year, secs, nanos) = s.decompose_timestamp().unwrap(); + assert_eq!(year, 2023); + assert_eq!(nanos, 250); + assert_eq!( + ArchiverSample::timestamp_from_epoch_parts(year, secs, nanos), + Some(ts) + ); + } + + #[test] + fn decompose_timestamp_errors_instead_of_panicking_out_of_range() { + // ~1.4e11 years past the epoch: representable as SystemTime, + // far outside chrono's ±262143-year calendar. + let ts = SystemTime::UNIX_EPOCH + .checked_add(Duration::from_secs(1 << 62)) + .expect("SystemTime holds i64 seconds on this platform"); + let s = ArchiverSample::new(ts, ArchiverValue::ScalarDouble(0.0)); + let err = s.decompose_timestamp().unwrap_err(); + assert!(err.to_string().contains("outside the representable")); + assert!(utc_datetime_checked(ts).is_none()); + } + + #[test] + fn utc_datetime_checked_handles_pre_epoch_fractions() { + let ts = SystemTime::UNIX_EPOCH - Duration::new(1, 500_000_000); + let dt = utc_datetime_checked(ts).unwrap(); + assert_eq!(dt.timestamp(), -2); + assert_eq!(dt.timestamp_subsec_nanos(), 500_000_000); + assert_eq!(dt.year(), 1969); + } +} diff --git a/tests/plainpb_compat.rs b/tests/plainpb_compat.rs index 790aa58..e3710b8 100644 --- a/tests/plainpb_compat.rs +++ b/tests/plainpb_compat.rs @@ -795,6 +795,25 @@ async fn ghost_file_path_records_loss() { ); } +/// A timestamp outside chrono's calendar must fail the append with an +/// error, not panic the shard's spawn_blocking task. +#[tokio::test] +async fn append_with_unrepresentable_timestamp_errors_without_panic() { + let dir = temp_dir(); + let plugin = + PlainPbStoragePlugin::new("test", dir.path().to_path_buf(), PartitionGranularity::Hour); + let huge = SystemTime::UNIX_EPOCH + std::time::Duration::from_secs(1 << 62); + let s = ArchiverSample::new(huge, ArchiverValue::ScalarDouble(1.0)); + let err = plugin + .append_event("TEST:HugeTs", ArchDbType::ScalarDouble, &s) + .await + .expect_err("out-of-range timestamp must be refused"); + assert!( + err.to_string().contains("outside the representable"), + "unexpected error: {err:#}" + ); +} + /// rename_pv must not clobber a partition that already exists under /// the destination name; both PVs' data must survive the refusal. #[tokio::test] From 373bb4e85010ae3c53396da13b9b1ab4b4c85693 Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 09:53:30 +0900 Subject: [PATCH 19/55] storage: convert stored partitions in changeTypeForPV changeTypeForPV only flipped the registry row, so after a retype the current partition still carried the old header and every new sample was refused until it rolled. StoragePlugin::convert_pv_type rewrites each partition through ArchiverValue::convert_to (Java's ThruNumberConversion rules: numeric via f64, string parse/format, scalar <-> first element), to a temp sibling replaced by one rename, and skips partitions already of the new type so a re-run after a crash completes. The handler converts before the registry flips. --- crates/archiver-api/src/handlers/mgmt/p2.rs | 17 ++ .../archiver-core/src/storage/plainpb/mod.rs | 93 ++++++++ crates/archiver-core/src/storage/tiered.rs | 7 + crates/archiver-core/src/storage/traits.rs | 12 ++ crates/archiver-core/src/types.rs | 201 ++++++++++++++++++ tests/api_mgmt.rs | 48 +++++ tests/plainpb_compat.rs | 113 ++++++++++ 7 files changed, 491 insertions(+) diff --git a/crates/archiver-api/src/handlers/mgmt/p2.rs b/crates/archiver-api/src/handlers/mgmt/p2.rs index 6dbae6b..c600e94 100644 --- a/crates/archiver-api/src/handlers/mgmt/p2.rs +++ b/crates/archiver-api/src/handlers/mgmt/p2.rs @@ -193,6 +193,22 @@ pub async fn change_type_for_pv( return ApiError::BadRequest(format!("invalid newtype: {}", q.newtype)).into_response(); } }; + // Java parity: changeTypeForPV converts the PV's stored data to the + // new type (thru-number conversion) BEFORE the registry flips, so no + // partition is left whose header disagrees with the archived type + // (the storage layer refuses such appends). Files are the truth: a + // crash between the conversion and the registry update is repaired + // by re-issuing the request — converted partitions are skipped. + let converted = match state.storage.convert_pv_type(&canonical, new_type).await { + Ok(n) => n, + Err(e) => { + return ApiError::internal(e.context(format!( + "converting stored data of '{}' to {new_type:?}", + q.pv + ))) + .into_response(); + } + }; if let Err(e) = state.pv_cmd.import_pv( &canonical, new_type, @@ -213,6 +229,7 @@ pub async fn change_type_for_pv( "pv": canonical, "oldType": record.dbr_type as i32, "newType": q.newtype, + "partitionsConverted": converted, })) .into_response() } diff --git a/crates/archiver-core/src/storage/plainpb/mod.rs b/crates/archiver-core/src/storage/plainpb/mod.rs index 53c4706..16a86dc 100644 --- a/crates/archiver-core/src/storage/plainpb/mod.rs +++ b/crates/archiver-core/src/storage/plainpb/mod.rs @@ -2005,6 +2005,64 @@ fn list_pv_pb_files(root: &Path, pv: &str) -> anyhow::Result> { Ok(files) } +/// Rewrite one partition to `new_type` (see `convert_pv_type`). The +/// converted stream goes to a `.pb.convert_tmp` sibling that replaces +/// the original with one rename, so a crash mid-rewrite leaves the +/// original intact and at most a stray temp file (not a `.pb`, so no +/// listing sees it; the next attempt truncates it). Returns `false` +/// when the partition already holds `new_type`. +fn convert_partition_file( + path: &Path, + pv: &str, + new_type: ArchDbType, + fsync: bool, +) -> anyhow::Result { + let mut reader = PbFileReader::open(path)?; + let desc = reader.description().clone(); + if desc.db_type == new_type { + return Ok(false); + } + // The header's element count describes the shape; it only changes + // when the conversion crosses scalar ↔ waveform. + let element_count = if new_type.is_waveform() == desc.db_type.is_waveform() { + desc.element_count + } else { + Some(1) + }; + let tmp = path.with_extension("pb.convert_tmp"); + let mut write = || -> anyhow::Result<()> { + let mut out = BufWriter::with_capacity(64 * 1024, std::fs::File::create(&tmp)?); + let header = + writer::build_payload_info(pv, new_type, desc.year, element_count, &desc.headers); + let mut frame = codec::escape(&header.encode_to_vec()); + frame.push(codec::NEWLINE); + out.write_all(&frame)?; + while let Some(sample) = reader.next_event()? { + let converted = ArchiverSample { + value: sample.value.convert_to(new_type)?, + ..sample + }; + let mut frame = codec::escape(&writer::encode_sample(new_type, &converted)?); + frame.push(codec::NEWLINE); + out.write_all(&frame)?; + } + let file = out.into_inner().map_err(|e| e.into_error())?; + if fsync { + file.sync_all()?; + } + Ok(()) + }; + if let Err(e) = write() { + let _ = std::fs::remove_file(&tmp); + return Err(e.context(format!("converting {path:?} to {new_type:?}"))); + } + std::fs::rename(&tmp, path)?; + if fsync && let Some(parent) = path.parent() { + PlainPbStoragePlugin::sync_dir(parent)?; + } + Ok(true) +} + #[async_trait] impl StoragePlugin for PlainPbStoragePlugin { fn name(&self) -> &str { @@ -2313,6 +2371,41 @@ impl StoragePlugin for PlainPbStoragePlugin { }]) } + async fn convert_pv_type(&self, pv: &str, new_type: ArchDbType) -> anyhow::Result { + // Same three-phase slot handling as delete_pv_data / rename_pv: + // tombstone so a concurrent append bails while partitions are + // being replaced, and drop the cached writer — its fd would + // otherwise keep pointing at the replaced inode. The RAII guard + // un-tombstones on every return path. + let _cleanup = TombstoneCleanupGuard { + plugin: self, + pv: pv.to_string(), + }; + let slot_arc = self.slot_for(pv); + { + let mut slot = slot_arc.lock().unwrap_or_else(|e| e.into_inner()); + slot.dead = true; + if let Some(cached) = slot.writer.take() { + let _ = self.drop_dirty_writer(pv, cached); + } + } + + let files = list_pv_pb_files(&self.root_folder, pv)?; + let mut converted = 0u64; + for path in files { + let pv = pv.to_string(); + let fsync = self.fsync_on_flush; + let rewritten = tokio::task::spawn_blocking(move || { + convert_partition_file(&path, &pv, new_type, fsync) + }) + .await??; + if rewritten { + converted += 1; + } + } + Ok(converted) + } + async fn rename_pv(&self, from: &str, to: &str) -> anyhow::Result { // Phase 1: tombstone the SOURCE slot in cache (don't // remove it yet — see delete_pv_data for the same race diff --git a/crates/archiver-core/src/storage/tiered.rs b/crates/archiver-core/src/storage/tiered.rs index 16a9912..9d43e00 100644 --- a/crates/archiver-core/src/storage/tiered.rs +++ b/crates/archiver-core/src/storage/tiered.rs @@ -275,6 +275,13 @@ impl StoragePlugin for TieredStorage { let l = self.lts.rename_pv(from, to).await?; Ok(s + m + l) } + + async fn convert_pv_type(&self, pv: &str, new_type: ArchDbType) -> anyhow::Result { + let s = self.sts.convert_pv_type(pv, new_type).await?; + let m = self.mts.convert_pv_type(pv, new_type).await?; + let l = self.lts.convert_pv_type(pv, new_type).await?; + Ok(s + m + l) + } } #[cfg(test)] diff --git a/crates/archiver-core/src/storage/traits.rs b/crates/archiver-core/src/storage/traits.rs index 4c6daa5..4ecd25f 100644 --- a/crates/archiver-core/src/storage/traits.rs +++ b/crates/archiver-core/src/storage/traits.rs @@ -196,6 +196,18 @@ pub trait StoragePlugin: Send + Sync { async fn rename_pv(&self, _from: &str, _to: &str) -> anyhow::Result { anyhow::bail!("rename_pv not implemented for this storage plugin") } + + /// Rewrite every partition of `pv` to `new_type`, converting each + /// sample "through a number" (`ArchiverValue::convert_to`). This is + /// what `changeTypeForPV` runs before the registry flips the PV's + /// type, so no partition is left whose header disagrees with the + /// type the PV is archived as. Partitions already of `new_type` are + /// skipped, so a re-run after a crash finishes the job. Returns the + /// number of partitions rewritten. Defaults to error so missing + /// implementations surface explicitly. + async fn convert_pv_type(&self, _pv: &str, _new_type: ArchDbType) -> anyhow::Result { + anyhow::bail!("convert_pv_type not implemented for this storage plugin") + } } /// Post-processor trait for data reduction (mean, max, min, etc.). diff --git a/crates/archiver-core/src/types.rs b/crates/archiver-core/src/types.rs index 66e3640..8bbcd33 100644 --- a/crates/archiver-core/src/types.rs +++ b/crates/archiver-core/src/types.rs @@ -106,6 +106,85 @@ impl ArchiverValue { } } + /// Convert to another archived type "through a number" — the rule + /// `changeTypeForPV` applies to a PV's existing partitions (Java + /// parity: `ThruNumberConversion`). Numeric families convert via + /// `f64` with `as` semantics (truncation toward zero, saturation at + /// the target's bounds); strings parse and format; a scalar becomes + /// a one-element waveform and a waveform collapses to its first + /// element. `V4GenericBytes` is opaque and converts neither way. + pub fn convert_to(&self, target: ArchDbType) -> anyhow::Result { + if self.db_type() == target { + return Ok(self.clone()); + } + let elems = self.elements()?; + let first = || -> anyhow::Result<&Elem> { + elems.first().ok_or_else(|| { + anyhow::anyhow!("cannot convert an empty waveform to scalar {target:?}") + }) + }; + let all_num = || -> anyhow::Result> { elems.iter().map(Elem::to_f64).collect() }; + Ok(match target { + ArchDbType::ScalarString => Self::ScalarString(first()?.to_text()), + ArchDbType::ScalarByte => Self::ScalarByte(vec![first()?.to_f64()? as i8 as u8]), + ArchDbType::ScalarShort => Self::ScalarShort(first()?.to_f64()? as i16 as i32), + ArchDbType::ScalarEnum => Self::ScalarEnum(first()?.to_f64()? as i16 as i32), + ArchDbType::ScalarInt => Self::ScalarInt(first()?.to_f64()? as i32), + ArchDbType::ScalarFloat => Self::ScalarFloat(first()?.to_f64()? as f32), + ArchDbType::ScalarDouble => Self::ScalarDouble(first()?.to_f64()?), + ArchDbType::WaveformString => { + Self::VectorString(elems.iter().map(Elem::to_text).collect()) + } + ArchDbType::WaveformByte => { + Self::VectorChar(all_num()?.into_iter().map(|f| f as i8 as u8).collect()) + } + ArchDbType::WaveformShort => { + Self::VectorShort(all_num()?.into_iter().map(|f| f as i16 as i32).collect()) + } + ArchDbType::WaveformEnum => { + Self::VectorEnum(all_num()?.into_iter().map(|f| f as i16 as i32).collect()) + } + ArchDbType::WaveformInt => { + Self::VectorInt(all_num()?.into_iter().map(|f| f as i32).collect()) + } + ArchDbType::WaveformFloat => { + Self::VectorFloat(all_num()?.into_iter().map(|f| f as f32).collect()) + } + ArchDbType::WaveformDouble => Self::VectorDouble(all_num()?), + ArchDbType::V4GenericBytes => { + anyhow::bail!( + "cannot convert {:?} to opaque V4GenericBytes", + self.db_type() + ) + } + }) + } + + /// Element-wise view for `convert_to`. Bytes are signed (EPICS + /// DBR_CHAR); a scalar is one element. + fn elements(&self) -> anyhow::Result> { + Ok(match self { + Self::ScalarString(s) => vec![Elem::Text(s.clone())], + Self::ScalarByte(b) | Self::VectorChar(b) => { + b.iter().map(|x| Elem::Num(f64::from(*x as i8))).collect() + } + Self::ScalarShort(v) | Self::ScalarInt(v) | Self::ScalarEnum(v) => { + vec![Elem::Num(f64::from(*v))] + } + Self::ScalarFloat(v) => vec![Elem::Num(f64::from(*v))], + Self::ScalarDouble(v) => vec![Elem::Num(*v)], + Self::VectorString(v) => v.iter().map(|s| Elem::Text(s.clone())).collect(), + Self::VectorShort(v) | Self::VectorInt(v) | Self::VectorEnum(v) => { + v.iter().map(|x| Elem::Num(f64::from(*x))).collect() + } + Self::VectorFloat(v) => v.iter().map(|x| Elem::Num(f64::from(*x))).collect(), + Self::VectorDouble(v) => v.iter().map(|x| Elem::Num(*x)).collect(), + Self::V4GenericBytes(_) => { + anyhow::bail!("cannot convert opaque V4GenericBytes to another type") + } + }) + } + /// Try to extract a f64 representation (for postprocessors like mean/max/min). pub fn as_f64(&self) -> Option { match self { @@ -195,6 +274,31 @@ impl ArchiverSample { } } +/// One element of an [`ArchiverValue`] as `convert_to` sees it. +enum Elem { + Num(f64), + Text(String), +} + +impl Elem { + fn to_f64(&self) -> anyhow::Result { + match self { + Self::Num(f) => Ok(*f), + Self::Text(s) => s + .trim() + .parse::() + .map_err(|e| anyhow::anyhow!("cannot convert string {s:?} to a number: {e}")), + } + } + + fn to_text(&self) -> String { + match self { + Self::Num(f) => f.to_string(), + Self::Text(s) => s.clone(), + } + } +} + /// Non-panicking `SystemTime` → `DateTime` conversion. `From` in /// chrono unwraps and panics once the year leaves ±262143; every /// timestamp that reaches storage or retrieval goes through here (or @@ -295,6 +399,103 @@ mod tests { use super::*; use std::time::Duration; + #[test] + fn convert_to_numeric_truncates_toward_zero_and_saturates() { + let d = ArchiverValue::ScalarDouble(-1.9); + assert_eq!( + d.convert_to(ArchDbType::ScalarInt).unwrap(), + ArchiverValue::ScalarInt(-1) + ); + assert_eq!( + ArchiverValue::ScalarDouble(70000.0) + .convert_to(ArchDbType::ScalarShort) + .unwrap(), + ArchiverValue::ScalarShort(i16::MAX as i32) + ); + assert_eq!( + ArchiverValue::ScalarInt(300) + .convert_to(ArchDbType::ScalarByte) + .unwrap(), + ArchiverValue::ScalarByte(vec![i8::MAX as u8]) + ); + assert_eq!( + ArchiverValue::ScalarByte(vec![0xFF]) + .convert_to(ArchDbType::ScalarInt) + .unwrap(), + ArchiverValue::ScalarInt(-1), + "bytes are signed DBR_CHAR" + ); + assert_eq!( + ArchiverValue::ScalarDouble(1.5) + .convert_to(ArchDbType::ScalarFloat) + .unwrap(), + ArchiverValue::ScalarFloat(1.5) + ); + } + + #[test] + fn convert_to_string_formats_and_parses() { + assert_eq!( + ArchiverValue::ScalarDouble(2.5) + .convert_to(ArchDbType::ScalarString) + .unwrap(), + ArchiverValue::ScalarString("2.5".into()) + ); + assert_eq!( + ArchiverValue::ScalarString(" 42 ".into()) + .convert_to(ArchDbType::ScalarInt) + .unwrap(), + ArchiverValue::ScalarInt(42) + ); + let err = ArchiverValue::ScalarString("abc".into()) + .convert_to(ArchDbType::ScalarDouble) + .unwrap_err(); + assert!(err.to_string().contains("cannot convert string")); + } + + #[test] + fn convert_to_changes_shape_like_java_thru_number() { + assert_eq!( + ArchiverValue::ScalarInt(7) + .convert_to(ArchDbType::WaveformDouble) + .unwrap(), + ArchiverValue::VectorDouble(vec![7.0]) + ); + assert_eq!( + ArchiverValue::VectorDouble(vec![3.7, 9.9]) + .convert_to(ArchDbType::ScalarInt) + .unwrap(), + ArchiverValue::ScalarInt(3), + "waveform → scalar takes the first element" + ); + let err = ArchiverValue::VectorDouble(vec![]) + .convert_to(ArchDbType::ScalarDouble) + .unwrap_err(); + assert!(err.to_string().contains("empty waveform")); + assert_eq!( + ArchiverValue::VectorInt(vec![1, 2]) + .convert_to(ArchDbType::WaveformString) + .unwrap(), + ArchiverValue::VectorString(vec!["1".into(), "2".into()]) + ); + } + + #[test] + fn convert_to_same_type_and_opaque_bytes() { + let v = ArchiverValue::VectorFloat(vec![1.0, 2.0]); + assert_eq!(v.convert_to(ArchDbType::WaveformFloat).unwrap(), v); + assert!( + ArchiverValue::V4GenericBytes(vec![1]) + .convert_to(ArchDbType::ScalarDouble) + .is_err() + ); + assert!( + ArchiverValue::ScalarDouble(1.0) + .convert_to(ArchDbType::V4GenericBytes) + .is_err() + ); + } + #[test] fn decompose_timestamp_round_trips_through_epoch_parts() { let ts = SystemTime::UNIX_EPOCH + Duration::new(1_700_000_000, 250); diff --git a/tests/api_mgmt.rs b/tests/api_mgmt.rs index 2bfc4f8..57126bf 100644 --- a/tests/api_mgmt.rs +++ b/tests/api_mgmt.rs @@ -1601,6 +1601,54 @@ async fn test_p2_change_type_for_pv_requires_pause() { assert_eq!(body["newType"], 2); } +/// changeTypeForPV converts the stored partitions (Java parity), not +/// only the registry row. +#[tokio::test] +async fn test_p2_change_type_for_pv_converts_stored_data() { + use archiver_core::storage::plainpb::reader::PbFileReader; + use archiver_core::storage::traits::EventStream as _; + use archiver_core::types::{ArchDbType, ArchiverSample, ArchiverValue}; + + let (app, reg, dir) = build_test_app_with_pvs().await; + // Seed a Double partition through a second plugin over the same root. + let seed = + PlainPbStoragePlugin::new("sts", dir.path().to_path_buf(), PartitionGranularity::Hour); + let ts = SystemTime::UNIX_EPOCH + Duration::from_secs(1_700_000_000); + seed.append_event( + "SIM:Sine", + ArchDbType::ScalarDouble, + &ArchiverSample::new(ts, ArchiverValue::ScalarDouble(3.7)), + ) + .await + .unwrap(); + seed.flush_writes().await.unwrap(); + + let req = get_request("/mgmt/bpl/pauseArchivingPV?pv=SIM:Sine"); + assert_eq!( + app.clone().oneshot(req).await.unwrap().status(), + StatusCode::OK + ); + let newtype = ArchDbType::ScalarInt as i32; + let req = get_request(&format!( + "/mgmt/bpl/changeTypeForPV?pv=SIM:Sine&newtype={newtype}" + )); + let resp = app.clone().oneshot(req).await.unwrap(); + assert_eq!(resp.status(), StatusCode::OK); + let body = body_to_json(resp.into_body()).await; + assert_eq!(body["partitionsConverted"], 1); + + let mut rdr = PbFileReader::open(&seed.file_path_for("SIM:Sine", ts)).unwrap(); + assert_eq!(rdr.description().db_type, ArchDbType::ScalarInt); + assert_eq!( + rdr.next_event().unwrap().unwrap().value, + ArchiverValue::ScalarInt(3) + ); + assert_eq!( + reg.get_pv("SIM:Sine").unwrap().unwrap().dbr_type, + ArchDbType::ScalarInt + ); +} + #[tokio::test] async fn test_p2_aggregated_appliance_info_standalone() { let (app, _reg, _dir) = build_test_app_with_pvs().await; diff --git a/tests/plainpb_compat.rs b/tests/plainpb_compat.rs index e3710b8..84230ce 100644 --- a/tests/plainpb_compat.rs +++ b/tests/plainpb_compat.rs @@ -795,6 +795,119 @@ async fn ghost_file_path_records_loss() { ); } +/// changeTypeForPV parity: `convert_pv_type` rewrites every partition +/// of the PV to the new type (thru-number), atomically per file, keeps +/// timestamps/severity/status, is idempotent on re-run, and leaves the +/// PV writable under the new type. +#[tokio::test] +async fn convert_pv_type_rewrites_partitions_thru_number() { + let dir = temp_dir(); + let plugin = + PlainPbStoragePlugin::new("test", dir.path().to_path_buf(), PartitionGranularity::Hour); + let pv = "TEST:Retype"; + let ts1: SystemTime = Utc.with_ymd_and_hms(2026, 9, 1, 12, 0, 0).unwrap().into(); + let ts2 = ts1 + std::time::Duration::from_secs(2 * 3600); + let mut s1 = ArchiverSample::new(ts1, ArchiverValue::ScalarDouble(1.9)); + s1.severity = 2; + s1.status = 3; + let s2 = ArchiverSample::new(ts2, ArchiverValue::ScalarDouble(-2.9)); + for s in [&s1, &s2] { + plugin + .append_event(pv, ArchDbType::ScalarDouble, s) + .await + .unwrap(); + } + plugin.flush_writes().await.unwrap(); + + let n = plugin + .convert_pv_type(pv, ArchDbType::ScalarInt) + .await + .unwrap(); + assert_eq!(n, 2, "both hour partitions rewritten"); + + for (ts, expect, sev) in [ + (ts1, ArchiverValue::ScalarInt(1), 2), + (ts2, ArchiverValue::ScalarInt(-2), 0), + ] { + let path = plugin.file_path_for(pv, ts); + assert!( + !path.with_extension("pb.convert_tmp").exists(), + "temp file must be renamed away" + ); + let mut rdr = PbFileReader::open(&path).unwrap(); + assert_eq!(rdr.description().db_type, ArchDbType::ScalarInt); + assert_eq!(rdr.description().pv_name, pv); + let s = rdr.next_event().unwrap().unwrap(); + assert_eq!(s.value, expect); + assert_eq!(s.timestamp, ts); + assert_eq!(s.severity, sev); + assert!(rdr.next_event().unwrap().is_none()); + } + + // Idempotent: nothing left to convert. + assert_eq!( + plugin + .convert_pv_type(pv, ArchDbType::ScalarInt) + .await + .unwrap(), + 0 + ); + + // The PV stays writable under the new type (slot un-tombstoned, + // header matches). + let s3 = ArchiverSample::new( + ts2 + std::time::Duration::from_secs(60), + ArchiverValue::ScalarInt(5), + ); + plugin + .append_event(pv, ArchDbType::ScalarInt, &s3) + .await + .expect("append under the new type"); + plugin.flush_writes().await.unwrap(); + let mut rdr = PbFileReader::open(&plugin.file_path_for(pv, ts2)).unwrap(); + let mut vals = Vec::new(); + while let Some(s) = rdr.next_event().unwrap() { + vals.push(s.value); + } + assert_eq!( + vals, + vec![ArchiverValue::ScalarInt(-2), ArchiverValue::ScalarInt(5)] + ); +} + +/// A conversion that cannot be represented (opaque V4 bytes) must +/// fail without touching the partition. +#[tokio::test] +async fn convert_pv_type_failure_leaves_partition_intact() { + let dir = temp_dir(); + let plugin = + PlainPbStoragePlugin::new("test", dir.path().to_path_buf(), PartitionGranularity::Hour); + let pv = "TEST:RetypeFail"; + let ts1: SystemTime = Utc.with_ymd_and_hms(2026, 9, 1, 12, 0, 0).unwrap().into(); + let s1 = ArchiverSample::new(ts1, ArchiverValue::ScalarString("abc".into())); + plugin + .append_event(pv, ArchDbType::ScalarString, &s1) + .await + .unwrap(); + plugin.flush_writes().await.unwrap(); + let path = plugin.file_path_for(pv, ts1); + let len_before = std::fs::metadata(&path).unwrap().len(); + + let err = plugin + .convert_pv_type(pv, ArchDbType::ScalarDouble) + .await + .expect_err("'abc' cannot become a number"); + assert!(err.to_string().contains("converting"), "{err:#}"); + assert_eq!(std::fs::metadata(&path).unwrap().len(), len_before); + assert!(!path.with_extension("pb.convert_tmp").exists()); + let mut rdr = PbFileReader::open(&path).unwrap(); + assert_eq!(rdr.description().db_type, ArchDbType::ScalarString); + assert_eq!( + rdr.next_event().unwrap().unwrap().value, + ArchiverValue::ScalarString("abc".into()) + ); +} + /// A timestamp outside chrono's calendar must fail the append with an /// error, not panic the shard's spawn_blocking task. #[tokio::test] From f950733453c459824f01b9015038e72f80f80b09 Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 09:54:14 +0900 Subject: [PATCH 20/55] storage: check the partition type on the cached-writer path too The header comparison ran only when write_cached (re)opened a file; a writer already cached accepted any dbr_type, so the one-type-per- partition rule held only across restarts. CachedWriter now carries the partition's type and every append is checked against it. --- .../archiver-core/src/storage/plainpb/mod.rs | 15 +++++++ tests/plainpb_compat.rs | 42 +++++++++++++++++++ 2 files changed, 57 insertions(+) diff --git a/crates/archiver-core/src/storage/plainpb/mod.rs b/crates/archiver-core/src/storage/plainpb/mod.rs index 16a86dc..76b4be6 100644 --- a/crates/archiver-core/src/storage/plainpb/mod.rs +++ b/crates/archiver-core/src/storage/plainpb/mod.rs @@ -121,6 +121,11 @@ impl Drop for WriterFdGuard { struct CachedWriter { path: PathBuf, writer: BufWriter, + /// The type the partition's header declares (or the type we wrote + /// the header with). Every append is checked against it, so the + /// "one type per partition" rule holds on the cached-writer path + /// too, not only when the file is (re)opened. + dbr_type: ArchDbType, /// `true` between writes and the next successful flush. Lets /// `flush_writes` skip writers that have nothing pending so a /// reader-side `get_data` doesn't pay an O(N) syscall storm @@ -555,6 +560,7 @@ impl PlainPbStoragePlugin { let CachedWriter { path, mut writer, + dbr_type: _, dirty, last_used: _, _fd_guard, @@ -1550,6 +1556,7 @@ impl PlainPbStoragePlugin { slot.writer = Some(CachedWriter { path: path_buf, writer: bw, + dbr_type, // Header bytes (if any) are buffered but not yet // flushed. Mark dirty so the periodic flush picks // them up — without this, a PV that gets created @@ -1566,6 +1573,14 @@ impl PlainPbStoragePlugin { } let cached = slot.writer.as_mut().expect("just inserted"); + if cached.dbr_type != dbr_type { + return Err(anyhow::anyhow!( + "PB partition {path:?} holds {:?} samples; refusing to append a \ + {dbr_type:?} sample (type changed — the partition keeps its \ + header type until it rolls)", + cached.dbr_type + )); + } cached.last_used = SystemTime::now(); // Atomic-at-buffer-layer sample frame: a single `write_all` // means the BufWriter never splits the sample/newline pair diff --git a/tests/plainpb_compat.rs b/tests/plainpb_compat.rs index 84230ce..48791f0 100644 --- a/tests/plainpb_compat.rs +++ b/tests/plainpb_compat.rs @@ -1027,6 +1027,48 @@ async fn append_refuses_type_that_differs_from_partition_header() { assert_eq!(vals, vec![ArchiverValue::ScalarDouble(1.0)]); } +/// The same rule on the cached-writer path: a foreign-type sample for +/// a partition whose writer is still cached (no restart) is refused +/// too, and the buffered good sample still lands. +#[tokio::test] +async fn append_refuses_type_that_differs_from_cached_writer() { + let dir = temp_dir(); + let plugin = + PlainPbStoragePlugin::new("test", dir.path().to_path_buf(), PartitionGranularity::Hour); + let ts1: SystemTime = Utc.with_ymd_and_hms(2026, 9, 1, 12, 0, 0).unwrap().into(); + let pv = "TEST:RetypedCached"; + plugin + .append_event( + pv, + ArchDbType::ScalarDouble, + &ArchiverSample::new(ts1, ArchiverValue::ScalarDouble(1.0)), + ) + .await + .unwrap(); + let err = plugin + .append_event( + pv, + ArchDbType::ScalarInt, + &ArchiverSample::new( + ts1 + std::time::Duration::from_secs(1), + ArchiverValue::ScalarInt(2), + ), + ) + .await + .expect_err("a foreign-type append must be refused on the cached writer"); + assert!( + err.to_string().contains("holds ScalarDouble"), + "unexpected error: {err:#}" + ); + plugin.flush_writes().await.unwrap(); + let mut rdr = PbFileReader::open(&plugin.file_path_for(pv, ts1)).unwrap(); + let mut vals = Vec::new(); + while let Some(s) = rdr.next_event().unwrap() { + vals.push(s.value); + } + assert_eq!(vals, vec![ArchiverValue::ScalarDouble(1.0)]); +} + /// The partition directory is removed out from under a plugin that /// already created it (operator cleanup). The next append must /// recreate it and land, not fail on the stale `known_dirs` entry From 392ebdeb29fcf54a9182ac3afba975cc54cc5f2e Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 10:19:50 +0900 Subject: [PATCH 21/55] engine: make the shard type-change gate in shard_handle_sample stateless PvSample.dbr_type is the registry type captured once per archiving task, so the first-seen map could only ever differ for a new task started with another registry type (changeTypeForPV + resume, deletePV + re-archive) and then dropped every sample until restart. Compare the value's own type instead: that is the pair encode_sample refuses, and it is what latest_observed_dbr is meant to report. --- crates/archiver-engine/src/channel_manager.rs | 59 ++++++------ tests/write_loop_failure.rs | 90 +++++++++++++++++++ 2 files changed, 117 insertions(+), 32 deletions(-) diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index 43a9a7e..a3fb7b1 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -3341,11 +3341,11 @@ fn record_closed_drop(shard: usize, s: &PvSample, phase: &'static str) { /// the worker panicked; restart it and retry once rather than /// dropping this shard's ~1/N of all PVs forever. /// -/// The respawned shard starts with empty in-shard ordering maps -/// (`last_ts` / `last_dbr_type`). That is safe: the PB reader -/// tolerates non-monotonic timestamps (clock backsteps, late -/// backfills), and the per-PV writer-slot Mutex inside the storage -/// layer — not the shard's maps — is the real append-ordering owner. +/// The respawned shard starts with an empty in-shard ordering map +/// (`last_ts`). That is safe: the PB reader tolerates non-monotonic +/// timestamps (clock backsteps, late backfills), and the per-PV +/// writer-slot Mutex inside the storage layer — not the shard's map — +/// is the real append-ordering owner. /// How long a spent respawn budget takes to refill. const RESPAWN_WINDOW: Duration = Duration::from_secs(60); /// Max respawns allowed per shard within [`RESPAWN_WINDOW`]. @@ -3915,10 +3915,10 @@ pub async fn write_loop_with_config( /// the next flush cycle. /// /// The shard worker holds NO ts_updates / flush state of its own. -/// It tracks `last_ts` and `last_dbr_type` only for the -/// per-PV ordering / type-change drop checks; those are local -/// to the shard because the dispatcher's consistent hash pins each -/// PV to one shard. +/// It tracks `last_ts` only for the per-PV ordering drop check; +/// that map is local to the shard because the dispatcher's +/// consistent hash pins each PV to one shard. The type-change +/// check is stateless (see `shard_handle_sample`). async fn shard_append_loop( shard_idx: usize, storage: Arc, @@ -3929,13 +3929,11 @@ async fn shard_append_loop( ) { let append_timeout = cfg.append_timeout; info!(shard = shard_idx, "Shard append loop started"); - // Per-PV state for the in-shard sanity drops. The dispatcher's - // consistent hash keeps each PV in one shard, so these maps - // never see cross-shard PVs. + // Per-PV state for the in-shard ordering drop. The dispatcher's + // consistent hash keeps each PV in one shard, so this map + // never sees cross-shard PVs. let mut last_ts: std::collections::HashMap = std::collections::HashMap::new(); - let mut last_dbr_type: std::collections::HashMap = - std::collections::HashMap::new(); loop { tokio::select! { @@ -3946,7 +3944,6 @@ async fn shard_append_loop( &pending, pv_sample, &mut last_ts, - &mut last_dbr_type, append_timeout, ) .await; @@ -3958,7 +3955,6 @@ async fn shard_append_loop( &pending, &mut sample_rx, &mut last_ts, - &mut last_dbr_type, &cfg, ) .await; @@ -3979,7 +3975,6 @@ async fn shard_handle_sample( pending: &Arc, pv_sample: PvSample, last_ts: &mut std::collections::HashMap, - last_dbr_type: &mut std::collections::HashMap, append_timeout: Duration, ) { let ts = pv_sample.sample.timestamp; @@ -4005,31 +4000,33 @@ async fn shard_handle_sample( return; } - // Type-change drop. The first sample defines the PV's wire - // type; later samples with a different DBR get dropped - // (operator must changeTypeForPV first). - let prev_type = last_dbr_type.insert(pv_sample.pv_name.clone(), pv_sample.dbr_type); - if let Some(prev) = prev_type - && prev != pv_sample.dbr_type - { + // Type-change drop, stateless. `pv_sample.dbr_type` is the + // registry type captured once per archiving task, so it cannot + // change within a task; a remembered first-seen type therefore + // only ever differed for a NEW task started with a different + // registry type (changeTypeForPV + resume, deletePV + re-archive) + // and then dropped every sample for the rest of the process. + // The real signal is the value the IOC now sends disagreeing + // with the archived type — the exact pair `writer::encode_sample` + // refuses — so compare the value's own type. The operator must + // changeTypeForPV to accept the new type. + let observed = pv_sample.sample.value.db_type(); + if observed != pv_sample.dbr_type { if let Some(ref c) = pv_sample.counters { c.type_change_drops.fetch_add(1, Ordering::Relaxed); // Java parity (9f2234f): record the latest observed // DBR so the dropped-events report can show what the // IOC is now sending vs the archived type. c.latest_observed_dbr - .store(pv_sample.dbr_type as i32, Ordering::Relaxed); + .store(observed as i32, Ordering::Relaxed); } debug!( shard = shard_idx, pv = pv_sample.pv_name, - ?prev, - new = ?pv_sample.dbr_type, + archived = ?pv_sample.dbr_type, + ?observed, "Dropping type-changed sample" ); - // Restore prev_type so a single mismatched sample doesn't - // permanently flip our recorded type. - last_dbr_type.insert(pv_sample.pv_name.clone(), prev); return; } @@ -4158,7 +4155,6 @@ async fn shard_drain_on_shutdown( pending: &Arc, sample_rx: &mut mpsc::Receiver, last_ts: &mut std::collections::HashMap, - last_dbr_type: &mut std::collections::HashMap, cfg: &WriteLoopConfig, ) { let drain_per_sample_timeout = cfg.drain_per_sample_timeout; @@ -4186,7 +4182,6 @@ async fn shard_drain_on_shutdown( pending, pv_sample, last_ts, - last_dbr_type, drain_per_sample_timeout, ) .await; diff --git a/tests/write_loop_failure.rs b/tests/write_loop_failure.rs index b442332..e6f529f 100644 --- a/tests/write_loop_failure.rs +++ b/tests/write_loop_failure.rs @@ -1702,3 +1702,93 @@ async fn clean_flush_persists_all_timestamps() { shutdown(sd_tx, join).await; } + +/// `PvSample.dbr_type` is the registry type captured once per +/// archiving task, so a differing type can only arrive from a NEW +/// task: changeTypeForPV + resume, or deletePV + re-archive with a +/// new native type. The shard's first-seen map treated exactly that +/// as a type change and dropped every later sample for the rest of +/// the process while the PV stayed Active. +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn type_gate_accepts_new_registry_type_from_a_new_task() { + let storage = InjectingStorage::new(); + let registry = Arc::new(PvRegistry::in_memory().unwrap()); + registry + .register_pv("A", ArchDbType::ScalarDouble, &SampleMode::Monitor, 1) + .unwrap(); + let (tx, sd_tx, join) = spawn_loop(storage.clone(), registry.clone(), fast_cfg()); + + let gen1 = Arc::new(PvCounters::default()); + tx.send(pv_sample("A", ts(100), 1.0, &gen1)).await.unwrap(); + + // changeTypeForPV flipped the registry to Int; the resumed task + // carries the new type and matching values. + let gen2 = Arc::new(PvCounters::default()); + tx.send(PvSample { + pv_name: "A".to_string(), + dbr_type: ArchDbType::ScalarInt, + sample: ArchiverSample::new(ts(200), ArchiverValue::ScalarInt(2)), + element_count: Some(1), + counters: Some(gen2.clone()), + }) + .await + .unwrap(); + tokio::time::sleep(Duration::from_millis(200)).await; + + let a: Vec<_> = storage + .appends_snapshot() + .into_iter() + .filter(|r| r.pv == "A") + .collect(); + assert_eq!(a.len(), 2, "both generations must reach storage: {a:?}"); + assert_eq!( + gen2.type_change_drops.load(Ordering::Relaxed), + 0, + "a new task's registry type is not a type change" + ); + shutdown(sd_tx, join).await; +} + +/// The real type-change signal is the value the IOC now sends +/// disagreeing with the archived type — the exact pair the PB codec +/// refuses. That must be dropped at the shard with the observed type +/// recorded, not surface as a storage write error. +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn type_gate_drops_value_that_disagrees_with_registry_type() { + let storage = InjectingStorage::new(); + let registry = Arc::new(PvRegistry::in_memory().unwrap()); + registry + .register_pv("A", ArchDbType::ScalarDouble, &SampleMode::Monitor, 1) + .unwrap(); + let (tx, sd_tx, join) = spawn_loop(storage.clone(), registry.clone(), fast_cfg()); + + let counters = Arc::new(PvCounters::default()); + tx.send(pv_sample("A", ts(100), 1.0, &counters)) + .await + .unwrap(); + tx.send(PvSample { + pv_name: "A".to_string(), + dbr_type: ArchDbType::ScalarDouble, + sample: ArchiverSample::new(ts(200), ArchiverValue::ScalarInt(7)), + element_count: Some(1), + counters: Some(counters.clone()), + }) + .await + .unwrap(); + tokio::time::sleep(Duration::from_millis(200)).await; + + let a: Vec<_> = storage + .appends_snapshot() + .into_iter() + .filter(|r| r.pv == "A") + .collect(); + assert_eq!(a.len(), 1, "mismatched value must not reach storage: {a:?}"); + assert_eq!(counters.type_change_drops.load(Ordering::Relaxed), 1); + assert_eq!( + counters.latest_observed_dbr.load(Ordering::Relaxed), + ArchDbType::ScalarInt as i32, + "latest_observed_dbr must name the type the IOC now sends" + ); + assert_eq!(counters.storage_write_errors.load(Ordering::Relaxed), 0); + shutdown(sd_tx, join).await; +} From d520dd919596473cc5c5ceb756d683e643796ae5 Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 10:23:02 +0900 Subject: [PATCH 22/55] engine: keep the shard ordering high-water in PvCounters, not a shard map The shard-lifetime last_ts map outlived the archiving task it described: after deletePV + re-archive the new task inherited the dead task's last timestamp and dropped its first samples as out-of-order. ordering_last_ts_nanos lives in the task's PvCounters, so it dies with the task; the storage writer-slot Mutex stays the real ordering owner. --- crates/archiver-engine/src/channel_manager.rs | 93 +++++++++++-------- tests/write_loop_failure.rs | 45 ++++++++- 2 files changed, 95 insertions(+), 43 deletions(-) diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index a3fb7b1..149a3ab 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -228,6 +228,16 @@ pub struct PvCounters { /// appended. Distinct from `shard_closed_drops` (channel closed): the /// channel was open and draining, but time ran out. pub shutdown_abandoned_drops: AtomicU64, + /// Ordering high-water for the shard's out-of-order drop: the + /// newest timestamp the write pool has handed to storage for this + /// archiving task, as unix nanoseconds (0 = none yet). It lives + /// with the task rather than in a shard-lifetime map so it dies + /// with the task: a deletePV + re-archive, or a resume, starts + /// from a clean slate instead of inheriting a dead task's last + /// timestamp and dropping its first samples. Written only by the + /// shard that owns the PV (the dispatcher's consistent hash pins + /// each PV to one shard); not part of the reported snapshot. + pub ordering_last_ts_nanos: AtomicU64, } impl Default for PvCounters { @@ -249,6 +259,7 @@ impl Default for PvCounters { storage_write_errors: AtomicU64::new(0), shard_closed_drops: AtomicU64::new(0), shutdown_abandoned_drops: AtomicU64::new(0), + ordering_last_ts_nanos: AtomicU64::new(0), } } } @@ -2186,6 +2197,15 @@ async fn monitor_loop( } } +/// Unix nanoseconds for the shard ordering high-water; 0 for any +/// pre-epoch time (the "none yet" sentinel — such timestamps never +/// pass `accept_ioc_timestamp`), saturating past the u64 range. +fn unix_nanos(t: SystemTime) -> u64 { + t.duration_since(SystemTime::UNIX_EPOCH) + .map(|d| u64::try_from(d.as_nanos()).unwrap_or(u64::MAX)) + .unwrap_or(0) +} + fn unix_secs(t: SystemTime) -> i64 { t.duration_since(SystemTime::UNIX_EPOCH) .map(|d| d.as_secs() as i64) @@ -3341,11 +3361,11 @@ fn record_closed_drop(shard: usize, s: &PvSample, phase: &'static str) { /// the worker panicked; restart it and retry once rather than /// dropping this shard's ~1/N of all PVs forever. /// -/// The respawned shard starts with an empty in-shard ordering map -/// (`last_ts`). That is safe: the PB reader tolerates non-monotonic -/// timestamps (clock backsteps, late backfills), and the per-PV -/// writer-slot Mutex inside the storage layer — not the shard's map — -/// is the real append-ordering owner. +/// The respawned shard carries no per-PV state of its own: the +/// ordering high-water lives in each task's `PvCounters`, and the +/// per-PV writer-slot Mutex inside the storage layer — not the shard +/// — is the real append-ordering owner (the PB reader tolerates +/// non-monotonic timestamps: clock backsteps, late backfills). /// How long a spent respawn budget takes to refill. const RESPAWN_WINDOW: Duration = Duration::from_secs(60); /// Max respawns allowed per shard within [`RESPAWN_WINDOW`]. @@ -3914,11 +3934,10 @@ pub async fn write_loop_with_config( /// [`PendingReports`] map so the global flush owner sees it on /// the next flush cycle. /// -/// The shard worker holds NO ts_updates / flush state of its own. -/// It tracks `last_ts` only for the per-PV ordering drop check; -/// that map is local to the shard because the dispatcher's -/// consistent hash pins each PV to one shard. The type-change -/// check is stateless (see `shard_handle_sample`). +/// The shard worker holds NO per-PV state of its own: the ordering +/// drop reads and bumps `PvCounters::ordering_last_ts_nanos` carried +/// by each sample, and the type-change check is stateless (see +/// `shard_handle_sample`). async fn shard_append_loop( shard_idx: usize, storage: Arc, @@ -3929,11 +3948,6 @@ async fn shard_append_loop( ) { let append_timeout = cfg.append_timeout; info!(shard = shard_idx, "Shard append loop started"); - // Per-PV state for the in-shard ordering drop. The dispatcher's - // consistent hash keeps each PV in one shard, so this map - // never sees cross-shard PVs. - let mut last_ts: std::collections::HashMap = - std::collections::HashMap::new(); loop { tokio::select! { @@ -3943,7 +3957,6 @@ async fn shard_append_loop( &storage, &pending, pv_sample, - &mut last_ts, append_timeout, ) .await; @@ -3954,7 +3967,6 @@ async fn shard_append_loop( &storage, &pending, &mut sample_rx, - &mut last_ts, &cfg, ) .await; @@ -3974,30 +3986,28 @@ async fn shard_handle_sample( storage: &Arc, pending: &Arc, pv_sample: PvSample, - last_ts: &mut std::collections::HashMap, append_timeout: Duration, ) { let ts = pv_sample.sample.timestamp; - - // Out-of-order timestamp drop. Storage requires monotonic - // appends per PV; an older timestamp would produce a corrupt - // partition. Tracked via shard-local `last_ts` so the check - // survives flush cycles (the legacy ts_updates-based check - // only caught within-cycle reorderings). - if let Some(prev_ts) = last_ts.get(&pv_sample.pv_name) - && ts < *prev_ts - { - if let Some(ref c) = pv_sample.counters { + let ts_nanos = unix_nanos(ts); + + // Out-of-order timestamp drop. The high-water lives in the + // archiving task's `PvCounters` (see `ordering_last_ts_nanos`), + // so it survives flush cycles but not the task: a sample without + // counters has no task and no ordering state. + if let Some(ref c) = pv_sample.counters { + let prev = c.ordering_last_ts_nanos.load(Ordering::Relaxed); + if prev != 0 && ts_nanos < prev { c.timestamp_drops.fetch_add(1, Ordering::Relaxed); + debug!( + shard = shard_idx, + pv = pv_sample.pv_name, + ?ts, + prev_nanos = prev, + "Dropping out-of-order sample" + ); + return; } - debug!( - shard = shard_idx, - pv = pv_sample.pv_name, - ?ts, - ?prev_ts, - "Dropping out-of-order sample" - ); - return; } // Type-change drop, stateless. `pv_sample.dbr_type` is the @@ -4081,8 +4091,8 @@ async fn shard_handle_sample( }); let res = tokio::time::timeout(append_timeout, join).await; // Conservative ordering high-water (principle 5). Bump - // `last_ts` on EVERY storage-layer return path — success, - // error, panic, timeout — not just on Ok and timeout. + // `ordering_last_ts_nanos` on EVERY storage-layer return path — + // success, error, panic, timeout — not just on Ok and timeout. // // Why all four: // * Ok — sample on disk; obvious bump. @@ -4099,7 +4109,10 @@ async fn shard_handle_sample( // out-of-order and type-changed samples BEFORE we get here, // so this bump only ever sets the high-water to a ts the // shard explicitly accepted into the storage layer. - last_ts.insert(pv_name_for_post.clone(), ts); + if let Some(ref c) = counters_for_post { + c.ordering_last_ts_nanos + .fetch_max(ts_nanos, Ordering::Relaxed); + } match res { Ok(Ok(Ok(()))) => { // Report already went out from inside the closure. @@ -4154,7 +4167,6 @@ async fn shard_drain_on_shutdown( storage: &Arc, pending: &Arc, sample_rx: &mut mpsc::Receiver, - last_ts: &mut std::collections::HashMap, cfg: &WriteLoopConfig, ) { let drain_per_sample_timeout = cfg.drain_per_sample_timeout; @@ -4181,7 +4193,6 @@ async fn shard_drain_on_shutdown( storage, pending, pv_sample, - last_ts, drain_per_sample_timeout, ) .await; diff --git a/tests/write_loop_failure.rs b/tests/write_loop_failure.rs index e6f529f..f66ebb3 100644 --- a/tests/write_loop_failure.rs +++ b/tests/write_loop_failure.rs @@ -1392,12 +1392,16 @@ async fn sharded_pool_routes_all_pvs_correctly() { run_sharded_write_pool(storage_for_pool, registry_for_pool, rx, sd_rx, cfg).await }); - let counters = Arc::new(PvCounters::default()); + // One PvCounters per PV: counters belong to one archiving task, + // and the shard's ordering high-water lives in them. + let counters: Vec> = (0..N_PVS) + .map(|_| Arc::new(PvCounters::default())) + .collect(); // Send 5 samples per PV with strictly increasing timestamps. for s in 0..SAMPLES_PER_PV { for (i, name) in pv_names.iter().enumerate() { let t = ts(10_000 + (i as u64) * 100 + s as u64); - tx.send(pv_sample(name, t, s as f64, &counters)) + tx.send(pv_sample(name, t, s as f64, &counters[i])) .await .unwrap(); } @@ -1792,3 +1796,40 @@ async fn type_gate_drops_value_that_disagrees_with_registry_type() { assert_eq!(counters.storage_write_errors.load(Ordering::Relaxed), 0); shutdown(sd_tx, join).await; } + +/// Ordering state belongs to the archiving task. After deletePV + +/// re-archive the new task must not inherit the dead task's last +/// timestamp and drop its (legitimately older) first samples. +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn ordering_state_dies_with_the_archiving_task() { + let storage = InjectingStorage::new(); + let registry = Arc::new(PvRegistry::in_memory().unwrap()); + registry + .register_pv("A", ArchDbType::ScalarDouble, &SampleMode::Monitor, 1) + .unwrap(); + let (tx, sd_tx, join) = spawn_loop(storage.clone(), registry.clone(), fast_cfg()); + + let gen1 = Arc::new(PvCounters::default()); + tx.send(pv_sample("A", ts(1000), 1.0, &gen1)).await.unwrap(); + tokio::time::sleep(Duration::from_millis(100)).await; + + // Same task: an older sample is still an ordering drop. + tx.send(pv_sample("A", ts(900), 9.0, &gen1)).await.unwrap(); + tokio::time::sleep(Duration::from_millis(100)).await; + assert_eq!(gen1.timestamp_drops.load(Ordering::Relaxed), 1); + + // New task (deletePV + re-archive): older than gen1's last, must be stored. + let gen2 = Arc::new(PvCounters::default()); + tx.send(pv_sample("A", ts(900), 2.0, &gen2)).await.unwrap(); + tokio::time::sleep(Duration::from_millis(200)).await; + + let a: Vec<_> = storage + .appends_snapshot() + .into_iter() + .filter(|r| r.pv == "A") + .map(|r| r.timestamp) + .collect(); + assert_eq!(a, vec![ts(1000), ts(900)], "appends: {a:?}"); + assert_eq!(gen2.timestamp_drops.load(Ordering::Relaxed), 0); + shutdown(sd_tx, join).await; +} From 0ec0105f86eb9ca0952bbaa69f8ef16114e2730f Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 10:24:56 +0900 Subject: [PATCH 23/55] main: treat the write pool as a critical task in RuntimeSupervisor If run_sharded_write_pool ends before shutdown was requested, every producer sees a closed channel and stops silently while the API keeps reporting the PVs as Active. spawn_critical turns that exit (or panic) into a shutdown request and a non-zero exit from main, and the HTTP graceful-shutdown trigger now watches the supervisor as well as SIGINT. --- src/main.rs | 36 +++++++---- src/supervisor.rs | 149 ++++++++++++++++++++++++++++++++++++++++++++-- 2 files changed, 168 insertions(+), 17 deletions(-) diff --git a/src/main.rs b/src/main.rs index dd776e4..8bbfe73 100644 --- a/src/main.rs +++ b/src/main.rs @@ -128,8 +128,8 @@ async fn main() -> anyhow::Result<()> { let storage: Arc = tiered.clone(); // Shutdown signal. - let (shutdown_tx, shutdown_rx) = watch::channel(false); - let mut supervisor = RuntimeSupervisor::new(shutdown_rx); + let (shutdown_tx, _shutdown_rx) = watch::channel(false); + let mut supervisor = RuntimeSupervisor::new(shutdown_tx.clone()); // Load policy config if specified. let policy = @@ -193,7 +193,12 @@ async fn main() -> anyhow::Result<()> { ?flush_period, "Spawning sharded write pool" ); - supervisor.spawn("write_loop", async move { + // Critical: if the pool ends before shutdown was requested, every + // producer sees a closed channel and stops silently while the API + // keeps reporting the PVs as Active. The supervisor turns that + // into a shutdown and a non-zero exit so the service manager + // restarts the process. + supervisor.spawn_critical("write_loop", async move { channel_manager::run_sharded_write_pool( write_storage, write_registry, @@ -382,6 +387,19 @@ async fn main() -> anyhow::Result<()> { let addr = format!("{}:{}", config.listen_addr, config.listen_port); + // Graceful-shutdown trigger for the HTTP server: the OS signal, or + // the supervisor flipping the watch because a critical task died. + let mut supervisor_shutdown = supervisor.shutdown_rx(); + let shutdown_signal = async move { + tokio::select! { + _ = tokio::signal::ctrl_c() => info!("Shutdown signal received"), + _ = supervisor_shutdown.changed() => { + tracing::warn!("Shutdown requested by the supervisor"); + } + } + let _ = shutdown_tx.send(true); + }; + // Serve with optional TLS. if let Some(ref tls) = config.tls { // Install ring crypto provider for rustls (ignore if already installed). @@ -395,9 +413,7 @@ async fn main() -> anyhow::Result<()> { let handle = axum_server::Handle::new(); let shutdown_handle = handle.clone(); tokio::spawn(async move { - tokio::signal::ctrl_c().await.ok(); - info!("Shutdown signal received"); - let _ = shutdown_tx.send(true); + shutdown_signal.await; shutdown_handle.graceful_shutdown(Some(Duration::from_secs(30))); }); @@ -414,15 +430,11 @@ async fn main() -> anyhow::Result<()> { listener, app.into_make_service_with_connect_info::(), ) - .with_graceful_shutdown(async move { - tokio::signal::ctrl_c().await.ok(); - info!("Shutdown signal received"); - let _ = shutdown_tx.send(true); - }); + .with_graceful_shutdown(shutdown_signal); server.await?; } - supervisor.shutdown(write_pool_shutdown_budget).await; + supervisor.shutdown(write_pool_shutdown_budget).await?; info!("Archiver stopped"); Ok(()) } diff --git a/src/supervisor.rs b/src/supervisor.rs index 8083378..cc82ab3 100644 --- a/src/supervisor.rs +++ b/src/supervisor.rs @@ -1,3 +1,4 @@ +use std::sync::{Arc, Mutex}; use std::time::Duration; use tokio::sync::watch; @@ -5,22 +6,32 @@ use tokio::task::{AbortHandle, JoinHandle}; use tracing::{error, info, warn}; pub struct RuntimeSupervisor { - shutdown_rx: watch::Receiver, + shutdown_tx: watch::Sender, handles: Vec<(String, JoinHandle<()>)>, abort_handles: Vec<(String, AbortHandle)>, + /// Name of the first critical task that exited (or panicked) + /// before shutdown was requested. Set by the wrapper spawned in + /// [`spawn_critical`](Self::spawn_critical); read by + /// [`shutdown`](Self::shutdown) to fail the process. + critical_exit: Arc>>, } impl RuntimeSupervisor { - pub fn new(shutdown_rx: watch::Receiver) -> Self { + /// `shutdown_tx` is the process-wide shutdown watch. The + /// supervisor sends `true` on it when a critical task dies, so + /// whoever drives the HTTP server's graceful shutdown must watch + /// the receiver side as well as the OS signal. + pub fn new(shutdown_tx: watch::Sender) -> Self { Self { - shutdown_rx, + shutdown_tx, handles: Vec::new(), abort_handles: Vec::new(), + critical_exit: Arc::new(Mutex::new(None)), } } pub fn shutdown_rx(&self) -> watch::Receiver { - self.shutdown_rx.clone() + self.shutdown_tx.subscribe() } pub fn spawn( @@ -35,7 +46,52 @@ impl RuntimeSupervisor { self.handles.push((name.to_string(), handle)); } - pub async fn shutdown(self, timeout: Duration) { + /// Spawn a task the process cannot run without. If it returns or + /// panics while the shutdown watch still reads `false`, the + /// supervisor requests shutdown and [`shutdown`](Self::shutdown) + /// reports the failure, so the process exits non-zero instead of + /// idling with the task's work silently undone. An exit after + /// shutdown was requested is the normal case and is not reported. + pub fn spawn_critical( + &mut self, + name: &str, + fut: impl std::future::Future + Send + 'static, + ) { + let inner = tokio::spawn(fut); + // Registered so a shutdown-timeout abort reaches the real + // task, not only the wrapper awaiting it. + self.abort_handles + .push((name.to_string(), inner.abort_handle())); + let task_name = name.to_string(); + let shutdown_tx = self.shutdown_tx.clone(); + let critical_exit = self.critical_exit.clone(); + self.spawn(name, async move { + let result = inner.await; + if *shutdown_tx.borrow() { + return; + } + match result { + Ok(()) => error!( + task = task_name, + "Critical task exited before shutdown was requested; shutting down" + ), + Err(e) => error!( + task = task_name, + "Critical task died before shutdown was requested: {e}; shutting down" + ), + } + critical_exit + .lock() + .unwrap_or_else(|e| e.into_inner()) + .get_or_insert(task_name); + let _ = shutdown_tx.send(true); + }); + } + + /// Wait for every task to finish, aborting stragglers after + /// `timeout`. Returns an error naming the critical task if one + /// exited before shutdown was requested. + pub async fn shutdown(self, timeout: Duration) -> anyhow::Result<()> { let count = self.handles.len(); info!(tasks = count, "Waiting for background tasks to complete"); @@ -62,5 +118,88 @@ impl RuntimeSupervisor { } } } + + let failed = self + .critical_exit + .lock() + .unwrap_or_else(|e| e.into_inner()) + .take(); + match failed { + Some(name) => { + anyhow::bail!("critical task `{name}` exited before shutdown was requested") + } + None => Ok(()), + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + async fn wait_for_shutdown(mut rx: watch::Receiver) -> bool { + tokio::time::timeout(Duration::from_secs(2), async move { + while !*rx.borrow_and_update() { + if rx.changed().await.is_err() { + return false; + } + } + true + }) + .await + .unwrap_or(false) + } + + #[tokio::test] + async fn critical_task_early_exit_requests_shutdown_and_fails() { + let (tx, rx) = watch::channel(false); + let mut sup = RuntimeSupervisor::new(tx); + sup.spawn_critical("pool", async {}); + assert!( + wait_for_shutdown(rx).await, + "early exit must request shutdown" + ); + let err = sup + .shutdown(Duration::from_secs(1)) + .await + .expect_err("shutdown must report the early exit"); + assert!(err.to_string().contains("pool"), "{err}"); + } + + #[tokio::test] + async fn critical_task_panic_requests_shutdown_and_fails() { + let (tx, rx) = watch::channel(false); + let mut sup = RuntimeSupervisor::new(tx); + sup.spawn_critical("pool", async { panic!("boom") }); + assert!(wait_for_shutdown(rx).await, "panic must request shutdown"); + assert!(sup.shutdown(Duration::from_secs(1)).await.is_err()); + } + + #[tokio::test] + async fn critical_task_exit_after_shutdown_request_is_clean() { + let (tx, _rx) = watch::channel(false); + let mut sup = RuntimeSupervisor::new(tx.clone()); + let mut watch_rx = sup.shutdown_rx(); + sup.spawn_critical("pool", async move { + while !*watch_rx.borrow_and_update() { + if watch_rx.changed().await.is_err() { + return; + } + } + }); + let _ = tx.send(true); + sup.shutdown(Duration::from_secs(1)) + .await + .expect("exit after a requested shutdown is normal"); + } + + #[tokio::test] + async fn non_critical_task_exit_is_ignored() { + let (tx, rx) = watch::channel(false); + let mut sup = RuntimeSupervisor::new(tx); + sup.spawn("cleanup", async {}); + tokio::time::sleep(Duration::from_millis(50)).await; + assert!(!*rx.borrow(), "plain tasks never request shutdown"); + sup.shutdown(Duration::from_secs(1)).await.unwrap(); } } From 41f027bc89c8c09e9f547fbd05a1211b0c197d7b Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 10:28:18 +0900 Subject: [PATCH 24/55] storage: probe the partition header after the fd reservation in write_cached MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit partition_header opens the partition before the writer does, so under fd exhaustion it was the first call to fail — outside the EMFILE evict-and-retry that only guarded the writer open, and before the budget reservation that evicts an LRU writer. Every retry then hit the same wall until fds freed elsewhere. Reserve first, and give the probe the same evict-LRU-and-retry rule. --- Cargo.lock | 1 + Cargo.toml | 2 + .../archiver-core/src/storage/plainpb/mod.rs | 74 +++++++---- tests/plainpb_compat.rs | 118 ++++++++++++++++++ 4 files changed, 169 insertions(+), 26 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index bac1cdc..f0bee42 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -826,6 +826,7 @@ dependencies = [ "epics-rs", "futures", "hyper", + "libc", "metrics-exporter-prometheus", "prost", "reqwest", diff --git a/Cargo.toml b/Cargo.toml index 083d4af..be9aa9c 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -155,3 +155,5 @@ urlencoding = { workspace = true } epics-rs = { workspace = true } async-trait = { workspace = true } anyhow = { workspace = true } +# RLIMIT_NOFILE clamping in the PlainPB fd-exhaustion tests. +libc = "0.2" diff --git a/crates/archiver-core/src/storage/plainpb/mod.rs b/crates/archiver-core/src/storage/plainpb/mod.rs index 76b4be6..cd37bd0 100644 --- a/crates/archiver-core/src/storage/plainpb/mod.rs +++ b/crates/archiver-core/src/storage/plainpb/mod.rs @@ -1401,32 +1401,6 @@ impl PlainPbStoragePlugin { } if slot.writer.is_none() { - // A partition's header fixes the type readers decode every - // frame with. Appending a sample of another type (the PV - // was retyped at the IOC, or changeTypeForPV ran without a - // partition roll) would write frames the reader cannot - // decode — and, worse, ones it stops at, hiding every - // later sample in the file. Refuse here, at the one place - // partitions are opened for append, so the file stays - // decodable and the loss is a counted write error. - let header_type = partition_header(path)?; - if let Some(existing) = header_type - && existing != dbr_type - { - return Err(anyhow::anyhow!( - "PB partition {path:?} holds {existing:?} samples; refusing to \ - append a {dbr_type:?} sample (type changed — the partition \ - keeps its header type until it rolls)" - )); - } - let needs_header = header_type.is_none(); - // Whether the open below will CREATE the file (add a new - // directory entry). Reliable under the per-PV slot lock: no - // other writer targets this path. Gates the parent-dir - // fsync so a plain reopen (post-eviction) of an - // already-durable file skips the syscall. - let is_new_file = self.fsync_on_flush && !path.exists(); - // Atomic fd-cap reservation: loop trying to reserve a // permit; each failed reservation triggers one LRU // eviction (which drops a CachedWriter, decrementing @@ -1452,6 +1426,54 @@ impl PlainPbStoragePlugin { } }; + // A partition's header fixes the type readers decode every + // frame with. Appending a sample of another type (the PV + // was retyped at the IOC, or changeTypeForPV ran without a + // partition roll) would write frames the reader cannot + // decode — and, worse, ones it stops at, hiding every + // later sample in the file. Refuse here, at the one place + // partitions are opened for append, so the file stays + // decodable and the loss is a counted write error. + // Probe the header only with a slot reserved, and with the + // same EMFILE recovery as the writer open below. The probe + // opens the partition before the writer does, so under fd + // exhaustion it is the first call to fail; without eviction + // here every retry hit the same wall until fds freed + // elsewhere. + let header_type = loop { + match partition_header(path) { + Ok(t) => break t, + Err(e) + if e.downcast_ref::() + .is_some_and(is_too_many_open_files) + && self.evict_lru_writer(pv) => + { + tracing::warn!( + ?path, + "Hit OS file-handle limit probing the partition \ + header; evicted LRU writer and retrying" + ); + } + Err(e) => return Err(e), + } + }; + if let Some(existing) = header_type + && existing != dbr_type + { + return Err(anyhow::anyhow!( + "PB partition {path:?} holds {existing:?} samples; refusing to \ + append a {dbr_type:?} sample (type changed — the partition \ + keeps its header type until it rolls)" + )); + } + let needs_header = header_type.is_none(); + // Whether the open below will CREATE the file (add a new + // directory entry). Reliable under the per-PV slot lock: no + // other writer targets this path. Gates the parent-dir + // fsync so a plain reopen (post-eviction) of an + // already-durable file skips the syscall. + let is_new_file = self.fsync_on_flush && !path.exists(); + // Open with EMFILE/ENFILE recovery: even with our // internal reservation honoured, the OS-wide fd table // can still be exhausted (other processes, other tiers diff --git a/tests/plainpb_compat.rs b/tests/plainpb_compat.rs index 48791f0..888f967 100644 --- a/tests/plainpb_compat.rs +++ b/tests/plainpb_compat.rs @@ -1905,3 +1905,121 @@ async fn fsync_on_flush_rename_persists_across_prefix() { "old name must hold no data after rename" ); } + +/// Clamp `RLIMIT_NOFILE` to the highest fd currently open and fill +/// every free slot below it, so the next `open(2)` fails with EMFILE +/// until some fd is closed. Restored on drop. Relies on nextest's +/// one-process-per-test isolation: under a threaded `cargo test` it +/// would starve concurrently running tests of fds for its duration. +#[cfg(unix)] +struct FdClamp { + orig: libc::rlimit, + _fill: Vec, +} + +#[cfg(unix)] +impl FdClamp { + fn new() -> Self { + let mut orig = libc::rlimit { + rlim_cur: 0, + rlim_max: 0, + }; + // SAFETY: plain libc calls with a valid out-pointer. + assert_eq!( + unsafe { libc::getrlimit(libc::RLIMIT_NOFILE, &mut orig) }, + 0 + ); + let scan_max = i32::try_from(orig.rlim_cur.min(65_536)).unwrap(); + let highest_open = (0..scan_max) + // SAFETY: F_GETFD on an arbitrary fd number is harmless; it + // only reports whether the fd is open. + .filter(|&fd| unsafe { libc::fcntl(fd, libc::F_GETFD) } != -1) + .max() + .expect("at least stdio is open"); + let clamped = libc::rlimit { + rlim_cur: libc::rlim_t::try_from(highest_open + 1).unwrap(), + rlim_max: orig.rlim_max, + }; + // SAFETY: lowering the soft limit of the calling process. + assert_eq!(unsafe { libc::setrlimit(libc::RLIMIT_NOFILE, &clamped) }, 0); + let mut fill = Vec::new(); + loop { + match std::fs::File::open("/dev/null") { + Ok(f) => fill.push(f), + Err(e) => { + assert_eq!(e.raw_os_error(), Some(libc::EMFILE), "{e}"); + break; + } + } + } + Self { orig, _fill: fill } + } +} + +#[cfg(unix)] +impl Drop for FdClamp { + fn drop(&mut self) { + // SAFETY: restoring the limit read in `new`. + unsafe { libc::setrlimit(libc::RLIMIT_NOFILE, &self.orig) }; + } +} + +/// EMFILE at the header probe of a re-opened partition. The probe +/// opens the file before the writer does, so under fd exhaustion it +/// fails first; it must trigger the same evict-LRU-and-retry as the +/// writer open instead of failing the append until fds free elsewhere. +#[cfg(unix)] +#[tokio::test] +async fn append_recovers_from_emfile_at_the_header_probe() { + let dir = temp_dir(); + let plugin = PlainPbStoragePlugin::with_fd_budget( + "test", + dir.path().to_path_buf(), + PartitionGranularity::Hour, + FdBudget::new(3), + ); + let ts1: SystemTime = Utc.with_ymd_and_hms(2026, 9, 1, 12, 0, 0).unwrap().into(); + let ts2 = ts1 + std::time::Duration::from_secs(60); + let (a, b, c) = ("TEST:EmfileA", "TEST:EmfileB", "TEST:EmfileC"); + for (pv, v) in [(a, 1.0), (b, 2.0), (c, 3.0)] { + plugin + .append_event( + pv, + ArchDbType::ScalarDouble, + &ArchiverSample::new(ts1, ArchiverValue::ScalarDouble(v)), + ) + .await + .unwrap(); + } + plugin.flush_writes().await.unwrap(); + // A's partition exists with a header but has no cached writer, so + // the next append re-probes it. B and C stay open: one free budget + // slot, zero free OS fds. + let a_path = plugin.file_path_for(a, ts1); + assert!(plugin.evict_writer_for_path(&a_path)); + + let clamp = FdClamp::new(); + let appended = plugin + .append_event( + a, + ArchDbType::ScalarDouble, + &ArchiverSample::new(ts2, ArchiverValue::ScalarDouble(4.0)), + ) + .await; + drop(clamp); + appended.expect("header probe must evict an LRU writer on EMFILE and retry"); + plugin.flush_writes().await.unwrap(); + + let mut rdr = PbFileReader::open(&a_path).unwrap(); + let mut vals = Vec::new(); + while let Some(s) = rdr.next_event().unwrap() { + vals.push(s.value); + } + assert_eq!( + vals, + vec![ + ArchiverValue::ScalarDouble(1.0), + ArchiverValue::ScalarDouble(4.0) + ] + ); +} From 286e7e0c3726f316fdcb9fc4931453e138272124 Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 10:30:00 +0900 Subject: [PATCH 25/55] engine: give the flush owner's in-flight wait its own shutdown_flush_timeout window phase2_deadline was fixed before the shards_done wait, so a drain that spent the whole drain_total_budget left no time to wait for a ticker flush still in flight, and the final flush was skipped for a flush that was about to finish. main's supervisor budget grows by the same window. --- crates/archiver-engine/src/channel_manager.rs | 21 +++++---- src/main.rs | 12 ++--- tests/write_loop_failure.rs | 45 +++++++++++++++++++ 3 files changed, 65 insertions(+), 13 deletions(-) diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index 149a3ab..b38460c 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -4296,13 +4296,17 @@ async fn flush_owner_loop( // short-circuit, and skip every entry in `pending` — // data on disk but no registry commit. // - // All windows fit inside the operator-configured - // `drain_total_budget`. If a drain or flush is genuinely wedged - // past the budget, we proceed to final flush; that call will - // see `in_flight` still set and short-circuit, but at least we - // didn't pin the process forever. Future samples on restart - // will re-populate `pending` and the registry catches up. - let phase2_deadline = std::time::Instant::now() + drain_total_budget; + // Windows 0 and 1 fit inside the operator-configured + // `drain_total_budget`; window 2 gets its own + // `shutdown_flush_timeout`, because a drain that spends the whole + // budget must not also spend the in-flight wait — the final flush + // would then be skipped for a flush that was about to finish. If + // a drain or flush is genuinely wedged past its window we proceed + // to the final flush; that call sees `in_flight` still set and + // short-circuits, but at least the process isn't pinned forever. + // Future samples on restart re-populate `pending` and the + // registry catches up. `main` sizes the supervisor budget as + // drain_total_budget + 2 × shutdown_flush_timeout (+ margin). if tokio::time::timeout(drain_total_budget, shards_done) .await .is_err() @@ -4317,7 +4321,8 @@ async fn flush_owner_loop( if !min_grace.is_zero() { tokio::time::sleep(min_grace).await; } - while flush_in_flight.load(Ordering::Acquire) && std::time::Instant::now() < phase2_deadline { + let inflight_deadline = std::time::Instant::now() + shutdown_flush_timeout; + while flush_in_flight.load(Ordering::Acquire) && std::time::Instant::now() < inflight_deadline { tokio::time::sleep(Duration::from_millis(50)).await; } if flush_in_flight.load(Ordering::Acquire) { diff --git a/src/main.rs b/src/main.rs index 8bbfe73..35483d9 100644 --- a/src/main.rs +++ b/src/main.rs @@ -180,12 +180,14 @@ async fn main() -> anyhow::Result<()> { ..Default::default() }, }; - // The write pool's worst-case shutdown is the shard drain budget - // followed by the final flush + registry commit. The supervisor - // must outlast both: aborting the pool mid-final-flush leaves the - // drained tail on disk with its timestamps never committed. + // The write pool's worst-case shutdown is the shard drain budget, + // then a wait of up to shutdown_flush_timeout for a ticker flush + // still in flight, then the final flush + registry commit under + // the same timeout. The supervisor must outlast all three: + // aborting the pool mid-final-flush leaves the drained tail on + // disk with its timestamps never committed. let write_pool_shutdown_budget = pool_cfg.write_loop.drain_total_budget - + pool_cfg.write_loop.shutdown_flush_timeout + + 2 * pool_cfg.write_loop.shutdown_flush_timeout + Duration::from_secs(5); info!( shards = pool_cfg.shards, diff --git a/tests/write_loop_failure.rs b/tests/write_loop_failure.rs index f66ebb3..435f300 100644 --- a/tests/write_loop_failure.rs +++ b/tests/write_loop_failure.rs @@ -1833,3 +1833,48 @@ async fn ordering_state_dies_with_the_archiving_task() { assert_eq!(gen2.timestamp_drops.load(Ordering::Relaxed), 0); shutdown(sd_tx, join).await; } + +/// A ticker flush still in flight when shutdown arrives gets its own +/// wait window (`shutdown_flush_timeout`), not whatever the shard +/// drain left of `drain_total_budget`. With the drain budget spent +/// the final flush was skipped and the tail never committed. +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn shutdown_waits_for_in_flight_flush_beyond_drain_budget() { + let cfg = WriteLoopConfig { + flush_period: Duration::from_millis(100), + append_timeout: Duration::from_millis(300), + flush_timeout: Duration::from_millis(200), + drain_per_sample_timeout: Duration::from_millis(300), + drain_total_budget: Duration::from_millis(250), + shutdown_flush_timeout: Duration::from_secs(3), + }; + let storage = InjectingStorage::new(); + // Every flush outlives flush_timeout, so the ticker flush started + // by the first sample is still in flight when shutdown arrives. + storage.set_flush_hang(Duration::from_millis(900)); + + let registry = Arc::new(PvRegistry::in_memory().unwrap()); + registry + .register_pv("A", ArchDbType::ScalarDouble, &SampleMode::Monitor, 1) + .unwrap(); + let counters = Arc::new(PvCounters::default()); + let (tx, sd_tx, join) = spawn_loop(storage.clone(), registry.clone(), cfg); + + tx.send(pv_sample("A", ts(100), 1.0, &counters)) + .await + .unwrap(); + tokio::time::sleep(Duration::from_millis(150)).await; + tx.send(pv_sample("A", ts(200), 2.0, &counters)) + .await + .unwrap(); + tokio::time::sleep(Duration::from_millis(50)).await; + + shutdown(sd_tx, join).await; + + let rec = registry.get_pv("A").unwrap().unwrap(); + assert_eq!( + rec.last_timestamp, + Some(ts(200)), + "final flush must run once the in-flight flush finishes" + ); +} From 75be628c468f04b6bf0653a0374271210315b3df Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 10:31:02 +0900 Subject: [PATCH 26/55] registry: refuse register_pv_with_protocol on an alias row The UPSERT keeps alias_for, so registering an alias name produced a row that archived in-process but was filtered out of every restore query (alias_for IS NULL) and vanished at the next restart. The HTTP handler already refuses this; the registry now does too, for every caller. --- crates/archiver-core/src/registry.rs | 46 ++++++++++++++++++++++++++++ 1 file changed, 46 insertions(+) diff --git a/crates/archiver-core/src/registry.rs b/crates/archiver-core/src/registry.rs index 9c06d38..8d03f43 100644 --- a/crates/archiver-core/src/registry.rs +++ b/crates/archiver-core/src/registry.rs @@ -352,6 +352,21 @@ impl PvRegistry { anyhow::bail!("invalid PV name: {pv_name:?}"); } let conn = self.lock_conn()?; + // Registering makes the name a real PV; an alias row must not + // be turned into one silently. The UPSERT below keeps + // alias_for, so such a row would archive in-process yet be + // filtered out of every restore query (`alias_for IS NULL`) + // and vanish at the next restart. + let alias_target: Option> = conn + .query_row( + "SELECT alias_for FROM pv_info WHERE pv_name = ?1", + params![pv_name], + |r| r.get(0), + ) + .optional()?; + if let Some(Some(target)) = alias_target { + anyhow::bail!("'{pv_name}' is an alias for '{target}'; archive the target PV instead"); + } let now = Utc::now().to_rfc3339(); let (mode_str, period) = sample_mode.to_db(); @@ -1482,6 +1497,37 @@ mod tests { assert!(reg.get_pv("PV:A").unwrap().is_some()); } + #[test] + fn register_pv_refuses_an_alias_name() { + let reg = PvRegistry::in_memory().unwrap(); + reg.register_pv("PV:Real", ArchDbType::ScalarDouble, &SampleMode::Monitor, 1) + .unwrap(); + reg.add_alias("PV:Alias", "PV:Real").unwrap(); + + let err = reg + .register_pv( + "PV:Alias", + ArchDbType::ScalarDouble, + &SampleMode::Monitor, + 1, + ) + .unwrap_err(); + assert!(err.to_string().contains("alias for 'PV:Real'"), "{err}"); + + // The alias row is untouched: still routed, still absent from + // the restore query. + assert_eq!( + reg.resolve_alias("PV:Alias").unwrap().as_deref(), + Some("PV:Real") + ); + assert!( + reg.pvs_by_status(PvStatus::Active) + .unwrap() + .iter() + .all(|r| r.pv_name != "PV:Alias") + ); + } + #[test] fn test_silent_pvs() { let reg = PvRegistry::in_memory().unwrap(); From ad54c4b24b9e6c68886e3531be7e82446a10ef28 Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 10:31:17 +0900 Subject: [PATCH 27/55] registry: refuse a dbr_type change in register_pv_with_protocol A re-archive that found another native type at the IOC flipped the registry type while the current partition kept its header type, so every append was refused until that partition rolled. Only changeTypeForPV (import_pv, after converting the stored partitions) may change a PV's type. --- crates/archiver-core/src/registry.rs | 52 +++++++++++++++++++++++++--- 1 file changed, 48 insertions(+), 4 deletions(-) diff --git a/crates/archiver-core/src/registry.rs b/crates/archiver-core/src/registry.rs index 8d03f43..9925586 100644 --- a/crates/archiver-core/src/registry.rs +++ b/crates/archiver-core/src/registry.rs @@ -357,16 +357,30 @@ impl PvRegistry { // alias_for, so such a row would archive in-process yet be // filtered out of every restore query (`alias_for IS NULL`) // and vanish at the next restart. - let alias_target: Option> = conn + let existing: Option<(i32, Option)> = conn .query_row( - "SELECT alias_for FROM pv_info WHERE pv_name = ?1", + "SELECT dbr_type, alias_for FROM pv_info WHERE pv_name = ?1", params![pv_name], - |r| r.get(0), + |r| Ok((r.get(0)?, r.get(1)?)), ) .optional()?; - if let Some(Some(target)) = alias_target { + if let Some((_, Some(target))) = &existing { anyhow::bail!("'{pv_name}' is an alias for '{target}'; archive the target PV instead"); } + // Only changeTypeForPV (import_pv, after converting the stored + // partitions) may change a PV's type. A re-archive that finds + // another native type at the IOC would flip the registry while + // the current partition keeps its header type, and every append + // would be refused until that partition rolls. + if let Some((existing_type, None)) = existing + && existing_type != dbr_type as i32 + { + anyhow::bail!( + "'{pv_name}' is registered as {:?}; re-archiving cannot change it to \ + {dbr_type:?} — use changeTypeForPV", + ArchDbType::from_i32(existing_type) + ); + } let now = Utc::now().to_rfc3339(); let (mode_str, period) = sample_mode.to_db(); @@ -1497,6 +1511,36 @@ mod tests { assert!(reg.get_pv("PV:A").unwrap().is_some()); } + #[test] + fn register_pv_refuses_a_type_change() { + let reg = PvRegistry::in_memory().unwrap(); + reg.register_pv( + "PV:Typed", + ArchDbType::ScalarDouble, + &SampleMode::Monitor, + 1, + ) + .unwrap(); + let err = reg + .register_pv("PV:Typed", ArchDbType::ScalarInt, &SampleMode::Monitor, 1) + .unwrap_err(); + assert!(err.to_string().contains("changeTypeForPV"), "{err}"); + assert_eq!( + reg.get_pv("PV:Typed").unwrap().unwrap().dbr_type, + ArchDbType::ScalarDouble + ); + // Same type: mode and element count may still be re-registered. + reg.register_pv( + "PV:Typed", + ArchDbType::ScalarDouble, + &SampleMode::Scan { period_secs: 2.0 }, + 4, + ) + .unwrap(); + let rec = reg.get_pv("PV:Typed").unwrap().unwrap(); + assert_eq!(rec.element_count, 4); + } + #[test] fn register_pv_refuses_an_alias_name() { let reg = PvRegistry::in_memory().unwrap(); From 4fcc34a7af649fdb78123f509677ab4877daa0e9 Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 10:33:14 +0900 Subject: [PATCH 28/55] api: hold the ETL move gates across change_type_for_pv's conversion The ETL skips paused PVs only when a run starts, so a move that began before the pause could delete the source partition after the conversion opened it; the converted rewrite then resurrected the partition next to the coarser tier's old-type copy. EtlExecutor now exposes its move gate and the handler takes every distinct gate in chain order for the duration of convert_pv_type. --- crates/archiver-api/src/handlers/mgmt/p2.rs | 19 +++++ crates/archiver-core/src/etl/executor.rs | 8 ++ tests/api_mgmt.rs | 93 +++++++++++++++++++++ 3 files changed, 120 insertions(+) diff --git a/crates/archiver-api/src/handlers/mgmt/p2.rs b/crates/archiver-api/src/handlers/mgmt/p2.rs index c600e94..f607ea4 100644 --- a/crates/archiver-api/src/handlers/mgmt/p2.rs +++ b/crates/archiver-api/src/handlers/mgmt/p2.rs @@ -199,6 +199,24 @@ pub async fn change_type_for_pv( // (the storage layer refuses such appends). Files are the truth: a // crash between the conversion and the registry update is repaired // by re-issuing the request — converted partitions are skipped. + // + // Held under the ETL chain's move gates: the ETL skips paused PVs + // only when a run starts, so a move that began before the pause + // could delete the source partition after the conversion opened + // it, and the converted rewrite would then resurrect the partition + // next to the coarser tier's old-type copy. main shares one gate + // across the chain; distinct gates are taken in chain order. + let mut gates: Vec>> = Vec::new(); + for exec in &state.etl_chain { + let gate = exec.move_gate(); + if !gates.iter().any(|g| std::sync::Arc::ptr_eq(g, &gate)) { + gates.push(gate); + } + } + let mut no_moves_in_flight = Vec::with_capacity(gates.len()); + for gate in &gates { + no_moves_in_flight.push(gate.lock().await); + } let converted = match state.storage.convert_pv_type(&canonical, new_type).await { Ok(n) => n, Err(e) => { @@ -209,6 +227,7 @@ pub async fn change_type_for_pv( .into_response(); } }; + drop(no_moves_in_flight); if let Err(e) = state.pv_cmd.import_pv( &canonical, new_type, diff --git a/crates/archiver-core/src/etl/executor.rs b/crates/archiver-core/src/etl/executor.rs index 7ba3bd1..464790e 100644 --- a/crates/archiver-core/src/etl/executor.rs +++ b/crates/archiver-core/src/etl/executor.rs @@ -106,6 +106,14 @@ impl EtlExecutor { self } + /// The move gate this executor serializes its moves on. Held by + /// operations outside the ETL that rewrite a partition in place + /// (changeTypeForPV's conversion) so they cannot interleave with a + /// move of the same file. + pub fn move_gate(&self) -> Arc> { + self.move_gate.clone() + } + /// Wire a PV registry so the executor can skip paused PVs in /// `run_once`. Java parity (92db337): without this, PB files for a /// paused PV continue to migrate out of the STS, which surprises diff --git a/tests/api_mgmt.rs b/tests/api_mgmt.rs index 57126bf..55fcf90 100644 --- a/tests/api_mgmt.rs +++ b/tests/api_mgmt.rs @@ -1649,6 +1649,99 @@ async fn test_p2_change_type_for_pv_converts_stored_data() { ); } +/// changeTypeForPV must serialize with the ETL chain's move gate. The +/// ETL skips paused PVs only when a run starts, so a move that began +/// before the pause could delete the source partition after the +/// conversion opened it; the converted rewrite would then resurrect +/// the partition next to the coarser tier's old-type copy. +#[tokio::test] +async fn test_p2_change_type_for_pv_waits_for_the_etl_move_gate() { + use archiver_core::etl::executor::EtlExecutor; + use archiver_core::types::{ArchDbType, ArchiverSample, ArchiverValue}; + + let dir = tempfile::tempdir().unwrap(); + let sts = Arc::new(PlainPbStoragePlugin::new( + "sts", + dir.path().join("sts"), + PartitionGranularity::Hour, + )); + let mts = Arc::new(PlainPbStoragePlugin::new( + "mts", + dir.path().join("mts"), + PartitionGranularity::Day, + )); + let registry = Arc::new(PvRegistry::in_memory().unwrap()); + registry + .register_pv( + "SIM:Sine", + ArchDbType::ScalarDouble, + &SampleMode::Monitor, + 1, + ) + .unwrap(); + registry.set_status("SIM:Sine", PvStatus::Paused).unwrap(); + let ts = SystemTime::UNIX_EPOCH + Duration::from_secs(1_700_000_000); + sts.append_event( + "SIM:Sine", + ArchDbType::ScalarDouble, + &ArchiverSample::new(ts, ArchiverValue::ScalarDouble(3.7)), + ) + .await + .unwrap(); + sts.flush_writes().await.unwrap(); + + let executor = Arc::new(EtlExecutor::new(sts.clone(), mts.clone(), 3600, 5, 3)); + let gate = executor.move_gate(); + let (channel_mgr, _rx) = ChannelManager::new(sts.clone(), registry.clone(), None) + .await + .unwrap(); + let repo = Arc::new(RegistryRepository::new(registry.clone())); + let archiver = Arc::new(ChannelArchiverControl::new(Arc::new(channel_mgr))); + let state = AppState { + storage: sts.clone(), + pv_query: repo.clone(), + pv_cmd: repo, + archiver_query: archiver.clone(), + archiver_cmd: archiver, + cluster: None, + api_keys: None, + cluster_api_key: None, + metrics_handle: None, + rate_limiter: None, + trust_proxy_headers: false, + failover: None, + etl_chain: vec![executor], + reassign_appliance_enabled: false, + }; + let app = build_router(state, &SecurityConfig::default()); + + // An ETL move in flight holds the gate. + let in_flight_move = gate.lock().await; + let app_for_req = app.clone(); + let newtype = ArchDbType::ScalarInt as i32; + let request = tokio::spawn(async move { + app_for_req + .oneshot(get_request(&format!( + "/mgmt/bpl/changeTypeForPV?pv=SIM:Sine&newtype={newtype}" + ))) + .await + .unwrap() + }); + tokio::time::sleep(Duration::from_millis(300)).await; + assert!( + !request.is_finished(), + "changeTypeForPV must wait for the in-flight ETL move" + ); + drop(in_flight_move); + let resp = tokio::time::timeout(Duration::from_secs(5), request) + .await + .expect("request completes once the move gate is released") + .unwrap(); + assert_eq!(resp.status(), StatusCode::OK); + let body = body_to_json(resp.into_body()).await; + assert_eq!(body["partitionsConverted"], 1); +} + #[tokio::test] async fn test_p2_aggregated_appliance_info_standalone() { let (app, _reg, _dir) = build_test_app_with_pvs().await; From 6ab47a1f3ea7ffbcdf5e60e93f44e67dfd86a559 Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 10:52:26 +0900 Subject: [PATCH 29/55] supervisor: add shutdown_requested so a critical death is seen regardless of subscribe order main subscribed the HTTP graceful-shutdown trigger only just before serving, and watch::Sender::subscribe marks the current value as seen, so a write_loop that died during ETL/cleanup setup never stopped the server. wait_for(|v| *v) resolves on state, not on an edge. --- src/main.rs | 4 ++-- src/supervisor.rs | 31 +++++++++++++++++++++++++++++++ 2 files changed, 33 insertions(+), 2 deletions(-) diff --git a/src/main.rs b/src/main.rs index 35483d9..daa88e1 100644 --- a/src/main.rs +++ b/src/main.rs @@ -391,11 +391,11 @@ async fn main() -> anyhow::Result<()> { // Graceful-shutdown trigger for the HTTP server: the OS signal, or // the supervisor flipping the watch because a critical task died. - let mut supervisor_shutdown = supervisor.shutdown_rx(); + let supervisor_shutdown = supervisor.shutdown_requested(); let shutdown_signal = async move { tokio::select! { _ = tokio::signal::ctrl_c() => info!("Shutdown signal received"), - _ = supervisor_shutdown.changed() => { + _ = supervisor_shutdown => { tracing::warn!("Shutdown requested by the supervisor"); } } diff --git a/src/supervisor.rs b/src/supervisor.rs index cc82ab3..aee354e 100644 --- a/src/supervisor.rs +++ b/src/supervisor.rs @@ -34,6 +34,20 @@ impl RuntimeSupervisor { self.shutdown_tx.subscribe() } + /// Resolves once shutdown has been requested — including when the + /// request happened before this call. `changed()` on a receiver + /// obtained here cannot express that case: `subscribe` marks the + /// current value as seen, so a critical task that died before the + /// caller subscribed would never wake it. + pub fn shutdown_requested(&self) -> impl std::future::Future + Send + 'static { + let mut rx = self.shutdown_tx.subscribe(); + async move { + // Err means every sender is gone, which only happens on + // the way out — treat it as "requested". + let _ = rx.wait_for(|requested| *requested).await; + } + } + pub fn spawn( &mut self, name: &str, @@ -193,6 +207,23 @@ mod tests { .expect("exit after a requested shutdown is normal"); } + #[tokio::test] + async fn shutdown_requested_resolves_for_a_death_before_the_call() { + let (tx, rx) = watch::channel(false); + let mut sup = RuntimeSupervisor::new(tx); + sup.spawn_critical("pool", async {}); + assert!( + wait_for_shutdown(rx).await, + "early exit must request shutdown" + ); + // The death is already recorded; a waiter created only now + // must still resolve. + tokio::time::timeout(Duration::from_secs(1), sup.shutdown_requested()) + .await + .expect("a waiter created after the death must resolve"); + assert!(sup.shutdown(Duration::from_secs(1)).await.is_err()); + } + #[tokio::test] async fn non_critical_task_exit_is_ignored() { let (tx, rx) = watch::channel(false); From b084c4d86c7cf9b4db78d6ee33997611eee6d1fb Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 10:55:50 +0900 Subject: [PATCH 30/55] engine: charge flush-time losses to the PV's flush_losses counter A PV whose buffered bytes were lost at flush (flush_ingest_writes failed, or a dirty-writer eviction via take_loss_markers) had already counted those samples in events_stored, and the flush owner only logged the loss. PendingReports now carries the appending task's counters so remove_failed can attribute the loss; getPVStatus exposes it as flushLosses and archiver_storage_flush_losses_total counts the reports. --- .../archiver-api/src/handlers/mgmt/engine.rs | 1 + .../src/services/fakes/archiver_control.rs | 1 + .../src/services/impls/archiver_control.rs | 1 + crates/archiver-api/src/services/traits.rs | 3 + crates/archiver-engine/src/channel_manager.rs | 57 +++++++++++---- tests/write_loop_failure.rs | 73 +++++++++++++++++++ 6 files changed, 120 insertions(+), 16 deletions(-) diff --git a/crates/archiver-api/src/handlers/mgmt/engine.rs b/crates/archiver-api/src/handlers/mgmt/engine.rs index 767ecf6..f182a1a 100644 --- a/crates/archiver-api/src/handlers/mgmt/engine.rs +++ b/crates/archiver-api/src/handlers/mgmt/engine.rs @@ -317,6 +317,7 @@ pub async fn pv_status_action( "storageWriteErrors": c.storage_write_errors, "shardClosedDrops": c.shard_closed_drops, "shutdownAbandonedDrops": c.shutdown_abandoned_drops, + "flushLosses": c.flush_losses, "disconnectCount": c.disconnect_count, "lastDisconnectEpochSecs": c.last_disconnect_unix_secs, }) diff --git a/crates/archiver-api/src/services/fakes/archiver_control.rs b/crates/archiver-api/src/services/fakes/archiver_control.rs index 37f0437..871a1ba 100644 --- a/crates/archiver-api/src/services/fakes/archiver_control.rs +++ b/crates/archiver-api/src/services/fakes/archiver_control.rs @@ -108,6 +108,7 @@ impl ArchiverQuery for FakeArchiverControl { storage_write_errors: 0, shard_closed_drops: 0, shutdown_abandoned_drops: 0, + flush_losses: 0, }, ) }) diff --git a/crates/archiver-api/src/services/impls/archiver_control.rs b/crates/archiver-api/src/services/impls/archiver_control.rs index d909470..503153b 100644 --- a/crates/archiver-api/src/services/impls/archiver_control.rs +++ b/crates/archiver-api/src/services/impls/archiver_control.rs @@ -61,6 +61,7 @@ impl ArchiverQuery for ChannelArchiverControl { storage_write_errors: c.storage_write_errors, shard_closed_drops: c.shard_closed_drops, shutdown_abandoned_drops: c.shutdown_abandoned_drops, + flush_losses: c.flush_losses, }, ) }) diff --git a/crates/archiver-api/src/services/traits.rs b/crates/archiver-api/src/services/traits.rs index 1815175..1f160a2 100644 --- a/crates/archiver-api/src/services/traits.rs +++ b/crates/archiver-api/src/services/traits.rs @@ -127,6 +127,9 @@ pub struct PvCountersDto { /// Buffered events abandoned during graceful shutdown because the /// drain budget expired before they could be appended. pub shutdown_abandoned_drops: u64, + /// Flush cycles in which this PV's buffered bytes were lost after + /// their appends had already been counted in `events_stored`. + pub flush_losses: u64, } // --- ArchiverCommand (async — write operations on archiver engine) --- diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index b38460c..739a5bc 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -228,6 +228,12 @@ pub struct PvCounters { /// appended. Distinct from `shard_closed_drops` (channel closed): the /// channel was open and draining, but time ran out. pub shutdown_abandoned_drops: AtomicU64, + /// Flush cycles in which this PV's buffered bytes were lost: + /// `flush_ingest_writes` reported the PV failed, or its writer was + /// evicted dirty (`take_loss_markers`). Those samples were counted + /// in `events_stored` when their append succeeded, so this is the + /// only per-PV trace that they never reached disk. + pub flush_losses: AtomicU64, /// Ordering high-water for the shard's out-of-order drop: the /// newest timestamp the write pool has handed to storage for this /// archiving task, as unix nanoseconds (0 = none yet). It lives @@ -259,6 +265,7 @@ impl Default for PvCounters { storage_write_errors: AtomicU64::new(0), shard_closed_drops: AtomicU64::new(0), shutdown_abandoned_drops: AtomicU64::new(0), + flush_losses: AtomicU64::new(0), ordering_last_ts_nanos: AtomicU64::new(0), } } @@ -285,6 +292,7 @@ pub struct PvCountersSnapshot { pub storage_write_errors: u64, pub shard_closed_drops: u64, pub shutdown_abandoned_drops: u64, + pub flush_losses: u64, } impl From<&PvCounters> for PvCountersSnapshot { @@ -314,6 +322,7 @@ impl From<&PvCounters> for PvCountersSnapshot { storage_write_errors: c.storage_write_errors.load(Ordering::Relaxed), shard_closed_drops: c.shard_closed_drops.load(Ordering::Relaxed), shutdown_abandoned_drops: c.shutdown_abandoned_drops.load(Ordering::Relaxed), + flush_losses: c.flush_losses.load(Ordering::Relaxed), } } } @@ -3697,6 +3706,8 @@ async fn run_flush_and_commit( // occasional under-commit that the next sample // resolves. if !failed.is_empty() { + metrics::counter!("archiver_storage_flush_losses_total") + .increment(failed.len() as u64); pending.remove_failed(&failed); error!( "STS flush dropped {} PV(s) from timestamp commit \ @@ -3825,7 +3836,14 @@ async fn run_flush_and_commit( /// committed value. #[derive(Default)] pub struct PendingReports { - inner: dashmap::DashMap, + inner: dashmap::DashMap, +} + +struct PendingEntry { + ts: SystemTime, + /// Counters of the archiving task that appended the pending + /// bytes, so a flush-time loss is attributed to the PV. + counters: Option>, } impl PendingReports { @@ -3836,18 +3854,21 @@ impl PendingReports { /// Coalescing report. If `pv` already has a newer timestamp, /// no-op. Always succeeds — this is the channel-replacement /// invariant that makes silent-PV under-commit impossible. - pub fn report(&self, pv: &str, ts: SystemTime) { - // DashMap's entry() locks one shard internally. Fast-path - // updates use `and_modify`; brand-new PVs go through - // `or_insert`. No external lock contention across shards. - self.inner + pub fn report(&self, pv: &str, ts: SystemTime, counters: Option>) { + // DashMap's entry() locks one shard internally. No external + // lock contention across shards. + let mut entry = self + .inner .entry(pv.to_string()) - .and_modify(|cur| { - if *cur < ts { - *cur = ts; - } - }) - .or_insert(ts); + .or_insert_with(|| PendingEntry { ts, counters: None }); + if entry.ts < ts { + entry.ts = ts; + } + // The latest reporter owns the attribution: a respawned task + // for the same PV must not credit its losses to a dead one. + if counters.is_some() { + entry.counters = counters; + } } /// Snapshot the entire map for a flush cycle. Returns a @@ -3856,7 +3877,7 @@ impl PendingReports { pub fn snapshot(&self) -> std::collections::HashMap { self.inner .iter() - .map(|kv| (kv.key().clone(), *kv.value())) + .map(|kv| (kv.key().clone(), kv.value().ts)) .collect() } @@ -3868,7 +3889,7 @@ impl PendingReports { for (pv, &committed_ts) in committed { // remove_if applies the predicate atomically inside // the DashMap shard lock. - let _ = self.inner.remove_if(pv, |_, v| *v == committed_ts); + let _ = self.inner.remove_if(pv, |_, v| v.ts == committed_ts); } } @@ -3891,7 +3912,11 @@ impl PendingReports { /// over-commit is never acceptable. pub fn remove_failed(&self, failed: &[String]) { for pv in failed { - let _ = self.inner.remove(pv); + if let Some((_, entry)) = self.inner.remove(pv) + && let Some(c) = entry.counters + { + c.flush_losses.fetch_add(1, Ordering::Relaxed); + } } } @@ -4085,7 +4110,7 @@ async fn shard_handle_sample( // a PV that goes silent right after a successful // append still gets its `last_event` committed once // the owner gets to its next flush cycle. - pending_for_task.report(&pv_sample.pv_name, ts); + pending_for_task.report(&pv_sample.pv_name, ts, counters_in_task); } res }); diff --git a/tests/write_loop_failure.rs b/tests/write_loop_failure.rs index 435f300..799d0d6 100644 --- a/tests/write_loop_failure.rs +++ b/tests/write_loop_failure.rs @@ -1878,3 +1878,76 @@ async fn shutdown_waits_for_in_flight_flush_beyond_drain_budget() { "final flush must run once the in-flight flush finishes" ); } + +/// A flush-time loss is charged to the PV that lost bytes and to no +/// other, once per loss: the storage reports B failed on every +/// flush, but B has pending bytes only in the first cycle. +#[tokio::test] +async fn flush_failed_pv_charges_only_its_flush_losses_counter() { + let storage = InjectingStorage::new(); + storage.set_flush_failed(vec!["B".to_string()]); + let registry = Arc::new(PvRegistry::in_memory().unwrap()); + for pv in ["A", "B", "C", "T"] { + registry + .register_pv(pv, ArchDbType::ScalarDouble, &SampleMode::Monitor, 1) + .unwrap(); + } + let a = Arc::new(PvCounters::default()); + let b = Arc::new(PvCounters::default()); + let c = Arc::new(PvCounters::default()); + let t = Arc::new(PvCounters::default()); + let cfg = fast_cfg(); + let (tx, sd_tx, join) = spawn_loop(storage.clone(), registry.clone(), cfg.clone()); + tx.send(pv_sample("A", ts(10), 1.0, &a)).await.unwrap(); + tx.send(pv_sample("B", ts(20), 2.0, &b)).await.unwrap(); + tx.send(pv_sample("C", ts(30), 3.0, &c)).await.unwrap(); + for i in 0..2u64 { + tokio::time::sleep(cfg.flush_period + Duration::from_millis(50)).await; + tx.send(pv_sample("T", ts(40 + i), 9.0, &t)).await.unwrap(); + tokio::time::sleep(Duration::from_millis(200)).await; + } + assert_eq!( + b.flush_losses.load(Ordering::Relaxed), + 1, + "B lost bytes in one flush" + ); + assert_eq!(a.flush_losses.load(Ordering::Relaxed), 0); + assert_eq!(c.flush_losses.load(Ordering::Relaxed), 0); + assert_eq!( + b.events_stored.load(Ordering::Relaxed), + 1, + "the append succeeded; flush_losses is the only trace of the loss" + ); + shutdown(sd_tx, join).await; +} + +/// A dirty-writer eviction reaches the flush owner through +/// `take_loss_markers`; it charges the same counter as a failed flush. +#[tokio::test] +async fn dirty_eviction_loss_marker_charges_flush_losses() { + let storage = InjectingStorage::new(); + let registry = Arc::new(PvRegistry::in_memory().unwrap()); + for pv in ["A", "T"] { + registry + .register_pv(pv, ArchDbType::ScalarDouble, &SampleMode::Monitor, 1) + .unwrap(); + } + let a = Arc::new(PvCounters::default()); + let t = Arc::new(PvCounters::default()); + let cfg = fast_cfg(); + let (tx, sd_tx, join) = spawn_loop(storage.clone(), registry.clone(), cfg.clone()); + // Marker queued before the append so the first flush finds A + // pending and lost at the same time. + storage.push_loss("A"); + tx.send(pv_sample("A", ts(10), 1.0, &a)).await.unwrap(); + tokio::time::sleep(cfg.flush_period + Duration::from_millis(50)).await; + tx.send(pv_sample("T", ts(40), 9.0, &t)).await.unwrap(); + tokio::time::sleep(Duration::from_millis(200)).await; + assert_eq!(a.flush_losses.load(Ordering::Relaxed), 1); + assert_eq!( + registry.get_pv("A").unwrap().unwrap().last_timestamp, + None, + "a PV whose bytes were lost must not have its timestamp committed" + ); + shutdown(sd_tx, join).await; +} From 6b52feb76878a6f911008cbcb478c02ff261ca38 Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 10:57:54 +0900 Subject: [PATCH 31/55] config: default per_shard_buffer to the main channel capacity split across write_shards With write_shards > 1 the dispatcher drains the 500k main channel instantly into 4096-slot shard channels and drops on Full, so enabling shards cut the burst the engine absorbs before losing samples from 500k to 4096 per shard. Unset now means auto_per_shard_buffer(shards); an explicit value still overrides. Worst-case memory stays at the main channel's order (the shard total equals its capacity). --- crates/archiver-core/src/config.rs | 48 +++++++++++++++---- crates/archiver-engine/src/channel_manager.rs | 28 ++++++++++- src/main.rs | 8 +++- 3 files changed, 71 insertions(+), 13 deletions(-) diff --git a/crates/archiver-core/src/config.rs b/crates/archiver-core/src/config.rs index ec62b25..b8482af 100644 --- a/crates/archiver-core/src/config.rs +++ b/crates/archiver-core/src/config.rs @@ -130,9 +130,13 @@ pub struct EngineConfig { /// shard channel — when a shard is saturated its overflow is /// dropped and recorded on the per-PV `buffer_overflow_drops` /// counter, while OTHER shards keep flowing (per-shard - /// isolation). - #[serde(default = "default_per_shard_buffer")] - pub per_shard_buffer: usize, + /// isolation). Unset (the default) sizes every shard to the main + /// sample channel's capacity divided by `write_shards`, so the + /// shard layer as a whole holds as many samples as the + /// single-worker layout before it starts dropping; set a value + /// to override that. + #[serde(default)] + pub per_shard_buffer: Option, } fn default_write_period() -> u64 { @@ -147,10 +151,6 @@ fn default_write_shards() -> usize { 1 } -fn default_per_shard_buffer() -> usize { - 4096 -} - impl Default for EngineConfig { // `#[serde(default)]` on the outer `engine` field falls back to // this when the TOML omits `[engine]` entirely. Without a manual @@ -165,7 +165,7 @@ impl Default for EngineConfig { policy_file: None, server_ioc_drift_secs: default_server_ioc_drift_secs(), write_shards: default_write_shards(), - per_shard_buffer: default_per_shard_buffer(), + per_shard_buffer: None, } } } @@ -347,7 +347,7 @@ impl ArchiverConfig { "engine.write_shards must be > 0 (use 1 for the legacy single-worker layout)" ); } - if self.engine.per_shard_buffer == 0 && self.engine.write_shards > 1 { + if self.engine.per_shard_buffer == Some(0) && self.engine.write_shards > 1 { anyhow::bail!( "engine.per_shard_buffer must be > 0 when write_shards > 1; \ a 0-capacity shard channel would drop every sample" @@ -429,6 +429,36 @@ partition_granularity = "year" assert!(config.cluster.is_none()); } + #[test] + fn per_shard_buffer_defaults_to_auto_and_rejects_zero_with_shards() { + let base = r#" +[storage.sts] +root_folder = "/tmp/sts" +partition_granularity = "hour" + +[storage.mts] +root_folder = "/tmp/mts" +partition_granularity = "day" + +[storage.lts] +root_folder = "/tmp/lts" +partition_granularity = "year" + +[engine] +write_shards = 2 +"#; + let config = ArchiverConfig::from_toml(base).unwrap(); + assert_eq!(config.engine.per_shard_buffer, None); + config.validate().unwrap(); + + let zero = format!("{base}per_shard_buffer = 0\n"); + let err = ArchiverConfig::from_toml(&zero) + .unwrap() + .validate() + .unwrap_err(); + assert!(err.to_string().contains("per_shard_buffer"), "{err}"); + } + #[test] fn parse_config_with_cluster() { let toml = r#" diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index 739a5bc..b22b69a 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -3171,6 +3171,8 @@ pub struct ShardedWritePoolConfig { /// the dispatcher; samples lost to a closed channel are recorded /// separately under `archiver_dispatcher_shard_closed_drops_total` /// (a distinct cause, kept out of `buffer_overflow_drops`). + /// Production sizes this with [`auto_per_shard_buffer`] unless + /// `engine.per_shard_buffer` is set. pub per_shard_buffer: usize, /// Per-worker config (timeouts, flush period). Cloned into /// each shard. @@ -3181,12 +3183,34 @@ impl Default for ShardedWritePoolConfig { fn default() -> Self { Self { shards: 1, - per_shard_buffer: 4096, + per_shard_buffer: auto_per_shard_buffer(1), write_loop: WriteLoopConfig::default(), } } } +/// Default per-shard channel capacity for `shards` workers: the main +/// sample channel's capacity split evenly. The dispatcher drains the +/// main channel as fast as it can route, so the shard channels are +/// the only buffer in the multi-shard layout; sizing them to the same +/// total keeps the burst the engine absorbs before dropping +/// independent of `shards`. +pub fn auto_per_shard_buffer(shards: usize) -> usize { + (SAMPLE_CHANNEL_CAPACITY / shards.max(1)).max(1) +} + +#[cfg(test)] +mod shard_sizing_tests { + use super::*; + + #[test] + fn auto_per_shard_buffer_keeps_the_aggregate_at_the_main_capacity() { + assert_eq!(auto_per_shard_buffer(1), SAMPLE_CHANNEL_CAPACITY); + assert_eq!(auto_per_shard_buffer(4) * 4, SAMPLE_CHANNEL_CAPACITY); + assert_eq!(auto_per_shard_buffer(0), SAMPLE_CHANNEL_CAPACITY); + } +} + /// Hash a PV name to a shard index in `0..n` — the dispatcher's /// consistent-hash router. Pinning each PV to one shard keeps its /// samples ordered and its per-PV writer slot touched by a single @@ -3944,7 +3968,7 @@ pub async fn write_loop_with_config( let pool_cfg = ShardedWritePoolConfig { shards: 1, // unused on the 1-shard fast path (rx is forwarded directly) - per_shard_buffer: 4096, + per_shard_buffer: auto_per_shard_buffer(1), write_loop: cfg, }; run_sharded_write_pool(storage, registry, rx, shutdown, pool_cfg).await; diff --git a/src/main.rs b/src/main.rs index daa88e1..40651ed 100644 --- a/src/main.rs +++ b/src/main.rs @@ -172,9 +172,13 @@ async fn main() -> anyhow::Result<()> { let write_storage = storage.clone(); let write_registry = registry.clone(); let write_shutdown = supervisor.shutdown_rx(); + let shards = config.engine.write_shards.max(1); let pool_cfg = channel_manager::ShardedWritePoolConfig { - shards: config.engine.write_shards.max(1), - per_shard_buffer: config.engine.per_shard_buffer.max(1), + shards, + per_shard_buffer: config.engine.per_shard_buffer.map_or_else( + || channel_manager::auto_per_shard_buffer(shards), + |buffer| buffer.max(1), + ), write_loop: channel_manager::WriteLoopConfig { flush_period, ..Default::default() From 6ceb10af189f283ef0330942d5d3c29a3c4396b7 Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 11:05:56 +0900 Subject: [PATCH 32/55] registry: promote array PVs to the waveform type in register_pv_with_protocol The CA path registered a DBF_DOUBLE waveform as ScalarDouble because dbr_field_to_arch_type never saw the element count, while epics-ca-rs decodes every count > 1 event as an array, so the per-sample type gate (and before it, encode_sample) refused every CA array sample. One rule now lives in ArchDbType::with_element_count and the registry applies it to every row it writes; init_schema promotes legacy rows once, which needs no partition conversion because no scalar-header file was ever created for them. A scalar channel re-archived with element_count > 1 is now a refused type change (its samples would be waveforms). --- crates/archiver-core/src/registry.rs | 126 ++++++++++++- crates/archiver-core/src/types.rs | 50 +++++ crates/archiver-engine/src/channel_manager.rs | 10 +- tests/save_path_live_ca.rs | 172 ++++++++++++++++++ 4 files changed, 352 insertions(+), 6 deletions(-) create mode 100644 tests/save_path_live_ca.rs diff --git a/crates/archiver-core/src/registry.rs b/crates/archiver-core/src/registry.rs index 9925586..1e79945 100644 --- a/crates/archiver-core/src/registry.rs +++ b/crates/archiver-core/src/registry.rs @@ -9,7 +9,7 @@ use std::time::{Duration, SystemTime}; use chrono::{DateTime, Utc}; use rusqlite::{Connection, OptionalExtension, params}; -use tracing::info; +use tracing::{info, warn}; use crate::types::ArchDbType; @@ -322,6 +322,39 @@ impl PvRegistry { "CREATE INDEX IF NOT EXISTS idx_pv_alias \ ON pv_info(alias_for) WHERE alias_for IS NOT NULL;", )?; + // Step 4: rows registered before the element-count rule lived + // here hold a scalar dbr_type with element_count > 1 (CA + // arrays). Every sample of such a PV is refused as a type + // change, so promote them once. No partition needs converting: + // the encoder refused every array sample before a file with + // the scalar header could be created. + let legacy: Vec<(String, i32, i32)> = { + let mut stmt = conn.prepare( + "SELECT pv_name, dbr_type, element_count FROM pv_info WHERE element_count > 1", + )?; + let rows = stmt.query_map([], |r| Ok((r.get(0)?, r.get(1)?, r.get(2)?)))?; + rows.collect::>()? + }; + let now = Utc::now().to_rfc3339(); + for (pv_name, stored, element_count) in legacy { + let Some(current) = ArchDbType::from_i32(stored) else { + continue; + }; + let wanted = current.with_element_count(element_count); + if wanted != current { + conn.execute( + "UPDATE pv_info SET dbr_type = ?1, updated_at = ?2 WHERE pv_name = ?3", + params![wanted as i32, now, pv_name], + )?; + warn!( + pv = pv_name, + from = ?current, + to = ?wanted, + element_count, + "Promoted array PV to its waveform type" + ); + } + } info!("PV registry schema initialized"); Ok(()) } @@ -351,6 +384,10 @@ impl PvRegistry { if !is_valid_pv_name(pv_name) { anyhow::bail!("invalid PV name: {pv_name:?}"); } + // Arrays archive as waveforms whatever scalar type the caller + // derived from the channel; this is the one place the rule + // is applied, so no writer can store a scalar array row. + let dbr_type = dbr_type.with_element_count(element_count); let conn = self.lock_conn()?; // Registering makes the name a real PV; an alias row must not // be turned into one silently. The UPSERT below keeps @@ -694,6 +731,7 @@ impl PvRegistry { { anyhow::bail!("invalid alias target: {target:?}"); } + let dbr_type = dbr_type.with_element_count(element_count); let conn = self.lock_conn()?; let now = Utc::now().to_rfc3339(); let (mode_str, period) = sample_mode.to_db(); @@ -1529,16 +1567,33 @@ mod tests { reg.get_pv("PV:Typed").unwrap().unwrap().dbr_type, ArchDbType::ScalarDouble ); - // Same type: mode and element count may still be re-registered. + // Same type: the mode may still be re-registered. reg.register_pv( "PV:Typed", ArchDbType::ScalarDouble, &SampleMode::Scan { period_secs: 2.0 }, - 4, + 1, ) .unwrap(); - let rec = reg.get_pv("PV:Typed").unwrap().unwrap(); - assert_eq!(rec.element_count, 4); + // A scalar channel that comes back as an array is a type + // change too: its samples would be waveforms. + let err = reg + .register_pv( + "PV:Typed", + ArchDbType::ScalarDouble, + &SampleMode::Monitor, + 4, + ) + .unwrap_err(); + assert!(err.to_string().contains("WaveformDouble"), "{err}"); + // An array PV may change length within the waveform form. + reg.register_pv("PV:Arr", ArchDbType::ScalarDouble, &SampleMode::Monitor, 4) + .unwrap(); + reg.register_pv("PV:Arr", ArchDbType::ScalarDouble, &SampleMode::Monitor, 8) + .unwrap(); + let rec = reg.get_pv("PV:Arr").unwrap().unwrap(); + assert_eq!(rec.dbr_type, ArchDbType::WaveformDouble); + assert_eq!(rec.element_count, 8); } #[test] @@ -1572,6 +1627,67 @@ mod tests { ); } + #[test] + fn register_pv_promotes_arrays_to_the_waveform_type() { + let reg = PvRegistry::in_memory().unwrap(); + reg.register_pv("PV:Wf", ArchDbType::ScalarDouble, &SampleMode::Monitor, 3) + .unwrap(); + assert_eq!( + reg.get_pv("PV:Wf").unwrap().unwrap().dbr_type, + ArchDbType::WaveformDouble + ); + reg.register_pv("PV:Sc", ArchDbType::ScalarDouble, &SampleMode::Monitor, 1) + .unwrap(); + assert_eq!( + reg.get_pv("PV:Sc").unwrap().unwrap().dbr_type, + ArchDbType::ScalarDouble + ); + // A re-archive derives the scalar form again; that is the same + // type after promotion, not a refused type change. + reg.register_pv("PV:Wf", ArchDbType::ScalarDouble, &SampleMode::Monitor, 3) + .unwrap(); + // The import writer applies the same rule. + reg.import_pv( + "PV:Imp", + ArchDbType::ScalarInt, + &SampleMode::Monitor, + 5, + PvStatus::Active, + None, + None, + None, + None, + &[], + None, + ) + .unwrap(); + assert_eq!( + reg.get_pv("PV:Imp").unwrap().unwrap().dbr_type, + ArchDbType::WaveformInt + ); + } + + #[test] + fn open_promotes_legacy_array_rows_to_the_waveform_type() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("registry.db"); + { + let reg = PvRegistry::open(&path).unwrap(); + let conn = reg.lock_conn().unwrap(); + let now = Utc::now().to_rfc3339(); + conn.execute( + "INSERT INTO pv_info (pv_name, dbr_type, element_count, created_at, updated_at) \ + VALUES ('PV:Legacy', ?1, 3, ?2, ?2)", + params![ArchDbType::ScalarDouble as i32, now], + ) + .unwrap(); + } + let reg = PvRegistry::open(&path).unwrap(); + let rec = reg.get_pv("PV:Legacy").unwrap().unwrap(); + assert_eq!(rec.dbr_type, ArchDbType::WaveformDouble); + assert_eq!(rec.element_count, 3); + } + #[test] fn test_silent_pvs() { let reg = PvRegistry::in_memory().unwrap(); diff --git a/crates/archiver-core/src/types.rs b/crates/archiver-core/src/types.rs index 8bbcd33..35794f0 100644 --- a/crates/archiver-core/src/types.rs +++ b/crates/archiver-core/src/types.rs @@ -63,6 +63,27 @@ impl ArchDbType { | Self::WaveformDouble ) } + + /// The type a PV of this type archives as when its channel carries + /// `element_count` elements: arrays take the waveform form of a + /// scalar type; scalars, waveforms and the V4 types are unchanged. + /// The one element-count rule — the registry applies it to every + /// row it writes, so callers derive only the native scalar type. + pub fn with_element_count(self, element_count: i32) -> Self { + if element_count <= 1 { + return self; + } + match self { + Self::ScalarString => Self::WaveformString, + Self::ScalarShort => Self::WaveformShort, + Self::ScalarFloat => Self::WaveformFloat, + Self::ScalarEnum => Self::WaveformEnum, + Self::ScalarByte => Self::WaveformByte, + Self::ScalarInt => Self::WaveformInt, + Self::ScalarDouble => Self::WaveformDouble, + other => other, + } + } } /// The unified value type for all archived data. @@ -531,3 +552,32 @@ mod tests { assert_eq!(dt.year(), 1969); } } + +#[cfg(test)] +mod element_count_tests { + use super::ArchDbType; + + #[test] + fn with_element_count_promotes_only_scalars_and_only_for_arrays() { + assert_eq!( + ArchDbType::ScalarDouble.with_element_count(1), + ArchDbType::ScalarDouble + ); + assert_eq!( + ArchDbType::ScalarDouble.with_element_count(2), + ArchDbType::WaveformDouble + ); + assert_eq!( + ArchDbType::ScalarByte.with_element_count(0), + ArchDbType::ScalarByte + ); + assert_eq!( + ArchDbType::WaveformDouble.with_element_count(1), + ArchDbType::WaveformDouble + ); + assert_eq!( + ArchDbType::V4GenericBytes.with_element_count(8), + ArchDbType::V4GenericBytes + ); + } +} diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index b22b69a..0692fe0 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -602,7 +602,12 @@ impl ChannelManager { self.start_archiving_internal(&record).await?; metrics::gauge!("archiver_pvs_active").increment(1.0); - info!(pv = pv_name, ?dbr_type, element_count, "Started archiving"); + info!( + pv = pv_name, + dbr_type = ?record.dbr_type, + element_count, + "Started archiving" + ); Ok(()) } @@ -3028,6 +3033,9 @@ fn pv_field_extract_timestamp(field: &PvField) -> SystemTime { } /// Convert epics-base-rs DbFieldType to archiver ArchDbType. +/// Scalar form only. Arrays (`element_count > 1`) are promoted to the +/// waveform form by the registry (`ArchDbType::with_element_count`) +/// when the PV is registered. fn dbr_field_to_arch_type(field_type: DbFieldType) -> ArchDbType { match field_type { DbFieldType::String => ArchDbType::ScalarString, diff --git a/tests/save_path_live_ca.rs b/tests/save_path_live_ca.rs new file mode 100644 index 0000000..01651cd --- /dev/null +++ b/tests/save_path_live_ca.rs @@ -0,0 +1,172 @@ +//! Live save-path smoke over Channel Access: an in-process CA server +//! serving a DOUBLE waveform record → `ChannelManager` CA monitor → +//! sharded write pool → PlainPB on disk + on-disk SQLite registry → +//! read-back through `query_data`. +//! +//! The archiver's own `CaClient` is built from the environment, so the +//! server is reached through `EPICS_CA_ADDR_LIST`. Sets process env; +//! run under nextest (one process per test). + +use std::sync::Arc; +use std::time::{Duration, SystemTime}; + +use archiver_core::registry::{Protocol, PvRegistry, SampleMode}; +use archiver_core::retrieval::query::query_data; +use archiver_core::storage::partition::PartitionGranularity; +use archiver_core::storage::plainpb::PlainPbStoragePlugin; +use archiver_core::storage::traits::StoragePlugin; +use archiver_core::types::{ArchDbType, ArchiverValue}; +use archiver_engine::channel_manager::{ + ChannelManager, PvCountersSnapshot, ShardedWritePoolConfig, WriteLoopConfig, + run_sharded_write_pool, +}; +use epics_rs::base::server::records::waveform::WaveformRecord; +use epics_rs::base::types::{DbFieldType, EpicsValue}; +use epics_rs::ca::client::CaClient; +use epics_rs::ca::server::CaServer; + +const PV: &str = "SMOKE:CA:WF"; +const NELM: i32 = 3; + +fn counters(mgr: &ChannelManager) -> PvCountersSnapshot { + mgr.all_pv_counters() + .into_iter() + .find(|(n, _)| n == PV) + .map(|(_, c)| c) + .expect("counters for PV") +} + +/// A CA array channel registers as the waveform type and its array +/// samples reach the disk as `VectorDouble`, with no type-change drops. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn ca_waveform_samples_land_as_waveform_double() { + // Quiet unless RUST_LOG is set (e.g. archiver_engine=debug). + let _ = tracing_subscriber::fmt() + .with_env_filter(tracing_subscriber::EnvFilter::from_default_env()) + .with_test_writer() + .try_init(); + let dir = tempfile::tempdir().unwrap(); + let storage: Arc = Arc::new(PlainPbStoragePlugin::new( + "sts", + dir.path().join("sts"), + PartitionGranularity::Year, + )); + let registry = Arc::new(PvRegistry::open(&dir.path().join("registry.db")).unwrap()); + + let server = CaServer::builder() + .port(0) + .record(PV, WaveformRecord::new(NELM, DbFieldType::Double)) + .build() + .await + .expect("in-process CA server"); + let port = server.udp_port(); + let server_task = tokio::spawn(async move { server.run().await }); + + // SAFETY: nextest runs each test in its own process; nothing else + // reads these variables concurrently, and they are set before any + // CaClient snapshots its resolver configuration. + unsafe { + std::env::set_var("EPICS_CA_ADDR_LIST", format!("127.0.0.1:{port}")); + std::env::set_var("EPICS_CA_AUTO_ADDR_LIST", "NO"); + std::env::set_var("EPICS_CA_SERVER_PORT", port.to_string()); + } + + let (mgr, rx) = ChannelManager::new(storage.clone(), registry.clone(), None) + .await + .unwrap(); + let mgr = Arc::new(mgr); + let (shutdown_tx, shutdown_rx) = tokio::sync::watch::channel(false); + let pool = tokio::spawn(run_sharded_write_pool( + storage.clone(), + registry.clone(), + rx, + shutdown_rx, + ShardedWritePoolConfig { + shards: 1, + per_shard_buffer: 1024, + write_loop: WriteLoopConfig { + flush_period: Duration::from_millis(300), + ..Default::default() + }, + }, + )); + + // Process the record once before archiving: a never-processed + // record reports TIME = 0, which the drift filter rejects, so the + // connect-time event would otherwise be a timestamp drop rather + // than the first stored sample. + let writer = CaClient::new().await.expect("writer client"); + let ch = writer.create_channel(PV); + ch.wait_connected(Duration::from_secs(10)) + .await + .expect("writer connects"); + ch.put(&EpicsValue::DoubleArray(vec![0.0; NELM as usize])) + .await + .expect("seed caput"); + + let t0 = SystemTime::now(); + mgr.archive_pv(PV, &SampleMode::Monitor, Protocol::Ca) + .await + .expect("archive_pv over CA"); + let rec = registry.get_pv(PV).unwrap().expect("registry row"); + assert_eq!(rec.dbr_type, ArchDbType::WaveformDouble, "{rec:?}"); + assert_eq!(rec.element_count, NELM, "{rec:?}"); + assert_eq!(rec.protocol, Protocol::Ca); + + // Each put processes the record and stamps the value; the + // archiver's monitor sees it as a DoubleArray event. + let arrays: [[f64; 3]; 3] = [[1.0, 2.0, 3.0], [4.0, 5.0, 6.0], [7.0, 8.0, 9.0]]; + for a in &arrays { + ch.put(&EpicsValue::DoubleArray(a.to_vec())) + .await + .expect("caput"); + tokio::time::sleep(Duration::from_millis(100)).await; + } + + let from = t0 - Duration::from_secs(3600); + let to = t0 + Duration::from_secs(3600); + let deadline = tokio::time::Instant::now() + Duration::from_secs(20); + let stored = loop { + let mut stream = query_data(&*storage, PV, from, to, None) + .await + .expect("query_data"); + let mut got: Vec> = Vec::new(); + while let Some(s) = stream.next_event().expect("next_event") { + match s.value { + ArchiverValue::VectorDouble(v) => got.push(v), + other => panic!("stored value is not a waveform: {other:?}"), + } + } + if arrays.iter().all(|a| got.iter().any(|g| g == a)) { + break got; + } + assert!( + tokio::time::Instant::now() < deadline, + "timed out waiting for {arrays:?} on disk; have {got:?}; counters {:?}", + counters(&mgr) + ); + tokio::time::sleep(Duration::from_millis(200)).await; + }; + // Stream order is timestamp order; the connect-time value may + // precede the posted arrays. + let posted: Vec> = stored + .into_iter() + .filter(|g| arrays.iter().any(|a| a == g.as_slice())) + .collect(); + let want: Vec> = arrays.iter().map(|a| a.to_vec()).collect(); + assert_eq!(posted, want, "on-disk sample order/content"); + + let c = counters(&mgr); + assert_eq!(c.type_change_drops, 0, "{c:?}"); + assert_eq!(c.storage_write_errors, 0, "{c:?}"); + assert_eq!(c.timestamp_drops, 0, "{c:?}"); + assert!(c.events_stored >= arrays.len() as u64, "{c:?}"); + + mgr.stop_pv(PV).await.unwrap(); + shutdown_tx.send(true).unwrap(); + tokio::time::timeout(Duration::from_secs(10), pool) + .await + .expect("write pool exits on shutdown") + .unwrap(); + server_task.abort(); +} From 286a62cbeefa13184c036820dfb2c25420737dbe Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 11:18:37 +0900 Subject: [PATCH 33/55] engine: promote count-1 CA events to the waveform form in epics_value_to_archiver CA autosize subscriptions carry the record's current NORD, and a count of 1 decodes as the scalar variant, so a waveform PV holding one element produced scalar samples that the shard type gate dropped. The converter now takes the registered ArchDbType and wraps a scalar into its one-element vector (ArchiverValue::into_vector, the same pairing as ArchDbType::with_element_count). --- crates/archiver-core/src/types.rs | 73 ++++++++++++++++++ crates/archiver-engine/src/channel_manager.rs | 74 +++++++++++++++++-- tests/save_path_live_ca.rs | 14 ++-- 3 files changed, 146 insertions(+), 15 deletions(-) diff --git a/crates/archiver-core/src/types.rs b/crates/archiver-core/src/types.rs index 35794f0..a09adda 100644 --- a/crates/archiver-core/src/types.rs +++ b/crates/archiver-core/src/types.rs @@ -127,6 +127,23 @@ impl ArchiverValue { } } + /// The one-element waveform form of a scalar; waveforms and + /// `V4GenericBytes` are returned unchanged. The scalar → vector + /// pairing is the one [`ArchDbType::with_element_count`] uses, so a + /// promoted value always matches the registered waveform type. + pub fn into_vector(self) -> Self { + match self { + Self::ScalarString(s) => Self::VectorString(vec![s]), + Self::ScalarByte(b) => Self::VectorChar(b), + Self::ScalarShort(v) => Self::VectorShort(vec![v]), + Self::ScalarInt(v) => Self::VectorInt(vec![v]), + Self::ScalarEnum(v) => Self::VectorEnum(vec![v]), + Self::ScalarFloat(v) => Self::VectorFloat(vec![v]), + Self::ScalarDouble(v) => Self::VectorDouble(vec![v]), + other => other, + } + } + /// Convert to another archived type "through a number" — the rule /// `changeTypeForPV` applies to a PV's existing partitions (Java /// parity: `ThruNumberConversion`). Numeric families convert via @@ -581,3 +598,59 @@ mod element_count_tests { ); } } + +#[cfg(test)] +mod into_vector_tests { + use super::*; + + /// Every scalar promotes to the vector variant whose `db_type` is + /// the registry's waveform promotion of the scalar's type. + #[test] + fn scalars_become_one_element_waveforms_of_the_matching_type() { + let cases = [ + ( + ArchiverValue::ScalarString("a".into()), + ArchiverValue::VectorString(vec!["a".into()]), + ), + ( + ArchiverValue::ScalarByte(vec![7]), + ArchiverValue::VectorChar(vec![7]), + ), + ( + ArchiverValue::ScalarShort(1), + ArchiverValue::VectorShort(vec![1]), + ), + ( + ArchiverValue::ScalarInt(2), + ArchiverValue::VectorInt(vec![2]), + ), + ( + ArchiverValue::ScalarEnum(3), + ArchiverValue::VectorEnum(vec![3]), + ), + ( + ArchiverValue::ScalarFloat(1.5), + ArchiverValue::VectorFloat(vec![1.5]), + ), + ( + ArchiverValue::ScalarDouble(2.5), + ArchiverValue::VectorDouble(vec![2.5]), + ), + ]; + for (scalar, vector) in cases { + assert_eq!(scalar.db_type().with_element_count(2), vector.db_type()); + assert_eq!(scalar.into_vector(), vector); + } + } + + #[test] + fn waveforms_and_v4_are_unchanged() { + for v in [ + ArchiverValue::VectorDouble(vec![]), + ArchiverValue::VectorString(vec!["x".into()]), + ArchiverValue::V4GenericBytes(vec![1, 2]), + ] { + assert_eq!(v.clone().into_vector(), v); + } + } +} diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index 0692fe0..bcf4897 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -1174,7 +1174,10 @@ impl ChannelManager { pv_name: &str, timeout: Duration, ) -> Option> { - let channel_opt = self.channels.get(pv_name)?.channel.clone(); + let (channel_opt, dbr_type) = { + let handle = self.channels.get(pv_name)?; + (handle.channel.clone(), handle.dbr_type) + }; let Some(channel) = channel_opt else { // PVA-acquired PV — live_value via pvget_full so we get // the canonical channel descriptor alongside the value; @@ -1199,7 +1202,7 @@ impl ChannelManager { ))); } match tokio::time::timeout(timeout, channel.get()).await { - Ok(Ok((_dbr_type, val))) => Some(Ok(epics_value_to_archiver(&val))), + Ok(Ok((_dbr_type, val))) => Some(Ok(epics_value_to_archiver(dbr_type, &val))), Ok(Err(e)) => Some(Err(anyhow::anyhow!("CA get failed: {e}"))), Err(_) => Some(Err(anyhow::anyhow!("CA get timed out after {timeout:?}"))), } @@ -2159,7 +2162,8 @@ async fn monitor_loop( Ordering::Relaxed, Ordering::Relaxed, ); - let archiver_val = epics_value_to_archiver(&snapshot.value); + let archiver_val = + epics_value_to_archiver(dbr_type, &snapshot.value); let mut sample = ArchiverSample::new(snapshot.timestamp.into(), archiver_val); attach_extras(&extras, &mut sample); @@ -2347,7 +2351,7 @@ async fn scan_loop( Ordering::Relaxed, Ordering::Relaxed, ); - let archiver_val = epics_value_to_archiver(&epics_val); + let archiver_val = epics_value_to_archiver(dbr_type, &epics_val); let mut sample = ArchiverSample::new(now, archiver_val); attach_extras(&extras, &mut sample); if first_after_connect { @@ -3059,9 +3063,17 @@ fn dbr_field_to_arch_type(field_type: DbFieldType) -> ArchDbType { } } -/// Convert epics-base-rs EpicsValue to archiver ArchiverValue. -fn epics_value_to_archiver(val: &EpicsValue) -> ArchiverValue { - match val { +/// Convert an epics-base-rs `EpicsValue` into the `ArchiverValue` of a +/// PV registered as `dbr_type`. +/// +/// CA delivers a waveform whose current element count (NORD) is 1 as +/// the scalar variant — autosize subscriptions and gets carry count 1, +/// which decodes as a scalar — so a scalar arriving for a waveform PV +/// is promoted to its one-element vector here. The shard type gate +/// then compares the registered type against itself. An array arriving +/// for a scalar PV is left alone: that is a real type change. +fn epics_value_to_archiver(dbr_type: ArchDbType, val: &EpicsValue) -> ArchiverValue { + let value = match val { EpicsValue::String(s) => ArchiverValue::ScalarString(s.to_string()), EpicsValue::Short(v) => ArchiverValue::ScalarShort(*v as i32), EpicsValue::Float(v) => ArchiverValue::ScalarFloat(*v), @@ -3104,6 +3116,54 @@ fn epics_value_to_archiver(val: &EpicsValue) -> ArchiverValue { EpicsValue::StringArray(v) => { ArchiverValue::VectorString(v.iter().map(|s| s.to_string()).collect()) } + }; + if dbr_type.is_waveform() { + value.into_vector() + } else { + value + } +} + +#[cfg(test)] +mod ca_value_tests { + use super::*; + + /// A count-1 CA event on a waveform PV decodes as a scalar; the + /// converter stores it as the PV's one-element waveform. + #[test] + fn scalar_event_on_a_waveform_pv_is_a_one_element_vector() { + assert_eq!( + epics_value_to_archiver(ArchDbType::WaveformDouble, &EpicsValue::Double(5.0)), + ArchiverValue::VectorDouble(vec![5.0]) + ); + assert_eq!( + epics_value_to_archiver(ArchDbType::WaveformInt, &EpicsValue::Long(7)), + ArchiverValue::VectorInt(vec![7]) + ); + } + + #[test] + fn scalar_pvs_and_array_events_are_unchanged() { + assert_eq!( + epics_value_to_archiver(ArchDbType::ScalarDouble, &EpicsValue::Double(5.0)), + ArchiverValue::ScalarDouble(5.0) + ); + assert_eq!( + epics_value_to_archiver( + ArchDbType::WaveformDouble, + &EpicsValue::DoubleArray(vec![1.0, 2.0]) + ), + ArchiverValue::VectorDouble(vec![1.0, 2.0]) + ); + // An array on a scalar PV stays a type change for the gate. + assert_eq!( + epics_value_to_archiver( + ArchDbType::ScalarDouble, + &EpicsValue::DoubleArray(vec![1.0]) + ) + .db_type(), + ArchDbType::WaveformDouble + ); } } diff --git a/tests/save_path_live_ca.rs b/tests/save_path_live_ca.rs index 01651cd..7be9b7c 100644 --- a/tests/save_path_live_ca.rs +++ b/tests/save_path_live_ca.rs @@ -38,6 +38,8 @@ fn counters(mgr: &ChannelManager) -> PvCountersSnapshot { /// A CA array channel registers as the waveform type and its array /// samples reach the disk as `VectorDouble`, with no type-change drops. +/// The one-element put sets NORD to 1, which CA delivers as a scalar +/// event; it must still be stored as a one-element waveform. #[tokio::test(flavor = "multi_thread", worker_threads = 4)] async fn ca_waveform_samples_land_as_waveform_double() { // Quiet unless RUST_LOG is set (e.g. archiver_engine=debug). @@ -115,9 +117,9 @@ async fn ca_waveform_samples_land_as_waveform_double() { // Each put processes the record and stamps the value; the // archiver's monitor sees it as a DoubleArray event. - let arrays: [[f64; 3]; 3] = [[1.0, 2.0, 3.0], [4.0, 5.0, 6.0], [7.0, 8.0, 9.0]]; + let arrays: Vec> = vec![vec![1.0, 2.0, 3.0], vec![5.0], vec![7.0, 8.0, 9.0]]; for a in &arrays { - ch.put(&EpicsValue::DoubleArray(a.to_vec())) + ch.put(&EpicsValue::DoubleArray(a.clone())) .await .expect("caput"); tokio::time::sleep(Duration::from_millis(100)).await; @@ -149,12 +151,8 @@ async fn ca_waveform_samples_land_as_waveform_double() { }; // Stream order is timestamp order; the connect-time value may // precede the posted arrays. - let posted: Vec> = stored - .into_iter() - .filter(|g| arrays.iter().any(|a| a == g.as_slice())) - .collect(); - let want: Vec> = arrays.iter().map(|a| a.to_vec()).collect(); - assert_eq!(posted, want, "on-disk sample order/content"); + let posted: Vec> = stored.into_iter().filter(|g| arrays.contains(g)).collect(); + assert_eq!(posted, arrays, "on-disk sample order/content"); let c = counters(&mgr); assert_eq!(c.type_change_drops, 0, "{c:?}"); From 8aaf110043ceb9502da1975e0c1713ce7d15c732 Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 11:20:40 +0900 Subject: [PATCH 34/55] mgmt: refuse a scalar target in change_type_for_pv for array PVs The registry promotes any type imported with element_count > 1 to the waveform form, so converting an array PV's partitions to a scalar type left the files and the registry row disagreeing and every later append failing the partition header check. Apply the same ArchDbType::with_element_count rule before the conversion starts. --- crates/archiver-api/src/handlers/mgmt/p2.rs | 14 ++++++ tests/api_mgmt.rs | 47 +++++++++++++++++++++ 2 files changed, 61 insertions(+) diff --git a/crates/archiver-api/src/handlers/mgmt/p2.rs b/crates/archiver-api/src/handlers/mgmt/p2.rs index f607ea4..e7d3803 100644 --- a/crates/archiver-api/src/handlers/mgmt/p2.rs +++ b/crates/archiver-api/src/handlers/mgmt/p2.rs @@ -193,6 +193,20 @@ pub async fn change_type_for_pv( return ApiError::BadRequest(format!("invalid newtype: {}", q.newtype)).into_response(); } }; + // The registry stores an array PV only in the waveform form + // (`ArchDbType::with_element_count`), so a scalar target here would + // convert the partitions to the scalar type while the registry row + // came back as the waveform type, and every later append would fail + // the partition header check. Refuse before touching the files. + let stored_type = new_type.with_element_count(record.element_count); + if stored_type != new_type { + return ApiError::BadRequest(format!( + "PV '{}' has element_count {}; {new_type:?} is a scalar type \ + and would be stored as {stored_type:?}", + q.pv, record.element_count + )) + .into_response(); + } // Java parity: changeTypeForPV converts the PV's stored data to the // new type (thru-number conversion) BEFORE the registry flips, so no // partition is left whose header disagrees with the archived type diff --git a/tests/api_mgmt.rs b/tests/api_mgmt.rs index 55fcf90..f4f213f 100644 --- a/tests/api_mgmt.rs +++ b/tests/api_mgmt.rs @@ -1649,6 +1649,53 @@ async fn test_p2_change_type_for_pv_converts_stored_data() { ); } +/// changeTypeForPV refuses a scalar target for an array PV: the +/// registry would promote the imported type back to the waveform form +/// after the partitions had been converted to the scalar type. +#[tokio::test] +async fn test_p2_change_type_for_pv_refuses_a_scalar_type_for_an_array_pv() { + use archiver_core::types::ArchDbType; + + let (app, reg, _dir) = build_test_app_with_pvs().await; + reg.register_pv( + "SIM:Wave", + ArchDbType::WaveformDouble, + &SampleMode::Monitor, + 4, + ) + .unwrap(); + let req = get_request("/mgmt/bpl/pauseArchivingPV?pv=SIM:Wave"); + assert_eq!( + app.clone().oneshot(req).await.unwrap().status(), + StatusCode::OK + ); + + let newtype = ArchDbType::ScalarInt as i32; + let req = get_request(&format!( + "/mgmt/bpl/changeTypeForPV?pv=SIM:Wave&newtype={newtype}" + )); + let resp = app.clone().oneshot(req).await.unwrap(); + assert_eq!(resp.status(), StatusCode::BAD_REQUEST); + assert_eq!( + reg.get_pv("SIM:Wave").unwrap().unwrap().dbr_type, + ArchDbType::WaveformDouble + ); + + // The waveform form of the target family is accepted. + let newtype = ArchDbType::WaveformInt as i32; + let req = get_request(&format!( + "/mgmt/bpl/changeTypeForPV?pv=SIM:Wave&newtype={newtype}" + )); + assert_eq!( + app.clone().oneshot(req).await.unwrap().status(), + StatusCode::OK + ); + assert_eq!( + reg.get_pv("SIM:Wave").unwrap().unwrap().dbr_type, + ArchDbType::WaveformInt + ); +} + /// changeTypeForPV must serialize with the ETL chain's move gate. The /// ETL skips paused PVs only when a run starts, so a move that began /// before the pause could delete the source partition after the From 83f8048c9aa71f16ec7984b0a733f3647cb4042f Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 11:20:40 +0900 Subject: [PATCH 35/55] engine: type untyped PVA scalar arrays from the introspection The decoder leaves string arrays untyped, and both pv_field_to_arch_db_type and pv_field_scalar_to_archiver inferred the element type from the first element, so an empty string array could neither register nor be stored. The channel FieldDesc names the element type; the first element is only the fallback when it does not. --- crates/archiver-engine/src/channel_manager.rs | 85 +++++++++++++++---- 1 file changed, 68 insertions(+), 17 deletions(-) diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index bcf4897..6fd2394 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -11,7 +11,7 @@ use epics_rs::pva::client_native::PvaClient; use epics_rs::pva::client_native::ops_v2::MonitorConnEvent; use epics_rs::pva::proto::ByteOrder; use epics_rs::pva::pvdata::encode::{encode_pv_field, encode_type_desc}; -use epics_rs::pva::pvdata::{PvField, ScalarType, ScalarValue, TypedScalarArray}; +use epics_rs::pva::pvdata::{FieldDesc, PvField, ScalarType, ScalarValue, TypedScalarArray}; use tokio::sync::mpsc; use tokio_util::sync::CancellationToken; use tracing::{debug, error, info, warn}; @@ -654,7 +654,7 @@ impl ChannelManager { .map_err(|_| anyhow::anyhow!("PVA pvget for {pv_name} timed out"))? .map_err(|e| anyhow::anyhow!("Failed to pvget {pv_name}: {e}"))?; let (dbr_type, element_count) = - pv_field_to_arch_db_type(&initial.value).ok_or_else(|| { + pv_field_to_arch_db_type(&initial.value, &initial.introspection).ok_or_else(|| { anyhow::anyhow!("PV {pv_name}: empty PVA scalar-array; cannot infer element type") })?; @@ -2606,6 +2606,28 @@ fn pv_value_field(field: &PvField) -> &PvField { } } +/// Descriptor-side twin of [`pv_value_field`]: the `value` member of a +/// structure descriptor, or the descriptor itself. +fn pv_value_desc(desc: &FieldDesc) -> &FieldDesc { + match desc { + FieldDesc::Structure { fields, .. } => fields + .iter() + .find(|(name, _)| name == "value") + .map_or(desc, |(_, d)| d), + _ => desc, + } +} + +/// Element type of an untyped scalar array: the channel introspection +/// names it even when the array is empty; the first element is only +/// the fallback when the descriptor does not describe a scalar array. +fn untyped_array_element_type(items: &[ScalarValue], desc: &FieldDesc) -> Option { + match pv_value_desc(desc) { + FieldDesc::ScalarArray(st) => Some(*st), + _ => items.first().map(ScalarValue::scalar_type), + } +} + /// Map a PVA `ScalarValue` to an archiver `ArchiverValue`. fn scalar_value_to_archiver(s: &ScalarValue) -> ArchiverValue { match s { @@ -2652,10 +2674,10 @@ fn pv_field_scalar_to_archiver( PvField::Scalar(s) => Some(scalar_value_to_archiver(s)), PvField::ScalarArrayTyped(arr) => Some(typed_scalar_array_to_archiver(arr)), PvField::ScalarArray(items) => { - // Legacy untyped scalar-array path (rare with pvxs 0.14+, - // which decodes into `ScalarArrayTyped`). Promote to the - // typed form so a single converter covers both. - let st = items.first()?.scalar_type(); + // Untyped scalar-array path: the decoder leaves string + // arrays untyped. Promote to the typed form so a single + // converter covers both; an empty array is a valid sample. + let st = untyped_array_element_type(items, canonical_desc)?; let typed = TypedScalarArray::from_scalar_values(items, st)?; Some(typed_scalar_array_to_archiver(&typed)) } @@ -2678,7 +2700,7 @@ fn pv_field_scalar_to_archiver( /// archived as [`ArchDbType::V4GenericBytes`] with element_count = 1 /// (the structure as a whole is one sample; inner cardinality is /// carried inside the wire-encoded bytes). -fn pv_field_to_arch_db_type(field: &PvField) -> Option<(ArchDbType, i32)> { +fn pv_field_to_arch_db_type(field: &PvField, desc: &FieldDesc) -> Option<(ArchDbType, i32)> { // NTEnum: register as ScalarEnum (the index is the archived // value); choices live in extras_cache rather than the dbr_type. if nt_enum_parts(field).is_some() { @@ -2688,11 +2710,8 @@ fn pv_field_to_arch_db_type(field: &PvField) -> Option<(ArchDbType, i32)> { Some(match value { PvField::Scalar(s) => (scalar_value_to_arch_db_type(s), 1), PvField::ScalarArray(arr) => { - let elem = arr.first()?; - ( - scalar_type_to_waveform(elem.scalar_type()), - arr.len() as i32, - ) + let st = untyped_array_element_type(arr, desc)?; + (scalar_type_to_waveform(st), arr.len() as i32) } PvField::ScalarArrayTyped(arr) => { (scalar_type_to_waveform(arr.scalar_type()), arr.len() as i32) @@ -4477,6 +4496,38 @@ mod pva_mapping_tests { use epics_rs::pva::pvdata::encode::{decode_pv_field, decode_type_desc}; use std::io::Cursor; + /// String arrays arrive untyped from the decoder; an empty one has + /// no first element to infer from, so the element type must come + /// from the channel introspection, at registration and per sample. + #[test] + fn empty_untyped_string_array_is_typed_by_the_introspection() { + let canonical = FieldDesc::Structure { + struct_id: "epics:nt/NTScalarArray:1.0".into(), + fields: vec![("value".into(), FieldDesc::ScalarArray(ScalarType::String))], + }; + let wrap = |items: Vec| { + let mut root = PvStructure::new("epics:nt/NTScalarArray:1.0"); + root.fields + .push(("value".into(), PvField::ScalarArray(items))); + PvField::Structure(root) + }; + + let empty = wrap(vec![]); + let (db, ec) = pv_field_to_arch_db_type(&empty, &canonical).expect("classified"); + assert_eq!(db, ArchDbType::WaveformString); + assert_eq!(ec, 0); + assert_eq!( + pv_field_scalar_to_archiver(&empty, &canonical).expect("converted"), + ArchiverValue::VectorString(vec![]) + ); + + let one = wrap(vec![ScalarValue::String("a".into())]); + assert_eq!( + pv_field_scalar_to_archiver(&one, &canonical).expect("converted"), + ArchiverValue::VectorString(vec!["a".into()]) + ); + } + /// Build a synthetic NTTable with two double columns. Mirrors what /// a live IOC publishes for a typical waveform-table channel. fn make_nttable() -> PvField { @@ -4581,7 +4632,7 @@ mod pva_mapping_tests { #[test] fn nttable_classifies_as_v4_generic_bytes() { let pv = make_nttable(); - let (db, ec) = pv_field_to_arch_db_type(&pv).expect("classified"); + let (db, ec) = pv_field_to_arch_db_type(&pv, &pv.descriptor()).expect("classified"); assert_eq!(db, ArchDbType::V4GenericBytes); assert_eq!(ec, 1); } @@ -4642,7 +4693,7 @@ mod pva_mapping_tests { )); let pv = PvField::Structure(wrapper); - let (db, ec) = pv_field_to_arch_db_type(&pv).expect("classified"); + let (db, ec) = pv_field_to_arch_db_type(&pv, &pv.descriptor()).expect("classified"); assert_eq!(db, ArchDbType::WaveformDouble); assert_eq!(ec, 3); let av = pv_field_scalar_to_archiver(&pv, &pv.descriptor()).expect("converted"); @@ -4658,7 +4709,7 @@ mod pva_mapping_tests { #[test] fn nt_enum_classifies_as_scalar_enum() { let pv = make_nt_enum(2, &["Zero", "One", "Two"]); - let (db, ec) = pv_field_to_arch_db_type(&pv).expect("classified"); + let (db, ec) = pv_field_to_arch_db_type(&pv, &pv.descriptor()).expect("classified"); assert_eq!(db, ArchDbType::ScalarEnum); assert_eq!(ec, 1); } @@ -4907,7 +4958,7 @@ mod pva_mapping_tests { root.fields .push(("value".into(), PvField::Structure(inner))); let pv = PvField::Structure(root); - let (db, _) = pv_field_to_arch_db_type(&pv).expect("classified"); + let (db, _) = pv_field_to_arch_db_type(&pv, &pv.descriptor()).expect("classified"); assert_eq!(db, ArchDbType::V4GenericBytes); } @@ -4936,7 +4987,7 @@ mod pva_mapping_tests { .fields .push(("value".into(), PvField::Scalar(ScalarValue::Double(42.5)))); let pv = PvField::Structure(wrapper); - let (db, ec) = pv_field_to_arch_db_type(&pv).expect("classified"); + let (db, ec) = pv_field_to_arch_db_type(&pv, &pv.descriptor()).expect("classified"); assert_eq!(db, ArchDbType::ScalarDouble); assert_eq!(ec, 1); match pv_field_scalar_to_archiver(&pv, &pv.descriptor()).expect("converted") { From d426318fbb8d78cf0792c2c68b5e4fc2d97f8984 Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 11:34:17 +0900 Subject: [PATCH 36/55] engine: make the shutdown drain see the complete queue tail Two producers of uncounted shutdown loss. Shards drained on the same global flag as the dispatcher, so a shard could empty its channel and exit while the dispatcher was still moving the main queue's tail into it; the dispatcher now owns a per-shard drain signal (ShardSlot) that it flips only after that move. And nothing stopped the PV tasks, which kept sending into the pool during the drain and lost those samples when the receiver went away; ChannelManager::shutdown cancels every tracked producer, waits for them, and refuses later starts, and the binary runs it before flipping the pool's shutdown flag. --- crates/archiver-engine/src/channel_manager.rs | 178 ++++++++----- src/main.rs | 13 + tests/save_path_live_ca.rs | 247 ++++++++++++------ tests/write_loop_failure.rs | 66 +++++ 4 files changed, 355 insertions(+), 149 deletions(-) diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index 6fd2394..df165c0 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -14,6 +14,7 @@ use epics_rs::pva::pvdata::encode::{encode_pv_field, encode_type_desc}; use epics_rs::pva::pvdata::{FieldDesc, PvField, ScalarType, ScalarValue, TypedScalarArray}; use tokio::sync::mpsc; use tokio_util::sync::CancellationToken; +use tokio_util::task::TaskTracker; use tracing::{debug, error, info, warn}; use archiver_core::registry::{Protocol, PvRecord, PvRegistry, PvStatus, SampleMode}; @@ -406,6 +407,14 @@ pub struct ChannelManager { policy: Option, /// Per-site IOC drift bound (Java parity 6538631). server_ioc_drift_secs: u64, + /// Every sample producer (monitor/scan loops, PVA refresh and + /// watchdog tasks) is spawned through this tracker so + /// [`Self::shutdown`] can wait for all of them to exit. + tasks: TaskTracker, + /// `true` once [`Self::shutdown`] ran. A start holds the read side + /// while it spawns; shutdown takes the write side to flip it, so no + /// producer can be spawned after the sweep that cancels them. + lifecycle: tokio::sync::RwLock, } /// A sample ready to be written to storage. @@ -453,6 +462,8 @@ impl ChannelManager { sample_tx: tx, policy, server_ioc_drift_secs, + tasks: TaskTracker::new(), + lifecycle: tokio::sync::RwLock::new(false), }; Ok((mgr, rx)) @@ -472,6 +483,24 @@ impl ChannelManager { .clone() } + /// Stop every sample producer and wait for them to exit. The binary + /// calls this before the write pool is told to drain, so the queue + /// tail the pool moves to disk is the whole tail: a producer still + /// sending during the drain lost those samples, uncounted, once the + /// pool's receiver went away. Starts requested after this point are + /// refused; the PV handles (and their counters) stay visible. + pub async fn shutdown(&self) { + { + let mut closed = self.lifecycle.write().await; + *closed = true; + for entry in self.channels.iter() { + entry.value().cancel_token.cancel(); + } + } + self.tasks.close(); + self.tasks.wait().await; + } + /// Restore all active PVs from the registry (called on startup). /// /// `pvs_by_status(Active)` already filters out alias rows (they carry @@ -701,6 +730,15 @@ impl ChannelManager { /// `record.protocol`; the CA branch keeps the historical behaviour, /// the PVA branch goes through `start_archiving_internal_pva`. async fn start_archiving_internal(&self, record: &PvRecord) -> anyhow::Result<()> { + // Held until the producer is spawned: `shutdown` cannot sweep + // between the check and the spawn. + let lifecycle = self.lifecycle.read().await; + if *lifecycle { + anyhow::bail!( + "channel manager is shut down; refusing to start {}", + record.pv_name + ); + } if record.protocol == Protocol::Pva { return self.start_archiving_internal_pva(record).await; } @@ -763,7 +801,7 @@ impl ChannelManager { let drift = self.server_ioc_drift_secs; match &record.sample_mode { SampleMode::Monitor => { - tokio::spawn(async move { + self.tasks.spawn(async move { monitor_loop( pv_name, dbr_type, @@ -782,7 +820,7 @@ impl ChannelManager { } SampleMode::Scan { period_secs } => { let period = *period_secs; - tokio::spawn(async move { + self.tasks.spawn(async move { scan_loop( pv_name, dbr_type, @@ -850,7 +888,7 @@ impl ChannelManager { let pv_name_loop = pv_name.clone(); let archive_fields_loop = record.archive_fields.clone(); let extras_for_loop = extras.clone(); - tokio::spawn(async move { + self.tasks.spawn(async move { monitor_loop_pva( pv_name_loop, dbr_type, @@ -872,7 +910,7 @@ impl ChannelManager { let pv_name_loop = pv_name.clone(); let archive_fields_loop = record.archive_fields.clone(); let extras_for_loop = extras.clone(); - tokio::spawn(async move { + self.tasks.spawn(async move { scan_loop_pva( pv_name_loop, dbr_type, @@ -901,7 +939,7 @@ impl ChannelManager { let registry = self.registry.clone(); let cancel = cancel_token.clone(); let counters_for_refresh = counters.clone(); - tokio::spawn(async move { + self.tasks.spawn(async move { pva_metadata_refresh_loop( pv_name, pva_client, @@ -918,7 +956,7 @@ impl ChannelManager { let counters_for_watch = counters.clone(); let cancel = cancel_token.clone(); let sample_mode = record.sample_mode.clone(); - tokio::spawn(async move { + self.tasks.spawn(async move { pva_state_watchdog(pv_name, conn_info, counters_for_watch, cancel, sample_mode) .await; }); @@ -3385,10 +3423,10 @@ pub async fn run_sharded_write_pool( ShardSpawnCtx { storage: storage.clone(), pending: pending.clone(), - shutdown: shutdown.clone(), per_shard_buffer: cfg.per_shard_buffer, write_loop: cfg.write_loop.clone(), }, + shutdown.clone(), ) .await; } @@ -3405,26 +3443,39 @@ pub async fn run_sharded_write_pool( struct ShardSpawnCtx { storage: Arc, pending: Arc, - shutdown: tokio::sync::watch::Receiver, per_shard_buffer: usize, write_loop: WriteLoopConfig, } -/// Spawn one shard worker; return its input channel and join handle. -fn spawn_shard( - shard_idx: usize, - ctx: &ShardSpawnCtx, -) -> (mpsc::Sender, tokio::task::JoinHandle<()>) { +/// One shard worker as the dispatcher sees it: its input channel, its +/// task, and the drain signal the dispatcher flips once it has moved +/// the main queue's tail into `tx`. The shard drains on that signal, +/// not on the global shutdown flag: sharing the flag let a shard empty +/// its channel and exit while the dispatcher was still filling it, so +/// the tail landed in a closed (or dropped) channel. +struct ShardSlot { + tx: mpsc::Sender, + handle: tokio::task::JoinHandle<()>, + drain_tx: tokio::sync::watch::Sender, +} + +/// Spawn one shard worker; the dispatcher owns the returned slot. +fn spawn_shard(shard_idx: usize, ctx: &ShardSpawnCtx) -> ShardSlot { let (tx, rx) = mpsc::channel::(ctx.per_shard_buffer); + let (drain_tx, drain_rx) = tokio::sync::watch::channel(false); let handle = tokio::spawn(shard_append_loop( shard_idx, ctx.storage.clone(), rx, ctx.pending.clone(), - ctx.shutdown.clone(), + drain_rx, ctx.write_loop.clone(), )); - (tx, handle) + ShardSlot { + tx, + handle, + drain_tx, + } } /// Account a sample dropped because its shard channel was full (the @@ -3535,13 +3586,12 @@ impl RespawnGovernor { fn route_sample( sample: PvSample, - shard_txs: &mut [mpsc::Sender], - shard_handles: &mut [tokio::task::JoinHandle<()>], + shards: &mut [ShardSlot], governor: &mut RespawnGovernor, ctx: &ShardSpawnCtx, ) { - let idx = shard_for_pv(&sample.pv_name, shard_txs.len()); - match shard_txs[idx].try_send(sample) { + let idx = shard_for_pv(&sample.pv_name, shards.len()); + match shards[idx].tx.try_send(sample) { Ok(()) => {} Err(mpsc::error::TrySendError::Full(s)) => record_overflow_drop(idx, &s, "steady"), Err(mpsc::error::TrySendError::Closed(s)) => { @@ -3567,13 +3617,11 @@ fn route_sample( "shard" => idx.to_string(), ) .increment(1); - let (tx, handle) = spawn_shard(idx, ctx); // The old task is finished (its receiver was dropped, - // which is why we saw Closed); overwriting the handle - // detaches it. - shard_txs[idx] = tx; - shard_handles[idx] = handle; - match shard_txs[idx].try_send(s) { + // which is why we saw Closed); overwriting the slot + // detaches its handle. + shards[idx] = spawn_shard(idx, ctx); + match shards[idx].tx.try_send(s) { Ok(()) => {} Err(mpsc::error::TrySendError::Full(s2)) => { record_overflow_drop(idx, &s2, "respawn_retry") @@ -3598,18 +3646,16 @@ fn route_sample( /// `send().await` would block the dispatcher on the slow shard, /// back-pressuring the upstream main channel and starving every other /// shard's PVs — the exact failure mode the sharded layout prevents. -async fn dispatch_loop(mut rx: mpsc::Receiver, n: usize, ctx: ShardSpawnCtx) { - let mut shutdown = ctx.shutdown.clone(); +async fn dispatch_loop( + mut rx: mpsc::Receiver, + n: usize, + ctx: ShardSpawnCtx, + mut shutdown: tokio::sync::watch::Receiver, +) { // The dispatcher owns its shard workers so it can respawn a dead - // one. A shard task only exits normally on shutdown, so a closed - // channel before then is a panic — see [`route_sample`]. - let mut shard_txs: Vec> = Vec::with_capacity(n); - let mut shard_handles: Vec> = Vec::with_capacity(n); - for shard_idx in 0..n { - let (tx, handle) = spawn_shard(shard_idx, &ctx); - shard_txs.push(tx); - shard_handles.push(handle); - } + // one. A shard task only exits normally on its drain signal, so a + // closed channel before then is a panic — see [`route_sample`]. + let mut shards: Vec = (0..n).map(|idx| spawn_shard(idx, &ctx)).collect(); // Bounds respawns per shard so a deterministically-dying shard // can't storm (see [`RespawnGovernor`]). let mut governor = RespawnGovernor::new(n); @@ -3623,15 +3669,15 @@ async fn dispatch_loop(mut rx: mpsc::Receiver, n: usize, ctx: ShardSpa // even under a sustained sample storm. _ = shutdown.changed() => { if *shutdown.borrow() { - // Drain remaining samples (try_send for the same - // reason as the steady-state branch). No respawn - // during shutdown — a closed shard has already - // exited its loop on the same signal. Count BOTH - // Full and Closed drops so shutdown-time loss is - // visible to operators rather than silent. + // Move the main queue's tail into the shard + // channels (try_send for the same reason as the + // steady-state branch). No respawn during + // shutdown. Count BOTH Full and Closed drops so + // shutdown-time loss is visible to operators + // rather than silent. while let Ok(sample) = rx.try_recv() { let idx = shard_for_pv(&sample.pv_name, n); - match shard_txs[idx].try_send(sample) { + match shards[idx].tx.try_send(sample) { Ok(()) => {} Err(mpsc::error::TrySendError::Full(s)) => { record_overflow_drop(idx, &s, "shutdown_drain") @@ -3647,13 +3693,7 @@ async fn dispatch_loop(mut rx: mpsc::Receiver, n: usize, ctx: ShardSpa maybe = rx.recv() => { match maybe { Some(sample) => { - route_sample( - sample, - &mut shard_txs, - &mut shard_handles, - &mut governor, - &ctx, - ); + route_sample(sample, &mut shards, &mut governor, &ctx); } None => break, // Upstream closed. } @@ -3662,11 +3702,15 @@ async fn dispatch_loop(mut rx: mpsc::Receiver, n: usize, ctx: ShardSpa } info!("Sharded write dispatcher exiting"); - // Single owner joins the shard workers on the way out. The - // dispatcher exited → shard sample receivers see EOS or the - // shutdown signal directly via their watch::Receiver. - for handle in shard_handles { - let _ = handle.await; + // Both exits (shutdown flag, upstream closed) release the shards + // here, after the last try_send into their channels, so the drain + // a shard runs on its signal sees its complete input. Single owner + // joins the workers on the way out. + for slot in &shards { + let _ = slot.drain_tx.send(true); + } + for slot in shards { + let _ = slot.handle.await; } } @@ -5052,11 +5096,9 @@ mod shard_lifecycle_tests { /// PVs permanently after a single shard panic. #[tokio::test] async fn route_sample_respawns_dead_shard_and_delivers() { - let (sd_tx, sd_rx) = tokio::sync::watch::channel(false); let ctx = ShardSpawnCtx { storage: Arc::new(OkStorage), pending: Arc::new(PendingReports::new()), - shutdown: sd_rx, per_shard_buffer: 16, write_loop: WriteLoopConfig { append_timeout: Duration::from_secs(2), @@ -5068,10 +5110,14 @@ mod shard_lifecycle_tests { // receiver was dropped reports Closed on try_send. let (dead_tx, dead_rx) = mpsc::channel::(1); drop(dead_rx); - let mut shard_txs = vec![dead_tx]; - let mut shard_handles = vec![tokio::spawn(async {})]; + let (drain_tx, _drain_rx) = tokio::sync::watch::channel(false); + let mut shards = vec![ShardSlot { + tx: dead_tx, + handle: tokio::spawn(async {}), + drain_tx, + }]; assert!( - shard_txs[0].is_closed(), + shards[0].tx.is_closed(), "precondition: shard 0's worker is dead (channel closed)" ); @@ -5088,17 +5134,11 @@ mod shard_lifecycle_tests { }; let mut governor = RespawnGovernor::new(1); - route_sample( - sample, - &mut shard_txs, - &mut shard_handles, - &mut governor, - &ctx, - ); + route_sample(sample, &mut shards, &mut governor, &ctx); // Respawn must have replaced the dead sender with a live one. assert!( - !shard_txs[0].is_closed(), + !shards[0].tx.is_closed(), "shard 0 must be respawned with a live channel" ); @@ -5117,7 +5157,7 @@ mod shard_lifecycle_tests { ); // Tear the respawned shard down cleanly. - let _ = sd_tx.send(true); + let _ = shards[0].drain_tx.send(true); } #[test] diff --git a/src/main.rs b/src/main.rs index 40651ed..db79902 100644 --- a/src/main.rs +++ b/src/main.rs @@ -396,6 +396,8 @@ async fn main() -> anyhow::Result<()> { // Graceful-shutdown trigger for the HTTP server: the OS signal, or // the supervisor flipping the watch because a critical task died. let supervisor_shutdown = supervisor.shutdown_requested(); + let producers = channel_mgr.clone(); + let producer_stop_timeout = Duration::from_secs(10); let shutdown_signal = async move { tokio::select! { _ = tokio::signal::ctrl_c() => info!("Shutdown signal received"), @@ -403,6 +405,17 @@ async fn main() -> anyhow::Result<()> { tracing::warn!("Shutdown requested by the supervisor"); } } + // Stop the sample producers before the write pool is told to + // drain, so the queue tail it moves to disk is the whole tail. + if tokio::time::timeout(producer_stop_timeout, producers.shutdown()) + .await + .is_err() + { + tracing::warn!( + timeout = ?producer_stop_timeout, + "PV tasks did not stop in time; draining the write pool anyway" + ); + } let _ = shutdown_tx.send(true); }; diff --git a/tests/save_path_live_ca.rs b/tests/save_path_live_ca.rs index 7be9b7c..b8c5931 100644 --- a/tests/save_path_live_ca.rs +++ b/tests/save_path_live_ca.rs @@ -22,95 +22,139 @@ use archiver_engine::channel_manager::{ }; use epics_rs::base::server::records::waveform::WaveformRecord; use epics_rs::base::types::{DbFieldType, EpicsValue}; -use epics_rs::ca::client::CaClient; +use epics_rs::ca::client::{CaChannel, CaClient}; use epics_rs::ca::server::CaServer; const PV: &str = "SMOKE:CA:WF"; +const OTHER_PV: &str = "SMOKE:CA:OTHER"; const NELM: i32 = 3; -fn counters(mgr: &ChannelManager) -> PvCountersSnapshot { +fn counters(mgr: &ChannelManager, pv: &str) -> PvCountersSnapshot { mgr.all_pv_counters() .into_iter() - .find(|(n, _)| n == PV) + .find(|(n, _)| n == pv) .map(|(_, c)| c) .expect("counters for PV") } +/// In-process CA server, archiver manager, write pool, and a writer +/// client connected to `PV` (seeded once so the record carries a real +/// timestamp: a never-processed record reports TIME = 0, which the +/// drift filter rejects). +struct Stack { + _dir: tempfile::TempDir, + storage: Arc, + registry: Arc, + mgr: Arc, + pool: tokio::task::JoinHandle<()>, + pool_shutdown: tokio::sync::watch::Sender, + server_task: tokio::task::JoinHandle<()>, + _writer: CaClient, + ch: CaChannel, +} + +impl Stack { + async fn start() -> Self { + // Quiet unless RUST_LOG is set (e.g. archiver_engine=debug). + let _ = tracing_subscriber::fmt() + .with_env_filter(tracing_subscriber::EnvFilter::from_default_env()) + .with_test_writer() + .try_init(); + let dir = tempfile::tempdir().unwrap(); + let storage: Arc = Arc::new(PlainPbStoragePlugin::new( + "sts", + dir.path().join("sts"), + PartitionGranularity::Year, + )); + let registry = Arc::new(PvRegistry::open(&dir.path().join("registry.db")).unwrap()); + + let server = CaServer::builder() + .port(0) + .record(PV, WaveformRecord::new(NELM, DbFieldType::Double)) + .record(OTHER_PV, WaveformRecord::new(1, DbFieldType::Double)) + .build() + .await + .expect("in-process CA server"); + let port = server.udp_port(); + let server_task = tokio::spawn(async move { + let _ = server.run().await; + }); + + // SAFETY: nextest runs each test in its own process; nothing + // else reads these variables concurrently, and they are set + // before any CaClient snapshots its resolver configuration. + unsafe { + std::env::set_var("EPICS_CA_ADDR_LIST", format!("127.0.0.1:{port}")); + std::env::set_var("EPICS_CA_AUTO_ADDR_LIST", "NO"); + std::env::set_var("EPICS_CA_SERVER_PORT", port.to_string()); + } + + let (mgr, rx) = ChannelManager::new(storage.clone(), registry.clone(), None) + .await + .unwrap(); + let mgr = Arc::new(mgr); + let (pool_shutdown, shutdown_rx) = tokio::sync::watch::channel(false); + let pool = tokio::spawn(run_sharded_write_pool( + storage.clone(), + registry.clone(), + rx, + shutdown_rx, + ShardedWritePoolConfig { + shards: 1, + per_shard_buffer: 1024, + write_loop: WriteLoopConfig { + flush_period: Duration::from_millis(300), + ..Default::default() + }, + }, + )); + + let writer = CaClient::new().await.expect("writer client"); + let ch = writer.create_channel(PV); + ch.wait_connected(Duration::from_secs(10)) + .await + .expect("writer connects"); + ch.put(&EpicsValue::DoubleArray(vec![0.0; NELM as usize])) + .await + .expect("seed caput"); + + Self { + _dir: dir, + storage, + registry, + mgr, + pool, + pool_shutdown, + server_task, + _writer: writer, + ch, + } + } + + async fn finish(self) { + self.pool_shutdown.send(true).unwrap(); + tokio::time::timeout(Duration::from_secs(10), self.pool) + .await + .expect("write pool exits on shutdown") + .unwrap(); + self.server_task.abort(); + } +} + /// A CA array channel registers as the waveform type and its array /// samples reach the disk as `VectorDouble`, with no type-change drops. /// The one-element put sets NORD to 1, which CA delivers as a scalar /// event; it must still be stored as a one-element waveform. #[tokio::test(flavor = "multi_thread", worker_threads = 4)] async fn ca_waveform_samples_land_as_waveform_double() { - // Quiet unless RUST_LOG is set (e.g. archiver_engine=debug). - let _ = tracing_subscriber::fmt() - .with_env_filter(tracing_subscriber::EnvFilter::from_default_env()) - .with_test_writer() - .try_init(); - let dir = tempfile::tempdir().unwrap(); - let storage: Arc = Arc::new(PlainPbStoragePlugin::new( - "sts", - dir.path().join("sts"), - PartitionGranularity::Year, - )); - let registry = Arc::new(PvRegistry::open(&dir.path().join("registry.db")).unwrap()); - - let server = CaServer::builder() - .port(0) - .record(PV, WaveformRecord::new(NELM, DbFieldType::Double)) - .build() - .await - .expect("in-process CA server"); - let port = server.udp_port(); - let server_task = tokio::spawn(async move { server.run().await }); - - // SAFETY: nextest runs each test in its own process; nothing else - // reads these variables concurrently, and they are set before any - // CaClient snapshots its resolver configuration. - unsafe { - std::env::set_var("EPICS_CA_ADDR_LIST", format!("127.0.0.1:{port}")); - std::env::set_var("EPICS_CA_AUTO_ADDR_LIST", "NO"); - std::env::set_var("EPICS_CA_SERVER_PORT", port.to_string()); - } - - let (mgr, rx) = ChannelManager::new(storage.clone(), registry.clone(), None) - .await - .unwrap(); - let mgr = Arc::new(mgr); - let (shutdown_tx, shutdown_rx) = tokio::sync::watch::channel(false); - let pool = tokio::spawn(run_sharded_write_pool( - storage.clone(), - registry.clone(), - rx, - shutdown_rx, - ShardedWritePoolConfig { - shards: 1, - per_shard_buffer: 1024, - write_loop: WriteLoopConfig { - flush_period: Duration::from_millis(300), - ..Default::default() - }, - }, - )); - - // Process the record once before archiving: a never-processed - // record reports TIME = 0, which the drift filter rejects, so the - // connect-time event would otherwise be a timestamp drop rather - // than the first stored sample. - let writer = CaClient::new().await.expect("writer client"); - let ch = writer.create_channel(PV); - ch.wait_connected(Duration::from_secs(10)) - .await - .expect("writer connects"); - ch.put(&EpicsValue::DoubleArray(vec![0.0; NELM as usize])) - .await - .expect("seed caput"); + let s = Stack::start().await; let t0 = SystemTime::now(); - mgr.archive_pv(PV, &SampleMode::Monitor, Protocol::Ca) + s.mgr + .archive_pv(PV, &SampleMode::Monitor, Protocol::Ca) .await .expect("archive_pv over CA"); - let rec = registry.get_pv(PV).unwrap().expect("registry row"); + let rec = s.registry.get_pv(PV).unwrap().expect("registry row"); assert_eq!(rec.dbr_type, ArchDbType::WaveformDouble, "{rec:?}"); assert_eq!(rec.element_count, NELM, "{rec:?}"); assert_eq!(rec.protocol, Protocol::Ca); @@ -119,7 +163,7 @@ async fn ca_waveform_samples_land_as_waveform_double() { // archiver's monitor sees it as a DoubleArray event. let arrays: Vec> = vec![vec![1.0, 2.0, 3.0], vec![5.0], vec![7.0, 8.0, 9.0]]; for a in &arrays { - ch.put(&EpicsValue::DoubleArray(a.clone())) + s.ch.put(&EpicsValue::DoubleArray(a.clone())) .await .expect("caput"); tokio::time::sleep(Duration::from_millis(100)).await; @@ -129,12 +173,12 @@ async fn ca_waveform_samples_land_as_waveform_double() { let to = t0 + Duration::from_secs(3600); let deadline = tokio::time::Instant::now() + Duration::from_secs(20); let stored = loop { - let mut stream = query_data(&*storage, PV, from, to, None) + let mut stream = query_data(&*s.storage, PV, from, to, None) .await .expect("query_data"); let mut got: Vec> = Vec::new(); - while let Some(s) = stream.next_event().expect("next_event") { - match s.value { + while let Some(e) = stream.next_event().expect("next_event") { + match e.value { ArchiverValue::VectorDouble(v) => got.push(v), other => panic!("stored value is not a waveform: {other:?}"), } @@ -145,7 +189,7 @@ async fn ca_waveform_samples_land_as_waveform_double() { assert!( tokio::time::Instant::now() < deadline, "timed out waiting for {arrays:?} on disk; have {got:?}; counters {:?}", - counters(&mgr) + counters(&s.mgr, PV) ); tokio::time::sleep(Duration::from_millis(200)).await; }; @@ -154,17 +198,60 @@ async fn ca_waveform_samples_land_as_waveform_double() { let posted: Vec> = stored.into_iter().filter(|g| arrays.contains(g)).collect(); assert_eq!(posted, arrays, "on-disk sample order/content"); - let c = counters(&mgr); + let c = counters(&s.mgr, PV); assert_eq!(c.type_change_drops, 0, "{c:?}"); assert_eq!(c.storage_write_errors, 0, "{c:?}"); assert_eq!(c.timestamp_drops, 0, "{c:?}"); assert!(c.events_stored >= arrays.len() as u64, "{c:?}"); - mgr.stop_pv(PV).await.unwrap(); - shutdown_tx.send(true).unwrap(); - tokio::time::timeout(Duration::from_secs(10), pool) + s.mgr.stop_pv(PV).await.unwrap(); + s.finish().await; +} + +/// `ChannelManager::shutdown` stops every producer and refuses new +/// starts, so the write pool's drain that follows sees a fixed queue +/// tail. The channel keeps posting; nothing may be received after +/// shutdown returned. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn shutdown_stops_producers_and_refuses_new_starts() { + let s = Stack::start().await; + s.mgr + .archive_pv(PV, &SampleMode::Monitor, Protocol::Ca) .await - .expect("write pool exits on shutdown") - .unwrap(); - server_task.abort(); + .expect("archive_pv over CA"); + s.ch.put(&EpicsValue::DoubleArray(vec![1.0, 2.0, 3.0])) + .await + .expect("caput"); + let deadline = tokio::time::Instant::now() + Duration::from_secs(10); + while counters(&s.mgr, PV).events_received == 0 { + assert!( + tokio::time::Instant::now() < deadline, + "monitor never delivered" + ); + tokio::time::sleep(Duration::from_millis(50)).await; + } + + tokio::time::timeout(Duration::from_secs(10), s.mgr.shutdown()) + .await + .expect("producers stop"); + + let before = counters(&s.mgr, PV).events_received; + s.ch.put(&EpicsValue::DoubleArray(vec![4.0, 5.0, 6.0])) + .await + .expect("caput after shutdown"); + tokio::time::sleep(Duration::from_millis(500)).await; + assert_eq!( + counters(&s.mgr, PV).events_received, + before, + "a producer received a sample after shutdown" + ); + + let err = s + .mgr + .archive_pv(OTHER_PV, &SampleMode::Monitor, Protocol::Ca) + .await + .expect_err("start after shutdown must be refused"); + assert!(err.to_string().contains("shut down"), "{err}"); + + s.finish().await; } diff --git a/tests/write_loop_failure.rs b/tests/write_loop_failure.rs index 799d0d6..ce84fda 100644 --- a/tests/write_loop_failure.rs +++ b/tests/write_loop_failure.rs @@ -1951,3 +1951,69 @@ async fn dirty_eviction_loss_marker_charges_flush_losses() { ); shutdown(sd_tx, join).await; } + +/// Samples still in the main queue when shutdown flips are moved into +/// the shard channels by the dispatcher, and every shard must drain +/// them. Shards used to drain on the global shutdown flag, racing that +/// move: a shard that emptied its channel first exited, and the rest +/// of the tail landed in a closed (or dropped) channel. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn shutdown_drain_delivers_the_main_queue_tail_to_every_shard() { + let storage = InjectingStorage::new(); + let registry = Arc::new(PvRegistry::in_memory().unwrap()); + + const N_PVS: usize = 64; + const SAMPLES_PER_PV: usize = 50; + const TOTAL: usize = N_PVS * SAMPLES_PER_PV; + let pv_names: Vec = (0..N_PVS).map(|i| format!("pv{i}")).collect(); + for name in &pv_names { + registry + .register_pv(name, ArchDbType::ScalarDouble, &SampleMode::Monitor, 1) + .unwrap(); + } + let counters: Vec> = (0..N_PVS) + .map(|_| Arc::new(PvCounters::default())) + .collect(); + + // The whole load sits in the main queue, and shutdown is already + // requested, before the pool starts: its first poll is the drain. + let (tx, rx) = mpsc::channel::(TOTAL); + for s in 0..SAMPLES_PER_PV { + for (i, name) in pv_names.iter().enumerate() { + let t = ts(10_000 + (i as u64) * 100 + s as u64); + tx.try_send(pv_sample(name, t, s as f64, &counters[i])) + .unwrap(); + } + } + let (sd_tx, sd_rx) = watch::channel(false); + sd_tx.send(true).unwrap(); + + let cfg = ShardedWritePoolConfig { + shards: 4, + per_shard_buffer: TOTAL, + write_loop: fast_cfg(), + }; + let pool = tokio::spawn(run_sharded_write_pool( + storage.clone(), + registry.clone(), + rx, + sd_rx, + cfg, + )); + tokio::time::timeout(Duration::from_secs(20), pool) + .await + .expect("pool exits on shutdown") + .unwrap(); + + let sum = |f: &dyn Fn(&PvCounters) -> u64| counters.iter().map(|c| f(c)).sum::(); + let stored = sum(&|c| c.events_stored.load(Ordering::Relaxed)); + let closed = sum(&|c| c.shard_closed_drops.load(Ordering::Relaxed)); + let overflow = sum(&|c| c.buffer_overflow_drops.load(Ordering::Relaxed)); + let abandoned = sum(&|c| c.shutdown_abandoned_drops.load(Ordering::Relaxed)); + assert_eq!( + (stored, closed, overflow, abandoned), + (TOTAL as u64, 0, 0, 0), + "every queued sample must reach a shard drain" + ); + assert_eq!(storage.appends_snapshot().len(), TOTAL); +} From 304408e2e5d949bfc1d05556fb2d11d3831d6ddc Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 11:46:05 +0900 Subject: [PATCH 37/55] engine: seed the shard_handle_sample ordering gate from the store A resumed or restarted task is handed the current value again (CA initial monitor event, PVA initial get) with the timestamp of the sample already on disk; the gate started at zero and only rejected older timestamps, so that value was stored twice. task_counters seeds it from the registry and the last stored event, and an equal timestamp is now dropped as the same event redelivered. The CA tests wait for the connect-time event before putting: the in-process server can report a put that races the subscribe as the new value under the previous timestamp. --- crates/archiver-engine/src/channel_manager.rs | 64 +++++++--- tests/save_path_live_ca.rs | 114 ++++++++++++++++-- tests/write_loop_failure.rs | 40 ++++++ 3 files changed, 192 insertions(+), 26 deletions(-) diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index df165c0..7752b06 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -237,13 +237,15 @@ pub struct PvCounters { pub flush_losses: AtomicU64, /// Ordering high-water for the shard's out-of-order drop: the /// newest timestamp the write pool has handed to storage for this - /// archiving task, as unix nanoseconds (0 = none yet). It lives - /// with the task rather than in a shard-lifetime map so it dies - /// with the task: a deletePV + re-archive, or a resume, starts - /// from a clean slate instead of inheriting a dead task's last - /// timestamp and dropping its first samples. Written only by the - /// shard that owns the PV (the dispatcher's consistent hash pins - /// each PV to one shard); not part of the reported snapshot. + /// archiving task, as unix nanoseconds (0 = none). A sample must be + /// strictly newer to be stored. It lives with the task rather than + /// in a shard-lifetime map so a dead task's state cannot leak into + /// the next one; a new task is seeded from what is already archived + /// (`ChannelManager::task_counters`), so a deletePV + re-archive + /// starts at zero while a resume or restart drops the current value + /// the connect redelivers. Written only by the shard that owns the + /// PV (the dispatcher's consistent hash pins each PV to one shard); + /// not part of the reported snapshot. pub ordering_last_ts_nanos: AtomicU64, } @@ -396,8 +398,8 @@ pub struct ChannelManager { /// PV. Without this, e.g. `pause_pv` racing with `resume_pv` can leave /// the registry status and the channel map disagreeing. op_locks: DashMap>>, - /// Storage backend. - #[allow(dead_code)] + /// Storage backend; read for the last stored sample when a task + /// starts (see [`Self::task_counters`]). storage: Arc, /// PV metadata registry. registry: Arc, @@ -501,6 +503,34 @@ impl ChannelManager { self.tasks.wait().await; } + /// Counters for a new archiving task, with the ordering high-water + /// seeded from what is already archived: the newer of the + /// registry's committed `last_timestamp` and the last sample the + /// store holds. A task starting at zero would store the current + /// value CA and PVA redeliver on connect (its timestamp is the one + /// already on disk) a second time, and would accept the older + /// timestamps a reconnect backfill carries, which the drift filter + /// waives, behind newer samples in the same partition. + async fn task_counters(&self, record: &PvRecord) -> Arc { + let stored = match self.storage.get_last_known_event(&record.pv_name).await { + Ok(sample) => sample.map(|s| s.timestamp), + Err(e) => { + warn!( + pv = record.pv_name, + "Could not read the last stored sample; ordering seeded from the registry only: {e}" + ); + None + } + }; + let counters = Arc::new(PvCounters::default()); + if let Some(ts) = record.last_timestamp.into_iter().chain(stored).max() { + counters + .ordering_last_ts_nanos + .store(unix_nanos(ts), Ordering::Relaxed); + } + counters + } + /// Restore all active PVs from the registry (called on startup). /// /// `pvs_by_status(Active)` already filters out alias rows (they carry @@ -751,7 +781,7 @@ impl ChannelManager { let extras: Arc = Arc::new(DashMap::new()); let field_tokens: Arc> = Arc::new(DashMap::new()); let update_lock = Arc::new(tokio::sync::Mutex::new(())); - let counters = Arc::new(PvCounters::default()); + let counters = self.task_counters(record).await; // Hold update_lock around the whole insert+spawn block so a // concurrent update_archive_fields can't observe the empty @@ -860,7 +890,7 @@ impl ChannelManager { let extras: Arc = Arc::new(DashMap::new()); let field_tokens: Arc> = Arc::new(DashMap::new()); let update_lock = Arc::new(tokio::sync::Mutex::new(())); - let counters = Arc::new(PvCounters::default()); + let counters = self.task_counters(record).await; self.channels.insert( pv_name.clone(), @@ -4171,13 +4201,15 @@ async fn shard_handle_sample( let ts = pv_sample.sample.timestamp; let ts_nanos = unix_nanos(ts); - // Out-of-order timestamp drop. The high-water lives in the - // archiving task's `PvCounters` (see `ordering_last_ts_nanos`), - // so it survives flush cycles but not the task: a sample without - // counters has no task and no ordering state. + // Out-of-order timestamp drop; an equal timestamp is the same + // event redelivered (CA and PVA resend the current value on every + // connect). The high-water lives in the archiving task's + // `PvCounters` (see `ordering_last_ts_nanos`), so it survives + // flush cycles but not the task: a sample without counters has no + // task and no ordering state. if let Some(ref c) = pv_sample.counters { let prev = c.ordering_last_ts_nanos.load(Ordering::Relaxed); - if prev != 0 && ts_nanos < prev { + if prev != 0 && ts_nanos <= prev { c.timestamp_drops.fetch_add(1, Ordering::Relaxed); debug!( shard = shard_idx, diff --git a/tests/save_path_live_ca.rs b/tests/save_path_live_ca.rs index b8c5931..a58ded0 100644 --- a/tests/save_path_live_ca.rs +++ b/tests/save_path_live_ca.rs @@ -131,6 +131,26 @@ impl Stack { } } + /// Start archiving `PV` and wait for the connect-time event, which + /// proves the monitor subscription is active. A put that lands + /// while the subscription is still being set up can be reported by + /// the in-process server as the new value under the previous + /// timestamp, followed by the properly stamped event. + async fn archive(&self) { + self.mgr + .archive_pv(PV, &SampleMode::Monitor, Protocol::Ca) + .await + .expect("archive_pv over CA"); + let deadline = tokio::time::Instant::now() + Duration::from_secs(10); + while counters(&self.mgr, PV).events_received == 0 { + assert!( + tokio::time::Instant::now() < deadline, + "connect-time event never delivered" + ); + tokio::time::sleep(Duration::from_millis(20)).await; + } + } + async fn finish(self) { self.pool_shutdown.send(true).unwrap(); tokio::time::timeout(Duration::from_secs(10), self.pool) @@ -150,10 +170,7 @@ async fn ca_waveform_samples_land_as_waveform_double() { let s = Stack::start().await; let t0 = SystemTime::now(); - s.mgr - .archive_pv(PV, &SampleMode::Monitor, Protocol::Ca) - .await - .expect("archive_pv over CA"); + s.archive().await; let rec = s.registry.get_pv(PV).unwrap().expect("registry row"); assert_eq!(rec.dbr_type, ArchDbType::WaveformDouble, "{rec:?}"); assert_eq!(rec.element_count, NELM, "{rec:?}"); @@ -208,6 +225,86 @@ async fn ca_waveform_samples_land_as_waveform_double() { s.finish().await; } +/// Every `VectorDouble` sample on disk for `PV`, in stream order. +async fn stored_arrays(s: &Stack, t0: SystemTime) -> Vec> { + let from = t0 - Duration::from_secs(3600); + let to = t0 + Duration::from_secs(3600); + let mut stream = query_data(&*s.storage, PV, from, to, None) + .await + .expect("query_data"); + let mut got = Vec::new(); + while let Some(e) = stream.next_event().expect("next_event") { + match e.value { + ArchiverValue::VectorDouble(v) => got.push(v), + other => panic!("stored value is not a waveform: {other:?}"), + } + } + got +} + +/// Poll the disk until `want` is present; panic with the counters +/// after the deadline. +async fn wait_on_disk(s: &Stack, t0: SystemTime, want: &[f64]) -> Vec> { + let deadline = tokio::time::Instant::now() + Duration::from_secs(20); + loop { + let got = stored_arrays(s, t0).await; + if got.iter().any(|g| g == want) { + return got; + } + assert!( + tokio::time::Instant::now() < deadline, + "timed out waiting for {want:?} on disk; have {got:?}; counters {:?}", + counters(&s.mgr, PV) + ); + tokio::time::sleep(Duration::from_millis(200)).await; + } +} + +/// Pause and resume start a new archiving task, and the connect-time +/// event redelivers the current value with the timestamp already on +/// disk. The new task's ordering gate is seeded from the store, so the +/// redelivery is dropped instead of stored a second time. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn resume_does_not_re_store_the_current_value() { + let s = Stack::start().await; + let t0 = SystemTime::now(); + s.archive().await; + let first = vec![1.0, 2.0, 3.0]; + s.ch.put(&EpicsValue::DoubleArray(first.clone())) + .await + .expect("caput"); + wait_on_disk(&s, t0, &first).await; + + s.mgr.pause_pv(PV).await.unwrap(); + s.mgr.resume_pv(PV).await.unwrap(); + // The resumed task receives the connect-time redelivery of `first`. + let deadline = tokio::time::Instant::now() + Duration::from_secs(10); + while counters(&s.mgr, PV).events_received == 0 { + assert!( + tokio::time::Instant::now() < deadline, + "resumed monitor never delivered" + ); + tokio::time::sleep(Duration::from_millis(50)).await; + } + + let second = vec![4.0, 5.0, 6.0]; + s.ch.put(&EpicsValue::DoubleArray(second.clone())) + .await + .expect("caput"); + let got = wait_on_disk(&s, t0, &second).await; + let copies = got.iter().filter(|g| **g == first).count(); + assert_eq!( + copies, 1, + "the redelivered current value was stored again: {got:?}" + ); + let c = counters(&s.mgr, PV); + assert_eq!(c.timestamp_drops, 1, "{c:?}"); + assert_eq!(c.storage_write_errors, 0, "{c:?}"); + + s.mgr.stop_pv(PV).await.unwrap(); + s.finish().await; +} + /// `ChannelManager::shutdown` stops every producer and refuses new /// starts, so the write pool's drain that follows sees a fixed queue /// tail. The channel keeps posting; nothing may be received after @@ -215,18 +312,15 @@ async fn ca_waveform_samples_land_as_waveform_double() { #[tokio::test(flavor = "multi_thread", worker_threads = 4)] async fn shutdown_stops_producers_and_refuses_new_starts() { let s = Stack::start().await; - s.mgr - .archive_pv(PV, &SampleMode::Monitor, Protocol::Ca) - .await - .expect("archive_pv over CA"); + s.archive().await; s.ch.put(&EpicsValue::DoubleArray(vec![1.0, 2.0, 3.0])) .await .expect("caput"); let deadline = tokio::time::Instant::now() + Duration::from_secs(10); - while counters(&s.mgr, PV).events_received == 0 { + while counters(&s.mgr, PV).events_received < 2 { assert!( tokio::time::Instant::now() < deadline, - "monitor never delivered" + "monitor never delivered the put" ); tokio::time::sleep(Duration::from_millis(50)).await; } diff --git a/tests/write_loop_failure.rs b/tests/write_loop_failure.rs index ce84fda..ab7e249 100644 --- a/tests/write_loop_failure.rs +++ b/tests/write_loop_failure.rs @@ -2017,3 +2017,43 @@ async fn shutdown_drain_delivers_the_main_queue_tail_to_every_shard() { ); assert_eq!(storage.appends_snapshot().len(), TOTAL); } + +/// The ordering gate requires a strictly newer timestamp: a sample +/// carrying the last accepted timestamp is the same event redelivered +/// (CA and PVA resend the current value on every connect). +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn equal_timestamp_sample_is_dropped_as_a_redelivery() { + let cfg = fast_cfg(); + let storage = InjectingStorage::new(); + let registry = Arc::new(PvRegistry::in_memory().unwrap()); + registry + .register_pv("A", ArchDbType::ScalarDouble, &SampleMode::Monitor, 1) + .unwrap(); + let counters = Arc::new(PvCounters::default()); + let (tx, sd_tx, join) = spawn_loop(storage.clone(), registry.clone(), cfg); + + let t1 = ts(1000); + let t2 = ts(1001); + for (t, v) in [(t1, 1.0), (t1, 2.0), (t2, 3.0)] { + tx.send(pv_sample("A", t, v, &counters)).await.unwrap(); + } + let deadline = tokio::time::Instant::now() + Duration::from_secs(5); + while counters.events_stored.load(Ordering::Relaxed) < 2 { + assert!( + tokio::time::Instant::now() < deadline, + "samples never stored" + ); + tokio::time::sleep(Duration::from_millis(20)).await; + } + tokio::time::sleep(Duration::from_millis(100)).await; + + assert_eq!(counters.timestamp_drops.load(Ordering::Relaxed), 1); + let stored: Vec = storage + .appends_snapshot() + .into_iter() + .map(|r| r.timestamp) + .collect(); + assert_eq!(stored, vec![t1, t2]); + + shutdown(sd_tx, join).await; +} From 1360bee37ba926daf2b9c6575d0d0c2df55868f2 Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 11:59:26 +0900 Subject: [PATCH 38/55] engine: keep PvCounters for the process lifetime in task_counters Seeding the ordering gate from the store cost a partition tail read per PV and per tier on the restore path, and a full-cache flush per call. A PV's counters now outlive its task, so a resume continues the high-water exactly; the first task in a process is seeded from the registry's committed last_timestamp alone, which the flush owner advances only for bytes on disk. destroy_pv discards the entry. --- crates/archiver-engine/src/channel_manager.rs | 82 ++++++++-------- tests/save_path_live_ca.rs | 94 ++++++++++++++++++- 2 files changed, 134 insertions(+), 42 deletions(-) diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index 7752b06..d17dd12 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -237,15 +237,15 @@ pub struct PvCounters { pub flush_losses: AtomicU64, /// Ordering high-water for the shard's out-of-order drop: the /// newest timestamp the write pool has handed to storage for this - /// archiving task, as unix nanoseconds (0 = none). A sample must be - /// strictly newer to be stored. It lives with the task rather than - /// in a shard-lifetime map so a dead task's state cannot leak into - /// the next one; a new task is seeded from what is already archived - /// (`ChannelManager::task_counters`), so a deletePV + re-archive - /// starts at zero while a resume or restart drops the current value - /// the connect redelivers. Written only by the shard that owns the - /// PV (the dispatcher's consistent hash pins each PV to one shard); - /// not part of the reported snapshot. + /// PV, as unix nanoseconds (0 = none). A sample must be strictly + /// newer to be stored. The counters outlive the archiving task + /// (`ChannelManager::task_counters`), so a resume continues the + /// high-water and drops the current value the connect redelivers; + /// the first task in a process is seeded from the registry's + /// committed `last_timestamp`, and a deletePV + re-archive starts + /// at zero. Written only by the shard that owns the PV (the + /// dispatcher's consistent hash pins each PV to one shard); not + /// part of the reported snapshot. pub ordering_last_ts_nanos: AtomicU64, } @@ -398,8 +398,8 @@ pub struct ChannelManager { /// PV. Without this, e.g. `pause_pv` racing with `resume_pv` can leave /// the registry status and the channel map disagreeing. op_locks: DashMap>>, - /// Storage backend; read for the last stored sample when a task - /// starts (see [`Self::task_counters`]). + /// Storage backend. + #[allow(dead_code)] storage: Arc, /// PV metadata registry. registry: Arc, @@ -417,6 +417,10 @@ pub struct ChannelManager { /// while it spawns; shutdown takes the write side to flip it, so no /// producer can be spawned after the sweep that cancels them. lifecycle: tokio::sync::RwLock, + /// Per-PV counters for the life of the process, so a pause and + /// resume continue the PV's statistics and ordering high-water + /// (see [`Self::task_counters`]). `destroy_pv` removes the entry. + counters: DashMap>, } /// A sample ready to be written to storage. @@ -466,6 +470,7 @@ impl ChannelManager { server_ioc_drift_secs, tasks: TaskTracker::new(), lifecycle: tokio::sync::RwLock::new(false), + counters: DashMap::new(), }; Ok((mgr, rx)) @@ -503,32 +508,28 @@ impl ChannelManager { self.tasks.wait().await; } - /// Counters for a new archiving task, with the ordering high-water - /// seeded from what is already archived: the newer of the - /// registry's committed `last_timestamp` and the last sample the - /// store holds. A task starting at zero would store the current - /// value CA and PVA redeliver on connect (its timestamp is the one - /// already on disk) a second time, and would accept the older - /// timestamps a reconnect backfill carries, which the drift filter - /// waives, behind newer samples in the same partition. - async fn task_counters(&self, record: &PvRecord) -> Arc { - let stored = match self.storage.get_last_known_event(&record.pv_name).await { - Ok(sample) => sample.map(|s| s.timestamp), - Err(e) => { - warn!( - pv = record.pv_name, - "Could not read the last stored sample; ordering seeded from the registry only: {e}" - ); - None - } - }; - let counters = Arc::new(PvCounters::default()); - if let Some(ts) = record.last_timestamp.into_iter().chain(stored).max() { - counters - .ordering_last_ts_nanos - .store(unix_nanos(ts), Ordering::Relaxed); - } - counters + /// Counters for an archiving task. A PV's counters are kept for + /// the life of the process: a resumed task continues the ordering + /// high-water, so the current value CA and PVA redeliver on + /// connect, whose timestamp the previous task already stored, is + /// dropped rather than stored a second time. The first task for a + /// PV in this process is seeded from the registry's committed + /// `last_timestamp`, which the flush owner advances only for bytes + /// on disk. The store is not consulted: that would be a partition + /// tail read per PV and per tier on the restore path. + fn task_counters(&self, record: &PvRecord) -> Arc { + self.counters + .entry(record.pv_name.clone()) + .or_insert_with(|| { + let counters = PvCounters::default(); + if let Some(ts) = record.last_timestamp { + counters + .ordering_last_ts_nanos + .store(unix_nanos(ts), Ordering::Relaxed); + } + Arc::new(counters) + }) + .clone() } /// Restore all active PVs from the registry (called on startup). @@ -781,7 +782,7 @@ impl ChannelManager { let extras: Arc = Arc::new(DashMap::new()); let field_tokens: Arc> = Arc::new(DashMap::new()); let update_lock = Arc::new(tokio::sync::Mutex::new(())); - let counters = self.task_counters(record).await; + let counters = self.task_counters(record); // Hold update_lock around the whole insert+spawn block so a // concurrent update_archive_fields can't observe the empty @@ -890,7 +891,7 @@ impl ChannelManager { let extras: Arc = Arc::new(DashMap::new()); let field_tokens: Arc> = Arc::new(DashMap::new()); let update_lock = Arc::new(tokio::sync::Mutex::new(())); - let counters = self.task_counters(record).await; + let counters = self.task_counters(record); self.channels.insert( pv_name.clone(), @@ -1155,6 +1156,9 @@ impl ChannelManager { metrics::gauge!("archiver_extra_field_tasks").decrement(extra_count); } } + // The next task under this name is a new PV: its high-water + // and statistics must not carry over. + self.counters.remove(pv_name); self.registry.remove_pv(pv_name)?; // Don't remove the op_locks entry here: a concurrent caller may have // already taken a clone of this Arc and be queued on it; removing diff --git a/tests/save_path_live_ca.rs b/tests/save_path_live_ca.rs index a58ded0..dc92fda 100644 --- a/tests/save_path_live_ca.rs +++ b/tests/save_path_live_ca.rs @@ -262,8 +262,9 @@ async fn wait_on_disk(s: &Stack, t0: SystemTime, want: &[f64]) -> Vec> /// Pause and resume start a new archiving task, and the connect-time /// event redelivers the current value with the timestamp already on -/// disk. The new task's ordering gate is seeded from the store, so the -/// redelivery is dropped instead of stored a second time. +/// disk. The PV's counters, and with them the ordering gate, outlive +/// the task, so the redelivery is dropped instead of stored a second +/// time. #[tokio::test(flavor = "multi_thread", worker_threads = 4)] async fn resume_does_not_re_store_the_current_value() { let s = Stack::start().await; @@ -275,11 +276,12 @@ async fn resume_does_not_re_store_the_current_value() { .expect("caput"); wait_on_disk(&s, t0, &first).await; + let before = counters(&s.mgr, PV).events_received; s.mgr.pause_pv(PV).await.unwrap(); s.mgr.resume_pv(PV).await.unwrap(); // The resumed task receives the connect-time redelivery of `first`. let deadline = tokio::time::Instant::now() + Duration::from_secs(10); - while counters(&s.mgr, PV).events_received == 0 { + while counters(&s.mgr, PV).events_received == before { assert!( tokio::time::Instant::now() < deadline, "resumed monitor never delivered" @@ -305,6 +307,92 @@ async fn resume_does_not_re_store_the_current_value() { s.finish().await; } +/// A restart runs the PV's first task of a new process, whose ordering +/// gate is seeded from the registry's committed `last_timestamp`. The +/// connect-time redelivery of the value already on disk is dropped, +/// not stored again. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn restart_does_not_re_store_the_committed_value() { + let s = Stack::start().await; + let t0 = SystemTime::now(); + s.archive().await; + let first = vec![1.0, 2.0, 3.0]; + s.ch.put(&EpicsValue::DoubleArray(first.clone())) + .await + .expect("caput"); + wait_on_disk(&s, t0, &first).await; + // The flush owner commits `last_timestamp` on its next flush. + let deadline = tokio::time::Instant::now() + Duration::from_secs(10); + loop { + let rec = s.registry.get_pv(PV).unwrap().expect("registry row"); + if rec.last_timestamp.is_some_and(|t| t >= t0) { + break; + } + assert!( + tokio::time::Instant::now() < deadline, + "last_timestamp never committed: {rec:?}" + ); + tokio::time::sleep(Duration::from_millis(50)).await; + } + + // Restart: the first manager's producers stop, and a second + // manager with its own write pool restores from the same registry + // and store. + tokio::time::timeout(Duration::from_secs(10), s.mgr.shutdown()) + .await + .expect("producers stop"); + let (mgr2, rx2) = ChannelManager::new(s.storage.clone(), s.registry.clone(), None) + .await + .unwrap(); + let (pool2_shutdown, shutdown_rx2) = tokio::sync::watch::channel(false); + let pool2 = tokio::spawn(run_sharded_write_pool( + s.storage.clone(), + s.registry.clone(), + rx2, + shutdown_rx2, + ShardedWritePoolConfig { + shards: 1, + per_shard_buffer: 1024, + write_loop: WriteLoopConfig { + flush_period: Duration::from_millis(300), + ..Default::default() + }, + }, + )); + assert_eq!(mgr2.restore_from_registry().await.unwrap(), 1); + let deadline = tokio::time::Instant::now() + Duration::from_secs(10); + while counters(&mgr2, PV).events_received == 0 { + assert!( + tokio::time::Instant::now() < deadline, + "restored monitor never delivered" + ); + tokio::time::sleep(Duration::from_millis(50)).await; + } + + let second = vec![4.0, 5.0, 6.0]; + s.ch.put(&EpicsValue::DoubleArray(second.clone())) + .await + .expect("caput"); + let got = wait_on_disk(&s, t0, &second).await; + let copies = got.iter().filter(|g| **g == first).count(); + assert_eq!( + copies, 1, + "the redelivered committed value was stored again: {got:?}" + ); + let c = counters(&mgr2, PV); + assert_eq!(c.timestamp_drops, 1, "{c:?}"); + assert_eq!(c.events_stored, 1, "{c:?}"); + assert_eq!(c.storage_write_errors, 0, "{c:?}"); + + mgr2.stop_pv(PV).await.unwrap(); + pool2_shutdown.send(true).unwrap(); + tokio::time::timeout(Duration::from_secs(10), pool2) + .await + .expect("second write pool exits on shutdown") + .unwrap(); + s.finish().await; +} + /// `ChannelManager::shutdown` stops every producer and refuses new /// starts, so the write pool's drain that follows sees a fixed queue /// tail. The channel keeps posting; nothing may be received after From 31f9c36bdfe6a5deca9e6f6954023dbdbcb1e5ec Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 12:06:41 +0900 Subject: [PATCH 39/55] engine: park a PVA sample the write queue refuses in OverflowSlot pva_handle_event runs on the pvAccess reactor task and dropped every sample while the queue was full, so a PV that changed once during a write stall lost that change until its next one; the CA producer waits instead, and the CA client coalesces to the latest value. The slot keeps the newest refused sample per PV, drain_overflow_slot delivers it with backpressure, and only a replaced parked sample counts as an overflow drop. --- crates/archiver-engine/src/channel_manager.rs | 274 +++++++++++++++++- 1 file changed, 262 insertions(+), 12 deletions(-) diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index d17dd12..8ff2f0c 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -919,6 +919,9 @@ impl ChannelManager { let pv_name_loop = pv_name.clone(); let archive_fields_loop = record.archive_fields.clone(); let extras_for_loop = extras.clone(); + let slot = OverflowSlot::new(); + self.tasks + .spawn(drain_overflow_slot(slot.clone(), tx.clone(), token.clone())); self.tasks.spawn(async move { monitor_loop_pva( pv_name_loop, @@ -926,6 +929,7 @@ impl ChannelManager { element_count, pva_client, tx, + slot, token, ci, counters_for_loop, @@ -1440,6 +1444,7 @@ fn pva_handle_event( pv_name: &str, dbr_type: ArchDbType, tx: &mpsc::Sender, + slot: &OverflowSlot, conn_info: &Mutex, counters: &Arc, extras: &ExtraFieldsCache, @@ -1501,20 +1506,22 @@ fn pva_handle_event( element_count: Some(elem_count), counters: Some(counters.clone()), }; - // Non-blocking by design (this runs on the epics-rs reactor task), - // so a full channel drops the sample: count it AND say so. A - // closed channel means the write pool is gone (shutdown); the - // sample is dropped either way, and a silent drop of both cases - // was indistinguishable from a healthy PV in the logs. - match tx.try_send(pv_sample) { - Ok(()) => {} - Err(tokio::sync::mpsc::error::TrySendError::Full(_)) => { + // This runs on the epics-rs reactor task and cannot wait for + // queue space. A sample the queue will not take is parked for + // `drain_overflow_slot`; only the parked sample it replaces is + // lost. A closed channel means the write pool is gone (shutdown). + match slot.offer(tx, pv_sample) { + Offered::Kept => {} + Offered::Replaced => { counters .buffer_overflow_drops .fetch_add(1, Ordering::Relaxed); - debug!(pv = pv_name, "write channel full; PVA sample dropped"); + debug!( + pv = pv_name, + "write channel full; older parked PVA sample dropped" + ); } - Err(tokio::sync::mpsc::error::TrySendError::Closed(_)) => { + Offered::Closed => { debug!(pv = pv_name, "write channel closed; PVA sample dropped"); } } @@ -1522,8 +1529,9 @@ fn pva_handle_event( /// PVA monitor loop: subscribes once and parks until cancellation. /// Each fan-in event is decoded inline, packaged as a [`PvSample`], -/// and non-blocking-pushed into the storage write_loop's channel — -/// blocking would stall the pvAccess reactor thread. +/// and pushed into the storage write_loop's channel without waiting +/// (waiting would stall the pvAccess reactor task); what the channel +/// will not take is parked in `slot` for [`drain_overflow_slot`]. /// /// When `archive_fields` is non-empty, the subscription requests an /// explicit pvRequest with the mapped sub-field paths so the IOC @@ -1536,6 +1544,7 @@ async fn monitor_loop_pva( element_count: i32, pva_client: PvaClient, tx: mpsc::Sender, + slot: Arc, cancel_token: CancellationToken, conn_info: Arc>, counters: Arc, @@ -1604,6 +1613,7 @@ async fn monitor_loop_pva( }; let pv_name_cb = pv_name.clone(); let tx_cb = tx.clone(); + let slot_cb = slot.clone(); let conn_info_cb = conn_info.clone(); let counters_cb = counters.clone(); let extras_cb = extras.clone(); @@ -1616,6 +1626,7 @@ async fn monitor_loop_pva( &pv_name_cb, dbr_type, &tx_cb, + &slot_cb, &conn_info_cb, &counters_cb, &extras_cb, @@ -1656,6 +1667,7 @@ async fn monitor_loop_pva( // shapes that `PvField::descriptor()` would degrade. let pv_name_cb = pv_name.clone(); let tx_cb = tx.clone(); + let slot_cb = slot.clone(); let conn_info_cb = conn_info.clone(); let counters_cb = counters.clone(); let extras_cb = extras.clone(); @@ -1667,6 +1679,7 @@ async fn monitor_loop_pva( &pv_name_cb, dbr_type, &tx_cb, + &slot_cb, &conn_info_cb, &counters_cb, &extras_cb, @@ -2330,6 +2343,243 @@ async fn send_with_backpressure( } } +/// The newest PVA sample the write queue would not take. The PVA +/// monitor callback runs on the pvAccess reactor task and cannot wait +/// for queue space the way the CA producer does, so it parks the +/// sample here and [`drain_overflow_slot`] delivers it once the queue +/// drains; a newer sample replaces the parked one, and the replaced +/// sample is the one lost to overflow. This is the CA client's own +/// overflow rule (`CoalesceSlot`, after `dbEvent.c`'s replace-last): +/// a PV that changes once during a write stall still has its new +/// value archived, instead of the change being lost until the next +/// one. +struct OverflowSlot { + parked: std::sync::Mutex>, + ready: tokio::sync::Notify, +} + +/// What [`OverflowSlot::offer`] did with a sample. +enum Offered { + /// Queued, or parked with nothing lost. + Kept, + /// Parked in place of an older parked sample, which is lost. + Replaced, + /// The write pool is gone; the sample is dropped. + Closed, +} + +impl OverflowSlot { + fn new() -> Arc { + Arc::new(Self { + parked: std::sync::Mutex::new(None), + ready: tokio::sync::Notify::new(), + }) + } + + /// Queue `sample`, or park it when the queue is full. While a + /// sample is parked, later samples go through the slot too, so + /// delivery stays in timestamp order. + fn offer(&self, tx: &mpsc::Sender, sample: PvSample) -> Offered { + let mut parked = self.parked.lock().unwrap_or_else(|e| e.into_inner()); + if parked.is_some() { + *parked = Some(sample); + return Offered::Replaced; + } + match tx.try_send(sample) { + Ok(()) => Offered::Kept, + Err(tokio::sync::mpsc::error::TrySendError::Full(sample)) => { + *parked = Some(sample); + self.ready.notify_one(); + Offered::Kept + } + Err(tokio::sync::mpsc::error::TrySendError::Closed(_)) => Offered::Closed, + } + } + + fn take(&self) -> Option { + self.parked.lock().unwrap_or_else(|e| e.into_inner()).take() + } +} + +/// Deliver parked samples with backpressure, as the CA producer does +/// for its own. Ends when the write pool is gone or, after a final +/// pass over the slot, on cancel. +async fn drain_overflow_slot( + slot: Arc, + tx: mpsc::Sender, + cancel: CancellationToken, +) { + loop { + while let Some(sample) = slot.take() { + if tx.send(sample).await.is_err() { + return; + } + } + tokio::select! { + _ = cancel.cancelled() => { + if let Some(sample) = slot.take() { + let _ = tx.send(sample).await; + } + return; + } + _ = slot.ready.notified() => {} + } + } +} + +#[cfg(test)] +mod overflow_slot_tests { + use super::*; + use epics_rs::pva::pvdata::{PvStructure, ScalarValue}; + + fn nt_double(v: f64, secs: i64) -> PvField { + let mut ts = PvStructure::new("time_t"); + ts.fields.push(( + "secondsPastEpoch".into(), + PvField::Scalar(ScalarValue::Long(secs)), + )); + let mut nt = PvStructure::new("epics:nt/NTScalar:1.0"); + nt.fields + .push(("value".into(), PvField::Scalar(ScalarValue::Double(v)))); + nt.fields.push(("timeStamp".into(), PvField::Structure(ts))); + PvField::Structure(nt) + } + + struct Producer { + tx: mpsc::Sender, + slot: Arc, + counters: Arc, + conn: Mutex, + extras: ExtraFieldsCache, + now: i64, + } + + impl Producer { + fn new(tx: mpsc::Sender) -> Self { + Self { + tx, + slot: OverflowSlot::new(), + counters: Arc::new(PvCounters::default()), + conn: Mutex::new(ConnectionInfo::default()), + extras: ExtraFieldsCache::new(), + now: unix_secs(SystemTime::now()), + } + } + + /// One monitor event carrying `v`, `age` seconds old. + fn event(&self, v: f64, age: i64) { + let field = nt_double(v, self.now - age); + pva_handle_event( + &field, + &field.descriptor(), + "PV", + ArchDbType::ScalarDouble, + &self.tx, + &self.slot, + &self.conn, + &self.counters, + &self.extras, + &[], + 1800, + ); + } + + fn overflow_drops(&self) -> u64 { + self.counters.buffer_overflow_drops.load(Ordering::Relaxed) + } + } + + fn value(s: PvSample) -> ArchiverValue { + s.sample.value + } + + #[tokio::test] + async fn full_queue_parks_the_newest_sample_and_the_drainer_delivers_it() { + let (tx, mut rx) = mpsc::channel(1); + let p = Producer::new(tx.clone()); + p.event(1.0, 3); + p.event(2.0, 2); + p.event(3.0, 1); + assert_eq!(p.overflow_drops(), 1, "only the replaced sample is lost"); + assert_eq!(p.counters.events_received.load(Ordering::Relaxed), 3); + + let cancel = CancellationToken::new(); + let drainer = tokio::spawn(drain_overflow_slot(p.slot.clone(), tx, cancel.clone())); + assert_eq!( + value(rx.recv().await.unwrap()), + ArchiverValue::ScalarDouble(1.0) + ); + assert_eq!( + value(rx.recv().await.unwrap()), + ArchiverValue::ScalarDouble(3.0) + ); + cancel.cancel(); + drainer.await.unwrap(); + assert_eq!(p.overflow_drops(), 1); + } + + /// A sample that arrives while an older one is parked must not + /// overtake it through the queue, even when the queue has room. + #[tokio::test] + async fn a_parked_sample_keeps_later_samples_behind_it() { + let (tx, mut rx) = mpsc::channel(1); + let p = Producer::new(tx.clone()); + p.event(1.0, 3); + p.event(2.0, 2); + assert_eq!( + value(rx.recv().await.unwrap()), + ArchiverValue::ScalarDouble(1.0) + ); + p.event(3.0, 1); + assert!(rx.try_recv().is_err(), "3.0 went past the parked 2.0"); + assert_eq!(p.overflow_drops(), 1); + + let cancel = CancellationToken::new(); + let drainer = tokio::spawn(drain_overflow_slot(p.slot.clone(), tx, cancel.clone())); + assert_eq!( + value(rx.recv().await.unwrap()), + ArchiverValue::ScalarDouble(3.0) + ); + cancel.cancel(); + drainer.await.unwrap(); + } + + /// Cancel with a sample parked after the drainer's last pass: the + /// final pass still delivers it. + #[tokio::test] + async fn cancel_delivers_the_sample_parked_last() { + let (tx, mut rx) = mpsc::channel(1); + let p = Producer::new(tx.clone()); + let cancel = CancellationToken::new(); + let drainer = tokio::spawn(drain_overflow_slot(p.slot.clone(), tx, cancel.clone())); + tokio::task::yield_now().await; + p.event(1.0, 2); + p.event(2.0, 1); + cancel.cancel(); + // The final pass waits for queue space like any delivery, so + // the consumer must drain before the drainer can end. + assert_eq!( + value(rx.recv().await.unwrap()), + ArchiverValue::ScalarDouble(1.0) + ); + assert_eq!( + value(rx.recv().await.unwrap()), + ArchiverValue::ScalarDouble(2.0) + ); + drainer.await.unwrap(); + } + + #[tokio::test] + async fn closed_queue_drops_without_an_overflow_count() { + let (tx, rx) = mpsc::channel(1); + drop(rx); + let p = Producer::new(tx); + p.event(1.0, 1); + assert_eq!(p.overflow_drops(), 0); + assert!(p.slot.take().is_none()); + } +} + /// Scan loop: periodically read a channel value. #[allow(clippy::too_many_arguments)] async fn scan_loop( From 90d34390dc3a0ca6ae99c362338785395f80a844 Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 12:08:52 +0900 Subject: [PATCH 40/55] core: keep the writer when flush_dirty_writers fails at the write step A failed write leaves every unwritten byte in the BufWriter, yet the flush evicted the writer and counted up to 64 KiB of samples per PV as lost on a transient EIO or ENOSPC. The write step now reports the PV deferred, the same "still buffered, retry next cycle" the owner already handles for a busy slot; a failed fsync still evicts, since the kernel drops the pages it reported an error for. --- .../archiver-core/src/storage/plainpb/mod.rs | 72 ++++++++++---- crates/archiver-core/src/storage/traits.rs | 7 +- tests/plainpb_compat.rs | 99 +++++++++++++++++++ 3 files changed, 157 insertions(+), 21 deletions(-) diff --git a/crates/archiver-core/src/storage/plainpb/mod.rs b/crates/archiver-core/src/storage/plainpb/mod.rs index cd37bd0..298b339 100644 --- a/crates/archiver-core/src/storage/plainpb/mod.rs +++ b/crates/archiver-core/src/storage/plainpb/mod.rs @@ -190,21 +190,29 @@ impl<'a> Drop for TombstoneCleanupGuard<'a> { /// give the read-side and the write-side flush surfaces different /// semantics over the same underlying iteration. struct FlushOutcome { - /// PVs whose `flush()` syscall errored. Their cached writers - /// have been evicted (buffered bytes discarded via `into_parts`) - /// and their entries removed from the map. Surfaced to the - /// write_loop so it drops their `last_event` from the registry - /// commit batch. + /// PVs whose buffered bytes were discarded: fsync errored, or + /// the flush landed in a file that is gone. Their cached writers + /// have been evicted (`into_parts`) and their entries removed + /// from the map. Surfaced to the write_loop so it drops their + /// `last_event` from the registry commit batch. failed: Vec, - /// PVs whose slot was already locked (an `append` or another - /// flush is in flight). Their dirty bytes remain buffered and - /// will be picked up on the next flush cycle. Surfaced to the + /// PVs whose dirty bytes are still buffered: the slot was locked + /// (an `append` or another flush is in flight), or the `write` + /// behind the flush errored and the BufWriter kept what it did + /// not take. Picked up on the next flush cycle. Surfaced to the /// write_loop alongside `failed` so the registry doesn't claim /// `last_event` for samples whose bytes are still in BufWriter /// memory — under-commit, never over-commit. deferred: Vec, } +/// Which step of [`PlainPbStoragePlugin::flush_and_maybe_sync`] +/// failed; see there for why the two are not the same loss. +enum FlushFailure { + Write(std::io::Error), + Sync(std::io::Error), +} + /// Wraps a `PbFileReader` and clamps emitted samples to `[start, end]`. /// /// Java parity (e3b4471 + 88c7601): `binary_search_pb_file` returns @@ -352,14 +360,21 @@ impl PlainPbStoragePlugin { } /// Flush a dirty cached writer, then `sync_all` it to disk when - /// `fsync_on_flush` is set. A single fallible step so callers get - /// one `io::Result` to classify: a flush OR fsync failure is - /// treated identically (the bytes may not be durable). No-op fsync + /// `fsync_on_flush` is set. The two steps fail differently, which + /// is why the error says which one did: after a failed `write` + /// the `BufWriter` still holds every byte the kernel did not + /// take, so the writer can retry; after a failed `sync_all` the + /// bytes sit in the page cache with their error already reported, + /// which the kernel then drops, so they are lost. No-op fsync /// (and zero syscall) when the flag is off. - fn flush_and_maybe_sync(&self, cached: &mut CachedWriter) -> std::io::Result<()> { - cached.writer.flush()?; + fn flush_and_maybe_sync(&self, cached: &mut CachedWriter) -> Result<(), FlushFailure> { + cached.writer.flush().map_err(FlushFailure::Write)?; if self.fsync_on_flush { - cached.writer.get_ref().sync_all()?; + cached + .writer + .get_ref() + .sync_all() + .map_err(FlushFailure::Sync)?; } Ok(()) } @@ -406,8 +421,11 @@ impl PlainPbStoragePlugin { /// stuck PV from blocking the flush of every other PV — those /// are reported in `deferred` and retried next cycle. /// - /// Errored flushes evict the writer (buffered bytes dropped - /// via `into_parts`) and are recorded BOTH in the returned + /// A failed `write` keeps the writer: the `BufWriter` retains + /// the bytes, so the PV is reported in `deferred` like a busy + /// slot and retried next cycle. A failed fsync, or a flush that + /// landed in a deleted file, evicts the writer (buffered bytes + /// dropped via `into_parts`) and is recorded BOTH in the returned /// `failed` list (for the immediate caller) AND in /// `evicted_with_loss` (so the next ingest-side flush_owner /// pass can pick them up even if the immediate caller was the @@ -482,8 +500,26 @@ impl PlainPbStoragePlugin { cached.dirty = false; } } - Err(e) => { - tracing::warn!(pv, path = ?cached.path, "Failed to flush/fsync cached writer: {e}"); + Err(FlushFailure::Write(e)) => { + // Nothing is lost yet: the BufWriter holds every + // byte the failed `write` did not take. Keep the + // writer and its dirty bit; the pending registry + // timestamps stay uncommitted until a later cycle + // gets the bytes out. + tracing::warn!( + pv, + path = ?cached.path, + "Failed to write out cached writer; bytes stay buffered for retry: {e}" + ); + metrics::counter!( + "archiver_pb_flush_failures_total", + "tier" => self.plugin_name.clone(), + ) + .increment(1); + deferred.push(pv); + } + Err(FlushFailure::Sync(e)) => { + tracing::warn!(pv, path = ?cached.path, "Failed to fsync cached writer: {e}"); metrics::counter!( "archiver_pb_flush_failures_total", "tier" => self.plugin_name.clone(), diff --git a/crates/archiver-core/src/storage/traits.rs b/crates/archiver-core/src/storage/traits.rs index 4ecd25f..bf821e7 100644 --- a/crates/archiver-core/src/storage/traits.rs +++ b/crates/archiver-core/src/storage/traits.rs @@ -57,9 +57,10 @@ pub struct AppendMeta { /// loss queue and are drained by the owner via /// [`StoragePlugin::take_loss_markers`]. /// -/// * **deferred** — the writer's per-PV slot was already locked by -/// an in-flight append. The bytes are still buffered and will be -/// flushed on the next cycle. The caller MUST skip these from +/// * **deferred** — the bytes are still buffered: the writer's +/// per-PV slot was locked by an in-flight append, or the `write` +/// behind the flush errored and the BufWriter kept what it did not +/// take. They will be flushed on the next cycle. The caller MUST skip these from /// THIS cycle's commit but MUST keep them in `ts_updates` so the /// timestamp commits on a later cycle. Treating deferred as a /// permanent failure permanently loses the registry timestamp diff --git a/tests/plainpb_compat.rs b/tests/plainpb_compat.rs index 888f967..cfdec66 100644 --- a/tests/plainpb_compat.rs +++ b/tests/plainpb_compat.rs @@ -2023,3 +2023,102 @@ async fn append_recovers_from_emfile_at_the_header_probe() { ] ); } + +/// Caps this process's file size (`RLIMIT_FSIZE`) so a write past +/// `bytes` fails with EFBIG; restores the limit on drop. +#[cfg(unix)] +struct FileSizeCap { + orig: libc::rlimit, +} + +#[cfg(unix)] +impl FileSizeCap { + fn new(bytes: u64) -> Self { + // SAFETY: SIGXFSZ's default action terminates the process; + // ignored, the write returns EFBIG instead. + unsafe { libc::signal(libc::SIGXFSZ, libc::SIG_IGN) }; + let mut orig = libc::rlimit { + rlim_cur: 0, + rlim_max: 0, + }; + // SAFETY: plain libc calls with a valid out-pointer. + assert_eq!(unsafe { libc::getrlimit(libc::RLIMIT_FSIZE, &mut orig) }, 0); + let capped = libc::rlimit { + rlim_cur: bytes as libc::rlim_t, + rlim_max: orig.rlim_max, + }; + // SAFETY: lowering the soft limit of the calling process. + assert_eq!(unsafe { libc::setrlimit(libc::RLIMIT_FSIZE, &capped) }, 0); + Self { orig } + } +} + +#[cfg(unix)] +impl Drop for FileSizeCap { + fn drop(&mut self) { + // SAFETY: restoring the limit read in `new`. + unsafe { libc::setrlimit(libc::RLIMIT_FSIZE, &self.orig) }; + } +} + +/// A `write` that fails leaves its bytes in the BufWriter. The flush +/// must report the PV deferred and keep the writer, so that once +/// writes succeed again every sample is on disk and nothing was +/// counted as lost. +#[cfg(unix)] +#[tokio::test] +async fn failed_write_keeps_the_buffered_samples_for_the_next_flush() { + let dir = temp_dir(); + let plugin = + PlainPbStoragePlugin::new("sts", dir.path().to_path_buf(), PartitionGranularity::Year); + let pv = "TEST:FSIZE"; + let base = Utc.with_ymd_and_hms(2024, 6, 1, 0, 0, 0).unwrap(); + let sample = |secs: i64, v: f64| { + ArchiverSample::new( + SystemTime::from(base + chrono::Duration::seconds(secs)), + ArchiverValue::ScalarDouble(v), + ) + }; + plugin + .append_event(pv, ArchDbType::ScalarDouble, &sample(0, 1.0)) + .await + .unwrap(); + let clean = plugin.flush_ingest_writes().await.unwrap(); + assert!( + clean.failed.is_empty() && clean.deferred.is_empty(), + "{clean:?}" + ); + let path = plugin.file_path_for(pv, sample(0, 1.0).timestamp); + let len = std::fs::metadata(&path).unwrap().len(); + + let cap = FileSizeCap::new(len); + plugin + .append_event(pv, ArchDbType::ScalarDouble, &sample(1, 2.0)) + .await + .unwrap(); + let blocked = plugin.flush_ingest_writes().await.unwrap(); + assert_eq!(blocked.failed, Vec::::new()); + assert_eq!(blocked.deferred, vec![pv.to_string()]); + assert_eq!(plugin.open_writer_count(), 1, "the writer was evicted"); + assert_eq!(std::fs::metadata(&path).unwrap().len(), len); + drop(cap); + + let retried = plugin.flush_ingest_writes().await.unwrap(); + assert!( + retried.failed.is_empty() && retried.deferred.is_empty(), + "{retried:?}" + ); + assert!(plugin.take_loss_markers().is_empty()); + let mut rdr = PbFileReader::open(&path).unwrap(); + let mut values = Vec::new(); + while let Some(s) = rdr.next_event().unwrap() { + values.push(s.value); + } + assert_eq!( + values, + vec![ + ArchiverValue::ScalarDouble(1.0), + ArchiverValue::ScalarDouble(2.0) + ] + ); +} From 10ef415d0c55fe98a4e0f4eb44508b3892783944 Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 13:30:44 +0900 Subject: [PATCH 41/55] engine: make stop_tasks wait until a stopped PV has nothing in flight pause_pv only cancelled the token, so renamePV, changeTypeForPV, reassignAppliance and deletePV acted while the PV's tail could still be queued or inside an append: the late sample was stranded under the old name, refused by the converted partition, or deleted unmigrated. PvSample now carries an InFlight guard on PvCounters::in_flight, so every exit path settles by ownership, and PvHandle tracks its tasks; pause/stop/destroy wait for both, bounded by QUIESCE_TIMEOUT, and leave the registry status untouched on timeout. --- crates/archiver-engine/src/channel_manager.rs | 433 +++++++++++++----- tests/save_path_live_ca.rs | 113 ++++- tests/write_loop_failure.rs | 42 +- 3 files changed, 428 insertions(+), 160 deletions(-) diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index 8ff2f0c..39d6328 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -29,6 +29,12 @@ const CA_CONNECT_TIMEOUT: Duration = Duration::from_secs(10); const CA_RECONNECT_TIMEOUT: Duration = Duration::from_secs(30); /// Delay before retrying a failed CA subscription. const CA_RETRY_DELAY: Duration = Duration::from_secs(5); +/// Longest `ChannelManager::stop_tasks` waits for a PV's tasks to exit +/// and its queued samples to reach the store. Every producer wait is +/// shorter (a CA `get` is bounded at 30 s, the PVA client at 5 s) and +/// so is `WriteLoopConfig::append_timeout`, so only a storage append +/// that hangs past that can exhaust it. +const QUIESCE_TIMEOUT: Duration = Duration::from_secs(60); /// Hard floor on accepted timestamps. Mirrors Java's `PAST_CUTOFF_TIMESTAMP` /// of 1991-01-01 — earlier than that, the timestamp is almost certainly a @@ -247,6 +253,15 @@ pub struct PvCounters { /// dispatcher's consistent hash pins each PV to one shard); not /// part of the reported snapshot. pub ordering_last_ts_nanos: AtomicU64, + /// Samples of this PV that exist anywhere between the producer and + /// the end of their storage append: queued, parked, or being + /// written. Maintained only by [`InFlight`], which every + /// [`PvSample`] carries, so the count is exact on every exit path. + /// Zero after the PV's tasks have exited means nothing of the PV + /// can still land in the store. + in_flight: AtomicU64, + /// Woken when `in_flight` reaches zero; see [`Self::wait_in_flight`]. + settled: tokio::sync::Notify, } impl Default for PvCounters { @@ -270,6 +285,24 @@ impl Default for PvCounters { shutdown_abandoned_drops: AtomicU64::new(0), flush_losses: AtomicU64::new(0), ordering_last_ts_nanos: AtomicU64::new(0), + in_flight: AtomicU64::new(0), + settled: tokio::sync::Notify::new(), + } + } +} + +impl PvCounters { + /// Resolves once no sample of this PV is in flight. The `Notified` + /// is registered before the count is read, and the count is + /// `SeqCst` on both sides, so a release between the read and the + /// await still wakes this future. + async fn wait_in_flight(&self) { + loop { + let settled = self.settled.notified(); + if self.in_flight.load(Ordering::SeqCst) == 0 { + return; + } + settled.await; } } } @@ -357,6 +390,11 @@ struct PvHandle { /// connection reports. Lock-free reads; updates from the producer /// (monitor/scan) and the writer happen on different threads. counters: Arc, + /// Every task of this PV generation: the producer loop, the PVA + /// overflow drainer and refreshers, and the extra-field monitors. + /// `stop_tasks` waits on it so a stopped PV has no task left that + /// could still queue a sample. + tasks: TaskTracker, } /// Thread-safe cache of latest extra-field values for one PV. @@ -423,7 +461,8 @@ pub struct ChannelManager { counters: DashMap>, } -/// A sample ready to be written to storage. +/// A sample ready to be written to storage. Built only through +/// [`PvSample::new`], which registers it in its PV's in-flight count. pub struct PvSample { pub pv_name: String, pub dbr_type: ArchDbType, @@ -433,6 +472,56 @@ pub struct PvSample { /// type-change drops. None on samples produced before counter /// support was wired up — write_loop tolerates the absence. pub counters: Option>, + /// Released when the sample is dropped, on whichever path. + _in_flight: InFlight, +} + +impl PvSample { + pub fn new( + pv_name: String, + dbr_type: ArchDbType, + sample: ArchiverSample, + element_count: Option, + counters: Option>, + ) -> Self { + let _in_flight = InFlight::new(counters.clone()); + Self { + pv_name, + dbr_type, + sample, + element_count, + counters, + _in_flight, + } + } +} + +/// One unit of `PvCounters::in_flight`, held by a [`PvSample`] from +/// construction until the sample is dropped: after its append +/// returned, when a shard gate refuses it, when the dispatcher or a +/// dead shard drops it, when a newer sample replaces it in an +/// [`OverflowSlot`], or when the queue is abandoned at shutdown. +/// Accounting by ownership instead of at each of those sites is what +/// keeps the count exact. +struct InFlight(Option>); + +impl InFlight { + fn new(counters: Option>) -> Self { + if let Some(c) = &counters { + c.in_flight.fetch_add(1, Ordering::SeqCst); + } + Self(counters) + } +} + +impl Drop for InFlight { + fn drop(&mut self) { + if let Some(c) = &self.0 + && c.in_flight.fetch_sub(1, Ordering::SeqCst) == 1 + { + c.settled.notify_waiters(); + } + } } impl ChannelManager { @@ -532,6 +621,17 @@ impl ChannelManager { .clone() } + /// Spawn a task of one PV generation: tracked by `pv_tasks` so + /// [`Self::stop_tasks`] can wait for it, and by `self.tasks` for + /// [`Self::shutdown`]. + fn spawn_pv_task( + &self, + pv_tasks: &TaskTracker, + fut: impl std::future::Future + Send + 'static, + ) { + self.tasks.spawn(pv_tasks.track_future(fut)); + } + /// Restore all active PVs from the registry (called on startup). /// /// `pvs_by_status(Active)` already filters out alias rows (they carry @@ -783,6 +883,7 @@ impl ChannelManager { let field_tokens: Arc> = Arc::new(DashMap::new()); let update_lock = Arc::new(tokio::sync::Mutex::new(())); let counters = self.task_counters(record); + let pv_tasks = TaskTracker::new(); // Hold update_lock around the whole insert+spawn block so a // concurrent update_archive_fields can't observe the empty @@ -802,6 +903,7 @@ impl ChannelManager { field_tokens: field_tokens.clone(), update_lock: update_lock.clone(), counters: counters.clone(), + tasks: pv_tasks.clone(), }, ); @@ -810,8 +912,8 @@ impl ChannelManager { for field in &record.archive_fields { let child = cancel_token.child_token(); field_tokens.insert(field.clone(), child.clone()); - spawn_extra_field_monitor( - &self.ca_client, + self.spawn_extra_field_monitor( + &pv_tasks, &pv_name, field, extras.clone(), @@ -832,7 +934,7 @@ impl ChannelManager { let drift = self.server_ioc_drift_secs; match &record.sample_mode { SampleMode::Monitor => { - self.tasks.spawn(async move { + self.spawn_pv_task(&pv_tasks, async move { monitor_loop( pv_name, dbr_type, @@ -851,7 +953,7 @@ impl ChannelManager { } SampleMode::Scan { period_secs } => { let period = *period_secs; - self.tasks.spawn(async move { + self.spawn_pv_task(&pv_tasks, async move { scan_loop( pv_name, dbr_type, @@ -892,6 +994,7 @@ impl ChannelManager { let field_tokens: Arc> = Arc::new(DashMap::new()); let update_lock = Arc::new(tokio::sync::Mutex::new(())); let counters = self.task_counters(record); + let pv_tasks = TaskTracker::new(); self.channels.insert( pv_name.clone(), @@ -904,6 +1007,7 @@ impl ChannelManager { field_tokens: field_tokens.clone(), update_lock, counters: counters.clone(), + tasks: pv_tasks.clone(), }, ); @@ -920,9 +1024,11 @@ impl ChannelManager { let archive_fields_loop = record.archive_fields.clone(); let extras_for_loop = extras.clone(); let slot = OverflowSlot::new(); - self.tasks - .spawn(drain_overflow_slot(slot.clone(), tx.clone(), token.clone())); - self.tasks.spawn(async move { + self.spawn_pv_task( + &pv_tasks, + drain_overflow_slot(slot.clone(), tx.clone(), token.clone()), + ); + self.spawn_pv_task(&pv_tasks, async move { monitor_loop_pva( pv_name_loop, dbr_type, @@ -945,7 +1051,7 @@ impl ChannelManager { let pv_name_loop = pv_name.clone(); let archive_fields_loop = record.archive_fields.clone(); let extras_for_loop = extras.clone(); - self.tasks.spawn(async move { + self.spawn_pv_task(&pv_tasks, async move { scan_loop_pva( pv_name_loop, dbr_type, @@ -974,7 +1080,7 @@ impl ChannelManager { let registry = self.registry.clone(); let cancel = cancel_token.clone(); let counters_for_refresh = counters.clone(); - self.tasks.spawn(async move { + self.spawn_pv_task(&pv_tasks, async move { pva_metadata_refresh_loop( pv_name, pva_client, @@ -991,7 +1097,7 @@ impl ChannelManager { let counters_for_watch = counters.clone(); let cancel = cancel_token.clone(); let sample_mode = record.sample_mode.clone(); - self.tasks.spawn(async move { + self.spawn_pv_task(&pv_tasks, async move { pva_state_watchdog(pv_name, conn_info, counters_for_watch, cancel, sample_mode) .await; }); @@ -1016,7 +1122,7 @@ impl ChannelManager { // If the PV isn't currently active there's nothing more to do — // start_archiving_internal will pick up the new fields on resume. - let (parent_token, extras, field_tokens, update_lock, counters) = { + let (parent_token, extras, field_tokens, update_lock, counters, pv_tasks) = { let Some(handle) = self.channels.get(pv_name) else { return Ok(()); }; @@ -1026,6 +1132,7 @@ impl ChannelManager { handle.field_tokens.clone(), handle.update_lock.clone(), handle.counters.clone(), + handle.tasks.clone(), ) }; @@ -1058,8 +1165,8 @@ impl ChannelManager { if !field_tokens.contains_key(f) { let child = parent_token.child_token(); field_tokens.insert(f.clone(), child.clone()); - spawn_extra_field_monitor( - &self.ca_client, + self.spawn_extra_field_monitor( + &pv_tasks, pv_name, f, extras.clone(), @@ -1076,10 +1183,16 @@ impl ChannelManager { Ok(()) } - /// Pause archiving for a PV. - pub async fn pause_pv(&self, pv_name: &str) -> anyhow::Result<()> { - let lock = self.op_lock(pv_name); - let _g = lock.lock().await; + /// Stop the PV's tasks and wait until nothing of it is in flight: + /// the tasks have exited and every sample they queued has been + /// appended or dropped. On return the store holds all the samples + /// this task generation will ever produce, so the caller may set + /// the registry status and a later rename, type change, reassign + /// or delete acts on the whole data set. Fails, with the status + /// untouched, if the wait exceeds [`QUIESCE_TIMEOUT`]; the tasks + /// stay cancelled and a retry waits again. + async fn stop_tasks(&self, pv_name: &str) -> anyhow::Result<()> { + let deadline = tokio::time::Instant::now() + QUIESCE_TIMEOUT; if let Some((_key, handle)) = self.channels.remove(pv_name) { let extra_count = handle.field_tokens.len() as f64; handle.cancel_token.cancel(); @@ -1087,7 +1200,36 @@ impl ChannelManager { if extra_count > 0.0 { metrics::gauge!("archiver_extra_field_tasks").decrement(extra_count); } + handle.tasks.close(); + if tokio::time::timeout_at(deadline, handle.tasks.wait()) + .await + .is_err() + { + anyhow::bail!("{pv_name}: archiving tasks did not stop within {QUIESCE_TIMEOUT:?}"); + } } + let Some(counters) = self.counters.get(pv_name).map(|c| c.value().clone()) else { + return Ok(()); + }; + if tokio::time::timeout_at(deadline, counters.wait_in_flight()) + .await + .is_err() + { + anyhow::bail!( + "{pv_name}: {} samples still in flight after {QUIESCE_TIMEOUT:?}", + counters.in_flight.load(Ordering::SeqCst) + ); + } + Ok(()) + } + + /// Pause archiving for a PV. Returns once the PV has nothing in + /// flight (see [`Self::stop_tasks`]), so a `Paused` registry status + /// means the store holds every sample the PV produced. + pub async fn pause_pv(&self, pv_name: &str) -> anyhow::Result<()> { + let lock = self.op_lock(pv_name); + let _g = lock.lock().await; + self.stop_tasks(pv_name).await?; self.registry.set_status(pv_name, PvStatus::Paused)?; info!(pv = pv_name, "Paused archiving"); Ok(()) @@ -1135,14 +1277,7 @@ impl ChannelManager { pub async fn stop_pv(&self, pv_name: &str) -> anyhow::Result<()> { let lock = self.op_lock(pv_name); let _g = lock.lock().await; - if let Some((_key, handle)) = self.channels.remove(pv_name) { - let extra_count = handle.field_tokens.len() as f64; - handle.cancel_token.cancel(); - metrics::gauge!("archiver_pvs_active").decrement(1.0); - if extra_count > 0.0 { - metrics::gauge!("archiver_extra_field_tasks").decrement(extra_count); - } - } + self.stop_tasks(pv_name).await?; self.registry.set_status(pv_name, PvStatus::Inactive)?; info!(pv = pv_name, "Stopped archiving (inactive)"); Ok(()) @@ -1152,14 +1287,7 @@ impl ChannelManager { pub async fn destroy_pv(&self, pv_name: &str) -> anyhow::Result<()> { let lock = self.op_lock(pv_name); let _g = lock.lock().await; - if let Some((_key, handle)) = self.channels.remove(pv_name) { - let extra_count = handle.field_tokens.len() as f64; - handle.cancel_token.cancel(); - metrics::gauge!("archiver_pvs_active").decrement(1.0); - if extra_count > 0.0 { - metrics::gauge!("archiver_extra_field_tasks").decrement(extra_count); - } - } + self.stop_tasks(pv_name).await?; // The next task under this name is a new PV: its high-water // and statistics must not carry over. self.counters.remove(pv_name); @@ -1499,13 +1627,13 @@ fn pva_handle_event( } let mut sample = ArchiverSample::new(ts, value); attach_extras(extras, &mut sample); - let pv_sample = PvSample { - pv_name: pv_name.to_string(), + let pv_sample = PvSample::new( + pv_name.to_string(), dbr_type, sample, - element_count: Some(elem_count), - counters: Some(counters.clone()), - }; + Some(elem_count), + Some(counters.clone()), + ); // This runs on the epics-rs reactor task and cannot wait for // queue space. A sample the queue will not take is parked for // `drain_overflow_slot`; only the parked sample it replaces is @@ -1910,13 +2038,13 @@ async fn scan_loop_pva( } let mut sample = ArchiverSample::new(now, value); attach_extras(&extras, &mut sample); - let pv_sample = PvSample { - pv_name: pv_name.clone(), + let pv_sample = PvSample::new( + pv_name.clone(), dbr_type, sample, - element_count: Some(elem_count), - counters: Some(counters.clone()), - }; + Some(elem_count), + Some(counters.clone()), + ); if let Err(rejected) = send_with_backpressure(&tx, pv_sample).await { // Channel closed (write_loop down). Cooperative shutdown. let _ = rejected; @@ -2258,13 +2386,13 @@ async fn monitor_loop( .load(Ordering::Relaxed); attach_cnx_lost_headers(&mut sample, lost_secs, now_secs); } - let pv_sample = PvSample { - pv_name: pv_name.clone(), + let pv_sample = PvSample::new( + pv_name.clone(), dbr_type, sample, - element_count: Some(element_count), - counters: Some(counters.clone()), - }; + Some(element_count), + Some(counters.clone()), + ); if let Err(pv_sample) = send_with_backpressure(&tx, pv_sample).await { let _ = pv_sample; return; // Write loop shut down @@ -2680,13 +2808,13 @@ async fn scan_loop( let lost_secs = counters.last_disconnect_unix_secs.load(Ordering::Relaxed); attach_cnx_lost_headers(&mut sample, lost_secs, now_secs); } - let pv_sample = PvSample { - pv_name: pv_name.clone(), + let pv_sample = PvSample::new( + pv_name.clone(), dbr_type, sample, - element_count: Some(element_count), - counters: Some(counters.clone()), - }; + Some(element_count), + Some(counters.clone()), + ); if send_with_backpressure(&tx, pv_sample).await.is_err() { return; } @@ -2768,47 +2896,51 @@ fn epics_value_to_field_string(val: &EpicsValue) -> String { } } -/// Spawn a long-running task that subscribes to `.` and updates -/// `extras` with each event. Owned by `parent_token` so pause/destroy cleans -/// it up alongside the main PV. -fn spawn_extra_field_monitor( - ca_client: &CaClient, - pv_name: &str, - field: &str, - extras: Arc, - parent_token: CancellationToken, - counters: Arc, -) { - let full_name = format!("{pv_name}.{field}"); - let channel = ca_client.create_channel(&full_name); - let field_owned = field.to_string(); - let pv_owned = pv_name.to_string(); - - // Catch-unwind boundary: a panic from the CA client (e.g. malformed - // wire frame, allocation failure inside epics_rs) would propagate to - // the runtime and abort sibling tasks of this worker thread. Trap it - // here, log with PV+field context, and return normally so the runtime - // remains healthy. - let panic_pv = pv_owned.clone(); - let panic_field = field_owned.clone(); - tokio::spawn(async move { - let body = std::panic::AssertUnwindSafe(extra_field_monitor_body( - channel, - pv_owned, - field_owned, - extras, - parent_token, - counters, - )); - if let Err(payload) = futures::FutureExt::catch_unwind(body).await { - let msg = panic_payload_msg(&payload); - error!( - pv = panic_pv, - field = panic_field, - "Extra-field monitor panicked: {msg}" - ); - } - }); +impl ChannelManager { + /// Spawn a long-running task that subscribes to `.` and updates + /// `extras` with each event. Owned by `parent_token` so pause/destroy cleans + /// it up alongside the main PV, and tracked by `pv_tasks` so `stop_tasks` + /// waits for it. + fn spawn_extra_field_monitor( + &self, + pv_tasks: &TaskTracker, + pv_name: &str, + field: &str, + extras: Arc, + parent_token: CancellationToken, + counters: Arc, + ) { + let full_name = format!("{pv_name}.{field}"); + let channel = self.ca_client.create_channel(&full_name); + let field_owned = field.to_string(); + let pv_owned = pv_name.to_string(); + + // Catch-unwind boundary: a panic from the CA client (e.g. malformed + // wire frame, allocation failure inside epics_rs) would propagate to + // the runtime and abort sibling tasks of this worker thread. Trap it + // here, log with PV+field context, and return normally so the runtime + // remains healthy. + let panic_pv = pv_owned.clone(); + let panic_field = field_owned.clone(); + self.tasks.spawn(pv_tasks.track_future(async move { + let body = std::panic::AssertUnwindSafe(extra_field_monitor_body( + channel, + pv_owned, + field_owned, + extras, + parent_token, + counters, + )); + if let Err(payload) = futures::FutureExt::catch_unwind(body).await { + let msg = panic_payload_msg(&payload); + error!( + pv = panic_pv, + field = panic_field, + "Extra-field monitor panicked: {msg}" + ); + } + })); + } } /// Body of the spawned extra-field monitor. Split out so the spawn site @@ -5408,16 +5540,7 @@ mod shard_lifecycle_tests { ); let counters = Arc::new(PvCounters::default()); - let sample = PvSample { - pv_name: "pv:respawn".to_string(), - dbr_type: ArchDbType::ScalarDouble, - sample: ArchiverSample::new( - SystemTime::UNIX_EPOCH + Duration::from_secs(1_700_000_000), - ArchiverValue::ScalarDouble(1.0), - ), - element_count: Some(1), - counters: Some(counters.clone()), - }; + let sample = sample_for("pv:respawn", &counters); let mut governor = RespawnGovernor::new(1); route_sample(sample, &mut shards, &mut governor, &ctx); @@ -5480,16 +5603,7 @@ mod shard_lifecycle_tests { #[test] fn record_closed_drop_increments_per_pv_closed_counter_only() { let counters = Arc::new(PvCounters::default()); - let sample = PvSample { - pv_name: "pv:closed".to_string(), - dbr_type: ArchDbType::ScalarDouble, - sample: ArchiverSample::new( - SystemTime::UNIX_EPOCH + Duration::from_secs(1_700_000_000), - ArchiverValue::ScalarDouble(1.0), - ), - element_count: Some(1), - counters: Some(counters.clone()), - }; + let sample = sample_for("pv:closed", &counters); // Every closed-drop phase routes through record_closed_drop and // bumps the same per-PV counter. @@ -5501,4 +5615,93 @@ mod shard_lifecycle_tests { assert_eq!(counters.buffer_overflow_drops.load(Ordering::Relaxed), 0); assert_eq!(counters.shutdown_abandoned_drops.load(Ordering::Relaxed), 0); } + + fn sample_for(pv: &str, counters: &Arc) -> PvSample { + PvSample::new( + pv.to_string(), + ArchDbType::ScalarDouble, + ArchiverSample::new( + SystemTime::UNIX_EPOCH + Duration::from_secs(1_700_000_000), + ArchiverValue::ScalarDouble(1.0), + ), + Some(1), + Some(counters.clone()), + ) + } + + /// Boundary: a sample the dispatcher drops on a full shard leaves + /// the in-flight count, and a sample sitting in the shard queue + /// leaves it only when the queue lets go of it. + #[tokio::test] + async fn in_flight_settles_when_the_dispatcher_drops_on_a_full_shard() { + let ctx = ShardSpawnCtx { + storage: Arc::new(OkStorage), + pending: Arc::new(PendingReports::new()), + per_shard_buffer: 1, + write_loop: WriteLoopConfig::default(), + }; + // A live shard channel nobody reads: one queued sample fills it. + let (full_tx, full_rx) = mpsc::channel::(1); + let (drain_tx, _drain_rx) = tokio::sync::watch::channel(false); + let mut shards = vec![ShardSlot { + tx: full_tx, + handle: tokio::spawn(async {}), + drain_tx, + }]; + let counters = Arc::new(PvCounters::default()); + shards[0] + .tx + .try_send(sample_for("pv:full", &counters)) + .expect("first sample fills the shard queue"); + assert_eq!(counters.in_flight.load(Ordering::SeqCst), 1); + + let mut governor = RespawnGovernor::new(1); + route_sample( + sample_for("pv:full", &counters), + &mut shards, + &mut governor, + &ctx, + ); + assert_eq!(counters.buffer_overflow_drops.load(Ordering::Relaxed), 1); + assert_eq!( + counters.in_flight.load(Ordering::SeqCst), + 1, + "the dropped sample settled; the queued one is still in flight" + ); + + drop(shards); + drop(full_rx); + assert_eq!( + counters.in_flight.load(Ordering::SeqCst), + 0, + "a sample the queue lets go of settles" + ); + } + + /// Boundary: zero in flight resolves at once; one in flight holds + /// the waiter until that sample is dropped. + #[tokio::test] + async fn wait_in_flight_wakes_when_the_last_sample_is_dropped() { + let counters = Arc::new(PvCounters::default()); + tokio::time::timeout(Duration::from_secs(1), counters.wait_in_flight()) + .await + .expect("nothing in flight resolves immediately"); + + let sample = sample_for("pv:wait", &counters); + let waiter = { + let counters = counters.clone(); + tokio::spawn(async move { counters.wait_in_flight().await }) + }; + tokio::time::sleep(Duration::from_millis(50)).await; + assert!( + !waiter.is_finished(), + "one sample in flight keeps the waiter pending" + ); + + drop(sample); + tokio::time::timeout(Duration::from_secs(1), waiter) + .await + .expect("dropping the last sample wakes the waiter") + .unwrap(); + } } diff --git a/tests/save_path_live_ca.rs b/tests/save_path_live_ca.rs index dc92fda..da3dff3 100644 --- a/tests/save_path_live_ca.rs +++ b/tests/save_path_live_ca.rs @@ -10,14 +10,14 @@ use std::sync::Arc; use std::time::{Duration, SystemTime}; -use archiver_core::registry::{Protocol, PvRegistry, SampleMode}; +use archiver_core::registry::{Protocol, PvRegistry, PvStatus, SampleMode}; use archiver_core::retrieval::query::query_data; use archiver_core::storage::partition::PartitionGranularity; use archiver_core::storage::plainpb::PlainPbStoragePlugin; use archiver_core::storage::traits::StoragePlugin; use archiver_core::types::{ArchDbType, ArchiverValue}; use archiver_engine::channel_manager::{ - ChannelManager, PvCountersSnapshot, ShardedWritePoolConfig, WriteLoopConfig, + ChannelManager, PvCountersSnapshot, PvSample, ShardedWritePoolConfig, WriteLoopConfig, run_sharded_write_pool, }; use epics_rs::base::server::records::waveform::WaveformRecord; @@ -46,7 +46,9 @@ struct Stack { storage: Arc, registry: Arc, mgr: Arc, - pool: tokio::task::JoinHandle<()>, + /// The write pool's input, until `start_pool` spawns the pool. + pool_input: Option>, + pool: Option>, pool_shutdown: tokio::sync::watch::Sender, server_task: tokio::task::JoinHandle<()>, _writer: CaClient, @@ -55,6 +57,14 @@ struct Stack { impl Stack { async fn start() -> Self { + let mut s = Self::start_without_pool().await; + s.start_pool(); + s + } + + /// Everything but the write pool: the samples the producers queue + /// stay in the channel until `start_pool`. + async fn start_without_pool() -> Self { // Quiet unless RUST_LOG is set (e.g. archiver_engine=debug). let _ = tracing_subscriber::fmt() .with_env_filter(tracing_subscriber::EnvFilter::from_default_env()) @@ -93,21 +103,7 @@ impl Stack { .await .unwrap(); let mgr = Arc::new(mgr); - let (pool_shutdown, shutdown_rx) = tokio::sync::watch::channel(false); - let pool = tokio::spawn(run_sharded_write_pool( - storage.clone(), - registry.clone(), - rx, - shutdown_rx, - ShardedWritePoolConfig { - shards: 1, - per_shard_buffer: 1024, - write_loop: WriteLoopConfig { - flush_period: Duration::from_millis(300), - ..Default::default() - }, - }, - )); + let (pool_shutdown, _) = tokio::sync::watch::channel(false); let writer = CaClient::new().await.expect("writer client"); let ch = writer.create_channel(PV); @@ -123,7 +119,8 @@ impl Stack { storage, registry, mgr, - pool, + pool_input: Some(rx), + pool: None, pool_shutdown, server_task, _writer: writer, @@ -131,6 +128,24 @@ impl Stack { } } + fn start_pool(&mut self) { + let rx = self.pool_input.take().expect("write pool already started"); + self.pool = Some(tokio::spawn(run_sharded_write_pool( + self.storage.clone(), + self.registry.clone(), + rx, + self.pool_shutdown.subscribe(), + ShardedWritePoolConfig { + shards: 1, + per_shard_buffer: 1024, + write_loop: WriteLoopConfig { + flush_period: Duration::from_millis(300), + ..Default::default() + }, + }, + ))); + } + /// Start archiving `PV` and wait for the connect-time event, which /// proves the monitor subscription is active. A put that lands /// while the subscription is still being set up can be reported by @@ -152,11 +167,13 @@ impl Stack { } async fn finish(self) { - self.pool_shutdown.send(true).unwrap(); - tokio::time::timeout(Duration::from_secs(10), self.pool) - .await - .expect("write pool exits on shutdown") - .unwrap(); + if let Some(pool) = self.pool { + self.pool_shutdown.send(true).unwrap(); + tokio::time::timeout(Duration::from_secs(10), pool) + .await + .expect("write pool exits on shutdown") + .unwrap(); + } self.server_task.abort(); } } @@ -437,3 +454,51 @@ async fn shutdown_stops_producers_and_refuses_new_starts() { s.finish().await; } + +/// `pause_pv` returns only once every sample the PV produced has +/// reached the store: with no write pool running, the connect-time +/// sample sits in the queue and the pause must wait for it. A +/// `Paused` PV therefore has nothing in flight, and the mgmt +/// operations that require `Paused` (rename, type change, reassign, +/// delete) act on the whole data set. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn pause_waits_for_the_queued_sample_to_land() { + let mut s = Stack::start_without_pool().await; + let t0 = SystemTime::now(); + s.archive().await; + + let mgr = s.mgr.clone(); + let pause = tokio::spawn(async move { mgr.pause_pv(PV).await }); + tokio::time::sleep(Duration::from_millis(500)).await; + assert!( + !pause.is_finished(), + "pause returned while the connect-time sample was still queued" + ); + assert!(stored_arrays(&s, t0).await.is_empty()); + + s.start_pool(); + tokio::time::timeout(Duration::from_secs(10), pause) + .await + .expect("pause completes once the queued sample lands") + .unwrap() + .expect("pause_pv"); + let rec = s.registry.get_pv(PV).unwrap().expect("registry row"); + assert_eq!(rec.status, PvStatus::Paused); + + // The paused PV's counters are no longer reported, so the + // evidence is the disk: the ticker flush lands the seed value. + let deadline = tokio::time::Instant::now() + Duration::from_secs(20); + loop { + let got = stored_arrays(&s, t0).await; + if got == vec![vec![0.0; NELM as usize]] { + break; + } + assert!( + tokio::time::Instant::now() < deadline, + "timed out waiting for the seed value on disk; have {got:?}" + ); + tokio::time::sleep(Duration::from_millis(200)).await; + } + + s.finish().await; +} diff --git a/tests/write_loop_failure.rs b/tests/write_loop_failure.rs index ab7e249..00ee1ab 100644 --- a/tests/write_loop_failure.rs +++ b/tests/write_loop_failure.rs @@ -267,13 +267,13 @@ fn sample_at(ts: SystemTime, value: f64) -> ArchiverSample { } fn pv_sample(pv: &str, ts: SystemTime, value: f64, counters: &Arc) -> PvSample { - PvSample { - pv_name: pv.to_string(), - dbr_type: ArchDbType::ScalarDouble, - sample: sample_at(ts, value), - element_count: Some(1), - counters: Some(counters.clone()), - } + PvSample::new( + pv.to_string(), + ArchDbType::ScalarDouble, + sample_at(ts, value), + Some(1), + Some(counters.clone()), + ) } fn ts(secs_since_2020: u64) -> SystemTime { @@ -1728,13 +1728,13 @@ async fn type_gate_accepts_new_registry_type_from_a_new_task() { // changeTypeForPV flipped the registry to Int; the resumed task // carries the new type and matching values. let gen2 = Arc::new(PvCounters::default()); - tx.send(PvSample { - pv_name: "A".to_string(), - dbr_type: ArchDbType::ScalarInt, - sample: ArchiverSample::new(ts(200), ArchiverValue::ScalarInt(2)), - element_count: Some(1), - counters: Some(gen2.clone()), - }) + tx.send(PvSample::new( + "A".to_string(), + ArchDbType::ScalarInt, + ArchiverSample::new(ts(200), ArchiverValue::ScalarInt(2)), + Some(1), + Some(gen2.clone()), + )) .await .unwrap(); tokio::time::sleep(Duration::from_millis(200)).await; @@ -1770,13 +1770,13 @@ async fn type_gate_drops_value_that_disagrees_with_registry_type() { tx.send(pv_sample("A", ts(100), 1.0, &counters)) .await .unwrap(); - tx.send(PvSample { - pv_name: "A".to_string(), - dbr_type: ArchDbType::ScalarDouble, - sample: ArchiverSample::new(ts(200), ArchiverValue::ScalarInt(7)), - element_count: Some(1), - counters: Some(counters.clone()), - }) + tx.send(PvSample::new( + "A".to_string(), + ArchDbType::ScalarDouble, + ArchiverSample::new(ts(200), ArchiverValue::ScalarInt(7)), + Some(1), + Some(counters.clone()), + )) .await .unwrap(); tokio::time::sleep(Duration::from_millis(200)).await; From ec08a863cbe993d701fcb6918084697efca47693 Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 13:34:56 +0900 Subject: [PATCH 42/55] core: keep the committed last_timestamp across import_pv, renamePV and receivePVMigration import_pv_with_protocol used INSERT OR REPLACE, which re-inserts the row without last_timestamp, so putPVTypeInfo and changeTypeForPV on an archived PV dropped its high-water and the next restart stored the connect-time redelivery again. renamePV left the destination row at NULL although the moved partitions end at the source's, and the migration receiver appends past the write pool's flush owner, so its row stayed NULL too. The import is now an upsert over the columns it owns; the two handlers commit the tail through update_last_timestamp, the migration one only after its flush succeeded. --- crates/archiver-api/src/handlers/mgmt/p2.rs | 16 ++++- .../src/handlers/mgmt/type_info.rs | 8 +++ .../src/services/fakes/pv_repository.rs | 13 ++++ .../src/services/impls/registry_repo.rs | 8 +++ crates/archiver-api/src/services/traits.rs | 7 +++ crates/archiver-core/src/registry.rs | 61 ++++++++++++++++++- tests/api_mgmt.rs | 11 ++++ 7 files changed, 120 insertions(+), 4 deletions(-) diff --git a/crates/archiver-api/src/handlers/mgmt/p2.rs b/crates/archiver-api/src/handlers/mgmt/p2.rs index e7d3803..36bcccb 100644 --- a/crates/archiver-api/src/handlers/mgmt/p2.rs +++ b/crates/archiver-api/src/handlers/mgmt/p2.rs @@ -567,6 +567,7 @@ pub async fn receive_pv_migration( } let mut written = 0usize; + let mut newest: Option = None; for s in samples { let secs_i = s["secs"].as_i64().unwrap_or(0); if secs_i < 0 { @@ -604,9 +605,20 @@ pub async fn receive_pv_migration( return ApiError::internal(e).into_response(); } written += 1; + newest = Some(newest.map_or(ts, |n| n.max(ts))); } - if let Err(e) = state.storage.flush_writes().await { - tracing::warn!(pv = pv_name, "flush after migration failed: {e}"); + match state.storage.flush_writes().await { + Ok(()) => { + // The migrated tail is on disk; commit it so a resume here + // drops the connect-time redelivery instead of storing it + // again. + if let Some(ts) = newest + && let Err(e) = state.pv_cmd.update_last_timestamp(&pv_name, ts) + { + return ApiError::internal(e).into_response(); + } + } + Err(e) => tracing::warn!(pv = pv_name, "flush after migration failed: {e}"), } axum::Json(serde_json::json!({ diff --git a/crates/archiver-api/src/handlers/mgmt/type_info.rs b/crates/archiver-api/src/handlers/mgmt/type_info.rs index 4df4bac..e6da105 100644 --- a/crates/archiver-api/src/handlers/mgmt/type_info.rs +++ b/crates/archiver-api/src/handlers/mgmt/type_info.rs @@ -529,6 +529,14 @@ pub async fn rename_pv( ) { return ApiError::internal(e).into_response(); } + // The moved partitions end where the source's committed + // last_timestamp says; the new row starts there so a resume drops + // the connect-time redelivery instead of storing it again. + if let Some(ts) = source.last_timestamp + && let Err(e) = state.pv_cmd.update_last_timestamp(&q.newname, ts) + { + return ApiError::internal(e).into_response(); + } if let Err(e) = state.pv_cmd.set_status(&q.pv, PvStatus::Paused) { return ApiError::internal(e).into_response(); diff --git a/crates/archiver-api/src/services/fakes/pv_repository.rs b/crates/archiver-api/src/services/fakes/pv_repository.rs index 8495d33..00b573b 100644 --- a/crates/archiver-api/src/services/fakes/pv_repository.rs +++ b/crates/archiver-api/src/services/fakes/pv_repository.rs @@ -241,6 +241,19 @@ impl PvCommandRepository for InMemoryPvRepository { } } + fn update_last_timestamp( + &self, + pv: &str, + timestamp: std::time::SystemTime, + ) -> anyhow::Result<()> { + if let Some(record) = self.pvs.lock().unwrap().get_mut(pv) + && record.last_timestamp.is_none_or(|cur| timestamp > cur) + { + record.last_timestamp = Some(timestamp); + } + Ok(()) + } + fn update_sample_mode(&self, pv: &str, mode: &SampleMode) -> anyhow::Result { if let Some(record) = self.pvs.lock().unwrap().get_mut(pv) { record.sample_mode = mode.clone(); diff --git a/crates/archiver-api/src/services/impls/registry_repo.rs b/crates/archiver-api/src/services/impls/registry_repo.rs index 573fa9c..f44c1b0 100644 --- a/crates/archiver-api/src/services/impls/registry_repo.rs +++ b/crates/archiver-api/src/services/impls/registry_repo.rs @@ -93,6 +93,14 @@ impl PvCommandRepository for RegistryRepository { self.inner.set_status(pv, status) } + fn update_last_timestamp( + &self, + pv: &str, + timestamp: std::time::SystemTime, + ) -> anyhow::Result<()> { + self.inner.update_last_timestamp(pv, timestamp) + } + fn update_sample_mode(&self, pv: &str, mode: &SampleMode) -> anyhow::Result { self.inner.update_sample_mode(pv, mode) } diff --git a/crates/archiver-api/src/services/traits.rs b/crates/archiver-api/src/services/traits.rs index 1f160a2..53add52 100644 --- a/crates/archiver-api/src/services/traits.rs +++ b/crates/archiver-api/src/services/traits.rs @@ -41,6 +41,13 @@ pub trait PvCommandRepository: Send + Sync { ) -> anyhow::Result<()>; fn remove_pv(&self, pv: &str) -> anyhow::Result; fn set_status(&self, pv: &str, status: PvStatus) -> anyhow::Result; + /// Advance the committed `last_timestamp`; never regresses it (the + /// registry owns that rule). + fn update_last_timestamp( + &self, + pv: &str, + timestamp: std::time::SystemTime, + ) -> anyhow::Result<()>; fn update_sample_mode(&self, pv: &str, mode: &SampleMode) -> anyhow::Result; fn update_metadata( &self, diff --git a/crates/archiver-core/src/registry.rs b/crates/archiver-core/src/registry.rs index 1e79945..f26b5d4 100644 --- a/crates/archiver-core/src/registry.rs +++ b/crates/archiver-core/src/registry.rs @@ -742,11 +742,30 @@ impl PvRegistry { Some(serde_json::to_string(archive_fields)?) }; + // UPSERT, not INSERT OR REPLACE (see `register_pv`): REPLACE + // re-inserted the row without last_timestamp, so putPVTypeInfo + // and changeTypeForPV on an archived PV dropped its committed + // high-water and the next restart stored the connect-time + // redelivery again. An import owns every column but that one. conn.execute( - "INSERT OR REPLACE INTO pv_info + "INSERT INTO pv_info (pv_name, dbr_type, sample_mode, sample_period, status, element_count, created_at, updated_at, prec, egu, alias_for, archive_fields, policy_name, protocol) - VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, ?8, ?9, ?10, ?11, ?12, ?13, ?14)", + VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, ?8, ?9, ?10, ?11, ?12, ?13, ?14) + ON CONFLICT(pv_name) DO UPDATE SET + dbr_type = excluded.dbr_type, + sample_mode = excluded.sample_mode, + sample_period = excluded.sample_period, + status = excluded.status, + element_count = excluded.element_count, + created_at = excluded.created_at, + updated_at = excluded.updated_at, + prec = excluded.prec, + egu = excluded.egu, + alias_for = excluded.alias_for, + archive_fields = excluded.archive_fields, + policy_name = excluded.policy_name, + protocol = excluded.protocol", params![ pv_name, dbr_type as i32, @@ -1423,6 +1442,44 @@ mod tests { assert_eq!(r.egu.as_deref(), Some("mA")); } + /// An import owns the metadata columns only: the committed + /// last_timestamp of an archived PV survives putPVTypeInfo and + /// changeTypeForPV, which both go through `import_pv`. + #[test] + fn import_pv_keeps_the_committed_last_timestamp() { + let reg = PvRegistry::in_memory().unwrap(); + reg.register_pv( + "PV:Imported", + ArchDbType::ScalarDouble, + &SampleMode::Monitor, + 1, + ) + .unwrap(); + let ts = SystemTime::UNIX_EPOCH + std::time::Duration::from_secs(1_700_000_000); + reg.update_last_timestamp("PV:Imported", ts).unwrap(); + + reg.import_pv( + "PV:Imported", + ArchDbType::ScalarInt, + &SampleMode::Monitor, + 1, + PvStatus::Paused, + None, + Some("0"), + None, + None, + &[], + None, + ) + .unwrap(); + + let r = reg.get_pv("PV:Imported").unwrap().unwrap(); + assert_eq!(r.last_timestamp, Some(ts)); + assert_eq!(r.dbr_type, ArchDbType::ScalarInt); + assert_eq!(r.status, PvStatus::Paused); + assert_eq!(r.prec.as_deref(), Some("0")); + } + #[test] fn test_migration_from_old_schema() { // Build a connection with the v0.1.4 schema (no alias/archive_fields/policy) diff --git a/tests/api_mgmt.rs b/tests/api_mgmt.rs index f4f213f..eba82cf 100644 --- a/tests/api_mgmt.rs +++ b/tests/api_mgmt.rs @@ -1099,6 +1099,10 @@ async fn test_rename_pv_after_pause_creates_destination() { StatusCode::OK ); + // The source's committed high-water moves with its data. + let ts = std::time::UNIX_EPOCH + std::time::Duration::from_secs(1_700_000_000); + registry.update_last_timestamp("SIM:Cosine", ts).unwrap(); + let req = get_request("/mgmt/bpl/renamePV?pv=SIM:Cosine&newname=SIM:CosineRenamed"); let resp = app.clone().oneshot(req).await.unwrap(); assert_eq!(resp.status(), StatusCode::OK); @@ -1112,6 +1116,7 @@ async fn test_rename_pv_after_pause_creates_destination() { let dst = registry.get_pv("SIM:CosineRenamed").unwrap().unwrap(); assert_eq!(dst.status, archiver_core::registry::PvStatus::Paused); assert_eq!(dst.dbr_type, src.dbr_type); + assert_eq!(dst.last_timestamp, Some(ts)); } #[tokio::test] @@ -2109,4 +2114,10 @@ async fn test_receive_pv_migration_imports_with_proxied_header() { let r = registry.get_pv("MIGRATED:PV").unwrap().unwrap(); assert_eq!(r.archive_fields, vec!["HIHI".to_string()]); assert_eq!(r.status, archiver_core::registry::PvStatus::Paused); + // The migrated tail is committed, so a resume here does not store + // the connect-time redelivery of the newest sample again. + assert_eq!( + r.last_timestamp, + Some(std::time::UNIX_EPOCH + std::time::Duration::from_secs(1_700_000_010)) + ); } From 277f7def0e1e226120400187c46dafd715d1266e Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 13:39:11 +0900 Subject: [PATCH 43/55] engine: observe cancellation at the producers' bounded connect and get waits stop_tasks now waits for a PV's tasks, so a wait that is bounded but not raced against the cancel token turns into pause latency on a disconnected PV: 5 s for the scan connect wait, 10 s for an extra field's, 30 s for a CA get, 5 s for the PVA client. pvmonitor_handle stays unraced on purpose: dropping it mid-subscribe could leave a server-side monitor running with no handle to stop it. --- crates/archiver-engine/src/channel_manager.rs | 37 ++++++++++++++++--- tests/save_path_live_ca.rs | 37 +++++++++++++++++++ 2 files changed, 68 insertions(+), 6 deletions(-) diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index 39d6328..e051aef 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -1723,7 +1723,11 @@ async fn monitor_loop_pva( // archiver invariant is that every V4 sample on disk // carries the channel-INIT descriptor, so degrading to // value-recovery is not allowed. - let canonical = match pva_client.pvinfo(&pv_name).await { + let info = tokio::select! { + _ = cancel_token.cancelled() => return, + r = pva_client.pvinfo(&pv_name) => r, + }; + let canonical = match info { Ok(d) => Arc::new(d), Err(e) => { counters @@ -1864,6 +1868,9 @@ async fn monitor_loop_pva( } } }; + // Not raced against cancellation: dropping this future + // mid-subscribe could leave the server-side monitor running + // with no handle to stop it. The client timeout bounds it. let handle = match pva_client.pvmonitor_handle(&pv_name, cb, on_conn).await { Ok(h) => h, Err(e) => { @@ -1954,7 +1961,10 @@ async fn scan_loop_pva( // the value, so V4GenericBytes encoding can preserve Union / // UnionArray / Variant schemas that `PvField::descriptor()` // would otherwise degrade. - let res = tokio::time::timeout(pvget_timeout, pva_client.pvget_full(&pv_name)).await; + let res = tokio::select! { + _ = cancel_token.cancelled() => return, + r = tokio::time::timeout(pvget_timeout, pva_client.pvget_full(&pv_name)) => r, + }; let (field, canonical) = match res { Ok(Ok(r)) => (r.value, r.introspection), Ok(Err(e)) => { @@ -2082,7 +2092,10 @@ async fn pva_metadata_refresh_loop( _ = tick.tick() => {} } - let res = tokio::time::timeout(FETCH_TIMEOUT, pva_client.pvget(&pv_name)).await; + let res = tokio::select! { + _ = cancel_token.cancelled() => return, + r = tokio::time::timeout(FETCH_TIMEOUT, pva_client.pvget(&pv_name)) => r, + }; let field = match res { Ok(Ok(f)) => f, _ => { @@ -2735,7 +2748,11 @@ async fn scan_loop( _ = interval.tick() => {} } - if channel.wait_connected(CA_RETRY_DELAY).await.is_err() { + let connected = tokio::select! { + _ = cancel_token.cancelled() => return, + r = channel.wait_connected(CA_RETRY_DELAY) => r, + }; + if connected.is_err() { let was_connected = { let mut ci = conn_info.lock().unwrap_or_else(|e| e.into_inner()); let prev = ci.is_connected; @@ -2779,7 +2796,11 @@ async fn scan_loop( metadata_done = true; } - match channel.get().await { + let got = tokio::select! { + _ = cancel_token.cancelled() => return, + r = channel.get() => r, + }; + match got { Ok((_dbr_type, epics_val)) => { let now = SystemTime::now(); let first_after_connect = { @@ -2955,7 +2976,11 @@ async fn extra_field_monitor_body( ) { // Initial connect attempt — failure here is non-fatal (the field may // not exist on every IOC; we just leave the cache empty). - if channel.wait_connected(CA_CONNECT_TIMEOUT).await.is_err() { + let connected = tokio::select! { + _ = parent_token.cancelled() => return, + r = channel.wait_connected(CA_CONNECT_TIMEOUT) => r, + }; + if connected.is_err() { debug!( pv = pv_owned, field = field_owned, diff --git a/tests/save_path_live_ca.rs b/tests/save_path_live_ca.rs index da3dff3..44f75bc 100644 --- a/tests/save_path_live_ca.rs +++ b/tests/save_path_live_ca.rs @@ -27,6 +27,8 @@ use epics_rs::ca::server::CaServer; const PV: &str = "SMOKE:CA:WF"; const OTHER_PV: &str = "SMOKE:CA:OTHER"; +/// Not served by the in-process server: its channels never connect. +const MISSING_PV: &str = "SMOKE:CA:MISSING"; const NELM: i32 = 3; fn counters(mgr: &ChannelManager, pv: &str) -> PvCountersSnapshot { @@ -502,3 +504,38 @@ async fn pause_waits_for_the_queued_sample_to_land() { s.finish().await; } + +/// A PV that never connects keeps its scan loop and its extra-field +/// monitor inside connect waits. Those waits observe cancellation, so +/// `pause_pv` returns at once instead of after the wait's bound. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn pause_returns_promptly_while_producers_wait_to_connect() { + let s = Stack::start().await; + s.registry + .register_pv( + MISSING_PV, + ArchDbType::ScalarDouble, + &SampleMode::Scan { period_secs: 1.0 }, + 1, + ) + .unwrap(); + s.registry + .update_archive_fields(MISSING_PV, &["HIHI".to_string()]) + .unwrap(); + assert_eq!(s.mgr.restore_from_registry().await.unwrap(), 1); + // Let both tasks reach their connect waits. + tokio::time::sleep(Duration::from_millis(200)).await; + + let started = tokio::time::Instant::now(); + s.mgr.pause_pv(MISSING_PV).await.expect("pause_pv"); + let took = started.elapsed(); + assert!(took < Duration::from_secs(2), "pause took {took:?}"); + let rec = s + .registry + .get_pv(MISSING_PV) + .unwrap() + .expect("registry row"); + assert_eq!(rec.status, PvStatus::Paused); + + s.finish().await; +} From 6ed07192e6c781dcc08dbd3d63c2c32c5fccf8a8 Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 13:47:27 +0900 Subject: [PATCH 44/55] engine: drop the unused storage field from ChannelManager Nothing in the manager reads it: every write goes through the sample channel to the write pool, which owns the StoragePlugin. Removing it takes the storage parameter off ChannelManager::new and new_with_drift and the #[allow(dead_code)] that hid the dead field. --- crates/archiver-engine/src/channel_manager.rs | 8 +-- src/main.rs | 1 - tests/api_cluster.rs | 34 +++------ tests/api_failover.rs | 8 +-- tests/api_mgmt.rs | 69 ++++--------------- tests/api_retrieval.rs | 4 +- tests/save_path_live.rs | 4 +- tests/save_path_live_ca.rs | 8 +-- 8 files changed, 33 insertions(+), 103 deletions(-) diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index e051aef..a1a5a21 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -436,9 +436,6 @@ pub struct ChannelManager { /// PV. Without this, e.g. `pause_pv` racing with `resume_pv` can leave /// the registry status and the channel map disagreeing. op_locks: DashMap>>, - /// Storage backend. - #[allow(dead_code)] - storage: Arc, /// PV metadata registry. registry: Arc, /// Sample sender for the write thread. @@ -526,18 +523,16 @@ impl Drop for InFlight { impl ChannelManager { pub async fn new( - storage: Arc, registry: Arc, policy: Option, ) -> anyhow::Result<(Self, mpsc::Receiver)> { - Self::new_with_drift(storage, registry, policy, 30 * 60).await + Self::new_with_drift(registry, policy, 30 * 60).await } /// Construct with an explicit IOC drift bound. Java parity (6538631): /// keeps tests + sites that don't surface `EngineConfig` on the /// existing default while letting the daemon plumb a configured value. pub async fn new_with_drift( - storage: Arc, registry: Arc, policy: Option, server_ioc_drift_secs: u64, @@ -552,7 +547,6 @@ impl ChannelManager { channels: DashMap::new(), pending_archives: DashMap::new(), op_locks: DashMap::new(), - storage, registry, sample_tx: tx, policy, diff --git a/src/main.rs b/src/main.rs index db79902..4443a8a 100644 --- a/src/main.rs +++ b/src/main.rs @@ -155,7 +155,6 @@ async fn main() -> anyhow::Result<()> { // Initialize channel manager + write loop. Drift bound is per-site // configurable (Java parity 6538631). let (channel_mgr, sample_rx) = ChannelManager::new_with_drift( - storage.clone(), registry.clone(), policy, config.engine.server_ioc_drift_secs, diff --git a/tests/api_cluster.rs b/tests/api_cluster.rs index e2b4132..7cf0a7d 100644 --- a/tests/api_cluster.rs +++ b/tests/api_cluster.rs @@ -21,9 +21,7 @@ async fn start_mock_peer( registry: Arc, storage: Arc, ) -> (String, tokio::task::JoinHandle<()>) { - let (channel_mgr, _rx) = ChannelManager::new(storage.clone(), registry.clone(), None) - .await - .unwrap(); + let (channel_mgr, _rx) = ChannelManager::new(registry.clone(), None).await.unwrap(); let channel_mgr = Arc::new(channel_mgr); let repo = Arc::new(RegistryRepository::new(registry)); let archiver = Arc::new(ChannelArchiverControl::new(channel_mgr)); @@ -58,10 +56,9 @@ async fn build_cluster_test_app( local_storage: Arc, peer_url: &str, ) -> axum::Router { - let (channel_mgr, _rx) = - ChannelManager::new(local_storage.clone(), local_registry.clone(), None) - .await - .unwrap(); + let (channel_mgr, _rx) = ChannelManager::new(local_registry.clone(), None) + .await + .unwrap(); let channel_mgr = Arc::new(channel_mgr); let cluster_config = ClusterConfig { @@ -659,10 +656,9 @@ async fn build_cluster_test_app_with_auth( local_storage: Arc, peer_url: &str, ) -> axum::Router { - let (channel_mgr, _rx) = - ChannelManager::new(local_storage.clone(), local_registry.clone(), None) - .await - .unwrap(); + let (channel_mgr, _rx) = ChannelManager::new(local_registry.clone(), None) + .await + .unwrap(); let channel_mgr = Arc::new(channel_mgr); let cluster_config = ClusterConfig { @@ -714,9 +710,7 @@ async fn start_mock_peer_with_auth( registry: Arc, storage: Arc, ) -> (String, tokio::task::JoinHandle<()>) { - let (channel_mgr, _rx) = ChannelManager::new(storage.clone(), registry.clone(), None) - .await - .unwrap(); + let (channel_mgr, _rx) = ChannelManager::new(registry.clone(), None).await.unwrap(); let channel_mgr = Arc::new(channel_mgr); let repo = Arc::new(RegistryRepository::new(registry)); let archiver = Arc::new(ChannelArchiverControl::new(channel_mgr)); @@ -889,9 +883,7 @@ async fn start_mock_peer_with_own_key( storage: Arc, inbound_key: &str, ) -> (String, tokio::task::JoinHandle<()>) { - let (channel_mgr, _rx) = ChannelManager::new(storage.clone(), registry.clone(), None) - .await - .unwrap(); + let (channel_mgr, _rx) = ChannelManager::new(registry.clone(), None).await.unwrap(); let channel_mgr = Arc::new(channel_mgr); let repo = Arc::new(RegistryRepository::new(registry)); let archiver = Arc::new(ChannelArchiverControl::new(channel_mgr)); @@ -937,9 +929,7 @@ async fn test_cluster_proxy_authenticates_with_per_peer_key() { start_mock_peer_with_own_key(peer_reg.clone(), peer_storage, PEER_SPECIFIC_KEY).await; let (local_reg, local_storage, _local_dir) = new_registry_and_storage(); - let (channel_mgr, _rx) = ChannelManager::new(local_storage.clone(), local_reg.clone(), None) - .await - .unwrap(); + let (channel_mgr, _rx) = ChannelManager::new(local_reg.clone(), None).await.unwrap(); let channel_mgr = Arc::new(channel_mgr); let cluster_config = ClusterConfig { @@ -1033,9 +1023,7 @@ async fn test_different_peers_get_different_keys() { start_mock_peer_with_own_key(peer1_reg.clone(), peer1_storage, PEER1_KEY).await; let (local_reg, local_storage, _local_dir) = new_registry_and_storage(); - let (channel_mgr, _rx) = ChannelManager::new(local_storage.clone(), local_reg.clone(), None) - .await - .unwrap(); + let (channel_mgr, _rx) = ChannelManager::new(local_reg.clone(), None).await.unwrap(); let channel_mgr = Arc::new(channel_mgr); let cluster_config = ClusterConfig { diff --git a/tests/api_failover.rs b/tests/api_failover.rs index 23758a4..d49e897 100644 --- a/tests/api_failover.rs +++ b/tests/api_failover.rs @@ -151,9 +151,7 @@ async fn failover_merges_and_dedupes_peer_samples() { ); let (peer_url, _peer_task) = start_fake_peer(peer_body).await; - let (channel_mgr, _rx) = ChannelManager::new(storage.clone(), registry.clone(), None) - .await - .unwrap(); + let (channel_mgr, _rx) = ChannelManager::new(registry.clone(), None).await.unwrap(); let channel_mgr = Arc::new(channel_mgr); let repo = Arc::new(RegistryRepository::new(registry.clone())); let archiver = Arc::new(ChannelArchiverControl::new(channel_mgr)); @@ -237,9 +235,7 @@ async fn failover_tolerates_unreachable_peer() { .unwrap(); storage.flush_writes().await.unwrap(); - let (channel_mgr, _rx) = ChannelManager::new(storage.clone(), registry.clone(), None) - .await - .unwrap(); + let (channel_mgr, _rx) = ChannelManager::new(registry.clone(), None).await.unwrap(); let channel_mgr = Arc::new(channel_mgr); let repo = Arc::new(RegistryRepository::new(registry.clone())); let archiver = Arc::new(ChannelArchiverControl::new(channel_mgr)); diff --git a/tests/api_mgmt.rs b/tests/api_mgmt.rs index eba82cf..c66c759 100644 --- a/tests/api_mgmt.rs +++ b/tests/api_mgmt.rs @@ -22,9 +22,7 @@ async fn build_test_app() -> (axum::Router, tempfile::TempDir) { PartitionGranularity::Hour, )); let registry = Arc::new(PvRegistry::in_memory().unwrap()); - let (channel_mgr, _rx) = ChannelManager::new(storage.clone(), registry.clone(), None) - .await - .unwrap(); + let (channel_mgr, _rx) = ChannelManager::new(registry.clone(), None).await.unwrap(); let channel_mgr = Arc::new(channel_mgr); let repo = Arc::new(RegistryRepository::new(registry)); let archiver = Arc::new(ChannelArchiverControl::new(channel_mgr)); @@ -82,9 +80,7 @@ async fn build_test_app_with_pvs() -> (axum::Router, Arc, tempfile:: ) .unwrap(); - let (channel_mgr, _rx) = ChannelManager::new(storage.clone(), registry.clone(), None) - .await - .unwrap(); + let (channel_mgr, _rx) = ChannelManager::new(registry.clone(), None).await.unwrap(); let channel_mgr = Arc::new(channel_mgr); let repo = Arc::new(RegistryRepository::new(registry.clone())); let archiver = Arc::new(ChannelArchiverControl::new(channel_mgr)); @@ -509,9 +505,7 @@ async fn test_export_import_preserves_status() { PartitionGranularity::Hour, )); let registry2 = Arc::new(PvRegistry::in_memory().unwrap()); - let (channel_mgr2, _rx2) = ChannelManager::new(storage2.clone(), registry2.clone(), None) - .await - .unwrap(); + let (channel_mgr2, _rx2) = ChannelManager::new(registry2.clone(), None).await.unwrap(); let channel_mgr2 = Arc::new(channel_mgr2); let repo2 = Arc::new(RegistryRepository::new(registry2.clone())); let archiver2 = Arc::new(ChannelArchiverControl::new(channel_mgr2)); @@ -594,9 +588,7 @@ async fn build_test_app_with_auth() -> (axum::Router, tempfile::TempDir) { 1, ) .unwrap(); - let (channel_mgr, _rx) = ChannelManager::new(storage.clone(), registry.clone(), None) - .await - .unwrap(); + let (channel_mgr, _rx) = ChannelManager::new(registry.clone(), None).await.unwrap(); let channel_mgr = Arc::new(channel_mgr); let repo = Arc::new(RegistryRepository::new(registry)); let archiver = Arc::new(ChannelArchiverControl::new(channel_mgr)); @@ -740,9 +732,7 @@ async fn test_rate_limiter_blocks_excess_requests() { PartitionGranularity::Hour, )); let registry = Arc::new(PvRegistry::in_memory().unwrap()); - let (channel_mgr, _rx) = ChannelManager::new(storage.clone(), registry.clone(), None) - .await - .unwrap(); + let (channel_mgr, _rx) = ChannelManager::new(registry.clone(), None).await.unwrap(); let channel_mgr = Arc::new(channel_mgr); let repo = Arc::new(RegistryRepository::new(registry)); let archiver = Arc::new(ChannelArchiverControl::new(channel_mgr)); @@ -808,9 +798,7 @@ async fn test_rate_limiter_isolates_by_ip() { PartitionGranularity::Hour, )); let registry = Arc::new(PvRegistry::in_memory().unwrap()); - let (channel_mgr, _rx) = ChannelManager::new(storage.clone(), registry.clone(), None) - .await - .unwrap(); + let (channel_mgr, _rx) = ChannelManager::new(registry.clone(), None).await.unwrap(); let channel_mgr = Arc::new(channel_mgr); let repo = Arc::new(RegistryRepository::new(registry)); let archiver = Arc::new(ChannelArchiverControl::new(channel_mgr)); @@ -1314,18 +1302,10 @@ async fn test_alias_excluded_from_count_and_status_queries() { #[tokio::test] async fn test_restore_from_registry_skips_aliases() { use archiver_core::registry::{PvRegistry, PvStatus, SampleMode}; - use archiver_core::storage::partition::PartitionGranularity; - use archiver_core::storage::plainpb::PlainPbStoragePlugin; use archiver_core::types::ArchDbType; use archiver_engine::channel_manager::ChannelManager; use std::sync::Arc; - let dir = tempfile::tempdir().unwrap(); - let storage = Arc::new(PlainPbStoragePlugin::new( - "sts", - dir.path().to_path_buf(), - PartitionGranularity::Hour, - )); let registry = Arc::new(PvRegistry::in_memory().unwrap()); registry .register_pv("REAL:PV", ArchDbType::ScalarDouble, &SampleMode::Monitor, 1) @@ -1336,9 +1316,7 @@ async fn test_restore_from_registry_skips_aliases() { let alias_row = registry.get_pv("ALIAS:PV").unwrap().unwrap(); assert_eq!(alias_row.status, PvStatus::Alias); - let (mgr, _rx) = ChannelManager::new(storage.clone(), registry.clone(), None) - .await - .unwrap(); + let (mgr, _rx) = ChannelManager::new(registry.clone(), None).await.unwrap(); // restore_from_registry must NOT try to archive the alias. We can't // verify CA channel creation in a unit test (no IOC), but we can // verify pvs_by_status(Active) returns only REAL:PV. @@ -1366,13 +1344,10 @@ async fn test_consolidate_forwards_to_peer_for_remote_pv() { )); let registry = Arc::new(archiver_core::registry::PvRegistry::in_memory().unwrap()); // Local has nothing — REMOTE:PV is owned by a peer. - let (channel_mgr, _rx) = archiver_engine::channel_manager::ChannelManager::new( - storage.clone(), - registry.clone(), - None, - ) - .await - .unwrap(); + let (channel_mgr, _rx) = + archiver_engine::channel_manager::ChannelManager::new(registry.clone(), None) + .await + .unwrap(); let channel_mgr = Arc::new(channel_mgr); let repo = Arc::new(RegistryRepository::new(registry.clone())); let archiver = Arc::new(ChannelArchiverControl::new(channel_mgr)); @@ -1744,9 +1719,7 @@ async fn test_p2_change_type_for_pv_waits_for_the_etl_move_gate() { let executor = Arc::new(EtlExecutor::new(sts.clone(), mts.clone(), 3600, 5, 3)); let gate = executor.move_gate(); - let (channel_mgr, _rx) = ChannelManager::new(sts.clone(), registry.clone(), None) - .await - .unwrap(); + let (channel_mgr, _rx) = ChannelManager::new(registry.clone(), None).await.unwrap(); let repo = Arc::new(RegistryRepository::new(registry.clone())); let archiver = Arc::new(ChannelArchiverControl::new(Arc::new(channel_mgr))); let state = AppState { @@ -1901,9 +1874,7 @@ async fn test_d_get_data_at_time_stage_2_walkback() { } storage.flush_writes().await.unwrap(); - let (channel_mgr, _rx) = ChannelManager::new(storage.clone(), registry.clone(), None) - .await - .unwrap(); + let (channel_mgr, _rx) = ChannelManager::new(registry.clone(), None).await.unwrap(); let channel_mgr = Arc::new(channel_mgr); let repo = Arc::new(RegistryRepository::new(registry)); let archiver = Arc::new(ChannelArchiverControl::new(channel_mgr)); @@ -1959,18 +1930,8 @@ async fn test_d_update_archive_fields_concurrent_no_duplicate_tasks() { // // We can't observe field_tokens directly through the public API, but we // can at least verify the registry persists the last update correctly. - let cm = Arc::new({ - let dir = tempfile::tempdir().unwrap(); - let storage = Arc::new(PlainPbStoragePlugin::new( - "sts", - dir.path().to_path_buf(), - archiver_core::storage::partition::PartitionGranularity::Hour, - )); - let (cm, _rx) = ChannelManager::new(storage, registry.clone(), None) - .await - .unwrap(); - cm - }); + let (cm, _rx) = ChannelManager::new(registry.clone(), None).await.unwrap(); + let cm = Arc::new(cm); let mut handles = Vec::new(); for i in 0..16 { diff --git a/tests/api_retrieval.rs b/tests/api_retrieval.rs index b0fdc9b..4083fc6 100644 --- a/tests/api_retrieval.rs +++ b/tests/api_retrieval.rs @@ -49,9 +49,7 @@ async fn build_retrieval_app() -> (axum::Router, SystemTime, SystemTime, tempfil let data_start = base_time; let data_end = base_time + Duration::from_secs(19); - let (channel_mgr, _rx) = ChannelManager::new(storage.clone(), registry.clone(), None) - .await - .unwrap(); + let (channel_mgr, _rx) = ChannelManager::new(registry.clone(), None).await.unwrap(); let repo = Arc::new(RegistryRepository::new(registry)); let archiver = Arc::new(ChannelArchiverControl::new(Arc::new(channel_mgr))); let state = AppState { diff --git a/tests/save_path_live.rs b/tests/save_path_live.rs index 8e17deb..c452222 100644 --- a/tests/save_path_live.rs +++ b/tests/save_path_live.rs @@ -106,9 +106,7 @@ impl Live { std::env::set_var("EPICS_PVA_ADDR_LIST", ""); } - let (mgr, rx) = ChannelManager::new(storage.clone(), registry.clone(), None) - .await - .unwrap(); + let (mgr, rx) = ChannelManager::new(registry.clone(), None).await.unwrap(); let mgr = Arc::new(mgr); let (shutdown_tx, shutdown_rx) = tokio::sync::watch::channel(false); let pool = tokio::spawn(run_sharded_write_pool( diff --git a/tests/save_path_live_ca.rs b/tests/save_path_live_ca.rs index 44f75bc..010618c 100644 --- a/tests/save_path_live_ca.rs +++ b/tests/save_path_live_ca.rs @@ -101,9 +101,7 @@ impl Stack { std::env::set_var("EPICS_CA_SERVER_PORT", port.to_string()); } - let (mgr, rx) = ChannelManager::new(storage.clone(), registry.clone(), None) - .await - .unwrap(); + let (mgr, rx) = ChannelManager::new(registry.clone(), None).await.unwrap(); let mgr = Arc::new(mgr); let (pool_shutdown, _) = tokio::sync::watch::channel(false); @@ -360,9 +358,7 @@ async fn restart_does_not_re_store_the_committed_value() { tokio::time::timeout(Duration::from_secs(10), s.mgr.shutdown()) .await .expect("producers stop"); - let (mgr2, rx2) = ChannelManager::new(s.storage.clone(), s.registry.clone(), None) - .await - .unwrap(); + let (mgr2, rx2) = ChannelManager::new(s.registry.clone(), None).await.unwrap(); let (pool2_shutdown, shutdown_rx2) = tokio::sync::watch::channel(false); let pool2 = tokio::spawn(run_sharded_write_pool( s.storage.clone(), From 13859ab38fe11dab883f19b3e531b8f932ccfd28 Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 13:49:10 +0900 Subject: [PATCH 45/55] engine: report paused PVs from all_pv_counters and pv_counters Both read the active-channel map, so a pause removed the PV from the drop, lost-connection and event-rate reports and from pvStatusAction, which is where an operator looks for why it was paused. The counters map already lives for the process (task_counters) and is cleared by destroy_pv, which is the Java lifetime of a stopped channel's stats. --- crates/archiver-engine/src/channel_manager.rs | 25 +++++++++-------- tests/save_path_live_ca.rs | 27 +++++++++++++++++++ 2 files changed, 39 insertions(+), 13 deletions(-) diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index a1a5a21..43f3e0f 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -1342,24 +1342,23 @@ impl ChannelManager { .collect() } - /// Snapshot the diagnostic counters for one PV. Returns None if the - /// PV isn't actively archived. The returned Arc is the live counter - /// — callers read with `Ordering::Relaxed`. + /// The live counters of one PV: any PV this process has started + /// since boot, paused included, until `destroy_pv` removes it. The + /// returned Arc is the live counter — callers read with + /// `Ordering::Relaxed`. pub fn pv_counters(&self, pv_name: &str) -> Option> { - self.channels.get(pv_name).map(|h| h.counters.clone()) + self.counters.get(pv_name).map(|c| c.value().clone()) } - /// Snapshot every active PV's counters. Returns `(pv_name, - /// PvCountersSnapshot)` so callers don't have to handle Arc. + /// Snapshot the counters of every PV this process has started, + /// paused included: Java keeps a stopped channel in the engine's + /// channel list, so the drop and event-rate reports still cover it. + /// Returns `(pv_name, PvCountersSnapshot)` so callers don't have to + /// handle Arc. pub fn all_pv_counters(&self) -> Vec<(String, PvCountersSnapshot)> { - self.channels + self.counters .iter() - .map(|e| { - ( - e.key().clone(), - PvCountersSnapshot::from(&*e.value().counters), - ) - }) + .map(|e| (e.key().clone(), PvCountersSnapshot::from(&**e.value()))) .collect() } diff --git a/tests/save_path_live_ca.rs b/tests/save_path_live_ca.rs index 010618c..eb609d8 100644 --- a/tests/save_path_live_ca.rs +++ b/tests/save_path_live_ca.rs @@ -535,3 +535,30 @@ async fn pause_returns_promptly_while_producers_wait_to_connect() { s.finish().await; } + +/// A paused PV keeps its process-lifetime counters in the reports: Java +/// keeps the stopped channel in the engine's channel list, so the drop +/// and event-rate reports still show it. `destroy_pv` is what removes it. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn paused_pv_stays_in_the_counters_reports() { + let s = Stack::start().await; + let t0 = SystemTime::now(); + s.archive().await; + let first = vec![1.0, 2.0, 3.0]; + s.ch.put(&EpicsValue::DoubleArray(first.clone())) + .await + .expect("caput"); + wait_on_disk(&s, t0, &first).await; + let before = counters(&s.mgr, PV); + + s.mgr.pause_pv(PV).await.unwrap(); + let after = counters(&s.mgr, PV); + assert_eq!(after.events_received, before.events_received, "{after:?}"); + assert!(s.mgr.pv_counters(PV).is_some()); + + s.mgr.destroy_pv(PV).await.unwrap(); + assert!(s.mgr.pv_counters(PV).is_none()); + assert!(s.mgr.all_pv_counters().iter().all(|(n, _)| n != PV)); + + s.finish().await; +} From bd644d125b5880c4fdf28e94eea7c3519e0cf063 Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 13:50:04 +0900 Subject: [PATCH 46/55] engine: race pvmonitor_handle with the cancel token in monitor_loop_pva It was the one producer await left outside the cancel select on the belief that dropping it mid-subscribe could leak a server-side monitor. epics-pva-rs 0.28.0 shows otherwise: the call awaits only the channel lookup, and op_monitor_handle spawns the subscription task synchronously before returning the handle, so nothing outlives a drop. --- crates/archiver-engine/src/channel_manager.rs | 13 +++++++++---- 1 file changed, 9 insertions(+), 4 deletions(-) diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index 43f3e0f..c5f455e 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -1861,10 +1861,15 @@ async fn monitor_loop_pva( } } }; - // Not raced against cancellation: dropping this future - // mid-subscribe could leave the server-side monitor running - // with no handle to stop it. The client timeout bounds it. - let handle = match pva_client.pvmonitor_handle(&pv_name, cb, on_conn).await { + // Safe to drop mid-call: its only await is the channel + // lookup. The subscription task, and with it the server-side + // monitor, is spawned synchronously right before the handle + // is returned, so a cancelled call has no monitor to leak. + let subscribed = tokio::select! { + _ = cancel_token.cancelled() => return, + r = pva_client.pvmonitor_handle(&pv_name, cb, on_conn) => r, + }; + let handle = match subscribed { Ok(h) => h, Err(e) => { counters From 689270b2a3115443254d039e2904902ae44ff4c1 Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 14:29:48 +0900 Subject: [PATCH 47/55] engine: run the stop transition in a task that owns the op lock stop_tasks removed the channel and cancelled the producers, then awaited the quiesce before the caller set the registry status, so a caller that hit QUIESCE_TIMEOUT or was dropped mid-wait (the HTTP client gone) left the registry Active for a PV that no longer archived, silently in the second case. stop_and_finalize spawns the wait and the status commit as a tracked task holding the owned lock guard; the caller only waits on that task, bounded, and the status lands when the PV's queue drains. --- crates/archiver-engine/src/channel_manager.rs | 149 ++++++++++-------- tests/save_path_live_ca.rs | 43 +++++ 2 files changed, 130 insertions(+), 62 deletions(-) diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index c5f455e..4325ab2 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -29,11 +29,14 @@ const CA_CONNECT_TIMEOUT: Duration = Duration::from_secs(10); const CA_RECONNECT_TIMEOUT: Duration = Duration::from_secs(30); /// Delay before retrying a failed CA subscription. const CA_RETRY_DELAY: Duration = Duration::from_secs(5); -/// Longest `ChannelManager::stop_tasks` waits for a PV's tasks to exit -/// and its queued samples to reach the store. Every producer wait is -/// shorter (a CA `get` is bounded at 30 s, the PVA client at 5 s) and -/// so is `WriteLoopConfig::append_timeout`, so only a storage append -/// that hangs past that can exhaust it. +/// Longest a caller of `ChannelManager::stop_and_finalize` waits for a +/// PV's stop transition before it gets an error. The transition itself +/// is not bounded: it sets the registry status when the PV's tasks +/// have exited and its queued samples reached the store, however long +/// a stalled store takes. Every producer wait is shorter (a CA `get` +/// is bounded at 30 s, the PVA client at 5 s) and so is +/// `WriteLoopConfig::append_timeout`, so only a storage append that +/// hangs past that can exhaust it. const QUIESCE_TIMEOUT: Duration = Duration::from_secs(60); /// Hard floor on accepted timestamps. Mirrors Java's `PAST_CUTOFF_TIMESTAMP` @@ -392,7 +395,7 @@ struct PvHandle { counters: Arc, /// Every task of this PV generation: the producer loop, the PVA /// overflow drainer and refreshers, and the extra-field monitors. - /// `stop_tasks` waits on it so a stopped PV has no task left that + /// `stop_and_finalize` waits on it so a stopped PV has no task left that /// could still queue a sample. tasks: TaskTracker, } @@ -455,7 +458,8 @@ pub struct ChannelManager { /// Per-PV counters for the life of the process, so a pause and /// resume continue the PV's statistics and ordering high-water /// (see [`Self::task_counters`]). `destroy_pv` removes the entry. - counters: DashMap>, + /// Shared with the stop transitions, which run in their own tasks. + counters: Arc>>, } /// A sample ready to be written to storage. Built only through @@ -553,7 +557,7 @@ impl ChannelManager { server_ioc_drift_secs, tasks: TaskTracker::new(), lifecycle: tokio::sync::RwLock::new(false), - counters: DashMap::new(), + counters: Arc::new(DashMap::new()), }; Ok((mgr, rx)) @@ -616,7 +620,7 @@ impl ChannelManager { } /// Spawn a task of one PV generation: tracked by `pv_tasks` so - /// [`Self::stop_tasks`] can wait for it, and by `self.tasks` for + /// [`Self::stop_and_finalize`] can wait for it, and by `self.tasks` for /// [`Self::shutdown`]. fn spawn_pv_task( &self, @@ -1177,17 +1181,29 @@ impl ChannelManager { Ok(()) } - /// Stop the PV's tasks and wait until nothing of it is in flight: - /// the tasks have exited and every sample they queued has been - /// appended or dropped. On return the store holds all the samples - /// this task generation will ever produce, so the caller may set - /// the registry status and a later rename, type change, reassign - /// or delete acts on the whole data set. Fails, with the status - /// untouched, if the wait exceeds [`QUIESCE_TIMEOUT`]; the tasks - /// stay cancelled and a retry waits again. - async fn stop_tasks(&self, pv_name: &str) -> anyhow::Result<()> { - let deadline = tokio::time::Instant::now() + QUIESCE_TIMEOUT; - if let Some((_key, handle)) = self.channels.remove(pv_name) { + /// Stop the PV's tasks and, once nothing of it is in flight, run + /// `finalize`: the tasks have exited and every sample they queued + /// has been appended or dropped, so the store holds all the + /// samples this task generation will ever produce and the registry + /// status `finalize` commits describes the whole data set for a + /// later rename, type change, reassign or delete. + /// + /// The transition is owned by a tracked task that holds the PV's + /// op lock until `finalize` ran, so a caller that gives up after + /// [`QUIESCE_TIMEOUT`] or is dropped mid-wait (the HTTP client went + /// away) cannot leave the registry saying Active for a PV whose + /// producers are stopped. The lock keeps resume and the other + /// transitions out until then; the caller gets an error after the + /// bound, and the status lands when the PV's queue drains. + async fn stop_and_finalize( + &self, + pv_name: &str, + finalize: impl FnOnce() -> anyhow::Result<()> + Send + 'static, + ) -> anyhow::Result<()> { + let guard = self.op_lock(pv_name).lock_owned().await; + // Nothing below awaits before the task owns the transition. + let handle = self.channels.remove(pv_name).map(|(_key, handle)| handle); + if let Some(handle) = &handle { let extra_count = handle.field_tokens.len() as f64; handle.cancel_token.cancel(); metrics::gauge!("archiver_pvs_active").decrement(1.0); @@ -1195,38 +1211,42 @@ impl ChannelManager { metrics::gauge!("archiver_extra_field_tasks").decrement(extra_count); } handle.tasks.close(); - if tokio::time::timeout_at(deadline, handle.tasks.wait()) - .await - .is_err() - { - anyhow::bail!("{pv_name}: archiving tasks did not stop within {QUIESCE_TIMEOUT:?}"); - } } - let Some(counters) = self.counters.get(pv_name).map(|c| c.value().clone()) else { - return Ok(()); - }; - if tokio::time::timeout_at(deadline, counters.wait_in_flight()) - .await - .is_err() - { - anyhow::bail!( - "{pv_name}: {} samples still in flight after {QUIESCE_TIMEOUT:?}", - counters.in_flight.load(Ordering::SeqCst) - ); + let counters = self.counters.get(pv_name).map(|c| c.value().clone()); + let transition = self.tasks.spawn(async move { + let _guard = guard; + if let Some(handle) = handle { + handle.tasks.wait().await; + } + if let Some(counters) = counters { + counters.wait_in_flight().await; + } + finalize() + }); + match tokio::time::timeout(QUIESCE_TIMEOUT, transition).await { + Ok(finished) => { + finished.map_err(|e| anyhow::anyhow!("{pv_name}: stop transition failed: {e}"))? + } + Err(_) => anyhow::bail!( + "{pv_name}: still quiescing after {QUIESCE_TIMEOUT:?}; \ + its status is set when the queue drains" + ), } - Ok(()) } /// Pause archiving for a PV. Returns once the PV has nothing in - /// flight (see [`Self::stop_tasks`]), so a `Paused` registry status - /// means the store holds every sample the PV produced. + /// flight and is Paused (see [`Self::stop_and_finalize`]), so a + /// `Paused` registry status means the store holds every sample the + /// PV produced. pub async fn pause_pv(&self, pv_name: &str) -> anyhow::Result<()> { - let lock = self.op_lock(pv_name); - let _g = lock.lock().await; - self.stop_tasks(pv_name).await?; - self.registry.set_status(pv_name, PvStatus::Paused)?; - info!(pv = pv_name, "Paused archiving"); - Ok(()) + let registry = self.registry.clone(); + let pv = pv_name.to_string(); + self.stop_and_finalize(pv_name, move || { + registry.set_status(&pv, PvStatus::Paused)?; + info!(pv = %pv, "Paused archiving"); + Ok(()) + }) + .await } /// Resume a paused PV. Only paused or error PVs can be resumed; @@ -1269,30 +1289,35 @@ impl ChannelManager { /// Stop archiving a PV without removing it from the registry. /// Sets the PV status to Inactive (data retained, monitoring stopped). pub async fn stop_pv(&self, pv_name: &str) -> anyhow::Result<()> { - let lock = self.op_lock(pv_name); - let _g = lock.lock().await; - self.stop_tasks(pv_name).await?; - self.registry.set_status(pv_name, PvStatus::Inactive)?; - info!(pv = pv_name, "Stopped archiving (inactive)"); - Ok(()) + let registry = self.registry.clone(); + let pv = pv_name.to_string(); + self.stop_and_finalize(pv_name, move || { + registry.set_status(&pv, PvStatus::Inactive)?; + info!(pv = %pv, "Stopped archiving (inactive)"); + Ok(()) + }) + .await } /// Remove a PV from archiving entirely. pub async fn destroy_pv(&self, pv_name: &str) -> anyhow::Result<()> { - let lock = self.op_lock(pv_name); - let _g = lock.lock().await; - self.stop_tasks(pv_name).await?; - // The next task under this name is a new PV: its high-water - // and statistics must not carry over. - self.counters.remove(pv_name); - self.registry.remove_pv(pv_name)?; + let registry = self.registry.clone(); + let counters = self.counters.clone(); + let pv = pv_name.to_string(); // Don't remove the op_locks entry here: a concurrent caller may have // already taken a clone of this Arc and be queued on it; removing // would let a fresh caller obtain a new mutex and the queued one // would race them. The map is bounded by the lifetime universe of // PV names, which is acceptable. - info!(pv = pv_name, "Destroyed archiving channel"); - Ok(()) + self.stop_and_finalize(pv_name, move || { + // The next task under this name is a new PV: its high-water + // and statistics must not carry over. + counters.remove(&pv); + registry.remove_pv(&pv)?; + info!(pv = %pv, "Destroyed archiving channel"); + Ok(()) + }) + .await } /// List all currently archived PV names (from registry). @@ -2918,7 +2943,7 @@ fn epics_value_to_field_string(val: &EpicsValue) -> String { impl ChannelManager { /// Spawn a long-running task that subscribes to `.` and updates /// `extras` with each event. Owned by `parent_token` so pause/destroy cleans - /// it up alongside the main PV, and tracked by `pv_tasks` so `stop_tasks` + /// it up alongside the main PV, and tracked by `pv_tasks` so `stop_and_finalize` /// waits for it. fn spawn_extra_field_monitor( &self, diff --git a/tests/save_path_live_ca.rs b/tests/save_path_live_ca.rs index eb609d8..f14ae2e 100644 --- a/tests/save_path_live_ca.rs +++ b/tests/save_path_live_ca.rs @@ -562,3 +562,46 @@ async fn paused_pv_stays_in_the_counters_reports() { s.finish().await; } + +/// A pause whose caller gives up (the HTTP client went away, or the +/// 60 s bound passed) still completes: the transition runs in a task +/// that owns the op lock, so the registry cannot stay Active for a PV +/// whose producers are stopped. The pool starts only after the caller +/// is gone; the queued sample lands and the status follows. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn abandoned_pause_still_completes() { + let mut s = Stack::start_without_pool().await; + let t0 = SystemTime::now(); + s.archive().await; + + // The caller gives up while the connect-time sample is queued. + let gave_up = tokio::time::timeout(Duration::from_millis(500), s.mgr.pause_pv(PV)).await; + assert!( + gave_up.is_err(), + "pause returned without a pool: {gave_up:?}" + ); + let rec = s.registry.get_pv(PV).unwrap().expect("registry row"); + assert_eq!(rec.status, PvStatus::Active); + + s.start_pool(); + let deadline = tokio::time::Instant::now() + Duration::from_secs(10); + loop { + let rec = s.registry.get_pv(PV).unwrap().expect("registry row"); + if rec.status == PvStatus::Paused { + break; + } + assert!( + tokio::time::Instant::now() < deadline, + "abandoned pause never completed: {rec:?}" + ); + tokio::time::sleep(Duration::from_millis(50)).await; + } + assert_eq!(stored_arrays(&s, t0).await, vec![vec![0.0; NELM as usize]]); + + // The transition released the lock: a resume goes through. + tokio::time::timeout(Duration::from_secs(10), s.mgr.resume_pv(PV)) + .await + .expect("resume is not blocked by the finished transition") + .expect("resume_pv"); + s.finish().await; +} From d4a4a5e3b23ae820793fdb6655cf38918d2dd6fa Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 14:35:15 +0900 Subject: [PATCH 48/55] engine: seed the ordering gate from the store's tail in seed_ordering_gate task_counters seeded a PV's first gate of the process from the registry's committed last_timestamp, which can sit on either side of the store: behind it after a crash between a flush and the owner's commit, so the connect-time redelivery was stored twice, and ahead of it after a power loss with fsync_on_flush off, so that redelivery, the one sample still recoverable, was dropped. The owning shard now reads the store's tail once, at the PV's first sample, off the restore path. PlainPB's read-side flush is per PV for that, so a seed no longer flushes every other writer, the cost 1360bee moved away from. --- .../archiver-core/src/storage/plainpb/mod.rs | 41 ++++- crates/archiver-engine/src/channel_manager.rs | 73 ++++++-- tests/save_path_live_ca.rs | 164 ++++++++++++++++++ tests/write_loop_failure.rs | 153 +++++++++++++++- 4 files changed, 410 insertions(+), 21 deletions(-) diff --git a/crates/archiver-core/src/storage/plainpb/mod.rs b/crates/archiver-core/src/storage/plainpb/mod.rs index 298b339..9e6eb89 100644 --- a/crates/archiver-core/src/storage/plainpb/mod.rs +++ b/crates/archiver-core/src/storage/plainpb/mod.rs @@ -449,7 +449,38 @@ impl PlainPbStoragePlugin { let cache = self.write_cache.lock().unwrap_or_else(|e| e.into_inner()); cache.iter().map(|(k, v)| (k.clone(), v.clone())).collect() }; + self.flush_slots(snapshot) + } + + /// Flush one PV's cached writer, if it has one. The read paths use + /// it so a reader of one PV sees that PV's buffered bytes without + /// paying for every other writer's flush: the write pool's + /// ordering-gate seed reads each PV's tail once as the PVs + /// connect after a restart. + fn flush_pv_writer(&self, pv: &str) -> FlushOutcome { + let slot = { + let cache = self.write_cache.lock().unwrap_or_else(|e| e.into_inner()); + cache.get(pv).map(|slot| (pv.to_string(), slot.clone())) + }; + self.flush_slots(slot.into_iter().collect()) + } + + /// Read-side flush of one PV (`get_data` and the last-event reads): + /// only real I/O errors surface. A deferred writer (an `append` + /// holds the slot) is not a failure — its bytes reach disk on the + /// next cycle and the reader sees everything already flushed. Loss + /// markers are not drained here; see `flush_writes`. + fn flush_pv_for_read(&self, pv: &str) -> anyhow::Result<()> { + let outcome = self.flush_pv_writer(pv); + if !outcome.failed.is_empty() { + anyhow::bail!("writer flush failed for pv={pv}"); + } + Ok(()) + } + /// The flush step behind `flush_dirty_writers` and + /// `flush_pv_writer`; see the former for the contract. + fn flush_slots(&self, snapshot: Vec<(String, Arc>)>) -> FlushOutcome { let mut failed = Vec::new(); let mut deferred = Vec::new(); let mut to_remove = Vec::new(); @@ -2177,8 +2208,8 @@ impl StoragePlugin for PlainPbStoragePlugin { start: SystemTime, end: SystemTime, ) -> anyhow::Result>> { - // Flush cached writes so readers see the latest data. - self.flush_writes().await?; + // Flush this PV's cached writes so the reader sees its latest data. + self.flush_pv_for_read(pv)?; let files = self.list_files_for_range(pv, start, end); @@ -2218,8 +2249,8 @@ impl StoragePlugin for PlainPbStoragePlugin { } async fn get_last_known_event(&self, pv: &str) -> anyhow::Result> { - // Flush cached writes so readers can see the latest data. - self.flush_writes().await?; + // Flush this PV's cached writes so the reader sees its latest data. + self.flush_pv_for_read(pv)?; let pb_files = list_pv_pb_files(&self.root_folder, pv)?; @@ -2237,7 +2268,7 @@ impl StoragePlugin for PlainPbStoragePlugin { pv: &str, target: SystemTime, ) -> anyhow::Result> { - self.flush_writes().await?; + self.flush_pv_for_read(pv)?; let pb_files = list_pv_pb_files(&self.root_folder, pv)?; diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index 4325ab2..5895848 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -1,5 +1,5 @@ use std::collections::VecDeque; -use std::sync::atomic::{AtomicI32, AtomicI64, AtomicU64, Ordering}; +use std::sync::atomic::{AtomicBool, AtomicI32, AtomicI64, AtomicU64, Ordering}; use std::sync::{Arc, Mutex}; use std::time::{Duration, Instant, SystemTime}; @@ -250,12 +250,17 @@ pub struct PvCounters { /// newer to be stored. The counters outlive the archiving task /// (`ChannelManager::task_counters`), so a resume continues the /// high-water and drops the current value the connect redelivers; - /// the first task in a process is seeded from the registry's - /// committed `last_timestamp`, and a deletePV + re-archive starts - /// at zero. Written only by the shard that owns the PV (the - /// dispatcher's consistent hash pins each PV to one shard); not - /// part of the reported snapshot. + /// the first task in a process starts from the registry's + /// committed `last_timestamp` until the shard replaces it with the + /// store's tail (`seed_ordering_gate`), and a deletePV + + /// re-archive starts over. Written only by the shard that owns the + /// PV (the dispatcher's consistent hash pins each PV to one + /// shard); not part of the reported snapshot. pub ordering_last_ts_nanos: AtomicU64, + /// `true` once the owning shard seeded `ordering_last_ts_nanos` + /// from the store (`seed_ordering_gate`): once per life of the + /// counters, so a resume does not read the store again. + gate_seeded: AtomicBool, /// Samples of this PV that exist anywhere between the producer and /// the end of their storage append: queued, parked, or being /// written. Maintained only by [`InFlight`], which every @@ -288,6 +293,7 @@ impl Default for PvCounters { shutdown_abandoned_drops: AtomicU64::new(0), flush_losses: AtomicU64::new(0), ordering_last_ts_nanos: AtomicU64::new(0), + gate_seeded: AtomicBool::new(false), in_flight: AtomicU64::new(0), settled: tokio::sync::Notify::new(), } @@ -600,10 +606,10 @@ impl ChannelManager { /// high-water, so the current value CA and PVA redeliver on /// connect, whose timestamp the previous task already stored, is /// dropped rather than stored a second time. The first task for a - /// PV in this process is seeded from the registry's committed - /// `last_timestamp`, which the flush owner advances only for bytes - /// on disk. The store is not consulted: that would be a partition - /// tail read per PV and per tier on the restore path. + /// PV in this process starts from the registry's committed + /// `last_timestamp`; the shard that owns the PV replaces it with + /// the store's tail at the PV's first sample (`seed_ordering_gate`), + /// off the restore path and one PV at a time. fn task_counters(&self, record: &PvRecord) -> Arc { self.counters .entry(record.pv_name.clone()) @@ -4621,6 +4627,46 @@ async fn shard_append_loop( info!(shard = shard_idx, "Shard append loop exited"); } +/// Seed a PV's ordering gate from the store before the shard's first +/// decision for it in this process. `task_counters` starts the gate +/// at the registry's committed `last_timestamp`, and the registry can +/// sit on either side of the store: behind it when the process died +/// between a flush and the owner's commit, so the current value the +/// IOC redelivers on connect would be stored a second time; ahead of +/// it when power was lost with `fsync_on_flush` off, so that +/// redelivery, the one sample of the lost window still recoverable, +/// would be dropped. A duplicate or a gap is measured against the +/// store, so the store has the final word. A read error keeps the +/// registry seed; it is logged once and not retried per sample. +async fn seed_ordering_gate( + shard_idx: usize, + storage: &Arc, + pv: &str, + c: &PvCounters, +) { + if c.gate_seeded.load(Ordering::Relaxed) { + return; + } + match storage.get_last_known_event(pv).await { + Ok(last) => { + let nanos = last.map_or(0, |s| unix_nanos(s.timestamp)); + c.ordering_last_ts_nanos.store(nanos, Ordering::Relaxed); + debug!( + shard = shard_idx, + pv, + last_nanos = nanos, + "Seeded the ordering gate from the store" + ); + } + Err(e) => warn!( + shard = shard_idx, + pv, + "Could not read the last stored event; the ordering gate keeps the registry seed: {e}" + ), + } + c.gate_seeded.store(true, Ordering::Relaxed); +} + /// Process one sample on the shard's hot path. Extracted so the /// steady-state and shutdown-drain branches can share the same /// "drop checks → spawn_blocking append → report on success" @@ -4637,11 +4683,12 @@ async fn shard_handle_sample( // Out-of-order timestamp drop; an equal timestamp is the same // event redelivered (CA and PVA resend the current value on every - // connect). The high-water lives in the archiving task's - // `PvCounters` (see `ordering_last_ts_nanos`), so it survives - // flush cycles but not the task: a sample without counters has no + // connect). The high-water lives in the PV's `PvCounters` (see + // `ordering_last_ts_nanos`), seeded from the store at the PV's + // first sample in this process; a sample without counters has no // task and no ordering state. if let Some(ref c) = pv_sample.counters { + seed_ordering_gate(shard_idx, storage, &pv_sample.pv_name, c).await; let prev = c.ordering_last_ts_nanos.load(Ordering::Relaxed); if prev != 0 && ts_nanos <= prev { c.timestamp_drops.fetch_add(1, Ordering::Relaxed); diff --git a/tests/save_path_live_ca.rs b/tests/save_path_live_ca.rs index f14ae2e..7b2b049 100644 --- a/tests/save_path_live_ca.rs +++ b/tests/save_path_live_ca.rs @@ -21,6 +21,7 @@ use archiver_engine::channel_manager::{ run_sharded_write_pool, }; use epics_rs::base::server::records::waveform::WaveformRecord; +use epics_rs::base::server::snapshot::DbrClass; use epics_rs::base::types::{DbFieldType, EpicsValue}; use epics_rs::ca::client::{CaChannel, CaClient}; use epics_rs::ca::server::CaServer; @@ -605,3 +606,166 @@ async fn abandoned_pause_still_completes() { .expect("resume_pv"); s.finish().await; } + +/// A second manager and write pool over `s`'s store, restored from +/// `registry`: the restart half of the restart tests. +async fn restart( + s: &Stack, + registry: Arc, +) -> ( + ChannelManager, + tokio::task::JoinHandle<()>, + tokio::sync::watch::Sender, +) { + let (mgr2, rx2) = ChannelManager::new(registry.clone(), None).await.unwrap(); + let (pool2_shutdown, shutdown_rx2) = tokio::sync::watch::channel(false); + let pool2 = tokio::spawn(run_sharded_write_pool( + s.storage.clone(), + registry, + rx2, + shutdown_rx2, + ShardedWritePoolConfig { + shards: 1, + per_shard_buffer: 1024, + write_loop: WriteLoopConfig { + flush_period: Duration::from_millis(300), + ..Default::default() + }, + }, + )); + assert_eq!(mgr2.restore_from_registry().await.unwrap(), 1); + let deadline = tokio::time::Instant::now() + Duration::from_secs(10); + while counters(&mgr2, PV).events_received == 0 { + assert!( + tokio::time::Instant::now() < deadline, + "restored monitor never delivered" + ); + tokio::time::sleep(Duration::from_millis(50)).await; + } + (mgr2, pool2, pool2_shutdown) +} + +async fn stop_restarted( + mgr2: ChannelManager, + pool2: tokio::task::JoinHandle<()>, + pool2_shutdown: tokio::sync::watch::Sender, +) { + mgr2.stop_pv(PV).await.unwrap(); + pool2_shutdown.send(true).unwrap(); + tokio::time::timeout(Duration::from_secs(10), pool2) + .await + .expect("second write pool exits on shutdown") + .unwrap(); +} + +/// The record's TIME once `value` is its processed current value: two +/// consecutive DBR_TIME gets agree on it. +async fn record_time_of(ch: &CaChannel, value: &[f64]) -> SystemTime { + let want = EpicsValue::DoubleArray(value.to_vec()); + let deadline = tokio::time::Instant::now() + Duration::from_secs(10); + let mut seen: Option = None; + loop { + let snap = ch + .get_with_metadata(DbrClass::Time) + .await + .expect("caget with TIME"); + let ts = SystemTime::from(snap.timestamp); + if snap.value == want && seen == Some(ts) { + return ts; + } + seen = (snap.value == want).then_some(ts); + assert!( + tokio::time::Instant::now() < deadline, + "put of {value:?} never processed" + ); + tokio::time::sleep(Duration::from_millis(50)).await; + } +} + +/// The registry's committed `last_timestamp` can be behind the store +/// (the process died between a flush and the owner's commit). The +/// restarted task's gate is seeded from the store's tail, so the +/// connect-time redelivery of the value on disk is dropped although +/// the registry never heard of it; a registry without the high-water +/// at all is the extreme of that lag. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn restart_with_a_registry_behind_the_store_does_not_re_store_the_value() { + let s = Stack::start().await; + let t0 = SystemTime::now(); + s.archive().await; + let first = vec![1.0, 2.0, 3.0]; + s.ch.put(&EpicsValue::DoubleArray(first.clone())) + .await + .expect("caput"); + wait_on_disk(&s, t0, &first).await; + tokio::time::timeout(Duration::from_secs(10), s.mgr.shutdown()) + .await + .expect("producers stop"); + + let rec = s.registry.get_pv(PV).unwrap().expect("registry row"); + let behind = Arc::new(PvRegistry::open(&s._dir.path().join("registry2.db")).unwrap()); + behind + .register_pv(PV, rec.dbr_type, &rec.sample_mode, rec.element_count) + .unwrap(); + assert_eq!(behind.get_pv(PV).unwrap().unwrap().last_timestamp, None); + + let (mgr2, pool2, pool2_shutdown) = restart(&s, behind).await; + let second = vec![4.0, 5.0, 6.0]; + s.ch.put(&EpicsValue::DoubleArray(second.clone())) + .await + .expect("caput"); + let got = wait_on_disk(&s, t0, &second).await; + let copies = got.iter().filter(|g| **g == first).count(); + assert_eq!( + copies, 1, + "the redelivered value on disk was stored again: {got:?}" + ); + let c = counters(&mgr2, PV); + assert_eq!(c.timestamp_drops, 1, "{c:?}"); + assert_eq!(c.events_stored, 1, "{c:?}"); + + stop_restarted(mgr2, pool2, pool2_shutdown).await; + s.finish().await; +} + +/// The registry can also be ahead of the store: with `fsync_on_flush` +/// off, a power loss discards page-cache bytes the flush owner had +/// already committed. The IOC redelivers its current value on connect, +/// the one sample of the lost window still recoverable; seeded from +/// the store's tail, the gate stores it instead of trusting the +/// registry's claim that it is already on disk. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn restart_with_a_registry_ahead_of_the_store_stores_the_current_value() { + let s = Stack::start().await; + let t0 = SystemTime::now(); + s.archive().await; + let first = vec![1.0, 2.0, 3.0]; + s.ch.put(&EpicsValue::DoubleArray(first.clone())) + .await + .expect("caput"); + wait_on_disk(&s, t0, &first).await; + tokio::time::timeout(Duration::from_secs(10), s.mgr.shutdown()) + .await + .expect("producers stop"); + + // The IOC moves on while nothing archives, and the registry + // claims the new value is on disk, as it does after losing the + // bytes. + let lost = vec![4.0, 5.0, 6.0]; + s.ch.put(&EpicsValue::DoubleArray(lost.clone())) + .await + .expect("caput"); + let lost_ts = record_time_of(&s.ch, &lost).await; + s.registry.update_last_timestamp(PV, lost_ts).unwrap(); + + let (mgr2, pool2, pool2_shutdown) = restart(&s, s.registry.clone()).await; + let got = wait_on_disk(&s, t0, &lost).await; + let copies = got.iter().filter(|g| **g == lost).count(); + assert_eq!(copies, 1, "{got:?}"); + let c = counters(&mgr2, PV); + assert_eq!(c.timestamp_drops, 0, "{c:?}"); + assert_eq!(c.events_stored, 1, "{c:?}"); + + stop_restarted(mgr2, pool2, pool2_shutdown).await; + s.finish().await; +} diff --git a/tests/write_loop_failure.rs b/tests/write_loop_failure.rs index 00ee1ab..2f16315 100644 --- a/tests/write_loop_failure.rs +++ b/tests/write_loop_failure.rs @@ -12,7 +12,7 @@ use std::collections::HashMap; use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; use std::sync::{Arc, Mutex}; -use std::time::{Duration, SystemTime}; +use std::time::{Duration, SystemTime, UNIX_EPOCH}; use archiver_core::registry::{PvRegistry, SampleMode}; use archiver_core::storage::partition::PartitionGranularity; @@ -73,6 +73,9 @@ struct InjectingStorage { /// regression test assert the owner drains markers ONLY on the /// observed-success path (never on a flush timeout/panic). take_loss_calls: AtomicUsize, + /// What `get_last_known_event` reports per PV: the store's tail + /// the shard seeds its ordering gate from. + tail: Mutex>, } impl InjectingStorage { @@ -90,9 +93,14 @@ impl InjectingStorage { flush_ingest_calls: AtomicUsize::new(0), loss_queue: Mutex::new(Vec::new()), take_loss_calls: AtomicUsize::new(0), + tail: Mutex::new(HashMap::new()), }) } + fn set_tail(&self, pv: &str, sample: ArchiverSample) { + self.tail.lock().unwrap().insert(pv.to_string(), sample); + } + /// Push a dirty-write loss marker, simulating a loss recorded /// outside a flush pass (LRU eviction, ghost-file, etc.). fn push_loss(&self, pv: &str) { @@ -227,8 +235,8 @@ impl StoragePlugin for InjectingStorage { Ok(Vec::new()) } - async fn get_last_known_event(&self, _pv: &str) -> anyhow::Result> { - Ok(None) + async fn get_last_known_event(&self, pv: &str) -> anyhow::Result> { + Ok(self.tail.lock().unwrap().get(pv).cloned()) } async fn flush_writes(&self) -> anyhow::Result<()> { @@ -2057,3 +2065,142 @@ async fn equal_timestamp_sample_is_dropped_as_a_redelivery() { shutdown(sd_tx, join).await; } + +fn nanos(t: SystemTime) -> u64 { + t.duration_since(UNIX_EPOCH).unwrap().as_nanos() as u64 +} + +/// Registry seed, store tail, samples to send: returns the stored +/// timestamps and the drop count once `expect_stored` samples landed. +async fn seed_case( + registry_seed: Option, + tail: Option, + send: &[(SystemTime, f64)], + expect_stored: u64, +) -> (Vec, u64) { + let storage = InjectingStorage::new(); + let registry = Arc::new(PvRegistry::in_memory().unwrap()); + registry + .register_pv("A", ArchDbType::ScalarDouble, &SampleMode::Monitor, 1) + .unwrap(); + let counters = Arc::new(PvCounters::default()); + if let Some(seed) = registry_seed { + counters + .ordering_last_ts_nanos + .store(nanos(seed), Ordering::Relaxed); + } + if let Some(tail) = tail { + storage.set_tail("A", tail); + } + let (tx, sd_tx, join) = spawn_loop(storage.clone(), registry, fast_cfg()); + for (t, v) in send { + tx.send(pv_sample("A", *t, *v, &counters)).await.unwrap(); + } + let deadline = tokio::time::Instant::now() + Duration::from_secs(5); + while counters.events_stored.load(Ordering::Relaxed) < expect_stored { + assert!( + tokio::time::Instant::now() < deadline, + "samples never stored: {counters:?}" + ); + tokio::time::sleep(Duration::from_millis(20)).await; + } + tokio::time::sleep(Duration::from_millis(100)).await; + let stored = storage + .appends_snapshot() + .into_iter() + .map(|r| r.timestamp) + .collect(); + let drops = counters.timestamp_drops.load(Ordering::Relaxed); + shutdown(sd_tx, join).await; + (stored, drops) +} + +/// The shard seeds the ordering gate from the store's tail at the +/// PV's first sample in the process. A registry seed behind the store +/// (the process died between a flush and the owner's commit) must not +/// let the connect-time redelivery of the tail be stored again. +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn gate_seeds_from_a_store_tail_ahead_of_the_registry() { + let (stored, drops) = seed_case( + Some(ts(999)), + Some(sample_at(ts(1000), 1.0)), + &[(ts(1000), 1.0), (ts(1001), 2.0)], + 1, + ) + .await; + assert_eq!(stored, vec![ts(1001)]); + assert_eq!(drops, 1); +} + +/// A registry seed ahead of the store (power loss with +/// `fsync_on_flush` off): the redelivered current value is newer than +/// anything on disk and must be stored, although the registry claims +/// it already is. +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn gate_seeds_from_a_store_tail_behind_the_registry() { + let (stored, drops) = seed_case( + Some(ts(1001)), + Some(sample_at(ts(1000), 1.0)), + &[(ts(1001), 2.0)], + 1, + ) + .await; + assert_eq!(stored, vec![ts(1001)]); + assert_eq!(drops, 0); +} + +/// A store with nothing for the PV outranks a registry seed: the +/// first sample is stored whatever the registry claims. +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn gate_seeds_to_zero_from_an_empty_store() { + let (stored, drops) = seed_case(Some(ts(1001)), None, &[(ts(1000), 1.0)], 1).await; + assert_eq!(stored, vec![ts(1000)]); + assert_eq!(drops, 0); +} + +/// The store is read once per life of the counters: after the seed +/// the gate follows the samples the shard stored, not the store. +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn gate_is_seeded_once() { + let storage = InjectingStorage::new(); + let registry = Arc::new(PvRegistry::in_memory().unwrap()); + registry + .register_pv("A", ArchDbType::ScalarDouble, &SampleMode::Monitor, 1) + .unwrap(); + let counters = Arc::new(PvCounters::default()); + storage.set_tail("A", sample_at(ts(1000), 1.0)); + let (tx, sd_tx, join) = spawn_loop(storage.clone(), registry, fast_cfg()); + + tx.send(pv_sample("A", ts(1001), 2.0, &counters)) + .await + .unwrap(); + let deadline = tokio::time::Instant::now() + Duration::from_secs(5); + while counters.events_stored.load(Ordering::Relaxed) < 1 { + assert!( + tokio::time::Instant::now() < deadline, + "sample never stored" + ); + tokio::time::sleep(Duration::from_millis(20)).await; + } + // A tail newer than the next sample would drop it if re-read. + storage.set_tail("A", sample_at(ts(1005), 5.0)); + tx.send(pv_sample("A", ts(1002), 3.0, &counters)) + .await + .unwrap(); + let deadline = tokio::time::Instant::now() + Duration::from_secs(5); + while counters.events_stored.load(Ordering::Relaxed) < 2 { + assert!( + tokio::time::Instant::now() < deadline, + "second sample never stored: {counters:?}" + ); + tokio::time::sleep(Duration::from_millis(20)).await; + } + let stored: Vec = storage + .appends_snapshot() + .into_iter() + .map(|r| r.timestamp) + .collect(); + assert_eq!(stored, vec![ts(1001), ts(1002)]); + assert_eq!(counters.timestamp_drops.load(Ordering::Relaxed), 0); + shutdown(sd_tx, join).await; +} From 34224073b2394756768ecc55c20667b5cb355fbf Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 14:45:53 +0900 Subject: [PATCH 49/55] engine: isolate the seed_ordering_gate store read like the append The tail read ran inline on the shard's runtime worker with no bound, while the append is on the blocking pool under append_timeout so a stalled store parks a blocking thread. The read now takes the same route and bound; on timeout the registry seed stands and the PV counts as seeded, so the shard does not block on the store again per sample. --- crates/archiver-engine/src/channel_manager.rs | 35 +++++++++-- tests/write_loop_failure.rs | 58 +++++++++++++++++++ 2 files changed, 87 insertions(+), 6 deletions(-) diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index 5895848..a6abf7f 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -4636,19 +4636,32 @@ async fn shard_append_loop( /// it when power was lost with `fsync_on_flush` off, so that /// redelivery, the one sample of the lost window still recoverable, /// would be dropped. A duplicate or a gap is measured against the -/// store, so the store has the final word. A read error keeps the -/// registry seed; it is logged once and not retried per sample. +/// store, so the store has the final word. +/// +/// The read is isolated like the append: it runs on the blocking pool +/// and the shard waits at most `read_timeout` for it, so a store that +/// stalls parks a blocking thread, not a runtime worker. A read that +/// fails or times out keeps the registry seed; it is logged once, and +/// the PV counts as seeded so its later samples do not wait on the +/// store again. async fn seed_ordering_gate( shard_idx: usize, storage: &Arc, pv: &str, c: &PvCounters, + read_timeout: Duration, ) { if c.gate_seeded.load(Ordering::Relaxed) { return; } - match storage.get_last_known_event(pv).await { - Ok(last) => { + let storage_for_task = storage.clone(); + let pv_for_task = pv.to_string(); + let read = tokio::task::spawn_blocking(move || { + tokio::runtime::Handle::current() + .block_on(storage_for_task.get_last_known_event(&pv_for_task)) + }); + match tokio::time::timeout(read_timeout, read).await { + Ok(Ok(Ok(last))) => { let nanos = last.map_or(0, |s| unix_nanos(s.timestamp)); c.ordering_last_ts_nanos.store(nanos, Ordering::Relaxed); debug!( @@ -4658,11 +4671,21 @@ async fn seed_ordering_gate( "Seeded the ordering gate from the store" ); } - Err(e) => warn!( + Ok(Ok(Err(e))) => warn!( shard = shard_idx, pv, "Could not read the last stored event; the ordering gate keeps the registry seed: {e}" ), + Ok(Err(e)) => warn!( + shard = shard_idx, + pv, "The last stored event read failed; the ordering gate keeps the registry seed: {e}" + ), + Err(_) => warn!( + shard = shard_idx, + pv, + ?read_timeout, + "The last stored event read did not return in time; the ordering gate keeps the registry seed" + ), } c.gate_seeded.store(true, Ordering::Relaxed); } @@ -4688,7 +4711,7 @@ async fn shard_handle_sample( // first sample in this process; a sample without counters has no // task and no ordering state. if let Some(ref c) = pv_sample.counters { - seed_ordering_gate(shard_idx, storage, &pv_sample.pv_name, c).await; + seed_ordering_gate(shard_idx, storage, &pv_sample.pv_name, c, append_timeout).await; let prev = c.ordering_last_ts_nanos.load(Ordering::Relaxed); if prev != 0 && ts_nanos <= prev { c.timestamp_drops.fetch_add(1, Ordering::Relaxed); diff --git a/tests/write_loop_failure.rs b/tests/write_loop_failure.rs index 2f16315..aba3b09 100644 --- a/tests/write_loop_failure.rs +++ b/tests/write_loop_failure.rs @@ -76,6 +76,9 @@ struct InjectingStorage { /// What `get_last_known_event` reports per PV: the store's tail /// the shard seeds its ordering gate from. tail: Mutex>, + /// When set, `get_last_known_event` blocks its thread for this + /// long first, the way a stalled file read would. + tail_hang_for: Mutex>, } impl InjectingStorage { @@ -94,9 +97,14 @@ impl InjectingStorage { loss_queue: Mutex::new(Vec::new()), take_loss_calls: AtomicUsize::new(0), tail: Mutex::new(HashMap::new()), + tail_hang_for: Mutex::new(None), }) } + fn set_tail_hang(&self, hang: Duration) { + *self.tail_hang_for.lock().unwrap() = Some(hang); + } + fn set_tail(&self, pv: &str, sample: ArchiverSample) { self.tail.lock().unwrap().insert(pv.to_string(), sample); } @@ -236,6 +244,10 @@ impl StoragePlugin for InjectingStorage { } async fn get_last_known_event(&self, pv: &str) -> anyhow::Result> { + let hang = *self.tail_hang_for.lock().unwrap(); + if let Some(hang) = hang { + std::thread::sleep(hang); + } Ok(self.tail.lock().unwrap().get(pv).cloned()) } @@ -2204,3 +2216,49 @@ async fn gate_is_seeded_once() { assert_eq!(counters.timestamp_drops.load(Ordering::Relaxed), 0); shutdown(sd_tx, join).await; } + +/// The seed read is isolated like the append: a store that stalls on +/// the tail read holds the shard for `append_timeout`, not for the +/// stall, and the registry seed stands. Counted as seeded, the PV's +/// later samples do not wait on the store again. +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn gate_seed_read_is_bounded_by_append_timeout() { + let storage = InjectingStorage::new(); + let registry = Arc::new(PvRegistry::in_memory().unwrap()); + registry + .register_pv("A", ArchDbType::ScalarDouble, &SampleMode::Monitor, 1) + .unwrap(); + let counters = Arc::new(PvCounters::default()); + counters + .ordering_last_ts_nanos + .store(nanos(ts(999)), Ordering::Relaxed); + storage.set_tail("A", sample_at(ts(1000), 1.0)); + storage.set_tail_hang(Duration::from_secs(3)); + let (tx, sd_tx, join) = spawn_loop(storage.clone(), registry, fast_cfg()); + + for (t, v, expect) in [(ts(1000), 1.0, 1), (ts(1001), 2.0, 2)] { + let started = tokio::time::Instant::now(); + tx.send(pv_sample("A", t, v, &counters)).await.unwrap(); + let deadline = started + Duration::from_secs(8); + while counters.events_stored.load(Ordering::Relaxed) < expect { + assert!( + tokio::time::Instant::now() < deadline, + "sample {expect} never stored: {counters:?}" + ); + tokio::time::sleep(Duration::from_millis(20)).await; + } + let took = started.elapsed(); + assert!( + took < Duration::from_millis(1500), + "sample {expect} waited {took:?} on the stalled tail read" + ); + } + let stored: Vec = storage + .appends_snapshot() + .into_iter() + .map(|r| r.timestamp) + .collect(); + assert_eq!(stored, vec![ts(1000), ts(1001)]); + assert_eq!(counters.timestamp_drops.load(Ordering::Relaxed), 0); + shutdown(sd_tx, join).await; +} From 58abdbf64c86ea00c267c59be122f2bcd9fe4a2f Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 14:46:41 +0900 Subject: [PATCH 50/55] engine: bound the op-lock wait in lock_op by QUIESCE_TIMEOUT archive_pv, resume_pv and stop_and_finalize took the per-PV lock with no bound, so while a stop transition drained a stalled store the next operation on that PV queued for the whole stall, although the first caller had been given an error after 60 s. The same bound now applies to the lock wait, with an error naming the operation in progress. --- crates/archiver-engine/src/channel_manager.rs | 29 ++++++++++++++----- 1 file changed, 21 insertions(+), 8 deletions(-) diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index a6abf7f..1f6009c 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -569,9 +569,9 @@ impl ChannelManager { Ok((mgr, rx)) } - /// Get-or-insert the per-PV operation mutex. The returned `Arc` - /// is what callers should `.lock().await` on; holding the entry guard - /// (via `entry().or_insert_with`) across the await would deadlock the + /// Get-or-insert the per-PV operation mutex, taken through + /// [`Self::lock_op`]; holding the entry guard (via + /// `entry().or_insert_with`) across the await would deadlock the /// DashMap shard. fn op_lock(&self, pv_name: &str) -> Arc> { if let Some(existing) = self.op_locks.get(pv_name) { @@ -583,6 +583,21 @@ impl ChannelManager { .clone() } + /// Take the PV's op lock, waiting at most [`QUIESCE_TIMEOUT`]. A stop + /// transition draining a stalled store holds the lock for the whole + /// stall (see [`Self::stop_and_finalize`]); a start, resume or + /// further transition queued behind it gets an error after the + /// bound instead of waiting the stall out. + async fn lock_op(&self, pv_name: &str) -> anyhow::Result> { + match tokio::time::timeout(QUIESCE_TIMEOUT, self.op_lock(pv_name).lock_owned()).await { + Ok(guard) => Ok(guard), + Err(_) => anyhow::bail!( + "{pv_name}: another operation on this PV is still in progress \ + after {QUIESCE_TIMEOUT:?}" + ), + } + } + /// Stop every sample producer and wait for them to exit. The binary /// calls this before the write pool is told to drain, so the queue /// tail the pool moves to disk is the whole tail: a producer still @@ -684,8 +699,7 @@ impl ChannelManager { protocol: Protocol, ) -> anyhow::Result<()> { // Serialise with pause/resume/stop/destroy on the same PV. - let lock = self.op_lock(pv_name); - let _g = lock.lock().await; + let _g = self.lock_op(pv_name).await?; if self.channels.contains_key(pv_name) { anyhow::bail!("PV {pv_name} is already being archived"); @@ -1206,7 +1220,7 @@ impl ChannelManager { pv_name: &str, finalize: impl FnOnce() -> anyhow::Result<()> + Send + 'static, ) -> anyhow::Result<()> { - let guard = self.op_lock(pv_name).lock_owned().await; + let guard = self.lock_op(pv_name).await?; // Nothing below awaits before the task owns the transition. let handle = self.channels.remove(pv_name).map(|(_key, handle)| handle); if let Some(handle) = &handle { @@ -1258,8 +1272,7 @@ impl ChannelManager { /// Resume a paused PV. Only paused or error PVs can be resumed; /// calling resume on an already-active PV is a no-op (returns Ok). pub async fn resume_pv(&self, pv_name: &str) -> anyhow::Result<()> { - let lock = self.op_lock(pv_name); - let _g = lock.lock().await; + let _g = self.lock_op(pv_name).await?; let record = self .registry From 67a4caba1e2ae56f86f10be2cc0e7750999772ed Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 15:36:12 +0900 Subject: [PATCH 51/55] build: bump epics-rs to 0.28.1 Its CA server takes the record lock for subscription snapshots, so an initial event no longer reports a racing put's value under the pre-put timestamp. Stack::archive keeps waiting for the connect-time event: the live tests expect the seed value stored first. --- Cargo.lock | 28 ++++++++++++++-------------- Cargo.toml | 2 +- tests/save_path_live_ca.rs | 9 ++++----- 3 files changed, 19 insertions(+), 20 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index f0bee42..3f3d5f0 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -844,9 +844,9 @@ dependencies = [ [[package]] name = "epics-base-rs" -version = "0.28.0" +version = "0.28.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ffc5fb8c868e1114e589680551fd5dcc48df339a51c042e92a5415285a83cb3e" +checksum = "827b6713066f08d70cca1d3ea7de40810f20672976d1cb643bdcccab61d1daf0" dependencies = [ "arc-swap", "async-trait", @@ -870,9 +870,9 @@ dependencies = [ [[package]] name = "epics-ca-rs" -version = "0.28.0" +version = "0.28.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "07d093efafb055a711069a2e2985004f29a55198167892e52410a1d053eeea51" +checksum = "320382f1975f480de77e0f35aea4c1447d5f9dccff49d0d3dcd00e6dd91855b4" dependencies = [ "arc-swap", "async-trait", @@ -895,9 +895,9 @@ dependencies = [ [[package]] name = "epics-libcom-rs" -version = "0.28.0" +version = "0.28.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6a5794b250cda69b1fe724af5826d37831f8c9bf2a333359025fc9bae47d4b2c" +checksum = "0630b3bee193aa67e5816ce6bcca08e30028f80bec3691c63e2efaf0b9bd1049" dependencies = [ "chrono", "epics-rtems-boot", @@ -912,9 +912,9 @@ dependencies = [ [[package]] name = "epics-macros-rs" -version = "0.28.0" +version = "0.28.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d4574f2e2c8c5317e6f0983276feccae4cc5d45535620f9d9bb42987c663e229" +checksum = "a935efdf4a6729fa82a4ce14ff1e087e12e57cc496de9a49f490c5515b691d46" dependencies = [ "proc-macro-crate", "proc-macro2", @@ -924,9 +924,9 @@ dependencies = [ [[package]] name = "epics-pva-rs" -version = "0.28.0" +version = "0.28.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "67defd7fb25c002a023aa1750617c10ace4eba91400c82c61b7632fcf0525352" +checksum = "62215f9bc6ac5340d7d3b1f2da44b0adec1c18cb7fa8b0493684eeea0fa10e75" dependencies = [ "bytes", "chrono", @@ -963,9 +963,9 @@ dependencies = [ [[package]] name = "epics-rs" -version = "0.28.0" +version = "0.28.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "17a8d56c5d94ee6c8cf7c7e206427244a485638809c870f8a5007dd8c896805c" +checksum = "80e1a275a96929ceacbf38b4201409df72581382cd390606e0eb0bb7bcd792b6" dependencies = [ "epics-base-rs", "epics-ca-rs", @@ -974,9 +974,9 @@ dependencies = [ [[package]] name = "epics-rtems-boot" -version = "0.28.0" +version = "0.28.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ac25b14a69d693209a239812b356096de7c8051ec87af3f98da29ac21fdd7ff8" +checksum = "9d84dbcc9a4ede9dc56b1c4c1fd401d7765ab4dc5582cfa1268e98683d7c973b" dependencies = [ "cc", "libc", diff --git a/Cargo.toml b/Cargo.toml index be9aa9c..a49b098 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -74,7 +74,7 @@ tokio-util = { version = "0.7", features = ["rt"] } rusqlite = { version = "0.34", features = ["bundled"] } # EPICS Channel Access -epics-rs = { version = "0.28.0", features = ["pva"] } +epics-rs = { version = "0.28.1", features = ["pva"] } # URL encoding urlencoding = "2" diff --git a/tests/save_path_live_ca.rs b/tests/save_path_live_ca.rs index 7b2b049..12784be 100644 --- a/tests/save_path_live_ca.rs +++ b/tests/save_path_live_ca.rs @@ -147,11 +147,10 @@ impl Stack { ))); } - /// Start archiving `PV` and wait for the connect-time event, which - /// proves the monitor subscription is active. A put that lands - /// while the subscription is still being set up can be reported by - /// the in-process server as the new value under the previous - /// timestamp, followed by the properly stamped event. + /// Start archiving `PV` and wait for the connect-time event, so the + /// seed value is the first sample stored and the puts that follow + /// are ordered after it. A put that lands before the subscription's + /// initial snapshot would be the connect-time event instead. async fn archive(&self) { self.mgr .archive_pv(PV, &SampleMode::Monitor, Protocol::Ca) From 07f6b52e3b47745c20114a7a1ed344938dac440c Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 16:51:45 +0900 Subject: [PATCH 52/55] engine: return WriteQueueClosed from send_with_backpressure Every caller discarded the refused sample, so the 152-byte PvSample in the Err variant only tripped clippy::result_large_err on rustc 1.98. --- crates/archiver-engine/src/channel_manager.rs | 16 +++++++++------- 1 file changed, 9 insertions(+), 7 deletions(-) diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index 1f6009c..e152a4a 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -2097,9 +2097,8 @@ async fn scan_loop_pva( Some(elem_count), Some(counters.clone()), ); - if let Err(rejected) = send_with_backpressure(&tx, pv_sample).await { + if send_with_backpressure(&tx, pv_sample).await.is_err() { // Channel closed (write_loop down). Cooperative shutdown. - let _ = rejected; return; } } @@ -2448,8 +2447,7 @@ async fn monitor_loop( Some(element_count), Some(counters.clone()), ); - if let Err(pv_sample) = send_with_backpressure(&tx, pv_sample).await { - let _ = pv_sample; + if send_with_backpressure(&tx, pv_sample).await.is_err() { return; // Write loop shut down } } @@ -2513,19 +2511,23 @@ fn unix_secs(t: SystemTime) -> i64 { async fn send_with_backpressure( tx: &mpsc::Sender, pv_sample: PvSample, -) -> Result<(), PvSample> { +) -> Result<(), WriteQueueClosed> { match tx.try_send(pv_sample) { Ok(()) => Ok(()), Err(tokio::sync::mpsc::error::TrySendError::Full(pv_sample)) => { metrics::counter!("archiver_write_channel_backpressure_stalls_total").increment(1); // Backpressure: await until the writer drains space. The // sample is delivered, not dropped — so no drop counter. - tx.send(pv_sample).await.map_err(|e| e.0) + tx.send(pv_sample).await.map_err(|_| WriteQueueClosed) } - Err(tokio::sync::mpsc::error::TrySendError::Closed(pv_sample)) => Err(pv_sample), + Err(tokio::sync::mpsc::error::TrySendError::Closed(_)) => Err(WriteQueueClosed), } } +/// The write pool is gone; the sample it refused is dropped with the +/// producer, which has nothing left to deliver to. +struct WriteQueueClosed; + /// The newest PVA sample the write queue would not take. The PVA /// monitor callback runs on the pvAccess reactor task and cannot wait /// for queue space the way the CA producer does, so it parks the From 1d149650ea73f28cd4fc7b0c1ff98e2fa35a482d Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 16:55:01 +0900 Subject: [PATCH 53/55] core: add StoragePlugin::get_last_stored_event for the ordering gate seed TieredStorage::get_last_known_event honors SKIP__FOR_RETRIEVAL, so a gate seeded while a tier was routed around read an older tier's tail and stored the connect-time redelivery a second time. The stored view walks the tiers without the flags; seed_ordering_gate reads that. --- crates/archiver-core/src/storage/tiered.rs | 55 ++++++++++++++++ crates/archiver-core/src/storage/traits.rs | 9 +++ crates/archiver-engine/src/channel_manager.rs | 7 ++- tests/write_loop_failure.rs | 62 +++++++++++++++++-- 4 files changed, 126 insertions(+), 7 deletions(-) diff --git a/crates/archiver-core/src/storage/tiered.rs b/crates/archiver-core/src/storage/tiered.rs index 9d43e00..8eb96e5 100644 --- a/crates/archiver-core/src/storage/tiered.rs +++ b/crates/archiver-core/src/storage/tiered.rs @@ -179,6 +179,16 @@ impl StoragePlugin for TieredStorage { self.lts.get_last_known_event(pv).await } + async fn get_last_stored_event(&self, pv: &str) -> anyhow::Result> { + // The write path's view: newest tier first, no retrieval flags. + for tier in [&self.sts, &self.mts, &self.lts] { + if let Some(sample) = tier.get_last_known_event(pv).await? { + return Ok(Some(sample)); + } + } + Ok(None) + } + async fn get_last_event_before( &self, pv: &str, @@ -320,4 +330,49 @@ mod tests { assert!(tiered.mts.take_loss_markers().is_empty()); assert!(tiered.lts.take_loss_markers().is_empty()); } + + /// `SKIP__FOR_RETRIEVAL` hides a tier from readers only: the + /// stored view the write path seeds its ordering gate from still + /// reports the hidden tier's tail. + #[tokio::test] + async fn last_stored_event_ignores_the_retrieval_skip_flag() { + use std::time::{Duration, UNIX_EPOCH}; + + use crate::types::ArchiverValue; + + let dir = tempfile::tempdir().unwrap(); + let tier = |name: &str| { + Arc::new(PlainPbStoragePlugin::new( + name, + dir.path().join(name), + PartitionGranularity::Year, + )) + }; + let tiered = TieredStorage { + sts: tier("STS"), + mts: tier("MTS"), + lts: tier("LTS"), + }; + let at = |secs: u64| UNIX_EPOCH + Duration::from_secs(1_700_000_000 + secs); + let older = ArchiverSample::new(at(0), ArchiverValue::ScalarDouble(1.0)); + let newer = ArchiverSample::new(at(100), ArchiverValue::ScalarDouble(2.0)); + tiered + .mts + .append_event("TIER:PV", ArchDbType::ScalarDouble, &older) + .await + .unwrap(); + tiered + .sts + .append_event("TIER:PV", ArchDbType::ScalarDouble, &newer) + .await + .unwrap(); + + crate::flags::set("SKIP_STS_FOR_RETRIEVAL", true); + let read = tiered.get_last_known_event("TIER:PV").await; + let stored = tiered.get_last_stored_event("TIER:PV").await; + crate::flags::set("SKIP_STS_FOR_RETRIEVAL", false); + + assert_eq!(read.unwrap().unwrap().timestamp, older.timestamp); + assert_eq!(stored.unwrap().unwrap().timestamp, newer.timestamp); + } } diff --git a/crates/archiver-core/src/storage/traits.rs b/crates/archiver-core/src/storage/traits.rs index bf821e7..abc9dda 100644 --- a/crates/archiver-core/src/storage/traits.rs +++ b/crates/archiver-core/src/storage/traits.rs @@ -109,6 +109,15 @@ pub trait StoragePlugin: Send + Sync { /// Get the most recent known event for a PV. async fn get_last_known_event(&self, pv: &str) -> anyhow::Result>; + /// The newest event on disk for `pv`, ignoring retrieval routing. + /// [`Self::get_last_known_event`] is the retrieval view and may hide + /// tiers (`SKIP__FOR_RETRIEVAL`); the write path seeds its + /// ordering gate from this, so a tier hidden from readers still + /// counts as stored. Defaults to the retrieval view. + async fn get_last_stored_event(&self, pv: &str) -> anyhow::Result> { + self.get_last_known_event(pv).await + } + /// Get the last sample whose timestamp is strictly before `target`. /// Used by retrieval to prepend a continuity sample when the user's /// query window starts in a gap between samples (Java's diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index e152a4a..7a35c13 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -4651,7 +4651,10 @@ async fn shard_append_loop( /// it when power was lost with `fsync_on_flush` off, so that /// redelivery, the one sample of the lost window still recoverable, /// would be dropped. A duplicate or a gap is measured against the -/// store, so the store has the final word. +/// store, so the store has the final word. The read is the stored +/// view, not the retrieval view a `SKIP__FOR_RETRIEVAL` flag +/// routes around: a tier hidden from readers still holds the samples +/// the gate must not store again. /// /// The read is isolated like the append: it runs on the blocking pool /// and the shard waits at most `read_timeout` for it, so a store that @@ -4673,7 +4676,7 @@ async fn seed_ordering_gate( let pv_for_task = pv.to_string(); let read = tokio::task::spawn_blocking(move || { tokio::runtime::Handle::current() - .block_on(storage_for_task.get_last_known_event(&pv_for_task)) + .block_on(storage_for_task.get_last_stored_event(&pv_for_task)) }); match tokio::time::timeout(read_timeout, read).await { Ok(Ok(Ok(last))) => { diff --git a/tests/write_loop_failure.rs b/tests/write_loop_failure.rs index aba3b09..646fe89 100644 --- a/tests/write_loop_failure.rs +++ b/tests/write_loop_failure.rs @@ -73,12 +73,16 @@ struct InjectingStorage { /// regression test assert the owner drains markers ONLY on the /// observed-success path (never on a flush timeout/panic). take_loss_calls: AtomicUsize, - /// What `get_last_known_event` reports per PV: the store's tail - /// the shard seeds its ordering gate from. + /// What the tail reads report per PV: the store's tail the shard + /// seeds its ordering gate from. tail: Mutex>, - /// When set, `get_last_known_event` blocks its thread for this - /// long first, the way a stalled file read would. + /// When set, a tail read blocks its thread for this long first, + /// the way a stalled file read would. tail_hang_for: Mutex>, + /// When set, `get_last_known_event` reports nothing, the way a + /// tier under `SKIP__FOR_RETRIEVAL` is hidden from readers; + /// `get_last_stored_event` still reports the tail. + tail_hidden_from_retrieval: AtomicBool, } impl InjectingStorage { @@ -98,6 +102,7 @@ impl InjectingStorage { take_loss_calls: AtomicUsize::new(0), tail: Mutex::new(HashMap::new()), tail_hang_for: Mutex::new(None), + tail_hidden_from_retrieval: AtomicBool::new(false), }) } @@ -109,6 +114,11 @@ impl InjectingStorage { self.tail.lock().unwrap().insert(pv.to_string(), sample); } + fn hide_tail_from_retrieval(&self) { + self.tail_hidden_from_retrieval + .store(true, Ordering::SeqCst); + } + /// Push a dirty-write loss marker, simulating a loss recorded /// outside a flush pass (LRU eviction, ghost-file, etc.). fn push_loss(&self, pv: &str) { @@ -244,6 +254,13 @@ impl StoragePlugin for InjectingStorage { } async fn get_last_known_event(&self, pv: &str) -> anyhow::Result> { + if self.tail_hidden_from_retrieval.load(Ordering::SeqCst) { + return Ok(None); + } + self.get_last_stored_event(pv).await + } + + async fn get_last_stored_event(&self, pv: &str) -> anyhow::Result> { let hang = *self.tail_hang_for.lock().unwrap(); if let Some(hang) = hang { std::thread::sleep(hang); @@ -2090,7 +2107,23 @@ async fn seed_case( send: &[(SystemTime, f64)], expect_stored: u64, ) -> (Vec, u64) { - let storage = InjectingStorage::new(); + seed_case_with( + InjectingStorage::new(), + registry_seed, + tail, + send, + expect_stored, + ) + .await +} + +async fn seed_case_with( + storage: Arc, + registry_seed: Option, + tail: Option, + send: &[(SystemTime, f64)], + expect_stored: u64, +) -> (Vec, u64) { let registry = Arc::new(PvRegistry::in_memory().unwrap()); registry .register_pv("A", ArchDbType::ScalarDouble, &SampleMode::Monitor, 1) @@ -2144,6 +2177,25 @@ async fn gate_seeds_from_a_store_tail_ahead_of_the_registry() { assert_eq!(drops, 1); } +/// A tier under `SKIP__FOR_RETRIEVAL` is hidden from readers, +/// not from the gate: the seed reads the stored view, so the +/// connect-time redelivery of the hidden tail is still dropped. +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn gate_seeds_from_a_tail_hidden_from_retrieval() { + let storage = InjectingStorage::new(); + storage.hide_tail_from_retrieval(); + let (stored, drops) = seed_case_with( + storage, + None, + Some(sample_at(ts(1000), 1.0)), + &[(ts(1000), 1.0), (ts(1001), 2.0)], + 1, + ) + .await; + assert_eq!(stored, vec![ts(1001)]); + assert_eq!(drops, 1); +} + /// A registry seed ahead of the store (power loss with /// `fsync_on_flush` off): the redelivered current value is newer than /// anything on disk and must be stored, although the registry claims From 2943a000f4772fa115a0142f43b7680dfbabdbbb Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 16:58:54 +0900 Subject: [PATCH 54/55] engine: check the QUIESCE_TIMEOUT bounds of lock_op and stop_and_finalize Under tokio's paused clock the 60 s elapse as soon as the runtime is idle, so the bounds are checked at their real value; without either timeout the matching test hangs. --- crates/archiver-engine/Cargo.toml | 3 + crates/archiver-engine/src/channel_manager.rs | 72 +++++++++++++++++++ 2 files changed, 75 insertions(+) diff --git a/crates/archiver-engine/Cargo.toml b/crates/archiver-engine/Cargo.toml index 96e04ed..579d437 100644 --- a/crates/archiver-engine/Cargo.toml +++ b/crates/archiver-engine/Cargo.toml @@ -22,3 +22,6 @@ epics-rs = { workspace = true } tokio-util = { workspace = true } metrics = { workspace = true } futures = { workspace = true } + +[dev-dependencies] +tokio = { workspace = true, features = ["test-util"] } diff --git a/crates/archiver-engine/src/channel_manager.rs b/crates/archiver-engine/src/channel_manager.rs index 7a35c13..bd33758 100644 --- a/crates/archiver-engine/src/channel_manager.rs +++ b/crates/archiver-engine/src/channel_manager.rs @@ -5841,3 +5841,75 @@ mod shard_lifecycle_tests { .unwrap(); } } + +#[cfg(test)] +mod quiesce_bound_tests { + use super::*; + use archiver_core::registry::{PvRegistry, PvStatus, SampleMode}; + + const PV: &str = "BOUND:PV"; + + /// A manager over an in-memory registry with `PV` registered. Under + /// the paused clock [`QUIESCE_TIMEOUT`] elapses as soon as the + /// runtime is idle, so the bounds are checked at their real value. + async fn manager() -> Arc { + let registry = Arc::new(PvRegistry::in_memory().unwrap()); + registry + .register_pv(PV, ArchDbType::ScalarDouble, &SampleMode::Monitor, 1) + .unwrap(); + let (mgr, _rx) = ChannelManager::new(registry, None).await.unwrap(); + Arc::new(mgr) + } + + fn status(mgr: &ChannelManager) -> PvStatus { + mgr.registry.get_pv(PV).unwrap().unwrap().status + } + + /// An operation queued behind a held op lock fails after the bound + /// instead of waiting the holder out, and leaves the PV untouched. + #[tokio::test(start_paused = true)] + async fn an_operation_behind_a_held_op_lock_fails_after_the_bound() { + let mgr = manager().await; + let before = status(&mgr); + let _held = mgr.op_lock(PV).lock_owned().await; + + let started = tokio::time::Instant::now(); + let err = mgr.pause_pv(PV).await.expect_err("bounded lock wait"); + assert!(err.to_string().contains("still in progress after"), "{err}"); + assert!(started.elapsed() >= QUIESCE_TIMEOUT); + assert_eq!(status(&mgr), before); + } + + /// A stop transition still quiescing after the bound reports it, + /// keeps the op lock, and still finalizes when the queue drains. + #[tokio::test(start_paused = true)] + async fn a_stop_transition_still_quiescing_after_the_bound_completes_later() { + let mgr = manager().await; + let counters = Arc::new(PvCounters::default()); + mgr.counters.insert(PV.to_string(), counters.clone()); + let in_flight = InFlight::new(Some(counters)); + + let started = tokio::time::Instant::now(); + let err = mgr.pause_pv(PV).await.expect_err("bounded quiesce"); + assert!(err.to_string().contains("still quiescing after"), "{err}"); + assert!(started.elapsed() >= QUIESCE_TIMEOUT); + assert_ne!(status(&mgr), PvStatus::Paused); + // The parked transition still owns the lock. + let err = mgr + .resume_pv(PV) + .await + .expect_err("lock held by the transition"); + assert!(err.to_string().contains("still in progress after"), "{err}"); + + drop(in_flight); + let deadline = tokio::time::Instant::now() + Duration::from_secs(10); + while status(&mgr) != PvStatus::Paused { + assert!( + tokio::time::Instant::now() < deadline, + "transition never finalized" + ); + tokio::time::sleep(Duration::from_millis(10)).await; + } + assert!(mgr.lock_op(PV).await.is_ok(), "lock not released"); + } +} From af2a7c4620cab5cb23bd239bce1cf7ff6280b0c6 Mon Sep 17 00:00:00 2001 From: Sang Woo Kim Date: Thu, 3 Sep 2026 19:42:09 +0900 Subject: [PATCH 55/55] =?UTF-8?q?chore(release):=20v0.4.2=20=E2=80=94=20wr?= =?UTF-8?q?ite-path=20integrity=20fixes=20+=20epics-rs=200.28.1?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Bump the workspace, the four internal-dep pins and the decoupled binary package to 0.4.2 (aligned again: epics-rs 0.20.4 → 0.28.1 touches every crate) and record the release in CHANGELOG.md. --- CHANGELOG.md | 105 +++++++++++++++++++++++++++++++++++++++++++++++++++ Cargo.lock | 10 ++--- Cargo.toml | 16 ++++---- 3 files changed, 118 insertions(+), 13 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index c52ecbb..18fbfab 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,110 @@ # Changelog +## v0.4.2 — 2026-09-03 + +Write-path integrity release. Nine review rounds over the data-saving +path (ingest producers → sharded write pool → PlainPB → registry), each +finding fixed at its cause in its own commit. The result: a sample the +archiver is handed is stored once and only once across restarts, +resumes, retypes and shutdown, and every loss the engine cannot avoid +is counted per PV instead of logged or silent. Also bumps `epics-rs` +0.20.4 → 0.28.1. + +### Added + +- **Per-PV `storageWriteErrors` and `flushLosses` counters.** A failed + or panicked append and a flush-time loss (failed + `flush_ingest_writes`, dirty-writer eviction) are attributed to the + PV on `getPVStatus` and in Prometheus + (`archiver_storage_flush_losses_total`); before, a PV losing every + sample to ENOSPC looked healthy. +- **Per-PV quiesce for pause / stop / delete.** `PvSample` carries an + in-flight guard, so `pausePV`, `stopArchivingPV`, `deletePV`, + `renamePV`, `changeTypeForPV` and `reassignAppliance` act only once + the PV's tail is out of the write pool. The stop transition runs in + a task that owns the PV's op lock and commits the registry status + when the queue drains; the caller's wait and the op-lock wait are + both bounded by 60 s with an error, never a hang, and a caller + dropped mid-wait no longer leaves the registry `Active` for a PV + that stopped archiving. +- **PVA overflow slot.** A PVA sample the full write queue refuses is + parked and delivered with backpressure instead of dropped; only a + parked sample replaced by a newer one counts as an overflow drop. +- **`changeTypeForPV` converts the stored partitions** + (`StoragePlugin::convert_pv_type`, Java's ThruNumberConversion + rules) before the registry flips, so appends are not refused until + the partition rolls; the ETL move gates are held across the + conversion. A scalar target for an array PV is refused. +- **Write pool is a critical task.** Its exit or panic requests + shutdown and fails `main`, instead of every producer stopping + silently while the API reports the PVs `Active`. + +### Fixed + +- **Connect-time redelivery stored twice, or dropped.** The shard's + ordering gate is seeded from the store's tail (read once per PV, on + the blocking pool, bounded by `append_timeout`, ignoring + `SKIP__FOR_RETRIEVAL`) and an equal timestamp is dropped as + the same event; the registry's `last_timestamp` survives + `archivePV` re-imports, `putPVTypeInfo`, `changeTypeForPV`, + `renamePV` and `receivePVMigration`, and can only move forward. +- **Shard state outliving the archiving task.** The type-change gate + is stateless (compares the value's own type), and the ordering + high-water lives in the PV's counters, so `deletePV` + re-archive + and `changeTypeForPV` + resume no longer drop every sample until + restart. +- **Shutdown loss.** The flush owner's final flush runs after the + shard drain (with its own in-flight window), the dispatcher flips + each shard's drain only after moving the main queue's tail into it, + and the producers are cancelled and awaited before the pool's + shutdown flag flips. +- **CA arrays.** Array PVs register as the waveform type + (`ArchDbType::with_element_count`, legacy rows promoted once), and a + count-1 CA event on a waveform PV is stored as a one-element + waveform; before, every CA array sample was refused by the type + gate. +- **PVA.** The monitor re-subscribes when the subscription task ends + (server-side close, fatal error); an unrepresentable `timeStamp` + falls back to now instead of panicking the reactor task; untyped + (empty) string arrays are typed from the introspection; dropped + samples are logged; every producer wait is raced with the cancel + token so a pause is not delayed by a disconnected PV. +- **Drift window.** Only the past side is waived for the first sample + after connect; a far-future first stamp no longer poisons the + ordering gate until restart. +- **PlainPB.** A partition is truncated only on an undecodable header + (a transient EMFILE/EIO on the probe no longer wipes it); a cached + writer's buffer is discarded only when its file is positively + absent; a removed partition directory is recreated on the next + append; a failed write keeps the writer (deferred, retried next + cycle) instead of counting up to 64 KiB per PV as lost; the header + probe runs after the fd reservation with the same evict-and-retry; + appends whose type differs from the partition header are refused on + both the open and cached-writer paths; `rename_pv` refuses an + existing destination partition; timestamps outside chrono's range + fail instead of panicking. +- **Registry.** `register_pv_with_protocol` upserts (no more + `INSERT OR REPLACE` nulling `last_timestamp`, `prec`, `egu`, + aliases and policy), refuses alias rows and native-type changes; + `/` is rejected in PV names so the path encoding is injective; + paused PVs stay in the counters reports. +- **ETL.** `move_file` routes by the path-derived PV name, so a + renamed PV's partitions migrate under the new name. + +### Changed + +- **Bumped `epics-rs` 0.20.4 → 0.28.1.** PVA array-of-composite + elements carry per-element nullability; `pvmonitor_handle` reports + connection state through `on_conn`; the 0.28.1 CA server no longer + emits a subscription's initial event with a racing put's value + under the pre-put timestamp. +- **`per_shard_buffer` default** is the main channel capacity split + across `write_shards`; unset no longer means 4096 per shard. +- **Library API.** `ChannelManager::new` / `new_with_drift` drop the + storage parameter; `StoragePlugin` gains `convert_pv_type` and + `get_last_stored_event`; `epics_value_to_archiver` takes the + registered `ArchDbType`. + ## v0.4.1 — 2026-07-02 Data-integrity release. The headline is a from-scratch redesign of the diff --git a/Cargo.lock b/Cargo.lock index 3f3d5f0..f258e08 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -134,7 +134,7 @@ dependencies = [ [[package]] name = "archiver-api" -version = "0.4.1" +version = "0.4.2" dependencies = [ "anyhow", "archiver-core", @@ -164,7 +164,7 @@ dependencies = [ [[package]] name = "archiver-core" -version = "0.4.1" +version = "0.4.2" dependencies = [ "anyhow", "archiver-proto", @@ -185,7 +185,7 @@ dependencies = [ [[package]] name = "archiver-engine" -version = "0.4.1" +version = "0.4.2" dependencies = [ "anyhow", "archiver-core", @@ -205,7 +205,7 @@ dependencies = [ [[package]] name = "archiver-proto" -version = "0.4.1" +version = "0.4.2" dependencies = [ "prost", "prost-build", @@ -812,7 +812,7 @@ checksum = "869b0adbda23651a9c5c0c3d270aac9fcb52e8622a8f2b17e57802d7791962f2" [[package]] name = "epics-archiver" -version = "0.4.1" +version = "0.4.2" dependencies = [ "anyhow", "archiver-api", diff --git a/Cargo.toml b/Cargo.toml index a49b098..3c1f53b 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -9,7 +9,7 @@ members = [ resolver = "2" [workspace.package] -version = "0.4.1" +version = "0.4.2" edition = "2024" rust-version = "1.85" license = "MIT" @@ -87,20 +87,20 @@ tempfile = "3" tower = { version = "0.5", features = ["util"] } # Internal crates -archiver-proto = { path = "crates/archiver-proto", version = "0.4.1" } -archiver-core = { path = "crates/archiver-core", version = "0.4.1" } -archiver-engine = { path = "crates/archiver-engine", version = "0.4.1" } -archiver-api = { path = "crates/archiver-api", version = "0.4.1" } +archiver-proto = { path = "crates/archiver-proto", version = "0.4.2" } +archiver-core = { path = "crates/archiver-core", version = "0.4.2" } +archiver-engine = { path = "crates/archiver-engine", version = "0.4.2" } +archiver-api = { path = "crates/archiver-api", version = "0.4.2" } [package] name = "epics-archiver" # Decoupled from workspace.package.version so the binary can ship # bug-fix releases (e.g. CLI ergonomics) without churning the four -# library crates whose code is unchanged. v0.4.1 keeps binary and -# the four library crates aligned (the epics-rs 0.16.2 → 0.20.3 bump +# library crates whose code is unchanged. v0.4.2 keeps binary and +# the four library crates aligned (the epics-rs 0.20.4 → 0.28.1 bump # touches every crate); future binary-only patch releases may diverge # again. -version = "0.4.1" +version = "0.4.2" edition.workspace = true license.workspace = true repository.workspace = true