From b22d3d88a654a6f72a9e5800d39200b321107173 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 3 Sep 2026 08:24:57 +0900 Subject: [PATCH 01/49] test: reproduce pg-erd response slow-drip lifetime gap --- tests/pg_erd_slow_drip_response_traffic.rs | 271 +++++++++++++++++++++ 1 file changed, 271 insertions(+) create mode 100644 tests/pg_erd_slow_drip_response_traffic.rs diff --git a/tests/pg_erd_slow_drip_response_traffic.rs b/tests/pg_erd_slow_drip_response_traffic.rs new file mode 100644 index 00000000..443b04db --- /dev/null +++ b/tests/pg_erd_slow_drip_response_traffic.rs @@ -0,0 +1,271 @@ +//! Real-listener RED contract for a pg-erd upstream that continuously drips response-body bytes. +//! +//! Pingora's peer `read_timeout` is an inactivity timeout that resets after each successful read. +//! This fixture therefore keeps each origin write inside `read_ms` while extending the response +//! beyond an explicit migration-owned response-body lifetime. The current contract cannot express +//! that lifetime; the test is expected to stay RED until the versioned admin/runtime boundary owns +//! and enforces it without retrying or failing over after the response has been committed. + +use std::io::{ErrorKind, Read, Write}; +use std::net::{SocketAddr, TcpListener, TcpStream}; +use std::process::{Child, Command, Stdio}; +use std::thread; +use std::time::{Duration, Instant}; + +use tempfile::NamedTempFile; + +struct GatewayProcess(Child); + +impl Drop for GatewayProcess { + fn drop(&mut self) { + let _ = self.0.kill(); + let _ = self.0.wait(); + } +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum DownstreamTermination { + Eof, + ConnectionReset, +} + +fn reserve_loopback() -> SocketAddr { + TcpListener::bind("127.0.0.1:0") + .expect("loopback port should be reservable") + .local_addr() + .expect("reservation should expose an address") +} + +fn write_config( + listener: SocketAddr, + metrics_listener: SocketAddr, + backend: SocketAddr, + frontend: SocketAddr, +) -> NamedTempFile { + let mut file = NamedTempFile::new().expect("temporary config should be writable"); + writeln!( + file, + "version: 2\nlistener: {listener}\nmetrics_listener: {metrics_listener}\nmax_request_body_bytes: 8\nmax_in_flight_requests: 8\nmax_upstream_response_body_ms: 300\nupstream_keepalive_pool_size: 4\nupstreams:\n - name: backend\n address: {backend}\n tls: false\n timeouts:\n connection_ms: 200\n total_connection_ms: 400\n read_ms: 120\n write_ms: 1000\n idle_ms: 5000\n - name: frontend\n address: {frontend}\n tls: false\n timeouts:\n connection_ms: 200\n total_connection_ms: 400\n read_ms: 1000\n write_ms: 1000\n idle_ms: 5000" + ) + .expect("migration config should be written"); + file +} + +fn wait_until_listening(address: SocketAddr, process: &mut Child) { + let deadline = Instant::now() + Duration::from_secs(10); + loop { + if let Some(status) = process + .try_wait() + .expect("gateway process state should be readable") + { + panic!("gateway exited before accepting traffic: {status}"); + } + if TcpStream::connect_timeout(&address, Duration::from_millis(100)).is_ok() { + return; + } + assert!(Instant::now() < deadline, "gateway did not start within 10s"); + thread::sleep(Duration::from_millis(25)); + } +} + +fn start_gateway( + config: &NamedTempFile, + gateway_address: SocketAddr, + metrics_address: SocketAddr, +) -> GatewayProcess { + let mut child = Command::new(env!("CARGO_BIN_EXE_cwl-pingora-pg-erd-migration")) + .args(["--config", config.path().to_str().expect("UTF-8 temp path")]) + .stdin(Stdio::null()) + .stdout(Stdio::null()) + .stderr(Stdio::null()) + .spawn() + .expect("compiled pg-erd migration binary should start"); + wait_until_listening(gateway_address, &mut child); + wait_until_listening(metrics_address, &mut child); + GatewayProcess(child) +} + +fn raw_request(address: SocketAddr, request: &[u8]) -> String { + let mut downstream = TcpStream::connect(address).expect("gateway should accept traffic"); + downstream + .set_read_timeout(Some(Duration::from_secs(5))) + .expect("downstream timeout should be configurable"); + downstream + .write_all(request) + .expect("downstream request should be writable"); + let mut response = String::new(); + downstream + .read_to_string(&mut response) + .expect("gateway response should be readable"); + response +} + +fn raw_request_until_terminal( + address: SocketAddr, + request: &[u8], +) -> (Vec, DownstreamTermination, Duration) { + let started = Instant::now(); + let mut downstream = TcpStream::connect(address).expect("gateway should accept traffic"); + downstream + .set_read_timeout(Some(Duration::from_secs(2))) + .expect("downstream timeout should be configurable"); + downstream + .write_all(request) + .expect("downstream request should be writable"); + + let mut response = Vec::new(); + let mut buffer = [0_u8; 1024]; + loop { + match downstream.read(&mut buffer) { + Ok(0) => return (response, DownstreamTermination::Eof, started.elapsed()), + Ok(read) => response.extend_from_slice(&buffer[..read]), + Err(error) if error.kind() == ErrorKind::ConnectionReset => { + return ( + response, + DownstreamTermination::ConnectionReset, + started.elapsed(), + ); + } + Err(error) => panic!("slow-drip downstream response should terminate, not stall: {error}"), + } + } +} + +fn get(address: SocketAddr, path: &str) -> String { + raw_request( + address, + format!("GET {path} HTTP/1.1\r\nHost: app.example:8080\r\nConnection: close\r\n\r\n") + .as_bytes(), + ) +} + +fn read_request_headers(stream: &mut TcpStream) -> String { + let mut bytes = Vec::new(); + let mut buffer = [0_u8; 1024]; + loop { + let read = stream.read(&mut buffer).expect("origin request should be readable"); + assert!(read > 0, "gateway closed origin request before headers completed"); + bytes.extend_from_slice(&buffer[..read]); + if bytes.windows(4).any(|window| window == b"\r\n\r\n") { + return String::from_utf8_lossy(&bytes).into_owned(); + } + } +} + +#[test] +fn compiled_pg_erd_terminates_continuous_response_drip_without_poisoning_other_routes() { + let backend = TcpListener::bind("127.0.0.1:0").expect("backend fixture should bind"); + let backend_address = backend.local_addr().expect("backend address should exist"); + let backend_origin = thread::spawn(move || { + let (mut stream, _) = backend + .accept() + .expect("routed request should reach the characterized backend authority"); + let request = read_request_headers(&mut stream); + assert!(request.starts_with("GET /api/slow-drip HTTP/1.1\r\n")); + stream + .write_all( + b"HTTP/1.1 200 OK\r\nContent-Length: 20\r\nConnection: close\r\n\r\n", + ) + .expect("backend response header should be writable"); + + let mut writes = 0_usize; + for _ in 0..20 { + match stream.write_all(b"x") { + Ok(()) => writes += 1, + Err(error) + if matches!( + error.kind(), + ErrorKind::BrokenPipe | ErrorKind::ConnectionReset | ErrorKind::NotConnected + ) => + { + break; + } + Err(error) => panic!("unexpected slow-drip origin write failure: {error}"), + } + thread::sleep(Duration::from_millis(60)); + } + writes + }); + + let frontend = TcpListener::bind("127.0.0.1:0").expect("frontend fixture should bind"); + let frontend_address = frontend.local_addr().expect("frontend address should exist"); + let frontend_origin = thread::spawn(move || { + let (mut stream, _) = frontend + .accept() + .expect("fallback request should reach the independent frontend authority"); + let request = read_request_headers(&mut stream); + assert!(request.starts_with("GET /after-slow-drip HTTP/1.1\r\n")); + stream + .write_all( + b"HTTP/1.1 200 OK\r\nContent-Length: 9\r\nConnection: close\r\n\r\nrecovered", + ) + .expect("frontend recovery response should be writable"); + }); + + let gateway_address = reserve_loopback(); + let metrics_address = reserve_loopback(); + let config = write_config( + gateway_address, + metrics_address, + backend_address, + frontend_address, + ); + let _process = start_gateway(&config, gateway_address, metrics_address); + + let (partial, termination, elapsed) = raw_request_until_terminal( + gateway_address, + b"GET /api/slow-drip HTTP/1.1\r\nHost: app.example:8080\r\nConnection: close\r\n\r\n", + ); + assert!( + matches!( + termination, + DownstreamTermination::Eof | DownstreamTermination::ConnectionReset + ), + "an over-budget response body must terminate the downstream connection" + ); + assert!( + elapsed < Duration::from_secs(1), + "the 300ms response-body budget must stop a continuously progressing response instead of allowing the full 1.2s drip: {elapsed:?}" + ); + + let header_end = partial + .windows(4) + .position(|window| window == b"\r\n\r\n") + .map(|position| position + 4) + .expect("slow-drip response must commit a complete header block before termination"); + let headers = String::from_utf8_lossy(&partial[..header_end]).to_ascii_lowercase(); + assert!( + headers.starts_with("http/1.1 200"), + "a post-commit lifetime failure cannot be rewritten as a second status: {headers:?}" + ); + let body = &partial[header_end..]; + assert!( + body.len() < 20, + "the configured response-body budget must terminate before the declared body completes" + ); + + let readiness = get(gateway_address, "/readyz"); + assert!( + readiness.starts_with("HTTP/1.1 200"), + "one slow-drip origin must not poison process readiness: {readiness:?}" + ); + let metrics = get(metrics_address, "/metrics"); + assert!( + metrics.contains("cwl_pingora_gateway_request_errors_total 1"), + "response-lifetime enforcement must remain visible through low-cardinality error telemetry: {metrics:?}" + ); + let recovered = get(gateway_address, "/after-slow-drip"); + assert!(recovered.starts_with("HTTP/1.1 200")); + assert!(recovered.ends_with("\r\n\r\nrecovered")); + + frontend_origin + .join() + .expect("frontend recovery fixture should complete"); + let writes = backend_origin + .join() + .expect("slow-drip backend fixture should complete"); + assert!( + writes < 20, + "the gateway must close the over-budget origin response before all drip bytes are accepted" + ); +} From 75a181e609712cf599bf36e315898e25c7338a18 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 3 Sep 2026 08:26:55 +0900 Subject: [PATCH 02/49] feat: add response-body lifetime isolation budget --- src/runtime_isolation.rs | 142 +++++++++++++++++++++++++++++++++++++-- 1 file changed, 138 insertions(+), 4 deletions(-) diff --git a/src/runtime_isolation.rs b/src/runtime_isolation.rs index c81ec372..8da32e2b 100644 --- a/src/runtime_isolation.rs +++ b/src/runtime_isolation.rs @@ -1,10 +1,12 @@ //! Transport-neutral runtime-isolation budgets shared by Pingora delivery adapters. //! -//! This bounded context owns request-body and concurrent-request admission limits. It does not -//! select routes, mutate HTTP policy, authenticate callers, or make product-domain decisions. +//! This bounded context owns request-body, concurrent-request admission, and optional upstream +//! response-body lifetime limits. It does not select routes, mutate HTTP policy, authenticate +//! callers, or make product-domain decisions. use std::sync::atomic::{AtomicUsize, Ordering}; use std::sync::Arc; +use std::time::{Duration, Instant}; use thiserror::Error; @@ -17,6 +19,9 @@ pub enum RuntimeIsolationConfigError { /// A zero in-flight limit would reject every proxied request. #[error("max_in_flight_requests must be greater than zero")] ZeroMaxInFlightRequests, + /// A zero response-body lifetime would reject every non-empty upstream response immediately. + #[error("max_upstream_response_body_ms must be greater than zero")] + ZeroMaxUpstreamResponseBodyMs, } /// Immutable request-isolation limits shared by gateway delivery adapters. @@ -24,13 +29,38 @@ pub enum RuntimeIsolationConfigError { pub struct RuntimeIsolationLimits { max_request_body_bytes: u64, max_in_flight_requests: usize, + max_upstream_response_body_ms: Option, } impl RuntimeIsolationLimits { /// Validates explicit non-zero body and in-flight request budgets. + /// + /// The generic v1 contract has no response-body lifetime field, so this constructor preserves + /// that versioned behavior rather than inventing a hidden timeout. pub fn try_new( max_request_body_bytes: u64, max_in_flight_requests: usize, + ) -> Result { + Self::try_new_internal(max_request_body_bytes, max_in_flight_requests, None) + } + + /// Validates request isolation plus an explicit upstream response-body lifetime budget. + pub(crate) fn try_new_with_response_body_limit( + max_request_body_bytes: u64, + max_in_flight_requests: usize, + max_upstream_response_body_ms: u64, + ) -> Result { + Self::try_new_internal( + max_request_body_bytes, + max_in_flight_requests, + Some(max_upstream_response_body_ms), + ) + } + + fn try_new_internal( + max_request_body_bytes: u64, + max_in_flight_requests: usize, + max_upstream_response_body_ms: Option, ) -> Result { if max_request_body_bytes == 0 { return Err(RuntimeIsolationConfigError::ZeroMaxRequestBodyBytes); @@ -38,9 +68,13 @@ impl RuntimeIsolationLimits { if max_in_flight_requests == 0 { return Err(RuntimeIsolationConfigError::ZeroMaxInFlightRequests); } + if max_upstream_response_body_ms == Some(0) { + return Err(RuntimeIsolationConfigError::ZeroMaxUpstreamResponseBodyMs); + } Ok(Self { max_request_body_bytes, max_in_flight_requests, + max_upstream_response_body_ms, }) } @@ -51,6 +85,7 @@ impl RuntimeIsolationLimits { Self { max_request_body_bytes, max_in_flight_requests, + max_upstream_response_body_ms: None, } } @@ -63,6 +98,11 @@ impl RuntimeIsolationLimits { pub fn max_in_flight_requests(self) -> usize { self.max_in_flight_requests } + + /// Returns the configured upstream response-body lifetime, when the active contract owns one. + pub fn max_upstream_response_body_ms(self) -> Option { + self.max_upstream_response_body_ms + } } #[derive(Debug, Clone, PartialEq, Eq)] @@ -148,11 +188,59 @@ impl RequestBodyBudget { } } +/// Elapsed-time guard for the body phase of one admitted upstream response. +#[derive(Debug)] +pub(crate) struct ResponseBodyLifetimeBudget { + limit: Option, + started_at: Option, +} + +/// Evidence that a response-body progress callback arrived after its configured lifetime. +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct ResponseBodyLifetimeExceeded { + pub(crate) elapsed: Duration, + pub(crate) limit: Duration, +} + +impl ResponseBodyLifetimeBudget { + /// Creates a dormant response-body budget from the active runtime-isolation contract. + pub(crate) fn new(limits: RuntimeIsolationLimits) -> Self { + Self { + limit: limits.max_upstream_response_body_ms().map(Duration::from_millis), + started_at: None, + } + } + + /// Starts the body lifetime once; repeated response-filter callbacks cannot reset the deadline. + pub(crate) fn start(&mut self, now: Instant) { + if self.limit.is_some() && self.started_at.is_none() { + self.started_at = Some(now); + } + } + + /// Rejects the first observed body-progress boundary at or beyond the configured lifetime. + pub(crate) fn reject_if_expired( + &self, + now: Instant, + ) -> Result<(), ResponseBodyLifetimeExceeded> { + let (Some(limit), Some(started_at)) = (self.limit, self.started_at) else { + return Ok(()); + }; + let elapsed = now.saturating_duration_since(started_at); + if elapsed >= limit { + return Err(ResponseBodyLifetimeExceeded { elapsed, limit }); + } + Ok(()) + } +} + #[cfg(test)] mod tests { + use std::time::{Duration, Instant}; + use super::{ - BodyLimitExceeded, RequestAdmissionBudget, RequestBodyBudget, RuntimeIsolationConfigError, - RuntimeIsolationLimits, + BodyLimitExceeded, RequestAdmissionBudget, RequestBodyBudget, ResponseBodyLifetimeBudget, + ResponseBodyLifetimeExceeded, RuntimeIsolationConfigError, RuntimeIsolationLimits, }; #[test] @@ -165,10 +253,21 @@ mod tests { RuntimeIsolationLimits::try_new(1, 0), Err(RuntimeIsolationConfigError::ZeroMaxInFlightRequests) ); + assert_eq!( + RuntimeIsolationLimits::try_new_with_response_body_limit(1, 1, 0), + Err(RuntimeIsolationConfigError::ZeroMaxUpstreamResponseBodyMs) + ); let limits = RuntimeIsolationLimits::try_new(1024, 2).expect("non-zero limits are valid"); assert_eq!(limits.max_request_body_bytes(), 1024); assert_eq!(limits.max_in_flight_requests(), 2); + assert_eq!(limits.max_upstream_response_body_ms(), None); + + let bounded = RuntimeIsolationLimits::try_new_with_response_body_limit(2048, 3, 750) + .expect("explicit positive response lifetime must be valid"); + assert_eq!(bounded.max_request_body_bytes(), 2048); + assert_eq!(bounded.max_in_flight_requests(), 3); + assert_eq!(bounded.max_upstream_response_body_ms(), Some(750)); } #[test] @@ -207,4 +306,39 @@ mod tests { } ); } + + #[test] + fn response_body_lifetime_is_dormant_without_a_versioned_budget() { + let limits = RuntimeIsolationLimits::try_new(4, 1).expect("fixture limits are valid"); + let mut budget = ResponseBodyLifetimeBudget::new(limits); + let now = Instant::now(); + + budget.start(now); + assert!(budget + .reject_if_expired(now + Duration::from_secs(60)) + .is_ok()); + } + + #[test] + fn response_body_lifetime_starts_once_and_rejects_at_the_limit() { + let limits = RuntimeIsolationLimits::try_new_with_response_body_limit(4, 1, 300) + .expect("fixture limits are valid"); + let mut budget = ResponseBodyLifetimeBudget::new(limits); + let started = Instant::now(); + budget.start(started); + budget.start(started + Duration::from_millis(250)); + + assert!(budget + .reject_if_expired(started + Duration::from_millis(299)) + .is_ok()); + assert_eq!( + budget + .reject_if_expired(started + Duration::from_millis(300)) + .unwrap_err(), + ResponseBodyLifetimeExceeded { + elapsed: Duration::from_millis(300), + limit: Duration::from_millis(300), + } + ); + } } From ea59921d616ad4fdd63fde968871117be588295d Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 3 Sep 2026 08:27:28 +0900 Subject: [PATCH 03/49] feat: version pg-erd response-body lifetime config --- src/migration_admin.rs | 62 ++++++++++++++++++++++++++++++++++-------- 1 file changed, 50 insertions(+), 12 deletions(-) diff --git a/src/migration_admin.rs b/src/migration_admin.rs index 0179ac01..e15d06e7 100644 --- a/src/migration_admin.rs +++ b/src/migration_admin.rs @@ -22,8 +22,10 @@ use crate::migration_plan::EdgeMigrationPlan; use crate::migration_proxy::MigrationGatewayProxy; use crate::runtime_isolation::{RuntimeIsolationConfigError, RuntimeIsolationLimits}; -/// Version of the bounded `pg-erd-cloud` migration admin configuration. -pub const PG_ERD_MIGRATION_CONFIG_VERSION: u32 = 1; +const PG_ERD_LEGACY_MIGRATION_CONFIG_VERSION: u32 = 1; + +/// Current version of the bounded `pg-erd-cloud` migration admin configuration. +pub const PG_ERD_MIGRATION_CONFIG_VERSION: u32 = 2; /// Fail-closed admin configuration for the characterized `pg-erd-cloud` migration runtime. #[derive(Debug, Clone, Deserialize, PartialEq, Eq)] @@ -34,6 +36,8 @@ pub struct PgErdMigrationConfig { metrics_listener: SocketAddr, max_request_body_bytes: u64, max_in_flight_requests: usize, + #[serde(default)] + max_upstream_response_body_ms: Option, upstream_keepalive_pool_size: usize, upstreams: Vec, } @@ -47,6 +51,12 @@ pub enum PgErdMigrationConfigError { /// The configuration requests a migration-admin version this binary does not implement. #[error("unsupported pg-erd migration configuration version {0}")] UnsupportedVersion(u32), + /// Current version 2 requires an explicit response-body lifetime instead of a hidden default. + #[error("pg-erd migration config version 2 requires max_upstream_response_body_ms")] + MissingUpstreamResponseBodyLifetime, + /// Legacy version 1 cannot silently acquire semantics introduced by version 2. + #[error("max_upstream_response_body_ms requires pg-erd migration config version 2")] + ResponseBodyLifetimeRequiresVersion2, /// A port-zero traffic listener would delegate the public authority to an ephemeral OS port. #[error("listener must use a non-zero port")] ZeroListenerPort, @@ -121,6 +131,11 @@ impl PgErdMigrationConfig { self.metrics_listener } + /// Returns the current response-body lifetime budget, or `None` for legacy version 1. + pub fn max_upstream_response_body_ms(&self) -> Option { + self.max_upstream_response_body_ms + } + /// Returns the validated Pingora upstream keepalive-pool budget. pub fn upstream_keepalive_pool_size(&self) -> usize { self.upstream_keepalive_pool_size @@ -133,16 +148,23 @@ impl PgErdMigrationConfig { /// Any custom TLS trust bundle is read by Pingora delivery during this single materialization. pub fn build_proxy(&self) -> Result { let delivery = self.build_delivery()?; - let limits = RuntimeIsolationLimits::try_new( - self.max_request_body_bytes, - self.max_in_flight_requests, - )?; + let limits = self.runtime_isolation_limits()?; Ok(MigrationGatewayProxy::new(delivery, limits)) } fn validate(&self) -> Result<(), PgErdMigrationConfigError> { - if self.version != PG_ERD_MIGRATION_CONFIG_VERSION { - return Err(PgErdMigrationConfigError::UnsupportedVersion(self.version)); + match self.version { + PG_ERD_LEGACY_MIGRATION_CONFIG_VERSION => { + if self.max_upstream_response_body_ms.is_some() { + return Err(PgErdMigrationConfigError::ResponseBodyLifetimeRequiresVersion2); + } + } + PG_ERD_MIGRATION_CONFIG_VERSION => { + if self.max_upstream_response_body_ms.is_none() { + return Err(PgErdMigrationConfigError::MissingUpstreamResponseBodyLifetime); + } + } + unsupported => return Err(PgErdMigrationConfigError::UnsupportedVersion(unsupported)), } if self.listener.port() == 0 { return Err(PgErdMigrationConfigError::ZeroListenerPort); @@ -157,13 +179,29 @@ impl PgErdMigrationConfig { return Err(PgErdMigrationConfigError::InvalidUpstreamKeepalivePoolSize); } - RuntimeIsolationLimits::try_new( - self.max_request_body_bytes, - self.max_in_flight_requests, - )?; + self.runtime_isolation_limits()?; self.validate_transport_authority(&pg_erd_migration_plan()) } + fn runtime_isolation_limits(&self) -> Result { + self.max_upstream_response_body_ms.map_or_else( + || { + RuntimeIsolationLimits::try_new( + self.max_request_body_bytes, + self.max_in_flight_requests, + ) + }, + |max_upstream_response_body_ms| { + RuntimeIsolationLimits::try_new_with_response_body_limit( + self.max_request_body_bytes, + self.max_in_flight_requests, + max_upstream_response_body_ms, + ) + }, + ) + .map_err(Into::into) + } + fn validate_transport_authority( &self, plan: &EdgeMigrationPlan, From 48c89851ff0b0a74b73f8f51971abfb0d44fd995 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 3 Sep 2026 08:28:11 +0900 Subject: [PATCH 04/49] fix: enforce pg-erd response-body lifetime --- src/migration_proxy.rs | 51 +++++++++++++++++++++++++++++++++++++----- 1 file changed, 45 insertions(+), 6 deletions(-) diff --git a/src/migration_proxy.rs b/src/migration_proxy.rs index e9b933d7..f2bb2e4c 100644 --- a/src/migration_proxy.rs +++ b/src/migration_proxy.rs @@ -4,6 +4,8 @@ //! isolation, trusted forwarding metadata, and shared transport observability. It does not //! introduce product authorization, service discovery, or business logic. +use std::time::{Duration, Instant}; + use async_trait::async_trait; use bytes::Bytes; use pingora::prelude::{ @@ -18,7 +20,7 @@ use crate::pingora_delivery::reject_uncharacterized_http1_protocol_transition; use crate::process_health::{respond_healthy, LIVENESS_PATH, READINESS_PATH}; use crate::runtime_isolation::{ BodyLimitExceeded, RequestAdmission, RequestAdmissionBudget, RequestBodyBudget, - RuntimeIsolationLimits, + ResponseBodyLifetimeBudget, ResponseBodyLifetimeExceeded, RuntimeIsolationLimits, }; /// Fail-closed callback errors for a characterized migration runtime. @@ -36,6 +38,7 @@ pub enum MigrationGatewayProxyError { #[derive(Debug)] pub struct MigrationRequestContext { request_body: RequestBodyBudget, + response_body_lifetime: ResponseBodyLifetimeBudget, admission: Option, } @@ -43,6 +46,7 @@ impl MigrationRequestContext { fn new(limits: RuntimeIsolationLimits) -> Self { Self { request_body: RequestBodyBudget::new(limits), + response_body_lifetime: ResponseBodyLifetimeBudget::new(limits), admission: None, } } @@ -196,6 +200,11 @@ fn body_rejection_to_pingora(rejection: BodyLimitExceeded) -> Box { ) } +fn response_body_lifetime_to_pingora(rejection: ResponseBodyLifetimeExceeded) -> Box { + let _ = (rejection.elapsed, rejection.limit); + Error::new_up(ErrorType::Custom("UpstreamResponseBodyLifetimeExceeded")) +} + fn unmatched_route_to_pingora(_error: MigrationGatewayProxyError) -> Box { Error::explain( ErrorType::HTTPStatus(404), @@ -276,14 +285,30 @@ impl ProxyHttp for MigrationGatewayProxy { &self, _session: &mut Session, upstream_response: &mut ResponseHeader, - _ctx: &mut Self::CTX, + ctx: &mut Self::CTX, ) -> pingora::Result<()> where Self::CTX: Send + Sync, { + if !upstream_response.status.is_informational() { + ctx.response_body_lifetime.start(Instant::now()); + } self.apply_response_headers(upstream_response) } + fn upstream_response_body_filter( + &self, + _session: &mut Session, + _body: &mut Option, + _end_of_stream: bool, + ctx: &mut Self::CTX, + ) -> pingora::Result> { + ctx.response_body_lifetime + .reject_if_expired(Instant::now()) + .map_err(response_body_lifetime_to_pingora)?; + Ok(None) + } + async fn logging(&self, session: &mut Session, error: Option<&Error>, ctx: &mut Self::CTX) where Self::CTX: Send + Sync, @@ -294,13 +319,17 @@ impl ProxyHttp for MigrationGatewayProxy { #[cfg(test)] mod tests { - use pingora::prelude::ErrorType; + use std::time::Duration; + + use pingora::prelude::{ErrorSource, ErrorType}; use super::{ - body_rejection_to_pingora, unmatched_route_to_pingora, MigrationGatewayProxyError, - MigrationRequestContext, + body_rejection_to_pingora, response_body_lifetime_to_pingora, + unmatched_route_to_pingora, MigrationGatewayProxyError, MigrationRequestContext, + }; + use crate::runtime_isolation::{ + BodyLimitExceeded, ResponseBodyLifetimeExceeded, RuntimeIsolationLimits, }; - use crate::runtime_isolation::{BodyLimitExceeded, RuntimeIsolationLimits}; #[test] fn migration_context_starts_without_an_admission_lease() { @@ -322,5 +351,15 @@ mod tests { request_path: "/missing".to_string(), }); assert_eq!(route_error.etype, ErrorType::HTTPStatus(404)); + + let lifetime_error = response_body_lifetime_to_pingora(ResponseBodyLifetimeExceeded { + elapsed: Duration::from_millis(301), + limit: Duration::from_millis(300), + }); + assert_eq!( + lifetime_error.etype, + ErrorType::Custom("UpstreamResponseBodyLifetimeExceeded") + ); + assert_eq!(lifetime_error.esource, ErrorSource::Upstream); } } From 94c4ec59e5b9b3d07fdec21abca8f67f4bb2a899 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 3 Sep 2026 08:29:08 +0900 Subject: [PATCH 05/49] fix: preserve pg-erd v1 while admitting explicit v2 lifetime --- src/migration_admin.rs | 22 +++++++++++----------- 1 file changed, 11 insertions(+), 11 deletions(-) diff --git a/src/migration_admin.rs b/src/migration_admin.rs index e15d06e7..fcb0e160 100644 --- a/src/migration_admin.rs +++ b/src/migration_admin.rs @@ -22,10 +22,11 @@ use crate::migration_plan::EdgeMigrationPlan; use crate::migration_proxy::MigrationGatewayProxy; use crate::runtime_isolation::{RuntimeIsolationConfigError, RuntimeIsolationLimits}; -const PG_ERD_LEGACY_MIGRATION_CONFIG_VERSION: u32 = 1; +/// Original bounded `pg-erd-cloud` migration configuration version without a body-lifetime budget. +pub const PG_ERD_MIGRATION_CONFIG_VERSION: u32 = 1; -/// Current version of the bounded `pg-erd-cloud` migration admin configuration. -pub const PG_ERD_MIGRATION_CONFIG_VERSION: u32 = 2; +/// Opt-in pg-erd configuration version that requires an explicit response-body lifetime budget. +pub const PG_ERD_RESPONSE_LIFETIME_CONFIG_VERSION: u32 = 2; /// Fail-closed admin configuration for the characterized `pg-erd-cloud` migration runtime. #[derive(Debug, Clone, Deserialize, PartialEq, Eq)] @@ -51,10 +52,7 @@ pub enum PgErdMigrationConfigError { /// The configuration requests a migration-admin version this binary does not implement. #[error("unsupported pg-erd migration configuration version {0}")] UnsupportedVersion(u32), - /// Current version 2 requires an explicit response-body lifetime instead of a hidden default. - #[error("pg-erd migration config version 2 requires max_upstream_response_body_ms")] - MissingUpstreamResponseBodyLifetime, - /// Legacy version 1 cannot silently acquire semantics introduced by version 2. + /// Version 1 cannot silently acquire semantics introduced by the version-2 contract. #[error("max_upstream_response_body_ms requires pg-erd migration config version 2")] ResponseBodyLifetimeRequiresVersion2, /// A port-zero traffic listener would delegate the public authority to an ephemeral OS port. @@ -131,7 +129,7 @@ impl PgErdMigrationConfig { self.metrics_listener } - /// Returns the current response-body lifetime budget, or `None` for legacy version 1. + /// Returns the explicit response-body lifetime when the version-2 contract is active. pub fn max_upstream_response_body_ms(&self) -> Option { self.max_upstream_response_body_ms } @@ -154,14 +152,16 @@ impl PgErdMigrationConfig { fn validate(&self) -> Result<(), PgErdMigrationConfigError> { match self.version { - PG_ERD_LEGACY_MIGRATION_CONFIG_VERSION => { + PG_ERD_MIGRATION_CONFIG_VERSION => { if self.max_upstream_response_body_ms.is_some() { return Err(PgErdMigrationConfigError::ResponseBodyLifetimeRequiresVersion2); } } - PG_ERD_MIGRATION_CONFIG_VERSION => { + PG_ERD_RESPONSE_LIFETIME_CONFIG_VERSION => { if self.max_upstream_response_body_ms.is_none() { - return Err(PgErdMigrationConfigError::MissingUpstreamResponseBodyLifetime); + return Err(PgErdMigrationConfigError::UnsupportedVersion( + PG_ERD_RESPONSE_LIFETIME_CONFIG_VERSION, + )); } } unsupported => return Err(PgErdMigrationConfigError::UnsupportedVersion(unsupported)), From 3c84f1ef5c81daf58fe1c2602c0edafcbd1bbacc Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 3 Sep 2026 08:29:21 +0900 Subject: [PATCH 06/49] test: cover pg-erd response lifetime config version --- tests/pg_erd_response_lifetime_config.rs | 59 ++++++++++++++++++++++++ 1 file changed, 59 insertions(+) create mode 100644 tests/pg_erd_response_lifetime_config.rs diff --git a/tests/pg_erd_response_lifetime_config.rs b/tests/pg_erd_response_lifetime_config.rs new file mode 100644 index 00000000..b4c11290 --- /dev/null +++ b/tests/pg_erd_response_lifetime_config.rs @@ -0,0 +1,59 @@ +use cwl_pingora_gateway::migration_admin::{ + PgErdMigrationConfig, PgErdMigrationConfigError, PG_ERD_MIGRATION_CONFIG_VERSION, + PG_ERD_RESPONSE_LIFETIME_CONFIG_VERSION, +}; +use cwl_pingora_gateway::runtime_isolation::RuntimeIsolationConfigError; + +fn config_yaml(version: u32, lifetime_line: &str) -> String { + format!( + "version: {version}\nlistener: 127.0.0.1:18080\nmetrics_listener: 127.0.0.1:19090\nmax_request_body_bytes: 1024\nmax_in_flight_requests: 8\n{lifetime_line}upstream_keepalive_pool_size: 4\nupstreams:\n - name: backend\n address: 127.0.0.1:18000\n tls: false\n timeouts:\n connection_ms: 100\n total_connection_ms: 200\n read_ms: 300\n write_ms: 400\n idle_ms: 500\n - name: frontend\n address: 127.0.0.1:13000\n tls: false\n timeouts:\n connection_ms: 100\n total_connection_ms: 200\n read_ms: 300\n write_ms: 400\n idle_ms: 500\n" + ) +} + +#[test] +fn version_two_requires_and_preserves_an_explicit_positive_response_body_lifetime() { + let configured = PgErdMigrationConfig::from_yaml(&config_yaml( + PG_ERD_RESPONSE_LIFETIME_CONFIG_VERSION, + "max_upstream_response_body_ms: 750\n", + )) + .expect("version 2 with an explicit positive lifetime must parse"); + assert_eq!(configured.max_upstream_response_body_ms(), Some(750)); + configured + .build_proxy() + .expect("version-2 runtime budgets must materialize before listeners open"); + + assert_eq!( + PgErdMigrationConfig::from_yaml(&config_yaml( + PG_ERD_RESPONSE_LIFETIME_CONFIG_VERSION, + "max_upstream_response_body_ms: 0\n", + )), + Err(PgErdMigrationConfigError::RuntimeIsolation( + RuntimeIsolationConfigError::ZeroMaxUpstreamResponseBodyMs, + )) + ); + + assert_eq!( + PgErdMigrationConfig::from_yaml(&config_yaml( + PG_ERD_RESPONSE_LIFETIME_CONFIG_VERSION, + "", + )), + Err(PgErdMigrationConfigError::UnsupportedVersion( + PG_ERD_RESPONSE_LIFETIME_CONFIG_VERSION, + )) + ); +} + +#[test] +fn version_one_cannot_silently_acquire_version_two_response_semantics() { + let legacy = PgErdMigrationConfig::from_yaml(&config_yaml(PG_ERD_MIGRATION_CONFIG_VERSION, "")) + .expect("legacy version 1 remains readable without a hidden response lifetime"); + assert_eq!(legacy.max_upstream_response_body_ms(), None); + + assert_eq!( + PgErdMigrationConfig::from_yaml(&config_yaml( + PG_ERD_MIGRATION_CONFIG_VERSION, + "max_upstream_response_body_ms: 750\n", + )), + Err(PgErdMigrationConfigError::ResponseBodyLifetimeRequiresVersion2) + ); +} From eeb669bc228caef572236598aa56f206f1d09a90 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 3 Sep 2026 08:30:09 +0900 Subject: [PATCH 07/49] docs: version pg-erd response lifetime contract --- API_CONFIG_CONTRACT.md | 17 ++++++++++++----- 1 file changed, 12 insertions(+), 5 deletions(-) diff --git a/API_CONFIG_CONTRACT.md b/API_CONFIG_CONTRACT.md index 9a1de672..e8f43947 100644 --- a/API_CONFIG_CONTRACT.md +++ b/API_CONFIG_CONTRACT.md @@ -1,6 +1,6 @@ -# Version 1 Configuration Contracts +# Configuration Contracts -## Generic `cwl-pingora-gateway` +## Generic `cwl-pingora-gateway` version 1 ```yaml version: 1 @@ -25,7 +25,7 @@ upstreams: Unknown fields are rejected. `version` must be `1`. `listener`, `metrics_listener`, and `address` are socket addresses with non-zero ports. Port zero is rejected because this deployment contract requires stable operator-declared listener authority and a concrete connectable upstream rather than OS-selected ephemeral listener ports or unusable upstream destinations. Traffic and metrics listeners must not overlap one effective socket authority, and no upstream may overlap either gateway-owned listener. Equal addresses fail, same-port IPv4/IPv6 wildcard aliases fail, and an IPv6 wildcard plus an IPv4 authority on the same port fails closed because dual-stack bind behavior is platform-dependent. Distinct concrete IP addresses may use the same port. Rejecting listener/upstream overlap prevents a configured origin from recursively targeting the gateway's own traffic socket and prevents ordinary application routing from targeting the internal metrics surface. This does not prohibit two product-owned upstream identities from sharing an endpoint in a future multi-upstream contract; generic v1 still admits exactly one upstream. `max_request_body_bytes`, `max_in_flight_requests`, and `upstream_keepalive_pool_size` must all be positive. Generic v1 requires exactly one upstream and a non-empty stable upstream name. Every timeout must be positive. -The timeout fields map directly to the pinned Pingora peer options rather than defining a second gateway timer model. In particular, `read_ms` is a **per-read inactivity budget**: Pingora waits at most that long for each individual upstream `read()` and resets the timer after a successful read. It is not a total-response deadline. A connected upstream that sends no response bytes is therefore bounded by `read_ms`, while a slow-drip response can remain alive across multiple successful reads. Whole-response lifetime remains an explicit open runtime-isolation requirement and must not be inferred from `read_ms`. +The timeout fields map directly to the pinned Pingora peer options rather than defining a second gateway timer model. In particular, `read_ms` is a **per-read inactivity budget**: Pingora waits at most that long for each individual upstream `read()` and resets the timer after a successful read. It is not a total-response deadline. A connected upstream that sends no response bytes is therefore bounded by `read_ms`, while a slow-drip response can remain alive across multiple successful reads. Generic v1 still has no whole-response lifetime and must not infer one from `read_ms`. `max_in_flight_requests` is a process-local backpressure boundary for non-health downstream requests. When the budget is exhausted, the runtime fails fast with HTTP 503 instead of admitting unbounded work. `/livez` and `/readyz` bypass this application admission budget so saturation does not hide process health. The admission lease is released when the request context ends, including failed requests. `upstream_keepalive_pool_size` is wired directly into Pingora's `ServerConf`; the runtime does not inherit Pingora's framework default of 128 reusable upstream connections. @@ -37,14 +37,15 @@ Generic v1 downstream transport is cleartext TCP. Before proxying, the generic a ## Bounded `cwl-pingora-pg-erd-migration` candidate -The dedicated pg-erd migration binary consumes a different, migration-specific Admin Config profile. It deliberately reuses the same top-level deployment value names while admitting exactly two fixed transport authorities: +The dedicated pg-erd migration binary consumes a different, migration-specific Admin Config profile. Version 1 remains readable only to preserve the existing unreleased characterization stack. Version 2 is the opt-in response-lifetime increment and requires an explicit `max_upstream_response_body_ms`; version 1 rejects that field so the old contract cannot silently acquire new timing semantics. ```yaml -version: 1 +version: 2 listener: 0.0.0.0:6188 metrics_listener: 127.0.0.1:6192 max_request_body_bytes: 1048576 max_in_flight_requests: 128 +max_upstream_response_body_ms: 30000 upstream_keepalive_pool_size: 32 upstreams: - name: backend @@ -67,6 +68,12 @@ upstreams: idle_ms: 10000 ``` +The numeric value above is an illustrative configuration example, not a pg-erd production SLO. A deployment owner must choose the value from its observed long-response contract before canary or cutover. Version 2 rejects zero or a missing response-body lifetime rather than substituting a hidden default. + +`max_upstream_response_body_ms` starts when Pingora reaches the final downstream response-header filter for the admitted upstream response. At each upstream response-body progress callback, Runtime Isolation compares elapsed monotonic time with the configured lifetime. Once the lifetime is reached, the callback raises an upstream-scoped fatal error. If the response status/header was already committed, the gateway terminates that downstream response instead of inventing a second status or silently routing to the other pg-erd origin. The ordinary request context then drops its in-flight admission lease. + +This callback guard is deliberately not described as an exact timer interrupt. At the pinned Pingora revision, `read_ms` still applies independently to each upstream read and resets after a successful read. A continuously progressing body is therefore stopped at the first body callback at or beyond `max_upstream_response_body_ms`; a response that becomes quiescent is bounded by `read_ms`. The current callback surface does not wake a pending read at the absolute body-lifetime instant, and slow-drip of an incomplete **response header** remains a separate transport gap. Neither limitation may be hidden in parity or production-SLO claims. + This is not a generic multi-route configuration language. Operator input can bind only concrete transport/TLS values for the compiled `backend` and `frontend` identities. Missing, extra, duplicate, renamed, port-zero, or otherwise invalid transport authority fails closed before listener activation. The migration profile reuses the same effective network-authority invariant as the generic contract: traffic and metrics listeners cannot overlap, and neither characterized upstream may overlap the traffic listener or metrics listener through an exact address, a same-port wildcard alias, or the conservative IPv6-wildcard/IPv4 dual-stack case. Distinct concrete IP authorities on the same port remain valid. Port zero is rejected because this deployment contract requires stable operator-declared socket authority rather than an OS-selected ephemeral listener or an unusable upstream destination. Routes and edge-owned response fields are not configurable: the characterized profile fixes exact `/healthz -> backend`, raw `PathPrefix(`/api`) -> backend` semantics including `/apiary`, fallback `/ -> frontend`, and the four captured response fields `X-Content-Type-Options: nosniff`, `X-Frame-Options: DENY`, `Referrer-Policy: no-referrer`, and `Permissions-Policy: geolocation=(), microphone=(), camera=()`. Admin parsing validates only deterministic configuration and authority invariants. It does not read custom trust-bundle bytes. If an admitted TLS upstream supplies `trust_bundle_file`, the canonical Pingora peer adapter reads and parses that material exactly once during `build_proxy`, still before listeners are registered. An unreadable or invalid bundle therefore blocks activation without a validate-then-reload trust-file window. From 472e1872e11015f3be9d8f7c4cb3101fe666d287 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 3 Sep 2026 08:30:40 +0900 Subject: [PATCH 08/49] docs: define pg-erd response-body runtime budget --- TRD.md | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/TRD.md b/TRD.md index ef8cebac..2403a8cc 100644 --- a/TRD.md +++ b/TRD.md @@ -8,11 +8,11 @@ Rust edition 2021 with minimum Rust `1.98.0`. Pingora and `pingora-prometheus` a Configuration version 1 is strict YAML with unknown fields rejected. The generic aggregate requires an explicit non-zero traffic listener, a distinct non-zero metrics listener, positive `max_request_body_bytes`, `max_in_flight_requests`, and `upstream_keepalive_pool_size`, plus exactly one explicit upstream. Listener/metrics network authority rejects equal sockets, same-port wildcard aliases, and the platform-dependent IPv6-wildcard/IPv4 same-port overlap while preserving distinct concrete non-zero IP authorities. Every admitted upstream must also remain disjoint from both gateway-owned listeners under the same effective socket-authority semantics. -Each upstream has a stable name, concrete non-zero socket address, `tls`, optional `sni`, optional absolute PEM trust-bundle path, and explicit positive connection/total-connection/read/write/idle timeout budgets. TLS upstreams require SNI; cleartext upstreams reject SNI and custom trust bundles. Peer materialization keeps certificate and hostname verification enabled and uses HTTP/1.1 upstream transport for the current contract. No downstream request may choose an upstream dynamically. +Each upstream has a stable name, concrete non-zero socket address, `tls`, optional `sni`, optional absolute PEM trust-bundle path, and explicit positive connection/total-connection/read/write/idle timeout budgets. TLS upstreams require SNI; cleartext upstreams reject SNI and custom trust bundles. Peer materialization keeps certificate and hostname verification enabled and uses HTTP/1.1 upstream transport for the current contract. No downstream request may choose an upstream dynamically. Pingora `read_timeout` remains a per-read inactivity timer, not an overall response deadline; generic v1 therefore still has no response-body lifetime contract. ## Bounded pg-erd migration config -`PgErdMigrationConfig` is not a generic multi-route language. It exposes only deployment-variable listener/metrics sockets, positive body/in-flight/keepalive budgets, and concrete transport/TLS bindings for the fixed `backend` and `frontend` identities admitted by the characterized pg-erd migration plan. Route precedence, response-security policy, admitted upstream names, product authentication/authorization, service discovery, and business routing are not operator-configurable. Parsing validates authority without loading trust bytes; `build_proxy` materializes all peers and custom trust before listeners may obtain network authority. Both admitted pg-erd upstreams must remain disjoint from traffic and metrics listener authority under the shared exact/wildcard/dual-stack overlap rule. +`PgErdMigrationConfig` is not a generic multi-route language. Version 1 remains the existing unreleased characterization profile. Version 2 is a narrow runtime-isolation increment: it accepts the same deployment-variable listener/metrics sockets, positive body/in-flight/keepalive budgets, and concrete transport/TLS bindings for the fixed `backend` and `frontend` identities, and additionally requires a positive explicit `max_upstream_response_body_ms`. Version 1 rejects that field, so timing semantics cannot change without changing the configuration version. Route precedence, response-security policy, admitted upstream names, product authentication/authorization, service discovery, and business routing are not operator-configurable. Parsing validates authority without loading trust bytes; `build_proxy` materializes all peers and custom trust before listeners may obtain network authority. Both admitted pg-erd upstreams must remain disjoint from traffic and metrics listener authority under the shared exact/wildcard/dual-stack overlap rule. ## Request and forwarding policy @@ -24,6 +24,8 @@ The pg-erd migration uses the separate `forwarding_policy` trust boundary. It di Requests with a parseable `Content-Length` larger than `max_request_body_bytes` fail with HTTP 413 before upstream selection. Streamed chunks are counted against the same bound and fail with HTTP 413 when the cumulative body crosses it. Non-health application requests share the process in-flight admission budget and fail fast with HTTP 503 when it is exhausted; `/livez` and `/readyz` remain outside that application-capacity budget. A configurable smaller header budget, per-route budget, and origin-capacity budget are not yet implemented. +Pg-erd config version 2 starts its monotonic response-body lifetime when the final upstream response reaches `response_filter`. Each `upstream_response_body_filter` progress callback rejects an elapsed lifetime at or beyond `max_upstream_response_body_ms` as an upstream-scoped fatal error. This complements, rather than replaces, the per-read `read_ms` inactivity timer. With the pinned Pingora callback API the body-lifetime guard is progress-driven: it stops continuous slow-drip at the first body callback after the deadline, while a quiescent pending read can run until `read_ms`. It therefore must not be described as an exact timer interrupt. Slow-drip of an incomplete response header remains a separate transport gap. + ## Protocol-transition admission HTTP/1 connection-wide protocol transitions are not part of generic v1 or the bounded pg-erd candidate. A request containing an `Upgrade` field, or containing a case-insensitive `upgrade` token in any comma-delimited `Connection` field value, fails with HTTP 501 before application admission, route/upstream selection, or origin contact. Unrelated tokens such as `x-upgrade` do not match. Gateway-local `/livez` and `/readyz` remain local HTTP responses and therefore do not acquire origin network authority. Independently, every materialized Pingora peer uses `HttpUpstreamRequestPolicy::deny_upgrades()`, so a later callback or composition refactor cannot silently fall back to Pingora's default `H1UpgradePolicy::WebSocketOnly` below this admission boundary. @@ -36,7 +38,7 @@ This fail-closed policy is intentional rather than a claim that WebSocket is uns The version-1 process policy allows one total upstream attempt and therefore no automatic gateway retry. It overrides Pingora's keepalive-pool default from validated configuration and uses a 5-second grace period plus 10-second Pingora runtime shutdown timeout inside a 30-second external termination budget. Product idempotency/replay policy remains outside the gateway. -Failure semantics are phase-aware. Before an upstream response header is committed downstream, transport failure may be represented as the gateway's fail-closed error response under the configured attempt policy. After a valid response header has already been committed, a later upstream framing/body failure cannot be rewritten into a second HTTP status or silently failed over; the downstream response is terminated, low-cardinality error telemetry records the failed request, process readiness remains available, and independent routes must remain usable. This is a transport invariant, not product retry policy. +Failure semantics are phase-aware. Before an upstream response header is committed downstream, transport failure may be represented as the gateway's fail-closed error response under the configured attempt policy. After a valid response header has already been committed, a later upstream framing/body failure—including a version-2 response-body lifetime breach—cannot be rewritten into a second HTTP status or silently failed over; the downstream response is terminated, low-cardinality error telemetry records the failed request, process readiness remains available, the request context releases its in-flight lease, and independent routes must remain usable. This is a transport invariant, not product retry policy. ## Performance evidence From b7ade63020fa8126cf2b94de291054e8c4b6b10f Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 3 Sep 2026 08:31:22 +0900 Subject: [PATCH 09/49] docs: add pg-erd slow-drip acceptance --- TEST_STRATEGY.md | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/TEST_STRATEGY.md b/TEST_STRATEGY.md index ec83d3dd..e28ae319 100644 --- a/TEST_STRATEGY.md +++ b/TEST_STRATEGY.md @@ -20,13 +20,15 @@ Tests are organized by bounded responsibility and traffic phase rather than by i `tests/pg_erd_route_contract.rs` freezes the observed Traefik route precedence, including literal raw `/api` prefix behavior. `tests/pg_erd_http_policy_contract.rs` freezes the separately owned response-security fields and values, case-insensitive field identity, absent lookup, duplicate-field rejection, empty/invalid name/value rejection, and CR/LF injection rejection. `tests/pg_erd_migration_plan_contract.rs`, `tests/pg_erd_upstream_binding_contract.rs`, `tests/pg_erd_forwarding_contract.rs`, and `tests/pg_erd_runtime_proxy_contract.rs` prove the transport-neutral plan, exact upstream binding, forwarding-trust boundary, response-policy composition, and shared runtime-isolation/observability callback behavior. -The bounded Admin Config transition is separately executable. `tests/pg_erd_admin_config_contract.rs` rejects unknown/future configuration, traffic/metrics listener collision, zero runtime/keepalive budgets, missing/extra/duplicate/renamed authority, invalid concrete transport configuration, and any characterized `backend`/`frontend` transport authority that overlaps the gateway's traffic or metrics listener under the shared exact/wildcard/dual-stack semantics. `tests/pg_erd_binary_startup.rs` exercises the dedicated compiled process and requires fail-closed behavior for omitted/unreadable configuration, invalid Admin Config, custom TLS trust material that cannot be materialized before listener activation, and listener/upstream self-loop before network activation. +The bounded Admin Config transition is separately executable. `tests/pg_erd_admin_config_contract.rs` preserves the version-1 characterization contract and rejects unknown/future configuration, traffic/metrics listener collision, zero runtime/keepalive budgets, missing/extra/duplicate/renamed authority, invalid concrete transport configuration, and any characterized `backend`/`frontend` transport authority that overlaps the gateway's traffic or metrics listener under the shared exact/wildcard/dual-stack semantics. `tests/pg_erd_response_lifetime_config.rs` covers the opt-in version-2 increment: it requires an explicit positive `max_upstream_response_body_ms`, rejects zero, and proves version 1 cannot silently acquire version-2 timing semantics. `tests/pg_erd_binary_startup.rs` exercises the dedicated compiled process and requires fail-closed behavior for omitted/unreadable configuration, invalid Admin Config, custom TLS trust material that cannot be materialized before listener activation, and listener/upstream self-loop before network activation. On Unix, `tests/pg_erd_local_ca_tls.rs` separately proves the bounded composition root consumes its admitted upstream TLS contract rather than inheriting generic evidence. It generates an ephemeral one-day local CA and `backend.test` certificate, configures only `backend` as TLS and keeps `frontend` a distinct clear-text authority, then requires characterized `/api` traffic to succeed with matching explicit trust and SNI. A second process keeps the CA valid while changing only backend SNI; the routed backend request must fail as HTTP 502 without route failover, gateway-local `/readyz` must remain 200, and a later fallback request must still reach `frontend`. The fixture proves gateway-to-upstream TLS trust/hostname verification only. It does not claim downstream TLS termination, certificate issuance/rotation ownership, or production-network TLS latency. `tests/pg_erd_production_path.rs` starts real loopback `backend` and `frontend` origins plus `cwl-pingora-pg-erd-migration`. Gateway-local `/livez` and `/readyz` remain local, consumer `/healthz` and raw `/apiary` reach `backend`, fallback product traffic reaches `frontend`, hostile forwarding identity is discarded and rebuilt from accepted loopback transport/Host evidence, an upstream `X-Frame-Options: SAMEORIGIN` is replaced by characterized `DENY` together with the other response fields, and an over-limit declared body fails with 413 before origin delivery. -`tests/pg_erd_runtime_isolation_traffic.rs` adds chunked overflow and in-flight saturation/recovery. `tests/pg_erd_upstream_failure_traffic.rs` covers connection refusal: HTTP 502 within the connection budget, healthy readiness, low-cardinality error telemetry, and independent fallback recovery. `tests/pg_erd_read_stall_traffic.rs` covers an accepted connection that emits no response bytes: the configured per-read `read_ms` inactivity budget must fail as 502 within a conservative outer bound, without transferring evidence to slow-drip or whole-response lifetime behavior. +`tests/pg_erd_runtime_isolation_traffic.rs` adds chunked overflow and in-flight saturation/recovery. `tests/pg_erd_upstream_failure_traffic.rs` covers connection refusal: HTTP 502 within the connection budget, healthy readiness, low-cardinality error telemetry, and independent fallback recovery. `tests/pg_erd_read_stall_traffic.rs` covers an accepted connection that emits no response bytes: the configured per-read `read_ms` inactivity budget must fail as 502 within a conservative outer bound. + +`tests/pg_erd_slow_drip_response_traffic.rs` distinguishes that inactivity contract from the version-2 response-body lifetime. The backend commits HTTP 200 with `Content-Length: 20` and sends one body byte every 60 ms while `read_ms` is 120 ms, so every individual read can keep making progress. The version-2 config separately declares `max_upstream_response_body_ms: 300`; the gateway must terminate the committed response before all 20 bytes arrive, retain the original committed status rather than inventing a second status/failover, increment request-error telemetry, keep `/readyz` available, and serve a later independent frontend route. The elapsed-time guard is callback-driven: the fixture proves continuous body slow-drip is bounded, not that Pingora can interrupt a pending read at an exact absolute deadline. Slow-drip of an incomplete response header remains outside this test. `tests/pg_erd_partial_response_traffic.rs` covers the orderly post-header phase: the backend commits HTTP 200 plus `Content-Length: 20`, sends only `partial`, then closes. The downstream must preserve the committed status/framing and terminate incomplete rather than receiving a second status or silent failover; readiness, error telemetry, and an independent route remain usable. On Linux, `tests/pg_erd_upstream_reset_traffic.rs` covers a real pre-header TCP RST using `SO_LINGER(0)`, requiring prompt 502 without failover or waiting for the read inactivity budget. `tests/pg_erd_post_commit_reset_traffic.rs` moves the same abortive close past downstream response commitment and requires the already-committed response boundary to be preserved. These tests characterize ordinary HTTP failure phases; they do not provide WebSocket or generic long-lived-stream support. @@ -46,4 +48,4 @@ The OCI gate builds one digest-pinned distroless image and exercises both compos Every legacy migration begins with behavior characterization and then equivalent Pingora production-path evidence. Static-serving consumers must cover applicable route precedence, SPA fallback, MIME, ETag/cache, Range/HEAD/304/416, redirects/security fields, and compression. Proxy consumers must cover Host/SNI/TLS, forwarding trust, request limits, timeout/retry behavior, streaming/uploads, saturation/backpressure, errors, health/readiness, drain, and process-wide payload-safe diagnostics. WebSocket/Upgrade is a separate versioned capability and cannot be inferred from a supplier feature flag or a lightly loaded happy-path test. -Release-quality gaps remain: no downstream TLS listener contract; no HTTP/2 or HTTP/3 parity; no tracing evidence; no slow-drip/whole-response lifetime control; no representative origin-capacity/TLS/network/multi-hop/container-orchestrator benchmark; no immutable published registry digest/provenance with rehearsed rollback; no shadow/canary/cutover/legacy-removal evidence; and no benchmark against the replaced Nginx/Traefik production path. Supplier/security/dependency blockers remain fail-closed. All source acceptances, including pg-erd upstream TLS trust/hostname-failure recovery, protocol-transition rejection, peer-level Upgrade denial, dependency-diagnostic redaction, and the Rust-origin latency lane, must be reacquired on the exact final head after any source or documentation movement. +Release-quality gaps remain: no downstream TLS listener contract; no HTTP/2 or HTTP/3 parity; no tracing evidence; no response-header slow-drip deadline and no exact timer-interrupt guarantee beyond the combined response-body progress guard plus `read_ms`; no representative origin-capacity/TLS/network/multi-hop/container-orchestrator benchmark; no immutable published registry digest/provenance with rehearsed rollback; no shadow/canary/cutover/legacy-removal evidence; and no benchmark against the replaced Nginx/Traefik production path. Supplier/security/dependency blockers remain fail-closed. All source acceptances, including the new pg-erd version-2 response-body lifetime, upstream TLS trust/hostname-failure recovery, protocol-transition rejection, peer-level Upgrade denial, dependency-diagnostic redaction, and the Rust-origin latency lane, must be reacquired on the exact final head after any source or documentation movement. From 15dd1f0b448f4b7a32748dc2efe09414f05b693f Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 3 Sep 2026 08:32:10 +0900 Subject: [PATCH 10/49] test: avoid origin-close timing assertion --- tests/pg_erd_slow_drip_response_traffic.rs | 17 +++++------------ 1 file changed, 5 insertions(+), 12 deletions(-) diff --git a/tests/pg_erd_slow_drip_response_traffic.rs b/tests/pg_erd_slow_drip_response_traffic.rs index 443b04db..7668240b 100644 --- a/tests/pg_erd_slow_drip_response_traffic.rs +++ b/tests/pg_erd_slow_drip_response_traffic.rs @@ -1,10 +1,9 @@ -//! Real-listener RED contract for a pg-erd upstream that continuously drips response-body bytes. +//! Real-listener RED→GREEN contract for a pg-erd upstream that continuously drips response-body bytes. //! //! Pingora's peer `read_timeout` is an inactivity timeout that resets after each successful read. //! This fixture therefore keeps each origin write inside `read_ms` while extending the response -//! beyond an explicit migration-owned response-body lifetime. The current contract cannot express -//! that lifetime; the test is expected to stay RED until the versioned admin/runtime boundary owns -//! and enforces it without retrying or failing over after the response has been committed. +//! beyond an explicit migration-owned response-body lifetime. The version-2 admin/runtime boundary +//! must terminate that body without retrying or failing over after the response has been committed. use std::io::{ErrorKind, Read, Write}; use std::net::{SocketAddr, TcpListener, TcpStream}; @@ -168,10 +167,9 @@ fn compiled_pg_erd_terminates_continuous_response_drip_without_poisoning_other_r ) .expect("backend response header should be writable"); - let mut writes = 0_usize; for _ in 0..20 { match stream.write_all(b"x") { - Ok(()) => writes += 1, + Ok(()) => {} Err(error) if matches!( error.kind(), @@ -184,7 +182,6 @@ fn compiled_pg_erd_terminates_continuous_response_drip_without_poisoning_other_r } thread::sleep(Duration::from_millis(60)); } - writes }); let frontend = TcpListener::bind("127.0.0.1:0").expect("frontend fixture should bind"); @@ -261,11 +258,7 @@ fn compiled_pg_erd_terminates_continuous_response_drip_without_poisoning_other_r frontend_origin .join() .expect("frontend recovery fixture should complete"); - let writes = backend_origin + backend_origin .join() .expect("slow-drip backend fixture should complete"); - assert!( - writes < 20, - "the gateway must close the over-budget origin response before all drip bytes are accepted" - ); } From f80e3eccea9cd1a09c33fc96d38359632e650bbb Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 3 Sep 2026 08:35:01 +0900 Subject: [PATCH 11/49] test: cover dormant configured response lifetime --- src/runtime_isolation.rs | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/src/runtime_isolation.rs b/src/runtime_isolation.rs index 8da32e2b..abec4ebd 100644 --- a/src/runtime_isolation.rs +++ b/src/runtime_isolation.rs @@ -325,6 +325,10 @@ mod tests { .expect("fixture limits are valid"); let mut budget = ResponseBodyLifetimeBudget::new(limits); let started = Instant::now(); + assert!(budget + .reject_if_expired(started + Duration::from_secs(1)) + .is_ok()); + budget.start(started); budget.start(started + Duration::from_millis(250)); From 4f6da1def9878a947ff71a6b973aaa02a05a7062 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 3 Sep 2026 08:35:32 +0900 Subject: [PATCH 12/49] fix: start lifetime before upstream body callbacks --- src/migration_proxy.rs | 61 +++++++++++++++++++++++++++++++++++++----- 1 file changed, 55 insertions(+), 6 deletions(-) diff --git a/src/migration_proxy.rs b/src/migration_proxy.rs index f2bb2e4c..af799c21 100644 --- a/src/migration_proxy.rs +++ b/src/migration_proxy.rs @@ -205,6 +205,16 @@ fn response_body_lifetime_to_pingora(rejection: ResponseBodyLifetimeExceeded) -> Error::new_up(ErrorType::Custom("UpstreamResponseBodyLifetimeExceeded")) } +fn start_response_body_lifetime( + is_informational: bool, + ctx: &mut MigrationRequestContext, + now: Instant, +) { + if !is_informational { + ctx.response_body_lifetime.start(now); + } +} + fn unmatched_route_to_pingora(_error: MigrationGatewayProxyError) -> Box { Error::explain( ErrorType::HTTPStatus(404), @@ -281,7 +291,7 @@ impl ProxyHttp for MigrationGatewayProxy { self.apply_upstream_request_policy(upstream_request, &forwarding) } - async fn response_filter( + async fn upstream_response_filter( &self, _session: &mut Session, upstream_response: &mut ResponseHeader, @@ -290,9 +300,23 @@ impl ProxyHttp for MigrationGatewayProxy { where Self::CTX: Send + Sync, { - if !upstream_response.status.is_informational() { - ctx.response_body_lifetime.start(Instant::now()); - } + start_response_body_lifetime( + upstream_response.status.is_informational(), + ctx, + Instant::now(), + ); + Ok(()) + } + + async fn response_filter( + &self, + _session: &mut Session, + upstream_response: &mut ResponseHeader, + _ctx: &mut Self::CTX, + ) -> pingora::Result<()> + where + Self::CTX: Send + Sync, + { self.apply_response_headers(upstream_response) } @@ -319,13 +343,14 @@ impl ProxyHttp for MigrationGatewayProxy { #[cfg(test)] mod tests { - use std::time::Duration; + use std::time::{Duration, Instant}; use pingora::prelude::{ErrorSource, ErrorType}; use super::{ body_rejection_to_pingora, response_body_lifetime_to_pingora, - unmatched_route_to_pingora, MigrationGatewayProxyError, MigrationRequestContext, + start_response_body_lifetime, unmatched_route_to_pingora, MigrationGatewayProxyError, + MigrationRequestContext, }; use crate::runtime_isolation::{ BodyLimitExceeded, ResponseBodyLifetimeExceeded, RuntimeIsolationLimits, @@ -339,6 +364,30 @@ mod tests { assert!(ctx.admission.is_none()); } + #[test] + fn informational_headers_do_not_start_the_response_body_lifetime() { + let limits = RuntimeIsolationLimits::try_new_with_response_body_limit(8, 1, 300) + .expect("fixture limits are valid"); + let mut ctx = MigrationRequestContext::new(limits); + let now = Instant::now(); + + start_response_body_lifetime(true, &mut ctx, now); + assert!(ctx + .response_body_lifetime + .reject_if_expired(now + Duration::from_secs(1)) + .is_ok()); + + start_response_body_lifetime(false, &mut ctx, now); + assert!(ctx + .response_body_lifetime + .reject_if_expired(now + Duration::from_millis(299)) + .is_ok()); + assert!(ctx + .response_body_lifetime + .reject_if_expired(now + Duration::from_millis(300)) + .is_err()); + } + #[test] fn delivery_errors_map_to_fail_closed_http_errors() { let body_error = body_rejection_to_pingora(BodyLimitExceeded { From 2f207c5206aafef5397fbfc2a0ba35186d437da7 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 3 Sep 2026 08:36:05 +0900 Subject: [PATCH 13/49] test: isolate lifetime from per-read timeout --- tests/pg_erd_slow_drip_response_traffic.rs | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/pg_erd_slow_drip_response_traffic.rs b/tests/pg_erd_slow_drip_response_traffic.rs index 7668240b..11c4725e 100644 --- a/tests/pg_erd_slow_drip_response_traffic.rs +++ b/tests/pg_erd_slow_drip_response_traffic.rs @@ -1,7 +1,7 @@ //! Real-listener RED→GREEN contract for a pg-erd upstream that continuously drips response-body bytes. //! //! Pingora's peer `read_timeout` is an inactivity timeout that resets after each successful read. -//! This fixture therefore keeps each origin write inside `read_ms` while extending the response +//! This fixture therefore keeps each origin write well inside `read_ms` while extending the response //! beyond an explicit migration-owned response-body lifetime. The version-2 admin/runtime boundary //! must terminate that body without retrying or failing over after the response has been committed. @@ -44,7 +44,7 @@ fn write_config( let mut file = NamedTempFile::new().expect("temporary config should be writable"); writeln!( file, - "version: 2\nlistener: {listener}\nmetrics_listener: {metrics_listener}\nmax_request_body_bytes: 8\nmax_in_flight_requests: 8\nmax_upstream_response_body_ms: 300\nupstream_keepalive_pool_size: 4\nupstreams:\n - name: backend\n address: {backend}\n tls: false\n timeouts:\n connection_ms: 200\n total_connection_ms: 400\n read_ms: 120\n write_ms: 1000\n idle_ms: 5000\n - name: frontend\n address: {frontend}\n tls: false\n timeouts:\n connection_ms: 200\n total_connection_ms: 400\n read_ms: 1000\n write_ms: 1000\n idle_ms: 5000" + "version: 2\nlistener: {listener}\nmetrics_listener: {metrics_listener}\nmax_request_body_bytes: 8\nmax_in_flight_requests: 8\nmax_upstream_response_body_ms: 300\nupstream_keepalive_pool_size: 4\nupstreams:\n - name: backend\n address: {backend}\n tls: false\n timeouts:\n connection_ms: 200\n total_connection_ms: 400\n read_ms: 500\n write_ms: 1000\n idle_ms: 5000\n - name: frontend\n address: {frontend}\n tls: false\n timeouts:\n connection_ms: 200\n total_connection_ms: 400\n read_ms: 1000\n write_ms: 1000\n idle_ms: 5000" ) .expect("migration config should be written"); file From 8fc9e22a9ae207b9ea9ec7e2953f8c4899f25ee4 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 3 Sep 2026 08:36:36 +0900 Subject: [PATCH 14/49] docs: record pg-erd response lifetime decision --- ...0-version-pg-erd-response-body-lifetime.md | 62 +++++++++++++++++++ 1 file changed, 62 insertions(+) create mode 100644 docs/adr/0010-version-pg-erd-response-body-lifetime.md diff --git a/docs/adr/0010-version-pg-erd-response-body-lifetime.md b/docs/adr/0010-version-pg-erd-response-body-lifetime.md new file mode 100644 index 00000000..9df857ba --- /dev/null +++ b/docs/adr/0010-version-pg-erd-response-body-lifetime.md @@ -0,0 +1,62 @@ +# ADR 0010: Version the pg-erd upstream response-body lifetime + +- Status: Proposed +- Date: 2026-09-03 +- Owners: Runtime Isolation / bounded pg-erd migration + +## Problem + +The characterized pg-erd candidate configures Pingora `read_timeout`, but at the pinned Pingora revision `09696b51bc59315353d96686355861604d0bb48c` that setting applies to each individual upstream read and resets after a successful read. A backend can therefore retain an admitted request indefinitely by continuing to send response-body progress before each inactivity timeout. This can consume the process-local in-flight budget even though no single read stalls. + +The existing version-1 pg-erd Admin Config has no whole-response field. Reinterpreting `read_ms`, deriving a hidden multiplier, or reusing `total_connection_ms` would change an existing contract without an explicit version boundary; `total_connection_ms` also belongs to connection establishment, including TLS, rather than response delivery. + +## Constraints + +- Generic gateway v1 must not silently gain product-specific long-response semantics. +- The pg-erd migration may add only edge/runtime authority. Product retry, idempotency, authentication, business routing, Wardnet/EgressWeave verdicts, and Keyverse identity remain outside the gateway. +- Existing pg-erd version-1 fixtures and predecessor PRs must remain readable while the stack is unreleased. +- A failure after the downstream response header is committed cannot be converted into a second HTTP status or silently failed over. +- The pinned Pingora callback API exposes response-header and response-body progress callbacks but does not expose an exact absolute-deadline interrupt for a currently pending upstream read. + +## Options considered + +### Keep only `read_ms` + +Rejected. It protects inactivity, not total response-body lifetime, so continuous slow-drip remains unbounded. + +### Reinterpret `read_ms` or derive a multiplier + +Rejected. This creates undocumented semantics, couples two distinct failure modes, and makes operator intent impossible to audit. + +### Reuse `total_connection_ms` + +Rejected. Pingora applies that budget to connection establishment; changing its meaning at the gateway layer would conflict with supplier semantics. + +### Add an explicit pg-erd version-2 body-lifetime budget + +Selected. Version 2 requires positive `max_upstream_response_body_ms`. Version 1 rejects that field and otherwise retains its existing behavior. Runtime Isolation owns monotonic elapsed-time accounting, while the migration adapter starts the budget on the first non-informational upstream response header and checks it on each upstream response-body callback. + +## Decision + +Introduce pg-erd Admin Config version 2 with mandatory positive `max_upstream_response_body_ms`. Preserve version 1 without a hidden lifetime. On a version-2 request, start the response-body budget at the first non-informational upstream response header. If a later body-progress callback arrives at or beyond the budget, raise an upstream-scoped fatal error. Preserve the existing post-commit invariant: terminate the incomplete downstream response, record low-cardinality request-error telemetry, release the request admission lease with its context, and do not invent retry or failover. + +This is a progress-driven bound, not an exact timer interrupt. A continuously progressing body is stopped at the first body callback at or after the configured lifetime. A quiescent upstream is still bounded by its independent per-read `read_ms`. Slow-drip of an incomplete response header remains a separate gap because the current callback guard has not yet established an absolute header-read deadline. + +## Effects and risks + +The selected design makes slow-drip body retention explicit and auditable and keeps it in the Runtime Isolation bounded context. It avoids changing generic v1 and preserves the pg-erd v1 characterization stack. Operators must choose the version-2 value from observed long-response requirements before canary or cutover; example values are not production SLOs. + +The remaining limitation is timer precision: scheduler delay and Pingora read cadence can move actual termination past the configured instant until the next body callback. A future supplier/runtime capability may justify an exact timer boundary, but this ADR does not claim one. + +## Verification + +- deterministic Runtime Isolation tests cover absent, dormant, active, and expired response-body budgets; +- versioned Admin Config tests cover v1 compatibility, v1 rejection of v2 fields, v2 explicit positive values, zero rejection, and incomplete v2 schema rejection; +- real-listener pg-erd traffic commits HTTP 200 and drips one body byte every 60 ms while `read_ms` is 500 ms and `max_upstream_response_body_ms` is 300 ms; the response must terminate before the declared 20-byte body completes, without a second status or route failover, while error telemetry, readiness, and an independent route recover; +- exact-head format, compile, clippy, rustdoc, 100% owned-production coverage, load, OCI, security, supply-chain, and independent-review evidence remain required before this ADR may move from Proposed to Accepted. + +## References + +Cloudflare. (2026). *Pingora peer configuration and timeout semantics* (revision `09696b51bc59315353d96686355861604d0bb48c`). GitHub. + +Cloudflare. (2026). *ProxyHttp response filtering callbacks* (revision `09696b51bc59315353d96686355861604d0bb48c`). GitHub. From 2f69875b1c4876e304e683b5580610b4044e9619 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 3 Sep 2026 08:37:31 +0900 Subject: [PATCH 15/49] docs: document pg-erd response lifetime operation --- OPERABILITY.md | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/OPERABILITY.md b/OPERABILITY.md index ca3f618b..721f7fdd 100644 --- a/OPERABILITY.md +++ b/OPERABILITY.md @@ -6,16 +6,22 @@ Run the generic process as `cwl-pingora-gateway --config /path/to/gateway.yaml`. The pg-erd migration candidate is a separate executable: `cwl-pingora-pg-erd-migration --config /path/to/pg-erd-migration.yaml`. It consumes the bounded `PgErdMigrationConfig` profile rather than widening generic `GatewayConfig` v1. The profile admits exactly the compiled `backend` and `frontend` transport identities plus deployment-variable sockets, runtime budgets and upstream transport/TLS values. Route tables, response-policy fields, product authentication/business rules, Keyverse identity, Wardnet/EgressWeave verdicts, service discovery and arbitrary destinations are not operator-configurable. +Pg-erd configuration version 1 preserves the existing unreleased characterization behavior and has no response-body lifetime. Version 2 is opt-in and requires a positive `max_upstream_response_body_ms`; version 1 rejects that field rather than silently acquiring new timing semantics. Choose the version-2 value from observed application long-response requirements before canary or cutover. Documentation/example values are not production SLOs. A zero or incomplete version-2 budget fails before listeners open. + Admin parsing is side-effect free with respect to custom trust bytes. It validates the exact transport-authority set and `UpstreamConfig` invariants first. `build_proxy` then materializes Pingora peers and any custom PEM trust bundle once, still before listener registration. This avoids reading mutable trust material in a validation pass and reading it again for activation. Trust bundles are deployment input, not certificate-authority ownership. Mount them read-only from the platform or canonical secret/certificate owner and rotate them by replacing the deployment revision. The gateway does not issue certificates, manage ACME, or write trust material. -## Health and backpressure +## Health, backpressure, and response lifetime Both process identities reserve `/livez` and `/readyz` and return 200 with `Cache-Control: no-store` through the Pingora serving path. Readiness is process/configuration readiness, not upstream reachability. Do not use it as proof that a dependent application is healthy. For the pg-erd migration, consumer `/healthz` remains routed application traffic to `backend`; it is intentionally distinct from the gateway-local probes. `max_in_flight_requests` limits concurrently admitted non-health requests for one gateway process. At capacity the gateway fails new application traffic fast with HTTP 503 and increments `cwl_pingora_gateway_backpressure_rejections_total`; it does not queue unbounded work. Process health probes bypass that admission budget so operators can distinguish process health from traffic saturation. The request lease is released when the Pingora request context ends, including error paths, and a subsequent request is admissible again. +For pg-erd version 2, `max_upstream_response_body_ms` limits the elapsed body phase after the first non-informational upstream response header. It complements the peer's `read_ms`; it does not replace it. `read_ms` remains a per-read inactivity timeout that resets whenever Pingora successfully reads more upstream data. Continuous response-body slow-drip is stopped at the first body-progress callback at or beyond the explicit body lifetime, producing an upstream-scoped request error. If HTTP status/headers were already committed, the gateway terminates that incomplete response rather than sending a second status or switching to `frontend`. The request context then releases its in-flight lease. + +This control is not an exact wall-clock interrupt. With the pinned Pingora callback API, an already pending read is not awakened solely because the body-lifetime instant elapsed; a quiescent response remains bounded by `read_ms`, and continuously progressing traffic is checked when body progress reaches the callback. Slow-drip of an incomplete response header is still a separate gap. Do not advertise the configured number as a strict production deadline until representative deployment traffic has measured the actual scheduler/read-callback behavior. + `upstream_keepalive_pool_size` is copied into Pingora `ServerConf` before bootstrap. Choose it with expected upstream concurrency, origin capacity, instance count and connection reuse in mind. It limits retained reusable upstream connections; it is not a substitute for the downstream in-flight admission limit and does not create product-domain load-balancing semantics. ## Forwarding and protocol boundary @@ -36,7 +42,7 @@ The runtime does not inherit Pingora's retry, keepalive-pool, or drain defaults. SIGTERM uses Pingora graceful termination with an explicit 5-second request-drain grace period and a 10-second runtime-shutdown timeout. The pinned Pingora server calls Tokio `Runtime::shutdown_timeout` with that timeout and then sleeps for the same timeout while service runtimes are shut down in parallel. The policy therefore requires a 30-second supervisor hard-kill budget: its modeled worst-case Pingora process budget is 25 seconds plus scheduler/process-exit overhead. A Kubernetes-style deployment must set `terminationGracePeriodSeconds` to at least 30 or provide an equivalent supervisor budget; a shorter external kill deadline is not an admitted deployment contract. -`tests/graceful_shutdown.rs` exercises the generic compiled binary with a held upstream response. `tests/production_path.rs` covers generic saturation, health and failure recovery. `tests/pg_erd_production_path.rs` exercises the dedicated pg-erd process with real loopback backend/frontend origins, including process-local health, characterized route/response-header behavior, transport-derived forwarding replacement and declared body rejection. `tests/pg_erd_binary_startup.rs` requires missing/invalid configuration and unreadable trust material to fail before listener activation. `tests/pingora_diagnostic_log_safety.rs` runs the compiled generic process with broad trace diagnostics and requires request URI/Authorization/Cookie sentinels to reach the origin without appearing in process stderr. These source contracts become evidence only after terminal success on the exact current head; predecessor success never transfers. +`tests/graceful_shutdown.rs` exercises the generic compiled binary with a held upstream response. `tests/production_path.rs` covers generic saturation, health and failure recovery. `tests/pg_erd_production_path.rs` exercises the dedicated pg-erd process with real loopback backend/frontend origins, including process-local health, characterized route/response-header behavior, transport-derived forwarding replacement and declared body rejection. `tests/pg_erd_slow_drip_response_traffic.rs` separately proves the version-2 body-lifetime behavior with progress faster than `read_ms`. `tests/pg_erd_binary_startup.rs` requires missing/invalid configuration and unreadable trust material to fail before listener activation. `tests/pingora_diagnostic_log_safety.rs` runs the compiled generic process with broad trace diagnostics and requires request URI/Authorization/Cookie sentinels to reach the origin without appearing in process stderr. These source contracts become evidence only after terminal success on the exact current head; predecessor success never transfers. ## Container From 0a66dc8afc463e817ed070dec57c371118dd5fe4 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 3 Sep 2026 08:38:34 +0900 Subject: [PATCH 16/49] fix: distinguish incomplete v2 from future config --- src/migration_admin.rs | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/src/migration_admin.rs b/src/migration_admin.rs index fcb0e160..dca33b29 100644 --- a/src/migration_admin.rs +++ b/src/migration_admin.rs @@ -52,6 +52,9 @@ pub enum PgErdMigrationConfigError { /// The configuration requests a migration-admin version this binary does not implement. #[error("unsupported pg-erd migration configuration version {0}")] UnsupportedVersion(u32), + /// Version 2 must state the response-body lifetime rather than inheriting a hidden default. + #[error("pg-erd migration config version 2 requires max_upstream_response_body_ms")] + MissingUpstreamResponseBodyLifetime, /// Version 1 cannot silently acquire semantics introduced by the version-2 contract. #[error("max_upstream_response_body_ms requires pg-erd migration config version 2")] ResponseBodyLifetimeRequiresVersion2, @@ -159,9 +162,7 @@ impl PgErdMigrationConfig { } PG_ERD_RESPONSE_LIFETIME_CONFIG_VERSION => { if self.max_upstream_response_body_ms.is_none() { - return Err(PgErdMigrationConfigError::UnsupportedVersion( - PG_ERD_RESPONSE_LIFETIME_CONFIG_VERSION, - )); + return Err(PgErdMigrationConfigError::MissingUpstreamResponseBodyLifetime); } } unsupported => return Err(PgErdMigrationConfigError::UnsupportedVersion(unsupported)), From d0acaa8c48e72dffdaa4b229840d8a06742d8c18 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 3 Sep 2026 08:39:02 +0900 Subject: [PATCH 17/49] test: distinguish pg-erd v2 from future config --- tests/pg_erd_admin_config_contract.rs | 16 +++++++++++++--- 1 file changed, 13 insertions(+), 3 deletions(-) diff --git a/tests/pg_erd_admin_config_contract.rs b/tests/pg_erd_admin_config_contract.rs index b90a950f..fbe143c8 100644 --- a/tests/pg_erd_admin_config_contract.rs +++ b/tests/pg_erd_admin_config_contract.rs @@ -1,6 +1,7 @@ use cwl_pingora_gateway::edge_contract::GatewayConfigError; use cwl_pingora_gateway::migration_admin::{ PgErdMigrationConfig, PgErdMigrationConfigError, PG_ERD_MIGRATION_CONFIG_VERSION, + PG_ERD_RESPONSE_LIFETIME_CONFIG_VERSION, }; fn config_yaml(upstreams: &str) -> String { @@ -279,7 +280,7 @@ fn pg_erd_admin_config_rejects_invalid_concrete_transport_contract() { } #[test] -fn pg_erd_admin_config_rejects_unknown_fields_and_future_versions() { +fn pg_erd_admin_config_rejects_unknown_incomplete_and_future_versions() { let unknown = valid_yaml().replace( "max_request_body_bytes: 1048576", "max_request_body_bytes: 1048576\nproduct_auth_mode: embedded", @@ -289,12 +290,21 @@ fn pg_erd_admin_config_rejects_unknown_fields_and_future_versions() { Err(PgErdMigrationConfigError::Parse(_)) )); + let incomplete_v2 = valid_yaml().replace( + &format!("version: {PG_ERD_MIGRATION_CONFIG_VERSION}"), + &format!("version: {PG_ERD_RESPONSE_LIFETIME_CONFIG_VERSION}"), + ); + assert_eq!( + PgErdMigrationConfig::from_yaml(&incomplete_v2), + Err(PgErdMigrationConfigError::MissingUpstreamResponseBodyLifetime) + ); + let future = valid_yaml().replace( &format!("version: {PG_ERD_MIGRATION_CONFIG_VERSION}"), - "version: 2", + "version: 3", ); assert_eq!( PgErdMigrationConfig::from_yaml(&future), - Err(PgErdMigrationConfigError::UnsupportedVersion(2)) + Err(PgErdMigrationConfigError::UnsupportedVersion(3)) ); } From b327ec992cfb017d59925dafdc71f780a724b0ef Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 3 Sep 2026 08:39:15 +0900 Subject: [PATCH 18/49] test: require explicit pg-erd v2 lifetime --- tests/pg_erd_response_lifetime_config.rs | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/tests/pg_erd_response_lifetime_config.rs b/tests/pg_erd_response_lifetime_config.rs index b4c11290..2293fe14 100644 --- a/tests/pg_erd_response_lifetime_config.rs +++ b/tests/pg_erd_response_lifetime_config.rs @@ -37,9 +37,7 @@ fn version_two_requires_and_preserves_an_explicit_positive_response_body_lifetim PG_ERD_RESPONSE_LIFETIME_CONFIG_VERSION, "", )), - Err(PgErdMigrationConfigError::UnsupportedVersion( - PG_ERD_RESPONSE_LIFETIME_CONFIG_VERSION, - )) + Err(PgErdMigrationConfigError::MissingUpstreamResponseBodyLifetime) ); } From b9ffc855c8e8f074214bf8071d2840802fb55a19 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 3 Sep 2026 08:39:48 +0900 Subject: [PATCH 19/49] docs: align response lifetime with upstream callback order --- API_CONFIG_CONTRACT.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/API_CONFIG_CONTRACT.md b/API_CONFIG_CONTRACT.md index e8f43947..50589382 100644 --- a/API_CONFIG_CONTRACT.md +++ b/API_CONFIG_CONTRACT.md @@ -70,7 +70,7 @@ upstreams: The numeric value above is an illustrative configuration example, not a pg-erd production SLO. A deployment owner must choose the value from its observed long-response contract before canary or cutover. Version 2 rejects zero or a missing response-body lifetime rather than substituting a hidden default. -`max_upstream_response_body_ms` starts when Pingora reaches the final downstream response-header filter for the admitted upstream response. At each upstream response-body progress callback, Runtime Isolation compares elapsed monotonic time with the configured lifetime. Once the lifetime is reached, the callback raises an upstream-scoped fatal error. If the response status/header was already committed, the gateway terminates that downstream response instead of inventing a second status or silently routing to the other pg-erd origin. The ordinary request context then drops its in-flight admission lease. +`max_upstream_response_body_ms` starts when Pingora invokes the upstream-response-header filter for the first non-informational response, before its body-progress callbacks are processed. At each upstream response-body progress callback, Runtime Isolation compares elapsed monotonic time with the configured lifetime. Once the lifetime is reached, the callback raises an upstream-scoped fatal error. If the response status/header was already committed, the gateway terminates that downstream response instead of inventing a second status or silently routing to the other pg-erd origin. The ordinary request context then drops its in-flight admission lease. This callback guard is deliberately not described as an exact timer interrupt. At the pinned Pingora revision, `read_ms` still applies independently to each upstream read and resets after a successful read. A continuously progressing body is therefore stopped at the first body callback at or beyond `max_upstream_response_body_ms`; a response that becomes quiescent is bounded by `read_ms`. The current callback surface does not wake a pending read at the absolute body-lifetime instant, and slow-drip of an incomplete **response header** remains a separate transport gap. Neither limitation may be hidden in parity or production-SLO claims. From d5e85c606e7e06a2466bab1e7f68678a9c879fe7 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 3 Sep 2026 08:40:30 +0900 Subject: [PATCH 20/49] docs: align response lifetime callback boundary --- TRD.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/TRD.md b/TRD.md index 2403a8cc..5a1963e2 100644 --- a/TRD.md +++ b/TRD.md @@ -24,7 +24,7 @@ The pg-erd migration uses the separate `forwarding_policy` trust boundary. It di Requests with a parseable `Content-Length` larger than `max_request_body_bytes` fail with HTTP 413 before upstream selection. Streamed chunks are counted against the same bound and fail with HTTP 413 when the cumulative body crosses it. Non-health application requests share the process in-flight admission budget and fail fast with HTTP 503 when it is exhausted; `/livez` and `/readyz` remain outside that application-capacity budget. A configurable smaller header budget, per-route budget, and origin-capacity budget are not yet implemented. -Pg-erd config version 2 starts its monotonic response-body lifetime when the final upstream response reaches `response_filter`. Each `upstream_response_body_filter` progress callback rejects an elapsed lifetime at or beyond `max_upstream_response_body_ms` as an upstream-scoped fatal error. This complements, rather than replaces, the per-read `read_ms` inactivity timer. With the pinned Pingora callback API the body-lifetime guard is progress-driven: it stops continuous slow-drip at the first body callback after the deadline, while a quiescent pending read can run until `read_ms`. It therefore must not be described as an exact timer interrupt. Slow-drip of an incomplete response header remains a separate transport gap. +Pg-erd config version 2 starts its monotonic response-body lifetime in `upstream_response_filter` on the first non-informational upstream response header, before body-progress callbacks are processed. Each `upstream_response_body_filter` progress callback rejects an elapsed lifetime at or beyond `max_upstream_response_body_ms` as an upstream-scoped fatal error. This complements, rather than replaces, the per-read `read_ms` inactivity timer. With the pinned Pingora callback API the body-lifetime guard is progress-driven: it stops continuous slow-drip at the first body callback after the deadline, while a quiescent pending read can run until `read_ms`. It therefore must not be described as an exact timer interrupt. Slow-drip of an incomplete response header remains a separate transport gap. ## Protocol-transition admission From fc5429ad6fda0b75297c47f5be2979297e47b61c Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 3 Sep 2026 08:41:24 +0900 Subject: [PATCH 21/49] docs: isolate slow-drip test from read timeout --- TEST_STRATEGY.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/TEST_STRATEGY.md b/TEST_STRATEGY.md index e28ae319..f7e20fa7 100644 --- a/TEST_STRATEGY.md +++ b/TEST_STRATEGY.md @@ -20,7 +20,7 @@ Tests are organized by bounded responsibility and traffic phase rather than by i `tests/pg_erd_route_contract.rs` freezes the observed Traefik route precedence, including literal raw `/api` prefix behavior. `tests/pg_erd_http_policy_contract.rs` freezes the separately owned response-security fields and values, case-insensitive field identity, absent lookup, duplicate-field rejection, empty/invalid name/value rejection, and CR/LF injection rejection. `tests/pg_erd_migration_plan_contract.rs`, `tests/pg_erd_upstream_binding_contract.rs`, `tests/pg_erd_forwarding_contract.rs`, and `tests/pg_erd_runtime_proxy_contract.rs` prove the transport-neutral plan, exact upstream binding, forwarding-trust boundary, response-policy composition, and shared runtime-isolation/observability callback behavior. -The bounded Admin Config transition is separately executable. `tests/pg_erd_admin_config_contract.rs` preserves the version-1 characterization contract and rejects unknown/future configuration, traffic/metrics listener collision, zero runtime/keepalive budgets, missing/extra/duplicate/renamed authority, invalid concrete transport configuration, and any characterized `backend`/`frontend` transport authority that overlaps the gateway's traffic or metrics listener under the shared exact/wildcard/dual-stack semantics. `tests/pg_erd_response_lifetime_config.rs` covers the opt-in version-2 increment: it requires an explicit positive `max_upstream_response_body_ms`, rejects zero, and proves version 1 cannot silently acquire version-2 timing semantics. `tests/pg_erd_binary_startup.rs` exercises the dedicated compiled process and requires fail-closed behavior for omitted/unreadable configuration, invalid Admin Config, custom TLS trust material that cannot be materialized before listener activation, and listener/upstream self-loop before network activation. +The bounded Admin Config transition is separately executable. `tests/pg_erd_admin_config_contract.rs` preserves the version-1 characterization contract and rejects unknown/incomplete/future configuration, traffic/metrics listener collision, zero runtime/keepalive budgets, missing/extra/duplicate/renamed authority, invalid concrete transport configuration, and any characterized `backend`/`frontend` transport authority that overlaps the gateway's traffic or metrics listener under the shared exact/wildcard/dual-stack semantics. `tests/pg_erd_response_lifetime_config.rs` covers the opt-in version-2 increment: it requires an explicit positive `max_upstream_response_body_ms`, rejects zero or a missing version-2 value, and proves version 1 cannot silently acquire version-2 timing semantics. `tests/pg_erd_binary_startup.rs` exercises the dedicated compiled process and requires fail-closed behavior for omitted/unreadable configuration, invalid Admin Config, custom TLS trust material that cannot be materialized before listener activation, and listener/upstream self-loop before network activation. On Unix, `tests/pg_erd_local_ca_tls.rs` separately proves the bounded composition root consumes its admitted upstream TLS contract rather than inheriting generic evidence. It generates an ephemeral one-day local CA and `backend.test` certificate, configures only `backend` as TLS and keeps `frontend` a distinct clear-text authority, then requires characterized `/api` traffic to succeed with matching explicit trust and SNI. A second process keeps the CA valid while changing only backend SNI; the routed backend request must fail as HTTP 502 without route failover, gateway-local `/readyz` must remain 200, and a later fallback request must still reach `frontend`. The fixture proves gateway-to-upstream TLS trust/hostname verification only. It does not claim downstream TLS termination, certificate issuance/rotation ownership, or production-network TLS latency. @@ -28,7 +28,7 @@ On Unix, `tests/pg_erd_local_ca_tls.rs` separately proves the bounded compositio `tests/pg_erd_runtime_isolation_traffic.rs` adds chunked overflow and in-flight saturation/recovery. `tests/pg_erd_upstream_failure_traffic.rs` covers connection refusal: HTTP 502 within the connection budget, healthy readiness, low-cardinality error telemetry, and independent fallback recovery. `tests/pg_erd_read_stall_traffic.rs` covers an accepted connection that emits no response bytes: the configured per-read `read_ms` inactivity budget must fail as 502 within a conservative outer bound. -`tests/pg_erd_slow_drip_response_traffic.rs` distinguishes that inactivity contract from the version-2 response-body lifetime. The backend commits HTTP 200 with `Content-Length: 20` and sends one body byte every 60 ms while `read_ms` is 120 ms, so every individual read can keep making progress. The version-2 config separately declares `max_upstream_response_body_ms: 300`; the gateway must terminate the committed response before all 20 bytes arrive, retain the original committed status rather than inventing a second status/failover, increment request-error telemetry, keep `/readyz` available, and serve a later independent frontend route. The elapsed-time guard is callback-driven: the fixture proves continuous body slow-drip is bounded, not that Pingora can interrupt a pending read at an exact absolute deadline. Slow-drip of an incomplete response header remains outside this test. +`tests/pg_erd_slow_drip_response_traffic.rs` distinguishes that inactivity contract from the version-2 response-body lifetime. The backend commits HTTP 200 with `Content-Length: 20` and sends one body byte every 60 ms while `read_ms` is 500 ms, leaving substantial scheduling margin so the old per-read timeout cannot be mistaken for the new lifetime. The version-2 config separately declares `max_upstream_response_body_ms: 300`; the gateway must terminate the committed response before all 20 bytes arrive, retain the original committed status rather than inventing a second status/failover, increment request-error telemetry, keep `/readyz` available, and serve a later independent frontend route. The lifetime starts on the first non-informational `upstream_response_filter` callback and is checked on body-progress callbacks. The fixture proves continuous body slow-drip is bounded, not that Pingora can interrupt a pending read at an exact absolute deadline. Slow-drip of an incomplete response header remains outside this test. `tests/pg_erd_partial_response_traffic.rs` covers the orderly post-header phase: the backend commits HTTP 200 plus `Content-Length: 20`, sends only `partial`, then closes. The downstream must preserve the committed status/framing and terminate incomplete rather than receiving a second status or silent failover; readiness, error telemetry, and an independent route remain usable. On Linux, `tests/pg_erd_upstream_reset_traffic.rs` covers a real pre-header TCP RST using `SO_LINGER(0)`, requiring prompt 502 without failover or waiting for the read inactivity budget. `tests/pg_erd_post_commit_reset_traffic.rs` moves the same abortive close past downstream response commitment and requires the already-committed response boundary to be preserved. These tests characterize ordinary HTTP failure phases; they do not provide WebSocket or generic long-lived-stream support. From c757942a33e9839cbaae29421da02900b0061b23 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 3 Sep 2026 08:51:12 +0900 Subject: [PATCH 22/49] docs: make response lifetime gap baseline code-current --- docs/product-technical-gap-baseline.md | 24 ++++++++++++------------ 1 file changed, 12 insertions(+), 12 deletions(-) diff --git a/docs/product-technical-gap-baseline.md b/docs/product-technical-gap-baseline.md index 2b14763c..d5c89e8c 100644 --- a/docs/product-technical-gap-baseline.md +++ b/docs/product-technical-gap-baseline.md @@ -7,26 +7,26 @@ This baseline is code-current for the Pingora migration stack. Exact source head | Area | State | Evidence / gap | | --- | --- | --- | | Executable Pingora path | Implemented on ancestor branch | Generic production binary composes `GatewayCommand` -> `GatewayConfig` -> `GatewayProxy` -> `http_proxy_service`; the pg-erd candidate adds a separate bounded composition root rather than widening generic v1. Every changed head must reacquire hosted evidence | -| DDD ownership | Implemented | Edge admission invariants, including non-zero declared listener/upstream transport authority, effective listener/metrics socket-authority overlap, and separation of every admitted upstream from both gateway-owned listener authorities, live in `edge_contract`; consumer-derived route characterization in `edge_routing`; response-header characterization in `http_policy`; transport-neutral HTTP/1 protocol-transition classification in `protocol_transition_policy`; `migration_plan` composes route/header contracts with explicit stable upstream authority; `migration_delivery` binds admitted identities to explicit prevalidated Pingora peers; `pingora_delivery` translates admitted transport/protocol policy into Pingora types and now encodes the current HTTP/1 Upgrade denial again in immutable peer options; `migration_admin` owns only the fixed pg-erd profile plus deployment-variable listener/runtime/transport values; `forwarding_policy` owns accepted-transport -> compatibility-forwarding trust; `runtime_isolation` owns body/concurrency admission; `observability` owns payload-free low-cardinality completion facts/counters/log shape; `logging_policy` owns process-wide payload minimization for Pingora dependency diagnostics; internal `process_health` owns gateway-local `/livez` and `/readyz`; `gateway_proxy` and `migration_proxy` are Pingora callback adapters. Product auth/business policy, certificate issuance/rotation, Wardnet/EgressWeave decisions, Keyverse identity, and consumer/product logging remain outside this boundary | +| DDD ownership | Implemented | Edge admission invariants, including non-zero declared listener/upstream transport authority, effective listener/metrics socket-authority overlap, and separation of every admitted upstream from both gateway-owned listener authorities, live in `edge_contract`; consumer-derived route characterization in `edge_routing`; response-header characterization in `http_policy`; transport-neutral HTTP/1 protocol-transition classification in `protocol_transition_policy`; `migration_plan` composes route/header contracts with explicit stable upstream authority; `migration_delivery` binds admitted identities to explicit prevalidated Pingora peers; `pingora_delivery` translates admitted transport/protocol policy into Pingora types and now encodes the current HTTP/1 Upgrade denial again in immutable peer options; `migration_admin` owns only the fixed pg-erd profile plus deployment-variable listener/runtime/transport values and the version boundary for an explicit response-body lifetime; `forwarding_policy` owns accepted-transport -> compatibility-forwarding trust; `runtime_isolation` owns request-body/concurrency admission plus the opt-in monotonic response-body progress lifetime; `observability` owns payload-free low-cardinality completion facts/counters/log shape; `logging_policy` owns process-wide payload minimization for Pingora dependency diagnostics; internal `process_health` owns gateway-local `/livez` and `/readyz`; `gateway_proxy` and `migration_proxy` are Pingora callback adapters. Product auth/business policy, certificate issuance/rotation, Wardnet/EgressWeave decisions, Keyverse identity, and consumer/product logging remain outside this boundary | | Fail-closed generic config | Implemented, exact-head hosted GREEN pending | Strict YAML, version/body/upstream/TLS/trust-path/timeout validation; generic `GatewayConfig` v1 deliberately remains exactly one upstream and is not silently repurposed as a multi-route configuration language. PR #14 rejects port-zero traffic/metrics/upstream authority, equal listener sockets, same-port same-family wildcard aliases, and platform-dependent IPv6-wildcard/IPv4 same-port overlap while preserving distinct concrete non-zero IP authorities. Draft PR #27 extends that same effective socket-authority invariant to the configured upstream: an origin overlapping either the traffic listener or metrics listener through exact, wildcard, or conservative dual-stack authority fails before listener activation, preventing recursive self-proxy and accidental application routing into the internal metrics service | -| Admin Config | Listener-capable source + compiled traffic contract, hosted GREEN pending | Draft PR #12 adds strict `PgErdMigrationConfig` and a separate `cwl-pingora-pg-erd-migration` composition root. Operator input may set listener/metrics sockets, non-zero body/in-flight/keepalive budgets and concrete `backend`/`frontend` transport/TLS values, but cannot define routes, response headers, product auth/business logic, service discovery, or new migration authorities. Parsing validates exact authority without loading custom trust bytes; peer/trust materialization occurs once during `build_proxy` before listeners. Draft child PR #14 found that the same equality-only listener defect and generic port-zero authority gap existed outside the bounded migration profile, moved effective listener overlap into the shared `edge_contract`, made shared `UpstreamConfig` reject port zero, and applies non-zero generic traffic/metrics admission without widening generic routing semantics. Draft PR #27 reuses the same shared listener/upstream separation from the bounded `backend`/`frontend` Admin Config after stable identity validation and adds compiled fail-closed startup acceptance for both composition roots. Source presence remains incomplete until exact-head hosted execution | +| Admin Config | Listener-capable source + compiled traffic contract, hosted GREEN pending | Draft PR #12 adds strict `PgErdMigrationConfig` and a separate `cwl-pingora-pg-erd-migration` composition root. Operator input may set listener/metrics sockets, non-zero body/in-flight/keepalive budgets and concrete `backend`/`frontend` transport/TLS values, but cannot define routes, response headers, product auth/business logic, service discovery, or new migration authorities. Parsing validates exact authority without loading custom trust bytes; peer/trust materialization occurs once during `build_proxy` before listeners. Draft child PR #14 found that the same equality-only listener defect and generic port-zero authority gap existed outside the bounded migration profile, moved effective listener overlap into the shared `edge_contract`, made shared `UpstreamConfig` reject port zero, and applies non-zero generic traffic/metrics admission without widening generic routing semantics. Draft PR #27 reuses the same shared listener/upstream separation from the bounded `backend`/`frontend` Admin Config after stable identity validation and adds compiled fail-closed startup acceptance for both composition roots. Draft PR #39 versions only the pg-erd contract: v1 rejects the new field and preserves its existing semantics, while v2 requires a positive `max_upstream_response_body_ms`; generic v1 is unchanged. Source presence remains incomplete until exact-head hosted execution | | Edge Routing | Dedicated compiled-listener contract present, hosted GREEN pending | Draft parent PR #5 captures `pg-erd-cloud` Traefik precedence as executable exact/prefix selection. PR #11 consumes it from Pingora callbacks. PR #12's dedicated loopback process contract requires `/healthz -> backend`, raw `/apiary -> backend`, and fallback product traffic -> `frontend`. Exact-head compile/test/coverage must pass before parity is credited | | HTTP Policy | Dedicated compiled-listener contract present, hosted GREEN pending | Draft parent PR #6 captures the four pg-erd response-security fields. PR #11 applies them with replacement semantics. PR #12's loopback process contract requires an origin `X-Frame-Options: SAMEORIGIN` value to become characterized `DENY` together with the other three fields. Draft PR #29 adds a separate transport-neutral protocol-transition policy rather than treating Pingora's supplier Upgrade behavior as an implicit product feature. Draft PR #33 aligns the immutable Pingora peer beneath that policy with the same fail-closed decision rather than retaining supplier WebSocket forwarding by default. Exact-head hosted execution remains mandatory | | Migration Plan | Characterized and listener-composable | Draft parent PR #7 composes the pg-erd route and response-policy contracts with an explicit normalized upstream-authority set. Only `backend` and `frontend` are admitted; unknown route targets fail closed. This does not widen `GatewayConfig` v1 | | Migration Delivery | Bound and listener-composable | PR #10 requires exactly one explicit validated `UpstreamConfig` / Pingora peer for every identity admitted by an `EdgeMigrationPlan`, rejecting missing, duplicate, undeclared, or non-concrete port-zero transport authority. PR #11 selects only those prevalidated peers. PR #12 keeps the exact authority bijection in Admin Config and materializes peers once before listener creation; no arbitrary per-request destination or service-discovery inference exists. Draft PR #33 makes all peers materialized through the shared `pingora_delivery` adapter use Pingora's `deny_upgrades()` policy so callback changes cannot silently widen the current HTTP/1 protocol surface | -| Runtime Isolation | Dedicated compiled traffic acceptance added, hosted GREEN pending | `RuntimeIsolationLimits` is shared by the generic and multi-route adapters. Non-zero body/in-flight limits are explicit; declared and streamed body overflow fail closed at HTTP 413 and exhausted admission at HTTP 503. Draft PR #16 adds dedicated compiled pg-erd loopback traffic requiring chunked overflow to return 413 with readiness recovery, and a one-request saturation scenario requiring routed 503 backpressure, live `/readyz`/metrics, rejection telemetry, and post-release capacity recovery. Draft PR #29 rejects uncharacterized HTTP/1 transition requests before they consume application admission capacity or acquire origin authority. This source evidence is not parity until the exact head executes terminal-success | +| Runtime Isolation | Dedicated compiled traffic acceptance added, hosted GREEN pending | `RuntimeIsolationLimits` is shared by the generic and multi-route adapters. Non-zero body/in-flight limits are explicit; declared and streamed body overflow fail closed at HTTP 413 and exhausted admission at HTTP 503. Draft PR #16 adds dedicated compiled pg-erd loopback traffic requiring chunked overflow to return 413 with readiness recovery, and a one-request saturation scenario requiring routed 503 backpressure, live `/readyz`/metrics, rejection telemetry, and post-release capacity recovery. Draft PR #29 rejects uncharacterized HTTP/1 transition requests before they consume application admission capacity or acquire origin authority. Draft PR #39 adds an opt-in pg-erd v2 response-body progress lifetime: the budget starts at the first non-informational upstream response header and cannot be reset by later body callbacks; an over-budget body callback becomes an upstream-scoped fatal error. This is intentionally progress-driven rather than an exact interrupt of a pending read, so quiescent reads remain governed by `read_ms` and slow delivery of an incomplete response header remains open. This source evidence is not parity until the exact head executes terminal-success | | Upstream TLS | Generic and bounded pg-erd source acceptance present; exact-head hosted GREEN pending | Generic compiled local-CA/hostname verification proves custom trust and SNI mismatch behavior. The pg-erd Admin Config avoids trust preload and reads a custom bundle once during peer materialization; dedicated startup requires unreadable trust material to block listener activation. Draft PR #37 adds composition-root-specific real-listener evidence: matching explicit CA plus `backend.test` SNI must route characterized `/api` traffic successfully to the TLS backend; with the same valid CA and a mismatched SNI, the backend route must fail as HTTP 502 without route failover, `/readyz` must remain healthy, and an independent `frontend` fallback request must still succeed. This is gateway-to-upstream TLS source acceptance only. The captured legacy downstream entryPoint remains clear-text, downstream TLS termination/certificate rotation are unclaimed, and representative network/TLS performance remains open | | HTTP protocol scope | HTTP/1 ordinary request path admitted; connection-wide Upgrade fail-closed in admission and peer transport | Upstream peers explicitly use HTTP/1.1. Draft PR #29 makes generic v1 and the bounded pg-erd candidate return HTTP 501 for an `Upgrade` field or case-insensitive `upgrade` token in `Connection` before upstream selection/contact. Draft PR #33 closes the lower-layer mismatch: pinned Pingora `standard()` defaults to `H1UpgradePolicy::WebSocketOnly`, so `pingora_delivery` now uses `HttpUpstreamRequestPolicy::deny_upgrades()` while retaining the supplier's standard hop-by-hop/connection-nomination sanitization. This is deliberate because WebSocket is a PRD non-goal and fresh pinned-supplier evidence has open Pingora issue #946 with proposed fix #947 still unmerged. No WebSocket, HTTP/2 Extended CONNECT, HTTP/2 general parity, HTTP/3, or QUIC claim exists without a later versioned contract and executable traffic evidence | | Hop-by-hop / forwarding trust | Generic trust repair + dedicated migration contract, exact-head hosted GREEN pending | Generic v1 strips request-controlled `Forwarded`, `X-Forwarded-For`, `X-Forwarded-Host`, `X-Forwarded-Port`, `X-Forwarded-Proto`, `X-Forwarded-Server`, and `X-Real-IP` before emitting only gateway-owned `Forwarded: proto=http`; it does not assert client identity. The dedicated `forwarding_policy` separately removes those request-controlled fields and rebuilds characterized `X-Forwarded-For`, `X-Real-IP`, `X-Forwarded-Host`, `X-Forwarded-Port` and `X-Forwarded-Proto` from accepted downstream transport/request authority. The dedicated loopback contract supplies hostile proxy identity and requires direct loopback/Host-derived values with no attacker identity. PR #29 prevents HTTP/1 Upgrade from crossing this boundary at all in the current version, and PR #33 sets the underlying peer to Pingora `deny_upgrades()` while retaining standard hop-by-hop stripping. The characterized Traefik entryPoint is clear-text `web`, so `http` is explicit; HTTPS needs its own TLS-derived scheme contract | | Retry policy | Implemented, intentionally minimal | `max_retries=1` means one total upstream attempt and zero generic automatic retries; domain idempotency/replay policy stays with the product owner | | Request limits | Partial | Declared and streamed/chunked body size plus process-wide in-flight backpressure are bounded. Configurable header, connection, per-route and origin-capacity budgets remain gaps | -| Failure recovery | Refused-origin, connected-silent-origin, pre-header TCP-reset, post-header partial-response, post-commit TCP-reset, and TLS-identity-failure source acceptance added; hosted GREEN pending | The generic compiled loopback contract requires a refused origin connection to return HTTP 502 within configured connection budget and proves `/readyz` remains healthy afterward. PR #17 adds the dedicated multi-route refused-backend equivalent with bounded 502 failure, readiness, low-cardinality error telemetry, and independent `frontend` recovery. PR #18 adds the next pre-header transport phase: the characterized backend accepts the request but emits no response bytes, so Pingora's configured per-read `read_ms` inactivity budget must produce HTTP 502, preserve readiness/error telemetry, and leave the independent route usable. PR #21 adds the distinct orderly post-header phase: `backend` commits HTTP 200 with `Content-Length: 20`, sends only `partial`, then closes; the downstream must retain the committed status/framing and terminate before body completion, while readiness, error telemetry, and an independent `frontend` request recover. PR #24 adds a Linux real-listener pre-header abort: after receiving `/api/reset`, the characterized backend configures `SO_LINGER(0)` and closes with a TCP RST; the gateway must return HTTP 502 promptly without failover or waiting for the per-read inactivity budget, preserve readiness/error telemetry, and leave the independent `frontend` route usable. PR #25 advances the abortive-close phase past downstream commitment: the backend first writes HTTP 200, `Content-Length: 20`, and `partial`, then closes with `SO_LINGER(0)`; the downstream must preserve that status/framing and terminate the incomplete response rather than receiving a second status or failover, while readiness/error telemetry and the independent route remain usable. PR #37 adds the distinct upstream TLS identity phase: a hostname mismatch under otherwise valid explicit CA trust must fail the routed backend request as 502, preserve readiness, and leave the independent route usable. Slow-drip/whole-response-lifetime and broader admitted long-lived streaming cases remain gaps. WebSocket failure recovery is not credited because WebSocket is explicitly rejected by the current admission and peer transport contract. No source contract is credited until exact-head terminal execution | -| Health | Shared process boundary implemented; dedicated exact-head execution pending | `/livez` and `/readyz` are served locally through a shared internal Pingora process-health boundary in both adapters and bypass migration fallback routing. Consumer `/healthz` remains characterized application traffic to `backend`. The dedicated process contract explicitly distinguishes them; PR #16 additionally requires readiness to remain available after streamed-body rejection and while routed application capacity is saturated, PR #17 after refused upstream transport, PR #18 after connected read-timeout failure, PR #21 after a post-header truncated response, PR #24 after a pre-header TCP reset, PR #25 after a post-commit TCP reset, PR #29 after an HTTP/1 Upgrade rejection that must contact no origin, and PR #37 after upstream TLS hostname verification failure. Terminal exact-head execution is still required | +| Failure recovery | Refused-origin, connected-silent-origin, pre-header TCP-reset, post-header partial-response, post-commit TCP-reset, TLS-identity-failure, and slow-drip body-progress source acceptance added; hosted GREEN pending | The generic compiled loopback contract requires a refused origin connection to return HTTP 502 within configured connection budget and proves `/readyz` remains healthy afterward. PR #17 adds the dedicated multi-route refused-backend equivalent with bounded 502 failure, readiness, low-cardinality error telemetry, and independent `frontend` recovery. PR #18 adds the next pre-header transport phase: the characterized backend accepts the request but emits no response bytes, so Pingora's configured per-read `read_ms` inactivity budget must produce HTTP 502, preserve readiness/error telemetry, and leave the independent route usable. PR #21 adds the distinct orderly post-header phase: `backend` commits HTTP 200 with `Content-Length: 20`, sends only `partial`, then closes; the downstream must retain the committed status/framing and terminate before body completion, while readiness, error telemetry, and an independent `frontend` request recover. PR #24 adds a Linux real-listener pre-header abort: after receiving `/api/reset`, the characterized backend configures `SO_LINGER(0)` and closes with a TCP RST; the gateway must return HTTP 502 promptly without failover or waiting for the per-read inactivity budget, preserve readiness/error telemetry, and leave the independent `frontend` route usable. PR #25 advances the abortive-close phase past downstream commitment: the backend first writes HTTP 200, `Content-Length: 20`, and `partial`, then closes with `SO_LINGER(0)`; the downstream must preserve that status/framing and terminate the incomplete response rather than receiving a second status or failover, while readiness/error telemetry and the independent route remain usable. PR #37 adds the distinct upstream TLS identity phase: a hostname mismatch under otherwise valid explicit CA trust must fail the routed backend request as 502, preserve readiness, and leave the independent route usable. PR #39 adds a versioned pg-erd slow-drip body contract: a backend commits HTTP 200 and writes one byte every 60 ms while each write remains inside `read_ms`; v2's 300 ms progress budget must terminate the incomplete body without retry/failover, keep readiness healthy, advance error telemetry, and leave the independent route usable. This does not establish an exact absolute timer while a Pingora read is pending, nor an absolute response-header deadline or arbitrary long-lived streaming parity. WebSocket failure recovery is not credited because WebSocket is explicitly rejected by the current admission and peer transport contract. No source contract is credited until exact-head terminal execution | +| Health | Shared process boundary implemented; dedicated exact-head execution pending | `/livez` and `/readyz` are served locally through a shared internal Pingora process-health boundary in both adapters and bypass migration fallback routing. Consumer `/healthz` remains characterized application traffic to `backend`. The dedicated process contract explicitly distinguishes them; PR #16 additionally requires readiness to remain available after streamed-body rejection and while routed application capacity is saturated, PR #17 after refused upstream transport, PR #18 after connected read-timeout failure, PR #21 after a post-header truncated response, PR #24 after a pre-header TCP reset, PR #25 after a post-commit TCP reset, PR #29 after an HTTP/1 Upgrade rejection that must contact no origin, PR #37 after upstream TLS hostname verification failure, and PR #39 after response-body lifetime enforcement. Terminal exact-head execution is still required | | Graceful drain | Dedicated routed source acceptance added; hosted GREEN pending | Both composition roots use shared explicit Pingora server policy: 5 s grace plus 10 s runtime shutdown timeout inside a 30 s external termination budget. Generic drain is tested. PR #20 adds the consumer-root equivalent: a characterized `/api` request must reach `backend` and remain in flight before SIGTERM, then complete HTTP 200 when released during the grace period, after which the migration process must exit successfully inside the external termination budget. Generic evidence is not transferred; exact-head execution remains mandatory | -| Logs / metrics / traces | Shared/process payload minimization source acceptance added; hosted GREEN and tracing pending | Shared `observability` owns low-cardinality request/error/body/backpressure counters and credential/cookie-safe coarse logs. Both adapters delegate to it; paths, query strings, headers/cookies, credentials, customer payloads and product identifiers are excluded. PR #16 requires dedicated saturation to advance the backpressure counter while metrics remain reachable; PRs #17, #18, #21, #24, and #25 require the request-error counter to advance after refused, connected-silent, orderly post-header truncated, pre-header reset, and post-commit reset upstream failures. PR #23 adds a compiled pg-erd request carrying unique URI/query, Host, Authorization, Cookie, and product-context sentinels; the backend must receive them while the shared observability target must emit only `status`/`outcome`/`request_body_bytes` and none of those sentinels. Draft PR #31 extends this to the process dependency boundary: both production roots install `logging_policy`, operator `RUST_LOG` selection is preserved, but Pingora-family message bodies are replaced with a static marker before formatting; a compiled generic run under `RUST_LOG=trace` requires URI/query/Authorization/Cookie sentinels to reach the origin while remaining absent from all process stderr. Consumer/product logging stays with its owner; distributed tracing remains a separate gap | +| Logs / metrics / traces | Shared/process payload minimization source acceptance added; hosted GREEN and tracing pending | Shared `observability` owns low-cardinality request/error/body/backpressure counters and credential/cookie-safe coarse logs. Both adapters delegate to it; paths, query strings, headers/cookies, credentials, customer payloads and product identifiers are excluded. PR #16 requires dedicated saturation to advance the backpressure counter while metrics remain reachable; PRs #17, #18, #21, #24, #25 and #39 require the request-error counter to advance after refused, connected-silent, orderly post-header truncated, pre-header reset, post-commit reset and response-lifetime failures. PR #23 adds a compiled pg-erd request carrying unique URI/query, Host, Authorization, Cookie, and product-context sentinels; the backend must receive them while the shared observability target must emit only `status`/`outcome`/`request_body_bytes` and none of those sentinels. Draft PR #31 extends this to the process dependency boundary: both production roots install `logging_policy`, operator `RUST_LOG` selection is preserved, but Pingora-family message bodies are replaced with a static marker before formatting; a compiled generic run under `RUST_LOG=trace` requires URI/query/Authorization/Cookie sentinels to reach the origin while remaining absent from all process stderr. Consumer/product logging stays with its owner; distributed tracing remains a separate gap | | OCI isolation | Dedicated image invocation source acceptance added; hosted GREEN pending | Runtime remains uid/gid 65532, read-only-root compatible, capability-free and `no-new-privileges`; base images are digest-pinned. PR #19 packages both Rust composition roots in the same exact image while retaining generic v1 as the default entrypoint, then invokes `cwl-pingora-pg-erd-migration` explicitly with a read-only fixed-profile config under the same rootless/read-only/capability-free boundary and requires its process-health plus metrics listeners to become reachable. Source presence is not deployment evidence until the exact OCI job executes terminal-success | | Dependency policy | Release-blocked | `.github#1605` owns the exact-release Pingora vs patched-`lru` decision and disposition of unmaintained `derivative 2.2.0` / `RUSTSEC-2024-0388`; `.github#810` independently owns the public non-fork Dependency Review compare-API HTTP 403 incident. Known-unsound downgrade, blanket advisory waiver, fail-open 403 handling, or substitute-scanner promotion is prohibited. Open supplier Upgrade bug `cloudflare/pingora#946/#947` is an additional future-WebSocket capability blocker but does not justify weakening current dependency gates | -| Coverage / public API docs | Gates implemented, current head pending | Owned production line/region coverage is required at 100%; `#![deny(missing_docs)]` and warning-denied rustdoc cover public APIs. Fixed-profile/callback construction was modeled as infallible after validated boundaries instead of keeping structurally impossible error branches merely to evade coverage. PR #14 includes decision-path coverage for shared wildcard/equality listener authority plus generic zero-port traffic/metrics/upstream admission; PR #27 adds exact traffic/metrics upstream-collision, wildcard/dual-stack, distinct-concrete acceptance and compiled generic/pg-erd fail-closed startup paths around the shared network-authority separation. The forwarding-trust successor adds an executable generic regression for the full client-controlled proxy-identity set. PR #29 adds transport-neutral token classification plus real-listener generic and pg-erd Upgrade rejection before origin contact. PR #31 adds pure target-classification coverage plus compiled broad-diagnostics data-minimization acceptance. PR #33 adds a public `build_peer()` regression that asserts the immutable peer request policy is exactly Pingora `deny_upgrades()` rather than the supplier's `WebSocketOnly` default. PRs #16-#18, #20, #21, #23, #24, #25, and #37 add integration-only traffic/observability/TLS acceptance, while PR #19 adds OCI packaging/invocation acceptance and PR #22/#35 maintain the CI/load-fixture acceptance path. Current exact head still must pass the unchanged 100% line/region gate; source tests are not GREEN evidence | +| Coverage / public API docs | Gates implemented, current head pending | Owned production line/region coverage is required at 100%; `#![deny(missing_docs)]` and warning-denied rustdoc cover public APIs. Fixed-profile/callback construction was modeled as infallible after validated boundaries instead of keeping structurally impossible error branches merely to evade coverage. PR #14 includes decision-path coverage for shared wildcard/equality listener authority plus generic zero-port traffic/metrics/upstream admission; PR #27 adds exact traffic/metrics upstream-collision, wildcard/dual-stack, distinct-concrete acceptance and compiled generic/pg-erd fail-closed startup paths around the shared network-authority separation. The forwarding-trust successor adds an executable generic regression for the full client-controlled proxy-identity set. PR #29 adds transport-neutral token classification plus real-listener generic and pg-erd Upgrade rejection before origin contact. PR #31 adds pure target-classification coverage plus compiled broad-diagnostics data-minimization acceptance. PR #33 adds a public `build_peer()` regression that asserts the immutable peer request policy is exactly Pingora `deny_upgrades()` rather than the supplier's `WebSocketOnly` default. PR #39 adds deterministic response-lifetime domain/config tests plus integration traffic, while PRs #16-#18, #20, #21, #23, #24, #25, and #37 add other integration-only traffic/observability/TLS acceptance; PR #19 adds OCI packaging/invocation acceptance and PR #22/#35 maintain the CI/load-fixture acceptance path. Current exact head still must pass the unchanged 100% line/region gate; source tests are not GREEN evidence | | Load / 20 ms p95 | Generic and bounded pg-erd local regression source acceptance; exact-head GREEN pending | Checksum-pinned k6 2.2.0 keeps the generic 400-request/four-VU release-binary loopback gate unchanged. PR #22 adds a separate exact-release `cwl-pingora-pg-erd-migration` contract with distinct backend/frontend origins and 400 requests across four VUs alternating characterized backend and fallback routes; every response must preserve the route-specific body, request failures must remain zero, and loopback `http_req_duration` p95 must remain <20 ms. PR #35 replaces the Python origin in both measured paths with a deterministic std-only Rust HTTP/1.1 fixture while leaving VU/request counts, assertions, threshold, and no-application-route-warm-up policy unchanged. Fixture and process startup probes do not pre-exercise application routes. This is not production SLO evidence: larger origin-capacity, TLS/network/multi-hop and representative deployment measurements remain required | | Rollback | Documented, not rehearsed | Rehearsal requires an immutable protected release artifact/digest and a real consumer traffic transition | @@ -37,7 +37,7 @@ Fresh organization code evidence finds no OpenResty usage. Responsibility class, | Repository / evidence | Classification | Migration consequence | | --- | --- | --- | | `linux-cluster-ops/docs/architecture/nginx-routing-inventory.md` plus Nginx/Certbot recovery evidence | ACTIVE_RUNTIME / CURRENT_OPERATOR_DOC | True shared-edge candidate, but current multi-vhost routing, static/PHP-FPM and certificate-adjacent operations exceed Pingora v1. Split authority and freeze executable traffic/TLS contracts first | -| `pg-erd-cloud/deploy/traefik/dynamic.yaml` at protected `main@8dc746920c12988f082e914879d95e13c9693535` | ACTIVE_DEPLOYMENT / PLAUSIBLE_CONSUMER | Fresh 2026-09-03 read confirms ordered exact `/healthz -> backend`, raw-prefix `/api -> backend`, fallback `/ -> frontend` plus four response-security fields are unchanged. Consumer code also conditionally trusts sanitized `X-Forwarded-For`, so forwarding identity is migration behavior rather than cosmetic metadata. PRs #5/#6/#7/#10 characterize route/header/authority/peer binding, PR #11 composes them in Pingora callbacks, PR #12 adds bounded Admin Config, dedicated process startup tests, shared process-health separation and a real loopback backend/frontend traffic contract, PR #14 closes shared wildcard-collision plus generic port-zero network-authority gaps before activation, PR #15 closes the residual generic forwarding-identity spoofing set, PR #16 adds dedicated streamed-body and saturation/recovery traffic acceptance, PR #17 adds refused-backend recovery acceptance, PR #18 adds connected-silent-origin read-timeout acceptance, PR #19 packages/invokes the bounded migration root under the same hardened OCI boundary, PR #20 adds routed SIGTERM drain acceptance, PR #21 adds orderly post-header partial-response failure/recovery acceptance, PR #22 adds a routed multi-authority k6 regression, PR #23 adds payload-free shared access-log acceptance, PR #24 adds pre-header TCP-reset failure/recovery acceptance, PR #25 adds post-commit TCP-reset failure/recovery acceptance, PR #27 repairs listener/upstream self-loop and metrics-surface authority overlap in the shared owner boundary, PR #29 explicitly rejects uncharacterized HTTP/1 Upgrade before origin contact because pg-erd has no proven WebSocket requirement, PR #31 makes Pingora dependency diagnostics process-wide payload-safe in both composition roots, PR #33 aligns all materialized Pingora peers with that same Upgrade-denial contract, PR #35 removes Python from measured loopback latency evidence, and PR #37 adds bounded real-listener upstream TLS trust/hostname-failure recovery evidence without changing the captured clear-text downstream contract. No terminal exact-head pg-erd parity, production performance, shadow/canary or cutover is claimed yet | +| `pg-erd-cloud/deploy/traefik/dynamic.yaml` at protected `main@8dc746920c12988f082e914879d95e13c9693535` | ACTIVE_DEPLOYMENT / PLAUSIBLE_CONSUMER | Fresh 2026-09-03 read confirms ordered exact `/healthz -> backend`, raw-prefix `/api -> backend`, fallback `/ -> frontend` plus four response-security fields are unchanged. Consumer code also conditionally trusts sanitized `X-Forwarded-For`, so forwarding identity is migration behavior rather than cosmetic metadata. PRs #5/#6/#7/#10 characterize route/header/authority/peer binding, PR #11 composes them in Pingora callbacks, PR #12 adds bounded Admin Config, dedicated process startup tests, shared process-health separation and a real loopback backend/frontend traffic contract, PR #14 closes shared wildcard-collision plus generic port-zero network-authority gaps before activation, PR #15 closes the residual generic forwarding-identity spoofing set, PR #16 adds dedicated streamed-body and saturation/recovery traffic acceptance, PR #17 adds refused-backend recovery acceptance, PR #18 adds connected-silent-origin read-timeout acceptance, PR #19 packages/invokes the bounded migration root under the same hardened OCI boundary, PR #20 adds routed SIGTERM drain acceptance, PR #21 adds orderly post-header partial-response failure/recovery acceptance, PR #22 adds a routed multi-authority k6 regression, PR #23 adds payload-free shared access-log acceptance, PR #24 adds pre-header TCP-reset failure/recovery acceptance, PR #25 adds post-commit TCP-reset failure/recovery acceptance, PR #27 repairs listener/upstream self-loop and metrics-surface authority overlap in the shared owner boundary, PR #29 explicitly rejects uncharacterized HTTP/1 Upgrade before origin contact because pg-erd has no proven WebSocket requirement, PR #31 makes Pingora dependency diagnostics process-wide payload-safe in both composition roots, PR #33 aligns all materialized Pingora peers with that same Upgrade-denial contract, PR #35 removes Python from measured loopback latency evidence, PR #37 adds bounded real-listener upstream TLS trust/hostname-failure recovery evidence without changing the captured clear-text downstream contract, and PR #39 versions pg-erd response-body progress lifetime and adds real slow-drip failure/recovery acceptance. No terminal exact-head pg-erd parity, production performance, shadow/canary or cutover is claimed yet | | `naruon` NGINX ingress/live-E2E plus Traefik evaluation | ACTIVE_DEPLOYMENT / TEST_RUNTIME | Its Nginx proxy contract includes HTTP/1.1, long read/send timeouts, WebSocket Upgrade/Connection and forwarded identity semantics. Keycloak/authentication stays outside Pingora. Because current generic/pg-erd runtime now rejects Upgrade at both admission and peer transport boundaries, Naruon cannot inherit that runtime as WebSocket parity; a separate versioned WebSocket transport increment with supplier-race disposition and realistic tunnel evidence is a prerequisite if the live owner contract proves WebSocket is required | | `scopeweave`, `LineageWeave`, `inkspan` Nginx static-serving images/config | ACTIVE_STATIC_RUNTIME | Static hosting is not automatically a shared-edge migration; prove gateway responsibility before queueing | | `life-os` ClusterIP-only base manifests with separately managed edge namespace | DELEGATED EDGE | Repository base manifests do not prove an embedded legacy edge to migrate | @@ -56,10 +56,10 @@ The EA owner path must consume that released contract and project each approved 1. Reacquire exact-current-head CI, 100% owned production line/region coverage, rustdoc, k6, OCI, SAST and supply-chain evidence after every source or documentation movement; repair only evidence-backed repository defects. Organization runner acquisition is separately tracked in `.github#712`; queued jobs are not source GREEN. 2. Keep `.github#1605` and `.github#810` fail-closed until their respective policy and GitHub Dependency Review availability owner paths are resolved; do not suppress `RUSTSEC-2024-0388` generically. -3. Keep PRs #5, #6, #7, #10, #11, #12, #14, #15, #16, #17, #18, #19, #20, #21, #22, #23, #24, #25, #27, #29, #31, #33, #35 and #37 pre-traffic until exact-head quality/security evidence and coherent dependency ancestry are terminal. No child evidence repairs a parent release blocker. -4. PR #12 contains the bounded startup/Admin Config transition plus a dedicated compiled-process loopback traffic contract; PR #14 tightens shared generic/migration network-authority admission; PR #15 closes residual generic forwarding-identity spoofing; PR #16 adds dedicated streamed-body and in-flight saturation/readiness/telemetry/recovery traffic; PR #17 adds refused-backend failure/recovery; PR #18 adds connected-silent-origin per-read timeout/recovery while documenting that `read_ms` is not a whole-response budget; PR #19 packages and explicitly invokes the bounded migration binary under the same hardened OCI image/runtime boundary; PR #20 adds routed in-flight SIGTERM drain acceptance; PR #21 adds the orderly committed-header/truncated-body failure phase without inventing post-commit retry semantics; PR #22 adds route-correct four-VU/400-request local k6 evidence for the bounded binary; PR #23 adds compiled shared-log redaction evidence with non-vacuous request-sensitive sentinels; PR #24 adds an explicit Linux pre-header TCP-reset failure phase; PR #25 adds the Linux post-commit TCP-reset phase while preserving the already-committed response boundary; PR #27 closes a newly found shared network-authority gap by preventing characterized or generic upstream sockets from overlapping the gateway's traffic or metrics listeners, with generic and pg-erd process-startup rejection evidence; PR #29 closes the implicit-protocol-capability gap by rejecting HTTP/1 Upgrade as 501 before route/upstream selection or origin contact in both composition roots; PR #31 closes the process-logging gap by redacting every Pingora-family dependency record message before formatting while retaining CWL-owned bounded observability and operator-selected verbosity; PR #33 closes the residual transport-policy mismatch by configuring every shared Pingora peer with `HttpUpstreamRequestPolicy::deny_upgrades()` and locking that option with a public peer-level regression; PR #35 removes Python interpreter/server variance from the measured generic and pg-erd local latency paths while preserving the existing k6 contract; PR #37 adds real-listener upstream TLS trust/SNI success and hostname-failure recovery specifically through the bounded pg-erd composition root. The immediate gate remains terminal exact-head execution of these stacked contracts together with fmt/compile/clippy/rustdoc/100% coverage and applicable security/supply-chain checks. Any deterministic failure must be repaired causally; source test or packaging presence alone is not parity. -5. After the basic listener/runtime-isolation/refused-origin/read-stall/partial-response/pre-header-reset/post-commit-reset/upstream-TLS/drain/routed-load/shared-log/process-diagnostic-redaction/protocol-transition-rejection/peer-level-upgrade-denial contracts and listener/upstream authority separation are GREEN, extend dedicated pg-erd acceptance with slow-drip/whole-response-lifetime handling, larger origin-capacity stress, distributed tracing, and representative TLS/network/deployment measurements before adopting a production 20 ms p95 objective. Add WebSocket only if fresh consumer evidence requires it, and then only as a separate versioned increment after Pingora #946/#947 is immutably disposed or a reviewed safe supplier patch is pinned; HTTP/1 tunnel concurrency/disconnect/backpressure/drain and HTTP/2 Extended CONNECT require independent executable evidence. A future HTTPS listener must separately prove TLS-derived forwarded scheme instead of reusing the current clear-text `web` assumption. +3. Keep PRs #5, #6, #7, #10, #11, #12, #14, #15, #16, #17, #18, #19, #20, #21, #22, #23, #24, #25, #27, #29, #31, #33, #35, #37 and #39 pre-traffic until exact-head quality/security evidence and coherent dependency ancestry are terminal. No child evidence repairs a parent release blocker. +4. PR #12 contains the bounded startup/Admin Config transition plus a dedicated compiled-process loopback traffic contract; PR #14 tightens shared generic/migration network-authority admission; PR #15 closes residual generic forwarding-identity spoofing; PR #16 adds dedicated streamed-body and in-flight saturation/readiness/telemetry/recovery traffic; PR #17 adds refused-backend failure/recovery; PR #18 adds connected-silent-origin per-read timeout/recovery while documenting that `read_ms` is not a whole-response budget; PR #19 packages and explicitly invokes the bounded migration binary under the same hardened OCI image/runtime boundary; PR #20 adds routed in-flight SIGTERM drain acceptance; PR #21 adds the orderly committed-header/truncated-body failure phase without inventing post-commit retry semantics; PR #22 adds route-correct four-VU/400-request local k6 evidence for the bounded binary; PR #23 adds compiled shared-log redaction evidence with non-vacuous request-sensitive sentinels; PR #24 adds an explicit Linux pre-header TCP-reset failure phase; PR #25 adds the Linux post-commit TCP-reset phase while preserving the already-committed response boundary; PR #27 closes a newly found shared network-authority gap by preventing characterized or generic upstream sockets from overlapping the gateway's traffic or metrics listeners, with generic and pg-erd process-startup rejection evidence; PR #29 closes the implicit-protocol-capability gap by rejecting HTTP/1 Upgrade as 501 before route/upstream selection or origin contact in both composition roots; PR #31 closes the process-logging gap by redacting every Pingora-family dependency record message before formatting while retaining CWL-owned bounded observability and operator-selected verbosity; PR #33 closes the residual transport-policy mismatch by configuring every shared Pingora peer with `HttpUpstreamRequestPolicy::deny_upgrades()` and locking that option with a public peer-level regression; PR #35 removes Python interpreter/server variance from the measured generic and pg-erd local latency paths while preserving the existing k6 contract; PR #37 adds real-listener upstream TLS trust/SNI success and hostname-failure recovery specifically through the bounded pg-erd composition root; PR #39 introduces explicit pg-erd config v2 response-body progress lifetime without changing v1/generic semantics, and adds real slow-drip post-commit failure/recovery evidence. The immediate gate remains terminal exact-head execution of these stacked contracts together with fmt/compile/clippy/rustdoc/100% coverage and applicable security/supply-chain checks. Any deterministic failure must be repaired causally; source test or packaging presence alone is not parity. +5. After the basic listener/runtime-isolation/refused-origin/read-stall/partial-response/pre-header-reset/post-commit-reset/upstream-TLS/response-body-progress-lifetime/drain/routed-load/shared-log/process-diagnostic-redaction/protocol-transition-rejection/peer-level-upgrade-denial contracts and listener/upstream authority separation are GREEN, extend dedicated pg-erd acceptance with incomplete-response-header slow-drip/absolute header-read handling, broader admitted long-lived streaming semantics where consumer evidence requires them, larger origin-capacity stress, distributed tracing, and representative TLS/network/deployment measurements before adopting a production 20 ms p95 objective. Add WebSocket only if fresh consumer evidence requires it, and then only as a separate versioned increment after Pingora #946/#947 is immutably disposed or a reviewed safe supplier patch is pinned; HTTP/1 tunnel concurrency/disconnect/backpressure/drain and HTTP/2 Extended CONNECT require independent executable evidence. A future HTTPS listener must separately prove TLS-derived forwarded scheme instead of reusing the current clear-text `web` assumption. 6. Once the dedicated OCI invocation is terminal GREEN, add a protected release path publishing an immutable image digest with SBOM/provenance/reproducibility evidence and rehearse rollback against that exact digest. 7. Satisfy then-live protected-branch review/governance without self-approval, bot-as-human claims, stale evidence transfer, or routine administrator bypass. 8. Wait for an immutable released Context Graph bundle with source-bound consumer-verifiable release evidence and a coherent compatible GREEN EA admission path before asserting authoritative architecture execution state. -9. Only then move `pg-erd-cloud` through explicit parity -> shadow/canary -> cutover -> rollback -> legacy removal. Other Nginx/Traefik surfaces remain separate responsibility-bound migration candidates and are not inherited automatically from this consumer profile. +9. Only then move `pg-erd-cloud` through explicit parity -> shadow/canary -> cutover -> rollback -> legacy removal. Other Nginx/Traefik surfaces remain separate responsibility-bound migration candidates and are not inherited automatically from this consumer profile. \ No newline at end of file From 6077ec85a907e27105c08442fcecc84e44f16451 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Thu, 3 Sep 2026 08:51:59 +0900 Subject: [PATCH 23/49] docs: record pg-erd response lifetime contract --- CHANGELOG.md | 9 +++++---- 1 file changed, 5 insertions(+), 4 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 496b39fd..cbb2309d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -26,10 +26,11 @@ All notable changes are tracked here. No release has been published yet. - Added process-local fail-fast backpressure: non-health requests above the in-flight budget receive HTTP 503, health remains observable, rejection telemetry increments, and capacity is released after request completion or failure. - Added dedicated compiled pg-erd runtime-isolation traffic acceptance: chunked bodies that cross `max_request_body_bytes` must return 413 without poisoning `/readyz`; a held routed request at an in-flight budget of one must force the next routed request to 503 while `/readyz` and metrics remain available, increment the backpressure counter, and release capacity for a subsequent routed request. - Added dedicated compiled pg-erd refused-origin recovery acceptance: a characterized `backend` connection refusal must return HTTP 502 within the configured connection budget, preserve `/readyz`, increment low-cardinality request-error telemetry, and leave the independent fallback `frontend` route able to complete successfully. This is source acceptance pending terminal exact-head execution, not a parity or cutover claim. -- Added dedicated compiled pg-erd connected-but-silent origin acceptance: the backend accepts the routed request but emits no response bytes, so the configured Pingora per-read `read_ms` budget must produce HTTP 502, preserve `/readyz`, record low-cardinality request-error telemetry, and leave the independent fallback route usable. The configuration contract now explicitly states that Pingora resets this timer after each successful upstream read; slow-drip/whole-response lifetime remains a separate open isolation requirement. -- Added dedicated compiled pg-erd partial-response failure acceptance: the backend commits HTTP 200 with a declared 20-byte body, sends only `partial`, then closes. The downstream must retain the already-committed status/framing and terminate before body completion rather than receiving an invented second status or silent failover; `/readyz`, low-cardinality error telemetry, and the independent `frontend` route must remain usable. Explicit TCP-reset phases, upgraded/WebSocket failure, broader streaming failure, and slow-drip/whole-response lifetime remain separate contracts. +- Added dedicated compiled pg-erd connected-but-silent origin acceptance: the backend accepts the routed request but emits no response bytes, so the configured Pingora per-read `read_ms` budget must produce HTTP 502, preserve `/readyz`, record low-cardinality request-error telemetry, and leave the independent fallback route usable. The configuration contract explicitly states that Pingora resets this timer after each successful upstream read; a separate versioned response-body progress lifetime is required to bound a continuously progressing body. +- Added opt-in pg-erd Admin Config version 2 with mandatory positive `max_upstream_response_body_ms`, leaving pg-erd v1 and generic v1 semantics unchanged. Runtime Isolation starts the monotonic budget at the first non-informational upstream response header and does not reset it on later chunks; an over-budget body callback becomes an upstream-scoped fatal error. Real-listener acceptance commits HTTP 200 and drips one byte every 60 ms while each read stays inside `read_ms`, then requires termination before the declared body completes, low-cardinality error telemetry, healthy `/readyz`, and independent-route recovery. This is a progress-driven bound rather than an exact interrupt of a pending Pingora read; incomplete-response-header slow delivery and arbitrary admitted long-lived streaming still need separate contracts. +- Added dedicated compiled pg-erd partial-response failure acceptance: the backend commits HTTP 200 with a declared 20-byte body, sends only `partial`, then closes. The downstream must retain the already-committed status/framing and terminate before body completion rather than receiving an invented second status or silent failover; `/readyz`, low-cardinality error telemetry, and the independent `frontend` route must remain usable. Explicit TCP-reset phases, upgraded/WebSocket failure, broader streaming failure, and incomplete-response-header slow delivery remain separate contracts. - Added dedicated compiled pg-erd pre-header TCP-reset acceptance on Linux: the characterized backend receives `/api/reset`, configures an abortive `SO_LINGER(0)` close, and emits a real RST before any response header. The gateway must return HTTP 502 without waiting for the read inactivity budget or silently failing over, preserve `/readyz`, increment low-cardinality request-error telemetry, and leave the independent `frontend` route usable. -- Added dedicated compiled pg-erd post-commit TCP-reset acceptance on Linux: the characterized backend first writes HTTP 200, `Content-Length: 20`, and the `partial` body prefix, then closes abortively with `SO_LINGER(0)`. The downstream must preserve the committed status/framing and terminate the incomplete body instead of receiving a second status or silent failover; `/readyz`, low-cardinality request-error telemetry, and the independent `frontend` route must remain usable. Upgraded/WebSocket failure, broader streaming failure, and slow-drip/whole-response lifetime remain separate contracts. +- Added dedicated compiled pg-erd post-commit TCP-reset acceptance on Linux: the characterized backend first writes HTTP 200, `Content-Length: 20`, and the `partial` body prefix, then closes abortively with `SO_LINGER(0)`. The downstream must preserve the committed status/framing and terminate the incomplete body instead of receiving a second status or silent failover; `/readyz`, low-cardinality request-error telemetry, and the independent `frontend` route must remain usable. Upgraded/WebSocket failure, broader streaming failure, and incomplete-response-header slow delivery remain separate contracts. - Added dedicated routed pg-erd graceful-drain acceptance: a characterized `/api` backend request is held open, SIGTERM is sent only after the backend has accepted it, the response is released within the shared grace period, the downstream must still receive HTTP 200, and the migration process must exit successfully inside the external termination budget. Generic drain evidence is not transferred to this composition root. - Added a dedicated pg-erd routed k6 regression alongside the existing generic loopback load contract. The exact release binary now receives 400 requests across four VUs split between characterized backend and frontend routes, requires route-specific response bodies, zero request failures, and loopback p95 below 20 ms. The deterministic origin fixture is parameterized at process startup so both authorities remain distinct without adding request-path measurement overhead. This is source acceptance for a local regression bound, not a production/network SLO claim. - Packaged the bounded `cwl-pingora-pg-erd-migration` composition root in the same digest-pinned distroless OCI image as generic v1 while keeping `cwl-pingora-gateway` as the default entrypoint. Exact-image CI now invokes both composition roots as uid/gid 65532 with a read-only root filesystem, all Linux capabilities dropped, `no-new-privileges`, and read-only versioned configuration; the pg-erd invocation must expose its process-health and metrics listeners before the OCI gate passes. @@ -48,4 +49,4 @@ All notable changes are tracked here. No release has been published yet. - Added missing-public-rustdoc enforcement and documentation builds with warnings denied. - Added DDD, product, technical, security, threat, test, operability, configuration, migration-gap, and primary-source traceability documentation. -Release remains blocked on the organization decision for the exact Pingora release versus `RUSTSEC-2026-0253` and the separate time-bounded disposition of unmaintained `derivative 2.2.0` / `RUSTSEC-2024-0388` (`ContextualWisdomLab/.github#1605`), restoration of authoritative public non-fork Dependency Review evidence (`ContextualWisdomLab/.github#810`), terminal exact-current-head CI/supply-chain/security/review evidence, representative compiled-binary pg-erd streaming-failure/slow-drip/origin-capacity and representative deployment/network benchmark evidence, an immutable registry digest with provenance and rehearsed rollback, and protected-branch integration. HTTP/1 Upgrade now fails closed before origin contact and at immutable Pingora peer construction rather than being counted as WebSocket parity; any later WebSocket claim also requires supplier disposition for `cloudflare/pingora#946/#947`, consumer-derived handshake/tunnel evidence, concurrency/race, disconnect/backpressure/drain, and separate HTTP/2 semantics. Refused-backend, connected-silent-backend, pre-header and post-commit TCP reset, partial-response, pg-erd upstream TLS trust/hostname-failure recovery, payload-free shared/process dependency logging, routed graceful drain, dedicated pg-erd OCI invocation, and routed loopback latency now have source acceptance but remain uncredited until terminal exact-head execution. No consumer migration, canary, cutover, or legacy removal is claimed before those release and traffic-contract gates are satisfied. +Release remains blocked on the organization decision for the exact Pingora release versus `RUSTSEC-2026-0253` and the separate time-bounded disposition of unmaintained `derivative 2.2.0` / `RUSTSEC-2024-0388` (`ContextualWisdomLab/.github#1605`), restoration of authoritative public non-fork Dependency Review evidence (`ContextualWisdomLab/.github#810`), terminal exact-current-head CI/supply-chain/security/review evidence, representative compiled-binary pg-erd incomplete-header slow-delivery/broader long-lived-stream/origin-capacity and representative deployment/network benchmark evidence, an immutable registry digest with provenance and rehearsed rollback, and protected-branch integration. HTTP/1 Upgrade now fails closed before origin contact and at immutable Pingora peer construction rather than being counted as WebSocket parity; any later WebSocket claim also requires supplier disposition for `cloudflare/pingora#946/#947`, consumer-derived handshake/tunnel evidence, concurrency/race, disconnect/backpressure/drain, and separate HTTP/2 semantics. Refused-backend, connected-silent-backend, pre-header and post-commit TCP reset, partial-response, response-body progress-lifetime, pg-erd upstream TLS trust/hostname-failure recovery, payload-free shared/process dependency logging, routed graceful drain, dedicated pg-erd OCI invocation, and routed loopback latency now have source acceptance but remain uncredited until terminal exact-head execution. No consumer migration, canary, cutover, or legacy removal is claimed before those release and traffic-contract gates are satisfied. \ No newline at end of file From c92d670fa9eb2811855bf49f80e7132e511f2d81 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Wed, 9 Sep 2026 02:29:28 +0900 Subject: [PATCH 24/49] test: harden pg-erd slow-drip traffic evidence --- tests/pg_erd_slow_drip_response_traffic.rs | 127 ++++++++++++++++++--- 1 file changed, 111 insertions(+), 16 deletions(-) diff --git a/tests/pg_erd_slow_drip_response_traffic.rs b/tests/pg_erd_slow_drip_response_traffic.rs index 11c4725e..95073106 100644 --- a/tests/pg_erd_slow_drip_response_traffic.rs +++ b/tests/pg_erd_slow_drip_response_traffic.rs @@ -13,6 +13,9 @@ use std::time::{Duration, Instant}; use tempfile::NamedTempFile; +const MAX_REQUEST_HEADER_BYTES: usize = 64 * 1024; +const SOCKET_IO_TIMEOUT: Duration = Duration::from_secs(5); + struct GatewayProcess(Child); impl Drop for GatewayProcess { @@ -28,13 +31,18 @@ enum DownstreamTermination { ConnectionReset, } -fn reserve_loopback() -> SocketAddr { - TcpListener::bind("127.0.0.1:0") - .expect("loopback port should be reservable") - .local_addr() - .expect("reservation should expose an address") +/// Holds traffic and metrics ports simultaneously so sequential bind/drop cannot reuse one port. +fn reserve_distinct_loopback_listeners() -> (TcpListener, TcpListener) { + let traffic = TcpListener::bind("127.0.0.1:0").expect("traffic port should be reservable"); + let metrics = TcpListener::bind("127.0.0.1:0").expect("metrics port should be reservable"); + assert_ne!( + traffic.local_addr().expect("traffic reservation address"), + metrics.local_addr().expect("metrics reservation address") + ); + (traffic, metrics) } +/// Writes the version-2 bounded response-lifetime contract while route authority remains compiled. fn write_config( listener: SocketAddr, metrics_listener: SocketAddr, @@ -50,6 +58,7 @@ fn write_config( file } +/// Waits for one real listener while failing immediately if process activation aborts. fn wait_until_listening(address: SocketAddr, process: &mut Child) { let deadline = Instant::now() + Duration::from_secs(10); loop { @@ -67,6 +76,7 @@ fn wait_until_listening(address: SocketAddr, process: &mut Child) { } } +/// Starts the compiled pg-erd gateway after its traffic/metrics reservations are released. fn start_gateway( config: &NamedTempFile, gateway_address: SocketAddr, @@ -84,10 +94,21 @@ fn start_gateway( GatewayProcess(child) } +/// Applies finite origin I/O bounds before any fixture read or write can stall the hosted lane. +fn set_origin_deadlines(stream: &TcpStream) { + stream + .set_read_timeout(Some(SOCKET_IO_TIMEOUT)) + .expect("origin read timeout should be configurable"); + stream + .set_write_timeout(Some(SOCKET_IO_TIMEOUT)) + .expect("origin write timeout should be configurable"); +} + +/// Sends a bounded raw request used for readiness, metrics, and independent-route recovery checks. fn raw_request(address: SocketAddr, request: &[u8]) -> String { let mut downstream = TcpStream::connect(address).expect("gateway should accept traffic"); downstream - .set_read_timeout(Some(Duration::from_secs(5))) + .set_read_timeout(Some(SOCKET_IO_TIMEOUT)) .expect("downstream timeout should be configurable"); downstream .write_all(request) @@ -99,6 +120,7 @@ fn raw_request(address: SocketAddr, request: &[u8]) -> String { response } +/// Captures the committed response until EOF/RST and records the wall-clock termination boundary. fn raw_request_until_terminal( address: SocketAddr, request: &[u8], @@ -130,6 +152,7 @@ fn raw_request_until_terminal( } } +/// Sends one characterized close-delimited GET through a gateway or metrics listener. fn get(address: SocketAddr, path: &str) -> String { raw_request( address, @@ -138,19 +161,57 @@ fn get(address: SocketAddr, path: &str) -> String { ) } +/// Reads exactly one finite HTTP/1 request header block from an origin connection. fn read_request_headers(stream: &mut TcpStream) -> String { + set_origin_deadlines(stream); let mut bytes = Vec::new(); let mut buffer = [0_u8; 1024]; loop { - let read = stream.read(&mut buffer).expect("origin request should be readable"); + let read = stream + .read(&mut buffer) + .expect("origin request should be readable inside the fixture deadline"); assert!(read > 0, "gateway closed origin request before headers completed"); bytes.extend_from_slice(&buffer[..read]); + assert!( + bytes.len() <= MAX_REQUEST_HEADER_BYTES, + "origin request headers exceeded the 64 KiB fixture bound" + ); if bytes.windows(4).any(|window| window == b"\r\n\r\n") { return String::from_utf8_lossy(&bytes).into_owned(); } } } +/// Parses only an exact case-sensitive HTTP/1.1 three-digit status token. +fn http11_status(response: &str) -> Option { + let mut tokens = response.lines().next()?.split_ascii_whitespace(); + if tokens.next()? != "HTTP/1.1" { + return None; + } + let code = tokens.next()?; + if code.len() != 3 || !code.bytes().all(|byte| byte.is_ascii_digit()) { + return None; + } + code.parse().ok() +} + +/// Returns semantically exact values for one HTTP field name, rejecting lookalike field names. +fn header_values(headers: &str, field_name: &str) -> Vec { + headers + .lines() + .skip(1) + .filter_map(|line| line.split_once(':')) + .filter(|(name, _)| name.eq_ignore_ascii_case(field_name)) + .map(|(_, value)| value.trim().to_string()) + .collect() +} + +/// Requires one exact unlabelled Prometheus sample so numeric-prefix values cannot create GREEN. +fn has_exact_metric_sample(metrics: &str, sample: &str) -> bool { + metrics.lines().any(|line| line.trim() == sample) +} + +/// Proves progress inside `read_ms` cannot evade the versioned whole-body lifetime boundary. #[test] fn compiled_pg_erd_terminates_continuous_response_drip_without_poisoning_other_routes() { let backend = TcpListener::bind("127.0.0.1:0").expect("backend fixture should bind"); @@ -199,14 +260,21 @@ fn compiled_pg_erd_terminates_continuous_response_drip_without_poisoning_other_r .expect("frontend recovery response should be writable"); }); - let gateway_address = reserve_loopback(); - let metrics_address = reserve_loopback(); + let (traffic_reservation, metrics_reservation) = reserve_distinct_loopback_listeners(); + let gateway_address = traffic_reservation + .local_addr() + .expect("traffic reservation address"); + let metrics_address = metrics_reservation + .local_addr() + .expect("metrics reservation address"); let config = write_config( gateway_address, metrics_address, backend_address, frontend_address, ); + drop(traffic_reservation); + drop(metrics_reservation); let _process = start_gateway(&config, gateway_address, metrics_address); let (partial, termination, elapsed) = raw_request_until_terminal( @@ -230,11 +298,17 @@ fn compiled_pg_erd_terminates_continuous_response_drip_without_poisoning_other_r .position(|window| window == b"\r\n\r\n") .map(|position| position + 4) .expect("slow-drip response must commit a complete header block before termination"); - let headers = String::from_utf8_lossy(&partial[..header_end]).to_ascii_lowercase(); - assert!( - headers.starts_with("http/1.1 200"), + let headers = String::from_utf8_lossy(&partial[..header_end]); + assert_eq!( + http11_status(&headers), + Some(200), "a post-commit lifetime failure cannot be rewritten as a second status: {headers:?}" ); + assert_eq!( + header_values(&headers, "Content-Length"), + vec!["20"], + "the committed framing must remain the characterized single Content-Length field" + ); let body = &partial[header_end..]; assert!( body.len() < 20, @@ -242,17 +316,18 @@ fn compiled_pg_erd_terminates_continuous_response_drip_without_poisoning_other_r ); let readiness = get(gateway_address, "/readyz"); - assert!( - readiness.starts_with("HTTP/1.1 200"), + assert_eq!( + http11_status(&readiness), + Some(200), "one slow-drip origin must not poison process readiness: {readiness:?}" ); let metrics = get(metrics_address, "/metrics"); assert!( - metrics.contains("cwl_pingora_gateway_request_errors_total 1"), + has_exact_metric_sample(&metrics, "cwl_pingora_gateway_request_errors_total 1"), "response-lifetime enforcement must remain visible through low-cardinality error telemetry: {metrics:?}" ); let recovered = get(gateway_address, "/after-slow-drip"); - assert!(recovered.starts_with("HTTP/1.1 200")); + assert_eq!(http11_status(&recovered), Some(200)); assert!(recovered.ends_with("\r\n\r\nrecovered")); frontend_origin @@ -262,3 +337,23 @@ fn compiled_pg_erd_terminates_continuous_response_drip_without_poisoning_other_r .join() .expect("slow-drip backend fixture should complete"); } + +/// Prevents status, header-name, and metric-value lookalikes from satisfying the traffic oracle. +#[test] +fn slow_drip_evidence_parsers_reject_lookalikes() { + assert_eq!(http11_status("HTTP/1.1 200 OK\r\n\r\n"), Some(200)); + assert_eq!(http11_status("HTTP/1.1 2000 Weird\r\n\r\n"), None); + assert_eq!(http11_status("http/1.1 200 OK\r\n\r\n"), None); + + let headers = "HTTP/1.1 200 OK\r\nX-Content-Length: 20\r\ncontent-length: 20\r\n\r\n"; + assert_eq!(header_values(headers, "Content-Length"), vec!["20"]); + + assert!(has_exact_metric_sample( + "cwl_pingora_gateway_request_errors_total 1\n", + "cwl_pingora_gateway_request_errors_total 1" + )); + assert!(!has_exact_metric_sample( + "cwl_pingora_gateway_request_errors_total 10\n", + "cwl_pingora_gateway_request_errors_total 1" + )); +} From 368aab21af64c1816afbfd765e2185ea7e9dc8d9 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Wed, 9 Sep 2026 03:30:41 +0900 Subject: [PATCH 25/49] docs: align config contract with pg-erd response lifetime v2 --- API_CONFIG_CONTRACT.md | 17 ++++++++++++----- 1 file changed, 12 insertions(+), 5 deletions(-) diff --git a/API_CONFIG_CONTRACT.md b/API_CONFIG_CONTRACT.md index fd629f2b..fa4b370d 100644 --- a/API_CONFIG_CONTRACT.md +++ b/API_CONFIG_CONTRACT.md @@ -1,6 +1,6 @@ -# Version 1 Configuration Contracts +# Configuration Contracts -## Generic `cwl-pingora-gateway` +## Generic `cwl-pingora-gateway` version 1 ```yaml version: 1 @@ -25,7 +25,7 @@ upstreams: Unknown fields are rejected. `version` must be `1`. `listener`, `metrics_listener`, and `address` are socket addresses with non-zero ports. Port zero is rejected because this deployment contract requires stable operator-declared listener authority and a concrete connectable upstream rather than OS-selected ephemeral listener ports or unusable upstream destinations. Traffic and metrics listeners must not overlap one effective socket authority. Equal addresses, same-port same-family wildcard/concrete aliases, exact IPv4-mapped IPv6 aliases of the same IPv4 address, an IPv4 wildcard paired with any mapped IPv4 authority, a mapped IPv4 wildcard paired with any native or mapped IPv4 authority, and the platform-dependent same-port IPv6-wildcard/IPv4 combination all fail closed; distinct concrete non-aliased addresses may use the same port. `max_request_body_bytes`, `max_in_flight_requests`, and `upstream_keepalive_pool_size` must all be positive. Generic v1 requires exactly one upstream and a non-empty stable upstream name. Every timeout must be positive. -The timeout fields map directly to the pinned Pingora peer options rather than defining a second gateway timer model. In particular, `read_ms` is a **per-read inactivity budget**: Pingora waits at most that long for each individual upstream `read()` and resets the timer after a successful read. It is not a total-response deadline. A connected upstream that sends no response bytes is therefore bounded by `read_ms`, while a slow-drip response can remain alive across multiple successful reads. Whole-response lifetime remains an explicit open runtime-isolation requirement and must not be inferred from `read_ms`. +The timeout fields map directly to the pinned Pingora peer options rather than defining a second gateway timer model. In particular, `read_ms` is a **per-read inactivity budget**: Pingora waits at most that long for each individual upstream `read()` and resets the timer after a successful read. It is not a total-response deadline. A connected upstream that sends no response bytes is therefore bounded by `read_ms`, while a slow-drip response can remain alive across multiple successful reads. Generic v1 still has no whole-response lifetime and must not infer one from `read_ms`. `max_in_flight_requests` is a process-local backpressure boundary for non-health downstream requests. When the budget is exhausted, the runtime fails fast with HTTP 503 instead of admitting unbounded work. `/livez` and `/readyz` bypass this application admission budget so saturation does not hide process health. The admission lease is released when the request context ends, including failed requests. `upstream_keepalive_pool_size` is wired directly into Pingora's `ServerConf`; the runtime does not inherit Pingora's framework default of 128 reusable upstream connections. @@ -37,14 +37,15 @@ Generic v1 downstream transport is cleartext TCP. Before proxying, the generic a ## Bounded `cwl-pingora-pg-erd-migration` candidate -The dedicated pg-erd migration binary consumes a different, migration-specific Admin Config profile. It deliberately reuses the same top-level deployment value names while admitting exactly two fixed transport authorities: +The dedicated pg-erd migration binary consumes a different, migration-specific Admin Config profile. Version 1 remains readable only to preserve the existing unreleased characterization stack. Version 2 is the opt-in response-lifetime increment and requires an explicit positive `max_upstream_response_body_ms`; version 1 rejects that field so the old contract cannot silently acquire new timing semantics. ```yaml -version: 1 +version: 2 listener: 0.0.0.0:6188 metrics_listener: 127.0.0.1:6192 max_request_body_bytes: 1048576 max_in_flight_requests: 128 +max_upstream_response_body_ms: 30000 upstream_keepalive_pool_size: 32 upstreams: - name: backend @@ -67,6 +68,12 @@ upstreams: idle_ms: 10000 ``` +The numeric value above is an illustrative configuration example, not a pg-erd production SLO. A deployment owner must choose the version-2 value from its observed long-response contract before canary or cutover. Version 2 rejects zero or a missing response-body lifetime rather than substituting a hidden default. + +`max_upstream_response_body_ms` starts when Pingora invokes the upstream-response-header filter for the first non-informational response, before body-progress callbacks are processed. Runtime Isolation compares elapsed monotonic time only when a non-empty upstream body chunk is actually observed. Once that progress boundary is at or beyond the configured lifetime, the callback raises an upstream-scoped fatal error. Empty/end-of-stream bookkeeping callbacks do not create a false timeout. If the response status/header was already committed, the gateway terminates that incomplete downstream response instead of inventing a second status or silently routing to the other pg-erd origin. The ordinary request context then drops its in-flight admission lease. + +This callback guard is deliberately not described as an exact timer interrupt. At the pinned Pingora revision, `read_ms` still applies independently to each upstream read and resets after a successful read. A continuously progressing body is therefore stopped at the first non-empty body callback at or beyond `max_upstream_response_body_ms`; a response that becomes quiescent is bounded by `read_ms`. The current callback surface does not wake a pending read at the absolute body-lifetime instant, and slow-drip of an incomplete **response header** remains a separate transport gap. Neither limitation may be hidden in parity or production-SLO claims. + This is not a generic multi-route configuration language. Operator input can bind only concrete transport/TLS values for the compiled `backend` and `frontend` identities. Missing, extra, duplicate, renamed, port-zero, or otherwise invalid listener/metrics/upstream transport authorities fail closed before listener activation. Port zero is rejected because this deployment contract requires stable operator-declared socket authority rather than an OS-selected ephemeral listener or an unusable upstream destination. Listener and metrics authority use the same effective-authority invariant as generic v1, including exact and wildcard IPv4-mapped aliases, same-port same-family wildcard/concrete aliases, and the platform-dependent same-port IPv6-wildcard/IPv4 combination; distinct concrete non-aliased addresses remain admissible. Routes and edge-owned response fields are not configurable: the characterized profile fixes exact `/healthz -> backend`, raw `PathPrefix(`/api`) -> backend` semantics including `/apiary`, fallback `/ -> frontend`, and the four captured response fields `X-Content-Type-Options: nosniff`, `X-Frame-Options: DENY`, `Referrer-Policy: no-referrer`, and `Permissions-Policy: geolocation=(), microphone=(), camera=()`. Admin parsing validates only deterministic configuration and authority invariants. It does not read custom trust-bundle bytes. If an admitted TLS upstream supplies `trust_bundle_file`, the canonical Pingora peer adapter reads and parses that material exactly once during `build_proxy`, still before listeners are registered. An unreadable or invalid bundle therefore blocks activation without a validate-then-reload trust-file window. From 485bd65eb06b807ee74e941d6899787191069e45 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Wed, 9 Sep 2026 03:31:14 +0900 Subject: [PATCH 26/49] docs: make response lifetime operability code-current --- OPERABILITY.md | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/OPERABILITY.md b/OPERABILITY.md index c201b3e5..e880158f 100644 --- a/OPERABILITY.md +++ b/OPERABILITY.md @@ -6,16 +6,22 @@ Run the generic process as `cwl-pingora-gateway --config /path/to/gateway.yaml`. The pg-erd migration candidate is a separate executable: `cwl-pingora-pg-erd-migration --config /path/to/pg-erd-migration.yaml`. It consumes the bounded `PgErdMigrationConfig` profile rather than widening generic `GatewayConfig` v1. The profile admits exactly the compiled `backend` and `frontend` transport identities plus deployment-variable sockets, runtime budgets and upstream transport/TLS values. Route tables, response-policy fields, product authentication/business rules, Keyverse identity, Wardnet/EgressWeave verdicts, service discovery and arbitrary destinations are not operator-configurable. +Pg-erd configuration version 1 preserves the existing unreleased characterization behavior and has no response-body lifetime. Version 2 is opt-in and requires a positive `max_upstream_response_body_ms`; version 1 rejects that field rather than silently acquiring new timing semantics. Choose the version-2 value from observed application long-response requirements before canary or cutover. Documentation/example values are not production SLOs. A zero or incomplete version-2 budget fails before listeners open. + Admin parsing is side-effect free with respect to custom trust bytes. It validates the exact transport-authority set and `UpstreamConfig` invariants first. `build_proxy` then materializes Pingora peers and any custom PEM trust bundle once, still before listener registration. This avoids reading mutable trust material in a validation pass and reading it again for activation. Trust bundles are deployment input, not certificate-authority ownership. Mount them read-only from the platform or canonical secret/certificate owner and rotate them by replacing the deployment revision. The gateway does not issue certificates, manage ACME, or write trust material. -## Health and backpressure +## Health, backpressure, and response lifetime Both process identities reserve `/livez` and `/readyz` and return 200 with `Cache-Control: no-store` through the Pingora serving path. Readiness is process/configuration readiness, not upstream reachability. Do not use it as proof that a dependent application is healthy. For the pg-erd migration, consumer `/healthz` remains routed application traffic to `backend`; it is intentionally distinct from the gateway-local probes. `max_in_flight_requests` limits concurrently admitted non-health requests for one gateway process. At capacity the gateway fails new application traffic fast with HTTP 503 and increments `cwl_pingora_gateway_backpressure_rejections_total`; it does not queue unbounded work. Process health probes bypass that admission budget so operators can distinguish process health from traffic saturation. The request lease is released when the Pingora request context ends, including error paths, and a subsequent request is admissible again. +For pg-erd version 2, `max_upstream_response_body_ms` limits elapsed body delivery after the first non-informational upstream response header. It complements the peer's `read_ms`; it does not replace it. `read_ms` remains a per-read inactivity timeout that resets whenever Pingora successfully reads more upstream data. Continuous response-body slow-drip is stopped at the first **non-empty body-progress callback** at or beyond the explicit lifetime, producing an upstream-scoped request error. Empty/end-of-stream bookkeeping callbacks are not treated as body progress. If HTTP status/headers were already committed, the gateway terminates that incomplete response rather than sending a second status or switching to `frontend`. The request context then releases its in-flight lease. + +This control is not an exact wall-clock interrupt. With the pinned Pingora callback API, an already pending read is not awakened solely because the body-lifetime instant elapsed; a quiescent response remains bounded by `read_ms`, and continuously progressing traffic is checked when non-empty body progress reaches the callback. Slow-drip of an incomplete response header is still a separate gap. Do not advertise the configured number as a strict production deadline until representative deployment traffic has measured the actual scheduler/read-callback behavior. + `upstream_keepalive_pool_size` is copied into Pingora `ServerConf` before bootstrap. Choose it with expected upstream concurrency, origin capacity, instance count and connection reuse in mind. It limits retained reusable upstream connections; it is not a substitute for the downstream in-flight admission limit and does not create product-domain load-balancing semantics. ## Forwarding and protocol boundary @@ -36,7 +42,7 @@ The runtime does not inherit Pingora's retry, keepalive-pool, or drain defaults. SIGTERM uses Pingora graceful termination with an explicit 5-second request-drain grace period and a 10-second runtime-shutdown timeout. The pinned Pingora server calls Tokio `Runtime::shutdown_timeout` with that timeout and then sleeps for the same timeout while service runtimes are shut down in parallel. The policy therefore requires a 30-second supervisor hard-kill budget: its modeled worst-case Pingora process budget is 25 seconds plus scheduler/process-exit overhead. A Kubernetes-style deployment must set `terminationGracePeriodSeconds` to at least 30 or provide an equivalent supervisor budget; a shorter external kill deadline is not an admitted deployment contract. -`tests/graceful_shutdown.rs` exercises the generic compiled binary with a held upstream response. `tests/production_path.rs` covers generic saturation, health and failure recovery. `tests/pg_erd_production_path.rs` exercises the dedicated pg-erd process with real loopback backend/frontend origins, including process-local health, characterized route/response-header behavior, transport-derived forwarding replacement and declared body rejection. `tests/pg_erd_binary_startup.rs` requires missing/invalid configuration and unreadable trust material to fail before listener activation. `tests/pingora_diagnostic_log_safety.rs` runs the compiled generic process with broad trace diagnostics, proves the secret-bearing URI/Host/Authorization/Cookie request reaches the origin, requires a new Pingora redaction marker after the readiness probes, and requires none of those sentinels in process stderr. These source contracts become evidence only after terminal success on the exact current head; predecessor success never transfers. +`tests/graceful_shutdown.rs` exercises the generic compiled binary with a held upstream response. `tests/production_path.rs` covers generic saturation, health and failure recovery. `tests/pg_erd_production_path.rs` exercises the dedicated pg-erd process with real loopback backend/frontend origins, including process-local health, characterized route/response-header behavior, transport-derived forwarding replacement and declared body rejection. `tests/pg_erd_slow_drip_response_traffic.rs` separately proves the version-2 body-lifetime behavior with continuous progress faster than `read_ms`. `tests/pg_erd_binary_startup.rs` requires missing/invalid configuration and unreadable trust material to fail before listener activation. `tests/pingora_diagnostic_log_safety.rs` runs the compiled generic process with broad trace diagnostics, proves the secret-bearing URI/Host/Authorization/Cookie request reaches the origin, requires a new Pingora redaction marker after the readiness probes, and requires none of those sentinels in process stderr. These source contracts become evidence only after terminal success on the exact current head; predecessor success never transfers. ## Container From d138cd3439e6d98404a7eb059abf683110474653 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Wed, 9 Sep 2026 03:32:00 +0900 Subject: [PATCH 27/49] docs: add pg-erd response lifetime acceptance --- TEST_STRATEGY.md | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/TEST_STRATEGY.md b/TEST_STRATEGY.md index 0a107c7e..d1c008b1 100644 --- a/TEST_STRATEGY.md +++ b/TEST_STRATEGY.md @@ -12,7 +12,7 @@ The generic `load-contract` is deliberately separate from functional production- `tests/pg_erd_route_contract.rs` freezes observed `pg-erd-cloud` path precedence, including literal raw `/api` prefix behavior. `tests/pg_erd_http_policy_contract.rs` freezes separately owned response-policy fields and rejects malformed or ambiguous header identity. Migration-plan, upstream-binding, forwarding and runtime-proxy suites prove the transport-neutral plan, exact upstream authority, forwarding-trust boundary, response-policy composition and shared isolation/observability callbacks. -`tests/pg_erd_admin_config_contract.rs` proves that operator configuration is fail closed: unknown/future fields, listener collision, invalid runtime/keepalive budgets, missing/extra/duplicate/renamed authorities and invalid transport bindings are rejected. Only concrete `backend` and `frontend` transport authorities may be bound; route selection remains compiled into the characterized migration plan rather than becoming an operator route DSL. `tests/pg_erd_binary_startup.rs` exercises the dedicated compiled process and trust-material activation failures before listener authority is granted. +`tests/pg_erd_admin_config_contract.rs` proves that operator configuration is fail closed: unknown/incomplete/future configuration, listener collision, invalid runtime/keepalive budgets, missing/extra/duplicate/renamed authorities and invalid transport bindings are rejected. Only concrete `backend` and `frontend` transport authorities may be bound; route selection remains compiled into the characterized migration plan rather than becoming an operator route DSL. `tests/pg_erd_response_lifetime_config.rs` covers the opt-in version-2 increment: it requires an explicit positive `max_upstream_response_body_ms`, rejects zero or a missing version-2 value, and proves version 1 cannot silently acquire version-2 timing semantics. `tests/pg_erd_binary_startup.rs` exercises the dedicated compiled process and trust-material activation failures before listener authority is granted. `tests/pg_erd_local_ca_tls.rs` separately proves the bounded migration composition root consumes its own admitted upstream TLS contract rather than inheriting generic evidence. A one-day local CA and `backend.test` certificate are generated at test time. Matching explicit CA plus SNI must carry characterized `/api/tls` traffic to the TLS backend and preserve a distinct clear-text fallback route. With the same valid CA but mismatched SNI, the backend route must fail as exact HTTP 502 without route failover, `/readyz` must remain 200, the TLS backend must receive no HTTP request bytes, and a later independent fallback request must succeed. Traffic/metrics ports remain simultaneously reserved until startup; accepted origin sockets use five-second I/O deadlines; request headers are capped at 64 KiB; exact case-sensitive HTTP/1.1 three-digit status parsing rejects `2000`/protocol-case false GREEN. This is gateway-to-upstream TLS trust/hostname evidence only, not downstream TLS termination, certificate lifecycle ownership or representative TLS performance. @@ -20,7 +20,7 @@ The generic `load-contract` is deliberately separate from functional production- `tests/pg_erd_production_path.rs` drives real loopback `backend` and `frontend` origins through `cwl-pingora-pg-erd-migration`. It covers gateway-local live/readiness endpoints, backend and fallback routing, forwarding identity reconstruction, characterized response-policy replacement and pre-origin 413 rejection. -Runtime and failure traffic are split by causal phase. `tests/pg_erd_runtime_isolation_traffic.rs` covers streamed body overflow plus in-flight saturation/recovery and exact rejection telemetry. `tests/pg_erd_upstream_failure_traffic.rs` creates deterministic `ECONNREFUSED`, requires bounded 502 recovery, exact request-error telemetry and later frontend success. `tests/pg_erd_read_stall_traffic.rs` keeps an accepted origin connection silent and open so read inactivity cannot be confused with origin closure. `tests/pg_erd_graceful_shutdown.rs` holds a routed request in flight, fixes one SIGTERM-relative external termination deadline, releases the response during the shared grace period, requires downstream 200 and clean process exit before that same deadline. +Runtime and failure traffic are split by causal phase. `tests/pg_erd_runtime_isolation_traffic.rs` covers streamed body overflow plus in-flight saturation/recovery and exact rejection telemetry. `tests/pg_erd_upstream_failure_traffic.rs` creates deterministic `ECONNREFUSED`, requires bounded 502 recovery, exact request-error telemetry and later frontend success. `tests/pg_erd_read_stall_traffic.rs` keeps an accepted origin connection silent and open so read inactivity cannot be confused with origin closure. `tests/pg_erd_slow_drip_response_traffic.rs` distinguishes that inactivity contract from the version-2 response-body lifetime: the backend commits HTTP 200 with `Content-Length: 20` and sends one byte every 60 ms while `read_ms` is 500 ms, but `max_upstream_response_body_ms` is 300 ms. The gateway must terminate before all 20 bytes arrive, preserve the committed 200/framing rather than inventing a second status/failover, increment exact request-error telemetry, keep `/readyz` available, and serve a later independent frontend route. The lifetime starts at the first non-informational upstream response header and is enforced only on non-empty body-progress callbacks, so empty/end-of-stream bookkeeping cannot manufacture a timeout. The fixture proves continuous body slow-drip is bounded; it does not claim an exact timer interrupt of a pending read or an absolute deadline for incomplete response headers. `tests/pg_erd_graceful_shutdown.rs` holds a routed request in flight, fixes one SIGTERM-relative external termination deadline, releases the response during the shared grace period, requires downstream 200 and clean process exit before that same deadline. `tests/pg_erd_partial_response_traffic.rs` covers the post-header orderly-close phase. The backend sends HTTP 200 with `Content-Length: 20` and only the seven-byte `partial` prefix, then waits until the downstream has observed the complete header block plus that exact prefix before closing. Acceptance requires the committed status/framing to remain visible without a fabricated second status or silent failover, `/readyz` to stay 200, exact `cwl_pingora_gateway_request_errors_total 1`, and an independent frontend recovery request. The framing oracle parses header lines, matches `Content-Length` field identity case-insensitively, trims field-value whitespace, requires exactly one value equal to `20`, and rejects lookalike or duplicate/conflicting fields. Exact #21 has terminal exact-head CI/Supply Chain evidence; descendants must revalidate rather than transfer that receipt. @@ -62,4 +62,4 @@ Supply-chain evidence must remain exact-source-bound and include dependency audi ## Remaining gaps -Open acceptance still includes versioned WebSocket enablement and upgraded-connection failure behavior, slow-drip/whole-response lifetime, downstream TLS/H2, H2→H1 Cookie handling, Extended CONNECT, explicit H3/QUIC disposition, tracing, property/fuzz testing, representative routed TLS/origin-capacity load, shadow/canary and rollback. Pg-erd upstream TLS trust/hostname success/failure now has a dedicated source acceptance but must close on the unchanged current exact head before it is credited. Nginx/OpenResty/legacy removal is permitted only after parity, immutable release, canary/cutover and rollback evidence are all current on protected ancestry. Process-wide diagnostic redaction must be reacquired on every changed descendant; source presence or a predecessor trace run is never release evidence. +Open acceptance still includes versioned WebSocket enablement and upgraded-connection failure behavior, incomplete-response-header slow-drip/absolute header-read handling, broader admitted long-lived streaming semantics where consumer evidence requires them, downstream TLS/H2, H2→H1 Cookie handling, Extended CONNECT, explicit H3/QUIC disposition, tracing, property/fuzz testing, representative routed TLS/origin-capacity load, shadow/canary and rollback. Pg-erd upstream TLS trust/hostname success/failure has terminal source-gate evidence on final #37; the new pg-erd v2 response-body progress lifetime must independently close on its unchanged current head before credit. Nginx/OpenResty/legacy removal is permitted only after parity, immutable release, canary/cutover and rollback evidence are all current on protected ancestry. Process-wide diagnostic redaction must be reacquired on every changed descendant; source presence or a predecessor trace run is never release evidence. From a501b90a0c78aadd1a7da6a39a3b712f15f2942f Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Wed, 9 Sep 2026 03:32:32 +0900 Subject: [PATCH 28/49] docs: align TRD with versioned response lifetime --- TRD.md | 12 +++++++----- 1 file changed, 7 insertions(+), 5 deletions(-) diff --git a/TRD.md b/TRD.md index 12b38e2a..74c9da44 100644 --- a/TRD.md +++ b/TRD.md @@ -14,11 +14,11 @@ Generic configuration version 1 is strict YAML with `deny_unknown_fields`. Requi Traffic and metrics listeners must use non-zero, non-overlapping effective socket authority. Validation rejects same-family wildcard/concrete overlap, IPv6-wildcard/IPv4 dual-stack ambiguity, native IPv4 versus IPv4-mapped IPv6 aliases, and native/mapped or mapped-to-mapped IPv4 wildcard aliases while preserving distinct concrete non-aliased addresses. Request-body, in-flight, and keepalive-pool budgets must be positive. -Each upstream has a stable non-empty name, a non-zero concrete socket address, `tls`, optional `sni`, optional absolute `trust_bundle_file`, and explicit positive `connection_ms`, `total_connection_ms`, `read_ms`, `write_ms`, and `idle_ms` budgets. TLS upstreams require non-empty SNI. Pingora `HttpPeer` enables certificate and hostname verification. When `trust_bundle_file` is configured, the loaded PEM certificates become that peer's CA store rather than being silently merged with platform roots. Clear-text upstreams may define neither SNI nor a trust bundle. The gateway does not issue, renew, or rotate certificates. +Each upstream has a stable non-empty name, a non-zero concrete socket address, `tls`, optional `sni`, optional absolute `trust_bundle_file`, and explicit positive `connection_ms`, `total_connection_ms`, `read_ms`, `write_ms`, and `idle_ms` budgets. TLS upstreams require non-empty SNI. Pingora `HttpPeer` enables certificate and hostname verification. When `trust_bundle_file` is configured, the loaded PEM certificates become that peer's CA store rather than being silently merged with platform roots. Clear-text upstreams may define neither SNI nor a trust bundle. The gateway does not issue, renew, or rotate certificates. Pingora `read_ms` remains a per-read inactivity timer, not an overall response deadline; generic v1 therefore still has no response-body lifetime contract. -`PgErdMigrationConfig` is a bounded Admin Config contract, not a second generic router. It admits operator-supplied listener/metrics sockets, non-zero runtime budgets, and concrete transport/TLS data only for the already characterized `backend` and `frontend` identities. Route selection and response-security policy remain compiled migration contracts. Missing, extra, renamed, zero-port, or overlapping transport authority fails before listener activation. +`PgErdMigrationConfig` is a bounded Admin Config contract, not a second generic router. Version 1 preserves the existing unreleased characterization profile. Version 2 is a narrow Runtime Isolation increment: it accepts the same operator-supplied listener/metrics sockets, non-zero request/in-flight/keepalive budgets, and concrete transport/TLS data for only the characterized `backend` and `frontend` identities, and additionally requires a positive explicit `max_upstream_response_body_ms`. Version 1 rejects that field, so timing semantics cannot change without a configuration-version change. Route selection and response-security policy remain compiled migration contracts. Missing, extra, renamed, zero-port, or overlapping transport authority fails before listener activation. -`PgErdMigrationConfig` also derives public Serde `Deserialize`; callers therefore are not forced through `PgErdMigrationConfig::from_yaml`. The public `build_proxy()` activation boundary revalidates the complete deterministic Admin Config contract before delivery peers or runtime limits are materialized. Only after that revalidation may `RuntimeIsolationLimits::from_validated` reuse the proven-positive budgets, so direct deserialization cannot bypass version, listener-authority, runtime, keepalive, or transport-authority invariants. +`PgErdMigrationConfig` also derives public Serde `Deserialize`; callers therefore are not forced through `PgErdMigrationConfig::from_yaml`. The public `build_proxy()` activation boundary revalidates the complete deterministic Admin Config contract before delivery peers or runtime limits are materialized. Only after that revalidation may the infallible `RuntimeIsolationLimits::from_validated*` constructors reuse the proven-positive budgets, so direct deserialization cannot bypass version, listener-authority, runtime, keepalive, or transport-authority invariants. ## Request policy @@ -30,9 +30,11 @@ The pg-erd migration adapter also discards request-controlled forwarding identit Non-health requests acquire the process `max_in_flight_requests` budget before upstream selection and fail closed with HTTP 503 at capacity. Requests with a parseable `Content-Length` above `max_request_body_bytes` fail with HTTP 413 before upstream selection; streamed body bytes are counted and fail with 413 if the same bound is exceeded. Pingora's parser retains its own finite protocol limits, but an operator-controlled smaller HTTP/1 header byte/count budget remains a separate edge-policy gap. +Pg-erd config version 2 starts its monotonic response-body lifetime in `upstream_response_filter` on the first non-informational upstream response header. `upstream_response_body_filter` checks the elapsed lifetime only when a non-empty body chunk is observed; empty/end-of-stream bookkeeping callbacks cannot create a false timeout. A non-empty progress callback at or beyond `max_upstream_response_body_ms` becomes an upstream-scoped fatal error. This complements, rather than replaces, the per-read `read_ms` inactivity timer. With the pinned Pingora callback API the body-lifetime guard is progress-driven: it stops continuous slow-drip at the first real body callback after the deadline, while a quiescent pending read can run until `read_ms`. It therefore must not be described as an exact timer interrupt. Slow-drip of an incomplete response header remains a separate transport gap. + Generic v1 makes one prevalidated upstream peer available per request. The pg-erd migration adapter selects only peers already bound by `MigrationDeliveryPlan`; neither path performs request-controlled service discovery. Domain retries, failover, and idempotency policy are not invented by this runtime. -Failure handling is phase-aware. Before an upstream response header is committed downstream, transport failure may still be represented by the gateway's fail-closed error response under the one-attempt policy. After a valid response header has been committed, a later upstream framing/body failure cannot be rewritten into a second HTTP status or silently failed over: the incomplete downstream response terminates, low-cardinality error telemetry records the failed request, process readiness remains available, and independent routes must remain usable. This is an edge transport invariant, not product retry authority. +Failure handling is phase-aware. Before an upstream response header is committed downstream, transport failure may still be represented by the gateway's fail-closed error response under the one-attempt policy. After a valid response header has been committed, a later upstream framing/body failure—including a version-2 response-body lifetime breach—cannot be rewritten into a second HTTP status or silently failed over: the incomplete downstream response terminates, low-cardinality error telemetry records the failed request, process readiness remains available, the request context releases its in-flight lease, and independent routes must remain usable. This is an edge transport invariant, not product retry authority. ## Health and observability @@ -52,4 +54,4 @@ The candidate supply-chain lane builds and vulnerability-scans both admitted ima Generic v1 is a clear-text downstream HTTP proxy with one explicit upstream per process. HTTP/1 Upgrade is explicitly denied both before request admission and at immutable peer construction; that is non-support evidence, not WebSocket parity. Downstream TLS termination, HTTP/2 admission, H2→H1 Cookie normalization, HTTP/3/QUIC, versioned WebSocket/Extended CONNECT, dynamic reload, Kubernetes Gateway API, and consumer-specific multi-route behavior are separate increments with realistic RED→GREEN evidence. -The concrete pg-erd migration stack is a bounded consumer-characterization adapter and does not widen generic v1. Source presence is not parity. Promotion still requires unchanged exact-head formatting, compile/test, strict Clippy, rustdoc, owned-production coverage, routed traffic/load/failure evidence, terminal dedicated OCI/supply-chain execution, immutable release identity, consumer deployment pin, shadow/canary, rollback rehearsal, protected cutover, and verified legacy removal. +The concrete pg-erd migration stack is a bounded consumer-characterization adapter and does not widen generic v1. The response-body progress lifetime does not close incomplete-response-header slow delivery, exact pending-read interruption, or arbitrary long-lived-stream semantics. Source presence is not parity. Promotion still requires unchanged exact-head formatting, compile/test, strict Clippy, rustdoc, owned-production coverage, routed traffic/load/failure evidence, terminal dedicated OCI/supply-chain execution, immutable release identity, consumer deployment pin, shadow/canary, rollback rehearsal, protected cutover, and verified legacy removal. From 0581caad8e291d565694f49d11e420db82a953d6 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Wed, 9 Sep 2026 03:33:23 +0900 Subject: [PATCH 29/49] docs: record pg-erd response lifetime delta --- CHANGELOG.md | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 3bdd5063..aaa12f17 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -26,8 +26,9 @@ All notable changes are tracked here. No release has been published yet. - Added process-local fail-fast backpressure: non-health requests above the in-flight budget receive HTTP 503, health remains observable, rejection telemetry increments, and capacity is released after request completion or failure. - Added dedicated compiled pg-erd traffic acceptance for streamed/chunked body overflow and routed in-flight saturation/recovery: the migration process must return 413 above the shared body budget, return 503 in less than one second above the in-flight budget, keep `/readyz` observable, expose the exact single-rejection Prometheus sample, and admit a later routed request after capacity is released. - Added dedicated compiled pg-erd refused-origin recovery acceptance using a Linux TCP socket bound to the characterized backend address without entering LISTEN state. The fixture first proves direct `ECONNREFUSED` while retaining exclusive port ownership, then requires the migration gateway to return 502 within a conservative one-second envelope around the configured 200/400 ms connection budgets, keep `/readyz` 200, expose the exact single-error Prometheus sample, and allow a later independent frontend route to recover. Connected read stall, TCP reset, partial-response/streaming failure, retry and failover behavior remain separate gaps. -- Added dedicated compiled pg-erd connected read-stall acceptance with `read_ms=100`: the backend accepts the routed request and remains open without response bytes until the gateway has already failed it, preventing fixture closure from faking timeout behavior. The contract requires 502 inside a conservative one-second envelope, preserved `/readyz`, the exact single-error Prometheus sample, and independent frontend recovery. Pingora `read_timeout` remains a per-read inactivity budget, not a whole-response lifetime; reset, partial-response and slow-drip cases remain open. -- Added dedicated compiled pg-erd post-header partial-response acceptance: the backend commits HTTP 200 with exactly one `Content-Length: 20` field, writes only `partial`, and closes only after the downstream has observed that committed header/body prefix. The downstream must retain the committed status and the single exact framing field, then terminate before body completion instead of receiving an invented second status or silent failover; `/readyz`, exact low-cardinality request-error telemetry, and an independent `frontend` route must remain usable. `X-Content-Length` and duplicate/conflicting `Content-Length` fields cannot satisfy the framing oracle. Explicit TCP reset, broader streaming/upgraded failure, and slow-drip/whole-response lifetime remain separate gaps. +- Added dedicated compiled pg-erd connected read-stall acceptance with `read_ms=100`: the backend accepts the routed request and remains open without response bytes until the gateway has already failed it, preventing fixture closure from faking timeout behavior. The contract requires 502 inside a conservative one-second envelope, preserved `/readyz`, the exact single-error Prometheus sample, and independent frontend recovery. Pingora `read_timeout` remains a per-read inactivity budget, not a whole-response lifetime. +- Added opt-in pg-erd Admin Config version 2 with mandatory positive `max_upstream_response_body_ms`, leaving pg-erd v1 and generic v1 semantics unchanged. Runtime Isolation starts the monotonic budget at the first non-informational upstream response header and checks it only on non-empty body progress, so empty/end-of-stream bookkeeping cannot create a false timeout. A continuously progressing body that reaches the budget becomes an upstream-scoped fatal error without retry/failover after response commitment. Real-listener acceptance commits HTTP 200 and drips one byte every 60 ms while each read stays inside `read_ms`, then requires termination before the declared body completes, exact low-cardinality error telemetry, healthy `/readyz`, and independent-route recovery. This remains a progress-driven bound rather than an exact interrupt of a pending Pingora read; incomplete-response-header slow delivery and broader long-lived streaming remain separate gaps. +- Added dedicated compiled pg-erd post-header partial-response acceptance: the backend commits HTTP 200 with exactly one `Content-Length: 20` field, writes only `partial`, and closes only after the downstream has observed that committed header/body prefix. The downstream must retain the committed status and the single exact framing field, then terminate before body completion instead of receiving an invented second status or silent failover; `/readyz`, exact low-cardinality request-error telemetry, and an independent `frontend` route must remain usable. `X-Content-Length` and duplicate/conflicting `Content-Length` fields cannot satisfy the framing oracle. Explicit TCP reset and broader streaming/upgraded failure remain separate gaps. - Added dedicated compiled pg-erd pre-header TCP-reset acceptance on Linux. The characterized backend must first receive `/api/reset`, then apply abortive `SO_LINGER(0)` and close before emitting any response header. Acceptance requires HTTP 502 within two seconds despite a configured five-second read inactivity budget, no silent frontend failover, preserved `/readyz`, exactly one low-cardinality request-error sample, and a later independent frontend HTTP 200. The fixture bounds origin header reads to five seconds/64 KiB, uses exact HTTP/1.1 status parsing, and matches complete Prometheus sample lines so forwarding defects, protocol/status lookalikes or numeric-prefix counters cannot manufacture GREEN. - Added dedicated compiled pg-erd post-commit TCP-reset acceptance on Linux. The characterized backend writes HTTP 200 with exactly one `Content-Length: 20` field and the seven-byte `partial` prefix, waits until the downstream has actually observed that committed prefix, then applies abortive `SO_LINGER(0)`. Acceptance preserves the committed status/framing, requires termination before the declared body completes, forbids a fabricated second status or silent failover, keeps `/readyz` 200, exposes exactly one low-cardinality request-error sample, and allows independent frontend recovery. The fixture reserves listener ports concurrently, bounds origin reads to five seconds/64 KiB, rejects `X-Content-Length` as framing authority, uses exact HTTP/1.1 status parsing, and matches whole Prometheus sample lines. - Added dedicated routed pg-erd graceful-drain acceptance: a characterized `/api/held` backend request is held in flight, SIGTERM is sent only after the backend has accepted it, the response is released during the shared grace period, the downstream must still receive HTTP 200, and the migration process must exit successfully inside the external termination budget measured from SIGTERM. Generic drain evidence is not transferred to this composition root. @@ -48,4 +49,4 @@ All notable changes are tracked here. No release has been published yet. - Added missing-public-rustdoc enforcement and documentation builds with warnings denied. - Added DDD, product, technical, security, threat, test, operability, configuration, migration-gap, and primary-source traceability documentation. -Release remains blocked on the exact Pingora supplier disposition, including unmaintained `derivative 2.2.0` / `RUSTSEC-2024-0388`, restoration of authoritative dependency-review evidence, terminal exact-current-head CI/supply-chain/security/review evidence for the current TLS child and later descendants, representative pg-erd TLS/protocol/failure/concurrency performance, immutable registry/package identity with release-bound SBOM/provenance/reproducibility, rollback rehearsal, and protected-branch integration. The ordinary pg-erd chain through #33 has retained unchanged-head hosted technical evidence; the current pg-erd TLS child must independently reacquire exact-head hosted/current-range review evidence. HTTP/1 Upgrade remains intentionally unsupported, and no consumer migration, canary, cutover, rollback or legacy removal is claimed before the release and traffic-contract gates are satisfied. +Release remains blocked on the exact Pingora supplier disposition, including unmaintained `derivative 2.2.0` / `RUSTSEC-2024-0388`, restoration of authoritative dependency-review evidence, terminal exact-current-head CI/supply-chain/security/review evidence for the response-lifetime descendant, representative pg-erd incomplete-header/long-lived-stream/origin-capacity/TLS/network performance, immutable registry/package identity with release-bound SBOM/provenance/reproducibility, rollback rehearsal, and protected-branch integration. Final #37 has unchanged-head upstream-TLS hosted technical evidence; the response-lifetime descendant must independently reacquire its own exact-head receipts. HTTP/1 Upgrade remains intentionally unsupported, and no consumer migration, canary, cutover, rollback or legacy removal is claimed before the release and traffic-contract gates are satisfied. From 27c094c4df4558593588ef1513030f5c044554d7 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Wed, 9 Sep 2026 03:34:34 +0900 Subject: [PATCH 30/49] docs: advance gap baseline through response lifetime --- docs/product-technical-gap-baseline.md | 26 ++++++++++++++++---------- 1 file changed, 16 insertions(+), 10 deletions(-) diff --git a/docs/product-technical-gap-baseline.md b/docs/product-technical-gap-baseline.md index 2c2045a0..09ae21a7 100644 --- a/docs/product-technical-gap-baseline.md +++ b/docs/product-technical-gap-baseline.md @@ -6,7 +6,7 @@ This file is the code-current migration baseline for `ContextualWisdomLab/pingor `pingora-gateway` owns shared Ingress, Edge Routing, TLS consumption, HTTP Policy, Load Balancing mechanics, Observability, Admin Config and Runtime Isolation behavior that is genuinely edge/runtime responsibility. Product authentication and business logic remain product-owned. Keyverse remains identity authority; Wardnet/EgressWeave remain their policy authorities. The gateway consumes released/versioned contracts or explicit operator transport inputs and does not copy sibling source, issue cross-service SQL, or depend on mutable sibling PR heads. -Generic v1 remains a one-upstream Rust/Pingora process. The bounded `cwl-pingora-pg-erd-migration` process is a separate composition root for the characterized `pg-erd-cloud` route/header contract; it is not a general product-routing DSL. `PgErdMigrationConfig` deliberately denies unknown fields and admits only concrete listener, runtime budget and characterized `backend`/`frontend` transport bindings. Route authority stays compiled into the migration plan rather than becoming operator-configurable product routing. +Generic v1 remains a one-upstream Rust/Pingora process. The bounded `cwl-pingora-pg-erd-migration` process is a separate composition root for the characterized `pg-erd-cloud` route/header contract; it is not a general product-routing DSL. `PgErdMigrationConfig` deliberately denies unknown fields and admits only concrete listener, runtime budget and characterized `backend`/`frontend` transport bindings. Route authority stays compiled into the migration plan rather than becoming operator-configurable product routing. Pg-erd configuration version 2 adds only an explicit Runtime Isolation response-body progress lifetime; version 1 and generic v1 do not silently acquire that semantic. ## Dependency and promotion root @@ -72,27 +72,33 @@ The peer setting preserves standard hop-by-hop and `Connection`-nomination sanit ### `#37` pg-erd upstream TLS trust and hostname acceptance -The bounded pg-erd composition root must independently prove the TLS contract already characterized by generic v1 rather than inheriting generic evidence. The current #37 child ordinarily/non-force adopts final #33 and changes only its dedicated TLS acceptance plus code-current documentation. `tests/pg_erd_local_ca_tls.rs` generates a short-lived local CA and `backend.test` certificate at test time, configures an explicit trust bundle and SNI for the characterized `backend`, and keeps `frontend` as the separate clear-text authority. +The bounded pg-erd composition root independently proves the TLS contract already characterized by generic v1 rather than inheriting generic evidence. Final #37 ordinarily/non-force adopts final #33 and changes only its dedicated TLS acceptance plus code-current documentation. `tests/pg_erd_local_ca_tls.rs` generates a short-lived local CA and `backend.test` certificate at test time, configures an explicit trust bundle and SNI for the characterized `backend`, and keeps `frontend` as the separate clear-text authority. -Matching CA plus SNI must carry `/api/tls` through the real migration listener to the TLS backend and preserve an independent fallback request. Reusing the valid CA with a mismatched SNI must fail the backend route as exact HTTP 502 without route failover, keep `/readyz` HTTP 200, send zero HTTP request bytes to the TLS backend after the failed hostname check, and preserve an independent frontend recovery request. Traffic/metrics reservations remain simultaneous until startup, accepted sockets carry five-second I/O deadlines, request headers are capped at 64 KiB, and exact HTTP/1.1 three-digit status parsing rejects protocol-case and numeric-prefix lookalikes. This is upstream TLS trust/hostname evidence only; downstream TLS termination, certificate issuance/rotation authority and representative TLS performance remain separate gaps. Because this child has changed after its ordinary restack, it must independently close exact-head hosted and current-range technical-review evidence before TLS acceptance is credited to a release candidate. +Matching CA plus SNI carries `/api/tls` through the real migration listener to the TLS backend and preserves an independent fallback request. Reusing the valid CA with a mismatched SNI fails the backend route as exact HTTP 502 without route failover, keeps `/readyz` HTTP 200, sends zero HTTP request bytes to the TLS backend after the failed hostname check, and preserves an independent frontend recovery request. Traffic/metrics reservations remain simultaneous until startup, accepted sockets carry five-second I/O deadlines, request headers are capped at 64 KiB, and exact HTTP/1.1 three-digit status parsing rejects protocol-case and numeric-prefix lookalikes. This is upstream TLS trust/hostname evidence only; downstream TLS termination, certificate issuance/rotation authority and representative TLS performance remain separate gaps. Final #37 has terminal unchanged-head CI/Supply Chain and current-range technical COMMENT evidence; that does not replace #56 independent human approval. + +### `#39` pg-erd response-body progress lifetime + +This phase ordinarily/non-force adopts final #37 while preserving the historical response-lifetime child as ancestry and reapplying only the valid versioned Runtime Isolation delta. Pg-erd version 2 requires a positive `max_upstream_response_body_ms`; version 1 rejects that field and generic v1 remains unchanged. The budget starts at the first non-informational upstream response header and cannot be reset by later headers. Runtime enforcement checks elapsed monotonic time only on non-empty upstream body progress, so empty/end-of-stream bookkeeping callbacks cannot manufacture a timeout. A non-empty progress callback at or beyond the configured budget becomes an upstream-scoped fatal error. If status/framing is already committed, the gateway terminates the incomplete response rather than inventing a second status, retry or failover. + +`tests/pg_erd_slow_drip_response_traffic.rs` separates this control from Pingora `read_ms`: the characterized backend commits HTTP 200 with `Content-Length: 20`, writes one byte every 60 ms while `read_ms` is 500 ms, and version 2 declares a 300 ms body-progress lifetime. Acceptance requires termination before the full 20-byte/1.2-second drip can complete, exact committed 200/framing, exact request-error telemetry, healthy `/readyz`, and later independent frontend recovery. Traffic/metrics reservations are simultaneous; origin reads are bounded to five seconds/64 KiB; HTTP status, `Content-Length`, and Prometheus metric oracles reject protocol/name/numeric-prefix lookalikes. This is deliberately a progress-driven bound, not an exact timer interrupt of a currently pending read. Incomplete response-header slow delivery and broader admitted long-lived-stream semantics remain separate gaps. ADR 0010 remains Proposed until exact-head hosted/review/governance evidence closes. ## Capability state and buyer-visible gaps | Area | Current state | Remaining acceptance | | --- | --- | --- | -| Admin Config / network authority | Characterized pg-erd transport binding is fail closed; routes remain compiled; final #27 rejects upstream aliases of traffic/metrics listener authority in both composition roots | Revalidate the invariant on every changed descendant and release candidate | +| Admin Config / network authority | Characterized pg-erd transport binding is fail closed; routes remain compiled; final #27 rejects upstream aliases of traffic/metrics listener authority; #39 versions only the explicit response-body lifetime | Revalidate the invariant and v1/v2 boundary on every changed descendant/release candidate | | Generic forwarding trust | Sanitizer/reconstruction invariant is inherited through the current parent stack | No client-IP/trusted-proxy claim until separately characterized | -| Runtime isolation / recovery | Body/in-flight rejection, refused origin, read stall, partial response, graceful drain, pre-header reset and post-commit reset have retained exact-head evidence | Broader streaming/upgraded failure, slow-drip/whole-response lifetime and rollback traffic remain unproven | -| Upstream TLS | Generic local-CA/SNI verification is retained; #37 adds dedicated pg-erd matching-CA/SNI success and hostname-mismatch fail-closed acceptance | Close #37 on one unchanged exact head; representative TLS/network performance and downstream TLS remain unproven | +| Runtime isolation / recovery | Body/in-flight rejection, refused origin, read stall, partial response, graceful drain, pre-header reset, post-commit reset and source-level body-progress lifetime are present | Close #39 exact hosted/review; incomplete-header slow delivery, broader admitted long-lived streaming and rollback traffic remain unproven | +| Upstream TLS | Final #37 has dedicated pg-erd matching-CA/SNI success and hostname-mismatch fail-closed hosted evidence | Representative TLS/network performance and downstream TLS remain unproven | | Protocols | Final #29 rejects uncharacterized HTTP/1 Upgrade before admission/route/upstream selection; final #33 additionally denies Upgrade at immutable peer construction | Downstream TLS/H2, H2→H1 Cookie behavior, versioned WebSocket/Extended CONNECT and explicit H3/QUIC disposition remain separate contracts | -| OCI / supply chain | Final #33 and prior pg-erd phases have exact dual-profile OCI/Supply Chain evidence | #37 and every changed descendant must independently revalidate; immutable registry digest, signing/attestation/provenance, release-bound SBOM, reproducibility receipt and rollback rehearsal remain gaps | +| OCI / supply chain | Final #37 and prior pg-erd phases have exact dual-profile OCI/Supply Chain evidence | #39 and every changed descendant must independently revalidate; immutable registry digest, signing/attestation/provenance, release-bound SBOM, reproducibility receipt and rollback rehearsal remain gaps | | Performance | #22 routed Rust-origin traffic passes aggregate/per-route/sample-floor acceptance and descendant load lanes remain required | Controlled loopback is not production SLO proof; TLS/multi-hop/container scheduling/origin-capacity deployment measurements remain required | -| Observability | #23 proves payload-free shared observability; final #31 proves process-wide Pingora-family diagnostic redaction around the same bounded contract | Revalidate logging on every changed descendant; tracing remains separately unproven and product logging stays product-owned | -| Documentation / review | Baseline is aligned through final #33 and the current #37 TLS acceptance | Current #37 changed head needs exact hosted/current-range technical evidence; bot/owner technical comments do not replace #56 independent human approval | +| Observability | #23 proves payload-free shared observability; final #31 proves process-wide Pingora-family diagnostic redaction; #39 requires exact request-error telemetry on lifetime breach | Revalidate on #39/descendants; tracing remains separately unproven and product logging stays product-owned | +| Documentation / review | Baseline is code-current through final #37 and the #39 response-lifetime source contract | #39 requires unchanged exact-head hosted/current-range technical evidence; bot/owner technical comments do not replace #56 independent human approval | | Release / migration | No protected release or consumer cutover credit | Immutable release → parity → shadow/canary → rollback rehearsal → cutover → verified Nginx/OpenResty/legacy removal | ## Execution order -The current dependency order is `#54 derivative RED + #62 exact semantics/load GREEN → maintainer-integrated release-qualified supplier repair → gateway supplier bump and committed lock regeneration → #54 GREEN + preserved #62 GREEN → #56 independent APPROVED/governance → protected foundation integration → #12 → #14 → #15 → #16 → #17 → #18 → #19 → #20 → #21 exact hosted GREEN → #22 exact hosted/technical-review GREEN → #23 exact hosted/technical-review GREEN → #24 exact hosted/technical-review GREEN → #25 exact hosted/owner-technical closure → #27 exact hosted/current-range technical closure → #29 exact hosted/current-range technical closure → #31 exact hosted/current-range technical closure → #33 exact hosted/current-range technical closure → #37 pg-erd upstream TLS exact hosted/current-range technical closure → remaining failure/protocol acceptance → immutable gateway release/SBOM/provenance/reproducibility/rollback → shadow/canary → cutover → verified legacy removal`. +The current dependency order is `#54 derivative RED + #62 exact semantics/load GREEN → maintainer-integrated release-qualified supplier repair → gateway supplier bump and committed lock regeneration → #54 GREEN + preserved #62 GREEN → #56 independent APPROVED/governance → protected foundation integration → #12 → #14 → #15 → #16 → #17 → #18 → #19 → #20 → #21 exact hosted GREEN → #22 exact hosted/technical-review GREEN → #23 exact hosted/technical-review GREEN → #24 exact hosted/technical-review GREEN → #25 exact hosted/owner-technical closure → #27 exact hosted/current-range technical closure → #29 exact hosted/current-range technical closure → #31 exact hosted/current-range technical closure → #33 exact hosted/current-range technical closure → #37 pg-erd upstream TLS exact hosted/current-range technical closure → #39 response-body progress lifetime exact hosted/current-range technical closure → remaining failure/protocol acceptance → immutable gateway release/SBOM/provenance/reproducibility/rollback → shadow/canary → cutover → verified legacy removal`. No Draft state, predecessor receipt, bot review, local image ID, mutable supplier PR, queue state or controlled loopback measurement is treated as protected merge, release, canary, cutover, rollback or legacy-removal evidence. From 03c9ab1e9cd4623024266983f94ad3302b5a4bc0 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Wed, 9 Sep 2026 03:39:16 +0900 Subject: [PATCH 31/49] style: apply hosted rustfmt to response lifetime proxy --- src/migration_proxy.rs | 9 +++++---- 1 file changed, 5 insertions(+), 4 deletions(-) diff --git a/src/migration_proxy.rs b/src/migration_proxy.rs index 70ae61ea..00cfd14a 100644 --- a/src/migration_proxy.rs +++ b/src/migration_proxy.rs @@ -341,8 +341,8 @@ mod tests { use super::{ body_rejection_to_pingora, enforce_response_body_lifetime, - response_body_lifetime_to_pingora, start_response_body_lifetime, MigrationGatewayProxyError, - MigrationRequestContext, + response_body_lifetime_to_pingora, start_response_body_lifetime, + MigrationGatewayProxyError, MigrationRequestContext, }; use crate::runtime_isolation::{ BodyLimitExceeded, ResponseBodyLifetimeExceeded, RuntimeIsolationLimits, @@ -392,8 +392,9 @@ mod tests { assert!(enforce_response_body_lifetime(&None, &ctx, expired).is_ok()); assert!(enforce_response_body_lifetime(&Some(Bytes::new()), &ctx, expired).is_ok()); - assert!(enforce_response_body_lifetime(&Some(Bytes::from_static(b"x")), &ctx, expired) - .is_err()); + assert!( + enforce_response_body_lifetime(&Some(Bytes::from_static(b"x")), &ctx, expired).is_err() + ); } #[test] From 5246b30574420325a084009513e2fdb9487fe135 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Wed, 9 Sep 2026 03:39:52 +0900 Subject: [PATCH 32/49] style: apply hosted rustfmt to runtime isolation --- src/runtime_isolation.rs | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/src/runtime_isolation.rs b/src/runtime_isolation.rs index 9fdb93dc..95b4873c 100644 --- a/src/runtime_isolation.rs +++ b/src/runtime_isolation.rs @@ -218,7 +218,9 @@ impl ResponseBodyLifetimeBudget { /// Creates a dormant response-body budget from the active runtime-isolation contract. pub(crate) fn new(limits: RuntimeIsolationLimits) -> Self { Self { - limit: limits.max_upstream_response_body_ms().map(Duration::from_millis), + limit: limits + .max_upstream_response_body_ms() + .map(Duration::from_millis), started_at: None, } } @@ -281,7 +283,8 @@ mod tests { assert_eq!(bounded.max_in_flight_requests(), 3); assert_eq!(bounded.max_upstream_response_body_ms(), Some(750)); - let validated = RuntimeIsolationLimits::from_validated_with_response_body_limit(4096, 4, 900); + let validated = + RuntimeIsolationLimits::from_validated_with_response_body_limit(4096, 4, 900); assert_eq!(validated.max_request_body_bytes(), 4096); assert_eq!(validated.max_in_flight_requests(), 4); assert_eq!(validated.max_upstream_response_body_ms(), Some(900)); From c295d91f5026731c7f1e55b573a84c854575c49c Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Wed, 9 Sep 2026 03:40:06 +0900 Subject: [PATCH 33/49] style: apply hosted rustfmt to response lifetime config test --- tests/pg_erd_response_lifetime_config.rs | 5 +---- 1 file changed, 1 insertion(+), 4 deletions(-) diff --git a/tests/pg_erd_response_lifetime_config.rs b/tests/pg_erd_response_lifetime_config.rs index 2293fe14..82b560f8 100644 --- a/tests/pg_erd_response_lifetime_config.rs +++ b/tests/pg_erd_response_lifetime_config.rs @@ -33,10 +33,7 @@ fn version_two_requires_and_preserves_an_explicit_positive_response_body_lifetim ); assert_eq!( - PgErdMigrationConfig::from_yaml(&config_yaml( - PG_ERD_RESPONSE_LIFETIME_CONFIG_VERSION, - "", - )), + PgErdMigrationConfig::from_yaml(&config_yaml(PG_ERD_RESPONSE_LIFETIME_CONFIG_VERSION, "",)), Err(PgErdMigrationConfigError::MissingUpstreamResponseBodyLifetime) ); } From cd0c3582d25aeaa59f4f3b8385ac4c07beb5e9fc Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Wed, 9 Sep 2026 03:40:49 +0900 Subject: [PATCH 34/49] style: apply hosted rustfmt to slow-drip acceptance --- tests/pg_erd_slow_drip_response_traffic.rs | 26 +++++++++++++++------- 1 file changed, 18 insertions(+), 8 deletions(-) diff --git a/tests/pg_erd_slow_drip_response_traffic.rs b/tests/pg_erd_slow_drip_response_traffic.rs index 95073106..e0307fab 100644 --- a/tests/pg_erd_slow_drip_response_traffic.rs +++ b/tests/pg_erd_slow_drip_response_traffic.rs @@ -71,7 +71,10 @@ fn wait_until_listening(address: SocketAddr, process: &mut Child) { if TcpStream::connect_timeout(&address, Duration::from_millis(100)).is_ok() { return; } - assert!(Instant::now() < deadline, "gateway did not start within 10s"); + assert!( + Instant::now() < deadline, + "gateway did not start within 10s" + ); thread::sleep(Duration::from_millis(25)); } } @@ -147,7 +150,9 @@ fn raw_request_until_terminal( started.elapsed(), ); } - Err(error) => panic!("slow-drip downstream response should terminate, not stall: {error}"), + Err(error) => { + panic!("slow-drip downstream response should terminate, not stall: {error}") + } } } } @@ -170,7 +175,10 @@ fn read_request_headers(stream: &mut TcpStream) -> String { let read = stream .read(&mut buffer) .expect("origin request should be readable inside the fixture deadline"); - assert!(read > 0, "gateway closed origin request before headers completed"); + assert!( + read > 0, + "gateway closed origin request before headers completed" + ); bytes.extend_from_slice(&buffer[..read]); assert!( bytes.len() <= MAX_REQUEST_HEADER_BYTES, @@ -223,9 +231,7 @@ fn compiled_pg_erd_terminates_continuous_response_drip_without_poisoning_other_r let request = read_request_headers(&mut stream); assert!(request.starts_with("GET /api/slow-drip HTTP/1.1\r\n")); stream - .write_all( - b"HTTP/1.1 200 OK\r\nContent-Length: 20\r\nConnection: close\r\n\r\n", - ) + .write_all(b"HTTP/1.1 200 OK\r\nContent-Length: 20\r\nConnection: close\r\n\r\n") .expect("backend response header should be writable"); for _ in 0..20 { @@ -234,7 +240,9 @@ fn compiled_pg_erd_terminates_continuous_response_drip_without_poisoning_other_r Err(error) if matches!( error.kind(), - ErrorKind::BrokenPipe | ErrorKind::ConnectionReset | ErrorKind::NotConnected + ErrorKind::BrokenPipe + | ErrorKind::ConnectionReset + | ErrorKind::NotConnected ) => { break; @@ -246,7 +254,9 @@ fn compiled_pg_erd_terminates_continuous_response_drip_without_poisoning_other_r }); let frontend = TcpListener::bind("127.0.0.1:0").expect("frontend fixture should bind"); - let frontend_address = frontend.local_addr().expect("frontend address should exist"); + let frontend_address = frontend + .local_addr() + .expect("frontend address should exist"); let frontend_origin = thread::spawn(move || { let (mut stream, _) = frontend .accept() From 9b135db0947f88dfd525c07f34687fba92490592 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Wed, 9 Sep 2026 03:44:46 +0900 Subject: [PATCH 35/49] test: avoid proxy equality in direct-deserialize regression --- tests/pg_erd_admin_config_contract.rs | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/pg_erd_admin_config_contract.rs b/tests/pg_erd_admin_config_contract.rs index bf9339ba..d157325a 100644 --- a/tests/pg_erd_admin_config_contract.rs +++ b/tests/pg_erd_admin_config_contract.rs @@ -188,10 +188,10 @@ fn pg_erd_admin_config_build_proxy_revalidates_direct_deserialization() { ); let config: PgErdMigrationConfig = serde_yaml::from_str(&incomplete_v2) .expect("direct deserialization may construct an incomplete version-2 value"); - assert_eq!( + assert!(matches!( config.build_proxy(), Err(PgErdMigrationConfigError::MissingUpstreamResponseBodyLifetime) - ); + )); } #[test] From d8156eed6c4622b135daee5a7ddce209cfa82acb Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Wed, 9 Sep 2026 03:57:04 +0900 Subject: [PATCH 36/49] test: cover nonexpired response body progress --- src/migration_proxy.rs | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/src/migration_proxy.rs b/src/migration_proxy.rs index 00cfd14a..a831954b 100644 --- a/src/migration_proxy.rs +++ b/src/migration_proxy.rs @@ -388,10 +388,17 @@ mod tests { let mut ctx = MigrationRequestContext::new(limits); let started = Instant::now(); start_response_body_lifetime(false, &mut ctx, started); + let before_expiry = started + Duration::from_millis(299); let expired = started + Duration::from_millis(300); assert!(enforce_response_body_lifetime(&None, &ctx, expired).is_ok()); assert!(enforce_response_body_lifetime(&Some(Bytes::new()), &ctx, expired).is_ok()); + assert!(enforce_response_body_lifetime( + &Some(Bytes::from_static(b"x")), + &ctx, + before_expiry + ) + .is_ok()); assert!( enforce_response_body_lifetime(&Some(Bytes::from_static(b"x")), &ctx, expired).is_err() ); From 0bdc7f703075ef4db5d629b0d2c63140b7e1d48e Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Wed, 9 Sep 2026 04:09:42 +0900 Subject: [PATCH 37/49] test: close response filter region coverage --- src/migration_proxy.rs | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/src/migration_proxy.rs b/src/migration_proxy.rs index a831954b..82bb526b 100644 --- a/src/migration_proxy.rs +++ b/src/migration_proxy.rs @@ -192,11 +192,13 @@ fn enforce_response_body_lifetime( body: &Option, ctx: &MigrationRequestContext, now: Instant, -) -> Result<(), ResponseBodyLifetimeExceeded> { +) -> pingora::Result> { if body.as_ref().is_some_and(|chunk| !chunk.is_empty()) { - ctx.response_body_lifetime.reject_if_expired(now)?; + ctx.response_body_lifetime + .reject_if_expired(now) + .map_err(response_body_lifetime_to_pingora)?; } - Ok(()) + Ok(None) } #[async_trait] @@ -319,8 +321,6 @@ impl ProxyHttp for MigrationGatewayProxy { ctx: &mut Self::CTX, ) -> pingora::Result> { enforce_response_body_lifetime(body, ctx, Instant::now()) - .map_err(response_body_lifetime_to_pingora)?; - Ok(None) } /// Emits only the shared low-cardinality completion observation for the finished request. From b4d7387fc148b3daa114e30dcd6dcb90b99da98a Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Wed, 9 Sep 2026 04:19:02 +0900 Subject: [PATCH 38/49] docs: align response lifetime execution condition --- docs/adr/0010-version-pg-erd-response-body-lifetime.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/adr/0010-version-pg-erd-response-body-lifetime.md b/docs/adr/0010-version-pg-erd-response-body-lifetime.md index 9df857ba..d116b405 100644 --- a/docs/adr/0010-version-pg-erd-response-body-lifetime.md +++ b/docs/adr/0010-version-pg-erd-response-body-lifetime.md @@ -34,7 +34,7 @@ Rejected. Pingora applies that budget to connection establishment; changing its ### Add an explicit pg-erd version-2 body-lifetime budget -Selected. Version 2 requires positive `max_upstream_response_body_ms`. Version 1 rejects that field and otherwise retains its existing behavior. Runtime Isolation owns monotonic elapsed-time accounting, while the migration adapter starts the budget on the first non-informational upstream response header and checks it on each upstream response-body callback. +Selected. Version 2 requires positive `max_upstream_response_body_ms`. Version 1 rejects that field and otherwise retains its existing behavior. Runtime Isolation owns monotonic elapsed-time accounting, while the migration adapter starts the budget on the first non-informational upstream response header and checks it only on non-empty upstream response-body progress callbacks. ## Decision From 942c58933bfdd3f6f6beefe41c96bf456e237f7b Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Wed, 9 Sep 2026 04:19:45 +0900 Subject: [PATCH 39/49] test: prove response lifetime is causal --- tests/pg_erd_slow_drip_response_traffic.rs | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/tests/pg_erd_slow_drip_response_traffic.rs b/tests/pg_erd_slow_drip_response_traffic.rs index e0307fab..aedd734e 100644 --- a/tests/pg_erd_slow_drip_response_traffic.rs +++ b/tests/pg_erd_slow_drip_response_traffic.rs @@ -302,6 +302,10 @@ fn compiled_pg_erd_terminates_continuous_response_drip_without_poisoning_other_r elapsed < Duration::from_secs(1), "the 300ms response-body budget must stop a continuously progressing response instead of allowing the full 1.2s drip: {elapsed:?}" ); + assert!( + elapsed >= Duration::from_millis(300), + "termination must be caused by the 300ms body-progress budget, not by an immediate post-header failure: {elapsed:?}" + ); let header_end = partial .windows(4) @@ -320,6 +324,10 @@ fn compiled_pg_erd_terminates_continuous_response_drip_without_poisoning_other_r "the committed framing must remain the characterized single Content-Length field" ); let body = &partial[header_end..]; + assert!( + !body.is_empty(), + "the committed response must deliver body progress before the budget terminates it" + ); assert!( body.len() < 20, "the configured response-body budget must terminate before the declared body completes" From 56c7c4822d35c0506d660c71ace54dade104ab0e Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Wed, 9 Sep 2026 04:29:12 +0900 Subject: [PATCH 40/49] test: prove response lifetime starts at response header --- tests/pg_erd_slow_drip_response_traffic.rs | 12 +++++++++--- 1 file changed, 9 insertions(+), 3 deletions(-) diff --git a/tests/pg_erd_slow_drip_response_traffic.rs b/tests/pg_erd_slow_drip_response_traffic.rs index aedd734e..24550b82 100644 --- a/tests/pg_erd_slow_drip_response_traffic.rs +++ b/tests/pg_erd_slow_drip_response_traffic.rs @@ -15,6 +15,8 @@ use tempfile::NamedTempFile; const MAX_REQUEST_HEADER_BYTES: usize = 64 * 1024; const SOCKET_IO_TIMEOUT: Duration = Duration::from_secs(5); +const PRE_RESPONSE_HEADER_DELAY: Duration = Duration::from_millis(150); +const RESPONSE_BODY_LIFETIME: Duration = Duration::from_millis(300); struct GatewayProcess(Child); @@ -230,6 +232,10 @@ fn compiled_pg_erd_terminates_continuous_response_drip_without_poisoning_other_r .expect("routed request should reach the characterized backend authority"); let request = read_request_headers(&mut stream); assert!(request.starts_with("GET /api/slow-drip HTTP/1.1\r\n")); + + // The delay stays below read_ms but makes a request-start lifetime distinguishable from the + // selected first-final-response-header lifetime on the real listener path. + thread::sleep(PRE_RESPONSE_HEADER_DELAY); stream .write_all(b"HTTP/1.1 200 OK\r\nContent-Length: 20\r\nConnection: close\r\n\r\n") .expect("backend response header should be writable"); @@ -300,11 +306,11 @@ fn compiled_pg_erd_terminates_continuous_response_drip_without_poisoning_other_r ); assert!( elapsed < Duration::from_secs(1), - "the 300ms response-body budget must stop a continuously progressing response instead of allowing the full 1.2s drip: {elapsed:?}" + "the response-body budget must stop the delayed-header continuous drip instead of allowing it to run to completion: {elapsed:?}" ); assert!( - elapsed >= Duration::from_millis(300), - "termination must be caused by the 300ms body-progress budget, not by an immediate post-header failure: {elapsed:?}" + elapsed >= PRE_RESPONSE_HEADER_DELAY + RESPONSE_BODY_LIFETIME, + "termination must occur only after the 150ms pre-header delay plus the 300ms body-progress budget, proving the budget starts at the response header: {elapsed:?}" ); let header_end = partial From ba02ba4c77f3cc77cf5cc921b51f433decb23ddd Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 10:54:07 +0900 Subject: [PATCH 41/49] runtime: preserve current proxy semantics while adding response lifetime --- src/migration_proxy.rs | 231 +++++++++++++++++++++++++++++++---------- 1 file changed, 175 insertions(+), 56 deletions(-) diff --git a/src/migration_proxy.rs b/src/migration_proxy.rs index 82bb526b..e0ed19a5 100644 --- a/src/migration_proxy.rs +++ b/src/migration_proxy.rs @@ -8,9 +8,13 @@ use std::time::{Duration, Instant}; use async_trait::async_trait; use bytes::Bytes; +use log::error; use pingora::prelude::{ Error, ErrorType, HttpPeer, ProxyHttp, RequestHeader, ResponseHeader, Session, }; +use pingora::protocols::http::ServerSession; +use pingora::proxy::FailToProxy; +use pingora::ErrorSource; use thiserror::Error; use crate::forwarding_policy::{DownstreamScheme, ForwardingContext}; @@ -34,17 +38,6 @@ pub enum MigrationGatewayProxyError { }, } -impl MigrationGatewayProxyError { - /// Maps an adapter-local route miss to the bounded HTTP response exposed at the edge. - pub fn into_pingora(self) -> Box { - let Self::UnmatchedRoute { .. } = self; - Error::explain( - ErrorType::HTTPStatus(404), - "request path does not match a characterized edge route", - ) - } -} - /// Per-request state for the characterized multi-route Pingora adapter. #[derive(Debug)] pub struct MigrationRequestContext { @@ -54,7 +47,6 @@ pub struct MigrationRequestContext { } impl MigrationRequestContext { - /// Creates isolated per-request accounting with no in-flight admission lease yet acquired. fn new(limits: RuntimeIsolationLimits) -> Self { Self { request_body: RequestBodyBudget::new(limits), @@ -74,9 +66,6 @@ pub struct MigrationGatewayProxy { impl MigrationGatewayProxy { /// Creates callbacks over an already validated delivery plan and runtime-isolation contract. - /// - /// Construction has no remaining fallible work: peer materialization already happened in - /// `MigrationDeliveryPlan`, and `RuntimeIsolationLimits` can exist only after validation. pub fn new(delivery: MigrationDeliveryPlan, limits: RuntimeIsolationLimits) -> Self { Self { delivery, @@ -85,10 +74,7 @@ impl MigrationGatewayProxy { } } - /// Backward-compatible constructor for callers that still consume the earlier result shape. - /// - /// No runtime error can be produced at this boundary; later fail-closed errors are request - /// routing, body, forwarding, transport, or upstream failures handled by Pingora callbacks. + /// Backward-compatible constructor retaining the earlier result-shaped API. pub fn try_new( delivery: MigrationDeliveryPlan, limits: RuntimeIsolationLimits, @@ -119,15 +105,12 @@ impl MigrationGatewayProxy { /// Applies every characterized edge-owned response header using replacement semantics. pub fn apply_response_headers(&self, response: &mut ResponseHeader) -> pingora::Result<()> { - for rule in self.delivery.response_header_rules() { - response - .insert_header(rule.name.clone(), rule.value.as_str()) - .expect("validated response-header policy must remain representable at delivery"); - } - Ok(()) + self.delivery + .response_header_rules() + .iter() + .try_for_each(|rule| response.insert_header(rule.name.clone(), rule.value.as_str())) } - /// Acquires one process-local in-flight lease or rejects before upstream work begins. fn admit_request(&self, ctx: &mut MigrationRequestContext) -> pingora::Result<()> { if let Some(admission) = self.admission_budget.acquire() { ctx.admission = Some(admission); @@ -141,7 +124,6 @@ impl MigrationGatewayProxy { )) } - /// Rejects an already-declared request body that exceeds the configured byte budget. fn reject_oversize_declared_body( session: &Session, ctx: &MigrationRequestContext, @@ -161,7 +143,20 @@ impl MigrationGatewayProxy { } } -/// Maps the transport-neutral body-limit violation to the stable fail-closed HTTP status. +fn pg_erd_forwarding_context( + session: &Session, + upstream_request: &RequestHeader, +) -> pingora::Result { + // The characterized pg-erd Traefik configuration exposes only the clear-text `web` + // entryPoint. Downstream TLS is a separate migration contract and must not be invented here. + ForwardingContext::from_downstream_transport( + session.client_addr(), + upstream_request, + session.req_header(), + DownstreamScheme::Http, + ) +} + fn body_rejection_to_pingora(rejection: BodyLimitExceeded) -> Box { let _ = (rejection.observed, rejection.limit); Error::explain( @@ -170,13 +165,11 @@ fn body_rejection_to_pingora(rejection: BodyLimitExceeded) -> Box { ) } -/// Maps a versioned response-body lifetime breach to an upstream-scoped Pingora failure. fn response_body_lifetime_to_pingora(rejection: ResponseBodyLifetimeExceeded) -> Box { let _ = (rejection.elapsed, rejection.limit); Error::new_up(ErrorType::Custom("UpstreamResponseBodyLifetimeExceeded")) } -/// Starts the lifetime on the first final upstream response header without reset on later headers. fn start_response_body_lifetime( is_informational: bool, ctx: &mut MigrationRequestContext, @@ -187,7 +180,6 @@ fn start_response_body_lifetime( } } -/// Applies the lifetime only to actual body progress, not empty/end-of-stream bookkeeping callbacks. fn enforce_response_body_lifetime( body: &Option, ctx: &MigrationRequestContext, @@ -201,16 +193,36 @@ fn enforce_response_body_lifetime( Ok(None) } +fn unmatched_route_to_pingora(_error: MigrationGatewayProxyError) -> Box { + Error::explain( + ErrorType::HTTPStatus(404), + "request path does not match a characterized edge route", + ) +} + +fn proxy_error_status(error: &Error) -> u16 { + if let ErrorType::HTTPStatus(code) = &error.etype { + return *code; + } + + match &error.esource { + ErrorSource::Upstream => 502, + ErrorSource::Downstream => match &error.etype { + ErrorType::WriteError | ErrorType::ReadError | ErrorType::ConnectionClosed => 0, + _ => 400, + }, + ErrorSource::Internal | ErrorSource::Unset => 500, + } +} + #[async_trait] impl ProxyHttp for MigrationGatewayProxy { type CTX = MigrationRequestContext; - /// Creates request-local body and admission accounting for a new downstream exchange. fn new_ctx(&self) -> Self::CTX { MigrationRequestContext::new(self.limits) } - /// Serves process health locally and admits ordinary traffic before any upstream selection. async fn request_filter( &self, session: &mut Session, @@ -233,7 +245,6 @@ impl ProxyHttp for MigrationGatewayProxy { } } - /// Accounts streamed request bytes so chunked bodies cannot bypass the declared-length gate. async fn request_body_filter( &self, _session: &mut Session, @@ -244,13 +255,12 @@ impl ProxyHttp for MigrationGatewayProxy { where Self::CTX: Send + Sync, { - let chunk_bytes = body.as_ref().map_or(0_usize, Bytes::len) as u64; + let chunk_bytes = body.as_ref().map_or(0_u64, |chunk| chunk.len() as u64); ctx.request_body .observe_chunk(chunk_bytes) .map_err(body_rejection_to_pingora) } - /// Resolves only a prevalidated peer admitted by the immutable characterized route plan. async fn upstream_peer( &self, session: &mut Session, @@ -258,10 +268,9 @@ impl ProxyHttp for MigrationGatewayProxy { ) -> pingora::Result> { self.build_upstream_peer(session.req_header().uri.path()) .map(Box::new) - .map_err(MigrationGatewayProxyError::into_pingora) + .map_err(unmatched_route_to_pingora) } - /// Rebuilds forwarding identity from the accepted socket and Host authority before origin I/O. async fn upstream_request_filter( &self, session: &mut Session, @@ -271,17 +280,10 @@ impl ProxyHttp for MigrationGatewayProxy { where Self::CTX: Send + Sync, { - let forwarding = ForwardingContext::from_downstream_transport( - session.client_addr(), - session.server_addr(), - upstream_request, - session.req_header(), - DownstreamScheme::Http, - )?; + let forwarding = pg_erd_forwarding_context(session, upstream_request)?; self.apply_upstream_request_policy(upstream_request, &forwarding) } - /// Starts the versioned response-body lifetime at the first non-informational upstream header. async fn upstream_response_filter( &self, _session: &mut Session, @@ -299,7 +301,6 @@ impl ProxyHttp for MigrationGatewayProxy { Ok(()) } - /// Applies the characterized edge-owned response fields with replacement semantics. async fn response_filter( &self, _session: &mut Session, @@ -312,7 +313,6 @@ impl ProxyHttp for MigrationGatewayProxy { self.apply_response_headers(upstream_response) } - /// Enforces the versioned lifetime at real upstream body-progress boundaries only. fn upstream_response_body_filter( &self, _session: &mut Session, @@ -323,7 +323,36 @@ impl ProxyHttp for MigrationGatewayProxy { enforce_response_body_lifetime(body, ctx, Instant::now()) } - /// Emits only the shared low-cardinality completion observation for the finished request. + async fn fail_to_proxy( + &self, + session: &mut Session, + error_value: &Error, + _ctx: &mut Self::CTX, + ) -> FailToProxy { + let status = proxy_error_status(error_value); + if status > 0 { + let mut response = ServerSession::generate_error(status); + if let Err(policy_error) = self.apply_response_headers(&mut response) { + error!( + "validated pg-erd response policy could not be applied to local error response: {policy_error}" + ); + } + if let Err(write_error) = session + .write_response_header(Box::new(response), true) + .await + { + error!( + "failed to send policy-complete error response to downstream: {write_error}" + ); + } + } + + FailToProxy { + error_code: status, + can_reuse_downstream: false, + } + } + async fn logging(&self, session: &mut Session, error: Option<&Error>, ctx: &mut Self::CTX) where Self::CTX: Send + Sync, @@ -334,20 +363,67 @@ impl ProxyHttp for MigrationGatewayProxy { #[cfg(test)] mod tests { + use std::net::SocketAddr; use std::time::{Duration, Instant}; use bytes::Bytes; - use pingora::prelude::{ErrorSource, ErrorType}; + use pingora::prelude::{Error, ErrorType}; + use pingora::ErrorSource; use super::{ - body_rejection_to_pingora, enforce_response_body_lifetime, - response_body_lifetime_to_pingora, start_response_body_lifetime, - MigrationGatewayProxyError, MigrationRequestContext, + body_rejection_to_pingora, enforce_response_body_lifetime, proxy_error_status, + response_body_lifetime_to_pingora, start_response_body_lifetime, unmatched_route_to_pingora, + MigrationGatewayProxy, MigrationGatewayProxyError, MigrationRequestContext, }; + use crate::edge_contract::{UpstreamConfig, UpstreamTimeouts}; + use crate::edge_routing::{RouteMatch, RouteRule}; + use crate::http_policy::ResponseHeaderRule; + use crate::migration_delivery::MigrationDeliveryPlan; + use crate::migration_plan::EdgeMigrationPlan; use crate::runtime_isolation::{ BodyLimitExceeded, ResponseBodyLifetimeExceeded, RuntimeIsolationLimits, }; + fn admission_proxy() -> (MigrationGatewayProxy, RuntimeIsolationLimits) { + let plan = EdgeMigrationPlan::try_new( + vec!["backend".to_string()], + vec![RouteRule { + name: "backend".to_string(), + priority: 100, + matcher: RouteMatch::Exact("/".to_string()), + upstream: "backend".to_string(), + }], + vec![ResponseHeaderRule { + name: "X-Content-Type-Options".to_string(), + value: "nosniff".to_string(), + }], + ) + .expect("admission fixture migration plan must be valid"); + let delivery = MigrationDeliveryPlan::try_new( + plan, + vec![UpstreamConfig { + name: "backend".to_string(), + address: SocketAddr::from(([127, 0, 0, 1], 18081)), + tls: false, + sni: None, + trust_bundle_file: None, + timeouts: UpstreamTimeouts { + connection_ms: 100, + total_connection_ms: 200, + read_ms: 200, + write_ms: 200, + idle_ms: 500, + }, + }], + ) + .expect("admission fixture delivery must be valid"); + let limits = RuntimeIsolationLimits::try_new(8, 1) + .expect("admission fixture isolation limits must be valid"); + let proxy = MigrationGatewayProxy::try_new(delivery, limits) + .expect("admission fixture proxy must activate"); + (proxy, limits) + } + #[test] fn migration_context_starts_without_an_admission_lease() { let limits = RuntimeIsolationLimits::try_new(8, 1).expect("fixture limits are valid"); @@ -356,6 +432,28 @@ mod tests { assert!(ctx.admission.is_none()); } + #[test] + fn admission_budget_rejects_and_recovers_after_lease_release() { + let (proxy, limits) = admission_proxy(); + let mut held = MigrationRequestContext::new(limits); + let mut rejected = MigrationRequestContext::new(limits); + + proxy + .admit_request(&mut held) + .expect("first request must acquire the single admission lease"); + let error = proxy + .admit_request(&mut rejected) + .expect_err("second concurrent request must fail closed"); + assert_eq!(error.etype, ErrorType::HTTPStatus(503)); + assert!(rejected.admission.is_none()); + + drop(held); + proxy + .admit_request(&mut rejected) + .expect("released admission lease must make capacity reusable"); + assert!(rejected.admission.is_some()); + } + #[test] fn informational_headers_do_not_start_or_reset_the_response_body_lifetime() { let limits = RuntimeIsolationLimits::try_new_with_response_body_limit(8, 1, 300) @@ -412,10 +510,9 @@ mod tests { }); assert_eq!(body_error.etype, ErrorType::HTTPStatus(413)); - let route_error = MigrationGatewayProxyError::UnmatchedRoute { + let route_error = unmatched_route_to_pingora(MigrationGatewayProxyError::UnmatchedRoute { request_path: "/missing".to_string(), - } - .into_pingora(); + }); assert_eq!(route_error.etype, ErrorType::HTTPStatus(404)); let lifetime_error = response_body_lifetime_to_pingora(ResponseBodyLifetimeExceeded { @@ -428,4 +525,26 @@ mod tests { ); assert_eq!(lifetime_error.esource, ErrorSource::Upstream); } + + #[test] + fn proxy_error_status_preserves_pingora_failure_mapping() { + let explicit = Error::explain(ErrorType::HTTPStatus(503), "saturated"); + assert_eq!(proxy_error_status(&explicit), 503); + + let mut upstream = Error::explain(ErrorType::ConnectError, "origin unavailable"); + upstream.esource = ErrorSource::Upstream; + assert_eq!(proxy_error_status(&upstream), 502); + + let mut downstream = Error::explain(ErrorType::InvalidHTTPHeader, "bad request"); + downstream.esource = ErrorSource::Downstream; + assert_eq!(proxy_error_status(&downstream), 400); + + let mut closed = Error::explain(ErrorType::ConnectionClosed, "peer closed"); + closed.esource = ErrorSource::Downstream; + assert_eq!(proxy_error_status(&closed), 0); + + let mut internal = Error::explain(ErrorType::InternalError, "internal failure"); + internal.esource = ErrorSource::Internal; + assert_eq!(proxy_error_status(&internal), 500); + } } From b805e28fa9f7c4899a52116f9c7d2d5412fc6f46 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 10:58:19 +0900 Subject: [PATCH 42/49] runtime: suppress second error status after committed response --- src/migration_proxy.rs | 18 ++++++++++++++---- 1 file changed, 14 insertions(+), 4 deletions(-) diff --git a/src/migration_proxy.rs b/src/migration_proxy.rs index e0ed19a5..a87f7022 100644 --- a/src/migration_proxy.rs +++ b/src/migration_proxy.rs @@ -215,6 +215,13 @@ fn proxy_error_status(error: &Error) -> u16 { } } +fn proxy_error_response_status(error: &Error, response_already_written: bool) -> u16 { + if response_already_written { + return 0; + } + proxy_error_status(error) +} + #[async_trait] impl ProxyHttp for MigrationGatewayProxy { type CTX = MigrationRequestContext; @@ -329,7 +336,7 @@ impl ProxyHttp for MigrationGatewayProxy { error_value: &Error, _ctx: &mut Self::CTX, ) -> FailToProxy { - let status = proxy_error_status(error_value); + let status = proxy_error_response_status(error_value, session.response_written().is_some()); if status > 0 { let mut response = ServerSession::generate_error(status); if let Err(policy_error) = self.apply_response_headers(&mut response) { @@ -371,9 +378,10 @@ mod tests { use pingora::ErrorSource; use super::{ - body_rejection_to_pingora, enforce_response_body_lifetime, proxy_error_status, - response_body_lifetime_to_pingora, start_response_body_lifetime, unmatched_route_to_pingora, - MigrationGatewayProxy, MigrationGatewayProxyError, MigrationRequestContext, + body_rejection_to_pingora, enforce_response_body_lifetime, proxy_error_response_status, + proxy_error_status, response_body_lifetime_to_pingora, start_response_body_lifetime, + unmatched_route_to_pingora, MigrationGatewayProxy, MigrationGatewayProxyError, + MigrationRequestContext, }; use crate::edge_contract::{UpstreamConfig, UpstreamTimeouts}; use crate::edge_routing::{RouteMatch, RouteRule}; @@ -524,6 +532,8 @@ mod tests { ErrorType::Custom("UpstreamResponseBodyLifetimeExceeded") ); assert_eq!(lifetime_error.esource, ErrorSource::Upstream); + assert_eq!(proxy_error_response_status(&lifetime_error, true), 0); + assert_eq!(proxy_error_response_status(&lifetime_error, false), 502); } #[test] From 00cafa9c1f18be0e3e18b819fbd70e19fd4df506 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 10:59:36 +0900 Subject: [PATCH 43/49] docs: reconcile response lifetime config with current contracts --- API_CONFIG_CONTRACT.md | 20 ++++++++++---------- 1 file changed, 10 insertions(+), 10 deletions(-) diff --git a/API_CONFIG_CONTRACT.md b/API_CONFIG_CONTRACT.md index fa4b370d..2bb3b986 100644 --- a/API_CONFIG_CONTRACT.md +++ b/API_CONFIG_CONTRACT.md @@ -1,4 +1,4 @@ -# Configuration Contracts +# Versioned Configuration Contracts ## Generic `cwl-pingora-gateway` version 1 @@ -23,13 +23,13 @@ upstreams: idle_ms: 10000 ``` -Unknown fields are rejected. `version` must be `1`. `listener`, `metrics_listener`, and `address` are socket addresses with non-zero ports. Port zero is rejected because this deployment contract requires stable operator-declared listener authority and a concrete connectable upstream rather than OS-selected ephemeral listener ports or unusable upstream destinations. Traffic and metrics listeners must not overlap one effective socket authority. Equal addresses, same-port same-family wildcard/concrete aliases, exact IPv4-mapped IPv6 aliases of the same IPv4 address, an IPv4 wildcard paired with any mapped IPv4 authority, a mapped IPv4 wildcard paired with any native or mapped IPv4 authority, and the platform-dependent same-port IPv6-wildcard/IPv4 combination all fail closed; distinct concrete non-aliased addresses may use the same port. `max_request_body_bytes`, `max_in_flight_requests`, and `upstream_keepalive_pool_size` must all be positive. Generic v1 requires exactly one upstream and a non-empty stable upstream name. Every timeout must be positive. +Unknown fields are rejected. `version` must be `1`. `listener`, `metrics_listener`, and `address` are socket addresses with non-zero ports. Port zero is rejected because production authority must remain operator-declared rather than OS-selected or unusable. Traffic and metrics listeners must not overlap one effective socket authority: equal sockets, same-port same-family wildcard/concrete aliases, exact native/IPv4-mapped aliases, native or mapped IPv4 wildcard aliases, and the platform-dependent same-port IPv6-wildcard/IPv4 combination fail closed. Distinct concrete non-aliased addresses may share a non-zero port. `max_request_body_bytes`, `max_in_flight_requests`, and `upstream_keepalive_pool_size` must all be positive. Generic v1 requires exactly one upstream and a non-empty stable upstream name. Every timeout must be positive. -The timeout fields map directly to the pinned Pingora peer options rather than defining a second gateway timer model. In particular, `read_ms` is a **per-read inactivity budget**: Pingora waits at most that long for each individual upstream `read()` and resets the timer after a successful read. It is not a total-response deadline. A connected upstream that sends no response bytes is therefore bounded by `read_ms`, while a slow-drip response can remain alive across multiple successful reads. Generic v1 still has no whole-response lifetime and must not infer one from `read_ms`. +The timeout fields map directly to the pinned Pingora peer options rather than defining a second gateway timer model. In particular, `read_ms` is a **per-read inactivity budget**: Pingora waits at most that long for each individual upstream `read()` and resets the timer after a successful read. It is not a total-response deadline. A connected upstream that sends no response bytes is therefore bounded by `read_ms`, while a slow-drip response can remain alive across multiple successful reads. Generic v1 has no whole-response lifetime and must not infer one from `read_ms`. `max_in_flight_requests` is a process-local backpressure boundary for non-health downstream requests. When the budget is exhausted, the runtime fails fast with HTTP 503 instead of admitting unbounded work. `/livez` and `/readyz` bypass this application admission budget so saturation does not hide process health. The admission lease is released when the request context ends, including failed requests. `upstream_keepalive_pool_size` is wired directly into Pingora's `ServerConf`; the runtime does not inherit Pingora's framework default of 128 reusable upstream connections. -`tls: true` requires non-empty `sni`, which Pingora uses for SNI and hostname verification together with certificate verification. `trust_bundle_file` is optional and, when present, must be an absolute path to a non-empty PEM certificate bundle readable during peer activation before listeners open. The bundle supplies trust anchors for that upstream instead of changing certificate-authority ownership: issuance and rotation remain external responsibilities. If `trust_bundle_file` is omitted, Pingora uses platform trust roots. `tls: false` forbids both `sni` and `trust_bundle_file`. +`tls: true` requires a non-empty RFC 6066 SNI `HostName`: an ASCII DNS hostname without a trailing root dot or literal IPv4/IPv6 address. Each DNS label must be 1–63 ASCII letters, digits, or hyphens, with an alphanumeric first and last byte; the complete textual name is limited to 253 bytes. Internationalized names must therefore arrive as ASCII IDNA A-labels rather than raw Unicode U-labels. Invalid SNI identity fails during transport-neutral configuration validation before Pingora peer construction or listener activation. Pingora then uses the admitted name for SNI and hostname verification together with certificate verification. `trust_bundle_file` is optional and, when present, must be an absolute path to a non-empty PEM certificate bundle readable during peer activation before listeners open. The bundle supplies trust anchors for that upstream instead of changing certificate-authority ownership: issuance and rotation remain external responsibilities. If `trust_bundle_file` is omitted, Pingora uses platform trust roots. `tls: false` forbids both `sni` and `trust_bundle_file`. The generic contract does not include route tables, user-selected destinations, credentials, downstream certificates, ACME, retry counts, static roots, WebSocket switches, or load-balancer policy. Adding one of those fields changes public semantics and requires a versioned contract/ADR plus behavior tests. @@ -37,7 +37,7 @@ Generic v1 downstream transport is cleartext TCP. Before proxying, the generic a ## Bounded `cwl-pingora-pg-erd-migration` candidate -The dedicated pg-erd migration binary consumes a different, migration-specific Admin Config profile. Version 1 remains readable only to preserve the existing unreleased characterization stack. Version 2 is the opt-in response-lifetime increment and requires an explicit positive `max_upstream_response_body_ms`; version 1 rejects that field so the old contract cannot silently acquire new timing semantics. +The dedicated pg-erd migration binary consumes a different, migration-specific Admin Config profile. Version 1 preserves the existing unreleased characterization semantics and rejects `max_upstream_response_body_ms`. Version 2 is the explicit response-lifetime increment and requires a positive `max_upstream_response_body_ms`; the runtime never injects a hidden default. ```yaml version: 2 @@ -68,17 +68,17 @@ upstreams: idle_ms: 10000 ``` -The numeric value above is an illustrative configuration example, not a pg-erd production SLO. A deployment owner must choose the version-2 value from its observed long-response contract before canary or cutover. Version 2 rejects zero or a missing response-body lifetime rather than substituting a hidden default. +The numeric response-lifetime value above is illustrative configuration, not a production SLO. A deployment owner must choose the version-2 value from its observed long-response contract before canary or cutover. -`max_upstream_response_body_ms` starts when Pingora invokes the upstream-response-header filter for the first non-informational response, before body-progress callbacks are processed. Runtime Isolation compares elapsed monotonic time only when a non-empty upstream body chunk is actually observed. Once that progress boundary is at or beyond the configured lifetime, the callback raises an upstream-scoped fatal error. Empty/end-of-stream bookkeeping callbacks do not create a false timeout. If the response status/header was already committed, the gateway terminates that incomplete downstream response instead of inventing a second status or silently routing to the other pg-erd origin. The ordinary request context then drops its in-flight admission lease. +`max_upstream_response_body_ms` starts at the first non-informational upstream response header. Runtime Isolation compares elapsed monotonic time only when a non-empty upstream body chunk is observed. Progress at or beyond the configured lifetime raises an upstream-scoped fatal error; empty/end-of-stream bookkeeping callbacks do not manufacture a timeout. If the final response has already been written, `fail_to_proxy` observes Pingora's `Session::response_written()` commitment state and emits no second status. Pre-commit upstream failures retain the existing policy-complete local error response behavior. No route failover is introduced by this contract. -This callback guard is deliberately not described as an exact timer interrupt. At the pinned Pingora revision, `read_ms` still applies independently to each upstream read and resets after a successful read. A continuously progressing body is therefore stopped at the first non-empty body callback at or beyond `max_upstream_response_body_ms`; a response that becomes quiescent is bounded by `read_ms`. The current callback surface does not wake a pending read at the absolute body-lifetime instant, and slow-drip of an incomplete **response header** remains a separate transport gap. Neither limitation may be hidden in parity or production-SLO claims. +This callback guard is not an exact timer interrupt. `read_ms` remains a per-read inactivity budget that resets after a successful read. A continuously progressing body is stopped at the first non-empty body callback at or beyond `max_upstream_response_body_ms`; a response that becomes quiescent is bounded by `read_ms`. The current callback surface does not wake a pending read exactly at the body-lifetime instant, and slow delivery of an incomplete response header remains a separate transport gap. -This is not a generic multi-route configuration language. Operator input can bind only concrete transport/TLS values for the compiled `backend` and `frontend` identities. Missing, extra, duplicate, renamed, port-zero, or otherwise invalid listener/metrics/upstream transport authorities fail closed before listener activation. Port zero is rejected because this deployment contract requires stable operator-declared socket authority rather than an OS-selected ephemeral listener or an unusable upstream destination. Listener and metrics authority use the same effective-authority invariant as generic v1, including exact and wildcard IPv4-mapped aliases, same-port same-family wildcard/concrete aliases, and the platform-dependent same-port IPv6-wildcard/IPv4 combination; distinct concrete non-aliased addresses remain admissible. Routes and edge-owned response fields are not configurable: the characterized profile fixes exact `/healthz -> backend`, raw `PathPrefix(`/api`) -> backend` semantics including `/apiary`, fallback `/ -> frontend`, and the four captured response fields `X-Content-Type-Options: nosniff`, `X-Frame-Options: DENY`, `Referrer-Policy: no-referrer`, and `Permissions-Policy: geolocation=(), microphone=(), camera=()`. +This is not a generic multi-route configuration language. Operator input can bind only concrete transport/TLS values for the compiled `backend` and `frontend` identities. Missing, extra, duplicate, renamed, port-zero, or otherwise invalid listener/metrics/upstream transport authorities fail closed before listener activation. Listener and metrics sockets consume the same effective-authority invariant as generic v1, while the migration profile keeps its specific zero-transport-authority error contract. Routes and edge-owned response fields are not configurable: the characterized profile fixes exact `/healthz -> backend`, raw `PathPrefix(`/api`) -> backend` semantics including `/apiary`, fallback `/ -> frontend`, and the four captured response fields `X-Content-Type-Options: nosniff`, `X-Frame-Options: DENY`, `Referrer-Policy: no-referrer`, and `Permissions-Policy: geolocation=(), microphone=(), camera=()`. Admin parsing validates only deterministic configuration and authority invariants. It does not read custom trust-bundle bytes. If an admitted TLS upstream supplies `trust_bundle_file`, the canonical Pingora peer adapter reads and parses that material exactly once during `build_proxy`, still before listeners are registered. An unreadable or invalid bundle therefore blocks activation without a validate-then-reload trust-file window. -The migration adapter reserves `/livez` and `/readyz` as process-local Pingora health endpoints and does not route them to either consumer origin. The legacy consumer `/healthz` remains distinct routed application traffic to `backend`. Hostile request-controlled `Forwarded`, `X-Forwarded-*`, `X-Real-IP`, and `X-Forwarded-Server` identity is not trusted; characterized compatibility fields are rebuilt from accepted downstream transport/request authority. The current captured Traefik entryPoint is cleartext `web`, so this candidate emits downstream scheme `http`. HTTPS/TLS listener behavior requires a separate executable contract. +The migration adapter reserves `/livez` and `/readyz` as process-local Pingora health endpoints and does not route them to either consumer origin. The legacy consumer `/healthz` remains distinct routed application traffic to `backend`. Hostile request-controlled `Forwarded`, `X-Forwarded-*`, `X-Real-IP`, and `X-Forwarded-Server` identity is not trusted. `X-Forwarded-For` and `X-Real-IP` are rebuilt from the accepted client socket; `X-Forwarded-Host` preserves the original Host authority; `X-Forwarded-Port` uses the explicit Host port when present or the admitted scheme default otherwise. The process listener bind port is deliberately not external authority because container, Service, NAT, and port-publish layers may expose a different public port. The current captured Traefik entryPoint is cleartext `web`, so this candidate emits downstream scheme `http`. HTTPS/TLS listener behavior requires a separate executable contract. The migration profile cannot configure product authentication/business rules, Keyverse identity, Wardnet/EgressWeave verdicts, certificate issuance/rotation, service discovery, arbitrary destinations, or Context Graph/EA state. Source-level listener capability is not release, deployment, parity, canary, cutover, or legacy-removal evidence. From 371421f658c2ea292a9a1d826303513a92fa1b6a Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 11:00:22 +0900 Subject: [PATCH 44/49] docs: reconcile response lifetime operability with current parent --- OPERABILITY.md | 22 ++++++++++------------ 1 file changed, 10 insertions(+), 12 deletions(-) diff --git a/OPERABILITY.md b/OPERABILITY.md index e880158f..55abebcf 100644 --- a/OPERABILITY.md +++ b/OPERABILITY.md @@ -2,11 +2,11 @@ ## Start -Run the generic process as `cwl-pingora-gateway --config /path/to/gateway.yaml`. Configuration is read and validated before the listener is registered. Invalid or missing configuration exits non-zero. `max_request_body_bytes`, `max_in_flight_requests`, and `upstream_keepalive_pool_size` are mandatory positive deployment inputs; the process will not start with a zero value or silently inherit a Pingora keepalive default. If a TLS upstream declares `trust_bundle_file`, that absolute PEM bundle is read and parsed during peer activation before listeners open; unreadable, empty, or malformed trust material prevents activation. +Run the generic process as `cwl-pingora-gateway --config /path/to/gateway.yaml`. Configuration is read and validated before the listener is registered. Invalid or missing configuration exits non-zero. `max_request_body_bytes`, `max_in_flight_requests`, and `upstream_keepalive_pool_size` are mandatory positive deployment inputs; the process will not start with a zero value or silently inherit a Pingora keepalive default. Traffic, metrics and upstream transport ports must be non-zero, and traffic/metrics listeners must not overlap the same effective socket authority. If a TLS upstream declares `trust_bundle_file`, that absolute PEM bundle is read and parsed during peer activation before listeners open; unreadable, empty, or malformed trust material prevents activation. The pg-erd migration candidate is a separate executable: `cwl-pingora-pg-erd-migration --config /path/to/pg-erd-migration.yaml`. It consumes the bounded `PgErdMigrationConfig` profile rather than widening generic `GatewayConfig` v1. The profile admits exactly the compiled `backend` and `frontend` transport identities plus deployment-variable sockets, runtime budgets and upstream transport/TLS values. Route tables, response-policy fields, product authentication/business rules, Keyverse identity, Wardnet/EgressWeave verdicts, service discovery and arbitrary destinations are not operator-configurable. -Pg-erd configuration version 1 preserves the existing unreleased characterization behavior and has no response-body lifetime. Version 2 is opt-in and requires a positive `max_upstream_response_body_ms`; version 1 rejects that field rather than silently acquiring new timing semantics. Choose the version-2 value from observed application long-response requirements before canary or cutover. Documentation/example values are not production SLOs. A zero or incomplete version-2 budget fails before listeners open. +Pg-erd configuration version 1 preserves the existing unreleased characterization behavior and rejects `max_upstream_response_body_ms`. Version 2 requires an explicit positive `max_upstream_response_body_ms`; zero or omission fails before listener activation. Choose the value from observed application long-response requirements before canary or cutover. Example values are not production SLOs. Admin parsing is side-effect free with respect to custom trust bytes. It validates the exact transport-authority set and `UpstreamConfig` invariants first. `build_proxy` then materializes Pingora peers and any custom PEM trust bundle once, still before listener registration. This avoids reading mutable trust material in a validation pass and reading it again for activation. @@ -18,17 +18,17 @@ Both process identities reserve `/livez` and `/readyz` and return 200 with `Cach `max_in_flight_requests` limits concurrently admitted non-health requests for one gateway process. At capacity the gateway fails new application traffic fast with HTTP 503 and increments `cwl_pingora_gateway_backpressure_rejections_total`; it does not queue unbounded work. Process health probes bypass that admission budget so operators can distinguish process health from traffic saturation. The request lease is released when the Pingora request context ends, including error paths, and a subsequent request is admissible again. -For pg-erd version 2, `max_upstream_response_body_ms` limits elapsed body delivery after the first non-informational upstream response header. It complements the peer's `read_ms`; it does not replace it. `read_ms` remains a per-read inactivity timeout that resets whenever Pingora successfully reads more upstream data. Continuous response-body slow-drip is stopped at the first **non-empty body-progress callback** at or beyond the explicit lifetime, producing an upstream-scoped request error. Empty/end-of-stream bookkeeping callbacks are not treated as body progress. If HTTP status/headers were already committed, the gateway terminates that incomplete response rather than sending a second status or switching to `frontend`. The request context then releases its in-flight lease. +For pg-erd version 2, `max_upstream_response_body_ms` begins at the first non-informational upstream response header. Runtime Isolation checks elapsed monotonic time only on non-empty upstream body progress. `read_ms` remains a separate per-read inactivity timeout and resets after successful reads. A continuously progressing body is stopped at the first non-empty body callback at or beyond the configured lifetime; a quiescent response remains bounded by `read_ms`. -This control is not an exact wall-clock interrupt. With the pinned Pingora callback API, an already pending read is not awakened solely because the body-lifetime instant elapsed; a quiescent response remains bounded by `read_ms`, and continuously progressing traffic is checked when non-empty body progress reaches the callback. Slow-drip of an incomplete response header is still a separate gap. Do not advertise the configured number as a strict production deadline until representative deployment traffic has measured the actual scheduler/read-callback behavior. +The response-lifetime guard is not an exact wall-clock interrupt: the pinned Pingora callback surface does not wake an already-pending read only because the lifetime instant elapsed, and slow delivery of an incomplete response header is a separate gap. When a final response is already committed, `fail_to_proxy` observes `Session::response_written()` and does not emit a second status. Pre-commit upstream failures continue to use the existing policy-complete local error response. No route failover is introduced by the lifetime contract. `upstream_keepalive_pool_size` is copied into Pingora `ServerConf` before bootstrap. Choose it with expected upstream concurrency, origin capacity, instance count and connection reuse in mind. It limits retained reusable upstream connections; it is not a substitute for the downstream in-flight admission limit and does not create product-domain load-balancing semantics. ## Forwarding and protocol boundary -The generic v1 adapter and the pg-erd migration adapter have different compatibility forwarding contracts. Generic v1 strips inbound proxy-identity fields and emits only `Forwarded: proto=http`. The pg-erd migration adapter removes request-controlled `Forwarded`, `X-Forwarded-*`, `X-Real-IP` and legacy `X-Forwarded-Server` authority, then rebuilds only the characterized `X-Forwarded-For`, `X-Real-IP`, `X-Forwarded-Host`, `X-Forwarded-Port`, and `X-Forwarded-Proto` fields from accepted downstream transport/request authority. Product identity, tenant identity and authorization are never derived from these transport fields by the shared gateway. +The generic v1 adapter and the pg-erd migration adapter have different compatibility forwarding contracts. Generic v1 strips inbound proxy-identity fields and emits only `Forwarded: proto=http`. The pg-erd migration adapter removes request-controlled `Forwarded`, `X-Forwarded-*`, `X-Real-IP` and legacy `X-Forwarded-Server` authority. It rebuilds `X-Forwarded-For` and `X-Real-IP` from the accepted client socket, preserves the original Host authority in `X-Forwarded-Host`, derives `X-Forwarded-Port` from the explicit Host port or the admitted scheme default, and emits the characterized `X-Forwarded-Proto`. Do not substitute the process listener bind port for the original authority: Kubernetes Services, containers, NAT and port publishing may expose a different external port. Product identity, tenant identity and authorization are never derived from these transport fields by the shared gateway. -The captured pg-erd Traefik entryPoint is clear-text `web`, so this migration profile currently uses downstream scheme `http`. The current versioned contract rejects HTTP/1 protocol-transition attempts with 501 before request admission and route/upstream selection; compiled traffic separately observes no dedicated-origin connection for a bounded 500 ms post-response window. Do not deploy this profile behind a TLS listener and assume `X-Forwarded-Proto: https` parity. Downstream TLS, HTTP/2, HTTP/3, versioned WebSocket/Extended CONNECT enablement and broader long-lived streaming behavior require separate executable contracts before they become migration behavior. +The captured pg-erd Traefik entryPoint is clear-text `web`, so this migration profile currently uses downstream scheme `http`. Do not deploy it behind a TLS listener and assume `X-Forwarded-Proto: https` parity. Downstream TLS, HTTP/2, HTTP/3, WebSocket/upgrade and broader streaming behavior require separate executable contracts before they become admitted migration behavior. ## Logging @@ -42,17 +42,15 @@ The runtime does not inherit Pingora's retry, keepalive-pool, or drain defaults. SIGTERM uses Pingora graceful termination with an explicit 5-second request-drain grace period and a 10-second runtime-shutdown timeout. The pinned Pingora server calls Tokio `Runtime::shutdown_timeout` with that timeout and then sleeps for the same timeout while service runtimes are shut down in parallel. The policy therefore requires a 30-second supervisor hard-kill budget: its modeled worst-case Pingora process budget is 25 seconds plus scheduler/process-exit overhead. A Kubernetes-style deployment must set `terminationGracePeriodSeconds` to at least 30 or provide an equivalent supervisor budget; a shorter external kill deadline is not an admitted deployment contract. -`tests/graceful_shutdown.rs` exercises the generic compiled binary with a held upstream response. `tests/production_path.rs` covers generic saturation, health and failure recovery. `tests/pg_erd_production_path.rs` exercises the dedicated pg-erd process with real loopback backend/frontend origins, including process-local health, characterized route/response-header behavior, transport-derived forwarding replacement and declared body rejection. `tests/pg_erd_slow_drip_response_traffic.rs` separately proves the version-2 body-lifetime behavior with continuous progress faster than `read_ms`. `tests/pg_erd_binary_startup.rs` requires missing/invalid configuration and unreadable trust material to fail before listener activation. `tests/pingora_diagnostic_log_safety.rs` runs the compiled generic process with broad trace diagnostics, proves the secret-bearing URI/Host/Authorization/Cookie request reaches the origin, requires a new Pingora redaction marker after the readiness probes, and requires none of those sentinels in process stderr. These source contracts become evidence only after terminal success on the exact current head; predecessor success never transfers. +`tests/graceful_shutdown.rs` exercises the generic compiled binary with a held upstream response. `tests/production_path.rs` covers generic saturation, health and failure recovery. `tests/pg_erd_production_path.rs` exercises the dedicated pg-erd process with real loopback backend/frontend origins, including process-local health, characterized route/response-header behavior, Host-authority forwarding replacement and declared body rejection. `tests/pg_erd_slow_drip_response_traffic.rs` separately exercises version-2 continuously progressing response bodies that stay inside each `read_ms` interval but cross the explicit lifetime; the downstream must keep its committed status/framing, terminate incomplete, avoid route failover or a second status, preserve `/readyz`, record the bounded error telemetry, and allow independent-route recovery. `tests/pg_erd_binary_startup.rs` requires missing/invalid configuration and unreadable trust material to fail before listener activation. `tests/pingora_diagnostic_log_safety.rs` runs the compiled generic process with broad trace diagnostics, proves the secret-bearing URI/Host/Authorization/Cookie request reaches the origin, requires a new Pingora redaction marker after the readiness probes, and requires none of those sentinels in process stderr. These source contracts become evidence only after terminal success on the exact current head; predecessor success never transfers. ## Container Run as a non-root user and prefer a read-only root filesystem. Mount only the versioned config and any required upstream trust bundle read-only. The runtime does not intentionally write logs or state files; stdout/stderr should be collected by the platform. Do not bake secrets or private keys into the image or config. -The Dockerfile has one build-time process selector, `CWL_GATEWAY_BIN`, with an explicit allowlist of `cwl-pingora-gateway` and `cwl-pingora-pg-erd-migration`. The selected executable is copied to one fixed distroless runtime path, so the final image contains neither a shell selector nor both product profiles. The default remains the generic runtime. Build the bounded pg-erd candidate with `docker build --build-arg CWL_GATEWAY_BIN=cwl-pingora-pg-erd-migration -t cwl-pingora-pg-erd-migration: .`; any other selector value fails the image build. +The Dockerfile exposes one build-time selector, `CWL_GATEWAY_BIN`, fail-closed to exactly `cwl-pingora-gateway` or `cwl-pingora-pg-erd-migration`. The selected executable is copied to one fixed distroless runtime path, so the final image contains one admitted process identity and no shell-based runtime selector. The default remains the generic runtime. Build the pg-erd profile with `docker build --build-arg CWL_GATEWAY_BIN=cwl-pingora-pg-erd-migration -t cwl-pingora-pg-erd-migration: .`. -CI now defines executable OCI acceptance for both admitted images. Each image must declare uid/gid `65532`, start with a read-only root filesystem, all Linux capabilities dropped and `no-new-privileges`, and reach its process-local `/livez` endpoint from a read-only mounted versioned configuration. `examples/pg-erd-migration.yaml` is an OCI smoke fixture only: its loopback origins are placeholders because `/livez` does not establish product-origin health. Dedicated routed origin/load/failure evidence remains a separate gate. Until the exact current head actually runs these jobs to terminal success, this source-defined acceptance is not hosted GREEN evidence. - -The supply-chain candidate lane builds both admitted images, scans each fail-closed for HIGH/CRITICAL OS/library vulnerabilities, records both local image IDs against the exact source SHA, and uploads the common committed-lock/dependency SBOM plus per-image scan results. These are unreleased candidate receipts; release still requires immutable registry digests, release-bound SBOM/provenance/reproducibility evidence, and rollback rehearsal. +CI defines executable OCI acceptance for both admitted images. Each must declare uid/gid `65532`, start with a read-only root filesystem, all Linux capabilities dropped and `no-new-privileges`, consume only a read-only configuration mount, and expose its process-local `/livez`. `examples/pg-erd-migration.yaml` is an OCI smoke fixture only; its loopback origins are placeholders because `/livez` does not establish product-origin health. The supply-chain lane builds and fail-closed scans both candidate images, binds both local image IDs and per-image scan outputs to the exact source SHA, and keeps failure diagnostics separate from the promotion-shaped success artifact. These are unreleased candidate receipts, not registry release identities. ## Cutover and rollback @@ -60,4 +58,4 @@ A consumer migration must keep the last known-good deployment manifest/image dig Roll back by restoring the exact protected prior deployment revision, not by editing a live container. Certificate management, identity, product authorization/business policy, and security-verdict ownership must remain with their existing bounded owners during edge-runtime rollback. -No consumer may pin `pingora-gateway` until a protected release publishes an immutable image digest and rollback has been rehearsed. Dedicated pg-erd image source/acceptance now exists on the candidate branch, but it still needs terminal exact-head OCI/supply-chain execution plus routed load/failure evidence before release/canary eligibility. +No consumer may pin `pingora-gateway` until a protected release publishes an immutable image digest and rollback has been rehearsed. Dedicated pg-erd image source/acceptance exists on this candidate, but it still requires terminal exact-head OCI/supply-chain execution plus routed load/failure evidence before release/canary eligibility. From 6d4e909896781016c9070f2cba58f431d67f1f1d Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 11:03:09 +0900 Subject: [PATCH 45/49] docs: reconcile response lifetime test strategy with current evidence --- TEST_STRATEGY.md | 58 ++++++++++++++++++++++-------------------------- 1 file changed, 26 insertions(+), 32 deletions(-) diff --git a/TEST_STRATEGY.md b/TEST_STRATEGY.md index d1c008b1..0a885013 100644 --- a/TEST_STRATEGY.md +++ b/TEST_STRATEGY.md @@ -1,65 +1,59 @@ # Test Strategy -Tests are organized by responsibility and evidence boundary rather than by implementation layer. A source test is not acceptance evidence until the unchanged exact head passes the relevant hosted gates; parent receipts are never transferred to a changed child. +Tests are organized by responsibility and evidence boundary. A source test is not promotion evidence until the unchanged exact head passes the applicable hosted gates; parent or historical receipts never transfer to a changed child. ## Generic gateway contracts -`tests/config_contract.rs`, `tests/trust_bundle_contract.rs`, `tests/startup_contract.rs`, `tests/pingora_peer_adapter.rs`, `tests/gateway_proxy.rs`, and `tests/runtime_policy.rs` cover strict Admin Config, TLS trust loading, fail-closed startup, Pingora peer mapping, admission budgets, no-retry policy, keepalive and bounded shutdown. Compiled-process suites cover generic startup, trusted local-CA/SNI behavior, production-path health and forwarding reconstruction, declared/streamed request-body rejection, saturation/backpressure recovery, and graceful SIGTERM drain. +`tests/config_contract.rs`, `tests/trust_bundle_contract.rs`, `tests/tls_sni_hostname_contract.rs`, `tests/startup_contract.rs`, `tests/pingora_peer_adapter.rs`, `tests/gateway_proxy.rs`, and `tests/runtime_policy.rs` cover strict Admin Config, non-zero and non-overlapping network authority, RFC 6066 SNI hostname admission, custom trust material, fail-closed startup, Pingora peer mapping, request/in-flight/keepalive budgets, one-attempt policy, and bounded shutdown. -The generic `load-contract` is deliberately separate from functional production-path tests. It builds the release-mode gateway plus the bounded Rust origin in `tests/load/load_origin.rs`, format-checks and directly tests that origin, and executes checksum-pinned k6 2.2.0. The measured load job contains no Python invocation. Generic `tests/load/gateway_smoke.js` sends 400 requests across four VUs, requires exact status/body preservation and zero HTTP failures, and gates controlled loopback `http_req_duration` p95 below 20 ms. This remains regression evidence for the local generic path, not a production SLO claim. +`tests/binary_startup.rs`, `tests/trust_bundle_startup.rs`, `tests/production_path.rs`, `tests/local_ca_tls.rs`, and `tests/graceful_shutdown.rs` exercise the compiled generic process. Production-path readiness requires a complete bounded `/readyz` HTTP/1.1 200 carrying `Cache-Control: no-store`, not a bare TCP accept. Traffic and metrics loopback reservations are retained through config construction and released only at child-bind handoff so port reuse cannot manufacture readiness evidence. -## Characterized pg-erd migration contracts +The generic load lane builds the release-mode gateway and bounded Rust origin in `tests/load/load_origin.rs`, format-checks and directly tests that origin, then uses checksum-pinned k6. `tests/load/gateway_smoke.js` requires exact status/body preservation, zero HTTP failures, and controlled-loopback p95 below 20 ms. This is a regression bound, not a production SLO. -`tests/pg_erd_route_contract.rs` freezes observed `pg-erd-cloud` path precedence, including literal raw `/api` prefix behavior. `tests/pg_erd_http_policy_contract.rs` freezes separately owned response-policy fields and rejects malformed or ambiguous header identity. Migration-plan, upstream-binding, forwarding and runtime-proxy suites prove the transport-neutral plan, exact upstream authority, forwarding-trust boundary, response-policy composition and shared isolation/observability callbacks. +## Characterized pg-erd migration contracts -`tests/pg_erd_admin_config_contract.rs` proves that operator configuration is fail closed: unknown/incomplete/future configuration, listener collision, invalid runtime/keepalive budgets, missing/extra/duplicate/renamed authorities and invalid transport bindings are rejected. Only concrete `backend` and `frontend` transport authorities may be bound; route selection remains compiled into the characterized migration plan rather than becoming an operator route DSL. `tests/pg_erd_response_lifetime_config.rs` covers the opt-in version-2 increment: it requires an explicit positive `max_upstream_response_body_ms`, rejects zero or a missing version-2 value, and proves version 1 cannot silently acquire version-2 timing semantics. `tests/pg_erd_binary_startup.rs` exercises the dedicated compiled process and trust-material activation failures before listener authority is granted. +`tests/pg_erd_route_contract.rs` freezes the observed `pg-erd-cloud` route precedence, including literal raw `/api` prefix behavior. `tests/pg_erd_http_policy_contract.rs` freezes the separately owned response-security fields and RFC 9110-compatible header admission. `tests/pg_erd_migration_plan_contract.rs`, `tests/pg_erd_upstream_binding_contract.rs`, `tests/pg_erd_forwarding_contract.rs`, and `tests/pg_erd_runtime_proxy_contract.rs` prove the transport-neutral route/policy composition, exact upstream-authority binding, forwarding-trust boundary, failure-response policy, runtime isolation, and shared observability callbacks. Characterization alone is not listener parity evidence. -`tests/pg_erd_local_ca_tls.rs` separately proves the bounded migration composition root consumes its own admitted upstream TLS contract rather than inheriting generic evidence. A one-day local CA and `backend.test` certificate are generated at test time. Matching explicit CA plus SNI must carry characterized `/api/tls` traffic to the TLS backend and preserve a distinct clear-text fallback route. With the same valid CA but mismatched SNI, the backend route must fail as exact HTTP 502 without route failover, `/readyz` must remain 200, the TLS backend must receive no HTTP request bytes, and a later independent fallback request must succeed. Traffic/metrics ports remain simultaneously reserved until startup; accepted origin sockets use five-second I/O deadlines; request headers are capped at 64 KiB; exact case-sensitive HTTP/1.1 three-digit status parsing rejects `2000`/protocol-case false GREEN. This is gateway-to-upstream TLS trust/hostname evidence only, not downstream TLS termination, certificate lifecycle ownership or representative TLS performance. +`tests/pg_erd_admin_config_contract.rs` proves fail-closed bounded Admin Config. Unknown or future versions, listener collision, zero runtime/keepalive budgets, missing/extra/duplicate/renamed upstream authorities, and invalid concrete transport/TLS data are rejected. Only `backend` and `frontend` may bind the compiled route profile. `tests/listener_authority_contract.rs`, `tests/network_authority_self_loop_contract.rs`, and `tests/upstream_unicast_authority_contract.rs` freeze zero-port, wildcard/native/mapped alias, recursive self-loop, broadcast, and multicast rejection while retaining distinct concrete unicast authority where valid. `tests/pg_erd_binary_startup.rs` requires invalid configuration or unusable trust material to fail before listener activation. -`tests/network_authority_self_loop_contract.rs` covers the additional gateway-owned authority invariant shared by generic and pg-erd configuration: an upstream may not alias either the public traffic listener or the internal metrics listener through exact, wildcard, dual-stack, or mapped/native socket equivalence. Distinct concrete non-aliased IP authorities on the same port remain admitted. Public construction paths are revalidated after direct deserialization, and both compiled composition roots must fail closed before listener activation on recursive authority. +`tests/pg_erd_local_ca_tls.rs` proves the bounded migration composition root consumes its own admitted upstream TLS contract. A short-lived local CA and `backend.test` certificate must succeed only with matching CA/SNI; a mismatched SNI with the same CA must fail as exact HTTP 502 without route failover, preserve `/readyz`, emit no HTTP request bytes to the rejected TLS backend, and leave an independent clear-text fallback route usable. This is upstream TLS trust/hostname evidence only, not downstream TLS termination or certificate-lifecycle authority. -`tests/pg_erd_production_path.rs` drives real loopback `backend` and `frontend` origins through `cwl-pingora-pg-erd-migration`. It covers gateway-local live/readiness endpoints, backend and fallback routing, forwarding identity reconstruction, characterized response-policy replacement and pre-origin 413 rejection. +## Protocol-transition boundary -Runtime and failure traffic are split by causal phase. `tests/pg_erd_runtime_isolation_traffic.rs` covers streamed body overflow plus in-flight saturation/recovery and exact rejection telemetry. `tests/pg_erd_upstream_failure_traffic.rs` creates deterministic `ECONNREFUSED`, requires bounded 502 recovery, exact request-error telemetry and later frontend success. `tests/pg_erd_read_stall_traffic.rs` keeps an accepted origin connection silent and open so read inactivity cannot be confused with origin closure. `tests/pg_erd_slow_drip_response_traffic.rs` distinguishes that inactivity contract from the version-2 response-body lifetime: the backend commits HTTP 200 with `Content-Length: 20` and sends one byte every 60 ms while `read_ms` is 500 ms, but `max_upstream_response_body_ms` is 300 ms. The gateway must terminate before all 20 bytes arrive, preserve the committed 200/framing rather than inventing a second status/failover, increment exact request-error telemetry, keep `/readyz` available, and serve a later independent frontend route. The lifetime starts at the first non-informational upstream response header and is enforced only on non-empty body-progress callbacks, so empty/end-of-stream bookkeeping cannot manufacture a timeout. The fixture proves continuous body slow-drip is bounded; it does not claim an exact timer interrupt of a pending read or an absolute deadline for incomplete response headers. `tests/pg_erd_graceful_shutdown.rs` holds a routed request in flight, fixes one SIGTERM-relative external termination deadline, releases the response during the shared grace period, requires downstream 200 and clean process exit before that same deadline. +`tests/protocol_transition_policy.rs` treats either an `Upgrade` field or a case-insensitive comma-delimited `upgrade` token in `Connection` as an uncharacterized HTTP/1 protocol-transition attempt. Ordinary HTTP and unrelated tokens such as `x-upgrade` remain admitted. Both generic and pg-erd request filters reject the transition with HTTP 501 before request admission or route/upstream selection. -`tests/pg_erd_partial_response_traffic.rs` covers the post-header orderly-close phase. The backend sends HTTP 200 with `Content-Length: 20` and only the seven-byte `partial` prefix, then waits until the downstream has observed the complete header block plus that exact prefix before closing. Acceptance requires the committed status/framing to remain visible without a fabricated second status or silent failover, `/readyz` to stay 200, exact `cwl_pingora_gateway_request_errors_total 1`, and an independent frontend recovery request. The framing oracle parses header lines, matches `Content-Length` field identity case-insensitively, trims field-value whitespace, requires exactly one value equal to `20`, and rejects lookalike or duplicate/conflicting fields. Exact #21 has terminal exact-head CI/Supply Chain evidence; descendants must revalidate rather than transfer that receipt. +`tests/peer_protocol_policy.rs` locks the same boundary at immutable peer construction through `HttpUpstreamRequestPolicy::deny_upgrades()`, rather than inheriting the supplier `WebSocketOnly` default. `tests/protocol_transition_traffic.rs` verifies both compiled roots return exact 501, make no dedicated-origin connection during a bounded observation window, and keep `/readyz` available. This is explicit non-support evidence, not WebSocket parity; WebSocket enablement, HTTP/2 Extended CONNECT, and HTTP/3/QUIC require separate versioned contracts and realistic tunnel/concurrency/disconnect/backpressure/drain tests. -On Linux, `tests/pg_erd_upstream_reset_traffic.rs` covers the distinct pre-header abortive-reset phase. The characterized backend must first receive the complete `/api/reset` request headers, then configure `SO_LINGER(0)` and close the established socket without sending response bytes. The gateway must return HTTP 502 within two seconds despite a configured five-second upstream read-inactivity budget, must not silently fail over to `frontend`, must keep `/readyz` 200, must expose exactly one `cwl_pingora_gateway_request_errors_total 1` sample, and must still route a later independent frontend request successfully. Traffic/metrics ports are held concurrently until process start, origin header reads are bounded to five seconds and 64 KiB, the metric oracle matches a complete sample line, and response status uses an exact HTTP/1.1 three-digit parser so unrelated port reuse, forwarding hangs, protocol-case lookalikes or numeric-prefix values cannot manufacture GREEN. Final #24 has unchanged-head CI/Supply Chain and fresh exact-head technical review; descendants must still revalidate. +## pg-erd production and failure traffic -On Linux, `tests/pg_erd_post_commit_reset_traffic.rs` covers the later abortive-reset phase after response commitment. The backend must receive the routed `/api/post-commit-reset` request, write HTTP 200 with exactly one `Content-Length: 20` field and the seven-byte `partial` prefix, then wait until the downstream has observed the complete committed header and that prefix before applying `SO_LINGER(0)`. Acceptance preserves the committed 200 and framing, requires termination before 20 response-body bytes complete, forbids a fabricated second status or silent failover, keeps `/readyz` 200, requires exactly one request-error Prometheus sample, and proves a later independent frontend route still succeeds. Listener reservations are concurrent; origin reads fail closed at five seconds/64 KiB; `Content-Length` lookup uses exact case-insensitive field identity; status parsing requires exact HTTP/1.1 plus one three-digit token; metric matching requires a complete sample line. Final #25 has unchanged-head hosted evidence; descendants must still revalidate rather than transfer its receipts. +`tests/pg_erd_production_path.rs` starts real loopback `backend` and `frontend` origins with `cwl-pingora-pg-erd-migration`. It requires gateway-local `/livez` and `/readyz`, routed consumer `/healthz`, raw-prefix `/api` behavior, fallback routing, hostile forwarded-header replacement from accepted transport/request authority, exact characterized response-policy replacement, and pre-origin declared-body rejection. -`tests/pg_erd_payload_free_observability.rs` covers the shared gateway observability boundary through the compiled migration process. A real routed request carries unique URI/query, Host, Authorization, Cookie and product-context sentinels; the backend must receive those values so the test cannot pass vacuously, while the `cwl_pingora_gateway::observability` target may emit only the bounded completion vocabulary and none of the sentinels. Traffic and metrics listener reservations are held concurrently until process start, origin header reads are bounded by five seconds and 64 KiB, HTTP field identity is matched case-insensitively by exact field name with OWS trimming, and the Prometheus oracle requires a complete exact sample line so a counter value such as `10` cannot satisfy an expected value of `1`. This proves only the shared gateway observability contract; it does not claim authority over product-owned logging, tracing, identity or third-party logger configuration. +`tests/pg_erd_runtime_isolation_traffic.rs` covers streamed request-body overflow plus process-local in-flight saturation/recovery and exact backpressure telemetry. `tests/pg_erd_upstream_failure_traffic.rs` creates deterministic `ECONNREFUSED`; `tests/pg_erd_read_stall_traffic.rs` keeps an accepted origin connection open without response bytes so per-read inactivity cannot be confused with origin closure. Both require bounded failure, exact error telemetry, preserved readiness, no silent route failover, and later independent-route recovery. -## Process-wide diagnostic logging +On Linux, `tests/pg_erd_upstream_reset_traffic.rs` covers abortive reset after request headers but before any response header. `tests/pg_erd_post_commit_reset_traffic.rs` covers abortive reset after HTTP 200 framing and a partial body have crossed the transport boundary. The latter uses bounded origin/downstream deadlines and Linux `SIOCOUTQ` to break timing cycles without pretending TCP acknowledgement proves Pingora application parsing. Acceptance preserves committed status/framing, terminates before the declared body completes, forbids a fabricated second status or silent failover, records exactly one request error, keeps readiness healthy, and proves independent-route recovery. -`tests/pingora_diagnostic_log_safety.rs` covers the dependency-log boundary that the shared `observability` target cannot prove. It starts the compiled generic runtime with `RUST_LOG=trace`, holds traffic and metrics reservations simultaneously until startup, and sends a request with unique URI/query, Host, Authorization and Cookie sentinels. The origin must receive the exact request target and semantically exact header values, with origin accept/header acquisition bounded to five seconds and 64 KiB, so a reject/strip path cannot manufacture GREEN. +`tests/pg_erd_partial_response_traffic.rs` covers orderly close after a committed HTTP 200 and partial body. Exact status, `Content-Length`, and Prometheus sample parsing reject protocol, field-name, duplicate-framing, and numeric-prefix lookalikes. `tests/pg_erd_graceful_shutdown.rs` holds a routed request in flight, sends SIGTERM only after bounded origin-header receipt, releases the response during the grace period, and requires downstream completion plus clean process exit inside one absolute external termination deadline. -Readiness probes occur before the characterized request and can themselves cause supplier diagnostics. The test therefore snapshots the static Pingora-redaction marker count only after both probes, requires the count to increase after the secret-bearing request reaches the origin and returns, waits for an exact bounded completion-log line suffix, and then requires none of the sentinels anywhere in captured process stderr. This proves that the characterized request traversed a redacted Pingora-family diagnostic path rather than merely observing an unrelated startup marker. Product/application loggers remain outside this process boundary. +`tests/pg_erd_payload_free_observability.rs` sends unique URI/query, Host, Authorization, Cookie, and product-context sentinels through the compiled migration process. The backend must receive them so the test is non-vacuous, while shared gateway stderr and low-cardinality metrics must not expose those sentinels. `tests/pingora_diagnostic_log_safety.rs` separately covers broad Pingora-family diagnostics under `RUST_LOG=trace`, requiring a new redaction marker attributable to the characterized request and forbidding request-derived secrets in process stderr. -## HTTP/1 protocol-transition admission +## Version-2 response-body lifetime -`tests/protocol_transition_policy.rs` freezes the transport-neutral v1 admission rule. An `Upgrade` field is sufficient to classify a protocol-transition attempt, and any `Connection` field may independently signal the same attempt through a comma-delimited case-insensitive `upgrade` token. Ordinary HTTP and unrelated tokens such as `x-upgrade` remain admitted. The policy is intentionally about transition evidence rather than WebSocket business semantics. +`tests/pg_erd_response_lifetime_config.rs` freezes the version transition. Pg-erd version 2 requires a positive explicit `max_upstream_response_body_ms`; version 2 rejects zero or omission, and version 1 rejects the field so timing semantics cannot change silently. -`tests/peer_protocol_policy.rs` locks the same non-support invariant at the immutable Pingora transport boundary. A peer returned by public `build_peer()` must use `HttpUpstreamRequestPolicy::deny_upgrades()` rather than the supplier `standard()`/`WebSocketOnly` default. This is a separate defense-in-depth oracle beneath callback admission; it must not be replaced by a mocked callback-only test or interpreted as WebSocket parity. +Unit coverage in `src/runtime_isolation.rs` and `src/migration_proxy.rs` proves the monotonic lifetime begins only at the first non-informational upstream response header, is not reset by later final-header callbacks, ignores empty/end-of-stream bookkeeping, and rejects non-empty body progress at or beyond the configured limit. A lifetime error is upstream-scoped. `fail_to_proxy` must preserve the existing local error response for pre-commit failures, but once `Session::response_written()` reports a final response it must return error code 0 and emit no second status. -Both compiled composition roots invoke the transition-rejection guard before request admission and any route/upstream selection. `tests/protocol_transition_traffic.rs` separately verifies the wire-visible boundary: WebSocket-shaped HTTP/1.1 requests must receive exact status 501, each dedicated origin listener must observe no connection during a fixed 500 ms window after that response, and gateway-local `/readyz` must remain HTTP 200 afterwards. Traffic and metrics listener ports remain simultaneously reserved until configuration is ready; response-header reads use one five-second whole-header deadline and a 64 KiB byte cap; status parsing accepts only case-sensitive `HTTP/1.1` plus one exact three-digit token and rejects numeric-prefix lookalikes such as `5010` or protocol-case variants. The 500 ms observation window is bounded traffic evidence, not proof that no arbitrarily delayed origin connection could ever occur. This is non-support acceptance only. Enabling WebSocket later requires a separate versioned contract plus realistic handshake/tunnel concurrency, disconnect/backpressure/drain behavior and supplier disposition; HTTP/2 Extended CONNECT is a separate protocol path. +`tests/pg_erd_slow_drip_response_traffic.rs` is the realistic wire oracle. The backend commits HTTP 200 with `Content-Length: 20` and emits body progress frequently enough that each read remains within `read_ms`, while a shorter `max_upstream_response_body_ms` expires. The downstream must retain the committed status/framing, terminate before all declared bytes arrive, receive no fabricated second status or silent failover, observe exact request-error telemetry, keep `/readyz` available, and allow a later independent frontend request. This proves continuous body slow-drip is bounded. It does not claim an exact timer interrupt for an already-pending read or an absolute deadline for incomplete response headers; those remain separate gaps. ## Routed performance acceptance -PR `#22` adds the first dedicated routed pg-erd concurrency/latency gate on the current #21 ancestry. `tests/load/load_origin.rs` is the only measured origin implementation: bounded std-only Rust, finite worker/queue capacity, 64 KiB header bound, deterministic Content-Length framing and direct parser/framing tests. The CI load lane compiles both admitted gateway binaries, runs `rustfmt` and `rustc -D warnings --test` on the origin, then builds the optimized fixture used by measured traffic. - -`tests/load/pg_erd_gateway_smoke.js` runs four VUs for 400 total iterations. It alternates `/api/load-contract` and `/load-contract`, tags each request `backend` or `frontend`, requires exact 200/body identity and zero HTTP failures, gates aggregate plus each route independently at p95 `<20 ms`, and requires at least 198 measured requests per route. `tests/pg_erd_routed_latency_contract.rs` freezes the route/path/body/tag binding and per-route thresholds/sample floors. `tests/rust_load_origin_workflow_contract.rs` prevents regression to interpreted measured-origin execution by inspecting only the `load-contract` job and requiring the bounded Rust build/test/start commands with no Python invocation. - -The routed CI fixture starts separate bounded Rust origins for the characterized `backend` and `frontend` authorities and the compiled pg-erd process with only admitted listener, metrics, runtime budget, keepalive and transport data. The workflow must not add configurable routes because `PgErdMigrationConfig` denies unknown fields and the route table is compiled into the migration plan. `k6-pg-erd-summary.json` is uploaded as exact-SHA evidence only after that unchanged child head executes it. - -The `<20 ms` threshold is a controlled loopback regression objective, not a production claim. Representative TLS, multi-hop, container/Kubernetes scheduling, origin-capacity and failure traffic must be measured separately before buyer-visible production p95 credit. Sample reduction, route omission, threshold removal or unrealistic warm-up is not an admissible performance repair. +`tests/load/pg_erd_gateway_smoke.js` and `tests/pg_erd_routed_latency_contract.rs` exercise characterized backend/frontend routes through bounded Rust origins. The measured lane requires exact status/body identity, zero HTTP failures, aggregate/backend/frontend p95 below 20 ms, and the configured minimum samples per route. `tests/rust_load_origin_workflow_contract.rs` prevents regression to interpreted measured-origin execution. Controlled loopback evidence is not representative TLS, multi-hop, container/Kubernetes scheduling, origin-capacity, or production-SLO evidence; sample reduction, route omission, threshold removal, or unrealistic warm-up is not an admissible repair. -## OCI, supply chain and release gates +## OCI, supply chain, coverage, and release evidence -The `oci-runtime` job builds the generic and pg-erd image profiles separately. Both must declare uid/gid `65532`, run read-only with all capabilities dropped and `no-new-privileges`, and consume only read-only versioned configuration. The generic profile must expose `/livez`; the pg-erd profile must expose its traffic `/livez` and a separately published Prometheus `/metrics` listener whose media type is `text/plain` after stripping only optional parameters. +The `oci-runtime` job builds the generic and pg-erd images separately. Both must declare uid/gid `65532`, run with a read-only root filesystem, all capabilities dropped, `no-new-privileges`, and a read-only versioned configuration mount. The generic image must expose `/livez`; pg-erd must expose its traffic `/livez` and a separately published Prometheus `/metrics` listener whose base media type is `text/plain`. -Supply-chain evidence must remain exact-source-bound and include dependency audit, both candidate image builds, SPDX SBOM evidence and image vulnerability scans. Release credit additionally requires immutable registry/package identity, signing/attestation/provenance, reproducibility evidence and rollback rehearsal; local image IDs or Draft PR checks are insufficient. +Supply-chain evidence must remain exact-source-bound and include dependency audit, both candidate image builds, SPDX SBOM evidence, and image vulnerability scans. Owned production code requires 100% line and region coverage without exclusions, missing-public-rustdoc enforcement, warning-denied documentation, formatting, all-target compile/test, and strict Clippy. Release credit additionally requires immutable registry/package identity, signing/attestation/provenance, reproducibility evidence, and rollback rehearsal; local image IDs or Draft PR checks are insufficient. ## Remaining gaps -Open acceptance still includes versioned WebSocket enablement and upgraded-connection failure behavior, incomplete-response-header slow-drip/absolute header-read handling, broader admitted long-lived streaming semantics where consumer evidence requires them, downstream TLS/H2, H2→H1 Cookie handling, Extended CONNECT, explicit H3/QUIC disposition, tracing, property/fuzz testing, representative routed TLS/origin-capacity load, shadow/canary and rollback. Pg-erd upstream TLS trust/hostname success/failure has terminal source-gate evidence on final #37; the new pg-erd v2 response-body progress lifetime must independently close on its unchanged current head before credit. Nginx/OpenResty/legacy removal is permitted only after parity, immutable release, canary/cutover and rollback evidence are all current on protected ancestry. Process-wide diagnostic redaction must be reacquired on every changed descendant; source presence or a predecessor trace run is never release evidence. +Open acceptance includes downstream TLS/H2, H2-to-H1 Cookie normalization, versioned WebSocket/Extended CONNECT, explicit H3/QUIC disposition, incomplete-response-header slow-drip/absolute header-read handling, broader admitted long-lived-stream semantics where consumer evidence requires them, dynamic reload, tracing, property/fuzz testing, representative routed TLS/origin-capacity load, shadow/canary, rollback, cutover, and verified legacy proxy removal. Every changed descendant must reacquire applicable exact-head evidence; neither source presence nor predecessor GREEN is release evidence. From 7b81e1c336f84545117366c3ce6c55ead76c2e52 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 11:03:41 +0900 Subject: [PATCH 46/49] docs: reconcile response lifetime TRD with current parent contracts --- TRD.md | 79 +++++++++++++++++++++++++++++++++++++--------------------- 1 file changed, 50 insertions(+), 29 deletions(-) diff --git a/TRD.md b/TRD.md index 74c9da44..ec70dcb4 100644 --- a/TRD.md +++ b/TRD.md @@ -1,57 +1,78 @@ -# Technical Requirements +# Technical Requirements Document -## Runtime +## Runtime and composition roots -Rust edition 2021 with manifest MSRV `1.98.0` on this branch. Cloudflare Pingora and `pingora-prometheus` are pinned to exact public upstream revision `09696b51bc59315353d96686355861604d0bb48c`; mutable branch or contributor-PR dependencies are not release authority. +This branch uses Rust edition 2021 with manifest MSRV `1.98.0`. Cloudflare Pingora crates remain pinned to the exact upstream revision admitted by the parent stack; mutable branches, tags, contributor PRs, or local forks are not release authority. -The generic production composition root is `src/bin/cwl-pingora-gateway.rs`. It parses one explicit `--config`, validates the transport-neutral contract before granting network authority, constructs `GatewayProxy`, exposes the dedicated metrics listener, adds the downstream TCP listener to Pingora `http_proxy_service`, and delegates serving and shutdown to Pingora's `Server` lifecycle. +There are two composition roots with intentionally different public contracts: -The characterized `pg-erd-cloud` migration uses the separate `src/bin/cwl-pingora-pg-erd-migration.rs` composition root and `PgErdMigrationConfig`. Keeping a separate binary prevents the generic v1 contract from being widened into a product-routing configuration language. Product authentication/authorization, business routing, certificate issuance/ACME, Wardnet/EgressWeave policy, and Keyverse identity remain outside this process boundary. +- `src/bin/cwl-pingora-gateway.rs` activates generic version-1 `GatewayConfig` and the single-upstream `GatewayProxy`. +- `src/bin/cwl-pingora-pg-erd-migration.rs` activates only bounded `PgErdMigrationConfig` and `MigrationGatewayProxy` for the characterized pg-erd edge surface. -## Contract +Both parse and validate an explicit configuration before creating listeners and delegate serving/shutdown to Pingora. The pg-erd binary does not widen generic v1 into a product routing language. Product authentication/authorization, business logic, certificate issuance/ACME, Keyverse identity, Wardnet/EgressWeave policy, and consumer service discovery remain outside this runtime. -Generic configuration version 1 is strict YAML with `deny_unknown_fields`. Required top-level fields are `version`, `listener`, `metrics_listener`, `max_request_body_bytes`, `max_in_flight_requests`, `upstream_keepalive_pool_size`, and `upstreams`. Version 1 accepts exactly one upstream; it does not provide a generic product route table or request-controlled destination. +## Network and Admin Config authority -Traffic and metrics listeners must use non-zero, non-overlapping effective socket authority. Validation rejects same-family wildcard/concrete overlap, IPv6-wildcard/IPv4 dual-stack ambiguity, native IPv4 versus IPv4-mapped IPv6 aliases, and native/mapped or mapped-to-mapped IPv4 wildcard aliases while preserving distinct concrete non-aliased addresses. Request-body, in-flight, and keepalive-pool budgets must be positive. +Generic v1 is strict YAML with unknown fields denied. It admits one explicit non-zero traffic listener, one distinct non-zero metrics listener, exactly one non-zero upstream authority, positive request-body/in-flight/keepalive budgets, and positive upstream I/O budgets. Listener and metrics validation rejects equal sockets, same-family wildcard/concrete aliases, native/IPv4-mapped IPv4 aliases, native or mapped wildcard aliases, and platform-dependent same-port IPv6-wildcard/IPv4 ambiguity while preserving distinct concrete non-aliased authority. -Each upstream has a stable non-empty name, a non-zero concrete socket address, `tls`, optional `sni`, optional absolute `trust_bundle_file`, and explicit positive `connection_ms`, `total_connection_ms`, `read_ms`, `write_ms`, and `idle_ms` budgets. TLS upstreams require non-empty SNI. Pingora `HttpPeer` enables certificate and hostname verification. When `trust_bundle_file` is configured, the loaded PEM certificates become that peer's CA store rather than being silently merged with platform roots. Clear-text upstreams may define neither SNI nor a trust bundle. The gateway does not issue, renew, or rotate certificates. Pingora `read_ms` remains a per-read inactivity timer, not an overall response deadline; generic v1 therefore still has no response-body lifetime contract. +TLS upstreams require an admitted RFC 6066 DNS hostname for SNI/hostname verification and may optionally consume one absolute PEM trust-bundle path before listeners open. Clear-text upstreams may define neither SNI nor a trust bundle. The gateway does not issue, renew, rotate, or persist trust material. -`PgErdMigrationConfig` is a bounded Admin Config contract, not a second generic router. Version 1 preserves the existing unreleased characterization profile. Version 2 is a narrow Runtime Isolation increment: it accepts the same operator-supplied listener/metrics sockets, non-zero request/in-flight/keepalive budgets, and concrete transport/TLS data for only the characterized `backend` and `frontend` identities, and additionally requires a positive explicit `max_upstream_response_body_ms`. Version 1 rejects that field, so timing semantics cannot change without a configuration-version change. Route selection and response-security policy remain compiled migration contracts. Missing, extra, renamed, zero-port, or overlapping transport authority fails before listener activation. +`PgErdMigrationConfig` is a bounded Admin Config contract, not a generic route DSL. Operators may provide only the traffic/metrics sockets, positive runtime and keepalive budgets, and concrete transport/TLS values for the compiled `backend` and `frontend` identities. Route precedence, response-security policy, product auth, business routing, and arbitrary destinations are not operator-configurable. Missing, duplicate, extra, renamed, zero-port, recursive, multicast/broadcast, or otherwise invalid transport authority fails before listener activation. -`PgErdMigrationConfig` also derives public Serde `Deserialize`; callers therefore are not forced through `PgErdMigrationConfig::from_yaml`. The public `build_proxy()` activation boundary revalidates the complete deterministic Admin Config contract before delivery peers or runtime limits are materialized. Only after that revalidation may the infallible `RuntimeIsolationLimits::from_validated*` constructors reuse the proven-positive budgets, so direct deserialization cannot bypass version, listener-authority, runtime, keepalive, or transport-authority invariants. +The type is publicly deserializable, so `build_proxy()` revalidates the complete deterministic contract before creating delivery peers or infallible runtime limits. Direct deserialization cannot bypass version, listener, runtime, keepalive, or transport-authority invariants. Custom trust bytes are materialized exactly once during peer construction after deterministic validation and before listener registration. -## Request policy +## Version-2 response-body lifetime -Every immutable Pingora upstream peer uses `HttpUpstreamRequestPolicy::deny_upgrades()`. This retains the pinned supplier's standard hop-by-hop and `Connection`-nomination sanitation but changes its HTTP/1 upgrade policy from the default `WebSocketOnly` behavior to `Deny`. The separate transport-neutral admission guard remains authoritative for returning HTTP 501 before application admission or origin selection. Keeping both boundaries aligned prevents callback/composition changes from implicitly enabling a supplier protocol capability that the versioned gateway contract does not admit. +Pg-erd version 1 preserves the existing unreleased characterization semantics and rejects `max_upstream_response_body_ms`. Version 2 requires an explicit positive `max_upstream_response_body_ms`; omission or zero is invalid. This version boundary prevents a timing policy from appearing through a hidden default. -The generic gateway additionally removes client-provided `Forwarded`, `X-Forwarded-For`, `X-Forwarded-Host`, `X-Forwarded-Port`, `X-Forwarded-Proto`, `X-Forwarded-Server`, and `X-Real-IP`, then emits only gateway-owned `Forwarded: proto=http` for the v1 clear-text downstream listener. Generic v1 deliberately makes no client-IP identity or downstream proxy-provenance claim. +`RuntimeIsolationLimits` carries the optional response-body lifetime only for the version-2 profile. `MigrationRequestContext` creates a `ResponseBodyLifetimeBudget` from those limits. `upstream_response_filter` starts the monotonic budget at the first non-informational upstream response header. Informational headers do not start it, and later final-header callbacks do not reset it. `upstream_response_body_filter` checks elapsed time only when a non-empty body chunk is observed; empty/end-of-stream bookkeeping cannot manufacture expiry. -The pg-erd migration adapter also discards request-controlled forwarding identity before rebuilding only the characterized compatibility fields from accepted transport/request authority. The current captured Traefik entry point is clear-text, so its forwarded scheme is explicitly `http`; HTTPS requires a separate TLS-derived contract rather than inference. +A body-progress callback at or beyond the configured lifetime becomes an upstream-scoped fatal error. The lifetime is deliberately independent of Pingora peer `read_ms`: `read_ms` remains a per-read inactivity timer that resets after successful reads, while the version-2 lifetime bounds continuously progressing bodies at their first non-empty callback after the absolute lifetime is reached. With the pinned callback surface this is not an exact interrupt of an already-pending read, and slow delivery of an incomplete response header remains a separate gap. -Non-health requests acquire the process `max_in_flight_requests` budget before upstream selection and fail closed with HTTP 503 at capacity. Requests with a parseable `Content-Length` above `max_request_body_bytes` fail with HTTP 413 before upstream selection; streamed body bytes are counted and fail with 413 if the same bound is exceeded. Pingora's parser retains its own finite protocol limits, but an operator-controlled smaller HTTP/1 header byte/count budget remains a separate edge-policy gap. +Failure handling is response-phase-aware. Before a final downstream response has been written, existing `fail_to_proxy` behavior may still emit the policy-complete local error response for an upstream failure. After `Session::response_written()` reports a final response, the runtime returns error code 0 and writes no second status. A post-commit lifetime breach therefore terminates the incomplete response instead of rewriting it to 502 or routing to another pg-erd origin. The ordinary request context then releases its in-flight admission lease. -Pg-erd config version 2 starts its monotonic response-body lifetime in `upstream_response_filter` on the first non-informational upstream response header. `upstream_response_body_filter` checks the elapsed lifetime only when a non-empty body chunk is observed; empty/end-of-stream bookkeeping callbacks cannot create a false timeout. A non-empty progress callback at or beyond `max_upstream_response_body_ms` becomes an upstream-scoped fatal error. This complements, rather than replaces, the per-read `read_ms` inactivity timer. With the pinned Pingora callback API the body-lifetime guard is progress-driven: it stops continuous slow-drip at the first real body callback after the deadline, while a quiescent pending read can run until `read_ms`. It therefore must not be described as an exact timer interrupt. Slow-drip of an incomplete response header remains a separate transport gap. +## Request, forwarding, and protocol policy -Generic v1 makes one prevalidated upstream peer available per request. The pg-erd migration adapter selects only peers already bound by `MigrationDeliveryPlan`; neither path performs request-controlled service discovery. Domain retries, failover, and idempotency policy are not invented by this runtime. +Every immutable Pingora peer uses `HttpUpstreamRequestPolicy::deny_upgrades()`. A separate transport-neutral request guard returns HTTP 501 for an HTTP/1 request carrying either `Upgrade` or a case-insensitive `upgrade` token in `Connection`, before request admission or upstream selection. This is deliberate protocol non-support, not WebSocket parity. -Failure handling is phase-aware. Before an upstream response header is committed downstream, transport failure may still be represented by the gateway's fail-closed error response under the one-attempt policy. After a valid response header has been committed, a later upstream framing/body failure—including a version-2 response-body lifetime breach—cannot be rewritten into a second HTTP status or silently failed over: the incomplete downstream response terminates, low-cardinality error telemetry records the failed request, process readiness remains available, the request context releases its in-flight lease, and independent routes must remain usable. This is an edge transport invariant, not product retry authority. +Generic v1 strips request-controlled `Forwarded`, `X-Forwarded-*`, `X-Real-IP`, and related proxy-identity fields before emitting only gateway-owned `Forwarded: proto=http`. It makes no client-identity or downstream proxy-provenance assertion. -## Health and observability +The pg-erd migration adapter also removes request-controlled forwarding identity. It rebuilds only the characterized compatibility fields from accepted client transport plus validated request authority: `X-Forwarded-For`, `X-Real-IP`, `X-Forwarded-Host`, `X-Forwarded-Port`, and `X-Forwarded-Proto`. The currently characterized consumer entry point is clear-text `web`, so the admitted downstream scheme is `http`; HTTPS forwarding semantics require a separate downstream-TLS contract. -`GET /livez` and `/readyz` return HTTP 200 with an empty, non-cacheable response through the process-local Pingora health boundary. Readiness proves validated configuration plus an active serving path, not product dependency health. In the pg-erd migration profile, consumer `/healthz` remains ordinary routed application traffic and is not confused with process liveness/readiness. +Non-health traffic acquires the process `max_in_flight_requests` lease before upstream selection. Saturation fails fast with HTTP 503 and increments bounded telemetry. A declared `Content-Length` above `max_request_body_bytes` fails with HTTP 413 before origin selection, and streamed body bytes are counted against the same limit. `/livez` and `/readyz` bypass application admission so process health remains observable under saturation. -The shared process exposes bounded Prometheus counters for request completion, request errors, observed request-body bytes, and backpressure rejection. The canonical gateway log vocabulary records only low-cardinality transport completion facts; authorization headers, cookies, credentials, customer payloads, and unbounded product route labels are outside this shared observability contract. +## Routing, response policy, and delivery -## Packaging +`EdgeMigrationPlan` owns the characterized transport-neutral pg-erd route/policy composition. The route contract admits exact `/healthz -> backend`, raw `/api` prefix behavior including `/apiary -> backend`, and fallback `/ -> frontend` according to the captured consumer edge semantics. The response-policy contract owns only the characterized gateway response fields; it does not take product-domain response ownership. -The Docker builder is digest-pinned `rust:1.98.0-bookworm`; the final image is digest-pinned `gcr.io/distroless/base-nossl-debian13:nonroot`. The pinned Pingora OpenSSL path is vendored, so the final image does not carry Debian `libssl`. The Dockerfile exposes only one build-time selector, `CWL_GATEWAY_BIN`, and fail-closes unless its value is exactly `cwl-pingora-gateway` or `cwl-pingora-pg-erd-migration`. The selected executable is normalized to one fixed runtime path before the distroless stage, so a final image contains one admitted process identity rather than both binaries or a runtime shell selector. +`MigrationDeliveryPlan` binds each admitted upstream identity to exactly one prevalidated Pingora `HttpPeer`. Missing, duplicate, undeclared, recursive, or invalid transport bindings fail closed. Request data cannot select an arbitrary destination or create service-discovery authority. -Both image profiles run as uid/gid `65532`, have no intentional application writes, and are required to remain compatible with a read-only root filesystem, all capabilities dropped, and `no-new-privileges`. The OCI gate builds the default generic image and an explicit pg-erd image, then independently starts each exact candidate under those restrictions with a read-only configuration mount and requires `/livez` to become reachable. `examples/pg-erd-migration.yaml` is bounded smoke configuration for this process-level OCI proof; it does not substitute for routed origin/load/failure acceptance. Until the exact current head reaches terminal success, the new workflow is source-defined acceptance rather than hosted GREEN evidence. +The migration proxy applies response-security fields through replacement semantics. Upstream transport failures before response commitment use the bounded local error mapping; post-commit truncation, reset, or lifetime failure preserves the committed response rather than inventing a second status or silent failover. -The candidate supply-chain lane builds and vulnerability-scans both admitted images and binds both local image IDs plus per-image scan outputs to the exact source SHA. Its dependency SBOM describes the shared committed Rust dependency graph. A protected release still requires registry-bound immutable image digests, release-bound SBOM/provenance/reproducibility evidence, and rollback rehearsal; no Draft PR head or local image ID is a deployable release identity. +## Health, observability, and graceful lifecycle -## Protocol and migration limits +`GET /livez` and `/readyz` are process-local and return non-cacheable HTTP 200 through Pingora. They indicate validated process/configuration readiness, not consumer dependency health. Pg-erd consumer `/healthz` remains routed application traffic to `backend`. -Generic v1 is a clear-text downstream HTTP proxy with one explicit upstream per process. HTTP/1 Upgrade is explicitly denied both before request admission and at immutable peer construction; that is non-support evidence, not WebSocket parity. Downstream TLS termination, HTTP/2 admission, H2→H1 Cookie normalization, HTTP/3/QUIC, versioned WebSocket/Extended CONNECT, dynamic reload, Kubernetes Gateway API, and consumer-specific multi-route behavior are separate increments with realistic RED→GREEN evidence. +Shared observability is low-cardinality and payload-free. The gateway records bounded request completion/error/body-byte/backpressure facts but excludes request paths, query strings, credentials, cookies, customer payloads, product identifiers, and unbounded labels. Pingora-family dependency diagnostics pass through the process-wide payload-safe logger so broad `RUST_LOG` settings cannot bypass this boundary. -The concrete pg-erd migration stack is a bounded consumer-characterization adapter and does not widen generic v1. The response-body progress lifetime does not close incomplete-response-header slow delivery, exact pending-read interruption, or arbitrary long-lived-stream semantics. Source presence is not parity. Promotion still requires unchanged exact-head formatting, compile/test, strict Clippy, rustdoc, owned-production coverage, routed traffic/load/failure evidence, terminal dedicated OCI/supply-chain execution, immutable release identity, consumer deployment pin, shadow/canary, rollback rehearsal, protected cutover, and verified legacy removal. +The shared server policy makes one total upstream attempt, configures the admitted keepalive pool, applies an explicit five-second drain grace period, and uses the bounded runtime shutdown policy documented in `OPERABILITY.md`. Consumer retry/failover requires idempotency and product knowledge and is not inferred by the gateway. + +## Packaging and supply chain + +The Docker build admits only `cwl-pingora-gateway` or `cwl-pingora-pg-erd-migration` as build-time process identities and normalizes the selected executable into one distroless non-root runtime image. Exact-head OCI acceptance requires uid/gid `65532`, read-only root, dropped capabilities, `no-new-privileges`, and read-only configuration mounts. The pg-erd image must expose the traffic health endpoint and the separately published Prometheus metrics listener. + +Supply-chain evidence remains exact-source-bound: committed lockfile, dependency/advisory policy, SBOM, both candidate image builds, and image vulnerability scans. These are candidate receipts, not immutable release identity. Protected promotion additionally requires immutable package/image digests, release-bound SBOM/provenance/attestation, reproducibility, rollback rehearsal, and current protected ancestry. + +## Test and performance requirements + +Every changed exact head must pass formatting, all-target compile/test, strict Clippy, warning-denied rustdoc, 100% owned-production line and region coverage without exclusions, realistic production-path traffic, OCI, supply-chain, and current-head review. Parent or historical GREEN never transfers after source, parent, or evidence changes. + +The version-2 lifetime path requires both focused unit/config coverage and real slow-drip traffic. The real fixture must keep each upstream read inside `read_ms` while total body progress crosses `max_upstream_response_body_ms`, preserve committed 200/framing, terminate before the declared body completes, emit no second status or route failover, record exact error telemetry, retain `/readyz`, and prove an independent route can recover. + +Applicable routed buyer paths use bounded Rust origins and k6/E2E with p95 below 20 ms on controlled loopback. Such loopback evidence is a regression gate only. Representative TLS, multi-hop, container/Kubernetes scheduling, origin-capacity, failure contention, and deployment traffic are required before production p95 credit; sample reduction, route omission, threshold weakening, or unrealistic warm-up are not repairs. + +## Known limits and migration boundary + +Generic v1 remains a clear-text downstream HTTP proxy with one explicit upstream per process. The pg-erd binary remains a bounded consumer-characterization adapter. Downstream TLS/H2, H2-to-H1 Cookie normalization, versioned WebSocket/Extended CONNECT, H3/QUIC, incomplete-response-header slow-drip, broader long-lived-stream semantics, dynamic reload, tracing, property/fuzz testing, and consumer-specific cutover behavior remain separate increments with their own RED-to-GREEN evidence. + +Source capability is not parity or deployment state. Promotion still requires exact protected lineage, normal integration, immutable release identity, representative traffic/security/performance evidence, consumer shadow/canary, observed rollback, cutover, and verified legacy reverse-proxy removal. From 14a650f23662aa80cb58777e4bbc665471279ba1 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 11:04:13 +0900 Subject: [PATCH 47/49] docs: reconcile changelog with current response lifetime ancestry --- CHANGELOG.md | 73 ++++++++++++++++++++-------------------------------- 1 file changed, 28 insertions(+), 45 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index aaa12f17..ccd1b363 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,49 +4,32 @@ All notable changes are tracked here. No release has been published yet. ## Unreleased -- Bootstrapped an executable Rust Pingora proxy through pull-request governance. -- Added strict v1 configuration, explicit one-upstream network authority, TLS identity verification, and explicit upstream I/O budgets. -- Added a transport-neutral Edge Routing characterization for the live `pg-erd-cloud` Traefik contract: exact `/healthz`, raw-prefix `/api`, fallback `/`, explicit numeric precedence, and fail-closed ambiguous-priority/malformed-route rejection. This does not yet claim traffic cutover. -- Added a separate transport-neutral HTTP Policy characterization for the live `pg-erd-cloud` Traefik response-security middleware: exact `X-Content-Type-Options`, `X-Frame-Options`, `Referrer-Policy`, and `Permissions-Policy` values, ASCII case-insensitive field identity, duplicate authority rejection, and fail-closed CR/LF values. -- Added a transport-neutral `EdgeMigrationPlan` that composes the characterized route and HTTP-policy contracts with an explicit normalized upstream-authority set. The pg-erd-cloud plan admits only `backend` and `frontend`, rejects undeclared route targets, and remains migration evidence rather than shadow/canary, cutover, or legacy-removal proof. -- Added `MigrationDeliveryPlan` to bind every characterized migration upstream identity to exactly one explicit, prevalidated Pingora `HttpPeer`; missing, duplicate, and undeclared concrete transport authority fails closed. This does not widen `GatewayConfig` v1. -- Added `MigrationGatewayProxy` as a Pingora callback adapter over the characterized route/header/peer contracts. It rejects unmatched routes, enforces shared body/concurrency isolation, and applies only the characterized response policy. -- Added bounded `PgErdMigrationConfig` and a separate `cwl-pingora-pg-erd-migration` composition root. The operator may configure listener/metrics sockets, non-zero runtime/keepalive budgets, and concrete `backend`/`frontend` transport/TLS values, but all listener, metrics and upstream transport ports must be non-zero and routes, response-policy fields, product auth/business logic, service discovery, or new migration authorities remain non-configurable. The generic `GatewayConfig` v1 public shape and one-upstream semantics remain unchanged. -- Tightened the shared edge network-authority invariant used by generic and pg-erd configuration. Generic v1 rejects port-zero traffic, metrics and upstream bindings instead of allowing ephemeral or unusable authority; both profiles reject same-port wildcard traffic/metrics aliases and IPv4-mapped/native IPv4 aliases, including native or mapped IPv4 wildcard combinations. IPv6 wildcard plus IPv4 same-port authority is rejected because dual-stack bind behavior is platform-dependent; distinct concrete non-aliased addresses on the same non-zero port remain configurable. -- Extended the shared network-authority invariant so a configured upstream cannot alias either gateway-owned traffic or metrics listener authority. Generic and bounded pg-erd startup fail closed on exact, wildcard, dual-stack, or mapped/native self-loop authority while still allowing distinct concrete IP authorities on the same port. This prevents recursive self-proxying and accidental metrics exposure without becoming product routing or service-discovery policy. -- Added a fail-closed HTTP/1 protocol-transition admission boundary for generic v1 and the bounded pg-erd composition root. Any `Upgrade` field or case-insensitive comma-delimited `upgrade` token in `Connection` returns HTTP 501; callback ordering places that rejection before request admission and route/upstream selection. Compiled real-listener acceptance then observes no origin connection during a fixed 500 ms window after the verified 501 response and requires `/readyz` to stay available. `x-upgrade` and other unrelated tokens do not match. This is a deliberate non-support contract, not WebSocket implementation; HTTP/2 Extended CONNECT and H3/QUIC remain separate versioned work. -- Aligned immutable Pingora peer construction with the same fail-closed HTTP/1 transition contract. `pingora_delivery` now uses `HttpUpstreamRequestPolicy::deny_upgrades()` instead of the supplier `standard()`/`WebSocketOnly` default while retaining standard hop-by-hop and `Connection`-nomination sanitization. A dedicated peer-level regression prevents a later callback/composition refactor from silently re-enabling WebSocket forwarding below request admission. -- Tightened the generic v1 forwarding trust boundary: request-controlled `Forwarded`, `X-Forwarded-For`, `X-Forwarded-Host`, `X-Forwarded-Port`, `X-Forwarded-Proto`, `X-Forwarded-Server`, and `X-Real-IP` are all stripped before emitting only gateway-owned `Forwarded: proto=http`. -- Kept migration admin parsing side-effect free for custom TLS trust material: exact transport-authority and upstream-contract validation happens during parse, while peer/trust-bundle materialization occurs once during `build_proxy` before listener creation. This removes an avoidable validate-then-reload trust-file window. -- Added a separate Ingress forwarding-policy boundary for the pg-erd migration. Request-controlled `Forwarded`, `X-Forwarded-*` and `X-Real-IP` values are removed, then the compatibility `X-Forwarded-For`, `X-Real-IP`, `X-Forwarded-Host`, `X-Forwarded-Port` and `X-Forwarded-Proto` fields are rebuilt from accepted transport metadata. The current characterized Traefik `web` entryPoint remains explicitly HTTP; TLS-derived scheme behavior is not claimed before a TLS listener contract exists. -- Added a shared `observability` bounded context so both `GatewayProxy` and `MigrationGatewayProxy` use the same low-cardinality request/error/body/backpressure counters and coarse access-log shape instead of duplicating telemetry. The public observation vocabulary contains only response status, `ok`/`error`, and observed request-body bytes; paths, query strings, headers/cookies, credentials, customer payloads and product identifiers stay out of the shared telemetry contract. -- Added dedicated compiled pg-erd payload-free access-log acceptance. A routed request carries unique URI/query, Host, Authorization, Cookie and product-context sentinels; the backend must receive those exact fields so the fixture is non-vacuous, while shared gateway stderr must emit only the bounded completion vocabulary and none of the sentinels. Test oracles use exact case-insensitive HTTP field matching and exact Prometheus sample-line matching so `X-Forwarded-Host` cannot satisfy `Host` and counter value `10` cannot satisfy the expected value `1`. -- Added a process-wide payload-safe dependency logging policy to both production binaries. Operator `RUST_LOG` still selects levels and targets, but Pingora-family dependency message bodies are replaced with a static marker before formatting so supplier trace/debug/error records cannot expose request URI, Host, Authorization, Cookie, payload, or other request-derived material. The compiled generic regression snapshots redaction activity after readiness probes, requires the secret-bearing request to reach the origin, requires a new redacted Pingora diagnostic after that request, and requires none of the sentinels in process stderr. Product/consumer logging remains outside this gateway-owned boundary. -- Added mandatory positive `max_in_flight_requests` and `upstream_keepalive_pool_size` capacity budgets; Pingora's framework keepalive default is overridden from the validated edge contract. -- Added process-local fail-fast backpressure: non-health requests above the in-flight budget receive HTTP 503, health remains observable, rejection telemetry increments, and capacity is released after request completion or failure. -- Added dedicated compiled pg-erd traffic acceptance for streamed/chunked body overflow and routed in-flight saturation/recovery: the migration process must return 413 above the shared body budget, return 503 in less than one second above the in-flight budget, keep `/readyz` observable, expose the exact single-rejection Prometheus sample, and admit a later routed request after capacity is released. -- Added dedicated compiled pg-erd refused-origin recovery acceptance using a Linux TCP socket bound to the characterized backend address without entering LISTEN state. The fixture first proves direct `ECONNREFUSED` while retaining exclusive port ownership, then requires the migration gateway to return 502 within a conservative one-second envelope around the configured 200/400 ms connection budgets, keep `/readyz` 200, expose the exact single-error Prometheus sample, and allow a later independent frontend route to recover. Connected read stall, TCP reset, partial-response/streaming failure, retry and failover behavior remain separate gaps. -- Added dedicated compiled pg-erd connected read-stall acceptance with `read_ms=100`: the backend accepts the routed request and remains open without response bytes until the gateway has already failed it, preventing fixture closure from faking timeout behavior. The contract requires 502 inside a conservative one-second envelope, preserved `/readyz`, the exact single-error Prometheus sample, and independent frontend recovery. Pingora `read_timeout` remains a per-read inactivity budget, not a whole-response lifetime. -- Added opt-in pg-erd Admin Config version 2 with mandatory positive `max_upstream_response_body_ms`, leaving pg-erd v1 and generic v1 semantics unchanged. Runtime Isolation starts the monotonic budget at the first non-informational upstream response header and checks it only on non-empty body progress, so empty/end-of-stream bookkeeping cannot create a false timeout. A continuously progressing body that reaches the budget becomes an upstream-scoped fatal error without retry/failover after response commitment. Real-listener acceptance commits HTTP 200 and drips one byte every 60 ms while each read stays inside `read_ms`, then requires termination before the declared body completes, exact low-cardinality error telemetry, healthy `/readyz`, and independent-route recovery. This remains a progress-driven bound rather than an exact interrupt of a pending Pingora read; incomplete-response-header slow delivery and broader long-lived streaming remain separate gaps. -- Added dedicated compiled pg-erd post-header partial-response acceptance: the backend commits HTTP 200 with exactly one `Content-Length: 20` field, writes only `partial`, and closes only after the downstream has observed that committed header/body prefix. The downstream must retain the committed status and the single exact framing field, then terminate before body completion instead of receiving an invented second status or silent failover; `/readyz`, exact low-cardinality request-error telemetry, and an independent `frontend` route must remain usable. `X-Content-Length` and duplicate/conflicting `Content-Length` fields cannot satisfy the framing oracle. Explicit TCP reset and broader streaming/upgraded failure remain separate gaps. -- Added dedicated compiled pg-erd pre-header TCP-reset acceptance on Linux. The characterized backend must first receive `/api/reset`, then apply abortive `SO_LINGER(0)` and close before emitting any response header. Acceptance requires HTTP 502 within two seconds despite a configured five-second read inactivity budget, no silent frontend failover, preserved `/readyz`, exactly one low-cardinality request-error sample, and a later independent frontend HTTP 200. The fixture bounds origin header reads to five seconds/64 KiB, uses exact HTTP/1.1 status parsing, and matches complete Prometheus sample lines so forwarding defects, protocol/status lookalikes or numeric-prefix counters cannot manufacture GREEN. -- Added dedicated compiled pg-erd post-commit TCP-reset acceptance on Linux. The characterized backend writes HTTP 200 with exactly one `Content-Length: 20` field and the seven-byte `partial` prefix, waits until the downstream has actually observed that committed prefix, then applies abortive `SO_LINGER(0)`. Acceptance preserves the committed status/framing, requires termination before the declared body completes, forbids a fabricated second status or silent failover, keeps `/readyz` 200, exposes exactly one low-cardinality request-error sample, and allows independent frontend recovery. The fixture reserves listener ports concurrently, bounds origin reads to five seconds/64 KiB, rejects `X-Content-Length` as framing authority, uses exact HTTP/1.1 status parsing, and matches whole Prometheus sample lines. -- Added dedicated routed pg-erd graceful-drain acceptance: a characterized `/api/held` backend request is held in flight, SIGTERM is sent only after the backend has accepted it, the response is released during the shared grace period, the downstream must still receive HTTP 200, and the migration process must exit successfully inside the external termination budget measured from SIGTERM. Generic drain evidence is not transferred to this composition root. -- Added routed pg-erd concurrency/latency acceptance with bounded Rust-only measured origins. The load lane format-checks and directly tests the std-only origin, builds an optimized fixture, runs 4 VUs for 400 iterations, requires exact status/body and zero HTTP failures, independently gates aggregate/backend/frontend p95 below 20 ms, and requires at least 198 samples per route. Route selection remains compiled into the pg-erd migration plan rather than becoming operator-configurable CI data. Controlled loopback evidence is not production SLO proof. -- Added optional per-upstream absolute PEM trust-bundle consumption without taking ownership of certificate issuance/rotation; trust material is loaded fail-closed before listeners open. -- Added an executable local-CA TLS test through the compiled generic gateway that holds CA trust constant and proves SNI/hostname mismatch is rejected. -- Added dedicated pg-erd real-listener upstream TLS acceptance without changing production routing/TLS code. An ephemeral one-day local CA plus `backend.test` certificate must allow characterized `/api` traffic only with matching explicit CA/SNI; the same CA with mismatched SNI must fail as exact HTTP 502 without route failover, preserve `/readyz`, and leave an independent `frontend` route usable. Fixture I/O is bounded to five seconds/64 KiB, traffic/metrics ports stay concurrently reserved until startup, and exact HTTP/1.1 status parsing rejects numeric-prefix/protocol-case false GREEN. This remains gateway-to-upstream TLS evidence only; downstream TLS termination, certificate issuance/rotation and representative TLS latency are not claimed. -- Added a focused transport-adapter regression proving an upstream without a custom trust bundle leaves Pingora's platform trust roots selected rather than replacing the CA store. -- Added fail-closed binary startup and real loopback production-path tests, including held-request saturation/recovery at an in-flight budget of one. -- Added `/livez` and `/readyz` through the Pingora serving path. -- Added request-body limits and a distrust-by-default forwarded-header policy. -- Added low-cardinality metrics plus credential/cookie-safe access logging through the production path. -- Overrode Pingora framework retry/drain defaults with one total upstream attempt, a 5-second SIGTERM grace period, and a 30-second graceful-shutdown timeout. -- Added non-root/read-only-root OCI packaging with an explicit build-time allowlist for the generic and bounded pg-erd process identities. Exact-head OCI acceptance builds and starts each image under uid/gid 65532, dropped capabilities and `no-new-privileges`; the supply-chain lane builds and vulnerability-scans both candidate images. -- Extended the dedicated pg-erd OCI acceptance so the least-privilege migration container is not accepted until its process-health `/livez` endpoint answers and the separately published `/metrics` listener identifies the Pingora Prometheus service through its `text/plain` response media type. The check deliberately does not require a metric family before application traffic has emitted one, and it preserves the current one-binary-per-image packaging boundary. -- Added a committed dependency lock, fail-closed license/source/advisory policy, exact-source SBOM and image-vulnerability evidence. -- Added an exact-head owned-production coverage gate that requires 100% lines and regions without filename/function/branch exclusions; repaired compiler-generated generic startup coverage and structurally impossible literal-header error regions rather than weakening the gate. -- Added missing-public-rustdoc enforcement and documentation builds with warnings denied. -- Added DDD, product, technical, security, threat, test, operability, configuration, migration-gap, and primary-source traceability documentation. +- Bootstrapped an executable Rust Pingora proxy through pull-request governance with strict versioned configuration and explicit network authority. +- Added exact upstream transport/TLS identity, optional pre-activation PEM trust bundles, RFC 6066 SNI hostname admission, explicit connection/read/write/idle budgets, and fail-closed invalid authority. +- Added transport-neutral pg-erd Edge Routing characterization for exact `/healthz`, raw-prefix `/api` semantics including `/apiary`, fallback `/`, explicit precedence, and ambiguous/malformed route rejection. +- Added the separate pg-erd HTTP Policy characterization for the captured response-security fields, case-insensitive field identity, duplicate authority rejection, and RFC 9110-compatible field-value admission. +- Added `EdgeMigrationPlan` and `MigrationDeliveryPlan` so the bounded pg-erd profile can consume only compiled `backend`/`frontend` route authority and prevalidated concrete Pingora peers without becoming product service discovery. +- Added the dedicated `cwl-pingora-pg-erd-migration` composition root and bounded `PgErdMigrationConfig`; generic `GatewayConfig` v1 remains a one-upstream contract. +- Added effective socket-authority validation for traffic, metrics, and upstream endpoints, including port-zero, wildcard/concrete, native/IPv4-mapped, dual-stack ambiguity, recursive self-loop, broadcast, and multicast rejection while preserving valid distinct concrete unicast authority. +- Added a fail-closed HTTP/1 protocol-transition boundary. `Upgrade` or a `Connection: upgrade` token returns HTTP 501 before admission/upstream selection, and immutable Pingora peers use `HttpUpstreamRequestPolicy::deny_upgrades()` so supplier WebSocket defaults cannot bypass the public contract. +- Tightened generic and pg-erd forwarding trust. Request-controlled `Forwarded`, `X-Forwarded-*`, and `X-Real-IP` authority is discarded; generic v1 emits only gateway-owned `Forwarded: proto=http`, while pg-erd rebuilds only the characterized compatibility fields from accepted transport/request authority. +- Kept migration Admin Config parsing side-effect free for custom trust bytes. Deterministic validation completes before `build_proxy()` materializes peer/trust state once and before listener creation. +- Added compiled pg-erd production-path traffic proving process-local `/livez`/`/readyz`, characterized backend/fallback routing, exact response-policy replacement, forwarding reconstruction, declared and streamed request-body rejection, and readiness recovery. +- Added shared process-local request isolation: positive `max_in_flight_requests`, positive `upstream_keepalive_pool_size`, fail-fast HTTP 503 backpressure, lease recovery, and low-cardinality rejection telemetry. +- Added shared payload-free observability for both adapters plus process-wide Pingora-family diagnostic redaction so broad `RUST_LOG` settings cannot expose request URI, Host, Authorization, Cookie, product context, or payload data. +- Added deterministic pg-erd failure traffic for refused origin, connected read stall, orderly post-header truncation, pre-header TCP reset, post-commit TCP reset, and later independent-route recovery. Post-commit cases preserve the committed status/framing and forbid an invented second status or silent failover. +- Hardened readiness and fixture deadlines. Traffic admission requires a complete bounded `/readyz` HTTP/1.1 200 rather than bare TCP connect; repeated connect/write/read/retry loops consume absolute deadlines so slow-drip progress cannot renew fixture evidence windows indefinitely. +- Added dedicated routed pg-erd graceful-drain acceptance with one SIGTERM-relative external deadline and held-request completion during the shared drain grace period. +- Added pg-erd upstream TLS acceptance using an ephemeral local CA and `backend.test` certificate. Matching CA/SNI must succeed; the same CA with mismatched SNI must fail closed as HTTP 502 without route failover while readiness and an independent clear-text route remain usable. +- Added controlled-loopback pg-erd routed load acceptance with bounded Rust origins, exact route/body checks, zero HTTP failures, per-route sample floors, and aggregate/backend/frontend p95 below 20 ms. This remains regression evidence rather than a production SLO. +- Added non-root/read-only-root OCI packaging with a fail-closed build-time process allowlist for the generic and pg-erd binaries. Exact candidate acceptance requires uid/gid 65532, dropped capabilities, `no-new-privileges`, read-only configuration, process health, and the separate pg-erd Prometheus listener identity. +- Added committed dependency lock, license/source/advisory policy, SBOM and image-vulnerability evidence, exact-source candidate identity, and missing-public-rustdoc/documentation builds with warnings denied. +- Added an exact-head owned-production coverage gate requiring 100% lines and regions without filename/function/branch exclusions; coverage findings are repaired causally rather than excluded or threshold-weakened. +- Added versioned pg-erd response-body lifetime control. Admin Config version 2 requires a positive explicit `max_upstream_response_body_ms`; version 1 rejects that field and retains its previous semantics. +- Added `ResponseBodyLifetimeBudget` to Runtime Isolation. It starts at the first non-informational upstream response header, is not reset by later final-header callbacks, ignores empty/end-of-stream bookkeeping, and rejects non-empty body progress at or beyond the configured monotonic lifetime. +- Kept the response-lifetime control separate from Pingora `read_ms`. `read_ms` remains a per-read inactivity timer; continuously progressing body traffic is stopped at the first non-empty body callback after the lifetime expires. The current callback surface does not claim an exact interrupt for a pending read or a deadline for incomplete response headers. +- Made response-lifetime failure phase-aware. Pre-commit upstream failures preserve the existing policy-complete local error response, while `Session::response_written()` suppresses any second local status after a final response is already committed. +- Added realistic pg-erd slow-drip response traffic: the origin commits HTTP 200 framing and sends body bytes frequently enough to stay within each `read_ms`, while the shorter version-2 lifetime must terminate the incomplete response without a second status or failover, record exact request-error telemetry, preserve `/readyz`, and allow independent-route recovery. +- Added DDD, architecture, security, threat, test, operability, configuration, ADR, migration-gap, and primary-source traceability documentation. Dedicated repository-wide product-gap baseline and TRACEABILITY authority remain with their owner lane rather than being copied into migration descendants. -Release remains blocked on the exact Pingora supplier disposition, including unmaintained `derivative 2.2.0` / `RUSTSEC-2024-0388`, restoration of authoritative dependency-review evidence, terminal exact-current-head CI/supply-chain/security/review evidence for the response-lifetime descendant, representative pg-erd incomplete-header/long-lived-stream/origin-capacity/TLS/network performance, immutable registry/package identity with release-bound SBOM/provenance/reproducibility, rollback rehearsal, and protected-branch integration. Final #37 has unchanged-head upstream-TLS hosted technical evidence; the response-lifetime descendant must independently reacquire its own exact-head receipts. HTTP/1 Upgrade remains intentionally unsupported, and no consumer migration, canary, cutover, rollback or legacy removal is claimed before the release and traffic-contract gates are satisfied. +Release remains blocked on maintainer-integrated and release-qualified disposition of the supplier `derivative 2.2.0` / `RUSTSEC-2024-0388` admission failure, terminal exact-current CI/Supply/Security/CodeQL/current-review evidence for the active ancestry, the compiler/supplier promotion chain, representative routed TLS/origin-capacity/failure performance, protected-source reproducibility and provenance, immutable package/image identity, rollback rehearsal, and protected integration. Consumer parity, shadow/canary, cutover, rollback, and verified legacy reverse-proxy removal are not claimed before those gates are satisfied. From 85cd1e6802dc0e3a0dafcb04e62e9a6e09f0b9d4 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 11:05:49 +0900 Subject: [PATCH 48/49] test: make pg-erd slow-drip evidence absolute-deadline bounded --- tests/pg_erd_slow_drip_response_traffic.rs | 238 ++++++++++++--------- 1 file changed, 136 insertions(+), 102 deletions(-) diff --git a/tests/pg_erd_slow_drip_response_traffic.rs b/tests/pg_erd_slow_drip_response_traffic.rs index 24550b82..6712ab79 100644 --- a/tests/pg_erd_slow_drip_response_traffic.rs +++ b/tests/pg_erd_slow_drip_response_traffic.rs @@ -2,8 +2,8 @@ //! //! Pingora's peer `read_timeout` is an inactivity timeout that resets after each successful read. //! This fixture therefore keeps each origin write well inside `read_ms` while extending the response -//! beyond an explicit migration-owned response-body lifetime. The version-2 admin/runtime boundary -//! must terminate that body without retrying or failing over after the response has been committed. +//! beyond an explicit migration-owned response-body lifetime. Fixture readiness and I/O use absolute +//! deadlines so partial progress cannot renew the evidence window. use std::io::{ErrorKind, Read, Write}; use std::net::{SocketAddr, TcpListener, TcpStream}; @@ -13,8 +13,10 @@ use std::time::{Duration, Instant}; use tempfile::NamedTempFile; -const MAX_REQUEST_HEADER_BYTES: usize = 64 * 1024; -const SOCKET_IO_TIMEOUT: Duration = Duration::from_secs(5); +const MAX_HEADER_BYTES: usize = 64 * 1024; +const FIXTURE_IO_TIMEOUT: Duration = Duration::from_secs(5); +const STARTUP_TIMEOUT: Duration = Duration::from_secs(10); +const TERMINATION_TIMEOUT: Duration = Duration::from_secs(2); const PRE_RESPONSE_HEADER_DELAY: Duration = Duration::from_millis(150); const RESPONSE_BODY_LIFETIME: Duration = Duration::from_millis(300); @@ -33,7 +35,6 @@ enum DownstreamTermination { ConnectionReset, } -/// Holds traffic and metrics ports simultaneously so sequential bind/drop cannot reuse one port. fn reserve_distinct_loopback_listeners() -> (TcpListener, TcpListener) { let traffic = TcpListener::bind("127.0.0.1:0").expect("traffic port should be reservable"); let metrics = TcpListener::bind("127.0.0.1:0").expect("metrics port should be reservable"); @@ -44,7 +45,6 @@ fn reserve_distinct_loopback_listeners() -> (TcpListener, TcpListener) { (traffic, metrics) } -/// Writes the version-2 bounded response-lifetime contract while route authority remains compiled. fn write_config( listener: SocketAddr, metrics_listener: SocketAddr, @@ -60,28 +60,91 @@ fn write_config( file } -/// Waits for one real listener while failing immediately if process activation aborts. -fn wait_until_listening(address: SocketAddr, process: &mut Child) { - let deadline = Instant::now() + Duration::from_secs(10); +fn remaining(deadline: Instant, context: &str) -> Duration { + deadline + .checked_duration_since(Instant::now()) + .filter(|duration| !duration.is_zero()) + .unwrap_or_else(|| panic!("absolute fixture deadline expired while {context}")) +} + +fn set_read_timeout_to_remaining(stream: &TcpStream, deadline: Instant, context: &str) { + stream + .set_read_timeout(Some(remaining(deadline, context))) + .expect("read timeout should be configurable"); +} + +fn set_write_timeout_to_remaining(stream: &TcpStream, deadline: Instant, context: &str) { + stream + .set_write_timeout(Some(remaining(deadline, context))) + .expect("write timeout should be configurable"); +} + +fn read_header_block( + stream: &mut TcpStream, + deadline: Instant, + context: &str, +) -> std::io::Result> { + let mut bytes = Vec::new(); + let mut buffer = [0_u8; 1024]; + loop { + set_read_timeout_to_remaining(stream, deadline, context); + let read = stream.read(&mut buffer)?; + if read == 0 { + return Err(std::io::Error::new( + ErrorKind::UnexpectedEof, + "connection closed before response headers completed", + )); + } + bytes.extend_from_slice(&buffer[..read]); + if bytes.len() > MAX_HEADER_BYTES { + return Err(std::io::Error::new( + ErrorKind::InvalidData, + "header block exceeded 64 KiB fixture bound", + )); + } + if bytes.windows(4).any(|window| window == b"\r\n\r\n") { + return Ok(bytes); + } + } +} + +fn probe_http_status(address: SocketAddr, path: &str, deadline: Instant) -> Option { + let connect_timeout = remaining(deadline, "connecting readiness probe") + .min(Duration::from_millis(100)); + let mut stream = TcpStream::connect_timeout(&address, connect_timeout).ok()?; + let request = format!( + "GET {path} HTTP/1.1\r\nHost: readiness.invalid\r\nConnection: close\r\n\r\n" + ); + set_write_timeout_to_remaining(&stream, deadline, "writing readiness probe"); + stream.write_all(request.as_bytes()).ok()?; + let headers = read_header_block(&mut stream, deadline, "reading readiness response").ok()?; + http11_status(&String::from_utf8_lossy(&headers)) +} + +fn wait_until_http_status(address: SocketAddr, path: &str, process: &mut Child, expected: u16) { + let deadline = Instant::now() + STARTUP_TIMEOUT; loop { if let Some(status) = process .try_wait() .expect("gateway process state should be readable") { - panic!("gateway exited before accepting traffic: {status}"); + panic!("gateway exited before HTTP readiness: {status}"); + } + + if Instant::now() >= deadline { + panic!("gateway did not return HTTP {expected} for {path} within 10s"); } - if TcpStream::connect_timeout(&address, Duration::from_millis(100)).is_ok() { + + if probe_http_status(address, path, deadline) == Some(expected) { return; } - assert!( - Instant::now() < deadline, - "gateway did not start within 10s" - ); - thread::sleep(Duration::from_millis(25)); + + let sleep_for = remaining(deadline, "waiting to retry readiness") + .min(Duration::from_millis(25)); + thread::sleep(sleep_for); } } -/// Starts the compiled pg-erd gateway after its traffic/metrics reservations are released. fn start_gateway( config: &NamedTempFile, gateway_address: SocketAddr, @@ -94,47 +157,50 @@ fn start_gateway( .stderr(Stdio::null()) .spawn() .expect("compiled pg-erd migration binary should start"); - wait_until_listening(gateway_address, &mut child); - wait_until_listening(metrics_address, &mut child); - GatewayProcess(child) -} -/// Applies finite origin I/O bounds before any fixture read or write can stall the hosted lane. -fn set_origin_deadlines(stream: &TcpStream) { - stream - .set_read_timeout(Some(SOCKET_IO_TIMEOUT)) - .expect("origin read timeout should be configurable"); - stream - .set_write_timeout(Some(SOCKET_IO_TIMEOUT)) - .expect("origin write timeout should be configurable"); + wait_until_http_status(gateway_address, "/readyz", &mut child, 200); + wait_until_http_status(metrics_address, "/metrics", &mut child, 200); + GatewayProcess(child) } -/// Sends a bounded raw request used for readiness, metrics, and independent-route recovery checks. fn raw_request(address: SocketAddr, request: &[u8]) -> String { - let mut downstream = TcpStream::connect(address).expect("gateway should accept traffic"); - downstream - .set_read_timeout(Some(SOCKET_IO_TIMEOUT)) - .expect("downstream timeout should be configurable"); + let deadline = Instant::now() + FIXTURE_IO_TIMEOUT; + let mut downstream = TcpStream::connect_timeout( + &address, + remaining(deadline, "connecting bounded request"), + ) + .expect("gateway should accept traffic"); + set_write_timeout_to_remaining(&downstream, deadline, "writing bounded request"); downstream .write_all(request) .expect("downstream request should be writable"); - let mut response = String::new(); - downstream - .read_to_string(&mut response) - .expect("gateway response should be readable"); - response + + let mut response = Vec::new(); + let mut buffer = [0_u8; 1024]; + loop { + set_read_timeout_to_remaining(&downstream, deadline, "reading bounded response"); + match downstream.read(&mut buffer) { + Ok(0) => break, + Ok(read) => response.extend_from_slice(&buffer[..read]), + Err(error) if error.kind() == ErrorKind::ConnectionReset => break, + Err(error) => panic!("gateway response should complete inside absolute deadline: {error}"), + } + } + String::from_utf8_lossy(&response).into_owned() } -/// Captures the committed response until EOF/RST and records the wall-clock termination boundary. fn raw_request_until_terminal( address: SocketAddr, request: &[u8], ) -> (Vec, DownstreamTermination, Duration) { let started = Instant::now(); - let mut downstream = TcpStream::connect(address).expect("gateway should accept traffic"); - downstream - .set_read_timeout(Some(Duration::from_secs(2))) - .expect("downstream timeout should be configurable"); + let deadline = started + TERMINATION_TIMEOUT; + let mut downstream = TcpStream::connect_timeout( + &address, + remaining(deadline, "connecting slow-drip downstream"), + ) + .expect("gateway should accept traffic"); + set_write_timeout_to_remaining(&downstream, deadline, "writing slow-drip request"); downstream .write_all(request) .expect("downstream request should be writable"); @@ -142,6 +208,7 @@ fn raw_request_until_terminal( let mut response = Vec::new(); let mut buffer = [0_u8; 1024]; loop { + set_read_timeout_to_remaining(&downstream, deadline, "reading slow-drip termination"); match downstream.read(&mut buffer) { Ok(0) => return (response, DownstreamTermination::Eof, started.elapsed()), Ok(read) => response.extend_from_slice(&buffer[..read]), @@ -152,14 +219,13 @@ fn raw_request_until_terminal( started.elapsed(), ); } - Err(error) => { - panic!("slow-drip downstream response should terminate, not stall: {error}") - } + Err(error) => panic!( + "slow-drip downstream response must terminate inside one absolute 2s deadline: {error}" + ), } } } -/// Sends one characterized close-delimited GET through a gateway or metrics listener. fn get(address: SocketAddr, path: &str) -> String { raw_request( address, @@ -168,31 +234,13 @@ fn get(address: SocketAddr, path: &str) -> String { ) } -/// Reads exactly one finite HTTP/1 request header block from an origin connection. fn read_request_headers(stream: &mut TcpStream) -> String { - set_origin_deadlines(stream); - let mut bytes = Vec::new(); - let mut buffer = [0_u8; 1024]; - loop { - let read = stream - .read(&mut buffer) - .expect("origin request should be readable inside the fixture deadline"); - assert!( - read > 0, - "gateway closed origin request before headers completed" - ); - bytes.extend_from_slice(&buffer[..read]); - assert!( - bytes.len() <= MAX_REQUEST_HEADER_BYTES, - "origin request headers exceeded the 64 KiB fixture bound" - ); - if bytes.windows(4).any(|window| window == b"\r\n\r\n") { - return String::from_utf8_lossy(&bytes).into_owned(); - } - } + let deadline = Instant::now() + FIXTURE_IO_TIMEOUT; + let bytes = read_header_block(stream, deadline, "reading origin request headers") + .expect("origin request headers should complete inside one absolute 5s deadline"); + String::from_utf8_lossy(&bytes).into_owned() } -/// Parses only an exact case-sensitive HTTP/1.1 three-digit status token. fn http11_status(response: &str) -> Option { let mut tokens = response.lines().next()?.split_ascii_whitespace(); if tokens.next()? != "HTTP/1.1" { @@ -205,7 +253,6 @@ fn http11_status(response: &str) -> Option { code.parse().ok() } -/// Returns semantically exact values for one HTTP field name, rejecting lookalike field names. fn header_values(headers: &str, field_name: &str) -> Vec { headers .lines() @@ -216,12 +263,10 @@ fn header_values(headers: &str, field_name: &str) -> Vec { .collect() } -/// Requires one exact unlabelled Prometheus sample so numeric-prefix values cannot create GREEN. fn has_exact_metric_sample(metrics: &str, sample: &str) -> bool { metrics.lines().any(|line| line.trim() == sample) } -/// Proves progress inside `read_ms` cannot evade the versioned whole-body lifetime boundary. #[test] fn compiled_pg_erd_terminates_continuous_response_drip_without_poisoning_other_routes() { let backend = TcpListener::bind("127.0.0.1:0").expect("backend fixture should bind"); @@ -230,11 +275,12 @@ fn compiled_pg_erd_terminates_continuous_response_drip_without_poisoning_other_r let (mut stream, _) = backend .accept() .expect("routed request should reach the characterized backend authority"); + stream + .set_write_timeout(Some(FIXTURE_IO_TIMEOUT)) + .expect("origin write timeout should be configurable"); let request = read_request_headers(&mut stream); assert!(request.starts_with("GET /api/slow-drip HTTP/1.1\r\n")); - // The delay stays below read_ms but makes a request-start lifetime distinguishable from the - // selected first-final-response-header lifetime on the real listener path. thread::sleep(PRE_RESPONSE_HEADER_DELAY); stream .write_all(b"HTTP/1.1 200 OK\r\nContent-Length: 20\r\nConnection: close\r\n\r\n") @@ -267,6 +313,9 @@ fn compiled_pg_erd_terminates_continuous_response_drip_without_poisoning_other_r let (mut stream, _) = frontend .accept() .expect("fallback request should reach the independent frontend authority"); + stream + .set_write_timeout(Some(FIXTURE_IO_TIMEOUT)) + .expect("origin write timeout should be configurable"); let request = read_request_headers(&mut stream); assert!(request.starts_with("GET /after-slow-drip HTTP/1.1\r\n")); stream @@ -297,20 +346,17 @@ fn compiled_pg_erd_terminates_continuous_response_drip_without_poisoning_other_r gateway_address, b"GET /api/slow-drip HTTP/1.1\r\nHost: app.example:8080\r\nConnection: close\r\n\r\n", ); - assert!( - matches!( - termination, - DownstreamTermination::Eof | DownstreamTermination::ConnectionReset - ), - "an over-budget response body must terminate the downstream connection" - ); + assert!(matches!( + termination, + DownstreamTermination::Eof | DownstreamTermination::ConnectionReset + )); assert!( elapsed < Duration::from_secs(1), - "the response-body budget must stop the delayed-header continuous drip instead of allowing it to run to completion: {elapsed:?}" + "response-body lifetime must stop the continuous drip instead of allowing completion: {elapsed:?}" ); assert!( elapsed >= PRE_RESPONSE_HEADER_DELAY + RESPONSE_BODY_LIFETIME, - "termination must occur only after the 150ms pre-header delay plus the 300ms body-progress budget, proving the budget starts at the response header: {elapsed:?}" + "lifetime must start at the final response header rather than request start: {elapsed:?}" ); let header_end = partial @@ -322,33 +368,22 @@ fn compiled_pg_erd_terminates_continuous_response_drip_without_poisoning_other_r assert_eq!( http11_status(&headers), Some(200), - "a post-commit lifetime failure cannot be rewritten as a second status: {headers:?}" - ); - assert_eq!( - header_values(&headers, "Content-Length"), - vec!["20"], - "the committed framing must remain the characterized single Content-Length field" + "post-commit lifetime failure cannot be rewritten as a second status: {headers:?}" ); + assert_eq!(header_values(&headers, "Content-Length"), vec!["20"]); let body = &partial[header_end..]; - assert!( - !body.is_empty(), - "the committed response must deliver body progress before the budget terminates it" - ); + assert!(!body.is_empty(), "body progress must cross the callback boundary"); assert!( body.len() < 20, - "the configured response-body budget must terminate before the declared body completes" + "configured lifetime must terminate before the declared body completes" ); let readiness = get(gateway_address, "/readyz"); - assert_eq!( - http11_status(&readiness), - Some(200), - "one slow-drip origin must not poison process readiness: {readiness:?}" - ); + assert_eq!(http11_status(&readiness), Some(200)); let metrics = get(metrics_address, "/metrics"); assert!( has_exact_metric_sample(&metrics, "cwl_pingora_gateway_request_errors_total 1"), - "response-lifetime enforcement must remain visible through low-cardinality error telemetry: {metrics:?}" + "lifetime enforcement must remain visible through bounded error telemetry: {metrics:?}" ); let recovered = get(gateway_address, "/after-slow-drip"); assert_eq!(http11_status(&recovered), Some(200)); @@ -362,7 +397,6 @@ fn compiled_pg_erd_terminates_continuous_response_drip_without_poisoning_other_r .expect("slow-drip backend fixture should complete"); } -/// Prevents status, header-name, and metric-value lookalikes from satisfying the traffic oracle. #[test] fn slow_drip_evidence_parsers_reject_lookalikes() { assert_eq!(http11_status("HTTP/1.1 200 OK\r\n\r\n"), Some(200)); From d8ed573176ce754909ccb1625a2199a1035c2f4d Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 16:26:11 +0900 Subject: [PATCH 49/49] test: apply rustfmt to slow-drip evidence --- tests/pg_erd_slow_drip_response_traffic.rs | 30 ++++++++++++---------- 1 file changed, 16 insertions(+), 14 deletions(-) diff --git a/tests/pg_erd_slow_drip_response_traffic.rs b/tests/pg_erd_slow_drip_response_traffic.rs index 6712ab79..25e1029e 100644 --- a/tests/pg_erd_slow_drip_response_traffic.rs +++ b/tests/pg_erd_slow_drip_response_traffic.rs @@ -109,12 +109,11 @@ fn read_header_block( } fn probe_http_status(address: SocketAddr, path: &str, deadline: Instant) -> Option { - let connect_timeout = remaining(deadline, "connecting readiness probe") - .min(Duration::from_millis(100)); + let connect_timeout = + remaining(deadline, "connecting readiness probe").min(Duration::from_millis(100)); let mut stream = TcpStream::connect_timeout(&address, connect_timeout).ok()?; - let request = format!( - "GET {path} HTTP/1.1\r\nHost: readiness.invalid\r\nConnection: close\r\n\r\n" - ); + let request = + format!("GET {path} HTTP/1.1\r\nHost: readiness.invalid\r\nConnection: close\r\n\r\n"); set_write_timeout_to_remaining(&stream, deadline, "writing readiness probe"); stream.write_all(request.as_bytes()).ok()?; let headers = read_header_block(&mut stream, deadline, "reading readiness response").ok()?; @@ -139,8 +138,8 @@ fn wait_until_http_status(address: SocketAddr, path: &str, process: &mut Child, return; } - let sleep_for = remaining(deadline, "waiting to retry readiness") - .min(Duration::from_millis(25)); + let sleep_for = + remaining(deadline, "waiting to retry readiness").min(Duration::from_millis(25)); thread::sleep(sleep_for); } } @@ -165,11 +164,9 @@ fn start_gateway( fn raw_request(address: SocketAddr, request: &[u8]) -> String { let deadline = Instant::now() + FIXTURE_IO_TIMEOUT; - let mut downstream = TcpStream::connect_timeout( - &address, - remaining(deadline, "connecting bounded request"), - ) - .expect("gateway should accept traffic"); + let mut downstream = + TcpStream::connect_timeout(&address, remaining(deadline, "connecting bounded request")) + .expect("gateway should accept traffic"); set_write_timeout_to_remaining(&downstream, deadline, "writing bounded request"); downstream .write_all(request) @@ -183,7 +180,9 @@ fn raw_request(address: SocketAddr, request: &[u8]) -> String { Ok(0) => break, Ok(read) => response.extend_from_slice(&buffer[..read]), Err(error) if error.kind() == ErrorKind::ConnectionReset => break, - Err(error) => panic!("gateway response should complete inside absolute deadline: {error}"), + Err(error) => { + panic!("gateway response should complete inside absolute deadline: {error}") + } } } String::from_utf8_lossy(&response).into_owned() @@ -372,7 +371,10 @@ fn compiled_pg_erd_terminates_continuous_response_drip_without_poisoning_other_r ); assert_eq!(header_values(&headers, "Content-Length"), vec!["20"]); let body = &partial[header_end..]; - assert!(!body.is_empty(), "body progress must cross the callback boundary"); + assert!( + !body.is_empty(), + "body progress must cross the callback boundary" + ); assert!( body.len() < 20, "configured lifetime must terminate before the declared body completes"