From 1ca2e56b5822a2d40b40fd6a8e829b7c34f8672c Mon Sep 17 00:00:00 2001 From: Nina Rei Date: Tue, 1 Sep 2026 12:33:54 -0400 Subject: [PATCH 1/9] feat(compat): add serverless-compat inventory reporter (SVLS-9604) Adds a fire-and-forget inventory reporter that makes the serverless-compat mini-agent appear in Fleet Automation as a serverless_compat_agent row. Supported workloads: azure_function, cloud_function (GCP Gen1). Lambda and Azure Spring Apps are explicitly skipped. Key design points: - Startup report fires immediately; periodic reports every 30 min - Process UUID is stable across all reports from the same process - Bounded retry (3 attempts, exponential backoff) for 429/5xx/transport errors - 4xx rejections are not retried - Gen2 Cloud Run Functions (FUNCTION_TARGET set) are excluded - DD_SERVERLESS_COMPAT_VERSION from language package is preferred over the Rust crate version for serverless_compat_version - DD_SERVERLESS_COMPAT_RUNTIME/RUNTIME_VERSION env vars supported for language package handoff, with fallback to Azure/GCP env vars - UUID absent from agent_metadata (only at payload top level) - platform_version field absent (not in REDAPL schema) - Hostname absent (activates ECS Fargate path in EPRW) - GCP metadata server fallback for Gen1 when region/project absent from env --- Cargo.lock | 2 + crates/datadog-serverless-compat/Cargo.toml | 5 +- .../src/inventory.rs | 915 ++++++++++++++++++ crates/datadog-serverless-compat/src/main.rs | 21 + 4 files changed, 942 insertions(+), 1 deletion(-) create mode 100644 crates/datadog-serverless-compat/src/inventory.rs diff --git a/Cargo.lock b/Cargo.lock index 5a599c9..87e4f9a 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -589,11 +589,13 @@ dependencies = [ "dogstatsd", "libdd-trace-utils 8.0.0", "reqwest", + "serde", "serde_json", "tokio", "tokio-util", "tracing", "tracing-subscriber", + "uuid", "zstd", ] diff --git a/crates/datadog-serverless-compat/Cargo.toml b/crates/datadog-serverless-compat/Cargo.toml index a1ec609..55c03b5 100644 --- a/crates/datadog-serverless-compat/Cargo.toml +++ b/crates/datadog-serverless-compat/Cargo.toml @@ -18,7 +18,9 @@ libdd-trace-utils = { git = "https://github.com/DataDog/libdatadog", rev = "a820 datadog-fips = { path = "../datadog-fips", default-features = false } dogstatsd = { path = "../dogstatsd", default-features = true } reqwest = { version = "0.12.4", default-features = false } -tokio = { version = "1", features = ["macros", "rt-multi-thread"] } +serde = { version = "1.0", default-features = false, features = ["derive"] } +serde_json = { version = "1.0", default-features = false, features = ["alloc"] } +tokio = { version = "1", features = ["macros", "rt-multi-thread", "time"] } tokio-util = { version = "0.7", default-features = false } tracing = { version = "0.1", default-features = false } tracing-subscriber = { version = "0.3", default-features = false, features = [ @@ -28,6 +30,7 @@ tracing-subscriber = { version = "0.3", default-features = false, features = [ "env-filter", "tracing-log", ] } +uuid = { version = "1", default-features = false, features = ["v4"] } zstd = { version = "0.13.3", default-features = false } [dev-dependencies] diff --git a/crates/datadog-serverless-compat/src/inventory.rs b/crates/datadog-serverless-compat/src/inventory.rs new file mode 100644 index 0000000..725ff04 --- /dev/null +++ b/crates/datadog-serverless-compat/src/inventory.rs @@ -0,0 +1,915 @@ +// Copyright 2023-Present Datadog, Inc. https://www.datadoghq.com/ +// SPDX-License-Identifier: Apache-2.0 + +use datadog_fips::reqwest_adapter::create_reqwest_client_builder; +use libdd_trace_utils::trace_utils::EnvironmentType; +use std::env; +use std::time::{Duration, SystemTime, UNIX_EPOCH}; +use tokio::time::interval; +use tracing::{info, warn}; + +/// How often to send a periodic inventory report while the mini-agent is running. +const INVENTORY_INTERVAL: Duration = Duration::from_secs(30 * 60); + +/// Maximum retry attempts for transient failures (429, 5xx, transport errors). +const MAX_RETRIES: u32 = 3; + +/// Minimum Datadog agent protocol version accepted by EPRW (7.x.x format). +/// Used only for HTTP transport headers. The actual Compat version is reported +/// as `agent_metadata.serverless_compat_version`. +const AGENT_VERSION: &str = "7.83.0"; + +/// Supported Compat workload types. Unsupported env types (Lambda, Azure Spring +/// Apps) are silently skipped so they never create a `serverless_compat_agent` row. +fn supported_workload_type(env_type: &EnvironmentType) -> Option<&'static str> { + match env_type { + EnvironmentType::AzureFunction => Some("azure_function"), + EnvironmentType::CloudFunction => Some("cloud_function"), + // Lambda and Azure Spring Apps are not supported by serverless_compat_agent. + EnvironmentType::LambdaFunction | EnvironmentType::AzureSpringApp => None, + } +} + +/// Runs the inventory reporter for the lifetime of the mini-agent. +/// +/// Sends a startup report immediately, then a periodic report every +/// [`INVENTORY_INTERVAL`]. Spawned as a background task — never panics, +/// never blocks agent startup. +pub async fn run_inventory_reporter( + api_key: &str, + dd_site: &str, + https_proxy: Option<&str>, + env_type: EnvironmentType, +) { + let Some(workload_type) = supported_workload_type(&env_type) else { + // Unsupported workload: skip silently. + return; + }; + + let client = match build_client(https_proxy) { + Ok(c) => c, + Err(e) => { + warn!("inventory: failed to create HTTP client: {e}"); + return; + } + }; + + // Stable process UUID — reused across all reports from this process. + let process_id = uuid::Uuid::new_v4().to_string(); + + // Startup report. + send_report( + &client, + api_key, + dd_site, + &env_type, + workload_type, + &process_id, + "startup", + ) + .await; + + // Periodic reports. + let mut ticker = interval(INVENTORY_INTERVAL); + ticker.tick().await; // consumes the immediate first tick + loop { + ticker.tick().await; + send_report( + &client, + api_key, + dd_site, + &env_type, + workload_type, + &process_id, + "periodic", + ) + .await; + } +} + +/// Builds and sends one inventory report with bounded retry for transient failures. +async fn send_report( + client: &reqwest::Client, + api_key: &str, + dd_site: &str, + env_type: &EnvironmentType, + workload_type: &str, + process_id: &str, + report_reason: &str, +) { + let (mut resource_id, resource_name) = build_resource_identity(env_type); + + // Gen1 Cloud Functions: if FUNCTION_NAME was present but region/project were + // absent from env vars, try the GCP instance metadata server to complete the + // resource_id. resource_name is non-empty only when a function name was found; + // if both are empty, this is an unrecognised environment — skip entirely. + if matches!(env_type, EnvironmentType::CloudFunction) + && resource_id.is_empty() + && !resource_name.is_empty() + { + let project_from_env = env::var("GCP_PROJECT") + .or_else(|_| env::var("GCLOUD_PROJECT")) + .or_else(|_| env::var("GOOGLE_CLOUD_PROJECT")) + .ok() + .filter(|s| !s.is_empty()); + + let (region_opt, project_opt) = if project_from_env.is_some() { + (fetch_gcp_region_from_metadata().await, project_from_env) + } else { + let (r, p) = tokio::join!( + fetch_gcp_region_from_metadata(), + fetch_gcp_project_from_metadata(), + ); + (r, p) + }; + + if let (Some(region), Some(project)) = (region_opt, project_opt) { + resource_id = format!( + "//cloudfunctions.googleapis.com/projects/{}/locations/{}/functions/{}", + project, region, resource_name + ); + } + } + + // EPRW rejects the serverless_compat_agent write when required fields are absent. + // Log and skip rather than sending a doomed payload. + if resource_id.is_empty() { + warn!( + "inventory: required identity unavailable, skipping report \ + (report_reason={report_reason}, workload_type={workload_type})" + ); + return; + } + + let body = match build_payload( + process_id, + workload_type, + report_reason, + &resource_id, + &resource_name, + env_type, + ) { + Ok(b) => b, + Err(e) => { + warn!("inventory: failed to serialize payload: {e}"); + return; + } + }; + + let url = format!("https://api.{dd_site}/api/v1/metadata"); + + for attempt in 0..=MAX_RETRIES { + match do_send(client, &url, api_key, body.clone()).await { + Ok(status) if status < 300 || status == 202 => { + info!( + "inventory: report sent \ + (report_reason={report_reason}, workload_type={workload_type}, \ + resource_id={resource_id}, process_id={process_id}, status={status})" + ); + return; + } + Ok(429) | Ok(500..=599) if attempt < MAX_RETRIES => { + let backoff = Duration::from_secs(1 << attempt); + warn!( + "inventory: transient failure, retrying in {backoff:?} \ + (report_reason={report_reason}, attempt={attempt})" + ); + tokio::time::sleep(backoff).await; + } + Ok(status) => { + warn!( + "inventory: intake rejected report \ + (report_reason={report_reason}, status={status}, \ + resource_id={resource_id}, process_id={process_id})" + ); + return; + } + Err(e) if attempt < MAX_RETRIES => { + let backoff = Duration::from_secs(1 << attempt); + warn!( + "inventory: transport error, retrying in {backoff:?} \ + (report_reason={report_reason}, attempt={attempt}, error={e})" + ); + tokio::time::sleep(backoff).await; + } + Err(e) => { + warn!( + "inventory: transport error after {attempt} attempts \ + (report_reason={report_reason}, error={e})" + ); + return; + } + } + } +} + +/// Sends the raw payload body, returning the HTTP status code or a transport error. +async fn do_send( + client: &reqwest::Client, + url: &str, + api_key: &str, + body: Vec, +) -> Result { + let resp = client + .post(url) + .header("DD-API-KEY", api_key) + .header("Content-Type", "application/json") + .header("DD-Agent-Version", AGENT_VERSION) + .header("User-Agent", format!("datadog-agent/{AGENT_VERSION}")) + .body(body) + .send() + .await?; + Ok(resp.status().as_u16()) +} + +/// Builds the serialized JSON inventory payload. +fn build_payload( + process_id: &str, + workload_type: &str, + report_reason: &str, + resource_id: &str, + resource_name: &str, + env_type: &EnvironmentType, +) -> Result, serde_json::Error> { + // Must be nanoseconds to match time.Now().UnixNano() expected by EPRW. + let timestamp = SystemTime::now() + .duration_since(UNIX_EPOCH) + .map(|d| d.as_nanos() as i64) + .unwrap_or(0); + + // serverless_compat_version: prefer DD_SERVERLESS_COMPAT_VERSION (set by the + // language package wrapping this binary); fall back to the Rust crate version + // when running standalone. + let compat_version = env::var("DD_SERVERLESS_COMPAT_VERSION") + .ok() + .filter(|v| !v.is_empty()) + .unwrap_or_else(|| env!("CARGO_PKG_VERSION").to_string()); + + let mut metadata = serde_json::json!({ + "flavor": "serverless-compat", + "workload_type": workload_type, + "report_reason": report_reason, + "resource_id": resource_id, + "resource_name": resource_name, + "serverless_compat_version": compat_version, + }); + + // DD unified service tags. + for (env_key, meta_key) in [ + ("DD_ENV", "dd_env"), + ("DD_SERVICE", "dd_service"), + ("DD_VERSION", "dd_version"), + ("DD_SITE", "dd_site"), + ] { + if let Ok(val) = env::var(env_key) + && !val.is_empty() + { + metadata[meta_key] = serde_json::Value::String(val); + } + } + + // Platform-specific optional fields (runtime, region, cloud IDs, etc.). + enrich_platform_fields(&mut metadata, env_type); + + // Hostname intentionally absent: setting it (even to "") causes EPRW to + // attempt a host_id lookup that fails for serverless workloads, rejecting + // the record. Omitting it activates the ECS Fargate UUID path in EPRW. + let payload = serde_json::json!({ + "uuid": process_id, + "timestamp": timestamp, + "agent_metadata": metadata, + }); + + serde_json::to_vec(&payload) +} + +/// Returns `(resource_id, resource_name)` for supported Compat workloads. +/// +/// `resource_id` is the canonical cloud resource identifier used as the primary +/// key in `serverless_compat_agent`. An empty `resource_id` means the required +/// environment variables are absent; the caller must skip the write. +fn build_resource_identity(env_type: &EnvironmentType) -> (String, String) { + match env_type { + EnvironmentType::AzureFunction => build_azure_function_identity(), + EnvironmentType::CloudFunction => build_cloud_function_identity(), + EnvironmentType::LambdaFunction | EnvironmentType::AzureSpringApp => { + (String::new(), String::new()) + } + } +} + +fn build_azure_function_identity() -> (String, String) { + let name = env::var("WEBSITE_SITE_NAME").unwrap_or_default(); + // WEBSITE_OWNER_NAME = "{subscription_guid}+{rg}-{region}webspace[-os]" + let owner_name = env::var("WEBSITE_OWNER_NAME").unwrap_or_default(); + + let sub = owner_name + .split('+') + .next() + .filter(|s| !s.is_empty()) + .map(str::to_string) + .unwrap_or_default(); + + // WEBSITE_RESOURCE_GROUP is not always injected; parse from WEBSITE_OWNER_NAME + // when absent. Format after '+': "{rg}-{region}webspace[-Linux|-Windows]" + let rg = env::var("WEBSITE_RESOURCE_GROUP") + .ok() + .filter(|s| !s.is_empty()) + .or_else(|| parse_rg_from_owner_name(&owner_name)) + .unwrap_or_default(); + + if name.is_empty() || rg.is_empty() || sub.is_empty() { + return (String::new(), name); + } + + let resource_id = format!( + "//microsoft.azure/functionApps/{}/{}/{}", + sub, + rg, + name.to_lowercase() + ); + (resource_id, name) +} + +/// Parses the resource group from `WEBSITE_OWNER_NAME`. +/// +/// Format: `"{sub}+{rg}-{region}webspace[-Linux|-Windows]"` +/// Strips the OS suffix, "webspace", then the trailing "-{region}" segment. +fn parse_rg_from_owner_name(owner_name: &str) -> Option { + let after_plus = owner_name.split('+').nth(1)?; + let stripped = after_plus + .strip_suffix("-Linux") + .or_else(|| after_plus.strip_suffix("-Windows")) + .unwrap_or(after_plus); + let without_webspace = stripped.strip_suffix("webspace")?; + let last_dash = without_webspace.rfind('-')?; + let rg = &without_webspace[..last_dash]; + if rg.is_empty() { None } else { Some(rg.to_string()) } +} + +fn build_cloud_function_identity() -> (String, String) { + // Gen2 Cloud Run Functions set FUNCTION_TARGET alongside K_SERVICE. + // These belong in serverless_init_agent, not serverless_compat_agent. + if env::var("FUNCTION_TARGET").map(|v| !v.is_empty()).unwrap_or(false) { + return (String::new(), String::new()); + } + + // Gen1: FUNCTION_NAME is canonical; newer Gen1 runtimes on Cloud Run infra + // may omit it and expose K_SERVICE instead. + let name = env::var("FUNCTION_NAME") + .or_else(|_| env::var("K_SERVICE")) + .unwrap_or_default(); + if name.is_empty() { + return (String::new(), String::new()); + } + + let region = env::var("FUNCTION_REGION") + .or_else(|_| env::var("GOOGLE_CLOUD_REGION")) + .or_else(|_| env::var("REGION_NAME")) + .unwrap_or_default(); + let project = env::var("GCP_PROJECT") + .or_else(|_| env::var("GCLOUD_PROJECT")) + .or_else(|_| env::var("GOOGLE_CLOUD_PROJECT")) + .unwrap_or_default(); + + if region.is_empty() || project.is_empty() { + // Region/project absent from env vars — caller may retry via metadata server. + return (String::new(), name); + } + + let resource_id = format!( + "//cloudfunctions.googleapis.com/projects/{}/locations/{}/functions/{}", + project, region, name + ); + (resource_id, name) +} + +/// Fetches a single field from the GCP instance metadata server. +async fn fetch_gcp_metadata_value( + path: &str, + label: &str, + parse: impl Fn(&str) -> Option, +) -> Option { + let client = create_reqwest_client_builder() + .and_then(|b| { + b.timeout(Duration::from_secs(2)).build().map_err(Into::into) + }) + .ok()?; + + let url = format!("http://metadata.google.internal/computeMetadata/v1/{path}"); + let resp = client + .get(&url) + .header("Metadata-Flavor", "Google") + .send() + .await + .ok()?; + + if !resp.status().is_success() { + warn!("inventory: GCP metadata server returned {} for {label}", resp.status()); + return None; + } + + let body = resp.text().await.ok()?; + let result = parse(body.trim()); + info!("inventory: GCP metadata server {label}: {:?}", result); + result +} + +async fn fetch_gcp_region_from_metadata() -> Option { + // Response: "projects//regions/" + fetch_gcp_metadata_value("instance/region", "region", |body| { + body.split('/').next_back().filter(|s| !s.is_empty()).map(str::to_string) + }) + .await +} + +async fn fetch_gcp_project_from_metadata() -> Option { + fetch_gcp_metadata_value("project/project-id", "project-id", |body| { + if body.is_empty() { None } else { Some(body.to_string()) } + }) + .await +} + +/// Adds platform-specific optional fields to `metadata`. +fn enrich_platform_fields(metadata: &mut serde_json::Value, env_type: &EnvironmentType) { + match env_type { + EnvironmentType::AzureFunction => enrich_azure_function_fields(metadata), + EnvironmentType::CloudFunction => enrich_cloud_function_fields(metadata), + EnvironmentType::LambdaFunction | EnvironmentType::AzureSpringApp => {} + } +} + +fn enrich_azure_function_fields(metadata: &mut serde_json::Value) { + let owner_name = env::var("WEBSITE_OWNER_NAME").unwrap_or_default(); + + // Region: prefer REGION_NAME; fall back to parsing WEBSITE_OWNER_NAME. + let region = env::var("REGION_NAME") + .ok() + .filter(|s| !s.is_empty()) + .or_else(|| { + let after_plus = owner_name.split('+').nth(1)?; + let without_webspace = after_plus + .strip_suffix("-Linux") + .or_else(|| after_plus.strip_suffix("-Windows")) + .unwrap_or(after_plus) + .strip_suffix("webspace")?; + without_webspace.split('-').next_back().map(str::to_string) + }); + if let Some(r) = region { + metadata["region"] = serde_json::Value::String(r); + } + + if let Some(sub) = owner_name.split('+').next().filter(|s| !s.is_empty()) { + metadata["azure_subscription_id"] = serde_json::Value::String(sub.to_string()); + } + if let Ok(rg) = env::var("WEBSITE_RESOURCE_GROUP") + && !rg.is_empty() + { + metadata["azure_resource_group"] = serde_json::Value::String(rg); + } + + // Runtime: prefer DD_SERVERLESS_COMPAT_RUNTIME (set by language package); + // fall back to FUNCTIONS_WORKER_RUNTIME injected by Azure. + let runtime = env::var("DD_SERVERLESS_COMPAT_RUNTIME") + .ok() + .filter(|s| !s.is_empty()) + .or_else(|| env::var("FUNCTIONS_WORKER_RUNTIME").ok().filter(|s| !s.is_empty())); + if let Some(rt) = runtime { + metadata["runtime"] = serde_json::Value::String(rt); + } + + // Runtime version: prefer DD_SERVERLESS_COMPAT_RUNTIME_VERSION (language package), + // then FUNCTIONS_WORKER_RUNTIME_VERSION, then language-specific vars. + let runtime_ver = env::var("DD_SERVERLESS_COMPAT_RUNTIME_VERSION") + .ok() + .filter(|s| !s.is_empty()) + .or_else(|| { + env::var("FUNCTIONS_WORKER_RUNTIME_VERSION").ok().filter(|s| !s.is_empty()) + }); + if let Some(v) = runtime_ver { + metadata["serverless_compat_runtime_version"] = serde_json::Value::String(v); + } +} + +fn enrich_cloud_function_fields(metadata: &mut serde_json::Value) { + let region = env::var("FUNCTION_REGION") + .or_else(|_| env::var("GOOGLE_CLOUD_REGION")) + .or_else(|_| env::var("REGION_NAME")) + .ok() + .filter(|s| !s.is_empty()); + if let Some(r) = region { + metadata["region"] = serde_json::Value::String(r); + } + + let project = env::var("GCP_PROJECT") + .or_else(|_| env::var("GCLOUD_PROJECT")) + .or_else(|_| env::var("GOOGLE_CLOUD_PROJECT")) + .ok() + .filter(|s| !s.is_empty()); + if let Some(p) = project { + metadata["gcp_project_id"] = serde_json::Value::String(p); + } + + // Runtime: prefer DD_SERVERLESS_COMPAT_RUNTIME (language package); fall back + // to detecting from well-known GCP Cloud Functions gen1 env vars. + let (lang, ver) = env::var("DD_SERVERLESS_COMPAT_RUNTIME") + .ok() + .filter(|s| !s.is_empty()) + .map(|rt| { + let ver = env::var("DD_SERVERLESS_COMPAT_RUNTIME_VERSION") + .ok() + .filter(|s| !s.is_empty()) + .unwrap_or_default(); + (rt, ver) + }) + .unwrap_or_else(detect_gcp_gen1_runtime); + + if !lang.is_empty() { + metadata["runtime"] = serde_json::Value::String(lang); + } + if !ver.is_empty() { + metadata["serverless_compat_runtime_version"] = serde_json::Value::String(ver); + } +} + +/// Infers the runtime language and version for GCP Cloud Functions Gen1 from +/// well-known environment variables injected by the GCP runtime. +fn detect_gcp_gen1_runtime() -> (String, String) { + for (lang, env_var) in [ + ("node", "NODE_VERSION"), + ("python", "PYTHON_VERSION"), + ("java", "JAVA_VERSION"), + ("go", "GO_VERSION"), + ] { + if let Ok(ver) = env::var(env_var) + && !ver.is_empty() + { + return (lang.to_string(), ver); + } + } + (String::new(), String::new()) +} + +fn build_client(https_proxy: Option<&str>) -> Result> { + let mut builder = + create_reqwest_client_builder()?.timeout(Duration::from_secs(10)); + + if let Some(proxy) = https_proxy { + builder = builder.proxy(reqwest::Proxy::https(proxy)?); + } + + Ok(builder.build()?) +} + +#[cfg(test)] +mod tests { + use super::*; + + /// Mutex to serialize tests that read or write process-global env vars. + static ENV_LOCK: std::sync::LazyLock> = + std::sync::LazyLock::new(|| std::sync::Mutex::new(())); + + // ── Workload type filtering ────────────────────────────────────────────── + + #[test] + fn supported_workloads_accepted() { + assert!(supported_workload_type(&EnvironmentType::AzureFunction).is_some()); + assert!(supported_workload_type(&EnvironmentType::CloudFunction).is_some()); + } + + #[test] + fn unsupported_workloads_skipped() { + assert!(supported_workload_type(&EnvironmentType::LambdaFunction).is_none()); + assert!(supported_workload_type(&EnvironmentType::AzureSpringApp).is_none()); + } + + #[test] + fn azure_function_workload_type() { + assert_eq!(supported_workload_type(&EnvironmentType::AzureFunction), Some("azure_function")); + } + + #[test] + fn cloud_function_workload_type() { + assert_eq!(supported_workload_type(&EnvironmentType::CloudFunction), Some("cloud_function")); + } + + // ── Azure Function identity ────────────────────────────────────────────── + + #[test] + fn azure_function_identity_full() { + let _lock = ENV_LOCK.lock().unwrap(); + unsafe { + env::set_var("WEBSITE_SITE_NAME", "my-func-app"); + env::set_var("WEBSITE_RESOURCE_GROUP", "my-rg"); + env::set_var("WEBSITE_OWNER_NAME", "abc123+my-rg-eastuswebspace"); + } + + let (id, name) = build_azure_function_identity(); + + assert_eq!(name, "my-func-app"); + assert_eq!(id, "//microsoft.azure/functionApps/abc123/my-rg/my-func-app"); + + unsafe { + env::remove_var("WEBSITE_SITE_NAME"); + env::remove_var("WEBSITE_RESOURCE_GROUP"); + env::remove_var("WEBSITE_OWNER_NAME"); + } + } + + #[test] + fn azure_function_identity_rg_from_owner_name() { + // WEBSITE_RESOURCE_GROUP absent; RG parsed from WEBSITE_OWNER_NAME. + let _lock = ENV_LOCK.lock().unwrap(); + unsafe { + env::set_var("WEBSITE_SITE_NAME", "my-func"); + env::remove_var("WEBSITE_RESOURCE_GROUP"); + env::set_var("WEBSITE_OWNER_NAME", "sub123+my-resource-group-westus2webspace-Linux"); + } + + let (id, name) = build_azure_function_identity(); + + assert_eq!(name, "my-func"); + assert!(id.contains("/my-resource-group/"), "expected RG in id: {id}"); + + unsafe { + env::remove_var("WEBSITE_SITE_NAME"); + env::remove_var("WEBSITE_OWNER_NAME"); + } + } + + #[test] + fn azure_function_identity_missing_name_returns_empty() { + let _lock = ENV_LOCK.lock().unwrap(); + unsafe { + env::remove_var("WEBSITE_SITE_NAME"); + env::set_var("WEBSITE_RESOURCE_GROUP", "my-rg"); + env::set_var("WEBSITE_OWNER_NAME", "abc123+my-rg-eastuswebspace"); + } + + let (id, _name) = build_azure_function_identity(); + assert!(id.is_empty(), "missing WEBSITE_SITE_NAME must produce empty resource_id"); + + unsafe { + env::remove_var("WEBSITE_RESOURCE_GROUP"); + env::remove_var("WEBSITE_OWNER_NAME"); + } + } + + // ── GCP Cloud Function identity ────────────────────────────────────────── + + #[test] + fn cloud_function_gen1_identity_full() { + let _lock = ENV_LOCK.lock().unwrap(); + unsafe { + env::remove_var("FUNCTION_TARGET"); + env::set_var("FUNCTION_NAME", "my-fn"); + env::set_var("FUNCTION_REGION", "us-central1"); + env::set_var("GCP_PROJECT", "my-project"); + } + + let (id, name) = build_cloud_function_identity(); + + assert_eq!(name, "my-fn"); + assert_eq!( + id, + "//cloudfunctions.googleapis.com/projects/my-project/locations/us-central1/functions/my-fn" + ); + + unsafe { + env::remove_var("FUNCTION_NAME"); + env::remove_var("FUNCTION_REGION"); + env::remove_var("GCP_PROJECT"); + } + } + + #[test] + fn cloud_function_gen1_region_name_fallback() { + // Gen1 on Cloud Run infra: FUNCTION_REGION absent, REGION_NAME present. + let _lock = ENV_LOCK.lock().unwrap(); + unsafe { + env::remove_var("FUNCTION_TARGET"); + env::set_var("FUNCTION_NAME", "my-fn"); + env::remove_var("FUNCTION_REGION"); + env::remove_var("GOOGLE_CLOUD_REGION"); + env::set_var("REGION_NAME", "us-central1"); + env::set_var("GCP_PROJECT", "my-project"); + } + + let (id, name) = build_cloud_function_identity(); + + assert_eq!(name, "my-fn"); + assert!(id.contains("us-central1"), "expected region in id: {id}"); + + unsafe { + env::remove_var("FUNCTION_NAME"); + env::remove_var("REGION_NAME"); + env::remove_var("GCP_PROJECT"); + } + } + + #[test] + fn cloud_function_gen2_with_function_target_skipped() { + // Gen2 Cloud Run Functions: FUNCTION_TARGET set → must not write to compat table. + let _lock = ENV_LOCK.lock().unwrap(); + unsafe { + env::set_var("K_SERVICE", "my-service"); + env::set_var("FUNCTION_TARGET", "my-handler"); + env::set_var("REGION_NAME", "us-central1"); + env::set_var("GCP_PROJECT", "my-project"); + } + + let (id, name) = build_cloud_function_identity(); + + assert!(id.is_empty(), "Gen2 must produce empty resource_id; got: {id}"); + assert!(name.is_empty(), "Gen2 must produce empty resource_name; got: {name}"); + + unsafe { + env::remove_var("K_SERVICE"); + env::remove_var("FUNCTION_TARGET"); + env::remove_var("REGION_NAME"); + env::remove_var("GCP_PROJECT"); + } + } + + #[test] + fn cloud_function_missing_name_skipped() { + let _lock = ENV_LOCK.lock().unwrap(); + unsafe { + env::remove_var("FUNCTION_TARGET"); + env::remove_var("FUNCTION_NAME"); + env::remove_var("K_SERVICE"); + env::set_var("FUNCTION_REGION", "us-central1"); + env::set_var("GCP_PROJECT", "my-project"); + } + + let (id, name) = build_cloud_function_identity(); + + assert!(id.is_empty()); + assert!(name.is_empty()); + + unsafe { + env::remove_var("FUNCTION_REGION"); + env::remove_var("GCP_PROJECT"); + } + } + + #[test] + fn cloud_function_missing_project_returns_empty_id() { + // Name present but project/region absent — resource_id empty pending metadata server. + let _lock = ENV_LOCK.lock().unwrap(); + unsafe { + env::remove_var("FUNCTION_TARGET"); + env::set_var("FUNCTION_NAME", "my-fn"); + env::remove_var("FUNCTION_REGION"); + env::remove_var("GOOGLE_CLOUD_REGION"); + env::remove_var("REGION_NAME"); + env::remove_var("GCP_PROJECT"); + env::remove_var("GCLOUD_PROJECT"); + env::remove_var("GOOGLE_CLOUD_PROJECT"); + } + + let (id, name) = build_cloud_function_identity(); + + assert!(id.is_empty(), "incomplete identity must produce empty resource_id"); + assert_eq!(name, "my-fn", "resource_name should still be set for metadata retry"); + + unsafe { + env::remove_var("FUNCTION_NAME"); + } + } + + // ── Payload structure ──────────────────────────────────────────────────── + + #[test] + fn payload_structure_azure_function() { + let _lock = ENV_LOCK.lock().unwrap(); + unsafe { + env::remove_var("DD_SERVERLESS_COMPAT_VERSION"); + } + + let process_id = "test-uuid-1234"; + let body = build_payload( + process_id, + "azure_function", + "startup", + "//microsoft.azure/functionApps/sub/rg/my-func", + "my-func", + &EnvironmentType::AzureFunction, + ) + .expect("build_payload must not fail"); + + let payload: serde_json::Value = serde_json::from_slice(&body).unwrap(); + let obj = payload.as_object().unwrap(); + + // Top-level structure. + assert_eq!(obj["uuid"], process_id); + assert!(obj["timestamp"].as_i64().unwrap() > 1_000_000_000_000_000_000_i64); + assert!(!obj.contains_key("hostname"), "hostname must be absent"); + + let meta = obj["agent_metadata"].as_object().unwrap(); + assert_eq!(meta["flavor"], "serverless-compat"); + assert_eq!(meta["workload_type"], "azure_function"); + assert_eq!(meta["report_reason"], "startup"); + assert_eq!(meta["resource_id"], "//microsoft.azure/functionApps/sub/rg/my-func"); + assert_eq!(meta["resource_name"], "my-func"); + assert!(meta.contains_key("serverless_compat_version")); + + // UUID must NOT appear inside agent_metadata. + assert!(!meta.contains_key("uuid"), "uuid must not be inside agent_metadata"); + + // platform_version must not appear — not in the REDAPL schema. + assert!(!meta.contains_key("platform_version")); + } + + #[test] + fn payload_uses_dd_serverless_compat_version_env() { + let _lock = ENV_LOCK.lock().unwrap(); + unsafe { + env::set_var("DD_SERVERLESS_COMPAT_VERSION", "3.7.1"); + } + + let body = build_payload( + "pid", + "azure_function", + "startup", + "//microsoft.azure/functionApps/s/r/f", + "f", + &EnvironmentType::AzureFunction, + ) + .unwrap(); + let payload: serde_json::Value = serde_json::from_slice(&body).unwrap(); + assert_eq!(payload["agent_metadata"]["serverless_compat_version"], "3.7.1"); + + unsafe { + env::remove_var("DD_SERVERLESS_COMPAT_VERSION"); + } + } + + #[test] + fn payload_falls_back_to_crate_version() { + let _lock = ENV_LOCK.lock().unwrap(); + unsafe { + env::remove_var("DD_SERVERLESS_COMPAT_VERSION"); + } + + let body = build_payload( + "pid", + "azure_function", + "startup", + "//microsoft.azure/functionApps/s/r/f", + "f", + &EnvironmentType::AzureFunction, + ) + .unwrap(); + let payload: serde_json::Value = serde_json::from_slice(&body).unwrap(); + let ver = payload["agent_metadata"]["serverless_compat_version"] + .as_str() + .unwrap(); + assert_eq!(ver, env!("CARGO_PKG_VERSION")); + } + + #[test] + fn periodic_report_reason_in_payload() { + let _lock = ENV_LOCK.lock().unwrap(); + unsafe { + env::remove_var("DD_SERVERLESS_COMPAT_VERSION"); + } + + let body = build_payload( + "pid", + "cloud_function", + "periodic", + "//cloudfunctions.googleapis.com/projects/p/locations/r/functions/fn", + "fn", + &EnvironmentType::CloudFunction, + ) + .unwrap(); + let payload: serde_json::Value = serde_json::from_slice(&body).unwrap(); + assert_eq!(payload["agent_metadata"]["report_reason"], "periodic"); + } + + // ── RG parsing ─────────────────────────────────────────────────────────── + + #[test] + fn parse_rg_linux_suffix() { + let rg = parse_rg_from_owner_name("sub+my-rg-eastuswebspace-Linux"); + assert_eq!(rg.as_deref(), Some("my-rg")); + } + + #[test] + fn parse_rg_windows_suffix() { + let rg = parse_rg_from_owner_name("sub+my-rg-westus2webspace-Windows"); + assert_eq!(rg.as_deref(), Some("my-rg")); + } + + #[test] + fn parse_rg_no_os_suffix() { + let rg = parse_rg_from_owner_name("sub+my-rg-eastuswebspace"); + assert_eq!(rg.as_deref(), Some("my-rg")); + } + + #[test] + fn parse_rg_missing_plus_returns_none() { + assert!(parse_rg_from_owner_name("noplushere").is_none()); + } +} diff --git a/crates/datadog-serverless-compat/src/main.rs b/crates/datadog-serverless-compat/src/main.rs index 69bfb9d..dae6432 100644 --- a/crates/datadog-serverless-compat/src/main.rs +++ b/crates/datadog-serverless-compat/src/main.rs @@ -28,6 +28,8 @@ use datadog_metrics_collector::azure_cpu::CpuMetricsCollector; use libdd_trace_utils::{config_utils::read_cloud_env, trace_utils::EnvironmentType}; +mod inventory; + use datadog_fips::reqwest_adapter::create_reqwest_client_builder; use datadog_logs_agent::{ AggregatorHandle as LogAggregatorHandle, AggregatorService as LogAggregatorService, @@ -165,6 +167,25 @@ pub async fn main() { debug!("Logging subsystem enabled"); + // Launch the inventory reporter. Runs independently of the trace/metrics + // path — a failure here never blocks agent startup or request handling. + if let Some(api_key) = dd_api_key.clone() { + let dd_site_inv = dd_site.clone(); + let https_proxy_inv = https_proxy.clone(); + let env_type_inv = env_type.clone(); + tokio::spawn(async move { + inventory::run_inventory_reporter( + &api_key, + &dd_site_inv, + https_proxy_inv.as_deref(), + env_type_inv, + ) + .await; + }); + } else { + warn!("DD_API_KEY not set, skipping inventory reporter"); + } + let env_verifier = Arc::new(env_verifier::ServerlessEnvVerifier::default()); let config = match config::Config::new() { From 5e2b730b185ac83f6035cd8e611e2d7304f2dc0e Mon Sep 17 00:00:00 2001 From: Nina Rei Date: Wed, 2 Sep 2026 13:25:51 -0400 Subject: [PATCH 2/9] fix(compat): rollout gate, ARM resource_id format, and gate tests - Add DD_SERVERLESS_COMPAT_INVENTORY_ENABLED gate (default off); extracted to is_inventory_enabled() so it can be unit-tested without async runtime - Fix Azure Function resource_id to lowercase ARM format: /subscriptions/{sub}/resourcegroups/{rg}/providers/microsoft.web/sites/{name} (was //microsoft.azure/functionApps/... which doesn't match crawler keys) - Add unit tests: gate off by default, on when "true", off for any other value --- .../src/inventory.rs | 44 +- scripts/svls9604/.gitignore | 2 + scripts/svls9604/README.md | 220 +++++ scripts/svls9604/azure/container-app.bicep | 89 ++ scripts/svls9604/azure/function.bicep | 60 ++ scripts/svls9604/azure/web-app-code.bicep | 80 ++ .../svls9604/azure/web-app-container.bicep | 84 ++ scripts/svls9604/fixtures/Dockerfile | 111 +++ scripts/svls9604/matrix.json | 34 + scripts/svls9604/report.py | 345 +++++++ scripts/svls9604/run.sh | 18 + scripts/svls9604/runner.py | 930 ++++++++++++++++++ 12 files changed, 2013 insertions(+), 4 deletions(-) create mode 100644 scripts/svls9604/.gitignore create mode 100644 scripts/svls9604/README.md create mode 100644 scripts/svls9604/azure/container-app.bicep create mode 100644 scripts/svls9604/azure/function.bicep create mode 100644 scripts/svls9604/azure/web-app-code.bicep create mode 100644 scripts/svls9604/azure/web-app-container.bicep create mode 100644 scripts/svls9604/fixtures/Dockerfile create mode 100644 scripts/svls9604/matrix.json create mode 100644 scripts/svls9604/report.py create mode 100755 scripts/svls9604/run.sh create mode 100755 scripts/svls9604/runner.py diff --git a/crates/datadog-serverless-compat/src/inventory.rs b/crates/datadog-serverless-compat/src/inventory.rs index 725ff04..e6a6011 100644 --- a/crates/datadog-serverless-compat/src/inventory.rs +++ b/crates/datadog-serverless-compat/src/inventory.rs @@ -35,12 +35,23 @@ fn supported_workload_type(env_type: &EnvironmentType) -> Option<&'static str> { /// Sends a startup report immediately, then a periodic report every /// [`INVENTORY_INTERVAL`]. Spawned as a background task — never panics, /// never blocks agent startup. +/// Returns true only when `DD_SERVERLESS_COMPAT_INVENTORY_ENABLED=true`. +/// Extracted so the gate logic can be unit-tested without an async runtime. +fn is_inventory_enabled() -> bool { + env::var("DD_SERVERLESS_COMPAT_INVENTORY_ENABLED").as_deref() == Ok("true") +} + pub async fn run_inventory_reporter( api_key: &str, dd_site: &str, https_proxy: Option<&str>, env_type: EnvironmentType, ) { + if !is_inventory_enabled() { + // Rollout gate: inventory is opt-in during the ramp. Default is off. + return; + } + let Some(workload_type) = supported_workload_type(&env_type) else { // Unsupported workload: skip silently. return; @@ -323,9 +334,9 @@ fn build_azure_function_identity() -> (String, String) { } let resource_id = format!( - "//microsoft.azure/functionApps/{}/{}/{}", - sub, - rg, + "/subscriptions/{}/resourcegroups/{}/providers/microsoft.web/sites/{}", + sub.to_lowercase(), + rg.to_lowercase(), name.to_lowercase() ); (resource_id, name) @@ -607,7 +618,7 @@ mod tests { let (id, name) = build_azure_function_identity(); assert_eq!(name, "my-func-app"); - assert_eq!(id, "//microsoft.azure/functionApps/abc123/my-rg/my-func-app"); + assert_eq!(id, "/subscriptions/abc123/resourcegroups/my-rg/providers/microsoft.web/sites/my-func-app"); unsafe { env::remove_var("WEBSITE_SITE_NAME"); @@ -912,4 +923,29 @@ mod tests { fn parse_rg_missing_plus_returns_none() { assert!(parse_rg_from_owner_name("noplushere").is_none()); } + + // ── Inventory gate ─────────────────────────────────────────────────────── + + #[test] + fn gate_off_by_default() { + let _lock = ENV_LOCK.lock().unwrap(); + unsafe { env::remove_var("DD_SERVERLESS_COMPAT_INVENTORY_ENABLED"); } + assert!(!is_inventory_enabled(), "gate must be off when env var is absent"); + } + + #[test] + fn gate_on_when_set_to_true() { + let _lock = ENV_LOCK.lock().unwrap(); + unsafe { env::set_var("DD_SERVERLESS_COMPAT_INVENTORY_ENABLED", "true"); } + assert!(is_inventory_enabled(), "gate must be on when env var is 'true'"); + unsafe { env::remove_var("DD_SERVERLESS_COMPAT_INVENTORY_ENABLED"); } + } + + #[test] + fn gate_off_when_set_to_other_value() { + let _lock = ENV_LOCK.lock().unwrap(); + unsafe { env::set_var("DD_SERVERLESS_COMPAT_INVENTORY_ENABLED", "false"); } + assert!(!is_inventory_enabled(), "gate must be off when env var is not 'true'"); + unsafe { env::remove_var("DD_SERVERLESS_COMPAT_INVENTORY_ENABLED"); } + } } diff --git a/scripts/svls9604/.gitignore b/scripts/svls9604/.gitignore new file mode 100644 index 0000000..960b43b --- /dev/null +++ b/scripts/svls9604/.gitignore @@ -0,0 +1,2 @@ +__pycache__/ +datadog-serverless-compat-*.tgz diff --git a/scripts/svls9604/README.md b/scripts/svls9604/README.md new file mode 100644 index 0000000..3580bdf --- /dev/null +++ b/scripts/svls9604/README.md @@ -0,0 +1,220 @@ +# SVLS-9604 independent workload runner + +This runner builds the candidate `serverless-init` binary from the local +`datadog-agent` checkout and the Compat package from the local +`serverless-components` and `datadog-serverless-compat-js` checkouts. It does +not invoke either self-monitoring repository. + +Authentication is always supplied internally through: + +```sh +dd-auth --site=datad0g.com --org-uuid=2 +``` + +Preview either exact resource matrix without creating resources: + +```sh +./scripts/svls9604/run.sh --profile gcp --plan +./scripts/svls9604/run.sh --profile azure --plan +``` + +Execute a profile: + +```sh +./scripts/svls9604/run.sh --profile gcp --yes +./scripts/svls9604/run.sh --profile azure --yes +``` + +Rerunning the Azure command with the same `--run-id` and `RESULTS_DIR` resumes +the partial manifest: resources already deployed with the same candidate Agent +image are reused, and only missing resources are deployed. This avoids +repeating the multi-hour portion of an interrupted Azure run. + +These commands run the full automated suite by default: + +- L0: one baseline request or execution per deployed resource +- L1: 10 concurrent requests per HTTP init resource +- L2: 50 concurrent requests per HTTP init resource +- L3: 100 concurrent requests per HTTP init resource +- L4: one concurrent request per distinct HTTP resource +- L5: 15 minutes of unchanged traffic at one-minute intervals +- L6: create a fresh representative revision with request concurrency set to 1, + immediately send 100 concurrent requests, and record the new revision CCRID +- L7: keep revisions A and B active, configure a 10/90 traffic split, send 100 + requests through the service URL, and probe each revision directly + +Add `--scaling-matrix` to create additional representative revisions for: + +- L8: minimum instances/replicas `0`, `5`, and `100`, with concurrency `1` +- L9: maximum instances/replicas `100`, `1000`, and `4000`, with concurrency + `1` and 100-request pressure per configured maximum + +L8 verifies whether each provider-started process sends startup inventory. L9 +verifies that scale-out reports keep one row per revision; 100 requests do not +prove that the service reached a configured maximum above 100. Use provider +instance-start metrics and producer startup logs for the actual instance count. +The scaling matrix is opt-in because minimum `100` can create material cloud +cost, and a provider may reject a requested maximum that exceeds the project, +subscription, region, or environment quota. + +```sh +./scripts/svls9604/run.sh --profile gcp --scaling-matrix --yes +./scripts/svls9604/run.sh --profile azure --scaling-matrix --yes +``` + +Use `--suite baseline` for deployment debugging without the load stages. + +L6 is cold-start *pressure*, not an inferred cold-start count. The authoritative +instance/start count must come from provider or Agent startup logs for revision +B. HTTP attempts and successful responses are reported separately. + +The GCP profile creates 44 resources: 14 Cloud Run in-container services, 14 +Cloud Run sidecars, one Cloud Run job, 14 real Gen2 Cloud Functions with a +sidecar attached to the backing Cloud Run service, and one Gen1 Compat +function. + +The Azure profile creates 63 resources: 14 Container Apps in-container, 14 +Container Apps sidecar, 14 App Service container in-container, 14 App Service +container sidecar, six App Service Linux-code sidecars, and one Node.js Compat +Function App. + +Azure defaults target the existing test infrastructure: + +- Azure Functions: resource group `dd-serverless-test-aca` (the App Service test + resource group does not permit Linux Dynamic workers) + +- Container Apps: `dd-serverless-test-aca/dd-serverless-env` +- ACR: `ddsvlstestaca.azurecr.io` +- App Service: `dd-serverless-test-aas` and its container, sidecar, and + Linux-code plans + +Every default is overridable through the corresponding CLI option or +environment variable. Each run writes a mode-`0700` evidence directory under +`/tmp/svls9604-RUN_ID` with a machine-readable deployment manifest. Generated +parameter files are mode `0600` because they contain temporary secure values. + +The runner sets `DD_ENV=svls9604-RUN_ID` and records `started_at`, `updated_at`, +and `completed_at` in the manifest. This makes a run isolatable in EPRW metrics +without using high-cardinality `resource_id` metric tags. + +Current automated scope is deployment inventory, endpoint/job baseline +execution, L1-L7 load/revision stages, revision-aware expected identities, and +report generation. An HTTP response alone is not proof that inventory traversed +EPRW or deduplicated in Iris. + +The runner always rebuilds the Serverless Compat binary before packaging it. +This prevents a stale `target/` artifact from being deployed under a new run ID. + +Each run produces `serverless-redapl-rc-results.md` and `report.json`. To add +table and pipeline evidence after the run, place these files in the evidence +directory and rerun `report.py`: + +- `serverless_init_agent.csv` +- `serverless_compat_agent.csv` +- `pipeline-evidence.json` + +```sh +python3 scripts/svls9604/report.py --manifest /path/to/run-manifest.json +``` + +For GCP, export Cloud Run logs containing `inventory report queued` as JSON, +then build the event-level process/instance ledger and stage summary: + +```sh +python3 scripts/svls9604/collect_producer_evidence.py \ + --manifest /path/to/run-manifest.json \ + --gcp-logs /path/to/producer-events.json +``` + +This writes `producer-instance-ledger.csv`, `producer-stage-summary.csv`, and +`producer-evidence.json`. The ledger preserves timestamp, stage, service, +revision, provider instance ID, Agent process ID, report reason, and resource +ID. It does not infer downstream EPRW or Iris outcomes. + +`pipeline-evidence.json` supports overall, per-stage, and per-revision counts: + +```json +{ + "eprw_commit": "deployed-sha", + "iris_commit": "deployed-sha", + "eprw_debug_tracking": true, + "iris_upsert_telemetry": true, + "producer_attempts": 0, + "producer_reasons": {"startup": 0, "periodic": 0, "refresh": 0}, + "decoder_accepts": 0, + "decoder_reasons": {"startup": 0, "periodic": 0, "refresh": 0}, + "resource_edge_successes": 0, + "resource_edge_failures": 0, + "iris_primary": {"CREATED": 0, "UPDATED": 0, "EXTENDED": 0, "IGNORED": 0, "ERROR": 0}, + "stages": { + "L6": {"producer_attempts": 0, "decoder_accepts": 0, "resource_edge_successes": 0, "resource_edge_failures": 0} + }, + "resources": { + "REVISION_CCRID": { + "producer_attempts": 0, + "decoder_accepts": 0, + "resource_edge_successes": 0, + "iris_primary": {"CREATED": 0, "UPDATED": 0, "EXTENDED": 0, "IGNORED": 0, "ERROR": 0} + } + } +} +``` + +Replace every example zero with the observed value. Omit measurements that +were not collected; the report treats missing values as `NOT MEASURED` and +never turns them into a pass. + +## Pipeline evidence for an RFC run + +Before the run, temporarily enable full EPRW debug tracking for org 2 and the +two resource types: + +```ini +[dd.event_platform_resource_writer.debug_tracking.track_type_resource_type:agentmetadata:serverless_init_agent:2] +sampling_rate = 1 + +[dd.event_platform_resource_writer.debug_tracking.track_type_resource_type:agentmetadata:serverless_compat_agent:2] +sampling_rate = 1 +``` + +Also enable the Iris experiment `serverless-inventory-upsert-telemetry`. It +emits one structured log per serverless upsert with `resource_id`, hashed +resource identity, result, origin Kafka partition/offset, and `shadow_mode`. +Disable these temporary controls after the validation window. + +Collect these counts for the manifest's exact time window: + +1. Producer reports: Cloud logs containing `inventory report queued`, grouped + by `reason` (`startup`, `periodic`, or `refresh`). Compat currently emits + only `Inventory payload sent (report_reason=startup`. Group by cloud + resource and revision. +2. Decoder accepts, grouped by `report_reason`: + `event_platform_resource_writer.agentmetadata.serverless_write.accepted` + filtered by the manifest's unique `dd_env`. +3. Resource-edge responses: + `event_platform_resource_writer.agentmetadata.serverless_write.resource_edge_response` + filtered by `dd_env`, grouped by `outcome`. +4. Iris outcomes: feature-gated `serverless inventory upsert result` logs, + restricted to `shadow_mode:false` and the manifest resource names, grouped + by `resource_id` and `result`. +5. Final rows: DDSQL rows from `udm.all.serverless_init_agent` and + `udm.all.serverless_compat_agent` filtered by the manifest's exact `dd_env`. + For Cloud Run and Azure Container Apps, compare revision `resource_id` and + stable `parent_resource_id` against every identity in the manifest. + +The hard gates are: + +```text +decoder accepts = resource-edge ok + resource-edge failures +resource-edge ok = primary Iris CREATED + UPDATED + EXTENDED + IGNORED + ERROR +unique DDSQL resource_id count = manifest expected_resource_ids count +duplicate DDSQL keys = 0 +``` + +Interpret Iris results per revision-scoped `resource_id`: `CREATED` is the first row, +`UPDATED` is a real configuration change, `EXTENDED` is an unchanged report +deduplicated to the existing row, and `IGNORED` is stale/out-of-order. A cold +start or new process UUID may increase report attempts, but must not increase +row cardinality within one revision. Creating revision B must create a second +row related to the same stable parent; instances of revision B must deduplicate +into that row. Use `--skip-burst` only for deployment debugging. diff --git a/scripts/svls9604/azure/container-app.bicep b/scripts/svls9604/azure/container-app.bicep new file mode 100644 index 0000000..38a3a66 --- /dev/null +++ b/scripts/svls9604/azure/container-app.bicep @@ -0,0 +1,89 @@ +targetScope = 'resourceGroup' + +param name string +param appEnvId string +param appImage string +param agentImage string +param registryServer string +param registryUsername string +@secure() +param registryPassword string +@secure() +param ddApiKey string +param ddSite string = 'datad0g.com' +param runtime string +param sidecar bool +param minReplicas int +param runId string +param location string = resourceGroup().location + +var commonEnv = [ + { name: 'DD_API_KEY', value: ddApiKey } + { name: 'DD_SITE', value: ddSite } + { name: 'DD_ENV', value: 'svls9604-${runId}' } + { name: 'DD_SERVICE', value: name } + { name: 'DD_SERVERLESS_DIAGNOSTIC_INFO', value: 'true' } + { name: 'DD_LOG_LEVEL', value: 'debug' } + { name: 'DD_AZURE_SUBSCRIPTION_ID', value: subscription().subscriptionId } + { name: 'DD_AZURE_RESOURCE_GROUP', value: resourceGroup().name } +] + +resource app 'Microsoft.App/containerApps@2024-03-01' = { + name: name + location: location + tags: { + svls9604: 'true' + 'svls9604-run': runId + runtime: runtime + 'deployment-model': sidecar ? 'sidecar' : 'in-container' + } + properties: { + managedEnvironmentId: appEnvId + configuration: { + activeRevisionsMode: 'Multiple' + ingress: { + external: true + targetPort: 8080 + transport: 'auto' + } + registries: [ + { + server: registryServer + username: registryUsername + passwordSecretRef: 'registry-password' + } + ] + secrets: [ + { name: 'registry-password', value: registryPassword } + ] + } + template: { + containers: concat([ + { + name: 'app' + image: appImage + resources: { cpu: json('0.5'), memory: '1Gi' } + env: sidecar ? [ + { name: 'DD_ENV', value: 'svls9604-${runId}' } + { name: 'DD_SERVICE', value: name } + ] : commonEnv + } + ], sidecar ? [ + { + name: 'datadog-sidecar' + image: agentImage + resources: { cpu: json('0.5'), memory: '1Gi' } + env: commonEnv + } + ] : []) + scale: { + minReplicas: minReplicas + maxReplicas: 100 + } + } + } +} + +output fqdn string = app.properties.configuration.ingress.fqdn +output resourceId string = app.id +output latestRevisionName string = app.properties.latestRevisionName diff --git a/scripts/svls9604/azure/function.bicep b/scripts/svls9604/azure/function.bicep new file mode 100644 index 0000000..b16ce7f --- /dev/null +++ b/scripts/svls9604/azure/function.bicep @@ -0,0 +1,60 @@ +targetScope = 'resourceGroup' + +param name string +param storageName string +@secure() +param ddApiKey string +param ddSite string = 'datad0g.com' +param runId string +param location string = resourceGroup().location + +resource storage 'Microsoft.Storage/storageAccounts@2023-05-01' = { + name: storageName + location: location + tags: { svls9604: 'true', 'svls9604-run': runId } + sku: { name: 'Standard_LRS' } + kind: 'StorageV2' + properties: { minimumTlsVersion: 'TLS1_2' } +} + +resource plan 'Microsoft.Web/serverfarms@2024-11-01' = { + name: '${name}-plan' + location: location + kind: 'functionapp' + sku: { name: 'Y1', tier: 'Dynamic' } + properties: { reserved: true } +} + +resource app 'Microsoft.Web/sites@2024-11-01' = { + name: name + location: location + kind: 'functionapp,linux' + tags: { + svls9604: 'true' + 'svls9604-run': runId + runtime: 'node' + 'deployment-model': 'compat' + } + properties: { + httpsOnly: true + serverFarmId: plan.id + siteConfig: { + linuxFxVersion: 'NODE|20' + appSettings: [ + { name: 'AzureWebJobsStorage', value: 'DefaultEndpointsProtocol=https;AccountName=${storage.name};AccountKey=${storage.listKeys().keys[0].value};EndpointSuffix=${environment().suffixes.storage}' } + { name: 'FUNCTIONS_EXTENSION_VERSION', value: '~4' } + { name: 'FUNCTIONS_WORKER_RUNTIME', value: 'node' } + { name: 'WEBSITE_NODE_DEFAULT_VERSION', value: '~20' } + { name: 'SCM_DO_BUILD_DURING_DEPLOYMENT', value: 'true' } + { name: 'ENABLE_ORYX_BUILD', value: 'true' } + { name: 'DD_API_KEY', value: ddApiKey } + { name: 'DD_SITE', value: ddSite } + { name: 'DD_ENV', value: 'svls9604-${runId}' } + { name: 'DD_SERVICE', value: name } + ] + } + } +} + +output hostname string = app.properties.defaultHostName +output resourceId string = app.id diff --git a/scripts/svls9604/azure/web-app-code.bicep b/scripts/svls9604/azure/web-app-code.bicep new file mode 100644 index 0000000..c66e58e --- /dev/null +++ b/scripts/svls9604/azure/web-app-code.bicep @@ -0,0 +1,80 @@ +targetScope = 'resourceGroup' + +param name string +param servicePlanId string +param agentImage string +param registryServer string +param registryUsername string +@secure() +param registryPassword string +@secure() +param ddApiKey string +param ddSite string = 'datad0g.com' +param runtime 'node' | 'dotnet' | 'python' +param alwaysOn bool +param runId string +param location string = resourceGroup().location + +var fxVersions = { + node: 'NODE|22-lts' + dotnet: 'DOTNETCORE|8.0' + python: 'PYTHON|3.12' +} +var commands = { + node: 'npm start' + dotnet: 'dotnet app.dll' + python: 'python app.py' +} +var settings = [ + { name: 'SCM_DO_BUILD_DURING_DEPLOYMENT', value: 'true' } + { name: 'DD_API_KEY', value: ddApiKey } + { name: 'DD_SITE', value: ddSite } + { name: 'DD_ENV', value: 'svls9604-${runId}' } + { name: 'DD_SERVICE', value: name } + { name: 'DD_SERVERLESS_DIAGNOSTIC_INFO', value: 'true' } + { name: 'DD_LOG_LEVEL', value: 'debug' } + { name: 'DD_AZURE_SUBSCRIPTION_ID', value: subscription().subscriptionId } + { name: 'DD_AZURE_RESOURCE_GROUP', value: resourceGroup().name } + { name: 'DOCKER_REGISTRY_SERVER_URL', value: 'https://${registryServer}' } + { name: 'DOCKER_REGISTRY_SERVER_USERNAME', value: registryUsername } + { name: 'DOCKER_REGISTRY_SERVER_PASSWORD', value: registryPassword } +] + +resource app 'Microsoft.Web/sites@2024-11-01' = { + name: name + location: location + kind: 'app,linux' + tags: { + svls9604: 'true' + 'svls9604-run': runId + runtime: runtime + 'deployment-model': 'sidecar-code' + } + properties: { + httpsOnly: true + serverFarmId: servicePlanId + siteConfig: { + linuxFxVersion: fxVersions[runtime] + appCommandLine: commands[runtime] + alwaysOn: alwaysOn + appSettings: settings + } + } +} + +resource agent 'Microsoft.Web/sites/sitecontainers@2024-11-01' = { + parent: app + name: 'datadog-sidecar' + properties: { + isMain: false + image: agentImage + targetPort: '8126' + authType: 'UserCredentials' + userName: registryUsername + passwordSecret: registryPassword + inheritAppSettingsAndConnectionStrings: true + } +} + +output hostname string = app.properties.defaultHostName +output resourceId string = app.id diff --git a/scripts/svls9604/azure/web-app-container.bicep b/scripts/svls9604/azure/web-app-container.bicep new file mode 100644 index 0000000..f829041 --- /dev/null +++ b/scripts/svls9604/azure/web-app-container.bicep @@ -0,0 +1,84 @@ +targetScope = 'resourceGroup' + +param name string +param servicePlanId string +param appImage string +param agentImage string +param registryServer string +param registryUsername string +@secure() +param registryPassword string +@secure() +param ddApiKey string +param ddSite string = 'datad0g.com' +param runtime string +param sidecar bool +param alwaysOn bool +param runId string +param location string = resourceGroup().location + +var settings = [ + { name: 'DD_API_KEY', value: ddApiKey } + { name: 'DD_SITE', value: ddSite } + { name: 'DD_ENV', value: 'svls9604-${runId}' } + { name: 'DD_SERVICE', value: name } + { name: 'DD_SERVERLESS_DIAGNOSTIC_INFO', value: 'true' } + { name: 'DD_LOG_LEVEL', value: 'debug' } + { name: 'DD_AZURE_SUBSCRIPTION_ID', value: subscription().subscriptionId } + { name: 'DD_AZURE_RESOURCE_GROUP', value: resourceGroup().name } + { name: 'DOCKER_REGISTRY_SERVER_URL', value: 'https://${registryServer}' } + { name: 'DOCKER_REGISTRY_SERVER_USERNAME', value: registryUsername } + { name: 'DOCKER_REGISTRY_SERVER_PASSWORD', value: registryPassword } + { name: 'WEBSITES_PORT', value: '8080' } +] + +resource app 'Microsoft.Web/sites@2024-11-01' = { + name: name + location: location + kind: 'app,linux,container' + tags: { + svls9604: 'true' + 'svls9604-run': runId + runtime: runtime + 'deployment-model': sidecar ? 'sidecar' : 'in-container' + } + properties: { + httpsOnly: true + serverFarmId: servicePlanId + siteConfig: { + linuxFxVersion: 'SITECONTAINERS' + alwaysOn: alwaysOn + appSettings: settings + } + } +} + +resource main 'Microsoft.Web/sites/sitecontainers@2024-11-01' = { + parent: app + name: 'main' + properties: { + isMain: true + image: appImage + targetPort: '8080' + authType: 'UserCredentials' + userName: registryUsername + passwordSecret: registryPassword + } +} + +resource agent 'Microsoft.Web/sites/sitecontainers@2024-11-01' = if (sidecar) { + parent: app + name: 'datadog-sidecar' + properties: { + isMain: false + image: agentImage + targetPort: '8126' + authType: 'UserCredentials' + userName: registryUsername + passwordSecret: registryPassword + inheritAppSettingsAndConnectionStrings: true + } +} + +output hostname string = app.properties.defaultHostName +output resourceId string = app.id diff --git a/scripts/svls9604/fixtures/Dockerfile b/scripts/svls9604/fixtures/Dockerfile new file mode 100644 index 0000000..a53a34d --- /dev/null +++ b/scripts/svls9604/fixtures/Dockerfile @@ -0,0 +1,111 @@ +ARG RUNTIME=python +ARG AGENT_IMAGE + +FROM python:3.12-slim AS python-plain +WORKDIR /app +RUN printf '%s\n' \ + 'from http.server import BaseHTTPRequestHandler, HTTPServer' \ + 'class H(BaseHTTPRequestHandler):' \ + ' def do_GET(self):' \ + ' self.send_response(200); self.end_headers(); self.wfile.write(b"Hello World!")' \ + 'HTTPServer(("0.0.0.0", 8080), H).serve_forever()' > app.py +EXPOSE 8080 +CMD ["python", "app.py"] + +FROM node:22-slim AS node-plain +WORKDIR /app +RUN printf '%s\n' \ + "const http=require('http');" \ + "http.createServer((q,r)=>{r.writeHead(200);r.end('Hello World!')}).listen(8080,'0.0.0.0');" > app.js +EXPOSE 8080 +CMD ["node", "app.js"] + +FROM golang:1.24-bookworm AS go-build +WORKDIR /src +RUN printf '%s\n' \ + 'package main' \ + 'import ("fmt"; "net/http")' \ + 'func main(){http.HandleFunc("/",func(w http.ResponseWriter,r *http.Request){fmt.Fprint(w,"Hello World!")});http.ListenAndServe(":8080",nil)}' > main.go \ + && go build -o /app main.go +FROM debian:bookworm-slim AS go-plain +COPY --from=go-build /app /app +EXPOSE 8080 +CMD ["/app"] + +FROM eclipse-temurin:21-jdk-jammy AS java-build +WORKDIR /src +RUN printf '%s\n' \ + 'import com.sun.net.httpserver.HttpServer;' \ + 'import java.net.InetSocketAddress;' \ + 'public class App { public static void main(String[] a) throws Exception {' \ + 'var s=HttpServer.create(new InetSocketAddress(8080),0); s.createContext("/",e->{byte[] b="Hello World!".getBytes();e.sendResponseHeaders(200,b.length);e.getResponseBody().write(b);e.close();});s.start();}}' > App.java \ + && javac App.java +FROM eclipse-temurin:21-jre-jammy AS java-plain +WORKDIR /app +COPY --from=java-build /src/App.class . +EXPOSE 8080 +CMD ["java", "App"] + +FROM mcr.microsoft.com/dotnet/sdk:8.0 AS dotnet-build +WORKDIR /src +RUN printf '%s\n' 'net8.0enable' > app.csproj \ + && printf '%s\n' 'var b=WebApplication.CreateBuilder(args);var a=b.Build();a.MapGet("/",()=>"Hello World!");a.Run("http://0.0.0.0:8080");' > Program.cs \ + && dotnet publish -c Release -o /out +FROM mcr.microsoft.com/dotnet/aspnet:8.0 AS dotnet-plain +WORKDIR /app +COPY --from=dotnet-build /out . +EXPOSE 8080 +CMD ["dotnet", "app.dll"] + +FROM ruby:3.3-slim AS ruby-plain +WORKDIR /app +RUN printf '%s\n' \ + 'require "socket"' \ + 's=TCPServer.new("0.0.0.0",8080)' \ + 'loop{c=s.accept;c.gets;while (l=c.gets);break if l=="\r\n";end;c.write "HTTP/1.1 200 OK\r\nContent-Length: 12\r\n\r\nHello World!";c.close}' > app.rb +EXPOSE 8080 +CMD ["ruby", "app.rb"] + +FROM php:8.4-cli AS php-plain +WORKDIR /app +RUN printf '%s\n' '' > index.php +EXPOSE 8080 +CMD ["php", "-S", "0.0.0.0:8080", "-t", "/app"] + +FROM ${AGENT_IMAGE} AS agent +FROM ${RUNTIME}-plain AS plain + +FROM python-plain AS python-init +COPY --from=agent /serverless-init /serverless-init +ENTRYPOINT ["/serverless-init"] +CMD ["python", "app.py"] + +FROM node-plain AS node-init +COPY --from=agent /serverless-init /serverless-init +ENTRYPOINT ["/serverless-init"] +CMD ["node", "app.js"] + +FROM go-plain AS go-init +COPY --from=agent /serverless-init /serverless-init +ENTRYPOINT ["/serverless-init"] +CMD ["/app"] + +FROM java-plain AS java-init +COPY --from=agent /serverless-init /serverless-init +ENTRYPOINT ["/serverless-init"] +CMD ["java", "App"] + +FROM dotnet-plain AS dotnet-init +COPY --from=agent /serverless-init /serverless-init +ENTRYPOINT ["/serverless-init"] +CMD ["dotnet", "app.dll"] + +FROM ruby-plain AS ruby-init +COPY --from=agent /serverless-init /serverless-init +ENTRYPOINT ["/serverless-init"] +CMD ["ruby", "app.rb"] + +FROM php-plain AS php-init +COPY --from=agent /serverless-init /serverless-init +ENTRYPOINT ["/serverless-init"] +CMD ["php", "-S", "0.0.0.0:8080", "-t", "/app"] diff --git a/scripts/svls9604/matrix.json b/scripts/svls9604/matrix.json new file mode 100644 index 0000000..7dadff4 --- /dev/null +++ b/scripts/svls9604/matrix.json @@ -0,0 +1,34 @@ +{ + "runtimes": ["python", "node", "go", "java", "dotnet", "ruby", "php"], + "profiles": { + "gcp": [ + {"id": "SI-01", "provider": "gcp", "workload_type": "cloud_run_service", "deployment_model": "in-container", "runtimes": "all", "variants": ["busy", "coldstart"]}, + {"id": "SI-02", "provider": "gcp", "workload_type": "cloud_run_service", "deployment_model": "sidecar", "runtimes": "all", "variants": ["busy", "coldstart"]}, + {"id": "SI-03", "provider": "gcp", "workload_type": "cloud_run_job", "deployment_model": "in-container", "runtimes": ["python"], "variants": ["execution"]}, + {"id": "SI-04", "provider": "gcp", "workload_type": "cloud_run_function", "deployment_model": "sidecar", "runtimes": "all", "variants": ["busy", "coldstart"]}, + {"id": "SC-02", "provider": "gcp", "workload_type": "cloud_function", "runtimes": ["node"], "variants": ["function"]} + ], + "gcp-sanity": [ + {"id": "SI-01", "provider": "gcp", "workload_type": "cloud_run_service", "deployment_model": "in-container", "runtimes": ["python"], "variants": ["busy"]}, + {"id": "SI-03", "provider": "gcp", "workload_type": "cloud_run_job", "deployment_model": "in-container", "runtimes": ["python"], "variants": ["execution"]}, + {"id": "SI-04", "provider": "gcp", "workload_type": "cloud_run_function", "deployment_model": "sidecar", "runtimes": ["python"], "variants": ["busy"]}, + {"id": "SC-02", "provider": "gcp", "workload_type": "cloud_function", "runtimes": ["node"], "variants": ["function"]} + ], + "azure-sanity": [ + {"id": "SI-05", "provider": "azure", "workload_type": "azure_container_app", "deployment_model": "in-container", "runtimes": ["python"], "variants": ["busy"]}, + {"id": "SI-06", "provider": "azure", "workload_type": "azure_container_app", "deployment_model": "sidecar", "runtimes": ["python"], "variants": ["busy"]}, + {"id": "SI-07", "provider": "azure", "workload_type": "azure_app_service", "deployment_model": "in-container", "runtimes": ["python"], "variants": ["busy"]}, + {"id": "SI-08", "provider": "azure", "workload_type": "azure_app_service", "deployment_model": "sidecar", "runtimes": ["python"], "variants": ["busy"]}, + {"id": "SI-09", "provider": "azure", "workload_type": "azure_app_service", "deployment_model": "sidecar", "runtimes": ["python"], "variants": ["busy"], "shape": "linux-code"}, + {"id": "SC-01", "provider": "azure", "workload_type": "azure_function", "runtimes": ["node"], "variants": ["function"]} + ], + "azure": [ + {"id": "SI-05", "provider": "azure", "workload_type": "azure_container_app", "deployment_model": "in-container", "runtimes": "all", "variants": ["busy", "coldstart"]}, + {"id": "SI-06", "provider": "azure", "workload_type": "azure_container_app", "deployment_model": "sidecar", "runtimes": "all", "variants": ["busy", "coldstart"]}, + {"id": "SI-07", "provider": "azure", "workload_type": "azure_app_service", "deployment_model": "in-container", "runtimes": "all", "variants": ["busy", "coldstart"]}, + {"id": "SI-08", "provider": "azure", "workload_type": "azure_app_service", "deployment_model": "sidecar", "runtimes": "all", "variants": ["busy", "coldstart"]}, + {"id": "SI-09", "provider": "azure", "workload_type": "azure_app_service", "deployment_model": "sidecar", "runtimes": ["node", "dotnet", "python"], "variants": ["busy", "coldstart"], "shape": "linux-code"}, + {"id": "SC-01", "provider": "azure", "workload_type": "azure_function", "runtimes": ["node"], "variants": ["function"]} + ] + } +} diff --git a/scripts/svls9604/report.py b/scripts/svls9604/report.py new file mode 100644 index 0000000..2120919 --- /dev/null +++ b/scripts/svls9604/report.py @@ -0,0 +1,345 @@ +#!/usr/bin/env python3 +import argparse +import csv +import json +import pathlib + + +NOT_MEASURED = "NOT MEASURED" + +def measured_number(value): + return isinstance(value,(int,float)) and not isinstance(value,bool) + + +def load_csv(path): + if not path or not path.exists(): + return [] + with path.open(newline="") as handle: + return list(csv.DictReader(handle)) + + +def status(ok, measured=True): + if not measured: + return NOT_MEASURED + return "PASS" if ok else "FAIL" + + +def aggregate_stage(stage): + if "resources" in stage and stage["resources"]: + resources=stage["resources"] + attempts=sum(item.get("attempts",1) for item in resources) + successes=sum(item.get("successes",1 if 200 <= item.get("http_status",0) < 400 else 0) + for item in resources) + failures=sum(item.get("failures",0 if 200 <= item.get("http_status",0) < 400 else 1) + for item in resources) + return attempts,successes,failures + if "rounds" in stage: + return (sum(item["attempts"] for item in stage["rounds"]), + sum(item["successes"] for item in stage["rounds"]), + sum(item["failures"] for item in stage["rounds"])) + return stage.get("attempts",0),stage.get("successes",0),stage.get("failures",0) + + +def find_default_csv(run_dir, name): + candidates=[run_dir/f"{name}.csv",run_dir/"pup"/f"{name}.csv"] + return next((path for path in candidates if path.exists()),None) + +def resource_expected_ids(resource): + return resource.get("expected_resource_ids") or [resource["resource_id"]] + +def table_evidence(resources, rows): + expected={resource_id.lower():resource for resource in resources + for resource_id in resource_expected_ids(resource)} + selected=[row for row in rows if row.get("resource_id","").lower() in expected] + ids=[row.get("resource_id","").lower() for row in selected if row.get("resource_id")] + distinct=set(ids) + missing=sorted(key for key in expected if key not in distinct) + unexpected=sorted(row.get("resource_id","") for row in rows + if row.get("resource_id") and row["resource_id"].lower() not in expected) + required=("resource_id","resource_name","workload_type") + nulls={field:sum(1 for row in selected if not row.get(field)) for field in required} + revision_rows=[row for row in selected if row.get("workload_type") in + ("cloud_run_service","cloud_run_function","azure_container_app")] + nulls["parent_resource_id"]=sum(1 for row in revision_rows if not row.get("parent_resource_id")) + nulls["deployment_id"]=sum(1 for row in revision_rows if not row.get("deployment_id")) + return {"rows":len(selected),"distinct_resource_ids":len(distinct),"missing":missing, + "unexpected":unexpected,"duplicates":len(ids)-len(distinct),"required_nulls":nulls} + + +def render(manifest, init_rows, compat_rows, pipeline): + resources=manifest.get("resources",[]) + init_resources=[r for r in resources if r["id"].startswith("SI-")] + compat_resources=[r for r in resources if r["id"].startswith("SC-")] + init=table_evidence(init_resources,init_rows) + compat=table_evidence(compat_resources,compat_rows) + table_measured=bool(init_rows or compat_rows) + acceptance_measured=all(measured_number(pipeline.get(key)) for key in ("producer_attempts","decoder_accepts")) + edge_measured=all(measured_number(pipeline.get(key)) for key in ("decoder_accepts","resource_edge_successes","resource_edge_failures")) + iris=pipeline.get("iris_primary",{}) + iris_measured=all(measured_number(iris.get(outcome)) for outcome in ("CREATED","UPDATED","EXTENDED","IGNORED","ERROR")) + iris_total=sum(iris.values()) if iris_measured else 0 + producer_reasons=pipeline.get("producer_reasons",{}) + decoder_reasons=pipeline.get("decoder_reasons",{}) + reason_names=("startup","periodic","refresh") + reasons_measured=all( + measured_number(producer_reasons.get(reason)) and measured_number(decoder_reasons.get(reason)) + for reason in reason_names + ) + deployed=len(resources) + expected_init_ids=sum(len(resource_expected_ids(r)) for r in init_resources) + expected_compat_ids=sum(len(resource_expected_ids(r)) for r in compat_resources) + revision_stages={stage.get("id"):stage for stage in manifest.get("load_stages",[]) if stage.get("id") in ("L6","L7")} + revision_measured=all(stage_id in revision_stages and revision_stages[stage_id].get("status") != NOT_MEASURED + for stage_id in ("L6","L7")) + stage_pipeline_complete=all( + all(measured_number(pipeline.get("stages",{}).get(stage.get("id"),{}).get(key)) + for key in ("producer_attempts","decoder_accepts","resource_edge_successes","resource_edge_failures")) + for stage in manifest.get("load_stages",[]) + ) + expected_revision_ids={resource_id.lower() for resource in init_resources for resource_id in resource_expected_ids(resource)} + resource_pipeline_complete=all( + resource_id in {key.lower() for key in pipeline.get("resources",{})} + for resource_id in expected_revision_ids + ) and all( + all(measured_number(evidence.get(key)) for key in ("producer_attempts","decoder_accepts","resource_edge_successes")) and + all(measured_number(evidence.get("iris_primary",{}).get(outcome)) for outcome in ("CREATED","UPDATED","EXTENDED","IGNORED","ERROR")) + for evidence in pipeline.get("resources",{}).values() + ) + tables_valid=(init["rows"]==expected_init_ids and compat["rows"]==expected_compat_ids and + init["duplicates"]==0 and compat["duplicates"]==0 and not init["missing"] and + not compat["missing"] and not init["unexpected"] and not compat["unexpected"] and + not any(init["required_nulls"].values()) and not any(compat["required_nulls"].values())) + reason_totals_valid=(reasons_measured and + sum(producer_reasons.values())==pipeline.get("producer_attempts") and + sum(decoder_reasons.values())==pipeline.get("decoder_accepts") and + all(producer_reasons[reason]==decoder_reasons[reason] for reason in reason_names)) + pipeline_valid=(acceptance_measured and edge_measured and iris_measured and reason_totals_valid and + pipeline.get("producer_attempts")==pipeline.get("decoder_accepts") and + pipeline.get("decoder_accepts")==pipeline.get("resource_edge_successes",0)+pipeline.get("resource_edge_failures",0) and + pipeline.get("resource_edge_failures")==0 and pipeline.get("resource_edge_successes")==iris_total) + complete=table_measured and tables_valid and pipeline_valid and revision_measured and stage_pipeline_complete and resource_pipeline_complete + triggered=sum(1 for r in resources if r.get("baseline",{}).get("status") in ("ok","executed")) + lines=[ + "# Serverless Agent REDAPL RC Results", + "", + f"**Status:** {'COMPLETE' if complete else 'PARTIAL — pipeline, revision, or DDSQL evidence still required'}", + "", + f"**Run ID:** `{manifest.get('run_id','')}`", + "", + f"**Environment:** `{manifest.get('dd_env','')}` on `datad0g.com`, org 2", + "", + f"**Start and end time:** {manifest.get('started_at',NOT_MEASURED)} to {manifest.get('completed_at',NOT_MEASURED)}", + "", + f"**Declared scope:** `{manifest.get('profile')}` / `{manifest.get('suite')}`", + "", + "## 1. Build and environment record", + "", + "| Item | Observed value |", + "|---|---|", + f"| Agent image tag and digest | `{manifest.get('agent_image',NOT_MEASURED)}` |", + f"| Datadog Agent commit | `{manifest.get('candidate_commits',{}).get('datadog_agent',manifest.get('agent_sha',NOT_MEASURED))}` |", + f"| Serverless Components commit | `{manifest.get('candidate_commits',{}).get('serverless_components',NOT_MEASURED)}` |", + f"| Compat JS commit | `{manifest.get('candidate_commits',{}).get('datadog_serverless_compat_js',NOT_MEASURED)}` |", + f"| EPRW decoder deployed commit | `{pipeline.get('eprw_commit',NOT_MEASURED)}` |", + f"| Iris deployed commit | `{pipeline.get('iris_commit',NOT_MEASURED)}` |", + f"| EPRW debug tracking | `{pipeline.get('eprw_debug_tracking',NOT_MEASURED)}` |", + f"| Iris upsert experiment | `{pipeline.get('iris_upsert_telemetry',NOT_MEASURED)}` |", + "", + "## 2. Executive results", + "", + "| Result | Expected | Observed | Status |", + "|---|---:|---:|---|", + f"| Resources deployed | {len(resources)} | {deployed} | {status(deployed==len(resources))} |", + f"| Baseline triggers | {deployed} | {triggered} | {status(triggered==deployed)} |", + f"| EPRW decoder accepts | Producer attempts | {pipeline.get('decoder_accepts',NOT_MEASURED)} | {status(pipeline.get('decoder_accepts')==pipeline.get('producer_attempts'),acceptance_measured)} |", + f"| Resource Edge failures | 0 | {pipeline.get('resource_edge_failures',NOT_MEASURED)} | {status(pipeline.get('resource_edge_failures')==0,'resource_edge_failures' in pipeline)} |", + f"| REDAPL init revision/workload rows | {expected_init_ids} | {init['rows'] if init_rows else NOT_MEASURED} | {status(init['rows']==expected_init_ids,bool(init_rows))} |", + f"| REDAPL compat rows | {expected_compat_ids} | {compat['rows'] if compat_rows else NOT_MEASURED} | {status(compat['rows']==expected_compat_ids,bool(compat_rows))} |", + f"| Duplicate resource keys | 0 | {(init['duplicates']+compat['duplicates']) if table_measured else NOT_MEASURED} | {status(init['duplicates']+compat['duplicates']==0,table_measured)} |", + "", + "## 3. Per-workload results", + "", + "| ID | Table | Workload | Runtime | Model | Resource | Baseline | REDAPL |", + "|---|---|---|---|---|---|---|---|", + ] + init_ids={row.get("resource_id","").lower() for row in init_rows} + compat_ids={row.get("resource_id","").lower() for row in compat_rows} + for resource in resources: + table="serverless_init_agent" if resource["id"].startswith("SI-") else "serverless_compat_agent" + ids=init_ids if table=="serverless_init_agent" else compat_ids + observed=status(all(resource_id.lower() in ids for resource_id in resource_expected_ids(resource)),bool(init_rows if table=="serverless_init_agent" else compat_rows)) + lines.append(f"| {resource['id']} | `{table}` | `{resource['workload_type']}` | {resource['runtime']} | {resource['deployment_model']} | `{resource['name']}` | {resource.get('baseline',{}).get('status',NOT_MEASURED)} | {observed} |") + lines.extend([ + "", + "## 4. Producer, EPRW, and Iris reconciliation", + "", + "| Measurement | Observed | Status |", + "|---|---:|---|", + f"| Producer inventory attempts | {pipeline.get('producer_attempts',NOT_MEASURED)} | {status(pipeline.get('producer_attempts')==pipeline.get('decoder_accepts'),acceptance_measured)} |", + f"| EPRW decoder accepts | {pipeline.get('decoder_accepts',NOT_MEASURED)} | {status(pipeline.get('decoder_accepts')==pipeline.get('producer_attempts'),acceptance_measured)} |", + f"| Resource Edge successes | {pipeline.get('resource_edge_successes',NOT_MEASURED)} | {status(pipeline.get('decoder_accepts')==pipeline.get('resource_edge_successes',0)+pipeline.get('resource_edge_failures',0),edge_measured)} |", + f"| Resource Edge failures | {pipeline.get('resource_edge_failures',NOT_MEASURED)} | {status(pipeline.get('resource_edge_failures')==0,'resource_edge_failures' in pipeline)} |", + ]) + for outcome in ("CREATED","UPDATED","EXTENDED","IGNORED","ERROR"): + lines.append(f"| Primary Iris {outcome} | {iris.get(outcome,NOT_MEASURED)} | {status(iris.get(outcome,0)==0 if outcome=='ERROR' else True,outcome in iris)} |") + lines.extend([ + "", + "### Collection reasons", + "", + "| Reason | Producer reports | EPRW accepts | Status |", + "|---|---:|---:|---|", + ]) + for reason in reason_names: + producer=producer_reasons.get(reason,NOT_MEASURED) + decoder=decoder_reasons.get(reason,NOT_MEASURED) + measured=measured_number(producer) and measured_number(decoder) + lines.append(f"| `{reason}` | {producer} | {decoder} | {status(producer==decoder,measured)} |") + lines.extend([ + "", + "Reconciliation gates:", + "", + "```text", + "decoder accepts = Resource Edge successes + Resource Edge failures", + "Resource Edge successes = primary Iris CREATED + UPDATED + EXTENDED + IGNORED + ERROR", + "```", + "", + "## 5. REDAPL identity and data results", + "", + "| Table | Expected IDs | Rows | Distinct IDs | Duplicates | Missing | Unexpected | Required nulls | Status |", + "|---|---:|---:|---:|---:|---:|---:|---:|---|", + f"| `serverless_init_agent` | {expected_init_ids} | {init['rows'] if init_rows else NOT_MEASURED} | {init['distinct_resource_ids'] if init_rows else NOT_MEASURED} | {init['duplicates'] if init_rows else NOT_MEASURED} | {len(init['missing']) if init_rows else NOT_MEASURED} | {len(init['unexpected']) if init_rows else NOT_MEASURED} | {sum(init['required_nulls'].values()) if init_rows else NOT_MEASURED} | {status(init['rows']==expected_init_ids and init['duplicates']==0 and not init['missing'] and not init['unexpected'] and not any(init['required_nulls'].values()),bool(init_rows))} |", + f"| `serverless_compat_agent` | {expected_compat_ids} | {compat['rows'] if compat_rows else NOT_MEASURED} | {compat['distinct_resource_ids'] if compat_rows else NOT_MEASURED} | {compat['duplicates'] if compat_rows else NOT_MEASURED} | {len(compat['missing']) if compat_rows else NOT_MEASURED} | {len(compat['unexpected']) if compat_rows else NOT_MEASURED} | {sum(compat['required_nulls'].values()) if compat_rows else NOT_MEASURED} | {status(compat['rows']==expected_compat_ids and compat['duplicates']==0 and not compat['missing'] and not compat['unexpected'] and not any(compat['required_nulls'].values()),bool(compat_rows))} |", + "", + "## 6. Load-stage results", + "", + "| Stage | Scenario | Sent | Successful | Failed | Pipeline evidence | Status |", + "|---|---|---:|---:|---:|---|---|", + ]) + for stage in manifest.get("load_stages",[]): + attempts,successes,failures=aggregate_stage(stage) + evidence=pipeline.get("stages",{}).get(stage["id"],{}) + stage_pipeline_measured=all(measured_number(evidence.get(key)) for key in ("producer_attempts","decoder_accepts","resource_edge_successes","resource_edge_failures")) + pipeline_summary=(f"producer={evidence['producer_attempts']}, decoder={evidence['decoder_accepts']}, " + f"edge_ok={evidence['resource_edge_successes']}, edge_failed={evidence['resource_edge_failures']}") if stage_pipeline_measured else NOT_MEASURED + if stage.get("status") == NOT_MEASURED: + stage_status=NOT_MEASURED + elif failures: + stage_status="FAIL" + elif not stage_pipeline_measured: + stage_status="PARTIAL" + else: + stage_status=status(evidence["producer_attempts"]==evidence["decoder_accepts"] and + evidence["decoder_accepts"]==evidence["resource_edge_successes"]+evidence["resource_edge_failures"] and + evidence["resource_edge_failures"]==0) + lines.append(f"| {stage['id']} | {stage['scenario']} | {attempts or NOT_MEASURED} | {successes or NOT_MEASURED} | {failures if attempts else NOT_MEASURED} | {pipeline_summary} | {stage_status} |") + lines.extend([ + "", + "### Per-revision reconciliation", + "", + "| Resource ID | Producer starts | Decoder accepts | Edge successes | Iris total | Status |", + "|---|---:|---:|---:|---:|---|", + ]) + for resource_id,evidence in sorted(pipeline.get("resources",{}).items()): + per_iris=evidence.get("iris_primary",{}) + per_iris_measured=all(measured_number(per_iris.get(outcome)) for outcome in ("CREATED","UPDATED","EXTENDED","IGNORED","ERROR")) + per_measured=all(measured_number(evidence.get(key)) for key in ("producer_attempts","decoder_accepts","resource_edge_successes")) and per_iris_measured + per_iris_total=sum(per_iris.values()) if per_iris_measured else NOT_MEASURED + per_status=status(evidence.get("producer_attempts")==evidence.get("decoder_accepts")==evidence.get("resource_edge_successes")==per_iris_total,per_measured) + lines.append(f"| `{resource_id}` | {evidence.get('producer_attempts',NOT_MEASURED)} | {evidence.get('decoder_accepts',NOT_MEASURED)} | {evidence.get('resource_edge_successes',NOT_MEASURED)} | {per_iris_total} | {per_status} |") + if not pipeline.get("resources"): + lines.append(f"| {NOT_MEASURED} | {NOT_MEASURED} | {NOT_MEASURED} | {NOT_MEASURED} | {NOT_MEASURED} | {NOT_MEASURED} |") + lines.extend([ + "", + "## 7. Staging evidence queries", + "", + "Use the run time window shown above and the following filters in `datad0g.com`:", + "", + "```text", + f"Environment: {manifest.get('dd_env','')}", + f"EPRW accepts: sum:event_platform_resource_writer.agentmetadata.serverless_write.accepted{{resource_type:serverless_init_agent}} by {{resource_type,workload_type,deployment_model,report_reason}}", + f"EPRW accepts (compat): sum:event_platform_resource_writer.agentmetadata.serverless_write.accepted{{resource_type:serverless_compat_agent}} by {{resource_type,workload_type,report_reason}}", + f"Producer startup log (Init): env:{manifest.get('dd_env','')} \"serverless-init: inventory report queued\"", + f"Iris primary logs: service:iris-node-go @message:\"serverless inventory upsert result\" @shadow_mode:false @resource_id:*{manifest.get('run_id','').lower()}*", + "EPRW debug logs: service:event-platform-resource-writer @resource_id:*RUN_ID*", + "```", + "", + "For L6, record provider startup/replica counts for `revision_b`; the request total is pressure applied, not the number of cold starts. For L7, confirm both revision resource IDs reach EPRW/Iris and both remain related to the same parent service/app.", + "", + "DDSQL exports:", + "", + "```sql", + "SELECT resource_id, parent_resource_id, resource_name, workload_type, deployment_model, deployment_id, runtime, dd_env, _first_seen_at, _modification_detected_at, _updated_at", + "FROM udm.all.serverless_init_agent", + f"WHERE dd_env = '{manifest.get('dd_env','')}'", + "ORDER BY parent_resource_id, deployment_id, resource_id;", + "", + "SELECT resource_id, resource_name, workload_type, runtime, dd_env, _first_seen_at, _modification_detected_at, _updated_at", + "FROM udm.all.serverless_compat_agent", + f"WHERE dd_env = '{manifest.get('dd_env','')}'", + "ORDER BY resource_id;", + "```", + "", + "## 8. Evidence still required", + "", + "- Export `serverless_init_agent.csv` and `serverless_compat_agent.csv` into this run directory, then rerun `report.py`.", + "- Export run-filtered EPRW and primary Iris counts into `pipeline-evidence.json`.", + "- Add producer and EPRW counts grouped by `report_reason` (`startup`, `periodic`, `refresh`).", + "- Add provider instance-start evidence for sequential cold starts and scale-out.", + "- Run the controlled `SeenAt`, revision A/B, TTL, crawler, Fleet, and UI checks.", + "", + "## 9. RFC approval gates", + "", + "| Gate | Status |", + "|---|---|", + f"| Every in-scope workload/revision reports a valid row | {status(init['rows']==expected_init_ids and compat['rows']==expected_compat_ids,bool(init_rows and compat_rows))} |", + f"| One row per `resource_id` | {status(init['duplicates']==0 and compat['duplicates']==0,table_measured)} |", + f"| EPRW/Iris counts reconcile | {status(edge_measured and iris_measured and pipeline.get('decoder_accepts')==pipeline.get('resource_edge_successes',0)+pipeline.get('resource_edge_failures',0) and pipeline.get('resource_edge_successes')==iris_total,edge_measured and iris_measured)} |", + f"| Producer reasons reconcile with EPRW | {status(all(producer_reasons.get(reason)==decoder_reasons.get(reason) for reason in reason_names),reasons_measured)} |", + f"| Older `SeenAt` cannot replace newer data | {NOT_MEASURED} |", + f"| Revision creation and traffic split executed | {status(revision_measured,revision_measured)} |", + f"| TTL and reactivation | {NOT_MEASURED} |", + f"| Crawler, Fleet, and UI | {NOT_MEASURED} |", + "", + "## 10. Conclusion", + "", + "The generated report distinguishes observed execution results from pipeline and UI evidence that has not yet been collected. Missing evidence is never converted into a pass.", + ]) + return "\n".join(lines)+"\n",{"init":init,"compat":compat} + + +def main(): + parser=argparse.ArgumentParser() + parser.add_argument("--manifest",type=pathlib.Path,required=True) + parser.add_argument("--init-csv",type=pathlib.Path) + parser.add_argument("--compat-csv",type=pathlib.Path) + parser.add_argument("--pipeline-evidence",type=pathlib.Path) + args=parser.parse_args() + run_dir=args.manifest.parent + manifest=json.loads(args.manifest.read_text()) + init_path=args.init_csv or find_default_csv(run_dir,"serverless_init_agent") + compat_path=args.compat_csv or find_default_csv(run_dir,"serverless_compat_agent") + pipeline_path=args.pipeline_evidence or run_dir/"pipeline-evidence.json" + pipeline=json.loads(pipeline_path.read_text()) if pipeline_path.exists() else {} + producer_path=run_dir/"producer-evidence.json" + if producer_path.exists(): + producer=json.loads(producer_path.read_text()) + pipeline.setdefault("producer_attempts",producer.get("events")) + pipeline.setdefault("producer_reasons",producer.get("reasons",{})) + pipeline.setdefault("stages",{}) + for stage_id,evidence in producer.get("stages",{}).items(): + pipeline["stages"].setdefault(stage_id,{}) + pipeline["stages"][stage_id].setdefault("producer_attempts",evidence.get("reports")) + pipeline.setdefault("resources",{}) + for resource_id,evidence in producer.get("resources",{}).items(): + pipeline["resources"].setdefault(resource_id,{}) + pipeline["resources"][resource_id].setdefault("producer_attempts",evidence.get("reports")) + markdown,tables=render(manifest,load_csv(init_path),load_csv(compat_path),pipeline) + (run_dir/"serverless-redapl-rc-results.md").write_text(markdown) + (run_dir/"report.json").write_text(json.dumps({"manifest":manifest,"tables":tables, + "pipeline":pipeline},indent=2)) + print(f"Report: {run_dir/'serverless-redapl-rc-results.md'}") + print(f"Machine-readable report: {run_dir/'report.json'}") + + +if __name__ == "__main__": + main() diff --git a/scripts/svls9604/run.sh b/scripts/svls9604/run.sh new file mode 100755 index 0000000..2bfcb4a --- /dev/null +++ b/scripts/svls9604/run.sh @@ -0,0 +1,18 @@ +#!/usr/bin/env bash +set -euo pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" + +for arg in "$@"; do + if [[ "${arg}" == "--plan" ]]; then + exec python3 "${SCRIPT_DIR}/runner.py" "$@" + fi +done + +if [[ "${SVLS9604_DD_AUTHENTICATED:-}" != "1" ]]; then + exec dd-auth --site=datad0g.com --org-uuid=2 -- env \ + SVLS9604_DD_AUTHENTICATED=1 \ + python3 "${SCRIPT_DIR}/runner.py" "$@" +fi + +exec python3 "${SCRIPT_DIR}/runner.py" "$@" diff --git a/scripts/svls9604/runner.py b/scripts/svls9604/runner.py new file mode 100755 index 0000000..43a76ff --- /dev/null +++ b/scripts/svls9604/runner.py @@ -0,0 +1,930 @@ +#!/usr/bin/env python3 +import argparse +import concurrent.futures +import datetime as dt +import hashlib +import json +import os +import pathlib +import re +import secrets +import shlex +import subprocess +import sys +import tempfile +import time +import urllib.request +import zipfile + +ROOT = pathlib.Path(__file__).resolve().parent +WORKSPACE = ROOT.parents[2] +AGENT_REPO = pathlib.Path(os.environ.get("DATADOG_AGENT_DIR", WORKSPACE / "datadog-agent")) +COMPONENTS_REPO = pathlib.Path(os.environ.get("SERVERLESS_COMPONENTS_DIR", WORKSPACE / "serverless-components")) +COMPAT_JS_REPO = pathlib.Path(os.environ.get("COMPAT_JS_DIR", WORKSPACE / "datadog-serverless-compat-js")) +MATRIX = json.loads((ROOT / "matrix.json").read_text()) +RUN_ENV = "svls9604" +RUN_STARTED_AT = None + +FULL_LOAD_STAGES = (("L1", 10), ("L2", 50), ("L3", 100)) + +def aggregate_stage(stage): + if stage.get("resources"): + return (sum(item.get("attempts",1) for item in stage["resources"]), + sum(item.get("successes",1 if 200 <= item.get("http_status",0) < 400 else 0) for item in stage["resources"]), + sum(item.get("failures",0 if 200 <= item.get("http_status",0) < 400 else 1) for item in stage["resources"])) + if stage.get("rounds"): + return (sum(item["attempts"] for item in stage["rounds"]), + sum(item["successes"] for item in stage["rounds"]), + sum(item["failures"] for item in stage["rounds"])) + return stage.get("attempts",0),stage.get("successes",0),stage.get("failures",0) + +def run(cmd, *, cwd=None, capture=False, env=None): + shown = " ".join(shlex.quote(str(x)) for x in cmd) + api_key = os.environ.get("DD_API_KEY") + app_key = os.environ.get("DD_APP_KEY") + acr_password = os.environ.get("SVLS9604_ACR_PASSWORD") + if api_key: + shown = shown.replace(api_key, "***DD_API_KEY***") + if app_key: + shown = shown.replace(app_key, "***DD_APP_KEY***") + if acr_password: + shown = shown.replace(acr_password, "***ACR_PASSWORD***") + print(f"+ {shown}", flush=True) + merged = os.environ.copy() + if env: + merged.update(env) + result = subprocess.run(cmd, cwd=cwd, env=merged, text=True, + stdout=subprocess.PIPE if capture else None, + stderr=subprocess.PIPE if capture else None) + if result.returncode: + if capture: + print(result.stdout, end="", file=sys.stderr) + print(result.stderr, end="", file=sys.stderr) + raise RuntimeError(f"command failed ({result.returncode}): {shown}") + return result.stdout.strip() if capture else "" + +def git_sha(repo): + return run(["git", "rev-parse", "HEAD"], cwd=repo, capture=True) + +def utc_now(): + return dt.datetime.now(dt.timezone.utc).isoformat() + + +def write_manifest(path, *, run_id, profile, agent_sha, agent_image, resources, + stages=None, suite="full", **provider): + path.write_text(json.dumps({ + "run_id":run_id, + "dd_env":RUN_ENV, + "profile":profile, + "started_at":RUN_STARTED_AT, + "updated_at":utc_now(), + "suite":suite, + "agent_sha":agent_sha, + "agent_image":agent_image, + **provider, + "resources":resources, + "load_stages":stages or [], + },indent=2)) + +def remote_image_exists(image): + return subprocess.run(["docker", "buildx", "imagetools", "inspect", image], + stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL).returncode == 0 + +def short_runtime(runtime): + return {"python":"py", "node":"node", "go":"go", "java":"java", "dotnet":"dotnet", "ruby":"ruby", "php":"php"}[runtime] + +def expand(profile): + resources=[] + for item in MATRIX["profiles"][profile]: + runtimes=MATRIX["runtimes"] if item["runtimes"] == "all" else item["runtimes"] + for runtime in runtimes: + for variant in item["variants"]: + resources.append({**item, "runtime":runtime, "variant":variant}) + return resources + +def name_for(run_id, r): + model={"in-container":"in", "sidecar":"sc", "compat":"compat"}.get(r.get("deployment_model","compat"),"compat") + if r["provider"] == "azure": + # Container Apps names are limited to 32 characters. These names are + # also globally unique enough for App Service because run_id is random. + return f"sv{run_id[:10]}-{r['id'].lower().replace('-', '')}-{model}-{short_runtime(r['runtime'])}-{r['variant'][:4]}"[:32].rstrip("-") + return f"sv-{run_id}-{r['id'].lower().replace('-', '')}-{model}-{short_runtime(r['runtime'])}-{r['variant'][:4]}"[:63] + +def preflight(args): + required=["docker", "git", "python3", "gcloud" if args.profile in ("gcp", "gcp-sanity") else "az"] + for tool in required: + run(["sh", "-c", f"command -v {shlex.quote(tool)} >/dev/null"]) + if not os.environ.get("DD_API_KEY") or os.environ.get("DD_SITE") != "datad0g.com": + raise RuntimeError("runner must be invoked through dd-auth for datad0g.com") + if args.profile in ("gcp", "gcp-sanity"): + account=run(["gcloud","auth","list","--filter=status:ACTIVE","--format=value(account)"],capture=True) + if not account: + raise RuntimeError("gcloud has no active account") + print(f"Authenticated gcloud account: {account}") + else: + account=run(["az","account","show","--query","{name:name,id:id}","-o","json"],capture=True) + print(f"Authenticated Azure subscription: {account}") + run(["docker","info","--format={{.ServerVersion}}"],capture=True) + print(f"Datadog site: {os.environ['DD_SITE']} (dd-auth org UUID 2)") + +def image_digest(image): + output=run(["docker","buildx","imagetools","inspect",image],capture=True) + match=re.search(r"^Digest:\s+(sha256:[0-9a-f]+)$",output,re.MULTILINE) + if not match: + raise RuntimeError(f"could not determine digest for {image}") + return match.group(1) + +def build_agent(project, region, registry, run_id): + image=f"{registry}/serverless-init:{run_id}-v3" + sha=git_sha(AGENT_REPO) + release=json.loads((AGENT_REPO/"release.json").read_text()) + version=f"{release['current_milestone']}-dev" + if remote_image_exists(image): + print(f"Reusing candidate image {image}") + else: + run(["docker","buildx","build","--platform=linux/amd64","--push", + "--build-arg",f"GIT_COMMIT={sha[:12]}","--build-arg",f"AGENT_VERSION={version}", + "--build-arg",f"SERVERLESS_INIT_VERSION={version}", + "-f",str(AGENT_REPO/"scripts/serverless-deploy/Dockerfile.serverless-init"), + "-t",image,str(AGENT_REPO)]) + digest=run(["gcloud","artifacts","docker","images","describe",image, + f"--project={project}",f"--format=value(image_summary.digest)"],capture=True) + return f"{registry}/serverless-init@{digest}", sha + +def build_runtime_images(registry, run_id, agent_image): + fixture=ROOT/"fixtures" + images={} + def build_one(pair): + runtime,target=pair + # v2 init images use runtime-specific stages with explicit CMD values; + # serverless-init needs those argv values to launch the wrapped app. + image_suffix=f"{target}-v3" if target == "init" else target + build_target=f"{runtime}-init" if target == "init" else "plain" + image=f"{registry}/fixture-{runtime}-{image_suffix}:{run_id}" + if remote_image_exists(image): + print(f"Reusing fixture image {image}") + else: + run(["docker","buildx","build","--platform=linux/amd64","--push", + "--build-arg",f"RUNTIME={runtime}","--build-arg",f"AGENT_IMAGE={agent_image}", + "--target",build_target,"-t",image,"-f",str(fixture/"Dockerfile"),str(fixture)]) + return (runtime,target,image) + pairs=[(r,t) for r in MATRIX["runtimes"] for t in ("plain","init")] + with concurrent.futures.ThreadPoolExecutor(max_workers=3) as pool: + for runtime,target,image in pool.map(build_one,pairs): + images[f"{runtime}:{target}"]=image + return images + +def ensure_registry(project, region): + repo="svls9604" + exists=subprocess.run(["gcloud","artifacts","repositories","describe",repo, + f"--location={region}",f"--project={project}"], + stdout=subprocess.DEVNULL,stderr=subprocess.DEVNULL).returncode == 0 + if not exists: + run(["gcloud","artifacts","repositories","create",repo,"--repository-format=docker", + f"--location={region}",f"--project={project}"]) + run(["gcloud","auth","configure-docker",f"{region}-docker.pkg.dev","--quiet"]) + return f"{region}-docker.pkg.dev/{project}/{repo}" + +def env_list(name): + return {"DD_API_KEY":os.environ["DD_API_KEY"],"DD_SITE":"datad0g.com","DD_ENV":RUN_ENV, + "DD_SERVICE":name,"DD_SERVERLESS_DIAGNOSTIC_INFO":"true","DD_LOG_LEVEL":"debug"} + +def env_arg(values): + return ",".join(f"{k}={v}" for k,v in values.items()) + +def service_json(project, name, app_image, agent_image, variant, runtime, function=False): + labels={"svls9604":"true","svls9604-run":name.split("-")[1],"runtime":runtime, + "workload":"function-gen2" if function else "cloud-run"} + annotations={"autoscaling.knative.dev/minScale":"1" if variant=="busy" else "0", + "autoscaling.knative.dev/maxScale":"100", + "run.googleapis.com/container-dependencies":json.dumps({"app":["datadog-sidecar"]})} + return {"apiVersion":"serving.knative.dev/v1","kind":"Service", + "metadata":{"name":name,"namespace":project,"labels":labels}, + "spec":{"template":{"metadata":{"annotations":annotations},"spec":{"containerConcurrency":1 if variant=="coldstart" else 80,"timeoutSeconds":300,"containers":[ + {"name":"app","image":app_image,"ports":[{"containerPort":8080}],"env":[{"name":"DD_ENV","value":RUN_ENV}]}, + {"name":"datadog-sidecar","image":agent_image,"startupProbe":{"tcpSocket":{"port":5555},"periodSeconds":3,"failureThreshold":20}, + "env":[{"name":k,"value":v} for k,v in {**env_list(name),"DD_HEALTH_PORT":"5555","DD_APM_NON_LOCAL_TRAFFIC":"true","DD_DOGSTATSD_NON_LOCAL_TRAFFIC":"true", **({"FUNCTION_TARGET":"main"} if function else {})}.items()]} + ]}}}} + +def gcp_service_identity(project, region, service, revision=None): + status=json.loads(run(["gcloud","run","services","describe",service, + f"--project={project}",f"--region={region}", + "--format=json(status.url,status.latestReadyRevisionName)"],capture=True)) + revision=revision or status["status"]["latestReadyRevisionName"] + return {"endpoint":status["status"]["url"], + "resource_id":f"//run.googleapis.com/projects/{project}/locations/{region}/revisions/{revision}", + "parent_resource_id":f"//run.googleapis.com/projects/{project}/locations/{region}/services/{service}", + "deployment_id":revision, + "expected_resource_ids":[f"//run.googleapis.com/projects/{project}/locations/{region}/revisions/{revision}"]} + +def gcp_function_source(runtime, name, run_dir): + source=run_dir/f"function-{name}"; source.mkdir(exist_ok=True) + specs={ + "python":("python312","main"), "node":("nodejs22","main"), "go":("go126","main"), + "java":("java21","functions.Main"), "dotnet":("dotnet10","Function"), + "ruby":("ruby33","main"), "php":("php84","main")} + if runtime == "python": + (source/"main.py").write_text("import functions_framework\n@functions_framework.http\ndef main(request): return 'Hello World!'\n") + (source/"requirements.txt").write_text("functions-framework==3.*\n") + elif runtime == "node": + (source/"index.js").write_text("const functions=require('@google-cloud/functions-framework');functions.http('main',(q,r)=>r.send('Hello World!'));\n") + (source/"package.json").write_text(json.dumps({"name":name,"version":"1.0.0","main":"index.js","dependencies":{"@google-cloud/functions-framework":"^3.4.0"}})) + elif runtime == "go": + (source/"go.mod").write_text("module example.com/svls9604\n\ngo 1.24\n\nrequire github.com/GoogleCloudPlatform/functions-framework-go v1.9.0\n") + (source/"function.go").write_text('package function\nimport("fmt";"net/http";"github.com/GoogleCloudPlatform/functions-framework-go/functions")\nfunc init(){functions.HTTP("main",handler)}\nfunc handler(w http.ResponseWriter,r *http.Request){fmt.Fprint(w,"Hello World!")}\n') + elif runtime == "java": + package=source/"src/main/java/functions"; package.mkdir(parents=True,exist_ok=True) + (source/"pom.xml").write_text('4.0.0functionssvls96041.021com.google.cloud.functionsfunctions-framework-api1.1.4') + (package/"Main.java").write_text('package functions;import com.google.cloud.functions.*;import java.io.*;public class Main implements HttpFunction{public void service(HttpRequest q,HttpResponse r)throws IOException{r.getWriter().write("Hello World!");}}\n') + elif runtime == "dotnet": + (source/"Function.csproj").write_text('net10.0Exeenable') + (source/"Function.cs").write_text('using Google.Cloud.Functions.Framework;using Microsoft.AspNetCore.Http;public class Function:IHttpFunction{public async Task HandleAsync(HttpContext c){await c.Response.WriteAsync("Hello World!");}}\n') + elif runtime == "ruby": + (source/"Gemfile").write_text("source 'https://rubygems.org'\ngem 'functions_framework', '~> 1.4'\n") + (source/"Gemfile.lock").write_text("""GEM + remote: https://rubygems.org/ + specs: + cloud_events (0.9.0) + functions_framework (1.7.0) + cloud_events (>= 0.7.0, < 2.a) + puma (>= 4.3.0, < 9.a) + rack (>= 2.1, < 4.a) + nio4r (2.7.5) + puma (8.0.2) + nio4r (~> 2.0) + rack (3.2.7) + +PLATFORMS + aarch64-linux + ruby + +DEPENDENCIES + functions_framework (~> 1.4) + +BUNDLED WITH + 2.5.22 +""") + (source/"app.rb").write_text("require 'functions_framework'\nFunctionsFramework.http 'main' do |_request|\n 'Hello World!'\nend\n") + else: + (source/"composer.json").write_text(json.dumps({"require":{"google/cloud-functions-framework":"^1.4"}})) + (source/"index.php").write_text("r.status(200).send('Hello World!'));\n") + run(["gcloud","functions","deploy",name,"--no-gen2",f"--project={project}",f"--region={region}","--runtime=nodejs20", + "--trigger-http","--allow-unauthenticated","--entry-point=main",f"--source={source}", + f"--set-env-vars={env_arg(env_list(name))}","--quiet","--format=value(name)"]) + run(["gcloud","functions","add-iam-policy-binding",name,f"--project={project}",f"--region={region}", + "--member=allUsers","--role=roles/cloudfunctions.invoker","--quiet"]) + url=run(["gcloud","functions","describe",name,f"--project={project}",f"--region={region}","--format=value(httpsTrigger.url)"],capture=True) + return {"name":name,"endpoint":url,"resource_id":f"//cloudfunctions.googleapis.com/projects/{project}/locations/{region}/functions/{name}"} + +def trigger(resource, project, region): + if resource["id"]=="SI-03": + run(["gcloud","run","jobs","execute",resource["name"],f"--project={project}",f"--region={region}","--wait"]) + return {"status":"executed"} + url=resource["endpoint"] + result=subprocess.run(["curl","-fsS","--max-time","60",url],text=True,capture_output=True) + return {"status":"ok" if result.returncode==0 else "failed","http_body":result.stdout[:200],"error":result.stderr[:300]} + +def http_request(url): + started=time.monotonic() + try: + with urllib.request.urlopen(url,timeout=60) as response: + response.read(256) + return response.status,round((time.monotonic()-started)*1000,1),None + except Exception as error: + return 0,round((time.monotonic()-started)*1000,1),str(error)[:300] + +def burst(resource, count=80): + with concurrent.futures.ThreadPoolExecutor(max_workers=count) as pool: + results=list(pool.map(http_request,[resource["endpoint"]]*count)) + latencies=sorted(x[1] for x in results) + successes=sum(1 for status,_,_ in results if 200 <= status < 400) + failures=[error or f"HTTP {status}" for status,_,error in results if not 200 <= status < 400] + def percentile(p): + return latencies[min(len(latencies)-1,round((len(latencies)-1)*p))] + return {"attempts":count,"successes":successes,"failures":count-successes, + "latency_ms":{"p50":percentile(.50),"p95":percentile(.95),"max":latencies[-1]}, + "sample_errors":failures[:5]} + + +def run_same_resource_stage(stage_id, targets, count): + started=utc_now() + results=[] + for i,resource in enumerate(targets,1): + print(f"[{stage_id} {i}/{len(targets)}] {count} requests -> {resource['name']}") + results.append({"resource_id":resource["resource_id"],"name":resource["name"], + **burst(resource,count)}) + return {"id":stage_id,"scenario":"same-resource concurrent requests", + "requests_per_resource":count,"started_at":started,"completed_at":utc_now(), + "resources":results} + + +def run_distributed_stage(targets): + started=utc_now() + with concurrent.futures.ThreadPoolExecutor(max_workers=min(100,len(targets))) as pool: + raw=list(pool.map(lambda resource: http_request(resource["endpoint"]),targets)) + resources=[] + for resource,(status,elapsed,error) in zip(targets,raw): + resources.append({"resource_id":resource["resource_id"],"name":resource["name"], + "http_status":status,"elapsed_ms":elapsed,"error":error}) + successes=sum(1 for result in resources if 200 <= result["http_status"] < 400) + return {"id":"L4","scenario":"one concurrent request per distinct resource_id", + "started_at":started,"completed_at":utc_now(),"attempts":len(resources), + "successes":successes,"failures":len(resources)-successes,"resources":resources} + + +def run_sustained_stage(targets, minutes, interval_seconds): + started=utc_now() + deadline=time.monotonic() + minutes*60 + rounds=[] + while True: + round_started=utc_now() + with concurrent.futures.ThreadPoolExecutor(max_workers=min(100,len(targets))) as pool: + raw=list(pool.map(lambda resource: http_request(resource["endpoint"]),targets)) + successes=sum(1 for status,_,_ in raw if 200 <= status < 400) + rounds.append({"started_at":round_started,"completed_at":utc_now(), + "attempts":len(raw),"successes":successes, + "failures":len(raw)-successes}) + remaining=deadline-time.monotonic() + if remaining <= 0: + break + time.sleep(min(interval_seconds,remaining)) + return {"id":"L5","scenario":"unchanged sustained reporting window", + "duration_minutes":minutes,"interval_seconds":interval_seconds, + "started_at":started,"completed_at":utc_now(),"rounds":rounds} + + +def run_full_load_suite(targets, *, sustained_minutes, sustained_interval): + stages=[] + for stage_id,count in FULL_LOAD_STAGES: + stages.append(run_same_resource_stage(stage_id,targets,count)) + stages.append(run_distributed_stage(targets)) + busy=[resource for resource in targets if resource.get("variant")=="busy"] + stages.append(run_sustained_stage(busy or targets,sustained_minutes,sustained_interval)) + return stages + +def add_expected_revision(resource, identity): + expected=resource.setdefault("expected_resource_ids",[resource["resource_id"]]) + if identity["resource_id"] not in expected: + expected.append(identity["resource_id"]) + resource.setdefault("observed_revisions",[]).append(identity) + +def gcp_revision_stages(project, region, targets, run_id): + target=next((r for r in targets if r["id"]=="SI-02" and r["runtime"]=="python" and r["variant"]=="busy"),None) + if not target: + return [{"id":"L6","scenario":"controlled revision cold-start pressure","status":"NOT MEASURED","reason":"representative SI-02 Python busy target missing"}, + {"id":"L7","scenario":"two active revisions with split traffic","status":"NOT MEASURED","reason":"representative SI-02 Python busy target missing"}] + revision_a=target["deployment_id"] + suffix=f"load-{run_id[-6:].lower()}"[:15] + started=utc_now() + run(["gcloud","run","services","update",target["name"],f"--project={project}",f"--region={region}", + "--concurrency=1","--min-instances=0","--max-instances=100",f"--revision-suffix={suffix}", + "--container=datadog-sidecar",f"--update-env-vars=DD_VERSION={suffix}"]) + revision_b=f"{target['name']}-{suffix}" + identity_b=gcp_service_identity(project,region,target["name"],revision_b) + add_expected_revision(target,identity_b) + pressure=burst({**target,"endpoint":identity_b["endpoint"]},100) + l6={"id":"L6","scenario":"fresh revision, concurrency=1, 100-request cold-start pressure", + "started_at":started,"completed_at":utc_now(),"provider":"gcp","resource":target["name"], + "parent_resource_id":target["parent_resource_id"],"revision_a":revision_a, + "revision_b":identity_b["deployment_id"],"resource_id_b":identity_b["resource_id"], + "provider_instance_starts":"REQUIRES LOG EVIDENCE",**pressure} + l7_started=utc_now() + run(["gcloud","run","services","update-traffic",target["name"],f"--project={project}",f"--region={region}", + f"--set-tags=old={revision_a},new={identity_b['deployment_id']}", + f"--to-revisions={revision_a}=10,{identity_b['deployment_id']}=90"]) + traffic=json.loads(run(["gcloud","run","services","describe",target["name"],f"--project={project}",f"--region={region}", + "--format=json(status.traffic)"],capture=True))["status"]["traffic"] + tagged={entry.get("tag"):entry.get("url") for entry in traffic if entry.get("tag") and entry.get("url")} + old_result=burst({**target,"endpoint":tagged["old"]},10) + new_result=burst({**target,"endpoint":tagged["new"]},10) + split_result=burst(target,100) + l7={"id":"L7","scenario":"two active revisions, 10/90 service traffic plus direct tagged revision probes", + "started_at":l7_started,"completed_at":utc_now(),"provider":"gcp","resource":target["name"], + "parent_resource_id":target["parent_resource_id"],"revision_a":revision_a, + "revision_b":identity_b["deployment_id"],"traffic":traffic, + "attempts":split_result["attempts"]+old_result["attempts"]+new_result["attempts"], + "successes":split_result["successes"]+old_result["successes"]+new_result["successes"], + "failures":split_result["failures"]+old_result["failures"]+new_result["failures"], + "service_traffic":split_result,"revision_a_direct":old_result,"revision_b_direct":new_result} + return [l6,l7] + +def gcp_scaling_stages(project, region, targets, run_id, maxima): + target=next((r for r in targets if r["id"]=="SI-02" and r["runtime"]=="python" and r["variant"]=="busy"),None) + if not target: + return [{"id":"L8","scenario":"minimum-instance report fan-out","status":"NOT MEASURED","reason":"representative SI-02 Python busy target missing"}, + {"id":"L9","scenario":"maximum-instance scale-out ceilings","status":"NOT MEASURED","reason":"representative SI-02 Python busy target missing"}] + + def configure_revision(label, minimum, maximum): + suffix=f"{label}-{run_id[-4:].lower()}"[:15] + started_at=utc_now() + run(["gcloud","run","services","update",target["name"],f"--project={project}",f"--region={region}", + "--concurrency=1",f"--min-instances={minimum}",f"--max-instances={maximum}", + f"--revision-suffix={suffix}","--container=datadog-sidecar", + f"--update-env-vars=DD_VERSION={suffix}"]) + revision=f"{target['name']}-{suffix}" + identity=gcp_service_identity(project,region,target["name"],revision) + add_expected_revision(target,identity) + run(["gcloud","run","services","update-traffic",target["name"],f"--project={project}",f"--region={region}", + f"--to-revisions={identity['deployment_id']}=100"]) + identity["started_at"]=started_at + identity["configured_at"]=utc_now() + return identity + + minimum_cases=[] + l8_started=utc_now() + for minimum in (0,5,100): + identity=configure_revision(f"min{minimum}",minimum,100) + probe=burst({**target,"endpoint":identity["endpoint"]},1) + minimum_cases.append({"minimum_instances":minimum,"maximum_instances":100, + "resource_id":identity["resource_id"],"deployment_id":identity["deployment_id"], + "started_at":identity["started_at"],"completed_at":utc_now(), + "provider_instance_starts":"REQUIRES PROVIDER/STARTUP LOG EVIDENCE",**probe}) + l8={"id":"L8","scenario":"minimum-instance report fan-out (0, 5, 100)", + "started_at":l8_started,"completed_at":utc_now(),"provider":"gcp","resource":target["name"], + "cases":minimum_cases,"resources":minimum_cases} + + maximum_cases=[] + l9_started=utc_now() + for maximum in maxima: + identity=configure_revision(f"max{maximum}",0,maximum) + pressure=burst({**target,"endpoint":identity["endpoint"]},100) + maximum_cases.append({"minimum_instances":0,"maximum_instances":maximum, + "resource_id":identity["resource_id"],"deployment_id":identity["deployment_id"], + "started_at":identity["started_at"],"completed_at":utc_now(), + "provider_instance_starts":"REQUIRES PROVIDER/STARTUP LOG EVIDENCE", + "note":"100 requests validate fan-out and identity, not attainment of the configured maximum", + **pressure}) + l9={"id":"L9","scenario":"maximum-instance configuration boundaries with 100-request pressure", + "started_at":l9_started,"completed_at":utc_now(),"provider":"gcp","resource":target["name"], + "cases":maximum_cases,"resources":maximum_cases} + return [l8,l9] + +def azure_revision_identity(app_id, revision, fqdn=None): + resource_id=f"{app_id.rstrip('/')}/revisions/{revision}".lower() + result={"resource_id":resource_id,"parent_resource_id":app_id.lower(), + "deployment_id":revision,"expected_resource_ids":[resource_id]} + if fqdn: + result["endpoint"]="https://"+fqdn + return result + +def azure_revision_stages(resource_group, targets, run_id): + target=next((r for r in targets if r["id"]=="SI-06" and r["runtime"]=="python" and r["variant"]=="busy"),None) + if not target: + return [{"id":"L6","scenario":"controlled revision cold-start pressure","status":"NOT MEASURED","reason":"representative SI-06 Python busy target missing"}, + {"id":"L7","scenario":"two active revisions with split traffic","status":"NOT MEASURED","reason":"representative SI-06 Python busy target missing"}] + revision_a=target["deployment_id"] + suffix=f"load{run_id[-6:].lower()}"[:10] + started=utc_now() + revision_b=run(["az","containerapp","revision","copy","--resource-group",resource_group,"--name",target["name"], + "--from-revision",revision_a,"--container-name","datadog-sidecar", + "--set-env-vars",f"DD_VERSION={suffix}","--revision-suffix",suffix, + "--min-replicas","0","--max-replicas","100","--scale-rule-name","http", + "--scale-rule-type","http","--scale-rule-http-concurrency","1", + "--query","properties.latestRevisionName","-o","tsv"],capture=True) + revisions=json.loads(run(["az","containerapp","revision","list","--resource-group",resource_group,"--name",target["name"],"-o","json"],capture=True)) + by_name={item["name"]:item for item in revisions} + identity_b=azure_revision_identity(target["parent_resource_id"],revision_b,by_name[revision_b]["properties"].get("fqdn")) + add_expected_revision(target,identity_b) + pressure=burst({**target,"endpoint":identity_b["endpoint"]},100) + l6={"id":"L6","scenario":"fresh revision, HTTP concurrency=1, 100-request cold-start pressure", + "started_at":started,"completed_at":utc_now(),"provider":"azure","resource":target["name"], + "parent_resource_id":target["parent_resource_id"],"revision_a":revision_a,"revision_b":revision_b, + "resource_id_b":identity_b["resource_id"],"provider_instance_starts":"REQUIRES REVISION/LOG EVIDENCE",**pressure} + l7_started=utc_now() + run(["az","containerapp","ingress","traffic","set","--resource-group",resource_group,"--name",target["name"], + "--revision-weight",f"{revision_a}=10",f"{revision_b}=90"]) + old_endpoint="https://"+by_name[revision_a]["properties"]["fqdn"] + new_endpoint=identity_b["endpoint"] + old_result=burst({**target,"endpoint":old_endpoint},10) + new_result=burst({**target,"endpoint":new_endpoint},10) + split_result=burst(target,100) + l7={"id":"L7","scenario":"two active revisions, 10/90 service traffic plus direct revision probes", + "started_at":l7_started,"completed_at":utc_now(),"provider":"azure","resource":target["name"], + "parent_resource_id":target["parent_resource_id"],"revision_a":revision_a,"revision_b":revision_b, + "traffic_weights":{revision_a:10,revision_b:90}, + "attempts":split_result["attempts"]+old_result["attempts"]+new_result["attempts"], + "successes":split_result["successes"]+old_result["successes"]+new_result["successes"], + "failures":split_result["failures"]+old_result["failures"]+new_result["failures"], + "service_traffic":split_result,"revision_a_direct":old_result,"revision_b_direct":new_result} + return [l6,l7] + +def azure_scaling_stages(resource_group, targets, run_id, maxima): + target=next((r for r in targets if r["id"]=="SI-06" and r["runtime"]=="python" and r["variant"]=="busy"),None) + if not target: + return [{"id":"L8","scenario":"minimum-replica report fan-out","status":"NOT MEASURED","reason":"representative SI-06 Python busy target missing"}, + {"id":"L9","scenario":"maximum-replica scale-out ceilings","status":"NOT MEASURED","reason":"representative SI-06 Python busy target missing"}] + + def copy_revision(label, minimum, maximum): + suffix=f"{label}{run_id[-4:].lower()}"[:10] + started_at=utc_now() + revision=run(["az","containerapp","revision","copy","--resource-group",resource_group,"--name",target["name"], + "--from-revision",target["deployment_id"],"--container-name","datadog-sidecar", + "--set-env-vars",f"DD_VERSION={suffix}","--revision-suffix",suffix, + "--min-replicas",str(minimum),"--max-replicas",str(maximum),"--scale-rule-name","http", + "--scale-rule-type","http","--scale-rule-http-concurrency","1", + "--query","properties.latestRevisionName","-o","tsv"],capture=True) + revisions=json.loads(run(["az","containerapp","revision","list","--resource-group",resource_group,"--name",target["name"],"-o","json"],capture=True)) + observed=next(item for item in revisions if item["name"]==revision) + identity=azure_revision_identity(target["parent_resource_id"],revision,observed["properties"].get("fqdn")) + identity["started_at"]=started_at + identity["configured_at"]=utc_now() + add_expected_revision(target,identity) + return identity + + minimum_cases=[] + l8_started=utc_now() + for minimum in (0,5,100): + identity=copy_revision(f"min{minimum}",minimum,100) + probe=burst({**target,"endpoint":identity["endpoint"]},1) + minimum_cases.append({"minimum_instances":minimum,"maximum_instances":100, + "resource_id":identity["resource_id"],"deployment_id":identity["deployment_id"], + "started_at":identity["started_at"],"completed_at":utc_now(), + "provider_instance_starts":"REQUIRES PROVIDER/STARTUP LOG EVIDENCE",**probe}) + l8={"id":"L8","scenario":"minimum-replica report fan-out (0, 5, 100)", + "started_at":l8_started,"completed_at":utc_now(),"provider":"azure","resource":target["name"], + "cases":minimum_cases,"resources":minimum_cases} + + maximum_cases=[] + l9_started=utc_now() + for maximum in maxima: + identity=copy_revision(f"max{maximum}",0,maximum) + pressure=burst({**target,"endpoint":identity["endpoint"]},100) + maximum_cases.append({"minimum_instances":0,"maximum_instances":maximum, + "resource_id":identity["resource_id"],"deployment_id":identity["deployment_id"], + "started_at":identity["started_at"],"completed_at":utc_now(), + "provider_instance_starts":"REQUIRES PROVIDER/STARTUP LOG EVIDENCE", + "note":"100 requests validate fan-out and identity, not attainment of the configured maximum", + **pressure}) + l9={"id":"L9","scenario":"maximum-replica configuration boundaries with 100-request pressure", + "started_at":l9_started,"completed_at":utc_now(),"provider":"azure","resource":target["name"], + "cases":maximum_cases,"resources":maximum_cases} + return [l8,l9] + +def azure_params(path, values): + body={"$schema":"https://schema.management.azure.com/schemas/2019-04-01/deploymentParameters.json#", + "contentVersion":"1.0.0.0","parameters":{k:{"value":v} for k,v in values.items()}} + path.write_text(json.dumps(body,indent=2)); path.chmod(0o600) + +def azure_deploy(template, resource_group, deployment, values, run_dir): + params=run_dir/f"{deployment}-parameters.json" + azure_params(params,values) + output=run(["az","deployment","group","create","--resource-group",resource_group, + "--name",deployment,"--template-file",str(ROOT/"azure"/template), + "--parameters",f"@{params}","--query","properties.outputs","-o","json"],capture=True) + return json.loads(output) + +def build_agent_azure(registry, run_id): + image=f"{registry}/svls9604/serverless-init:{run_id}-v3" + sha=git_sha(AGENT_REPO) + release=json.loads((AGENT_REPO/"release.json").read_text()) + version=f"{release['current_milestone']}-dev" + if not remote_image_exists(image): + run(["docker","buildx","build","--platform=linux/amd64","--push", + "--build-arg",f"GIT_COMMIT={sha[:12]}","--build-arg",f"AGENT_VERSION={version}", + "--build-arg",f"SERVERLESS_INIT_VERSION={version}", + "-f",str(AGENT_REPO/"scripts/serverless-deploy/Dockerfile.serverless-init"), + "-t",image,str(AGENT_REPO)]) + digest=image_digest(image) + return f"{registry}/svls9604/serverless-init@{digest}",sha + +def source_zip(runtime, name, run_dir): + source=run_dir/f"source-{name}"; source.mkdir(exist_ok=True) + if runtime == "node": + (source/"package.json").write_text(json.dumps({"name":name,"version":"1.0.0","scripts":{"start":"node app.js"}})) + (source/"app.js").write_text("const http=require('http');http.createServer((q,r)=>r.end('Hello World!')).listen(process.env.PORT||8080,'0.0.0.0');\n") + elif runtime == "python": + (source/"requirements.txt").write_text("") + (source/"app.py").write_text("from http.server import BaseHTTPRequestHandler,HTTPServer\nclass H(BaseHTTPRequestHandler):\n def do_GET(self): self.send_response(200);self.end_headers();self.wfile.write(b'Hello World!')\nHTTPServer(('0.0.0.0',int(__import__('os').environ.get('PORT','8000'))),H).serve_forever()\n") + elif runtime == "dotnet": + (source/"app.csproj").write_text('net8.0enable') + (source/"Program.cs").write_text('var b=WebApplication.CreateBuilder(args);var a=b.Build();a.MapGet("/",()=>"Hello World!");a.Run();\n') + archive=run_dir/f"{name}.zip" + with zipfile.ZipFile(archive,"w",zipfile.ZIP_DEFLATED) as z: + for child in source.iterdir(): z.write(child,child.name) + return archive + +def compat_azure_zip(name, run_dir): + package=build_compat_package(run_dir) + source=run_dir/f"source-{name}"; source.mkdir(exist_ok=True) + (source/"package.tgz").write_bytes(package.read_bytes()) + (source/"package.json").write_text(json.dumps({"name":name,"version":"1.0.0","main":"index.js","dependencies":{"@azure/functions":"4.7.2","@datadog/serverless-compat":"file:package.tgz","dd-trace":"5.45.0"}})) + (source/"host.json").write_text(json.dumps({"version":"2.0"})) + (source/"index.js").write_text("require('@datadog/serverless-compat/init');const{app}=require('@azure/functions');app.http('main',{methods:['GET'],authLevel:'anonymous',handler:async()=>({body:'Hello World!'})});\n") + archive=run_dir/f"{name}.zip" + with zipfile.ZipFile(archive,"w",zipfile.ZIP_DEFLATED) as z: + for child in source.iterdir(): z.write(child,child.name) + return archive + +def deploy_azure_resource(args, r, name, images, agent_image, acr, run_id, run_dir): + common={"name":name,"agentImage":agent_image,"registryServer":acr["server"], + "registryUsername":acr["username"],"registryPassword":acr["password"], + "ddApiKey":os.environ["DD_API_KEY"],"runtime":r["runtime"],"runId":run_id} + deployment=f"d-{name}"[:64] + if r["id"] in ("SI-05","SI-06"): + values={**common,"appEnvId":args.azure_containerapp_env_id, + "appImage":images[f"{r['runtime']}:{'init' if r['id']=='SI-05' else 'plain'}"], + "sidecar":r["id"]=="SI-06","minReplicas":1 if r["variant"]=="busy" else 0} + out=azure_deploy("container-app.bicep",args.azure_containerapp_resource_group,deployment,values,run_dir) + identity=azure_revision_identity(out["resourceId"]["value"],out["latestRevisionName"]["value"],out["fqdn"]["value"]) + return {"name":name,**identity} + if r["id"] in ("SI-07","SI-08"): + plan=args.azure_container_plan_id if r["id"]=="SI-07" else args.azure_sidecar_plan_id + values={**common,"servicePlanId":plan, + "appImage":images[f"{r['runtime']}:{'init' if r['id']=='SI-07' else 'plain'}"], + "sidecar":r["id"]=="SI-08","alwaysOn":r["variant"]=="busy"} + out=azure_deploy("web-app-container.bicep",args.azure_resource_group,deployment,values,run_dir) + return {"name":name,"endpoint":"https://"+out["hostname"]["value"],"resource_id":out["resourceId"]["value"]} + if r["id"]=="SI-09": + values={**common,"servicePlanId":args.azure_code_plan_id,"alwaysOn":r["variant"]=="busy"} + out=azure_deploy("web-app-code.bicep",args.azure_resource_group,deployment,values,run_dir) + package=source_zip(r["runtime"],name,run_dir) + try: + run(["az","webapp","deploy","--resource-group",args.azure_resource_group,"--name",name, + "--src-path",str(package),"--type","zip","--clean","true","--async","true"]) + except RuntimeError as e: + print(f"Warning: az webapp deploy exited non-zero ({e}); verifying via health check",flush=True) + endpoint="https://"+out["hostname"]["value"] + print(f"Waiting for {name} to become healthy...",flush=True) + for _ in range(72): + try: + with urllib.request.urlopen(endpoint,timeout=10) as resp: + if resp.status < 500: + print(f"{name} healthy (HTTP {resp.status})",flush=True) + break + except Exception: + pass + time.sleep(10) + return {"name":name,"endpoint":endpoint,"resource_id":out["resourceId"]["value"]} + storage=("sv"+hashlib.sha256(name.encode()).hexdigest()[:20])[:24] + values={"name":name,"storageName":storage,"ddApiKey":os.environ["DD_API_KEY"],"runId":run_id} + out=azure_deploy("function.bicep",args.azure_function_resource_group,deployment,values,run_dir) + package=compat_azure_zip(name,run_dir) + run(["az","functionapp","deployment","source","config-zip","--resource-group",args.azure_function_resource_group, + "--name",name,"--src",str(package),"--build-remote","true"]) + return {"name":name,"endpoint":"https://"+out["hostname"]["value"]+"/api/main","resource_id":out["resourceId"]["value"]} + +def baseline_stage(resources, started_at): + successes=sum(1 for resource in resources + if resource.get("baseline",{}).get("status") in ("ok","executed")) + return {"id":"L0","scenario":"one baseline request or execution per deployed resource", + "started_at":started_at,"completed_at":utc_now(),"attempts":len(resources), + "successes":successes,"failures":len(resources)-successes} + + +def run_azure(args, resources, run_id, run_dir): + run(["az","acr","login","--name",args.azure_acr]) + credential=json.loads(run(["az","acr","credential","show","--name",args.azure_acr,"-o","json"],capture=True)) + acr={"server":args.azure_registry,"username":credential["username"],"password":credential["passwords"][0]["value"]} + os.environ["SVLS9604_ACR_PASSWORD"]=acr["password"] + agent_image,agent_sha=build_agent_azure(args.azure_registry,run_id) + images=build_runtime_images(f"{args.azure_registry}/svls9604",run_id,agent_image) + manifest_path=run_dir/"run-manifest.json" + existing={} + if manifest_path.exists(): + previous=json.loads(manifest_path.read_text()) + if previous.get("profile") in ("azure","azure-sanity") and previous.get("agent_image")==agent_image: + candidates=[item for item in previous.get("resources",[]) + if item.get("agent_image")==agent_image and item.get("endpoint")] + with concurrent.futures.ThreadPoolExecutor(max_workers=min(20,len(candidates) or 1)) as pool: + checks=list(pool.map(lambda item: http_request(item["endpoint"]),candidates)) + for item,(status,_,error) in zip(candidates,checks): + if 200 <= status < 500: + existing[item["name"]]=item + else: + print(f"Azure resume will redeploy unhealthy {item['name']} (HTTP {status}: {error})") + deployed=[] + for i,r in enumerate(resources,1): + name=name_for(run_id,r) + if name in existing: + print(f"[{i}/{len(resources)}] reusing deployed {r['id']} {r['runtime']} {r['variant']} as {name}") + deployed.append(existing[name]) + continue + print(f"[{i}/{len(resources)}] deploying {r['id']} {r['runtime']} {r['variant']} as {name}") + try: + observed=deploy_azure_resource(args,r,name,images,agent_image,acr,run_id,run_dir) + except Exception as exc: + print(f" WARNING: deploy failed for {name}: {exc}", flush=True) + continue + deployed.append({**r,**observed,"agent_image":agent_image}) + write_manifest(manifest_path,run_id=run_id,profile=args.profile,agent_sha=agent_sha,agent_image=agent_image,resources=deployed) + baseline_started=utc_now() + for resource in deployed: + result=subprocess.run(["curl","-fsS","--max-time","60",resource["endpoint"]],text=True,capture_output=True) + resource["baseline"]={"status":"ok" if result.returncode==0 else "failed","http_body":result.stdout[:200],"error":result.stderr[:300]} + stages=[baseline_stage(deployed,baseline_started)] + if args.suite == "full" and not args.skip_burst: + targets=[r for r in deployed if r["id"].startswith("SI-") and r.get("endpoint")] + stages.extend(run_full_load_suite(targets,sustained_minutes=args.sustained_minutes, + sustained_interval=args.sustained_interval)) + stages.extend(azure_revision_stages(args.azure_containerapp_resource_group,targets,run_id)) + if args.scaling_matrix: + stages.extend(azure_scaling_stages(args.azure_containerapp_resource_group,targets,run_id,args.scaling_maxima)) + compat_targets=[r for r in deployed if r["id"].startswith("SC-") and r.get("endpoint")] + for stage_id,count in [("SC-L1",10),("SC-L2",50),("SC-L3",100)]: + stages.append(run_same_resource_stage(stage_id,compat_targets,count)) + write_manifest(manifest_path,run_id=run_id,profile=args.profile,agent_sha=agent_sha, + agent_image=agent_image,resources=deployed,stages=stages,suite=args.suite) + return deployed + +def run_gcp(args, resources, run_id, run_dir): + project=args.project or run(["gcloud","config","get-value","project"],capture=True) + region=args.region + registry=ensure_registry(project,region) + agent_image,agent_sha=build_agent(project,region,registry,run_id) + images=build_runtime_images(registry,run_id,agent_image) + manifest_path=run_dir/"run-manifest.json" + existing={} + if manifest_path.exists(): + previous=json.loads(manifest_path.read_text()) + # Revision-scoped identity is required for Cloud Run and Gen2 functions. + # Never resume a manifest produced by the earlier service-scoped runner. + valid=[] + for x in previous.get("resources",[]): + if x.get("agent_image") != agent_image: + continue + if x["id"] not in ("SI-01","SI-02","SI-04"): + valid.append(x) + elif "/revisions/" in x.get("resource_id","") and x.get("parent_resource_id"): + valid.append(x) + existing={x["name"]:x for x in valid} + deployed=[] + for i,r in enumerate(resources,1): + name=name_for(run_id,r) + if name in existing: + print(f"[{i}/{len(resources)}] reusing deployed {r['id']} {r['runtime']} {r['variant']} as {name}") + deployed.append(existing[name]); continue + print(f"[{i}/{len(resources)}] deploying {r['id']} {r['runtime']} {r['variant']} as {name}") + try: + if r["id"]=="SI-03": observed=deploy_job(project,region,r,name,images) + elif r["id"]=="SC-02": observed=deploy_compat_gcp(project,region,name,run_dir) + else: observed=deploy_service(project,region,r,name,images,agent_image,run_dir) + except Exception as exc: + print(f" WARNING: deploy failed for {name}: {exc}", flush=True) + continue + deployed.append({**r,**observed,"agent_image":agent_image}) + write_manifest(manifest_path,run_id=run_id,profile="gcp",project=project,region=region,agent_sha=agent_sha,agent_image=agent_image,resources=deployed) + baseline_started=utc_now() + for resource in deployed: + resource["baseline"]=trigger(resource,project,region) + stages=[baseline_stage(deployed,baseline_started)] + if args.suite == "full" and not args.skip_burst: + targets=[r for r in deployed if r["id"].startswith("SI-") and r.get("endpoint")] + stages.extend(run_full_load_suite(targets,sustained_minutes=args.sustained_minutes, + sustained_interval=args.sustained_interval)) + stages.extend(gcp_revision_stages(project,region,targets,run_id)) + if args.scaling_matrix: + stages.extend(gcp_scaling_stages(project,region,targets,run_id,args.scaling_maxima)) + compat_targets=[r for r in deployed if r["id"].startswith("SC-") and r.get("endpoint")] + for stage_id,count in [("SC-L1",10),("SC-L2",50),("SC-L3",100)]: + stages.append(run_same_resource_stage(stage_id,compat_targets,count)) + write_manifest(manifest_path,run_id=run_id,profile="gcp",project=project,region=region, + agent_sha=agent_sha,agent_image=agent_image,resources=deployed, + stages=stages,suite=args.suite) + return deployed + +def main(): + global RUN_ENV, RUN_STARTED_AT + parser=argparse.ArgumentParser() + parser.add_argument("--profile",choices=["gcp","azure","gcp-sanity","azure-sanity"],required=True) + parser.add_argument("--project",default=os.environ.get("GCP_PROJECT","datadog-serverless-gcp-demo")) + parser.add_argument("--region",default=os.environ.get("GCP_REGION","us-central1")) + parser.add_argument("--azure-resource-group",default=os.environ.get("AZURE_RESOURCE_GROUP","dd-serverless-test-aas")) + parser.add_argument("--azure-containerapp-resource-group",default=os.environ.get("AZURE_CONTAINERAPP_RESOURCE_GROUP","dd-serverless-test-aca")) + parser.add_argument("--azure-function-resource-group",default=os.environ.get("AZURE_FUNCTION_RESOURCE_GROUP","dd-serverless-test-aca")) + parser.add_argument("--azure-acr",default=os.environ.get("AZURE_ACR","ddsvlstestaca")) + parser.add_argument("--azure-registry",default=os.environ.get("AZURE_REGISTRY","ddsvlstestaca.azurecr.io")) + parser.add_argument("--azure-containerapp-env-id",default=os.environ.get("AZURE_CONTAINERAPP_ENV_ID","/subscriptions/1dd25961-a5c7-45bf-a5ba-c1475d365cc7/resourceGroups/dd-serverless-test-aca/providers/Microsoft.App/managedEnvironments/dd-serverless-env")) + parser.add_argument("--azure-container-plan-id",default=os.environ.get("AZURE_CONTAINER_PLAN_ID","/subscriptions/1dd25961-a5c7-45bf-a5ba-c1475d365cc7/resourceGroups/dd-serverless-test-aas/providers/Microsoft.Web/serverfarms/dd-test-plan-container")) + parser.add_argument("--azure-sidecar-plan-id",default=os.environ.get("AZURE_SIDECAR_PLAN_ID","/subscriptions/1dd25961-a5c7-45bf-a5ba-c1475d365cc7/resourceGroups/dd-serverless-test-aas/providers/Microsoft.Web/serverfarms/dd-test-plan-sidecar")) + parser.add_argument("--azure-code-plan-id",default=os.environ.get("AZURE_CODE_PLAN_ID","/subscriptions/1dd25961-a5c7-45bf-a5ba-c1475d365cc7/resourceGroups/dd-serverless-test-aas/providers/Microsoft.Web/serverfarms/dd-test-plan-linux-code")) + parser.add_argument("--run-id") + parser.add_argument("--suite",choices=["baseline","full"],default="full", + help="full runs L0-L5 plus revision cold-start pressure (L6) and split traffic (L7)") + parser.add_argument("--sustained-minutes",type=float,default=15, + help="duration of the L5 unchanged-report window") + parser.add_argument("--sustained-interval",type=float,default=60, + help="seconds between L5 trigger rounds") + parser.add_argument("--scaling-matrix",action="store_true", + help="add L8/L9 minimum and maximum instance/replica boundary cases") + parser.add_argument("--scaling-maxima",default="100,1000,4000", + help="comma-separated L9 maximum instance/replica settings") + parser.add_argument("--plan",action="store_true") + parser.add_argument("--yes",action="store_true") + parser.add_argument("--skip-burst",action="store_true",help="compatibility alias for --suite baseline") + args=parser.parse_args() + args.scaling_maxima=tuple(int(value) for value in args.scaling_maxima.split(",") if value) + if args.skip_burst: + args.suite="baseline" + resources=expand(args.profile) + run_id=args.run_id or dt.datetime.now(dt.timezone.utc).strftime("%m%d%H%M")+secrets.token_hex(2) + RUN_ENV=f"svls9604-{run_id.lower()}" + RUN_STARTED_AT=dt.datetime.now(dt.timezone.utc).isoformat() + print(f"Profile: {args.profile}\nRun ID: {run_id}\nSuite: {args.suite}\nExpected resources: {len(resources)}") + for r in resources: + print(f" {r['id']:5} {r['workload_type']:28} {r.get('deployment_model','compat'):12} {r['runtime']:7} {r['variant']}") + if args.plan: + return + preflight(args) + if not args.yes: + if input(f"Create {len(resources)} resources? [y/N] ").strip().lower() != "y": + raise SystemExit("cancelled") + run_dir=pathlib.Path(os.environ.get("RESULTS_DIR",f"/tmp/svls9604-{run_id}")); run_dir.mkdir(parents=True,exist_ok=True) + run_dir.chmod(0o700) + failure_exit=0 + if args.profile in ("gcp", "gcp-sanity"): + deployed=run_gcp(args,resources,run_id,run_dir) + failed=[r for r in deployed if r.get("baseline",{}).get("status") not in ("ok","executed")] + print(f"GCP profile deployed {len(deployed)}/{len(resources)}; baseline failures={len(failed)}") + print(f"Evidence: {run_dir}") + if failed: failure_exit=2 + else: + deployed=run_azure(args,resources,run_id,run_dir) + failed=[r for r in deployed if r.get("baseline",{}).get("status") != "ok"] + print(f"Azure profile deployed {len(deployed)}/{len(resources)}; baseline failures={len(failed)}") + print(f"Evidence: {run_dir}") + if failed: failure_exit=2 + + manifest_path=run_dir/"run-manifest.json" + manifest=json.loads(manifest_path.read_text()) + manifest["dd_env"]=RUN_ENV + manifest["started_at"]=RUN_STARTED_AT + manifest["completed_at"]=utc_now() + manifest["candidate_commits"]={ + "serverless_components":git_sha(COMPONENTS_REPO), + "datadog_agent":git_sha(AGENT_REPO), + "datadog_serverless_compat_js":git_sha(COMPAT_JS_REPO), + } + manifest_path.write_text(json.dumps(manifest,indent=2)) + run([sys.executable,str(ROOT/"report.py"),"--manifest",str(manifest_path)]) + for stage in manifest.get("load_stages",[]): + _,_,stage_failures=aggregate_stage(stage) + if stage_failures: + failure_exit=2 + if failure_exit: + raise SystemExit(failure_exit) + +if __name__=="__main__": + main() From 4e178dcc894f82c71cb75196968e6af279906486 Mon Sep 17 00:00:00 2001 From: Nina Rei Date: Wed, 2 Sep 2026 13:26:19 -0400 Subject: [PATCH 3/9] chore(svls9604): add baseline profiles, inventory gates, build.sh - Add gcp-baseline and azure-baseline minimal profiles (1 workload each) for quick REDAPL wire contract validation with a single resource - Set DD_SERVERLESS_INIT_INVENTORY_ENABLED and DD_SERVERLESS_COMPAT_INVENTORY_ENABLED in env_list() so all runner deploys enable inventory without manual az/gcloud override - Add scripts/serverless-compat-deploy/build.sh: builds the Rust compat binary for x86_64-unknown-linux-musl (required by runner.py before packaging) - Include azure-baseline in Azure manifest resume check --- scripts/serverless-compat-deploy/build.sh | 23 + .../serverless-compat-deploy/check-logs.sh | 318 +++++++ .../serverless-compat-deploy/demo-both-rc.sh | 801 ++++++++++++++++++ scripts/svls9604/matrix.json | 6 + scripts/svls9604/runner.py | 14 +- 5 files changed, 1156 insertions(+), 6 deletions(-) create mode 100755 scripts/serverless-compat-deploy/build.sh create mode 100755 scripts/serverless-compat-deploy/check-logs.sh create mode 100755 scripts/serverless-compat-deploy/demo-both-rc.sh diff --git a/scripts/serverless-compat-deploy/build.sh b/scripts/serverless-compat-deploy/build.sh new file mode 100755 index 0000000..a1179a8 --- /dev/null +++ b/scripts/serverless-compat-deploy/build.sh @@ -0,0 +1,23 @@ +#!/usr/bin/env bash +# Build the serverless-compat binary for x86_64-unknown-linux-musl. +# Called by scripts/svls9604/runner.py before packaging. +set -euo pipefail + +REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" + +# Ensure rustup-managed toolchain is on PATH (needed when invoked by runner.py) +export PATH="$HOME/.rustup/toolchains/stable-aarch64-apple-darwin/bin:$PATH" + +CARGO="${CARGO:-$HOME/.rustup/toolchains/stable-aarch64-apple-darwin/bin/cargo}" +if [[ ! -x "$CARGO" ]]; then + CARGO="$(command -v cargo)" +fi + +echo "Building datadog-serverless-compat for x86_64-unknown-linux-musl..." +"$CARGO" build \ + --manifest-path "$REPO_ROOT/Cargo.toml" \ + --package datadog-serverless-compat \ + --target x86_64-unknown-linux-musl \ + --release + +echo "Binary: $REPO_ROOT/target/x86_64-unknown-linux-musl/release/datadog-serverless-compat" diff --git a/scripts/serverless-compat-deploy/check-logs.sh b/scripts/serverless-compat-deploy/check-logs.sh new file mode 100755 index 0000000..c68f474 --- /dev/null +++ b/scripts/serverless-compat-deploy/check-logs.sh @@ -0,0 +1,318 @@ +#!/usr/bin/env bash +set -euo pipefail + +# --------------------------------------------------------------------------- +# check-logs.sh — diagnostic log collector for SVLS-9604 +# +# Collects inventory diagnostic output from deployed self-monitoring services +# for both serverless-init and serverless-compat, then prints the +# resource_id-based DDSQL queries to validate rows in both REDAPL tables. +# +# Resource identity: both REDAPL tables key on resource_id (CCRID), not UUID. +# Look for log lines containing "resource_id=" to find the actual key. +# +# Required env vars: +# GCP_PROJECT — GCP project for init Cloud Run services +# AZURE_SUBSCRIPTION_ID — Azure subscription (auto-detected if omitted) +# +# Service name overrides (defaults match the POC deploy scripts): +# INIT_CR_SERVICE — Cloud Run service name for init in-container (SI-01) +# INIT_CR_SIDECAR — Cloud Run service name for init sidecar (SI-02) +# INIT_CR_JOB — Cloud Run job name (SI-03, separate POC) +# INIT_CR_FN_SIDECAR — Cloud Run service name for Functions Gen 2 (SI-04, separate POC) +# INIT_ACA_INIT — Azure Container App name, in-container (SI-05) +# INIT_ACA_SIDECAR — Azure Container App name, sidecar (SI-06) +# INIT_AAS_CONTAINER — Azure App Service, Linux container (SI-07) +# INIT_AAS_SIDECAR — Azure App Service, SITECONTAINERS (SI-08) +# INIT_AAS_CODE — Azure App Service, Linux code (SI-09) +# COMPAT_AZURE_APP — Azure Function App name (SC-01) +# COMPAT_GCP_FN — GCP Cloud Function Gen 1 name (SC-02) +# COMPAT_GCP_PROJECT — GCP project for compat functions (may differ from init) +# AZURE_RG_ACA — Resource group for Container Apps +# AZURE_RG_AAS — Resource group for App Service apps +# COMPAT_AZURE_RG — Resource group for compat Azure Function +# +# Output: diagnostic-results-YYYYMMDD-HHMMSS.txt +# --------------------------------------------------------------------------- + +: "${GCP_PROJECT:?GCP_PROJECT must be set}" +AZURE_SUBSCRIPTION_ID="${AZURE_SUBSCRIPTION_ID:-$(az account show --query id -o tsv 2>/dev/null || echo '')}" + +# Service name defaults (override to match your deployed services) +INIT_CR_SERVICE="${INIT_CR_SERVICE:-nina-cloudrun-init}" +INIT_CR_SIDECAR="${INIT_CR_SIDECAR:-nina-cloudrun-sidecar}" +INIT_CR_JOB="${INIT_CR_JOB:-nina-cloudrun-job}" +INIT_CR_FN_SIDECAR="${INIT_CR_FN_SIDECAR:-nina-cloudrun-function-sidecar}" +INIT_ACA_INIT="${INIT_ACA_INIT:-nina-containerapp-init}" +INIT_ACA_SIDECAR="${INIT_ACA_SIDECAR:-nina-containerapp-sidecar}" +INIT_AAS_CONTAINER="${INIT_AAS_CONTAINER:-nina-webapp-container}" +INIT_AAS_SIDECAR="${INIT_AAS_SIDECAR:-nina-webapp-sidecar}" +INIT_AAS_CODE="${INIT_AAS_CODE:-nina-webapp-linux-code}" +COMPAT_AZURE_APP="${COMPAT_AZURE_APP:-nina-compat-inventory-node}" +COMPAT_GCP_FN="${COMPAT_GCP_FN:-nina-compat-inventory-nodejs}" +COMPAT_GCP_PROJECT="${COMPAT_GCP_PROJECT:-datadog-sandbox}" +AZURE_RG_ACA="${AZURE_RG_ACA:-dd-serverless-test-aca}" +AZURE_RG_AAS="${AZURE_RG_AAS:-dd-serverless-test-aas}" +COMPAT_AZURE_RG="${COMPAT_AZURE_RG:-self-monitoring-nina-dev}" + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")" && pwd)" +OUTPUT_FILE="${OUTPUT_FILE:-${SCRIPT_DIR}/diagnostic-results-$(date -u +%Y%m%d-%H%M%S).txt}" + +log() { echo "[$(date -u '+%Y-%m-%dT%H:%M:%SZ')] $*"; } + +# --------------------------------------------------------------------------- +# GCP log fetcher — matches inventory log lines containing resource_id +# --------------------------------------------------------------------------- +_gcp_logs() { + local label="$1" resource_type="$2" filter_key="$3" filter_val="$4" project="${5:-${GCP_PROJECT}}" + echo "" + echo "================================================================" + echo "# ${label}" + echo "================================================================" + gcloud logging read \ + "resource.type=${resource_type} AND resource.labels.${filter_key}=${filter_val} AND (textPayload:resource_id OR textPayload:SERVERLESS_DIAGNOSTIC OR textPayload:\"Inventory payload\")" \ + --project="${project}" \ + --limit=50 \ + --format="value(textPayload)" 2>/dev/null \ + || echo " (no matching logs — trigger a request and retry)" +} + +# --------------------------------------------------------------------------- +# Azure Container App log fetcher +# --------------------------------------------------------------------------- +_azure_containerapp_logs() { + local label="$1" app_name="$2" rg="$3" container="${4:-}" + [[ -z "${AZURE_SUBSCRIPTION_ID}" ]] && { echo " (AZURE_SUBSCRIPTION_ID not set — skipping)"; return; } + echo "" + echo "================================================================" + echo "# ${label}" + echo "================================================================" + local args=(--name "${app_name}" --resource-group "${rg}" \ + --subscription "${AZURE_SUBSCRIPTION_ID}" --tail 200) + [[ -n "${container}" ]] && args+=(--container "${container}") + az containerapp logs show "${args[@]}" 2>/dev/null \ + | grep -E "resource_id|SERVERLESS_DIAGNOSTIC|Inventory payload" \ + || echo " (no matching logs — trigger a request and retry)" +} + +# --------------------------------------------------------------------------- +# Azure Web App log fetcher (downloads archived logs) +# --------------------------------------------------------------------------- +_azure_webapp_logs() { + local label="$1" app_name="$2" rg="$3" + [[ -z "${AZURE_SUBSCRIPTION_ID}" ]] && { echo " (AZURE_SUBSCRIPTION_ID not set — skipping)"; return; } + echo "" + echo "================================================================" + echo "# ${label}" + echo "================================================================" + local tmp_zip + tmp_zip=$(mktemp /tmp/webapp-logs-XXXXXX.zip) + az webapp log download \ + --name "${app_name}" \ + --resource-group "${rg}" \ + --subscription "${AZURE_SUBSCRIPTION_ID}" \ + --log-file "${tmp_zip}" 2>/dev/null \ + && unzip -p "${tmp_zip}" 2>/dev/null \ + | grep -aE "resource_id|SERVERLESS_DIAGNOSTIC|Inventory payload" \ + | sort -u \ + || echo " (no matching logs — trigger a request and retry)" + rm -f "${tmp_zip}" +} + +# --------------------------------------------------------------------------- +# Azure Function log fetcher (compat) +# --------------------------------------------------------------------------- +_azure_function_logs() { + local label="$1" app_name="$2" rg="$3" + [[ -z "${AZURE_SUBSCRIPTION_ID}" ]] && { echo " (AZURE_SUBSCRIPTION_ID not set — skipping)"; return; } + echo "" + echo "================================================================" + echo "# ${label}" + echo "================================================================" + # az webapp log tail exits after --timeout seconds; we want a snapshot not a stream. + # Use log download instead for reliability. + local tmp_zip + tmp_zip=$(mktemp /tmp/fn-logs-XXXXXX.zip) + az webapp log download \ + --name "${app_name}" \ + --resource-group "${rg}" \ + --subscription "${AZURE_SUBSCRIPTION_ID}" \ + --log-file "${tmp_zip}" 2>/dev/null \ + && unzip -p "${tmp_zip}" 2>/dev/null \ + | grep -aE "resource_id|workload_type|Inventory payload sent" \ + | sort -u \ + || echo " (no matching logs — trigger a request and retry)" + rm -f "${tmp_zip}" +} + +# --------------------------------------------------------------------------- +# Run all checks +# --------------------------------------------------------------------------- +run_checks() { + echo "================================================================" + echo "SVLS-9604 — Serverless REDAPL Diagnostic Log Collection" + echo "Date: $(date -u '+%Y-%m-%d %H:%M:%S UTC')" + echo "================================================================" + echo "" + echo "What to look for: lines containing 'resource_id=//...' — this is" + echo "the CCRID used as the REDAPL table key, NOT the agent UUID." + echo "" + + echo "===== serverless_init_agent workloads =====" + + _gcp_logs "SI-01 Cloud Run Service (in-container, ${INIT_CR_SERVICE})" \ + cloud_run_revision service_name "${INIT_CR_SERVICE}" + + _gcp_logs "SI-02 Cloud Run Service (sidecar, ${INIT_CR_SIDECAR})" \ + cloud_run_revision service_name "${INIT_CR_SIDECAR}" + + echo "" + echo "# SI-03 Cloud Run Job — requires separate POC (not deployed via self-monitoring)" + _gcp_logs "SI-03 Cloud Run Job (${INIT_CR_JOB}) — if POC deployed" \ + cloud_run_job job_name "${INIT_CR_JOB}" + + echo "" + echo "# SI-04 Cloud Run Functions Gen 2 — requires separate POC" + _gcp_logs "SI-04 Cloud Run Functions Gen 2 (${INIT_CR_FN_SIDECAR}) — if POC deployed" \ + cloud_run_revision service_name "${INIT_CR_FN_SIDECAR}" + + _azure_containerapp_logs "SI-05 Azure Container App (in-container, ${INIT_ACA_INIT})" \ + "${INIT_ACA_INIT}" "${AZURE_RG_ACA}" + + _azure_containerapp_logs "SI-06 Azure Container App (sidecar, dd-agent, ${INIT_ACA_SIDECAR})" \ + "${INIT_ACA_SIDECAR}" "${AZURE_RG_ACA}" "dd-agent" + + _azure_webapp_logs "SI-07 Azure App Service Linux container (${INIT_AAS_CONTAINER})" \ + "${INIT_AAS_CONTAINER}" "${AZURE_RG_AAS}" + + _azure_webapp_logs "SI-08 Azure App Service SITECONTAINERS (${INIT_AAS_SIDECAR})" \ + "${INIT_AAS_SIDECAR}" "${AZURE_RG_AAS}" + + _azure_webapp_logs "SI-09 Azure App Service Linux code (${INIT_AAS_CODE})" \ + "${INIT_AAS_CODE}" "${AZURE_RG_AAS}" + + echo "" + echo "===== serverless_compat_agent workloads =====" + + _azure_function_logs "SC-01 Azure Functions Node.js (${COMPAT_AZURE_APP})" \ + "${COMPAT_AZURE_APP}" "${COMPAT_AZURE_RG}" + + _gcp_logs "SC-02 GCP Cloud Functions Gen 1 (${COMPAT_GCP_FN})" \ + cloud_function function_name "${COMPAT_GCP_FN}" "${COMPAT_GCP_PROJECT}" +} + +# --------------------------------------------------------------------------- +# Extract resource_ids and generate DDSQL queries +# --------------------------------------------------------------------------- +generate_sql() { + local output_file="$1" + + # Extract resource_ids — CCRID format: //run.googleapis.com/... or //cloudfunctions.googleapis.com/... etc. + local resource_ids + resource_ids=$(python3 -c " +import re, sys +text = open('${output_file}').read() +# resource_id= or resource_id: followed by a CCRID +rids = re.findall(r'resource_id[=:]\s*(//[^\s,\"\']+)', text) +print('\n'.join(sorted(set(rids)))) +" 2>/dev/null || true) + + echo "" + echo "================================================================" + echo "# DDSQL Queries (paste in go/redapl → Queries → SQL)" + echo "# Note: these tables do NOT have api_key_uuid — filter by resource_id." + echo "================================================================" + echo "" + + if [[ -z "${resource_ids}" ]]; then + echo "-- No resource_ids found in logs yet." + echo "-- Trigger the services, wait ~5 min for EPRW propagation, and retry." + echo "" + echo "-- Fallback: show all rows modified today" + echo "SELECT resource_id, workload_type, deployment_model, _modified_at" + echo "FROM udm.all.serverless_init_agent" + echo "WHERE _modified_at >= TIMESTAMP '$(date -u +%Y-%m-%dT00:00:00Z)'" + echo "ORDER BY _modified_at DESC LIMIT 20;" + echo "" + echo "SELECT resource_id, workload_type, _modified_at" + echo "FROM udm.all.serverless_compat_agent" + echo "WHERE _modified_at >= TIMESTAMP '$(date -u +%Y-%m-%dT00:00:00Z)'" + echo "ORDER BY _modified_at DESC LIMIT 20;" + return + fi + + # Build IN clause + local in_clause="" + while IFS= read -r rid; do + [[ -z "${rid}" ]] && continue + in_clause+=" '${rid}',\n" + done <<< "${resource_ids}" + + echo "-- serverless_init_agent: rows by extracted resource_id" + echo "SELECT _key, resource_id, resource_name, workload_type, deployment_model," + echo " agent_version_base, serverless_init_version, runtime," + echo " _first_seen_at, _modified_at" + echo "FROM udm.all.serverless_init_agent" + echo "WHERE resource_id IN (" + printf "%b" "${in_clause}" + echo ")" + echo "ORDER BY workload_type, _modified_at DESC;" + echo "" + + echo "-- serverless_compat_agent: rows by extracted resource_id" + echo "SELECT _key, resource_id, resource_name, workload_type," + echo " serverless_compat_version, serverless_compat_runtime_version," + echo " _first_seen_at, _modified_at" + echo "FROM udm.all.serverless_compat_agent" + echo "WHERE resource_id IN (" + printf "%b" "${in_clause}" + echo ")" + echo "ORDER BY workload_type, _modified_at DESC;" + echo "" + + echo "-- Cardinality: rows must equal distinct_resources" + echo "SELECT 'serverless_init_agent' AS flavor, workload_type," + echo " COUNT(*) AS total_rows, COUNT(DISTINCT resource_id) AS distinct_resources" + echo "FROM udm.all.serverless_init_agent" + echo "WHERE resource_id IN (" + printf "%b" "${in_clause}" + echo ")" + echo "GROUP BY workload_type" + echo "UNION ALL" + echo "SELECT 'serverless_compat_agent', workload_type," + echo " COUNT(*), COUNT(DISTINCT resource_id)" + echo "FROM udm.all.serverless_compat_agent" + echo "WHERE resource_id IN (" + printf "%b" "${in_clause}" + echo ")" + echo "GROUP BY workload_type" + echo "ORDER BY flavor, workload_type;" +} + +# --------------------------------------------------------------------------- +# Entry point +# --------------------------------------------------------------------------- +log "Collecting diagnostic logs → ${OUTPUT_FILE}" +run_checks | tee "${OUTPUT_FILE}" +generate_sql "${OUTPUT_FILE}" | tee -a "${OUTPUT_FILE}" + +echo "" +log "Done. Output: ${OUTPUT_FILE}" + +# Summarise what was found +found=$(python3 -c " +import re +text = open('${OUTPUT_FILE}').read() +rids = sorted(set(re.findall(r'resource_id[=:]\s*(//[^\s,\"\']+)', text))) +if rids: + print(f'{len(rids)} resource_id(s) found:') + print('\n'.join(' ' + r for r in rids)) +else: + print('No resource_ids found. Trigger the services and retry.') +" 2>/dev/null || echo "Could not parse output file.") + +log "${found}" + +if [[ "${found}" == "No resource_ids found."* ]]; then + exit 1 +fi diff --git a/scripts/serverless-compat-deploy/demo-both-rc.sh b/scripts/serverless-compat-deploy/demo-both-rc.sh new file mode 100755 index 0000000..3d48b4a --- /dev/null +++ b/scripts/serverless-compat-deploy/demo-both-rc.sh @@ -0,0 +1,801 @@ +#!/usr/bin/env bash +set -euo pipefail + +# --------------------------------------------------------------------------- +# demo-both-rc.sh — SVLS-9604 full RC validation +# +# End-to-end REDAPL staging test for BOTH serverless_init_agent AND +# serverless_compat_agent. +# +# What this script does: +# Phase 0 — Preflight: validate env, tools, and that both compat endpoints +# are already deployed and reachable (no deploy automation for compat) +# Phase 1 — Init RC deploy: discover → build → deploy via serverless-init-self-monitoring +# (runs npm run discover to regenerate apps.yml including Azure workloads, +# then build + deploy with AGENT_IMAGE=) +# Phase 2 — Baseline trigger (L0): one bounded pass across all workloads. +# Uses npm run test:once for init (exits nonzero on failure). +# Uses trigger.sh for compat. Blocks burst on any failure. +# Phase 3 — Burst test: serverless-init (npm run burst, WAVE_SIZES waves) +# Sets concurrency=1 on Cloud Run services before burst so each +# request forces a new instance. Restores concurrency after. +# Phase 4 — Burst test: serverless-compat (COMPAT_CONCURRENCY concurrent requests) +# Records application response latency (not EPRW/REDAPL ingestion lag). +# Phase 5 — MANUAL VALIDATION RUNBOOK: EPRW metrics and DDSQL cardinality queries. +# REDAPL query visibility lag must be measured manually — poll until rows +# appear and record wall-clock time. _modified_at shows when EPRW wrote +# the row, not when DDSQL first exposed it. +# +# Compat deploy: serverless-compat-self-monitoring has no root-level deployment +# automation. Deploy those functions manually before running this script. +# See: https://datadoghq.atlassian.net/wiki/spaces/SLS/pages/2977497119 +# +# Usage: +# export DD_API_KEY= # datad0g.com org key +# export AGENT_IMAGE=gcr.io/datadoghq/serverless-init: +# ./demo-both-rc.sh +# +# Skip flags (combine freely): +# SKIP_INIT_DEPLOY=true — skip discover + build + deploy for init +# SKIP_BURST=true — skip load stages; trigger once only (L0) +# LOAD_STAGE=L1 — compat burst stage: L0=1, L1=10, L2=50, L3=100 +# WAVE_SIZES=10,50,100 — init burst wave sizes (overrides defaults) +# +# Required before running: +# - Both compat functions deployed (azure + gcp). Set AZURE_FUNCTION_APP and +# GCP_FUNCTION_NAME to match your deployed endpoints. +# - gcloud CLI authenticated (gcloud auth login + application-default login) +# - az CLI authenticated (az login) +# - npm + tsx installed +# --------------------------------------------------------------------------- + +: "${DD_API_KEY:?DD_API_KEY must be set (datad0g.com staging org key)}" +: "${AGENT_IMAGE:?AGENT_IMAGE must be set (e.g. gcr.io/datadoghq/serverless-init:)}" + +DD_SITE="${DD_SITE:-datad0g.com}" +GCP_PROJECT="${GCP_PROJECT:-datadog-serverless-gcp-dev}" +GCP_REGION="${GCP_REGION:-us-central1}" +AZURE_SUBSCRIPTION_ID="${AZURE_SUBSCRIPTION_ID:-$(az account show --query id -o tsv 2>/dev/null || echo '')}" +AZURE_FUNCTION_APP="${AZURE_FUNCTION_APP:-nina-compat-inventory-node}" +GCP_FUNCTION_NAME="${GCP_FUNCTION_NAME:-nina-compat-inventory-nodejs}" + +SKIP_INIT_DEPLOY="${SKIP_INIT_DEPLOY:-false}" +SKIP_DISCOVER="${SKIP_DISCOVER:-false}" # set true to skip npm run discover (preserves cloud-run-v2) +SKIP_BURST="${SKIP_BURST:-false}" +LOAD_STAGE="${LOAD_STAGE:-L1}" # compat burst: L0=1 L1=10 L2=50 L3=100 +WAVE_SIZES="${WAVE_SIZES:-10,50,100}" # init burst waves + +# Largest wave size — used to set max-instances during burst so Cloud Run can actually scale out. +MAX_WAVE=$(echo "${WAVE_SIZES}" | tr ',' '\n' | sort -rn | head -1) + +INIT_SM_DIR="${INIT_SM_DIR:-${HOME}/go/src/github.com/DataDog/serverless-init-self-monitoring}" + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")" && pwd)" +RESULTS_DIR="${RESULTS_DIR:-/tmp/svls9604-rc-$(date +%Y%m%d-%H%M%S)}" +mkdir -p "${RESULTS_DIR}" + +log() { echo "[$(date -u '+%Y-%m-%dT%H:%M:%SZ')] $*" | tee -a "${RESULTS_DIR}/run.log"; } +header() { + local msg="$*" + echo "" | tee -a "${RESULTS_DIR}/run.log" + echo "==================================================" | tee -a "${RESULTS_DIR}/run.log" + echo " ${msg}" | tee -a "${RESULTS_DIR}/run.log" + echo "==================================================" | tee -a "${RESULTS_DIR}/run.log" +} +fail() { echo "ERROR: $*" >&2; exit 1; } + +# Portable millisecond timestamp (macOS-safe: date +%s%3N returns literal '3N' on macOS) +_ms_now() { python3 -c "import time; print(int(time.time() * 1000))"; } + +# --------------------------------------------------------------------------- +# Phase 0 — Preflight +# --------------------------------------------------------------------------- +header "Phase 0 — Preflight checks" + +[[ -d "${INIT_SM_DIR}" ]] || fail "serverless-init-self-monitoring not found at ${INIT_SM_DIR}. + Set INIT_SM_DIR or: gh repo clone DataDog/serverless-init-self-monitoring ${INIT_SM_DIR}" + +command -v npm >/dev/null 2>&1 || fail "npm not found" +command -v gcloud >/dev/null 2>&1 || fail "gcloud CLI not found" +command -v az >/dev/null 2>&1 || fail "az CLI not found" +command -v curl >/dev/null 2>&1 || fail "curl not found" +command -v python3 >/dev/null 2>&1 || fail "python3 not found" + +# tsx: prefer project-local, fall back to global +TSX_BIN="${INIT_SM_DIR}/node_modules/.bin/tsx" +if [[ ! -x "${TSX_BIN}" ]]; then + TSX_BIN=$(command -v tsx 2>/dev/null || echo "") +fi +[[ -n "${TSX_BIN}" ]] || fail "tsx not found — run: npm install inside ${INIT_SM_DIR} or npm install -g tsx" + +case "${LOAD_STAGE}" in + L0) COMPAT_CONCURRENCY=1 ;; + L1) COMPAT_CONCURRENCY=10 ;; + L2) COMPAT_CONCURRENCY=50 ;; + L3) COMPAT_CONCURRENCY=100 ;; + *) fail "Unknown LOAD_STAGE ${LOAD_STAGE} (use L0/L1/L2/L3)" ;; +esac + +if [[ "${LOAD_STAGE}" == "L3" ]]; then + log "WARNING: LOAD_STAGE=L3 (100 concurrent). Coordinate with RP Ingest before running." + log " See RFC Section 9 — do not advance past L2 without explicit approval." + read -r -p " Continue? [y/N] " confirm + [[ "${confirm}" =~ ^[Yy]$ ]] || exit 0 +fi + +# Preflight: verify compat endpoints are deployed and reachable +log "Checking Azure Function endpoint (${AZURE_FUNCTION_APP})..." +AZURE_BASE="https://${AZURE_FUNCTION_APP}.azurewebsites.net" +if ! curl -sf --max-time 15 "${AZURE_BASE}/api/httptest" -o /dev/null 2>/dev/null; then + fail "Azure Function not reachable: ${AZURE_BASE}/api/httptest + Deploy serverless-compat-self-monitoring azure_functions/compat_node manually first. + See: https://datadoghq.atlassian.net/wiki/spaces/SLS/pages/2977497119" +fi +log " Azure Function: OK" + +log "Checking GCP Cloud Function endpoint (${GCP_FUNCTION_NAME})..." +GCP_FN_URL=$(gcloud functions describe "${GCP_FUNCTION_NAME}" \ + --project="${GCP_PROJECT}" --region="${GCP_REGION}" \ + --format="value(httpsTrigger.url)" 2>/dev/null \ + || echo "https://${GCP_REGION}-${GCP_PROJECT}.cloudfunctions.net/${GCP_FUNCTION_NAME}") +GCP_TOKEN=$(gcloud auth print-identity-token 2>/dev/null || true) +if [[ -n "${GCP_TOKEN}" ]]; then + GCP_CURL_ARGS=(-H "Authorization: Bearer ${GCP_TOKEN}") +else + GCP_CURL_ARGS=() + log " WARNING: no GCP identity token — falling back to unauthenticated check" +fi +if ! curl -sf --max-time 15 "${GCP_CURL_ARGS[@]}" "${GCP_FN_URL}" -o /dev/null 2>/dev/null; then + fail "GCP Cloud Function not reachable: ${GCP_FN_URL} + Deploy serverless-compat-self-monitoring gcp_functions/nodejs manually first. + See: https://datadoghq.atlassian.net/wiki/spaces/SLS/pages/2977497119" +fi +log " GCP Cloud Function: OK" + +log "" +log "AGENT_IMAGE : ${AGENT_IMAGE}" +log "DD_SITE : ${DD_SITE}" +log "GCP_PROJECT : ${GCP_PROJECT}" +log "AZURE_FUNCTION_APP : ${AZURE_FUNCTION_APP}" +log "GCP_FUNCTION_NAME : ${GCP_FUNCTION_NAME}" +log "LOAD_STAGE : ${LOAD_STAGE} (compat concurrency=${COMPAT_CONCURRENCY})" +log "WAVE_SIZES (init) : ${WAVE_SIZES} (max wave = ${MAX_WAVE} → max-instances set to ${MAX_WAVE} during burst)" +log "INIT_SM_DIR : ${INIT_SM_DIR}" +log "SKIP_DISCOVER : ${SKIP_DISCOVER}" +log "Results dir : ${RESULTS_DIR}" + +START_TIME=$(date -u '+%Y-%m-%dT%H:%M:%SZ') +START_EPOCH=$(date +%s) + +# --------------------------------------------------------------------------- +# Phase 1 — serverless-init RC deploy +# +# Runs npm run discover to regenerate apps.yml from currently deployed resources +# (including Azure workloads — skipping this step leaves apps.yml stale with +# only GCP entries). Then builds all app images with AGENT_IMAGE= +# and deploys to GCP and Azure self-monitoring environments. +# +# Skippable when apps are already deployed (SKIP_INIT_DEPLOY=true). +# --------------------------------------------------------------------------- +if [[ "${SKIP_INIT_DEPLOY}" != "true" ]]; then + header "Phase 1 — serverless-init RC deploy (AGENT_IMAGE=${AGENT_IMAGE})" + + log "Running npm install..." + (cd "${INIT_SM_DIR}" && npm install --silent 2>&1 | tail -3) + + if [[ "${SKIP_DISCOVER}" == "true" ]]; then + log "Skipping npm run discover (SKIP_DISCOVER=true). Using existing apps.yml." + log " Preserves cloud-run-v2/sidecar entries that discover would remove." + else + log "Running npm run discover..." + log " WARNING: discover scans deploy/gcp/ and will REMOVE cloud-run-v2/sidecar entries" + log " because there is no deploy/gcp/cloud-run-v2/ directory. The next step" + log " (npm run ensure:v2) re-injects those entries automatically." + (cd "${INIT_SM_DIR}" && npm run discover 2>&1) \ + | tee "${RESULTS_DIR}/init-discover.log" + + # Verify cloud-run-v2 entries survived discovery. + # With deploy/gcp/cloud-run-v2/ stub files present, discover_apps.ts should find + # the cloud-run-v2 product and preserve those entries in apps.yml. If they are + # missing, the stub files may not be in place. + if ! grep -q 'product: cloud-run-v2' "${INIT_SM_DIR}/apps.yml" 2>/dev/null; then + log " WARNING: discover wiped cloud-run-v2 entries from apps.yml" + log " Attempting to restore via npm run ensure:v2..." + (cd "${INIT_SM_DIR}" && npm run ensure:v2 2>&1) \ + | tee -a "${RESULTS_DIR}/init-discover.log" + if ! grep -q 'product: cloud-run-v2' "${INIT_SM_DIR}/apps.yml" 2>/dev/null; then + fail "discover wiped cloud-run-v2 entries from apps.yml — deploy/gcp/cloud-run-v2/ stubs may be missing" + fi + log " cloud-run-v2 entries restored via ensure:v2" + else + log " cloud-run-v2 entries preserved in apps.yml" + fi + fi + + # Coverage preflight: fail if hard-required workloads are missing from apps.yml. + # Hard required: SI-01 (cloud-run/in-process), SI-02 (cloud-run/sidecar), + # SI-04 (cloud-run-v2/sidecar). + # Soft required (warn): SI-03 (jobs), SI-05..SI-09 (Azure — needs az login + Azure deploy). + log "Running coverage check (npm run coverage:check)..." + if ! (cd "${INIT_SM_DIR}" && WAVE_SIZES="${WAVE_SIZES}" npm run coverage:check 2>&1) \ + | tee "${RESULTS_DIR}/coverage-check.log"; then + fail "Coverage check failed — see ${RESULTS_DIR}/coverage-check.log for missing workloads." + fi + + log "Running npm run build (builds all app images with RC agent)..." + (cd "${INIT_SM_DIR}" && AGENT_IMAGE="${AGENT_IMAGE}" npm run build 2>&1) \ + | tee "${RESULTS_DIR}/init-build.log" + + log "Running npm run deploy..." + (cd "${INIT_SM_DIR}" && \ + AGENT_IMAGE="${AGENT_IMAGE}" \ + DD_API_KEY="${DD_API_KEY}" \ + DD_SITE="${DD_SITE}" \ + npm run deploy 2>&1) \ + | tee "${RESULTS_DIR}/init-deploy.log" + + log "Init RC deploy complete." +else + log "Skipping init deploy (SKIP_INIT_DEPLOY=true). Using existing deployment." + log "NOTE: run 'npm run discover && npm run ensure:v2' in ${INIT_SM_DIR} if apps.yml is stale." + + # Still run coverage check even when skipping deploy — the harness should know its gaps. + log "Running coverage check (npm run coverage:check)..." + (cd "${INIT_SM_DIR}" && WAVE_SIZES="${WAVE_SIZES}" npm run coverage:check 2>&1) \ + | tee "${RESULTS_DIR}/coverage-check.log" || true # warn only when not deploying +fi + +# --------------------------------------------------------------------------- +# Phase 2 — Baseline trigger (L0: one pass, all workloads) +# +# npm run test:once: sends exactly one request to each app (coldstart + busy). +# Exits nonzero if any request fails. Failure blocks burst phases. +# SI-03 (Cloud Run Jobs): needs Cloud Scheduler triggering the Jobs API — +# not an HTTP service, not covered by test:once. +# SI-04 (Cloud Run Functions Gen 2): covered as cloud-run-v2 product. +# --------------------------------------------------------------------------- +header "Phase 2 — Baseline trigger (L0: one pass, all workloads)" + +log "Triggering one pass across all init self-monitoring apps (npm run test:once)..." +log " SI-03 (Cloud Run Jobs) not covered — trigger via 'gcloud run jobs execute' separately." +(cd "${INIT_SM_DIR}" && npm run test:once 2>&1) \ + | tee "${RESULTS_DIR}/baseline-init.log" +log "Init baseline passed." + +# SI-03 (Cloud Run Jobs): trigger if JOB_NAME is set. +# Jobs have no HTTP endpoint — they are triggered via the Jobs API, not test:once. +log "SI-03 (Cloud Run Jobs): trigger if JOB_NAME is set..." +if [[ -n "${JOB_NAME:-}" ]]; then + gcloud run jobs execute "${JOB_NAME}" \ + --project="${GCP_PROJECT}" --region="${GCP_REGION}" \ + --wait 2>&1 | tee "${RESULTS_DIR}/job-trigger.log" + log " Cloud Run Job triggered" +else + log " WARNING: JOB_NAME not set — SI-03 not triggered. Set JOB_NAME= to cover SI-03." +fi + +log "Triggering compat functions once..." +GCP_PROJECT="${GCP_PROJECT}" \ +GCP_REGION="${GCP_REGION}" \ +AZURE_FUNCTION_APP="${AZURE_FUNCTION_APP}" \ +GCP_FUNCTION_NAME="${GCP_FUNCTION_NAME}" \ + "${SCRIPT_DIR}/trigger.sh" 2>&1 | tee "${RESULTS_DIR}/baseline-compat.log" + +BASELINE_END=$(date -u '+%Y-%m-%dT%H:%M:%SZ') +log "Baseline complete at ${BASELINE_END}." +log "Waiting 5 min for EPRW propagation before burst..." +sleep 300 + +# --------------------------------------------------------------------------- +# Phase 3 — Burst test: serverless-init +# +# Sets concurrency=1 on all self-monitoring Cloud Run services so each +# concurrent request forces a new instance (scale-out). Without this, +# Cloud Run's default concurrency=80 lets one instance absorb the whole wave, +# defeating the cardinality test. Restores concurrency after burst. +# +# Reports application response latency (not EPRW/REDAPL ingestion lag). +# REDAPL visibility is validated separately in Phase 5. +# --------------------------------------------------------------------------- +SELF_MON_SERVICES="" +if [[ "${SKIP_BURST}" != "true" ]]; then + header "Phase 3 — Burst test: serverless-init (WAVE_SIZES=${WAVE_SIZES})" + + log "Collecting self-monitoring Cloud Run services (label: selfmonitoring=true)..." + SELF_MON_SERVICES=$(gcloud run services list \ + --project="${GCP_PROJECT}" --region="${GCP_REGION}" \ + --filter="metadata.labels.selfmonitoring=true" \ + --format="value(metadata.name)" 2>/dev/null || echo "") + + if [[ -z "${SELF_MON_SERVICES}" ]]; then + log " WARNING: no self-monitoring Cloud Run services found (filter: labels.selfmonitoring=true)" + log " Burst will run but cannot force scale-out — cardinality result may not be meaningful." + fi + + # Save original concurrency and max-instances per service, then set burst values. + # concurrency=1 → each concurrent HTTP request requires a separate instance. + # max-instances → set to max(WAVE_SIZES) so Cloud Run can actually scale out. + # + # Without increasing max-instances (default=1 in in-process.yaml and gcp.ts:154), + # a 100-request wave still hits one instance regardless of concurrency. + declare -A ORIG_CONCURRENCY + declare -A ORIG_MAX_INSTANCES + + if [[ -n "${SELF_MON_SERVICES}" ]]; then + log "Saving original settings and setting concurrency=1 max-instances=${MAX_WAVE}..." + while IFS= read -r svc; do + [[ -z "${svc}" ]] && continue + + orig_conc=$(gcloud run services describe "${svc}" \ + --project="${GCP_PROJECT}" --region="${GCP_REGION}" \ + --format="value(spec.template.spec.containerConcurrency)" 2>/dev/null || echo "80") + orig_max=$(gcloud run services describe "${svc}" \ + --project="${GCP_PROJECT}" --region="${GCP_REGION}" \ + --format="value(spec.template.metadata.annotations['autoscaling.knative.dev/maxScale'])" \ + 2>/dev/null || echo "1") + + ORIG_CONCURRENCY["${svc}"]="${orig_conc:-80}" + ORIG_MAX_INSTANCES["${svc}"]="${orig_max:-1}" + + gcloud run services update "${svc}" \ + --concurrency=1 \ + --max-instances="${MAX_WAVE}" \ + --project="${GCP_PROJECT}" --region="${GCP_REGION}" --quiet 2>/dev/null || true + log " ${svc}: concurrency ${orig_conc:-?} → 1, max-instances ${orig_max:-?} → ${MAX_WAVE}" + done <<< "${SELF_MON_SERVICES}" + fi + + log "Running npm run burst (${WAVE_SIZES} waves, ${MAX_WAVE} max-instances per service)..." + log " HTTP requests exercised: $(echo "${SELF_MON_SERVICES}" | grep -c . || echo 0) services × sum(${WAVE_SIZES}) waves" + burst_exit=0 + (cd "${INIT_SM_DIR}" && \ + WAVE_SIZES="${WAVE_SIZES}" \ + BURST_WAIT_MS=90000 \ + DD_SITE="${DD_SITE}" \ + npm run burst 2>&1) \ + | tee "${RESULTS_DIR}/burst-init.log" || burst_exit=$? + + log "Restoring original concurrency and max-instances on Cloud Run services..." + if [[ -n "${SELF_MON_SERVICES}" ]]; then + while IFS= read -r svc; do + [[ -z "${svc}" ]] && continue + restore_conc="${ORIG_CONCURRENCY["${svc}"]:-80}" + restore_max="${ORIG_MAX_INSTANCES["${svc}"]:-1}" + gcloud run services update "${svc}" \ + --concurrency="${restore_conc}" \ + --max-instances="${restore_max}" \ + --project="${GCP_PROJECT}" --region="${GCP_REGION}" --quiet 2>/dev/null || true + log " ${svc}: restored concurrency=${restore_conc} max-instances=${restore_max}" + done <<< "${SELF_MON_SERVICES}" + fi + + if [[ "${burst_exit}" -ne 0 ]]; then + fail "Init burst had request failures (exit ${burst_exit}) — check ${RESULTS_DIR}/burst-init.log" + fi + log "Init burst complete." +else + log "Skipping burst (SKIP_BURST=true)" +fi + +# --------------------------------------------------------------------------- +# Phase 4 — Burst test: serverless-compat +# +# Fires COMPAT_CONCURRENCY concurrent requests at each compat function. +# Records per-request timing to CSV; computes p50/p95/p99 APPLICATION +# RESPONSE LATENCY. This is HTTP round-trip latency — not EPRW ingestion +# lag or REDAPL visibility lag. Those are measured in Phase 5. +# --------------------------------------------------------------------------- +if [[ "${SKIP_BURST}" != "true" ]]; then + header "Phase 4 — Burst test: serverless-compat (${LOAD_STAGE}=${COMPAT_CONCURRENCY} concurrent)" + log " NOTE: latency stats below are APPLICATION RESPONSE LATENCY (HTTP round-trip)." + log " REDAPL visibility lag is measured separately in Phase 5 using _modified_at." + + COMPAT_BURST_DIR="${RESULTS_DIR}/compat-burst" + mkdir -p "${COMPAT_BURST_DIR}" + + _trigger_one() { + local id="$1" url="$2" out_file="$3" auth_header="${4:-}" + local t_start elapsed status + + t_start=$(_ms_now) + local curl_args=(-sf -o /dev/null -w "%{http_code}" --max-time 30) + [[ -n "${auth_header}" ]] && curl_args+=(-H "${auth_header}") + status=$(curl "${curl_args[@]}" "${url}" 2>/dev/null || echo "000") + elapsed=$(( $(_ms_now) - t_start )) + echo "${id},${status},${elapsed}" >> "${out_file}" + } + + _run_wave() { + local label="$1" url="$2" auth_header="${3:-}" + local wave_file="${COMPAT_BURST_DIR}/${label}.csv" + echo "id,http_status,elapsed_ms" > "${wave_file}" + + log "Launching ${COMPAT_CONCURRENCY} concurrent requests → ${label}..." + local pids=() + for i in $(seq 1 "${COMPAT_CONCURRENCY}"); do + _trigger_one "${i}" "${url}" "${wave_file}" "${auth_header}" & + pids+=($!) + done + for pid in "${pids[@]}"; do wait "${pid}" 2>/dev/null || true; done + + python3 - "${wave_file}" "${label}" <<'PYEOF' +import sys, csv, statistics +rows = list(csv.DictReader(open(sys.argv[1]))) +label = sys.argv[2] +total = len(rows) +success = sum(1 for r in rows if r['http_status'].startswith('2')) +fail = total - success +lats = [int(r['elapsed_ms']) for r in rows if r['elapsed_ms'].isdigit()] +p50 = statistics.median(lats) if lats else 0 +p95 = sorted(lats)[max(0, int(len(lats)*0.95)-1)] if lats else 0 +p99 = sorted(lats)[max(0, int(len(lats)*0.99)-1)] if lats else 0 +print(f" {label}: {total} total | {success} OK | {fail} failed") +print(f" App response latency: p50={p50}ms p95={p95}ms p99={p99}ms") +if fail > 0: + sys.exit(1) +PYEOF + } + + GCP_TOKEN=$(gcloud auth print-identity-token 2>/dev/null || true) + GCP_AUTH_HEADER="" + [[ -n "${GCP_TOKEN}" ]] && GCP_AUTH_HEADER="Authorization: Bearer ${GCP_TOKEN}" + + compat_burst_exit=0 + _run_wave "azure-function" "${AZURE_BASE}/api/httptest" || compat_burst_exit=$? + _run_wave "gcp-cloud-function-gen1" "${GCP_FN_URL}" "${GCP_AUTH_HEADER}" || compat_burst_exit=$? + + if [[ "${compat_burst_exit}" -ne 0 ]]; then + fail "Compat burst had request failures — check ${COMPAT_BURST_DIR}/" + fi + + log "Compat burst complete. Waiting 5 min for EPRW propagation..." + sleep 300 +else + log "Skipping compat burst (SKIP_BURST=true)" +fi + +END_TIME=$(date -u '+%Y-%m-%dT%H:%M:%SZ') + +# --------------------------------------------------------------------------- +# Phase 4b — EPRW metric gates (automated, via Datadog API) +# +# Polls EPRW accepted-write and rejection metrics for the run window. +# Requires DD_APP_KEY in addition to DD_API_KEY. Skips gracefully if absent. +# --------------------------------------------------------------------------- +header "Phase 4b — EPRW metric gates" +eprw_exit=0 +if [[ -n "${DD_APP_KEY:-}" ]]; then + (cd "${INIT_SM_DIR}" && \ + DD_API_KEY="${DD_API_KEY}" \ + DD_APP_KEY="${DD_APP_KEY}" \ + DD_SITE="${DD_SITE}" \ + FROM="${START_TIME}" \ + npm run poll:eprw 2>&1) \ + | tee "${RESULTS_DIR}/eprw-poll.log" || eprw_exit=$? + + if [[ "${eprw_exit}" -ne 0 ]]; then + log "WARNING: EPRW metric gates failed — see ${RESULTS_DIR}/eprw-poll.log" + log " This is non-fatal; verify manually using Phase 5 queries." + fi +else + log "DD_APP_KEY not set — skipping automated EPRW metric poll." + log " Set DD_APP_KEY to enable: accepted write count gates and rejection metric gates." + log " Verify EPRW metrics manually using the Phase 5 queries below." +fi +# --------------------------------------------------------------------------- +# Phase 5 — MANUAL VALIDATION RUNBOOK +# +# The following queries must be run manually in go/redapl → Queries → SQL. +# This script cannot execute DDSQL queries automatically. +# +# REDAPL query visibility lag (the five-minute SLO): +# _modified_at tells you when EPRW wrote the row, not when DDSQL first +# exposed it. To measure actual query visibility lag: +# 1. Record wall-clock time when you first run Query C and rows appear. +# 2. Subtract START_TIME (printed below) from that wall-clock time. +# 3. That difference is the REDAPL query visibility lag for this run. +# Poll Query C every 60s after the trigger until rows appear. +# --------------------------------------------------------------------------- +header "Phase 5 — MANUAL VALIDATION RUNBOOK (go/redapl → Queries → SQL)" + +cat < distinct_resources → per-instance field leaked into key +SELECT workload_type, deployment_model, + COUNT(*) AS total_rows, + COUNT(DISTINCT resource_id) AS distinct_resources +FROM udm.all.serverless_init_agent +GROUP BY workload_type, deployment_model +ORDER BY workload_type, deployment_model; + +-- B. serverless_compat_agent: exactly 1 row per function regardless of burst size +-- PASS: azure_function=1, gcp_cloud_function_gen1=1 +-- FAIL: any total_rows > 1 → cold-start fan-out reaching REDAPL +SELECT workload_type, + COUNT(*) AS total_rows, + COUNT(DISTINCT resource_id) AS distinct_resources +FROM udm.all.serverless_compat_agent +GROUP BY workload_type +ORDER BY workload_type; + +-- C. REDAPL visibility: poll this every 60s until rows appear. +-- Record wall-clock time when rows first appear — subtract ${START_TIME} +-- to get REDAPL query visibility lag (must be ≤ 5 min at p95). +-- _modified_at = when EPRW wrote the row (not when DDSQL exposed it). +SELECT 'serverless_init_agent' AS flavor, resource_id, workload_type, + _modified_at, + TIMESTAMPDIFF(MINUTE, TIMESTAMP '${START_TIME}', _modified_at) AS eprw_write_lag_min +FROM udm.all.serverless_init_agent +WHERE _modified_at >= TIMESTAMP '${START_TIME}' +UNION ALL +SELECT 'serverless_compat_agent', resource_id, workload_type, + _modified_at, + TIMESTAMPDIFF(MINUTE, TIMESTAMP '${START_TIME}', _modified_at) +FROM udm.all.serverless_compat_agent +WHERE _modified_at >= TIMESTAMP '${START_TIME}' +ORDER BY flavor, eprw_write_lag_min ASC; + +-- D. _key correctness: must equal SanitizeString(resource_id), not uuid or composite. +SELECT _key, resource_id, workload_type +FROM udm.all.serverless_init_agent +WHERE _modified_at >= TIMESTAMP '${START_TIME}' +LIMIT 20; + +SELECT _key, resource_id, workload_type +FROM udm.all.serverless_compat_agent +WHERE _modified_at >= TIMESTAMP '${START_TIME}' +LIMIT 10; + +-- E. Crawler joins: both gcp_run_service AND gcp_run_revision checked. +-- The schema relationship is gcp_run_revision, sourced from +-- /metadata.key_overrides.gcp_run_revision_key with on_empty fallback to resource_id. +-- If the decoder provides a revision CCRID: gcp_run_revision resolves. +-- If not (service-level CCRID fallback): gcp_run_service resolves. +-- PASS: at least one of the two is non-NULL per row. +-- FAIL (R6): both NULL → CCRID format matches neither crawler table. +SELECT a.resource_id, a.workload_type, a._key, + svc._key AS gcp_run_service_key, + rev._key AS gcp_run_revision_key +FROM udm.all.serverless_init_agent a +LEFT JOIN udm.all.gcp_run_service svc ON a._key = svc._key +LEFT JOIN udm.all.gcp_run_revision rev ON a._key = rev._key +WHERE a.workload_type = 'cloud_run_service' + AND a._modified_at >= TIMESTAMP '${START_TIME}' +LIMIT 20; + +SELECT a.resource_id, a.workload_type, a._key, + c._key AS crawler_key +FROM udm.all.serverless_init_agent a +LEFT JOIN udm.all.azure_container_app c ON a._key = c._key +WHERE a.workload_type = 'azure_container_app' + AND a._modified_at >= TIMESTAMP '${START_TIME}' +LIMIT 20; + +SELECT a.resource_id, a.workload_type, a._key, + c._key AS crawler_key +FROM udm.all.serverless_init_agent a +LEFT JOIN udm.all.azure_app_service c ON a._key = c._key +WHERE a.workload_type = 'azure_app_service' + AND a._modified_at >= TIMESTAMP '${START_TIME}' +LIMIT 20; + +-- F. Legacy datadog_agent secondary write (confirm payload reached EPRW at all). +-- UUID-keyed; one row per cold start. NOT the per-flavor table rows. +SELECT _key AS uuid, hostname, agent_version, install_method_tool, _first_seen_at +FROM udm.all.datadog_agent +WHERE install_method_tool IN ('serverless-init', 'serverless-compat') + AND _first_seen_at >= TIMESTAMP '${START_TIME}' +ORDER BY _first_seen_at DESC +LIMIT 20; + +DDSQL + +# --------------------------------------------------------------------------- +# Machine-readable report +# Counts and pass/fail for each RFC workload — written to RESULTS_DIR/report.json +# --------------------------------------------------------------------------- +END_EPOCH=$(date +%s) +DURATION=$(( END_EPOCH - START_EPOCH )) + +python3 - "${RESULTS_DIR}" "${START_TIME}" "${END_TIME}" \ + "${AGENT_IMAGE}" "${LOAD_STAGE}" "${WAVE_SIZES}" \ + "${COMPAT_CONCURRENCY}" "${MAX_WAVE}" <<'REPORT_PY' +import json, sys, os, re +from datetime import datetime, timezone + +results_dir, start_time, end_time, agent_image, load_stage, wave_sizes, \ + compat_concurrency, max_wave = sys.argv[1:] + +def count_lines(path): + try: + with open(path) as f: + return sum(1 for _ in f) + except FileNotFoundError: + return None + +# Parse burst summary from log (count succeeded/failed lines) +burst_log = os.path.join(results_dir, 'burst-init.log') +burst_ok = burst_fail = 0 +if os.path.exists(burst_log): + with open(burst_log) as f: + for line in f: + m = re.search(r'✓ (\d+)/\d+ succeeded', line) + if m: + burst_ok += int(m.group(1)) + m2 = re.search(r'✗ (\d+) failed', line) + if m2: + burst_fail += int(m2.group(1)) + +# Coverage check result +coverage_log = os.path.join(results_dir, 'coverage-check.log') +coverage_hard_failed = False +if os.path.exists(coverage_log): + with open(coverage_log) as f: + content = f.read() + coverage_hard_failed = 'HARD FAIL' in content + +report = { + 'run': { + 'start': start_time, + 'end': end_time, + 'agent_image': agent_image, + 'load_stage': load_stage, + 'wave_sizes': wave_sizes, + 'max_instances_during_burst': int(max_wave), + }, + 'counts': { + 'note': 'HTTP requests only — instance starts and REDAPL rows require manual verification', + 'init_burst_http_ok': burst_ok, + 'init_burst_http_fail': burst_fail, + 'compat_burst_concurrency': int(compat_concurrency), + 'instance_starts_observed': 'not measured — requires Cloud Run log query', + 'eprw_accepted_writes': 'see eprw-poll.log or Phase 5 metrics', + 'eprw_rejected_writes': 'see eprw-poll.log or Phase 5 metrics', + 'distinct_resource_ids': 'not measured — requires DDSQL query (Phase 5)', + 'redapl_rows': 'not measured — requires DDSQL query (Phase 5)', + }, + 'rfc_workloads': [ + {'id': 'SI-01', 'workload': 'cloud_run_service / in-container', 'automated': True, 'result': 'HTTP triggered — REDAPL rows not auto-verified'}, + {'id': 'SI-02', 'workload': 'cloud_run_service / sidecar', 'automated': True, 'result': 'HTTP triggered — REDAPL rows not auto-verified'}, + {'id': 'SI-03', 'workload': 'cloud_run_job / in-container', 'automated': False, 'result': 'NOT TESTED — no HTTP endpoint; needs gcloud run jobs execute'}, + {'id': 'SI-04', 'workload': 'cloud_function_gen2 / sidecar', 'automated': True, 'result': 'HTTP triggered — REDAPL rows not auto-verified'}, + {'id': 'SI-05', 'workload': 'azure_container_app / in-container', 'automated': False, 'result': 'NOT TESTED — Azure not in apps.yml without discover + az login'}, + {'id': 'SI-06', 'workload': 'azure_container_app / sidecar', 'automated': False, 'result': 'NOT TESTED — Azure not in apps.yml without discover + az login'}, + {'id': 'SI-07', 'workload': 'azure_app_service / in-container', 'automated': False, 'result': 'NOT TESTED — Azure not in apps.yml without discover + az login'}, + {'id': 'SI-08', 'workload': 'azure_app_service / SITECONTAINERS', 'automated': False, 'result': 'NOT TESTED — Azure not in apps.yml without discover + az login'}, + {'id': 'SI-09', 'workload': 'azure_app_service / linux-code', 'automated': False, 'result': 'NOT TESTED — Azure not in apps.yml without discover + az login'}, + {'id': 'SC-01', 'workload': 'azure_function / compat', 'automated': True, 'result': 'HTTP triggered — REDAPL rows not auto-verified'}, + {'id': 'SC-02', 'workload': 'gcp_cloud_function_gen1 / compat', 'automated': True, 'result': 'HTTP triggered — REDAPL rows not auto-verified'}, + ], + 'unverified': [ + 'actual instance starts (not measured)', + 'inventory payload count per service (not measured)', + 'EPRW accepted writes per resource_id (requires poll:eprw)', + 'EPRW rejection counts (requires poll:eprw)', + 'REDAPL row count per resource_id (requires DDSQL, Phase 5)', + 'REDAPL visibility lag wall-clock time (requires polling, Phase 5)', + 'RFC 7.2 restart identity row stability (requires DDSQL)', + 'RFC 7.3 config upgrade row update (requires test:ordering + DDSQL)', + 'RFC 7.4 flip-flop convergence (requires test:ordering + DDSQL)', + 'crawler join correctness (requires DDSQL, Phase 5 Query E)', + 'TTL expiration and reactivation (not scripted)', + 'FleetQuerier and UI verification (not scripted)', + ], + 'coverage_preflight_hard_failed': coverage_hard_failed, +} + +out = os.path.join(results_dir, 'report.json') +with open(out, 'w') as f: + json.dump(report, f, indent=2) +print(f"Report written to: {out}") +REPORT_PY + +# --------------------------------------------------------------------------- +# Summary +# --------------------------------------------------------------------------- +header "RC validation complete" + +log "Start : ${START_TIME}" +log "Baseline complete : ${BASELINE_END:-skipped}" +log "End : ${END_TIME}" +log "Duration : ${DURATION}s" +log "AGENT_IMAGE : ${AGENT_IMAGE}" +log "LOAD_STAGE : ${LOAD_STAGE} (compat concurrency=${COMPAT_CONCURRENCY})" +log "WAVE_SIZES (init) : ${WAVE_SIZES} (max-instances set to ${MAX_WAVE} during burst)" +log "Results dir : ${RESULTS_DIR}/" +log "" +log "Coverage (HTTP-triggered only — REDAPL row count requires manual DDSQL verification):" +log " SI-01 cloud_run_service / in-container HTTP ✓ | instances NOT measured | REDAPL rows NOT auto-verified" +log " SI-02 cloud_run_service / sidecar HTTP ✓ | instances NOT measured | REDAPL rows NOT auto-verified" +log " SI-03 cloud_run_job / in-container NOT TESTED (no HTTP endpoint — needs gcloud run jobs execute)" +log " SI-04 cloud_function_gen2 / sidecar HTTP ✓ | instances NOT measured | REDAPL rows NOT auto-verified" +log " SI-05 azure_container_app / in-container NOT TESTED (Azure not in apps.yml without discover + az login)" +log " SI-06 azure_container_app / sidecar NOT TESTED (Azure not in apps.yml without discover + az login)" +log " SI-07 azure_app_service / in-container NOT TESTED (Azure not in apps.yml without discover + az login)" +log " SI-08 azure_app_service / SITECONTAINERS NOT TESTED (Azure not in apps.yml without discover + az login)" +log " SI-09 azure_app_service / linux-code NOT TESTED (Azure not in apps.yml without discover + az login)" +log " SC-01 azure_function HTTP ✓ | REDAPL rows NOT auto-verified" +log " SC-02 gcp_cloud_function_gen1 HTTP ✓ | REDAPL rows NOT auto-verified" +log "" +log "What this script does NOT verify automatically:" +log " - Actual instance starts (metric: container/instance_count or Cloud Run logs)" +log " - Inventory payload count per service" +log " - EPRW accepted write count (Phase 4b; requires DD_APP_KEY)" +log " - REDAPL row count == distinct resource_ids (requires Phase 5 DDSQL queries)" +log " - REDAPL visibility lag wall-clock time (poll Phase 5 Query C)" +log " - RFC 7.2/7.3/7.4 ordering guarantees (run: npm run test:ordering)" +log " - Crawler join correctness (Phase 5 Query E)" +log "" +log "Next steps:" +log " Full DDSQL query set: ${SCRIPT_DIR}/check-redapl.sh" +log " Log diagnostics: GCP_PROJECT=${GCP_PROJECT} ${SCRIPT_DIR}/check-logs.sh" +log " Ordering test: (cd ${INIT_SM_DIR} && npm run test:ordering)" +log "" +log "RFC approval gates (verify manually using Phase 5 queries above):" +log " Gate 1: rows == resources for every workload_type in both tables (Queries A + B)" +log " Gate 2: REDAPL query visibility lag ≤ 5 min — poll Query C until rows appear" +log " Gate 3: crawler_key NOT NULL for all rows (Query E)" +log " Gate 4: all rejection metrics zero (Phase 4b or manual)" +log " Gate 5: no EPRW CPU/memory/error regression during burst" +log "" +log "Machine-readable report: ${RESULTS_DIR}/report.json" + +# Write a concise JSON summary for quick pass/fail inspection. +# This is separate from the detailed report.json produced by the Python block above. +_cloud_run_v2_preserved=$(grep -q 'product: cloud-run-v2' "${INIT_SM_DIR}/apps.yml" 2>/dev/null && echo true || echo false) +_si03_covered=$([ -n "${JOB_NAME:-}" ] && echo true || echo false) + +cat > "${RESULTS_DIR}/rc-summary.json" </dev/null"]) if not os.environ.get("DD_API_KEY") or os.environ.get("DD_SITE") != "datad0g.com": raise RuntimeError("runner must be invoked through dd-auth for datad0g.com") - if args.profile in ("gcp", "gcp-sanity"): + if args.profile in ("gcp", "gcp-sanity", "gcp-baseline"): account=run(["gcloud","auth","list","--filter=status:ACTIVE","--format=value(account)"],capture=True) if not account: raise RuntimeError("gcloud has no active account") @@ -187,7 +187,9 @@ def ensure_registry(project, region): def env_list(name): return {"DD_API_KEY":os.environ["DD_API_KEY"],"DD_SITE":"datad0g.com","DD_ENV":RUN_ENV, - "DD_SERVICE":name,"DD_SERVERLESS_DIAGNOSTIC_INFO":"true","DD_LOG_LEVEL":"debug"} + "DD_SERVICE":name,"DD_SERVERLESS_DIAGNOSTIC_INFO":"true","DD_LOG_LEVEL":"debug", + "DD_SERVERLESS_INIT_INVENTORY_ENABLED":"true", + "DD_SERVERLESS_COMPAT_INVENTORY_ENABLED":"true"} def env_arg(values): return ",".join(f"{k}={v}" for k,v in values.items()) @@ -745,7 +747,7 @@ def run_azure(args, resources, run_id, run_dir): existing={} if manifest_path.exists(): previous=json.loads(manifest_path.read_text()) - if previous.get("profile") in ("azure","azure-sanity") and previous.get("agent_image")==agent_image: + if previous.get("profile") in ("azure","azure-sanity","azure-baseline") and previous.get("agent_image")==agent_image: candidates=[item for item in previous.get("resources",[]) if item.get("agent_image")==agent_image and item.get("endpoint")] with concurrent.futures.ThreadPoolExecutor(max_workers=min(20,len(candidates) or 1)) as pool: @@ -848,7 +850,7 @@ def run_gcp(args, resources, run_id, run_dir): def main(): global RUN_ENV, RUN_STARTED_AT parser=argparse.ArgumentParser() - parser.add_argument("--profile",choices=["gcp","azure","gcp-sanity","azure-sanity"],required=True) + parser.add_argument("--profile",choices=["gcp","azure","gcp-sanity","azure-sanity","gcp-baseline","azure-baseline"],required=True) parser.add_argument("--project",default=os.environ.get("GCP_PROJECT","datadog-serverless-gcp-demo")) parser.add_argument("--region",default=os.environ.get("GCP_REGION","us-central1")) parser.add_argument("--azure-resource-group",default=os.environ.get("AZURE_RESOURCE_GROUP","dd-serverless-test-aas")) @@ -894,7 +896,7 @@ def main(): run_dir=pathlib.Path(os.environ.get("RESULTS_DIR",f"/tmp/svls9604-{run_id}")); run_dir.mkdir(parents=True,exist_ok=True) run_dir.chmod(0o700) failure_exit=0 - if args.profile in ("gcp", "gcp-sanity"): + if args.profile in ("gcp", "gcp-sanity", "gcp-baseline"): deployed=run_gcp(args,resources,run_id,run_dir) failed=[r for r in deployed if r.get("baseline",{}).get("status") not in ("ok","executed")] print(f"GCP profile deployed {len(deployed)}/{len(resources)}; baseline failures={len(failed)}") From 511c4eda773839efcc4bc262c207dee63c032148 Mon Sep 17 00:00:00 2001 From: Nina Rei Date: Wed, 2 Sep 2026 18:15:43 -0400 Subject: [PATCH 4/9] chore: run cargo fmt on inventory.rs --- .../src/inventory.rs | 144 ++++++++++++++---- 1 file changed, 113 insertions(+), 31 deletions(-) diff --git a/crates/datadog-serverless-compat/src/inventory.rs b/crates/datadog-serverless-compat/src/inventory.rs index e6a6011..1381700 100644 --- a/crates/datadog-serverless-compat/src/inventory.rs +++ b/crates/datadog-serverless-compat/src/inventory.rs @@ -355,13 +355,20 @@ fn parse_rg_from_owner_name(owner_name: &str) -> Option { let without_webspace = stripped.strip_suffix("webspace")?; let last_dash = without_webspace.rfind('-')?; let rg = &without_webspace[..last_dash]; - if rg.is_empty() { None } else { Some(rg.to_string()) } + if rg.is_empty() { + None + } else { + Some(rg.to_string()) + } } fn build_cloud_function_identity() -> (String, String) { // Gen2 Cloud Run Functions set FUNCTION_TARGET alongside K_SERVICE. // These belong in serverless_init_agent, not serverless_compat_agent. - if env::var("FUNCTION_TARGET").map(|v| !v.is_empty()).unwrap_or(false) { + if env::var("FUNCTION_TARGET") + .map(|v| !v.is_empty()) + .unwrap_or(false) + { return (String::new(), String::new()); } @@ -403,7 +410,9 @@ async fn fetch_gcp_metadata_value( ) -> Option { let client = create_reqwest_client_builder() .and_then(|b| { - b.timeout(Duration::from_secs(2)).build().map_err(Into::into) + b.timeout(Duration::from_secs(2)) + .build() + .map_err(Into::into) }) .ok()?; @@ -416,7 +425,10 @@ async fn fetch_gcp_metadata_value( .ok()?; if !resp.status().is_success() { - warn!("inventory: GCP metadata server returned {} for {label}", resp.status()); + warn!( + "inventory: GCP metadata server returned {} for {label}", + resp.status() + ); return None; } @@ -429,14 +441,21 @@ async fn fetch_gcp_metadata_value( async fn fetch_gcp_region_from_metadata() -> Option { // Response: "projects//regions/" fetch_gcp_metadata_value("instance/region", "region", |body| { - body.split('/').next_back().filter(|s| !s.is_empty()).map(str::to_string) + body.split('/') + .next_back() + .filter(|s| !s.is_empty()) + .map(str::to_string) }) .await } async fn fetch_gcp_project_from_metadata() -> Option { fetch_gcp_metadata_value("project/project-id", "project-id", |body| { - if body.is_empty() { None } else { Some(body.to_string()) } + if body.is_empty() { + None + } else { + Some(body.to_string()) + } }) .await } @@ -484,7 +503,11 @@ fn enrich_azure_function_fields(metadata: &mut serde_json::Value) { let runtime = env::var("DD_SERVERLESS_COMPAT_RUNTIME") .ok() .filter(|s| !s.is_empty()) - .or_else(|| env::var("FUNCTIONS_WORKER_RUNTIME").ok().filter(|s| !s.is_empty())); + .or_else(|| { + env::var("FUNCTIONS_WORKER_RUNTIME") + .ok() + .filter(|s| !s.is_empty()) + }); if let Some(rt) = runtime { metadata["runtime"] = serde_json::Value::String(rt); } @@ -495,7 +518,9 @@ fn enrich_azure_function_fields(metadata: &mut serde_json::Value) { .ok() .filter(|s| !s.is_empty()) .or_else(|| { - env::var("FUNCTIONS_WORKER_RUNTIME_VERSION").ok().filter(|s| !s.is_empty()) + env::var("FUNCTIONS_WORKER_RUNTIME_VERSION") + .ok() + .filter(|s| !s.is_empty()) }); if let Some(v) = runtime_ver { metadata["serverless_compat_runtime_version"] = serde_json::Value::String(v); @@ -562,8 +587,7 @@ fn detect_gcp_gen1_runtime() -> (String, String) { } fn build_client(https_proxy: Option<&str>) -> Result> { - let mut builder = - create_reqwest_client_builder()?.timeout(Duration::from_secs(10)); + let mut builder = create_reqwest_client_builder()?.timeout(Duration::from_secs(10)); if let Some(proxy) = https_proxy { builder = builder.proxy(reqwest::Proxy::https(proxy)?); @@ -596,12 +620,18 @@ mod tests { #[test] fn azure_function_workload_type() { - assert_eq!(supported_workload_type(&EnvironmentType::AzureFunction), Some("azure_function")); + assert_eq!( + supported_workload_type(&EnvironmentType::AzureFunction), + Some("azure_function") + ); } #[test] fn cloud_function_workload_type() { - assert_eq!(supported_workload_type(&EnvironmentType::CloudFunction), Some("cloud_function")); + assert_eq!( + supported_workload_type(&EnvironmentType::CloudFunction), + Some("cloud_function") + ); } // ── Azure Function identity ────────────────────────────────────────────── @@ -618,7 +648,10 @@ mod tests { let (id, name) = build_azure_function_identity(); assert_eq!(name, "my-func-app"); - assert_eq!(id, "/subscriptions/abc123/resourcegroups/my-rg/providers/microsoft.web/sites/my-func-app"); + assert_eq!( + id, + "/subscriptions/abc123/resourcegroups/my-rg/providers/microsoft.web/sites/my-func-app" + ); unsafe { env::remove_var("WEBSITE_SITE_NAME"); @@ -634,13 +667,19 @@ mod tests { unsafe { env::set_var("WEBSITE_SITE_NAME", "my-func"); env::remove_var("WEBSITE_RESOURCE_GROUP"); - env::set_var("WEBSITE_OWNER_NAME", "sub123+my-resource-group-westus2webspace-Linux"); + env::set_var( + "WEBSITE_OWNER_NAME", + "sub123+my-resource-group-westus2webspace-Linux", + ); } let (id, name) = build_azure_function_identity(); assert_eq!(name, "my-func"); - assert!(id.contains("/my-resource-group/"), "expected RG in id: {id}"); + assert!( + id.contains("/my-resource-group/"), + "expected RG in id: {id}" + ); unsafe { env::remove_var("WEBSITE_SITE_NAME"); @@ -658,7 +697,10 @@ mod tests { } let (id, _name) = build_azure_function_identity(); - assert!(id.is_empty(), "missing WEBSITE_SITE_NAME must produce empty resource_id"); + assert!( + id.is_empty(), + "missing WEBSITE_SITE_NAME must produce empty resource_id" + ); unsafe { env::remove_var("WEBSITE_RESOURCE_GROUP"); @@ -731,8 +773,14 @@ mod tests { let (id, name) = build_cloud_function_identity(); - assert!(id.is_empty(), "Gen2 must produce empty resource_id; got: {id}"); - assert!(name.is_empty(), "Gen2 must produce empty resource_name; got: {name}"); + assert!( + id.is_empty(), + "Gen2 must produce empty resource_id; got: {id}" + ); + assert!( + name.is_empty(), + "Gen2 must produce empty resource_name; got: {name}" + ); unsafe { env::remove_var("K_SERVICE"); @@ -781,8 +829,14 @@ mod tests { let (id, name) = build_cloud_function_identity(); - assert!(id.is_empty(), "incomplete identity must produce empty resource_id"); - assert_eq!(name, "my-fn", "resource_name should still be set for metadata retry"); + assert!( + id.is_empty(), + "incomplete identity must produce empty resource_id" + ); + assert_eq!( + name, "my-fn", + "resource_name should still be set for metadata retry" + ); unsafe { env::remove_var("FUNCTION_NAME"); @@ -821,12 +875,18 @@ mod tests { assert_eq!(meta["flavor"], "serverless-compat"); assert_eq!(meta["workload_type"], "azure_function"); assert_eq!(meta["report_reason"], "startup"); - assert_eq!(meta["resource_id"], "//microsoft.azure/functionApps/sub/rg/my-func"); + assert_eq!( + meta["resource_id"], + "//microsoft.azure/functionApps/sub/rg/my-func" + ); assert_eq!(meta["resource_name"], "my-func"); assert!(meta.contains_key("serverless_compat_version")); // UUID must NOT appear inside agent_metadata. - assert!(!meta.contains_key("uuid"), "uuid must not be inside agent_metadata"); + assert!( + !meta.contains_key("uuid"), + "uuid must not be inside agent_metadata" + ); // platform_version must not appear — not in the REDAPL schema. assert!(!meta.contains_key("platform_version")); @@ -849,7 +909,10 @@ mod tests { ) .unwrap(); let payload: serde_json::Value = serde_json::from_slice(&body).unwrap(); - assert_eq!(payload["agent_metadata"]["serverless_compat_version"], "3.7.1"); + assert_eq!( + payload["agent_metadata"]["serverless_compat_version"], + "3.7.1" + ); unsafe { env::remove_var("DD_SERVERLESS_COMPAT_VERSION"); @@ -929,23 +992,42 @@ mod tests { #[test] fn gate_off_by_default() { let _lock = ENV_LOCK.lock().unwrap(); - unsafe { env::remove_var("DD_SERVERLESS_COMPAT_INVENTORY_ENABLED"); } - assert!(!is_inventory_enabled(), "gate must be off when env var is absent"); + unsafe { + env::remove_var("DD_SERVERLESS_COMPAT_INVENTORY_ENABLED"); + } + assert!( + !is_inventory_enabled(), + "gate must be off when env var is absent" + ); } #[test] fn gate_on_when_set_to_true() { let _lock = ENV_LOCK.lock().unwrap(); - unsafe { env::set_var("DD_SERVERLESS_COMPAT_INVENTORY_ENABLED", "true"); } - assert!(is_inventory_enabled(), "gate must be on when env var is 'true'"); - unsafe { env::remove_var("DD_SERVERLESS_COMPAT_INVENTORY_ENABLED"); } + unsafe { + env::set_var("DD_SERVERLESS_COMPAT_INVENTORY_ENABLED", "true"); + } + assert!( + is_inventory_enabled(), + "gate must be on when env var is 'true'" + ); + unsafe { + env::remove_var("DD_SERVERLESS_COMPAT_INVENTORY_ENABLED"); + } } #[test] fn gate_off_when_set_to_other_value() { let _lock = ENV_LOCK.lock().unwrap(); - unsafe { env::set_var("DD_SERVERLESS_COMPAT_INVENTORY_ENABLED", "false"); } - assert!(!is_inventory_enabled(), "gate must be off when env var is not 'true'"); - unsafe { env::remove_var("DD_SERVERLESS_COMPAT_INVENTORY_ENABLED"); } + unsafe { + env::set_var("DD_SERVERLESS_COMPAT_INVENTORY_ENABLED", "false"); + } + assert!( + !is_inventory_enabled(), + "gate must be off when env var is not 'true'" + ); + unsafe { + env::remove_var("DD_SERVERLESS_COMPAT_INVENTORY_ENABLED"); + } } } From d22f64d6607863e1d587941a05c87adf3fa39fb1 Mon Sep 17 00:00:00 2001 From: Nina Rei Date: Thu, 3 Sep 2026 14:13:54 -0400 Subject: [PATCH 5/9] chore: remove local testing scripts from PR --- scripts/serverless-compat-deploy/build.sh | 23 - .../serverless-compat-deploy/check-logs.sh | 318 ------ .../serverless-compat-deploy/demo-both-rc.sh | 801 --------------- scripts/svls9604/.gitignore | 2 - scripts/svls9604/README.md | 220 ----- scripts/svls9604/azure/container-app.bicep | 89 -- scripts/svls9604/azure/function.bicep | 60 -- scripts/svls9604/azure/web-app-code.bicep | 80 -- .../svls9604/azure/web-app-container.bicep | 84 -- scripts/svls9604/fixtures/Dockerfile | 111 --- scripts/svls9604/matrix.json | 40 - scripts/svls9604/report.py | 345 ------- scripts/svls9604/run.sh | 18 - scripts/svls9604/runner.py | 932 ------------------ 14 files changed, 3123 deletions(-) delete mode 100755 scripts/serverless-compat-deploy/build.sh delete mode 100755 scripts/serverless-compat-deploy/check-logs.sh delete mode 100755 scripts/serverless-compat-deploy/demo-both-rc.sh delete mode 100644 scripts/svls9604/.gitignore delete mode 100644 scripts/svls9604/README.md delete mode 100644 scripts/svls9604/azure/container-app.bicep delete mode 100644 scripts/svls9604/azure/function.bicep delete mode 100644 scripts/svls9604/azure/web-app-code.bicep delete mode 100644 scripts/svls9604/azure/web-app-container.bicep delete mode 100644 scripts/svls9604/fixtures/Dockerfile delete mode 100644 scripts/svls9604/matrix.json delete mode 100644 scripts/svls9604/report.py delete mode 100755 scripts/svls9604/run.sh delete mode 100755 scripts/svls9604/runner.py diff --git a/scripts/serverless-compat-deploy/build.sh b/scripts/serverless-compat-deploy/build.sh deleted file mode 100755 index a1179a8..0000000 --- a/scripts/serverless-compat-deploy/build.sh +++ /dev/null @@ -1,23 +0,0 @@ -#!/usr/bin/env bash -# Build the serverless-compat binary for x86_64-unknown-linux-musl. -# Called by scripts/svls9604/runner.py before packaging. -set -euo pipefail - -REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" - -# Ensure rustup-managed toolchain is on PATH (needed when invoked by runner.py) -export PATH="$HOME/.rustup/toolchains/stable-aarch64-apple-darwin/bin:$PATH" - -CARGO="${CARGO:-$HOME/.rustup/toolchains/stable-aarch64-apple-darwin/bin/cargo}" -if [[ ! -x "$CARGO" ]]; then - CARGO="$(command -v cargo)" -fi - -echo "Building datadog-serverless-compat for x86_64-unknown-linux-musl..." -"$CARGO" build \ - --manifest-path "$REPO_ROOT/Cargo.toml" \ - --package datadog-serverless-compat \ - --target x86_64-unknown-linux-musl \ - --release - -echo "Binary: $REPO_ROOT/target/x86_64-unknown-linux-musl/release/datadog-serverless-compat" diff --git a/scripts/serverless-compat-deploy/check-logs.sh b/scripts/serverless-compat-deploy/check-logs.sh deleted file mode 100755 index c68f474..0000000 --- a/scripts/serverless-compat-deploy/check-logs.sh +++ /dev/null @@ -1,318 +0,0 @@ -#!/usr/bin/env bash -set -euo pipefail - -# --------------------------------------------------------------------------- -# check-logs.sh — diagnostic log collector for SVLS-9604 -# -# Collects inventory diagnostic output from deployed self-monitoring services -# for both serverless-init and serverless-compat, then prints the -# resource_id-based DDSQL queries to validate rows in both REDAPL tables. -# -# Resource identity: both REDAPL tables key on resource_id (CCRID), not UUID. -# Look for log lines containing "resource_id=" to find the actual key. -# -# Required env vars: -# GCP_PROJECT — GCP project for init Cloud Run services -# AZURE_SUBSCRIPTION_ID — Azure subscription (auto-detected if omitted) -# -# Service name overrides (defaults match the POC deploy scripts): -# INIT_CR_SERVICE — Cloud Run service name for init in-container (SI-01) -# INIT_CR_SIDECAR — Cloud Run service name for init sidecar (SI-02) -# INIT_CR_JOB — Cloud Run job name (SI-03, separate POC) -# INIT_CR_FN_SIDECAR — Cloud Run service name for Functions Gen 2 (SI-04, separate POC) -# INIT_ACA_INIT — Azure Container App name, in-container (SI-05) -# INIT_ACA_SIDECAR — Azure Container App name, sidecar (SI-06) -# INIT_AAS_CONTAINER — Azure App Service, Linux container (SI-07) -# INIT_AAS_SIDECAR — Azure App Service, SITECONTAINERS (SI-08) -# INIT_AAS_CODE — Azure App Service, Linux code (SI-09) -# COMPAT_AZURE_APP — Azure Function App name (SC-01) -# COMPAT_GCP_FN — GCP Cloud Function Gen 1 name (SC-02) -# COMPAT_GCP_PROJECT — GCP project for compat functions (may differ from init) -# AZURE_RG_ACA — Resource group for Container Apps -# AZURE_RG_AAS — Resource group for App Service apps -# COMPAT_AZURE_RG — Resource group for compat Azure Function -# -# Output: diagnostic-results-YYYYMMDD-HHMMSS.txt -# --------------------------------------------------------------------------- - -: "${GCP_PROJECT:?GCP_PROJECT must be set}" -AZURE_SUBSCRIPTION_ID="${AZURE_SUBSCRIPTION_ID:-$(az account show --query id -o tsv 2>/dev/null || echo '')}" - -# Service name defaults (override to match your deployed services) -INIT_CR_SERVICE="${INIT_CR_SERVICE:-nina-cloudrun-init}" -INIT_CR_SIDECAR="${INIT_CR_SIDECAR:-nina-cloudrun-sidecar}" -INIT_CR_JOB="${INIT_CR_JOB:-nina-cloudrun-job}" -INIT_CR_FN_SIDECAR="${INIT_CR_FN_SIDECAR:-nina-cloudrun-function-sidecar}" -INIT_ACA_INIT="${INIT_ACA_INIT:-nina-containerapp-init}" -INIT_ACA_SIDECAR="${INIT_ACA_SIDECAR:-nina-containerapp-sidecar}" -INIT_AAS_CONTAINER="${INIT_AAS_CONTAINER:-nina-webapp-container}" -INIT_AAS_SIDECAR="${INIT_AAS_SIDECAR:-nina-webapp-sidecar}" -INIT_AAS_CODE="${INIT_AAS_CODE:-nina-webapp-linux-code}" -COMPAT_AZURE_APP="${COMPAT_AZURE_APP:-nina-compat-inventory-node}" -COMPAT_GCP_FN="${COMPAT_GCP_FN:-nina-compat-inventory-nodejs}" -COMPAT_GCP_PROJECT="${COMPAT_GCP_PROJECT:-datadog-sandbox}" -AZURE_RG_ACA="${AZURE_RG_ACA:-dd-serverless-test-aca}" -AZURE_RG_AAS="${AZURE_RG_AAS:-dd-serverless-test-aas}" -COMPAT_AZURE_RG="${COMPAT_AZURE_RG:-self-monitoring-nina-dev}" - -SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")" && pwd)" -OUTPUT_FILE="${OUTPUT_FILE:-${SCRIPT_DIR}/diagnostic-results-$(date -u +%Y%m%d-%H%M%S).txt}" - -log() { echo "[$(date -u '+%Y-%m-%dT%H:%M:%SZ')] $*"; } - -# --------------------------------------------------------------------------- -# GCP log fetcher — matches inventory log lines containing resource_id -# --------------------------------------------------------------------------- -_gcp_logs() { - local label="$1" resource_type="$2" filter_key="$3" filter_val="$4" project="${5:-${GCP_PROJECT}}" - echo "" - echo "================================================================" - echo "# ${label}" - echo "================================================================" - gcloud logging read \ - "resource.type=${resource_type} AND resource.labels.${filter_key}=${filter_val} AND (textPayload:resource_id OR textPayload:SERVERLESS_DIAGNOSTIC OR textPayload:\"Inventory payload\")" \ - --project="${project}" \ - --limit=50 \ - --format="value(textPayload)" 2>/dev/null \ - || echo " (no matching logs — trigger a request and retry)" -} - -# --------------------------------------------------------------------------- -# Azure Container App log fetcher -# --------------------------------------------------------------------------- -_azure_containerapp_logs() { - local label="$1" app_name="$2" rg="$3" container="${4:-}" - [[ -z "${AZURE_SUBSCRIPTION_ID}" ]] && { echo " (AZURE_SUBSCRIPTION_ID not set — skipping)"; return; } - echo "" - echo "================================================================" - echo "# ${label}" - echo "================================================================" - local args=(--name "${app_name}" --resource-group "${rg}" \ - --subscription "${AZURE_SUBSCRIPTION_ID}" --tail 200) - [[ -n "${container}" ]] && args+=(--container "${container}") - az containerapp logs show "${args[@]}" 2>/dev/null \ - | grep -E "resource_id|SERVERLESS_DIAGNOSTIC|Inventory payload" \ - || echo " (no matching logs — trigger a request and retry)" -} - -# --------------------------------------------------------------------------- -# Azure Web App log fetcher (downloads archived logs) -# --------------------------------------------------------------------------- -_azure_webapp_logs() { - local label="$1" app_name="$2" rg="$3" - [[ -z "${AZURE_SUBSCRIPTION_ID}" ]] && { echo " (AZURE_SUBSCRIPTION_ID not set — skipping)"; return; } - echo "" - echo "================================================================" - echo "# ${label}" - echo "================================================================" - local tmp_zip - tmp_zip=$(mktemp /tmp/webapp-logs-XXXXXX.zip) - az webapp log download \ - --name "${app_name}" \ - --resource-group "${rg}" \ - --subscription "${AZURE_SUBSCRIPTION_ID}" \ - --log-file "${tmp_zip}" 2>/dev/null \ - && unzip -p "${tmp_zip}" 2>/dev/null \ - | grep -aE "resource_id|SERVERLESS_DIAGNOSTIC|Inventory payload" \ - | sort -u \ - || echo " (no matching logs — trigger a request and retry)" - rm -f "${tmp_zip}" -} - -# --------------------------------------------------------------------------- -# Azure Function log fetcher (compat) -# --------------------------------------------------------------------------- -_azure_function_logs() { - local label="$1" app_name="$2" rg="$3" - [[ -z "${AZURE_SUBSCRIPTION_ID}" ]] && { echo " (AZURE_SUBSCRIPTION_ID not set — skipping)"; return; } - echo "" - echo "================================================================" - echo "# ${label}" - echo "================================================================" - # az webapp log tail exits after --timeout seconds; we want a snapshot not a stream. - # Use log download instead for reliability. - local tmp_zip - tmp_zip=$(mktemp /tmp/fn-logs-XXXXXX.zip) - az webapp log download \ - --name "${app_name}" \ - --resource-group "${rg}" \ - --subscription "${AZURE_SUBSCRIPTION_ID}" \ - --log-file "${tmp_zip}" 2>/dev/null \ - && unzip -p "${tmp_zip}" 2>/dev/null \ - | grep -aE "resource_id|workload_type|Inventory payload sent" \ - | sort -u \ - || echo " (no matching logs — trigger a request and retry)" - rm -f "${tmp_zip}" -} - -# --------------------------------------------------------------------------- -# Run all checks -# --------------------------------------------------------------------------- -run_checks() { - echo "================================================================" - echo "SVLS-9604 — Serverless REDAPL Diagnostic Log Collection" - echo "Date: $(date -u '+%Y-%m-%d %H:%M:%S UTC')" - echo "================================================================" - echo "" - echo "What to look for: lines containing 'resource_id=//...' — this is" - echo "the CCRID used as the REDAPL table key, NOT the agent UUID." - echo "" - - echo "===== serverless_init_agent workloads =====" - - _gcp_logs "SI-01 Cloud Run Service (in-container, ${INIT_CR_SERVICE})" \ - cloud_run_revision service_name "${INIT_CR_SERVICE}" - - _gcp_logs "SI-02 Cloud Run Service (sidecar, ${INIT_CR_SIDECAR})" \ - cloud_run_revision service_name "${INIT_CR_SIDECAR}" - - echo "" - echo "# SI-03 Cloud Run Job — requires separate POC (not deployed via self-monitoring)" - _gcp_logs "SI-03 Cloud Run Job (${INIT_CR_JOB}) — if POC deployed" \ - cloud_run_job job_name "${INIT_CR_JOB}" - - echo "" - echo "# SI-04 Cloud Run Functions Gen 2 — requires separate POC" - _gcp_logs "SI-04 Cloud Run Functions Gen 2 (${INIT_CR_FN_SIDECAR}) — if POC deployed" \ - cloud_run_revision service_name "${INIT_CR_FN_SIDECAR}" - - _azure_containerapp_logs "SI-05 Azure Container App (in-container, ${INIT_ACA_INIT})" \ - "${INIT_ACA_INIT}" "${AZURE_RG_ACA}" - - _azure_containerapp_logs "SI-06 Azure Container App (sidecar, dd-agent, ${INIT_ACA_SIDECAR})" \ - "${INIT_ACA_SIDECAR}" "${AZURE_RG_ACA}" "dd-agent" - - _azure_webapp_logs "SI-07 Azure App Service Linux container (${INIT_AAS_CONTAINER})" \ - "${INIT_AAS_CONTAINER}" "${AZURE_RG_AAS}" - - _azure_webapp_logs "SI-08 Azure App Service SITECONTAINERS (${INIT_AAS_SIDECAR})" \ - "${INIT_AAS_SIDECAR}" "${AZURE_RG_AAS}" - - _azure_webapp_logs "SI-09 Azure App Service Linux code (${INIT_AAS_CODE})" \ - "${INIT_AAS_CODE}" "${AZURE_RG_AAS}" - - echo "" - echo "===== serverless_compat_agent workloads =====" - - _azure_function_logs "SC-01 Azure Functions Node.js (${COMPAT_AZURE_APP})" \ - "${COMPAT_AZURE_APP}" "${COMPAT_AZURE_RG}" - - _gcp_logs "SC-02 GCP Cloud Functions Gen 1 (${COMPAT_GCP_FN})" \ - cloud_function function_name "${COMPAT_GCP_FN}" "${COMPAT_GCP_PROJECT}" -} - -# --------------------------------------------------------------------------- -# Extract resource_ids and generate DDSQL queries -# --------------------------------------------------------------------------- -generate_sql() { - local output_file="$1" - - # Extract resource_ids — CCRID format: //run.googleapis.com/... or //cloudfunctions.googleapis.com/... etc. - local resource_ids - resource_ids=$(python3 -c " -import re, sys -text = open('${output_file}').read() -# resource_id= or resource_id: followed by a CCRID -rids = re.findall(r'resource_id[=:]\s*(//[^\s,\"\']+)', text) -print('\n'.join(sorted(set(rids)))) -" 2>/dev/null || true) - - echo "" - echo "================================================================" - echo "# DDSQL Queries (paste in go/redapl → Queries → SQL)" - echo "# Note: these tables do NOT have api_key_uuid — filter by resource_id." - echo "================================================================" - echo "" - - if [[ -z "${resource_ids}" ]]; then - echo "-- No resource_ids found in logs yet." - echo "-- Trigger the services, wait ~5 min for EPRW propagation, and retry." - echo "" - echo "-- Fallback: show all rows modified today" - echo "SELECT resource_id, workload_type, deployment_model, _modified_at" - echo "FROM udm.all.serverless_init_agent" - echo "WHERE _modified_at >= TIMESTAMP '$(date -u +%Y-%m-%dT00:00:00Z)'" - echo "ORDER BY _modified_at DESC LIMIT 20;" - echo "" - echo "SELECT resource_id, workload_type, _modified_at" - echo "FROM udm.all.serverless_compat_agent" - echo "WHERE _modified_at >= TIMESTAMP '$(date -u +%Y-%m-%dT00:00:00Z)'" - echo "ORDER BY _modified_at DESC LIMIT 20;" - return - fi - - # Build IN clause - local in_clause="" - while IFS= read -r rid; do - [[ -z "${rid}" ]] && continue - in_clause+=" '${rid}',\n" - done <<< "${resource_ids}" - - echo "-- serverless_init_agent: rows by extracted resource_id" - echo "SELECT _key, resource_id, resource_name, workload_type, deployment_model," - echo " agent_version_base, serverless_init_version, runtime," - echo " _first_seen_at, _modified_at" - echo "FROM udm.all.serverless_init_agent" - echo "WHERE resource_id IN (" - printf "%b" "${in_clause}" - echo ")" - echo "ORDER BY workload_type, _modified_at DESC;" - echo "" - - echo "-- serverless_compat_agent: rows by extracted resource_id" - echo "SELECT _key, resource_id, resource_name, workload_type," - echo " serverless_compat_version, serverless_compat_runtime_version," - echo " _first_seen_at, _modified_at" - echo "FROM udm.all.serverless_compat_agent" - echo "WHERE resource_id IN (" - printf "%b" "${in_clause}" - echo ")" - echo "ORDER BY workload_type, _modified_at DESC;" - echo "" - - echo "-- Cardinality: rows must equal distinct_resources" - echo "SELECT 'serverless_init_agent' AS flavor, workload_type," - echo " COUNT(*) AS total_rows, COUNT(DISTINCT resource_id) AS distinct_resources" - echo "FROM udm.all.serverless_init_agent" - echo "WHERE resource_id IN (" - printf "%b" "${in_clause}" - echo ")" - echo "GROUP BY workload_type" - echo "UNION ALL" - echo "SELECT 'serverless_compat_agent', workload_type," - echo " COUNT(*), COUNT(DISTINCT resource_id)" - echo "FROM udm.all.serverless_compat_agent" - echo "WHERE resource_id IN (" - printf "%b" "${in_clause}" - echo ")" - echo "GROUP BY workload_type" - echo "ORDER BY flavor, workload_type;" -} - -# --------------------------------------------------------------------------- -# Entry point -# --------------------------------------------------------------------------- -log "Collecting diagnostic logs → ${OUTPUT_FILE}" -run_checks | tee "${OUTPUT_FILE}" -generate_sql "${OUTPUT_FILE}" | tee -a "${OUTPUT_FILE}" - -echo "" -log "Done. Output: ${OUTPUT_FILE}" - -# Summarise what was found -found=$(python3 -c " -import re -text = open('${OUTPUT_FILE}').read() -rids = sorted(set(re.findall(r'resource_id[=:]\s*(//[^\s,\"\']+)', text))) -if rids: - print(f'{len(rids)} resource_id(s) found:') - print('\n'.join(' ' + r for r in rids)) -else: - print('No resource_ids found. Trigger the services and retry.') -" 2>/dev/null || echo "Could not parse output file.") - -log "${found}" - -if [[ "${found}" == "No resource_ids found."* ]]; then - exit 1 -fi diff --git a/scripts/serverless-compat-deploy/demo-both-rc.sh b/scripts/serverless-compat-deploy/demo-both-rc.sh deleted file mode 100755 index 3d48b4a..0000000 --- a/scripts/serverless-compat-deploy/demo-both-rc.sh +++ /dev/null @@ -1,801 +0,0 @@ -#!/usr/bin/env bash -set -euo pipefail - -# --------------------------------------------------------------------------- -# demo-both-rc.sh — SVLS-9604 full RC validation -# -# End-to-end REDAPL staging test for BOTH serverless_init_agent AND -# serverless_compat_agent. -# -# What this script does: -# Phase 0 — Preflight: validate env, tools, and that both compat endpoints -# are already deployed and reachable (no deploy automation for compat) -# Phase 1 — Init RC deploy: discover → build → deploy via serverless-init-self-monitoring -# (runs npm run discover to regenerate apps.yml including Azure workloads, -# then build + deploy with AGENT_IMAGE=) -# Phase 2 — Baseline trigger (L0): one bounded pass across all workloads. -# Uses npm run test:once for init (exits nonzero on failure). -# Uses trigger.sh for compat. Blocks burst on any failure. -# Phase 3 — Burst test: serverless-init (npm run burst, WAVE_SIZES waves) -# Sets concurrency=1 on Cloud Run services before burst so each -# request forces a new instance. Restores concurrency after. -# Phase 4 — Burst test: serverless-compat (COMPAT_CONCURRENCY concurrent requests) -# Records application response latency (not EPRW/REDAPL ingestion lag). -# Phase 5 — MANUAL VALIDATION RUNBOOK: EPRW metrics and DDSQL cardinality queries. -# REDAPL query visibility lag must be measured manually — poll until rows -# appear and record wall-clock time. _modified_at shows when EPRW wrote -# the row, not when DDSQL first exposed it. -# -# Compat deploy: serverless-compat-self-monitoring has no root-level deployment -# automation. Deploy those functions manually before running this script. -# See: https://datadoghq.atlassian.net/wiki/spaces/SLS/pages/2977497119 -# -# Usage: -# export DD_API_KEY= # datad0g.com org key -# export AGENT_IMAGE=gcr.io/datadoghq/serverless-init: -# ./demo-both-rc.sh -# -# Skip flags (combine freely): -# SKIP_INIT_DEPLOY=true — skip discover + build + deploy for init -# SKIP_BURST=true — skip load stages; trigger once only (L0) -# LOAD_STAGE=L1 — compat burst stage: L0=1, L1=10, L2=50, L3=100 -# WAVE_SIZES=10,50,100 — init burst wave sizes (overrides defaults) -# -# Required before running: -# - Both compat functions deployed (azure + gcp). Set AZURE_FUNCTION_APP and -# GCP_FUNCTION_NAME to match your deployed endpoints. -# - gcloud CLI authenticated (gcloud auth login + application-default login) -# - az CLI authenticated (az login) -# - npm + tsx installed -# --------------------------------------------------------------------------- - -: "${DD_API_KEY:?DD_API_KEY must be set (datad0g.com staging org key)}" -: "${AGENT_IMAGE:?AGENT_IMAGE must be set (e.g. gcr.io/datadoghq/serverless-init:)}" - -DD_SITE="${DD_SITE:-datad0g.com}" -GCP_PROJECT="${GCP_PROJECT:-datadog-serverless-gcp-dev}" -GCP_REGION="${GCP_REGION:-us-central1}" -AZURE_SUBSCRIPTION_ID="${AZURE_SUBSCRIPTION_ID:-$(az account show --query id -o tsv 2>/dev/null || echo '')}" -AZURE_FUNCTION_APP="${AZURE_FUNCTION_APP:-nina-compat-inventory-node}" -GCP_FUNCTION_NAME="${GCP_FUNCTION_NAME:-nina-compat-inventory-nodejs}" - -SKIP_INIT_DEPLOY="${SKIP_INIT_DEPLOY:-false}" -SKIP_DISCOVER="${SKIP_DISCOVER:-false}" # set true to skip npm run discover (preserves cloud-run-v2) -SKIP_BURST="${SKIP_BURST:-false}" -LOAD_STAGE="${LOAD_STAGE:-L1}" # compat burst: L0=1 L1=10 L2=50 L3=100 -WAVE_SIZES="${WAVE_SIZES:-10,50,100}" # init burst waves - -# Largest wave size — used to set max-instances during burst so Cloud Run can actually scale out. -MAX_WAVE=$(echo "${WAVE_SIZES}" | tr ',' '\n' | sort -rn | head -1) - -INIT_SM_DIR="${INIT_SM_DIR:-${HOME}/go/src/github.com/DataDog/serverless-init-self-monitoring}" - -SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")" && pwd)" -RESULTS_DIR="${RESULTS_DIR:-/tmp/svls9604-rc-$(date +%Y%m%d-%H%M%S)}" -mkdir -p "${RESULTS_DIR}" - -log() { echo "[$(date -u '+%Y-%m-%dT%H:%M:%SZ')] $*" | tee -a "${RESULTS_DIR}/run.log"; } -header() { - local msg="$*" - echo "" | tee -a "${RESULTS_DIR}/run.log" - echo "==================================================" | tee -a "${RESULTS_DIR}/run.log" - echo " ${msg}" | tee -a "${RESULTS_DIR}/run.log" - echo "==================================================" | tee -a "${RESULTS_DIR}/run.log" -} -fail() { echo "ERROR: $*" >&2; exit 1; } - -# Portable millisecond timestamp (macOS-safe: date +%s%3N returns literal '3N' on macOS) -_ms_now() { python3 -c "import time; print(int(time.time() * 1000))"; } - -# --------------------------------------------------------------------------- -# Phase 0 — Preflight -# --------------------------------------------------------------------------- -header "Phase 0 — Preflight checks" - -[[ -d "${INIT_SM_DIR}" ]] || fail "serverless-init-self-monitoring not found at ${INIT_SM_DIR}. - Set INIT_SM_DIR or: gh repo clone DataDog/serverless-init-self-monitoring ${INIT_SM_DIR}" - -command -v npm >/dev/null 2>&1 || fail "npm not found" -command -v gcloud >/dev/null 2>&1 || fail "gcloud CLI not found" -command -v az >/dev/null 2>&1 || fail "az CLI not found" -command -v curl >/dev/null 2>&1 || fail "curl not found" -command -v python3 >/dev/null 2>&1 || fail "python3 not found" - -# tsx: prefer project-local, fall back to global -TSX_BIN="${INIT_SM_DIR}/node_modules/.bin/tsx" -if [[ ! -x "${TSX_BIN}" ]]; then - TSX_BIN=$(command -v tsx 2>/dev/null || echo "") -fi -[[ -n "${TSX_BIN}" ]] || fail "tsx not found — run: npm install inside ${INIT_SM_DIR} or npm install -g tsx" - -case "${LOAD_STAGE}" in - L0) COMPAT_CONCURRENCY=1 ;; - L1) COMPAT_CONCURRENCY=10 ;; - L2) COMPAT_CONCURRENCY=50 ;; - L3) COMPAT_CONCURRENCY=100 ;; - *) fail "Unknown LOAD_STAGE ${LOAD_STAGE} (use L0/L1/L2/L3)" ;; -esac - -if [[ "${LOAD_STAGE}" == "L3" ]]; then - log "WARNING: LOAD_STAGE=L3 (100 concurrent). Coordinate with RP Ingest before running." - log " See RFC Section 9 — do not advance past L2 without explicit approval." - read -r -p " Continue? [y/N] " confirm - [[ "${confirm}" =~ ^[Yy]$ ]] || exit 0 -fi - -# Preflight: verify compat endpoints are deployed and reachable -log "Checking Azure Function endpoint (${AZURE_FUNCTION_APP})..." -AZURE_BASE="https://${AZURE_FUNCTION_APP}.azurewebsites.net" -if ! curl -sf --max-time 15 "${AZURE_BASE}/api/httptest" -o /dev/null 2>/dev/null; then - fail "Azure Function not reachable: ${AZURE_BASE}/api/httptest - Deploy serverless-compat-self-monitoring azure_functions/compat_node manually first. - See: https://datadoghq.atlassian.net/wiki/spaces/SLS/pages/2977497119" -fi -log " Azure Function: OK" - -log "Checking GCP Cloud Function endpoint (${GCP_FUNCTION_NAME})..." -GCP_FN_URL=$(gcloud functions describe "${GCP_FUNCTION_NAME}" \ - --project="${GCP_PROJECT}" --region="${GCP_REGION}" \ - --format="value(httpsTrigger.url)" 2>/dev/null \ - || echo "https://${GCP_REGION}-${GCP_PROJECT}.cloudfunctions.net/${GCP_FUNCTION_NAME}") -GCP_TOKEN=$(gcloud auth print-identity-token 2>/dev/null || true) -if [[ -n "${GCP_TOKEN}" ]]; then - GCP_CURL_ARGS=(-H "Authorization: Bearer ${GCP_TOKEN}") -else - GCP_CURL_ARGS=() - log " WARNING: no GCP identity token — falling back to unauthenticated check" -fi -if ! curl -sf --max-time 15 "${GCP_CURL_ARGS[@]}" "${GCP_FN_URL}" -o /dev/null 2>/dev/null; then - fail "GCP Cloud Function not reachable: ${GCP_FN_URL} - Deploy serverless-compat-self-monitoring gcp_functions/nodejs manually first. - See: https://datadoghq.atlassian.net/wiki/spaces/SLS/pages/2977497119" -fi -log " GCP Cloud Function: OK" - -log "" -log "AGENT_IMAGE : ${AGENT_IMAGE}" -log "DD_SITE : ${DD_SITE}" -log "GCP_PROJECT : ${GCP_PROJECT}" -log "AZURE_FUNCTION_APP : ${AZURE_FUNCTION_APP}" -log "GCP_FUNCTION_NAME : ${GCP_FUNCTION_NAME}" -log "LOAD_STAGE : ${LOAD_STAGE} (compat concurrency=${COMPAT_CONCURRENCY})" -log "WAVE_SIZES (init) : ${WAVE_SIZES} (max wave = ${MAX_WAVE} → max-instances set to ${MAX_WAVE} during burst)" -log "INIT_SM_DIR : ${INIT_SM_DIR}" -log "SKIP_DISCOVER : ${SKIP_DISCOVER}" -log "Results dir : ${RESULTS_DIR}" - -START_TIME=$(date -u '+%Y-%m-%dT%H:%M:%SZ') -START_EPOCH=$(date +%s) - -# --------------------------------------------------------------------------- -# Phase 1 — serverless-init RC deploy -# -# Runs npm run discover to regenerate apps.yml from currently deployed resources -# (including Azure workloads — skipping this step leaves apps.yml stale with -# only GCP entries). Then builds all app images with AGENT_IMAGE= -# and deploys to GCP and Azure self-monitoring environments. -# -# Skippable when apps are already deployed (SKIP_INIT_DEPLOY=true). -# --------------------------------------------------------------------------- -if [[ "${SKIP_INIT_DEPLOY}" != "true" ]]; then - header "Phase 1 — serverless-init RC deploy (AGENT_IMAGE=${AGENT_IMAGE})" - - log "Running npm install..." - (cd "${INIT_SM_DIR}" && npm install --silent 2>&1 | tail -3) - - if [[ "${SKIP_DISCOVER}" == "true" ]]; then - log "Skipping npm run discover (SKIP_DISCOVER=true). Using existing apps.yml." - log " Preserves cloud-run-v2/sidecar entries that discover would remove." - else - log "Running npm run discover..." - log " WARNING: discover scans deploy/gcp/ and will REMOVE cloud-run-v2/sidecar entries" - log " because there is no deploy/gcp/cloud-run-v2/ directory. The next step" - log " (npm run ensure:v2) re-injects those entries automatically." - (cd "${INIT_SM_DIR}" && npm run discover 2>&1) \ - | tee "${RESULTS_DIR}/init-discover.log" - - # Verify cloud-run-v2 entries survived discovery. - # With deploy/gcp/cloud-run-v2/ stub files present, discover_apps.ts should find - # the cloud-run-v2 product and preserve those entries in apps.yml. If they are - # missing, the stub files may not be in place. - if ! grep -q 'product: cloud-run-v2' "${INIT_SM_DIR}/apps.yml" 2>/dev/null; then - log " WARNING: discover wiped cloud-run-v2 entries from apps.yml" - log " Attempting to restore via npm run ensure:v2..." - (cd "${INIT_SM_DIR}" && npm run ensure:v2 2>&1) \ - | tee -a "${RESULTS_DIR}/init-discover.log" - if ! grep -q 'product: cloud-run-v2' "${INIT_SM_DIR}/apps.yml" 2>/dev/null; then - fail "discover wiped cloud-run-v2 entries from apps.yml — deploy/gcp/cloud-run-v2/ stubs may be missing" - fi - log " cloud-run-v2 entries restored via ensure:v2" - else - log " cloud-run-v2 entries preserved in apps.yml" - fi - fi - - # Coverage preflight: fail if hard-required workloads are missing from apps.yml. - # Hard required: SI-01 (cloud-run/in-process), SI-02 (cloud-run/sidecar), - # SI-04 (cloud-run-v2/sidecar). - # Soft required (warn): SI-03 (jobs), SI-05..SI-09 (Azure — needs az login + Azure deploy). - log "Running coverage check (npm run coverage:check)..." - if ! (cd "${INIT_SM_DIR}" && WAVE_SIZES="${WAVE_SIZES}" npm run coverage:check 2>&1) \ - | tee "${RESULTS_DIR}/coverage-check.log"; then - fail "Coverage check failed — see ${RESULTS_DIR}/coverage-check.log for missing workloads." - fi - - log "Running npm run build (builds all app images with RC agent)..." - (cd "${INIT_SM_DIR}" && AGENT_IMAGE="${AGENT_IMAGE}" npm run build 2>&1) \ - | tee "${RESULTS_DIR}/init-build.log" - - log "Running npm run deploy..." - (cd "${INIT_SM_DIR}" && \ - AGENT_IMAGE="${AGENT_IMAGE}" \ - DD_API_KEY="${DD_API_KEY}" \ - DD_SITE="${DD_SITE}" \ - npm run deploy 2>&1) \ - | tee "${RESULTS_DIR}/init-deploy.log" - - log "Init RC deploy complete." -else - log "Skipping init deploy (SKIP_INIT_DEPLOY=true). Using existing deployment." - log "NOTE: run 'npm run discover && npm run ensure:v2' in ${INIT_SM_DIR} if apps.yml is stale." - - # Still run coverage check even when skipping deploy — the harness should know its gaps. - log "Running coverage check (npm run coverage:check)..." - (cd "${INIT_SM_DIR}" && WAVE_SIZES="${WAVE_SIZES}" npm run coverage:check 2>&1) \ - | tee "${RESULTS_DIR}/coverage-check.log" || true # warn only when not deploying -fi - -# --------------------------------------------------------------------------- -# Phase 2 — Baseline trigger (L0: one pass, all workloads) -# -# npm run test:once: sends exactly one request to each app (coldstart + busy). -# Exits nonzero if any request fails. Failure blocks burst phases. -# SI-03 (Cloud Run Jobs): needs Cloud Scheduler triggering the Jobs API — -# not an HTTP service, not covered by test:once. -# SI-04 (Cloud Run Functions Gen 2): covered as cloud-run-v2 product. -# --------------------------------------------------------------------------- -header "Phase 2 — Baseline trigger (L0: one pass, all workloads)" - -log "Triggering one pass across all init self-monitoring apps (npm run test:once)..." -log " SI-03 (Cloud Run Jobs) not covered — trigger via 'gcloud run jobs execute' separately." -(cd "${INIT_SM_DIR}" && npm run test:once 2>&1) \ - | tee "${RESULTS_DIR}/baseline-init.log" -log "Init baseline passed." - -# SI-03 (Cloud Run Jobs): trigger if JOB_NAME is set. -# Jobs have no HTTP endpoint — they are triggered via the Jobs API, not test:once. -log "SI-03 (Cloud Run Jobs): trigger if JOB_NAME is set..." -if [[ -n "${JOB_NAME:-}" ]]; then - gcloud run jobs execute "${JOB_NAME}" \ - --project="${GCP_PROJECT}" --region="${GCP_REGION}" \ - --wait 2>&1 | tee "${RESULTS_DIR}/job-trigger.log" - log " Cloud Run Job triggered" -else - log " WARNING: JOB_NAME not set — SI-03 not triggered. Set JOB_NAME= to cover SI-03." -fi - -log "Triggering compat functions once..." -GCP_PROJECT="${GCP_PROJECT}" \ -GCP_REGION="${GCP_REGION}" \ -AZURE_FUNCTION_APP="${AZURE_FUNCTION_APP}" \ -GCP_FUNCTION_NAME="${GCP_FUNCTION_NAME}" \ - "${SCRIPT_DIR}/trigger.sh" 2>&1 | tee "${RESULTS_DIR}/baseline-compat.log" - -BASELINE_END=$(date -u '+%Y-%m-%dT%H:%M:%SZ') -log "Baseline complete at ${BASELINE_END}." -log "Waiting 5 min for EPRW propagation before burst..." -sleep 300 - -# --------------------------------------------------------------------------- -# Phase 3 — Burst test: serverless-init -# -# Sets concurrency=1 on all self-monitoring Cloud Run services so each -# concurrent request forces a new instance (scale-out). Without this, -# Cloud Run's default concurrency=80 lets one instance absorb the whole wave, -# defeating the cardinality test. Restores concurrency after burst. -# -# Reports application response latency (not EPRW/REDAPL ingestion lag). -# REDAPL visibility is validated separately in Phase 5. -# --------------------------------------------------------------------------- -SELF_MON_SERVICES="" -if [[ "${SKIP_BURST}" != "true" ]]; then - header "Phase 3 — Burst test: serverless-init (WAVE_SIZES=${WAVE_SIZES})" - - log "Collecting self-monitoring Cloud Run services (label: selfmonitoring=true)..." - SELF_MON_SERVICES=$(gcloud run services list \ - --project="${GCP_PROJECT}" --region="${GCP_REGION}" \ - --filter="metadata.labels.selfmonitoring=true" \ - --format="value(metadata.name)" 2>/dev/null || echo "") - - if [[ -z "${SELF_MON_SERVICES}" ]]; then - log " WARNING: no self-monitoring Cloud Run services found (filter: labels.selfmonitoring=true)" - log " Burst will run but cannot force scale-out — cardinality result may not be meaningful." - fi - - # Save original concurrency and max-instances per service, then set burst values. - # concurrency=1 → each concurrent HTTP request requires a separate instance. - # max-instances → set to max(WAVE_SIZES) so Cloud Run can actually scale out. - # - # Without increasing max-instances (default=1 in in-process.yaml and gcp.ts:154), - # a 100-request wave still hits one instance regardless of concurrency. - declare -A ORIG_CONCURRENCY - declare -A ORIG_MAX_INSTANCES - - if [[ -n "${SELF_MON_SERVICES}" ]]; then - log "Saving original settings and setting concurrency=1 max-instances=${MAX_WAVE}..." - while IFS= read -r svc; do - [[ -z "${svc}" ]] && continue - - orig_conc=$(gcloud run services describe "${svc}" \ - --project="${GCP_PROJECT}" --region="${GCP_REGION}" \ - --format="value(spec.template.spec.containerConcurrency)" 2>/dev/null || echo "80") - orig_max=$(gcloud run services describe "${svc}" \ - --project="${GCP_PROJECT}" --region="${GCP_REGION}" \ - --format="value(spec.template.metadata.annotations['autoscaling.knative.dev/maxScale'])" \ - 2>/dev/null || echo "1") - - ORIG_CONCURRENCY["${svc}"]="${orig_conc:-80}" - ORIG_MAX_INSTANCES["${svc}"]="${orig_max:-1}" - - gcloud run services update "${svc}" \ - --concurrency=1 \ - --max-instances="${MAX_WAVE}" \ - --project="${GCP_PROJECT}" --region="${GCP_REGION}" --quiet 2>/dev/null || true - log " ${svc}: concurrency ${orig_conc:-?} → 1, max-instances ${orig_max:-?} → ${MAX_WAVE}" - done <<< "${SELF_MON_SERVICES}" - fi - - log "Running npm run burst (${WAVE_SIZES} waves, ${MAX_WAVE} max-instances per service)..." - log " HTTP requests exercised: $(echo "${SELF_MON_SERVICES}" | grep -c . || echo 0) services × sum(${WAVE_SIZES}) waves" - burst_exit=0 - (cd "${INIT_SM_DIR}" && \ - WAVE_SIZES="${WAVE_SIZES}" \ - BURST_WAIT_MS=90000 \ - DD_SITE="${DD_SITE}" \ - npm run burst 2>&1) \ - | tee "${RESULTS_DIR}/burst-init.log" || burst_exit=$? - - log "Restoring original concurrency and max-instances on Cloud Run services..." - if [[ -n "${SELF_MON_SERVICES}" ]]; then - while IFS= read -r svc; do - [[ -z "${svc}" ]] && continue - restore_conc="${ORIG_CONCURRENCY["${svc}"]:-80}" - restore_max="${ORIG_MAX_INSTANCES["${svc}"]:-1}" - gcloud run services update "${svc}" \ - --concurrency="${restore_conc}" \ - --max-instances="${restore_max}" \ - --project="${GCP_PROJECT}" --region="${GCP_REGION}" --quiet 2>/dev/null || true - log " ${svc}: restored concurrency=${restore_conc} max-instances=${restore_max}" - done <<< "${SELF_MON_SERVICES}" - fi - - if [[ "${burst_exit}" -ne 0 ]]; then - fail "Init burst had request failures (exit ${burst_exit}) — check ${RESULTS_DIR}/burst-init.log" - fi - log "Init burst complete." -else - log "Skipping burst (SKIP_BURST=true)" -fi - -# --------------------------------------------------------------------------- -# Phase 4 — Burst test: serverless-compat -# -# Fires COMPAT_CONCURRENCY concurrent requests at each compat function. -# Records per-request timing to CSV; computes p50/p95/p99 APPLICATION -# RESPONSE LATENCY. This is HTTP round-trip latency — not EPRW ingestion -# lag or REDAPL visibility lag. Those are measured in Phase 5. -# --------------------------------------------------------------------------- -if [[ "${SKIP_BURST}" != "true" ]]; then - header "Phase 4 — Burst test: serverless-compat (${LOAD_STAGE}=${COMPAT_CONCURRENCY} concurrent)" - log " NOTE: latency stats below are APPLICATION RESPONSE LATENCY (HTTP round-trip)." - log " REDAPL visibility lag is measured separately in Phase 5 using _modified_at." - - COMPAT_BURST_DIR="${RESULTS_DIR}/compat-burst" - mkdir -p "${COMPAT_BURST_DIR}" - - _trigger_one() { - local id="$1" url="$2" out_file="$3" auth_header="${4:-}" - local t_start elapsed status - - t_start=$(_ms_now) - local curl_args=(-sf -o /dev/null -w "%{http_code}" --max-time 30) - [[ -n "${auth_header}" ]] && curl_args+=(-H "${auth_header}") - status=$(curl "${curl_args[@]}" "${url}" 2>/dev/null || echo "000") - elapsed=$(( $(_ms_now) - t_start )) - echo "${id},${status},${elapsed}" >> "${out_file}" - } - - _run_wave() { - local label="$1" url="$2" auth_header="${3:-}" - local wave_file="${COMPAT_BURST_DIR}/${label}.csv" - echo "id,http_status,elapsed_ms" > "${wave_file}" - - log "Launching ${COMPAT_CONCURRENCY} concurrent requests → ${label}..." - local pids=() - for i in $(seq 1 "${COMPAT_CONCURRENCY}"); do - _trigger_one "${i}" "${url}" "${wave_file}" "${auth_header}" & - pids+=($!) - done - for pid in "${pids[@]}"; do wait "${pid}" 2>/dev/null || true; done - - python3 - "${wave_file}" "${label}" <<'PYEOF' -import sys, csv, statistics -rows = list(csv.DictReader(open(sys.argv[1]))) -label = sys.argv[2] -total = len(rows) -success = sum(1 for r in rows if r['http_status'].startswith('2')) -fail = total - success -lats = [int(r['elapsed_ms']) for r in rows if r['elapsed_ms'].isdigit()] -p50 = statistics.median(lats) if lats else 0 -p95 = sorted(lats)[max(0, int(len(lats)*0.95)-1)] if lats else 0 -p99 = sorted(lats)[max(0, int(len(lats)*0.99)-1)] if lats else 0 -print(f" {label}: {total} total | {success} OK | {fail} failed") -print(f" App response latency: p50={p50}ms p95={p95}ms p99={p99}ms") -if fail > 0: - sys.exit(1) -PYEOF - } - - GCP_TOKEN=$(gcloud auth print-identity-token 2>/dev/null || true) - GCP_AUTH_HEADER="" - [[ -n "${GCP_TOKEN}" ]] && GCP_AUTH_HEADER="Authorization: Bearer ${GCP_TOKEN}" - - compat_burst_exit=0 - _run_wave "azure-function" "${AZURE_BASE}/api/httptest" || compat_burst_exit=$? - _run_wave "gcp-cloud-function-gen1" "${GCP_FN_URL}" "${GCP_AUTH_HEADER}" || compat_burst_exit=$? - - if [[ "${compat_burst_exit}" -ne 0 ]]; then - fail "Compat burst had request failures — check ${COMPAT_BURST_DIR}/" - fi - - log "Compat burst complete. Waiting 5 min for EPRW propagation..." - sleep 300 -else - log "Skipping compat burst (SKIP_BURST=true)" -fi - -END_TIME=$(date -u '+%Y-%m-%dT%H:%M:%SZ') - -# --------------------------------------------------------------------------- -# Phase 4b — EPRW metric gates (automated, via Datadog API) -# -# Polls EPRW accepted-write and rejection metrics for the run window. -# Requires DD_APP_KEY in addition to DD_API_KEY. Skips gracefully if absent. -# --------------------------------------------------------------------------- -header "Phase 4b — EPRW metric gates" -eprw_exit=0 -if [[ -n "${DD_APP_KEY:-}" ]]; then - (cd "${INIT_SM_DIR}" && \ - DD_API_KEY="${DD_API_KEY}" \ - DD_APP_KEY="${DD_APP_KEY}" \ - DD_SITE="${DD_SITE}" \ - FROM="${START_TIME}" \ - npm run poll:eprw 2>&1) \ - | tee "${RESULTS_DIR}/eprw-poll.log" || eprw_exit=$? - - if [[ "${eprw_exit}" -ne 0 ]]; then - log "WARNING: EPRW metric gates failed — see ${RESULTS_DIR}/eprw-poll.log" - log " This is non-fatal; verify manually using Phase 5 queries." - fi -else - log "DD_APP_KEY not set — skipping automated EPRW metric poll." - log " Set DD_APP_KEY to enable: accepted write count gates and rejection metric gates." - log " Verify EPRW metrics manually using the Phase 5 queries below." -fi -# --------------------------------------------------------------------------- -# Phase 5 — MANUAL VALIDATION RUNBOOK -# -# The following queries must be run manually in go/redapl → Queries → SQL. -# This script cannot execute DDSQL queries automatically. -# -# REDAPL query visibility lag (the five-minute SLO): -# _modified_at tells you when EPRW wrote the row, not when DDSQL first -# exposed it. To measure actual query visibility lag: -# 1. Record wall-clock time when you first run Query C and rows appear. -# 2. Subtract START_TIME (printed below) from that wall-clock time. -# 3. That difference is the REDAPL query visibility lag for this run. -# Poll Query C every 60s after the trigger until rows appear. -# --------------------------------------------------------------------------- -header "Phase 5 — MANUAL VALIDATION RUNBOOK (go/redapl → Queries → SQL)" - -cat < distinct_resources → per-instance field leaked into key -SELECT workload_type, deployment_model, - COUNT(*) AS total_rows, - COUNT(DISTINCT resource_id) AS distinct_resources -FROM udm.all.serverless_init_agent -GROUP BY workload_type, deployment_model -ORDER BY workload_type, deployment_model; - --- B. serverless_compat_agent: exactly 1 row per function regardless of burst size --- PASS: azure_function=1, gcp_cloud_function_gen1=1 --- FAIL: any total_rows > 1 → cold-start fan-out reaching REDAPL -SELECT workload_type, - COUNT(*) AS total_rows, - COUNT(DISTINCT resource_id) AS distinct_resources -FROM udm.all.serverless_compat_agent -GROUP BY workload_type -ORDER BY workload_type; - --- C. REDAPL visibility: poll this every 60s until rows appear. --- Record wall-clock time when rows first appear — subtract ${START_TIME} --- to get REDAPL query visibility lag (must be ≤ 5 min at p95). --- _modified_at = when EPRW wrote the row (not when DDSQL exposed it). -SELECT 'serverless_init_agent' AS flavor, resource_id, workload_type, - _modified_at, - TIMESTAMPDIFF(MINUTE, TIMESTAMP '${START_TIME}', _modified_at) AS eprw_write_lag_min -FROM udm.all.serverless_init_agent -WHERE _modified_at >= TIMESTAMP '${START_TIME}' -UNION ALL -SELECT 'serverless_compat_agent', resource_id, workload_type, - _modified_at, - TIMESTAMPDIFF(MINUTE, TIMESTAMP '${START_TIME}', _modified_at) -FROM udm.all.serverless_compat_agent -WHERE _modified_at >= TIMESTAMP '${START_TIME}' -ORDER BY flavor, eprw_write_lag_min ASC; - --- D. _key correctness: must equal SanitizeString(resource_id), not uuid or composite. -SELECT _key, resource_id, workload_type -FROM udm.all.serverless_init_agent -WHERE _modified_at >= TIMESTAMP '${START_TIME}' -LIMIT 20; - -SELECT _key, resource_id, workload_type -FROM udm.all.serverless_compat_agent -WHERE _modified_at >= TIMESTAMP '${START_TIME}' -LIMIT 10; - --- E. Crawler joins: both gcp_run_service AND gcp_run_revision checked. --- The schema relationship is gcp_run_revision, sourced from --- /metadata.key_overrides.gcp_run_revision_key with on_empty fallback to resource_id. --- If the decoder provides a revision CCRID: gcp_run_revision resolves. --- If not (service-level CCRID fallback): gcp_run_service resolves. --- PASS: at least one of the two is non-NULL per row. --- FAIL (R6): both NULL → CCRID format matches neither crawler table. -SELECT a.resource_id, a.workload_type, a._key, - svc._key AS gcp_run_service_key, - rev._key AS gcp_run_revision_key -FROM udm.all.serverless_init_agent a -LEFT JOIN udm.all.gcp_run_service svc ON a._key = svc._key -LEFT JOIN udm.all.gcp_run_revision rev ON a._key = rev._key -WHERE a.workload_type = 'cloud_run_service' - AND a._modified_at >= TIMESTAMP '${START_TIME}' -LIMIT 20; - -SELECT a.resource_id, a.workload_type, a._key, - c._key AS crawler_key -FROM udm.all.serverless_init_agent a -LEFT JOIN udm.all.azure_container_app c ON a._key = c._key -WHERE a.workload_type = 'azure_container_app' - AND a._modified_at >= TIMESTAMP '${START_TIME}' -LIMIT 20; - -SELECT a.resource_id, a.workload_type, a._key, - c._key AS crawler_key -FROM udm.all.serverless_init_agent a -LEFT JOIN udm.all.azure_app_service c ON a._key = c._key -WHERE a.workload_type = 'azure_app_service' - AND a._modified_at >= TIMESTAMP '${START_TIME}' -LIMIT 20; - --- F. Legacy datadog_agent secondary write (confirm payload reached EPRW at all). --- UUID-keyed; one row per cold start. NOT the per-flavor table rows. -SELECT _key AS uuid, hostname, agent_version, install_method_tool, _first_seen_at -FROM udm.all.datadog_agent -WHERE install_method_tool IN ('serverless-init', 'serverless-compat') - AND _first_seen_at >= TIMESTAMP '${START_TIME}' -ORDER BY _first_seen_at DESC -LIMIT 20; - -DDSQL - -# --------------------------------------------------------------------------- -# Machine-readable report -# Counts and pass/fail for each RFC workload — written to RESULTS_DIR/report.json -# --------------------------------------------------------------------------- -END_EPOCH=$(date +%s) -DURATION=$(( END_EPOCH - START_EPOCH )) - -python3 - "${RESULTS_DIR}" "${START_TIME}" "${END_TIME}" \ - "${AGENT_IMAGE}" "${LOAD_STAGE}" "${WAVE_SIZES}" \ - "${COMPAT_CONCURRENCY}" "${MAX_WAVE}" <<'REPORT_PY' -import json, sys, os, re -from datetime import datetime, timezone - -results_dir, start_time, end_time, agent_image, load_stage, wave_sizes, \ - compat_concurrency, max_wave = sys.argv[1:] - -def count_lines(path): - try: - with open(path) as f: - return sum(1 for _ in f) - except FileNotFoundError: - return None - -# Parse burst summary from log (count succeeded/failed lines) -burst_log = os.path.join(results_dir, 'burst-init.log') -burst_ok = burst_fail = 0 -if os.path.exists(burst_log): - with open(burst_log) as f: - for line in f: - m = re.search(r'✓ (\d+)/\d+ succeeded', line) - if m: - burst_ok += int(m.group(1)) - m2 = re.search(r'✗ (\d+) failed', line) - if m2: - burst_fail += int(m2.group(1)) - -# Coverage check result -coverage_log = os.path.join(results_dir, 'coverage-check.log') -coverage_hard_failed = False -if os.path.exists(coverage_log): - with open(coverage_log) as f: - content = f.read() - coverage_hard_failed = 'HARD FAIL' in content - -report = { - 'run': { - 'start': start_time, - 'end': end_time, - 'agent_image': agent_image, - 'load_stage': load_stage, - 'wave_sizes': wave_sizes, - 'max_instances_during_burst': int(max_wave), - }, - 'counts': { - 'note': 'HTTP requests only — instance starts and REDAPL rows require manual verification', - 'init_burst_http_ok': burst_ok, - 'init_burst_http_fail': burst_fail, - 'compat_burst_concurrency': int(compat_concurrency), - 'instance_starts_observed': 'not measured — requires Cloud Run log query', - 'eprw_accepted_writes': 'see eprw-poll.log or Phase 5 metrics', - 'eprw_rejected_writes': 'see eprw-poll.log or Phase 5 metrics', - 'distinct_resource_ids': 'not measured — requires DDSQL query (Phase 5)', - 'redapl_rows': 'not measured — requires DDSQL query (Phase 5)', - }, - 'rfc_workloads': [ - {'id': 'SI-01', 'workload': 'cloud_run_service / in-container', 'automated': True, 'result': 'HTTP triggered — REDAPL rows not auto-verified'}, - {'id': 'SI-02', 'workload': 'cloud_run_service / sidecar', 'automated': True, 'result': 'HTTP triggered — REDAPL rows not auto-verified'}, - {'id': 'SI-03', 'workload': 'cloud_run_job / in-container', 'automated': False, 'result': 'NOT TESTED — no HTTP endpoint; needs gcloud run jobs execute'}, - {'id': 'SI-04', 'workload': 'cloud_function_gen2 / sidecar', 'automated': True, 'result': 'HTTP triggered — REDAPL rows not auto-verified'}, - {'id': 'SI-05', 'workload': 'azure_container_app / in-container', 'automated': False, 'result': 'NOT TESTED — Azure not in apps.yml without discover + az login'}, - {'id': 'SI-06', 'workload': 'azure_container_app / sidecar', 'automated': False, 'result': 'NOT TESTED — Azure not in apps.yml without discover + az login'}, - {'id': 'SI-07', 'workload': 'azure_app_service / in-container', 'automated': False, 'result': 'NOT TESTED — Azure not in apps.yml without discover + az login'}, - {'id': 'SI-08', 'workload': 'azure_app_service / SITECONTAINERS', 'automated': False, 'result': 'NOT TESTED — Azure not in apps.yml without discover + az login'}, - {'id': 'SI-09', 'workload': 'azure_app_service / linux-code', 'automated': False, 'result': 'NOT TESTED — Azure not in apps.yml without discover + az login'}, - {'id': 'SC-01', 'workload': 'azure_function / compat', 'automated': True, 'result': 'HTTP triggered — REDAPL rows not auto-verified'}, - {'id': 'SC-02', 'workload': 'gcp_cloud_function_gen1 / compat', 'automated': True, 'result': 'HTTP triggered — REDAPL rows not auto-verified'}, - ], - 'unverified': [ - 'actual instance starts (not measured)', - 'inventory payload count per service (not measured)', - 'EPRW accepted writes per resource_id (requires poll:eprw)', - 'EPRW rejection counts (requires poll:eprw)', - 'REDAPL row count per resource_id (requires DDSQL, Phase 5)', - 'REDAPL visibility lag wall-clock time (requires polling, Phase 5)', - 'RFC 7.2 restart identity row stability (requires DDSQL)', - 'RFC 7.3 config upgrade row update (requires test:ordering + DDSQL)', - 'RFC 7.4 flip-flop convergence (requires test:ordering + DDSQL)', - 'crawler join correctness (requires DDSQL, Phase 5 Query E)', - 'TTL expiration and reactivation (not scripted)', - 'FleetQuerier and UI verification (not scripted)', - ], - 'coverage_preflight_hard_failed': coverage_hard_failed, -} - -out = os.path.join(results_dir, 'report.json') -with open(out, 'w') as f: - json.dump(report, f, indent=2) -print(f"Report written to: {out}") -REPORT_PY - -# --------------------------------------------------------------------------- -# Summary -# --------------------------------------------------------------------------- -header "RC validation complete" - -log "Start : ${START_TIME}" -log "Baseline complete : ${BASELINE_END:-skipped}" -log "End : ${END_TIME}" -log "Duration : ${DURATION}s" -log "AGENT_IMAGE : ${AGENT_IMAGE}" -log "LOAD_STAGE : ${LOAD_STAGE} (compat concurrency=${COMPAT_CONCURRENCY})" -log "WAVE_SIZES (init) : ${WAVE_SIZES} (max-instances set to ${MAX_WAVE} during burst)" -log "Results dir : ${RESULTS_DIR}/" -log "" -log "Coverage (HTTP-triggered only — REDAPL row count requires manual DDSQL verification):" -log " SI-01 cloud_run_service / in-container HTTP ✓ | instances NOT measured | REDAPL rows NOT auto-verified" -log " SI-02 cloud_run_service / sidecar HTTP ✓ | instances NOT measured | REDAPL rows NOT auto-verified" -log " SI-03 cloud_run_job / in-container NOT TESTED (no HTTP endpoint — needs gcloud run jobs execute)" -log " SI-04 cloud_function_gen2 / sidecar HTTP ✓ | instances NOT measured | REDAPL rows NOT auto-verified" -log " SI-05 azure_container_app / in-container NOT TESTED (Azure not in apps.yml without discover + az login)" -log " SI-06 azure_container_app / sidecar NOT TESTED (Azure not in apps.yml without discover + az login)" -log " SI-07 azure_app_service / in-container NOT TESTED (Azure not in apps.yml without discover + az login)" -log " SI-08 azure_app_service / SITECONTAINERS NOT TESTED (Azure not in apps.yml without discover + az login)" -log " SI-09 azure_app_service / linux-code NOT TESTED (Azure not in apps.yml without discover + az login)" -log " SC-01 azure_function HTTP ✓ | REDAPL rows NOT auto-verified" -log " SC-02 gcp_cloud_function_gen1 HTTP ✓ | REDAPL rows NOT auto-verified" -log "" -log "What this script does NOT verify automatically:" -log " - Actual instance starts (metric: container/instance_count or Cloud Run logs)" -log " - Inventory payload count per service" -log " - EPRW accepted write count (Phase 4b; requires DD_APP_KEY)" -log " - REDAPL row count == distinct resource_ids (requires Phase 5 DDSQL queries)" -log " - REDAPL visibility lag wall-clock time (poll Phase 5 Query C)" -log " - RFC 7.2/7.3/7.4 ordering guarantees (run: npm run test:ordering)" -log " - Crawler join correctness (Phase 5 Query E)" -log "" -log "Next steps:" -log " Full DDSQL query set: ${SCRIPT_DIR}/check-redapl.sh" -log " Log diagnostics: GCP_PROJECT=${GCP_PROJECT} ${SCRIPT_DIR}/check-logs.sh" -log " Ordering test: (cd ${INIT_SM_DIR} && npm run test:ordering)" -log "" -log "RFC approval gates (verify manually using Phase 5 queries above):" -log " Gate 1: rows == resources for every workload_type in both tables (Queries A + B)" -log " Gate 2: REDAPL query visibility lag ≤ 5 min — poll Query C until rows appear" -log " Gate 3: crawler_key NOT NULL for all rows (Query E)" -log " Gate 4: all rejection metrics zero (Phase 4b or manual)" -log " Gate 5: no EPRW CPU/memory/error regression during burst" -log "" -log "Machine-readable report: ${RESULTS_DIR}/report.json" - -# Write a concise JSON summary for quick pass/fail inspection. -# This is separate from the detailed report.json produced by the Python block above. -_cloud_run_v2_preserved=$(grep -q 'product: cloud-run-v2' "${INIT_SM_DIR}/apps.yml" 2>/dev/null && echo true || echo false) -_si03_covered=$([ -n "${JOB_NAME:-}" ] && echo true || echo false) - -cat > "${RESULTS_DIR}/rc-summary.json" < app.py -EXPOSE 8080 -CMD ["python", "app.py"] - -FROM node:22-slim AS node-plain -WORKDIR /app -RUN printf '%s\n' \ - "const http=require('http');" \ - "http.createServer((q,r)=>{r.writeHead(200);r.end('Hello World!')}).listen(8080,'0.0.0.0');" > app.js -EXPOSE 8080 -CMD ["node", "app.js"] - -FROM golang:1.24-bookworm AS go-build -WORKDIR /src -RUN printf '%s\n' \ - 'package main' \ - 'import ("fmt"; "net/http")' \ - 'func main(){http.HandleFunc("/",func(w http.ResponseWriter,r *http.Request){fmt.Fprint(w,"Hello World!")});http.ListenAndServe(":8080",nil)}' > main.go \ - && go build -o /app main.go -FROM debian:bookworm-slim AS go-plain -COPY --from=go-build /app /app -EXPOSE 8080 -CMD ["/app"] - -FROM eclipse-temurin:21-jdk-jammy AS java-build -WORKDIR /src -RUN printf '%s\n' \ - 'import com.sun.net.httpserver.HttpServer;' \ - 'import java.net.InetSocketAddress;' \ - 'public class App { public static void main(String[] a) throws Exception {' \ - 'var s=HttpServer.create(new InetSocketAddress(8080),0); s.createContext("/",e->{byte[] b="Hello World!".getBytes();e.sendResponseHeaders(200,b.length);e.getResponseBody().write(b);e.close();});s.start();}}' > App.java \ - && javac App.java -FROM eclipse-temurin:21-jre-jammy AS java-plain -WORKDIR /app -COPY --from=java-build /src/App.class . -EXPOSE 8080 -CMD ["java", "App"] - -FROM mcr.microsoft.com/dotnet/sdk:8.0 AS dotnet-build -WORKDIR /src -RUN printf '%s\n' 'net8.0enable' > app.csproj \ - && printf '%s\n' 'var b=WebApplication.CreateBuilder(args);var a=b.Build();a.MapGet("/",()=>"Hello World!");a.Run("http://0.0.0.0:8080");' > Program.cs \ - && dotnet publish -c Release -o /out -FROM mcr.microsoft.com/dotnet/aspnet:8.0 AS dotnet-plain -WORKDIR /app -COPY --from=dotnet-build /out . -EXPOSE 8080 -CMD ["dotnet", "app.dll"] - -FROM ruby:3.3-slim AS ruby-plain -WORKDIR /app -RUN printf '%s\n' \ - 'require "socket"' \ - 's=TCPServer.new("0.0.0.0",8080)' \ - 'loop{c=s.accept;c.gets;while (l=c.gets);break if l=="\r\n";end;c.write "HTTP/1.1 200 OK\r\nContent-Length: 12\r\n\r\nHello World!";c.close}' > app.rb -EXPOSE 8080 -CMD ["ruby", "app.rb"] - -FROM php:8.4-cli AS php-plain -WORKDIR /app -RUN printf '%s\n' '' > index.php -EXPOSE 8080 -CMD ["php", "-S", "0.0.0.0:8080", "-t", "/app"] - -FROM ${AGENT_IMAGE} AS agent -FROM ${RUNTIME}-plain AS plain - -FROM python-plain AS python-init -COPY --from=agent /serverless-init /serverless-init -ENTRYPOINT ["/serverless-init"] -CMD ["python", "app.py"] - -FROM node-plain AS node-init -COPY --from=agent /serverless-init /serverless-init -ENTRYPOINT ["/serverless-init"] -CMD ["node", "app.js"] - -FROM go-plain AS go-init -COPY --from=agent /serverless-init /serverless-init -ENTRYPOINT ["/serverless-init"] -CMD ["/app"] - -FROM java-plain AS java-init -COPY --from=agent /serverless-init /serverless-init -ENTRYPOINT ["/serverless-init"] -CMD ["java", "App"] - -FROM dotnet-plain AS dotnet-init -COPY --from=agent /serverless-init /serverless-init -ENTRYPOINT ["/serverless-init"] -CMD ["dotnet", "app.dll"] - -FROM ruby-plain AS ruby-init -COPY --from=agent /serverless-init /serverless-init -ENTRYPOINT ["/serverless-init"] -CMD ["ruby", "app.rb"] - -FROM php-plain AS php-init -COPY --from=agent /serverless-init /serverless-init -ENTRYPOINT ["/serverless-init"] -CMD ["php", "-S", "0.0.0.0:8080", "-t", "/app"] diff --git a/scripts/svls9604/matrix.json b/scripts/svls9604/matrix.json deleted file mode 100644 index 3cd2bd2..0000000 --- a/scripts/svls9604/matrix.json +++ /dev/null @@ -1,40 +0,0 @@ -{ - "runtimes": ["python", "node", "go", "java", "dotnet", "ruby", "php"], - "profiles": { - "gcp": [ - {"id": "SI-01", "provider": "gcp", "workload_type": "cloud_run_service", "deployment_model": "in-container", "runtimes": "all", "variants": ["busy", "coldstart"]}, - {"id": "SI-02", "provider": "gcp", "workload_type": "cloud_run_service", "deployment_model": "sidecar", "runtimes": "all", "variants": ["busy", "coldstart"]}, - {"id": "SI-03", "provider": "gcp", "workload_type": "cloud_run_job", "deployment_model": "in-container", "runtimes": ["python"], "variants": ["execution"]}, - {"id": "SI-04", "provider": "gcp", "workload_type": "cloud_run_function", "deployment_model": "sidecar", "runtimes": "all", "variants": ["busy", "coldstart"]}, - {"id": "SC-02", "provider": "gcp", "workload_type": "cloud_function", "runtimes": ["node"], "variants": ["function"]} - ], - "gcp-sanity": [ - {"id": "SI-01", "provider": "gcp", "workload_type": "cloud_run_service", "deployment_model": "in-container", "runtimes": ["python"], "variants": ["busy"]}, - {"id": "SI-03", "provider": "gcp", "workload_type": "cloud_run_job", "deployment_model": "in-container", "runtimes": ["python"], "variants": ["execution"]}, - {"id": "SI-04", "provider": "gcp", "workload_type": "cloud_run_function", "deployment_model": "sidecar", "runtimes": ["python"], "variants": ["busy"]}, - {"id": "SC-02", "provider": "gcp", "workload_type": "cloud_function", "runtimes": ["node"], "variants": ["function"]} - ], - "azure-sanity": [ - {"id": "SI-05", "provider": "azure", "workload_type": "azure_container_app", "deployment_model": "in-container", "runtimes": ["python"], "variants": ["busy"]}, - {"id": "SI-06", "provider": "azure", "workload_type": "azure_container_app", "deployment_model": "sidecar", "runtimes": ["python"], "variants": ["busy"]}, - {"id": "SI-07", "provider": "azure", "workload_type": "azure_app_service", "deployment_model": "in-container", "runtimes": ["python"], "variants": ["busy"]}, - {"id": "SI-08", "provider": "azure", "workload_type": "azure_app_service", "deployment_model": "sidecar", "runtimes": ["python"], "variants": ["busy"]}, - {"id": "SI-09", "provider": "azure", "workload_type": "azure_app_service", "deployment_model": "sidecar", "runtimes": ["python"], "variants": ["busy"], "shape": "linux-code"}, - {"id": "SC-01", "provider": "azure", "workload_type": "azure_function", "runtimes": ["node"], "variants": ["function"]} - ], - "gcp-baseline": [ - {"id": "SI-01", "provider": "gcp", "workload_type": "cloud_run_service", "deployment_model": "in-container", "runtimes": ["python"], "variants": ["busy"]} - ], - "azure-baseline": [ - {"id": "SI-07", "provider": "azure", "workload_type": "azure_app_service", "deployment_model": "in-container", "runtimes": ["python"], "variants": ["busy"]} - ], - "azure": [ - {"id": "SI-05", "provider": "azure", "workload_type": "azure_container_app", "deployment_model": "in-container", "runtimes": "all", "variants": ["busy", "coldstart"]}, - {"id": "SI-06", "provider": "azure", "workload_type": "azure_container_app", "deployment_model": "sidecar", "runtimes": "all", "variants": ["busy", "coldstart"]}, - {"id": "SI-07", "provider": "azure", "workload_type": "azure_app_service", "deployment_model": "in-container", "runtimes": "all", "variants": ["busy", "coldstart"]}, - {"id": "SI-08", "provider": "azure", "workload_type": "azure_app_service", "deployment_model": "sidecar", "runtimes": "all", "variants": ["busy", "coldstart"]}, - {"id": "SI-09", "provider": "azure", "workload_type": "azure_app_service", "deployment_model": "sidecar", "runtimes": ["node", "dotnet", "python"], "variants": ["busy", "coldstart"], "shape": "linux-code"}, - {"id": "SC-01", "provider": "azure", "workload_type": "azure_function", "runtimes": ["node"], "variants": ["function"]} - ] - } -} diff --git a/scripts/svls9604/report.py b/scripts/svls9604/report.py deleted file mode 100644 index 2120919..0000000 --- a/scripts/svls9604/report.py +++ /dev/null @@ -1,345 +0,0 @@ -#!/usr/bin/env python3 -import argparse -import csv -import json -import pathlib - - -NOT_MEASURED = "NOT MEASURED" - -def measured_number(value): - return isinstance(value,(int,float)) and not isinstance(value,bool) - - -def load_csv(path): - if not path or not path.exists(): - return [] - with path.open(newline="") as handle: - return list(csv.DictReader(handle)) - - -def status(ok, measured=True): - if not measured: - return NOT_MEASURED - return "PASS" if ok else "FAIL" - - -def aggregate_stage(stage): - if "resources" in stage and stage["resources"]: - resources=stage["resources"] - attempts=sum(item.get("attempts",1) for item in resources) - successes=sum(item.get("successes",1 if 200 <= item.get("http_status",0) < 400 else 0) - for item in resources) - failures=sum(item.get("failures",0 if 200 <= item.get("http_status",0) < 400 else 1) - for item in resources) - return attempts,successes,failures - if "rounds" in stage: - return (sum(item["attempts"] for item in stage["rounds"]), - sum(item["successes"] for item in stage["rounds"]), - sum(item["failures"] for item in stage["rounds"])) - return stage.get("attempts",0),stage.get("successes",0),stage.get("failures",0) - - -def find_default_csv(run_dir, name): - candidates=[run_dir/f"{name}.csv",run_dir/"pup"/f"{name}.csv"] - return next((path for path in candidates if path.exists()),None) - -def resource_expected_ids(resource): - return resource.get("expected_resource_ids") or [resource["resource_id"]] - -def table_evidence(resources, rows): - expected={resource_id.lower():resource for resource in resources - for resource_id in resource_expected_ids(resource)} - selected=[row for row in rows if row.get("resource_id","").lower() in expected] - ids=[row.get("resource_id","").lower() for row in selected if row.get("resource_id")] - distinct=set(ids) - missing=sorted(key for key in expected if key not in distinct) - unexpected=sorted(row.get("resource_id","") for row in rows - if row.get("resource_id") and row["resource_id"].lower() not in expected) - required=("resource_id","resource_name","workload_type") - nulls={field:sum(1 for row in selected if not row.get(field)) for field in required} - revision_rows=[row for row in selected if row.get("workload_type") in - ("cloud_run_service","cloud_run_function","azure_container_app")] - nulls["parent_resource_id"]=sum(1 for row in revision_rows if not row.get("parent_resource_id")) - nulls["deployment_id"]=sum(1 for row in revision_rows if not row.get("deployment_id")) - return {"rows":len(selected),"distinct_resource_ids":len(distinct),"missing":missing, - "unexpected":unexpected,"duplicates":len(ids)-len(distinct),"required_nulls":nulls} - - -def render(manifest, init_rows, compat_rows, pipeline): - resources=manifest.get("resources",[]) - init_resources=[r for r in resources if r["id"].startswith("SI-")] - compat_resources=[r for r in resources if r["id"].startswith("SC-")] - init=table_evidence(init_resources,init_rows) - compat=table_evidence(compat_resources,compat_rows) - table_measured=bool(init_rows or compat_rows) - acceptance_measured=all(measured_number(pipeline.get(key)) for key in ("producer_attempts","decoder_accepts")) - edge_measured=all(measured_number(pipeline.get(key)) for key in ("decoder_accepts","resource_edge_successes","resource_edge_failures")) - iris=pipeline.get("iris_primary",{}) - iris_measured=all(measured_number(iris.get(outcome)) for outcome in ("CREATED","UPDATED","EXTENDED","IGNORED","ERROR")) - iris_total=sum(iris.values()) if iris_measured else 0 - producer_reasons=pipeline.get("producer_reasons",{}) - decoder_reasons=pipeline.get("decoder_reasons",{}) - reason_names=("startup","periodic","refresh") - reasons_measured=all( - measured_number(producer_reasons.get(reason)) and measured_number(decoder_reasons.get(reason)) - for reason in reason_names - ) - deployed=len(resources) - expected_init_ids=sum(len(resource_expected_ids(r)) for r in init_resources) - expected_compat_ids=sum(len(resource_expected_ids(r)) for r in compat_resources) - revision_stages={stage.get("id"):stage for stage in manifest.get("load_stages",[]) if stage.get("id") in ("L6","L7")} - revision_measured=all(stage_id in revision_stages and revision_stages[stage_id].get("status") != NOT_MEASURED - for stage_id in ("L6","L7")) - stage_pipeline_complete=all( - all(measured_number(pipeline.get("stages",{}).get(stage.get("id"),{}).get(key)) - for key in ("producer_attempts","decoder_accepts","resource_edge_successes","resource_edge_failures")) - for stage in manifest.get("load_stages",[]) - ) - expected_revision_ids={resource_id.lower() for resource in init_resources for resource_id in resource_expected_ids(resource)} - resource_pipeline_complete=all( - resource_id in {key.lower() for key in pipeline.get("resources",{})} - for resource_id in expected_revision_ids - ) and all( - all(measured_number(evidence.get(key)) for key in ("producer_attempts","decoder_accepts","resource_edge_successes")) and - all(measured_number(evidence.get("iris_primary",{}).get(outcome)) for outcome in ("CREATED","UPDATED","EXTENDED","IGNORED","ERROR")) - for evidence in pipeline.get("resources",{}).values() - ) - tables_valid=(init["rows"]==expected_init_ids and compat["rows"]==expected_compat_ids and - init["duplicates"]==0 and compat["duplicates"]==0 and not init["missing"] and - not compat["missing"] and not init["unexpected"] and not compat["unexpected"] and - not any(init["required_nulls"].values()) and not any(compat["required_nulls"].values())) - reason_totals_valid=(reasons_measured and - sum(producer_reasons.values())==pipeline.get("producer_attempts") and - sum(decoder_reasons.values())==pipeline.get("decoder_accepts") and - all(producer_reasons[reason]==decoder_reasons[reason] for reason in reason_names)) - pipeline_valid=(acceptance_measured and edge_measured and iris_measured and reason_totals_valid and - pipeline.get("producer_attempts")==pipeline.get("decoder_accepts") and - pipeline.get("decoder_accepts")==pipeline.get("resource_edge_successes",0)+pipeline.get("resource_edge_failures",0) and - pipeline.get("resource_edge_failures")==0 and pipeline.get("resource_edge_successes")==iris_total) - complete=table_measured and tables_valid and pipeline_valid and revision_measured and stage_pipeline_complete and resource_pipeline_complete - triggered=sum(1 for r in resources if r.get("baseline",{}).get("status") in ("ok","executed")) - lines=[ - "# Serverless Agent REDAPL RC Results", - "", - f"**Status:** {'COMPLETE' if complete else 'PARTIAL — pipeline, revision, or DDSQL evidence still required'}", - "", - f"**Run ID:** `{manifest.get('run_id','')}`", - "", - f"**Environment:** `{manifest.get('dd_env','')}` on `datad0g.com`, org 2", - "", - f"**Start and end time:** {manifest.get('started_at',NOT_MEASURED)} to {manifest.get('completed_at',NOT_MEASURED)}", - "", - f"**Declared scope:** `{manifest.get('profile')}` / `{manifest.get('suite')}`", - "", - "## 1. Build and environment record", - "", - "| Item | Observed value |", - "|---|---|", - f"| Agent image tag and digest | `{manifest.get('agent_image',NOT_MEASURED)}` |", - f"| Datadog Agent commit | `{manifest.get('candidate_commits',{}).get('datadog_agent',manifest.get('agent_sha',NOT_MEASURED))}` |", - f"| Serverless Components commit | `{manifest.get('candidate_commits',{}).get('serverless_components',NOT_MEASURED)}` |", - f"| Compat JS commit | `{manifest.get('candidate_commits',{}).get('datadog_serverless_compat_js',NOT_MEASURED)}` |", - f"| EPRW decoder deployed commit | `{pipeline.get('eprw_commit',NOT_MEASURED)}` |", - f"| Iris deployed commit | `{pipeline.get('iris_commit',NOT_MEASURED)}` |", - f"| EPRW debug tracking | `{pipeline.get('eprw_debug_tracking',NOT_MEASURED)}` |", - f"| Iris upsert experiment | `{pipeline.get('iris_upsert_telemetry',NOT_MEASURED)}` |", - "", - "## 2. Executive results", - "", - "| Result | Expected | Observed | Status |", - "|---|---:|---:|---|", - f"| Resources deployed | {len(resources)} | {deployed} | {status(deployed==len(resources))} |", - f"| Baseline triggers | {deployed} | {triggered} | {status(triggered==deployed)} |", - f"| EPRW decoder accepts | Producer attempts | {pipeline.get('decoder_accepts',NOT_MEASURED)} | {status(pipeline.get('decoder_accepts')==pipeline.get('producer_attempts'),acceptance_measured)} |", - f"| Resource Edge failures | 0 | {pipeline.get('resource_edge_failures',NOT_MEASURED)} | {status(pipeline.get('resource_edge_failures')==0,'resource_edge_failures' in pipeline)} |", - f"| REDAPL init revision/workload rows | {expected_init_ids} | {init['rows'] if init_rows else NOT_MEASURED} | {status(init['rows']==expected_init_ids,bool(init_rows))} |", - f"| REDAPL compat rows | {expected_compat_ids} | {compat['rows'] if compat_rows else NOT_MEASURED} | {status(compat['rows']==expected_compat_ids,bool(compat_rows))} |", - f"| Duplicate resource keys | 0 | {(init['duplicates']+compat['duplicates']) if table_measured else NOT_MEASURED} | {status(init['duplicates']+compat['duplicates']==0,table_measured)} |", - "", - "## 3. Per-workload results", - "", - "| ID | Table | Workload | Runtime | Model | Resource | Baseline | REDAPL |", - "|---|---|---|---|---|---|---|---|", - ] - init_ids={row.get("resource_id","").lower() for row in init_rows} - compat_ids={row.get("resource_id","").lower() for row in compat_rows} - for resource in resources: - table="serverless_init_agent" if resource["id"].startswith("SI-") else "serverless_compat_agent" - ids=init_ids if table=="serverless_init_agent" else compat_ids - observed=status(all(resource_id.lower() in ids for resource_id in resource_expected_ids(resource)),bool(init_rows if table=="serverless_init_agent" else compat_rows)) - lines.append(f"| {resource['id']} | `{table}` | `{resource['workload_type']}` | {resource['runtime']} | {resource['deployment_model']} | `{resource['name']}` | {resource.get('baseline',{}).get('status',NOT_MEASURED)} | {observed} |") - lines.extend([ - "", - "## 4. Producer, EPRW, and Iris reconciliation", - "", - "| Measurement | Observed | Status |", - "|---|---:|---|", - f"| Producer inventory attempts | {pipeline.get('producer_attempts',NOT_MEASURED)} | {status(pipeline.get('producer_attempts')==pipeline.get('decoder_accepts'),acceptance_measured)} |", - f"| EPRW decoder accepts | {pipeline.get('decoder_accepts',NOT_MEASURED)} | {status(pipeline.get('decoder_accepts')==pipeline.get('producer_attempts'),acceptance_measured)} |", - f"| Resource Edge successes | {pipeline.get('resource_edge_successes',NOT_MEASURED)} | {status(pipeline.get('decoder_accepts')==pipeline.get('resource_edge_successes',0)+pipeline.get('resource_edge_failures',0),edge_measured)} |", - f"| Resource Edge failures | {pipeline.get('resource_edge_failures',NOT_MEASURED)} | {status(pipeline.get('resource_edge_failures')==0,'resource_edge_failures' in pipeline)} |", - ]) - for outcome in ("CREATED","UPDATED","EXTENDED","IGNORED","ERROR"): - lines.append(f"| Primary Iris {outcome} | {iris.get(outcome,NOT_MEASURED)} | {status(iris.get(outcome,0)==0 if outcome=='ERROR' else True,outcome in iris)} |") - lines.extend([ - "", - "### Collection reasons", - "", - "| Reason | Producer reports | EPRW accepts | Status |", - "|---|---:|---:|---|", - ]) - for reason in reason_names: - producer=producer_reasons.get(reason,NOT_MEASURED) - decoder=decoder_reasons.get(reason,NOT_MEASURED) - measured=measured_number(producer) and measured_number(decoder) - lines.append(f"| `{reason}` | {producer} | {decoder} | {status(producer==decoder,measured)} |") - lines.extend([ - "", - "Reconciliation gates:", - "", - "```text", - "decoder accepts = Resource Edge successes + Resource Edge failures", - "Resource Edge successes = primary Iris CREATED + UPDATED + EXTENDED + IGNORED + ERROR", - "```", - "", - "## 5. REDAPL identity and data results", - "", - "| Table | Expected IDs | Rows | Distinct IDs | Duplicates | Missing | Unexpected | Required nulls | Status |", - "|---|---:|---:|---:|---:|---:|---:|---:|---|", - f"| `serverless_init_agent` | {expected_init_ids} | {init['rows'] if init_rows else NOT_MEASURED} | {init['distinct_resource_ids'] if init_rows else NOT_MEASURED} | {init['duplicates'] if init_rows else NOT_MEASURED} | {len(init['missing']) if init_rows else NOT_MEASURED} | {len(init['unexpected']) if init_rows else NOT_MEASURED} | {sum(init['required_nulls'].values()) if init_rows else NOT_MEASURED} | {status(init['rows']==expected_init_ids and init['duplicates']==0 and not init['missing'] and not init['unexpected'] and not any(init['required_nulls'].values()),bool(init_rows))} |", - f"| `serverless_compat_agent` | {expected_compat_ids} | {compat['rows'] if compat_rows else NOT_MEASURED} | {compat['distinct_resource_ids'] if compat_rows else NOT_MEASURED} | {compat['duplicates'] if compat_rows else NOT_MEASURED} | {len(compat['missing']) if compat_rows else NOT_MEASURED} | {len(compat['unexpected']) if compat_rows else NOT_MEASURED} | {sum(compat['required_nulls'].values()) if compat_rows else NOT_MEASURED} | {status(compat['rows']==expected_compat_ids and compat['duplicates']==0 and not compat['missing'] and not compat['unexpected'] and not any(compat['required_nulls'].values()),bool(compat_rows))} |", - "", - "## 6. Load-stage results", - "", - "| Stage | Scenario | Sent | Successful | Failed | Pipeline evidence | Status |", - "|---|---|---:|---:|---:|---|---|", - ]) - for stage in manifest.get("load_stages",[]): - attempts,successes,failures=aggregate_stage(stage) - evidence=pipeline.get("stages",{}).get(stage["id"],{}) - stage_pipeline_measured=all(measured_number(evidence.get(key)) for key in ("producer_attempts","decoder_accepts","resource_edge_successes","resource_edge_failures")) - pipeline_summary=(f"producer={evidence['producer_attempts']}, decoder={evidence['decoder_accepts']}, " - f"edge_ok={evidence['resource_edge_successes']}, edge_failed={evidence['resource_edge_failures']}") if stage_pipeline_measured else NOT_MEASURED - if stage.get("status") == NOT_MEASURED: - stage_status=NOT_MEASURED - elif failures: - stage_status="FAIL" - elif not stage_pipeline_measured: - stage_status="PARTIAL" - else: - stage_status=status(evidence["producer_attempts"]==evidence["decoder_accepts"] and - evidence["decoder_accepts"]==evidence["resource_edge_successes"]+evidence["resource_edge_failures"] and - evidence["resource_edge_failures"]==0) - lines.append(f"| {stage['id']} | {stage['scenario']} | {attempts or NOT_MEASURED} | {successes or NOT_MEASURED} | {failures if attempts else NOT_MEASURED} | {pipeline_summary} | {stage_status} |") - lines.extend([ - "", - "### Per-revision reconciliation", - "", - "| Resource ID | Producer starts | Decoder accepts | Edge successes | Iris total | Status |", - "|---|---:|---:|---:|---:|---|", - ]) - for resource_id,evidence in sorted(pipeline.get("resources",{}).items()): - per_iris=evidence.get("iris_primary",{}) - per_iris_measured=all(measured_number(per_iris.get(outcome)) for outcome in ("CREATED","UPDATED","EXTENDED","IGNORED","ERROR")) - per_measured=all(measured_number(evidence.get(key)) for key in ("producer_attempts","decoder_accepts","resource_edge_successes")) and per_iris_measured - per_iris_total=sum(per_iris.values()) if per_iris_measured else NOT_MEASURED - per_status=status(evidence.get("producer_attempts")==evidence.get("decoder_accepts")==evidence.get("resource_edge_successes")==per_iris_total,per_measured) - lines.append(f"| `{resource_id}` | {evidence.get('producer_attempts',NOT_MEASURED)} | {evidence.get('decoder_accepts',NOT_MEASURED)} | {evidence.get('resource_edge_successes',NOT_MEASURED)} | {per_iris_total} | {per_status} |") - if not pipeline.get("resources"): - lines.append(f"| {NOT_MEASURED} | {NOT_MEASURED} | {NOT_MEASURED} | {NOT_MEASURED} | {NOT_MEASURED} | {NOT_MEASURED} |") - lines.extend([ - "", - "## 7. Staging evidence queries", - "", - "Use the run time window shown above and the following filters in `datad0g.com`:", - "", - "```text", - f"Environment: {manifest.get('dd_env','')}", - f"EPRW accepts: sum:event_platform_resource_writer.agentmetadata.serverless_write.accepted{{resource_type:serverless_init_agent}} by {{resource_type,workload_type,deployment_model,report_reason}}", - f"EPRW accepts (compat): sum:event_platform_resource_writer.agentmetadata.serverless_write.accepted{{resource_type:serverless_compat_agent}} by {{resource_type,workload_type,report_reason}}", - f"Producer startup log (Init): env:{manifest.get('dd_env','')} \"serverless-init: inventory report queued\"", - f"Iris primary logs: service:iris-node-go @message:\"serverless inventory upsert result\" @shadow_mode:false @resource_id:*{manifest.get('run_id','').lower()}*", - "EPRW debug logs: service:event-platform-resource-writer @resource_id:*RUN_ID*", - "```", - "", - "For L6, record provider startup/replica counts for `revision_b`; the request total is pressure applied, not the number of cold starts. For L7, confirm both revision resource IDs reach EPRW/Iris and both remain related to the same parent service/app.", - "", - "DDSQL exports:", - "", - "```sql", - "SELECT resource_id, parent_resource_id, resource_name, workload_type, deployment_model, deployment_id, runtime, dd_env, _first_seen_at, _modification_detected_at, _updated_at", - "FROM udm.all.serverless_init_agent", - f"WHERE dd_env = '{manifest.get('dd_env','')}'", - "ORDER BY parent_resource_id, deployment_id, resource_id;", - "", - "SELECT resource_id, resource_name, workload_type, runtime, dd_env, _first_seen_at, _modification_detected_at, _updated_at", - "FROM udm.all.serverless_compat_agent", - f"WHERE dd_env = '{manifest.get('dd_env','')}'", - "ORDER BY resource_id;", - "```", - "", - "## 8. Evidence still required", - "", - "- Export `serverless_init_agent.csv` and `serverless_compat_agent.csv` into this run directory, then rerun `report.py`.", - "- Export run-filtered EPRW and primary Iris counts into `pipeline-evidence.json`.", - "- Add producer and EPRW counts grouped by `report_reason` (`startup`, `periodic`, `refresh`).", - "- Add provider instance-start evidence for sequential cold starts and scale-out.", - "- Run the controlled `SeenAt`, revision A/B, TTL, crawler, Fleet, and UI checks.", - "", - "## 9. RFC approval gates", - "", - "| Gate | Status |", - "|---|---|", - f"| Every in-scope workload/revision reports a valid row | {status(init['rows']==expected_init_ids and compat['rows']==expected_compat_ids,bool(init_rows and compat_rows))} |", - f"| One row per `resource_id` | {status(init['duplicates']==0 and compat['duplicates']==0,table_measured)} |", - f"| EPRW/Iris counts reconcile | {status(edge_measured and iris_measured and pipeline.get('decoder_accepts')==pipeline.get('resource_edge_successes',0)+pipeline.get('resource_edge_failures',0) and pipeline.get('resource_edge_successes')==iris_total,edge_measured and iris_measured)} |", - f"| Producer reasons reconcile with EPRW | {status(all(producer_reasons.get(reason)==decoder_reasons.get(reason) for reason in reason_names),reasons_measured)} |", - f"| Older `SeenAt` cannot replace newer data | {NOT_MEASURED} |", - f"| Revision creation and traffic split executed | {status(revision_measured,revision_measured)} |", - f"| TTL and reactivation | {NOT_MEASURED} |", - f"| Crawler, Fleet, and UI | {NOT_MEASURED} |", - "", - "## 10. Conclusion", - "", - "The generated report distinguishes observed execution results from pipeline and UI evidence that has not yet been collected. Missing evidence is never converted into a pass.", - ]) - return "\n".join(lines)+"\n",{"init":init,"compat":compat} - - -def main(): - parser=argparse.ArgumentParser() - parser.add_argument("--manifest",type=pathlib.Path,required=True) - parser.add_argument("--init-csv",type=pathlib.Path) - parser.add_argument("--compat-csv",type=pathlib.Path) - parser.add_argument("--pipeline-evidence",type=pathlib.Path) - args=parser.parse_args() - run_dir=args.manifest.parent - manifest=json.loads(args.manifest.read_text()) - init_path=args.init_csv or find_default_csv(run_dir,"serverless_init_agent") - compat_path=args.compat_csv or find_default_csv(run_dir,"serverless_compat_agent") - pipeline_path=args.pipeline_evidence or run_dir/"pipeline-evidence.json" - pipeline=json.loads(pipeline_path.read_text()) if pipeline_path.exists() else {} - producer_path=run_dir/"producer-evidence.json" - if producer_path.exists(): - producer=json.loads(producer_path.read_text()) - pipeline.setdefault("producer_attempts",producer.get("events")) - pipeline.setdefault("producer_reasons",producer.get("reasons",{})) - pipeline.setdefault("stages",{}) - for stage_id,evidence in producer.get("stages",{}).items(): - pipeline["stages"].setdefault(stage_id,{}) - pipeline["stages"][stage_id].setdefault("producer_attempts",evidence.get("reports")) - pipeline.setdefault("resources",{}) - for resource_id,evidence in producer.get("resources",{}).items(): - pipeline["resources"].setdefault(resource_id,{}) - pipeline["resources"][resource_id].setdefault("producer_attempts",evidence.get("reports")) - markdown,tables=render(manifest,load_csv(init_path),load_csv(compat_path),pipeline) - (run_dir/"serverless-redapl-rc-results.md").write_text(markdown) - (run_dir/"report.json").write_text(json.dumps({"manifest":manifest,"tables":tables, - "pipeline":pipeline},indent=2)) - print(f"Report: {run_dir/'serverless-redapl-rc-results.md'}") - print(f"Machine-readable report: {run_dir/'report.json'}") - - -if __name__ == "__main__": - main() diff --git a/scripts/svls9604/run.sh b/scripts/svls9604/run.sh deleted file mode 100755 index 2bfcb4a..0000000 --- a/scripts/svls9604/run.sh +++ /dev/null @@ -1,18 +0,0 @@ -#!/usr/bin/env bash -set -euo pipefail - -SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" - -for arg in "$@"; do - if [[ "${arg}" == "--plan" ]]; then - exec python3 "${SCRIPT_DIR}/runner.py" "$@" - fi -done - -if [[ "${SVLS9604_DD_AUTHENTICATED:-}" != "1" ]]; then - exec dd-auth --site=datad0g.com --org-uuid=2 -- env \ - SVLS9604_DD_AUTHENTICATED=1 \ - python3 "${SCRIPT_DIR}/runner.py" "$@" -fi - -exec python3 "${SCRIPT_DIR}/runner.py" "$@" diff --git a/scripts/svls9604/runner.py b/scripts/svls9604/runner.py deleted file mode 100755 index bd8dfe5..0000000 --- a/scripts/svls9604/runner.py +++ /dev/null @@ -1,932 +0,0 @@ -#!/usr/bin/env python3 -import argparse -import concurrent.futures -import datetime as dt -import hashlib -import json -import os -import pathlib -import re -import secrets -import shlex -import subprocess -import sys -import tempfile -import time -import urllib.request -import zipfile - -ROOT = pathlib.Path(__file__).resolve().parent -WORKSPACE = ROOT.parents[2] -AGENT_REPO = pathlib.Path(os.environ.get("DATADOG_AGENT_DIR", WORKSPACE / "datadog-agent")) -COMPONENTS_REPO = pathlib.Path(os.environ.get("SERVERLESS_COMPONENTS_DIR", WORKSPACE / "serverless-components")) -COMPAT_JS_REPO = pathlib.Path(os.environ.get("COMPAT_JS_DIR", WORKSPACE / "datadog-serverless-compat-js")) -MATRIX = json.loads((ROOT / "matrix.json").read_text()) -RUN_ENV = "svls9604" -RUN_STARTED_AT = None - -FULL_LOAD_STAGES = (("L1", 10), ("L2", 50), ("L3", 100)) - -def aggregate_stage(stage): - if stage.get("resources"): - return (sum(item.get("attempts",1) for item in stage["resources"]), - sum(item.get("successes",1 if 200 <= item.get("http_status",0) < 400 else 0) for item in stage["resources"]), - sum(item.get("failures",0 if 200 <= item.get("http_status",0) < 400 else 1) for item in stage["resources"])) - if stage.get("rounds"): - return (sum(item["attempts"] for item in stage["rounds"]), - sum(item["successes"] for item in stage["rounds"]), - sum(item["failures"] for item in stage["rounds"])) - return stage.get("attempts",0),stage.get("successes",0),stage.get("failures",0) - -def run(cmd, *, cwd=None, capture=False, env=None): - shown = " ".join(shlex.quote(str(x)) for x in cmd) - api_key = os.environ.get("DD_API_KEY") - app_key = os.environ.get("DD_APP_KEY") - acr_password = os.environ.get("SVLS9604_ACR_PASSWORD") - if api_key: - shown = shown.replace(api_key, "***DD_API_KEY***") - if app_key: - shown = shown.replace(app_key, "***DD_APP_KEY***") - if acr_password: - shown = shown.replace(acr_password, "***ACR_PASSWORD***") - print(f"+ {shown}", flush=True) - merged = os.environ.copy() - if env: - merged.update(env) - result = subprocess.run(cmd, cwd=cwd, env=merged, text=True, - stdout=subprocess.PIPE if capture else None, - stderr=subprocess.PIPE if capture else None) - if result.returncode: - if capture: - print(result.stdout, end="", file=sys.stderr) - print(result.stderr, end="", file=sys.stderr) - raise RuntimeError(f"command failed ({result.returncode}): {shown}") - return result.stdout.strip() if capture else "" - -def git_sha(repo): - return run(["git", "rev-parse", "HEAD"], cwd=repo, capture=True) - -def utc_now(): - return dt.datetime.now(dt.timezone.utc).isoformat() - - -def write_manifest(path, *, run_id, profile, agent_sha, agent_image, resources, - stages=None, suite="full", **provider): - path.write_text(json.dumps({ - "run_id":run_id, - "dd_env":RUN_ENV, - "profile":profile, - "started_at":RUN_STARTED_AT, - "updated_at":utc_now(), - "suite":suite, - "agent_sha":agent_sha, - "agent_image":agent_image, - **provider, - "resources":resources, - "load_stages":stages or [], - },indent=2)) - -def remote_image_exists(image): - return subprocess.run(["docker", "buildx", "imagetools", "inspect", image], - stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL).returncode == 0 - -def short_runtime(runtime): - return {"python":"py", "node":"node", "go":"go", "java":"java", "dotnet":"dotnet", "ruby":"ruby", "php":"php"}[runtime] - -def expand(profile): - resources=[] - for item in MATRIX["profiles"][profile]: - runtimes=MATRIX["runtimes"] if item["runtimes"] == "all" else item["runtimes"] - for runtime in runtimes: - for variant in item["variants"]: - resources.append({**item, "runtime":runtime, "variant":variant}) - return resources - -def name_for(run_id, r): - model={"in-container":"in", "sidecar":"sc", "compat":"compat"}.get(r.get("deployment_model","compat"),"compat") - if r["provider"] == "azure": - # Container Apps names are limited to 32 characters. These names are - # also globally unique enough for App Service because run_id is random. - return f"sv{run_id[:10]}-{r['id'].lower().replace('-', '')}-{model}-{short_runtime(r['runtime'])}-{r['variant'][:4]}"[:32].rstrip("-") - return f"sv-{run_id}-{r['id'].lower().replace('-', '')}-{model}-{short_runtime(r['runtime'])}-{r['variant'][:4]}"[:63] - -def preflight(args): - required=["docker", "git", "python3", "gcloud" if args.profile in ("gcp", "gcp-sanity", "gcp-baseline") else "az"] - for tool in required: - run(["sh", "-c", f"command -v {shlex.quote(tool)} >/dev/null"]) - if not os.environ.get("DD_API_KEY") or os.environ.get("DD_SITE") != "datad0g.com": - raise RuntimeError("runner must be invoked through dd-auth for datad0g.com") - if args.profile in ("gcp", "gcp-sanity", "gcp-baseline"): - account=run(["gcloud","auth","list","--filter=status:ACTIVE","--format=value(account)"],capture=True) - if not account: - raise RuntimeError("gcloud has no active account") - print(f"Authenticated gcloud account: {account}") - else: - account=run(["az","account","show","--query","{name:name,id:id}","-o","json"],capture=True) - print(f"Authenticated Azure subscription: {account}") - run(["docker","info","--format={{.ServerVersion}}"],capture=True) - print(f"Datadog site: {os.environ['DD_SITE']} (dd-auth org UUID 2)") - -def image_digest(image): - output=run(["docker","buildx","imagetools","inspect",image],capture=True) - match=re.search(r"^Digest:\s+(sha256:[0-9a-f]+)$",output,re.MULTILINE) - if not match: - raise RuntimeError(f"could not determine digest for {image}") - return match.group(1) - -def build_agent(project, region, registry, run_id): - image=f"{registry}/serverless-init:{run_id}-v3" - sha=git_sha(AGENT_REPO) - release=json.loads((AGENT_REPO/"release.json").read_text()) - version=f"{release['current_milestone']}-dev" - if remote_image_exists(image): - print(f"Reusing candidate image {image}") - else: - run(["docker","buildx","build","--platform=linux/amd64","--push", - "--build-arg",f"GIT_COMMIT={sha[:12]}","--build-arg",f"AGENT_VERSION={version}", - "--build-arg",f"SERVERLESS_INIT_VERSION={version}", - "-f",str(AGENT_REPO/"scripts/serverless-deploy/Dockerfile.serverless-init"), - "-t",image,str(AGENT_REPO)]) - digest=run(["gcloud","artifacts","docker","images","describe",image, - f"--project={project}",f"--format=value(image_summary.digest)"],capture=True) - return f"{registry}/serverless-init@{digest}", sha - -def build_runtime_images(registry, run_id, agent_image): - fixture=ROOT/"fixtures" - images={} - def build_one(pair): - runtime,target=pair - # v2 init images use runtime-specific stages with explicit CMD values; - # serverless-init needs those argv values to launch the wrapped app. - image_suffix=f"{target}-v3" if target == "init" else target - build_target=f"{runtime}-init" if target == "init" else "plain" - image=f"{registry}/fixture-{runtime}-{image_suffix}:{run_id}" - if remote_image_exists(image): - print(f"Reusing fixture image {image}") - else: - run(["docker","buildx","build","--platform=linux/amd64","--push", - "--build-arg",f"RUNTIME={runtime}","--build-arg",f"AGENT_IMAGE={agent_image}", - "--target",build_target,"-t",image,"-f",str(fixture/"Dockerfile"),str(fixture)]) - return (runtime,target,image) - pairs=[(r,t) for r in MATRIX["runtimes"] for t in ("plain","init")] - with concurrent.futures.ThreadPoolExecutor(max_workers=3) as pool: - for runtime,target,image in pool.map(build_one,pairs): - images[f"{runtime}:{target}"]=image - return images - -def ensure_registry(project, region): - repo="svls9604" - exists=subprocess.run(["gcloud","artifacts","repositories","describe",repo, - f"--location={region}",f"--project={project}"], - stdout=subprocess.DEVNULL,stderr=subprocess.DEVNULL).returncode == 0 - if not exists: - run(["gcloud","artifacts","repositories","create",repo,"--repository-format=docker", - f"--location={region}",f"--project={project}"]) - run(["gcloud","auth","configure-docker",f"{region}-docker.pkg.dev","--quiet"]) - return f"{region}-docker.pkg.dev/{project}/{repo}" - -def env_list(name): - return {"DD_API_KEY":os.environ["DD_API_KEY"],"DD_SITE":"datad0g.com","DD_ENV":RUN_ENV, - "DD_SERVICE":name,"DD_SERVERLESS_DIAGNOSTIC_INFO":"true","DD_LOG_LEVEL":"debug", - "DD_SERVERLESS_INIT_INVENTORY_ENABLED":"true", - "DD_SERVERLESS_COMPAT_INVENTORY_ENABLED":"true"} - -def env_arg(values): - return ",".join(f"{k}={v}" for k,v in values.items()) - -def service_json(project, name, app_image, agent_image, variant, runtime, function=False): - labels={"svls9604":"true","svls9604-run":name.split("-")[1],"runtime":runtime, - "workload":"function-gen2" if function else "cloud-run"} - annotations={"autoscaling.knative.dev/minScale":"1" if variant=="busy" else "0", - "autoscaling.knative.dev/maxScale":"100", - "run.googleapis.com/container-dependencies":json.dumps({"app":["datadog-sidecar"]})} - return {"apiVersion":"serving.knative.dev/v1","kind":"Service", - "metadata":{"name":name,"namespace":project,"labels":labels}, - "spec":{"template":{"metadata":{"annotations":annotations},"spec":{"containerConcurrency":1 if variant=="coldstart" else 80,"timeoutSeconds":300,"containers":[ - {"name":"app","image":app_image,"ports":[{"containerPort":8080}],"env":[{"name":"DD_ENV","value":RUN_ENV}]}, - {"name":"datadog-sidecar","image":agent_image,"startupProbe":{"tcpSocket":{"port":5555},"periodSeconds":3,"failureThreshold":20}, - "env":[{"name":k,"value":v} for k,v in {**env_list(name),"DD_HEALTH_PORT":"5555","DD_APM_NON_LOCAL_TRAFFIC":"true","DD_DOGSTATSD_NON_LOCAL_TRAFFIC":"true", **({"FUNCTION_TARGET":"main"} if function else {})}.items()]} - ]}}}} - -def gcp_service_identity(project, region, service, revision=None): - status=json.loads(run(["gcloud","run","services","describe",service, - f"--project={project}",f"--region={region}", - "--format=json(status.url,status.latestReadyRevisionName)"],capture=True)) - revision=revision or status["status"]["latestReadyRevisionName"] - return {"endpoint":status["status"]["url"], - "resource_id":f"//run.googleapis.com/projects/{project}/locations/{region}/revisions/{revision}", - "parent_resource_id":f"//run.googleapis.com/projects/{project}/locations/{region}/services/{service}", - "deployment_id":revision, - "expected_resource_ids":[f"//run.googleapis.com/projects/{project}/locations/{region}/revisions/{revision}"]} - -def gcp_function_source(runtime, name, run_dir): - source=run_dir/f"function-{name}"; source.mkdir(exist_ok=True) - specs={ - "python":("python312","main"), "node":("nodejs22","main"), "go":("go126","main"), - "java":("java21","functions.Main"), "dotnet":("dotnet10","Function"), - "ruby":("ruby33","main"), "php":("php84","main")} - if runtime == "python": - (source/"main.py").write_text("import functions_framework\n@functions_framework.http\ndef main(request): return 'Hello World!'\n") - (source/"requirements.txt").write_text("functions-framework==3.*\n") - elif runtime == "node": - (source/"index.js").write_text("const functions=require('@google-cloud/functions-framework');functions.http('main',(q,r)=>r.send('Hello World!'));\n") - (source/"package.json").write_text(json.dumps({"name":name,"version":"1.0.0","main":"index.js","dependencies":{"@google-cloud/functions-framework":"^3.4.0"}})) - elif runtime == "go": - (source/"go.mod").write_text("module example.com/svls9604\n\ngo 1.24\n\nrequire github.com/GoogleCloudPlatform/functions-framework-go v1.9.0\n") - (source/"function.go").write_text('package function\nimport("fmt";"net/http";"github.com/GoogleCloudPlatform/functions-framework-go/functions")\nfunc init(){functions.HTTP("main",handler)}\nfunc handler(w http.ResponseWriter,r *http.Request){fmt.Fprint(w,"Hello World!")}\n') - elif runtime == "java": - package=source/"src/main/java/functions"; package.mkdir(parents=True,exist_ok=True) - (source/"pom.xml").write_text('4.0.0functionssvls96041.021com.google.cloud.functionsfunctions-framework-api1.1.4') - (package/"Main.java").write_text('package functions;import com.google.cloud.functions.*;import java.io.*;public class Main implements HttpFunction{public void service(HttpRequest q,HttpResponse r)throws IOException{r.getWriter().write("Hello World!");}}\n') - elif runtime == "dotnet": - (source/"Function.csproj").write_text('net10.0Exeenable') - (source/"Function.cs").write_text('using Google.Cloud.Functions.Framework;using Microsoft.AspNetCore.Http;public class Function:IHttpFunction{public async Task HandleAsync(HttpContext c){await c.Response.WriteAsync("Hello World!");}}\n') - elif runtime == "ruby": - (source/"Gemfile").write_text("source 'https://rubygems.org'\ngem 'functions_framework', '~> 1.4'\n") - (source/"Gemfile.lock").write_text("""GEM - remote: https://rubygems.org/ - specs: - cloud_events (0.9.0) - functions_framework (1.7.0) - cloud_events (>= 0.7.0, < 2.a) - puma (>= 4.3.0, < 9.a) - rack (>= 2.1, < 4.a) - nio4r (2.7.5) - puma (8.0.2) - nio4r (~> 2.0) - rack (3.2.7) - -PLATFORMS - aarch64-linux - ruby - -DEPENDENCIES - functions_framework (~> 1.4) - -BUNDLED WITH - 2.5.22 -""") - (source/"app.rb").write_text("require 'functions_framework'\nFunctionsFramework.http 'main' do |_request|\n 'Hello World!'\nend\n") - else: - (source/"composer.json").write_text(json.dumps({"require":{"google/cloud-functions-framework":"^1.4"}})) - (source/"index.php").write_text("r.status(200).send('Hello World!'));\n") - run(["gcloud","functions","deploy",name,"--no-gen2",f"--project={project}",f"--region={region}","--runtime=nodejs20", - "--trigger-http","--allow-unauthenticated","--entry-point=main",f"--source={source}", - f"--set-env-vars={env_arg(env_list(name))}","--quiet","--format=value(name)"]) - run(["gcloud","functions","add-iam-policy-binding",name,f"--project={project}",f"--region={region}", - "--member=allUsers","--role=roles/cloudfunctions.invoker","--quiet"]) - url=run(["gcloud","functions","describe",name,f"--project={project}",f"--region={region}","--format=value(httpsTrigger.url)"],capture=True) - return {"name":name,"endpoint":url,"resource_id":f"//cloudfunctions.googleapis.com/projects/{project}/locations/{region}/functions/{name}"} - -def trigger(resource, project, region): - if resource["id"]=="SI-03": - run(["gcloud","run","jobs","execute",resource["name"],f"--project={project}",f"--region={region}","--wait"]) - return {"status":"executed"} - url=resource["endpoint"] - result=subprocess.run(["curl","-fsS","--max-time","60",url],text=True,capture_output=True) - return {"status":"ok" if result.returncode==0 else "failed","http_body":result.stdout[:200],"error":result.stderr[:300]} - -def http_request(url): - started=time.monotonic() - try: - with urllib.request.urlopen(url,timeout=60) as response: - response.read(256) - return response.status,round((time.monotonic()-started)*1000,1),None - except Exception as error: - return 0,round((time.monotonic()-started)*1000,1),str(error)[:300] - -def burst(resource, count=80): - with concurrent.futures.ThreadPoolExecutor(max_workers=count) as pool: - results=list(pool.map(http_request,[resource["endpoint"]]*count)) - latencies=sorted(x[1] for x in results) - successes=sum(1 for status,_,_ in results if 200 <= status < 400) - failures=[error or f"HTTP {status}" for status,_,error in results if not 200 <= status < 400] - def percentile(p): - return latencies[min(len(latencies)-1,round((len(latencies)-1)*p))] - return {"attempts":count,"successes":successes,"failures":count-successes, - "latency_ms":{"p50":percentile(.50),"p95":percentile(.95),"max":latencies[-1]}, - "sample_errors":failures[:5]} - - -def run_same_resource_stage(stage_id, targets, count): - started=utc_now() - results=[] - for i,resource in enumerate(targets,1): - print(f"[{stage_id} {i}/{len(targets)}] {count} requests -> {resource['name']}") - results.append({"resource_id":resource["resource_id"],"name":resource["name"], - **burst(resource,count)}) - return {"id":stage_id,"scenario":"same-resource concurrent requests", - "requests_per_resource":count,"started_at":started,"completed_at":utc_now(), - "resources":results} - - -def run_distributed_stage(targets): - started=utc_now() - with concurrent.futures.ThreadPoolExecutor(max_workers=min(100,len(targets))) as pool: - raw=list(pool.map(lambda resource: http_request(resource["endpoint"]),targets)) - resources=[] - for resource,(status,elapsed,error) in zip(targets,raw): - resources.append({"resource_id":resource["resource_id"],"name":resource["name"], - "http_status":status,"elapsed_ms":elapsed,"error":error}) - successes=sum(1 for result in resources if 200 <= result["http_status"] < 400) - return {"id":"L4","scenario":"one concurrent request per distinct resource_id", - "started_at":started,"completed_at":utc_now(),"attempts":len(resources), - "successes":successes,"failures":len(resources)-successes,"resources":resources} - - -def run_sustained_stage(targets, minutes, interval_seconds): - started=utc_now() - deadline=time.monotonic() + minutes*60 - rounds=[] - while True: - round_started=utc_now() - with concurrent.futures.ThreadPoolExecutor(max_workers=min(100,len(targets))) as pool: - raw=list(pool.map(lambda resource: http_request(resource["endpoint"]),targets)) - successes=sum(1 for status,_,_ in raw if 200 <= status < 400) - rounds.append({"started_at":round_started,"completed_at":utc_now(), - "attempts":len(raw),"successes":successes, - "failures":len(raw)-successes}) - remaining=deadline-time.monotonic() - if remaining <= 0: - break - time.sleep(min(interval_seconds,remaining)) - return {"id":"L5","scenario":"unchanged sustained reporting window", - "duration_minutes":minutes,"interval_seconds":interval_seconds, - "started_at":started,"completed_at":utc_now(),"rounds":rounds} - - -def run_full_load_suite(targets, *, sustained_minutes, sustained_interval): - stages=[] - for stage_id,count in FULL_LOAD_STAGES: - stages.append(run_same_resource_stage(stage_id,targets,count)) - stages.append(run_distributed_stage(targets)) - busy=[resource for resource in targets if resource.get("variant")=="busy"] - stages.append(run_sustained_stage(busy or targets,sustained_minutes,sustained_interval)) - return stages - -def add_expected_revision(resource, identity): - expected=resource.setdefault("expected_resource_ids",[resource["resource_id"]]) - if identity["resource_id"] not in expected: - expected.append(identity["resource_id"]) - resource.setdefault("observed_revisions",[]).append(identity) - -def gcp_revision_stages(project, region, targets, run_id): - target=next((r for r in targets if r["id"]=="SI-02" and r["runtime"]=="python" and r["variant"]=="busy"),None) - if not target: - return [{"id":"L6","scenario":"controlled revision cold-start pressure","status":"NOT MEASURED","reason":"representative SI-02 Python busy target missing"}, - {"id":"L7","scenario":"two active revisions with split traffic","status":"NOT MEASURED","reason":"representative SI-02 Python busy target missing"}] - revision_a=target["deployment_id"] - suffix=f"load-{run_id[-6:].lower()}"[:15] - started=utc_now() - run(["gcloud","run","services","update",target["name"],f"--project={project}",f"--region={region}", - "--concurrency=1","--min-instances=0","--max-instances=100",f"--revision-suffix={suffix}", - "--container=datadog-sidecar",f"--update-env-vars=DD_VERSION={suffix}"]) - revision_b=f"{target['name']}-{suffix}" - identity_b=gcp_service_identity(project,region,target["name"],revision_b) - add_expected_revision(target,identity_b) - pressure=burst({**target,"endpoint":identity_b["endpoint"]},100) - l6={"id":"L6","scenario":"fresh revision, concurrency=1, 100-request cold-start pressure", - "started_at":started,"completed_at":utc_now(),"provider":"gcp","resource":target["name"], - "parent_resource_id":target["parent_resource_id"],"revision_a":revision_a, - "revision_b":identity_b["deployment_id"],"resource_id_b":identity_b["resource_id"], - "provider_instance_starts":"REQUIRES LOG EVIDENCE",**pressure} - l7_started=utc_now() - run(["gcloud","run","services","update-traffic",target["name"],f"--project={project}",f"--region={region}", - f"--set-tags=old={revision_a},new={identity_b['deployment_id']}", - f"--to-revisions={revision_a}=10,{identity_b['deployment_id']}=90"]) - traffic=json.loads(run(["gcloud","run","services","describe",target["name"],f"--project={project}",f"--region={region}", - "--format=json(status.traffic)"],capture=True))["status"]["traffic"] - tagged={entry.get("tag"):entry.get("url") for entry in traffic if entry.get("tag") and entry.get("url")} - old_result=burst({**target,"endpoint":tagged["old"]},10) - new_result=burst({**target,"endpoint":tagged["new"]},10) - split_result=burst(target,100) - l7={"id":"L7","scenario":"two active revisions, 10/90 service traffic plus direct tagged revision probes", - "started_at":l7_started,"completed_at":utc_now(),"provider":"gcp","resource":target["name"], - "parent_resource_id":target["parent_resource_id"],"revision_a":revision_a, - "revision_b":identity_b["deployment_id"],"traffic":traffic, - "attempts":split_result["attempts"]+old_result["attempts"]+new_result["attempts"], - "successes":split_result["successes"]+old_result["successes"]+new_result["successes"], - "failures":split_result["failures"]+old_result["failures"]+new_result["failures"], - "service_traffic":split_result,"revision_a_direct":old_result,"revision_b_direct":new_result} - return [l6,l7] - -def gcp_scaling_stages(project, region, targets, run_id, maxima): - target=next((r for r in targets if r["id"]=="SI-02" and r["runtime"]=="python" and r["variant"]=="busy"),None) - if not target: - return [{"id":"L8","scenario":"minimum-instance report fan-out","status":"NOT MEASURED","reason":"representative SI-02 Python busy target missing"}, - {"id":"L9","scenario":"maximum-instance scale-out ceilings","status":"NOT MEASURED","reason":"representative SI-02 Python busy target missing"}] - - def configure_revision(label, minimum, maximum): - suffix=f"{label}-{run_id[-4:].lower()}"[:15] - started_at=utc_now() - run(["gcloud","run","services","update",target["name"],f"--project={project}",f"--region={region}", - "--concurrency=1",f"--min-instances={minimum}",f"--max-instances={maximum}", - f"--revision-suffix={suffix}","--container=datadog-sidecar", - f"--update-env-vars=DD_VERSION={suffix}"]) - revision=f"{target['name']}-{suffix}" - identity=gcp_service_identity(project,region,target["name"],revision) - add_expected_revision(target,identity) - run(["gcloud","run","services","update-traffic",target["name"],f"--project={project}",f"--region={region}", - f"--to-revisions={identity['deployment_id']}=100"]) - identity["started_at"]=started_at - identity["configured_at"]=utc_now() - return identity - - minimum_cases=[] - l8_started=utc_now() - for minimum in (0,5,100): - identity=configure_revision(f"min{minimum}",minimum,100) - probe=burst({**target,"endpoint":identity["endpoint"]},1) - minimum_cases.append({"minimum_instances":minimum,"maximum_instances":100, - "resource_id":identity["resource_id"],"deployment_id":identity["deployment_id"], - "started_at":identity["started_at"],"completed_at":utc_now(), - "provider_instance_starts":"REQUIRES PROVIDER/STARTUP LOG EVIDENCE",**probe}) - l8={"id":"L8","scenario":"minimum-instance report fan-out (0, 5, 100)", - "started_at":l8_started,"completed_at":utc_now(),"provider":"gcp","resource":target["name"], - "cases":minimum_cases,"resources":minimum_cases} - - maximum_cases=[] - l9_started=utc_now() - for maximum in maxima: - identity=configure_revision(f"max{maximum}",0,maximum) - pressure=burst({**target,"endpoint":identity["endpoint"]},100) - maximum_cases.append({"minimum_instances":0,"maximum_instances":maximum, - "resource_id":identity["resource_id"],"deployment_id":identity["deployment_id"], - "started_at":identity["started_at"],"completed_at":utc_now(), - "provider_instance_starts":"REQUIRES PROVIDER/STARTUP LOG EVIDENCE", - "note":"100 requests validate fan-out and identity, not attainment of the configured maximum", - **pressure}) - l9={"id":"L9","scenario":"maximum-instance configuration boundaries with 100-request pressure", - "started_at":l9_started,"completed_at":utc_now(),"provider":"gcp","resource":target["name"], - "cases":maximum_cases,"resources":maximum_cases} - return [l8,l9] - -def azure_revision_identity(app_id, revision, fqdn=None): - resource_id=f"{app_id.rstrip('/')}/revisions/{revision}".lower() - result={"resource_id":resource_id,"parent_resource_id":app_id.lower(), - "deployment_id":revision,"expected_resource_ids":[resource_id]} - if fqdn: - result["endpoint"]="https://"+fqdn - return result - -def azure_revision_stages(resource_group, targets, run_id): - target=next((r for r in targets if r["id"]=="SI-06" and r["runtime"]=="python" and r["variant"]=="busy"),None) - if not target: - return [{"id":"L6","scenario":"controlled revision cold-start pressure","status":"NOT MEASURED","reason":"representative SI-06 Python busy target missing"}, - {"id":"L7","scenario":"two active revisions with split traffic","status":"NOT MEASURED","reason":"representative SI-06 Python busy target missing"}] - revision_a=target["deployment_id"] - suffix=f"load{run_id[-6:].lower()}"[:10] - started=utc_now() - revision_b=run(["az","containerapp","revision","copy","--resource-group",resource_group,"--name",target["name"], - "--from-revision",revision_a,"--container-name","datadog-sidecar", - "--set-env-vars",f"DD_VERSION={suffix}","--revision-suffix",suffix, - "--min-replicas","0","--max-replicas","100","--scale-rule-name","http", - "--scale-rule-type","http","--scale-rule-http-concurrency","1", - "--query","properties.latestRevisionName","-o","tsv"],capture=True) - revisions=json.loads(run(["az","containerapp","revision","list","--resource-group",resource_group,"--name",target["name"],"-o","json"],capture=True)) - by_name={item["name"]:item for item in revisions} - identity_b=azure_revision_identity(target["parent_resource_id"],revision_b,by_name[revision_b]["properties"].get("fqdn")) - add_expected_revision(target,identity_b) - pressure=burst({**target,"endpoint":identity_b["endpoint"]},100) - l6={"id":"L6","scenario":"fresh revision, HTTP concurrency=1, 100-request cold-start pressure", - "started_at":started,"completed_at":utc_now(),"provider":"azure","resource":target["name"], - "parent_resource_id":target["parent_resource_id"],"revision_a":revision_a,"revision_b":revision_b, - "resource_id_b":identity_b["resource_id"],"provider_instance_starts":"REQUIRES REVISION/LOG EVIDENCE",**pressure} - l7_started=utc_now() - run(["az","containerapp","ingress","traffic","set","--resource-group",resource_group,"--name",target["name"], - "--revision-weight",f"{revision_a}=10",f"{revision_b}=90"]) - old_endpoint="https://"+by_name[revision_a]["properties"]["fqdn"] - new_endpoint=identity_b["endpoint"] - old_result=burst({**target,"endpoint":old_endpoint},10) - new_result=burst({**target,"endpoint":new_endpoint},10) - split_result=burst(target,100) - l7={"id":"L7","scenario":"two active revisions, 10/90 service traffic plus direct revision probes", - "started_at":l7_started,"completed_at":utc_now(),"provider":"azure","resource":target["name"], - "parent_resource_id":target["parent_resource_id"],"revision_a":revision_a,"revision_b":revision_b, - "traffic_weights":{revision_a:10,revision_b:90}, - "attempts":split_result["attempts"]+old_result["attempts"]+new_result["attempts"], - "successes":split_result["successes"]+old_result["successes"]+new_result["successes"], - "failures":split_result["failures"]+old_result["failures"]+new_result["failures"], - "service_traffic":split_result,"revision_a_direct":old_result,"revision_b_direct":new_result} - return [l6,l7] - -def azure_scaling_stages(resource_group, targets, run_id, maxima): - target=next((r for r in targets if r["id"]=="SI-06" and r["runtime"]=="python" and r["variant"]=="busy"),None) - if not target: - return [{"id":"L8","scenario":"minimum-replica report fan-out","status":"NOT MEASURED","reason":"representative SI-06 Python busy target missing"}, - {"id":"L9","scenario":"maximum-replica scale-out ceilings","status":"NOT MEASURED","reason":"representative SI-06 Python busy target missing"}] - - def copy_revision(label, minimum, maximum): - suffix=f"{label}{run_id[-4:].lower()}"[:10] - started_at=utc_now() - revision=run(["az","containerapp","revision","copy","--resource-group",resource_group,"--name",target["name"], - "--from-revision",target["deployment_id"],"--container-name","datadog-sidecar", - "--set-env-vars",f"DD_VERSION={suffix}","--revision-suffix",suffix, - "--min-replicas",str(minimum),"--max-replicas",str(maximum),"--scale-rule-name","http", - "--scale-rule-type","http","--scale-rule-http-concurrency","1", - "--query","properties.latestRevisionName","-o","tsv"],capture=True) - revisions=json.loads(run(["az","containerapp","revision","list","--resource-group",resource_group,"--name",target["name"],"-o","json"],capture=True)) - observed=next(item for item in revisions if item["name"]==revision) - identity=azure_revision_identity(target["parent_resource_id"],revision,observed["properties"].get("fqdn")) - identity["started_at"]=started_at - identity["configured_at"]=utc_now() - add_expected_revision(target,identity) - return identity - - minimum_cases=[] - l8_started=utc_now() - for minimum in (0,5,100): - identity=copy_revision(f"min{minimum}",minimum,100) - probe=burst({**target,"endpoint":identity["endpoint"]},1) - minimum_cases.append({"minimum_instances":minimum,"maximum_instances":100, - "resource_id":identity["resource_id"],"deployment_id":identity["deployment_id"], - "started_at":identity["started_at"],"completed_at":utc_now(), - "provider_instance_starts":"REQUIRES PROVIDER/STARTUP LOG EVIDENCE",**probe}) - l8={"id":"L8","scenario":"minimum-replica report fan-out (0, 5, 100)", - "started_at":l8_started,"completed_at":utc_now(),"provider":"azure","resource":target["name"], - "cases":minimum_cases,"resources":minimum_cases} - - maximum_cases=[] - l9_started=utc_now() - for maximum in maxima: - identity=copy_revision(f"max{maximum}",0,maximum) - pressure=burst({**target,"endpoint":identity["endpoint"]},100) - maximum_cases.append({"minimum_instances":0,"maximum_instances":maximum, - "resource_id":identity["resource_id"],"deployment_id":identity["deployment_id"], - "started_at":identity["started_at"],"completed_at":utc_now(), - "provider_instance_starts":"REQUIRES PROVIDER/STARTUP LOG EVIDENCE", - "note":"100 requests validate fan-out and identity, not attainment of the configured maximum", - **pressure}) - l9={"id":"L9","scenario":"maximum-replica configuration boundaries with 100-request pressure", - "started_at":l9_started,"completed_at":utc_now(),"provider":"azure","resource":target["name"], - "cases":maximum_cases,"resources":maximum_cases} - return [l8,l9] - -def azure_params(path, values): - body={"$schema":"https://schema.management.azure.com/schemas/2019-04-01/deploymentParameters.json#", - "contentVersion":"1.0.0.0","parameters":{k:{"value":v} for k,v in values.items()}} - path.write_text(json.dumps(body,indent=2)); path.chmod(0o600) - -def azure_deploy(template, resource_group, deployment, values, run_dir): - params=run_dir/f"{deployment}-parameters.json" - azure_params(params,values) - output=run(["az","deployment","group","create","--resource-group",resource_group, - "--name",deployment,"--template-file",str(ROOT/"azure"/template), - "--parameters",f"@{params}","--query","properties.outputs","-o","json"],capture=True) - return json.loads(output) - -def build_agent_azure(registry, run_id): - image=f"{registry}/svls9604/serverless-init:{run_id}-v3" - sha=git_sha(AGENT_REPO) - release=json.loads((AGENT_REPO/"release.json").read_text()) - version=f"{release['current_milestone']}-dev" - if not remote_image_exists(image): - run(["docker","buildx","build","--platform=linux/amd64","--push", - "--build-arg",f"GIT_COMMIT={sha[:12]}","--build-arg",f"AGENT_VERSION={version}", - "--build-arg",f"SERVERLESS_INIT_VERSION={version}", - "-f",str(AGENT_REPO/"scripts/serverless-deploy/Dockerfile.serverless-init"), - "-t",image,str(AGENT_REPO)]) - digest=image_digest(image) - return f"{registry}/svls9604/serverless-init@{digest}",sha - -def source_zip(runtime, name, run_dir): - source=run_dir/f"source-{name}"; source.mkdir(exist_ok=True) - if runtime == "node": - (source/"package.json").write_text(json.dumps({"name":name,"version":"1.0.0","scripts":{"start":"node app.js"}})) - (source/"app.js").write_text("const http=require('http');http.createServer((q,r)=>r.end('Hello World!')).listen(process.env.PORT||8080,'0.0.0.0');\n") - elif runtime == "python": - (source/"requirements.txt").write_text("") - (source/"app.py").write_text("from http.server import BaseHTTPRequestHandler,HTTPServer\nclass H(BaseHTTPRequestHandler):\n def do_GET(self): self.send_response(200);self.end_headers();self.wfile.write(b'Hello World!')\nHTTPServer(('0.0.0.0',int(__import__('os').environ.get('PORT','8000'))),H).serve_forever()\n") - elif runtime == "dotnet": - (source/"app.csproj").write_text('net8.0enable') - (source/"Program.cs").write_text('var b=WebApplication.CreateBuilder(args);var a=b.Build();a.MapGet("/",()=>"Hello World!");a.Run();\n') - archive=run_dir/f"{name}.zip" - with zipfile.ZipFile(archive,"w",zipfile.ZIP_DEFLATED) as z: - for child in source.iterdir(): z.write(child,child.name) - return archive - -def compat_azure_zip(name, run_dir): - package=build_compat_package(run_dir) - source=run_dir/f"source-{name}"; source.mkdir(exist_ok=True) - (source/"package.tgz").write_bytes(package.read_bytes()) - (source/"package.json").write_text(json.dumps({"name":name,"version":"1.0.0","main":"index.js","dependencies":{"@azure/functions":"4.7.2","@datadog/serverless-compat":"file:package.tgz","dd-trace":"5.45.0"}})) - (source/"host.json").write_text(json.dumps({"version":"2.0"})) - (source/"index.js").write_text("require('@datadog/serverless-compat/init');const{app}=require('@azure/functions');app.http('main',{methods:['GET'],authLevel:'anonymous',handler:async()=>({body:'Hello World!'})});\n") - archive=run_dir/f"{name}.zip" - with zipfile.ZipFile(archive,"w",zipfile.ZIP_DEFLATED) as z: - for child in source.iterdir(): z.write(child,child.name) - return archive - -def deploy_azure_resource(args, r, name, images, agent_image, acr, run_id, run_dir): - common={"name":name,"agentImage":agent_image,"registryServer":acr["server"], - "registryUsername":acr["username"],"registryPassword":acr["password"], - "ddApiKey":os.environ["DD_API_KEY"],"runtime":r["runtime"],"runId":run_id} - deployment=f"d-{name}"[:64] - if r["id"] in ("SI-05","SI-06"): - values={**common,"appEnvId":args.azure_containerapp_env_id, - "appImage":images[f"{r['runtime']}:{'init' if r['id']=='SI-05' else 'plain'}"], - "sidecar":r["id"]=="SI-06","minReplicas":1 if r["variant"]=="busy" else 0} - out=azure_deploy("container-app.bicep",args.azure_containerapp_resource_group,deployment,values,run_dir) - identity=azure_revision_identity(out["resourceId"]["value"],out["latestRevisionName"]["value"],out["fqdn"]["value"]) - return {"name":name,**identity} - if r["id"] in ("SI-07","SI-08"): - plan=args.azure_container_plan_id if r["id"]=="SI-07" else args.azure_sidecar_plan_id - values={**common,"servicePlanId":plan, - "appImage":images[f"{r['runtime']}:{'init' if r['id']=='SI-07' else 'plain'}"], - "sidecar":r["id"]=="SI-08","alwaysOn":r["variant"]=="busy"} - out=azure_deploy("web-app-container.bicep",args.azure_resource_group,deployment,values,run_dir) - return {"name":name,"endpoint":"https://"+out["hostname"]["value"],"resource_id":out["resourceId"]["value"]} - if r["id"]=="SI-09": - values={**common,"servicePlanId":args.azure_code_plan_id,"alwaysOn":r["variant"]=="busy"} - out=azure_deploy("web-app-code.bicep",args.azure_resource_group,deployment,values,run_dir) - package=source_zip(r["runtime"],name,run_dir) - try: - run(["az","webapp","deploy","--resource-group",args.azure_resource_group,"--name",name, - "--src-path",str(package),"--type","zip","--clean","true","--async","true"]) - except RuntimeError as e: - print(f"Warning: az webapp deploy exited non-zero ({e}); verifying via health check",flush=True) - endpoint="https://"+out["hostname"]["value"] - print(f"Waiting for {name} to become healthy...",flush=True) - for _ in range(72): - try: - with urllib.request.urlopen(endpoint,timeout=10) as resp: - if resp.status < 500: - print(f"{name} healthy (HTTP {resp.status})",flush=True) - break - except Exception: - pass - time.sleep(10) - return {"name":name,"endpoint":endpoint,"resource_id":out["resourceId"]["value"]} - storage=("sv"+hashlib.sha256(name.encode()).hexdigest()[:20])[:24] - values={"name":name,"storageName":storage,"ddApiKey":os.environ["DD_API_KEY"],"runId":run_id} - out=azure_deploy("function.bicep",args.azure_function_resource_group,deployment,values,run_dir) - package=compat_azure_zip(name,run_dir) - run(["az","functionapp","deployment","source","config-zip","--resource-group",args.azure_function_resource_group, - "--name",name,"--src",str(package),"--build-remote","true"]) - return {"name":name,"endpoint":"https://"+out["hostname"]["value"]+"/api/main","resource_id":out["resourceId"]["value"]} - -def baseline_stage(resources, started_at): - successes=sum(1 for resource in resources - if resource.get("baseline",{}).get("status") in ("ok","executed")) - return {"id":"L0","scenario":"one baseline request or execution per deployed resource", - "started_at":started_at,"completed_at":utc_now(),"attempts":len(resources), - "successes":successes,"failures":len(resources)-successes} - - -def run_azure(args, resources, run_id, run_dir): - run(["az","acr","login","--name",args.azure_acr]) - credential=json.loads(run(["az","acr","credential","show","--name",args.azure_acr,"-o","json"],capture=True)) - acr={"server":args.azure_registry,"username":credential["username"],"password":credential["passwords"][0]["value"]} - os.environ["SVLS9604_ACR_PASSWORD"]=acr["password"] - agent_image,agent_sha=build_agent_azure(args.azure_registry,run_id) - images=build_runtime_images(f"{args.azure_registry}/svls9604",run_id,agent_image) - manifest_path=run_dir/"run-manifest.json" - existing={} - if manifest_path.exists(): - previous=json.loads(manifest_path.read_text()) - if previous.get("profile") in ("azure","azure-sanity","azure-baseline") and previous.get("agent_image")==agent_image: - candidates=[item for item in previous.get("resources",[]) - if item.get("agent_image")==agent_image and item.get("endpoint")] - with concurrent.futures.ThreadPoolExecutor(max_workers=min(20,len(candidates) or 1)) as pool: - checks=list(pool.map(lambda item: http_request(item["endpoint"]),candidates)) - for item,(status,_,error) in zip(candidates,checks): - if 200 <= status < 500: - existing[item["name"]]=item - else: - print(f"Azure resume will redeploy unhealthy {item['name']} (HTTP {status}: {error})") - deployed=[] - for i,r in enumerate(resources,1): - name=name_for(run_id,r) - if name in existing: - print(f"[{i}/{len(resources)}] reusing deployed {r['id']} {r['runtime']} {r['variant']} as {name}") - deployed.append(existing[name]) - continue - print(f"[{i}/{len(resources)}] deploying {r['id']} {r['runtime']} {r['variant']} as {name}") - try: - observed=deploy_azure_resource(args,r,name,images,agent_image,acr,run_id,run_dir) - except Exception as exc: - print(f" WARNING: deploy failed for {name}: {exc}", flush=True) - continue - deployed.append({**r,**observed,"agent_image":agent_image}) - write_manifest(manifest_path,run_id=run_id,profile=args.profile,agent_sha=agent_sha,agent_image=agent_image,resources=deployed) - baseline_started=utc_now() - for resource in deployed: - result=subprocess.run(["curl","-fsS","--max-time","60",resource["endpoint"]],text=True,capture_output=True) - resource["baseline"]={"status":"ok" if result.returncode==0 else "failed","http_body":result.stdout[:200],"error":result.stderr[:300]} - stages=[baseline_stage(deployed,baseline_started)] - if args.suite == "full" and not args.skip_burst: - targets=[r for r in deployed if r["id"].startswith("SI-") and r.get("endpoint")] - stages.extend(run_full_load_suite(targets,sustained_minutes=args.sustained_minutes, - sustained_interval=args.sustained_interval)) - stages.extend(azure_revision_stages(args.azure_containerapp_resource_group,targets,run_id)) - if args.scaling_matrix: - stages.extend(azure_scaling_stages(args.azure_containerapp_resource_group,targets,run_id,args.scaling_maxima)) - compat_targets=[r for r in deployed if r["id"].startswith("SC-") and r.get("endpoint")] - for stage_id,count in [("SC-L1",10),("SC-L2",50),("SC-L3",100)]: - stages.append(run_same_resource_stage(stage_id,compat_targets,count)) - write_manifest(manifest_path,run_id=run_id,profile=args.profile,agent_sha=agent_sha, - agent_image=agent_image,resources=deployed,stages=stages,suite=args.suite) - return deployed - -def run_gcp(args, resources, run_id, run_dir): - project=args.project or run(["gcloud","config","get-value","project"],capture=True) - region=args.region - registry=ensure_registry(project,region) - agent_image,agent_sha=build_agent(project,region,registry,run_id) - images=build_runtime_images(registry,run_id,agent_image) - manifest_path=run_dir/"run-manifest.json" - existing={} - if manifest_path.exists(): - previous=json.loads(manifest_path.read_text()) - # Revision-scoped identity is required for Cloud Run and Gen2 functions. - # Never resume a manifest produced by the earlier service-scoped runner. - valid=[] - for x in previous.get("resources",[]): - if x.get("agent_image") != agent_image: - continue - if x["id"] not in ("SI-01","SI-02","SI-04"): - valid.append(x) - elif "/revisions/" in x.get("resource_id","") and x.get("parent_resource_id"): - valid.append(x) - existing={x["name"]:x for x in valid} - deployed=[] - for i,r in enumerate(resources,1): - name=name_for(run_id,r) - if name in existing: - print(f"[{i}/{len(resources)}] reusing deployed {r['id']} {r['runtime']} {r['variant']} as {name}") - deployed.append(existing[name]); continue - print(f"[{i}/{len(resources)}] deploying {r['id']} {r['runtime']} {r['variant']} as {name}") - try: - if r["id"]=="SI-03": observed=deploy_job(project,region,r,name,images) - elif r["id"]=="SC-02": observed=deploy_compat_gcp(project,region,name,run_dir) - else: observed=deploy_service(project,region,r,name,images,agent_image,run_dir) - except Exception as exc: - print(f" WARNING: deploy failed for {name}: {exc}", flush=True) - continue - deployed.append({**r,**observed,"agent_image":agent_image}) - write_manifest(manifest_path,run_id=run_id,profile="gcp",project=project,region=region,agent_sha=agent_sha,agent_image=agent_image,resources=deployed) - baseline_started=utc_now() - for resource in deployed: - resource["baseline"]=trigger(resource,project,region) - stages=[baseline_stage(deployed,baseline_started)] - if args.suite == "full" and not args.skip_burst: - targets=[r for r in deployed if r["id"].startswith("SI-") and r.get("endpoint")] - stages.extend(run_full_load_suite(targets,sustained_minutes=args.sustained_minutes, - sustained_interval=args.sustained_interval)) - stages.extend(gcp_revision_stages(project,region,targets,run_id)) - if args.scaling_matrix: - stages.extend(gcp_scaling_stages(project,region,targets,run_id,args.scaling_maxima)) - compat_targets=[r for r in deployed if r["id"].startswith("SC-") and r.get("endpoint")] - for stage_id,count in [("SC-L1",10),("SC-L2",50),("SC-L3",100)]: - stages.append(run_same_resource_stage(stage_id,compat_targets,count)) - write_manifest(manifest_path,run_id=run_id,profile="gcp",project=project,region=region, - agent_sha=agent_sha,agent_image=agent_image,resources=deployed, - stages=stages,suite=args.suite) - return deployed - -def main(): - global RUN_ENV, RUN_STARTED_AT - parser=argparse.ArgumentParser() - parser.add_argument("--profile",choices=["gcp","azure","gcp-sanity","azure-sanity","gcp-baseline","azure-baseline"],required=True) - parser.add_argument("--project",default=os.environ.get("GCP_PROJECT","datadog-serverless-gcp-demo")) - parser.add_argument("--region",default=os.environ.get("GCP_REGION","us-central1")) - parser.add_argument("--azure-resource-group",default=os.environ.get("AZURE_RESOURCE_GROUP","dd-serverless-test-aas")) - parser.add_argument("--azure-containerapp-resource-group",default=os.environ.get("AZURE_CONTAINERAPP_RESOURCE_GROUP","dd-serverless-test-aca")) - parser.add_argument("--azure-function-resource-group",default=os.environ.get("AZURE_FUNCTION_RESOURCE_GROUP","dd-serverless-test-aca")) - parser.add_argument("--azure-acr",default=os.environ.get("AZURE_ACR","ddsvlstestaca")) - parser.add_argument("--azure-registry",default=os.environ.get("AZURE_REGISTRY","ddsvlstestaca.azurecr.io")) - parser.add_argument("--azure-containerapp-env-id",default=os.environ.get("AZURE_CONTAINERAPP_ENV_ID","/subscriptions/1dd25961-a5c7-45bf-a5ba-c1475d365cc7/resourceGroups/dd-serverless-test-aca/providers/Microsoft.App/managedEnvironments/dd-serverless-env")) - parser.add_argument("--azure-container-plan-id",default=os.environ.get("AZURE_CONTAINER_PLAN_ID","/subscriptions/1dd25961-a5c7-45bf-a5ba-c1475d365cc7/resourceGroups/dd-serverless-test-aas/providers/Microsoft.Web/serverfarms/dd-test-plan-container")) - parser.add_argument("--azure-sidecar-plan-id",default=os.environ.get("AZURE_SIDECAR_PLAN_ID","/subscriptions/1dd25961-a5c7-45bf-a5ba-c1475d365cc7/resourceGroups/dd-serverless-test-aas/providers/Microsoft.Web/serverfarms/dd-test-plan-sidecar")) - parser.add_argument("--azure-code-plan-id",default=os.environ.get("AZURE_CODE_PLAN_ID","/subscriptions/1dd25961-a5c7-45bf-a5ba-c1475d365cc7/resourceGroups/dd-serverless-test-aas/providers/Microsoft.Web/serverfarms/dd-test-plan-linux-code")) - parser.add_argument("--run-id") - parser.add_argument("--suite",choices=["baseline","full"],default="full", - help="full runs L0-L5 plus revision cold-start pressure (L6) and split traffic (L7)") - parser.add_argument("--sustained-minutes",type=float,default=15, - help="duration of the L5 unchanged-report window") - parser.add_argument("--sustained-interval",type=float,default=60, - help="seconds between L5 trigger rounds") - parser.add_argument("--scaling-matrix",action="store_true", - help="add L8/L9 minimum and maximum instance/replica boundary cases") - parser.add_argument("--scaling-maxima",default="100,1000,4000", - help="comma-separated L9 maximum instance/replica settings") - parser.add_argument("--plan",action="store_true") - parser.add_argument("--yes",action="store_true") - parser.add_argument("--skip-burst",action="store_true",help="compatibility alias for --suite baseline") - args=parser.parse_args() - args.scaling_maxima=tuple(int(value) for value in args.scaling_maxima.split(",") if value) - if args.skip_burst: - args.suite="baseline" - resources=expand(args.profile) - run_id=args.run_id or dt.datetime.now(dt.timezone.utc).strftime("%m%d%H%M")+secrets.token_hex(2) - RUN_ENV=f"svls9604-{run_id.lower()}" - RUN_STARTED_AT=dt.datetime.now(dt.timezone.utc).isoformat() - print(f"Profile: {args.profile}\nRun ID: {run_id}\nSuite: {args.suite}\nExpected resources: {len(resources)}") - for r in resources: - print(f" {r['id']:5} {r['workload_type']:28} {r.get('deployment_model','compat'):12} {r['runtime']:7} {r['variant']}") - if args.plan: - return - preflight(args) - if not args.yes: - if input(f"Create {len(resources)} resources? [y/N] ").strip().lower() != "y": - raise SystemExit("cancelled") - run_dir=pathlib.Path(os.environ.get("RESULTS_DIR",f"/tmp/svls9604-{run_id}")); run_dir.mkdir(parents=True,exist_ok=True) - run_dir.chmod(0o700) - failure_exit=0 - if args.profile in ("gcp", "gcp-sanity", "gcp-baseline"): - deployed=run_gcp(args,resources,run_id,run_dir) - failed=[r for r in deployed if r.get("baseline",{}).get("status") not in ("ok","executed")] - print(f"GCP profile deployed {len(deployed)}/{len(resources)}; baseline failures={len(failed)}") - print(f"Evidence: {run_dir}") - if failed: failure_exit=2 - else: - deployed=run_azure(args,resources,run_id,run_dir) - failed=[r for r in deployed if r.get("baseline",{}).get("status") != "ok"] - print(f"Azure profile deployed {len(deployed)}/{len(resources)}; baseline failures={len(failed)}") - print(f"Evidence: {run_dir}") - if failed: failure_exit=2 - - manifest_path=run_dir/"run-manifest.json" - manifest=json.loads(manifest_path.read_text()) - manifest["dd_env"]=RUN_ENV - manifest["started_at"]=RUN_STARTED_AT - manifest["completed_at"]=utc_now() - manifest["candidate_commits"]={ - "serverless_components":git_sha(COMPONENTS_REPO), - "datadog_agent":git_sha(AGENT_REPO), - "datadog_serverless_compat_js":git_sha(COMPAT_JS_REPO), - } - manifest_path.write_text(json.dumps(manifest,indent=2)) - run([sys.executable,str(ROOT/"report.py"),"--manifest",str(manifest_path)]) - for stage in manifest.get("load_stages",[]): - _,_,stage_failures=aggregate_stage(stage) - if stage_failures: - failure_exit=2 - if failure_exit: - raise SystemExit(failure_exit) - -if __name__=="__main__": - main() From ec88ef6acf20fd8c4cc059f27d4bc003c131ae92 Mon Sep 17 00:00:00 2001 From: Nina Rei Date: Tue, 8 Sep 2026 15:28:18 -0400 Subject: [PATCH 6/9] fix(compat-inventory): dead branch, log level, retry count, azure owner once, clock warning --- .../src/inventory.rs | 54 ++++++++++++------- 1 file changed, 34 insertions(+), 20 deletions(-) diff --git a/crates/datadog-serverless-compat/src/inventory.rs b/crates/datadog-serverless-compat/src/inventory.rs index 1381700..12bf60a 100644 --- a/crates/datadog-serverless-compat/src/inventory.rs +++ b/crates/datadog-serverless-compat/src/inventory.rs @@ -6,7 +6,7 @@ use libdd_trace_utils::trace_utils::EnvironmentType; use std::env; use std::time::{Duration, SystemTime, UNIX_EPOCH}; use tokio::time::interval; -use tracing::{info, warn}; +use tracing::{debug, warn}; /// How often to send a periodic inventory report while the mini-agent is running. const INVENTORY_INTERVAL: Duration = Duration::from_secs(30 * 60); @@ -108,7 +108,15 @@ async fn send_report( process_id: &str, report_reason: &str, ) { - let (mut resource_id, resource_name) = build_resource_identity(env_type); + // Read WEBSITE_OWNER_NAME once for Azure; reused by both identity derivation + // and payload enrichment to avoid two env reads for the same value. + let azure_owner_name = if matches!(env_type, EnvironmentType::AzureFunction) { + env::var("WEBSITE_OWNER_NAME").unwrap_or_default() + } else { + String::new() + }; + + let (mut resource_id, resource_name) = build_resource_identity(env_type, &azure_owner_name); // Gen1 Cloud Functions: if FUNCTION_NAME was present but region/project were // absent from env vars, try the GCP instance metadata server to complete the @@ -159,6 +167,7 @@ async fn send_report( &resource_id, &resource_name, env_type, + &azure_owner_name, ) { Ok(b) => b, Err(e) => { @@ -171,11 +180,11 @@ async fn send_report( for attempt in 0..=MAX_RETRIES { match do_send(client, &url, api_key, body.clone()).await { - Ok(status) if status < 300 || status == 202 => { - info!( + Ok(status) if status < 300 => { + debug!( "inventory: report sent \ (report_reason={report_reason}, workload_type={workload_type}, \ - resource_id={resource_id}, process_id={process_id}, status={status})" + status={status})" ); return; } @@ -205,8 +214,9 @@ async fn send_report( } Err(e) => { warn!( - "inventory: transport error after {attempt} attempts \ - (report_reason={report_reason}, error={e})" + "inventory: transport error after {} attempts \ + (report_reason={report_reason}, error={e})", + attempt + 1, ); return; } @@ -241,12 +251,16 @@ fn build_payload( resource_id: &str, resource_name: &str, env_type: &EnvironmentType, + azure_owner_name: &str, ) -> Result, serde_json::Error> { // Must be nanoseconds to match time.Now().UnixNano() expected by EPRW. let timestamp = SystemTime::now() .duration_since(UNIX_EPOCH) .map(|d| d.as_nanos() as i64) - .unwrap_or(0); + .unwrap_or_else(|e| { + warn!("inventory: system clock error, timestamp will be 0: {e}"); + 0 + }); // serverless_compat_version: prefer DD_SERVERLESS_COMPAT_VERSION (set by the // language package wrapping this binary); fall back to the Rust crate version @@ -280,7 +294,7 @@ fn build_payload( } // Platform-specific optional fields (runtime, region, cloud IDs, etc.). - enrich_platform_fields(&mut metadata, env_type); + enrich_platform_fields(&mut metadata, env_type, azure_owner_name); // Hostname intentionally absent: setting it (even to "") causes EPRW to // attempt a host_id lookup that fails for serverless workloads, rejecting @@ -299,9 +313,9 @@ fn build_payload( /// `resource_id` is the canonical cloud resource identifier used as the primary /// key in `serverless_compat_agent`. An empty `resource_id` means the required /// environment variables are absent; the caller must skip the write. -fn build_resource_identity(env_type: &EnvironmentType) -> (String, String) { +fn build_resource_identity(env_type: &EnvironmentType, azure_owner_name: &str) -> (String, String) { match env_type { - EnvironmentType::AzureFunction => build_azure_function_identity(), + EnvironmentType::AzureFunction => build_azure_function_identity(azure_owner_name), EnvironmentType::CloudFunction => build_cloud_function_identity(), EnvironmentType::LambdaFunction | EnvironmentType::AzureSpringApp => { (String::new(), String::new()) @@ -309,10 +323,8 @@ fn build_resource_identity(env_type: &EnvironmentType) -> (String, String) { } } -fn build_azure_function_identity() -> (String, String) { +fn build_azure_function_identity(owner_name: &str) -> (String, String) { let name = env::var("WEBSITE_SITE_NAME").unwrap_or_default(); - // WEBSITE_OWNER_NAME = "{subscription_guid}+{rg}-{region}webspace[-os]" - let owner_name = env::var("WEBSITE_OWNER_NAME").unwrap_or_default(); let sub = owner_name .split('+') @@ -326,7 +338,7 @@ fn build_azure_function_identity() -> (String, String) { let rg = env::var("WEBSITE_RESOURCE_GROUP") .ok() .filter(|s| !s.is_empty()) - .or_else(|| parse_rg_from_owner_name(&owner_name)) + .or_else(|| parse_rg_from_owner_name(owner_name)) .unwrap_or_default(); if name.is_empty() || rg.is_empty() || sub.is_empty() { @@ -461,17 +473,19 @@ async fn fetch_gcp_project_from_metadata() -> Option { } /// Adds platform-specific optional fields to `metadata`. -fn enrich_platform_fields(metadata: &mut serde_json::Value, env_type: &EnvironmentType) { +fn enrich_platform_fields( + metadata: &mut serde_json::Value, + env_type: &EnvironmentType, + azure_owner_name: &str, +) { match env_type { - EnvironmentType::AzureFunction => enrich_azure_function_fields(metadata), + EnvironmentType::AzureFunction => enrich_azure_function_fields(metadata, azure_owner_name), EnvironmentType::CloudFunction => enrich_cloud_function_fields(metadata), EnvironmentType::LambdaFunction | EnvironmentType::AzureSpringApp => {} } } -fn enrich_azure_function_fields(metadata: &mut serde_json::Value) { - let owner_name = env::var("WEBSITE_OWNER_NAME").unwrap_or_default(); - +fn enrich_azure_function_fields(metadata: &mut serde_json::Value, owner_name: &str) { // Region: prefer REGION_NAME; fall back to parsing WEBSITE_OWNER_NAME. let region = env::var("REGION_NAME") .ok() From 164e239d2fa520bd88e00081c688ffad7281a12d Mon Sep 17 00:00:00 2001 From: Nina Rei Date: Tue, 8 Sep 2026 16:37:29 -0400 Subject: [PATCH 7/9] fix(compat-inventory): fix test breakage from fix6 refactor (info->debug, owner_name args) --- .../src/inventory.rs | 21 +++++++------------ 1 file changed, 8 insertions(+), 13 deletions(-) diff --git a/crates/datadog-serverless-compat/src/inventory.rs b/crates/datadog-serverless-compat/src/inventory.rs index 12bf60a..a3badae 100644 --- a/crates/datadog-serverless-compat/src/inventory.rs +++ b/crates/datadog-serverless-compat/src/inventory.rs @@ -446,7 +446,7 @@ async fn fetch_gcp_metadata_value( let body = resp.text().await.ok()?; let result = parse(body.trim()); - info!("inventory: GCP metadata server {label}: {:?}", result); + debug!("inventory: GCP metadata server {label}: {:?}", result); result } @@ -656,10 +656,9 @@ mod tests { unsafe { env::set_var("WEBSITE_SITE_NAME", "my-func-app"); env::set_var("WEBSITE_RESOURCE_GROUP", "my-rg"); - env::set_var("WEBSITE_OWNER_NAME", "abc123+my-rg-eastuswebspace"); } - let (id, name) = build_azure_function_identity(); + let (id, name) = build_azure_function_identity("abc123+my-rg-eastuswebspace"); assert_eq!(name, "my-func-app"); assert_eq!( @@ -670,7 +669,6 @@ mod tests { unsafe { env::remove_var("WEBSITE_SITE_NAME"); env::remove_var("WEBSITE_RESOURCE_GROUP"); - env::remove_var("WEBSITE_OWNER_NAME"); } } @@ -681,13 +679,9 @@ mod tests { unsafe { env::set_var("WEBSITE_SITE_NAME", "my-func"); env::remove_var("WEBSITE_RESOURCE_GROUP"); - env::set_var( - "WEBSITE_OWNER_NAME", - "sub123+my-resource-group-westus2webspace-Linux", - ); } - let (id, name) = build_azure_function_identity(); + let (id, name) = build_azure_function_identity("sub123+my-resource-group-westus2webspace-Linux"); assert_eq!(name, "my-func"); assert!( @@ -697,7 +691,6 @@ mod tests { unsafe { env::remove_var("WEBSITE_SITE_NAME"); - env::remove_var("WEBSITE_OWNER_NAME"); } } @@ -707,10 +700,9 @@ mod tests { unsafe { env::remove_var("WEBSITE_SITE_NAME"); env::set_var("WEBSITE_RESOURCE_GROUP", "my-rg"); - env::set_var("WEBSITE_OWNER_NAME", "abc123+my-rg-eastuswebspace"); } - let (id, _name) = build_azure_function_identity(); + let (id, _name) = build_azure_function_identity("abc123+my-rg-eastuswebspace"); assert!( id.is_empty(), "missing WEBSITE_SITE_NAME must produce empty resource_id" @@ -718,7 +710,6 @@ mod tests { unsafe { env::remove_var("WEBSITE_RESOURCE_GROUP"); - env::remove_var("WEBSITE_OWNER_NAME"); } } @@ -874,6 +865,7 @@ mod tests { "//microsoft.azure/functionApps/sub/rg/my-func", "my-func", &EnvironmentType::AzureFunction, + "", ) .expect("build_payload must not fail"); @@ -920,6 +912,7 @@ mod tests { "//microsoft.azure/functionApps/s/r/f", "f", &EnvironmentType::AzureFunction, + "", ) .unwrap(); let payload: serde_json::Value = serde_json::from_slice(&body).unwrap(); @@ -947,6 +940,7 @@ mod tests { "//microsoft.azure/functionApps/s/r/f", "f", &EnvironmentType::AzureFunction, + "", ) .unwrap(); let payload: serde_json::Value = serde_json::from_slice(&body).unwrap(); @@ -970,6 +964,7 @@ mod tests { "//cloudfunctions.googleapis.com/projects/p/locations/r/functions/fn", "fn", &EnvironmentType::CloudFunction, + "", ) .unwrap(); let payload: serde_json::Value = serde_json::from_slice(&body).unwrap(); From 1326442d8c180e553dbaf88b6f9f9d1c3b8fa95b Mon Sep 17 00:00:00 2001 From: Nina Rei Date: Tue, 8 Sep 2026 17:07:45 -0400 Subject: [PATCH 8/9] fix(compat-inventory): rustfmt line wrap for build_azure_function_identity test call --- crates/datadog-serverless-compat/src/inventory.rs | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/crates/datadog-serverless-compat/src/inventory.rs b/crates/datadog-serverless-compat/src/inventory.rs index a3badae..65f9f95 100644 --- a/crates/datadog-serverless-compat/src/inventory.rs +++ b/crates/datadog-serverless-compat/src/inventory.rs @@ -681,7 +681,8 @@ mod tests { env::remove_var("WEBSITE_RESOURCE_GROUP"); } - let (id, name) = build_azure_function_identity("sub123+my-resource-group-westus2webspace-Linux"); + let (id, name) = + build_azure_function_identity("sub123+my-resource-group-westus2webspace-Linux"); assert_eq!(name, "my-func"); assert!( From 5a10d294f22e544cd522dcd992893ba5bbe056e1 Mon Sep 17 00:00:00 2001 From: Nina Rei Date: Thu, 24 Sep 2026 09:21:31 -0400 Subject: [PATCH 9/9] refactor(compat-inventory): extract and harden inventory reporter --- Cargo.lock | 16 +- .../Cargo.toml | 19 + .../src/lib.rs} | 535 +++++++++++------- crates/datadog-serverless-compat/Cargo.toml | 2 +- crates/datadog-serverless-compat/src/main.rs | 4 +- 5 files changed, 361 insertions(+), 215 deletions(-) create mode 100644 crates/datadog-serverless-compat-inventory/Cargo.toml rename crates/{datadog-serverless-compat/src/inventory.rs => datadog-serverless-compat-inventory/src/lib.rs} (67%) diff --git a/Cargo.lock b/Cargo.lock index 87e4f9a..f7b83c9 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -585,6 +585,7 @@ dependencies = [ "datadog-fips", "datadog-logs-agent", "datadog-metrics-collector", + "datadog-serverless-compat-inventory", "datadog-trace-agent", "dogstatsd", "libdd-trace-utils 8.0.0", @@ -595,10 +596,23 @@ dependencies = [ "tokio-util", "tracing", "tracing-subscriber", - "uuid", "zstd", ] +[[package]] +name = "datadog-serverless-compat-inventory" +version = "0.1.0" +dependencies = [ + "datadog-fips", + "libdd-common 4.2.0", + "libdd-trace-utils 8.0.0", + "reqwest", + "serde_json", + "tokio", + "tracing", + "uuid", +] + [[package]] name = "datadog-trace-agent" version = "0.1.0" diff --git a/crates/datadog-serverless-compat-inventory/Cargo.toml b/crates/datadog-serverless-compat-inventory/Cargo.toml new file mode 100644 index 0000000..bcc0818 --- /dev/null +++ b/crates/datadog-serverless-compat-inventory/Cargo.toml @@ -0,0 +1,19 @@ +# Copyright 2025-Present Datadog, Inc. https://www.datadoghq.com/ +# SPDX-License-Identifier: Apache-2.0 + +[package] +name = "datadog-serverless-compat-inventory" +version = "0.1.0" +edition.workspace = true +license.workspace = true +description = "Fleet Automation inventory reporter for the Serverless Compat mini-agent" + +[dependencies] +datadog-fips = { path = "../datadog-fips" } +libdd-common = { git = "https://github.com/DataDog/libdatadog", rev = "a820699426f28cbabb3a74d87c7309d030b52e7c", default-features = false } +libdd-trace-utils = { git = "https://github.com/DataDog/libdatadog", rev = "a820699426f28cbabb3a74d87c7309d030b52e7c" } +reqwest = { version = "0.12.4", default-features = false, features = ["rustls-tls"] } +serde_json = { version = "1.0", default-features = false, features = ["alloc"] } +tokio = { version = "1", features = ["macros", "rt-multi-thread", "time"] } +tracing = { version = "0.1", default-features = false } +uuid = { version = "1", default-features = false, features = ["v4"] } diff --git a/crates/datadog-serverless-compat/src/inventory.rs b/crates/datadog-serverless-compat-inventory/src/lib.rs similarity index 67% rename from crates/datadog-serverless-compat/src/inventory.rs rename to crates/datadog-serverless-compat-inventory/src/lib.rs index 65f9f95..3714f62 100644 --- a/crates/datadog-serverless-compat/src/inventory.rs +++ b/crates/datadog-serverless-compat-inventory/src/lib.rs @@ -2,6 +2,7 @@ // SPDX-License-Identifier: Apache-2.0 use datadog_fips::reqwest_adapter::create_reqwest_client_builder; +use libdd_common::azure_app_services::{AzureMetadata, QueryEnv, UNKNOWN_VALUE}; use libdd_trace_utils::trace_utils::EnvironmentType; use std::env; use std::time::{Duration, SystemTime, UNIX_EPOCH}; @@ -11,14 +12,10 @@ use tracing::{debug, warn}; /// How often to send a periodic inventory report while the mini-agent is running. const INVENTORY_INTERVAL: Duration = Duration::from_secs(30 * 60); -/// Maximum retry attempts for transient failures (429, 5xx, transport errors). +/// Maximum total send attempts for transient failures (429, 5xx, transport errors). +/// Includes the initial attempt — 3 means one try plus up to two retries. const MAX_RETRIES: u32 = 3; -/// Minimum Datadog agent protocol version accepted by EPRW (7.x.x format). -/// Used only for HTTP transport headers. The actual Compat version is reported -/// as `agent_metadata.serverless_compat_version`. -const AGENT_VERSION: &str = "7.83.0"; - /// Supported Compat workload types. Unsupported env types (Lambda, Azure Spring /// Apps) are silently skipped so they never create a `serverless_compat_agent` row. fn supported_workload_type(env_type: &EnvironmentType) -> Option<&'static str> { @@ -30,17 +27,17 @@ fn supported_workload_type(env_type: &EnvironmentType) -> Option<&'static str> { } } -/// Runs the inventory reporter for the lifetime of the mini-agent. -/// -/// Sends a startup report immediately, then a periodic report every -/// [`INVENTORY_INTERVAL`]. Spawned as a background task — never panics, -/// never blocks agent startup. /// Returns true only when `DD_SERVERLESS_COMPAT_INVENTORY_ENABLED=true`. /// Extracted so the gate logic can be unit-tested without an async runtime. fn is_inventory_enabled() -> bool { env::var("DD_SERVERLESS_COMPAT_INVENTORY_ENABLED").as_deref() == Ok("true") } +/// Runs the inventory reporter for the lifetime of the mini-agent. +/// +/// Sends a startup report immediately, then a periodic report every +/// [`INVENTORY_INTERVAL`]. Spawned as a background task — never panics, +/// never blocks agent startup. pub async fn run_inventory_reporter( api_key: &str, dd_site: &str, @@ -108,15 +105,16 @@ async fn send_report( process_id: &str, report_reason: &str, ) { - // Read WEBSITE_OWNER_NAME once for Azure; reused by both identity derivation - // and payload enrichment to avoid two env reads for the same value. - let azure_owner_name = if matches!(env_type, EnvironmentType::AzureFunction) { - env::var("WEBSITE_OWNER_NAME").unwrap_or_default() + // Reuse libdatadog's canonical Azure parsing for identity and enrichment. + // This includes Flex Consumption's DD_AZURE_RESOURCE_GROUP fallback. + let azure_metadata = if matches!(env_type, EnvironmentType::AzureFunction) { + AzureMetadata::new_function(ProcessEnv) } else { - String::new() + None }; - let (mut resource_id, resource_name) = build_resource_identity(env_type, &azure_owner_name); + let (mut resource_id, resource_name) = + build_resource_identity(env_type, azure_metadata.as_ref()); // Gen1 Cloud Functions: if FUNCTION_NAME was present but region/project were // absent from env vars, try the GCP instance metadata server to complete the @@ -167,7 +165,7 @@ async fn send_report( &resource_id, &resource_name, env_type, - &azure_owner_name, + azure_metadata.as_ref(), ) { Ok(b) => b, Err(e) => { @@ -178,7 +176,7 @@ async fn send_report( let url = format!("https://api.{dd_site}/api/v1/metadata"); - for attempt in 0..=MAX_RETRIES { + for attempt in 0..MAX_RETRIES { match do_send(client, &url, api_key, body.clone()).await { Ok(status) if status < 300 => { debug!( @@ -188,35 +186,36 @@ async fn send_report( ); return; } - Ok(429) | Ok(500..=599) if attempt < MAX_RETRIES => { + Ok(429) | Ok(500..=599) if attempt + 1 < MAX_RETRIES => { let backoff = Duration::from_secs(1 << attempt); warn!( "inventory: transient failure, retrying in {backoff:?} \ - (report_reason={report_reason}, attempt={attempt})" + (report_reason={report_reason}, attempt={})", + attempt + 1 ); tokio::time::sleep(backoff).await; } Ok(status) => { warn!( "inventory: intake rejected report \ - (report_reason={report_reason}, status={status}, \ - resource_id={resource_id}, process_id={process_id})" + (report_reason={report_reason}, workload_type={workload_type}, \ + status={status})" ); return; } - Err(e) if attempt < MAX_RETRIES => { + Err(e) if attempt + 1 < MAX_RETRIES => { let backoff = Duration::from_secs(1 << attempt); warn!( "inventory: transport error, retrying in {backoff:?} \ - (report_reason={report_reason}, attempt={attempt}, error={e})" + (report_reason={report_reason}, attempt={}, error={e})", + attempt + 1 ); tokio::time::sleep(backoff).await; } Err(e) => { warn!( - "inventory: transport error after {} attempts \ - (report_reason={report_reason}, error={e})", - attempt + 1, + "inventory: transport error after {MAX_RETRIES} attempts \ + (report_reason={report_reason}, error={e})" ); return; } @@ -235,8 +234,10 @@ async fn do_send( .post(url) .header("DD-API-KEY", api_key) .header("Content-Type", "application/json") - .header("DD-Agent-Version", AGENT_VERSION) - .header("User-Agent", format!("datadog-agent/{AGENT_VERSION}")) + .header( + "User-Agent", + format!("datadog-serverless-compat/{}", env!("CARGO_PKG_VERSION")), + ) .body(body) .send() .await?; @@ -251,7 +252,7 @@ fn build_payload( resource_id: &str, resource_name: &str, env_type: &EnvironmentType, - azure_owner_name: &str, + azure_metadata: Option<&AzureMetadata>, ) -> Result, serde_json::Error> { // Must be nanoseconds to match time.Now().UnixNano() expected by EPRW. let timestamp = SystemTime::now() @@ -294,7 +295,7 @@ fn build_payload( } // Platform-specific optional fields (runtime, region, cloud IDs, etc.). - enrich_platform_fields(&mut metadata, env_type, azure_owner_name); + enrich_platform_fields(&mut metadata, env_type, azure_metadata); // Hostname intentionally absent: setting it (even to "") causes EPRW to // attempt a host_id lookup that fails for serverless workloads, rejecting @@ -313,9 +314,12 @@ fn build_payload( /// `resource_id` is the canonical cloud resource identifier used as the primary /// key in `serverless_compat_agent`. An empty `resource_id` means the required /// environment variables are absent; the caller must skip the write. -fn build_resource_identity(env_type: &EnvironmentType, azure_owner_name: &str) -> (String, String) { +fn build_resource_identity( + env_type: &EnvironmentType, + azure_metadata: Option<&AzureMetadata>, +) -> (String, String) { match env_type { - EnvironmentType::AzureFunction => build_azure_function_identity(azure_owner_name), + EnvironmentType::AzureFunction => build_azure_function_identity(azure_metadata), EnvironmentType::CloudFunction => build_cloud_function_identity(), EnvironmentType::LambdaFunction | EnvironmentType::AzureSpringApp => { (String::new(), String::new()) @@ -323,66 +327,57 @@ fn build_resource_identity(env_type: &EnvironmentType, azure_owner_name: &str) - } } -fn build_azure_function_identity(owner_name: &str) -> (String, String) { - let name = env::var("WEBSITE_SITE_NAME").unwrap_or_default(); +struct ProcessEnv; - let sub = owner_name - .split('+') - .next() - .filter(|s| !s.is_empty()) - .map(str::to_string) - .unwrap_or_default(); +impl QueryEnv for ProcessEnv { + fn get_var(&self, name: &str) -> Option { + env::var(name) + .ok() + .map(|value| value.trim().to_string()) + .filter(|value| !value.is_empty()) + } +} - // WEBSITE_RESOURCE_GROUP is not always injected; parse from WEBSITE_OWNER_NAME - // when absent. Format after '+': "{rg}-{region}webspace[-Linux|-Windows]" - let rg = env::var("WEBSITE_RESOURCE_GROUP") - .ok() - .filter(|s| !s.is_empty()) - .or_else(|| parse_rg_from_owner_name(owner_name)) - .unwrap_or_default(); +fn known_azure_value(value: &str) -> Option<&str> { + (value != UNKNOWN_VALUE && !value.is_empty()).then_some(value) +} + +fn build_azure_function_identity(metadata: Option<&AzureMetadata>) -> (String, String) { + let Some(metadata) = metadata else { + return (String::new(), String::new()); + }; - if name.is_empty() || rg.is_empty() || sub.is_empty() { + let name = known_azure_value(metadata.get_site_name()) + .unwrap_or_default() + .to_string(); + let Some(base_resource_id) = known_azure_value(metadata.get_resource_id()) else { return (String::new(), name); - } + }; - let resource_id = format!( - "/subscriptions/{}/resourcegroups/{}/providers/microsoft.web/sites/{}", - sub.to_lowercase(), - rg.to_lowercase(), - name.to_lowercase() - ); - (resource_id, name) -} + // Non-production deployment slots share WEBSITE_SITE_NAME with the parent app but + // expose a distinct ARM path: /sites/{name}/slots/{slot}. Azure sets WEBSITE_SLOT_NAME + // for every slot; the production slot uses the value "production". Appending the slot + // path ensures each slot gets its own inventory row rather than overwriting the parent. + let slot = env::var("WEBSITE_SLOT_NAME") + .ok() + .map(|value| value.trim().to_string()) + .filter(|value| !value.is_empty() && !value.eq_ignore_ascii_case("production")); -/// Parses the resource group from `WEBSITE_OWNER_NAME`. -/// -/// Format: `"{sub}+{rg}-{region}webspace[-Linux|-Windows]"` -/// Strips the OS suffix, "webspace", then the trailing "-{region}" segment. -fn parse_rg_from_owner_name(owner_name: &str) -> Option { - let after_plus = owner_name.split('+').nth(1)?; - let stripped = after_plus - .strip_suffix("-Linux") - .or_else(|| after_plus.strip_suffix("-Windows")) - .unwrap_or(after_plus); - let without_webspace = stripped.strip_suffix("webspace")?; - let last_dash = without_webspace.rfind('-')?; - let rg = &without_webspace[..last_dash]; - if rg.is_empty() { - None - } else { - Some(rg.to_string()) - } + let resource_id = match slot { + Some(slot) => format!("{base_resource_id}/slots/{}", slot.to_lowercase()), + None => base_resource_id.to_string(), + }; + (resource_id, name) } fn build_cloud_function_identity() -> (String, String) { - // Gen2 Cloud Run Functions set FUNCTION_TARGET alongside K_SERVICE. - // These belong in serverless_init_agent, not serverless_compat_agent. - if env::var("FUNCTION_TARGET") - .map(|v| !v.is_empty()) - .unwrap_or(false) - { - return (String::new(), String::new()); - } + // The compat binary is only ever installed in Gen1 Cloud Functions. + // Gen2 (Cloud Run Functions) uses serverless-init via sidecar instead. + // We no longer guard on FUNCTION_TARGET here: newer Gen1 runtimes + // (Python 3.11, Node.js 20+) run on Cloud Run infrastructure and set + // FUNCTION_TARGET to the entry-point name — indistinguishable from Gen2 + // purely on env vars. Removing the guard is safe because the binary is + // only present when the user explicitly installed the compat package. // Gen1: FUNCTION_NAME is canonical; newer Gen1 runtimes on Cloud Run infra // may omit it and expose K_SERVICE instead. @@ -420,21 +415,32 @@ async fn fetch_gcp_metadata_value( label: &str, parse: impl Fn(&str) -> Option, ) -> Option { - let client = create_reqwest_client_builder() - .and_then(|b| { - b.timeout(Duration::from_secs(2)) - .build() - .map_err(Into::into) - }) - .ok()?; + let client = match create_reqwest_client_builder().and_then(|builder| { + builder + .timeout(Duration::from_secs(2)) + .build() + .map_err(Into::into) + }) { + Ok(client) => client, + Err(error) => { + warn!("inventory: failed to create GCP metadata client for {label}: {error}"); + return None; + } + }; let url = format!("http://metadata.google.internal/computeMetadata/v1/{path}"); - let resp = client + let resp = match client .get(&url) .header("Metadata-Flavor", "Google") .send() .await - .ok()?; + { + Ok(response) => response, + Err(error) => { + warn!("inventory: failed to fetch GCP metadata {label}: {error}"); + return None; + } + }; if !resp.status().is_success() { warn!( @@ -444,7 +450,13 @@ async fn fetch_gcp_metadata_value( return None; } - let body = resp.text().await.ok()?; + let body = match resp.text().await { + Ok(body) => body, + Err(error) => { + warn!("inventory: failed to read GCP metadata {label}: {error}"); + return None; + } + }; let result = parse(body.trim()); debug!("inventory: GCP metadata server {label}: {:?}", result); result @@ -476,68 +488,47 @@ async fn fetch_gcp_project_from_metadata() -> Option { fn enrich_platform_fields( metadata: &mut serde_json::Value, env_type: &EnvironmentType, - azure_owner_name: &str, + azure_metadata: Option<&AzureMetadata>, ) { match env_type { - EnvironmentType::AzureFunction => enrich_azure_function_fields(metadata, azure_owner_name), + EnvironmentType::AzureFunction => enrich_azure_function_fields(metadata, azure_metadata), EnvironmentType::CloudFunction => enrich_cloud_function_fields(metadata), EnvironmentType::LambdaFunction | EnvironmentType::AzureSpringApp => {} } } -fn enrich_azure_function_fields(metadata: &mut serde_json::Value, owner_name: &str) { - // Region: prefer REGION_NAME; fall back to parsing WEBSITE_OWNER_NAME. +fn enrich_azure_function_fields( + metadata: &mut serde_json::Value, + azure_metadata: Option<&AzureMetadata>, +) { + // REGION_NAME is the canonical Azure-provided region value. let region = env::var("REGION_NAME") .ok() - .filter(|s| !s.is_empty()) - .or_else(|| { - let after_plus = owner_name.split('+').nth(1)?; - let without_webspace = after_plus - .strip_suffix("-Linux") - .or_else(|| after_plus.strip_suffix("-Windows")) - .unwrap_or(after_plus) - .strip_suffix("webspace")?; - without_webspace.split('-').next_back().map(str::to_string) - }); + .map(|value| value.trim().to_string()) + .filter(|value| !value.is_empty()); if let Some(r) = region { metadata["region"] = serde_json::Value::String(r); } - if let Some(sub) = owner_name.split('+').next().filter(|s| !s.is_empty()) { - metadata["azure_subscription_id"] = serde_json::Value::String(sub.to_string()); - } - if let Ok(rg) = env::var("WEBSITE_RESOURCE_GROUP") - && !rg.is_empty() - { - metadata["azure_resource_group"] = serde_json::Value::String(rg); - } - - // Runtime: prefer DD_SERVERLESS_COMPAT_RUNTIME (set by language package); - // fall back to FUNCTIONS_WORKER_RUNTIME injected by Azure. - let runtime = env::var("DD_SERVERLESS_COMPAT_RUNTIME") - .ok() - .filter(|s| !s.is_empty()) - .or_else(|| { - env::var("FUNCTIONS_WORKER_RUNTIME") - .ok() - .filter(|s| !s.is_empty()) - }); - if let Some(rt) = runtime { - metadata["runtime"] = serde_json::Value::String(rt); - } - - // Runtime version: prefer DD_SERVERLESS_COMPAT_RUNTIME_VERSION (language package), - // then FUNCTIONS_WORKER_RUNTIME_VERSION, then language-specific vars. - let runtime_ver = env::var("DD_SERVERLESS_COMPAT_RUNTIME_VERSION") - .ok() - .filter(|s| !s.is_empty()) - .or_else(|| { - env::var("FUNCTIONS_WORKER_RUNTIME_VERSION") - .ok() - .filter(|s| !s.is_empty()) - }); - if let Some(v) = runtime_ver { - metadata["serverless_compat_runtime_version"] = serde_json::Value::String(v); + if let Some(azure_metadata) = azure_metadata { + for (field, value) in [ + ( + "azure_subscription_id", + azure_metadata.get_subscription_id(), + ), + ("azure_resource_group", azure_metadata.get_resource_group()), + // Preserve Azure's documented raw runtime values (for example node, + // python, dotnet, or dotnet-isolated); do not invent a second taxonomy. + ("runtime", azure_metadata.get_runtime()), + ( + "serverless_compat_runtime_version", + azure_metadata.get_runtime_version(), + ), + ] { + if let Some(value) = known_azure_value(value) { + metadata[field] = serde_json::Value::String(value.to_string()); + } + } } } @@ -560,19 +551,10 @@ fn enrich_cloud_function_fields(metadata: &mut serde_json::Value) { metadata["gcp_project_id"] = serde_json::Value::String(p); } - // Runtime: prefer DD_SERVERLESS_COMPAT_RUNTIME (language package); fall back - // to detecting from well-known GCP Cloud Functions gen1 env vars. - let (lang, ver) = env::var("DD_SERVERLESS_COMPAT_RUNTIME") - .ok() - .filter(|s| !s.is_empty()) - .map(|rt| { - let ver = env::var("DD_SERVERLESS_COMPAT_RUNTIME_VERSION") - .ok() - .filter(|s| !s.is_empty()) - .unwrap_or_default(); - (rt, ver) - }) - .unwrap_or_else(detect_gcp_gen1_runtime); + // These values come from the GCP runtime itself. The language packages set + // DD_SERVERLESS_COMPAT_VERSION, which is a package version and must not be + // confused with the language runtime version. + let (lang, ver) = detect_gcp_gen1_runtime(); if !lang.is_empty() { metadata["runtime"] = serde_json::Value::String(lang); @@ -618,6 +600,27 @@ mod tests { static ENV_LOCK: std::sync::LazyLock> = std::sync::LazyLock::new(|| std::sync::Mutex::new(())); + unsafe fn clear_azure_function_env() { + for name in [ + "DD_AZURE_RESOURCE_GROUP", + "FUNCTIONS_EXTENSION_VERSION", + "FUNCTIONS_WORKER_RUNTIME", + "FUNCTIONS_WORKER_RUNTIME_VERSION", + "REGION_NAME", + "WEBSITE_OWNER_NAME", + "WEBSITE_RESOURCE_GROUP", + "WEBSITE_SITE_NAME", + "WEBSITE_SKU", + "WEBSITE_SLOT_NAME", + ] { + unsafe { env::remove_var(name) }; + } + } + + fn current_azure_metadata() -> AzureMetadata { + AzureMetadata::new_function(ProcessEnv).expect("Azure Functions metadata should be built") + } + // ── Workload type filtering ────────────────────────────────────────────── #[test] @@ -654,11 +657,15 @@ mod tests { fn azure_function_identity_full() { let _lock = ENV_LOCK.lock().unwrap(); unsafe { + clear_azure_function_env(); + env::set_var("FUNCTIONS_WORKER_RUNTIME", "python"); + env::set_var("WEBSITE_OWNER_NAME", "abc123+my-rg-eastuswebspace"); env::set_var("WEBSITE_SITE_NAME", "my-func-app"); env::set_var("WEBSITE_RESOURCE_GROUP", "my-rg"); } - let (id, name) = build_azure_function_identity("abc123+my-rg-eastuswebspace"); + let azure_metadata = current_azure_metadata(); + let (id, name) = build_azure_function_identity(Some(&azure_metadata)); assert_eq!(name, "my-func-app"); assert_eq!( @@ -667,8 +674,7 @@ mod tests { ); unsafe { - env::remove_var("WEBSITE_SITE_NAME"); - env::remove_var("WEBSITE_RESOURCE_GROUP"); + clear_azure_function_env(); } } @@ -677,12 +683,17 @@ mod tests { // WEBSITE_RESOURCE_GROUP absent; RG parsed from WEBSITE_OWNER_NAME. let _lock = ENV_LOCK.lock().unwrap(); unsafe { + clear_azure_function_env(); + env::set_var("FUNCTIONS_WORKER_RUNTIME", "python"); + env::set_var( + "WEBSITE_OWNER_NAME", + "sub123+my-resource-group-westus2webspace-Linux", + ); env::set_var("WEBSITE_SITE_NAME", "my-func"); - env::remove_var("WEBSITE_RESOURCE_GROUP"); } - let (id, name) = - build_azure_function_identity("sub123+my-resource-group-westus2webspace-Linux"); + let azure_metadata = current_azure_metadata(); + let (id, name) = build_azure_function_identity(Some(&azure_metadata)); assert_eq!(name, "my-func"); assert!( @@ -691,7 +702,33 @@ mod tests { ); unsafe { - env::remove_var("WEBSITE_SITE_NAME"); + clear_azure_function_env(); + } + } + + #[test] + fn azure_function_identity_flex_consumption_uses_dd_resource_group() { + let _lock = ENV_LOCK.lock().unwrap(); + unsafe { + clear_azure_function_env(); + env::set_var("FUNCTIONS_WORKER_RUNTIME", "python"); + env::set_var("WEBSITE_OWNER_NAME", "abc123+flex-host"); + env::set_var("WEBSITE_SITE_NAME", "my-flex-func"); + env::set_var("WEBSITE_SKU", "FlexConsumption"); + env::set_var("DD_AZURE_RESOURCE_GROUP", "My-Flex-RG"); + } + + let azure_metadata = current_azure_metadata(); + let (id, name) = build_azure_function_identity(Some(&azure_metadata)); + + assert_eq!(name, "my-flex-func"); + assert_eq!( + id, + "/subscriptions/abc123/resourcegroups/my-flex-rg/providers/microsoft.web/sites/my-flex-func" + ); + + unsafe { + clear_azure_function_env(); } } @@ -699,18 +736,100 @@ mod tests { fn azure_function_identity_missing_name_returns_empty() { let _lock = ENV_LOCK.lock().unwrap(); unsafe { - env::remove_var("WEBSITE_SITE_NAME"); + clear_azure_function_env(); + env::set_var("FUNCTIONS_WORKER_RUNTIME", "python"); + env::set_var("WEBSITE_OWNER_NAME", "abc123+my-rg-eastuswebspace"); env::set_var("WEBSITE_RESOURCE_GROUP", "my-rg"); } - let (id, _name) = build_azure_function_identity("abc123+my-rg-eastuswebspace"); + let azure_metadata = current_azure_metadata(); + let (id, _name) = build_azure_function_identity(Some(&azure_metadata)); assert!( id.is_empty(), "missing WEBSITE_SITE_NAME must produce empty resource_id" ); unsafe { - env::remove_var("WEBSITE_RESOURCE_GROUP"); + clear_azure_function_env(); + } + } + + #[test] + fn azure_function_identity_non_production_slot() { + let _lock = ENV_LOCK.lock().unwrap(); + unsafe { + clear_azure_function_env(); + env::set_var("FUNCTIONS_WORKER_RUNTIME", "python"); + env::set_var("WEBSITE_OWNER_NAME", "abc123+my-rg-eastuswebspace"); + env::set_var("WEBSITE_SITE_NAME", "my-func-app"); + env::set_var("WEBSITE_RESOURCE_GROUP", "my-rg"); + env::set_var("WEBSITE_SLOT_NAME", "staging"); + } + + let azure_metadata = current_azure_metadata(); + let (id, name) = build_azure_function_identity(Some(&azure_metadata)); + + assert_eq!(name, "my-func-app"); + assert!( + id.ends_with("/slots/staging"), + "non-production slot must appear in resource_id; got: {id}" + ); + + unsafe { + clear_azure_function_env(); + } + } + + #[test] + fn azure_function_identity_production_slot_omitted() { + let _lock = ENV_LOCK.lock().unwrap(); + unsafe { + clear_azure_function_env(); + env::set_var("FUNCTIONS_WORKER_RUNTIME", "python"); + env::set_var("WEBSITE_OWNER_NAME", "abc123+my-rg-eastuswebspace"); + env::set_var("WEBSITE_SITE_NAME", "my-func-app"); + env::set_var("WEBSITE_RESOURCE_GROUP", "my-rg"); + env::set_var("WEBSITE_SLOT_NAME", "Production"); + } + + let azure_metadata = current_azure_metadata(); + let (id, _name) = build_azure_function_identity(Some(&azure_metadata)); + + assert!( + !id.contains("/slots/"), + "production slot must not appear in resource_id; got: {id}" + ); + + unsafe { + clear_azure_function_env(); + } + } + + #[test] + fn azure_function_enrichment_uses_platform_runtime_values() { + let _lock = ENV_LOCK.lock().unwrap(); + unsafe { + clear_azure_function_env(); + env::set_var("FUNCTIONS_WORKER_RUNTIME", "dotnet-isolated"); + env::set_var("FUNCTIONS_WORKER_RUNTIME_VERSION", "8.0"); + env::set_var("REGION_NAME", "East US 2"); + env::set_var("WEBSITE_OWNER_NAME", "abc123+my-rg-eastus2webspace"); + env::set_var("WEBSITE_RESOURCE_GROUP", "my-rg"); + env::set_var("WEBSITE_SITE_NAME", "my-func-app"); + } + + let azure_metadata = current_azure_metadata(); + let mut metadata = serde_json::json!({}); + enrich_azure_function_fields(&mut metadata, Some(&azure_metadata)); + + assert_eq!(metadata["runtime"], "dotnet-isolated"); + assert_eq!(metadata["serverless_compat_runtime_version"], "8.0"); + assert_eq!(metadata["region"], "East US 2"); + assert_eq!(metadata["azure_subscription_id"], "abc123"); + assert_eq!(metadata["azure_resource_group"], "my-rg"); + + unsafe { + clear_azure_function_env(); } } @@ -767,8 +886,10 @@ mod tests { } #[test] - fn cloud_function_gen2_with_function_target_skipped() { - // Gen2 Cloud Run Functions: FUNCTION_TARGET set → must not write to compat table. + fn cloud_function_newer_gen1_with_function_target_still_reported() { + // Newer Gen1 runtimes (Python 3.11+, Node.js 20+) run on Cloud Run + // infrastructure and set FUNCTION_TARGET alongside K_SERVICE. The compat + // binary's presence is the gate — we always report if installed. let _lock = ENV_LOCK.lock().unwrap(); unsafe { env::set_var("K_SERVICE", "my-service"); @@ -779,13 +900,10 @@ mod tests { let (id, name) = build_cloud_function_identity(); + assert_eq!(name, "my-service"); assert!( - id.is_empty(), - "Gen2 must produce empty resource_id; got: {id}" - ); - assert!( - name.is_empty(), - "Gen2 must produce empty resource_name; got: {name}" + id.contains("my-service"), + "newer Gen1 with FUNCTION_TARGET must still produce resource_id; got: {id}" ); unsafe { @@ -849,6 +967,27 @@ mod tests { } } + #[test] + fn cloud_function_runtime_comes_from_platform() { + let _lock = ENV_LOCK.lock().unwrap(); + unsafe { + env::remove_var("NODE_VERSION"); + env::remove_var("JAVA_VERSION"); + env::remove_var("GO_VERSION"); + env::set_var("PYTHON_VERSION", "3.12.7"); + } + + let mut metadata = serde_json::json!({}); + enrich_cloud_function_fields(&mut metadata); + + assert_eq!(metadata["runtime"], "python"); + assert_eq!(metadata["serverless_compat_runtime_version"], "3.12.7"); + + unsafe { + env::remove_var("PYTHON_VERSION"); + } + } + // ── Payload structure ──────────────────────────────────────────────────── #[test] @@ -859,14 +998,16 @@ mod tests { } let process_id = "test-uuid-1234"; + let resource_id = + "/subscriptions/sub/resourcegroups/rg/providers/microsoft.web/sites/my-func"; let body = build_payload( process_id, "azure_function", "startup", - "//microsoft.azure/functionApps/sub/rg/my-func", + resource_id, "my-func", &EnvironmentType::AzureFunction, - "", + None, ) .expect("build_payload must not fail"); @@ -882,10 +1023,7 @@ mod tests { assert_eq!(meta["flavor"], "serverless-compat"); assert_eq!(meta["workload_type"], "azure_function"); assert_eq!(meta["report_reason"], "startup"); - assert_eq!( - meta["resource_id"], - "//microsoft.azure/functionApps/sub/rg/my-func" - ); + assert_eq!(meta["resource_id"], resource_id); assert_eq!(meta["resource_name"], "my-func"); assert!(meta.contains_key("serverless_compat_version")); @@ -910,10 +1048,10 @@ mod tests { "pid", "azure_function", "startup", - "//microsoft.azure/functionApps/s/r/f", + "/subscriptions/s/resourcegroups/r/providers/microsoft.web/sites/f", "f", &EnvironmentType::AzureFunction, - "", + None, ) .unwrap(); let payload: serde_json::Value = serde_json::from_slice(&body).unwrap(); @@ -938,10 +1076,10 @@ mod tests { "pid", "azure_function", "startup", - "//microsoft.azure/functionApps/s/r/f", + "/subscriptions/s/resourcegroups/r/providers/microsoft.web/sites/f", "f", &EnvironmentType::AzureFunction, - "", + None, ) .unwrap(); let payload: serde_json::Value = serde_json::from_slice(&body).unwrap(); @@ -965,38 +1103,13 @@ mod tests { "//cloudfunctions.googleapis.com/projects/p/locations/r/functions/fn", "fn", &EnvironmentType::CloudFunction, - "", + None, ) .unwrap(); let payload: serde_json::Value = serde_json::from_slice(&body).unwrap(); assert_eq!(payload["agent_metadata"]["report_reason"], "periodic"); } - // ── RG parsing ─────────────────────────────────────────────────────────── - - #[test] - fn parse_rg_linux_suffix() { - let rg = parse_rg_from_owner_name("sub+my-rg-eastuswebspace-Linux"); - assert_eq!(rg.as_deref(), Some("my-rg")); - } - - #[test] - fn parse_rg_windows_suffix() { - let rg = parse_rg_from_owner_name("sub+my-rg-westus2webspace-Windows"); - assert_eq!(rg.as_deref(), Some("my-rg")); - } - - #[test] - fn parse_rg_no_os_suffix() { - let rg = parse_rg_from_owner_name("sub+my-rg-eastuswebspace"); - assert_eq!(rg.as_deref(), Some("my-rg")); - } - - #[test] - fn parse_rg_missing_plus_returns_none() { - assert!(parse_rg_from_owner_name("noplushere").is_none()); - } - // ── Inventory gate ─────────────────────────────────────────────────────── #[test] diff --git a/crates/datadog-serverless-compat/Cargo.toml b/crates/datadog-serverless-compat/Cargo.toml index 55c03b5..b28aa94 100644 --- a/crates/datadog-serverless-compat/Cargo.toml +++ b/crates/datadog-serverless-compat/Cargo.toml @@ -13,6 +13,7 @@ windows-enhanced-metrics = ["datadog-metrics-collector/windows-enhanced-metrics" [dependencies] datadog-logs-agent = { path = "../datadog-logs-agent" } datadog-metrics-collector = { path = "../datadog-metrics-collector" } +datadog-serverless-compat-inventory = { path = "../datadog-serverless-compat-inventory" } datadog-trace-agent = { path = "../datadog-trace-agent" } libdd-trace-utils = { git = "https://github.com/DataDog/libdatadog", rev = "a820699426f28cbabb3a74d87c7309d030b52e7c" } datadog-fips = { path = "../datadog-fips", default-features = false } @@ -30,7 +31,6 @@ tracing-subscriber = { version = "0.3", default-features = false, features = [ "env-filter", "tracing-log", ] } -uuid = { version = "1", default-features = false, features = ["v4"] } zstd = { version = "0.13.3", default-features = false } [dev-dependencies] diff --git a/crates/datadog-serverless-compat/src/main.rs b/crates/datadog-serverless-compat/src/main.rs index dae6432..9356da2 100644 --- a/crates/datadog-serverless-compat/src/main.rs +++ b/crates/datadog-serverless-compat/src/main.rs @@ -28,7 +28,7 @@ use datadog_metrics_collector::azure_cpu::CpuMetricsCollector; use libdd_trace_utils::{config_utils::read_cloud_env, trace_utils::EnvironmentType}; -mod inventory; +use datadog_serverless_compat_inventory::run_inventory_reporter; use datadog_fips::reqwest_adapter::create_reqwest_client_builder; use datadog_logs_agent::{ @@ -174,7 +174,7 @@ pub async fn main() { let https_proxy_inv = https_proxy.clone(); let env_type_inv = env_type.clone(); tokio::spawn(async move { - inventory::run_inventory_reporter( + run_inventory_reporter( &api_key, &dd_site_inv, https_proxy_inv.as_deref(),